initial commit

This commit is contained in:
2025-05-21 04:25:09 +02:00
commit f5b11100e5
4 changed files with 423 additions and 0 deletions

5
.gitignore vendored Normal file
View File

@@ -0,0 +1,5 @@
.git/
node_modules/
.DS_Store
*.log
.env

3
LICENSE Normal file
View File

@@ -0,0 +1,3 @@
CC0 1.0 Universal
The person...

58
README.md Normal file
View File

@@ -0,0 +1,58 @@
# search-and-analyze
**Author:** Victor Giers
# search-and-analyze
**Author:** Victor Giers
## Overview
This project contains a Python script that automates the process of generating search queries, performing web searches, extracting and analyzing content from web pages, and providing relevant information based on user input. The script uses advanced language models for query generation and analysis, ensures efficient web scraping with Playwright, and supports various configurations for different AI models.
## Features
1. **Query Generation:** Generates 5 optimized search queries based on user input using a specified LLM model.
2. **Web Search:** Searches the web using SearXNG to find relevant URLs.
3. **Content Extraction:** Extracts content from web pages, rendering JavaScript if necessary.
4. **Analysis:** Analyzes extracted text for relevance and provides summaries or specific information based on user queries.
5. **Fallback Mechanism:** Handles failures in data extraction or analysis by moving to the next URL.
6. **Parallel Processing:** Loads and analyzes up to 5 URLs concurrently.
7. **Language Support:** Automatically detects and supports multiple languages using `langid` and `langcodes`.
8. **NLTK Integration:** Supports NLTK for text summarization as an alternative to external LLM models.
## Requirements
- Python 3.x
- Required Python packages: `requests`, `asyncio`, `argparse`, `urllib`, `json`, `re`, `termios`, `atexit`, `signal`, `time`, `langid`, `langcodes`, `collections`, `newspaper3k`, `nltk`, `playwright`, `langchain_core`, `langchain_ollama`
- SearXNG running locally at `http://127.0.0.1:8888`
## Installation
```bash
pip install requests asyncio argparse urllib json re termios atexit signal time langid langcodes collections newspaper3k nltk playwright langchain_core langchain_ollama
playwright install chromium
```
## Usage
Run the script with a user prompt and optional arguments for analysis and query models.
```bash
python search-and-analyze.py "your search query" --analysis-model mistral-small3.1:24b --query-model mistral:latest
```
- `--analysis-model`: Specifies the AI model to use for content analysis (e.g., `mistral-small3.1:24b`). Use `NLTK` for a local summary using Newspaper3k.
- `--query-model`: Specifies the LLM model to use for generating search queries (default is `mistral:latest`).
## Example
```bash
python search-and-analyze.py "origin of Valentine’s Day" --analysis-model NLTK
```
This command will generate search queries related to the origin of Valentine's Day, perform web searches, extract and summarize content from relevant pages, and output the results.
## License
This project is licensed under the MIT License - see the [LICENSE](LICENSE) file for details.

357
search-and-analyze.py Normal file
View File

@@ -0,0 +1,357 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
Kombiniertes Skript mit dynamischer Fallback-Logik:
1. Generiert 5 Suchanfragen via LLM (query_model)
2. Sucht per SearXNG je 5 Treffer pro Query
3. Entfernt Dubletten und generiert einfache Kandidatenliste
4. Lädt Kandidaten parallel (max 5) mit Playwright, rendert JS, Timeout
5. Extrahiert Text via Newspaper3k (ohne Zusammenfassung)
6. Fallback: Bei Fehlschlag oder "failed" vom LLM wird die nächste URL geladen
7. Gibt für jede der ersten 5 validen Seiten sofort ein JSON-Objekt aus und startet parallel eine AI-Analyse (analysis_model)
8. Unterstützt schnelles NLTK-Summary, wenn --analysis-model NLTK
9. Säubert Inhalte per Regex und stellt sauberes Beenden sicher
"""
import sys
import os
import termios
import atexit, signal
import time
import json
import re
import requests
import asyncio
import argparse
from urllib.parse import urlparse, unquote
import langid
import langcodes
from collections import deque
from langchain_core.runnables import RunnableSequence
from langchain_core.prompts import PromptTemplate
from langchain_ollama import OllamaLLM
from playwright.async_api import async_playwright
from newspaper import Article
import nltk
parser = argparse.ArgumentParser(...)
parser.add_argument('prompt' , help='The user’s search query or input text to be researched and analyzed.')
parser.add_argument('--analysis-model', default='mistral-small3.1:24b', help='The AI model to use for content analysis (e.g. "mistral:7b"), or "NTLK" to use a local Newspaper3k summary.')
parser.add_argument('--query-model' , default='mistral:latest', help='The LLM model to use for generating search queries (e.g. "mistral:7b").')
args = parser.parse_args()
user_input = args.prompt
analysis_model = args.analysis_model
query_model = args.query_model
use_nltk = analysis_model.strip().upper() == 'NTLK'
# Sauberes Beenden
t0_settings = None
try:
t0_settings = termios.tcgetattr(sys.stdin)
except:
pass
def safe_terminal_restore():
if t0_settings:
termios.tcsetattr(sys.stdin, termios.TCSADRAIN, t0_settings)
# ensure sane mode
try:
os.system('stty sane')
except:
pass
atexit.register(safe_terminal_restore)
for sig in (signal.SIGINT, signal.SIGTERM):
signal.signal(sig, lambda *args: (safe_terminal_restore(), sys.exit(0)))
# Punkt-Tokenizer sicherstellen
try:
nltk.data.find('tokenizers/punkt')
except LookupError:
nltk.download('punkt')
# Startzeit
t0 = time.time()
# Fix namespace
#user_input = parser.parse_args().prompt
#analysis_model = parser.parse_args().analysis_model
#query_model = parser.parse_args().query_model
# Flag für NLTK
#use_nltk = analysis_model.strip().upper() == 'NTLK'
# === 1) Query-Generierung via LLM ===
query_llm = OllamaLLM(model=query_model)
query_prompt = PromptTemplate(
input_variables=["input", "lang"],
template="""You are a search-query generator for online search engines.
Your task is to transform any user input into a list of precise, search-engine-optimized queries.
Follow these steps:
1. Analyze the user’s text and extract the core terms and topics. Correct potential typos.
2. Formulate 5 concise search queries (each 2–6 words) that provide optimal entry points for web research.
3. Vary the queries slightly to cover different aspects and phrasing (e.g., synonyms, related terms, specific detail questions).
4. Return the result as a JSON object under the key "queries", for example:
Example 1:
Input: "give me information about the betobeto-san yokai"
Output:
{{
"queries": [
"Betobeto-san Yokai origin",
"Betobeto-san Japanese spirit legend",
"Betobeto-san Yokai stories",
"Betobeto-san appearance description",
"Betobeto-san folklore sources"
]
}}
Example 2:
Input: "how does quantum entanglement work?"
Output:
{{
"queries": [
"quantum entanglement explanation",
"how quantum entanglement works physics",
"quantum entanglement experiments examples",
"applications of quantum entanglement",
"quantum entanglement vs theory"
]
}}
Example 3:
Input: "tips for a vegetarian diet"
Output:
{{
"queries": [
"vegetarian diet tips",
"healthy vegetarian recipes",
"vegetarian grocery list",
"protein sources for vegetarians",
"vegetarian meal plans"
]
}}
Example 4:
Input: "explain the law of large numbers"
Output:
{{
"queries": [
"law of large numbers explanation",
"statistics law of large numbers example",
"law of large numbers sampling mathematics",
"stochastic law of large numbers theorem",
"applications of law of large numbers"
]
}}
Example 5:
Input: "best time to visit the Lofoten"
Output:
{{
"queries": [
"best time to visit Lofoten weather",
"Lofoten climate by season",
"Lofoten northern lights months",
"Lofoten summer activities",
"Lofoten winter polar lights"
]
}}
Example 6:
Input: "JavaScript Promise vs. Callback"
Output:
{{
"queries": [
"JavaScript promise vs callback difference",
"JS promise callback examples",
"asynchronous JS promises callbacks comparison",
"JavaScript callbacks vs promises tutorial",
"best practices promises callbacks JS"
]
}}
Example 7:
Input: "origin of Valentine’s Day"
Output:
{{
"queries": [
"origin of Valentine’s Day history",
"Saint Valentine legend",
"Valentine’s Day customs evolution",
"historical sources Valentine’s Day",
"spread of Valentine’s Day traditions"
]
}}
It's VERY important that the queries are supposed to be in {lang}, so make sure that you answer in {lang}!
Now process the following user input and return **only** the JSON object with the field "queries":
{input}
"""
)
query_chain = RunnableSequence(query_prompt, query_llm)
def generate_search_queries_and_lang(user_prompt: str):
lang_code, _ = langid.classify(user_prompt)
# Maximiere das Tag, sodass wir auch eine Region bekommen
lang_obj = langcodes.Language.get(lang_code).maximize()
language = lang_obj.language # z. B. "de"
region = lang_obj.region or language.upper() # z. B. "DE"
locale_tag = f"{language}-{region}" # ergibt "de-DE"
# Anzeige-Name für das LLM
lang_display = lang_obj.display_name() # z. B. "German"
raw = query_chain.invoke({"input": user_prompt, "lang": lang_display})
data = json.loads(raw)
return data.get("queries", []), lang_display, locale_tag
# === 2) Analysis-Chain mit variablem Model ===
if not use_nltk:
analysis_llm = OllamaLLM(model=analysis_model)
analysis_prompt = PromptTemplate(
input_variables=["question", "content", "lang"],
template=(
"You are a multilingual assistant. Answer in {lang}.\n"
"Determine whether the following text contains information relevant to the question \"{question}\". "
"If it does, summarize only that information. Don't mention the text in your reply, just display the information about the question \"{question}\". "
"IMPORTANT: If it does NOT contain any relevant information, respond only with 'failed' and nothing else!\n"
"Here is the text:\n\n{content}\n"
"---\n"
"IMPORTANT: Answer in {lang}!\n"
"IMPORTANT: If it does NOT contain any relevant information or no answer to the question is given, or if the text does not mention anything about \"{question}\", respond only with 'failed' and nothing else! Don't talk about the text itself! \n"
)
)
analysis_chain = RunnableSequence(analysis_prompt, analysis_llm)
# === 3) SearXNG-Suche ===
def searx_search(query: str, max_results: int = 5) -> list:
resp = requests.get('http://127.0.0.1:8888/search', params={'q': query, 'format':'json'})
resp.raise_for_status()
return [ {'url': e.get('url',''), 'title': e.get('title',''), 'snippet': e.get('content','')}
for e in resp.json().get('results',[])[:max_results] ]
# === 4) Kandidatenliste ===
def canonicalize_url(u: str) -> str:
p = urlparse(u)
return f"{p.scheme or 'http'}://{p.netloc.lower()}{unquote(p.path).lower().rstrip('/')}"
def select_candidate_list(all_results: dict) -> list:
seen = set(); candidates = []
for lst in all_results.values():
for r in lst:
can = canonicalize_url(r['url'])
if can and can not in seen:
seen.add(can); candidates.append(can)
return candidates
# === 5) Playwright fetch ===
async def fetch_html_safe(url: str, locale: str, timeout: float = 30.0) -> str:
browser = None
try:
async with async_playwright() as p:
browser = await p.chromium.launch(headless=True,
args=['--disable-blink-features=AutomationControlled','--no-sandbox','--disable-dev-shm-usage'])
ctx = await browser.new_context(
user_agent='Mozilla/5.0',
locale=locale, # Hier nutzen wir das übergebene locale
timezone_id='UTC'
)
page = await ctx.new_page()
resp = await page.goto(url, wait_until='domcontentloaded', timeout=timeout*1000)
if not resp or resp.status != 200:
raise RuntimeError(f'HTTP {resp.status}')
await page.evaluate("""() => {document.querySelectorAll('.cookie-notice, .cookie-banner, .cookie-consent').forEach(el=>el.remove());}""")
html = await page.content()
await ctx.close(); await browser.close()
if len(html) < 2000: raise RuntimeError('HTML too short')
return html
except Exception:
if browser:
try: await browser.close()
except: pass
raise
# === 6) Text-Extraktion ===
def extract_text(html: str, url:str) -> str:
art = Article(url); art.download(input_html=html); art.parse()
return (art.text or '').strip()
# === 7) Runner mit flexibler Analyse ===
def runner(candidates, question, lang, locale, max_out=5, max_concurrent=5):
async def _run():
loop = asyncio.get_running_loop()
loop.set_exception_handler(lambda l,c: None)
queue = deque(candidates)
count = 0
running = set()
def is_failed(ans: str) -> bool:
return bool(re.search(r'(^|\W)failed(\W|$)', ans, re.IGNORECASE))
async def proc(url: str):
nonlocal count
# fetch, extract, analyse...
try:
html = await asyncio.wait_for(fetch_html_safe(url, locale), timeout=20)
txt = extract_text(html, url)
clean = re.sub(r'https?://\S+', '', txt)
if use_nltk:
art = Article(url); art.download(input_html=html); art.parse(); art.nlp()
ans = art.summary or ''
else:
# offload the blocking LLM call
ans = await asyncio.to_thread(
analysis_chain.invoke,
{'question': question, 'content': clean, 'lang': lang}
)
if is_failed(ans):
raise RuntimeError('No information after analysis')
print(json.dumps({'url': url, 'content': clean, 'analysis': ans}, ensure_ascii=False))
count += 1
return True
except Exception as e:
print(f"Skipped {url}: {type(e).__name__} – {repr(e)}", file=sys.stderr)
return False
# helper to spawn next task if needed
def try_spawn():
if queue and count < max_out and len(running) < max_concurrent:
next_url = queue.popleft()
task = asyncio.create_task(proc(next_url))
running.add(task)
# initially fill up to concurrency
for _ in range(min(max_concurrent, len(queue))):
try_spawn()
# process tasks as they complete
while running:
done, running = await asyncio.wait(running, return_when=asyncio.FIRST_COMPLETED)
for task in done:
success = await task
# as soon as we have enough results, cancel everything
if count >= max_out:
for t in running:
t.cancel()
await asyncio.gather(*running, return_exceptions=True)
return
# otherwise, try to spawn one more
try_spawn()
try:
asyncio.run(_run())
finally:
safe_terminal_restore()
# === 8) Main ===
if __name__ == '__main__':
queries, lang, browser_locale = generate_search_queries_and_lang(user_input)
results = {q: searx_search(q) for q in queries}
candidates = select_candidate_list(results)
runner(candidates, user_input, lang, browser_locale)
print(f"\n/* Total Runtime: {time.time()-t0:.2f} s */")
sys.exit(0)