Files
search-and-analyze/search-and-analyze.py
2025-05-21 04:25:09 +02:00

357 lines
13 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
Kombiniertes Skript mit dynamischer Fallback-Logik:
1. Generiert 5 Suchanfragen via LLM (query_model)
2. Sucht per SearXNG je 5 Treffer pro Query
3. Entfernt Dubletten und generiert einfache Kandidatenliste
4. Lädt Kandidaten parallel (max 5) mit Playwright, rendert JS, Timeout
5. Extrahiert Text via Newspaper3k (ohne Zusammenfassung)
6. Fallback: Bei Fehlschlag oder "failed" vom LLM wird die nächste URL geladen
7. Gibt für jede der ersten 5 validen Seiten sofort ein JSON-Objekt aus und startet parallel eine AI-Analyse (analysis_model)
8. Unterstützt schnelles NLTK-Summary, wenn --analysis-model NLTK
9. Säubert Inhalte per Regex und stellt sauberes Beenden sicher
"""
import sys
import os
import termios
import atexit, signal
import time
import json
import re
import requests
import asyncio
import argparse
from urllib.parse import urlparse, unquote
import langid
import langcodes
from collections import deque
from langchain_core.runnables import RunnableSequence
from langchain_core.prompts import PromptTemplate
from langchain_ollama import OllamaLLM
from playwright.async_api import async_playwright
from newspaper import Article
import nltk
parser = argparse.ArgumentParser(...)
parser.add_argument('prompt' , help='The user’s search query or input text to be researched and analyzed.')
parser.add_argument('--analysis-model', default='mistral-small3.1:24b', help='The AI model to use for content analysis (e.g. "mistral:7b"), or "NTLK" to use a local Newspaper3k summary.')
parser.add_argument('--query-model' , default='mistral:latest', help='The LLM model to use for generating search queries (e.g. "mistral:7b").')
args = parser.parse_args()
user_input = args.prompt
analysis_model = args.analysis_model
query_model = args.query_model
use_nltk = analysis_model.strip().upper() == 'NTLK'
# Sauberes Beenden
t0_settings = None
try:
t0_settings = termios.tcgetattr(sys.stdin)
except:
pass
def safe_terminal_restore():
if t0_settings:
termios.tcsetattr(sys.stdin, termios.TCSADRAIN, t0_settings)
# ensure sane mode
try:
os.system('stty sane')
except:
pass
atexit.register(safe_terminal_restore)
for sig in (signal.SIGINT, signal.SIGTERM):
signal.signal(sig, lambda *args: (safe_terminal_restore(), sys.exit(0)))
# Punkt-Tokenizer sicherstellen
try:
nltk.data.find('tokenizers/punkt')
except LookupError:
nltk.download('punkt')
# Startzeit
t0 = time.time()
# Fix namespace
#user_input = parser.parse_args().prompt
#analysis_model = parser.parse_args().analysis_model
#query_model = parser.parse_args().query_model
# Flag für NLTK
#use_nltk = analysis_model.strip().upper() == 'NTLK'
# === 1) Query-Generierung via LLM ===
query_llm = OllamaLLM(model=query_model)
query_prompt = PromptTemplate(
input_variables=["input", "lang"],
template="""You are a search-query generator for online search engines.
Your task is to transform any user input into a list of precise, search-engine-optimized queries.
Follow these steps:
1. Analyze the user’s text and extract the core terms and topics. Correct potential typos.
2. Formulate 5 concise search queries (each 2–6 words) that provide optimal entry points for web research.
3. Vary the queries slightly to cover different aspects and phrasing (e.g., synonyms, related terms, specific detail questions).
4. Return the result as a JSON object under the key "queries", for example:
Example 1:
Input: "give me information about the betobeto-san yokai"
Output:
{{
"queries": [
"Betobeto-san Yokai origin",
"Betobeto-san Japanese spirit legend",
"Betobeto-san Yokai stories",
"Betobeto-san appearance description",
"Betobeto-san folklore sources"
]
}}
Example 2:
Input: "how does quantum entanglement work?"
Output:
{{
"queries": [
"quantum entanglement explanation",
"how quantum entanglement works physics",
"quantum entanglement experiments examples",
"applications of quantum entanglement",
"quantum entanglement vs theory"
]
}}
Example 3:
Input: "tips for a vegetarian diet"
Output:
{{
"queries": [
"vegetarian diet tips",
"healthy vegetarian recipes",
"vegetarian grocery list",
"protein sources for vegetarians",
"vegetarian meal plans"
]
}}
Example 4:
Input: "explain the law of large numbers"
Output:
{{
"queries": [
"law of large numbers explanation",
"statistics law of large numbers example",
"law of large numbers sampling mathematics",
"stochastic law of large numbers theorem",
"applications of law of large numbers"
]
}}
Example 5:
Input: "best time to visit the Lofoten"
Output:
{{
"queries": [
"best time to visit Lofoten weather",
"Lofoten climate by season",
"Lofoten northern lights months",
"Lofoten summer activities",
"Lofoten winter polar lights"
]
}}
Example 6:
Input: "JavaScript Promise vs. Callback"
Output:
{{
"queries": [
"JavaScript promise vs callback difference",
"JS promise callback examples",
"asynchronous JS promises callbacks comparison",
"JavaScript callbacks vs promises tutorial",
"best practices promises callbacks JS"
]
}}
Example 7:
Input: "origin of Valentine’s Day"
Output:
{{
"queries": [
"origin of Valentine’s Day history",
"Saint Valentine legend",
"Valentine’s Day customs evolution",
"historical sources Valentine’s Day",
"spread of Valentine’s Day traditions"
]
}}
It's VERY important that the queries are supposed to be in {lang}, so make sure that you answer in {lang}!
Now process the following user input and return **only** the JSON object with the field "queries":
{input}
"""
)
query_chain = RunnableSequence(query_prompt, query_llm)
def generate_search_queries_and_lang(user_prompt: str):
lang_code, _ = langid.classify(user_prompt)
# Maximiere das Tag, sodass wir auch eine Region bekommen
lang_obj = langcodes.Language.get(lang_code).maximize()
language = lang_obj.language # z. B. "de"
region = lang_obj.region or language.upper() # z. B. "DE"
locale_tag = f"{language}-{region}" # ergibt "de-DE"
# Anzeige-Name für das LLM
lang_display = lang_obj.display_name() # z. B. "German"
raw = query_chain.invoke({"input": user_prompt, "lang": lang_display})
data = json.loads(raw)
return data.get("queries", []), lang_display, locale_tag
# === 2) Analysis-Chain mit variablem Model ===
if not use_nltk:
analysis_llm = OllamaLLM(model=analysis_model)
analysis_prompt = PromptTemplate(
input_variables=["question", "content", "lang"],
template=(
"You are a multilingual assistant. Answer in {lang}.\n"
"Determine whether the following text contains information relevant to the question \"{question}\". "
"If it does, summarize only that information. Don't mention the text in your reply, just display the information about the question \"{question}\". "
"IMPORTANT: If it does NOT contain any relevant information, respond only with 'failed' and nothing else!\n"
"Here is the text:\n\n{content}\n"
"---\n"
"IMPORTANT: Answer in {lang}!\n"
"IMPORTANT: If it does NOT contain any relevant information or no answer to the question is given, or if the text does not mention anything about \"{question}\", respond only with 'failed' and nothing else! Don't talk about the text itself! \n"
)
)
analysis_chain = RunnableSequence(analysis_prompt, analysis_llm)
# === 3) SearXNG-Suche ===
def searx_search(query: str, max_results: int = 5) -> list:
resp = requests.get('http://127.0.0.1:8888/search', params={'q': query, 'format':'json'})
resp.raise_for_status()
return [ {'url': e.get('url',''), 'title': e.get('title',''), 'snippet': e.get('content','')}
for e in resp.json().get('results',[])[:max_results] ]
# === 4) Kandidatenliste ===
def canonicalize_url(u: str) -> str:
p = urlparse(u)
return f"{p.scheme or 'http'}://{p.netloc.lower()}{unquote(p.path).lower().rstrip('/')}"
def select_candidate_list(all_results: dict) -> list:
seen = set(); candidates = []
for lst in all_results.values():
for r in lst:
can = canonicalize_url(r['url'])
if can and can not in seen:
seen.add(can); candidates.append(can)
return candidates
# === 5) Playwright fetch ===
async def fetch_html_safe(url: str, locale: str, timeout: float = 30.0) -> str:
browser = None
try:
async with async_playwright() as p:
browser = await p.chromium.launch(headless=True,
args=['--disable-blink-features=AutomationControlled','--no-sandbox','--disable-dev-shm-usage'])
ctx = await browser.new_context(
user_agent='Mozilla/5.0',
locale=locale, # Hier nutzen wir das übergebene locale
timezone_id='UTC'
)
page = await ctx.new_page()
resp = await page.goto(url, wait_until='domcontentloaded', timeout=timeout*1000)
if not resp or resp.status != 200:
raise RuntimeError(f'HTTP {resp.status}')
await page.evaluate("""() => {document.querySelectorAll('.cookie-notice, .cookie-banner, .cookie-consent').forEach(el=>el.remove());}""")
html = await page.content()
await ctx.close(); await browser.close()
if len(html) < 2000: raise RuntimeError('HTML too short')
return html
except Exception:
if browser:
try: await browser.close()
except: pass
raise
# === 6) Text-Extraktion ===
def extract_text(html: str, url:str) -> str:
art = Article(url); art.download(input_html=html); art.parse()
return (art.text or '').strip()
# === 7) Runner mit flexibler Analyse ===
def runner(candidates, question, lang, locale, max_out=5, max_concurrent=5):
async def _run():
loop = asyncio.get_running_loop()
loop.set_exception_handler(lambda l,c: None)
queue = deque(candidates)
count = 0
running = set()
def is_failed(ans: str) -> bool:
return bool(re.search(r'(^|\W)failed(\W|$)', ans, re.IGNORECASE))
async def proc(url: str):
nonlocal count
# fetch, extract, analyse...
try:
html = await asyncio.wait_for(fetch_html_safe(url, locale), timeout=20)
txt = extract_text(html, url)
clean = re.sub(r'https?://\S+', '', txt)
if use_nltk:
art = Article(url); art.download(input_html=html); art.parse(); art.nlp()
ans = art.summary or ''
else:
# offload the blocking LLM call
ans = await asyncio.to_thread(
analysis_chain.invoke,
{'question': question, 'content': clean, 'lang': lang}
)
if is_failed(ans):
raise RuntimeError('No information after analysis')
print(json.dumps({'url': url, 'content': clean, 'analysis': ans}, ensure_ascii=False))
count += 1
return True
except Exception as e:
print(f"Skipped {url}: {type(e).__name__} – {repr(e)}", file=sys.stderr)
return False
# helper to spawn next task if needed
def try_spawn():
if queue and count < max_out and len(running) < max_concurrent:
next_url = queue.popleft()
task = asyncio.create_task(proc(next_url))
running.add(task)
# initially fill up to concurrency
for _ in range(min(max_concurrent, len(queue))):
try_spawn()
# process tasks as they complete
while running:
done, running = await asyncio.wait(running, return_when=asyncio.FIRST_COMPLETED)
for task in done:
success = await task
# as soon as we have enough results, cancel everything
if count >= max_out:
for t in running:
t.cancel()
await asyncio.gather(*running, return_exceptions=True)
return
# otherwise, try to spawn one more
try_spawn()
try:
asyncio.run(_run())
finally:
safe_terminal_restore()
# === 8) Main ===
if __name__ == '__main__':
queries, lang, browser_locale = generate_search_queries_and_lang(user_input)
results = {q: searx_search(q) for q in queries}
candidates = select_candidate_list(results)
runner(candidates, user_input, lang, browser_locale)
print(f"\n/* Total Runtime: {time.time()-t0:.2f} s */")
sys.exit(0)