357 lines
13 KiB
Python
357 lines
13 KiB
Python
#!/usr/bin/env python3
|
||
# -*- coding: utf-8 -*-
|
||
"""
|
||
Kombiniertes Skript mit dynamischer Fallback-Logik:
|
||
1. Generiert 5 Suchanfragen via LLM (query_model)
|
||
2. Sucht per SearXNG je 5 Treffer pro Query
|
||
3. Entfernt Dubletten und generiert einfache Kandidatenliste
|
||
4. Lädt Kandidaten parallel (max 5) mit Playwright, rendert JS, Timeout
|
||
5. Extrahiert Text via Newspaper3k (ohne Zusammenfassung)
|
||
6. Fallback: Bei Fehlschlag oder "failed" vom LLM wird die nächste URL geladen
|
||
7. Gibt für jede der ersten 5 validen Seiten sofort ein JSON-Objekt aus und startet parallel eine AI-Analyse (analysis_model)
|
||
8. Unterstützt schnelles NLTK-Summary, wenn --analysis-model NLTK
|
||
9. Säubert Inhalte per Regex und stellt sauberes Beenden sicher
|
||
"""
|
||
import sys
|
||
import os
|
||
import termios
|
||
import atexit, signal
|
||
import time
|
||
import json
|
||
import re
|
||
import requests
|
||
import asyncio
|
||
import argparse
|
||
from urllib.parse import urlparse, unquote
|
||
import langid
|
||
import langcodes
|
||
from collections import deque
|
||
from langchain_core.runnables import RunnableSequence
|
||
from langchain_core.prompts import PromptTemplate
|
||
from langchain_ollama import OllamaLLM
|
||
from playwright.async_api import async_playwright
|
||
from newspaper import Article
|
||
import nltk
|
||
|
||
parser = argparse.ArgumentParser(...)
|
||
parser.add_argument('prompt' , help='The user’s search query or input text to be researched and analyzed.')
|
||
parser.add_argument('--analysis-model', default='mistral-small3.1:24b', help='The AI model to use for content analysis (e.g. "mistral:7b"), or "NTLK" to use a local Newspaper3k summary.')
|
||
parser.add_argument('--query-model' , default='mistral:latest', help='The LLM model to use for generating search queries (e.g. "mistral:7b").')
|
||
args = parser.parse_args()
|
||
|
||
user_input = args.prompt
|
||
analysis_model = args.analysis_model
|
||
query_model = args.query_model
|
||
use_nltk = analysis_model.strip().upper() == 'NTLK'
|
||
|
||
# Sauberes Beenden
|
||
|
||
t0_settings = None
|
||
try:
|
||
t0_settings = termios.tcgetattr(sys.stdin)
|
||
except:
|
||
pass
|
||
|
||
def safe_terminal_restore():
|
||
if t0_settings:
|
||
termios.tcsetattr(sys.stdin, termios.TCSADRAIN, t0_settings)
|
||
# ensure sane mode
|
||
try:
|
||
os.system('stty sane')
|
||
except:
|
||
pass
|
||
|
||
atexit.register(safe_terminal_restore)
|
||
for sig in (signal.SIGINT, signal.SIGTERM):
|
||
signal.signal(sig, lambda *args: (safe_terminal_restore(), sys.exit(0)))
|
||
|
||
# Punkt-Tokenizer sicherstellen
|
||
try:
|
||
nltk.data.find('tokenizers/punkt')
|
||
except LookupError:
|
||
nltk.download('punkt')
|
||
|
||
# Startzeit
|
||
t0 = time.time()
|
||
|
||
|
||
# Fix namespace
|
||
#user_input = parser.parse_args().prompt
|
||
#analysis_model = parser.parse_args().analysis_model
|
||
#query_model = parser.parse_args().query_model
|
||
# Flag für NLTK
|
||
#use_nltk = analysis_model.strip().upper() == 'NTLK'
|
||
|
||
# === 1) Query-Generierung via LLM ===
|
||
query_llm = OllamaLLM(model=query_model)
|
||
query_prompt = PromptTemplate(
|
||
input_variables=["input", "lang"],
|
||
template="""You are a search-query generator for online search engines.
|
||
Your task is to transform any user input into a list of precise, search-engine-optimized queries.
|
||
Follow these steps:
|
||
|
||
1. Analyze the user’s text and extract the core terms and topics. Correct potential typos.
|
||
2. Formulate 5 concise search queries (each 2–6 words) that provide optimal entry points for web research.
|
||
3. Vary the queries slightly to cover different aspects and phrasing (e.g., synonyms, related terms, specific detail questions).
|
||
4. Return the result as a JSON object under the key "queries", for example:
|
||
|
||
Example 1:
|
||
Input: "give me information about the betobeto-san yokai"
|
||
Output:
|
||
{{
|
||
"queries": [
|
||
"Betobeto-san Yokai origin",
|
||
"Betobeto-san Japanese spirit legend",
|
||
"Betobeto-san Yokai stories",
|
||
"Betobeto-san appearance description",
|
||
"Betobeto-san folklore sources"
|
||
]
|
||
}}
|
||
|
||
Example 2:
|
||
Input: "how does quantum entanglement work?"
|
||
Output:
|
||
{{
|
||
"queries": [
|
||
"quantum entanglement explanation",
|
||
"how quantum entanglement works physics",
|
||
"quantum entanglement experiments examples",
|
||
"applications of quantum entanglement",
|
||
"quantum entanglement vs theory"
|
||
]
|
||
}}
|
||
|
||
Example 3:
|
||
Input: "tips for a vegetarian diet"
|
||
Output:
|
||
{{
|
||
"queries": [
|
||
"vegetarian diet tips",
|
||
"healthy vegetarian recipes",
|
||
"vegetarian grocery list",
|
||
"protein sources for vegetarians",
|
||
"vegetarian meal plans"
|
||
]
|
||
}}
|
||
|
||
Example 4:
|
||
Input: "explain the law of large numbers"
|
||
Output:
|
||
{{
|
||
"queries": [
|
||
"law of large numbers explanation",
|
||
"statistics law of large numbers example",
|
||
"law of large numbers sampling mathematics",
|
||
"stochastic law of large numbers theorem",
|
||
"applications of law of large numbers"
|
||
]
|
||
}}
|
||
|
||
Example 5:
|
||
Input: "best time to visit the Lofoten"
|
||
Output:
|
||
{{
|
||
"queries": [
|
||
"best time to visit Lofoten weather",
|
||
"Lofoten climate by season",
|
||
"Lofoten northern lights months",
|
||
"Lofoten summer activities",
|
||
"Lofoten winter polar lights"
|
||
]
|
||
}}
|
||
|
||
Example 6:
|
||
Input: "JavaScript Promise vs. Callback"
|
||
Output:
|
||
{{
|
||
"queries": [
|
||
"JavaScript promise vs callback difference",
|
||
"JS promise callback examples",
|
||
"asynchronous JS promises callbacks comparison",
|
||
"JavaScript callbacks vs promises tutorial",
|
||
"best practices promises callbacks JS"
|
||
]
|
||
}}
|
||
|
||
Example 7:
|
||
Input: "origin of Valentine’s Day"
|
||
Output:
|
||
{{
|
||
"queries": [
|
||
"origin of Valentine’s Day history",
|
||
"Saint Valentine legend",
|
||
"Valentine’s Day customs evolution",
|
||
"historical sources Valentine’s Day",
|
||
"spread of Valentine’s Day traditions"
|
||
]
|
||
}}
|
||
|
||
It's VERY important that the queries are supposed to be in {lang}, so make sure that you answer in {lang}!
|
||
Now process the following user input and return **only** the JSON object with the field "queries":
|
||
{input}
|
||
"""
|
||
)
|
||
query_chain = RunnableSequence(query_prompt, query_llm)
|
||
|
||
def generate_search_queries_and_lang(user_prompt: str):
|
||
lang_code, _ = langid.classify(user_prompt)
|
||
# Maximiere das Tag, sodass wir auch eine Region bekommen
|
||
lang_obj = langcodes.Language.get(lang_code).maximize()
|
||
language = lang_obj.language # z. B. "de"
|
||
region = lang_obj.region or language.upper() # z. B. "DE"
|
||
locale_tag = f"{language}-{region}" # ergibt "de-DE"
|
||
# Anzeige-Name für das LLM
|
||
lang_display = lang_obj.display_name() # z. B. "German"
|
||
raw = query_chain.invoke({"input": user_prompt, "lang": lang_display})
|
||
data = json.loads(raw)
|
||
return data.get("queries", []), lang_display, locale_tag
|
||
|
||
# === 2) Analysis-Chain mit variablem Model ===
|
||
if not use_nltk:
|
||
analysis_llm = OllamaLLM(model=analysis_model)
|
||
analysis_prompt = PromptTemplate(
|
||
input_variables=["question", "content", "lang"],
|
||
template=(
|
||
"You are a multilingual assistant. Answer in {lang}.\n"
|
||
"Determine whether the following text contains information relevant to the question \"{question}\". "
|
||
"If it does, summarize only that information. Don't mention the text in your reply, just display the information about the question \"{question}\". "
|
||
"IMPORTANT: If it does NOT contain any relevant information, respond only with 'failed' and nothing else!\n"
|
||
"Here is the text:\n\n{content}\n"
|
||
"---\n"
|
||
"IMPORTANT: Answer in {lang}!\n"
|
||
"IMPORTANT: If it does NOT contain any relevant information or no answer to the question is given, or if the text does not mention anything about \"{question}\", respond only with 'failed' and nothing else! Don't talk about the text itself! \n"
|
||
)
|
||
)
|
||
analysis_chain = RunnableSequence(analysis_prompt, analysis_llm)
|
||
|
||
# === 3) SearXNG-Suche ===
|
||
def searx_search(query: str, max_results: int = 5) -> list:
|
||
resp = requests.get('http://127.0.0.1:8888/search', params={'q': query, 'format':'json'})
|
||
resp.raise_for_status()
|
||
return [ {'url': e.get('url',''), 'title': e.get('title',''), 'snippet': e.get('content','')}
|
||
for e in resp.json().get('results',[])[:max_results] ]
|
||
|
||
# === 4) Kandidatenliste ===
|
||
def canonicalize_url(u: str) -> str:
|
||
p = urlparse(u)
|
||
return f"{p.scheme or 'http'}://{p.netloc.lower()}{unquote(p.path).lower().rstrip('/')}"
|
||
|
||
def select_candidate_list(all_results: dict) -> list:
|
||
seen = set(); candidates = []
|
||
for lst in all_results.values():
|
||
for r in lst:
|
||
can = canonicalize_url(r['url'])
|
||
if can and can not in seen:
|
||
seen.add(can); candidates.append(can)
|
||
return candidates
|
||
|
||
# === 5) Playwright fetch ===
|
||
async def fetch_html_safe(url: str, locale: str, timeout: float = 30.0) -> str:
|
||
browser = None
|
||
try:
|
||
async with async_playwright() as p:
|
||
browser = await p.chromium.launch(headless=True,
|
||
args=['--disable-blink-features=AutomationControlled','--no-sandbox','--disable-dev-shm-usage'])
|
||
ctx = await browser.new_context(
|
||
user_agent='Mozilla/5.0',
|
||
locale=locale, # Hier nutzen wir das übergebene locale
|
||
timezone_id='UTC'
|
||
)
|
||
page = await ctx.new_page()
|
||
resp = await page.goto(url, wait_until='domcontentloaded', timeout=timeout*1000)
|
||
if not resp or resp.status != 200:
|
||
raise RuntimeError(f'HTTP {resp.status}')
|
||
await page.evaluate("""() => {document.querySelectorAll('.cookie-notice, .cookie-banner, .cookie-consent').forEach(el=>el.remove());}""")
|
||
html = await page.content()
|
||
await ctx.close(); await browser.close()
|
||
if len(html) < 2000: raise RuntimeError('HTML too short')
|
||
return html
|
||
except Exception:
|
||
if browser:
|
||
try: await browser.close()
|
||
except: pass
|
||
raise
|
||
|
||
# === 6) Text-Extraktion ===
|
||
def extract_text(html: str, url:str) -> str:
|
||
art = Article(url); art.download(input_html=html); art.parse()
|
||
return (art.text or '').strip()
|
||
|
||
# === 7) Runner mit flexibler Analyse ===
|
||
def runner(candidates, question, lang, locale, max_out=5, max_concurrent=5):
|
||
async def _run():
|
||
loop = asyncio.get_running_loop()
|
||
loop.set_exception_handler(lambda l,c: None)
|
||
|
||
queue = deque(candidates)
|
||
count = 0
|
||
running = set()
|
||
|
||
def is_failed(ans: str) -> bool:
|
||
return bool(re.search(r'(^|\W)failed(\W|$)', ans, re.IGNORECASE))
|
||
|
||
async def proc(url: str):
|
||
nonlocal count
|
||
# fetch, extract, analyse...
|
||
try:
|
||
html = await asyncio.wait_for(fetch_html_safe(url, locale), timeout=20)
|
||
txt = extract_text(html, url)
|
||
clean = re.sub(r'https?://\S+', '', txt)
|
||
|
||
if use_nltk:
|
||
art = Article(url); art.download(input_html=html); art.parse(); art.nlp()
|
||
ans = art.summary or ''
|
||
else:
|
||
# offload the blocking LLM call
|
||
ans = await asyncio.to_thread(
|
||
analysis_chain.invoke,
|
||
{'question': question, 'content': clean, 'lang': lang}
|
||
)
|
||
if is_failed(ans):
|
||
raise RuntimeError('No information after analysis')
|
||
|
||
print(json.dumps({'url': url, 'content': clean, 'analysis': ans}, ensure_ascii=False))
|
||
count += 1
|
||
return True
|
||
except Exception as e:
|
||
print(f"Skipped {url}: {type(e).__name__} – {repr(e)}", file=sys.stderr)
|
||
return False
|
||
|
||
# helper to spawn next task if needed
|
||
def try_spawn():
|
||
if queue and count < max_out and len(running) < max_concurrent:
|
||
next_url = queue.popleft()
|
||
task = asyncio.create_task(proc(next_url))
|
||
running.add(task)
|
||
|
||
# initially fill up to concurrency
|
||
for _ in range(min(max_concurrent, len(queue))):
|
||
try_spawn()
|
||
|
||
# process tasks as they complete
|
||
while running:
|
||
done, running = await asyncio.wait(running, return_when=asyncio.FIRST_COMPLETED)
|
||
for task in done:
|
||
success = await task
|
||
# as soon as we have enough results, cancel everything
|
||
if count >= max_out:
|
||
for t in running:
|
||
t.cancel()
|
||
await asyncio.gather(*running, return_exceptions=True)
|
||
return
|
||
# otherwise, try to spawn one more
|
||
try_spawn()
|
||
|
||
try:
|
||
asyncio.run(_run())
|
||
finally:
|
||
safe_terminal_restore()
|
||
|
||
# === 8) Main ===
|
||
if __name__ == '__main__':
|
||
queries, lang, browser_locale = generate_search_queries_and_lang(user_input)
|
||
results = {q: searx_search(q) for q in queries}
|
||
candidates = select_candidate_list(results)
|
||
runner(candidates, user_input, lang, browser_locale)
|
||
print(f"\n/* Total Runtime: {time.time()-t0:.2f} s */")
|
||
sys.exit(0) |