initial commit
This commit is contained in:
5
.gitignore
vendored
Normal file
5
.gitignore
vendored
Normal file
@@ -0,0 +1,5 @@
|
||||
.git/
|
||||
node_modules/
|
||||
.DS_Store
|
||||
*.log
|
||||
.env
|
||||
58
README.md
Normal file
58
README.md
Normal file
@@ -0,0 +1,58 @@
|
||||
# search-and-analyze
|
||||
|
||||
**Author:** Victor Giers
|
||||
|
||||
# search-and-analyze
|
||||
|
||||
**Author:** Victor Giers
|
||||
|
||||
## Overview
|
||||
|
||||
This project contains a Python script that automates the process of generating search queries, performing web searches, extracting and analyzing content from web pages, and providing relevant information based on user input. The script uses advanced language models for query generation and analysis, ensures efficient web scraping with Playwright, and supports various configurations for different AI models.
|
||||
|
||||
## Features
|
||||
|
||||
1. **Query Generation:** Generates 5 optimized search queries based on user input using a specified LLM model.
|
||||
2. **Web Search:** Searches the web using SearXNG to find relevant URLs.
|
||||
3. **Content Extraction:** Extracts content from web pages, rendering JavaScript if necessary.
|
||||
4. **Analysis:** Analyzes extracted text for relevance and provides summaries or specific information based on user queries.
|
||||
5. **Fallback Mechanism:** Handles failures in data extraction or analysis by moving to the next URL.
|
||||
6. **Parallel Processing:** Loads and analyzes up to 5 URLs concurrently.
|
||||
7. **Language Support:** Automatically detects and supports multiple languages using `langid` and `langcodes`.
|
||||
8. **NLTK Integration:** Supports NLTK for text summarization as an alternative to external LLM models.
|
||||
|
||||
## Requirements
|
||||
|
||||
- Python 3.x
|
||||
- Required Python packages: `requests`, `asyncio`, `argparse`, `urllib`, `json`, `re`, `termios`, `atexit`, `signal`, `time`, `langid`, `langcodes`, `collections`, `newspaper3k`, `nltk`, `playwright`, `langchain_core`, `langchain_ollama`
|
||||
- SearXNG running locally at `http://127.0.0.1:8888`
|
||||
|
||||
## Installation
|
||||
|
||||
```bash
|
||||
pip install requests asyncio argparse urllib json re termios atexit signal time langid langcodes collections newspaper3k nltk playwright langchain_core langchain_ollama
|
||||
playwright install chromium
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
Run the script with a user prompt and optional arguments for analysis and query models.
|
||||
|
||||
```bash
|
||||
python search-and-analyze.py "your search query" --analysis-model mistral-small3.1:24b --query-model mistral:latest
|
||||
```
|
||||
|
||||
- `--analysis-model`: Specifies the AI model to use for content analysis (e.g., `mistral-small3.1:24b`). Use `NLTK` for a local summary using Newspaper3k.
|
||||
- `--query-model`: Specifies the LLM model to use for generating search queries (default is `mistral:latest`).
|
||||
|
||||
## Example
|
||||
|
||||
```bash
|
||||
python search-and-analyze.py "origin of Valentine’s Day" --analysis-model NLTK
|
||||
```
|
||||
|
||||
This command will generate search queries related to the origin of Valentine's Day, perform web searches, extract and summarize content from relevant pages, and output the results.
|
||||
|
||||
## License
|
||||
|
||||
This project is licensed under the MIT License - see the [LICENSE](LICENSE) file for details.
|
||||
357
search-and-analyze.py
Normal file
357
search-and-analyze.py
Normal file
@@ -0,0 +1,357 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""
|
||||
Kombiniertes Skript mit dynamischer Fallback-Logik:
|
||||
1. Generiert 5 Suchanfragen via LLM (query_model)
|
||||
2. Sucht per SearXNG je 5 Treffer pro Query
|
||||
3. Entfernt Dubletten und generiert einfache Kandidatenliste
|
||||
4. Lädt Kandidaten parallel (max 5) mit Playwright, rendert JS, Timeout
|
||||
5. Extrahiert Text via Newspaper3k (ohne Zusammenfassung)
|
||||
6. Fallback: Bei Fehlschlag oder "failed" vom LLM wird die nächste URL geladen
|
||||
7. Gibt für jede der ersten 5 validen Seiten sofort ein JSON-Objekt aus und startet parallel eine AI-Analyse (analysis_model)
|
||||
8. Unterstützt schnelles NLTK-Summary, wenn --analysis-model NLTK
|
||||
9. Säubert Inhalte per Regex und stellt sauberes Beenden sicher
|
||||
"""
|
||||
import sys
|
||||
import os
|
||||
import termios
|
||||
import atexit, signal
|
||||
import time
|
||||
import json
|
||||
import re
|
||||
import requests
|
||||
import asyncio
|
||||
import argparse
|
||||
from urllib.parse import urlparse, unquote
|
||||
import langid
|
||||
import langcodes
|
||||
from collections import deque
|
||||
from langchain_core.runnables import RunnableSequence
|
||||
from langchain_core.prompts import PromptTemplate
|
||||
from langchain_ollama import OllamaLLM
|
||||
from playwright.async_api import async_playwright
|
||||
from newspaper import Article
|
||||
import nltk
|
||||
|
||||
parser = argparse.ArgumentParser(...)
|
||||
parser.add_argument('prompt' , help='The user’s search query or input text to be researched and analyzed.')
|
||||
parser.add_argument('--analysis-model', default='mistral-small3.1:24b', help='The AI model to use for content analysis (e.g. "mistral:7b"), or "NTLK" to use a local Newspaper3k summary.')
|
||||
parser.add_argument('--query-model' , default='mistral:latest', help='The LLM model to use for generating search queries (e.g. "mistral:7b").')
|
||||
args = parser.parse_args()
|
||||
|
||||
user_input = args.prompt
|
||||
analysis_model = args.analysis_model
|
||||
query_model = args.query_model
|
||||
use_nltk = analysis_model.strip().upper() == 'NTLK'
|
||||
|
||||
# Sauberes Beenden
|
||||
|
||||
t0_settings = None
|
||||
try:
|
||||
t0_settings = termios.tcgetattr(sys.stdin)
|
||||
except:
|
||||
pass
|
||||
|
||||
def safe_terminal_restore():
|
||||
if t0_settings:
|
||||
termios.tcsetattr(sys.stdin, termios.TCSADRAIN, t0_settings)
|
||||
# ensure sane mode
|
||||
try:
|
||||
os.system('stty sane')
|
||||
except:
|
||||
pass
|
||||
|
||||
atexit.register(safe_terminal_restore)
|
||||
for sig in (signal.SIGINT, signal.SIGTERM):
|
||||
signal.signal(sig, lambda *args: (safe_terminal_restore(), sys.exit(0)))
|
||||
|
||||
# Punkt-Tokenizer sicherstellen
|
||||
try:
|
||||
nltk.data.find('tokenizers/punkt')
|
||||
except LookupError:
|
||||
nltk.download('punkt')
|
||||
|
||||
# Startzeit
|
||||
t0 = time.time()
|
||||
|
||||
|
||||
# Fix namespace
|
||||
#user_input = parser.parse_args().prompt
|
||||
#analysis_model = parser.parse_args().analysis_model
|
||||
#query_model = parser.parse_args().query_model
|
||||
# Flag für NLTK
|
||||
#use_nltk = analysis_model.strip().upper() == 'NTLK'
|
||||
|
||||
# === 1) Query-Generierung via LLM ===
|
||||
query_llm = OllamaLLM(model=query_model)
|
||||
query_prompt = PromptTemplate(
|
||||
input_variables=["input", "lang"],
|
||||
template="""You are a search-query generator for online search engines.
|
||||
Your task is to transform any user input into a list of precise, search-engine-optimized queries.
|
||||
Follow these steps:
|
||||
|
||||
1. Analyze the user’s text and extract the core terms and topics. Correct potential typos.
|
||||
2. Formulate 5 concise search queries (each 2–6 words) that provide optimal entry points for web research.
|
||||
3. Vary the queries slightly to cover different aspects and phrasing (e.g., synonyms, related terms, specific detail questions).
|
||||
4. Return the result as a JSON object under the key "queries", for example:
|
||||
|
||||
Example 1:
|
||||
Input: "give me information about the betobeto-san yokai"
|
||||
Output:
|
||||
{{
|
||||
"queries": [
|
||||
"Betobeto-san Yokai origin",
|
||||
"Betobeto-san Japanese spirit legend",
|
||||
"Betobeto-san Yokai stories",
|
||||
"Betobeto-san appearance description",
|
||||
"Betobeto-san folklore sources"
|
||||
]
|
||||
}}
|
||||
|
||||
Example 2:
|
||||
Input: "how does quantum entanglement work?"
|
||||
Output:
|
||||
{{
|
||||
"queries": [
|
||||
"quantum entanglement explanation",
|
||||
"how quantum entanglement works physics",
|
||||
"quantum entanglement experiments examples",
|
||||
"applications of quantum entanglement",
|
||||
"quantum entanglement vs theory"
|
||||
]
|
||||
}}
|
||||
|
||||
Example 3:
|
||||
Input: "tips for a vegetarian diet"
|
||||
Output:
|
||||
{{
|
||||
"queries": [
|
||||
"vegetarian diet tips",
|
||||
"healthy vegetarian recipes",
|
||||
"vegetarian grocery list",
|
||||
"protein sources for vegetarians",
|
||||
"vegetarian meal plans"
|
||||
]
|
||||
}}
|
||||
|
||||
Example 4:
|
||||
Input: "explain the law of large numbers"
|
||||
Output:
|
||||
{{
|
||||
"queries": [
|
||||
"law of large numbers explanation",
|
||||
"statistics law of large numbers example",
|
||||
"law of large numbers sampling mathematics",
|
||||
"stochastic law of large numbers theorem",
|
||||
"applications of law of large numbers"
|
||||
]
|
||||
}}
|
||||
|
||||
Example 5:
|
||||
Input: "best time to visit the Lofoten"
|
||||
Output:
|
||||
{{
|
||||
"queries": [
|
||||
"best time to visit Lofoten weather",
|
||||
"Lofoten climate by season",
|
||||
"Lofoten northern lights months",
|
||||
"Lofoten summer activities",
|
||||
"Lofoten winter polar lights"
|
||||
]
|
||||
}}
|
||||
|
||||
Example 6:
|
||||
Input: "JavaScript Promise vs. Callback"
|
||||
Output:
|
||||
{{
|
||||
"queries": [
|
||||
"JavaScript promise vs callback difference",
|
||||
"JS promise callback examples",
|
||||
"asynchronous JS promises callbacks comparison",
|
||||
"JavaScript callbacks vs promises tutorial",
|
||||
"best practices promises callbacks JS"
|
||||
]
|
||||
}}
|
||||
|
||||
Example 7:
|
||||
Input: "origin of Valentine’s Day"
|
||||
Output:
|
||||
{{
|
||||
"queries": [
|
||||
"origin of Valentine’s Day history",
|
||||
"Saint Valentine legend",
|
||||
"Valentine’s Day customs evolution",
|
||||
"historical sources Valentine’s Day",
|
||||
"spread of Valentine’s Day traditions"
|
||||
]
|
||||
}}
|
||||
|
||||
It's VERY important that the queries are supposed to be in {lang}, so make sure that you answer in {lang}!
|
||||
Now process the following user input and return **only** the JSON object with the field "queries":
|
||||
{input}
|
||||
"""
|
||||
)
|
||||
query_chain = RunnableSequence(query_prompt, query_llm)
|
||||
|
||||
def generate_search_queries_and_lang(user_prompt: str):
|
||||
lang_code, _ = langid.classify(user_prompt)
|
||||
# Maximiere das Tag, sodass wir auch eine Region bekommen
|
||||
lang_obj = langcodes.Language.get(lang_code).maximize()
|
||||
language = lang_obj.language # z. B. "de"
|
||||
region = lang_obj.region or language.upper() # z. B. "DE"
|
||||
locale_tag = f"{language}-{region}" # ergibt "de-DE"
|
||||
# Anzeige-Name für das LLM
|
||||
lang_display = lang_obj.display_name() # z. B. "German"
|
||||
raw = query_chain.invoke({"input": user_prompt, "lang": lang_display})
|
||||
data = json.loads(raw)
|
||||
return data.get("queries", []), lang_display, locale_tag
|
||||
|
||||
# === 2) Analysis-Chain mit variablem Model ===
|
||||
if not use_nltk:
|
||||
analysis_llm = OllamaLLM(model=analysis_model)
|
||||
analysis_prompt = PromptTemplate(
|
||||
input_variables=["question", "content", "lang"],
|
||||
template=(
|
||||
"You are a multilingual assistant. Answer in {lang}.\n"
|
||||
"Determine whether the following text contains information relevant to the question \"{question}\". "
|
||||
"If it does, summarize only that information. Don't mention the text in your reply, just display the information about the question \"{question}\". "
|
||||
"IMPORTANT: If it does NOT contain any relevant information, respond only with 'failed' and nothing else!\n"
|
||||
"Here is the text:\n\n{content}\n"
|
||||
"---\n"
|
||||
"IMPORTANT: Answer in {lang}!\n"
|
||||
"IMPORTANT: If it does NOT contain any relevant information or no answer to the question is given, or if the text does not mention anything about \"{question}\", respond only with 'failed' and nothing else! Don't talk about the text itself! \n"
|
||||
)
|
||||
)
|
||||
analysis_chain = RunnableSequence(analysis_prompt, analysis_llm)
|
||||
|
||||
# === 3) SearXNG-Suche ===
|
||||
def searx_search(query: str, max_results: int = 5) -> list:
|
||||
resp = requests.get('http://127.0.0.1:8888/search', params={'q': query, 'format':'json'})
|
||||
resp.raise_for_status()
|
||||
return [ {'url': e.get('url',''), 'title': e.get('title',''), 'snippet': e.get('content','')}
|
||||
for e in resp.json().get('results',[])[:max_results] ]
|
||||
|
||||
# === 4) Kandidatenliste ===
|
||||
def canonicalize_url(u: str) -> str:
|
||||
p = urlparse(u)
|
||||
return f"{p.scheme or 'http'}://{p.netloc.lower()}{unquote(p.path).lower().rstrip('/')}"
|
||||
|
||||
def select_candidate_list(all_results: dict) -> list:
|
||||
seen = set(); candidates = []
|
||||
for lst in all_results.values():
|
||||
for r in lst:
|
||||
can = canonicalize_url(r['url'])
|
||||
if can and can not in seen:
|
||||
seen.add(can); candidates.append(can)
|
||||
return candidates
|
||||
|
||||
# === 5) Playwright fetch ===
|
||||
async def fetch_html_safe(url: str, locale: str, timeout: float = 30.0) -> str:
|
||||
browser = None
|
||||
try:
|
||||
async with async_playwright() as p:
|
||||
browser = await p.chromium.launch(headless=True,
|
||||
args=['--disable-blink-features=AutomationControlled','--no-sandbox','--disable-dev-shm-usage'])
|
||||
ctx = await browser.new_context(
|
||||
user_agent='Mozilla/5.0',
|
||||
locale=locale, # Hier nutzen wir das übergebene locale
|
||||
timezone_id='UTC'
|
||||
)
|
||||
page = await ctx.new_page()
|
||||
resp = await page.goto(url, wait_until='domcontentloaded', timeout=timeout*1000)
|
||||
if not resp or resp.status != 200:
|
||||
raise RuntimeError(f'HTTP {resp.status}')
|
||||
await page.evaluate("""() => {document.querySelectorAll('.cookie-notice, .cookie-banner, .cookie-consent').forEach(el=>el.remove());}""")
|
||||
html = await page.content()
|
||||
await ctx.close(); await browser.close()
|
||||
if len(html) < 2000: raise RuntimeError('HTML too short')
|
||||
return html
|
||||
except Exception:
|
||||
if browser:
|
||||
try: await browser.close()
|
||||
except: pass
|
||||
raise
|
||||
|
||||
# === 6) Text-Extraktion ===
|
||||
def extract_text(html: str, url:str) -> str:
|
||||
art = Article(url); art.download(input_html=html); art.parse()
|
||||
return (art.text or '').strip()
|
||||
|
||||
# === 7) Runner mit flexibler Analyse ===
|
||||
def runner(candidates, question, lang, locale, max_out=5, max_concurrent=5):
|
||||
async def _run():
|
||||
loop = asyncio.get_running_loop()
|
||||
loop.set_exception_handler(lambda l,c: None)
|
||||
|
||||
queue = deque(candidates)
|
||||
count = 0
|
||||
running = set()
|
||||
|
||||
def is_failed(ans: str) -> bool:
|
||||
return bool(re.search(r'(^|\W)failed(\W|$)', ans, re.IGNORECASE))
|
||||
|
||||
async def proc(url: str):
|
||||
nonlocal count
|
||||
# fetch, extract, analyse...
|
||||
try:
|
||||
html = await asyncio.wait_for(fetch_html_safe(url, locale), timeout=20)
|
||||
txt = extract_text(html, url)
|
||||
clean = re.sub(r'https?://\S+', '', txt)
|
||||
|
||||
if use_nltk:
|
||||
art = Article(url); art.download(input_html=html); art.parse(); art.nlp()
|
||||
ans = art.summary or ''
|
||||
else:
|
||||
# offload the blocking LLM call
|
||||
ans = await asyncio.to_thread(
|
||||
analysis_chain.invoke,
|
||||
{'question': question, 'content': clean, 'lang': lang}
|
||||
)
|
||||
if is_failed(ans):
|
||||
raise RuntimeError('No information after analysis')
|
||||
|
||||
print(json.dumps({'url': url, 'content': clean, 'analysis': ans}, ensure_ascii=False))
|
||||
count += 1
|
||||
return True
|
||||
except Exception as e:
|
||||
print(f"Skipped {url}: {type(e).__name__} – {repr(e)}", file=sys.stderr)
|
||||
return False
|
||||
|
||||
# helper to spawn next task if needed
|
||||
def try_spawn():
|
||||
if queue and count < max_out and len(running) < max_concurrent:
|
||||
next_url = queue.popleft()
|
||||
task = asyncio.create_task(proc(next_url))
|
||||
running.add(task)
|
||||
|
||||
# initially fill up to concurrency
|
||||
for _ in range(min(max_concurrent, len(queue))):
|
||||
try_spawn()
|
||||
|
||||
# process tasks as they complete
|
||||
while running:
|
||||
done, running = await asyncio.wait(running, return_when=asyncio.FIRST_COMPLETED)
|
||||
for task in done:
|
||||
success = await task
|
||||
# as soon as we have enough results, cancel everything
|
||||
if count >= max_out:
|
||||
for t in running:
|
||||
t.cancel()
|
||||
await asyncio.gather(*running, return_exceptions=True)
|
||||
return
|
||||
# otherwise, try to spawn one more
|
||||
try_spawn()
|
||||
|
||||
try:
|
||||
asyncio.run(_run())
|
||||
finally:
|
||||
safe_terminal_restore()
|
||||
|
||||
# === 8) Main ===
|
||||
if __name__ == '__main__':
|
||||
queries, lang, browser_locale = generate_search_queries_and_lang(user_input)
|
||||
results = {q: searx_search(q) for q in queries}
|
||||
candidates = select_candidate_list(results)
|
||||
runner(candidates, user_input, lang, browser_locale)
|
||||
print(f"\n/* Total Runtime: {time.time()-t0:.2f} s */")
|
||||
sys.exit(0)
|
||||
Reference in New Issue
Block a user