From 7b07b344e91df587b9403c38597bb127ed677ea1 Mon Sep 17 00:00:00 2001 From: Victor Giers Date: Mon, 15 Jun 2026 20:45:57 +0200 Subject: [PATCH] Enhance web toolchain robustness with advanced filtering, topic matching, and refined query generation. --- backend/agent/tools/web.py | 280 +++++++++++++++++++++++++++++++++++-- 1 file changed, 268 insertions(+), 12 deletions(-) diff --git a/backend/agent/tools/web.py b/backend/agent/tools/web.py index b1d326c..d9107d2 100644 --- a/backend/agent/tools/web.py +++ b/backend/agent/tools/web.py @@ -4,7 +4,10 @@ import asyncio from datetime import datetime, timezone import hashlib import json +import re from typing import Any, Dict, List +from urllib.parse import unquote, urlparse +import unicodedata import httpx @@ -30,12 +33,238 @@ QUERY_FORMAT = { FETCH_TEXT_LIMIT = 12_000 +GENERIC_QUERY_TERMS = { + "a", + "about", + "actual", + "aktuell", + "aktuelle", + "aktuellen", + "an", + "and", + "are", + "article", + "articles", + "auf", + "bericht", + "berichte", + "current", + "das", + "dem", + "den", + "der", + "deutsch", + "developments", + "die", + "english", + "ereignisse", + "event", + "events", + "for", + "from", + "heute", + "headline", + "headlines", + "in", + "ist", + "latest", + "look", + "media", + "nachrichten", + "news", + "neueste", + "of", + "on", + "online", + "outlet", + "outlets", + "recent", + "search", + "story", + "stories", + "the", + "today", + "top", + "ubersetzung", + "uebersetzung", + "und", + "was", + "weather", + "what", + "whats", + "wie", + "zur", +} + +CURRENT_LOOKUP_TERMS = { + "actual", + "aktuell", + "aktuelle", + "aktuellen", + "breaking", + "current", + "forecast", + "headline", + "headlines", + "heute", + "latest", + "nachrichten", + "news", + "neueste", + "recent", + "today", + "weather", +} + +CONSENT_OR_UNUSABLE_HOSTS = ( + "consent.google.", + "accounts.google.", +) + +CURRENT_LOOKUP_NOISE_HOSTS = ( + "collinsdictionary.com", + "de.langenscheidt.com", + "dict.cc", + "dict.leo.org", + "dictionary.cambridge.org", + "duden.de", + "dwds.de", + "langenscheidt.com", + "leo.org", + "linguee.", + "news.google.", +) + +CURRENT_LOOKUP_NOISE_TITLE_PATTERNS = ( + "deutsch-ubersetzung", + "deutsch ubersetzung", + "deutsch-uebersetzung", + "deutsch uebersetzung", + "dictionary", + "englisch-deutsch", + "english-german", + "worterbuch", +) + + +def _normalize_text(value: Any) -> str: + text = unquote(str(value or "")) + text = unicodedata.normalize("NFKD", text) + return text.encode("ascii", "ignore").decode("ascii").lower() + + +def _word_tokens(value: Any) -> List[str]: + return [item for item in re.findall(r"[a-z0-9]+", _normalize_text(value)) if len(item) >= 2] + + +def _topic_terms(prompt: str) -> List[str]: + seen = set() + terms: List[str] = [] + for token in _word_tokens(prompt): + if token in GENERIC_QUERY_TERMS or len(token) < 3: + continue + if token not in seen: + seen.add(token) + terms.append(token) + return terms[:8] + + +def _term_aliases(term: str) -> List[str]: + aliases = [term] + if term.endswith("ian") and len(term) > 5: + aliases.append(term[:-3]) + if term.endswith("isch") and len(term) > 6: + aliases.append(term[:-4]) + return [alias for alias in aliases if len(alias) >= 3] + + +def _matches_topic(prompt: str, value: str) -> bool: + terms = _topic_terms(prompt) + if not terms: + return True + haystack = _normalize_text(value) + matches = 0 + for term in terms: + if any(alias in haystack for alias in _term_aliases(term)): + matches += 1 + required = 1 if len(terms) == 1 else min(2, len(terms)) + return matches >= required + + +def _is_current_lookup(value: str) -> bool: + tokens = set(_word_tokens(value)) + return bool(tokens & CURRENT_LOOKUP_TERMS) + + +def _host_for(url: str) -> str: + try: + return _normalize_text(urlparse(url).netloc) + except Exception: + return "" + + +def _is_noise_result(url: str, title: str = "", topic_text: str = "") -> bool: + host = _host_for(url) + title_norm = _normalize_text(title) + if any(marker in host for marker in CONSENT_OR_UNUSABLE_HOSTS): + return True + if "before you continue" in title_norm or "google consent" in title_norm: + return True + if not _is_current_lookup(topic_text): + return False + if any(marker in host for marker in CURRENT_LOOKUP_NOISE_HOSTS): + return True + return any(pattern in title_norm for pattern in CURRENT_LOOKUP_NOISE_TITLE_PATTERNS) + + +def _fallback_queries(prompt: str) -> List[str]: + cleaned_prompt = " ".join(str(prompt or "").split()).strip() + terms = _topic_terms(cleaned_prompt) + topic = " ".join(terms[:4]) + queries: List[str] = [] + if topic and any(token in set(_word_tokens(cleaned_prompt)) for token in {"news", "nachrichten", "headlines"}): + queries.extend([f"{topic} news today", f"{topic} latest news", f"{topic} current events"]) + elif topic and any(token in set(_word_tokens(cleaned_prompt)) for token in {"weather", "forecast"}): + queries.extend([f"{topic} weather today", f"{topic} current weather", f"{topic} weather forecast"]) + if cleaned_prompt: + queries.append(cleaned_prompt) + return queries + + +def _query_is_too_generic(query: str, prompt: str) -> bool: + query_terms = _topic_terms(query) + prompt_terms = _topic_terms(prompt) + if not query_terms: + return True + if prompt_terms and not _matches_topic(prompt, query): + return True + return False + + +def _clean_queries(prompt: str, generated: List[Any]) -> List[str]: + cleaned: List[str] = [] + seen = set() + for query in [*_fallback_queries(prompt), *generated]: + text = " ".join(str(query or "").split()).strip() + if not text or _query_is_too_generic(text, prompt): + continue + key = _normalize_text(text) + if key in seen: + continue + seen.add(key) + cleaned.append(text[:300]) + if len(cleaned) >= 5: + break + return cleaned or [str(prompt or "")[:300]] + async def generate_queries_handler(arguments: Dict[str, Any], context: ToolExecutionContext) -> Dict[str, Any]: recent = render_recent_context(arguments.get("messages") or [], char_limit=1600) prompt = ( "Generate 3 concise, diverse web-search queries for the current request. " - "Resolve references using recent conversation. Return only valid JSON matching the schema.\n\n" + "Resolve references using recent conversation. Every query must include the concrete topic, " + "entity, or location from the request. Never output standalone generic terms such as latest, " + "current, top, today, news, or weather. Return only valid JSON matching the schema.\n\n" f"Current request:\n{arguments['prompt']}\n\nRecent conversation:\n{recent}" ) result = await chat_typed( @@ -50,12 +279,7 @@ async def generate_queries_handler(arguments: Dict[str, Any], context: ToolExecu queries = parsed.get("queries") if isinstance(parsed, dict) else None except Exception: queries = None - cleaned = [] - for query in queries or [arguments["prompt"]]: - text = " ".join(str(query).split()).strip() - if text and text.lower() not in {item.lower() for item in cleaned}: - cleaned.append(text[:300]) - return {"queries": cleaned[:5] or [arguments["prompt"][:300]]} + return {"queries": _clean_queries(arguments["prompt"], queries or [])} async def web_search_handler(arguments: Dict[str, Any], _context: ToolExecutionContext) -> Dict[str, Any]: @@ -90,10 +314,16 @@ async def web_search_handler(arguments: Dict[str, Any], _context: ToolExecutionC for query, items in zip(unique_queries, responses): for item in items: url = str(item.get("url") or "") + title = str(item.get("title") or "") + result_blob = f"{title} {url}" if not url or url in seen: continue + if _is_noise_result(url, title, query): + continue + if not _matches_topic(query, result_blob): + continue seen.add(url) - results.append({"query": query, "url": url, "title": item.get("title") or "", "engine": item.get("engine")}) + results.append({"query": query, "url": url, "title": title, "engine": item.get("engine")}) if len(results) >= maximum: break return {"queries": unique_queries, "results": results, "unavailable": not results} @@ -108,7 +338,9 @@ async def web_fetch_handler(arguments: Dict[str, Any], _context: ToolExecutionCo for item in urls: url = item.get("url") if isinstance(item, dict) else item url = str(url or "").strip() - if url and url not in unique: + title = str(item.get("title") or "") if isinstance(item, dict) else "" + query = str(item.get("query") or "") if isinstance(item, dict) else "" + if url and url not in unique and not _is_noise_result(url, title, query): unique.append(url) semaphore = asyncio.Semaphore(4) @@ -118,10 +350,22 @@ async def web_fetch_handler(arguments: Dict[str, Any], _context: ToolExecutionCo snapshot = await fetch_website_snapshot(url) full_text = str(snapshot.get("text") or "") text = full_text[:FETCH_TEXT_LIMIT] + canonical_url = snapshot.get("url") or snapshot.get("final_url") or url + title = snapshot.get("title") or "" + if _is_noise_result(str(canonical_url), str(title), ""): + return { + "requested_url": snapshot.get("requested_url") or url, + "canonical_url": canonical_url, + "title": title, + "cleaned_text": "", + "fetch_timestamp": datetime.now(timezone.utc).isoformat(), + "content_hash": hashlib.sha256(full_text.encode("utf-8", errors="ignore")).hexdigest(), + "error": "filtered_noise_page", + } return { "requested_url": snapshot.get("requested_url") or url, - "canonical_url": snapshot.get("url") or snapshot.get("final_url") or url, - "title": snapshot.get("title") or "", + "canonical_url": canonical_url, + "title": title, "cleaned_text": text, "fetch_timestamp": datetime.now(timezone.utc).isoformat(), "content_hash": hashlib.sha256(full_text.encode("utf-8", errors="ignore")).hexdigest(), @@ -147,7 +391,14 @@ async def web_rerank_handler(arguments: Dict[str, Any], _context: ToolExecutionC for page in arguments.get("pages") or []: if page.get("error") or not str(page.get("cleaned_text") or "").strip(): continue - docs.append((str(page.get("canonical_url") or page.get("requested_url") or ""), str(page.get("cleaned_text") or ""))) + url = str(page.get("canonical_url") or page.get("requested_url") or "") + title = str(page.get("title") or "") + text = str(page.get("cleaned_text") or "") + if _is_noise_result(url, title, arguments["prompt"]): + continue + if not _matches_topic(arguments["prompt"], f"{title} {url} {text[:2400]}"): + continue + docs.append((url, text)) ranked = await rerank( arguments["prompt"], docs, @@ -166,6 +417,11 @@ async def web_rerank_handler(arguments: Dict[str, Any], _context: ToolExecutionC if score < minimum_score: continue page = page_by_url.get(url) or {} + title = str(page.get("title") or "") + if _is_noise_result(url, title, arguments["prompt"]): + continue + if not _matches_topic(arguments["prompt"], f"{title} {url} {text[:2400]}"): + continue selected.append({ "type": "web", "url": url,