Enhance RAG corpus enrichment: Update prompt version, improve response validation, increase token limits, and refactor data normalization logic.
This commit is contained in:
@@ -91,7 +91,7 @@ except Exception:
|
|||||||
# Constants & helpers
|
# Constants & helpers
|
||||||
# -------------------------
|
# -------------------------
|
||||||
|
|
||||||
PROMPT_VERSION = "v4.0-standard-deep"
|
PROMPT_VERSION = "v4.1-standard-deep"
|
||||||
|
|
||||||
ENTITY_CANON = {
|
ENTITY_CANON = {
|
||||||
"PERSON": "PERSON",
|
"PERSON": "PERSON",
|
||||||
@@ -301,8 +301,7 @@ def ollama_generate_json(
|
|||||||
return json_loads(m.group(0))
|
return json_loads(m.group(0))
|
||||||
except Exception:
|
except Exception:
|
||||||
pass
|
pass
|
||||||
# last resort minimal structure
|
raise ValueError("Ollama returned incomplete or invalid JSON.")
|
||||||
return {"headline": "", "summary": raw, "keywords": [], "entities": [], "qa": []}
|
|
||||||
|
|
||||||
def ollama_generate_text(
|
def ollama_generate_text(
|
||||||
host: str,
|
host: str,
|
||||||
@@ -379,6 +378,25 @@ def build_user_main(text: str, summary_lang: str, doc_hint: str, want_qa: int, *
|
|||||||
"TEXT:\n" + text
|
"TEXT:\n" + text
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def unwrap_enrichment_response(value: Any) -> Dict[str, Any]:
|
||||||
|
if not isinstance(value, dict):
|
||||||
|
raise ValueError("Enrichment response is not a JSON object.")
|
||||||
|
nested = value.get("summary")
|
||||||
|
if not value.get("headline") and isinstance(nested, str) and nested.lstrip().startswith("{"):
|
||||||
|
try:
|
||||||
|
decoded = json_loads(nested)
|
||||||
|
except Exception as exc:
|
||||||
|
raise ValueError("Enrichment response contains truncated nested JSON.") from exc
|
||||||
|
if isinstance(decoded, dict):
|
||||||
|
value = decoded
|
||||||
|
if not isinstance(value.get("summary"), str) or not str(value.get("summary") or "").strip():
|
||||||
|
raise ValueError("Enrichment response is missing a summary.")
|
||||||
|
for field in ("keywords", "entities", "qa"):
|
||||||
|
if field in value and not isinstance(value[field], list):
|
||||||
|
raise ValueError(f"Enrichment response field '{field}' must be a list.")
|
||||||
|
return value
|
||||||
|
|
||||||
def build_user_qa_topup(text: str, summary_lang: str, need: int) -> str:
|
def build_user_qa_topup(text: str, summary_lang: str, need: int) -> str:
|
||||||
need = max(1, min(3, int(need)))
|
need = max(1, min(3, int(need)))
|
||||||
return (
|
return (
|
||||||
@@ -786,7 +804,7 @@ def enrich_one(
|
|||||||
"temperature": 0.2,
|
"temperature": 0.2,
|
||||||
"repeat_penalty": 1.1,
|
"repeat_penalty": 1.1,
|
||||||
"top_p": 0.9,
|
"top_p": 0.9,
|
||||||
"num_predict": 360 if include_qa else 240,
|
"num_predict": 1000 if include_qa else 700,
|
||||||
}
|
}
|
||||||
|
|
||||||
with sem:
|
with sem:
|
||||||
@@ -795,10 +813,8 @@ def enrich_one(
|
|||||||
try:
|
try:
|
||||||
out = ollama_generate_json(args.ollama, args.model, system, user,
|
out = ollama_generate_json(args.ollama, args.model, system, user,
|
||||||
keep_alive=args.keep_alive, timeout=args.timeout, options=options)
|
keep_alive=args.keep_alive, timeout=args.timeout, options=options)
|
||||||
|
out = unwrap_enrichment_response(out)
|
||||||
# sanitize + normalize structure
|
# sanitize + normalize structure
|
||||||
if not isinstance(out, dict):
|
|
||||||
out = {"headline": "", "summary": sanitize_text(str(out)), "keywords": [], "entities": [], "qa": []}
|
|
||||||
else:
|
|
||||||
for k in ("headline", "summary"):
|
for k in ("headline", "summary"):
|
||||||
if k in out and isinstance(out[k], str):
|
if k in out and isinstance(out[k], str):
|
||||||
out[k] = sanitize_text(out[k])
|
out[k] = sanitize_text(out[k])
|
||||||
|
|||||||
Reference in New Issue
Block a user