#!/usr/bin/env python3 """ ai-answer-evidence: figure extraction for the main run. Lists the figures of every saved answer with the local model evidence-extractor:v1 (gemma4:12b with a standing briefing and two worked examples, gemma/). The functions below the line are the technical pilot's (pipeline.py of 23 September 2026); the folders and the engine list around them differ, a failed model call is repeated once before it is final, and the output names the model's digest. Writes work/extract///.json and skips existing files. Usage: python3 extract_figures.py --set main """ import argparse, json, re, time, urllib.request from chain_common import SETS, WORK, answers, log, model_digest, questions OLLAMA = "http://localhost:11434/api/chat" GEMMA = "gemma4:12b" EXTRACTOR = "evidence-extractor:v1" # ---------- the pilot's extraction, unchanged ---------- KINDS = ["price", "percentage", "rate", "count", "duration", "range", "other"] MAX_OUT = 2000 # output tokens; a longer generation is cut and parked as a failure CHUNK_CHARS = 5000 # segments per Gemma call are capped so input + output fit in 8,192 tokens def segments(text): """Answer lines split into sentences, keeping those with a digit. Each is an exact substring of the answer, so the quote that goes to the checker is verbatim by construction.""" out = [] for line in text.splitlines(): for part in re.split(r'(?<=[.!?])\s+(?=[A-Z0-9"\u201c(\[*$])', line.strip()): part = part.strip() if part and re.search(r"\d", part): out.append(part) return out def chunks(segs): part, size = [], 0 for i, seg in enumerate(segs): if part and size + len(seg) > CHUNK_CHARS: yield part part, size = [], 0 part.append((i + 1, seg)) size += len(seg) if part: yield part def gemma(question, part): """part = [(global segment number, text), ...]. Gemma labels which numbers are figure claims; the briefing and worked examples live in the EXTRACTOR model itself.""" schema = {"type": "object", "required": ["claims"], "properties": {"claims": {"type": "array", "items": { "type": "object", "required": ["s", "number", "kind", "subject"], "properties": { "s": {"type": "integer", "minimum": 1, "maximum": len(part)}, "number": {"type": "string"}, "kind": {"type": "string", "enum": KINDS}, "subject": {"type": "string"}}}}}} user = f"QUESTION: {question}\nSEGMENTS:\n" + "\n".join(f"[{k}] {seg}" for k, (_, seg) in enumerate(part, 1)) # think=False: gemma4 is a thinking model and by default spent 2,000-5,000 hidden tokens # reasoning before the JSON; the task needs none of it (70 tokens with thinking off). body = json.dumps({"model": EXTRACTOR, "stream": False, "format": schema, "think": False, "options": {"num_predict": MAX_OUT}, "messages": [{"role": "user", "content": user}]}).encode() req = urllib.request.Request(OLLAMA, data=body, headers={"Content-Type": "application/json"}) with urllib.request.urlopen(req, timeout=900) as r: resp = json.loads(r.read()) if resp.get("done_reason") == "length": raise ValueError(f"output hit the {MAX_OUT}-token cap") claims = json.loads(resp["message"]["content"]).get("claims", []) for c in claims: # local segment number -> global k = c.get("s") c["s"] = part[k - 1][0] if isinstance(k, int) and 1 <= k <= len(part) else None return claims def extract(which, budget=1): """One answer per call, so a slow Gemma never holds up the source downloads.""" qtext = {q["id"]: q["question"] for q in questions(which)} digest = model_digest(EXTRACTOR) # read before any model work; stops when it cannot be read n = 0 for a in answers(which): if n >= budget: break out = WORK / "extract" / a["engine"] / which / f"{a['id']}.json" fail = WORK / "extract-failed" / a["engine"] / which / f"{a['id']}.json" if out.exists() or fail.exists() or not a["text"].strip(): continue t0 = time.time() segs = segments(a["text"]) try: try: raw = [c for part in chunks(segs) for c in gemma(qtext.get(a["id"], ""), part)] except Exception: # one repeat; a second failure is final and is reported raw = [c for part in chunks(segs) for c in gemma(qtext.get(a["id"], ""), part)] except Exception as e: fail.parent.mkdir(parents=True, exist_ok=True) fail.write_text(json.dumps({"engine": a["engine"], "id": a["id"], "error": f"{type(e).__name__}: {e}", "seconds": round(time.time() - t0, 1)}, ensure_ascii=False, indent=1)) log(f"gemma error {a['engine']}/{a['id']}: {e} (parked in extract-failed)") continue kept, dropped, used = [], [], {} for c in raw: k, num = c.get("s"), str(c.get("number", "")).strip() seg = segs[k - 1] if k else "" # The same figure may measure two things in one segment ("$20" for Team and for # Enterprise), so allow as many entries as the figure occurs in the segment. if num and num in seg and digits(num) and used.get((k, num), 0) < seg.count(num): used[(k, num)] = used.get((k, num), 0) + 1 kept.append({"s": k, "quote": seg, "number": num, "kind": c.get("kind"), "subject": c.get("subject")}) else: dropped.append(c) # not in its segment (reformatted), repeated, or bad segment number out.parent.mkdir(parents=True, exist_ok=True) out.write_text(json.dumps({"engine": a["engine"], "id": a["id"], "model": EXTRACTOR, "base_model": GEMMA, "model_digest": digest, "segments": len(segs), "seconds": round(time.time() - t0, 1), "claims": kept, "dropped": dropped}, ensure_ascii=False, indent=1)) n += 1 log(f"extracted {a['engine']}/{a['id']}: {len(kept)} claims ({len(dropped)} dropped) in {time.time()-t0:.0f}s") return n def digits(s): """Numbers in a string, normalized: '$1,299.00' -> '1299.00', '15 %' -> '15'.""" return [x.replace(",", "") for x in re.findall(r"\d[\d,]*(?:\.\d+)?", s)] if __name__ == "__main__": ap = argparse.ArgumentParser() ap.add_argument("--set", required=True, choices=sorted(SETS)) which = ap.parse_args().set while extract(which, budget=1): pass