#!/usr/bin/env python3 """ ai-answer-evidence: the classifier of the main run (method note, section 8). The check is a screen. For every figure it did not find, this script gives one label: absent on no cited page in any form partial one end of a range, or the number without its qualifier rounded a cited page gives the value before rounding derived follows from numbers on a cited page by arithmetic present on a cited page, and the check missed it Three steps, and the model takes part in the second only. 1. Passages. The script cuts the places on the cited pages that hold (a) the digits of the figure in any unit, (b) a value within 10% of it, or (c) the subject's words. They are taken in turn from these three lists, at most PER_PAGE per page and list, up to BUDGET characters. 2. Reading. The local model evidence-classifier:v1 (gemma4:12b with a standing briefing and six worked examples, gemma/) names the passage that comes closest, copies a quote and the page's own figure from it, says whether that figure measures the same thing, and writes a calculation if the figure needs one. 3. Label. The script confirms the quote in the saved page text and the page figure in the quote, then compares the page figure with the answer's figure by the rules in label(). The model never names a label. Writes work/classify///.json and skips existing files. Usage: python3 classify_figures.py --set main """ import argparse, decimal, json, math, re, time, urllib.request from chain_common import SETS, WORK, answers, log, model_digest, page_texts, questions from check_figures import claim_comps, comps from extract_figures import digits OLLAMA = "http://localhost:11434/api/chat" GEMMA = "gemma4:12b" CLASSIFIER = "evidence-classifier:v1" LABELS = ["absent", "partial", "rounded", "derived", "present"] WIN = 220 # characters kept on each side of a hit PER_PAGE = 2 # passages per page from each of the three lists BUDGET = 6000 # characters of passages per figure, so input and output fit in 8,192 tokens NEAR = 0.10 # a page value within 10% of the figure's value is "near" QUOTE_MAX = 200 STOP = {"the", "and", "for", "per", "with", "from", "that", "this", "your", "you", "are", "can", "will", "have", "cost", "costs", "price", "prices", "average", "typical", "typically", "about", "around", "between", "most", "many", "some", "more", "than", "into", "over", "under", "month", "monthly", "year", "yearly"} def flat(s): return re.sub(r"\s+", " ", s).strip() def words(s): return [w for w in dict.fromkeys(re.findall(r"[a-z]{3,}", (s or "").lower())) if w not in STOP] def passages(pages, figure, subject, sentence): """pages = [(address, text)]. Returns [{"page", "url", "text", "why"}], ranked and capped.""" parts = claim_comps(figure) subj, sent = words(subject), words(sentence) lists = {"same digits": [], "near value": [], "subject words": []} for n, (url, text) in enumerate(pages, 1): text = flat(text) low = text.lower() def cut(pos, end, why): a, b = max(0, pos - WIN), min(len(text), end + WIN) piece = text[a:b] pl = piece.lower() score = 2 * sum(w in pl for w in subj) + sum(w in pl for w in sent) lists[why].append({"page": n, "url": url, "text": piece, "why": why, "at": a, "score": score}) for s in comps(text): for c in parts: if math.isclose(s["value"], c["value"], rel_tol=1e-9, abs_tol=1e-9): cut(s["start"], s["end"], "same digits") elif c["value"] and abs(s["value"] - c["value"]) <= NEAR * abs(c["value"]) \ and (s["unit"] == c["unit"] or not s["unit"] or not c["unit"]): cut(s["start"], s["end"], "near value") for w in subj: for m in list(re.finditer(r"\b" + re.escape(w), low))[:PER_PAGE]: cut(m.start(), m.end(), "subject words") out, taken, size = [], [], 0 for why in lists: # best first; one page gives at most PER_PAGE per list lists[why].sort(key=lambda p: -p["score"]) kept, per = [], {} for p in lists[why]: if per.get(p["page"], 0) < PER_PAGE: per[p["page"]] = per.get(p["page"], 0) + 1 kept.append(p) lists[why] = kept while any(lists.values()) and size < BUDGET: for why in lists: while lists[why]: p = lists[why].pop(0) # skip a passage that overlaps one already taken from the same page text if any(q["page"] == p["page"] and abs(q["at"] - p["at"]) < WIN for q in taken) \ or any(q["text"][:80] == p["text"][:80] for q in taken): continue taken.append(p) size += len(p["text"]) out.append({k: p[k] for k in ("page", "url", "text", "why")}) break if size >= BUDGET: break return out ABOUT = ["same", "other qualifier", "other thing"] PERIODS = {12.0, 52.0, 365.0} # months, weeks and days of a year may enter a calculation unquoted CALC = re.compile(r"^\s*\d[\d,]*(?:\.\d+)?(?:\s*[-+*/]\s*\d[\d,]*(?:\.\d+)?){1,3}\s*$") def gemma(question, claim, shown, think=False): schema = {"type": "object", "required": ["passage", "quote", "page_figure", "about", "arithmetic"], "properties": { "passage": {"type": "integer", "minimum": 0, "maximum": len(shown)}, "quote": {"type": "string"}, "page_figure": {"type": "string"}, "about": {"type": "string", "enum": ABOUT}, "arithmetic": {"type": "string"}}} user = (f"QUESTION: {question}\nANSWER SENTENCE: {flat(claim['quote'])[:600]}\nFIGURE: {claim['number']}\n" f"MEASURES: {claim.get('subject') or ''}\nPASSAGES:\n" + "\n".join(f"[{k}] (page {p['page']}) {p['text']}" for k, p in enumerate(shown, 1))) body = json.dumps({"model": CLASSIFIER, "stream": False, "format": schema, "think": think, "messages": [{"role": "user", "content": user}]}).encode() req = urllib.request.Request(OLLAMA, data=body, headers={"Content-Type": "application/json"}) with urllib.request.urlopen(req, timeout=900) as r: resp = json.loads(r.read()) if resp.get("done_reason") == "length": raise ValueError("output hit the token cap") return json.loads(resp["message"]["content"]) def rounds_to(page, answer): """A page value rounds to the answer's value: half up at the place of the answer's last non-zero digit ($149 -> $150, $6,021 -> $6,000), and no further than NEAR from it.""" if not answer or math.isclose(page, answer) or abs(page - answer) > NEAR * abs(answer): return False a = decimal.Decimal(repr(answer)).normalize() place = decimal.Decimal(1).scaleb(a.as_tuple().exponent) return (decimal.Decimal(repr(page)) / place).quantize(1, rounding=decimal.ROUND_HALF_UP) * place == a def fit(a, p): """How one page value fits one part of the answer's figure: "exact", "rounded" or None.""" if a["unit"] and p["unit"] and a["unit"] != p["unit"]: return None if math.isclose(a["value"], p["value"], rel_tol=1e-9, abs_tol=1e-9): return "exact" return "rounded" if rounds_to(p["value"], a["value"]) else None def calculated(expr, quote, sentence): """The value of the model's calculation, when every number in it is in the quote, in the answer sentence or a period of the year, and at least one is in the quote. Else None.""" if not CALC.match(expr or ""): return None nums = [float(x) for x in digits(expr)] on_page, in_answer = {float(x) for x in digits(quote)}, {float(x) for x in digits(sentence)} if not all(n in on_page or n in in_answer or n in PERIODS for n in nums) or not any(n in on_page for n in nums): return None try: return float(eval(expr.replace(",", ""), {"__builtins__": {}})) # digits and + - * / only (CALC) except Exception: # a division by zero, a number Python cannot read return None def label(figure, sentence, r, quote_confirmed): """The label that follows from the model's reading r. Returns (label, how each part fits).""" answer = claim_comps(figure) page = claim_comps(r.get("page_figure") or "") in_quote = set(digits(r.get("quote") or "")) if not quote_confirmed or not page or r.get("about") == "other thing" \ or not all(d in in_quote for d in digits(r.get("page_figure") or "")): return "absent", [] # A two-ended range against a two-ended range: low end with low end, high with high. if len(answer) == 2 and len(page) >= 2 and answer[0]["unit"] == answer[1]["unit"]: lo, hi = sorted(answer, key=lambda c: c["value"]) fits = [fit(lo, min(page, key=lambda c: c["value"])), fit(hi, max(page, key=lambda c: c["value"]))] else: fits = [next((f for f in (fit(a, p) for p in page) if f == "exact"), None) or next((f for f in (fit(a, p) for p in page) if f), None) for a in answer] if r.get("about") == "other qualifier": return ("partial" if any(fits) else "absent"), fits if all(fits): return ("present" if all(f == "exact" for f in fits) else "rounded"), fits value = calculated(r.get("arithmetic"), r.get("quote") or "", sentence) if value is not None: derived = [f or ("derived" if fit(a, {"unit": "", "value": value}) else None) for a, f in zip(answer, fits)] if all(derived): return "derived", derived return ("partial" if any(fits) else "absent"), fits def classify(question, claim, pages, think=False): """One not-found figure -> the model's reading, the checks on it, and the label that follows.""" shown = passages(pages, claim["number"], claim.get("subject"), claim["quote"]) if not shown: # nothing on any page to show: absent by rule, no model call return {"label": "absent", "by_rule": True, "passages_shown": 0, "passage": 0, "quote": "", "quote_confirmed": None, "page_figure": "", "about": None, "arithmetic": "", "fits": []} r = gemma(question, claim, shown, think) k, quote = r.get("passage") or 0, flat(r.get("quote") or "") page = shown[k - 1] if 1 <= k <= len(shown) else None text = flat(dict(pages).get(page["url"], "")) if page else "" confirmed = (bool(quote) and len(quote) <= QUOTE_MAX and quote in text) if page else None lab, fits = label(claim["number"], claim["quote"], {**r, "quote": quote}, confirmed) return {"label": lab, "by_rule": False, "passages_shown": len(shown), "passage": k, # absent only because the model's quote is not in the page as written: with the quote # taken as confirmed, the same reading would give another label "absent_quote_unconfirmed": bool(lab == "absent" and page and not confirmed and label( claim["number"], claim["quote"], {**r, "quote": quote}, True)[0] != "absent"), "passage_why": page["why"] if page else None, "url": page["url"] if page else None, "quote": quote[:QUOTE_MAX], "quote_confirmed": confirmed, "page_figure": flat(r.get("page_figure") or ""), "about": r.get("about"), "arithmetic": flat(r.get("arithmetic") or ""), "fits": fits} def run(which, budget=1): qtext = {q["id"]: q["question"] for q in questions(which)} texts, n = page_texts(which), 0 digest = model_digest(CLASSIFIER) # read before any model work; stops when it cannot be read for a in answers(which): if n >= budget: break chk = WORK / "check" / a["engine"] / which / f"{a['id']}.json" out = WORK / "classify" / a["engine"] / which / f"{a['id']}.json" if out.exists() or not chk.exists(): continue t0 = time.time() pages = [(u, texts[u]) for u in a["sources"] if texts.get(u)] rows = [] for i, c in enumerate(json.loads(chk.read_text(encoding="utf-8"))["claims"]): if c["outcome"] != "not_found": continue try: try: reading = classify(qtext.get(a["id"], ""), c, pages) except Exception: # one repeat; a second failure is final and is reported reading = classify(qtext.get(a["id"], ""), c, pages) rows.append({"figure_index": i, "s": c["s"], "number": c["number"], "subject": c.get("subject"), **reading}) except Exception as e: # recorded, never guessed rows.append({"figure_index": i, "s": c["s"], "number": c["number"], "subject": c.get("subject"), "label": None, "error": f"{type(e).__name__}: {e}"}) out.parent.mkdir(parents=True, exist_ok=True) out.write_text(json.dumps({"engine": a["engine"], "id": a["id"], "model": CLASSIFIER, "base_model": GEMMA, "model_digest": digest, "seconds": round(time.time() - t0, 1), "figures": rows}, ensure_ascii=False, indent=1)) n += 1 log(f"classified {a['engine']}/{a['id']}: {len(rows)} not-found figures in {time.time()-t0:.0f}s") return n if __name__ == "__main__": ap = argparse.ArgumentParser() ap.add_argument("--set", required=True, choices=sorted(SETS)) which = ap.parse_args().set while run(which, budget=1): pass