#!/usr/bin/env python3 """ ai-answer-evidence: the check of the main run. Offline: reads saved answers, extractions and page snapshots; no network. For every extracted figure it asks whether the pages the answer cites contain it: unit the value occurs with the same unit: $39 is not 39% or 39 users; 15 cents is $0.15; $1.2K is $1,200; "$50-100" is $50 to $100 together every part of a range or compound figure occurs within WINDOW characters attached the page whose citation marker sits in the figure's block contains it and gives the figure one outcome (method note, section 7): found_attached, found_elsewhere, not_found, no_source, access_failure, instrument_gap The number matching below the line (comps, claim_comps, positions, match) is the technical pilot's, byte for byte (check_strict.py of 23 September 2026). overlaps() is stricter than the pilot's: a marker is attached to a figure only when the figure's sentence lies inside the marker's block. A figure the extractor returned in a form that is not in its sentence cannot be checked. It stays in the denominator as an instrument gap. Writes work/check///.json. Usage: python3 check_figures.py --set main """ import argparse, json, math, re from chain_common import SETS, WORK, answers, page_address, page_texts from extract_figures import digits, segments WINDOW = 300 # ---------- the pilot's matching, unchanged ---------- FIG = re.compile(r""" (?:(?P
US\$|\$|€|£|\b(?:USD|EUR|GBP))\s?)?
    (?P\d{1,3}(?:,\d{3})+(?:\.\d+)?|\d+(?:\.\d+)?)
    (?:\s?(?PK\b|k\b|M\b|B\b|bn\b|thousand\b|million\b|billion\b))?
    (?:\s?(?P%|percent\b|per\s?cent\b|¢|cents?\b|x(?![\w\d])|×|USD\b|dollars?\b))?
""", re.X)
MULT = {"K": 1e3, "k": 1e3, "thousand": 1e3, "M": 1e6, "million": 1e6, "B": 1e9, "bn": 1e9, "billion": 1e9}
CUR = {"US$": "$", "$": "$", "USD": "$", "€": "€", "EUR": "€", "£": "£", "GBP": "£"}
RANGE_SEP = re.compile(r"^\s*(?:to|-|–|—|and)\s*$")


def comps(text, ranges="add"):
    out = []
    for m in FIG.finditer(text):
        value = float(m.group("num").replace(",", "")) * MULT.get(m.group("mult") or "", 1)
        pre = (m.group("pre") or "").strip()
        post = re.sub(r"\s", "", (m.group("post") or "").lower())
        unit, cents = CUR.get(pre, ""), False
        if post in ("%", "percent", "percent"):
            unit = "%"
        elif post in ("¢", "cent", "cents"):
            unit, cents, value = "$", True, value / 100
        elif post in ("x", "×"):
            unit = "x"
        elif post in ("usd", "dollar", "dollars"):
            unit = "$"
        out.append({"unit": unit, "value": value, "start": m.start(), "end": m.end(),
                    "mult": m.group("mult"), "cents": cents})
    # Two numbers joined by a range separator share unit and multiplier: "$50-100" is
    # $50 to $100, "10-15%" is 10% to 15%, "$1-2M" is $1M to $2M. A claim's own figure
    # is rewritten ("replace"); page text keeps its literal reading and gains the range
    # reading beside it ("add"), so a page's "$100-300" also counts as $300 (R9).
    extra = []
    for a, b in zip(out, out[1:]):
        if not RANGE_SEP.match(text[a["end"]:b["start"]]):
            continue
        na, nb = dict(a), dict(b)
        if b["mult"] and not a["mult"]:
            na["value"] *= MULT[b["mult"]]
        if b["unit"] and not a["unit"]:
            na["unit"] = b["unit"]
            na["value"] /= 100 if b["cents"] else 1
        elif a["unit"] and not b["unit"]:
            nb["unit"] = a["unit"]
            nb["value"] /= 100 if a["cents"] else 1
        if ranges == "replace":
            a.update(na)
            b.update(nb)
        else:
            extra += [x for x, y in ((na, a), (nb, b)) if x != y]
    return out + extra


def claim_comps(fig):
    """Components of one extracted figure, with unit and multiplier carried across a range."""
    return [{"unit": c["unit"], "value": c["value"]} for c in comps(fig, ranges="replace")]


def positions(page, c):
    return [s["start"] for s in page if s["unit"] == c["unit"]
            and math.isclose(s["value"], c["value"], rel_tol=1e-9, abs_tol=1e-9)]


def match(page, cs):
    """(every part present with its unit, every part within WINDOW characters of the first)"""
    if not cs or page is None:
        return None, None
    pos = [positions(page, c) for c in cs]
    if not all(pos):
        return False, False
    together = any(all(any(abs(q - p0) <= WINDOW for q in ps) for ps in pos[1:]) for p0 in pos[0])
    return True, together


def norm(s):
    return re.sub(r"\s+", " ", re.sub(r"[*_#>`]", "", s)).strip().lower()


def overlaps(seg, excerpt):
    a, b = norm(seg), norm(excerpt or "")
    if not a or not b:
        return False
    # The sentence lies inside the marker's block. Shared words alone do not attach: list items
    # of one template share most words.
    return a in b


# ---------- main run ----------
def outcome(parts, sources, unaddressed, readable, found_attached, found_any):
    if not parts:
        return "instrument_gap"
    if not sources and not unaddressed:
        return "no_source"
    if found_attached:
        return "found_attached"
    if found_any:
        return "found_elsewhere"
    if len(readable) < len(sources) or unaddressed:
        return "access_failure"
    return "not_found"


def changed_by_extractor(extraction, segs):
    """Entries the extractor returned in a form that is not in their sentence, unless the same
    sentence has a kept figure with the same digits (then the entry repeats a figure that is checked)."""
    kept = {(c["s"], tuple(digits(c["number"]))) for c in extraction["claims"]}
    out = []
    for d in extraction.get("dropped", []):
        k, num = d.get("s"), str(d.get("number", "")).strip()
        if isinstance(k, int) and 1 <= k <= len(segs) and digits(num) and num not in segs[k - 1] \
                and (k, tuple(digits(num))) not in kept:
            kept.add((k, tuple(digits(num))))        # the same changed figure returned twice counts once
            out.append({"s": k, "quote": segs[k - 1], "number": num, "kind": d.get("kind"), "subject": d.get("subject")})
    return out


def run(which):
    texts, cache, rows = page_texts(which), {}, []

    def page(url):
        if url not in cache:
            cache[url] = comps(texts[url]) if texts.get(url) else None
        return cache[url]

    for a in answers(which):
        ex = WORK / "extract" / a["engine"] / which / f"{a['id']}.json"
        if not ex.exists():
            continue
        extraction = json.loads(ex.read_text(encoding="utf-8"))
        claims = extraction["claims"]
        segs = segments(a["text"])
        readable = [u for u in a["sources"] if page(u) is not None]
        out = []
        for c in claims:
            parts = claim_comps(c["number"])
            seg = segs[c["s"] - 1]
            marks = [m for m in a["citations"] if overlaps(seg, m.get("claim_excerpt"))]
            attached = list(dict.fromkeys(page_address(m["url"]) for m in marks
                                          if m.get("url") and m.get("url_complete", True)))
            # A marker that names more pages than the interface lists for it (ChatGPT, guest).
            unseen = any(m.get("unseen_pages", 0) > 0 for m in marks)
            found_on = [u for u in readable if match(page(u), parts)[1]]
            found_attached = any(u in found_on for u in attached)
            out.append({**c, "components": parts, "cited_pages": len(a["sources"]) + a["unaddressed"],
                        "readable_pages": len(readable), "pages_without_address": a["unaddressed"],
                        "attached": attached, "found_on": found_on, "marker_names_unseen_page": unseen,
                        "outcome": outcome(parts, a["sources"], a["unaddressed"], readable, found_attached,
                                           bool(found_on))})
        for c in changed_by_extractor(extraction, segs):      # after the kept figures, so their index is stable
            marks = [m for m in a["citations"] if overlaps(c["quote"], m.get("claim_excerpt"))]
            out.append({**c, "components": [], "cited_pages": len(a["sources"]) + a["unaddressed"],
                        "readable_pages": len(readable), "pages_without_address": a["unaddressed"],
                        "attached": [], "found_on": [],
                        "marker_names_unseen_page": any(m.get("unseen_pages", 0) > 0 for m in marks),
                        "changed_by_extractor": True, "outcome": "instrument_gap"})
        o = WORK / "check" / a["engine"] / which / f"{a['id']}.json"
        o.parent.mkdir(parents=True, exist_ok=True)
        o.write_text(json.dumps({"engine": a["engine"], "id": a["id"], "window": WINDOW, "claims": out},
                                ensure_ascii=False, indent=1))
        rows.append((a["engine"], a["id"], len(out)))
    return rows


if __name__ == "__main__":
    ap = argparse.ArgumentParser()
    ap.add_argument("--set", required=True, choices=sorted(SETS))
    for engine, qid, n in run(ap.parse_args().set):
        print(f"{engine:11} {qid}  {n} figures")