#!/usr/bin/env python3 """ ai-answer-evidence: the summary of the main run (method note, section 9). frame per engine, every question of the set is accounted for: answered, no AI Overview, no answer, parse failed, no record; and every answer: with figures, without figures, empty text, extraction failed. An answer that is not checked yet stops the script unless --partial is given. primary per engine, the share of an answer's figures found on any page the answer cites; every figure the extractor listed is in the denominator and an unknown scores zero. The statistic is the median across answers, with a 95% percentile bootstrap interval over questions (10,000 resamples). also the count of every outcome per engine and per sector; the share found on the attached page; both shares without the unknowns; the share found in answers whose cited pages were all readable; the shares without the figures whose marker names a page the interface does not show, and without the answers that hold a "Sponsored" label; the classifier's label for every figure that was not found; and for the repeat run the first and the second answer to each question side by side. rules no percentage and no median for fewer than 5 units; an empty input stops the script. Reads raw/ records, work/check/ and work/classify/. Writes work/summary/.json and prints it. Usage: python3 summary_figures.py --set main [--partial] """ import argparse, collections, json, random, statistics, sys from chain_common import ENGINES, SETS, WORK, frame, questions, records from classify_figures import LABELS OUTCOMES = ["found_attached", "found_elsewhere", "not_found", "no_source", "access_failure", "instrument_gap"] FOUND, UNKNOWN = OUTCOMES[:2], OUTCOMES[4:] STATUSES = ["ok", "no_aio", "no_answer", "parse_failed", "no_record"] ANSWERS = ["with_figures", "without_figures", "empty_text", "extraction_failed", "not_checked"] SEED = 1109872877 # the first 32 bits of the SHA-256 of questions-v1.tsv (method note, section 3) RESAMPLES = 10000 MIN_UNITS = 5 def share(n, d): return None if d < MIN_UNITS else round(n / d, 4) def part(figures): """Found figures of a set of figures: n, of, share.""" n = sum(c["outcome"] in FOUND for c in figures) return {"n": n, "of": len(figures), "share": share(n, len(figures))} def show(n, d): return f"{n}/{d}" + (f" ({100 * n / d:.0f}%)" if d >= MIN_UNITS else "") def interval(by_question, rng): """Percentile bootstrap interval of the median of the per-answer shares, resampling questions.""" ids = sorted(by_question) if len(ids) < MIN_UNITS: return None medians = sorted(statistics.median(by_question[rng.choice(ids)] for _ in ids) for _ in range(RESAMPLES)) return [round(medians[int(0.025 * RESAMPLES)], 4), round(medians[int(0.975 * RESAMPLES) - 1], 4)] def checked(engine, which, qid): f = WORK / "check" / engine / which / f"{qid}.json" return json.loads(f.read_text(encoding="utf-8"))["claims"] if f.exists() else None def found_share(engine, which, qid): """[figures found, figures] of one answer, or None when the answer has no checked figures.""" claims = checked(engine, which, qid) return [sum(c["outcome"] in FOUND for c in claims), len(claims)] if claims else None def pairs(engine, ids): """Per repeated question: the first answer (main run) and the second (repeat run). Never pooled.""" return [{"id": qid, "first": found_share(engine, "main", qid), "second": found_share(engine, "repeat", qid)} for qid in ids] def labels_of(engine, which, qid): """figure index -> the classifier's row, for the figures of one answer.""" f = WORK / "classify" / engine / which / f"{qid}.json" rows = json.loads(f.read_text(encoding="utf-8"))["figures"] if f.exists() else [] return {r["figure_index"]: r for r in rows} def main(which, partial): ids = frame(which) if not ids: sys.exit("no question in the frame: nothing was collected for this set") # never an empty table sector = {q["id"]: q.get("sector", "") for q in questions(which)} recs = {(d["engine"], d["id"]): d for d in records(which)} result = {"set": which, "questions": len(ids), "engines": {}} for engine in ENGINES: status, state = collections.Counter(), collections.Counter() figures, per_answer, labels, sponsored = [], {}, collections.Counter(), collections.Counter() for qid in ids: d = recs.get((engine, qid)) status[d["status"] if d else "no_record"] += 1 if not d or d["status"] != "ok": continue marks = d.get("sponsored_labels") or {} sponsored["on_page"] += bool(marks.get("on_page")) sponsored["in_answer"] += bool(marks.get("in_answer")) sponsored["cards"] += bool(d.get("shopping_cards")) claims = checked(engine, which, qid) if not (d.get("answer_text") or "").strip(): state["empty_text"] += 1 elif (WORK / "extract-failed" / engine / which / f"{qid}.json").exists(): state["extraction_failed"] += 1 elif claims is None: state["not_checked"] += 1 elif not claims: state["without_figures"] += 1 else: state["with_figures"] += 1 per_answer[qid] = sum(c["outcome"] in FOUND for c in claims) / len(claims) rows = labels_of(engine, which, qid) for i, c in enumerate(claims): figures.append({**c, "id": qid, "sponsored": bool(marks.get("in_answer")), "all_readable": c["cited_pages"] > 0 and c["readable_pages"] == c["cited_pages"]}) if c["outcome"] == "not_found": # a label counts only for a figure that is not found now row = rows.get(i) or {} if row and row.get("number") != c["number"]: # a label given to another figure row = {} labels[row.get("label") or "unlabelled"] += 1 labels["absent_quote_unconfirmed"] += bool(row.get("absent_quote_unconfirmed")) if set(status) - set(STATUSES): sys.exit(f"{engine}: unknown record status {sorted(set(status) - set(STATUSES))}") if state["not_checked"] and not partial: sys.exit(f"{engine}: {state['not_checked']} answers are not extracted and checked yet " "(run extract_figures.py and check_figures.py, or pass --partial)") r = {"questions": {k: status[k] for k in STATUSES}, "answers": {k: state[k] for k in ANSWERS}, "answers_with_sponsored_label": dict(on_page=sponsored["on_page"], in_answer=sponsored["in_answer"]), "answers_with_shopping_cards_removed": sponsored["cards"], "figures": len(figures)} if figures: count = {k: sum(c["outcome"] == k for c in figures) for k in OUTCOMES} known = [c for c in figures if c["outcome"] not in UNKNOWN] sectors = {} for c in figures: row = sectors.setdefault(sector.get(c["id"], ""), {"figures": 0, "found": 0, **{k: 0 for k in OUTCOMES}}) row["figures"] += 1 row["found"] += c["outcome"] in FOUND row[c["outcome"]] += 1 enough = len(per_answer) >= MIN_UNITS r.update({ "outcomes": count, "found": part(figures), "found_attached": {"n": count["found_attached"], "of": len(figures), "share": share(count["found_attached"], len(figures))}, # A figure is unknown only when it was not found, so these two shares lean upward. "found_without_unknowns": part(known), "found_attached_without_unknowns": {"n": count["found_attached"], "of": len(known), "share": share(count["found_attached"], len(known))}, # Does not depend on the outcome: the answers whose cited pages were all readable. "found_in_fully_readable_answers": part([c for c in figures if c["all_readable"]]), "found_without_unseen_markers": part([c for c in figures if not c.get("marker_names_unseen_page")]), "found_without_sponsored_answers": part([c for c in figures if not c["sponsored"]]), "not_found_labels": {**{k: labels[k] for k in LABELS}, "unlabelled": labels["unlabelled"], "absent_quote_unconfirmed": labels["absent_quote_unconfirmed"]}, "median_share_per_answer": round(statistics.median(per_answer.values()), 4) if enough else None, "median_interval_95": interval(per_answer, random.Random(SEED)), "by_sector": {k: {**v, "share": share(v["found"], v["figures"])} for k, v in sorted(sectors.items())}, }) if which == "repeat": # method note, section 10 r["first_and_second_answer"] = pairs(engine, ids) result["engines"][engine] = r if not any(r["figures"] for r in result["engines"].values()): sys.exit("no checked figures: run check_figures.py first") # never an empty table out = WORK / "summary" / f"{which}.json" out.parent.mkdir(parents=True, exist_ok=True) out.write_text(json.dumps(result, indent=1) + "\n", encoding="utf-8") print(f"{len(ids)} questions per engine") for engine, r in result["engines"].items(): print(f"{engine:11} questions: " + ", ".join(f"{k} {v}" for k, v in r["questions"].items()) + " | answers: " + ", ".join(f"{k} {v}" for k, v in r["answers"].items())) print(f"\n{'engine':11} {'figures':>7} " + " ".join(f"{k:>15}" for k in OUTCOMES)) for engine, r in result["engines"].items(): if r["figures"]: print(f"{engine:11} {r['figures']:>7} " + " ".join(f"{show(r['outcomes'][k], r['figures']):>15}" for k in OUTCOMES)) print() for engine, r in result["engines"].items(): if not r["figures"]: continue ci, median = r["median_interval_95"], r["median_share_per_answer"] views = [("found", "found"), ("attached", "found_attached"), ("without unknowns", "found_without_unknowns"), ("fully readable answers", "found_in_fully_readable_answers"), ("without unseen markers", "found_without_unseen_markers"), ("without sponsored answers", "found_without_sponsored_answers")] print(f"{engine:11} " + " | ".join(f"{name} {show(r[key]['n'], r[key]['of'])}" for name, key in views)) print(f"{'':11} median per answer " + (f"{median:.0%} [{ci[0]:.0%}, {ci[1]:.0%}]" if ci else f"not given: {r['answers']['with_figures']} answers") + f" | not found {r['outcomes']['not_found']}: " + ", ".join(f"{k} {v}" for k, v in r["not_found_labels"].items())) if which == "repeat": print() for engine, r in result["engines"].items(): for row in r["first_and_second_answer"]: first, second = (show(*x) if x else "no figures" for x in (row["first"], row["second"])) print(f"{engine:11} {row['id']} first {first:>13} | second {second:>13}") if __name__ == "__main__": ap = argparse.ArgumentParser() ap.add_argument("--set", required=True, choices=sorted(SETS)) ap.add_argument("--partial", action="store_true", help="allow answers that are not checked yet") a = ap.parse_args() main(a.set, a.partial)