#!/usr/bin/env python3 """ ai-answer-evidence: views beside the registered result (method note, section 11: an additional computation may appear beside the registered result, never in its place). Not part of the frozen instrument. It reads the frozen outputs (raw/, work/check/, work/summary/, work/validation/, snapshots/) and imports the frozen modules without changing them. Every interval follows the registered rule (summary_figures.interval): 10,000 resamples of questions drawn by a generator seeded with the study's seed, percentile bounds. The registered intervals are reproduced as a check. Shares, medians and bounds are stored unrounded, so a table rounds each value once. The placebo draws use the same seed. common_questions the questions where all three answers state a figure citing_answers the answers that cite at least one page (each engine on its own questions) common_citing_questions the questions where all three answers state a figure and cite a page units_common_questions figures that carry a unit, on the questions where all three answers hold one paired Google AI Overviews minus ChatGPT on the same questions: the difference of the medians with its interval, and the questions where the first is higher, lower or equal share_by_page_band figures found by the number of cited pages and of readable pages of the answer; the (figure, page) pair share by the same bands two_or_more_readable_pages the answers with two or more readable pages pages_and_share rank correlation of the number of pages with the per-answer share pages_per_answer cited and readable pages per answer; the share of (figure, readable cited page) pairs in which the page holds the figure cut_to_k_pages the exact expectation of the share found when each answer keeps k of its readable pages, drawn at random without replacement (k = 1, 2, 3) exactly_two_pages the answers that have exactly two readable pages placebo each answer's pages replaced by the same number of readable pages cited for questions of other sectors (seeded draws, and the exact expectation) units figures with a currency, percent or multiple sign, and figures matched on the number alone attached figures whose sentence has no source attached, and the share found on the attached source where there is one not_on_a_readable_page not found plus access failure, and every figure that was not found fully_readable the answers whose cited pages were all readable, as medians per answer sectors answers per sector cell, and the figures of answers that cite no source validation the sample by engine, the readings with a quote, number-only figures frame totals, ChatGPT's answers without a source, the page frame, the time zone perplexity_and_claude the two engines added on 6 October 2026, after the preregistration (deviation log, entries 3, 5 and 6), from the check outputs the frozen functions wrote in addendum/figures/work/check//addendum/: unrounded medians and bounds on the questions all five engines answered with a figure, how the intervals lie, the paired differences of every pair, outcomes, pages per answer, the cuts, units, Perplexity without SVC-07, and the dates of the answers and of the page fetches in UTC+3. The values addendum/figures/summary.json stores are reproduced as a check. Chance matching: the placebo above for all five engines, with the pool of the main set and with a second pool drawn from the pages Perplexity and Claude cite. The like-for-like comparison in four views: all figures, figures with a unit, each answer cut to two readable pages (median per answer), the share minus the answer's placebo, and the questions where both answers have two or more readable pages. five engines from the check outputs of all five engines, by the same rules as above: the answers that cite a page, the views by number of pages, the rank correlations, the cuts, the placebo with its sector cells, units, attached sources (not available for Perplexity), the fully readable answers and the sector cells are computed for Google AI Overviews, ChatGPT, Gemini, Perplexity and Claude. The key five_engines holds the totals of the frame and of the classifier's labels, and the sources the five engines cite and share (the functions of addendum/analysis/source_overlap.py, imported; its four-engine values are reproduced as a check). Perplexity and Claude were added on 6 October 2026, after the preregistration (deviation log, entries 3, 5 and 6). Writes report/side-views.json. Usage (from the study folder): python3 report/side_views.py """ import bisect, collections, datetime, json, math, pathlib, random, re, statistics, sys HERE = pathlib.Path(__file__).resolve().parents[1] sys.dont_write_bytecode = True sys.path.insert(0, str(HERE)) import summary_figures as sf # noqa: E402 (frozen; imported, not changed) import check_figures as cf # noqa: E402 (frozen; imported, not changed) from chain_common import ENGINES, MIN_PAGE_CHARS, RAW, SNAPSHOTS, WORK, answers, frame, page_texts, questions # noqa: E402 OUT = pathlib.Path(__file__).resolve().parent / "side-views.json" SEED = sf.SEED # 1109872877 DRAWS = 1000 # placebo draws per answer FOUND = set(sf.FOUND) ADDED = ["perplexity", "claude"] # added on 6 October 2026 (deviation log, entries 3, 5 and 6) FIVE = ENGINES + ADDED # the fixed order of the engines in every table ADDED_FIG = HERE / "addendum" / "figures" SPLIT_ANSWER = ("perplexity", "SVC-07") # deviation log, run record of 6 October 2026 UTC3 = datetime.timezone(datetime.timedelta(hours=3)) # the time zone of the drafts def part(n, d): """n of d with the unrounded share; no share for a cell of fewer than 5 units.""" return {"n": n, "of": d, "share": n / d if d >= sf.MIN_UNITS else None} def resampled(ids, statistic): """The registered resampling rule for any statistic of a set of questions: 10,000 resamples of the sorted question list, drawn with replacement by a generator seeded with the study's seed; the bounds are the percentile elements summary_figures.interval takes. A resample in which the statistic is not defined is left out. Returns the unrounded bounds and the resamples used.""" ids, rng, values = sorted(ids), random.Random(SEED), [] for _ in range(sf.RESAMPLES): v = statistic([rng.choice(ids) for _ in ids]) if v is not None: values.append(v) values.sort() return [values[int(0.025 * len(values))], values[int(0.975 * len(values)) - 1]], len(values) def interval(per): """summary_figures.interval, unrounded: the interval of the median of the per-answer shares.""" if len(per) < sf.MIN_UNITS: return None return resampled(per, lambda ids: statistics.median(per[q] for q in ids))[0] def median_view(per): """Median of the per-answer shares with the registered interval; none for fewer than 5 answers.""" enough = len(per) >= sf.MIN_UNITS return {"answers": len(per), "median": statistics.median(per.values()) if enough else None, "interval_95": interval(per)} def paired(a, b, ids, centre="median"): """First engine minus second on the same questions. centre "median": the difference of the two medians; "mean": the mean of the per-question differences. Each resample draws one list of questions and uses it for both engines.""" ids = sorted(ids) if centre == "median": def stat(qs): return statistics.median(a[q] for q in qs) - statistics.median(b[q] for q in qs) else: def stat(qs): return statistics.mean(a[q] - b[q] for q in qs) return {"questions": len(ids), "statistic": "difference of the medians" if centre == "median" else "mean of the per-question differences", "first_median": statistics.median(a[q] for q in ids), "second_median": statistics.median(b[q] for q in ids), "difference": stat(ids), "interval_95": resampled(ids, stat)[0], "first_higher": part(sum(a[q] > b[q] for q in ids), len(ids)), "first_lower": part(sum(a[q] < b[q] for q in ids), len(ids)), "equal": part(sum(a[q] == b[q] for q in ids), len(ids))} def ranks(xs): """Average ranks: tied values share the mean of their positions.""" order, out, i = sorted(range(len(xs)), key=lambda k: xs[k]), [0.0] * len(xs), 0 while i < len(order): j = i while j + 1 < len(order) and xs[order[j + 1]] == xs[order[i]]: j += 1 for k in range(i, j + 1): out[order[k]] = (i + j) / 2 + 1 i = j + 1 return out def rank_correlation(x, y): """Spearman: the Pearson correlation of the average ranks; none when one side is constant.""" a, b = ranks(x), ranks(y) ma, mb = statistics.mean(a), statistics.mean(b) den = math.sqrt(sum((p - ma) ** 2 for p in a) * sum((q - mb) ** 2 for q in b)) return sum((p - ma) * (q - mb) for p, q in zip(a, b)) / den if den else None def position(view): """How the intervals of Google AI Overviews and ChatGPT lie: apart, meeting at one value, or overlapping.""" low, high = view["google_aio"]["interval_95"][0], view["chatgpt"]["interval_95"][1] return "do not overlap" if low > high else "meet at one value" if low == high else "overlap" def lie(x, y): """How two intervals lie, in either order: apart, meeting at one value, or overlapping.""" low, high = max(x[0], y[0]), min(x[1], y[1]) return "do not overlap" if low > high else "meet at one value" if low == high else "overlap" def stamp(text): """A time stamp of a record (+0300) or of a manifest (-04:00) as a datetime with its offset.""" return datetime.datetime.fromisoformat(re.sub(r"([+-]\d\d)(\d\d)$", r"\1:\2", text)) def utc3(text): return stamp(text).astimezone(UTC3).isoformat(timespec="minutes") BANDS = ["0", "1", "2", "3 to 4", "5 to 7", "8 or more"] def band(n): return str(n) if n <= 2 else "3 to 4" if n <= 4 else "5 to 7" if n <= 7 else "8 or more" def number_only(c): return bool(c["components"]) and all(p["unit"] == "" for p in c["components"]) def with_unit(c): return bool(c["components"]) and not number_only(c) def found(c): return c["outcome"] in FOUND # ---------- the frozen match, indexed (the result is asserted equal to the frozen function) ---------- class Page: """A page's figures by unit, sorted by value, so one figure is looked up without a scan.""" def __init__(self, text): self.comps = cf.comps(text) by_unit = collections.defaultdict(list) for s in self.comps: by_unit[s["unit"]].append((s["value"], s["start"])) self.values = {u: sorted(v) for u, v in by_unit.items()} def positions(self, c): rows = self.values.get(c["unit"], []) slack = 2e-9 * abs(c["value"]) + 2e-9 lo = bisect.bisect_left(rows, (c["value"] - slack, -1)) out = [] for value, start in rows[lo:]: if value > c["value"] + slack: break if math.isclose(value, c["value"], rel_tol=1e-9, abs_tol=1e-9): out.append(start) return out def holds(self, parts): """check_figures.match(page, parts)[1]: every part present with its unit and within the window.""" if not parts: return None pos = [self.positions(c) for c in parts] if not all(pos): return False return any(all(any(abs(q - p0) <= cf.WINDOW for q in ps) for ps in pos[1:]) for p0 in pos[0]) def cut_to(claims, k): """Expected number of found figures when k of the answer's readable pages are kept at random.""" total = 0.0 for c in claims: n, f = c["readable_pages"], len(c["found_on"]) if n == 0 or f == 0: continue kk = min(k, n) total += 1.0 if n - f < kk else 1 - math.comb(n - f, kk) / math.comb(n, kk) return total def saved_page_marks(d): f = RAW / d["engine"] / "main" / f"{d['id']}.a{d['attempt']}.html" html = f.read_text(encoding="utf-8", errors="replace") if f.exists() else "" return {"sources_controls": len(re.findall(r">Sources<", html)), "tagged_outbound_links": len(re.findall(r"utm_source=chatgpt\.com", html))} def first_fetches(manifest): """address -> the start of its first fetch in a manifest.""" out = {} for raw in manifest.read_text(encoding="utf-8").splitlines(): line = json.loads(raw) if raw.strip() else {} if line.get("started_at"): t = stamp(line["started_at"]) out[line["url_fetched"]] = min(out.get(line["url_fetched"], t), t) return out def hours(deltas): """The shortest and the longest of some time differences, in hours.""" values = sorted(d.total_seconds() / 3600 for d in deltas) return [values[0], values[-1]] def added_claims(ids): """The check outputs of Perplexity and Claude (addendum/figures/work/check//addendum/), written by the frozen check: the same fields as work/check/ holds for the other three engines.""" out = {} for e in ADDED: out[e] = {} for qid in ids: f = ADDED_FIG / "work" / "check" / e / "addendum" / f"{qid}.json" out[e][qid] = json.loads(f.read_text(encoding="utf-8"))["claims"] if f.exists() else [] return out def added_set(): """The page texts and the answers of Perplexity and Claude, read by the frozen readers. Their module points the frozen readers at addendum/figures/ and adds the reused page texts (addendum/figures/addendum_common.py); the names it changes are put back before this returns, so nothing else in this script reads from that folder. Nothing is written.""" sys.path.insert(0, str(ADDED_FIG)) import addendum_common as ac # noqa: E402 (imported, not changed) import chain_common as cc # noqa: E402 (frozen; imported, not changed) names = ("RAW", "WORK", "SNAPSHOTS", "LOG", "ENGINES", "SETS") kept = {n: getattr(cc, n) for n in names} try: ac.wire() return ac.page_texts(ac.SET), cc.answers(ac.SET) finally: for n, v in kept.items(): setattr(cc, n, v) assert all(getattr(cc, n) is kept[n] for n in names) def expectation(pool_size, holding, n): """The exact chance that at least one of n pages drawn without replacement from a pool holds a figure that `holding` pages of the pool hold: the rule of the registered placebo above.""" if not n or not holding: return 0.0 if pool_size - holding < n: return 1.0 return 1 - math.comb(pool_size - holding, n) / math.comb(pool_size, n) GROUPS = {"all": lambda c: True, "number_only": number_only, "with_unit": with_unit} def placebo_for(engines, ids, sector, claims, pool, pages): """The placebo for some engines with one pool: per engine the pooled exact expectation (all figures, figures with a unit, figures matched on the number alone) and, per answer, the mean expectation over the answer's figures. An answer draws as many pages as it has readable pages.""" holds, rows, per = {}, {}, {} for e in engines: total, per[e] = {g: [0.0, 0, 0] for g in GROUPS}, {} by_sector = collections.defaultdict(lambda: [0.0, 0, 0]) for qid in ids: cl = claims[e][qid] if not cl: continue s, n, mine = sector[qid], cl[0]["readable_pages"], 0.0 for c in cl: key = (tuple((p["unit"], p["value"]) for p in c["components"]), s) if key not in holds: holds[key] = sum(bool(pages[u].holds(c["components"])) for u in pool[s]) if c["components"] else 0 p = expectation(len(pool[s]), holds[key], n) mine += p by_sector[s][0] += p by_sector[s][1] += 1 by_sector[s][2] += found(c) for g, test in GROUPS.items(): if test(c): total[g][0] += p total[g][1] += 1 total[g][2] += found(c) per[e][qid] = mine / len(cl) rows[e] = {g: {"figures": v[1], "placebo_share_exact_expectation": v[0] / v[1], "actual": part(v[2], v[1])} for g, v in total.items()} rows[e]["by_sector"] = {s: {"figures": v[1], "placebo_share_exact_expectation": v[0] / v[1], "actual": part(v[2], v[1])} for s, v in sorted(by_sector.items())} return rows, per def perplexity_and_claude(ids, registered_claims, registered_share, sector, registered_pages, registered_pool, registered_placebo, registered_texts): """Perplexity and Claude beside the three engines the preregistration names. The first three engines are read as in main(); Perplexity and Claude from the check outputs of addendum/figures/. The values of the first three engines are not changed by anything here.""" summary = json.loads((ADDED_FIG / "summary.json").read_text(encoding="utf-8")) prepared = json.loads((ADDED_FIG / "prepare-report.json").read_text(encoding="utf-8")) like, five = summary["like_for_like"], ENGINES + ADDED claims = {**{e: registered_claims[e] for e in ENGINES}, **added_claims(ids)} share = {e: {q: sum(map(found, cl)) / len(cl) for q, cl in claims[e].items() if cl} for e in five} assert all(share[e] == registered_share[e] for e in five) figs = {e: [c for q in ids for c in claims[e][q]] for e in five} first = {e: {q: claims[e][q][0] for q in share[e]} for e in five} def same(mine, theirs): # the unrounded view equals the four decimals summary.json stores assert mine["answers"] == theirs["answers"] assert round(mine["median"], 4) == theirs["median"] assert [round(x, 4) for x in mine["interval_95"]] == theirs["interval_95"] out = {"note": "Perplexity and Claude were added on 6 October 2026, after the preregistration, and their answers " "went through the same four steps on 7 and 8 October (deviation log, entries 3, 5 and 6). Read from " "addendum/figures/. The values of Google AI Overviews, ChatGPT and Gemini are unchanged.", "engines": five} # medians: each engine on its own answers, and the questions all five answered with a figure out["all_answers_with_figures"] = {e: median_view(share[e]) for e in five} for e in five: same(out["all_answers_with_figures"][e], like["all_answers_with_figures"][e]) common = [q for q in ids if all(q in share[e] for e in five)] assert common == like["common_questions_five_engines"]["ids"] view = {e: median_view({q: share[e][q] for q in common}) for e in five} for e in five: same(view[e], like["common_questions_five_engines"][e]) three = [q for q in ids if all(q in share[e] for e in ENGINES)] out["common_questions_five_engines"] = { "questions": len(common), **view, "intervals": {f"{a} and {b}": lie(view[a]["interval_95"], view[b]["interval_95"]) for i, a in enumerate(five) for b in five[i + 1:]}, "questions_of_the_three_registered_engines": len(three), "of_them_not_among_the_five_engine_questions": sorted(set(three) - set(common)), "added_engines_without_a_figure_there": { e: sorted(q for q in set(three) - set(common) if q not in share[e]) for e in ADDED}, # the answers of the registered engines to the questions that leave the set (counts; a share # only for an answer with 5 or more figures) "registered_answers_to_the_questions_that_leave": { q: {e: {**part(sum(map(found, claims[e][q])), len(claims[e][q])), "cited_pages": first[e][q]["cited_pages"], "readable_pages": first[e][q]["readable_pages"]} for e in ENGINES} for q in sorted(set(three) - set(common))}} out["paired_on_the_five_engine_questions"] = { f"{a} minus {b}": paired(share[a], share[b], common) for a, b in (("perplexity", "chatgpt"), ("claude", "chatgpt"), ("google_aio", "chatgpt"), ("perplexity", "google_aio"), ("claude", "google_aio"))} # every pair of the five engines, the first of the fixed order minus the second out["paired_every_pair_on_the_five_engine_questions"] = { f"{a} minus {b}": paired(share[a], share[b], common) for i, a in enumerate(five) for b in five[i + 1:]} # like for like with a cited page: the questions where all five answers state a figure and cite a page citing = [q for q in common if all(first[e][q]["cited_pages"] > 0 for e in five)] cview = {e: median_view({q: share[e][q] for q in citing}) for e in five} out["common_citing_questions_five_engines"] = { "questions": len(citing), **cview, "intervals": {f"{a} and {b}": lie(cview[a]["interval_95"], cview[b]["interval_95"]) for i, a in enumerate(five) for b in five[i + 1:]}} # the outcomes of the two added engines out["outcomes"], out["attached"], out["not_on_a_readable_page"] = {}, {}, {} out["fully_readable"], out["units"], out["without_a_source"] = {}, {}, {} for e in ADDED: s, n = summary["engines"][e], len(figs[e]) o = dict(collections.Counter(c["outcome"] for c in figs[e])) assert n == s["figures"] and all(o.get(k, 0) == s["outcomes"][k] for k in sf.OUTCOMES) o = s["outcomes"] out["outcomes"][e] = { "figures": n, "answers_with_figures": len(share[e]), "found": part(sum(map(found, figs[e])), n), "found_attached": part(o["found_attached"], n), "found_elsewhere": part(o["found_elsewhere"], n), "not_found": part(o["not_found"], n), "unknown": part(o["no_source"] + o["access_failure"] + o["instrument_gap"], n)} has = [c for c in figs[e] if c["attached"]] citing = [c for c in figs[e] if c["cited_pages"] > 0] assert n - len(citing) == o["no_source"] # Perplexity's records tie no passage to a source address: no figure has an attached source. out["attached"][e] = {"available": bool(has)} if has: out["attached"][e].update({ "no_source_attached_to_the_sentence": part(n - len(has), n), "the_same_in_answers_that_cite_a_page": part(sum(not c["attached"] for c in citing), len(citing)), "found_on_the_attached_source_where_one_is_attached": part( sum(c["outcome"] == "found_attached" for c in has), len(has)), "found_elsewhere_with_no_source_attached": part( sum(c["outcome"] == "found_elsewhere" and not c["attached"] for c in figs[e]), o["found_elsewhere"])}) out["not_on_a_readable_page"][e] = { **part(o["not_found"] + o["access_failure"], n), "not_found": o["not_found"], "access_failure": o["access_failure"], "in_answers_that_cite_no_source": part(o["no_source"], n), "figures_the_check_could_not_read": o["instrument_gap"], "every_figure_that_was_not_found": part(n - sum(map(found, figs[e])), n), "answers_with_an_unreadable_page": part( sum(x["cited_pages"] > x["readable_pages"] for x in first[e].values()), len(share[e]))} full = {q: v for q, v in share[e].items() if 0 < first[e][q]["cited_pages"] == first[e][q]["readable_pages"]} pooled = part(sum(found(c) for q in full for c in claims[e][q]), sum(len(claims[e][q]) for q in full)) assert [pooled["n"], pooled["of"]] == [s["found_in_fully_readable_answers"][k] for k in ("n", "of")] out["fully_readable"][e] = {**median_view(full), "pooled": pooled} unit, bare = [c for c in figs[e] if with_unit(c)], [c for c in figs[e] if number_only(c)] out["units"][e] = {"with_unit": part(sum(map(found, unit)), len(unit)), "number_only": part(sum(map(found, bare)), len(bare)), "found_figures_matched_on_the_number_alone": part(sum(map(found, bare)), sum(map(found, figs[e])))} sourceless = sorted(q for q in share[e] if first[e][q]["cited_pages"] == 0) out["without_a_source"][e] = {"answers": sourceless, "figures": sum(len(claims[e][q]) for q in sourceless)} # pages per answer, five engines side by side (the registered engines as in the views above) out["pages_per_answer"] = {} for e in five: heads, rows = list(first[e].values()), collections.defaultdict(lambda: [0, 0, 0]) for q in share[e]: n = first[e][q]["readable_pages"] key = "at most 1" if n <= 1 else "2 or more" rows[key][0] += 1 rows[key][1] += sum(map(found, claims[e][q])) rows[key][2] += len(claims[e][q]) stored = like["by_readable_pages"][e] assert statistics.median(c["cited_pages"] for c in heads) == stored["median_cited_pages"] assert statistics.median(c["readable_pages"] for c in heads) == stored["median_readable_pages"] assert rows["at most 1"][0] == stored["answers_with_at_most_one_readable_page"]["n"] assert [rows["2 or more"][1], rows["2 or more"][2]] == [ stored["found_by_readable_pages"]["two_or_more"][k] for k in ("n", "of")] out["pages_per_answer"][e] = { "answers": len(heads), "median_cited_pages": statistics.median(c["cited_pages"] for c in heads), "median_readable_pages": statistics.median(c["readable_pages"] for c in heads), "answers_with_at_most_one_readable_page": part(rows["at most 1"][0], len(heads)), "found_in_answers_with_two_or_more_readable_pages": { "answers": rows["2 or more"][0], **part(rows["2 or more"][1], rows["2 or more"][2])}, "figure_page_pairs": part(sum(len(c["found_on"]) for c in figs[e]), sum(c["readable_pages"] for c in figs[e])), "cut_to_k_pages": {str(k): { "pooled_share": cut_to(figs[e], k) / len(figs[e]), "answers_with_fewer_readable_pages": part(sum(c["readable_pages"] < k for c in heads), len(heads))} for k in (2, 3)}} # Perplexity with and without the answer read from the stored thread engine, qid = SPLIT_ANSWER rest = {q: v for q, v in share[engine].items() if q != qid} stored = like["perplexity_without_SVC-07"] without = {**median_view(rest), "found": part(sum(found(c) for q in rest for c in claims[engine][q]), sum(len(claims[engine][q]) for q in rest)), "the_answer_has_figures": qid in share[engine], "figures_of_the_answer": part(sum(map(found, claims[engine][qid])), len(claims[engine][qid])), "among_the_five_engine_questions": qid in common, "registered_engines_without_a_figure_for_it": [e for e in ENGINES if qid not in share[e]]} same(without, stored) assert [without["found"]["n"], without["found"]["of"]] == [stored["found"]["n"], stored["found"]["of"]] out["perplexity_without_SVC-07"] = without # dates in UTC+3, and the time from an answer to the fetch of its pages pg = summary["pages"] asked = {} for e in ADDED: asked[e] = {} for f in sorted((ADDED_FIG / "raw" / e / "addendum").glob("???-??.json")): d = json.loads(f.read_text(encoding="utf-8")) asked[e][d["id"]] = (stamp(d["collected_at"]), [x["url"] for x in d["sources"]]) own = first_fetches(ADDED_FIG / "snapshots" / "addendum" / "manifest.jsonl") reused = {} for raw in (ADDED_FIG / "reused.jsonl").read_text(encoding="utf-8").splitlines(): row = json.loads(raw) reused[row["address"]] = min(stamp(x["started_at"]) for x in row["fetches"]) cited = {u for e in ADDED for _, urls in asked[e].values() for u in urls} assert len(cited) == pg["distinct_addresses_cited"] and cited == set(own) | set(reused) and not set(own) & set(reused) assert len(reused) == pg["reused_from_the_registered_sets"]["addresses"] assert len(own) == pg["fetched_for_this_analysis"]["addresses"] main_fetch = first_fetches(SNAPSHOTS / "main" / "manifest.jsonl") first_cited, before, pairs = {}, 0, 0 for a in answers("main"): t = stamp(json.loads((RAW / a["engine"] / "main" / f"{a['id']}.json").read_text(encoding="utf-8"))["collected_at"]) for u in a["sources"]: first_cited[u] = min(first_cited.get(u, t), t) pairs += 1 before += main_fetch[u] < t px = prepared["engines"]["perplexity"] cx = prepared["engines"]["claude"] # Reused addresses whose fetch is earlier than an added answer that cites them. earlier = {u: [t for e in ADDED for t, urls in asked[e].values() if u in urls] for u in reused} assert all(earlier.values()) # Perplexity's collection records (read only): the source entries per answer, and the notice that a # preview of the advanced search was switched on (true: seen; a note says where it could not be observed). collected = [json.loads(f.read_text(encoding="utf-8")) for f in sorted((HERE / "addendum" / "perplexity" / "records").glob("*.json"))] assert len(collected) == px["records_read"] and sum(len(r["sources"]) for r in collected) == px["source_entries"] unobserved = [r["id"] for r in collected if not r["pro_preview_notice"] and "'not observed'" in (r.get("note") or "")] # Perplexity's addresses whose query string was not kept, and the figures found on such a page. stripped = collections.defaultdict(set) for row in px["sources_whose_query_string_was_not_kept"]["list"]: stripped[row["id"]].add(row["address_without_query"]) on_stripped = [c for q, urls in stripped.items() for c in claims["perplexity"][q] if set(c["found_on"]) & urls] only_stripped = [c for q, urls in stripped.items() for c in claims["perplexity"][q] if c["found_on"] and set(c["found_on"]) <= urls] validation = json.loads((WORK / "validation" / "main" / "validation.json").read_text(encoding="utf-8"))["rows"] # The first and the last line of the added chain's log (the machine's time, UTC+3, as the records). events = [x[:19] for x in (ADDED_FIG / "chain.log").read_text(encoding="utf-8").splitlines() if " === " in x] out["pages_and_dates"] = { "time_zone": "UTC+3", "figure_chain_first_and_last_log_line": [events[0].replace(" ", "T")[:16], events[-1].replace(" ", "T")[:16]], "answers_collected": {e: [utc3(x) for x in pg["answers_collected"][e]] for e in ADDED}, "distinct_addresses_cited": pg["distinct_addresses_cited"], "readable_addresses": part(pg["readable_addresses"], pg["distinct_addresses_cited"]), "reused_from_the_registered_sets": { **part(len(reused), len(cited)), "readable": pg["reused_from_the_registered_sets"]["readable"], "fetched_from": utc3(pg["reused_from_the_registered_sets"]["fetched_from"]), "fetched_until": utc3(pg["reused_from_the_registered_sets"]["fetched_until"]), "hours_from_an_answer_to_the_fetch": hours( reused[u] - t for e in ADDED for t, urls in asked[e].values() for u in urls if u in reused), "fetched_before_an_added_answer_that_cites_the_address": part( sum(any(reused[u] < t for t in ts) for u, ts in earlier.items()), len(reused)), "fetched_before_every_added_answer_that_cites_the_address": part( sum(all(reused[u] < t for t in ts) for u, ts in earlier.items()), len(reused))}, "fetched_for_this_analysis": { **part(len(own), len(cited)), "readable": pg["fetched_for_this_analysis"]["readable"], "fetched_from": utc3(pg["fetched_for_this_analysis"]["fetched_from"]), "fetched_until": utc3(pg["fetched_for_this_analysis"]["fetched_until"]), "hours_from_an_answer_to_the_fetch": hours( own[u] - t for e in ADDED for t, urls in asked[e].values() for u in urls if u in own)}, "registered_main_set": { "addresses": len(first_cited), "hours_from_the_first_answer_that_cites_an_address_to_its_fetch": hours( main_fetch[u] - t for u, t in first_cited.items()), "citations_of_an_address_fetched_before_the_answer": part(before, pairs)}, "perplexity": {"source_entries": px["source_entries"], "distinct_addresses": px["distinct_addresses"], "median_source_entries_per_answer": statistics.median(len(r["sources"]) for r in collected), "records_with_the_preview_notice": part(sum(bool(r["pro_preview_notice"]) for r in collected), len(collected)), "records_whose_note_says_the_notice_could_not_be_observed": len(unobserved), "source_entries_whose_query_string_was_not_kept": part( px["sources_whose_query_string_was_not_kept"]["count"], px["source_entries"]), "addresses_without_their_query_string": len({u for urls in stripped.values() for u in urls}), "answers_with_such_an_address": len(stripped), "found_figures_on_such_a_page": part(len(on_stripped), sum(map(found, figs["perplexity"]))), "found_figures_only_on_such_pages": part(len(only_stripped), sum(map(found, figs["perplexity"]))), "site_name_chip_lines": px["chip_lines_removed_from_text"]["lines"] - px["chip_lines_removed_from_text"]["counter_lines"], "citation_chip_lines_left_out_of_the_text": px["chip_lines_removed_from_text"]["lines"], "of_them_counter_lines": px["chip_lines_removed_from_text"]["counter_lines"]}, "claude": {"source_entries": cx["source_entries"], "distinct_addresses": cx["distinct_addresses"], "answers_with_citations": part(cx["answers_with_citations"], cx["answers_prepared"]), "cited_blocks": cx["cited_blocks"], "citation_entries": cx["citation_entries"]}, "figures_of_the_added_engines_in_the_validation_sample": sum(r["engine"] in ADDED for r in validation)} # Chance matching. The pages of the added set, read by the frozen readers; the indexed match is # asserted equal to the check outputs of the two added engines, figure by figure. texts, added_answers = added_set() added_pages = {u: Page(t) for u, t in texts.items() if t} assert len(texts) == pg["distinct_addresses_cited"] and len(added_pages) == pg["readable_addresses"] # The addresses of all five engines. An address the main set could not read was fetched again for Perplexity # and Claude, so its readability can differ between the two fetches; each answer is checked against its own set. both = set(texts) & set(registered_texts) again = {u for u in both if u in own} assert all((u in added_pages) == (u in registered_pages) for u in both - again) assert not any(u in registered_pages for u in again) out["pages_and_dates"]["five_engines"] = { "distinct_addresses_cited": len(set(texts) | set(registered_texts)), "addresses_cited_in_the_main_set_and_by_perplexity_or_claude": len(both), "of_them_not_readable_in_the_main_set_and_fetched_again": len(again), "of_those_readable_in_the_later_fetch": sum(u in added_pages for u in again), "readable_in_at_least_one_fetch": part(len(set(added_pages) | set(registered_pages)), len(set(texts) | set(registered_texts)))} by_added, cited_added, compared = {(a["engine"], a["id"]): a for a in added_answers}, collections.defaultdict(set), 0 for (e, qid), a in by_added.items(): cited_added[sector[qid]].update(a["sources"]) readable = [u for u in a["sources"] if u in added_pages] for c in claims[e][qid]: assert c["readable_pages"] == len(readable), (e, qid) assert [u for u in readable if added_pages[u].holds(c["components"])] == c["found_on"], (e, qid, c["number"]) compared += 1 assert compared == sum(len(figs[e]) for e in ADDED) sectors = sorted(set(sector.values())) added_pool = {s: sorted(u for u in added_pages if u not in cited_added[s]) for s in sectors} pools = { "registered_pool": { "pool": "The pool of the registered placebo: the readable pages of the registered main set, without the " "pages that Google AI Overviews, ChatGPT or Gemini cite for a question of the answer's sector.", "pages": registered_pool, "index": registered_pages}, "added_pool": { "pool": "A second pool: the readable pages cited by Perplexity or Claude, without the pages either of " "them cites for a question of the answer's sector.", "pages": added_pool, "index": added_pages}} out["placebo"], per_answer_placebo = {}, {} for name, x in pools.items(): rows, per_answer_placebo[name] = placebo_for(five, ids, sector, claims, x["pages"], x["index"]) out["placebo"][name] = { "pool": x["pool"], "pool_pages_by_sector": {s: len(x["pages"][s]) for s in sectors}, "pool_pages_an_added_engine_cites_in_the_sector": { s: len(set(x["pages"][s]) & cited_added[s]) for s in sectors}, **rows} for e in five: # the registered pool reproduces the placebo computed in main() for g in GROUPS: assert math.isclose(out["placebo"]["registered_pool"][e][g]["placebo_share_exact_expectation"], registered_placebo[e][g]["placebo_share_exact_expectation"], rel_tol=1e-12, abs_tol=1e-12) out["placebo"]["added_figures_compared_with_the_check_outputs"] = compared # Robustness of the difference between the added engines and ChatGPT. def view(per, qs): v = {e: median_view({q: per[e][q] for q in qs}) for e in five} return {"questions": len(qs), **v, "intervals": {f"{a} and {b}": lie(v[a]["interval_95"], v[b]["interval_95"]) for i, a in enumerate(five) for b in five[i + 1:]}, "intervals_against_chatgpt": {e: lie(v[e]["interval_95"], v["chatgpt"]["interval_95"]) for e in five if e != "chatgpt"}, "paired_against_chatgpt": {e: paired(per[e], per["chatgpt"], qs) for e in five if e != "chatgpt"}} unit_share = {e: {q: sum(found(c) for c in claims[e][q] if with_unit(c)) / sum(map(with_unit, claims[e][q])) for q in share[e] if any(map(with_unit, claims[e][q]))} for e in five} unit_common = [q for q in ids if all(q in unit_share[e] for e in five)] cut = {e: {q: cut_to(claims[e][q], 2) / len(claims[e][q]) for q in share[e]} for e in five} out["robustness"] = { "note": "Each view is a median per answer with the registered interval, and the paired difference of the " "medians against ChatGPT on the same questions. The last view is the mean of the per-question " "differences. The intervals of two engines are apart when they do not overlap.", "all_figures": view(share, common), "figures_with_a_unit": {**view(unit_share, unit_common), "ids": sorted(unit_common)}, "cut_to_two_readable_pages": { **view(cut, common), # an answer with fewer than two readable pages keeps all of them: it is not cut "answers_with_fewer_readable_pages": { e: part(sum(first[e][q]["readable_pages"] < 2 for q in common), len(common)) for e in five}}, "share_minus_placebo": { name: view({e: {q: share[e][q] - per[e][q] for q in share[e]} for e in five}, common) | {"median_placebo_per_answer": {e: statistics.median(per[e][q] for q in common) for e in five}} for name, per in per_answer_placebo.items()}, "both_answers_have_two_or_more_readable_pages": { e: paired(share[e], share["chatgpt"], [q for q in ids if q in share[e] and q in share["chatgpt"] and first[e][q]["readable_pages"] >= 2 and first["chatgpt"][q]["readable_pages"] >= 2], "mean") for e in ["google_aio", "gemini"] + ADDED}} return out def sources_five(): """The sources all five engines cite, with the page key and the domain rule of the analysis of sources: the functions of addendum/analysis/source_overlap.py are imported and not changed. That script writes the overlap of five engines and the ten pairs; this adds what it writes for four engines only (the ceiling the list sizes allow, the sources by the number of engines that cite them, the questions whose ChatGPT list was fully seen, the totals). The four-engine values are recomputed here and asserted equal to addendum/analysis/source-overlap.json.""" sys.path.insert(0, str(HERE / "addendum" / "analysis")) import source_overlap as so # noqa: E402 (imported, not changed) stored = json.loads(so.OUT.read_text(encoding="utf-8")) data, _ = so.load() keys = ("questions", "mean_jaccard", "median_jaccard", "questions_with_a_shared_source", "pooled_shared", "pooled_union") def brief(o): return {k: o[k] for k in keys} for level in ("pages", "domains"): # the stored values, reproduced assert brief(so.overlap(data, so.FOUR, level)) == brief(stored["four_engines"][level]) assert brief(so.overlap(data, so.FIVE, level)) == brief(stored["five_engines_with_claude"][level]) mine = so.ceiling(data, so.FOUR, level) assert all(mine[k] == stored["four_engines"][f"ceiling_{level}"][k] for k in mine) assert so.by_engines(data, so.FOUR, level) == stored["four_engines"][f"{level}_by_engines"] assert brief(so.overlap(data, so.FOUR, level, skip=[q["id"] for q in so.questions() if q["id"] not in so.FULLY_SEEN])) \ == brief(stored["four_engines_fully_seen_chatgpt_lists"][level]) ten = so.pairwise(data, so.FIVE, level) for k, v in {**stored["four_engines"][f"pairwise_{level}"], **stored["five_engines_with_claude"][f"pairwise_{level}_with_claude"]}.items(): assert ten[k] == v cited = so.most_cited(data, so.FIVE) four_cited = so.most_cited(data, so.FOUR) assert json.loads(json.dumps(cited)) == stored["five_engines_with_claude"]["most_cited"] assert four_cited["cited_by_every_engine"] == stored["four_engines"]["most_cited"]["cited_by_every_engine"] assert sum(len(data[e]) for e in so.FOUR) == stored["four_engines"]["answers"] five_ids = {r["id"] for r in so.overlap(data, so.FIVE, "pages")["rows"]} four_ids = {r["id"] for r in so.overlap(data, so.FOUR, "pages")["rows"]} not_seen = [q["id"] for q in so.questions() if q["id"] not in so.FULLY_SEEN] others = [e for e in so.FIVE if e != "chatgpt"] out = {"engines": so.FIVE, "answers": sum(len(data[e]) for e in so.FIVE), "answers_with_sources": sum(1 for e in so.FIVE for a in data[e].values() if a["pages"]), "distinct_pages": len(set().union(*[a["pages"] for e in so.FIVE for a in data[e].values()])), "questions_of_the_four_engine_row_that_are_not_in_the_five_engine_row": sorted(four_ids - five_ids), "lists_fully_seen": { "five_engine_questions": len(five_ids), "five_engine_questions_with_the_chatgpt_list_fully_seen": len(five_ids & so.FULL["chatgpt"]), "five_engine_questions_with_chatgpt_and_gemini_lists_fully_seen": len(five_ids & so.FULL["chatgpt"] & so.FULL["gemini"])}, "without_chatgpt": {"engines": others}} for level in ("pages", "domains"): out[level] = brief(so.overlap(data, so.FIVE, level)) out[f"ceiling_{level}"] = so.ceiling(data, so.FIVE, level) out[f"{level}_by_engines"] = so.by_engines(data, so.FIVE, level) out[f"{level}_without_interrupted_question"] = brief(so.overlap(data, so.FIVE, level, skip=(so.INTERRUPTED,))) out[f"{level}_chatgpt_list_fully_seen"] = brief(so.overlap(data, so.FIVE, level, skip=not_seen)) ten = so.pairwise(data, so.FIVE, level) values = [v["mean_jaccard"] for v in ten.values()] out[f"pairwise_{level}"] = ten out[f"pairwise_{level}_lowest_and_highest_mean"] = [min(values), max(values)] out["without_chatgpt"][level] = brief(so.overlap(data, others, level)) out["without_chatgpt"][f"ceiling_{level}"] = so.ceiling(data, others, level) reached = cited["domains_by_engines_reached"] out["most_cited"] = {"distinct_domains": cited["distinct_domains"], "domains_by_engines_reached": {str(k): v for k, v in reached.items()}, "share_cited_by_one_engine_only": reached[1] / cited["distinct_domains"], "cited_by_every_engine": cited["cited_by_every_engine"], "top": cited["top"][:20]} return out def main(): ids = frame("main") sector = {q["id"]: q["sector"] for q in questions("main")} summary = json.loads((WORK / "summary" / "main.json").read_text(encoding="utf-8"))["engines"] # Perplexity and Claude: the same summary and the same check outputs, written by the frozen functions # in addendum/figures/. From here on every view that the outputs allow is computed for all five engines. summary.update({e: v for e, v in json.loads((ADDED_FIG / "summary.json").read_text(encoding="utf-8"))["engines"].items() if e in ADDED}) claims = {e: {qid: sf.checked(e, "main", qid) or [] for qid in ids} for e in ENGINES} claims.update(added_claims(ids)) figs = {e: [c for qid in ids for c in claims[e][qid]] for e in FIVE} share = {e: {qid: sum(map(found, cl)) / len(cl) for qid, cl in claims[e].items() if cl} for e in FIVE} res = {"seed": SEED, "note": "Views beside the registered result. Intervals: the rule of summary_figures.interval, " "10,000 resamples of questions, seeded with the study's seed. Values are unrounded."} first = {e: {q: claims[e][q][0] for q in share[e]} for e in FIVE} # cited_pages and readable_pages of an answer for e in FIVE: assert len(figs[e]) == summary[e]["figures"] and len(share[e]) == summary[e]["answers"]["with_figures"] # The registered result, recomputed here as a check on this script's reading of the outputs. res["registered"] = {e: median_view(share[e]) for e in ENGINES} for e in ENGINES: assert len(figs[e]) == summary[e]["figures"] assert round(res["registered"][e]["median"], 4) == summary[e]["median_share_per_answer"] assert [round(x, 4) for x in res["registered"][e]["interval_95"]] == summary[e]["median_interval_95"] assert sf.interval(share[e], random.Random(SEED)) == summary[e]["median_interval_95"] # (a) the questions where all three answers state a figure common = [qid for qid in ids if all(qid in share[e] for e in ENGINES)] res["common_questions"] = {"questions": len(common), **{e: median_view({q: share[e][q] for q in common}) for e in ENGINES}} # (b) the answers that cite at least one page res["citing_answers"] = {} for e in FIVE: citing = {q: v for q, v in share[e].items() if claims[e][q][0]["cited_pages"] > 0} res["citing_answers"][e] = {**median_view(citing), "answers_with_figures_that_cite_no_page": sorted(set(share[e]) - set(citing)), "pooled": part(sum(found(c) for q in citing for c in claims[e][q]), sum(len(claims[e][q]) for q in citing))} # (b2) like for like: the questions where all three answers state a figure and cite a page cites = {e: {q for q in share[e] if first[e][q]["cited_pages"] > 0} for e in FIVE} common_citing = sorted(set.intersection(*[cites[e] for e in ENGINES])) res["common_citing_questions"] = {"questions": len(common_citing), **{e: median_view({q: share[e][q] for q in common_citing}) for e in ENGINES}} res["common_citing_questions"]["intervals_of_google_aio_and_chatgpt"] = position(res["common_citing_questions"]) res["common_questions"]["intervals_of_google_aio_and_chatgpt"] = position(res["common_questions"]) # (b3) paired: Google AI Overviews minus ChatGPT on the same questions unit_share = {} for e in FIVE: unit_share[e] = {} for q in share[e]: mine = [x for x in claims[e][q] if with_unit(x)] if mine: unit_share[e][q] = sum(map(found, mine)) / len(mine) unit_common = sorted(set.intersection(*[set(unit_share[e]) for e in ENGINES])) res["units_common_questions"] = {"questions": len(unit_common), "ids": unit_common, **{e: median_view({q: unit_share[e][q] for q in unit_common}) for e in ENGINES}} res["units_common_questions"]["intervals_of_google_aio_and_chatgpt"] = position(res["units_common_questions"]) two_plus = {e: {q for q in share[e] if first[e][q]["readable_pages"] >= 2} for e in FIVE} a, b = "google_aio", "chatgpt" res["paired"] = {"first": a, "second": b, "sets": { "common_questions": paired(share[a], share[b], common), "both_answers_cite_a_page": paired(share[a], share[b], cites[a] & cites[b]), "unit_bearing_figures": paired(unit_share[a], unit_share[b], set(unit_share[a]) & set(unit_share[b])), "both_answers_have_two_or_more_readable_pages": paired(share[a], share[b], two_plus[a] & two_plus[b], "mean")}} # (c) pages per answer, and the share of (figure, readable cited page) pairs that match res["pages_per_answer"] = {} for e in FIVE: heads = list(first[e].values()) pairs = sum(c["readable_pages"] for c in figs[e]) res["pages_per_answer"][e] = { "answers": len(heads), "median_cited_pages": statistics.median(c["cited_pages"] for c in heads), "median_readable_pages": statistics.median(c["readable_pages"] for c in heads), "figure_page_pairs": part(sum(len(c["found_on"]) for c in figs[e]), pairs), # The same per figure: the share of its answer's readable pages that hold it (0 without a # readable page), averaged over the engine's figures. "mean_share_of_readable_pages_holding_a_figure": sum(len(c["found_on"]) / c["readable_pages"] for c in figs[e] if c["readable_pages"]) / len(figs[e]), } # (d) every answer cut to k random readable pages (exact expectation), and the answers with exactly two res["cut_to_k_pages"] = {} for k in (1, 2, 3): res["cut_to_k_pages"][str(k)] = { e: {"expected_found": cut_to(figs[e], k), "figures": len(figs[e]), "pooled_share": cut_to(figs[e], k) / len(figs[e]), "median_share_per_answer": statistics.median( cut_to(claims[e][q], k) / len(claims[e][q]) for q in share[e]), # an answer with fewer than k readable pages keeps all of them: it is not cut "answers_with_fewer_readable_pages": part( sum(first[e][q]["readable_pages"] < k for q in share[e]), len(share[e]))} for e in FIVE} res["exactly_two_pages"] = {} for e in FIVE: two = [q for q in share[e] if claims[e][q][0]["readable_pages"] == 2] res["exactly_two_pages"][e] = {"answers": len(two), **part(sum(found(c) for q in two for c in claims[e][q]), sum(len(claims[e][q]) for q in two))} res["share_by_readable_pages"] = {} for e in FIVE: rows = collections.defaultdict(lambda: [0, 0, 0]) for q in share[e]: n = claims[e][q][0]["readable_pages"] key = str(n) if n <= 2 else "3 or more" rows[key][0] += 1 rows[key][1] += sum(map(found, claims[e][q])) rows[key][2] += len(claims[e][q]) res["share_by_readable_pages"][e] = {k: {"answers": rows[k][0], **part(rows[k][1], rows[k][2])} for k in ("0", "1", "2", "3 or more")} res["share_by_readable_pages"][e]["answers_with_at_most_one_readable_page"] = part( rows["0"][0] + rows["1"][0], len(share[e])) res["share_by_readable_pages"][e]["found_in_answers_with_at_most_one_readable_page"] = { "answers": rows["0"][0] + rows["1"][0], **part(rows["0"][1] + rows["1"][1], rows["0"][2] + rows["1"][2])} few = [q for q in share[e] if claims[e][q][0]["readable_pages"] <= 1] res["share_by_readable_pages"][e]["at_most_one_readable_page_and_an_unreadable_cited_page"] = part( sum(claims[e][q][0]["cited_pages"] > claims[e][q][0]["readable_pages"] for q in few), len(few)) # (d2) by band of pages: figures found (pooled), and the (figure, readable page) pair share res["share_by_page_band"] = {} for field in ("cited_pages", "readable_pages"): res["share_by_page_band"][field] = {} for e in FIVE: rows = {k: [0, 0, 0, 0, 0] for k in BANDS} # answers, found, figures, matching pairs, pairs for q in share[e]: r = rows[band(first[e][q][field])] r[0] += 1 r[1] += sum(map(found, claims[e][q])) r[2] += len(claims[e][q]) r[3] += sum(len(x["found_on"]) for x in claims[e][q]) r[4] += first[e][q]["readable_pages"] * len(claims[e][q]) res["share_by_page_band"][field][e] = { k: {"answers": r[0], **part(r[1], r[2]), **({"figure_page_pairs": part(r[3], r[4])} if field == "readable_pages" else {})} for k, r in rows.items()} res["two_or_more_readable_pages"] = {} for e in FIVE: per = {q: share[e][q] for q in two_plus[e]} res["two_or_more_readable_pages"][e] = { **median_view(per), "pooled": part(sum(found(x) for q in per for x in claims[e][q]), sum(len(claims[e][q]) for q in per))} # (d3) rank correlation of the number of pages with the per-answer share, with the registered resampling res["pages_and_share"] = {} for field in ("cited_pages", "readable_pages"): for least in (1, 2): rows = {} for e in FIVE: qs = sorted(q for q in share[e] if first[e][q][field] >= least) bounds, used = resampled(qs, lambda s, e=e: rank_correlation( [first[e][q][field] for q in s], [share[e][q] for q in s])) rows[e] = {"answers": len(qs), "rank_correlation": rank_correlation( [first[e][q][field] for q in qs], [share[e][q] for q in qs]), "interval_95": bounds, "resamples_used": used, "interval_includes_zero": bounds[0] <= 0 <= bounds[1]} res["pages_and_share"][f"{field}, answers with at least {least}"] = rows res["pages_and_share"]["citing_answers_by_cited_pages"] = { e: {"three_or_more": part(sum(first[e][q]["cited_pages"] >= 3 for q in cites[e]), len(cites[e])), "one_or_two": part(sum(first[e][q]["cited_pages"] <= 2 for q in cites[e]), len(cites[e]))} for e in FIVE} # (e) placebo: the same number of readable pages, cited for questions of other sectors texts = page_texts("main") pages = {u: Page(t) for u, t in texts.items() if t} by_answer = {(a["engine"], a["id"]): a for a in answers("main")} cited_in = collections.defaultdict(set) for (e, qid), a in by_answer.items(): cited_in[sector[qid]].update(a["sources"]) rng_check, checked_pairs = random.Random(SEED), 0 for e in ENGINES: # the indexed match reproduces the frozen check, figure by figure for qid in ids: readable = [u for u in by_answer.get((e, qid), {}).get("sources", []) if u in pages] for c in claims[e][qid]: assert c["readable_pages"] == len(readable) assert [u for u in readable if pages[u].holds(c["components"])] == c["found_on"], (e, qid, c["number"]) for u in rng_check.sample(sorted(pages), 3): # and the frozen function on unrelated pages assert pages[u].holds(c["components"]) == cf.match(pages[u].comps, c["components"])[1] checked_pairs += 1 pool = {s: sorted(u for u in pages if u not in cited_in[s]) for s in set(sector.values())} holds = {} # (figure components, sector) -> the pool pages that hold the figure def holding(c, s): key = (tuple((p["unit"], p["value"]) for p in c["components"]), s) if key not in holds: holds[key] = {u for u in pool[s] if pages[u].holds(c["components"])} if c["components"] else set() return holds[key] rng = random.Random(SEED) groups = {"all": lambda c: True, "number_only": number_only, "with_unit": with_unit} res["placebo"] = {"draws_per_answer": DRAWS, "seed": SEED, "pool": "The readable pages of the main set, without the pages that Google AI Overviews, ChatGPT " "or Gemini cite for a question of the answer's sector. The same pool for all five engines.", "pool_pages_by_sector": {s: len(v) for s, v in sorted(pool.items())}, "check_pairs_compared_with_the_frozen_match": checked_pairs} for e in FIVE: # the three engines of the main set first, so their seeded draws are unchanged drawn = {g: 0 for g in groups} exact = {g: 0.0 for g in groups} actual = {g: [0, 0] for g in groups} per_sector = collections.defaultdict(lambda: [0, 0, 0, 0.0]) # draws found, figures, actually found, exact for qid in ids: cl = claims[e][qid] if not cl: continue s, n = sector[qid], cl[0]["readable_pages"] sets = [holding(c, s) for c in cl] for c, m in zip(cl, sets): p = 0.0 if not n or not m else (1.0 if len(pool[s]) - len(m) < n else 1 - math.comb(len(pool[s]) - len(m), n) / math.comb(len(pool[s]), n)) for g, test in groups.items(): if test(c): exact[g] += p actual[g][0] += found(c) actual[g][1] += 1 per_sector[s][1] += 1 per_sector[s][2] += found(c) per_sector[s][3] += p for _ in range(DRAWS): sample = set(rng.sample(pool[s], n)) if n else set() for c, m in zip(cl, sets): hit = bool(m & sample) per_sector[s][0] += hit for g, test in groups.items(): drawn[g] += hit and test(c) res["placebo"][e] = { g: {"figures": actual[g][1], "placebo_share": drawn[g] / DRAWS / actual[g][1], "placebo_share_exact_expectation": exact[g] / actual[g][1], "actual": part(*actual[g])} for g in groups} res["placebo"][e]["by_sector"] = {s: {"figures": v[1], "placebo_share": v[0] / DRAWS / v[1], "placebo_share_exact_expectation": v[3] / v[1], "actual": part(v[2], v[1])} for s, v in sorted(per_sector.items())} # (f) figures with a unit, and figures matched on the number alone res["units"] = {} all_found = [c for e in ENGINES for c in figs[e] if found(c)] checkable = [c for e in ENGINES for c in figs[e] if c["components"]] res["units"]["all_engines"] = { "figures_the_check_could_read": len(checkable), "number_only": part(sum(map(number_only, checkable)), len(checkable)), "found_figures_matched_on_the_number_alone": part(sum(map(number_only, all_found)), len(all_found))} found5 = [c for e in FIVE for c in figs[e] if found(c)] readable5 = [c for e in FIVE for c in figs[e] if c["components"]] res["units"]["five_engines"] = { "figures_the_check_could_read": len(readable5), "number_only": part(sum(map(number_only, readable5)), len(readable5)), "found_figures_matched_on_the_number_alone": part(sum(map(number_only, found5)), len(found5))} for e in FIVE: unit, bare = [c for c in figs[e] if with_unit(c)], [c for c in figs[e] if number_only(c)] res["units"][e] = { "with_unit": part(sum(map(found, unit)), len(unit)), "number_only": part(sum(map(found, bare)), len(bare)), "found_figures_matched_on_the_number_alone": part(sum(map(found, bare)), sum(map(found, figs[e]))), "with_unit_per_answer": median_view(unit_share[e])} # attached sources res["attached"] = {} for e in FIVE: none = [c for c in figs[e] if not c["attached"]] has = [c for c in figs[e] if c["attached"]] citing_figs = [c for c in figs[e] if c["cited_pages"] > 0] assert len(figs[e]) - len(citing_figs) == summary[e]["outcomes"]["no_source"] if not has: # Perplexity: its records tie no passage to a source address res["attached"][e] = {"available": False} continue res["attached"][e] = { "available": True, "no_source_attached_to_the_sentence": part(len(none), len(figs[e])), "the_same_in_answers_that_cite_a_page": part(sum(not c["attached"] for c in citing_figs), len(citing_figs)), "found_on_the_attached_source_where_one_is_attached": part( sum(c["outcome"] == "found_attached" for c in has), len(has)), "found_elsewhere_with_no_source_attached": part( sum(c["outcome"] == "found_elsewhere" for c in none), sum(c["outcome"] == "found_elsewhere" for c in figs[e]))} # not found on any readable page: not found plus access failure res["not_on_a_readable_page"] = {} for e in FIVE: o = summary[e]["outcomes"] res["not_on_a_readable_page"][e] = { **part(o["not_found"] + o["access_failure"], len(figs[e])), "not_found": o["not_found"], "access_failure": o["access_failure"], "in_answers_that_cite_no_source": part(o["no_source"], len(figs[e])), "figures_the_check_could_not_read": o["instrument_gap"], "every_figure_that_was_not_found": part(len(figs[e]) - sum(map(found, figs[e])), len(figs[e])), "answers_with_an_unreadable_page": part( sum(claims[e][q][0]["cited_pages"] > claims[e][q][0]["readable_pages"] for q in share[e]), len(share[e]))} # the answers whose cited pages were all readable, as medians per answer res["fully_readable"] = {} for e in FIVE: full = {q: v for q, v in share[e].items() if 0 < claims[e][q][0]["cited_pages"] == claims[e][q][0]["readable_pages"]} pooled = part(sum(found(c) for q in full for c in claims[e][q]), sum(len(claims[e][q]) for q in full)) assert [pooled["n"], pooled["of"]] == [summary[e]["found_in_fully_readable_answers"][k] for k in ("n", "of")] res["fully_readable"][e] = {**median_view(full), "pooled": pooled} # sectors: answers per cell, and the figures of answers that cite no source res["sectors"] = {} for e in FIVE: rows = {} for s in sorted(set(sector.values())): mine = [q for q in share[e] if sector[q] == s] sourceless = sorted(q for q in mine if claims[e][q][0]["cited_pages"] == 0) fig = [c for q in mine for c in claims[e][q]] citing = [c for q in mine if q not in sourceless for c in claims[e][q]] rows[s] = {"answers_with_figures": len(mine), "found": part(sum(map(found, fig)), len(fig)), "answers_that_cite_no_source": sourceless, "figures_in_answers_that_cite_no_source": part(len(fig) - len(citing), len(fig)), "found_in_answers_that_cite_a_source": part(sum(map(found, citing)), len(citing))} # the sector cell equals the cell the frozen summary wrote assert [rows[s]["found"]["n"], rows[s]["found"]["of"]] == [summary[e]["by_sector"][s][k] for k in ("found", "figures")] res["sectors"][e] = rows cells = [r["answers_with_figures"] for e in FIVE for r in res["sectors"][e].values()] res["sectors"]["answers_per_cell"] = {"fewest": min(cells), "most": max(cells)} # validation sample v = json.loads((WORK / "validation" / "main" / "validation.json").read_text(encoding="utf-8")) rows = v["rows"] strata = {s: dict(collections.Counter(r["engine"] for r in rows if r["stratum"] == s)) for s in ("found", "not_found")} sampled_found = [r for r in rows if r["stratum"] == "found"] bare = [r for r in sampled_found if number_only(claims[r["engine"]][r["id"]][r["figure_index"]])] no_quote = [r for r in rows if not (r.get("quote") or "").strip()] labels = {k: sum(summary[e]["not_found_labels"][k] for e in ENGINES) for k in ("absent", "partial", "rounded", "derived", "present", "unlabelled", "absent_quote_unconfirmed")} absent = [t for t in v["classifier"]["table"] if t["classifier"] == "absent"] moved = {t["reading"]: t["figures"] for t in absent if t["reading"] != "absent"} res["validation"] = { "figures": len(rows), "by_stratum_and_engine": strata, "sampled_figures_the_classifier_called_absent": { "read_as_something_else": part(sum(moved.values()), sum(t["figures"] for t in absent)), "readings": moved, "disagreements_in_the_not_found_sample": v["classifier"]["figures"] - v["classifier"]["same_label"]}, "readings_with_a_confirmed_quote": part(len(rows) - len(no_quote), len(rows)), "readings_without_a_quote": {"n": len(no_quote), "readings": dict(collections.Counter(r["reading"] for r in no_quote))}, "found_sample_number_only": part(len(bare), len(sampled_found)), "found_sample_number_only_readings": dict(collections.Counter(r["reading"] for r in bare)), "found_sample_readings": dict(collections.Counter(r["reading"] for r in sampled_found)), "classifier_labels_all_engines": {**labels, "not_found": sum(summary[e]["outcomes"]["not_found"] for e in ENGINES)}} # the frame: totals, ChatGPT's answers without a source, the page frame, the time zone recs = {e: {} for e in ENGINES} for e in ENGINES: for f in sorted((RAW / e / "main").glob("???-??.json")): d = json.loads(f.read_text(encoding="utf-8")) recs[e][d["id"]] = d sourceless = [d for d in recs["chatgpt"].values() if d["status"] == "ok" and not d.get("sources") and not d.get("inline_citations")] stamps = sorted(d["collected_at"] for e in ENGINES for d in recs[e].values() if d["status"] == "ok") again = sorted(json.loads(f.read_text(encoding="utf-8")).get("collected_at") or "" for e in ENGINES for f in (RAW / e / "repeat").glob("???-??.json")) again = [t for t in again if t] repeat = json.loads((WORK / "summary" / "repeat.json").read_text(encoding="utf-8"))["engines"] # The fetch log counts an address as readable when a fetch returned text; the check also sets aside a # text under 500 characters and a check page (chain_common.page_texts). got_text, short = set(), collections.defaultdict(list) for raw in (SNAPSHOTS / "main" / "manifest.jsonl").read_text(encoding="utf-8").splitlines(): line = json.loads(raw) if line["outcome"] == "text" and line.get("text_file"): got_text.add(line["url_fetched"]) short[line["url_fetched"]].append(line["text_chars"] < MIN_PAGE_CHARS) aside = sorted(got_text - set(pages)) fetched_text = {"n": len(got_text), "of": len(texts), "set_aside_by_the_check": len(aside), "every_text_under_500_characters": sum(all(short[u]) for u in aside), "a_check_page_among_the_texts": sum(not all(short[u]) for u in aside)} assert set(pages) <= got_text and len(got_text) - len(aside) == len(pages) res["frame"] = { "answers_collected": part(sum(summary[e]["questions"]["ok"] for e in ENGINES), len(ENGINES) * len(ids)), "answers_with_figures": sum(len(share[e]) for e in ENGINES), "figures": sum(len(figs[e]) for e in ENGINES), "found": sum(sum(map(found, figs[e])) for e in ENGINES), "repeat_answers_collected": part(sum(repeat[e]["questions"]["ok"] for e in ENGINES), len(ENGINES) * len(frame("repeat"))), "chatgpt_answers_without_a_source": { "n": len(sourceless), "saved_pages": {d["id"]: saved_page_marks(d) for d in sourceless}, "saved_page_of_a_sourced_answer_for_reference": { d["id"]: saved_page_marks(d) for d in list(recs["chatgpt"].values())[:1] if d.get("sources")}, "without_a_figure": sorted(d["id"] for d in sourceless if d["id"] not in share["chatgpt"]), "with_figures": sorted(d["id"] for d in sourceless if d["id"] in share["chatgpt"]), "figures_in_them": sum(len(claims["chatgpt"][d["id"]]) for d in sourceless)}, "cited_addresses": len(texts), "readable_addresses": len(pages), "addresses_with_text_in_a_fetch": fetched_text, "utc_offsets_of_the_timestamps": sorted({t[-5:] for t in stamps + again}), "main_set_first_and_last_answer": [stamps[0], stamps[-1]], "repeat_set_first_and_last_answer": [again[0], again[-1]]} # Perplexity and Claude, and the views of all five engines on the same questions res["perplexity_and_claude"] = perplexity_and_claude(ids, claims, share, sector, pages, pool, res["placebo"], texts) for e in ADDED: # the views of main() equal the ones perplexity_and_claude() checks against summary.json ad = res["perplexity_and_claude"] assert res["fully_readable"][e] == ad["fully_readable"][e] and res["not_on_a_readable_page"][e] == ad["not_on_a_readable_page"][e] assert all(res["units"][e][k] == ad["units"][e][k] for k in ad["units"][e]) assert all(res["placebo"][e]["by_sector"][s2]["actual"] == res["sectors"][e][s2]["found"] for s2 in res["sectors"][e]) for s2, row in ad["placebo"]["registered_pool"][e]["by_sector"].items(): assert math.isclose(row["placebo_share_exact_expectation"], res["placebo"][e]["by_sector"][s2]["placebo_share_exact_expectation"], rel_tol=1e-12, abs_tol=1e-12) # all five engines together: totals of the frame, of the classifier's labels, and the sources they cite label_keys = ("absent", "partial", "rounded", "derived", "present", "unlabelled", "absent_quote_unconfirmed") res["five_engines"] = { "engines": FIVE, "note": "Google AI Overviews, ChatGPT and Gemini are the engines the preregistration of 5 October 2026 names. " "Perplexity and Claude were added on 6 October 2026, and their answers went through the same four steps " "on 7 and 8 October (deviation log, entries 3, 5 and 6).", "frame": { "answers_collected": part(sum(summary[e]["questions"]["ok"] for e in FIVE), len(FIVE) * len(ids)), "answers_with_figures": sum(len(share[e]) for e in FIVE), "figures": sum(len(figs[e]) for e in FIVE), "found": part(sum(sum(map(found, figs[e])) for e in FIVE), sum(len(figs[e]) for e in FIVE)), "not_found_every_cited_page_readable": sum(summary[e]["outcomes"]["not_found"] for e in FIVE), # the first and the last complete answer of each engine, in UTC+3 as the records carry it "first_and_last_answer": { **{e: [utc3(min(d["collected_at"] for d in recs[e].values() if d["status"] == "ok")), utc3(max(d["collected_at"] for d in recs[e].values() if d["status"] == "ok"))] for e in ENGINES}, **res["perplexity_and_claude"]["pages_and_dates"]["answers_collected"]}}, "classifier_labels": {**{k: sum(summary[e]["not_found_labels"][k] for e in FIVE) for k in label_keys}, "not_found": sum(summary[e]["outcomes"]["not_found"] for e in FIVE)}, "sources": sources_five()} OUT.write_text(json.dumps(res, indent=1, ensure_ascii=False) + "\n", encoding="utf-8") def four(x): return [round(y, 4) for y in x] for e in ENGINES: c, a, p, u = res["common_questions"][e], res["common_citing_questions"][e], res["placebo"][e]["all"], res["units"][e] print(f"{e:11} common {c['median']:.4f} {four(c['interval_95'])} | common citing n={a['answers']} {a['median']:.4f} " f"{four(a['interval_95'])} | two or more readable {res['two_or_more_readable_pages'][e]['median']:.4f} | " f"placebo exact {p['placebo_share_exact_expectation']:.4f} | with unit {u['with_unit_per_answer']['median']:.4f} " f"{four(u['with_unit_per_answer']['interval_95'])}") for name, x in res["paired"]["sets"].items(): print(f"paired, {name}: n={x['questions']} {x['statistic']} {x['difference']:+.4f} {four(x['interval_95'])} " f"higher/lower/equal {x['first_higher']['n']}/{x['first_lower']['n']}/{x['equal']['n']}") for name, rows in res["pages_and_share"].items(): if "answers with" in name: print(f"rank correlation, {name}: " + "; ".join( f"{e} n={r['answers']} {r['rank_correlation']:+.3f} {[round(y, 3) for y in r['interval_95']]}" for e, r in rows.items())) ad = res["perplexity_and_claude"] for e in ad["engines"]: c = ad["common_questions_five_engines"][e] print(f"five engines, {ad['common_questions_five_engines']['questions']} questions: {e:11} " f"{c['median']:.4f} {four(c['interval_95'])}") for name, x in ad["paired_on_the_five_engine_questions"].items(): print(f"five engines, paired, {name}: n={x['questions']} {x['difference']:+.4f} {four(x['interval_95'])} " f"higher/lower/equal {x['first_higher']['n']}/{x['first_lower']['n']}/{x['equal']['n']}") for name, x in ad["placebo"].items(): if isinstance(x, dict): print(f"five engines, placebo, {name}: " + "; ".join( f"{e} all {x[e]['all']['placebo_share_exact_expectation']:.4f} unit " f"{x[e]['with_unit']['placebo_share_exact_expectation']:.4f} bare " f"{x[e]['number_only']['placebo_share_exact_expectation']:.4f}" for e in ad["engines"])) rb = ad["robustness"] views = [("all figures", rb["all_figures"]), ("with a unit", rb["figures_with_a_unit"]), ("cut to two", rb["cut_to_two_readable_pages"])] + [ (f"share minus placebo, {k}", v) for k, v in rb["share_minus_placebo"].items()] for name, x in views: print(f"five engines, views, {name}, n={x['questions']}: " + "; ".join( f"{e} {x[e]['median']:.4f} {four(x[e]['interval_95'])}" for e in ad["engines"])) print(" against chatgpt: " + "; ".join( f"{e} {x['intervals_against_chatgpt'][e]}, paired {p['difference']:+.4f} {four(p['interval_95'])} " f"{p['first_higher']['n']}/{p['first_lower']['n']}/{p['equal']['n']}" for e, p in x["paired_against_chatgpt"].items())) for e, x in rb["both_answers_have_two_or_more_readable_pages"].items(): print(f"five engines, views, both answers with two or more readable pages, {e} minus chatgpt: " f"n={x['questions']} mean {x['difference']:+.4f} {four(x['interval_95'])} " f"{x['first_higher']['n']}/{x['first_lower']['n']}/{x['equal']['n']}") print("written:", OUT.relative_to(HERE)) if __name__ == "__main__": main()