#!/usr/bin/env python3 """ ai-answer-evidence: the report's tables, written from the outputs of the frozen scripts. Not part of the frozen instrument. It computes one view the method note asks for and the summary script does not write (Gemini's primary measure without the eight answers saved after the last passing exit check of block 1, deviation 1) and sums over the engines. Everything else is printed from files other scripts wrote. Rounding: every percentage is rounded once, from an unrounded value (a count over a count, or an unrounded share, median or bound of report/side-views.json and addendum/analysis/source-overlap.json). The registered medians and intervals are printed from the unrounded recomputation in report/side-views.json, after a check that it equals the frozen summary to the four decimals the summary stores. Every table shows the five engines together, in one order: Google AI Overviews, ChatGPT, Gemini, Perplexity, Claude (API). The preregistration of 5 October 2026 names the first three; Perplexity and Claude were added on 6 October and their answers went through the same four steps on 7 and 8 October (deviation log, entries 3, 5, 6 and 7). The values of the three engines the preregistration names are printed unchanged. A table that exists for three engines only (the repeat run, the validation sample) says so in its heading. Reads work/summary/main.json, work/summary/repeat.json, work/validation/main/validation.json, work/check/, addendum/figures/summary.json (the same summary for Perplexity and Claude), report/side-views.json (report/side_views.py: the views beside the primary measure and the five-engine views of the sources) and addendum/analysis/source-overlap.json (the sources the engines cite and share). Writes report/tables-main.md. Usage (from the study folder), after the two scripts above: python3 report/report_tables.py """ import datetime, json, pathlib, random, statistics, sys HERE = pathlib.Path(__file__).resolve().parents[1] sys.dont_write_bytecode = True sys.path.insert(0, str(HERE)) sys.path.insert(0, str(HERE / "report")) import summary_figures as sf # noqa: E402 (frozen; imported, not changed) import side_views as sv # noqa: E402 (the unrounded interval) NAMES = {"google_aio": "Google AI Overviews", "chatgpt": "ChatGPT", "gemini": "Gemini", "perplexity": "Perplexity", "claude": "Claude (API)"} SECTORS = {"ai_tools": "AI tools", "business_software": "Business software", "consumer_electronics": "Consumer electronics", "legal_local_services": "Legal and local services", "personal_finance": "Personal finance", "travel": "Travel"} FLAGGED = ["BIZ-04", "TRV-05", "AIT-04", "FIN-01", "FIN-02", "ELC-07", "AIT-17", "TRV-06"] # deviation 1 ENGINES = ["google_aio", "chatgpt", "gemini"] ADDED = ["perplexity", "claude"] # added on 6 October 2026 (deviation log, entries 3, 5 and 6) FIVE = ENGINES + ADDED def pct(n, d): return "n/a" if d < sf.MIN_UNITS else f"{100 * n / d:.0f}%" def cell(n, d): return f"{n} of {d} ({pct(n, d)})" if d >= sf.MIN_UNITS else f"{n} of {d}" def part(x): return cell(x["n"], x["of"]) def count(n, d): """cell(), without the percentage for a count that is not zero and would print as 0%.""" return f"{n} of {d}" if n and f"{100 * n / d:.0f}" == "0" else cell(n, d) def med(x): """A median per answer with its interval; nothing for fewer than 5 answers.""" if x["median"] is None or not x["interval_95"]: return f"not given ({x['answers']} answers)" return f"{x['median']:.0%} ({x['interval_95'][0]:.0%} to {x['interval_95'][1]:.0%})" def one(x, questions=None): """A mean or median overlap: one decimal, and a whole percentage for a cell of fewer than 10 questions.""" return f"{100 * x:.0f}%" if questions is not None and questions < 10 else f"{100 * x:.1f}%" def points(x): """A difference of two shares, in percentage points, with its sign.""" return f"{100 * x:+.0f}" def span(x): return f"{points(x[0])} to {points(x[1])} points" def corr(x): return f"{x['rank_correlation']:+.2f} ({x['interval_95'][0]:+.2f} to {x['interval_95'][1]:+.2f}), {x['answers']} answers" def table(head, rows): out = ["| " + " | ".join(head) + " |", "|" + "|".join("---" for _ in head) + "|"] return "\n".join(out + ["| " + " | ".join(str(c) for c in r) + " |" for r in rows]) def gemini_without_flagged(): rows = (HERE / "questions-main-v1.tsv").read_text(encoding="utf-8").splitlines()[1:] ids = [r.split("\t")[1] for r in rows] per, per_all = {}, {} for qid in ids: s = sf.found_share("gemini", "main", qid) if s and s[1]: per_all[qid] = s[0] / s[1] if qid not in FLAGGED: per[qid] = s[0] / s[1] return per_all, per def when(text, day=True): """A stored time stamp as the texts write it: 6 October 2026, 17:07.""" d = datetime.datetime.fromisoformat(text) return (f"{d.day} {d:%B} {d.year}, " if day else "") + f"{d:%H:%M}" def names(keys): return " and ".join(NAMES[x] for x in keys.split(" and ")) def lying(intervals, questions): """Which pairs of intervals lie apart, in one sentence. intervals: "a and b" -> how the two lie.""" apart = [names(k) for k, v in intervals.items() if v == "do not overlap"] meet = [names(k) for k, v in intervals.items() if v == "meet at one value"] out = (f"Intervals on the {questions} questions. Do not overlap: " + ("; ".join(apart) if apart else "no pair") + "." + (" Meet at one value: " + "; ".join(meet) + "." if meet else "")) rest = len(intervals) - len(apart) - len(meet) return out + (f" The other {rest} of the {len(intervals)} pairs overlap." if apart or meet else f" All {len(intervals)} pairs overlap.") def flipped(x): """A paired difference written the other way round: second minus first.""" return {**x, "first_median": x["second_median"], "second_median": x["first_median"], "difference": 0.0 - x["difference"], "interval_95": [0.0 - x["interval_95"][1], 0.0 - x["interval_95"][0]], "first_higher": x["first_lower"], "first_lower": x["first_higher"]} NOTE = ("The preregistration of 5 October 2026 names Google AI Overviews, ChatGPT and Gemini. Perplexity and Claude (API) " "were added on 6 October, and their answers went through the same four steps on 7 and 8 October, after the results of the first three had been " "computed, as an exploratory addition (deviation log, " "entries 3, 5, 6 and 7); see the method.") def main_tables(s, v, w): ad, fe = w["perplexity_and_claude"], w["five_engines"] pg = ad["pages_and_dates"] md = ["# Tables of the main run", "", "Written by `report/report_tables.py` from the outputs of the frozen scripts. Five engines in one order: " + ", ".join(NAMES[e] for e in FIVE) + ". " + NOTE + " The values of the three engines the preregistration " "names are printed unchanged. No percentage is given for a cell with fewer than 5 units, and no median for " "fewer than 5 answers. A count that is not zero and would print as 0% is given without a percentage.", ""] fr, f5 = w["frame"], fe["frame"] seen = f5["first_and_last_answer"] md += ["## 1. The frame: 50 questions per engine", "", table(["Engine", "Answered", "No AI Overview", "No answer", "Parse failed", "No record", "Answers with figures", "Answers without figures", "Extraction failed", "Figures", "Left unlabelled", "First answer (UTC+3)", "Last answer (UTC+3)"], [[NAMES[e], s[e]["questions"]["ok"], s[e]["questions"].get("no_aio", 0), s[e]["questions"]["no_answer"], s[e]["questions"].get("parse_failed", 0), s[e]["questions"]["no_record"], s[e]["answers"]["with_figures"], s[e]["answers"]["without_figures"], s[e]["answers"]["extraction_failed"], s[e]["figures"], s[e]["not_found_labels"]["unlabelled"], when(seen[e][0]), when(seen[e][1])] for e in FIVE]), "", f"All five engines: {f5['answers_collected']['n']} of {f5['answers_collected']['of']} answers collected, " f"{f5['answers_with_figures']} answers with figures, {f5['figures']} figures, {part(f5['found'])} of them found. " f"Google AI Overviews, ChatGPT and Gemini: {fr['answers_collected']['n']} of {fr['answers_collected']['of']} " f"answers collected, {fr['answers_with_figures']} answers with figures, {fr['figures']} figures, {fr['found']} of " f"them found. Perplexity and Claude (API): {sum(s[e]['questions']['ok'] for e in ADDED)} of {50 * len(ADDED)} " f"answers collected, {sum(s[e]['answers']['with_figures'] for e in ADDED)} answers with figures, " f"{sum(s[e]['figures'] for e in ADDED)} figures, {sum(s[e]['found']['n'] for e in ADDED)} of them found.", ""] c = fr["chatgpt_answers_without_a_source"] ref = next(iter(c["saved_page_of_a_sourced_answer_for_reference"].items())) nos = ad["without_a_source"] md += [f"ChatGPT answers that cite no source: {c['n']} ({', '.join(c['saved_pages'])}). Their saved pages hold " f"{sum(x['sources_controls'] for x in c['saved_pages'].values())} \"Sources\" controls and " f"{sum(x['tagged_outbound_links'] for x in c['saved_pages'].values())} outbound links " f"(a sourced answer, {ref[0]}: {ref[1]['sources_controls']} and {ref[1]['tagged_outbound_links']}), " f"which indicates that ChatGPT answered them without a search. {len(c['without_a_figure'])} of them state no figure " f"({', '.join(c['without_a_figure'])}); the other {len(c['with_figures'])} " f"({', '.join(c['with_figures'])}) hold {c['figures_in_them']} figures. Claude (API) answers that ran no search " f"and cite no source: {len(nos['claude']['answers'])} ({', '.join(nos['claude']['answers'])}), with " f"{nos['claude']['figures']} figures. Perplexity: every answer cites a source.", ""] reuse, mine, reg, all5 = (pg["reused_from_the_registered_sets"], pg["fetched_for_this_analysis"], pg["registered_main_set"], pg["five_engines"]) before = reuse["fetched_before_an_added_answer_that_cites_the_address"] always = reuse["fetched_before_every_added_answer_that_cites_the_address"] md += [f"Cited pages. The answers of Google AI Overviews, ChatGPT and Gemini cite {fr['cited_addresses']} addresses, " f"{fr['readable_addresses']} readable as the check counts them (a fetch returned the page's text, and the text " f"is not a stub or a check page). A fetch returned text for {part(fr['addresses_with_text_in_a_fetch'])} " f"addresses, the count of the run record; the check sets aside " f"{fr['addresses_with_text_in_a_fetch']['set_aside_by_the_check']} of them, because a fetched text under 500 " "characters, or a check page, is not the page's text " f"({fr['addresses_with_text_in_a_fetch']['every_text_under_500_characters']} with every text under 500 characters, " f"{fr['addresses_with_text_in_a_fetch']['a_check_page_among_the_texts']} with a check page). Each of these " f"addresses was fetched {reg['hours_from_the_first_answer_that_cites_an_address_to_its_fetch'][0]:.1f} to " f"{reg['hours_from_the_first_answer_that_cites_an_address_to_its_fetch'][1]:.1f} hours after the first answer " "that cites it.", "", f"The answers of Perplexity and Claude (API) cite {pg['distinct_addresses_cited']} addresses, " f"{part(pg['readable_addresses'])} readable as the check counts them. {reuse['n']} of {reuse['of']} addresses " f"reuse the fetch made for the main and the repeat set ({reuse['readable']} readable), made from " f"{when(reuse['fetched_from'])} to {when(reuse['fetched_until'])} (UTC+3): a reused fetch lies between " f"{-reuse['hours_from_an_answer_to_the_fetch'][0]:.0f} hours before and " f"{reuse['hours_from_an_answer_to_the_fetch'][1]:.0f} hours after the answer of Perplexity or Claude (API) that cites " f"it, and {before['n']} of {before['of']} reused addresses were fetched before such an answer " f"({always['n']} of {always['of']} before every such answer). The other {mine['n']} of {mine['of']} were " f"fetched from {when(mine['fetched_from'])} to {when(mine['fetched_until'])} (UTC+3) ({mine['readable']} " f"readable): {mine['hours_from_an_answer_to_the_fetch'][0]:.0f} to " f"{mine['hours_from_an_answer_to_the_fetch'][1]:.0f} hours after the answers that cite them. The four steps " f"on these answers ran from {when(pg['figure_chain_first_and_last_log_line'][0])} to " f"{when(pg['figure_chain_first_and_last_log_line'][1])} (`addendum/figures/chain.log`, the machine's time, " "UTC+3).", "", f"All five engines: {all5['distinct_addresses_cited']} distinct addresses; " f"{all5['addresses_cited_in_the_main_set_and_by_perplexity_or_claude']} of them are cited both by one of the " "first three engines and by Perplexity or Claude (API). " f"{all5['of_them_not_readable_in_the_main_set_and_fetched_again']} of those had not been readable in the fetch " f"of the main set and were fetched again; {all5['of_those_readable_in_the_later_fetch']} of the " f"{all5['of_them_not_readable_in_the_main_set_and_fetched_again']} were readable in the later fetch. Each answer " "is checked against the fetch of its own set.", ""] px, cx = pg["perplexity"], pg["claude"] notice, kept = px["records_with_the_preview_notice"], px["source_entries_whose_query_string_was_not_kept"] md += [f"Perplexity: {px['source_entries']} source entries (every entry of the source list the interface shows; a " f"median of {px['median_source_entries_per_answer']:g} per answer), {px['distinct_addresses']} distinct " "addresses. The records hold host and path without the query string: " f"{kept['n']} of {kept['of']} source entries had a query string that was not kept " f"({px['addresses_without_their_query_string']} addresses in {px['answers_with_such_an_address']} answers), and " "their pages are fetched without it, so the page read for them may differ from the page cited; " f"{part(px['found_figures_on_such_a_page'])} of the found figures were found on such a page and " f"{count(px['found_figures_only_on_such_pages']['n'], px['found_figures_only_on_such_pages']['of'])} only on " "such pages. No record ties a passage to a source address, so no figure has " f"an attached source. {px['citation_chip_lines_left_out_of_the_text']} citation chip lines " f"({px['site_name_chip_lines']} site-name lines and {px['of_them_counter_lines']} counter lines) are left out " f"of the answer text before the extraction. {notice['n']} of {notice['of']} records carry the interface's " "notice that a preview of the advanced search was switched on; the notes of " f"{px['records_whose_note_says_the_notice_could_not_be_observed']} further records say that the notice could " "not be observed. " f"Claude (API): {cx['source_entries']} source entries (the pages that a citation of the response names), " f"{cx['distinct_addresses']} distinct addresses; " f"{cx['answers_with_citations']['n']} of {cx['answers_with_citations']['of']} answers have citations " f"({cx['cited_blocks']} cited text blocks, {cx['citation_entries']} citation entries).", "", f"Time stamps: every one carries the offset {', '.join(fr['utc_offsets_of_the_timestamps'])} (UTC+3). " f"Main set, first and last answer: {fr['main_set_first_and_last_answer'][0]} and " f"{fr['main_set_first_and_last_answer'][1]}. Repeat set: {fr['repeat_set_first_and_last_answer'][0]} and " f"{fr['repeat_set_first_and_last_answer'][1]}.", ""] reg, own = w["registered"], ad["all_answers_with_figures"] for e in ENGINES: # the unrounded recomputation equals the frozen summary, and the five-engine view equals it assert round(reg[e]["median"], 4) == s[e]["median_share_per_answer"] assert [round(x, 4) for x in reg[e]["interval_95"]] == s[e]["median_interval_95"] assert own[e] == reg[e] md += ["## 2. Primary measure: share of an answer's figures found on a page the answer cites", "", "Each engine on its own answers with figures.", "", table(["Engine", "Answers with figures", "Median share per answer", "95% bootstrap interval"], [[NAMES[e], s[e]["answers"]["with_figures"], f"{own[e]['median']:.0%}", f"{own[e]['interval_95'][0]:.0%} to {own[e]['interval_95'][1]:.0%}"] for e in FIVE]), "", "The rows of Google AI Overviews, ChatGPT and Gemini are the primary measure as registered. Perplexity and " "Claude (API) were added after the preregistration; see the method. Each engine stands on its own questions here; the " "comparison on the same questions is in table S1.", ""] per_all, per = gemini_without_flagged() assert round(statistics.median(per_all.values()), 4) == s["gemini"]["median_share_per_answer"] ci = sv.interval(per) assert [round(x, 4) for x in ci] == sf.interval(per, random.Random(sf.SEED)) wo = ad["perplexity_without_SVC-07"] md += ["Gemini without the eight answers of block 1 that were saved after the block's last passing exit check " f"(deviation 1): {len(per)} answers, median {statistics.median(per.values()):.0%}, " f"interval {ci[0]:.0%} to {ci[1]:.0%}.", "", "Perplexity without its answer to SVC-07, which was read from the account's stored thread after an " f"interruption (deviation log, run record of 6 October 2026): {wo['answers']} answers, median " f"{wo['median']:.0%}, interval {wo['interval_95'][0]:.0%} to {wo['interval_95'][1]:.0%}, {part(wo['found'])} " f"figures found; with it, {own['perplexity']['answers']} answers, {med(own['perplexity'])}, " f"{cell(s['perplexity']['found']['n'], s['perplexity']['found']['of'])}. The answer holds " f"{wo['figures_of_the_answer']['of']} figures, {wo['figures_of_the_answer']['n']} of them found. SVC-07 is " + ("" if wo["among_the_five_engine_questions"] else "not ") + "among the questions of the like-for-like row of " "table S1" + ("" if wo["among_the_five_engine_questions"] else " (no figure in the answer of " + ", ".join(NAMES[x] for x in wo["registered_engines_without_a_figure_for_it"]) + "), so that row is the same with and without it") + ".", ""] def unknown(e): return sum(s[e]["outcomes"][k] for k in ("no_source", "access_failure", "instrument_gap")) def outcome(e, k): if not w["attached"][e]["available"] and k == "found_attached": return "not available" return count(s[e]["outcomes"][k], s[e]["figures"]) def total(label, engines): return ([label, sum(s[e]["figures"] for e in engines)] + [sum(s[e]["outcomes"][k] for e in engines) for k in sf.OUTCOMES] + [sum(unknown(e) for e in engines), sum(s[e]["found"]["n"] for e in engines)]) md += ["## 3. Outcomes of every figure", "", table(["Engine", "Figures", "Found on the attached page", "Found on another cited page", "Not found", "No source cited", "Page not readable", "Instrument gap", "Unknown: no source cited, page not readable or instrument gap", "Found on any cited page"], [[NAMES[e], s[e]["figures"]] + [outcome(e, k) for k in sf.OUTCOMES] + [count(unknown(e), s[e]["figures"]), cell(s[e]["found"]["n"], s[e]["found"]["of"])] for e in FIVE] + [total("All five engines", FIVE), total("Google AI Overviews, ChatGPT and Gemini", ENGINES)]), "", "Perplexity: its records tie no passage to a source address, so the attached view is not available and " "every found figure stands under \"found on another cited page\", which there means found on any cited page.", ""] views = [("Found on any cited page", "found", True), ("Found on the attached page", "found_attached", True), ("Found, without the unknowns", "found_without_unknowns", True), ("Found on the attached page, without the unknowns", "found_attached_without_unknowns", True), ("Found, in answers whose cited pages were all readable", "found_in_fully_readable_answers", True), ("Found, without figures whose marker names an unseen page", "found_without_unseen_markers", False), ("Found, without answers that hold a Sponsored label", "found_without_sponsored_answers", False)] def pooled(e, key, recorded): if e in ADDED and not recorded: return "not recorded" if "attached" in key and not w["attached"][e]["available"]: return "not available" return cell(s[e][key]["n"], s[e][key]["of"]) md += ["## 4. Pooled shares of figures, and the views beside them", "", table(["View"] + [NAMES[e] for e in FIVE], [[name] + [pooled(e, key, recorded) for e in FIVE] for name, key, recorded in views]), "", "The collections of Perplexity and Claude (API) recorded neither markers that name an unlisted page nor Sponsored " "labels. Answers whose saved page holds a \"Sponsored\" label: " + "; ".join( f"{NAMES[e]} {s[e]['answers_with_sponsored_label']['on_page']} on the page, " f"{s[e]['answers_with_sponsored_label']['in_answer']} inside the answer text" for e in ENGINES) + ".", ""] labels = ["absent", "partial", "rounded", "derived", "present", "unlabelled"] def label_total(label, engines): return ([label, sum(s[e]["outcomes"]["not_found"] for e in engines)] + [sum(s[e]["not_found_labels"][x] for e in engines) for x in labels + ["absent_quote_unconfirmed"]]) cl5 = fe["classifier_labels"] assert label_total("", FIVE)[1:] == [cl5["not_found"]] + [cl5[x] for x in labels + ["absent_quote_unconfirmed"]] md += ["## 5. The classifier's labels for the figures not found", "", table(["Engine", "Not found"] + [x.capitalize() for x in labels] + ["Absent only because the quote was not confirmed"], [[NAMES[e], s[e]["outcomes"]["not_found"]] + [s[e]["not_found_labels"][x] for x in labels] + [s[e]["not_found_labels"]["absent_quote_unconfirmed"]] for e in FIVE] + [label_total("All five engines", FIVE), label_total("Google AI Overviews, ChatGPT and Gemini", ENGINES)]), "", f"All five engines: {cl5['absent']} of {cl5['not_found']} figures that were not found are labelled absent, " f"{cl5['absent_quote_unconfirmed']} of them only because the quote was not confirmed.", ""] sec = w["sectors"] md += ["## 6. Found on any cited page, by sector", "", table(["Sector"] + [NAMES[e] for e in FIVE], [[SECTORS[k]] + [cell(s[e]["by_sector"][k]["found"], s[e]["by_sector"][k]["figures"]) for e in FIVE] for k in SECTORS]), "", f"A sector cell holds {sec['answers_per_cell']['fewest']} to {sec['answers_per_cell']['most']} answers with " "figures (table S5). The cells of Perplexity and Claude (API) were also counted from their check outputs and equal " "the cells of their summary.", ""] r, c, val = v["readings"], v["classifier"], w["validation"] md += ["## 7. Validation sample: 40 figures read by " + v["reader"] + " on " + v["read_on"], "", "The sample was drawn from the figures of Google AI Overviews, ChatGPT and Gemini, the three engines the " f"preregistration names. Figures of Perplexity or Claude (API) in the sample: " f"{pg['figures_of_the_added_engines_in_the_validation_sample']}.", "", table(["Stratum", "Reading", "Figures"], [["Found (20)", k.replace("_", " "), n] for k, n in r["found"].items()] + [["Not found (20)", k, n] for k, n in r["not_found"].items()]), "", f"Classifier and reader gave the same label to {c['same_label']} of {c['figures']} not-found figures, and agreed on " f"absent or not absent for {c['same_absent_or_not']} of {c['figures']}. Unlabelled: {c['unlabelled']}.", "", table(["Reader", "Classifier", "Figures"], [[t["reading"], t["classifier"], t["figures"]] for t in c["table"]]), "", table(["Stratum"] + [NAMES[e] for e in ENGINES], [[name] + [val["by_stratum_and_engine"][k].get(e, 0) for e in ENGINES] for name, k in (("Found (20)", "found"), ("Not found (20)", "not_found"))]), "", f"Readings with a quote that the script confirmed in the saved page: {part(val['readings_with_a_confirmed_quote'])}. " f"Readings without a quote: {val['readings_without_a_quote']['n']}, all read as " f"{', '.join(val['readings_without_a_quote']['readings'])}.", "", "Sampled figures the classifier called absent: " f"{val['sampled_figures_the_classifier_called_absent']['read_as_something_else']['of']}; the reader read " f"{val['sampled_figures_the_classifier_called_absent']['read_as_something_else']['n']} of " f"{val['sampled_figures_the_classifier_called_absent']['read_as_something_else']['of']} of them as something else (" + ", ".join(f"{k} {n}" for k, n in sorted(val["sampled_figures_the_classifier_called_absent"]["readings"].items())) + f"). These are all {val['sampled_figures_the_classifier_called_absent']['disagreements_in_the_not_found_sample']} " "disagreements of the not-found sample.", "", f"Found figures of the sample that carry no unit: {part(val['found_sample_number_only'])}, read as " + ", ".join(f"{k} ({n})" for k, n in val["found_sample_number_only_readings"].items()) + f". The label \"supports weakly\" was given to {r['found'].get('supports_weakly', 0)} figures.", ""] rp = json.loads((HERE / "work/summary/repeat.json").read_text(encoding="utf-8"))["engines"] ids = [x["id"] for x in rp["gemini"]["first_and_second_answer"]] pair = {e: {x["id"]: x for x in rp[e]["first_and_second_answer"]} for e in ENGINES} def show(x): return "no figures" if not x else cell(x[0], x[1]) rows = [[qid] + [c for e in ENGINES for c in (show(pair[e][qid]["first"]), show(pair[e][qid]["second"]))] for qid in ids] total = ["Questions with figures in both answers"] for e in ENGINES: both = [x for x in pair[e].values() if x["first"] and x["second"]] total += [cell(sum(x["first"][0] for x in both), sum(x["first"][1] for x in both)) + f", {len(both)} questions", cell(sum(x["second"][0] for x in both), sum(x["second"][1] for x in both))] md += ["## 8. Repeat run (Google AI Overviews, ChatGPT and Gemini): figures found in the first and in the second " "answer to the same question", "", "The first ten questions of the asking order, asked a second time of the three engines the preregistration " "names. Perplexity and Claude (API) were asked each question once. Not pooled with the main result. " "\"No figures\" also stands for a question without a complete answer.", "", table(["Question"] + [f"{NAMES[e]}, {x}" for e in ENGINES for x in ("first", "second")], rows + [total]), "", "Repeat answers by engine: " + "; ".join( f"{NAMES[e]} {rp[e]['questions']['ok']} answers, {rp[e]['figures']} figures, " f"{cell(rp[e]['found']['n'], rp[e]['found']['of'])} found" for e in ENGINES) + f". Complete second answers: {fr['repeat_answers_collected']['n']} of {fr['repeat_answers_collected']['of']}.", ""] return md def side_tables(s, w): ad = w["perplexity_and_claude"] md = ["# Views beside the primary measure", "", "Written from `report/side-views.json` (`report/side_views.py`). Additional computations: they stand beside " "the tables above and replace none of them. Every interval follows the registered rule (10,000 " "resamples of questions, seed 1109872877, percentile bounds).", ""] def answers(view, engines=FIVE): return ", ".join(str(view[e]["answers"]) for e in engines) def three(view): return [med(view[e]) for e in ENGINES] + ["not in the set"] * len(ADDED) own, com5, rb = ad["all_answers_with_figures"], ad["common_questions_five_engines"], ad["robustness"] uc5, cc5 = rb["figures_with_a_unit"], ad["common_citing_questions_five_engines"] com, cit, ful = w["common_questions"], w["citing_answers"], w["fully_readable"] cc, two, uc = w["common_citing_questions"], w["two_or_more_readable_pages"], w["units_common_questions"] uni = {e: w["units"][e]["with_unit_per_answer"] for e in FIVE} assert uc["ids"] == uc5["ids"] and all(uc[e] == uc5[e] for e in ENGINES) # the same 39 questions def apart(view): return view["intervals_of_google_aio_and_chatgpt"] gone = com5["of_them_not_among_the_five_engine_questions"] md += ["## S1. Median share per answer: the same questions for all engines, and other sets", "", "Like for like: the same questions for every engine of the row. Each engine on its own answers: the medians of " "a row are not compared with each other. Perplexity and Claude (API) were added after the preregistration; see the " "method.", "", table(["Set", "Answers"] + [NAMES[e] for e in FIVE], [[f"Like for like: the {com5['questions']} questions where all five answers state a figure", answers(com5)] + [med(com5[e]) for e in FIVE], [f"Like for like: figures that carry a unit only, on the {uc5['questions']} questions where all five answers " "hold one", answers(uc5)] + [med(uc5[e]) for e in FIVE], [f"Like for like: the {cc5['questions']} questions where all five answers state a figure and cite a page", answers(cc5)] + [med(cc5[e]) for e in FIVE], [f"Like for like, three engines: the {com['questions']} questions where the answers of Google AI Overviews, " "ChatGPT and Gemini state a figure", answers(com, ENGINES)] + three(com), [f"Like for like, three engines: the {cc['questions']} questions where those three answers state a figure " "and cite a page", answers(cc, ENGINES)] + three(cc), ["Each engine on its own answers: all answers with figures (table 2)", answers(own)] + [med(own[e]) for e in FIVE], ["Each engine on its own answers: answers that cite at least one page", answers(cit)] + [med(cit[e]) for e in FIVE], ["Each engine on its own answers: answers with two or more readable pages", answers(two)] + [med(two[e]) for e in FIVE], ["Each engine on its own answers: answers whose cited pages were all readable", answers(ful)] + [med(ful[e]) for e in FIVE], ["Each engine on its own answers: figures that carry a unit only", answers(uni)] + [med(uni[e]) for e in FIVE]]), "", "All figures. " + lying(com5["intervals"], com5["questions"]), "", "Figures that carry a unit only. " + lying(uc5["intervals"], uc5["questions"]) + f" The same {uc['questions']} questions are the ones where the answers of Google AI Overviews, ChatGPT and " "Gemini alone hold a figure with a unit.", "", "Answers that state a figure and cite a page. " + lying(cc5["intervals"], cc5["questions"]), "", f"Google AI Overviews and ChatGPT: on the {com5['questions']} questions of the five engines their intervals " f"{com5['intervals']['google_aio and chatgpt']}; on the {com['questions']} questions of the three-engine row they " f"{apart(com)}; on the {cc['questions']} questions they {apart(cc)}; for the figures that carry a unit, on the " f"{uc['questions']} questions, they {apart(uc)}. The {com5['questions']} questions are the {com['questions']} " "without " + ", ".join(gone) + ": " + "; ".join(f"{NAMES[e]} states no figure for {', '.join(v)}" for e, v in com5["added_engines_without_a_figure_there"].items() if v) + ". Figures found in the answers to those questions (" + ", ".join(NAMES[e] for e in ENGINES) + "): " + "; ".join(f"{q}: " + ", ".join(part(x[e]) for e in ENGINES) for q, x in com5["registered_answers_to_the_questions_that_leave"].items()) + ".", "", "Answers with figures that cite no page: " + "; ".join( f"{NAMES[e]} {', '.join(cit[e]['answers_with_figures_that_cite_no_page']) or 'none'}" for e in FIVE) + ".", ""] def prow(name, x): return [name, x["questions"], x["statistic"], f"{x['first_median']:.0%}", f"{x['second_median']:.0%}", f"{points(x['difference'])} points", span(x["interval_95"]), f"{x['first_higher']['n']} of {x['questions']}", f"{x['first_lower']['n']} of {x['questions']}", f"{x['equal']['n']} of {x['questions']}"] rows = [] for k, x in ad["paired_every_pair_on_the_five_engine_questions"].items(): a, b = k.split(" minus ") if x["difference"] < 0: a, b, x = b, a, flipped(x) rows.append(prow(f"{NAMES[a]} minus {NAMES[b]}", x)) for k, x in ad["paired_on_the_five_engine_questions"].items(): # the five pairs stored one way round equal them a, b = k.split(" minus ") assert prow(f"{NAMES[a]} minus {NAMES[b]}", x) in rows head = ["Pair", "Questions", "Statistic", "Median, first", "Median, second", "Difference", "95% interval of the difference", "First higher", "Lower", "Equal"] pr = w["paired"]["sets"] sets = [("common_questions", f"The {com['questions']} questions where the answers of Google AI Overviews, ChatGPT and " "Gemini state a figure"), ("both_answers_cite_a_page", "Questions where both answers state a figure and cite a page"), ("unit_bearing_figures", "Figures that carry a unit only, questions where both answers hold one"), ("both_answers_have_two_or_more_readable_pages", "Questions where both answers have two or more readable pages")] md += ["## S1b. Paired differences on the same questions", "", "The first engine minus the second on the same questions. Each resample draws one list of questions and uses " "it for both engines. Differences are in percentage points.", "", f"Every pair of the five engines on the {com5['questions']} questions where all five answers state a figure. " "Each pair is written with the engine of the higher median first.", "", table(head, rows), "", "Google AI Overviews minus ChatGPT on other sets of questions:", "", table(["Questions of the set"] + head[1:], [prow(name, pr[k]) for k, name in sets]), ""] pp, cut, by, band = w["pages_per_answer"], w["cut_to_k_pages"], w["share_by_readable_pages"], w["share_by_page_band"] def group(x, pairs=False): if not x["answers"]: return "no answer" y = x["figure_page_pairs"] if pairs else x return f"{part(y)}, {x['answers']} answer{'s' if x['answers'] > 1 else ''}" low = {e: by[e]["found_in_answers_with_at_most_one_readable_page"] for e in FIVE} rows = [["Median cited pages per answer"] + [f"{pp[e]['median_cited_pages']:g}" for e in FIVE], ["Median readable pages per answer"] + [f"{pp[e]['median_readable_pages']:g}" for e in FIVE], ["Answers with figures that have at most one readable page"] + [part(by[e]["answers_with_at_most_one_readable_page"]) for e in FIVE], ["Figures found in those answers"] + [group(low[e]) for e in FIVE], ["Of those answers, with a cited page that could not be read"] + [part(by[e]["at_most_one_readable_page_and_an_unreadable_cited_page"]) if low[e]["answers"] else "no answer" for e in FIVE]] for k, name in (("0", "no readable page"), ("1", "one readable page"), ("2", "two readable pages"), ("3 or more", "three or more readable pages")): rows += [[f"Answers with {name}: found"] + [group(by[e][k]) for e in FIVE]] rows += [["Answers with two or more readable pages: found"] + [group({**two[e]["pooled"], "answers": two[e]["answers"]}) for e in FIVE]] for k in ("3 to 4", "5 to 7", "8 or more"): rows += [[f"Answers with {k} readable pages: found"] + [group(band["readable_pages"][e][k]) for e in FIVE]] for k in ("1", "2", "3 to 4", "5 to 7", "8 or more"): rows += [[f"Answers that cite {k} page{'' if k == '1' else 's'}: found"] + [group(band["cited_pages"][e][k]) for e in FIVE]] md += ["## S2. Pages per answer, and figures found by the number of pages", "", "Each engine on its own answers with figures. Pooled counts of figures, with the number of answers of the cell.", "", table(["View"] + [NAMES[e] for e in FIVE], rows), ""] ps = w["pages_and_share"] md += ["Rank correlation of the number of pages of an answer with its share found (Spearman, with the interval of " "the registered resampling rule):", "", table(["Answers"] + [NAMES[e] for e in FIVE], [[name] + [corr(ps[k][e]) for e in FIVE] for name, k in ( ("One or more readable pages, by readable pages", "readable_pages, answers with at least 1"), ("Two or more readable pages, by readable pages", "readable_pages, answers with at least 2"), ("One or more cited pages, by cited pages", "cited_pages, answers with at least 1"), ("Two or more cited pages, by cited pages", "cited_pages, answers with at least 2"))]), "", "Two or more readable pages, by readable pages: the interval includes zero for " + ", ".join(NAMES[e] for e in FIVE if ps["readable_pages, answers with at least 2"][e]["interval_includes_zero"]) + " and does not include zero for " + (", ".join(NAMES[e] for e in FIVE if not ps["readable_pages, answers with at least 2"][e]["interval_includes_zero"]) or "no engine") + ".", "", "Citing answers by the number of pages they cite: " + "; ".join( f"{NAMES[e]} {part(ps['citing_answers_by_cited_pages'][e]['three_or_more'])} cite three or more and " f"{part(ps['citing_answers_by_cited_pages'][e]['one_or_two'])} one or two" for e in FIVE) + ".", ""] rows = [["(Figure, readable cited page) pairs in which the page holds the figure, pooled"] + [part(pp[e]["figure_page_pairs"]) for e in FIVE]] for k in ("1", "2", "3 to 4", "5 to 7", "8 or more"): rows += [[f"The same in answers with {k} readable page{'' if k == '1' else 's'}"] + [group(band["readable_pages"][e][k], pairs=True) for e in FIVE]] for k in ("2", "3"): rows += [[f"Each answer cut to {k} of its readable pages at random: expected share found, pooled over figures"] + [f"{cut[k][e]['pooled_share']:.0%}" for e in FIVE], ["The same, median per answer"] + [f"{cut[k][e]['median_share_per_answer']:.0%}" for e in FIVE], [f"Answers with fewer than {k} readable pages, which this cut leaves as they are"] + [part(cut[k][e]["answers_with_fewer_readable_pages"]) for e in FIVE]] rows += [["Each answer cut to 1 of its readable pages at random: expected share found, pooled over figures"] + [f"{cut['1'][e]['pooled_share']:.0%}" for e in FIVE], ["The same, median per answer"] + [f"{cut['1'][e]['median_share_per_answer']:.0%}" for e in FIVE]] md += ["## S2b. What one page contributes: the pair share and the cuts", "", table(["View"] + [NAMES[e] for e in FIVE], rows), "", "The pair share falls as lists grow, so it does not compare engines whose lists differ in length. " "The cut to k pages is an exact expectation: for a figure found on f of its answer's n readable pages, the " "chance that at least one of k pages drawn without replacement holds it. It is a pooled expectation, not a " "median of answers, and an answer with fewer than k readable pages keeps all of them. Each engine stands on " "its own answers here; the cut as a median per answer on the same questions is in table S6.", ""] pl, pl2 = w["placebo"], ad["placebo"]["added_pool"] sizes = pl["pool_pages_by_sector"].values() sizes2 = pl2["pool_pages_by_sector"].values() mp = rb["share_minus_placebo"] def exact(x): return f"{x['placebo_share_exact_expectation']:.0%}" rows = [] for name, g in (("All figures", "all"), ("Figures with a unit", "with_unit"), ("Figures matched on the number alone", "number_only")): rows += [[f"{name}: found on the cited pages"] + [part(pl[e][g]["actual"]) for e in FIVE], [f"{name}: placebo"] + [exact(pl[e][g]) for e in FIVE], [f"{name}: placebo, second pool"] + [exact(pl2[e][g]) for e in FIVE]] for k in SECTORS: rows += [[f"{SECTORS[k]}: found on the cited pages"] + [part(pl[e]["by_sector"][k]["actual"]) for e in FIVE], [f"{SECTORS[k]}: placebo"] + [exact(pl[e]["by_sector"][k]) for e in FIVE]] rows += [[f"Placebo per answer, median on the {com5['questions']} questions of table S1"] + [f"{mp['registered_pool']['median_placebo_per_answer'][e]:.0%}" for e in FIVE], [f"Placebo per answer, median on the {com5['questions']} questions of table S1, second pool"] + [f"{mp['added_pool']['median_placebo_per_answer'][e]:.0%}" for e in FIVE]] md += ["## S3. Placebo: pages cited for questions of other sectors", "", "Each answer's pages are replaced by the same number of readable pages cited for questions of other sectors: " "an answer draws as many pages as it has readable pages. The placebo share is the exact expectation of the " "part of the figures the check then finds, over every choice of such pages. The pool, the same for all five " "engines: the readable pages of the main set, without the pages that Google AI Overviews, ChatGPT or Gemini " f"cite for a question of the answer's sector ({min(sizes)} to {max(sizes)} pages per sector). Pages of this " "pool that Perplexity or Claude (API) cite for a question of the answer's sector: " f"{sum(ad['placebo']['registered_pool']['pool_pages_an_added_engine_cites_in_the_sector'].values())}. Second " "pool: the readable pages cited by Perplexity or Claude (API), without the pages either of them cites for a question " f"of the answer's sector ({min(sizes2)} to {max(sizes2)} pages per sector).", "", table(["Figures"] + [NAMES[e] for e in FIVE], rows), "", f"A check of the expectation by {pl['draws_per_answer']} seeded draws per answer, all figures: " + ", ".join( f"{NAMES[e]} {pl[e]['all']['placebo_share']:.1%} drawn against {pl[e]['all']['placebo_share_exact_expectation']:.1%}" for e in FIVE) + ". The match that computes the placebo was compared with the check outputs, figure by " f"figure, for all five engines ({ad['placebo']['added_figures_compared_with_the_check_outputs']} figures of " "Perplexity and Claude (API)): no difference.", ""] un, at, nr = w["units"], w["attached"], w["not_on_a_readable_page"] def attached(e, key): return part(at[e][key]) if at[e]["available"] else "not available" md += ["## S4. Units, attached sources, and figures not on a readable page", "", table(["View"] + [NAMES[e] for e in FIVE], [["Figures with a unit: found"] + [part(un[e]["with_unit"]) for e in FIVE], ["Figures without a unit (matched on the number alone): found"] + [part(un[e]["number_only"]) for e in FIVE], ["Found figures that were matched on the number alone"] + [part(un[e]["found_figures_matched_on_the_number_alone"]) for e in FIVE], ["Figures whose sentence has no source attached"] + [attached(e, "no_source_attached_to_the_sentence") for e in FIVE], ["The same, within the answers that cite a page"] + [attached(e, "the_same_in_answers_that_cite_a_page") for e in FIVE], ["Where a source is attached: found on it"] + [attached(e, "found_on_the_attached_source_where_one_is_attached") for e in FIVE], ["Found on another cited page: sentence has no source attached"] + [attached(e, "found_elsewhere_with_no_source_attached") for e in FIVE], ["Not found, in answers that cite a page (not found plus page not readable)"] + [part(nr[e]) for e in FIVE], ["Figures of answers that cite no source"] + [part(nr[e]["in_answers_that_cite_no_source"]) for e in FIVE], ["Figures the check could not read"] + [nr[e]["figures_the_check_could_not_read"] for e in FIVE], ["Every figure that was not found (the three rows above)"] + [part(nr[e]["every_figure_that_was_not_found"]) for e in FIVE], ["Answers with figures that cite at least one unreadable page"] + [part(nr[e]["answers_with_an_unreadable_page"]) for e in FIVE]]), "", f"All five engines: {part(un['five_engines']['found_figures_matched_on_the_number_alone'])} of the found figures " f"were matched on the number alone; {part(un['five_engines']['number_only'])} of the figures the check could read " f"carry no unit. Google AI Overviews, ChatGPT and Gemini: " f"{part(un['all_engines']['found_figures_matched_on_the_number_alone'])} and {part(un['all_engines']['number_only'])}.", ""] sec = w["sectors"] md += ["## S5. Sector cells: answers, and figures of answers that cite no source", "", table(["Sector"] + [f"{NAMES[e]}, {x}" for e in FIVE for x in ("answers with figures", "figures in answers that cite no source")], [[SECTORS[k]] + [c for e in FIVE for c in ( sec[e][k]["answers_with_figures"], part(sec[e][k]["figures_in_answers_that_cite_no_source"]) + (f" ({', '.join(sec[e][k]['answers_that_cite_no_source'])})" if sec[e][k]["answers_that_cite_no_source"] else ""))] for k in SECTORS]), "", f"A sector cell holds {sec['answers_per_cell']['fewest']} to {sec['answers_per_cell']['most']} answers with figures.", ""] def in_points(x): return (f"{100 * x['median']:.0f} points ({100 * x['interval_95'][0]:.0f} to " f"{100 * x['interval_95'][1]:.0f} points)") al, ct = rb["all_figures"], rb["cut_to_two_readable_pages"] pools = [("registered_pool", ""), ("added_pool", ", second pool")] views = [("All figures (table S1)", al, med), ("Figures with a unit only, on the questions where all five answers hold one", uc5, med), ("Each answer cut to 2 of its readable pages at random: expected share found, median per answer", ct, med)] + [ (f"Share found minus the answer's placebo{label}", mp[k], in_points) for k, label in pools] fewer = ct["answers_with_fewer_readable_pages"] md += ["## S6. The like-for-like comparison in four views", "", "Medians per answer with the registered interval, on the same questions for all five engines. The share minus " "the placebo is the answer's share found minus the mean placebo expectation of its figures (table S3), in " "percentage points.", "", table(["View", "Questions"] + [NAMES[e] for e in FIVE], [[name, x["questions"]] + [show(x[e]) for e in FIVE] for name, x, show in views]), ""] md += [f"{name.split(' (')[0].split(':')[0]}. " + lying(x["intervals"], x["questions"]) for name, x, _ in views] md[-len(views):] = [y for x in md[-len(views):] for y in (x, "")] md += ["The cut is an exact expectation over every choice of two pages, and an answer with fewer than two readable " f"pages keeps all of them and is not cut: on the {ct['questions']} questions, " + ", ".join(f"{NAMES[e]} {fewer[e]['n']} of {fewer[e]['of']}" for e in FIVE) + " answers.", ""] def crow(name, x, a, statistic): return [name, f"{NAMES[a]} minus ChatGPT", x["questions"], statistic, f"{points(x['difference'])} points", span(x["interval_95"]), f"{x['first_higher']['n']} of {x['questions']}", f"{x['first_lower']['n']} of {x['questions']}", f"{x['equal']['n']} of {x['questions']}"] others = [e for e in FIVE if e != "chatgpt"] rows = [] for name, x in [("All figures", al), ("Figures with a unit only", uc5), ("Each answer cut to 2 of its readable pages", ct)] + [ (f"Share found minus the answer's placebo{label}", mp[k]) for k, label in pools]: rows += [crow(name, x["paired_against_chatgpt"][a], a, "difference of the medians") for a in others] both = rb["both_answers_have_two_or_more_readable_pages"] rows += [crow("Questions where both answers have two or more readable pages", both[a], a, "mean of the differences") for a in others] md += ["Each engine against ChatGPT, the engine with the fewest pages per answer, on the same questions. Each resample " "draws one list of questions and uses it for both engines. Differences are in percentage points.", "", table(["View", "Pair", "Questions", "Statistic", "Difference", "95% interval of the difference", "First higher", "Lower", "Equal"], rows), ""] return md def source_tables(o, s5): """o: addendum/analysis/source-overlap.json; s5: the key five_engines.sources of report/side-views.json.""" e5, four = FIVE, FIVE[:4] notes = o["collection_notes"] md = ["# Tables of the sources the engines cite", "", "Written from `addendum/analysis/source-overlap.json` (`addendum/analysis/source_overlap.py`) and, for the " "views of all five engines that file does not hold, from the key `five_engines.sources` of " "`report/side-views.json` (`report/side_views.py`, with the functions of the first script). " f"Page key: {o['page_rule']}. The analysis of sources is exploratory: the preregistration does not contain it " "(deviation log, entry 3).", "", "## A1. Sources per engine", "", table(["Engine", "Answers", "Answers with sources", "Page citations", "Distinct pages", "Distinct domains", "Median pages per answer", "Answers with sources a marker names and the record does not list", "Such sources", "Answers that cite YouTube", "Distinct YouTube pages"], [[NAMES[e], o["engines"][e]["answers"], o["engines"][e]["answers_with_sources"], o["engines"][e]["page_citations"], o["engines"][e]["distinct_pages"], o["engines"][e]["distinct_domains"], f"{o['engines'][e]['median_pages_per_answer']:g}", notes[e].get("answers_with_unlisted_sources", "not recorded"), notes[e].get("unlisted_source_mentions", "not recorded"), o["engines"][e]["answers_citing_youtube"], o["engines"][e]["distinct_youtube_pages"]] for e in e5]), "", f"All five engines: {s5['answers_with_sources']} of {s5['answers']} answers cite at least one source, on " f"{s5['distinct_pages']} distinct pages. The four engines without Claude (API): " f"{o['four_engines']['answers_with_sources']} of {o['four_engines']['answers']} answers.", "", f"Google AI Overviews, ChatGPT and Gemini: {notes['registered_engines']['distinct_pages_by_the_page_key']} " "distinct pages by this key. The check of figures counts the same answers' sources as " f"{notes['registered_engines']['distinct_addresses_as_cited']} addresses, each address as it was cited, with its " "query string.", ""] c, fs, fs5 = notes["chatgpt"], notes["lists_fully_seen"], s5["lists_fully_seen"] with_sources = c["answers_with_sources"] md += [f"ChatGPT: {c['answers_without_sources']} answers cite no source (" + ", ".join(x["id"] for x in c["answers_without_sources_detail"]) + "); their saved pages hold " f"{sum(x['sources_controls'] for x in c['answers_without_sources_detail'])} \"Sources\" controls and " f"{sum(x['tagged_outbound_links'] for x in c['answers_without_sources_detail'])} outbound links, which " f"indicates that ChatGPT answered them without a search. In {c['answers_with_unlisted_sources']} of the other {with_sources} a marker " f"names more sources than the panel listed ({c['unlisted_source_mentions']} mentions). Gemini: " f"{notes['gemini']['answers_with_unlisted_sources']} answers hold markers that name " f"{notes['gemini']['unlisted_source_mentions']} sources the record has no address for.", "", f"Lists fully seen (sources cited, and no marker names an unlisted source): ChatGPT {fs['chatgpt_answers']} " f"answers, Gemini {fs['gemini_answers']}, Google AI Overviews {fs['google_aio_answers']}. ChatGPT's list is " f"fully seen in {fs5['five_engine_questions_with_the_chatgpt_list_fully_seen']} of the " f"{fs5['five_engine_questions']} five-engine questions, and both ChatGPT's and Gemini's in " f"{fs5['five_engine_questions_with_chatgpt_and_gemini_lists_fully_seen']} of the {fs5['five_engine_questions']} " f"(four engines: {fs['four_engine_questions_with_chatgpt_and_gemini_lists_fully_seen']} of the " f"{fs['four_engine_questions']}).", ""] p, cl = notes["perplexity"], notes["claude"] md += [f"Perplexity: {p['answers']} answers; interface language " + ", ".join(f"{k} ({n})" for k, n in p["interface_lang"].items()) + "; account " + ", ".join(f"{k} ({n})" for k, n in p["account"].items()) + f"; {p['pro_preview_notice'].get('True', 0)} of {p['answers']} records carry the notice that a preview of the " f"advanced search was switched on, and {p['records_whose_note_says_the_preview_notice_could_not_be_observed']} " f"further records note that it could not be observed reliably. The records hold {p['source_entries']} source entries; " f"{p['entries_that_repeat_a_page_of_the_same_answer']} of them repeat a page of the same answer under the page " f"key, which leaves {o['engines']['perplexity']['page_citations']} page citations. Claude (API): model " + ", ".join(f"{k} ({n})" for k, n in cl["model"].items()) + "; location setting " + ", ".join(f"{k} ({n})" for k, n in cl["location_setting"].items()) + "; searches allowed per answer " + ", ".join(cl["max_searches_allowed"]) + f"; {cl['searches'].get('1', 0)} of {cl['answers']} answers ran one " f"search, {cl['searches'].get('0', 0)} ran none, and none ran more: " + ", ".join(f"{k} in {n} answers" for k, n in cl["searches"].items()) + ".", ""] def orow(name, block, means=True): p, d = block["pages"], block["domains"] return [name, p["questions"], one(p["mean_jaccard"]) if means else "counts only", one(p["median_jaccard"]) if means else "", f"{p['questions_with_a_shared_source']} of {p['questions']}", f"{p['pooled_shared']} of {p['pooled_union']}", one(d["mean_jaccard"]) if means else "counts only", one(d["median_jaccard"]) if means else "", f"{d['questions_with_a_shared_source']} of {d['questions']}", f"{d['pooled_shared']} of {d['pooled_union']}"] wo = o["without_chatgpt"] seen5 = {"pages": s5["pages_chatgpt_list_fully_seen"], "domains": s5["domains_chatgpt_list_fully_seen"]} assert s5["pages"]["questions"] == s5["pages_without_interrupted_question"]["questions"] md += ["## A2. Overlap of the sources cited for the same question", "", "Jaccard over all engines of the row, on the questions where each of them cites at least one source: the " "mean and the median of the per-question ratios. Pooled: shared sources of all questions over the sources of " "all questions. A row of 11 questions gives counts only: one question holds its shared page. The row of four " "engines without Claude (API) is the view comparable with the two earlier rounds of the Cross-Engine Citation " "Study, which had those four engines.", "", table(["Engines", "Questions", "Mean overlap, pages", "Median, pages", "Questions with a shared page", "Pooled, pages", "Mean overlap, domains", "Median, domains", "Questions with a shared domain", "Pooled, domains"], [orow("All five: Google AI Overviews, ChatGPT, Gemini, Perplexity, Claude (API)", s5), orow("All five, only questions whose ChatGPT source list was fully seen", seen5, means=False), orow("Four without ChatGPT: Google AI Overviews, Gemini, Perplexity, Claude (API)", s5["without_chatgpt"]), orow("Four without Claude (API): Google AI Overviews, ChatGPT, Gemini, Perplexity", o["four_engines"]), orow("The same four, only questions whose ChatGPT source list was fully seen", o["four_engines_fully_seen_chatgpt_lists"], means=False), orow("Google AI Overviews, Gemini, Perplexity", wo), orow("Google AI Overviews, Gemini, Perplexity, without Perplexity's question SVC-07", {"pages": wo["pages_without_interrupted_question"], "domains": wo["domains_without_interrupted_question"]}), orow("Google AI Overviews, Gemini, Perplexity, only questions whose Gemini source list was fully seen", wo["gemini_lists_fully_seen"]), orow("Google AI Overviews, ChatGPT, Gemini", o["three_registered_engines"])]), "", f"The five-engine row holds the {o['four_engines']['pages']['questions']} questions of the four-engine row " "without " + ", ".join(s5["questions_of_the_four_engine_row_that_are_not_in_the_five_engine_row"]) + ", for which Claude (API) cites no source. Both rows without Perplexity's question SVC-07: " f"{s5['pages_without_interrupted_question']['questions']} and " f"{o['four_engines']['pages_without_interrupted_question']['questions']} questions, unchanged (ChatGPT cites " "no source for that question).", ""] def brow(name, x, n): ks = [str(k) for k in range(1, n + 1)] return ([name, x["questions"], x["sources_counted_per_question"]] + [count(x["cited_by_this_many_engines"][k]["n"], x["cited_by_this_many_engines"][k]["of"]) for k in ks] + [f"{x['questions_where_at_least_this_many_engines_share_a_source'][k]['n']} of {x['questions']}" for k in ks[1:]] + [x["distinct_sources_over_these_questions"]]) md += ["## A2b. Sources by the number of engines that cite them", "", "Counted per question: a source cited for two questions is counted twice. All five engines, on the questions " "of the five-engine row:", "", table(["Level", "Questions", "Sources, counted per question", "Cited by one engine", "By two", "By three", "By four", "By all five", "Questions where at least two engines share a source", "At least three", "At least four", "All five", "Distinct sources over these questions"], [brow("Pages", s5["pages_by_engines"], 5), brow("Domains", s5["domains_by_engines"], 5)]), "", "The four engines without Claude (API), on the questions of the four-engine row:", "", table(["Level", "Questions", "Sources, counted per question", "Cited by one engine", "By two", "By three", "By all four", "Questions where at least two engines share a source", "At least three", "All four", "Distinct sources over these questions"], [brow("Pages", o["four_engines"]["pages_by_engines"], 4), brow("Domains", o["four_engines"]["domains_by_engines"], 4)]), ""] rows = [] for block, label in ((s5, "All five engines"), (s5["without_chatgpt"], "Four engines without ChatGPT"), (o["four_engines"], "Four engines without Claude (API)"), (o["without_chatgpt"], "Google AI Overviews, Gemini and Perplexity")): for level in ("pages", "domains"): cg = block[f"ceiling_{level}"] rows.append([f"{label}, {level}", cg["questions"], one(block[level]["mean_jaccard"]), one(cg["mean_shortest_over_longest"]), one(cg["mean_shortest_over_observed_union"]), f"{cg['questions_where_the_shortest_list_has_one_source']} of {cg['questions']}"] + [f"{cg['questions_where_the_engine_has_the_shortest_list'][e]} of {cg['questions']}" if e in cg["questions_where_the_engine_has_the_shortest_list"] else "not in the row" for e in e5]) md += ["## A3. What the list sizes allow: the ceiling of the all-engine overlap", "", "The Jaccard of several lists cannot exceed the shortest list divided by the longest. Means over the questions " "of the row. An engine has the shortest list when no other list is shorter (ties count for each).", "", table(["Row", "Questions", "Observed mean overlap", "Highest mean overlap the list sizes allow", "The same, given the observed union", "Questions where the shortest list has one source"] + [f"Shortest list: {NAMES[e]}" for e in e5], rows), ""] pw_p, pw_d = s5["pairwise_pages"], s5["pairwise_domains"] cp, cd = o["containment_pages"], o["containment_domains"] def inside(block, a, b): x = block[f"{a} in {b}"] return cell(x["shared"], x["own"]) rows = [] for k in pw_d: a, b = k.split("+") rows.append([f"{NAMES[a]} and {NAMES[b]}", pw_d[k]["questions"], one(pw_p[k]["mean_jaccard"]), one(pw_d[k]["mean_jaccard"]), pw_p[k]["questions_with_a_shared_source"], pw_d[k]["questions_with_a_shared_source"], inside(cp, a, b), inside(cp, b, a), inside(cd, a, b), inside(cd, b, a)]) md += ["## A4. Overlap by pair of engines: Jaccard and containment", "", "All ten pairs of the five engines. Containment: of the sources the first engine cites, the share the second " "also cites for the same question, pooled over the questions where both cite at least one source. \"First\" " "and \"second\" follow the order of the pair's name.", "", table(["Pair", "Questions", "Mean overlap, pages", "Mean overlap, domains", "Questions with a shared page", "Questions with a shared domain", "Pages of the first also cited by the second", "Pages of the second also cited by the first", "Domains of the first also cited by the second", "Domains of the second also cited by the first"], rows), ""] lo, hi = s5["pairwise_pages_lowest_and_highest_mean"] dlo, dhi = s5["pairwise_domains_lowest_and_highest_mean"] cq = o["containment_on_common_questions"] def common(level, e): x = cq[level][e] return (f"{NAMES[e]}, {level}, {x['questions']} questions: " f"{cell(x[f'{e} in google_aio']['shared'], x[f'{e} in google_aio']['own'])} also cited by Google AI Overviews, " f"{cell(x[f'{e} in perplexity']['shared'], x[f'{e} in perplexity']['own'])} by Perplexity") md += [f"Mean overlap of two of the five engines, over the ten pairs: pages, lowest {one(lo)}, highest {one(hi)}; " f"domains, lowest {one(dlo)}, highest {one(dhi)}.", "", "Containment on one set of questions (the questions where the engine, Google AI Overviews and Perplexity all " "cite at least one source): " + "; ".join(common(level, e) for level in ("domains", "pages") for e in ("chatgpt", "claude")) + ".", ""] ry = o["repeat_yardstick_pages"] same, cross = ry["same_engine_two_askings"], ry["cross_engine_pairs_same_questions"] md += ["## A5. Yardstick (Google AI Overviews, ChatGPT and Gemini): the same engine asked the same question twice (pages)", "", "The repeat set, which exists for the three engines the preregistration names: the first ten questions of the " "asking order (" + ", ".join(ry["questions_of_the_repeat_set"]) + "). A row holds the questions where both lists cite at least one page.", "", table(["Comparison", "Questions", "Mean overlap, pages", "Median", "Questions with a shared page", "Questions without a shared page", "Pages of the first answer cited again"], [[f"{NAMES[e]}, first and second answer", same[e]["questions"], one(same[e]["mean_jaccard"], same[e]["questions"]), one(same[e]["median_jaccard"], same[e]["questions"]), f"{same[e]['questions_with_a_shared_source']} of {same[e]['questions']}", f"{same[e]['questions_without_a_shared_page']['n']} of {same[e]['questions']}", cell(same[e]["first_answer_pages_cited_again"], same[e]["first_answer_pages"])] for e in ENGINES] + [[" and ".join(NAMES[x] for x in k.split("+")) + ", first answers", cross[k]["questions"], one(cross[k]["mean_jaccard"], cross[k]["questions"]), one(cross[k]["median_jaccard"], cross[k]["questions"]), f"{cross[k]['questions_with_a_shared_source']} of {cross[k]['questions']}", f"{cross[k]['questions'] - cross[k]['questions_with_a_shared_source']} of {cross[k]['questions']}", "not applicable"] for k in cross]), "", "A mean or median over fewer than 10 questions is given as a whole percentage.", ""] lk = ry["like_for_like"] rows = [] for k, x in lk["pairs"].items(): a, b = k.split("+") n = x["questions"] rows.append([f"{NAMES[a]} and {NAMES[b]}", n, one(x["two_engines_first_answers"]["mean_jaccard"], n), one(x["two_engines_first_answers"]["median_jaccard"], n), one(x["same_engine_two_askings"][a]["mean_jaccard"], n), one(x["same_engine_two_askings"][a]["median_jaccard"], n), one(x["same_engine_two_askings"][b]["mean_jaccard"], n), one(x["same_engine_two_askings"][b]["median_jaccard"], n), part(x["questions_where_the_two_engines_share_less_than_either_shares_with_itself"])]) md += ["## A5b. The yardstick like for like (pages)", "", "For a pair of engines: the questions of the repeat set where both engines cite at least one page in both " "askings. The two engines' first answers stand beside each engine's own two answers on the same questions.", "", table(["Pair (first and second)", "Questions", "Two engines: mean overlap", "Two engines: median", "First engine asked twice: mean", "First engine asked twice: median", "Second engine asked twice: mean", "Second engine asked twice: median", "Questions where the two engines share less than either shares with itself"], rows), "", f"Across the three pairs the mean overlap of two engines runs from {one(lk['two_engines_lowest_and_highest_mean'][0], 0)} " f"to {one(lk['two_engines_lowest_and_highest_mean'][1], 0)}, and the mean overlap of one engine asked twice from " f"{one(lk['same_engine_lowest_and_highest_mean'][0], 0)} to {one(lk['same_engine_lowest_and_highest_mean'][1], 0)}.", ""] mc, m4 = s5["most_cited"], o["four_engines"]["most_cited"] by = mc["domains_by_engines_reached"] md += ["## A6. Most cited domains, five engines", "", f"{mc['distinct_domains']} distinct domains. Cited by one engine only: {by['1']} " f"({mc['share_cited_by_one_engine_only']:.0%}); by two: {by['2']}; by three: {by['3']}; by four: {by['4']}; " f"by all five: {by['5']} ({', '.join(mc['cited_by_every_engine'])}). " "Hosts counted under one platform domain: " + ", ".join(f"{d} {n}" for d, n in m4["hosts_under_platform_domains"].items()) + ".", "", table(["Domain", "Answers citing it", "Engines"] + [NAMES[e] for e in e5], [[t["domain"], t["answers"], t["engines"]] + [t[e] for e in e5] for t in mc["top"]]), "", f"The four engines without Claude (API): {m4['distinct_domains']} distinct domains. Cited by one engine only: " f"{m4['domains_by_engines_reached']['1']} ({m4['share_cited_by_one_engine_only']:.0%}); by two: " f"{m4['domains_by_engines_reached']['2']}; by three: {m4['domains_by_engines_reached']['3']}; by all four: " f"{m4['domains_by_engines_reached']['4']} ({', '.join(m4['cited_by_every_engine'])}).", ""] return md def main(): s = json.loads((HERE / "work/summary/main.json").read_text(encoding="utf-8"))["engines"] v = json.loads((HERE / "work/validation/main/validation.json").read_text(encoding="utf-8")) w = json.loads((HERE / "report/side-views.json").read_text(encoding="utf-8")) o = json.loads((HERE / "addendum/analysis/source-overlap.json").read_text(encoding="utf-8")) a = json.loads((HERE / "addendum/figures/summary.json").read_text(encoding="utf-8")) assert not a["partial"] and w["perplexity_and_claude"]["common_questions_five_engines"]["questions_of_the_three_registered_engines"] \ == w["common_questions"]["questions"] assert not set(s) & set(ADDED) and w["five_engines"]["engines"] == FIVE s.update({e: a["engines"][e] for e in ADDED}) # the same summary, written by the frozen function md = main_tables(s, v, w) + side_tables(s, w) + source_tables(o, w["five_engines"]["sources"]) out = HERE / "report" / "tables-main.md" out.write_text("\n".join(md), encoding="utf-8") print("written:", out.relative_to(HERE), "\n".join(md).count("\n") + 1, "lines") if __name__ == "__main__": main()