#!/usr/bin/env python3 """ ai-answer-evidence, added analysis (deviation 3): which sources the engines cite and how far the engines share them. Exploratory, outside the registration; it reads the collected records and changes nothing. Engines: Google AI Overviews, ChatGPT and Gemini (the registered collection, raw//main), Perplexity (addendum/perplexity) and Claude through the API (addendum/claude). The headline comparison uses the first four; Claude is reported apart because its access route differs. Source lists are incomplete in part. Some ChatGPT answers cite no source at all: the saved page holds no "Sources" control and no outbound link, which indicates an answer without a search. In other ChatGPT answers, and in Gemini answers, a citation marker names more sources than the record has an address for. The four-engine figures are also given without ChatGPT, and for the questions whose ChatGPT answer has a fully seen list. A page is host + path: the host in lower case without "www.", the path without a trailing slash, query and fragment dropped, with one exception: the video id of a youtube.com/watch address is kept, because there the query is the page. Perplexity's records hold no query; none of its sources is a YouTube address, so every engine is reduced to the same key. A domain is the registrable domain of the host (a short rule for two-part suffixes such as co.uk; no public-suffix list is used). Beside the Jaccard index the script writes: ceiling the highest all-engine Jaccard the list sizes allow (shortest list / longest list) containment for an ordered pair of engines, the share of the first engine's sources that the second also cites for the same question (pooled over questions) repeat the same engine asked the same question twice (the registered repeat set): the overlap of its two source lists, next to the cross-engine pairs on those questions, and the same comparison on the questions both rows share (like for like) by engines per question, every cited source counted by the number of engines that cite it Means, medians and shares are stored unrounded, so a table rounds each value once. Writes addendum/analysis/source-overlap.json and prints a short summary. Usage (from the study folder): python3 addendum/analysis/source_overlap.py """ import collections, itertools, json, pathlib, re, statistics, urllib.parse HERE = pathlib.Path(__file__).resolve().parents[2] OUT = pathlib.Path(__file__).resolve().parent / "source-overlap.json" REGISTERED = ["google_aio", "chatgpt", "gemini"] FOUR = REGISTERED + ["perplexity"] FIVE = FOUR + ["claude"] SECOND_LEVEL = {"co", "com", "org", "net", "gov", "ac", "edu"} INTERRUPTED = "SVC-07" # Perplexity's question 33: asked seconds before the exit dropped FULLY_SEEN = set() # ChatGPT answers whose sources panel listed every source the markers name FULL = {e: set() for e in REGISTERED} # answers with sources whose markers name no unlisted source REPEAT = 10 # the repeat set: the first ten questions of the asking order PLATFORMS = ["medium.com", "substack.com"] # one domain, many authors PAGE_RULE = ("host in lower case without www. + path without a trailing slash; query and fragment dropped, " "except the video id (v) of a youtube.com/watch address") def questions(): rows = (HERE / "questions-main-v1.tsv").read_text(encoding="utf-8").splitlines()[1:] return [dict(zip(["order", "id", "sector", "question"], r.split("\t"))) for r in rows] def domain(host): parts = host.split(".") if len(parts) >= 3 and len(parts[-1]) == 2 and parts[-2] in SECOND_LEVEL: return ".".join(parts[-3:]) return ".".join(parts[-2:]) def is_watch(host, path): return (host == "youtube.com" or host.endswith(".youtube.com")) and (path or "").rstrip("/") == "/watch" def page_key(host, path, query=""): """(page, domain, host). The query is dropped, except the video id of a YouTube watch address.""" host = host.lower().strip() if host.startswith("www."): host = host[4:] page = host + (path or "").rstrip("/") if is_watch(host, path): video = urllib.parse.parse_qs(query or "").get("v") if video: page += "?v=" + video[0] return page, domain(host), host def from_url(url): u = urllib.parse.urlsplit(url) return page_key(u.netloc.split("@")[-1].split(":")[0], u.path, u.query) if u.netloc else None def answer(keys): return {"pages": {k[0] for k in keys}, "domains": {k[1] for k in keys}, "hosts": {k[2] for k in keys}} def registered_answer(d): urls = [s.get("url") for s in d.get("sources") or []] urls += [c.get("url") for c in d.get("inline_citations") or []] return answer({k for k in (from_url(u) for u in urls if u) if k}) def unlisted(d): """Sources a citation marker names and the record has no address for.""" return sum(int(c.get("unseen_pages") or 0) for c in d.get("inline_citations") or []) def saved_page_marks(engine, which, d): """What the saved page of an answer holds: "Sources" controls and outbound links that carry ChatGPT's tag.""" f = HERE / "raw" / engine / which / f"{d['id']}.a{d['attempt']}.html" html = f.read_text(encoding="utf-8", errors="replace") if f.exists() else "" return {"sources_controls": len(re.findall(r">Sources<", html)), "tagged_outbound_links": len(re.findall(r"utm_source=chatgpt\.com", html))} def load(): """engine -> question id -> {"pages": set, "domains": set}, for complete answers only.""" data = {e: {} for e in FIVE} notes = {} addresses = set() for e in REGISTERED: seen, without, reference = collections.Counter(), [], None for f in sorted((HERE / "raw" / e / "main").glob("???-??.json")): d = json.loads(f.read_text(encoding="utf-8")) if d["status"] != "ok": continue data[e][d["id"]] = registered_answer(d) addresses.update(x["url"] for x in (d.get("sources") or []) + (d.get("inline_citations") or []) if x.get("url")) n = unlisted(d) seen["answers"] += 1 seen["answers_with_sources"] += bool(data[e][d["id"]]["pages"]) seen["answers_with_unlisted_sources"] += bool(n) seen["unlisted_source_mentions"] += n if data[e][d["id"]]["pages"] and not n: FULL[e].add(d["id"]) if e == "chatgpt": if not d.get("sources") and not d.get("inline_citations"): without.append({"id": d["id"], **saved_page_marks(e, "main", d), "note_panel_not_open": any("sources panel was not open" in x for x in d.get("notes") or [])}) elif reference is None: reference = {"id": d["id"], **saved_page_marks(e, "main", d)} notes[e] = dict(seen) if e == "chatgpt": # No "Sources" control and no outbound link on the saved page: ChatGPT answered without a search. notes[e]["answers_without_sources"] = len(without) notes[e]["answers_without_sources_detail"] = without notes[e]["saved_page_of_a_sourced_answer_for_reference"] = reference FULLY_SEEN.update(FULL["chatgpt"]) # The same answers as addresses (the unit of the registered check) and as pages by this script's key. notes["registered_engines"] = { "distinct_addresses_as_cited": len(addresses), "distinct_pages_by_the_page_key": len(set().union(*[a["pages"] for e in REGISTERED for a in data[e].values()]))} for e in ("perplexity", "claude"): fields, no_query, entries, unsure = collections.defaultdict(collections.Counter), 0, 0, 0 for f in sorted((HERE / "addendum" / e / "records").glob("*.json")): d = json.loads(f.read_text(encoding="utf-8")) if d["status"] != "complete": continue keys = set() for s in d.get("sources") or []: key = from_url(s["url"]) if s.get("url") else page_key(s["host"], s.get("path")) assert key[0].split("?")[0] == page_key(s["host"], s.get("path"))[0] # the record's own host and path # A watch address stored without its query cannot be keyed by video; none occurs. no_query += is_watch(key[2], s.get("path")) and "?v=" not in key[0] keys.add(key) data[e][d["id"]] = answer(keys) entries += len(d.get("sources") or []) # A record whose note says that the field stands for "not observed", not for "not shown". unsure += d.get("pro_preview_notice") is False and "not observed" in (d.get("note") or "") for name in ("interface_lang", "account", "pro_preview_notice", "model", "searches"): if name in d: fields[name][str(d[name])] += 1 if e == "claude": tool = d["request"]["tool"] fields["max_searches_allowed"][str(tool["max_uses"])] += 1 fields["location_setting"][f"{tool['user_location']['city']}, {tool['user_location']['country']}"] += 1 notes[e] = {k: dict(v) for k, v in fields.items()} notes[e]["answers"] = len(data[e]) notes[e]["watch_addresses_without_video_id"] = no_query # The record lists some pages twice in one answer; the page key counts such a page once. notes[e]["source_entries"] = entries if e == "perplexity": notes[e]["records_whose_note_says_the_preview_notice_could_not_be_observed"] = unsure notes[e]["entries_that_repeat_a_page_of_the_same_answer"] = entries - sum(len(a["pages"]) for a in data[e].values()) return data, notes def load_repeat(): """engine -> question id -> the same sets, for the complete second answers of the repeat set.""" out = {e: {} for e in REGISTERED} for e in REGISTERED: for f in sorted((HERE / "raw" / e / "repeat").glob("???-??.json")): d = json.loads(f.read_text(encoding="utf-8")) if d["status"] == "ok": out[e][d["id"]] = registered_answer(d) return out def jaccard(sets): union = set().union(*sets) return len(set.intersection(*sets)) / len(union) if union else None def describe(data, e): ans = data[e] pages = [len(a["pages"]) for a in ans.values()] doms = [len(a["domains"]) for a in ans.values()] return { "answers": len(ans), "answers_with_sources": sum(1 for n in pages if n), "page_citations": sum(pages), "distinct_pages": len(set().union(*[a["pages"] for a in ans.values()])) if ans else 0, "distinct_domains": len(set().union(*[a["domains"] for a in ans.values()])) if ans else 0, "median_pages_per_answer": statistics.median(pages) if pages else None, "median_domains_per_answer": statistics.median(doms) if doms else None, "answers_citing_youtube": sum(1 for a in ans.values() if "youtube.com" in a["domains"]), "distinct_youtube_pages": len({p for a in ans.values() for p in a["pages"] if p.split("/")[0].endswith("youtube.com")}), } def overlap(data, engines, level, skip=()): """Per question where every engine cites at least one source: Jaccard over all engines.""" rows = [] for q in questions(): if q["id"] in skip: continue sets = [data[e].get(q["id"], {}).get(level) for e in engines] if any(not s for s in sets): continue shared = set.intersection(*sets) rows.append({"id": q["id"], "sector": q["sector"], "jaccard": jaccard(sets), "shared": sorted(shared), "union": len(set().union(*sets))}) values = [r["jaccard"] for r in rows] return { "engines": engines, "level": level, "questions": len(rows), "mean_jaccard": statistics.mean(values) if values else None, "median_jaccard": statistics.median(values) if values else None, "questions_with_a_shared_source": sum(1 for r in rows if r["shared"]), "pooled_shared": sum(len(r["shared"]) for r in rows), "pooled_union": sum(r["union"] for r in rows), "rows": rows, } def ceiling(data, engines, level): """What the list sizes allow. The all-engine Jaccard of a question cannot exceed its shortest list divided by its longest (reached when the lists are nested), nor its shortest list divided by the union that was observed.""" by_size, by_union, shortest, single = [], [], collections.Counter(), 0 for q in questions(): sets = [data[e].get(q["id"], {}).get(level) for e in engines] if any(not s for s in sets): continue sizes = [len(s) for s in sets] by_size.append(min(sizes) / max(sizes)) by_union.append(min(sizes) / len(set().union(*sets))) single += min(sizes) == 1 for e, n in zip(engines, sizes): shortest[e] += n == min(sizes) # ties count for every engine that holds the shortest list return {"questions": len(by_size), "mean_shortest_over_longest": statistics.mean(by_size), "mean_shortest_over_observed_union": statistics.mean(by_union), "questions_where_the_shortest_list_has_one_source": single, "questions_where_the_engine_has_the_shortest_list": {e: shortest[e] for e in engines}} def containment(data, engines, level): """Ordered pair "a in b": of the sources a cites, the share b also cites for the same question, pooled over the questions where both cite at least one source.""" out = {} for a, b in itertools.permutations(engines, 2): own = shared = n = 0 for q in questions(): sa, sb = data[a].get(q["id"], {}).get(level), data[b].get(q["id"], {}).get(level) if sa and sb: n, own, shared = n + 1, own + len(sa), shared + len(sa & sb) out[f"{a} in {b}"] = {"questions": n, "shared": shared, "own": own, "share": shared / own} return out def repeat_yardstick(data, again): """The same engine, the same question, two askings (pages), and the cross-engine pairs on the same ten questions in the main set.""" ids = [q["id"] for q in sorted(questions(), key=lambda q: int(q["order"]))][:REPEAT] same = {} for e in REGISTERED: rows = [] for qid in ids: first, second = data[e].get(qid, {}).get("pages"), again[e].get(qid, {}).get("pages") if first and second: rows.append({"id": qid, "first": len(first), "second": len(second), "shared": len(first & second), "jaccard": len(first & second) / len(first | second)}) values = [r["jaccard"] for r in rows] same[e] = {"questions": len(rows), "mean_jaccard": statistics.mean(values), "median_jaccard": statistics.median(values), "questions_with_a_shared_source": sum(1 for r in rows if r["shared"]), "questions_without_a_shared_page": {"n": sum(1 for r in rows if not r["shared"]), "of": len(rows)}, "first_answer_pages": sum(r["first"] for r in rows), "first_answer_pages_cited_again": sum(r["shared"] for r in rows), "share_cited_again": sum(r["shared"] for r in rows) / sum(r["first"] for r in rows), "rows": rows} cross = {} for a, b in itertools.combinations(REGISTERED, 2): values, shared = [], 0 for qid in ids: sa, sb = data[a].get(qid, {}).get("pages"), data[b].get(qid, {}).get("pages") if sa and sb: values.append(len(sa & sb) / len(sa | sb)) shared += bool(sa & sb) cross[f"{a}+{b}"] = {"questions": len(values), "mean_jaccard": statistics.mean(values), "median_jaccard": statistics.median(values), "questions_with_a_shared_source": shared} # Like for like: a pair of engines on the questions where both engines have a source list in both askings. def jac(x, y): return len(x & y) / len(x | y) alike = {} for a, b in itertools.combinations(REGISTERED, 2): qs = [q for q in ids if all(src[e].get(q, {}).get("pages") for src in (data, again) for e in (a, b))] two = [jac(data[a][q]["pages"], data[b][q]["pages"]) for q in qs] own = {e: [jac(data[e][q]["pages"], again[e][q]["pages"]) for q in qs] for e in (a, b)} alike[f"{a}+{b}"] = { "questions": len(qs), "ids": qs, "two_engines_first_answers": {"mean_jaccard": statistics.mean(two), "median_jaccard": statistics.median(two)}, "same_engine_two_askings": {e: {"mean_jaccard": statistics.mean(own[e]), "median_jaccard": statistics.median(own[e])} for e in (a, b)}, "questions_where_the_two_engines_share_less_than_either_shares_with_itself": { "n": sum(x < min(y, z) for x, y, z in zip(two, own[a], own[b])), "of": len(qs)}} two_means = [x["two_engines_first_answers"]["mean_jaccard"] for x in alike.values()] own_means = [y["mean_jaccard"] for x in alike.values() for y in x["same_engine_two_askings"].values()] return {"questions_of_the_repeat_set": ids, "same_engine_two_askings": same, "cross_engine_pairs_same_questions": cross, "like_for_like": {"pairs": alike, "two_engines_lowest_and_highest_mean": [min(two_means), max(two_means)], "same_engine_lowest_and_highest_mean": [min(own_means), max(own_means)]}} def by_engines(data, engines, level): """On the questions where every engine cites at least one source: each source of a question counted by the number of engines that cite it for that question, and the questions where at least k engines share one.""" counts, at_least, n, distinct = collections.Counter(), collections.Counter(), 0, set() for q in questions(): sets = [data[e].get(q["id"], {}).get(level) for e in engines] if any(not s for s in sets): continue n += 1 cited = collections.Counter(x for s in sets for x in s) counts.update(cited.values()) distinct.update(cited) for k in range(2, len(engines) + 1): at_least[k] += max(cited.values()) >= k total = sum(counts.values()) return {"questions": n, "sources_counted_per_question": total, "cited_by_this_many_engines": {str(k): {"n": counts[k], "of": total, "share": counts[k] / total} for k in range(1, len(engines) + 1)}, "questions_where_at_least_this_many_engines_share_a_source": { str(k): {"n": at_least[k], "of": n} for k in range(2, len(engines) + 1)}, "distinct_sources_over_these_questions": len(distinct)} def containment_on_common_questions(data, own, partners, level): """Of the sources one engine cites, the share each partner also cites, on the questions where the engine and every partner cite at least one source, so that the shares stand on one set of questions.""" qs = [q["id"] for q in questions() if all(data[e].get(q["id"], {}).get(level) for e in [own] + partners)] total = sum(len(data[own][q][level]) for q in qs) return {"questions": len(qs), **{ f"{own} in {b}": {"shared": sum(len(data[own][q][level] & data[b][q][level]) for q in qs), "own": total, "share": sum(len(data[own][q][level] & data[b][q][level]) for q in qs) / total} for b in partners}} def pairwise(data, engines, level): out = {} for a, b in itertools.combinations(engines, 2): o = overlap(data, [a, b], level) out[f"{a}+{b}"] = {k: o[k] for k in ("questions", "mean_jaccard", "median_jaccard", "questions_with_a_shared_source")} return out def most_cited(data, engines, top=25): """Domains by the number of (engine, question) answers that cite them.""" by_engine = {e: collections.Counter() for e in engines} for e in engines: for a in data[e].values(): by_engine[e].update(a["domains"]) total = collections.Counter() for c in by_engine.values(): total.update(c) reach = collections.Counter({d: sum(1 for e in engines if by_engine[e][d]) for d in total}) table = [{"domain": d, "answers": n, "engines": reach[d], **{e: by_engine[e][d] for e in engines}} for d, n in sorted(total.items(), key=lambda x: (-x[1], x[0]))[:top]] hosts = collections.defaultdict(set) for e in engines: for a in data[e].values(): for h in a["hosts"]: hosts[domain(h)].add(h) return { "top": table, "domains_by_engines_reached": dict(sorted(collections.Counter(reach.values()).items())), "distinct_domains": len(total), "hosts_under_platform_domains": {d: len(hosts[d]) for d in PLATFORMS}, "cited_by_every_engine": sorted(d for d, n in reach.items() if n == len(engines)), "top_per_engine": {e: sorted(by_engine[e].items(), key=lambda x: (-x[1], x[0]))[:15] for e in engines}, } def by_sector(result): out = collections.defaultdict(list) for r in result["rows"]: out[r["sector"]].append(r["jaccard"]) return {s: {"questions": len(v), "mean_jaccard": statistics.mean(v)} for s, v in sorted(out.items())} def main(): data, notes = load() three = ["google_aio", "gemini", "perplexity"] four_ids = {r["id"] for r in overlap(data, FOUR, "pages")["rows"]} notes["lists_fully_seen"] = { "chatgpt_answers": len(FULL["chatgpt"]), "gemini_answers": len(FULL["gemini"]), "google_aio_answers": len(FULL["google_aio"]), "four_engine_questions": len(four_ids), "four_engine_questions_with_chatgpt_and_gemini_lists_fully_seen": len(four_ids & FULL["chatgpt"] & FULL["gemini"]), } res = { "page_rule": PAGE_RULE, "engines": {e: describe(data, e) for e in FIVE}, "collection_notes": notes, "four_engines": { "answers": sum(len(data[e]) for e in FOUR), "answers_with_sources": sum(1 for e in FOUR for a in data[e].values() if a["pages"]), "pages": overlap(data, FOUR, "pages"), "domains": overlap(data, FOUR, "domains"), "ceiling_pages": ceiling(data, FOUR, "pages"), "ceiling_domains": ceiling(data, FOUR, "domains"), "pages_by_engines": by_engines(data, FOUR, "pages"), "domains_by_engines": by_engines(data, FOUR, "domains"), "pages_without_interrupted_question": overlap(data, FOUR, "pages", skip=(INTERRUPTED,)), "domains_without_interrupted_question": overlap(data, FOUR, "domains", skip=(INTERRUPTED,)), "pairwise_pages": pairwise(data, FOUR, "pages"), "pairwise_domains": pairwise(data, FOUR, "domains"), "most_cited": most_cited(data, FOUR), }, "without_chatgpt": { "pages": overlap(data, three, "pages"), "domains": overlap(data, three, "domains"), "ceiling_pages": ceiling(data, three, "pages"), "ceiling_domains": ceiling(data, three, "domains"), "pages_without_interrupted_question": overlap(data, three, "pages", skip=(INTERRUPTED,)), "domains_without_interrupted_question": overlap(data, three, "domains", skip=(INTERRUPTED,)), "gemini_lists_fully_seen": { "pages": overlap(data, three, "pages", skip=[q["id"] for q in questions() if q["id"] not in FULL["gemini"]]), "domains": overlap(data, three, "domains", skip=[q["id"] for q in questions() if q["id"] not in FULL["gemini"]]), }, }, "four_engines_fully_seen_chatgpt_lists": { "chatgpt_answers": len(FULLY_SEEN), "pages": overlap(data, FOUR, "pages", skip=[q["id"] for q in questions() if q["id"] not in FULLY_SEEN]), "domains": overlap(data, FOUR, "domains", skip=[q["id"] for q in questions() if q["id"] not in FULLY_SEEN]), }, "three_registered_engines": { "pages": overlap(data, REGISTERED, "pages"), "domains": overlap(data, REGISTERED, "domains"), }, "five_engines_with_claude": { "pages": overlap(data, FIVE, "pages"), "domains": overlap(data, FIVE, "domains"), "pairwise_pages_with_claude": {k: v for k, v in pairwise(data, FIVE, "pages").items() if "claude" in k}, "pairwise_domains_with_claude": {k: v for k, v in pairwise(data, FIVE, "domains").items() if "claude" in k}, "most_cited": most_cited(data, FIVE), }, "containment_pages": containment(data, FIVE, "pages"), "containment_domains": containment(data, FIVE, "domains"), "containment_on_common_questions": { level: {e: containment_on_common_questions(data, e, ["google_aio", "perplexity"], level) for e in ("chatgpt", "claude")} for level in ("pages", "domains")}, "repeat_yardstick_pages": repeat_yardstick(data, load_repeat()), } one = res["four_engines"]["most_cited"]["domains_by_engines_reached"][1] res["four_engines"]["most_cited"]["share_cited_by_one_engine_only"] = ( one / res["four_engines"]["most_cited"]["distinct_domains"]) multi = {"four_engines": res["four_engines"], "without_chatgpt": res["without_chatgpt"], "three_registered_engines": res["three_registered_engines"], "five_engines_with_claude": res["five_engines_with_claude"]} res["median_jaccard_of_the_multi_engine_rows"] = { f"{name}, {level}": block[level]["median_jaccard"] for name, block in multi.items() for level in ("pages", "domains")} values = [v["mean_jaccard"] for v in res["four_engines"]["pairwise_pages"].values()] res["four_engines"]["pairwise_pages_lowest_and_highest_mean"] = [min(values), max(values)] for block in (res["four_engines"], res["without_chatgpt"]): for level in ("pages", "domains"): block[f"ceiling_{level}"]["observed_mean_overlap_as_a_share_of_the_ceiling"] = ( block[level]["mean_jaccard"] / block[f"ceiling_{level}"]["mean_shortest_over_longest"]) res["four_engines"]["domains_by_sector"] = by_sector(res["four_engines"]["domains"]) res["four_engines"]["pages_by_sector"] = by_sector(res["four_engines"]["pages"]) OUT.write_text(json.dumps(res, indent=1, ensure_ascii=False) + "\n", encoding="utf-8") print("engine answers with sources page citations pages domains median pages") for e in FIVE: d = res["engines"][e] print(f"{e:12} {d['answers']:7} {d['answers_with_sources']:12} {d['page_citations']:14} " f"{d['distinct_pages']:5} {d['distinct_domains']:7} {d['median_pages_per_answer']:12}") for name in ("pages", "domains", "pages_without_interrupted_question", "domains_without_interrupted_question"): o = res["four_engines"][name] print(f"four engines, {name}: {o['questions']} questions, mean Jaccard {o['mean_jaccard']}, " f"median {o['median_jaccard']}, a shared source in {o['questions_with_a_shared_source']}") for name in ("pages", "domains"): o = res["five_engines_with_claude"][name] print(f"five engines, {name}: {o['questions']} questions, mean Jaccard {o['mean_jaccard']}, " f"a shared source in {o['questions_with_a_shared_source']}") for block in ("without_chatgpt", "four_engines_fully_seen_chatgpt_lists"): for name in ("pages", "domains"): o = res[block][name] print(f"{block}, {name}: {o['questions']} questions, mean Jaccard {o['mean_jaccard']}, " f"a shared source in {o['questions_with_a_shared_source']}") for level in ("pages", "domains"): c = res["four_engines"][f"ceiling_{level}"] print(f"four engines, {level}: ceiling {c['mean_shortest_over_longest']} by list sizes, " f"{c['mean_shortest_over_observed_union']} by the observed union; shortest list has one source in " f"{c['questions_where_the_shortest_list_has_one_source']} of {c['questions']}") for e, r in res["repeat_yardstick_pages"]["same_engine_two_askings"].items(): print(f"repeat, {e}: {r['questions']} questions, mean Jaccard {r['mean_jaccard']}, median {r['median_jaccard']}, " f"a shared page in {r['questions_with_a_shared_source']}") print("collection notes:", json.dumps({k: {x: y for x, y in v.items() if not isinstance(y, list)} for k, v in notes.items()})) print("written:", OUT.relative_to(HERE)) if __name__ == "__main__": main()