""" ai-answer-evidence: shared paths and readers of the main-run chain. """ import json, pathlib, re, time, urllib.request HERE = pathlib.Path(__file__).resolve().parent RAW, WORK, SNAPSHOTS = HERE / "raw", HERE / "work", HERE / "snapshots" ENGINES = ["google_aio", "chatgpt", "gemini"] SETS = {"trial": "pilot-questions.tsv", "main": "questions-main-v1.tsv", "repeat": "questions-main-v1.tsv"} LOG = WORK / "chain.log" REPEAT = 10 # the repeat run asks the first 10 questions in the order of the main list again # A fetch's text is the text of the page only when it is long enough to be one and is not a check # page that was served as an ordinary page (method note, section 5). MIN_PAGE_CHARS = 500 CHALLENGE_BELOW = 2000 CHALLENGE = re.compile("|".join([ "verify you are human", "prove your humanity", "are you a robot", "not a robot", "unusual traffic", "just a moment", "checking your browser", "enable javascript and cookies", "access denied", "captcha", "request blocked", "complete the challenge"]), re.I) def log(msg): WORK.mkdir(parents=True, exist_ok=True) line = f"{time.strftime('%Y-%m-%d %H:%M:%S')} {msg}" print(line, flush=True) with LOG.open("a", encoding="utf-8") as f: f.write(line + "\n") def questions(which): rows = (HERE / SETS[which]).read_text(encoding="utf-8").splitlines() head = rows[0].split("\t") return [dict(zip(head, r.split("\t"))) for r in rows[1:] if r.strip()] def frame(which): """The question ids every engine owes an observation for: all 50 for the main run, the first 10 in asking order for the repeat run, and for the trial the ids that were asked.""" if which == "main": return [q["id"] for q in questions(which)] if which == "repeat": # the same rule as engines/blocks.py: the asking order, not the file order return [q["id"] for q in sorted(questions(which), key=lambda q: int(q["order"]))][:REPEAT] return sorted({d["id"] for d in records(which)}) def model_digest(name): """The digest Ollama holds for a model tag, so the output names the exact model it came from. Stops the script when the digest cannot be read: no output is written without it.""" try: with urllib.request.urlopen("http://localhost:11434/api/tags", timeout=10) as r: models = json.loads(r.read()).get("models", []) except (OSError, ValueError) as e: raise SystemExit(f"STOP: the model list of Ollama cannot be read ({e})") digest = next((m.get("digest") for m in models if m.get("name") == name), None) if not digest: raise SystemExit(f"STOP: Ollama has no model {name}: build it from its model file first") return digest def page_address(url): """The address that is fetched: the cited address without its fragment.""" return url.split("#", 1)[0] def records(which): """The answer records the engine parsers wrote, one per engine and question.""" for engine in ENGINES: for path in sorted((RAW / engine / which).glob("*.json")): if not path.name.endswith((".meta.json", ".links.json")): yield json.loads(path.read_text(encoding="utf-8")) def answers(which): """Every answered question: text, cited addresses, citation markers.""" out = [] for d in records(which): if d.get("status") != "ok": continue cites = d.get("inline_citations") or [] listed = d.get("sources", []) urls = [s["url"] for s in listed if s.get("url") and s.get("url_complete", True)] urls += [c["url"] for c in cites if c.get("url") and c.get("url_complete", True)] # Cited pages the record has no address for: a source shown with site and title only, or # a marker that carries no address at all. They are cited and unreadable. unaddressed = sum(1 for s in listed if not (s.get("url") and s.get("url_complete", True))) unaddressed += len({c.get("marker") for c in cites if not c.get("url")}) out.append({"engine": d["engine"], "id": d["id"], "text": d.get("answer_text") or "", "sources": list(dict.fromkeys(page_address(u) for u in urls if u.startswith("http"))), "unaddressed": unaddressed, "citations": cites, "sponsored_in_answer": bool((d.get("sponsored_labels") or {}).get("in_answer"))}) return out def page_texts(which): """address -> the text of every fetch that returned text, or None when no fetch did.""" root, texts = SNAPSHOTS / which, {} manifest = root / "manifest.jsonl" if not manifest.exists(): return texts for raw in manifest.read_text(encoding="utf-8").splitlines(): line = json.loads(raw) url = line["url_fetched"] texts.setdefault(url, None) if line["outcome"] == "text" and line.get("text_file"): text = (root / line["text_file"]).read_text(encoding="utf-8") if len(text) < MIN_PAGE_CHARS or (len(text) < CHALLENGE_BELOW and CHALLENGE.search(text)): continue # a stub or a check page, not the page texts[url] = (texts[url] + "\n" if texts[url] else "") + text return texts