#!/usr/bin/env python3 """ ai-answer-evidence: fetch the pages the answers cite (method note, section 5). Every address is fetched once per set, plain and rendered, by the snapshot fetcher (code/collection/snapshots), which writes an append-only manifest and never repeats a fetch that is in it. The fetcher says what it is in its user agent; a page that refuses it is recorded as it answered. The network exit must be New York (US). Writes snapshots//manifest.jsonl and snapshots//files/. Usage: python3 fetch_pages.py --set main """ import argparse, datetime, hashlib, os, sys from chain_common import HERE, SETS, SNAPSHOTS, answers, log # The snapshot fetcher lives in code/collection of the study folder (AAE_CODE overrides). sys.path[:0] = [str(HERE / "code" / "collection"), os.environ.get("AAE_CODE", str(HERE / "code" / "collection"))] from engines.exitcheck import exit_check from snapshots.exit_check import ExitCheck, ExitCheckProvider, ExitHold from snapshots.fetcher import SnapshotFetcher from snapshots.plain_fetch import PlainConfig from snapshots.plan import FetchJob from snapshots.render import RenderConfig USER_AGENT = ("Mozilla/5.0 (compatible; AIAnswerEvidenceIndex/1.0; automated research fetcher; " "+https://ismybrandinai.com)") PAGE_KEY_RULE = "sha256 of the address without its fragment, first 16 hex digits" def page_key(url): return hashlib.sha256(url.split("#", 1)[0].encode("utf-8")).hexdigest()[:16] class Checks(ExitCheckProvider): def __init__(self, which): self.which, self.last, self.count = which, None, 0 def latest(self): return self.last def run(self): self.count += 1 started = datetime.datetime.now().astimezone() # the check's time is when it began result = exit_check() self.last = ExitCheck(check_id="X-%s-%03d" % (self.which, self.count), at=started, passed=result["passed"]) print("exit check", self.last.check_id, "passed" if result["passed"] else "FAILED", flush=True) return self.last def main(which): fetcher = SnapshotFetcher( root=str(SNAPSHOTS / which), exit_checks=Checks(which), page_key_fn=page_key, page_key_rule=PAGE_KEY_RULE, plain_config=PlainConfig(user_agent=USER_AGENT), render_config=RenderConfig(extra_args=("--user-agent=" + USER_AGENT,))) seen = set() for a in answers(which): answer = "%s-%s-%s" % (a["engine"], which, a["id"]) for url in a["sources"]: if url in seen: # one fetch per address; every answer that cites it reads the same snapshot continue seen.add(url) # The manifest knows a fetch by attempt, page and kind. One attempt id per set makes an # address one fetch for the whole set, whichever answer cites it and in whichever block. job = FetchJob(attempt_id="pages-" + which, answer_id=answer, page_key=page_key(url), url_as_shown=url, url_resolved=url, url_fetched=url, list_origins=["answer"], attempt_ended_at=None) try: written = fetcher.fetch_job(job) except ExitHold: raise # the exit is not New York: stop, nothing is fetched from elsewhere except Exception as e: # one address the fetcher cannot handle must not stop the others log("fetch error %s: %s: %s" % (url[:120], type(e).__name__, e)) continue for line in written: print("%-9s %-13s %s %s" % (line["fetch_kind"], line["outcome"], line.get("http_status"), url[:80]), flush=True) if __name__ == "__main__": ap = argparse.ArgumentParser() ap.add_argument("--set", required=True, choices=sorted(SETS)) main(ap.parse_args().set)