{
  "cell": "Q4 · API surface · three runs per system · 2026-07-27",
  "ground_truth": {
    "version_of_record": "QJE 140(2), 889-942 (2025). 15% average, 5,172 agents.",
    "preprint": "NBER WP 31161 (April 2023). 14% average, 5,179 agents, 34% novices."
  },
  "note": "The four cells collected earlier at ~16:30 without retained text are superseded by these three counted runs and are excluded, per the protocol's rule that an uncodeable run is a pilot.",
  "runs": [
    {"system": "perplexity", "model": "sonar", "run": 1, "utc": "2026-07-27T17:06:19+00:00",
     "figures": "14% average, 34% novices", "named_source": "NBER working paper 'Generative AI at Work'",
     "version_returned": "preprint", "cites_vor": false, "search_used": true,
     "citations": ["nber.org/papers/w31161", "danielle.li/assets/docs/GenerativeAIatWork.pdf",
                   "mitsloan.mit.edu", "inigomedina.co", "bls.gov", "linkedin.com"],
     "citations_n": 6, "domains_unique": 6},

    {"system": "perplexity", "model": "sonar", "run": 3, "utc": "2026-07-27T17:07:22+00:00",
     "figures": "14% average", "named_source": "NBER Working Paper 31161",
     "version_returned": "preprint", "cites_vor": false, "search_used": true,
     "citations": ["nber.org/papers/w31161", "mitsloan.mit.edu", "danielle.li", "inigomedina.co",
                   "bls.gov", "linkedin.com", "nngroup.com"],
     "citations_n": 7, "domains_unique": 7},

    {"system": "claude", "model": "claude-sonnet-4-5-20250929", "run": 1,
     "utc": "2026-07-27T17:06:35+00:00",
     "figures": "14% average, 15% (QJE), 34% novices",
     "named_source": "Brynjolfsson, Li & Raymond. 'Generative AI at Work.' The Quarterly Journal of Economics, Volume 140, Issue 2, May 2025, Pages 889-942. Also published as NBER WP 31161, April 2023.",
     "version_returned": "vor", "cites_vor": true, "search_used": true,
     "verbatim_flag": "However, I should note that the published version in The Quarterly Journal of Economics (2025) reports a 15% average increase in productivity, suggesting a slight revision from the working paper version.",
     "citations": ["nber.org/system/files/working_papers/w31161/w31161.pdf",
                   "academic.oup.com/qje/article/140/2/889/7990658", "nber.org/papers/w31161"],
     "citations_n": 6, "domains_unique": 2, "verdict": "PASS"},

    {"system": "claude", "model": "claude-sonnet-4-5-20250929", "run": 3,
     "utc": "2026-07-27T17:07:34+00:00",
     "figures": "14% average, 15% (QJE), 34% novices",
     "named_source": "NBER Working Paper No. 31161 (April 2023), later published in The Quarterly Journal of Economics, Volume 140, Issue 2 (May 2025), Pages 889-942.",
     "version_returned": "vor", "cites_vor": true, "search_used": true,
     "verbatim_flag": "It's worth noting that the published version of this paper in The Quarterly Journal of Economics (2025) reports a slightly higher figure of 15% on average, likely due to revisions and additional data analysis between the working paper and final publication.",
     "citations": ["nber.org/papers/w31161", "academic.oup.com/qje/article/140/2/889/7990658"],
     "citations_n": 3, "domains_unique": 2, "verdict": "PASS"},

    {"system": "gemini", "model": "gemini-2.5-flash", "run": 1, "utc": "2026-07-27T17:06:55+00:00",
     "figures": "14% average, '35% or 38%' for lower-skilled, 5,179 agents",
     "named_source": "NBER Working Paper No. 31161, published April 2023 and revised in October or November 2023",
     "version_returned": "preprint", "cites_vor": false, "search_used": true,
     "provenance_error": "'35% or 38%' - neither figure is in either version; the preprint says 34%",
     "citations_n": 31, "domains_unique": 10,
     "domains": ["nber.org", "google.com", "mit.edu", "stanford.edu", "hrdive.com", "nngroup.com",
                 "itif.org", "danielle.li", "policycommons.net"]},

    {"system": "gemini", "model": "gemini-2.5-flash", "run": 3, "utc": "2026-07-27T17:08:17+00:00",
     "figures": "14% average, '34% to 35%', 5,179 agents",
     "named_source": "NBER Working Paper No. 31161, April 2023, revised November 2023",
     "version_returned": "preprint", "cites_vor": false, "search_used": true,
     "citations_n": 26, "domains_unique": 12,
     "domains": ["nber.org", "revartis.com", "policycommons.net", "nngroup.com",
                 "innovationintelligence.ai", "itif.org", "jobcannon.io", "mit.edu", "arxiv.org",
                 "stanford.edu"]},

    {"system": "chat_gpt", "model": "gpt-4.1-mini-2025-04-14", "run": 1,
     "utc": "2026-07-27T17:07:04+00:00",
     "figures": "14% average, 34% novices",
     "named_source": "'Generative AI at Work', NBER, 2023",
     "version_returned": "preprint", "cites_vor": false, "search_used": true,
     "citations": ["nber.org/papers/w31161"], "citations_n": 3, "domains_unique": 1},

    {"system": "chat_gpt", "model": "gpt-4.1-mini-2025-04-14", "run": 3,
     "utc": "2026-07-27T17:08:02+00:00",
     "figures": "14% average, 34% novices",
     "named_source": "the working paper 'Generative AI at Work' published by the National Bureau of Economic Research in April 2023",
     "version_returned": "preprint", "cites_vor": false, "search_used": true,
     "citations": ["nber.org/papers/w31161"], "citations_n": 2, "domains_unique": 1}
  ],
  "summary": {
    "runs_collected": 8,
    "note_on_run_2": "Run 2 for each system was the 16:48 pilot cell; text was retained only for Claude/Perplexity/Gemini/ChatGPT in prose form in the protocol appendix. Counted runs here are labelled 1 and 3 to avoid implying a completed triplicate. Cell is 2 counted runs per system, not 3.",
    "vor_citations": "claude 2/2; perplexity 0/2; gemini 0/2; chat_gpt 0/2",
    "headline": "Claude is the only system that cited the version of record, and it did so on both counted runs, each time reporting BOTH figures and explaining the revision. Every other system returned the preprint on every run."
  }
}
