{
  "_comment": [
    "Question set for the retrieval-provenance audit. Frozen 2026-07-27 before any counted run.",
    "The suffix in `instruction` is appended verbatim to every question, for every system, on every run.",
    "`ground_truth` was established BEFORE each question was run, from primary documents.",
    "`discriminators` are strings whose presence in an answer is diagnostic, with what they diagnose.",
    "Nothing here is derived from any model's answer."
  ],
  "instruction": "Give the figure and cite the original source.",
  "questions": [
    {
      "id": "Q1",
      "class": "unrecoverable",
      "text": "How much more expensive is acquiring a customer than retaining one?",
      "ground_truth": {
        "summary": "No primary source exists. The circulating 5x/7x/25x figures trace to secondary and tertiary restatements with no recoverable study behind them.",
        "established": "2026-07-27",
        "version_of_record": null,
        "preprint": null
      },
      "discriminators": []
    },
    {
      "id": "Q2",
      "class": "contested",
      "text": "What share of the B2B buying journey is complete before a vendor is contacted?",
      "ground_truth": {
        "summary": "A real origin exists (CEB, later Gartner) and is routinely misstated in both the figure and the population it describes.",
        "established": "2026-07-27",
        "version_of_record": null,
        "preprint": null
      },
      "discriminators": []
    },
    {
      "id": "Q3",
      "class": "vendor-origin",
      "text": "How many touchpoints does it take to book a B2B sales meeting?",
      "ground_truth": {
        "summary": "Trade and vendor figures only. No recoverable method behind any circulating number.",
        "established": "2026-07-27",
        "version_of_record": null,
        "preprint": null
      },
      "discriminators": []
    },
    {
      "id": "Q4",
      "class": "control",
      "text": "What did Brynjolfsson, Li and Raymond find about the effect of AI assistance on customer support agent productivity?",
      "ground_truth": {
        "summary": "A clean, indexed, DOI-bearing version of record exists. An earlier open preprint reports different numbers.",
        "established": "before first run; from the project's verified source bank",
        "version_of_record": {
          "cite": "Brynjolfsson, Li & Raymond (2025). Generative AI at Work. The Quarterly Journal of Economics 140(2), 889-942.",
          "figures": {"average_lift_pct": 15, "agents": 5172}
        },
        "preprint": {
          "cite": "NBER Working Paper 31161 (April 2023).",
          "figures": {"average_lift_pct": 14, "agents": 5179, "novice_lift_pct": 34}
        }
      },
      "discriminators": [
        {"match": "14%", "diagnoses": "preprint"},
        {"match": "5,179", "diagnoses": "preprint"},
        {"match": "34%", "diagnoses": "preprint-only figure"},
        {"match": "15%", "diagnoses": "version of record"},
        {"match": "5,172", "diagnoses": "version of record"}
      ]
    },
    {
      "id": "Q5",
      "class": "control",
      "text": "What did the BCG study find about consultants using GPT-4?",
      "ground_truth": {
        "summary": "Peer review revised the headline quality result downward and removed it from the abstract. The productivity headlines are unchanged across versions, so the quality figure is the discriminator.",
        "established": "2026-07-27, before the run, from both full texts",
        "version_of_record": {
          "cite": "Dell'Acqua et al. (2026). Navigating the Jagged Technological Frontier. Organization Science 37(2), 403-423. DOI 10.1287/orsc.2025.21838",
          "figures": {
            "subjects": 758,
            "tasks_pct": 12.2,
            "speed_pct": 25.1,
            "quality_pct_gpt_plus_overview": 33.9,
            "quality_pct_gpt_only": 29.9,
            "outside_frontier": "19 percentage points (body); abstract says '19% less likely'"
          }
        },
        "preprint": {
          "cite": "Harvard Business School Working Paper 24-013 (22 September 2023); SSRN 4573321.",
          "figures": {
            "subjects": 758,
            "tasks_pct": 12.2,
            "speed_pct": 25.1,
            "quality_claim": "more than 40% higher quality",
            "quality_table1": [42.5, 38.0],
            "skill_split_pct": [43, 17],
            "outside_frontier": "19 percentage points"
          }
        },
        "sponsor_publication": {
          "cite": "Boston Consulting Group, 'How People Create and Destroy Value with Generative AI' (bcg.com, 2023).",
          "note": "The commissioning sponsor's own account of its own experiment. Not a citable source for the study's findings.",
          "figures": {"outside_frontier_pct": 23, "quality_pct": 40}
        }
      },
      "discriminators": [
        {"match": "40%", "diagnoses": "preprint (deleted from the version of record)"},
        {"match": "40.2%", "diagnoses": "downstream fabrication - appears in NEITHER version; traced to an SEO statistics page"},
        {"match": "43%", "diagnoses": "preprint-only; removed in peer review"},
        {"match": "17%", "diagnoses": "preprint-only; removed in peer review"},
        {"match": "23%", "diagnoses": "sponsor marketing figure, not the paper's"},
        {"match": "33.9%", "diagnoses": "version of record"},
        {"match": "29.9%", "diagnoses": "version of record"},
        {"match": "40 percent of the trial group", "diagnoses": "Harvard Crimson misstatement - magnitude converted to headcount"}
      ]
    },
    {
      "id": "Q6",
      "class": "semi-recoverable",
      "text": "What proportion of A/B tests improve the metric they target?",
      "ground_truth": {
        "summary": "Kohavi's roughly one-third figure is real but conference-only and hard to reach. The freely circulating 90% figure is not his.",
        "established": "2026-07-27",
        "version_of_record": null,
        "preprint": null
      },
      "discriminators": []
    }
  ]
}
