{
  "version": "2026-09-13-r1",
  "title": "Secondary evidence ledger: attribution accuracy is not claim support",
  "checked_at": "2026-09-13T00:00:00+07:00",
  "scope": "A structured reading of two public studies. It does not merge their samples into one accuracy score because they tested different systems, dates and error types.",
  "studies": [
    {
      "study_id": "tow-center-2025",
      "title": "AI Search Has a Citation Problem",
      "publisher": "Tow Center for Digital Journalism / Columbia Journalism Review",
      "url": "https://www.cjr.org/tow_center/we-compared-eight-ai-search-engines-theyre-all-bad-at-citing-news.php",
      "published": "2025-03-06",
      "systems": 8,
      "publishers": 20,
      "articles_per_publisher": 10,
      "queries_total": 1600,
      "queries_per_system": 200,
      "task": "Given a direct excerpt, identify the source article, publisher, publication date and URL.",
      "chatgpt_incorrect_article_identifications": 134,
      "chatgpt_queries": 200,
      "chatgpt_incorrect_rate": 0.67,
      "chatgpt_uncertainty_signals_among_wrong_answers": 15,
      "chatgpt_declined_answers": 0,
      "what_it_measures": "Source/article attribution accuracy for an excerpt retrieval task",
      "what_it_does_not_measure": "Whether every factual claim in an ordinary ChatGPT answer is fully entailed by its inline source"
    },
    {
      "study_id": "liu-et-al-2023",
      "title": "Evaluating Verifiability in Generative Search Engines",
      "publisher": "arXiv / Stanford researchers",
      "url": "https://arxiv.org/abs/2304.09848",
      "published": "2023-04-20",
      "systems": ["Bing Chat", "NeevaAI", "perplexity.ai", "YouChat"],
      "chatgpt_included": false,
      "average_citation_recall": 0.515,
      "average_citation_precision": 0.745,
      "definitions": {
        "citation_recall": "Share of verification-worthy statements fully supported by their associated citations",
        "citation_precision": "Share of citations that support their associated statements under the paper's annotation rule"
      },
      "what_it_measures": "Human-annotated support coverage and citation support across four 2023 generative search engines",
      "what_it_does_not_measure": "Current ChatGPT Search accuracy or performance in Vietnam"
    }
  ],
  "interpretation_rules": [
    "Do not call the Tow Center's 67% source-identification error rate a claim-support error rate.",
    "Do not call the Liu et al. 51.5%/74.5% averages ChatGPT scores because ChatGPT was not included.",
    "Do not average the two studies because their systems, tasks, dates and denominators differ.",
    "A working URL, correct publisher attribution, and claim entailment are three separate checks."
  ],
  "buyer_checklist": [
    "Open the exact deep URL, not only the source domain.",
    "Locate the passage that supports the claim.",
    "Check whether the page is original, syndicated or copied.",
    "Verify date, geography, units and denominator.",
    "Separate full support, partial support, near-topic and unsupported.",
    "Archive the answer and source snapshot for later review."
  ]
}
