{
  "schema_version": "1.1",
  "id": "s13:https://arxiv.org/abs/2609.15964v1",
  "slug": "2609-15964v1-08frwh7",
  "url": "https://feed7.dev/p/2609-15964v1-08frwh7",
  "title": "Verifiable by Construction: Claim-Level Evaluation of Verbatim Citation in Clinical Question Answering",
  "why_included": "Claim-level quotes can look verifiable while failing to support the full claim. Builders of retrieval agents should score citation coverage, verbatim accuracy, and entailment separately.",
  "summary": "A harness evaluates **12 LLMs** on **222 synthetic clinical questions** drawn from four practice guidelines. It checks whether each factual claim has a citation, whether the cited text is verbatim, and whether that text fully supports the claim.",
  "practical_implication": "Builders should add claim-level entailment checks instead of treating citation presence or exact copying as proof. Retrieval-agent evaluations need separate measures for coverage, quotation fidelity, and complete substantiation.",
  "agent_context": "A harness evaluates **12 LLMs** on **222 synthetic clinical questions** drawn from four practice guidelines. It checks whether each factual claim has a citation, whether the cited text is verbatim, and whether that text fully supports the claim.\n\nBuilders should add claim-level entailment checks instead of treating citation presence or exact copying as proof. Retrieval-agent evaluations need separate measures for coverage, quotation fidelity, and complete substantiation.\n\nMost models quote source text for over 90% of claims, yet support can remain weak. Claude Opus 5 reaches **98.0% verbatim coverage** but only **37.1% full substantiation**; the clinical, synthetic setup limits direct generalization to other domains.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2609.15964v1",
    "published_at": "2026-09-14T17:53:51.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "benchmark",
  "domains": [
    "research"
  ],
  "topics": [
    "agent-evals",
    "retrieval",
    "benchmark-integrity"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "Most models quote source text for over 90% of claims, yet support can remain weak. Claude Opus 5 reaches **98.0% verbatim coverage** but only **37.1% full substantiation**; the clinical, synthetic setup limits direct generalization to other domains."
  ],
  "connected_context": {
    "meaning": "This replaces citation presence as a sufficient retrieval-eval signal with a three-part acceptance test: claim coverage, exact quotation, and full support. It reinforces prior warnings that apparently strong evaluators can miss domain-specific failures, while adding a directly inspectable evidence-to-claim check. Its clinical synthetic data supports the evaluation decomposition, not a universal performance estimate.",
    "corpus_size": 778,
    "generated_at": "2026-09-15T10:06:50.635Z",
    "connections": [
      {
        "title": "Domain-Specific Hallucination Detection in Large Language Models",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2609.11878v1",
        "feed7_url": "https://feed7.dev/p/2609-11878v1-1ltrudx",
        "reason": "Both show that a strong surface evaluation signal can conceal weak biomedical grounding, supporting domain-matched checks rather than reliance on a generic hallucination or citation score."
      },
      {
        "title": "Verifiable Environments for AI in Biology — Kenny Workman, LatchBio",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=3ZMUiFaQ3qg",
        "feed7_url": "https://feed7.dev/p/verifiable-environments-for-ai-in-biology-kenny-workman-latchbio-1vs6y66",
        "reason": "The claim-level support test supplies a deterministic component for scientific evaluation, while the biology evidence explains why expert review is still needed when valid interpretations exceed brittle graders."
      },
      {
        "title": "Beyond Scores: Understanding LLM-as-a-Judge Mechanisms in Summarization Evaluation",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2609.01604v1",
        "feed7_url": "https://feed7.dev/p/2609-01604v1-02vljon",
        "reason": "Claim-level decomposition provides observable failure categories that can help distinguish missed evidence from faulty final integration when auditing an LLM judge."
      },
      {
        "title": "Legibility is Not Interpretability: Comparing Judged and Actual Importance in Chain-Of-Thought Reasoning",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2609.04194v1",
        "feed7_url": "https://feed7.dev/p/2609-04194v1-0milybu",
        "reason": "Both caution against mistaking legible text for valid evidence: verbatim quotations may not substantiate claims, just as readable reasoning steps may not cause correct answers."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-09-14T17:53:51.000Z",
  "modified_at": "2026-09-14T17:53:51.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2609-15964v1-08frwh7",
    "json": "https://feed7.dev/p/2609-15964v1-08frwh7.json",
    "markdown": "https://feed7.dev/p/2609-15964v1-08frwh7.md"
  }
}