{
  "schema_version": "1.1",
  "id": "s13:https://arxiv.org/abs/2609.30205v1",
  "slug": "2609-30205v1-1msmwmk",
  "url": "https://feed7.dev/p/2609-30205v1-1msmwmk",
  "title": "A Living Benchmark for Information Retrieval from Electronic Health Records",
  "why_included": "BRIE generates refreshable EHR retrieval evaluations from longitudinal notes, addressing benchmark staleness and leakage while exposing omissions in multi-document clinical synthesis.",
  "summary": "BRIE automatically generates question-answer pairs from longitudinal health records, with its generator validated by **19 clinicians**. The study evaluates **nine LLMs** using **five inference strategies**.",
  "practical_implication": "Builders of retrieval agents can borrow the living-benchmark pattern: validate the generation process, refresh cases to reduce leakage, and allow multiple reference answers where expert reasoning legitimately varies.",
  "agent_context": "BRIE automatically generates question-answer pairs from longitudinal health records, with its generator validated by **19 clinicians**. The study evaluates **nine LLMs** using **five inference strategies**.\n\nBuilders of retrieval agents can borrow the living-benchmark pattern: validate the generation process, refresh cases to reduce leakage, and allow multiple reference answers where expert reasoning legitimately varies.\n\nThe evaluated systems frequently omitted important clinical details, especially when answers required synthesis across documents and encounters. The evidence is specific to electronic health records, where omissions carry unusually high stakes.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2609.30205v1",
    "published_at": "2026-09-24T17:41:16.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "benchmark",
  "domains": [
    "research",
    "data"
  ],
  "topics": [
    "retrieval",
    "agent-evals",
    "benchmark-integrity"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "The evaluated systems frequently omitted important clinical details, especially when answers required synthesis across documents and encounters. The evidence is specific to electronic health records, where omissions carry unusually high stakes."
  ],
  "connected_context": {
    "meaning": "This turns several retrieval-evaluation principles into a renewable clinical benchmark design: validate generation with experts, refresh cases against leakage, and accept legitimate answer plurality. Its omission findings make cross-document synthesis a concrete failure mode rather than a generic retrieval concern, while the EHR setting limits direct performance generalization and raises the cost of incomplete answers.",
    "corpus_size": 875,
    "generated_at": "2026-09-25T09:07:42.999Z",
    "connections": [
      {
        "title": "Verifiable by Construction: Claim-Level Evaluation of Verbatim Citation in Clinical Question Answering",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2609.15964v1",
        "feed7_url": "https://feed7.dev/p/2609-15964v1-08frwh7",
        "reason": "Claim-level coverage and entailment checks offer a direct way to distinguish BRIE’s clinically important omissions from answers that merely contain accurate cited fragments."
      },
      {
        "title": "Inside 847 Production Clinical AI Notes — Sebastian Fox, Composo",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=yqF6XhzbWBk",
        "feed7_url": "https://feed7.dev/p/inside-847-production-clinical-ai-notes-sebastian-fox-composo-0lrc8td",
        "reason": "Production clinical notes independently reinforce omissions as a consequential failure mode and support BRIE’s use of clinician judgment rather than generic model grading alone."
      },
      {
        "title": "Verifiable Environments for AI in Biology — Kenny Workman, LatchBio",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=3ZMUiFaQ3qg",
        "feed7_url": "https://feed7.dev/p/verifiable-environments-for-ai-in-biology-kenny-workman-latchbio-1vs6y66",
        "reason": "Both treat expert validation and multiple valid reasoning paths as necessary constraints on domain benchmarks, rather than assuming one brittle reference or grader captures correctness."
      },
      {
        "title": "Eval awareness in Claude Opus 4.6’s BrowseComp performance",
        "source_name": "Anthropic",
        "source_url": "https://www.anthropic.com/engineering/eval-awareness-browsecomp",
        "feed7_url": "https://feed7.dev/p/eval-awareness-browsecomp-1q6k277",
        "reason": "BrowseComp leakage demonstrates the benchmark-integrity risk that BRIE’s refreshed case generation is designed to reduce, though it does not establish that refreshes eliminate leakage."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-09-24T17:41:16.000Z",
  "modified_at": "2026-09-24T17:41:16.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2609-30205v1-1msmwmk",
    "json": "https://feed7.dev/p/2609-30205v1-1msmwmk.json",
    "markdown": "https://feed7.dev/p/2609-30205v1-1msmwmk.md"
  }
}