{
  "schema_version": "1.1",
  "id": "s13:https://arxiv.org/abs/2609.02797v1",
  "slug": "2609-02797v1-15ybio3",
  "url": "https://feed7.dev/p/2609-02797v1-15ybio3",
  "title": "Dutch Books for Language Models",
  "why_included": "A label-free Dutch-book test finds internally inconsistent probabilities from language models, especially when prompts add logical complexity or irrelevant context.",
  "summary": "The authors elicit forecasts for events derived from stock-return data, then use linear programs to find guaranteed betting profit against the model. The test needs **no outcome labels**, so it can assess unresolved events.",
  "practical_implication": "Treat agent-generated probabilities as claims to validate, not a coherent world model. Strip irrelevant prompt details and test related forecasts together; the paper reports that such details can raise incoherence by **an order of magnitude**.",
  "agent_context": "The authors elicit forecasts for events derived from stock-return data, then use linear programs to find guaranteed betting profit against the model. The test needs **no outcome labels**, so it can assess unresolved events.\n\nTreat agent-generated probabilities as claims to validate, not a coherent world model. Strip irrelevant prompt details and test related forecasts together; the paper reports that such details can raise incoherence by **an order of magnitude**.\n\nThis measures internal coherence, not whether forecasts are calibrated or accurate. The supplied abstract also does not identify the evaluated models or provide per-model results.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2609.02797v1",
    "published_at": "2026-09-02T16:31:47.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "benchmark",
  "domains": [
    "research",
    "data"
  ],
  "topics": [
    "agent-evals",
    "agent-reliability"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "This measures internal coherence, not whether forecasts are calibrated or accurate. The supplied abstract also does not identify the evaluated models or provide per-model results."
  ],
  "connected_context": {
    "meaning": "This adds a label-free reliability test for unresolved forecasts: evaluate whether related probabilities admit a guaranteed-loss betting strategy, independently of eventual accuracy or calibration. It therefore complements task-success and domain-validity benchmarks rather than replacing them. The reported sensitivity to irrelevant details also makes prompt variation part of coherence testing.",
    "corpus_size": 669,
    "generated_at": "2026-09-03T10:01:48.808Z",
    "connections": [
      {
        "title": "SocietyBench: Forecasting Counterfactual Social-World Evolution",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2608.04009v1",
        "feed7_url": "https://feed7.dev/p/2608-04009v1-05m5u8w",
        "reason": "SocietyBench separates calibration from temporal accuracy; Dutch-book testing adds a third, distinct property—internal coherence among forecasts—and can assess it before outcomes resolve."
      },
      {
        "title": "Fisher-R1: Training LLM Agents for Reliable Hypothesis Testing",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2608.07437v1",
        "feed7_url": "https://feed7.dev/p/2608-07437v1-0t4v96m",
        "reason": "Fisher-R1 checks whether an agent selects statistically valid methods, whereas this test checks whether its probability claims are mutually coherent; together they cover different failures in quantitative reasoning."
      },
      {
        "title": "MedPRESS: A Multi-turn Benchmark for Patient-Pressure-Induced Medical Sycophancy in LLMs",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2608.02520v1",
        "feed7_url": "https://feed7.dev/p/2608-02520v1-16negmk",
        "reason": "MedPRESS shows judgment can shift under escalating pressure, while this paper reports incoherence rising with irrelevant prompt details; both support evaluating reliability across prompt variations rather than a single static formulation."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-09-02T16:31:47.000Z",
  "modified_at": "2026-09-02T16:31:47.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2609-02797v1-15ybio3",
    "json": "https://feed7.dev/p/2609-02797v1-15ybio3.json",
    "markdown": "https://feed7.dev/p/2609-02797v1-15ybio3.md"
  }
}