{
  "schema_version": "1.1",
  "id": "auto-31ceb9467c",
  "slug": "quotebench-how-matched-scores-can-hide-command-path-fail-31ceb9467c",
  "url": "https://feed7.dev/p/quotebench-how-matched-scores-can-hide-command-path-fail-31ceb9467c",
  "title": "QuoteBench: How Matched Scores Can Hide Command-Path Failures",
  "why_included": "Test commands through the exact production transport and verify final state, since one parser cut task completion by 55.4–73.2 points.",
  "summary": "QuoteBench shows that shell-command scores can conceal failures introduced by serialization and reparsing. Agent evals should identify the execution path, not attribute every result to the model.",
  "practical_implication": "When the boundary was disclosed, six configurations recovered 30.4–60.7 points. Builders should test generated commands through the exact production transport and validate final state, especially where wrappers interpolate or reparse shell text.",
  "agent_context": "QuoteBench tests **56 one-shot tasks** from 14 incident-derived families across eight configurations. Adding one unescaped parser cut replayed-command success by **55.4–73.2 percentage points**.\n\nWhen the boundary was disclosed, six configurations recovered **30.4–60.7 points**. Builders should test generated commands through the exact production transport and validate final state, especially where wrappers interpolate or reparse shell text.\n\nAdaptation was uneven: two configurations recovered nothing or declined slightly. GPT-5.6-sol’s **−3.6-point matched gap** concealed −64.3 points of transport damage plus 60.7 points of compensation, so aggregate scores can misstate both model and harness quality.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2608.13547v1",
    "published_at": "2026-08-13T00:00:00.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "benchmark",
  "domains": [
    "coding",
    "security"
  ],
  "topics": [
    "agent-evals",
    "benchmark-integrity",
    "harness-engineering"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "Automatically selected from source material; feed7 has not independently tested the claim."
  ],
  "connected_context": null,
  "lifecycle": "New",
  "published_at": "2026-08-13T00:00:00.000Z",
  "modified_at": "2026-08-13T00:00:00.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/quotebench-how-matched-scores-can-hide-command-path-fail-31ceb9467c",
    "json": "https://feed7.dev/p/quotebench-how-matched-scores-can-hide-command-path-fail-31ceb9467c.json",
    "markdown": "https://feed7.dev/p/quotebench-how-matched-scores-can-hide-command-path-fail-31ceb9467c.md"
  }
}