{
  "schema_version": "1.1",
  "id": "s2:https://openai.com/index/how-two-settings-tripled-our-arc-agi-3-scores",
  "slug": "how-two-settings-tripled-our-arc-agi-3-scores-0vsgf4b",
  "url": "https://feed7.dev/p/how-two-settings-tripled-our-arc-agi-3-scores-0vsgf4b",
  "title": "How enabling two settings tripled our scores on the ARC-AGI-3 benchmark",
  "why_included": "Two API settings—reasoning retention and compaction—reportedly tripled GPT-5.6’s ARC-AGI-3 score. Agent evals should treat runtime configuration as part of the tested system.",
  "summary": "OpenAI says enabling **reasoning retention** and **compaction** produced **3× ARC-AGI-3 scores** for GPT-5.6 while also improving efficiency.",
  "practical_implication": "Record these settings alongside the model name in agent evaluations. Configuration can materially affect results, so defaults and explicit settings should not be compared as equivalent systems.",
  "agent_context": "OpenAI says enabling **reasoning retention** and **compaction** produced **3× ARC-AGI-3 scores** for GPT-5.6 while also improving efficiency.\n\nRecord these settings alongside the model name in agent evaluations. Configuration can materially affect results, so defaults and explicit settings should not be compared as equivalent systems.\n\nThe supplied material provides no absolute scores, token usage, latency, or experimental detail, leaving the size and generality of the efficiency gain unclear.",
  "source": {
    "name": "OpenAI",
    "url": "https://openai.com/index/how-two-settings-tripled-our-arc-agi-3-scores",
    "published_at": "2026-07-29T15:00:00.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Official Release",
  "layer": "benchmark",
  "domains": [],
  "topics": [
    "agent-evals",
    "benchmark-integrity",
    "context-caching"
  ],
  "verification": {
    "status": "official_source",
    "label": "Official Source",
    "method": "source_feed",
    "verified_at": null
  },
  "uncertainty": [
    "The supplied material provides no absolute scores, token usage, latency, or experimental detail, leaving the size and generality of the efficiency gain unclear."
  ],
  "connected_context": {
    "meaning": "This provides direct evidence that agent configuration can dominate a benchmark result: the same named model scored three times higher with reasoning retention and compaction enabled. It strengthens the case for treating model, harness, state-management settings, and resource conditions as one evaluated system, while absent absolute scores and experimental details limit generalization beyond ARC-AGI-3.",
    "corpus_size": 297,
    "generated_at": "2026-07-31T10:05:39.218Z",
    "connections": [
      {
        "title": "Quantifying infrastructure noise in agentic coding evals",
        "source_name": "Anthropic",
        "source_url": "https://www.anthropic.com/engineering/infrastructure-noise",
        "feed7_url": "https://feed7.dev/p/infrastructure-noise-1jyyyw1",
        "reason": "Both show that non-model setup can materially shift agent scores, extending reproducibility requirements from container resources to reasoning-state and context-management settings."
      },
      {
        "title": "State of Data — Sean Cai, Independent / State of Data",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=ZyIoTOAbRfs",
        "feed7_url": "https://feed7.dev/p/state-of-data-sean-cai-independent-state-of-data-0v9fy69",
        "reason": "The result supplies concrete support for the claim that scores are scaffold-dependent and makes cross-harness comparisons without configuration disclosure especially weak."
      },
      {
        "title": "Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2607.28576v1",
        "feed7_url": "https://feed7.dev/p/2607-28576v1-09h2m1u",
        "reason": "Because compaction and retained reasoning may change generated-token use, the equal-token accounting proposed here is a prerequisite for judging whether the reported efficiency improvement is comparable."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-07-29T15:00:00.000Z",
  "modified_at": "2026-07-29T15:00:00.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/how-two-settings-tripled-our-arc-agi-3-scores-0vsgf4b",
    "json": "https://feed7.dev/p/how-two-settings-tripled-our-arc-agi-3-scores-0vsgf4b.json",
    "markdown": "https://feed7.dev/p/how-two-settings-tripled-our-arc-agi-3-scores-0vsgf4b.md"
  }
}