{
  "schema_version": "1.1",
  "id": "auto-b3f2ae3bdf",
  "slug": "phantom-gains-auditing-self-improvement-against-a-measur-b3f2ae3bdf",
  "url": "https://feed7.dev/p/phantom-gains-auditing-self-improvement-against-a-measur-b3f2ae3bdf",
  "title": "Phantom Gains: Auditing Self-Improvement Against a Measured Null",
  "why_included": "Add frozen-baseline replicates and a measured null before treating one-decode gains or regressions as real.",
  "summary": "Per-problem self-improvement claims can arise from inference and evaluation noise. This audit argues every transition statistic needs a measured null from frozen baseline replicates.",
  "practical_implication": "For agent and model evaluations, do not treat problem-level gains and losses as ground truth from one decode. Measure each statistic’s null with baseline replicates, then use per-problem tests and false-discovery-rate control.",
  "agent_context": "The study ran three rounds of rank-32 LoRA self-training on Qwen3-8B alongside a frozen control using the same pipeline. It found **seven measurement failures** capable of reversing findings when the control was omitted.\n\nFor agent and model evaluations, do not treat problem-level gains and losses as ground truth from one decode. Measure each statistic’s null with baseline replicates, then use per-problem tests and false-discovery-rate control.\n\nA single greedy decode gave an untrained model an apparent expansion rate of **0.280**, while the replacement exact test detected **nothing on held-out replicates**. Results for problems the base model never solved remained inconclusive, and the method needs more baseline replicates than many studies possess.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2608.20290v1",
    "published_at": "2026-08-20T00:00:00.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "benchmark",
  "domains": [
    "research"
  ],
  "topics": [
    "benchmark-integrity",
    "agent-evals",
    "agent-reliability"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "Automatically selected from source material; feed7 has not independently tested the claim."
  ],
  "connected_context": null,
  "lifecycle": "New",
  "published_at": "2026-08-20T00:00:00.000Z",
  "modified_at": "2026-08-20T00:00:00.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/phantom-gains-auditing-self-improvement-against-a-measur-b3f2ae3bdf",
    "json": "https://feed7.dev/p/phantom-gains-auditing-self-improvement-against-a-measur-b3f2ae3bdf.json",
    "markdown": "https://feed7.dev/p/phantom-gains-auditing-self-improvement-against-a-measur-b3f2ae3bdf.md"
  }
}