{
  "schema_version": "1.1",
  "id": "s8:https://www.youtube.com/watch?v=PXj0p_mW9nI",
  "slug": "tokens-should-have-jobs-katelyn-lesse-angela-jiang-anthropic-1ov69ze",
  "url": "https://feed7.dev/p/tokens-should-have-jobs-katelyn-lesse-angela-jiang-anthropic-1ov69ze",
  "title": "Tokens Should Have Jobs — Katelyn Lesse & Angela Jiang, Anthropic",
  "why_included": "On Anthropic's financial-analysis bench, assigning tokens to adviser, grader, or memory roles beat pure execution at a fixed budget. Agent topology may matter as much as token count.",
  "summary": "Anthropic tested four token roles on financial-analysis tasks: execution, advising, grading, and dreaming into memory. At the same roughly **600,000-token budget**, execution scored **76** while advising scored **89**, showing that allocation changed results even when spend was held constant.",
  "practical_implication": "Benchmark agent topologies, not just models and larger budgets. Route some compute to advice when token efficiency matters, or to grading and iterative checks when perfect outputs matter; evaluate the cost of obtaining an acceptable result rather than a single run's average score.",
  "agent_context": "Anthropic tested four token roles on financial-analysis tasks: execution, advising, grading, and dreaming into memory. At the same roughly **600,000-token budget**, execution scored **76** while advising scored **89**, showing that allocation changed results even when spend was held constant.\n\nBenchmark agent topologies, not just models and larger budgets. Route some compute to advice when token efficiency matters, or to grading and iterative checks when perfect outputs matter; evaluate the cost of obtaining an acceptable result rather than a single run's average score.\n\nThis is one internally presented financial-task bench, and the material does not provide the dataset, rubric, variance, or replication details. The preferred strategy also changed with the target metric, so the reported ordering should not be generalized to coding-agent workloads without local evals.",
  "source": {
    "name": "AI Engineer",
    "url": "https://www.youtube.com/watch?v=PXj0p_mW9nI",
    "published_at": "2026-09-14T14:00:06.000Z"
  },
  "source_class": "video",
  "content_type": "Video",
  "layer": "benchmark",
  "domains": [
    "data"
  ],
  "topics": [
    "multi-agent",
    "agent-evals",
    "agent-reliability"
  ],
  "verification": {
    "status": "source_linked",
    "label": "Source Linked",
    "method": "source_feed",
    "verified_at": null
  },
  "uncertainty": [
    "This is one internally presented financial-task bench, and the material does not provide the dataset, rubric, variance, or replication details. The preferred strategy also changed with the target metric, so the reported ordering should not be generalized to coding-agent workloads without local evals."
  ],
  "connected_context": {
    "meaning": "This changes equal-budget evaluation from asking how many tokens a model receives to asking what work those tokens perform across an agent topology. The reported advising advantage challenges execution-only scaling, while grading may suit stricter acceptance targets. It also makes repeated sampling a necessary comparator, not an assumed optimum; the financial-only benchmark and missing variance prevent transferring the ordering to other domains.",
    "corpus_size": 778,
    "generated_at": "2026-09-15T10:06:19.289Z",
    "connections": [
      {
        "title": "Sample More, Reflect Less: Self-Refine and Reflexion Lose to Repeated Sampling at Equal Token Cost, from 1.5B to 7B",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2607.28576v1",
        "feed7_url": "https://feed7.dev/p/2607-28576v1-09h2m1u",
        "reason": "Repeated sampling is the direct token-matched baseline that advising, grading, and memory-dreaming topologies must beat before their coordination overhead is credited."
      },
      {
        "title": "The Future of Evals: From LLM as a Judge to Agent as a Judge — Aparna Dhinakaran, Arize AI",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=q2JrUKBMf0w",
        "feed7_url": "https://feed7.dev/p/the-future-of-evals-from-llm-as-a-judge-to-agent-as-a-judge-aparna-dhina-1fu560o",
        "reason": "Allocating tokens to grading expands evaluation into the inference topology, while the candidate warns that fixed judges may still miss failures across long trajectories."
      },
      {
        "title": "First Steps Toward Automated AI Research — Richard Socher, CEO Recursive AI",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=pWXUkLP9uWM",
        "feed7_url": "https://feed7.dev/p/first-steps-toward-automated-ai-research-richard-socher-ceo-recursive-ai-17qblkw",
        "reason": "Automated research uses several specialized agent roles, so this result implies that their token allocation should be evaluated against task-grounded rewards rather than topology alone."
      },
      {
        "title": "Domain-Specific Hallucination Detection in Large Language Models",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2609.11878v1",
        "feed7_url": "https://feed7.dev/p/2609-11878v1-1ltrudx",
        "reason": "The detector’s poor domain transfer reinforces the sourceBrief’s warning that a topology ranking from financial analysis requires local, domain-matched evaluation."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-09-14T14:00:06.000Z",
  "modified_at": "2026-09-14T14:00:06.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/tokens-should-have-jobs-katelyn-lesse-angela-jiang-anthropic-1ov69ze",
    "json": "https://feed7.dev/p/tokens-should-have-jobs-katelyn-lesse-angela-jiang-anthropic-1ov69ze.json",
    "markdown": "https://feed7.dev/p/tokens-should-have-jobs-katelyn-lesse-angela-jiang-anthropic-1ov69ze.md"
  }
}