{
  "schema_version": "1.0",
  "id": "s8:https://www.youtube.com/watch?v=q2JrUKBMf0w",
  "slug": "the-future-of-evals-from-llm-as-a-judge-to-agent-as-a-judge-aparna-dhina-1fu560o",
  "url": "https://feed7.dev/p/the-future-of-evals-from-llm-as-a-judge-to-agent-as-a-judge-aparna-dhina-1fu560o",
  "title": "The Future of Evals: From LLM as a Judge to Agent as a Judge — Aparna Dhinakaran, Arize AI",
  "why_included": "Fixed-rubric judges miss failures that emerge across long, variable agent trajectories. Arize argues for adding agent-based analysis while retaining deterministic and LLM-judge evals.",
  "summary": "Arize says it processes **over 100 million evals per month**; the average team runs about **12 eval jobs**, while top teams use more than 3,800 evaluators. As agents gained tools, memory, subagents, and long-running loops, their failures became harder to capture with fixed scores.",
  "practical_implication": "Keep deterministic checks and LLM judges for stable criteria, but add trajectory-aware analysis when the path itself matters. An evaluator agent can inspect traces for repeated tool calls, loops, forgotten context, incomplete work, or inefficient execution.",
  "agent_context": "Arize says it processes **over 100 million evals per month**; the average team runs about **12 eval jobs**, while top teams use more than 3,800 evaluators. As agents gained tools, memory, subagents, and long-running loops, their failures became harder to capture with fixed scores.\n\nKeep deterministic checks and LLM judges for stable criteria, but add trajectory-aware analysis when the path itself matters. An evaluator agent can inspect traces for repeated tool calls, loops, forgotten context, incomplete work, or inefficient execution.\n\nAgent-as-judge is presented as an additional eval type, not a replacement. The talk introduces Arize Signal as the implementation, but provides no comparative accuracy, cost, latency, or calibration results for deciding when it beats a fixed judge.",
  "source": {
    "name": "AI Engineer",
    "url": "https://www.youtube.com/watch?v=q2JrUKBMf0w",
    "published_at": "2026-07-24T20:00:06.000Z"
  },
  "source_class": "video",
  "content_type": "Video",
  "layer": "benchmark",
  "domains": [
    "coding"
  ],
  "topics": [
    "agent-evals",
    "agent-reliability",
    "multi-agent"
  ],
  "verification": {
    "status": "source_linked",
    "label": "Source Linked",
    "method": "source_feed",
    "verified_at": null
  },
  "uncertainty": [
    "Agent-as-judge is presented as an additional eval type, not a replacement. The talk introduces Arize Signal as the implementation, but provides no comparative accuracy, cost, latency, or calibration results for deciding when it beats a fixed judge."
  ],
  "lifecycle": "Current",
  "published_at": "2026-07-24T20:00:06.000Z",
  "modified_at": "2026-07-24T20:00:06.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/the-future-of-evals-from-llm-as-a-judge-to-agent-as-a-judge-aparna-dhina-1fu560o",
    "json": "https://feed7.dev/p/the-future-of-evals-from-llm-as-a-judge-to-agent-as-a-judge-aparna-dhina-1fu560o.json",
    "markdown": "https://feed7.dev/p/the-future-of-evals-from-llm-as-a-judge-to-agent-as-a-judge-aparna-dhina-1fu560o.md"
  }
}