{
  "schema_version": "1.1",
  "id": "s13:https://arxiv.org/abs/2607.28609v1",
  "slug": "2607-28609v1-0k011ot",
  "url": "https://feed7.dev/p/2607-28609v1-0k011ot",
  "title": "OSReward: Instituting Standardized Evaluation for Cross-Platform Computer-Use Reward Models",
  "why_included": "OSReward finds that VLM judges often approve failed computer-use runs. Its benchmark and open reward models offer a more grounded way to evaluate trajectories without paying frontier-model costs.",
  "summary": "OSReward evaluates vision-language judges on human-verified computer-use trajectories and adds **OSReward-Hard** and **OSReward-Multi**. The study finds a shared leniency bias: judges often classify failed runs as completed.",
  "practical_implication": "Do not treat one model-judge verdict as ground truth for browser or desktop agents. Calibrate against human-labeled failures, track false approvals, and consider the released **OS-Shepherd 9B and 35B** models for repeatable scoring.",
  "agent_context": "OSReward evaluates vision-language judges on human-verified computer-use trajectories and adds **OSReward-Hard** and **OSReward-Multi**. The study finds a shared leniency bias: judges often classify failed runs as completed.\n\nDo not treat one model-judge verdict as ground truth for browser or desktop agents. Calibrate against human-labeled failures, track false approvals, and consider the released **OS-Shepherd 9B and 35B** models for repeatable scoring.\n\nThe authors report commercial-level judging at **30–60% lower cost** than frontier models, but the supplied material gives no per-platform scores or evidence that these reward models generalize to a builder’s own UI and task mix.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2607.28609v1",
    "published_at": "2026-07-30T17:57:41.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "benchmark",
  "domains": [
    "coding"
  ],
  "topics": [
    "computer-use",
    "agent-evals",
    "agent-reliability"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "The authors report commercial-level judging at **30–60% lower cost** than frontier models, but the supplied material gives no per-platform scores or evidence that these reward models generalize to a builder’s own UI and task mix."
  ],
  "connected_context": {
    "meaning": "OSReward identifies false approval as a specific failure mode in computer-use judging and makes human-calibrated failure detection a prerequisite for trusting automated trajectory scores. It strengthens the prior case for replayable, step-aware evaluation while narrowing it: even a reproducible rollout is misleading if its judge systematically accepts failed runs.",
    "corpus_size": 297,
    "generated_at": "2026-07-31T10:07:52.083Z",
    "connections": [
      {
        "title": "Desktop-Delta Bench: Do Computer-Use Models Understand Desktop GUI Transitions?",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2607.26041v1",
        "feed7_url": "https://feed7.dev/p/2607-26041v1-1x1gw81",
        "reason": "Desktop-Delta Bench exposes missed GUI transitions; OSReward adds the consequence for evaluation pipelines, showing that trajectory judges themselves may approve failures and therefore need calibration."
      },
      {
        "title": "Everything Is a Rollout — Alex Shaw + Ryan Marten, Terminal-Bench, Harbor, Laude Institute",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=jRCpXUjz4CI",
        "feed7_url": "https://feed7.dev/p/everything-is-a-rollout-alex-shaw-ryan-marten-terminal-bench-harbor-laud-0iz4rgx",
        "reason": "Harbor’s reproducible rollout loop provides the evaluation setting, while OSReward shows that outcome verification within that loop must be checked against human-labeled failures."
      },
      {
        "title": "The Future of Evals: From LLM as a Judge to Agent as a Judge — Aparna Dhinakaran, Arize AI",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=q2JrUKBMf0w",
        "feed7_url": "https://feed7.dev/p/the-future-of-evals-from-llm-as-a-judge-to-agent-as-a-judge-aparna-dhina-1fu560o",
        "reason": "OSReward supplies evidence for retaining calibrated checks around model-based trajectory analysis: more flexible judges do not remove the risk of systematic leniency."
      },
      {
        "title": "From Agent Traces to Agent Simulations — Rustem Feyzkhanov, Snorkel AI",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=Ib5t2RLtxvM",
        "feed7_url": "https://feed7.dev/p/from-agent-traces-to-agent-simulations-rustem-feyzkhanov-snorkel-ai-0zwlzjq",
        "reason": "Replayable production traces can test reward models under a team’s actual UI and task mix, addressing OSReward’s unresolved generalization beyond its evaluated platforms."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-07-30T17:57:41.000Z",
  "modified_at": "2026-07-30T17:57:41.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2607-28609v1-0k011ot",
    "json": "https://feed7.dev/p/2607-28609v1-0k011ot.json",
    "markdown": "https://feed7.dev/p/2607-28609v1-0k011ot.md"
  }
}