{
  "schema_version": "1.1",
  "id": "s13:https://arxiv.org/abs/2609.26751v1",
  "slug": "2609-26751v1-0om13uz",
  "url": "https://feed7.dev/p/2609-26751v1-0om13uz",
  "title": "EquivSVA: A Formally Verified Dataset of Behavioral Assertions Across Equivalent RTL Implementations",
  "why_included": "EquivSVA tests whether generated hardware assertions describe interface behavior rather than quirks of one RTL implementation, exposing robustness gaps hidden by single-implementation evals.",
  "summary": "EquivSVA organizes **120 behavior families** into four equivalent RTL implementations each, totaling **480 implementations**, 914 gold properties, and 360 mutants. Every family passes a 17-job formal-validation suite.",
  "practical_implication": "Builders evaluating assertion agents should split by behavior family and test outputs across structurally different implementations. In the Qwen2.5-Coder-7B-Instruct case study, only **93 of 293** interface-only properties were formally sound.",
  "agent_context": "EquivSVA organizes **120 behavior families** into four equivalent RTL implementations each, totaling **480 implementations**, 914 gold properties, and 360 mutants. Every family passes a 17-job formal-validation suite.\n\nBuilders evaluating assertion agents should split by behavior family and test outputs across structurally different implementations. In the Qwen2.5-Coder-7B-Instruct case study, only **93 of 293** interface-only properties were formally sound.\n\nThis is a specialized hardware-verification dataset, not a broad coding-agent benchmark. Sound-property counts varied across equivalent implementations in **14 of 24** test families, but the paper reports only one model demonstration.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2609.26751v1",
    "published_at": "2026-09-22T17:33:09.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "benchmark",
  "domains": [
    "coding"
  ],
  "topics": [
    "agent-evals",
    "benchmark-integrity",
    "agent-reliability"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "This is a specialized hardware-verification dataset, not a broad coding-agent benchmark. Sound-property counts varied across equivalent implementations in **14 of 24** test families, but the paper reports only one model demonstration."
  ],
  "connected_context": {
    "meaning": "This adds formally verified, representation-varied evaluation to the benchmark-integrity toolkit. Family-level splits reduce leakage between equivalent implementations, while cross-implementation testing reveals whether assertions capture behavior rather than syntax. The single-model result exposes substantial soundness and robustness gaps but does not establish how other assertion agents perform.",
    "corpus_size": 856,
    "generated_at": "2026-09-23T09:06:24.727Z",
    "connections": [
      {
        "title": "When LLM Decompilers Recompile More and Preserve Less",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2609.05370v1",
        "feed7_url": "https://feed7.dev/p/2609-05370v1-03wuqdv",
        "reason": "Both require semantic equivalence checks beyond surface success: EquivSVA formally validates assertions across equivalent RTL, while the decompiler study differentially executes reconstructed code against the original."
      },
      {
        "title": "Computer Use at the Edge of the Statistical Precipice — Pierluca D'Oro, Programma Labs",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=CTLa_p6iOiY",
        "feed7_url": "https://feed7.dev/p/computer-use-at-the-edge-of-the-statistical-precipice-pierluca-d-oro-pro-16136hs",
        "reason": "Equivalent RTL implementations provide controlled environment variation, implementing the broader recommendation to vary benchmark state so fixed structural patterns cannot stand in for adaptation."
      },
      {
        "title": "When Will The Benchmaxxing Plague End? — Nick Heiner, Surge AI",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=-npY6XjM8CQ",
        "feed7_url": "https://feed7.dev/p/when-will-the-benchmaxxing-plague-end-nick-heiner-surge-ai-178gqcg",
        "reason": "The family-level split and formal validation directly address two benchmark risks highlighted there: leakage between related cases and weak verifiers that reward shortcuts."
      },
      {
        "title": "Metrics Failure in LLM-Based Code Vulnerability Repair: An Empirical Study and a Change-Aware Screen",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2609.26749v1",
        "feed7_url": "https://feed7.dev/p/2609-26749v1-1fj6jga",
        "reason": "Both show that convenient proxy metrics can misstate correctness and replace them with domain-grounded validation, formal soundness here and execution-grounded security checks for repair."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-09-22T17:33:09.000Z",
  "modified_at": "2026-09-22T17:33:09.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2609-26751v1-0om13uz",
    "json": "https://feed7.dev/p/2609-26751v1-0om13uz.json",
    "markdown": "https://feed7.dev/p/2609-26751v1-0om13uz.md"
  }
}