{
  "schema_version": "1.1",
  "id": "archive:https://arxiv.org/abs/2608.16868v1",
  "slug": "2608-16868v1-00830as",
  "url": "https://feed7.dev/p/2608-16868v1-00830as",
  "title": "Towards Computational Provenance: Carrying Causal-State Evidence in Generated Text",
  "why_included": "A controlled study encoded authenticated internal-state evidence into unchanged answers, suggesting generated text could carry provenance signals, but not that current models reveal them naturally.",
  "summary": "Researchers forced arithmetic models through one of two discrete internal paths, authenticated the chosen state, and encoded a subtle detectable pattern in otherwise equivalent output. Feed-forward and transformer systems passed **all 128 matched pairs** in public and sealed evaluations.",
  "practical_implication": "For builders, this sketches a future provenance mechanism in which output carries evidence about a causally relevant computation, not just the final answer. Replication across **five feed-forward models** and **three transformers** supports the controlled mechanism.",
  "agent_context": "Researchers forced arithmetic models through one of two discrete internal paths, authenticated the chosen state, and encoded a subtle detectable pattern in otherwise equivalent output. Feed-forward and transformer systems passed **all 128 matched pairs** in public and sealed evaluations.\n\nFor builders, this sketches a future provenance mechanism in which output carries evidence about a causally relevant computation, not just the final answer. Replication across **five feed-forward models** and **three transformers** supports the controlled mechanism.\n\nThis is a bounded proof of concept using deliberately trained architectures and an engineered signal. In a separate answer-only transformer experiment, linear probes did not recover a naturally learned intermediate state, so the work does not validate provenance for ordinary model outputs.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2608.16868v1",
    "published_at": "2026-08-17T17:50:04.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "benchmark",
  "domains": [
    "security"
  ],
  "topics": [
    "agent-evals",
    "benchmark-integrity"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "This is a bounded proof of concept using deliberately trained architectures and an engineered signal. In a separate answer-only transformer experiment, linear probes did not recover a naturally learned intermediate state, so the work does not validate provenance for ordinary model outputs."
  ],
  "connected_context": {
    "meaning": "This narrows provenance claims from inspecting outputs or probing naturally learned representations to deliberately training an authenticated causal state into the output. The controlled success shows that computation-path evidence can be carried, while the failed answer-only probe warns that ordinary models cannot yet be assumed to expose such provenance naturally.",
    "corpus_size": 479,
    "generated_at": "2026-08-18T10:05:01.713Z",
    "connections": [
      {
        "title": "What Do Compliance Detectors Read? An Audit of Activation Probes and Guard Models",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2608.16852v1",
        "feed7_url": "https://feed7.dev/p/2608-16852v1-0580uyd",
        "reason": "The compliance audit shows probes can appear accurate without reading the governing rule; this work contrasts that failure with an engineered signal tied to a forced, authenticated internal path."
      },
      {
        "title": "QuoteBench: How Matched Scores Can Hide Command-Path Failures",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2608.13547v1",
        "feed7_url": "https://feed7.dev/p/2608-13547v1-130h6xd",
        "reason": "QuoteBench shows final scores can conceal which command path produced them; computational provenance sketches a complementary way to attach evidence about a causally relevant internal path to the output itself."
      },
      {
        "title": "The Low Frequency Trap: Video Language Models Fail at Simple Event Bookkeeping",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2608.06361v1",
        "feed7_url": "https://feed7.dev/p/2608-06361v1-1n3dr85",
        "reason": "Both reject answer-only confidence: the video study requires event-level traces, while this work tests whether evidence of an intermediate computational state can travel with the answer."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-08-17T17:50:04.000Z",
  "modified_at": "2026-08-17T17:50:04.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2608-16868v1-00830as",
    "json": "https://feed7.dev/p/2608-16868v1-00830as.json",
    "markdown": "https://feed7.dev/p/2608-16868v1-00830as.md"
  }
}