{
  "schema_version": "1.0",
  "id": "s13:https://arxiv.org/abs/2607.22513v1",
  "slug": "2607-22513v1-0yf6na5",
  "url": "https://feed7.dev/p/2607-22513v1-0yf6na5",
  "title": "Opaque Epistemic Mediation: How LLM Deployment Configurations Shape the Validation of Pseudo-Science",
  "why_included": "The same model identifier produced sharply different judgments across API and web deployments. Treat model, interface, system configuration, and date as one versioned dependency.",
  "summary": "Researchers tested **four model families** from **October 2025 to February 2026** on a contested pseudo-scientific claim. Grok Fast scored it 70–75 versus 15–40 for the other families, while control prompts did not show the same gap.",
  "practical_implication": "For research agents, do not treat a model ID as a stable epistemic contract. Pin the deployment channel, record dates and configuration, and rerun domain-specific validation after silent service changes.",
  "agent_context": "Researchers tested **four model families** from **October 2025 to February 2026** on a contested pseudo-scientific claim. Grok Fast scored it 70–75 versus 15–40 for the other families, while control prompts did not show the same gap.\n\nFor research agents, do not treat a model ID as a stable epistemic contract. Pin the deployment channel, record dates and configuration, and rerun domain-specific validation after silent service changes.\n\nThe starkest mismatch was the same Grok identifier scoring **75 through the API and 5.5 on the web** three months later. This is a narrow case study, so it demonstrates deployment sensitivity rather than broad comparative model quality.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2607.22513v1",
    "published_at": "2026-07-24T17:32:43.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "benchmark",
  "domains": [
    "research"
  ],
  "topics": [
    "model-selection",
    "agent-reliability",
    "benchmark-integrity"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "The starkest mismatch was the same Grok identifier scoring **75 through the API and 5.5 on the web** three months later. This is a narrow case study, so it demonstrates deployment sensitivity rather than broad comparative model quality."
  ],
  "lifecycle": "Current",
  "published_at": "2026-07-24T17:32:43.000Z",
  "modified_at": "2026-07-24T17:32:43.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2607-22513v1-0yf6na5",
    "json": "https://feed7.dev/p/2607-22513v1-0yf6na5.json",
    "markdown": "https://feed7.dev/p/2607-22513v1-0yf6na5.md"
  }
}