{
  "schema_version": "1.1",
  "id": "s13:https://arxiv.org/abs/2609.20779v1",
  "slug": "2609-20779v1-0le7ma3",
  "url": "https://feed7.dev/p/2609-20779v1-0le7ma3",
  "title": "Harm Laundering in GPT Models: Evidence That Gender Discrimination Is Transformed Rather Than Reduced Across Safety-Trained Generations",
  "why_included": "Lower toxicity scores may hide changes in representational harm rather than its removal. Safety evals need topic and demographic analysis alongside surface-form classifiers.",
  "summary": "Across **450,000 completions** from **15 GPT-lineage models**, explicit discriminatory clusters declined while subtler representational disparities remained or emerged. Three classifiers marked a GPT-5 breast-cancer topic framed around men’s rights as non-toxic.",
  "practical_implication": "Do not use a falling toxicity score as the sole release gate for generative features. Compare topic coverage and representation across demographic conditions, then inspect whether safety tuning redistributes who receives positive, negative, or constrained portrayals.",
  "agent_context": "Across **450,000 completions** from **15 GPT-lineage models**, explicit discriminatory clusters declined while subtler representational disparities remained or emerged. Three classifiers marked a GPT-5 breast-cancer topic framed around men’s rights as non-toxic.\n\nDo not use a falling toxicity score as the sole release gate for generative features. Compare topic coverage and representation across demographic conditions, then inspect whether safety tuning redistributes who receives positive, negative, or constrained portrayals.\n\nWomen-directed topic diversity was **36% lower** than men-directed diversity at the GPT-4 alignment boundary. The study covers one model lineage and three demographic conditions, so its proposed detection protocol still needs validation across other models and forms of harm.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2609.20779v1",
    "published_at": "2026-09-17T17:49:28.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "benchmark",
  "domains": [],
  "topics": [
    "benchmark-integrity",
    "agent-evals"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "Women-directed topic diversity was **36% lower** than men-directed diversity at the GPT-4 alignment boundary. The study covers one model lineage and three demographic conditions, so its proposed detection protocol still needs validation across other models and forms of harm."
  ],
  "connected_context": {
    "meaning": "This identifies a release-evaluation failure mode that aggregate toxicity metrics can hide: safety tuning may change harm from explicit language into unequal topic coverage and representation. It therefore broadens benchmark integrity from detecting obvious harmful outputs to comparing distributions across demographic conditions, while limiting the conclusion to one model lineage and three tested conditions.",
    "corpus_size": 807,
    "generated_at": "2026-09-18T10:07:02.227Z",
    "connections": [
      {
        "title": "Beyond Scores: Understanding LLM-as-a-Judge Mechanisms in Summarization Evaluation",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2609.01604v1",
        "feed7_url": "https://feed7.dev/p/2609-01604v1-02vljon",
        "reason": "Both show that a scalar evaluator can conceal the underlying failure; the judge study motivates probing evaluation mechanisms, while this work requires demographic coverage and representation checks beyond toxicity labels."
      },
      {
        "title": "The Low Frequency Trap: Video Language Models Fail at Simple Event Bookkeeping",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2608.06361v1",
        "feed7_url": "https://feed7.dev/p/2608-06361v1-1n3dr85",
        "reason": "The video study likewise finds that improved aggregate scores need not reflect faithful underlying behavior, reinforcing the need for structured evidence checks rather than answer-level metrics alone."
      },
      {
        "title": "Inside the Unfair Judge: A Mechanistic Interpretability Account of LLM-as-Judge Bias",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2607.11871v1",
        "feed7_url": "https://feed7.dev/p/2607-11871v1-17vejh0",
        "reason": "The mechanistic account of judge bias supports scrutinizing evaluators themselves, which is consequential here because three classifiers failed to flag one reported form of representational harm."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-09-17T17:49:28.000Z",
  "modified_at": "2026-09-17T17:49:28.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2609-20779v1-0le7ma3",
    "json": "https://feed7.dev/p/2609-20779v1-0le7ma3.json",
    "markdown": "https://feed7.dev/p/2609-20779v1-0le7ma3.md"
  }
}