{
  "schema_version": "1.1",
  "id": "s13:https://arxiv.org/abs/2607.28617v1",
  "slug": "2607-28617v1-1cbh1qu",
  "url": "https://feed7.dev/p/2607-28617v1-1cbh1qu",
  "title": "AISPA: User-Centric System Prompt Auditing for Large Language Model Applications",
  "why_included": "AISPA turns system-prompt review into an eight-dimension audit. Its survey suggests builders should test prompts for user protection and conflicting instructions, not merely check that safeguards exist.",
  "summary": "AISPA classifies **3,249 instructions** from **88 commercial AI products** as protective or problematic across eight user-centered dimensions. Although 98.9% of products include a protection, only 24% cover every dimension.",
  "practical_implication": "Audit agent system prompts instruction by instruction, checking both coverage and conflicts. A long safety section is weak evidence when protective and user-hostile directives can coexist in the same prompt.",
  "agent_context": "AISPA classifies **3,249 instructions** from **88 commercial AI products** as protective or problematic across eight user-centered dimensions. Although 98.9% of products include a protection, only 24% cover every dimension.\n\nAudit agent system prompts instruction by instruction, checking both coverage and conflicts. A long safety section is weak evidence when protective and user-hostile directives can coexist in the same prompt.\n\nThe study reports that roughly **40% of products** contain at least one problematic instruction, but the abstract does not establish how its taxonomy transfers to private coding-agent harnesses or predicts runtime behavior.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2607.28617v1",
    "published_at": "2026-07-30T17:58:58.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "context",
  "domains": [
    "coding"
  ],
  "topics": [
    "prompting",
    "context-engineering",
    "agent-reliability"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "The study reports that roughly **40% of products** contain at least one problematic instruction, but the abstract does not establish how its taxonomy transfers to private coding-agent harnesses or predicts runtime behavior."
  ],
  "connected_context": {
    "meaning": "AISPA turns prompt review into a user-centered, instruction-level audit: broad safety language is insufficient when protections omit dimensions or coexist with harmful directives. Against the prior candidates, it adds a static governance check that complements—but cannot replace—task-level evaluation of runtime behavior and context sensitivity.",
    "corpus_size": 297,
    "generated_at": "2026-07-31T10:07:52.083Z",
    "connections": [
      {
        "title": "asgeirtj/system_prompts_leaks",
        "source_name": "GitHub",
        "source_url": "https://github.com/asgeirtj/system_prompts_leaks",
        "feed7_url": "https://feed7.dev/p/system-prompts-leaks-0pth2c5",
        "reason": "The prompt archive offers real product prompts to which AISPA’s instruction-level taxonomy could be applied, while AISPA supplies a structured audit method beyond informal comparison."
      },
      {
        "title": "How Evals and Prompts Shape Agent Behavior — Preetika Bhateja & Daniel Bump, YouTube Ads",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=xyL2Ltkh-SA",
        "feed7_url": "https://feed7.dev/p/how-evals-and-prompts-shape-agent-behavior-preetika-bhateja-daniel-bump-1cmecaw",
        "reason": "AISPA’s static conflict and coverage audit complements the production loop of trace review and evals, which is still needed because prompt contents alone do not establish runtime behavior."
      },
      {
        "title": "The Illusion of Robustness: Aggregate Accuracy Hides Prediction Flips under Task-Irrelevant Context",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2607.12963v1",
        "feed7_url": "https://feed7.dev/p/2607-12963v1-1oc0qmr",
        "reason": "The finding that irrelevant context can flip individual outputs reinforces evaluating audited prompts per task rather than assuming improved instruction coverage guarantees stable behavior."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-07-30T17:58:58.000Z",
  "modified_at": "2026-07-30T17:58:58.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2607-28617v1-1cbh1qu",
    "json": "https://feed7.dev/p/2607-28617v1-1cbh1qu.json",
    "markdown": "https://feed7.dev/p/2607-28617v1-1cbh1qu.md"
  }
}