{
  "id": 10950,
  "url": "https://arxiv.org/abs/2607.13149v1",
  "title": "Composable Trust for Language Models: A proven boundary and a measured defense",
  "summary": "In a language model, instructions and data share one token stream, so nothing inside the model's generation can keep untrusted text from steering it. We develop a trust model that places the authority to act outside the model, in code: a source's standing, not its content, decides which operation runs and whether it acts. A lower-trust source may inform an answer but not override a higher one. An unmodified model runs inside a deterministic pipeline that ranks inputs by source integrity, and a f",
  "authors": "Yakov Pyotr Shkolnikov",
  "category": "research",
  "topics": "military-security",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-14T18:01:08.000Z",
  "fetched_at": "2026-07-16T05:10:56.605Z",
  "source_slug": "x-arxiv-red-teaming-query",
  "source_name": "arXiv red teaming query",
  "source_homepage": "https://arxiv.org/a/redteam",
  "ethics_ai_record_url": "https://ethics.ai/record/10950",
  "original_url": "https://arxiv.org/abs/2607.13149v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}