{
  "id": 18791,
  "url": "https://arxiv.org/abs/2608.11705v1",
  "title": "Making Your LLMs More Objective: Stabilizing LLM Safety Behavior Across Traits with Trait-Invariant Safety Tuning",
  "summary": "Aligned large language models (LLMs) are expected to exhibit safety behavior based on the content of the user request: they should refuse unsafe requests and comply with safe ones. However, we show that the same request can elicit substantially different safety decisions under different traits assigned in the system prompt, a failure mode we call trait-induced safety variation. To measure this failure, we introduce refusal-based metrics: Trait-Induced Deviation measures dataset-level deviation f",
  "authors": "Lang Cao",
  "category": "research",
  "topics": null,
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-12T06:33:55.000Z",
  "fetched_at": "2026-08-13T05:10:37.786Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/18791",
  "original_url": "https://arxiv.org/abs/2608.11705v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}