{
  "slug": "open-weight-safeguard-durability",
  "url": "/research/open-weight-safeguard-durability/",
  "title": "Safeguard Durability in Open Weight Models",
  "type": "statement",
  "typeLabel": "Research statement",
  "lifecycle": "in-preparation",
  "epistemicStatus": "design-only",
  "statusLabel": "In preparation; design, cost measure and controls specified; no attack run, no model evaluated, no result",
  "programme": "deployment",
  "question": "When a model's weights are public, how much adversarial effort does each published safeguard actually impose, and how much of that cost survives a modest fine tuning budget?",
  "date": "2026-07-28",
  "version": "v0.1",
  "authors": [
    "Latent Minds Institute"
  ],
  "methods": [
    "adversarial red teaming across input, decoding, weight modification and composed surfaces (planned)",
    "attack cost curves against a fixed behaviour set (planned)",
    "capability controls separating a defeated safeguard from a degraded model (planned)",
    "attacker effort controls so null results read as bounded search (planned)"
  ],
  "models": [
    "published open weight models (planned)"
  ],
  "evidence": "none; this is a design statement published before any run",
  "code": null,
  "data": null,
  "provenance": "Published in advance of results so the question, the cost measure and the controls are on the record before any finding is",
  "related": [
    "model-entrenchment"
  ],
  "canonicalUrl": "https://latentmindsinstitute.com/research/open-weight-safeguard-durability/",
  "apiUrl": "https://latentmindsinstitute.com/api/research-objects/open-weight-safeguard-durability.json"
}
