{
  "slug": "agentic-misalignment-detection",
  "url": "/research/agentic-misalignment-detection/",
  "title": "Preemptive Detection of Agentic Misalignment, and Its Shelf Life",
  "type": "statement",
  "typeLabel": "Research statement",
  "lifecycle": "in-preparation",
  "epistemicStatus": "design-only",
  "statusLabel": "In preparation; two adjacent questions and a shared design specified; no probe fitted, no intervention run, no drift measured",
  "programme": "cognition",
  "question": "Can representation engineering detect agentic misalignment from internal state before the action is taken, and does the self representation such a detector reads remain stable as reinforcement learning horizons lengthen?",
  "date": "2026-07-28",
  "version": "v0.1",
  "authors": [
    "Latent Minds Institute"
  ],
  "methods": [
    "readouts over a fixed set of agentic settings, taken before the action is emitted (planned)",
    "intervention to separate a used internal variable from a merely decodable one (planned)",
    "re-evaluation across a reinforcement learning checkpoint series at matched behavioural performance (planned)",
    "surface form, capability and lexical baseline controls (planned)"
  ],
  "models": [
    "open weight models with reinforcement learning checkpoint series (planned)"
  ],
  "evidence": "none; this is a design statement published before any run",
  "code": null,
  "data": null,
  "provenance": "The two questions are posed together because the second sets the shelf life of any detector built under the first",
  "related": [
    "evaluation-state",
    "latent-signatures-strategic-games"
  ],
  "canonicalUrl": "https://latentmindsinstitute.com/research/agentic-misalignment-detection/",
  "apiUrl": "https://latentmindsinstitute.com/api/research-objects/agentic-misalignment-detection.json"
}
