{
  "schemaVersion": "1.0",
  "kind": "journalism_feature",
  "status": "published",
  "publisher": "The Amateur Limited",
  "published": "2026-09-26",
  "title": "Alignment to what? Why AI safety can look like whack-a-mole",
  "canonical": "https://theamateur.co.uk/research/ai/2026-09-26/alignment-to-what/",
  "standfirst": "The leading AI labs are getting better at suppressing individual failures. Their own results raise a deeper question: whether morality is ultimately something humans specify to an artificial mind, or something an intelligent agent should learn to discover.",
  "thesis": "A falsifiable research proposition: recurring alignment failure modes may partly reflect treating normativity as ultimately supplied by human specification rather than testing whether models can be trained to treat at least some moral truths as discoverable under uncertainty.",
  "irProblem": "intelligence:problem:3c8f1f43-3aa8-4c19-a615-4bb0d8127a25",
  "concepts": [
    "alignment whack-a-mole",
    "moral discovery",
    "moral realism",
    "A→A* dynamic task reinterpretation",
    "authority farming",
    "honour",
    "thumos",
    "phronesis",
    "natural law",
    "classical theism"
  ],
  "proposedExperimentalArms": [
    {
      "id": "S",
      "label": "Specification",
      "description": "Human-written safety and behavioural specification; model trained to reason over it."
    },
    {
      "id": "C",
      "label": "Reasons / character",
      "description": "Ethical reasons, admirable character, practical judgment and examples of acting well."
    },
    {
      "id": "MR",
      "label": "Moral discovery",
      "description": "Treat at least some moral propositions as potentially objective and discoverable under uncertainty; human instructions remain important but not definitionally infallible."
    },
    {
      "id": "CT",
      "label": "Classical virtue / theist",
      "description": "Add virtue, honour, practical wisdom, ordered goods, natural law, conscience and the classical-theist grounding of goodness."
    }
  ],
  "heldOutEvaluations": [
    "dynamic task reinterpretation A→A*",
    "corrupt completion / sunk cost",
    "human-authority farming",
    "private integrity",
    "moral invariance",
    "moral uncertainty",
    "multi-agent authority spoofing"
  ],
  "visuals": [
    {
      "file": "../assets/alignment-hero.webp",
      "role": "hero-summary",
      "caption": "From patching failures to moral discovery: repeated fixes, the A→A* transition, judgment and moral reasoning.",
      "rights": "AI-assisted original editorial illustration directed by The Amateur Limited.",
      "truthBoundary": "Conceptual editorial illustration; not documentary evidence and not an OpenAI or Anthropic graphic."
    },
    {
      "file": "../assets/whack-a-mole.svg",
      "role": "mechanism",
      "caption": "Conceptual diagram of detect → patch → repeat, leading to the deeper generalisation question.",
      "rights": "Original work by The Amateur Limited.",
      "truthBoundary": "Conceptual sequence; not a chronology of any one laboratory.",
      "mobileVariant": "../assets/whack-a-mole-mobile.svg"
    },
    {
      "file": "../assets/a-to-a-star.svg",
      "role": "mechanism",
      "caption": "A → A*: a morally salient environmental change can change what successful task execution means.",
      "rights": "Original work by The Amateur Limited.",
      "truthBoundary": "Illustrates the article's proposed construct.",
      "mobileVariant": "../assets/a-to-a-star-mobile.svg"
    },
    {
      "file": "../assets/experiment-map.svg",
      "role": "contrast",
      "caption": "Four proposed training conditions mapped against five held-out evaluation families.",
      "rights": "Original work by The Amateur Limited.",
      "truthBoundary": "Proposed experiment; not an OpenAI or Anthropic experiment.",
      "mobileVariant": "../assets/experiment-map-mobile.svg"
    }
  ],
  "sources": [
    {
      "role": "primary",
      "label": "Anthropic — Teaching Claude Why",
      "url": "https://alignment.anthropic.com/2026/teaching-claude-why/"
    },
    {
      "role": "primary",
      "label": "Anthropic — Claude's Constitution",
      "url": "https://www.anthropic.com/constitution"
    },
    {
      "role": "primary",
      "label": "Anthropic — Agentic Misalignment in Summer 2026",
      "url": "https://alignment.anthropic.com/2026/agentic-misalignment-summer-2026/"
    },
    {
      "role": "primary",
      "label": "OpenAI — Deliberative Alignment",
      "url": "https://openai.com/index/deliberative-alignment/"
    },
    {
      "role": "primary",
      "label": "OpenAI — Detecting and reducing scheming in AI models",
      "url": "https://openai.com/index/detecting-and-reducing-scheming-in-ai-models/"
    },
    {
      "role": "primary",
      "label": "OpenAI Alignment — How far does alignment midtraining generalize?",
      "url": "https://alignment.openai.com/how-far-does-alignment-midtraining-generalize/"
    },
    {
      "role": "reference",
      "label": "Stanford Encyclopedia of Philosophy — Moral Realism",
      "url": "https://plato.stanford.edu/entries/moral-realism/"
    },
    {
      "role": "reference",
      "label": "Stanford Encyclopedia of Philosophy — Ancient Ethical Theory",
      "url": "https://plato.stanford.edu/entries/ethics-ancient/"
    },
    {
      "role": "reference",
      "label": "Stanford Encyclopedia of Philosophy — Plato's Ethics: An Overview",
      "url": "https://plato.stanford.edu/entries/plato-ethics/"
    },
    {
      "role": "reference",
      "label": "Stanford Encyclopedia of Philosophy — Aristotle's Ethics",
      "url": "https://plato.stanford.edu/entries/aristotle-ethics/"
    },
    {
      "role": "reference",
      "label": "Stanford Encyclopedia of Philosophy — Natural Law Ethics",
      "url": "https://plato.stanford.edu/entries/natural-law-ethics/"
    }
  ],
  "heroImage": "https://theamateur.co.uk/research/ai/2026-09-26/assets/alignment-hero.webp"
}
