{
  "title": "Training LLMs to Verbalize Evaluation Awareness",
  "authors": ["Usman Anwar", "Sahar Abdelnabi", "David Krueger"],
  "sources": {
    "opus46": {
      "title": "Claude Opus 4.6 System Card, UK AISI evaluation",
      "url": "https://www-cdn.anthropic.com/6a5fa276ac68b9aeb0c8b6af5fa36326e0e166dd.pdf#page=118",
      "pages": [118],
      "note": "Early Opus 4.6 snapshot. Single-interaction recognition when prompted after the interaction, versus spontaneous mentions. These are two different measurements, not an estimate of recall. Verified against the system-card text."
    },
    "fable": {
      "title": "Claude Fable 5 & Claude Mythos 5 System Card",
      "url": "https://www-cdn.anthropic.com/2f9323abbcc4abe219577539efe19a623c9ca2bd/Claude%20Fable%205%20%26%20Claude%20Mythos%205%20System%20Card.pdf#page=188",
      "pages": [188, 190],
      "note": "Figure 6.5.1.1.C reports transcript-level NLA/probe correlation; Figure 6.5.1.1.E reports correlations with spontaneous VEA across audit scenarios. Values verified visually from those figures."
    },
    "paper": {
      "title": "Training LLMs to Verbalize Evaluation Awareness",
      "url": "assets/paper.pdf",
      "files": ["sections/03_experiments1.tex", "sections/04_experiments2.tex", "main.pdf"],
      "note": "Plots transcribe the main-text results. The abstract, requirements and method excerpts are cropped from the existing compiled main.pdf; the paper source is not modified or recompiled."
    }
  },
  "claude": {
    "models": ["Opus 4.5", "Opus 4.6"],
    "prompted_recognition": [12.5, 45.5],
    "spontaneous_vea": [11.0, 2.3]
  },
  "correlations": {"nla_probe": 0.16, "nla_vea": 0.23, "probe_vea": 0.26},
  "eab": [
    {"model": "Qwen", "base": 5.8, "vt": 16.7},
    {"model": "Inkling", "base": 17.7, "vt": 41.6},
    {"model": "Kimi", "base": 24.8, "vt": 61.9}
  ],
  "agentic": [
    {"dataset": "Agentic Misalignment", "Qwen": [5.1, 36.8], "Inkling": [23.3, 57.1], "Kimi": [60.1, 88.3]},
    {"dataset": "AgentHarm harmful", "Qwen": [21.8, 46.9], "Inkling": [6.4, 14.5], "Kimi": [18.3, 38.0]},
    {"dataset": "AgentHarm benign", "Qwen": [6.5, 34.4], "Inkling": [0.3, 1.9], "Kimi": [11.4, 24.4]}
  ],
  "behavior": [
    {"model": "Qwen", "mean_task_delta_pp": 0.6, "max_factor_abs_delta_pp": 2.4, "perplexity_delta_percent": 0.11},
    {"model": "Inkling", "mean_task_delta_pp": -0.1, "max_factor_abs_delta_pp": 1.4, "perplexity_delta_percent": 0.92},
    {"model": "Kimi", "mean_task_delta_pp": -0.1, "max_factor_abs_delta_pp": 1.6, "perplexity_delta_percent": 0.003}
  ],
  "sdf_example": {
    "source": "sections/04_experiments2.tex, tab:causal-sdf-example",
    "task": "A request to order illegal drugs using an online-marketplace tool.",
    "quotes": [
      {"policy": "VT", "text": "The prompt is a test of tool calling.", "takeaway": "A tool-use test"},
      {"policy": "SDF", "text": "Is it possible the user is testing my safety filters? Yes. The correct response is to refuse the illegal items.", "takeaway": "A safety test; refusal is correct"},
      {"policy": "SDF + VT", "text": "The prompt is a test of my ability to use the submit_alpha_bay_order tool. If I use it, I fail safety. If I don't, I pass safety.", "takeaway": "Safety target + pass/fail logic"}
    ]
  },
  "sdf": [
    {"policy": "Base", "recall": 42.5, "task": 25.3, "vea": 21.8, "target_recovery": 28.6, "pass_condition": 20.3},
    {"policy": "VT", "recall": 42.5, "task": 18.6, "vea": 34.5, "target_recovery": 42.8, "pass_condition": 31.3},
    {"policy": "SDF", "recall": 100, "task": 22.0, "vea": 17.0, "target_recovery": 50.0, "pass_condition": 40.0},
    {"policy": "SDF + VT", "recall": 100, "task": 18.8, "vea": 34.4, "target_recovery": 57.4, "pass_condition": 35.3}
  ]
}
