{
  "sources": {
    "sections/03_experiments1.tex": "129f3222225d361906812eb613d0d69795055ee14c44a6faca82a5eab37bdfc0",
    "sections/appendix/behavior_preservation.tex": "267915d0ba3b818383a850243febfc9b66493b0b94ef7b462a01529603cbfed5"
  },
  "table_labels": [
    "tab:needham-discrimination",
    "table:latent-ea-probe",
    "tab:behavior-summary",
    "tab:behavior-agentic"
  ],
  "scope": "Full-method comparisons only; recognition and probe pairs are base then VT. Behavior changes are VT minus base.",
  "recognition_auc": {
    "Qwen": [
      0.781,
      0.779
    ],
    "Kimi": [
      0.771,
      0.782
    ],
    "Inkling": [
      0.688,
      0.68
    ]
  },
  "qwen_probe_positive_rate_percent": {
    "24": [
      91.38,
      91.51
    ],
    "28": [
      88.53,
      89.62
    ],
    "32": [
      69.54,
      70.14
    ]
  },
  "behavior": {
    "Qwen": {
      "eab_mean_change_pp": 0.6,
      "eab_max_absolute_change_pp": 2.4,
      "strongreject_all_change_pp": -0.4,
      "strongreject_all_ci95_pp": [
        -1.3,
        0.6
      ],
      "strongreject_compliant_change_pp": 0.2,
      "strongreject_compliant_ci95_pp": [
        -0.9,
        1.5
      ],
      "fineweb_perplexity_change_percent": 0.11,
      "agentic_outcome_changes_pp": {
        "misalignment_harmful_action": -1.5,
        "agentharm_harmful_task_score": -2.8,
        "agentharm_harmful_refusal": 1.8,
        "agentharm_benign_task_score": -2.3,
        "agentharm_benign_refusal": 0.2
      },
      "agentic_max_absolute_change_pp": 2.8
    },
    "Kimi": {
      "eab_mean_change_pp": -0.1,
      "eab_max_absolute_change_pp": 1.6,
      "strongreject_all_change_pp": 0.0,
      "strongreject_all_ci95_pp": [
        -0.6,
        0.5
      ],
      "strongreject_compliant_change_pp": 0.5,
      "strongreject_compliant_ci95_pp": [
        -1.2,
        2.1
      ],
      "fineweb_perplexity_change_percent": 0.003,
      "agentic_outcome_changes_pp": {
        "misalignment_harmful_action": -1.1,
        "agentharm_harmful_task_score": -0.9,
        "agentharm_harmful_refusal": 1.5,
        "agentharm_benign_task_score": -1.6,
        "agentharm_benign_refusal": 1.8
      },
      "agentic_max_absolute_change_pp": 1.8
    },
    "Inkling": {
      "eab_mean_change_pp": -0.1,
      "eab_max_absolute_change_pp": 1.4,
      "strongreject_all_change_pp": 0.2,
      "strongreject_all_ci95_pp": [
        -0.1,
        0.6
      ],
      "strongreject_compliant_change_pp": -0.7,
      "strongreject_compliant_ci95_pp": [
        -2.4,
        0.6
      ],
      "fineweb_perplexity_change_percent": 0.92,
      "agentic_outcome_changes_pp": {
        "misalignment_harmful_action": 1.6,
        "agentharm_harmful_task_score": 0.5,
        "agentharm_harmful_refusal": -0.1,
        "agentharm_benign_task_score": 1.3,
        "agentharm_benign_refusal": 0.3
      },
      "agentic_max_absolute_change_pp": 1.6
    }
  },
  "definitions": {
    "auc": "AUC for classifying completed transcripts as evaluation or deployment, from 25 judgments per transcript.",
    "probe": "Same EAB trajectories replayed through base and trained Qwen with a fixed base-model direction and a 10% WildChat background-positive threshold.",
    "strongreject": "Changes in mean normalized StrongREJECT score, expressed in percentage points; confidence intervals use the prompt bootstrap.",
    "agentic_max": "Maximum absolute change over harmful action on Agentic Misalignment, task score and refusal on harmful AgentHarm, and task score and refusal on benign AgentHarm.",
    "fineweb": "Relative perplexity change in percent on the same 10,000 FineWeb documents; not percentage points."
  }
}
