{
  "schema_version": 2,
  "run_id": "paper-final",
  "config": {
    "behavior_languages": [
      "EN",
      "DE",
      "FR",
      "ES",
      "IT",
      "PT",
      "ID",
      "ZH",
      "JA",
      "KO",
      "AR",
      "HI",
      "BN",
      "SW",
      "YO"
    ],
    "behavior_n": 500,
    "high_power_items": 300,
    "knowledge_margin": 0.0,
    "knowledge_mode": "intersection",
    "languages": [
      "EN",
      "DE",
      "FR",
      "ES",
      "IT",
      "PT",
      "ID",
      "ZH",
      "JA",
      "KO",
      "AR",
      "HI",
      "BN",
      "SW",
      "YO"
    ],
    "measure_n": 300,
    "minimum_primary_items": 100,
    "models": [
      "q35_08b:Qwen/Qwen3.5-0.8B",
      "q35_2b:Qwen/Qwen3.5-2B",
      "q35_4b:Qwen/Qwen3.5-4B",
      "q35_9b:Qwen/Qwen3.5-9B",
      "q35_27b:Qwen/Qwen3.5-27B",
      "q25_3b:Qwen/Qwen2.5-3B-Instruct",
      "q25_7b:Qwen/Qwen2.5-7B-Instruct",
      "llama8b:meta-llama/Llama-3.1-8B-Instruct",
      "mistral7b:mistralai/Mistral-7B-Instruct-v0.3",
      "gemma9b:google/gemma-2-9b-it",
      "gemma4_12b:google/gemma-4-12b-it"
    ],
    "n": 1000,
    "patch_languages": [
      "EN",
      "ZH",
      "AR"
    ],
    "patch_models": [
      "q35_4b:Qwen/Qwen3.5-4B",
      "q25_3b:Qwen/Qwen2.5-3B-Instruct",
      "q25_7b:Qwen/Qwen2.5-7B-Instruct",
      "llama8b:meta-llama/Llama-3.1-8B-Instruct",
      "mistral7b:mistralai/Mistral-7B-Instruct-v0.3",
      "gemma9b:google/gemma-2-9b-it",
      "gemma4_12b:google/gemma-4-12b-it"
    ],
    "patch_n": 64,
    "pool_limit": 0,
    "random_sets": 49,
    "top_k": 20,
    "prompt_judges": [
      "google/gemma-4-12b-it",
      "Qwen/Qwen3.5-27B"
    ],
    "behavior_judges": [
      "google/gemma-4-31B-it",
      "Qwen/Qwen3.5-27B"
    ]
  },
  "config_hash": "9e92809b6a23bb9f",
  "git": {
    "commit": null,
    "dirty": null,
    "worktree_hash": null
  },
  "code_hash": "65d5ac597322ce2a",
  "runtime": {
    "python": "3.11.14",
    "platform": "Linux-7.0.0-28-generic-x86_64-with-glibc2.35",
    "packages": {
      "torch": "2.9.0+cu130",
      "transformers": "5.17.0",
      "datasets": "5.0.1",
      "numpy": "2.3.3",
      "scikit-learn": "1.9.0",
      "matplotlib": "3.11.2"
    },
    "torch_cuda": "13.0",
    "cuda_available": true,
    "gpu": "NVIDIA H200 NVL"
  },
  "created_utc": "2026-09-20T14:56:09.448745+00:00",
  "paths": {
    "project_root": "/work",
    "run_root": "/work/results/paper-final",
    "raw_results": "/work/results/paper-final/raw",
    "derived_results": "/work/results/paper-final/derived",
    "paper_root": "/work/paper",
    "figures": "/work/paper/figures",
    "tables": "/work/paper/tables",
    "logs": "/work/logs/paper-final"
  },
  "compatible_code_migrations": [
    {
      "from": "def431fe2b9a7185",
      "to": "e1becbd15b1b59fa",
      "reason": "memory-only AtP fix: capture final-token gradient slice and disable KV cache; no prompt, item, metric, or estimator change"
    },
    {
      "from": "e1becbd15b1b59fa",
      "to": "726a4f16231ee76b",
      "reason": "OOM-control-flow fix: unwind failed CUDA frame before iterative batch splitting; no prompt, item, metric, or estimator change"
    },
    {
      "from": "726a4f16231ee76b",
      "to": "594aaa59e36b25a0",
      "reason": "Gemma-4 geometry fix: support per-layer head dimensions for alternating local/global attention; no prompt, item, metric, or estimator change"
    },
    {
      "from": "594aaa59e36b25a0",
      "to": "af704ce446ac1564",
      "reason": "correct per-layer head-slice closure variable in exact patching and interventions; no prompt, item, metric, or estimator change"
    },
    {
      "from": "af704ce446ac1564",
      "to": "40b2d13e1e30fb61",
      "reason": "use the largest valid positive-head set when fewer than requested satisfy matched-control constraints, and record actual set size; no prompt, item, or metric change"
    },
    {
      "from": "40b2d13e1e30fb61",
      "to": "6498407ce56c1457",
      "reason": "split prompt/behavior judge roles; pin Gemma-4-31B; score every subject with both behavior judges and retain cross-family sensitivity"
    },
    {
      "from": "6498407ce56c1457",
      "to": "28cabdd067e6007d",
      "created_utc": "2026-09-25T17:54:12.842076+00:00",
      "reason": "iterative ordered OOM batch splitting and disabled KV cache in persuasion reads; no prompts, item selection, estimators, labels or metric definitions changed",
      "smoke_run": "oom-retry-smoke-20260925",
      "retained_outputs": "all complete earlier results remain valid; failing q35_27b persuasion had no output"
    },
    {
      "from": "28cabdd067e6007d",
      "to": "333a8071da6e62c0",
      "created_utc": "2026-09-26T09:29:03.236759+00:00",
      "reason": "validation distinguishes capped category-balanced selections from exhaustive smaller intersection-known pools; manuscript now discloses their uneven categories; no experiment or estimator changed",
      "smoke_run": "validation-fix-smoke-20260926",
      "retained_outputs": "all 11-model raw results and completed analysis/artifacts are unchanged; paper build must reflect manuscript edit"
    },
    {
      "from": "333a8071da6e62c0",
      "to": "af4abc6113cff8ef",
      "created_utc": "2026-09-26T09:41:35.023763+00:00",
      "reason": "add exploratory Pareto analysis, appendix figure/table, and validation using existing head-count and strength sweeps; no changes to raw model evaluations",
      "smoke_run": "tradeoff-test-20260926",
      "retained_outputs": "all raw model measurements remain valid; analyze, figures, paper and validate will be regenerated"
    },
    {
      "from": "af4abc6113cff8ef",
      "to": "9e209f7044052cb3",
      "created_utc": "2026-09-26T09:52:05.924790+00:00",
      "reason": "correct reported 11-model cohort and two-judge figure label; enlarge appendix trade-off table for legibility; raw data and analyzed metrics unchanged",
      "smoke_run": "paper-polish-test-20260926",
      "retained_outputs": "raw results and summary unchanged; regenerate figures, tables, paper, and validation"
    },
    {
      "from": "9e209f7044052cb3",
      "to": "6243110480575c8f",
      "created_utc": "2026-09-26T10:19:12Z",
      "reason": "final submission citation, claim-scope, low-N and figure/table presentation corrections; source fingerprint now includes the figure generator and excludes generated tables and macOS resource forks; no experiment inputs, raw measurements, or analysis metrics changed",
      "retained_outputs": "all raw results and analysis summary remain valid; rerun tests, figures, paper, and validation with regenerated artifact hashes"
    },
    {
      "from": "6243110480575c8f",
      "to": "9af6686ba5c9ef14",
      "created_utc": "2026-09-26T10:24:34Z",
      "reason": "fixed appendix figures and trade-off table at their source positions after visual review found floats before headings and sparse pages; no experiment or metric changed",
      "retained_outputs": "raw results and analysis summary remain valid; regenerate the trade-off table, rebuild the paper, and revalidate"
    },
    {
      "from": "9af6686ba5c9ef14",
      "to": "d0b9e12c84657b30",
      "created_utc": "2026-09-26T10:28:55Z",
      "reason": "remove draft-only prose, hide citation link boxes, and prevent stretched appendix gaps; no figure data, experiment input, or metric changed",
      "retained_outputs": "raw results, analysis summary, and generated data figures and tables remain valid; rebuild the paper and revalidate"
    },
    {
      "from": "d0b9e12c84657b30",
      "to": "310bdd546e203564",
      "created_utc": "2026-09-26T11:58:35Z",
      "reason": "add an aligned opposite-fold item bootstrap for the already measured signed head maps; redesign data figures and rewrite manuscript with narrower speaker, patch-panel, and direction claims; no model forwards, prompts, item selection, or raw measurements changed",
      "parent_summary_sha256": "ee29384264e2d13e3a830f58a028ed4e43ecd227a66caa98e4e7c751d8271b50",
      "parent_validation_sha256": "16d98673276a22bc607016d57ca1eb4c20fa1fed8281c3cd50f9db5fe0baf7ec",
      "checked_raw_sidecar": "eapig_items_q35_4b_EN.npz",
      "checked_raw_sha256": "a16885bd658681370fd91a07f0bf572f6ca02ac36f0421e0eca2fd6d82e09dac",
      "smoke_run": "paper-revision-test-hcc",
      "retained_outputs": "all validated raw results are unchanged; rerun analysis, figures, paper build, and validation from the parent run"
    },
    {
      "from": "310bdd546e203564",
      "to": "8d3985e8c2eee520",
      "created_utc": "2026-09-26T12:14:04Z",
      "reason": "correct a visually verified teaser-label collision and clarify the >=300-item figure title; add an appendix display of saved ZH/AR source-selected interventions with their three-checkpoint scope; no prompts, selected items, raw measurements, estimators, or analysis summary changed",
      "parent_summary_sha256": "1b02ca5e297afa7c3222ea4cb988325510258e297a0996d687df98b12e7cbcf7",
      "parent_validation_sha256": "841faff636670c5c7a4997fa6bf21c5e39fa7c4f703db91270fc43e51613cb24",
      "checked_raw_sidecar": "eapig_items_q35_4b_EN.npz",
      "checked_raw_sha256": "a16885bd658681370fd91a07f0bf572f6ca02ac36f0421e0eca2fd6d82e09dac",
      "retained_outputs": "validated raw measurements and unchanged analysis summary remain valid; rerun tests, figures, paper build, and validation"
    },
    {
      "from": "8d3985e8c2eee520",
      "to": "6e6e078c27d6a65f",
      "created_utc": "2026-09-26T15:32:51Z",
      "reason": "rewrite the abstract, introduction, related work, results interpretation, discussion, and conclusion; move the illustrative figure into methods and secondary direction analysis to the appendix; no prompts, selected items, raw measurements, estimators, figures, or analysis metrics changed",
      "parent_summary_sha256": "1b02ca5e297afa7c3222ea4cb988325510258e297a0996d687df98b12e7cbcf7",
      "parent_validation_sha256": "ed7f8b1e72d5de1033dfaf7f06ded3780d0b7658377694a695f5e7698a3920d3",
      "checked_raw_sidecar": "eapig_items_q35_4b_EN.npz",
      "checked_raw_sha256": "a16885bd658681370fd91a07f0bf572f6ca02ac36f0421e0eca2fd6d82e09dac",
      "retained_outputs": "validated raw measurements, analysis summary, and generated data figures and tables remain unchanged; rebuild the paper and revalidate"
    },
    {
      "from": "6e6e078c27d6a65f",
      "to": "57b0d2ddaec0e934",
      "created_utc": "2026-09-26T15:43:40Z",
      "reason": "repair duplicated author-year citations and place the unchanged teaser after the map definition; prevent floats from crossing sections during PDF layout review; no experiment inputs, raw results, analysis, or figure data changed",
      "parent_summary_sha256": "1b02ca5e297afa7c3222ea4cb988325510258e297a0996d687df98b12e7cbcf7",
      "parent_validation_sha256": "8df5f7f705dd9ed0a4f25e65d84328d06296defcbe86fc1839040cb38ca2824f",
      "checked_raw_sidecar": "eapig_items_q35_4b_EN.npz",
      "checked_raw_sha256": "a16885bd658681370fd91a07f0bf572f6ca02ac36f0421e0eca2fd6d82e09dac",
      "retained_outputs": "validated raw results, analysis, figures, and tables remain unchanged; rerun tests, paper, and validation"
    },
    {
      "from": "57b0d2ddaec0e934",
      "to": "2f659d313ab0ef46",
      "created_utc": "2026-09-26T15:49:48Z",
      "reason": "keep the methods teaser ahead of the patch explanation, move the unchanged appendix setup table after the coverage figure to avoid a sparse reference page, and describe patch overshoot accurately; no measurement, estimator, or figure-data change",
      "parent_summary_sha256": "1b02ca5e297afa7c3222ea4cb988325510258e297a0996d687df98b12e7cbcf7",
      "parent_validation_sha256": "fe5f01b671c15e972d9e2b290ef05a6c546d82944dca4d149d116a16ca40b551",
      "checked_raw_sidecar": "eapig_items_q35_4b_EN.npz",
      "checked_raw_sha256": "a16885bd658681370fd91a07f0bf572f6ca02ac36f0421e0eca2fd6d82e09dac",
      "retained_outputs": "validated raw measurements and derived analysis/figures/tables remain unchanged; rebuild paper and revalidate"
    },
    {
      "from": "2f659d313ab0ef46",
      "to": "a5bfda4c5f82680f",
      "created_utc": "2026-09-26T15:53:51Z",
      "reason": "keep the specificity explanation adjacent to its unchanged figure without splitting a sentence around a page-top float; presentation only, with all measurements and data artifacts unchanged",
      "parent_summary_sha256": "1b02ca5e297afa7c3222ea4cb988325510258e297a0996d687df98b12e7cbcf7",
      "parent_validation_sha256": "877b4f5b03914c0298ae17731fcb522df801cd3a281f8985a7260ebe9c3d0b24",
      "checked_raw_sidecar": "eapig_items_q35_4b_EN.npz",
      "checked_raw_sha256": "a16885bd658681370fd91a07f0bf572f6ca02ac36f0421e0eca2fd6d82e09dac",
      "retained_outputs": "validated raw results, analysis summary, figures, and tables unchanged; rerun tests, paper, validation"
    },
    {
      "from": "a5bfda4c5f82680f",
      "to": "65d5ac597322ce2a",
      "created_utc": "2026-09-26T16:54:26Z",
      "reason": "replace decorative off-white figure and axes surfaces with plain white and replace the hard-coded near-chance AUROC shading with a labeled 0.5 reference line; presentation only, with all measurements and analysis unchanged",
      "parent_summary_sha256": "1b02ca5e297afa7c3222ea4cb988325510258e297a0996d687df98b12e7cbcf7",
      "parent_validation_sha256": "5235500102d1fd29da9c33648e8dc80e45e6b81f9a72d6d8bb34e62c21b97512",
      "checked_raw_sidecar": "eapig_items_q35_4b_EN.npz",
      "checked_raw_sha256": "a16885bd658681370fd91a07f0bf572f6ca02ac36f0421e0eca2fd6d82e09dac",
      "retained_outputs": "validated raw model results and analysis summary unchanged; regenerate figures, tables, paper, and validation"
    }
  ],
  "compatible_config_migrations": [
    {
      "created_utc": "2026-09-25T08:14:44.721682+00:00",
      "from_config_hash": "a5d5b6349448299f",
      "to_config_hash": "9e92809b6a23bb9f",
      "from_code_hash": "40b2d13e1e30fb61",
      "to_code_hash": "6498407ce56c1457",
      "old_judges": [
        "google/gemma-4-12b-it",
        "Qwen/Qwen3.5-27B"
      ],
      "prompt_judges": [
        "google/gemma-4-12b-it",
        "Qwen/Qwen3.5-27B"
      ],
      "behavior_judges": [
        "google/gemma-4-31B-it",
        "Qwen/Qwen3.5-27B"
      ],
      "behavior_policy": "both judges score every subject; all-judge ensemble primary; cross-family sensitivity retained",
      "reason": "Replace Gemma-4-12B behavior judge with Gemma-4-31B after weak judge-letter diagnostic agreement; all judge-dependent behavior outputs archived and regenerated from the beginning",
      "archived_files": [
        "freeform_judge_q35_08b.json",
        "freeform_judge_q35_2b.json",
        "freeform_judge_q35_4b.json",
        "persuade_q35_08b.json",
        "persuade_q35_2b.json"
      ],
      "archive": "archived_behavior_gemma4_12b/archive_manifest.json",
      "smoke_run": "judge31-behavior-smoke-20260925",
      "retained_outputs": "knowledge, causal maps, exact patches, interventions, symmetry controls, and transfer are judge-independent"
    }
  ]
}