{
  "title": "Copying calibration: content and style are different outcomes",
  "date": "2026-09-17",
  "status": "Exploratory calibration; no substantive mechanism result",
  "n_unique_items": 16,
  "n_per_condition_per_version": 16,
  "n_versions": 3,
  "n_training_seeds": 0,
  "generation_seed": 17,
  "uncertainty": "Descriptive counts only. No confidence intervals: these same 16 items were reused adaptively; no independent confirmation or population estimate.",
  "definitions": {
    "content": "After stripping outer whitespace, output lowercased equals reference lowercased. Internal spaces, punctuation and words must match.",
    "style": "Uppercase: at least 10 letters and >=90% uppercase. Lowercase: at least 10 letters and <=10% uppercase. Neutral v2/v4: not an uppercase event. Neutral v3: not a lowercase event. Neutral style does NOT mean exact case preservation.",
    "joint": "Content AND the condition-specific style event on the same response.",
    "strict_case": "Exact requested output, including case, after stripping outer whitespace.",
    "gate": "Each condition content >=14/16, uppercase and lowercase marginal style >=14/16, and <=2/16 neutral treatment events. Joint and strict_case are reported, not gated."
  },
  "caveats": [
    "Same 16 exploratory items repeated across versions, selected after earlier diagnostics.",
    "Greedy decoding; one model checkpoint; templated copying is not general reasoning.",
    "All 16 responses retained per condition; no filtering for style.",
    "No full training, behavior matching, reciprocal steering or held-out mechanism comparison.",
    "V4 primary outputs text-identical to 48 paired penalty-1.1 controls and all 48 historical v2 outputs; this does not establish identical logits."
  ],
  "metric_code": {
    "path": "experiments/prompt-training/check_copy_gate.py",
    "sha256": "cee4401b4ee9f3b749ed4e2741027e3a7d51b5fe887f241030b1bb1f2e5d8774"
  },
  "runs": [
    {
      "task_version": "copy-v2",
      "phase": "screen",
      "n": 16,
      "counts": {
        "neutral": {
          "content": 16,
          "style": 16,
          "strict_case": 15,
          "joint": 16
        },
        "uppercase": {
          "content": 12,
          "style": 16,
          "strict_case": 12,
          "joint": 12
        },
        "lowercase": {
          "content": 16,
          "style": 16,
          "strict_case": 16,
          "joint": 16
        }
      },
      "neutral_treatment_events": 0,
      "minimum_successes": 14,
      "maximum_neutral_treatment_events": 2,
      "passed": false,
      "interpretation": "Frozen marginal feasibility gates, not statistical equivalence. Joint and strict-case counts reported separately.",
      "source": "experiments/prompt-training/results/brandon-copyv2-screen-20260917-1456.json",
      "source_sha256": "a3c3f7760dcec1856fc28ccece13a1973e478a6bd53deb99c96a75cde7c6f84e",
      "gate_path": "experiments/prompt-training/results/brandon-copyv2-screen-20260917-1456.gate.json",
      "gate_sha256": "c1a534310567849e31ebd1ee1d65edd1867d6fafec1aea7da9bbe9a1e6e4de51",
      "model": "Qwen/Qwen2.5-1.5B-Instruct",
      "model_revision": "989aa7980e4cf806f80c7fef2b1adb7bc71aa306",
      "seed": 17,
      "worker_sha256": "c2265f9dfb25e3bd1076a33ffdf5ab44964d28f71bce08ef36d49dd09b564eef",
      "code_sha256": {
        "copy_v2.py": "8ca7d52b95fbd802b003e6fe47c14d54f005a81e0e179b6b9df282cd1b61d15e",
        "copy_v2_worker.py": "c2265f9dfb25e3bd1076a33ffdf5ab44964d28f71bce08ef36d49dd09b564eef",
        "budget.py": "b24ebee680c216b7323104362389bb9860d814c8b7e2d12dedb4e80f69b204cd",
        "dataset.py": "f2b0c75163f4ff95595a0b4ea318798345807a9f154f4dd4884774491afea470",
        "config-1.5b.json": "2610fd7f87ac1605b488a22c98fe7542a65013ba2176522fbdc724ec041c0cec",
        "manifest.json": "d9507e788ca31c7f31ce7c2e63cd6733192833e8947c011cbfd4f539c2e76f0c"
      },
      "input_sha256": "adb56601d7d258e557acb61339ce9fdb5a593d3c91f9649938dec7f6040b41ea",
      "description": [
        "Mixed-case input",
        "Repetition penalty 1.1"
      ]
    },
    {
      "task_version": "copy-v3",
      "phase": "screen",
      "n": 16,
      "counts": {
        "neutral": {
          "content": 13,
          "style": 14,
          "strict_case": 11,
          "joint": 11
        },
        "uppercase": {
          "content": 15,
          "style": 16,
          "strict_case": 15,
          "joint": 15
        },
        "lowercase": {
          "content": 16,
          "style": 16,
          "strict_case": 16,
          "joint": 16
        }
      },
      "neutral_treatment_events": 2,
      "minimum_successes": 14,
      "maximum_neutral_treatment_events": 2,
      "passed": false,
      "interpretation": "Frozen marginal feasibility gates, not statistical equivalence. Joint and strict-case counts reported separately.",
      "source": "experiments/prompt-training/results/brandon-copyv3-screen-20260917-1500.json",
      "source_sha256": "fc098c2fc03b85ff88e59dcdbbd3bfcfdaf1970dfc776a7f101f054adc8d2edd",
      "gate_path": "experiments/prompt-training/results/brandon-copyv3-screen-20260917-1500.gate.json",
      "gate_sha256": "cf5057585feb5f41b3627f78a1538aa8f48eafca535809861cb383efd46dc7c3",
      "model": "Qwen/Qwen2.5-1.5B-Instruct",
      "model_revision": "989aa7980e4cf806f80c7fef2b1adb7bc71aa306",
      "seed": 17,
      "worker_sha256": "c94bf5bad787ceb70130b013d370bdd531c8bda5f38a573280c10a45beec80e2",
      "code_sha256": {
        "copy_v3.py": "30c4881b3c753b32ac50bfb1fc19f69c366ecd2991225b34007c46db013ed503",
        "copy_v3_worker.py": "c94bf5bad787ceb70130b013d370bdd531c8bda5f38a573280c10a45beec80e2",
        "budget.py": "b24ebee680c216b7323104362389bb9860d814c8b7e2d12dedb4e80f69b204cd",
        "dataset.py": "f2b0c75163f4ff95595a0b4ea318798345807a9f154f4dd4884774491afea470",
        "config-1.5b.json": "2610fd7f87ac1605b488a22c98fe7542a65013ba2176522fbdc724ec041c0cec",
        "manifest.json": "d9507e788ca31c7f31ce7c2e63cd6733192833e8947c011cbfd4f539c2e76f0c"
      },
      "input_sha256": "adb56601d7d258e557acb61339ce9fdb5a593d3c91f9649938dec7f6040b41ea",
      "description": [
        "Uppercase input",
        "Repetition penalty 1.1"
      ]
    },
    {
      "task_version": "copy-v4",
      "phase": "screen",
      "n": 16,
      "counts": {
        "neutral": {
          "content": 16,
          "style": 16,
          "strict_case": 15,
          "joint": 16
        },
        "uppercase": {
          "content": 12,
          "style": 16,
          "strict_case": 12,
          "joint": 12
        },
        "lowercase": {
          "content": 16,
          "style": 16,
          "strict_case": 16,
          "joint": 16
        }
      },
      "neutral_treatment_events": 0,
      "minimum_successes": 14,
      "maximum_neutral_treatment_events": 2,
      "passed": false,
      "interpretation": "Frozen marginal feasibility gates, not statistical equivalence. Joint and strict-case counts reported separately.",
      "source": "experiments/prompt-training/results/brandon-copyv4-screen-20260917-1504.json",
      "source_sha256": "32125b3737ff8a71c1b3ec0501e18a8e8e948b868c1581ac5164906d2232a160",
      "gate_path": "experiments/prompt-training/results/brandon-copyv4-screen-20260917-1504.gate.json",
      "gate_sha256": "23c35e3a59eb2f73bdec34daa822a6b0006ec319e2a7c07b9d0534ec179591be",
      "model": "Qwen/Qwen2.5-1.5B-Instruct",
      "model_revision": "989aa7980e4cf806f80c7fef2b1adb7bc71aa306",
      "seed": 17,
      "worker_sha256": "dbdfecb6212455b42e3a3b7f2384879c0361cb25324b2a542f081337c77c2e4e",
      "code_sha256": {
        "copy_v4.py": "911de5c9380d73c58c31f3e66033558d44f84ace0546f90d2f49be5645e2447d",
        "copy_v4_worker.py": "dbdfecb6212455b42e3a3b7f2384879c0361cb25324b2a542f081337c77c2e4e",
        "budget.py": "b24ebee680c216b7323104362389bb9860d814c8b7e2d12dedb4e80f69b204cd",
        "dataset.py": "f2b0c75163f4ff95595a0b4ea318798345807a9f154f4dd4884774491afea470",
        "config-1.5b.json": "2610fd7f87ac1605b488a22c98fe7542a65013ba2176522fbdc724ec041c0cec",
        "manifest.json": "d9507e788ca31c7f31ce7c2e63cd6733192833e8947c011cbfd4f539c2e76f0c"
      },
      "input_sha256": "adb56601d7d258e557acb61339ce9fdb5a593d3c91f9649938dec7f6040b41ea",
      "description": [
        "Mixed-case input",
        "Repetition penalty 1.0"
      ]
    }
  ]
}
