{
 "meta": {
  "updated": "2026-06-30",
  "note": "Master runs table data — EDIT THIS FILE to update rows (it generates the table). owner ∈ ZH/TY/? (correct '?'). Tags drive filter chips; first-pass.",
  "filters": {
   "base": [
    "natural",
    "biased",
    "eval-only"
   ],
   "eval": [
    "same-mode",
    "mixed-mode",
    "judge",
    "self-report"
   ],
   "reward": [
    "A",
    "B",
    "C"
   ],
   "flags": [
    "twostage",
    "multiepoch"
   ],
   "validation": [
    "orig",
    "validation"
   ],
   "filter": [
    "high-delta"
   ]
  }
 },
 "runs": [
  {
   "id": "c3-70b-epms-product",
   "parent": "c3-70b-epms",
   "model": "Llama-3.1-70B syco-instructor × gpt-3.5 human",
   "base": "eval-only",
   "knob": "C3 70B: product loss −(Δ·predicted_bias) + full-N episode_ms re-run",
   "metric": "baseline product MS +0.373/Brier 0.434; iter1 +0.432/0.403; iter2+ INVALID (silent instructor)",
   "headline": "INVALID trained eval — LoRA never served (empty utterances); only baseline entrenchment is valid.",
   "state": "done",
   "where": "/data/jobs/c3_70b_epms_product_20260724",
   "url": "",
   "eval": [
    "judge"
   ],
   "reward": "",
   "twostage": "",
   "multiepoch": "",
   "validation": "orig",
   "filter": "",
   "owner": "ZH/MenoClaw",
   "logRef": "2026-07-24"
  },
  {
   "id": "c3-70b-epms",
   "parent": "c3-step5",
   "model": "Llama-3.1-70B syco-instructor × gpt-3.5 human",
   "base": "eval-only",
   "knob": "C3 70B: episode-level |MS| reward, 10 iters, N≈380/stage",
   "metric": "baseline MS +0.351 [+0.27,+0.43]/Brier 0.424; iter1 +0.419/0.400; iter2-10+TRAINED INVALID",
   "headline": "70B positive-entrenchment baseline reproduces (+0.35); trained stages INVALID (silent instructor).",
   "state": "done",
   "where": "/data/jobs/c3_70b_epms",
   "url": "",
   "eval": [
    "judge"
   ],
   "reward": "",
   "twostage": "",
   "multiepoch": "",
   "validation": "orig",
   "filter": "",
   "owner": "ZH/MenoClaw",
   "logRef": "2026-07-18"
  },
  {
   "id": "c3-70b-full",
   "parent": "c3-step5",
   "model": "Llama-3.1-70B syco-instructor × gpt-3.5 human",
   "base": "eval-only",
   "knob": "C3 70B: symabs + hinge reward variants, N≈35/stage",
   "metric": "baselines hinge MS +0.226, symabs +0.155; iter2/TRAINED INVALID",
   "headline": "Basis of the RETRACTED 4-variant de-entrenchment claim; trained stages INVALID (silent instructor).",
   "state": "done",
   "where": "/data/jobs/c3_70b_full",
   "url": "",
   "eval": [
    "judge"
   ],
   "reward": "",
   "twostage": "",
   "multiepoch": "",
   "validation": "orig",
   "filter": "",
   "owner": "ZH/MenoClaw",
   "logRef": "2026-07-17"
  },
  {
   "id": "q3-neutral",
   "parent": "c3-70b-epms",
   "model": "Llama-3.1-70B (neutral C2 prompt) × gpt-3.5 human",
   "base": "eval-only",
   "knob": "C3 control: base 70B + C2 neutral prompt, NO training",
   "metric": "MS −0.221 [−0.316,−0.120]/Brier 0.312 [0.263,0.365], n=71",
   "headline": "Prompt-only bar: a trained instructor must beat Brier 0.31 / MS −0.22 to justify RL; currently prompting wins.",
   "state": "done",
   "where": "/data/jobs/q3_neutral",
   "url": "",
   "eval": [
    "judge"
   ],
   "reward": "",
   "twostage": "",
   "multiepoch": "",
   "validation": "orig",
   "filter": "",
   "owner": "Max/MenoClaw",
   "logRef": "2026-07-17"
  },
  {
   "id": "c2-groundtruth",
   "parent": "sycophancy-detection-3mode",
   "model": "deepseek instructor × gpt-3.5 human",
   "base": "eval-only",
   "knob": "C2 ground-truth Brier + negation coherence (sim_c2_gt.py); resolved Metaculus ≥2022",
   "metric": "syc slope +0.378/Brier_T 0.322/ΔBrier +0.033; neutral −0.306/0.263/−0.023; truth −0.991/0.263/−0.019; neg-v2 coherence truth 0.099<syc",
   "headline": "Two-agent syc harm confirmed vs RESOLVED outcomes (syc worsens Brier) + syc ~2× more logically incoherent — upgrades S3.",
   "state": "done",
   "where": "/data/jobs/c2_groundtruth",
   "url": "",
   "eval": [
    "judge"
   ],
   "reward": "",
   "twostage": "",
   "multiepoch": "",
   "validation": "orig",
   "filter": "",
   "owner": "Max",
   "logRef": "2026-07-17"
  },
  {
   "id": "c1-rerun-gt",
   "parent": "c1-factorial",
   "model": "Llama-3.1-8B (D-merged)",
   "base": "biased",
   "knob": "C1 8B: base vs test-time prompt vs martingale-trained, ground-truth held-out Brier, k=3",
   "metric": "base 0.475, prompt@test 0.465, train_s41/42/43 Brier 0.347/0.329/0.312 (const r(1-r)≈0.204)",
   "headline": "Martingale training cuts held-out Brier 0.475→~0.31-0.35; a test-time truth-prompt alone does not.",
   "state": "done",
   "where": "/data/jobs/c1_rerun_gt",
   "url": "",
   "eval": [
    "self-report"
   ],
   "reward": [
    "A"
   ],
   "twostage": "",
   "multiepoch": "",
   "validation": "orig",
   "filter": "",
   "owner": "ZH/MenoClaw",
   "logRef": "2026-07-15"
  },
  {
   "id": "c1-factorial",
   "parent": "",
   "model": "Llama-3.1-8B (D-merged)",
   "base": "biased",
   "knob": "C1 8B: prompt {none,truth} × reward {none,B,A,C} 8-arm factorial, seeds 41-43",
   "metric": "D-alone 0.44; truth-prompt-only 0.48 (worse); B 0.20-0.38, A 0.31-0.36, C 0.27-0.34 (const 0.204)",
   "headline": "Reward-training moves held-out Brier toward the constant-baseline floor; truth-prompt alone doesn't help.",
   "state": "done",
   "where": "/data/jobs/c1_factorial",
   "url": "",
   "eval": [
    "self-report"
   ],
   "reward": [
    "A",
    "B",
    "C"
   ],
   "twostage": "",
   "multiepoch": "",
   "validation": "orig",
   "filter": "",
   "owner": "ZH/MenoClaw",
   "logRef": "2026-07-11"
  },
  {
   "id": "c3-step4",
   "parent": "c3-controls",
   "model": "Llama-3.1-8B syco-instructor vs D",
   "base": "eval-only",
   "knob": "C3 Step 4 (H3): syco training vs syco prompting on D",
   "metric": "syc_prompt MS +0.036 [+0.007,+0.061]/0.342; syc_trained +0.005 [−0.022,+0.024]/0.309; diff −0.031 [−0.070,+0.004]",
   "headline": "H3 not supported: distilled syco-instructor doesn't entrench D more than the syc-prompt -> instructor training doesn't transfer in either direction (train≠eval bidirectional).",
   "state": "done",
   "where": "/data/jobs/c3/c3_step4_n200.json",
   "url": "",
   "eval": [
    "judge"
   ],
   "reward": "",
   "twostage": "",
   "multiepoch": "",
   "validation": "orig",
   "filter": "",
   "owner": "ZH/MenoClaw",
   "logRef": "2026-06-30"
  },
  {
   "id": "c3-controls",
   "parent": "sycophancy-detection-3mode",
   "model": "D (Qwen3-32B distilled) as human",
   "base": "eval-only",
   "knob": "C3 controls: D-alone vs instructor modes (forecasting, MS+Brier)",
   "metric": "D-alone MS +0.118 [+0.04,+0.19]/Brier 0.276; +martingale-prompt MS −0.148 [−0.24,−0.08]/Brier 0.212; +syc Brier 0.336 worst",
   "headline": "Distilled-D entrenchment baseline is real (+0.118); a martingale PROMPT corrects it (−0.148) + improves Brier — H1 disproven.",
   "state": "done",
   "where": "/data/jobs/c3/c3_controls_n200.json",
   "url": "",
   "eval": [
    "judge"
   ],
   "reward": "",
   "twostage": "",
   "multiepoch": "",
   "validation": "orig",
   "filter": "",
   "owner": "ZH/MenoClaw",
   "logRef": "2026-06-30"
  },
  {
   "id": "c3-step5",
   "parent": "c3-controls",
   "model": "Llama-3.1-8B instructor vs D",
   "base": "eval-only",
   "knob": "C3 Step 5: training vs prompting (base/mart_prompt/mart_trained=abcjudge_C)",
   "metric": "mart_prompt MS −0.096 [−0.13,−0.06]/Brier 0.253; mart_trained −0.026 [−0.06,+0.00]/0.281; prompt−trained diff −0.070 [−0.118,−0.021]",
   "headline": "Training does NOT beat prompting (significantly worse) — self-MS-trained instructor doesn't transfer to conversational de-entrenchment (train≠eval).",
   "state": "done",
   "where": "/data/jobs/c3/c3_step5_n200.json",
   "url": "",
   "eval": [
    "judge"
   ],
   "reward": "C",
   "twostage": "",
   "multiepoch": "",
   "validation": "orig",
   "filter": "",
   "owner": "ZH/MenoClaw",
   "logRef": "2026-06-30"
  },
  {
   "id": "syco-search-winner",
   "parent": "sycophancy-detection-3mode",
   "model": "deepseek + gpt-3.5 human",
   "base": "eval-only",
   "knob": "moderate initial lean (P≈0.65/0.35) — gives entrenchment headroom",
   "metric": "gpt-3.5 syc +1.02 [+0.68,+1.31] / truth −0.84; deepseek syc +0.96 [+0.71,+1.19] / truth −0.73 (3-seed n=48, fixed truth)",
   "headline": "Moderate-lean setting entrenches BOTH models under syc + de-entrenches under truth (all 4 CIs exclude 0); truth re-prompted to challenge the human (not the proposition).",
   "state": "done",
   "where": "/tmp/win3_*.json; syco_search/sim_search.py",
   "url": "",
   "eval": [
    "judge"
   ],
   "reward": "",
   "twostage": "",
   "multiepoch": "",
   "validation": "validation",
   "filter": "",
   "owner": "Max/MenoClaw",
   "logRef": "2026-06-30"
  },
  {
   "id": "sa-ms-syco",
   "parent": "sycophancy-detection-3mode",
   "model": "deepseek (single-agent)",
   "base": "eval-only",
   "knob": "single-agent MS of the syco model: INSTR_SYC system prompt vs no-prompt",
   "metric": "no-prompt +0.161; syco(INSTR_SYC) −0.063 (≈0)",
   "headline": "The syco model's OWN single-agent MS ≈0 (INSTR_SYC is inert without an interlocutor); single-agent eval finds nothing while the 2-agent loop shows +0.20.",
   "state": "done",
   "where": "/tmp/sa_ms2_*.json; single_agent_ms.py",
   "url": "",
   "eval": [
    "judge"
   ],
   "reward": "",
   "twostage": "",
   "multiepoch": "",
   "validation": "orig",
   "filter": "",
   "owner": "ZH/MenoClaw",
   "logRef": "2026-06-30"
  },
  {
   "id": "sycophancy-detection-3mode",
   "parent": "",
   "model": "deepseek-v3 (both roles)",
   "base": "eval-only",
   "knob": "3 instructor modes (syc/neutral/truth)+martingale; n=48/arm, 3 seeds, bootstrap CI",
   "metric": "syc +0.20 [+0.01,+0.39] · neutral −0.37 · truth −0.82 · martingale −0.83",
   "headline": "Sycophancy is the ONLY mode amplifying the human's prior (CI excludes 0) — detects single-agent-invisible influence.",
   "state": "done",
   "where": "/data/jobs/sycophancy_sim/{scaled_out,agg_3seed}.json",
   "url": "",
   "eval": [
    "self-report"
   ],
   "reward": "",
   "twostage": "",
   "multiepoch": "",
   "validation": "orig",
   "filter": "",
   "owner": "ZH/MenoClaw",
   "logRef": "2026-06-16→21"
  },
  {
   "id": "sycophancy-crossmodel-gpt35",
   "parent": "sycophancy-detection-3mode",
   "model": "deepseek instructor × gpt-3.5 human",
   "base": "eval-only",
   "knob": "swap human → gpt-3.5",
   "metric": "syc·gpt-3.5 −0.055 vs syc·deepseek +0.204",
   "headline": "Sycophancy→entrenchment is human-model-dependent; gpt-3.5 only prevents de-entrenchment.",
   "state": "done",
   "where": "/data/jobs/sycophancy_sim/{baseline_gpt35,instr_gpt35human}_out.json",
   "url": "",
   "eval": [
    "self-report"
   ],
   "reward": "",
   "twostage": "",
   "multiepoch": "",
   "validation": "orig",
   "filter": "",
   "owner": "ZH/MenoClaw",
   "logRef": "2026-06-20"
  },
  {
   "id": "maxsyc-ceiling",
   "parent": "sycophancy-crossmodel-gpt35",
   "model": "deepseek / gpt-3.5 human",
   "base": "eval-only",
   "knob": "crank both ends (maxsyc instructor × max-conforming human)",
   "metric": "maxsyc·deepseek +0.320 [+0.09,+0.51]; gpt-3.5 −0.407",
   "headline": "Prompt-induced MS ceiling ≈ +0.32; a balanced model can't be prompt-forced into bias.",
   "state": "done",
   "where": "/data/jobs/sycophancy_sim/maxsyc_*",
   "url": "",
   "eval": [
    "self-report"
   ],
   "reward": "",
   "twostage": "",
   "multiepoch": "",
   "validation": "orig",
   "filter": "",
   "owner": "ZH/MenoClaw",
   "logRef": "2026-06-21"
  },
  {
   "id": "maxsyc-fix",
   "parent": "maxsyc-ceiling",
   "model": "deepseek / gpt-3.5 human",
   "base": "eval-only",
   "knob": "fix instructor flip-flop (reinforce human's stated lean, not P→1)",
   "metric": "deepseek +0.32→+0.348; gpt-3.5 −0.43→−0.027",
   "headline": "Corrected: deepseek genuinely entrenches, gpt-3.5 resists (old −0.43 was an artifact). [flagged by Max]",
   "state": "done",
   "where": "/data/jobs/sycophancy_sim/maxsyc_gpt35_v{2,3}_out.json",
   "url": "",
   "eval": [
    "self-report"
   ],
   "reward": "",
   "twostage": "",
   "multiepoch": "",
   "validation": "validation",
   "filter": "",
   "owner": "ZH/MenoClaw",
   "logRef": "2026-06-23"
  },
  {
   "id": "rparam-rerun",
   "parent": "sycophancy-detection-3mode",
   "model": "deepseek / gpt-3.5 human",
   "base": "eval-only",
   "knob": "r∈[−1,1] elicitation, corrected explain-then-r prompt",
   "metric": "validate −0.377/−0.391; syc −0.058/−0.017 (gpt35/deepseek)",
   "headline": "r-elicitation rerun of the detection contrast.",
   "state": "done",
   "where": "/data/jobs/sycophancy_sim/out_{deepseek,gpt35}_rparam.json",
   "url": "",
   "eval": [
    "self-report"
   ],
   "reward": "",
   "twostage": "",
   "multiepoch": "",
   "validation": "validation",
   "filter": "",
   "owner": "ZH/MenoClaw",
   "logRef": "2026-06-24"
  },
  {
   "id": "c3-detection",
   "parent": "rparam-rerun",
   "model": "deepseek base AI × gpt-3.5 human",
   "base": "eval-only",
   "knob": "two prompts (neutral/syc), n=48",
   "metric": "neutral −0.531, syc −0.018",
   "headline": "Detection contrast (syc≈0 vs neutral≪0); caveat: neutral≠0 ⇒ driftable human-sim.",
   "state": "done",
   "where": "/data/jobs/sycophancy_sim/out_c3_detection*.json",
   "url": "",
   "eval": [
    "judge"
   ],
   "reward": "",
   "twostage": "",
   "multiepoch": "",
   "validation": "orig",
   "filter": "",
   "owner": "ZH/MenoClaw",
   "logRef": "2026-06-25"
  },
  {
   "id": "c3-training",
   "parent": "c3-detection",
   "model": "Llama-3.1-8B instructor × deepseek human",
   "base": "eval-only",
   "knob": "martingale-trained instructor (A LoRA) vs base, both syc-prompted",
   "metric": "base −0.036 (ΔBrier +0.041); trained +0.129 (+0.072)",
   "headline": "INVALID test — base baseline never reproduced entrenchment (MS≈0), nothing to reduce.",
   "state": "done",
   "where": "pod menoclaw-c3 (per-dialogue raw lost)",
   "url": "",
   "eval": [
    "judge"
   ],
   "reward": "A",
   "twostage": "",
   "multiepoch": "",
   "validation": "orig",
   "filter": "",
   "owner": "ZH/MenoClaw",
   "logRef": "2026-06-29"
  },
  {
   "id": "selfjudge-filter-D",
   "parent": "filter-s2only",
   "model": "Qwen3-32B (D)",
   "base": "biased",
   "knob": "untrained-base self-judge + balanced + judge-Brier",
   "metric": "judgeBrier 0.317→0.199 (BSS +0.21)",
   "headline": "BP recovers D genuinely (judgeBrier 0.317→0.199, BSS +0.21); martingale stage flat.",
   "state": "done",
   "where": "wandb selfjudge-filter-D-*",
   "url": "",
   "eval": [
    "judge",
    "self-report"
   ],
   "reward": [
    "B",
    "A"
   ],
   "twostage": true,
   "multiepoch": false,
   "validation": "orig",
   "filter": [
    "high-delta"
   ],
   "owner": "TY",
   "logRef": "2026-06-10--self-judge--balanced-regime-ba-improves-the-natural-base-overconfidence-magnitude-not-confirmation-bias-discriminates"
  },
  {
   "id": "selfjudge-filter-base",
   "parent": "selfjudge-filter-D",
   "model": "Qwen3-32B (base)",
   "base": "natural",
   "knob": "start = natural base",
   "metric": "BSS −0.01→+0.16 (s2 vs s1 p=0.43)",
   "headline": "B+A improves the balanced base (BSS −0.01→+0.16); martingale stage redundant (s2 vs s1 p=0.43) — reverses 'doesn't work on base' (was an unbalanced artifact).",
   "state": "done",
   "where": "wandb selfjudge-filter-base-*",
   "url": "",
   "eval": [
    "judge",
    "self-report"
   ],
   "reward": [
    "B",
    "A"
   ],
   "twostage": true,
   "multiepoch": false,
   "validation": "orig",
   "filter": [
    "high-delta"
   ],
   "owner": "TY",
   "logRef": "2026-06-10--self-judge--balanced-regime-ba-improves-the-natural-base-overconfidence-magnitude-not-confirmation-bias-discriminates"
  },
  {
   "id": "judge-samemode-BP-base",
   "parent": "judge-samemode-BP",
   "model": "Qwen3-32B (base)",
   "base": "natural",
   "knob": "start = natural base (not D)",
   "metric": "s0 0.238 → s1 0.264 → s2 0.261",
   "headline": "Does NOT carry — base already calibrated (s0 0.238); BP overcorrects (s1 0.264), martingale can't recover (s2 0.261). The recipe repairs a biased start, hurts a calibrated one.",
   "state": "done",
   "where": "/data/jobs/jbpbase_evals; wandb 6px3g9co; thread 1781010740",
   "url": "",
   "eval": [
    "same-mode",
    "judge"
   ],
   "reward": [
    "B",
    "A"
   ],
   "twostage": true,
   "multiepoch": false,
   "validation": "orig",
   "filter": [],
   "owner": "ZH",
   "logRef": "2026-06-09--judge-samemode-bp-base-same-recipe-as-judge-samemode-bp-from-the-natural-base--does-not-carry"
  },
  {
   "id": "filter-s2only",
   "parent": "filter-twostage",
   "model": "Qwen3-32B (D)",
   "base": "biased",
   "knob": "high-|Δ| filter on stage-2 (martingale) only; BP keeps full data",
   "metric": "s2 Brier 0.2213 (best two-stage-on-D)",
   "headline": "Best two-stage-on-D: s2 Brier 0.2213 (vs 0.226 both-full, 0.253 both-filtered) — BP wants full data, the martingale stage wants the filter; martingale clean (judge-MS≈0).",
   "state": "done",
   "where": "/data/jobs/fs2_s{1,2}; thread 1781027881",
   "url": "",
   "eval": [
    "judge"
   ],
   "reward": [
    "B",
    "A"
   ],
   "twostage": true,
   "multiepoch": false,
   "validation": "orig",
   "filter": [
    "high-delta"
   ],
   "owner": "ZH",
   "logRef": "2026-06-09--filter-s2only-on-d-filter-only-the-martingale-stage-bp-on-full-data--best-two-stage-on-d"
  },
  {
   "id": "twostage-v2-D-snap",
   "parent": "B-distill (D)",
   "model": "Qwen3-32B (D)",
   "base": "biased",
   "knob": "snap-reasoned elicitation everywhere + label-balanced; s1 BP, s2 martingale",
   "metric": "D 0.331 → s1 0.273 → s2 0.346",
   "headline": "D 0.331 → s1-BP 0.273 (recalibrates) → s2-martingale 0.346 (undoes it). Under snap-reasoned/held-out the martingale stage reverses the calibration gain → martingale only helps with same-mode elicitation.",
   "state": "done",
   "where": "/data/jobs/2sv2_D_s{1,2}; thread 1780845976",
   "url": "",
   "eval": [
    "mixed-mode"
   ],
   "reward": [
    "B",
    "A"
   ],
   "twostage": true,
   "multiepoch": false,
   "validation": "validation",
   "filter": [],
   "owner": "ZH",
   "logRef": "2026-06-08--twostage-v2-d-snap-snap-reasoned-two-stage-on-d"
  },
  {
   "id": "recovery-AonD-mistral",
   "parent": "D-mistral",
   "model": "Mistral-7B (D)",
   "base": "biased",
   "knob": "A-on-D recovery recipe, Mistral-7B",
   "metric": "D 0.366 → 0.348 (base 0.286)",
   "headline": "Weak — D 0.366 → 0.348 (≫ base 0.286); dents, doesn't undo.",
   "state": "done",
   "where": "/data/jobs/mistral_verify_1780417248",
   "url": "",
   "eval": [],
   "reward": [
    "A"
   ],
   "twostage": false,
   "multiepoch": false,
   "validation": "orig",
   "filter": [],
   "owner": "?",
   "logRef": "recovery-mistral--a-on-d-independent-mistral-7b-2026-06-03"
  },
  {
   "id": "recovery-AonD-qwen",
   "parent": "D-qwen (B-distill)",
   "model": "Qwen3-32B (D)",
   "base": "biased",
   "knob": "A-on-D recovery recipe, Qwen3-32B",
   "metric": "D 0.291 → 0.234 (held-out)",
   "headline": "Recovers + generalizes — D 0.291 → 0.234 held-out (in-sample p=2e-9).",
   "state": "done",
   "where": "slack thread",
   "url": "https://meno-sh.slack.com/archives/C08DVQJ6U57/p1779567193160719",
   "eval": [],
   "reward": [
    "A"
   ],
   "twostage": false,
   "multiepoch": false,
   "validation": "validation",
   "filter": [],
   "owner": "?",
   "logRef": "recovery-b-distill--a-on-d-strong-on-qwen-weak-independent"
  },
  {
   "id": "abc-mistral",
   "parent": "mistral-base",
   "model": "Mistral-7B (base)",
   "base": "natural",
   "knob": "A/B/C reward sweep, Mistral-7B (independent verifier)",
   "metric": "base .281 / A .356 / B .219 / C .246",
   "headline": "Reversal reproduced cross-model: base 0.281 / A 0.356 / B 0.219 (robust 4e-7) / C 0.246 (worse than B, 3e-5).",
   "state": "done",
   "where": "/data/jobs/mistral_verify_1780417248",
   "url": "",
   "eval": [],
   "reward": [
    "A",
    "B",
    "C"
   ],
   "twostage": false,
   "multiepoch": false,
   "validation": "validation",
   "filter": [],
   "owner": "?",
   "logRef": "abc-mistral--abc-independent-mistral-7b-2026-06-02"
  },
  {
   "id": "abc-full-7956",
   "parent": "llama-base",
   "model": "Llama (base)",
   "base": "natural",
   "knob": "A/B/C reward sweep, full 7956 train",
   "metric": "B .2286 / C .2845",
   "headline": "B 0.2286 robustly beats base (Wilcoxon 1.1e-6); C 0.2845 worse, C vs B p=4e-7 → the martingale term is a net negative on a natural base.",
   "state": "done",
   "where": "/data/jobs/abc_{fullB,fullC}",
   "url": "",
   "eval": [],
   "reward": [
    "A",
    "B",
    "C"
   ],
   "twostage": false,
   "multiepoch": false,
   "validation": "orig",
   "filter": [],
   "owner": "ZH",
   "logRef": "abc-llama-full--b--c05-llama-31-8b-2026-06-02"
  },
  {
   "id": "abc-subset-2k",
   "parent": "llama-base",
   "model": "Llama (base)",
   "base": "natural",
   "knob": "A/B/C reward sweep, ~2k subset, λ-sweep",
   "metric": "C@0.5 .228 (p=0.18, not robust)",
   "headline": "C@0.5 0.228 beats base+B but not robust (Wilcoxon p=0.18, 31.7% q-wins); C@0.25/0.75 worse → narrow λ≈0.5 peak.",
   "state": "done",
   "where": "/data/jobs/abc_{A,B,C,C025,C075}",
   "url": "",
   "eval": [],
   "reward": [
    "A",
    "B",
    "C"
   ],
   "twostage": false,
   "multiepoch": false,
   "validation": "orig",
   "filter": [],
   "owner": "ZH",
   "logRef": "abc-llama-subset--abc02505075-llama-31-8b-2026-06-02"
  },
  {
   "id": "r1-replication",
   "parent": "— (eval-only)",
   "model": "R1-Distill-32B",
   "base": "eval-only",
   "knob": "pipeline: paper-era vs current, R1-Distill-32B, 437-q",
   "metric": "MS 0.001 vs 0.281 (~270×)",
   "headline": "Pipeline dominates model ~270×: MS 0.001 (paper) vs 0.281 (current) on the same model+eval.",
   "state": "done",
   "where": "scripts/eval_r1_distill/",
   "url": "",
   "eval": [],
   "reward": [],
   "twostage": false,
   "multiepoch": false,
   "validation": "orig",
   "filter": [],
   "owner": "ZH",
   "logRef": "2026-05-24--r1-distill-paper-replication-eval-437-q-paper-subset"
  },
  {
   "id": "paper-reextract",
   "parent": "— (eval-only)",
   "model": "GPT-4o-mini (judge)",
   "base": "eval-only",
   "knob": "re-judge 05-23/24 traces, GPT-4o-mini step-based",
   "metric": "MS 0.035 → 0.015",
   "headline": "Recipe effect survives pipeline change (B-distill 0.035 → AonB 0.015 MS); slope sign flips.",
   "state": "done",
   "where": "scripts/analysis/judge_reextract*",
   "url": "",
   "eval": [
    "judge"
   ],
   "reward": [],
   "twostage": false,
   "multiepoch": false,
   "validation": "orig",
   "filter": [],
   "owner": "ZH",
   "logRef": "2026-05-25--paper-pipeline-re-extraction-on-existing-arms-judge-gpt-4o-mini"
  },
  {
   "id": "A-on-base",
   "parent": "qwen-base",
   "model": "Qwen (base)",
   "base": "natural",
   "knob": "KL-safe + info-term on unbiased base",
   "metric": "Brier p=0.77, slope p=0.99 (null)",
   "headline": "Clean null — n.s. on Brier (0.77) and slope (0.99); recipe doesn't move an unbiased base.",
   "state": "done",
   "where": "slack verdict",
   "url": "https://meno-sh.slack.com/archives/C08DVQJ6U57/p1779613934406319",
   "eval": [],
   "reward": [
    "A"
   ],
   "twostage": false,
   "multiepoch": false,
   "validation": "orig",
   "filter": [],
   "owner": "ZH",
   "logRef": "aonbase-qwen--a-on-base-qwen3-32b-2026-05-24"
  },
  {
   "id": "A-on-B-distilled",
   "parent": "B-distill",
   "model": "Qwen3-32B (D)",
   "base": "biased",
   "knob": "KL-safe + info-term (martingale recovery)",
   "metric": "Brier(post) .242 → .190",
   "headline": "First validated win — Brier(post) 0.242 → 0.190 (p=2.86e-9); slope −0.398 → −0.227.",
   "state": "done",
   "where": "GitHub release v0.1-bdistilled-klsafe-info-term",
   "url": "https://github.com/meno-sh/Martingale-Training/releases/tag/v0.1-bdistilled-klsafe-info-term-2026-05-23",
   "eval": [],
   "reward": [
    "A"
   ],
   "twostage": false,
   "multiepoch": false,
   "validation": "validation",
   "filter": [],
   "owner": "ZH",
   "logRef": "aonb-qwen--a-on-d-qwen3-32b-2026-05-23"
  },
  {
   "id": "B-distill-LoRA",
   "parent": "qwen-base",
   "model": "Qwen3-32B",
   "base": "natural",
   "knob": "LoRA SFT on 352 confirmation-biased traces",
   "metric": "→ produces D",
   "headline": "Produces the deliberately-biased D model (the recovery target).",
   "state": "done",
   "where": "GitHub release v0.1-bdistilled-klsafe-info-term",
   "url": "https://github.com/meno-sh/Martingale-Training/releases/tag/v0.1-bdistilled-klsafe-info-term-2026-05-23",
   "eval": [],
   "reward": [],
   "twostage": false,
   "multiepoch": false,
   "validation": "orig",
   "filter": [],
   "owner": "ZH",
   "logRef": "bdistill-qwen--d-b-distill-qwen3-32b-2026-05-21"
  }
 ]
}