{
 "schema": 1,
 "provenance": {
  "generated": "2026-10-01",
  "generated_at": "2026-10-01T17:43:00.261Z",
  "record_commit": "85f423841dcfef3dc83f50f890f5bba160733839",
  "record_commit_short": "85f4238",
  "branch": "feat/llm-engines",
  "version": "0.1.0",
  "command": "hd replay, then npm run build in site/",
  "repo": null,
  "note": "Built from the working tree: results scored after the named commit are included.",
  "warnings": []
 },
 "axes": [
  {
   "key": "depth",
   "heading": "proof depth (difficulty)"
  },
  {
   "key": "reference_label",
   "heading": "gold label"
  },
  {
   "key": "theory_kind",
   "heading": "theory kind"
  },
  {
   "key": "theory_negation",
   "heading": "negation in theory"
  },
  {
   "key": "statement_negated",
   "heading": "negated statement"
  },
  {
   "key": "strategy",
   "heading": "question strategy"
  },
  {
   "key": "paraphrased",
   "heading": "paraphrased rules"
  },
  {
   "key": "theory_max_depth",
   "heading": "theory max depth"
  },
  {
   "key": "words_bin",
   "heading": "theory length (words)"
  },
  {
   "key": "rules_bin",
   "heading": "rules in theory"
  },
  {
   "key": "facts_bin",
   "heading": "facts in theory"
  },
  {
   "key": "proof_size_bin",
   "heading": "proof size"
  }
 ],
 "machine": {
  "description": "the machine every local engine ran on",
  "model": "MacBookPro18,4 (Apple M1 Max)",
  "memory": "32 GB unified memory",
  "note": "on Apple silicon, model weights live in GPU (Metal) memory, which a process's RSS does not count; memory figures here are macOS physical footprint (footprint -p), current and peak, where peak includes loading."
 },
 "facts": {
  "jev": {
   "kind": "hosted decision model",
   "maker": "TypeSafe",
   "parameters": "undisclosed",
   "architecture": "undisclosed",
   "weights": "not available (hosted)",
   "where_run": "TypeSafe API (typesafe-sdk); version recorded per row (jev-1.13.0)",
   "hardware": "not applicable (hosted)",
   "price": "$42 per billion input tokens; output free"
  },
  "laya": {
   "kind": "open decision model",
   "maker": "Convai Innovations",
   "parameters": "421M",
   "architecture": "ModernBERT-large encoder with decision heads, trained with RLCD",
   "context_tokens": 512,
   "weights": "0.84 GB (model.safetensors, convaiinnovations/laya repo root)",
   "precision": "as loaded by the laya package (PyTorch)",
   "where_run": "this machine, laya 0.3.21 package, PyTorch on MPS",
   "hardware": "not stated by the maker",
   "source": "https://huggingface.co/convaiinnovations/laya (model card checkpoint table)"
  },
  "kev-0.8b": {
   "kind": "open decision model",
   "maker": "Jared Palmer (individual)",
   "parameters": "0.8B base (Qwen3.5-0.8B-Base) + rank-16 LoRA adapter and pointer head",
   "architecture": "Qwen3.5 hybrid (Gated DeltaNet + attention) base, frozen; LoRA r16; pointer head",
   "weights": "1.75 GB base (Qwen/Qwen3.5-0.8B-Base@dc7cdfe2) + 0.05 GB adapter and head (jaredpalmer/kev-0.8b@54f4f877)",
   "precision": "bfloat16 (MLX), fp32 pointer head",
   "where_run": "this machine, Kev server c9c1f855 on MLX",
   "hardware": "\"Any Apple Silicon Mac, L4\" (Kev README)",
   "source": "https://github.com/jaredpalmer/kev"
  },
  "kev-4b": {
   "kind": "open decision model",
   "maker": "Jared Palmer (individual)",
   "parameters": "4B base (Qwen3.5-4B-Base) + rank-16 LoRA adapter (33.8M trainable) and pointer head",
   "architecture": "as kev-0.8b",
   "weights": "9.32 GB base (Qwen/Qwen3.5-4B-Base@1001bb4d) + 0.14 GB adapter and head (jaredpalmer/kev-4b@139fdd94)",
   "precision": "bfloat16 (MLX), fp32 pointer head",
   "where_run": "this machine, Kev server c9c1f855 on MLX",
   "hardware": "\"32 GB Mac, L40S, H100\" (Kev README)",
   "source": "https://huggingface.co/jaredpalmer/kev-4b"
  },
  "kev-9b": {
   "kind": "open decision model",
   "maker": "Jared Palmer (individual)",
   "parameters": "9B base (Qwen3.5-9B-Base) + rank-16 LoRA adapter and pointer head",
   "architecture": "as kev-0.8b",
   "weights": "19.31 GB base (Qwen/Qwen3.5-9B-Base@68c46c4b) + 0.18 GB adapter and head (jaredpalmer/kev-9b@b5d8c18e)",
   "precision": "bfloat16 (MLX), fp32 pointer head",
   "where_run": "this machine, Kev server c9c1f855 on MLX",
   "hardware": "\"32 GB Mac, L40S, H100\" (Kev README)",
   "source": "https://huggingface.co/jaredpalmer/kev-9b"
  },
  "kev-27b": {
   "kind": "open decision model (not run)",
   "maker": "Jared Palmer (individual)",
   "parameters": "27B, every weight fine-tuned from Qwen3.8-27B (post-trained)",
   "weights": "51.26 GB full weights (jaredpalmer/kev-27b)",
   "hardware": "\"B200, H200, H100 80 GB; 96-128 GB Mac (expected)\" (Kev README, later revision)",
   "not_run": "does not fit this 32 GB machine; would need a rented 80 GB GPU",
   "source": "https://huggingface.co/jaredpalmer/kev-27b"
  },
  "openai-gpt-6-luna-effort-none": {
   "kind": "hosted LLM, used as a classifier with reasoning off",
   "maker": "OpenAI",
   "parameters": "undisclosed",
   "architecture": "undisclosed",
   "context_tokens": 1050000,
   "weights": "not available (hosted)",
   "where_run": "OpenAI Chat Completions API, reasoning_effort none, strict JSON schema",
   "hardware": "not applicable (hosted)",
   "price": "$0.10 input / $0.50 output per million tokens",
   "source": "https://developers.openai.com/api/docs/models/gpt-6-luna"
  }
 },
 "memory": {
  "kev-9b": {
   "phys_footprint_mb": 17408,
   "phys_footprint_peak_mb": 18432,
   "runs": [
    {
     "file": "answers/kev-9b/proofwriter-cwa.runs.jsonl",
     "task": "proofwriter-cwa",
     "phys_footprint_mb": 17408,
     "phys_footprint_peak_mb": 18432,
     "pid": 22803
    },
    {
     "file": "answers/kev-9b/proofwriter-owa.runs.jsonl",
     "task": "proofwriter-owa",
     "phys_footprint_mb": 17408,
     "phys_footprint_peak_mb": 18432,
     "pid": 22803
    },
    {
     "file": "answers/kev-9b/proofwriter-owa.runs.jsonl",
     "task": "proofwriter-owa",
     "phys_footprint_mb": 17408,
     "phys_footprint_peak_mb": 18432,
     "pid": 22803
    }
   ]
  },
  "kev-0.8b": {
   "phys_footprint_mb": 2176,
   "phys_footprint_peak_mb": 3316,
   "runs": [
    {
     "file": "timing/kev-0.8b/proofwriter-cwa.runs.jsonl",
     "task": "proofwriter-cwa",
     "phys_footprint_mb": 2162,
     "phys_footprint_peak_mb": 3316,
     "pid": 57457
    },
    {
     "file": "timing/kev-0.8b/proofwriter-owa.runs.jsonl",
     "task": "proofwriter-owa",
     "phys_footprint_mb": 2176,
     "phys_footprint_peak_mb": 3316,
     "pid": 57457
    }
   ]
  },
  "kev-4b": {
   "phys_footprint_mb": 8819,
   "phys_footprint_peak_mb": 17408,
   "runs": [
    {
     "file": "timing/kev-4b/proofwriter-cwa.runs.jsonl",
     "task": "proofwriter-cwa",
     "phys_footprint_mb": 8819,
     "phys_footprint_peak_mb": 17408,
     "pid": 1575
    },
    {
     "file": "timing/kev-4b/proofwriter-owa.runs.jsonl",
     "task": "proofwriter-owa",
     "phys_footprint_mb": 8819,
     "phys_footprint_peak_mb": 17408,
     "pid": 1575
    }
   ]
  },
  "laya": {
   "phys_footprint_mb": 10240,
   "phys_footprint_peak_mb": 10240,
   "runs": [
    {
     "file": "timing/laya/proofwriter-cwa.runs.jsonl",
     "task": "proofwriter-cwa",
     "phys_footprint_mb": 10240,
     "phys_footprint_peak_mb": 10240,
     "pid": 17800
    },
    {
     "file": "timing/laya/proofwriter-owa.runs.jsonl",
     "task": "proofwriter-owa",
     "phys_footprint_mb": 10240,
     "phys_footprint_peak_mb": 10240,
     "pid": 16278
    }
   ]
  }
 },
 "tasks": [
  {
   "slug": "proofwriter-cwa",
   "semantics": "CWA",
   "question": "Using only the facts and rules in the text, is the statement true or false? Assume that anything which cannot be derived from the facts and rules is false: a positive statement is true only if it can be derived, and a statement with 'not' is true when the unnegated statement cannot be derived. A rule applies when all of its conditions hold; 'not' in a condition holds when the unnegated condition cannot be derived.",
   "options": [
    "true",
    "false"
   ],
   "descriptions": {
    "true": "The statement holds under the closed-world assumption.",
    "false": "The statement does not hold under the closed-world assumption."
   },
   "n": 1800,
   "seed": 0,
   "configs": [
    "depth-0",
    "depth-1",
    "depth-2",
    "depth-3",
    "depth-5",
    "NatLang"
   ],
   "depths": [
    0,
    1,
    2,
    3,
    4,
    5
   ],
   "label_check": "every selected label recomputed by hard_decisions.solver; zero disagreements",
   "strata": [
    {
     "available": 30200,
     "depth": 0,
     "label": "false",
     "selected": 150
    },
    {
     "available": 24879,
     "depth": 0,
     "label": "true",
     "selected": 150
    },
    {
     "available": 12168,
     "depth": 1,
     "label": "false",
     "selected": 150
    },
    {
     "available": 16704,
     "depth": 1,
     "label": "true",
     "selected": 150
    },
    {
     "available": 6123,
     "depth": 2,
     "label": "false",
     "selected": 150
    },
    {
     "available": 6716,
     "depth": 2,
     "label": "true",
     "selected": 150
    },
    {
     "available": 3402,
     "depth": 3,
     "label": "false",
     "selected": 150
    },
    {
     "available": 3520,
     "depth": 3,
     "label": "true",
     "selected": 150
    },
    {
     "available": 1238,
     "depth": 4,
     "label": "false",
     "selected": 150
    },
    {
     "available": 1285,
     "depth": 4,
     "label": "true",
     "selected": 150
    },
    {
     "available": 1041,
     "depth": 5,
     "label": "false",
     "selected": 150
    },
    {
     "available": 1060,
     "depth": 5,
     "label": "true",
     "selected": 150
    }
   ],
   "pool": {
    "depth": {
     "0": 55079,
     "1": 28872,
     "2": 12839,
     "3": 6922,
     "4": 2523,
     "5": 2101
    },
    "facts_bin": {
     "0-3": 21068,
     "13+": 22206,
     "4-7": 27057,
     "8-12": 38005
    },
    "paraphrased": {
     "False": 100328,
     "True": 8008
    },
    "proof_size_bin": {
     "2": 10779,
     "0-1": 74380,
     "3-4": 13023,
     "5+": 10154
    },
    "reference_label": {
     "false": 54172,
     "true": 54164
    },
    "rules_bin": {
     "0-2": 20220,
     "3-5": 18060,
     "6-7": 32207,
     "8+": 37849
    },
    "statement_negated": {
     "False": 54172,
     "True": 54164
    },
    "strategy": {
     "inv-proof": 29700,
     "inv-random": 12157,
     "inv-rconc": 12307,
     "proof": 29700,
     "random": 17478,
     "rconc": 6994
    },
    "theory_kind": {
     "attribute": 57452,
     "relation": 50884
    },
    "theory_max_depth": {
     "0": 10222,
     "1": 24135,
     "2": 22375,
     "3": 27493,
     "4": 3648,
     "5": 18719,
     "6": 1512,
     "7": 136,
     "8": 48,
     "9": 48
    },
    "theory_negation": {
     "negation": 50032,
     "no-negation": 58304
    },
    "words_bin": {
     "0-49": 20016,
     "110+": 35228,
     "50-79": 26816,
     "80-109": 26276
    }
   },
   "selected": {
    "depth": {
     "0": 300,
     "1": 300,
     "2": 300,
     "3": 300,
     "4": 300,
     "5": 300
    },
    "facts_bin": {
     "0-3": 360,
     "13+": 203,
     "4-7": 425,
     "8-12": 812
    },
    "paraphrased": {
     "False": 1306,
     "True": 494
    },
    "proof_size_bin": {
     "2": 82,
     "0-1": 966,
     "3-4": 237,
     "5+": 515
    },
    "reference_label": {
     "false": 900,
     "true": 900
    },
    "rules_bin": {
     "0-2": 208,
     "3-5": 405,
     "6-7": 605,
     "8+": 582
    },
    "statement_negated": {
     "False": 882,
     "True": 918
    },
    "strategy": {
     "inv-proof": 501,
     "inv-random": 75,
     "inv-rconc": 342,
     "proof": 483,
     "random": 75,
     "rconc": 324
    },
    "theory_kind": {
     "attribute": 1016,
     "relation": 784
    },
    "theory_max_depth": {
     "0": 71,
     "1": 155,
     "2": 230,
     "3": 619,
     "4": 171,
     "5": 513,
     "6": 38,
     "7": 2,
     "8": 1
    },
    "theory_negation": {
     "negation": 870,
     "no-negation": 930
    },
    "words_bin": {
     "0-49": 256,
     "110+": 630,
     "50-79": 432,
     "80-109": 482
    }
   },
   "examples": {
    "0": {
     "id": "AttNoneg-CWA-D2-1376-Q9",
     "text": "Bob is cold. Bob is round. Fiona is quiet. Cold people are rough. Quiet people are cold.\n\nStatement: Bob is not white.",
     "label": "true",
     "depth": 0,
     "paraphrased": false
    },
    "1": {
     "id": "RelNeg-CWA-D1-1114-Q3",
     "text": "The dog is rough. Rough things are big.\n\nStatement: The dog is big.",
     "label": "true",
     "depth": 1,
     "paraphrased": false
    },
    "2": {
     "id": "RelNoneg-CWA-D2-1397-Q5",
     "text": "The lion is kind. Blue things are red. All kind things are blue.\n\nStatement: The lion is red.",
     "label": "true",
     "depth": 2,
     "paraphrased": false
    },
    "3": {
     "id": "RelNeg-CWA-D3-7-Q7",
     "text": "The bald eagle is cold. Big things are green. Cold things are blue. All blue, cold things are big.\n\nStatement: The bald eagle is green.",
     "label": "true",
     "depth": 3,
     "paraphrased": false
    },
    "4": {
     "id": "AttNoneg-CWA-D1-3169-Q5",
     "text": "Bob is white. Fiona is quiet. Harry is big. All quiet things are red. If something is nice then it is big. Rough, quiet things are big. If Bob is nice then Bob is blue. Big, blue things are red. Red things are nice.\n\nStatement: Bob is not big.",
     "label": "true",
     "depth": 4,
     "paraphrased": false
    },
    "5": {
     "id": "RelNoneg-CWA-D0-4382-Q3",
     "text": "The dog is young. All red things are blue. If something is red and blue then it is young. Kind things are rough. Red, blue things are kind. Rough, young things are red. All kind things are blue. If something is kind then it is rough. All blue, rough things are red.\n\nStatement: The dog is not red.",
     "label": "true",
     "depth": 5,
     "paraphrased": false
    }
   },
   "profile": {
    "depth": {
     "0": {
      "n": 300,
      "mean_depth": 0,
      "labels": {
       "true": 150,
       "false": 150
      }
     },
     "1": {
      "n": 300,
      "mean_depth": 1,
      "labels": {
       "true": 150,
       "false": 150
      }
     },
     "2": {
      "n": 300,
      "mean_depth": 2,
      "labels": {
       "true": 150,
       "false": 150
      }
     },
     "3": {
      "n": 300,
      "mean_depth": 3,
      "labels": {
       "true": 150,
       "false": 150
      }
     },
     "4": {
      "n": 300,
      "mean_depth": 4,
      "labels": {
       "true": 150,
       "false": 150
      }
     },
     "5": {
      "n": 300,
      "mean_depth": 5,
      "labels": {
       "true": 150,
       "false": 150
      }
     }
    },
    "reference_label": {
     "true": {
      "n": 900,
      "mean_depth": 2.5,
      "labels": {
       "true": 900
      }
     },
     "false": {
      "n": 900,
      "mean_depth": 2.5,
      "labels": {
       "false": 900
      }
     }
    },
    "theory_kind": {
     "attribute": {
      "n": 1016,
      "mean_depth": 2.6348425196850394,
      "labels": {
       "true": 510,
       "false": 506
      }
     },
     "relation": {
      "n": 784,
      "mean_depth": 2.3252551020408165,
      "labels": {
       "true": 390,
       "false": 394
      }
     }
    },
    "theory_negation": {
     "negation": {
      "n": 870,
      "mean_depth": 2.481609195402299,
      "labels": {
       "true": 436,
       "false": 434
      }
     },
     "no-negation": {
      "n": 930,
      "mean_depth": 2.5172043010752687,
      "labels": {
       "false": 466,
       "true": 464
      }
     }
    },
    "statement_negated": {
     "False": {
      "n": 882,
      "mean_depth": 2.450113378684807,
      "labels": {
       "true": 483,
       "false": 399
      }
     },
     "True": {
      "n": 918,
      "mean_depth": 2.547930283224401,
      "labels": {
       "false": 501,
       "true": 417
      }
     }
    },
    "strategy": {
     "proof": {
      "n": 483,
      "mean_depth": 2.5341614906832297,
      "labels": {
       "true": 483
      }
     },
     "inv-proof": {
      "n": 501,
      "mean_depth": 2.6207584830339323,
      "labels": {
       "false": 501
      }
     },
     "inv-rconc": {
      "n": 342,
      "mean_depth": 3,
      "labels": {
       "true": 342
      }
     },
     "rconc": {
      "n": 324,
      "mean_depth": 2.8919753086419755,
      "labels": {
       "false": 324
      }
     },
     "inv-random": {
      "n": 75,
      "mean_depth": 0,
      "labels": {
       "true": 75
      }
     },
     "random": {
      "n": 75,
      "mean_depth": 0,
      "labels": {
       "false": 75
      }
     }
    },
    "paraphrased": {
     "False": {
      "n": 1306,
      "mean_depth": 2.670750382848392,
      "labels": {
       "true": 650,
       "false": 656
      }
     },
     "True": {
      "n": 494,
      "mean_depth": 2.048582995951417,
      "labels": {
       "false": 244,
       "true": 250
      }
     }
    },
    "theory_max_depth": {
     "0": {
      "n": 71,
      "mean_depth": 1.1126760563380282,
      "labels": {
       "true": 54,
       "false": 17
      }
     },
     "1": {
      "n": 155,
      "mean_depth": 1.4129032258064516,
      "labels": {
       "false": 76,
       "true": 79
      }
     },
     "2": {
      "n": 230,
      "mean_depth": 1.9130434782608696,
      "labels": {
       "true": 121,
       "false": 109
      }
     },
     "3": {
      "n": 619,
      "mean_depth": 1.9709208400646203,
      "labels": {
       "true": 296,
       "false": 323
      }
     },
     "4": {
      "n": 171,
      "mean_depth": 3.0584795321637426,
      "labels": {
       "true": 87,
       "false": 84
      }
     },
     "5": {
      "n": 513,
      "mean_depth": 3.695906432748538,
      "labels": {
       "true": 244,
       "false": 269
      }
     },
     "6": {
      "n": 38,
      "mean_depth": 2.9210526315789473,
      "labels": {
       "true": 16,
       "false": 22
      }
     },
     "7": {
      "n": 2,
      "mean_depth": 3.5,
      "labels": {
       "true": 2
      }
     },
     "8": {
      "n": 1,
      "mean_depth": 5,
      "labels": {
       "true": 1
      }
     }
    },
    "words_bin": {
     "50-79": {
      "n": 432,
      "mean_depth": 2.5856481481481484,
      "labels": {
       "true": 215,
       "false": 217
      }
     },
     "110+": {
      "n": 630,
      "mean_depth": 2.8333333333333335,
      "labels": {
       "true": 318,
       "false": 312
      }
     },
     "80-109": {
      "n": 482,
      "mean_depth": 2.5622406639004147,
      "labels": {
       "false": 245,
       "true": 237
      }
     },
     "0-49": {
      "n": 256,
      "mean_depth": 1.41796875,
      "labels": {
       "true": 130,
       "false": 126
      }
     }
    },
    "rules_bin": {
     "6-7": {
      "n": 605,
      "mean_depth": 2.8495867768595042,
      "labels": {
       "true": 306,
       "false": 299
      }
     },
     "8+": {
      "n": 582,
      "mean_depth": 2.922680412371134,
      "labels": {
       "false": 293,
       "true": 289
      }
     },
     "0-2": {
      "n": 208,
      "mean_depth": 0.9567307692307693,
      "labels": {
       "true": 104,
       "false": 104
      }
     },
     "3-5": {
      "n": 405,
      "mean_depth": 2.162962962962963,
      "labels": {
       "false": 204,
       "true": 201
      }
     }
    },
    "facts_bin": {
     "8-12": {
      "n": 812,
      "mean_depth": 2.5246305418719213,
      "labels": {
       "true": 405,
       "false": 407
      }
     },
     "13+": {
      "n": 203,
      "mean_depth": 3.3251231527093594,
      "labels": {
       "true": 101,
       "false": 102
      }
     },
     "4-7": {
      "n": 425,
      "mean_depth": 2.5788235294117645,
      "labels": {
       "true": 207,
       "false": 218
      }
     },
     "0-3": {
      "n": 360,
      "mean_depth": 1.886111111111111,
      "labels": {
       "false": 173,
       "true": 187
      }
     }
    },
    "proof_size_bin": {
     "2": {
      "n": 82,
      "mean_depth": 1,
      "labels": {
       "true": 41,
       "false": 41
      }
     },
     "5+": {
      "n": 515,
      "mean_depth": 3.8679611650485435,
      "labels": {
       "true": 248,
       "false": 267
      }
     },
     "0-1": {
      "n": 966,
      "mean_depth": 2.0320910973084887,
      "labels": {
       "true": 492,
       "false": 474
      }
     },
     "3-4": {
      "n": 237,
      "mean_depth": 1.9535864978902953,
      "labels": {
       "false": 118,
       "true": 119
      }
     }
    }
   }
  },
  {
   "slug": "proofwriter-owa",
   "semantics": "OWA",
   "question": "Using only the facts and rules in the text, is the statement true, false, or unknown? A statement is true if it can be derived from the facts and rules, and false if its negation can be derived. If neither the statement nor its negation can be derived, it is unknown. A rule applies only when all of its conditions are established; 'not' in a condition requires the negation to be stated or derived.",
   "options": [
    "true",
    "false",
    "unknown"
   ],
   "descriptions": {
    "true": "The statement follows from the facts and rules.",
    "false": "The negation of the statement follows from the facts and rules.",
    "unknown": "Neither the statement nor its negation follows from the facts and rules."
   },
   "n": 1800,
   "seed": 0,
   "configs": [
    "depth-0",
    "depth-1",
    "depth-2",
    "depth-3",
    "depth-5",
    "NatLang"
   ],
   "depths": [
    0,
    1,
    2,
    3,
    4,
    5
   ],
   "label_check": "every selected label recomputed by hard_decisions.solver; zero disagreements",
   "strata": [
    {
     "available": 12625,
     "depth": 0,
     "label": "false",
     "selected": 100
    },
    {
     "available": 12625,
     "depth": 0,
     "label": "true",
     "selected": 100
    },
    {
     "available": 30178,
     "depth": 0,
     "label": "unknown",
     "selected": 100
    },
    {
     "available": 7236,
     "depth": 1,
     "label": "false",
     "selected": 101
    },
    {
     "available": 7236,
     "depth": 1,
     "label": "true",
     "selected": 101
    },
    {
     "available": 14937,
     "depth": 1,
     "label": "unknown",
     "selected": 100
    },
    {
     "available": 4629,
     "depth": 2,
     "label": "false",
     "selected": 101
    },
    {
     "available": 4629,
     "depth": 2,
     "label": "true",
     "selected": 101
    },
    {
     "available": 3276,
     "depth": 2,
     "label": "unknown",
     "selected": 101
    },
    {
     "available": 2835,
     "depth": 3,
     "label": "false",
     "selected": 101
    },
    {
     "available": 2835,
     "depth": 3,
     "label": "true",
     "selected": 101
    },
    {
     "available": 1024,
     "depth": 3,
     "label": "unknown",
     "selected": 101
    },
    {
     "available": 1016,
     "depth": 4,
     "label": "false",
     "selected": 101
    },
    {
     "available": 1016,
     "depth": 4,
     "label": "true",
     "selected": 101
    },
    {
     "available": 348,
     "depth": 4,
     "label": "unknown",
     "selected": 101
    },
    {
     "available": 954,
     "depth": 5,
     "label": "false",
     "selected": 101
    },
    {
     "available": 954,
     "depth": 5,
     "label": "true",
     "selected": 101
    },
    {
     "available": 87,
     "depth": 5,
     "label": "unknown",
     "selected": 87
    }
   ],
   "pool": {
    "depth": {
     "0": 55428,
     "1": 29409,
     "2": 12534,
     "3": 6694,
     "4": 2380,
     "5": 1995
    },
    "facts_bin": {
     "0-3": 21763,
     "13+": 20377,
     "4-7": 28904,
     "8-12": 37396
    },
    "paraphrased": {
     "False": 100432,
     "True": 8008
    },
    "proof_size_bin": {
     "2": 11566,
     "0-1": 75100,
     "3-4": 12431,
     "5+": 9343
    },
    "reference_label": {
     "false": 29295,
     "true": 29295,
     "unknown": 49850
    },
    "rules_bin": {
     "0-2": 19362,
     "3-5": 17588,
     "6-7": 32543,
     "8+": 38947
    },
    "statement_negated": {
     "False": 54160,
     "True": 54280
    },
    "strategy": {
     "inv-proof": 29295,
     "inv-random": 12439,
     "inv-rconc": 12482,
     "proof": 29295,
     "random": 17739,
     "rconc": 7190
    },
    "theory_kind": {
     "attribute": 57436,
     "relation": 51004
    },
    "theory_max_depth": {
     "0": 11638,
     "1": 24232,
     "2": 21462,
     "3": 27380,
     "4": 3410,
     "5": 18770,
     "6": 1308,
     "7": 120,
     "8": 96,
     "9": 24
    },
    "theory_negation": {
     "negation": 50136,
     "no-negation": 58304
    },
    "words_bin": {
     "0-49": 18970,
     "110+": 35900,
     "50-79": 27621,
     "80-109": 25949
    }
   },
   "selected": {
    "depth": {
     "0": 300,
     "1": 302,
     "2": 303,
     "3": 303,
     "4": 303,
     "5": 289
    },
    "facts_bin": {
     "0-3": 334,
     "13+": 159,
     "4-7": 435,
     "8-12": 872
    },
    "paraphrased": {
     "False": 1298,
     "True": 502
    },
    "proof_size_bin": {
     "2": 107,
     "0-1": 790,
     "3-4": 296,
     "5+": 607
    },
    "reference_label": {
     "false": 605,
     "true": 605,
     "unknown": 590
    },
    "rules_bin": {
     "0-2": 204,
     "3-5": 438,
     "6-7": 564,
     "8+": 594
    },
    "statement_negated": {
     "False": 899,
     "True": 901
    },
    "strategy": {
     "inv-proof": 605,
     "inv-random": 50,
     "inv-rconc": 249,
     "proof": 605,
     "random": 50,
     "rconc": 241
    },
    "theory_kind": {
     "attribute": 1059,
     "relation": 741
    },
    "theory_max_depth": {
     "0": 71,
     "1": 145,
     "2": 206,
     "3": 620,
     "4": 162,
     "5": 571,
     "6": 18,
     "7": 5,
     "8": 2
    },
    "theory_negation": {
     "negation": 881,
     "no-negation": 919
    },
    "words_bin": {
     "0-49": 247,
     "110+": 618,
     "50-79": 416,
     "80-109": 519
    }
   },
   "examples": {
    "0": {
     "id": "AttNeg-OWA-D0-4837-Q1",
     "text": "Gary is not white. If something is cold and big then it is not young.\n\nStatement: Gary is not white.",
     "label": "true",
     "depth": 0,
     "paraphrased": false
    },
    "1": {
     "id": "RelNeg-OWA-D1-1830-Q3",
     "text": "The lion is nice. If something is nice then it is not blue.\n\nStatement: The lion is not blue.",
     "label": "true",
     "depth": 1,
     "paraphrased": false
    },
    "2": {
     "id": "RelNoneg-OWA-D2-1397-Q5",
     "text": "The lion is kind. Blue things are red. All kind things are blue.\n\nStatement: The lion is red.",
     "label": "true",
     "depth": 2,
     "paraphrased": false
    },
    "3": {
     "id": "RelNoneg-OWA-D3-1056-Q7",
     "text": "The squirrel is kind. All kind people are big. Green, big people are red. Big, kind people are green.\n\nStatement: The squirrel is red.",
     "label": "true",
     "depth": 3,
     "paraphrased": false
    },
    "4": {
     "id": "AttNoneg-OWA-D5-1382-Q9",
     "text": "Bob is big. Bob is cold. Bob is kind. Bob is round. Bob is smart. Dave is cold. Erin is big. Erin is green. Fiona is big. Fiona is smart. Big, green things are round. If something is cold and blue then it is smart. Smart, round things are kind. Round, big things are cold. Cold things are blue.\n\nStatement: Erin is smart.",
     "label": "true",
     "depth": 4,
     "paraphrased": false
    },
    "5": {
     "id": "AttNoneg-OWA-D5-1382-Q11",
     "text": "Bob is big. Bob is cold. Bob is kind. Bob is round. Bob is smart. Dave is cold. Erin is big. Erin is green. Fiona is big. Fiona is smart. Big, green things are round. If something is cold and blue then it is smart. Smart, round things are kind. Round, big things are cold. Cold things are blue.\n\nStatement: Erin is kind.",
     "label": "true",
     "depth": 5,
     "paraphrased": false
    }
   },
   "profile": {
    "depth": {
     "0": {
      "n": 300,
      "mean_depth": 0,
      "labels": {
       "true": 100,
       "false": 100,
       "unknown": 100
      }
     },
     "1": {
      "n": 302,
      "mean_depth": 1,
      "labels": {
       "true": 101,
       "false": 101,
       "unknown": 100
      }
     },
     "2": {
      "n": 303,
      "mean_depth": 2,
      "labels": {
       "true": 101,
       "false": 101,
       "unknown": 101
      }
     },
     "3": {
      "n": 303,
      "mean_depth": 3,
      "labels": {
       "true": 101,
       "false": 101,
       "unknown": 101
      }
     },
     "4": {
      "n": 303,
      "mean_depth": 4,
      "labels": {
       "true": 101,
       "false": 101,
       "unknown": 101
      }
     },
     "5": {
      "n": 289,
      "mean_depth": 5,
      "labels": {
       "true": 101,
       "false": 101,
       "unknown": 87
      }
     }
    },
    "reference_label": {
     "true": {
      "n": 605,
      "mean_depth": 2.5041322314049586,
      "labels": {
       "true": 605
      }
     },
     "false": {
      "n": 605,
      "mean_depth": 2.5041322314049586,
      "labels": {
       "false": 605
      }
     },
     "unknown": {
      "n": 590,
      "mean_depth": 2.447457627118644,
      "labels": {
       "unknown": 590
      }
     }
    },
    "theory_kind": {
     "attribute": {
      "n": 1059,
      "mean_depth": 2.65533522190746,
      "labels": {
       "true": 350,
       "unknown": 358,
       "false": 351
      }
     },
     "relation": {
      "n": 741,
      "mean_depth": 2.242914979757085,
      "labels": {
       "false": 254,
       "true": 255,
       "unknown": 232
      }
     }
    },
    "theory_negation": {
     "negation": {
      "n": 881,
      "mean_depth": 2.4585698070374575,
      "labels": {
       "true": 309,
       "false": 308,
       "unknown": 264
      }
     },
     "no-negation": {
      "n": 919,
      "mean_depth": 2.511425462459195,
      "labels": {
       "unknown": 326,
       "true": 296,
       "false": 297
      }
     }
    },
    "statement_negated": {
     "False": {
      "n": 899,
      "mean_depth": 2.474972191323693,
      "labels": {
       "true": 373,
       "false": 235,
       "unknown": 291
      }
     },
     "True": {
      "n": 901,
      "mean_depth": 2.4961154273029966,
      "labels": {
       "false": 370,
       "unknown": 299,
       "true": 232
      }
     }
    },
    "strategy": {
     "proof": {
      "n": 605,
      "mean_depth": 2.5041322314049586,
      "labels": {
       "true": 605
      }
     },
     "inv-proof": {
      "n": 605,
      "mean_depth": 2.5041322314049586,
      "labels": {
       "false": 605
      }
     },
     "rconc": {
      "n": 241,
      "mean_depth": 2.912863070539419,
      "labels": {
       "unknown": 241
      }
     },
     "inv-rconc": {
      "n": 249,
      "mean_depth": 2.9799196787148596,
      "labels": {
       "unknown": 249
      }
     },
     "random": {
      "n": 50,
      "mean_depth": 0,
      "labels": {
       "unknown": 50
      }
     },
     "inv-random": {
      "n": 50,
      "mean_depth": 0,
      "labels": {
       "unknown": 50
      }
     }
    },
    "paraphrased": {
     "False": {
      "n": 1298,
      "mean_depth": 2.669491525423729,
      "labels": {
       "true": 425,
       "false": 425,
       "unknown": 448
      }
     },
     "True": {
      "n": 502,
      "mean_depth": 2.00996015936255,
      "labels": {
       "true": 180,
       "false": 180,
       "unknown": 142
      }
     }
    },
    "theory_max_depth": {
     "0": {
      "n": 71,
      "mean_depth": 0.8873239436619719,
      "labels": {
       "true": 15,
       "false": 13,
       "unknown": 43
      }
     },
     "1": {
      "n": 145,
      "mean_depth": 1.1448275862068966,
      "labels": {
       "true": 40,
       "false": 42,
       "unknown": 63
      }
     },
     "2": {
      "n": 206,
      "mean_depth": 1.6650485436893203,
      "labels": {
       "false": 61,
       "unknown": 81,
       "true": 64
      }
     },
     "3": {
      "n": 620,
      "mean_depth": 1.8903225806451613,
      "labels": {
       "unknown": 210,
       "false": 203,
       "true": 207
      }
     },
     "4": {
      "n": 162,
      "mean_depth": 2.802469135802469,
      "labels": {
       "true": 55,
       "false": 61,
       "unknown": 46
      }
     },
     "5": {
      "n": 571,
      "mean_depth": 3.831873905429072,
      "labels": {
       "true": 218,
       "false": 219,
       "unknown": 134
      }
     },
     "6": {
      "n": 18,
      "mean_depth": 3.3333333333333335,
      "labels": {
       "false": 5,
       "unknown": 8,
       "true": 5
      }
     },
     "7": {
      "n": 5,
      "mean_depth": 4,
      "labels": {
       "true": 1,
       "false": 1,
       "unknown": 3
      }
     },
     "8": {
      "n": 2,
      "mean_depth": 4,
      "labels": {
       "unknown": 2
      }
     }
    },
    "words_bin": {
     "80-109": {
      "n": 519,
      "mean_depth": 2.6396917148362236,
      "labels": {
       "true": 172,
       "unknown": 178,
       "false": 169
      }
     },
     "110+": {
      "n": 618,
      "mean_depth": 2.6812297734627832,
      "labels": {
       "false": 229,
       "true": 228,
       "unknown": 161
      }
     },
     "50-79": {
      "n": 416,
      "mean_depth": 2.5889423076923075,
      "labels": {
       "unknown": 153,
       "false": 133,
       "true": 130
      }
     },
     "0-49": {
      "n": 247,
      "mean_depth": 1.4979757085020242,
      "labels": {
       "unknown": 98,
       "false": 74,
       "true": 75
      }
     }
    },
    "rules_bin": {
     "8+": {
      "n": 594,
      "mean_depth": 2.991582491582492,
      "labels": {
       "true": 177,
       "false": 175,
       "unknown": 242
      }
     },
     "3-5": {
      "n": 438,
      "mean_depth": 2.310502283105023,
      "labels": {
       "true": 165,
       "unknown": 105,
       "false": 168
      }
     },
     "6-7": {
      "n": 564,
      "mean_depth": 2.645390070921986,
      "labels": {
       "true": 194,
       "false": 196,
       "unknown": 174
      }
     },
     "0-2": {
      "n": 204,
      "mean_depth": 0.946078431372549,
      "labels": {
       "false": 66,
       "unknown": 69,
       "true": 69
      }
     }
    },
    "facts_bin": {
     "8-12": {
      "n": 872,
      "mean_depth": 2.55848623853211,
      "labels": {
       "true": 312,
       "unknown": 252,
       "false": 308
      }
     },
     "13+": {
      "n": 159,
      "mean_depth": 3.006289308176101,
      "labels": {
       "false": 60,
       "unknown": 38,
       "true": 61
      }
     },
     "4-7": {
      "n": 435,
      "mean_depth": 2.671264367816092,
      "labels": {
       "unknown": 157,
       "false": 144,
       "true": 134
      }
     },
     "0-3": {
      "n": 334,
      "mean_depth": 1.8053892215568863,
      "labels": {
       "false": 93,
       "true": 98,
       "unknown": 143
      }
     }
    },
    "proof_size_bin": {
     "2": {
      "n": 107,
      "mean_depth": 1,
      "labels": {
       "true": 53,
       "false": 54
      }
     },
     "5+": {
      "n": 607,
      "mean_depth": 3.838550247116969,
      "labels": {
       "true": 303,
       "false": 304
      }
     },
     "0-1": {
      "n": 790,
      "mean_depth": 1.8278481012658228,
      "labels": {
       "unknown": 590,
       "true": 100,
       "false": 100
      }
     },
     "3-4": {
      "n": 296,
      "mean_depth": 2.0033783783783785,
      "labels": {
       "true": 149,
       "false": 147
      }
     }
    }
   }
  }
 ],
 "results": {
  "proofwriter-cwa": {
   "studies": {
    "engines": {
     "jev": {
      "file": "studies/proofwriter-cwa-jev.jsonl",
      "modified": "2026-10-01T16:36:17.268Z",
      "overall": {
       "n": 1800,
       "correct": 1607,
       "accuracy": 0.8927777777777778,
       "lo": 0.8783333333333333,
       "hi": 0.9066666666666666,
       "macro_f1": 0.8927761561826089,
       "invalid": 0,
       "best_constant": 0.5,
       "chance": 0.5,
       "recall": {
        "false": 0.8888888888888888,
        "true": 0.8966666666666666
       },
       "confusion": {
        "false": {
         "false": 800,
         "invalid": 0,
         "true": 100
        },
        "true": {
         "false": 93,
         "invalid": 0,
         "true": 807
        }
       },
       "model": [
        "jev-1.13.0"
       ]
      },
      "axes": {
       "depth": [
        {
         "value": "0",
         "n": 300,
         "correct": 298,
         "accuracy": 0.9933333333333333,
         "lo": 0.9833333333333333,
         "hi": 1,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 300,
         "correct": 281,
         "accuracy": 0.9366666666666666,
         "lo": 0.91,
         "hi": 0.9633333333333334,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 300,
         "correct": 276,
         "accuracy": 0.92,
         "lo": 0.8866666666666667,
         "hi": 0.9466666666666667,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 300,
         "correct": 255,
         "accuracy": 0.85,
         "lo": 0.8133333333333334,
         "hi": 0.8933333333333333,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 300,
         "correct": 229,
         "accuracy": 0.7633333333333333,
         "lo": 0.7166666666666667,
         "hi": 0.81,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 300,
         "correct": 268,
         "accuracy": 0.8933333333333333,
         "lo": 0.8566666666666667,
         "hi": 0.9233333333333333,
         "best_constant": 0.5,
         "invalid": 0
        }
       ],
       "reference_label": [
        {
         "value": "false",
         "n": 900,
         "correct": 800,
         "accuracy": 0.8888888888888888,
         "lo": 0.8688888888888889,
         "hi": 0.9088888888888889,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "true",
         "n": 900,
         "correct": 807,
         "accuracy": 0.8966666666666666,
         "lo": 0.8766666666666667,
         "hi": 0.9155555555555556,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "theory_kind": [
        {
         "value": "attribute",
         "n": 1016,
         "correct": 872,
         "accuracy": 0.8582677165354331,
         "lo": 0.8366141732283464,
         "hi": 0.8769685039370079,
         "best_constant": 0.5019685039370079,
         "invalid": 0
        },
        {
         "value": "relation",
         "n": 784,
         "correct": 735,
         "accuracy": 0.9375,
         "lo": 0.9209183673469388,
         "hi": 0.9540816326530612,
         "best_constant": 0.5025510204081632,
         "invalid": 0
        }
       ],
       "theory_negation": [
        {
         "value": "negation",
         "n": 870,
         "correct": 815,
         "accuracy": 0.9367816091954023,
         "lo": 0.9206896551724137,
         "hi": 0.9528735632183908,
         "best_constant": 0.5011494252873563,
         "invalid": 0
        },
        {
         "value": "no-negation",
         "n": 930,
         "correct": 792,
         "accuracy": 0.8516129032258064,
         "lo": 0.8279569892473119,
         "hi": 0.875268817204301,
         "best_constant": 0.5010752688172043,
         "invalid": 0
        }
       ],
       "statement_negated": [
        {
         "value": "False",
         "n": 882,
         "correct": 790,
         "accuracy": 0.8956916099773242,
         "lo": 0.8752834467120182,
         "hi": 0.9160997732426304,
         "best_constant": 0.5476190476190477,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 918,
         "correct": 817,
         "accuracy": 0.8899782135076253,
         "lo": 0.8703703703703703,
         "hi": 0.9095860566448801,
         "best_constant": 0.545751633986928,
         "invalid": 0
        }
       ],
       "strategy": [
        {
         "value": "inv-proof",
         "n": 501,
         "correct": 426,
         "accuracy": 0.8502994011976048,
         "lo": 0.8163672654690619,
         "hi": 0.8842315369261478,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-random",
         "n": 75,
         "correct": 75,
         "accuracy": 1,
         "lo": 1,
         "hi": 1,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-rconc",
         "n": 342,
         "correct": 316,
         "accuracy": 0.9239766081871345,
         "lo": 0.8976608187134503,
         "hi": 0.9502923976608187,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "proof",
         "n": 483,
         "correct": 416,
         "accuracy": 0.8612836438923396,
         "lo": 0.8302277432712215,
         "hi": 0.8902691511387164,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "random",
         "n": 75,
         "correct": 74,
         "accuracy": 0.9866666666666667,
         "lo": 0.96,
         "hi": 1,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "rconc",
         "n": 324,
         "correct": 300,
         "accuracy": 0.9259259259259259,
         "lo": 0.8981481481481481,
         "hi": 0.9537037037037037,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "paraphrased": [
        {
         "value": "False",
         "n": 1306,
         "correct": 1228,
         "accuracy": 0.9402756508422665,
         "lo": 0.9280245022970903,
         "hi": 0.9532924961715161,
         "best_constant": 0.5022970903522205,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 494,
         "correct": 379,
         "accuracy": 0.7672064777327935,
         "lo": 0.7307692307692307,
         "hi": 0.8036437246963563,
         "best_constant": 0.5060728744939271,
         "invalid": 0
        }
       ],
       "theory_max_depth": [
        {
         "value": "0",
         "n": 71,
         "correct": 69,
         "accuracy": 0.971830985915493,
         "lo": 0.9295774647887324,
         "hi": 1,
         "best_constant": 0.7605633802816901,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 155,
         "correct": 146,
         "accuracy": 0.9419354838709677,
         "lo": 0.9032258064516129,
         "hi": 0.9806451612903225,
         "best_constant": 0.5096774193548387,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 230,
         "correct": 219,
         "accuracy": 0.9521739130434783,
         "lo": 0.9217391304347826,
         "hi": 0.9782608695652174,
         "best_constant": 0.5260869565217391,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 619,
         "correct": 549,
         "accuracy": 0.8869143780290791,
         "lo": 0.8626817447495961,
         "hi": 0.9127625201938611,
         "best_constant": 0.5218093699515347,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 171,
         "correct": 111,
         "accuracy": 0.6491228070175439,
         "lo": 0.5789473684210527,
         "hi": 0.7134502923976608,
         "best_constant": 0.5087719298245614,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 513,
         "correct": 480,
         "accuracy": 0.935672514619883,
         "lo": 0.9122807017543859,
         "hi": 0.9551656920077972,
         "best_constant": 0.5243664717348928,
         "invalid": 0
        },
        {
         "value": "6",
         "n": 38,
         "correct": 30,
         "accuracy": 0.7894736842105263,
         "lo": 0.6578947368421053,
         "hi": 0.9210526315789473,
         "best_constant": 0.5789473684210527,
         "invalid": 0
        },
        {
         "value": "7",
         "n": 2,
         "correct": 2,
         "accuracy": 1,
         "lo": 1,
         "hi": 1,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "8",
         "n": 1,
         "correct": 1,
         "accuracy": 1,
         "lo": 1,
         "hi": 1,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "words_bin": [
        {
         "value": "0-49",
         "n": 256,
         "correct": 254,
         "accuracy": 0.9921875,
         "lo": 0.98046875,
         "hi": 1,
         "best_constant": 0.5078125,
         "invalid": 0
        },
        {
         "value": "50-79",
         "n": 432,
         "correct": 410,
         "accuracy": 0.9490740740740741,
         "lo": 0.9282407407407407,
         "hi": 0.9675925925925926,
         "best_constant": 0.5023148148148148,
         "invalid": 0
        },
        {
         "value": "80-109",
         "n": 482,
         "correct": 413,
         "accuracy": 0.8568464730290456,
         "lo": 0.8236514522821576,
         "hi": 0.8879668049792531,
         "best_constant": 0.508298755186722,
         "invalid": 0
        },
        {
         "value": "110+",
         "n": 630,
         "correct": 530,
         "accuracy": 0.8412698412698413,
         "lo": 0.8126984126984127,
         "hi": 0.8698412698412699,
         "best_constant": 0.5047619047619047,
         "invalid": 0
        }
       ],
       "rules_bin": [
        {
         "value": "0-2",
         "n": 208,
         "correct": 205,
         "accuracy": 0.9855769230769231,
         "lo": 0.9663461538461539,
         "hi": 1,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "3-5",
         "n": 405,
         "correct": 348,
         "accuracy": 0.8592592592592593,
         "lo": 0.8271604938271605,
         "hi": 0.8938271604938272,
         "best_constant": 0.5037037037037037,
         "invalid": 0
        },
        {
         "value": "6-7",
         "n": 605,
         "correct": 513,
         "accuracy": 0.8479338842975207,
         "lo": 0.8198347107438017,
         "hi": 0.8760330578512396,
         "best_constant": 0.5057851239669422,
         "invalid": 0
        },
        {
         "value": "8+",
         "n": 582,
         "correct": 541,
         "accuracy": 0.929553264604811,
         "lo": 0.9072164948453608,
         "hi": 0.9501718213058419,
         "best_constant": 0.5034364261168385,
         "invalid": 0
        }
       ],
       "facts_bin": [
        {
         "value": "0-3",
         "n": 360,
         "correct": 344,
         "accuracy": 0.9555555555555556,
         "lo": 0.9333333333333333,
         "hi": 0.975,
         "best_constant": 0.5194444444444445,
         "invalid": 0
        },
        {
         "value": "4-7",
         "n": 425,
         "correct": 393,
         "accuracy": 0.9247058823529412,
         "lo": 0.9011764705882352,
         "hi": 0.9482352941176471,
         "best_constant": 0.5129411764705882,
         "invalid": 0
        },
        {
         "value": "8-12",
         "n": 812,
         "correct": 685,
         "accuracy": 0.8435960591133005,
         "lo": 0.8189655172413793,
         "hi": 0.8694581280788177,
         "best_constant": 0.5012315270935961,
         "invalid": 0
        },
        {
         "value": "13+",
         "n": 203,
         "correct": 185,
         "accuracy": 0.9113300492610837,
         "lo": 0.8719211822660099,
         "hi": 0.9507389162561576,
         "best_constant": 0.5024630541871922,
         "invalid": 0
        }
       ],
       "proof_size_bin": [
        {
         "value": "0-1",
         "n": 966,
         "correct": 914,
         "accuracy": 0.9461697722567288,
         "lo": 0.9316770186335404,
         "hi": 0.9596273291925466,
         "best_constant": 0.5093167701863354,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 82,
         "correct": 80,
         "accuracy": 0.975609756097561,
         "lo": 0.9390243902439024,
         "hi": 1,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "3-4",
         "n": 237,
         "correct": 216,
         "accuracy": 0.9113924050632911,
         "lo": 0.8776371308016878,
         "hi": 0.9451476793248945,
         "best_constant": 0.5021097046413502,
         "invalid": 0
        },
        {
         "value": "5+",
         "n": 515,
         "correct": 397,
         "accuracy": 0.7708737864077669,
         "lo": 0.7339805825242719,
         "hi": 0.8058252427184466,
         "best_constant": 0.5184466019417475,
         "invalid": 0
        }
       ]
      }
     },
     "kev-0.8b": {
      "file": "studies/proofwriter-cwa-kev-0.8b.jsonl",
      "modified": "2026-10-01T16:36:18.582Z",
      "overall": {
       "n": 1800,
       "correct": 1004,
       "accuracy": 0.5577777777777778,
       "lo": 0.535,
       "hi": 0.5794444444444444,
       "macro_f1": 0.5530048630375454,
       "invalid": 0,
       "best_constant": 0.5,
       "chance": 0.5,
       "recall": {
        "false": 0.6611111111111111,
        "true": 0.45444444444444443
       },
       "confusion": {
        "false": {
         "false": 595,
         "invalid": 0,
         "true": 305
        },
        "true": {
         "false": 491,
         "invalid": 0,
         "true": 409
        }
       },
       "model": [
        "kev-latest"
       ]
      },
      "axes": {
       "depth": [
        {
         "value": "0",
         "n": 300,
         "correct": 204,
         "accuracy": 0.68,
         "lo": 0.6266666666666667,
         "hi": 0.7333333333333333,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 300,
         "correct": 179,
         "accuracy": 0.5966666666666667,
         "lo": 0.54,
         "hi": 0.6566666666666666,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 300,
         "correct": 154,
         "accuracy": 0.5133333333333333,
         "lo": 0.45666666666666667,
         "hi": 0.57,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 300,
         "correct": 159,
         "accuracy": 0.53,
         "lo": 0.4766666666666667,
         "hi": 0.5866666666666667,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 300,
         "correct": 153,
         "accuracy": 0.51,
         "lo": 0.45666666666666667,
         "hi": 0.57,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 300,
         "correct": 155,
         "accuracy": 0.5166666666666667,
         "lo": 0.46,
         "hi": 0.5733333333333334,
         "best_constant": 0.5,
         "invalid": 0
        }
       ],
       "reference_label": [
        {
         "value": "false",
         "n": 900,
         "correct": 595,
         "accuracy": 0.6611111111111111,
         "lo": 0.63,
         "hi": 0.6933333333333334,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "true",
         "n": 900,
         "correct": 409,
         "accuracy": 0.45444444444444443,
         "lo": 0.4211111111111111,
         "hi": 0.4855555555555556,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "theory_kind": [
        {
         "value": "attribute",
         "n": 1016,
         "correct": 531,
         "accuracy": 0.5226377952755905,
         "lo": 0.49311023622047245,
         "hi": 0.5551181102362205,
         "best_constant": 0.5019685039370079,
         "invalid": 0
        },
        {
         "value": "relation",
         "n": 784,
         "correct": 473,
         "accuracy": 0.6033163265306123,
         "lo": 0.5688775510204082,
         "hi": 0.6364795918367347,
         "best_constant": 0.5025510204081632,
         "invalid": 0
        }
       ],
       "theory_negation": [
        {
         "value": "negation",
         "n": 870,
         "correct": 499,
         "accuracy": 0.5735632183908046,
         "lo": 0.5413793103448276,
         "hi": 0.6068965517241379,
         "best_constant": 0.5011494252873563,
         "invalid": 0
        },
        {
         "value": "no-negation",
         "n": 930,
         "correct": 505,
         "accuracy": 0.543010752688172,
         "lo": 0.5096774193548387,
         "hi": 0.5741935483870968,
         "best_constant": 0.5010752688172043,
         "invalid": 0
        }
       ],
       "statement_negated": [
        {
         "value": "False",
         "n": 882,
         "correct": 473,
         "accuracy": 0.536281179138322,
         "lo": 0.5034013605442177,
         "hi": 0.5691609977324263,
         "best_constant": 0.5476190476190477,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 918,
         "correct": 531,
         "accuracy": 0.5784313725490197,
         "lo": 0.5468409586056645,
         "hi": 0.6089324618736384,
         "best_constant": 0.545751633986928,
         "invalid": 0
        }
       ],
       "strategy": [
        {
         "value": "inv-proof",
         "n": 501,
         "correct": 408,
         "accuracy": 0.8143712574850299,
         "lo": 0.782435129740519,
         "hi": 0.846307385229541,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-random",
         "n": 75,
         "correct": 13,
         "accuracy": 0.17333333333333334,
         "lo": 0.09333333333333334,
         "hi": 0.26666666666666666,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-rconc",
         "n": 342,
         "correct": 110,
         "accuracy": 0.3216374269005848,
         "lo": 0.26900584795321636,
         "hi": 0.3684210526315789,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "proof",
         "n": 483,
         "correct": 286,
         "accuracy": 0.5921325051759835,
         "lo": 0.546583850931677,
         "hi": 0.639751552795031,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "random",
         "n": 75,
         "correct": 47,
         "accuracy": 0.6266666666666667,
         "lo": 0.52,
         "hi": 0.72,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "rconc",
         "n": 324,
         "correct": 140,
         "accuracy": 0.43209876543209874,
         "lo": 0.37962962962962965,
         "hi": 0.4876543209876543,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "paraphrased": [
        {
         "value": "False",
         "n": 1306,
         "correct": 739,
         "accuracy": 0.5658499234303216,
         "lo": 0.5375191424196019,
         "hi": 0.5926493108728943,
         "best_constant": 0.5022970903522205,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 494,
         "correct": 265,
         "accuracy": 0.5364372469635628,
         "lo": 0.4959514170040486,
         "hi": 0.5809716599190283,
         "best_constant": 0.5060728744939271,
         "invalid": 0
        }
       ],
       "theory_max_depth": [
        {
         "value": "0",
         "n": 71,
         "correct": 39,
         "accuracy": 0.5492957746478874,
         "lo": 0.4225352112676056,
         "hi": 0.6619718309859155,
         "best_constant": 0.7605633802816901,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 155,
         "correct": 91,
         "accuracy": 0.5870967741935483,
         "lo": 0.5096774193548387,
         "hi": 0.6709677419354839,
         "best_constant": 0.5096774193548387,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 230,
         "correct": 124,
         "accuracy": 0.5391304347826087,
         "lo": 0.46956521739130436,
         "hi": 0.6086956521739131,
         "best_constant": 0.5260869565217391,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 619,
         "correct": 338,
         "accuracy": 0.5460420032310178,
         "lo": 0.505654281098546,
         "hi": 0.5848142164781907,
         "best_constant": 0.5218093699515347,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 171,
         "correct": 95,
         "accuracy": 0.5555555555555556,
         "lo": 0.47953216374269003,
         "hi": 0.631578947368421,
         "best_constant": 0.5087719298245614,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 513,
         "correct": 292,
         "accuracy": 0.5692007797270955,
         "lo": 0.5282651072124757,
         "hi": 0.6120857699805068,
         "best_constant": 0.5243664717348928,
         "invalid": 0
        },
        {
         "value": "6",
         "n": 38,
         "correct": 24,
         "accuracy": 0.631578947368421,
         "lo": 0.47368421052631576,
         "hi": 0.7894736842105263,
         "best_constant": 0.5789473684210527,
         "invalid": 0
        },
        {
         "value": "7",
         "n": 2,
         "correct": 1,
         "accuracy": 0.5,
         "lo": 0,
         "hi": 1,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "8",
         "n": 1,
         "correct": 0,
         "accuracy": 0,
         "lo": 0,
         "hi": 0,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "words_bin": [
        {
         "value": "0-49",
         "n": 256,
         "correct": 165,
         "accuracy": 0.64453125,
         "lo": 0.5859375,
         "hi": 0.6953125,
         "best_constant": 0.5078125,
         "invalid": 0
        },
        {
         "value": "50-79",
         "n": 432,
         "correct": 235,
         "accuracy": 0.5439814814814815,
         "lo": 0.4976851851851852,
         "hi": 0.5925925925925926,
         "best_constant": 0.5023148148148148,
         "invalid": 0
        },
        {
         "value": "80-109",
         "n": 482,
         "correct": 268,
         "accuracy": 0.5560165975103735,
         "lo": 0.5124481327800829,
         "hi": 0.6016597510373444,
         "best_constant": 0.508298755186722,
         "invalid": 0
        },
        {
         "value": "110+",
         "n": 630,
         "correct": 336,
         "accuracy": 0.5333333333333333,
         "lo": 0.49523809523809526,
         "hi": 0.5714285714285714,
         "best_constant": 0.5047619047619047,
         "invalid": 0
        }
       ],
       "rules_bin": [
        {
         "value": "0-2",
         "n": 208,
         "correct": 140,
         "accuracy": 0.6730769230769231,
         "lo": 0.6105769230769231,
         "hi": 0.7355769230769231,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "3-5",
         "n": 405,
         "correct": 252,
         "accuracy": 0.6222222222222222,
         "lo": 0.5728395061728395,
         "hi": 0.6641975308641975,
         "best_constant": 0.5037037037037037,
         "invalid": 0
        },
        {
         "value": "6-7",
         "n": 605,
         "correct": 306,
         "accuracy": 0.5057851239669422,
         "lo": 0.46611570247933887,
         "hi": 0.547107438016529,
         "best_constant": 0.5057851239669422,
         "invalid": 0
        },
        {
         "value": "8+",
         "n": 582,
         "correct": 306,
         "accuracy": 0.5257731958762887,
         "lo": 0.48281786941580757,
         "hi": 0.563573883161512,
         "best_constant": 0.5034364261168385,
         "invalid": 0
        }
       ],
       "facts_bin": [
        {
         "value": "0-3",
         "n": 360,
         "correct": 211,
         "accuracy": 0.5861111111111111,
         "lo": 0.5361111111111111,
         "hi": 0.6388888888888888,
         "best_constant": 0.5194444444444445,
         "invalid": 0
        },
        {
         "value": "4-7",
         "n": 425,
         "correct": 234,
         "accuracy": 0.5505882352941176,
         "lo": 0.5011764705882353,
         "hi": 0.5952941176470589,
         "best_constant": 0.5129411764705882,
         "invalid": 0
        },
        {
         "value": "8-12",
         "n": 812,
         "correct": 439,
         "accuracy": 0.5406403940886699,
         "lo": 0.5086206896551724,
         "hi": 0.5738916256157636,
         "best_constant": 0.5012315270935961,
         "invalid": 0
        },
        {
         "value": "13+",
         "n": 203,
         "correct": 120,
         "accuracy": 0.5911330049261084,
         "lo": 0.5172413793103449,
         "hi": 0.6551724137931034,
         "best_constant": 0.5024630541871922,
         "invalid": 0
        }
       ],
       "proof_size_bin": [
        {
         "value": "0-1",
         "n": 966,
         "correct": 454,
         "accuracy": 0.4699792960662526,
         "lo": 0.4409937888198758,
         "hi": 0.5020703933747412,
         "best_constant": 0.5093167701863354,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 82,
         "correct": 70,
         "accuracy": 0.8536585365853658,
         "lo": 0.7804878048780488,
         "hi": 0.926829268292683,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "3-4",
         "n": 237,
         "correct": 168,
         "accuracy": 0.7088607594936709,
         "lo": 0.6455696202531646,
         "hi": 0.7637130801687764,
         "best_constant": 0.5021097046413502,
         "invalid": 0
        },
        {
         "value": "5+",
         "n": 515,
         "correct": 312,
         "accuracy": 0.6058252427184466,
         "lo": 0.5611650485436893,
         "hi": 0.6427184466019418,
         "best_constant": 0.5184466019417475,
         "invalid": 0
        }
       ]
      }
     },
     "kev-4b": {
      "file": "studies/proofwriter-cwa-kev-4b.jsonl",
      "modified": "2026-10-01T16:36:19.906Z",
      "overall": {
       "n": 1800,
       "correct": 1058,
       "accuracy": 0.5877777777777777,
       "lo": 0.5655555555555556,
       "hi": 0.6105555555555555,
       "macro_f1": 0.5761609242966796,
       "invalid": 0,
       "best_constant": 0.5,
       "chance": 0.5,
       "recall": {
        "false": 0.7533333333333333,
        "true": 0.4222222222222222
       },
       "confusion": {
        "false": {
         "false": 678,
         "invalid": 0,
         "true": 222
        },
        "true": {
         "false": 520,
         "invalid": 0,
         "true": 380
        }
       },
       "model": [
        "kev-latest"
       ]
      },
      "axes": {
       "depth": [
        {
         "value": "0",
         "n": 300,
         "correct": 257,
         "accuracy": 0.8566666666666667,
         "lo": 0.8166666666666667,
         "hi": 0.8966666666666666,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 300,
         "correct": 205,
         "accuracy": 0.6833333333333333,
         "lo": 0.63,
         "hi": 0.7366666666666667,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 300,
         "correct": 163,
         "accuracy": 0.5433333333333333,
         "lo": 0.49,
         "hi": 0.5966666666666667,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 300,
         "correct": 168,
         "accuracy": 0.56,
         "lo": 0.5066666666666667,
         "hi": 0.6133333333333333,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 300,
         "correct": 135,
         "accuracy": 0.45,
         "lo": 0.39,
         "hi": 0.5033333333333333,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 300,
         "correct": 130,
         "accuracy": 0.43333333333333335,
         "lo": 0.38333333333333336,
         "hi": 0.49,
         "best_constant": 0.5,
         "invalid": 0
        }
       ],
       "reference_label": [
        {
         "value": "false",
         "n": 900,
         "correct": 678,
         "accuracy": 0.7533333333333333,
         "lo": 0.7277777777777777,
         "hi": 0.7822222222222223,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "true",
         "n": 900,
         "correct": 380,
         "accuracy": 0.4222222222222222,
         "lo": 0.39222222222222225,
         "hi": 0.45444444444444443,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "theory_kind": [
        {
         "value": "attribute",
         "n": 1016,
         "correct": 597,
         "accuracy": 0.5875984251968503,
         "lo": 0.5600393700787402,
         "hi": 0.6190944881889764,
         "best_constant": 0.5019685039370079,
         "invalid": 0
        },
        {
         "value": "relation",
         "n": 784,
         "correct": 461,
         "accuracy": 0.5880102040816326,
         "lo": 0.5548469387755102,
         "hi": 0.6237244897959183,
         "best_constant": 0.5025510204081632,
         "invalid": 0
        }
       ],
       "theory_negation": [
        {
         "value": "negation",
         "n": 870,
         "correct": 534,
         "accuracy": 0.6137931034482759,
         "lo": 0.5793103448275863,
         "hi": 0.6459770114942529,
         "best_constant": 0.5011494252873563,
         "invalid": 0
        },
        {
         "value": "no-negation",
         "n": 930,
         "correct": 524,
         "accuracy": 0.5634408602150538,
         "lo": 0.5301075268817205,
         "hi": 0.5924731182795699,
         "best_constant": 0.5010752688172043,
         "invalid": 0
        }
       ],
       "statement_negated": [
        {
         "value": "False",
         "n": 882,
         "correct": 517,
         "accuracy": 0.5861678004535147,
         "lo": 0.5532879818594104,
         "hi": 0.6167800453514739,
         "best_constant": 0.5476190476190477,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 918,
         "correct": 541,
         "accuracy": 0.5893246187363834,
         "lo": 0.5566448801742919,
         "hi": 0.6220043572984749,
         "best_constant": 0.545751633986928,
         "invalid": 0
        }
       ],
       "strategy": [
        {
         "value": "inv-proof",
         "n": 501,
         "correct": 330,
         "accuracy": 0.6586826347305389,
         "lo": 0.6187624750499002,
         "hi": 0.6986027944111777,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-random",
         "n": 75,
         "correct": 45,
         "accuracy": 0.6,
         "lo": 0.49333333333333335,
         "hi": 0.7066666666666667,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-rconc",
         "n": 342,
         "correct": 166,
         "accuracy": 0.4853801169590643,
         "lo": 0.4298245614035088,
         "hi": 0.5350877192982456,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "proof",
         "n": 483,
         "correct": 169,
         "accuracy": 0.3498964803312629,
         "lo": 0.3084886128364389,
         "hi": 0.391304347826087,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "random",
         "n": 75,
         "correct": 70,
         "accuracy": 0.9333333333333333,
         "lo": 0.88,
         "hi": 0.9866666666666667,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "rconc",
         "n": 324,
         "correct": 278,
         "accuracy": 0.8580246913580247,
         "lo": 0.8179012345679012,
         "hi": 0.8950617283950617,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "paraphrased": [
        {
         "value": "False",
         "n": 1306,
         "correct": 786,
         "accuracy": 0.6018376722817764,
         "lo": 0.5765696784073507,
         "hi": 0.6271056661562021,
         "best_constant": 0.5022970903522205,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 494,
         "correct": 272,
         "accuracy": 0.5506072874493927,
         "lo": 0.5040485829959515,
         "hi": 0.5951417004048583,
         "best_constant": 0.5060728744939271,
         "invalid": 0
        }
       ],
       "theory_max_depth": [
        {
         "value": "0",
         "n": 71,
         "correct": 43,
         "accuracy": 0.6056338028169014,
         "lo": 0.49295774647887325,
         "hi": 0.7183098591549296,
         "best_constant": 0.7605633802816901,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 155,
         "correct": 106,
         "accuracy": 0.6838709677419355,
         "lo": 0.6129032258064516,
         "hi": 0.7548387096774194,
         "best_constant": 0.5096774193548387,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 230,
         "correct": 156,
         "accuracy": 0.6782608695652174,
         "lo": 0.6173913043478261,
         "hi": 0.7391304347826086,
         "best_constant": 0.5260869565217391,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 619,
         "correct": 394,
         "accuracy": 0.6365105008077544,
         "lo": 0.5993537964458805,
         "hi": 0.6768982229402262,
         "best_constant": 0.5218093699515347,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 171,
         "correct": 83,
         "accuracy": 0.4853801169590643,
         "lo": 0.4152046783625731,
         "hi": 0.5555555555555556,
         "best_constant": 0.5087719298245614,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 513,
         "correct": 255,
         "accuracy": 0.49707602339181284,
         "lo": 0.45614035087719296,
         "hi": 0.5399610136452242,
         "best_constant": 0.5243664717348928,
         "invalid": 0
        },
        {
         "value": "6",
         "n": 38,
         "correct": 20,
         "accuracy": 0.5263157894736842,
         "lo": 0.3684210526315789,
         "hi": 0.6842105263157895,
         "best_constant": 0.5789473684210527,
         "invalid": 0
        },
        {
         "value": "7",
         "n": 2,
         "correct": 1,
         "accuracy": 0.5,
         "lo": 0,
         "hi": 1,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "8",
         "n": 1,
         "correct": 0,
         "accuracy": 0,
         "lo": 0,
         "hi": 0,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "words_bin": [
        {
         "value": "0-49",
         "n": 256,
         "correct": 182,
         "accuracy": 0.7109375,
         "lo": 0.65625,
         "hi": 0.765625,
         "best_constant": 0.5078125,
         "invalid": 0
        },
        {
         "value": "50-79",
         "n": 432,
         "correct": 261,
         "accuracy": 0.6041666666666666,
         "lo": 0.5555555555555556,
         "hi": 0.6504629629629629,
         "best_constant": 0.5023148148148148,
         "invalid": 0
        },
        {
         "value": "80-109",
         "n": 482,
         "correct": 285,
         "accuracy": 0.5912863070539419,
         "lo": 0.549792531120332,
         "hi": 0.6327800829875518,
         "best_constant": 0.508298755186722,
         "invalid": 0
        },
        {
         "value": "110+",
         "n": 630,
         "correct": 330,
         "accuracy": 0.5238095238095238,
         "lo": 0.48412698412698413,
         "hi": 0.5619047619047619,
         "best_constant": 0.5047619047619047,
         "invalid": 0
        }
       ],
       "rules_bin": [
        {
         "value": "0-2",
         "n": 208,
         "correct": 143,
         "accuracy": 0.6875,
         "lo": 0.6298076923076923,
         "hi": 0.7548076923076923,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "3-5",
         "n": 405,
         "correct": 232,
         "accuracy": 0.5728395061728395,
         "lo": 0.5259259259259259,
         "hi": 0.6197530864197531,
         "best_constant": 0.5037037037037037,
         "invalid": 0
        },
        {
         "value": "6-7",
         "n": 605,
         "correct": 312,
         "accuracy": 0.515702479338843,
         "lo": 0.47768595041322315,
         "hi": 0.5553719008264463,
         "best_constant": 0.5057851239669422,
         "invalid": 0
        },
        {
         "value": "8+",
         "n": 582,
         "correct": 371,
         "accuracy": 0.6374570446735395,
         "lo": 0.5979381443298969,
         "hi": 0.6752577319587629,
         "best_constant": 0.5034364261168385,
         "invalid": 0
        }
       ],
       "facts_bin": [
        {
         "value": "0-3",
         "n": 360,
         "correct": 245,
         "accuracy": 0.6805555555555556,
         "lo": 0.6333333333333333,
         "hi": 0.7333333333333333,
         "best_constant": 0.5194444444444445,
         "invalid": 0
        },
        {
         "value": "4-7",
         "n": 425,
         "correct": 271,
         "accuracy": 0.6376470588235295,
         "lo": 0.5905882352941176,
         "hi": 0.6847058823529412,
         "best_constant": 0.5129411764705882,
         "invalid": 0
        },
        {
         "value": "8-12",
         "n": 812,
         "correct": 448,
         "accuracy": 0.5517241379310345,
         "lo": 0.5184729064039408,
         "hi": 0.5862068965517241,
         "best_constant": 0.5012315270935961,
         "invalid": 0
        },
        {
         "value": "13+",
         "n": 203,
         "correct": 94,
         "accuracy": 0.4630541871921182,
         "lo": 0.39408866995073893,
         "hi": 0.5320197044334976,
         "best_constant": 0.5024630541871922,
         "invalid": 0
        }
       ],
       "proof_size_bin": [
        {
         "value": "0-1",
         "n": 966,
         "correct": 701,
         "accuracy": 0.7256728778467909,
         "lo": 0.6977225672877847,
         "hi": 0.7546583850931677,
         "best_constant": 0.5093167701863354,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 82,
         "correct": 65,
         "accuracy": 0.7926829268292683,
         "lo": 0.7073170731707317,
         "hi": 0.8780487804878049,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "3-4",
         "n": 237,
         "correct": 123,
         "accuracy": 0.5189873417721519,
         "lo": 0.45147679324894513,
         "hi": 0.5822784810126582,
         "best_constant": 0.5021097046413502,
         "invalid": 0
        },
        {
         "value": "5+",
         "n": 515,
         "correct": 169,
         "accuracy": 0.32815533980582523,
         "lo": 0.287378640776699,
         "hi": 0.37087378640776697,
         "best_constant": 0.5184466019417475,
         "invalid": 0
        }
       ]
      }
     },
     "kev-9b": {
      "file": "studies/proofwriter-cwa-kev-9b.jsonl",
      "modified": "2026-10-01T16:36:21.234Z",
      "overall": {
       "n": 1800,
       "correct": 1148,
       "accuracy": 0.6377777777777778,
       "lo": 0.615,
       "hi": 0.6611111111111111,
       "macro_f1": 0.6377491554888288,
       "invalid": 0,
       "best_constant": 0.5,
       "chance": 0.5,
       "recall": {
        "false": 0.6288888888888889,
        "true": 0.6466666666666666
       },
       "confusion": {
        "false": {
         "false": 566,
         "invalid": 0,
         "true": 334
        },
        "true": {
         "false": 318,
         "invalid": 0,
         "true": 582
        }
       },
       "model": [
        "kev-latest"
       ]
      },
      "axes": {
       "depth": [
        {
         "value": "0",
         "n": 300,
         "correct": 269,
         "accuracy": 0.8966666666666666,
         "lo": 0.86,
         "hi": 0.93,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 300,
         "correct": 231,
         "accuracy": 0.77,
         "lo": 0.7266666666666667,
         "hi": 0.8166666666666667,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 300,
         "correct": 192,
         "accuracy": 0.64,
         "lo": 0.59,
         "hi": 0.6933333333333334,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 300,
         "correct": 176,
         "accuracy": 0.5866666666666667,
         "lo": 0.53,
         "hi": 0.6433333333333333,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 300,
         "correct": 151,
         "accuracy": 0.5033333333333333,
         "lo": 0.44666666666666666,
         "hi": 0.5566666666666666,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 300,
         "correct": 129,
         "accuracy": 0.43,
         "lo": 0.37333333333333335,
         "hi": 0.4866666666666667,
         "best_constant": 0.5,
         "invalid": 0
        }
       ],
       "reference_label": [
        {
         "value": "false",
         "n": 900,
         "correct": 566,
         "accuracy": 0.6288888888888889,
         "lo": 0.5988888888888889,
         "hi": 0.6588888888888889,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "true",
         "n": 900,
         "correct": 582,
         "accuracy": 0.6466666666666666,
         "lo": 0.6166666666666667,
         "hi": 0.6755555555555556,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "theory_kind": [
        {
         "value": "attribute",
         "n": 1016,
         "correct": 666,
         "accuracy": 0.655511811023622,
         "lo": 0.6279527559055118,
         "hi": 0.6850393700787402,
         "best_constant": 0.5019685039370079,
         "invalid": 0
        },
        {
         "value": "relation",
         "n": 784,
         "correct": 482,
         "accuracy": 0.6147959183673469,
         "lo": 0.5816326530612245,
         "hi": 0.6492346938775511,
         "best_constant": 0.5025510204081632,
         "invalid": 0
        }
       ],
       "theory_negation": [
        {
         "value": "negation",
         "n": 870,
         "correct": 578,
         "accuracy": 0.664367816091954,
         "lo": 0.6333333333333333,
         "hi": 0.696551724137931,
         "best_constant": 0.5011494252873563,
         "invalid": 0
        },
        {
         "value": "no-negation",
         "n": 930,
         "correct": 570,
         "accuracy": 0.6129032258064516,
         "lo": 0.5827956989247312,
         "hi": 0.646236559139785,
         "best_constant": 0.5010752688172043,
         "invalid": 0
        }
       ],
       "statement_negated": [
        {
         "value": "False",
         "n": 882,
         "correct": 579,
         "accuracy": 0.6564625850340136,
         "lo": 0.6258503401360545,
         "hi": 0.6870748299319728,
         "best_constant": 0.5476190476190477,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 918,
         "correct": 569,
         "accuracy": 0.6198257080610022,
         "lo": 0.5893246187363834,
         "hi": 0.6503267973856209,
         "best_constant": 0.545751633986928,
         "invalid": 0
        }
       ],
       "strategy": [
        {
         "value": "inv-proof",
         "n": 501,
         "correct": 250,
         "accuracy": 0.499001996007984,
         "lo": 0.4550898203592814,
         "hi": 0.5429141716566867,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-random",
         "n": 75,
         "correct": 52,
         "accuracy": 0.6933333333333334,
         "lo": 0.5866666666666667,
         "hi": 0.8,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-rconc",
         "n": 342,
         "correct": 267,
         "accuracy": 0.7807017543859649,
         "lo": 0.7339181286549707,
         "hi": 0.8245614035087719,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "proof",
         "n": 483,
         "correct": 263,
         "accuracy": 0.5445134575569358,
         "lo": 0.4968944099378882,
         "hi": 0.5900621118012422,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "random",
         "n": 75,
         "correct": 70,
         "accuracy": 0.9333333333333333,
         "lo": 0.88,
         "hi": 0.9866666666666667,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "rconc",
         "n": 324,
         "correct": 246,
         "accuracy": 0.7592592592592593,
         "lo": 0.7129629629629629,
         "hi": 0.8055555555555556,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "paraphrased": [
        {
         "value": "False",
         "n": 1306,
         "correct": 846,
         "accuracy": 0.6477794793261868,
         "lo": 0.6202143950995406,
         "hi": 0.6745788667687596,
         "best_constant": 0.5022970903522205,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 494,
         "correct": 302,
         "accuracy": 0.611336032388664,
         "lo": 0.5668016194331984,
         "hi": 0.6538461538461539,
         "best_constant": 0.5060728744939271,
         "invalid": 0
        }
       ],
       "theory_max_depth": [
        {
         "value": "0",
         "n": 71,
         "correct": 56,
         "accuracy": 0.7887323943661971,
         "lo": 0.6901408450704225,
         "hi": 0.8732394366197183,
         "best_constant": 0.7605633802816901,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 155,
         "correct": 118,
         "accuracy": 0.7612903225806451,
         "lo": 0.6903225806451613,
         "hi": 0.8258064516129032,
         "best_constant": 0.5096774193548387,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 230,
         "correct": 177,
         "accuracy": 0.7695652173913043,
         "lo": 0.7130434782608696,
         "hi": 0.8217391304347826,
         "best_constant": 0.5260869565217391,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 619,
         "correct": 426,
         "accuracy": 0.6882067851373183,
         "lo": 0.6510500807754442,
         "hi": 0.7269789983844911,
         "best_constant": 0.5218093699515347,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 171,
         "correct": 87,
         "accuracy": 0.5087719298245614,
         "lo": 0.4327485380116959,
         "hi": 0.5906432748538012,
         "best_constant": 0.5087719298245614,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 513,
         "correct": 263,
         "accuracy": 0.5126705653021443,
         "lo": 0.47173489278752434,
         "hi": 0.557504873294347,
         "best_constant": 0.5243664717348928,
         "invalid": 0
        },
        {
         "value": "6",
         "n": 38,
         "correct": 19,
         "accuracy": 0.5,
         "lo": 0.34210526315789475,
         "hi": 0.6578947368421053,
         "best_constant": 0.5789473684210527,
         "invalid": 0
        },
        {
         "value": "7",
         "n": 2,
         "correct": 1,
         "accuracy": 0.5,
         "lo": 0,
         "hi": 1,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "8",
         "n": 1,
         "correct": 1,
         "accuracy": 1,
         "lo": 1,
         "hi": 1,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "words_bin": [
        {
         "value": "0-49",
         "n": 256,
         "correct": 189,
         "accuracy": 0.73828125,
         "lo": 0.6875,
         "hi": 0.7890625,
         "best_constant": 0.5078125,
         "invalid": 0
        },
        {
         "value": "50-79",
         "n": 432,
         "correct": 292,
         "accuracy": 0.6759259259259259,
         "lo": 0.6342592592592593,
         "hi": 0.7222222222222222,
         "best_constant": 0.5023148148148148,
         "invalid": 0
        },
        {
         "value": "80-109",
         "n": 482,
         "correct": 317,
         "accuracy": 0.6576763485477178,
         "lo": 0.6161825726141079,
         "hi": 0.7012448132780082,
         "best_constant": 0.508298755186722,
         "invalid": 0
        },
        {
         "value": "110+",
         "n": 630,
         "correct": 350,
         "accuracy": 0.5555555555555556,
         "lo": 0.5158730158730159,
         "hi": 0.5920634920634921,
         "best_constant": 0.5047619047619047,
         "invalid": 0
        }
       ],
       "rules_bin": [
        {
         "value": "0-2",
         "n": 208,
         "correct": 155,
         "accuracy": 0.7451923076923077,
         "lo": 0.6875,
         "hi": 0.8028846153846154,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "3-5",
         "n": 405,
         "correct": 243,
         "accuracy": 0.6,
         "lo": 0.5530864197530864,
         "hi": 0.6493827160493827,
         "best_constant": 0.5037037037037037,
         "invalid": 0
        },
        {
         "value": "6-7",
         "n": 605,
         "correct": 331,
         "accuracy": 0.547107438016529,
         "lo": 0.5107438016528926,
         "hi": 0.5851239669421487,
         "best_constant": 0.5057851239669422,
         "invalid": 0
        },
        {
         "value": "8+",
         "n": 582,
         "correct": 419,
         "accuracy": 0.7199312714776632,
         "lo": 0.6821305841924399,
         "hi": 0.7560137457044673,
         "best_constant": 0.5034364261168385,
         "invalid": 0
        }
       ],
       "facts_bin": [
        {
         "value": "0-3",
         "n": 360,
         "correct": 262,
         "accuracy": 0.7277777777777777,
         "lo": 0.6833333333333333,
         "hi": 0.7777777777777778,
         "best_constant": 0.5194444444444445,
         "invalid": 0
        },
        {
         "value": "4-7",
         "n": 425,
         "correct": 299,
         "accuracy": 0.7035294117647058,
         "lo": 0.6588235294117647,
         "hi": 0.7458823529411764,
         "best_constant": 0.5129411764705882,
         "invalid": 0
        },
        {
         "value": "8-12",
         "n": 812,
         "correct": 483,
         "accuracy": 0.5948275862068966,
         "lo": 0.5603448275862069,
         "hi": 0.6305418719211823,
         "best_constant": 0.5012315270935961,
         "invalid": 0
        },
        {
         "value": "13+",
         "n": 203,
         "correct": 104,
         "accuracy": 0.5123152709359606,
         "lo": 0.4433497536945813,
         "hi": 0.5812807881773399,
         "best_constant": 0.5024630541871922,
         "invalid": 0
        }
       ],
       "proof_size_bin": [
        {
         "value": "0-1",
         "n": 966,
         "correct": 782,
         "accuracy": 0.8095238095238095,
         "lo": 0.7857142857142857,
         "hi": 0.8343685300207039,
         "best_constant": 0.5093167701863354,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 82,
         "correct": 65,
         "accuracy": 0.7926829268292683,
         "lo": 0.7073170731707317,
         "hi": 0.8780487804878049,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "3-4",
         "n": 237,
         "correct": 139,
         "accuracy": 0.5864978902953587,
         "lo": 0.5232067510548524,
         "hi": 0.6497890295358649,
         "best_constant": 0.5021097046413502,
         "invalid": 0
        },
        {
         "value": "5+",
         "n": 515,
         "correct": 162,
         "accuracy": 0.3145631067961165,
         "lo": 0.2757281553398058,
         "hi": 0.3553398058252427,
         "best_constant": 0.5184466019417475,
         "invalid": 0
        }
       ]
      }
     },
     "laya": {
      "file": "studies/proofwriter-cwa-laya.jsonl",
      "modified": "2026-10-01T16:36:22.599Z",
      "overall": {
       "n": 1800,
       "correct": 1001,
       "accuracy": 0.5561111111111111,
       "lo": 0.5327777777777778,
       "hi": 0.5805555555555556,
       "macro_f1": 0.509477400046044,
       "invalid": 0,
       "best_constant": 0.5,
       "chance": 0.5,
       "recall": {
        "false": 0.8644444444444445,
        "true": 0.2477777777777778
       },
       "confusion": {
        "false": {
         "false": 778,
         "invalid": 0,
         "true": 122
        },
        "true": {
         "false": 677,
         "invalid": 0,
         "true": 223
        }
       },
       "model": [
        "laya-upstream:0.3.21"
       ]
      },
      "axes": {
       "depth": [
        {
         "value": "0",
         "n": 300,
         "correct": 216,
         "accuracy": 0.72,
         "lo": 0.6666666666666666,
         "hi": 0.7666666666666667,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 300,
         "correct": 153,
         "accuracy": 0.51,
         "lo": 0.45,
         "hi": 0.56,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 300,
         "correct": 164,
         "accuracy": 0.5466666666666666,
         "lo": 0.4866666666666667,
         "hi": 0.5966666666666667,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 300,
         "correct": 154,
         "accuracy": 0.5133333333333333,
         "lo": 0.45666666666666667,
         "hi": 0.5666666666666667,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 300,
         "correct": 176,
         "accuracy": 0.5866666666666667,
         "lo": 0.53,
         "hi": 0.6433333333333333,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 300,
         "correct": 138,
         "accuracy": 0.46,
         "lo": 0.4033333333333333,
         "hi": 0.51,
         "best_constant": 0.5,
         "invalid": 0
        }
       ],
       "reference_label": [
        {
         "value": "false",
         "n": 900,
         "correct": 778,
         "accuracy": 0.8644444444444445,
         "lo": 0.8411111111111111,
         "hi": 0.8855555555555555,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "true",
         "n": 900,
         "correct": 223,
         "accuracy": 0.2477777777777778,
         "lo": 0.21888888888888888,
         "hi": 0.27555555555555555,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "theory_kind": [
        {
         "value": "attribute",
         "n": 1016,
         "correct": 557,
         "accuracy": 0.5482283464566929,
         "lo": 0.5187007874015748,
         "hi": 0.5787401574803149,
         "best_constant": 0.5019685039370079,
         "invalid": 0
        },
        {
         "value": "relation",
         "n": 784,
         "correct": 444,
         "accuracy": 0.5663265306122449,
         "lo": 0.5318877551020408,
         "hi": 0.5994897959183674,
         "best_constant": 0.5025510204081632,
         "invalid": 0
        }
       ],
       "theory_negation": [
        {
         "value": "negation",
         "n": 870,
         "correct": 499,
         "accuracy": 0.5735632183908046,
         "lo": 0.5413793103448276,
         "hi": 0.6045977011494252,
         "best_constant": 0.5011494252873563,
         "invalid": 0
        },
        {
         "value": "no-negation",
         "n": 930,
         "correct": 502,
         "accuracy": 0.5397849462365591,
         "lo": 0.5053763440860215,
         "hi": 0.5731182795698925,
         "best_constant": 0.5010752688172043,
         "invalid": 0
        }
       ],
       "statement_negated": [
        {
         "value": "False",
         "n": 882,
         "correct": 500,
         "accuracy": 0.5668934240362812,
         "lo": 0.5351473922902494,
         "hi": 0.5975056689342404,
         "best_constant": 0.5476190476190477,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 918,
         "correct": 501,
         "accuracy": 0.545751633986928,
         "lo": 0.5130718954248366,
         "hi": 0.5773420479302832,
         "best_constant": 0.545751633986928,
         "invalid": 0
        }
       ],
       "strategy": [
        {
         "value": "inv-proof",
         "n": 501,
         "correct": 423,
         "accuracy": 0.844311377245509,
         "lo": 0.810379241516966,
         "hi": 0.8762475049900199,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-random",
         "n": 75,
         "correct": 13,
         "accuracy": 0.17333333333333334,
         "lo": 0.09333333333333334,
         "hi": 0.26666666666666666,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-rconc",
         "n": 342,
         "correct": 65,
         "accuracy": 0.19005847953216373,
         "lo": 0.15204678362573099,
         "hi": 0.2309941520467836,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "proof",
         "n": 483,
         "correct": 145,
         "accuracy": 0.3002070393374741,
         "lo": 0.2587991718426501,
         "hi": 0.33954451345755693,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "random",
         "n": 75,
         "correct": 65,
         "accuracy": 0.8666666666666667,
         "lo": 0.7866666666666666,
         "hi": 0.9333333333333333,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "rconc",
         "n": 324,
         "correct": 290,
         "accuracy": 0.8950617283950617,
         "lo": 0.8580246913580247,
         "hi": 0.9259259259259259,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "paraphrased": [
        {
         "value": "False",
         "n": 1306,
         "correct": 733,
         "accuracy": 0.5612557427258805,
         "lo": 0.5344563552833078,
         "hi": 0.5888208269525268,
         "best_constant": 0.5022970903522205,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 494,
         "correct": 268,
         "accuracy": 0.5425101214574899,
         "lo": 0.4979757085020243,
         "hi": 0.5890688259109311,
         "best_constant": 0.5060728744939271,
         "invalid": 0
        }
       ],
       "theory_max_depth": [
        {
         "value": "0",
         "n": 71,
         "correct": 34,
         "accuracy": 0.4788732394366197,
         "lo": 0.352112676056338,
         "hi": 0.6056338028169014,
         "best_constant": 0.7605633802816901,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 155,
         "correct": 95,
         "accuracy": 0.6129032258064516,
         "lo": 0.535483870967742,
         "hi": 0.6838709677419355,
         "best_constant": 0.5096774193548387,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 230,
         "correct": 130,
         "accuracy": 0.5652173913043478,
         "lo": 0.5043478260869565,
         "hi": 0.6347826086956522,
         "best_constant": 0.5260869565217391,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 619,
         "correct": 345,
         "accuracy": 0.5573505654281099,
         "lo": 0.518578352180937,
         "hi": 0.5945072697899838,
         "best_constant": 0.5218093699515347,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 171,
         "correct": 98,
         "accuracy": 0.5730994152046783,
         "lo": 0.49707602339181284,
         "hi": 0.6432748538011696,
         "best_constant": 0.5087719298245614,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 513,
         "correct": 273,
         "accuracy": 0.5321637426900585,
         "lo": 0.49122807017543857,
         "hi": 0.5789473684210527,
         "best_constant": 0.5243664717348928,
         "invalid": 0
        },
        {
         "value": "6",
         "n": 38,
         "correct": 25,
         "accuracy": 0.6578947368421053,
         "lo": 0.5,
         "hi": 0.7894736842105263,
         "best_constant": 0.5789473684210527,
         "invalid": 0
        },
        {
         "value": "7",
         "n": 2,
         "correct": 1,
         "accuracy": 0.5,
         "lo": 0,
         "hi": 1,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "8",
         "n": 1,
         "correct": 0,
         "accuracy": 0,
         "lo": 0,
         "hi": 0,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "words_bin": [
        {
         "value": "0-49",
         "n": 256,
         "correct": 157,
         "accuracy": 0.61328125,
         "lo": 0.55859375,
         "hi": 0.671875,
         "best_constant": 0.5078125,
         "invalid": 0
        },
        {
         "value": "50-79",
         "n": 432,
         "correct": 241,
         "accuracy": 0.5578703703703703,
         "lo": 0.5069444444444444,
         "hi": 0.6041666666666666,
         "best_constant": 0.5023148148148148,
         "invalid": 0
        },
        {
         "value": "80-109",
         "n": 482,
         "correct": 266,
         "accuracy": 0.5518672199170125,
         "lo": 0.5103734439834025,
         "hi": 0.5975103734439834,
         "best_constant": 0.508298755186722,
         "invalid": 0
        },
        {
         "value": "110+",
         "n": 630,
         "correct": 337,
         "accuracy": 0.5349206349206349,
         "lo": 0.49682539682539684,
         "hi": 0.573015873015873,
         "best_constant": 0.5047619047619047,
         "invalid": 0
        }
       ],
       "rules_bin": [
        {
         "value": "0-2",
         "n": 208,
         "correct": 122,
         "accuracy": 0.5865384615384616,
         "lo": 0.5192307692307693,
         "hi": 0.6538461538461539,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "3-5",
         "n": 405,
         "correct": 225,
         "accuracy": 0.5555555555555556,
         "lo": 0.5111111111111111,
         "hi": 0.6049382716049383,
         "best_constant": 0.5037037037037037,
         "invalid": 0
        },
        {
         "value": "6-7",
         "n": 605,
         "correct": 332,
         "accuracy": 0.5487603305785124,
         "lo": 0.509090909090909,
         "hi": 0.5867768595041323,
         "best_constant": 0.5057851239669422,
         "invalid": 0
        },
        {
         "value": "8+",
         "n": 582,
         "correct": 322,
         "accuracy": 0.5532646048109966,
         "lo": 0.5120274914089347,
         "hi": 0.5945017182130584,
         "best_constant": 0.5034364261168385,
         "invalid": 0
        }
       ],
       "facts_bin": [
        {
         "value": "0-3",
         "n": 360,
         "correct": 211,
         "accuracy": 0.5861111111111111,
         "lo": 0.5361111111111111,
         "hi": 0.6361111111111111,
         "best_constant": 0.5194444444444445,
         "invalid": 0
        },
        {
         "value": "4-7",
         "n": 425,
         "correct": 251,
         "accuracy": 0.5905882352941176,
         "lo": 0.5458823529411765,
         "hi": 0.6329411764705882,
         "best_constant": 0.5129411764705882,
         "invalid": 0
        },
        {
         "value": "8-12",
         "n": 812,
         "correct": 428,
         "accuracy": 0.5270935960591133,
         "lo": 0.49137931034482757,
         "hi": 0.562807881773399,
         "best_constant": 0.5012315270935961,
         "invalid": 0
        },
        {
         "value": "13+",
         "n": 203,
         "correct": 111,
         "accuracy": 0.5467980295566502,
         "lo": 0.47783251231527096,
         "hi": 0.6157635467980296,
         "best_constant": 0.5024630541871922,
         "invalid": 0
        }
       ],
       "proof_size_bin": [
        {
         "value": "0-1",
         "n": 966,
         "correct": 571,
         "accuracy": 0.5910973084886129,
         "lo": 0.5590062111801242,
         "hi": 0.6231884057971014,
         "best_constant": 0.5093167701863354,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 82,
         "correct": 46,
         "accuracy": 0.5609756097560976,
         "lo": 0.45121951219512196,
         "hi": 0.6707317073170732,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "3-4",
         "n": 237,
         "correct": 124,
         "accuracy": 0.5232067510548524,
         "lo": 0.459915611814346,
         "hi": 0.5864978902953587,
         "best_constant": 0.5021097046413502,
         "invalid": 0
        },
        {
         "value": "5+",
         "n": 515,
         "correct": 260,
         "accuracy": 0.5048543689320388,
         "lo": 0.4640776699029126,
         "hi": 0.5495145631067961,
         "best_constant": 0.5184466019417475,
         "invalid": 0
        }
       ]
      }
     },
     "openai-gpt-6-luna-effort-none": {
      "file": "studies/proofwriter-cwa-openai-gpt-6-luna-effort-none.jsonl",
      "modified": "2026-10-01T16:36:24.035Z",
      "overall": {
       "n": 1800,
       "correct": 1170,
       "accuracy": 0.65,
       "lo": 0.6277777777777778,
       "hi": 0.6738888888888889,
       "macro_f1": 0.6499477082378973,
       "invalid": 0,
       "best_constant": 0.5,
       "chance": 0.5,
       "recall": {
        "false": 0.6377777777777778,
        "true": 0.6622222222222223
       },
       "confusion": {
        "false": {
         "false": 574,
         "invalid": 0,
         "true": 326
        },
        "true": {
         "false": 304,
         "invalid": 0,
         "true": 596
        }
       },
       "model": [
        "gpt-6-luna"
       ]
      },
      "axes": {
       "depth": [
        {
         "value": "0",
         "n": 300,
         "correct": 294,
         "accuracy": 0.98,
         "lo": 0.9633333333333334,
         "hi": 0.9933333333333333,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 300,
         "correct": 230,
         "accuracy": 0.7666666666666667,
         "lo": 0.7233333333333334,
         "hi": 0.81,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 300,
         "correct": 196,
         "accuracy": 0.6533333333333333,
         "lo": 0.6,
         "hi": 0.7066666666666667,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 300,
         "correct": 171,
         "accuracy": 0.57,
         "lo": 0.5166666666666667,
         "hi": 0.63,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 300,
         "correct": 145,
         "accuracy": 0.48333333333333334,
         "lo": 0.43,
         "hi": 0.5366666666666666,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 300,
         "correct": 134,
         "accuracy": 0.44666666666666666,
         "lo": 0.38666666666666666,
         "hi": 0.5033333333333333,
         "best_constant": 0.5,
         "invalid": 0
        }
       ],
       "reference_label": [
        {
         "value": "false",
         "n": 900,
         "correct": 574,
         "accuracy": 0.6377777777777778,
         "lo": 0.6055555555555555,
         "hi": 0.6688888888888889,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "true",
         "n": 900,
         "correct": 596,
         "accuracy": 0.6622222222222223,
         "lo": 0.6311111111111111,
         "hi": 0.6922222222222222,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "theory_kind": [
        {
         "value": "attribute",
         "n": 1016,
         "correct": 645,
         "accuracy": 0.6348425196850394,
         "lo": 0.6062992125984252,
         "hi": 0.6633858267716536,
         "best_constant": 0.5019685039370079,
         "invalid": 0
        },
        {
         "value": "relation",
         "n": 784,
         "correct": 525,
         "accuracy": 0.6696428571428571,
         "lo": 0.639030612244898,
         "hi": 0.701530612244898,
         "best_constant": 0.5025510204081632,
         "invalid": 0
        }
       ],
       "theory_negation": [
        {
         "value": "negation",
         "n": 870,
         "correct": 597,
         "accuracy": 0.6862068965517242,
         "lo": 0.6540229885057471,
         "hi": 0.7183908045977011,
         "best_constant": 0.5011494252873563,
         "invalid": 0
        },
        {
         "value": "no-negation",
         "n": 930,
         "correct": 573,
         "accuracy": 0.6161290322580645,
         "lo": 0.5860215053763441,
         "hi": 0.6473118279569893,
         "best_constant": 0.5010752688172043,
         "invalid": 0
        }
       ],
       "statement_negated": [
        {
         "value": "False",
         "n": 882,
         "correct": 595,
         "accuracy": 0.6746031746031746,
         "lo": 0.6439909297052154,
         "hi": 0.7052154195011338,
         "best_constant": 0.5476190476190477,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 918,
         "correct": 575,
         "accuracy": 0.6263616557734205,
         "lo": 0.593681917211329,
         "hi": 0.6579520697167756,
         "best_constant": 0.545751633986928,
         "invalid": 0
        }
       ],
       "strategy": [
        {
         "value": "inv-proof",
         "n": 501,
         "correct": 260,
         "accuracy": 0.5189620758483033,
         "lo": 0.47704590818363274,
         "hi": 0.564870259481038,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-random",
         "n": 75,
         "correct": 74,
         "accuracy": 0.9866666666666667,
         "lo": 0.96,
         "hi": 1,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-rconc",
         "n": 342,
         "correct": 241,
         "accuracy": 0.7046783625730995,
         "lo": 0.652046783625731,
         "hi": 0.7514619883040936,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "proof",
         "n": 483,
         "correct": 281,
         "accuracy": 0.5817805383022774,
         "lo": 0.5383022774327122,
         "hi": 0.629399585921325,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "random",
         "n": 75,
         "correct": 75,
         "accuracy": 1,
         "lo": 1,
         "hi": 1,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "rconc",
         "n": 324,
         "correct": 239,
         "accuracy": 0.7376543209876543,
         "lo": 0.6882716049382716,
         "hi": 0.7870370370370371,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "paraphrased": [
        {
         "value": "False",
         "n": 1306,
         "correct": 879,
         "accuracy": 0.6730474732006125,
         "lo": 0.6477794793261868,
         "hi": 0.6990811638591118,
         "best_constant": 0.5022970903522205,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 494,
         "correct": 291,
         "accuracy": 0.5890688259109311,
         "lo": 0.5465587044534413,
         "hi": 0.6336032388663968,
         "best_constant": 0.5060728744939271,
         "invalid": 0
        }
       ],
       "theory_max_depth": [
        {
         "value": "0",
         "n": 71,
         "correct": 58,
         "accuracy": 0.8169014084507042,
         "lo": 0.7183098591549296,
         "hi": 0.9014084507042254,
         "best_constant": 0.7605633802816901,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 155,
         "correct": 112,
         "accuracy": 0.7225806451612903,
         "lo": 0.6516129032258065,
         "hi": 0.7935483870967742,
         "best_constant": 0.5096774193548387,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 230,
         "correct": 184,
         "accuracy": 0.8,
         "lo": 0.7478260869565218,
         "hi": 0.8478260869565217,
         "best_constant": 0.5260869565217391,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 619,
         "correct": 428,
         "accuracy": 0.691437802907916,
         "lo": 0.6575121163166397,
         "hi": 0.72859450726979,
         "best_constant": 0.5218093699515347,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 171,
         "correct": 82,
         "accuracy": 0.47953216374269003,
         "lo": 0.4093567251461988,
         "hi": 0.5497076023391813,
         "best_constant": 0.5087719298245614,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 513,
         "correct": 282,
         "accuracy": 0.5497076023391813,
         "lo": 0.50682261208577,
         "hi": 0.594541910331384,
         "best_constant": 0.5243664717348928,
         "invalid": 0
        },
        {
         "value": "6",
         "n": 38,
         "correct": 22,
         "accuracy": 0.5789473684210527,
         "lo": 0.42105263157894735,
         "hi": 0.7368421052631579,
         "best_constant": 0.5789473684210527,
         "invalid": 0
        },
        {
         "value": "7",
         "n": 2,
         "correct": 2,
         "accuracy": 1,
         "lo": 1,
         "hi": 1,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "8",
         "n": 1,
         "correct": 0,
         "accuracy": 0,
         "lo": 0,
         "hi": 0,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "words_bin": [
        {
         "value": "0-49",
         "n": 256,
         "correct": 217,
         "accuracy": 0.84765625,
         "lo": 0.8046875,
         "hi": 0.890625,
         "best_constant": 0.5078125,
         "invalid": 0
        },
        {
         "value": "50-79",
         "n": 432,
         "correct": 289,
         "accuracy": 0.6689814814814815,
         "lo": 0.625,
         "hi": 0.7129629629629629,
         "best_constant": 0.5023148148148148,
         "invalid": 0
        },
        {
         "value": "80-109",
         "n": 482,
         "correct": 304,
         "accuracy": 0.6307053941908713,
         "lo": 0.5871369294605809,
         "hi": 0.6742738589211619,
         "best_constant": 0.508298755186722,
         "invalid": 0
        },
        {
         "value": "110+",
         "n": 630,
         "correct": 360,
         "accuracy": 0.5714285714285714,
         "lo": 0.5349206349206349,
         "hi": 0.6111111111111112,
         "best_constant": 0.5047619047619047,
         "invalid": 0
        }
       ],
       "rules_bin": [
        {
         "value": "0-2",
         "n": 208,
         "correct": 174,
         "accuracy": 0.8365384615384616,
         "lo": 0.7884615384615384,
         "hi": 0.8846153846153846,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "3-5",
         "n": 405,
         "correct": 262,
         "accuracy": 0.6469135802469136,
         "lo": 0.6024691358024692,
         "hi": 0.691358024691358,
         "best_constant": 0.5037037037037037,
         "invalid": 0
        },
        {
         "value": "6-7",
         "n": 605,
         "correct": 346,
         "accuracy": 0.571900826446281,
         "lo": 0.5338842975206611,
         "hi": 0.6115702479338843,
         "best_constant": 0.5057851239669422,
         "invalid": 0
        },
        {
         "value": "8+",
         "n": 582,
         "correct": 388,
         "accuracy": 0.6666666666666666,
         "lo": 0.627147766323024,
         "hi": 0.7044673539518901,
         "best_constant": 0.5034364261168385,
         "invalid": 0
        }
       ],
       "facts_bin": [
        {
         "value": "0-3",
         "n": 360,
         "correct": 259,
         "accuracy": 0.7194444444444444,
         "lo": 0.6722222222222223,
         "hi": 0.7666666666666667,
         "best_constant": 0.5194444444444445,
         "invalid": 0
        },
        {
         "value": "4-7",
         "n": 425,
         "correct": 305,
         "accuracy": 0.7176470588235294,
         "lo": 0.6729411764705883,
         "hi": 0.76,
         "best_constant": 0.5129411764705882,
         "invalid": 0
        },
        {
         "value": "8-12",
         "n": 812,
         "correct": 500,
         "accuracy": 0.6157635467980296,
         "lo": 0.583743842364532,
         "hi": 0.6514778325123153,
         "best_constant": 0.5012315270935961,
         "invalid": 0
        },
        {
         "value": "13+",
         "n": 203,
         "correct": 106,
         "accuracy": 0.5221674876847291,
         "lo": 0.45320197044334976,
         "hi": 0.5862068965517241,
         "best_constant": 0.5024630541871922,
         "invalid": 0
        }
       ],
       "proof_size_bin": [
        {
         "value": "0-1",
         "n": 966,
         "correct": 774,
         "accuracy": 0.8012422360248447,
         "lo": 0.7753623188405797,
         "hi": 0.8271221532091098,
         "best_constant": 0.5093167701863354,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 82,
         "correct": 64,
         "accuracy": 0.7804878048780488,
         "lo": 0.6829268292682927,
         "hi": 0.8658536585365854,
         "best_constant": 0.5,
         "invalid": 0
        },
        {
         "value": "3-4",
         "n": 237,
         "correct": 150,
         "accuracy": 0.6329113924050633,
         "lo": 0.5738396624472574,
         "hi": 0.6962025316455697,
         "best_constant": 0.5021097046413502,
         "invalid": 0
        },
        {
         "value": "5+",
         "n": 515,
         "correct": 182,
         "accuracy": 0.3533980582524272,
         "lo": 0.3145631067961165,
         "hi": 0.39805825242718446,
         "best_constant": 0.5184466019417475,
         "invalid": 0
        }
       ]
      }
     }
    },
    "pairs": {
     "jev|kev-0.8b": {
      "a": "jev",
      "b": "kev-0.8b",
      "file": "studies/proofwriter-cwa-jev-vs-kev-0.8b.jsonl",
      "modified": "2026-10-01T16:36:24.417Z",
      "rows": [
       {
        "axis": "overall",
        "ci_high": 0.36333333333333334,
        "ci_low": 0.30666666666666664,
        "diff": 0.335,
        "n": 1800,
        "value": "all"
       },
       {
        "axis": "depth",
        "ci_high": 0.36666666666666664,
        "ci_low": 0.26,
        "diff": 0.31333333333333335,
        "n": 300,
        "value": "0"
       },
       {
        "axis": "depth",
        "ci_high": 0.4,
        "ci_low": 0.27666666666666667,
        "diff": 0.34,
        "n": 300,
        "value": "1"
       },
       {
        "axis": "depth",
        "ci_high": 0.4766666666666667,
        "ci_low": 0.34,
        "diff": 0.4066666666666667,
        "n": 300,
        "value": "2"
       },
       {
        "axis": "depth",
        "ci_high": 0.39,
        "ci_low": 0.25,
        "diff": 0.32,
        "n": 300,
        "value": "3"
       },
       {
        "axis": "depth",
        "ci_high": 0.32666666666666666,
        "ci_low": 0.18,
        "diff": 0.25333333333333335,
        "n": 300,
        "value": "4"
       },
       {
        "axis": "depth",
        "ci_high": 0.44,
        "ci_low": 0.31,
        "diff": 0.37666666666666665,
        "n": 300,
        "value": "5"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.2633333333333333,
        "ci_low": 0.18888888888888888,
        "diff": 0.22777777777777777,
        "n": 900,
        "value": "false"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.4822222222222222,
        "ci_low": 0.40444444444444444,
        "diff": 0.44222222222222224,
        "n": 900,
        "value": "true"
       }
      ]
     },
     "jev|kev-4b": {
      "a": "jev",
      "b": "kev-4b",
      "file": "studies/proofwriter-cwa-jev-vs-kev-4b.jsonl",
      "modified": "2026-10-01T16:36:24.799Z",
      "rows": [
       {
        "axis": "overall",
        "ci_high": 0.33055555555555555,
        "ci_low": 0.2788888888888889,
        "diff": 0.305,
        "n": 1800,
        "value": "all"
       },
       {
        "axis": "depth",
        "ci_high": 0.18,
        "ci_low": 0.09666666666666666,
        "diff": 0.13666666666666666,
        "n": 300,
        "value": "0"
       },
       {
        "axis": "depth",
        "ci_high": 0.31,
        "ci_low": 0.19333333333333333,
        "diff": 0.25333333333333335,
        "n": 300,
        "value": "1"
       },
       {
        "axis": "depth",
        "ci_high": 0.43666666666666665,
        "ci_low": 0.32,
        "diff": 0.37666666666666665,
        "n": 300,
        "value": "2"
       },
       {
        "axis": "depth",
        "ci_high": 0.3566666666666667,
        "ci_low": 0.22666666666666666,
        "diff": 0.29,
        "n": 300,
        "value": "3"
       },
       {
        "axis": "depth",
        "ci_high": 0.38666666666666666,
        "ci_low": 0.24666666666666667,
        "diff": 0.31333333333333335,
        "n": 300,
        "value": "4"
       },
       {
        "axis": "depth",
        "ci_high": 0.5233333333333333,
        "ci_low": 0.39666666666666667,
        "diff": 0.46,
        "n": 300,
        "value": "5"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.16444444444444445,
        "ci_low": 0.10444444444444445,
        "diff": 0.13555555555555557,
        "n": 900,
        "value": "false"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.51,
        "ci_low": 0.4388888888888889,
        "diff": 0.47444444444444445,
        "n": 900,
        "value": "true"
       }
      ]
     },
     "jev|kev-9b": {
      "a": "jev",
      "b": "kev-9b",
      "file": "studies/proofwriter-cwa-jev-vs-kev-9b.jsonl",
      "modified": "2026-10-01T16:36:25.187Z",
      "rows": [
       {
        "axis": "overall",
        "ci_high": 0.2783333333333333,
        "ci_low": 0.23,
        "diff": 0.255,
        "n": 1800,
        "value": "all"
       },
       {
        "axis": "depth",
        "ci_high": 0.13333333333333333,
        "ci_low": 0.06333333333333334,
        "diff": 0.09666666666666666,
        "n": 300,
        "value": "0"
       },
       {
        "axis": "depth",
        "ci_high": 0.22,
        "ci_low": 0.11666666666666667,
        "diff": 0.16666666666666666,
        "n": 300,
        "value": "1"
       },
       {
        "axis": "depth",
        "ci_high": 0.3333333333333333,
        "ci_low": 0.22,
        "diff": 0.28,
        "n": 300,
        "value": "2"
       },
       {
        "axis": "depth",
        "ci_high": 0.3333333333333333,
        "ci_low": 0.2,
        "diff": 0.2633333333333333,
        "n": 300,
        "value": "3"
       },
       {
        "axis": "depth",
        "ci_high": 0.32666666666666666,
        "ci_low": 0.19666666666666666,
        "diff": 0.26,
        "n": 300,
        "value": "4"
       },
       {
        "axis": "depth",
        "ci_high": 0.5233333333333333,
        "ci_low": 0.4033333333333333,
        "diff": 0.4633333333333333,
        "n": 300,
        "value": "5"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.29555555555555557,
        "ci_low": 0.22777777777777777,
        "diff": 0.26,
        "n": 900,
        "value": "false"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.28444444444444444,
        "ci_low": 0.21666666666666667,
        "diff": 0.25,
        "n": 900,
        "value": "true"
       }
      ]
     },
     "jev|laya": {
      "a": "jev",
      "b": "laya",
      "file": "studies/proofwriter-cwa-jev-vs-laya.jsonl",
      "modified": "2026-10-01T16:36:25.567Z",
      "rows": [
       {
        "axis": "overall",
        "ci_high": 0.365,
        "ci_low": 0.30722222222222223,
        "diff": 0.33666666666666667,
        "n": 1800,
        "value": "all"
       },
       {
        "axis": "depth",
        "ci_high": 0.33,
        "ci_low": 0.22333333333333333,
        "diff": 0.2733333333333333,
        "n": 300,
        "value": "0"
       },
       {
        "axis": "depth",
        "ci_high": 0.49333333333333335,
        "ci_low": 0.36666666666666664,
        "diff": 0.4266666666666667,
        "n": 300,
        "value": "1"
       },
       {
        "axis": "depth",
        "ci_high": 0.44,
        "ci_low": 0.31333333333333335,
        "diff": 0.37333333333333335,
        "n": 300,
        "value": "2"
       },
       {
        "axis": "depth",
        "ci_high": 0.41,
        "ci_low": 0.27,
        "diff": 0.33666666666666667,
        "n": 300,
        "value": "3"
       },
       {
        "axis": "depth",
        "ci_high": 0.25,
        "ci_low": 0.1,
        "diff": 0.17666666666666667,
        "n": 300,
        "value": "4"
       },
       {
        "axis": "depth",
        "ci_high": 0.5033333333333333,
        "ci_low": 0.37,
        "diff": 0.43333333333333335,
        "n": 300,
        "value": "5"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.058888888888888886,
        "ci_low": -0.006666666666666667,
        "diff": 0.024444444444444446,
        "n": 900,
        "value": "false"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.6833333333333333,
        "ci_low": 0.6155555555555555,
        "diff": 0.6488888888888888,
        "n": 900,
        "value": "true"
       }
      ]
     },
     "jev|openai-gpt-6-luna-effort-none": {
      "a": "jev",
      "b": "openai-gpt-6-luna-effort-none",
      "file": "studies/proofwriter-cwa-jev-vs-openai-gpt-6-luna-effort-none.jsonl",
      "modified": "2026-10-01T16:36:25.950Z",
      "rows": [
       {
        "axis": "overall",
        "ci_high": 0.26555555555555554,
        "ci_low": 0.21944444444444444,
        "diff": 0.2427777777777778,
        "n": 1800,
        "value": "all"
       },
       {
        "axis": "depth",
        "ci_high": 0.03,
        "ci_low": 0,
        "diff": 0.013333333333333334,
        "n": 300,
        "value": "0"
       },
       {
        "axis": "depth",
        "ci_high": 0.21666666666666667,
        "ci_low": 0.12333333333333334,
        "diff": 0.17,
        "n": 300,
        "value": "1"
       },
       {
        "axis": "depth",
        "ci_high": 0.32,
        "ci_low": 0.21666666666666667,
        "diff": 0.26666666666666666,
        "n": 300,
        "value": "2"
       },
       {
        "axis": "depth",
        "ci_high": 0.34,
        "ci_low": 0.22333333333333333,
        "diff": 0.28,
        "n": 300,
        "value": "3"
       },
       {
        "axis": "depth",
        "ci_high": 0.3433333333333333,
        "ci_low": 0.22,
        "diff": 0.28,
        "n": 300,
        "value": "4"
       },
       {
        "axis": "depth",
        "ci_high": 0.5133333333333333,
        "ci_low": 0.38,
        "diff": 0.44666666666666666,
        "n": 300,
        "value": "5"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.2833333333333333,
        "ci_low": 0.22,
        "diff": 0.2511111111111111,
        "n": 900,
        "value": "false"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.2677777777777778,
        "ci_low": 0.20333333333333334,
        "diff": 0.23444444444444446,
        "n": 900,
        "value": "true"
       }
      ]
     }
    },
    "retest": {
     "jev": {
      "file": "studies/retest/proofwriter-cwa.jsonl",
      "depth": [
       {
        "value": "0",
        "n": 300,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "1",
        "n": 300,
        "changed": 3,
        "rate": 0.01
       },
       {
        "value": "2",
        "n": 300,
        "changed": 5,
        "rate": 0.016666666666666666
       },
       {
        "value": "3",
        "n": 300,
        "changed": 6,
        "rate": 0.02
       },
       {
        "value": "4",
        "n": 300,
        "changed": 12,
        "rate": 0.04
       },
       {
        "value": "5",
        "n": 300,
        "changed": 12,
        "rate": 0.04
       }
      ],
      "n": 1800,
      "agreement": 0.9788888888888889,
      "ac1": 0.9577782469083678,
      "ac1_lo": 0.9444447187915121,
      "ac1_hi": 0.9711390456644431,
      "kappa": 0.9577781426580264,
      "changed": 38,
      "accuracy_run1": 0.8927777777777778,
      "accuracy_run2": 0.8972222222222223,
      "prob_shift_mean": 0.020144444444444444,
      "prob_shift_max": 0.9
     },
     "kev-0.8b": {
      "file": "studies/retest/proofwriter-cwa.jsonl",
      "depth": [
       {
        "value": "0",
        "n": 300,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "1",
        "n": 300,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "2",
        "n": 300,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "3",
        "n": 300,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "4",
        "n": 300,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "5",
        "n": 300,
        "changed": 0,
        "rate": 0
       }
      ],
      "n": 1800,
      "agreement": 1,
      "ac1": 1,
      "ac1_lo": 1,
      "ac1_hi": 1,
      "kappa": 1,
      "changed": 0,
      "accuracy_run1": 0.5577777777777778,
      "accuracy_run2": 0.5577777777777778,
      "prob_shift_mean": 0,
      "prob_shift_max": 0
     },
     "kev-4b": {
      "file": "studies/retest/proofwriter-cwa.jsonl",
      "depth": [
       {
        "value": "0",
        "n": 300,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "1",
        "n": 300,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "2",
        "n": 300,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "3",
        "n": 300,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "4",
        "n": 300,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "5",
        "n": 300,
        "changed": 0,
        "rate": 0
       }
      ],
      "n": 1800,
      "agreement": 1,
      "ac1": 1,
      "ac1_lo": 1,
      "ac1_hi": 1,
      "kappa": 1,
      "changed": 0,
      "accuracy_run1": 0.5877777777777777,
      "accuracy_run2": 0.5877777777777777,
      "prob_shift_mean": 0,
      "prob_shift_max": 0
     },
     "laya": {
      "file": "studies/retest/proofwriter-cwa.jsonl",
      "depth": [
       {
        "value": "0",
        "n": 300,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "1",
        "n": 300,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "2",
        "n": 300,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "3",
        "n": 300,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "4",
        "n": 300,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "5",
        "n": 300,
        "changed": 0,
        "rate": 0
       }
      ],
      "n": 1800,
      "agreement": 1,
      "ac1": 1,
      "ac1_lo": 1,
      "ac1_hi": 1,
      "kappa": 1,
      "changed": 0,
      "accuracy_run1": 0.5561111111111111,
      "accuracy_run2": 0.5561111111111111,
      "prob_shift_mean": 0,
      "prob_shift_max": 0
     },
     "openai-gpt-6-luna-effort-none": {
      "file": "studies/retest/proofwriter-cwa.jsonl",
      "depth": [
       {
        "value": "0",
        "n": 300,
        "changed": 2,
        "rate": 0.006666666666666667
       },
       {
        "value": "1",
        "n": 300,
        "changed": 22,
        "rate": 0.07333333333333333
       },
       {
        "value": "2",
        "n": 300,
        "changed": 35,
        "rate": 0.11666666666666667
       },
       {
        "value": "3",
        "n": 300,
        "changed": 36,
        "rate": 0.12
       },
       {
        "value": "4",
        "n": 300,
        "changed": 34,
        "rate": 0.11333333333333333
       },
       {
        "value": "5",
        "n": 300,
        "changed": 40,
        "rate": 0.13333333333333333
       }
      ],
      "n": 1800,
      "agreement": 0.9061111111111111,
      "ac1": 0.812329321939424,
      "ac1_lo": 0.7843732002303707,
      "ac1_hi": 0.8377795802268863,
      "kappa": 0.8121150582183911,
      "changed": 169,
      "accuracy_run1": 0.65,
      "accuracy_run2": 0.6483333333333333,
      "prob_shift_mean": null,
      "prob_shift_max": null
     }
    }
   },
   "status": {
    "jev": {
     "status": "complete",
     "scored": 1800,
     "answered": 1800,
     "of": 1800
    },
    "laya": {
     "status": "complete",
     "scored": 1800,
     "answered": 1800,
     "of": 1800
    },
    "kev-0.8b": {
     "status": "complete",
     "scored": 1800,
     "answered": 1800,
     "of": 1800
    },
    "kev-4b": {
     "status": "complete",
     "scored": 1800,
     "answered": 1800,
     "of": 1800
    },
    "kev-9b": {
     "status": "complete",
     "scored": 1800,
     "answered": 1800,
     "of": 1800
    },
    "kev-27b": {
     "status": "pending",
     "scored": 0,
     "answered": 0,
     "of": 1800
    },
    "openai-gpt-6-luna-effort-none": {
     "status": "complete",
     "scored": 1800,
     "answered": 1800,
     "of": 1800
    }
   },
   "derived": {
    "jev": {
     "cross": [
      {
       "depth": "5",
       "label": "true",
       "n": 150,
       "correct": 134,
       "said": {
        "true": 134,
        "false": 16
       }
      },
      {
       "depth": "5",
       "label": "false",
       "n": 150,
       "correct": 134,
       "said": {
        "false": 134,
        "true": 16
       }
      },
      {
       "depth": "4",
       "label": "true",
       "n": 150,
       "correct": 115,
       "said": {
        "true": 115,
        "false": 35
       }
      },
      {
       "depth": "4",
       "label": "false",
       "n": 150,
       "correct": 114,
       "said": {
        "false": 114,
        "true": 36
       }
      },
      {
       "depth": "3",
       "label": "true",
       "n": 150,
       "correct": 130,
       "said": {
        "true": 130,
        "false": 20
       }
      },
      {
       "depth": "3",
       "label": "false",
       "n": 150,
       "correct": 125,
       "said": {
        "false": 125,
        "true": 25
       }
      },
      {
       "depth": "2",
       "label": "true",
       "n": 150,
       "correct": 135,
       "said": {
        "true": 135,
        "false": 15
       }
      },
      {
       "depth": "2",
       "label": "false",
       "n": 150,
       "correct": 141,
       "said": {
        "false": 141,
        "true": 9
       }
      },
      {
       "depth": "1",
       "label": "true",
       "n": 150,
       "correct": 143,
       "said": {
        "true": 143,
        "false": 7
       }
      },
      {
       "depth": "1",
       "label": "false",
       "n": 150,
       "correct": 138,
       "said": {
        "false": 138,
        "true": 12
       }
      },
      {
       "depth": "0",
       "label": "true",
       "n": 150,
       "correct": 150,
       "said": {
        "true": 150
       }
      },
      {
       "depth": "0",
       "label": "false",
       "n": 150,
       "correct": 148,
       "said": {
        "false": 148,
        "true": 2
       }
      }
     ],
     "calibration": {
      "n": 1800,
      "ece": 0.028711111111115954,
      "mean_confidence": 0.9210555555555607,
      "accuracy": 0.8927777777777778,
      "bins": [
       {
        "lo": 0,
        "hi": 0.5,
        "n": 0,
        "accuracy": null,
        "confidence": null
       },
       {
        "lo": 0.5,
        "hi": 0.6,
        "n": 80,
        "accuracy": 0.4625,
        "confidence": 0.548875
       },
       {
        "lo": 0.6,
        "hi": 0.7,
        "n": 95,
        "accuracy": 0.6,
        "confidence": 0.6469473684210525
       },
       {
        "lo": 0.7,
        "hi": 0.8,
        "n": 104,
        "accuracy": 0.75,
        "confidence": 0.7462500000000003
       },
       {
        "lo": 0.8,
        "hi": 0.9,
        "n": 157,
        "accuracy": 0.7834394904458599,
        "confidence": 0.843885350318471
       },
       {
        "lo": 0.9,
        "hi": 1,
        "n": 1364,
        "accuracy": 0.9618768328445748,
        "confidence": 0.9841862170088042
       }
      ]
     },
     "models": [
      "jev-1.13.0"
     ],
     "file": "answers/jev/proofwriter-cwa.jsonl.gz"
    },
    "kev-0.8b": {
     "cross": [
      {
       "depth": "5",
       "label": "true",
       "n": 150,
       "correct": 61,
       "said": {
        "false": 89,
        "true": 61
       }
      },
      {
       "depth": "5",
       "label": "false",
       "n": 150,
       "correct": 94,
       "said": {
        "false": 94,
        "true": 56
       }
      },
      {
       "depth": "4",
       "label": "true",
       "n": 150,
       "correct": 64,
       "said": {
        "true": 64,
        "false": 86
       }
      },
      {
       "depth": "4",
       "label": "false",
       "n": 150,
       "correct": 89,
       "said": {
        "true": 61,
        "false": 89
       }
      },
      {
       "depth": "3",
       "label": "true",
       "n": 150,
       "correct": 68,
       "said": {
        "false": 82,
        "true": 68
       }
      },
      {
       "depth": "3",
       "label": "false",
       "n": 150,
       "correct": 91,
       "said": {
        "false": 91,
        "true": 59
       }
      },
      {
       "depth": "2",
       "label": "true",
       "n": 150,
       "correct": 60,
       "said": {
        "false": 90,
        "true": 60
       }
      },
      {
       "depth": "2",
       "label": "false",
       "n": 150,
       "correct": 94,
       "said": {
        "false": 94,
        "true": 56
       }
      },
      {
       "depth": "1",
       "label": "true",
       "n": 150,
       "correct": 70,
       "said": {
        "false": 80,
        "true": 70
       }
      },
      {
       "depth": "1",
       "label": "false",
       "n": 150,
       "correct": 109,
       "said": {
        "true": 41,
        "false": 109
       }
      },
      {
       "depth": "0",
       "label": "true",
       "n": 150,
       "correct": 86,
       "said": {
        "true": 86,
        "false": 64
       }
      },
      {
       "depth": "0",
       "label": "false",
       "n": 150,
       "correct": 118,
       "said": {
        "false": 118,
        "true": 32
       }
      }
     ],
     "calibration": {
      "n": 1800,
      "ece": 0.17032711111111132,
      "mean_confidence": 0.7281048888888897,
      "accuracy": 0.5577777777777778,
      "bins": [
       {
        "lo": 0,
        "hi": 0.5,
        "n": 0,
        "accuracy": null,
        "confidence": null
       },
       {
        "lo": 0.5,
        "hi": 0.6,
        "n": 374,
        "accuracy": 0.5,
        "confidence": 0.5506828877005351
       },
       {
        "lo": 0.6,
        "hi": 0.7,
        "n": 413,
        "accuracy": 0.46973365617433416,
        "confidence": 0.6524733656174333
       },
       {
        "lo": 0.7,
        "hi": 0.8,
        "n": 434,
        "accuracy": 0.5852534562211982,
        "confidence": 0.7498437788018436
       },
       {
        "lo": 0.8,
        "hi": 0.9,
        "n": 363,
        "accuracy": 0.6446280991735537,
        "confidence": 0.8488426997245181
       },
       {
        "lo": 0.9,
        "hi": 1,
        "n": 216,
        "accuracy": 0.625,
        "confidence": 0.9333324074074077
       }
      ]
     },
     "models": [
      "kev-latest"
     ],
     "file": "answers/kev-0.8b/proofwriter-cwa.jsonl.gz"
    },
    "kev-4b": {
     "cross": [
      {
       "depth": "5",
       "label": "true",
       "n": 150,
       "correct": 42,
       "said": {
        "false": 108,
        "true": 42
       }
      },
      {
       "depth": "5",
       "label": "false",
       "n": 150,
       "correct": 88,
       "said": {
        "true": 62,
        "false": 88
       }
      },
      {
       "depth": "4",
       "label": "true",
       "n": 150,
       "correct": 35,
       "said": {
        "false": 115,
        "true": 35
       }
      },
      {
       "depth": "4",
       "label": "false",
       "n": 150,
       "correct": 100,
       "said": {
        "true": 50,
        "false": 100
       }
      },
      {
       "depth": "3",
       "label": "true",
       "n": 150,
       "correct": 52,
       "said": {
        "false": 98,
        "true": 52
       }
      },
      {
       "depth": "3",
       "label": "false",
       "n": 150,
       "correct": 116,
       "said": {
        "false": 116,
        "true": 34
       }
      },
      {
       "depth": "2",
       "label": "true",
       "n": 150,
       "correct": 54,
       "said": {
        "false": 96,
        "true": 54
       }
      },
      {
       "depth": "2",
       "label": "false",
       "n": 150,
       "correct": 109,
       "said": {
        "false": 109,
        "true": 41
       }
      },
      {
       "depth": "1",
       "label": "true",
       "n": 150,
       "correct": 81,
       "said": {
        "false": 69,
        "true": 81
       }
      },
      {
       "depth": "1",
       "label": "false",
       "n": 150,
       "correct": 124,
       "said": {
        "false": 124,
        "true": 26
       }
      },
      {
       "depth": "0",
       "label": "true",
       "n": 150,
       "correct": 116,
       "said": {
        "true": 116,
        "false": 34
       }
      },
      {
       "depth": "0",
       "label": "false",
       "n": 150,
       "correct": 141,
       "said": {
        "false": 141,
        "true": 9
       }
      }
     ],
     "calibration": {
      "n": 1800,
      "ece": 0.07630583333333359,
      "mean_confidence": 0.6640836111111111,
      "accuracy": 0.5877777777777777,
      "bins": [
       {
        "lo": 0,
        "hi": 0.5,
        "n": 0,
        "accuracy": null,
        "confidence": null
       },
       {
        "lo": 0.5,
        "hi": 0.6,
        "n": 704,
        "accuracy": 0.5042613636363636,
        "confidence": 0.5503846590909101
       },
       {
        "lo": 0.6,
        "hi": 0.7,
        "n": 444,
        "accuracy": 0.5945945945945946,
        "confidence": 0.6483407657657656
       },
       {
        "lo": 0.7,
        "hi": 0.8,
        "n": 360,
        "accuracy": 0.6361111111111111,
        "confidence": 0.7461869444444444
       },
       {
        "lo": 0.8,
        "hi": 0.9,
        "n": 235,
        "accuracy": 0.6936170212765957,
        "confidence": 0.8453782978723396
       },
       {
        "lo": 0.9,
        "hi": 1,
        "n": 57,
        "accuracy": 0.8245614035087719,
        "confidence": 0.92500350877193
       }
      ]
     },
     "models": [
      "kev-latest"
     ],
     "file": "answers/kev-4b/proofwriter-cwa.jsonl.gz"
    },
    "kev-9b": {
     "cross": [
      {
       "depth": "5",
       "label": "true",
       "n": 150,
       "correct": 77,
       "said": {
        "false": 73,
        "true": 77
       }
      },
      {
       "depth": "5",
       "label": "false",
       "n": 150,
       "correct": 52,
       "said": {
        "true": 98,
        "false": 52
       }
      },
      {
       "depth": "4",
       "label": "true",
       "n": 150,
       "correct": 76,
       "said": {
        "true": 76,
        "false": 74
       }
      },
      {
       "depth": "4",
       "label": "false",
       "n": 150,
       "correct": 75,
       "said": {
        "true": 75,
        "false": 75
       }
      },
      {
       "depth": "3",
       "label": "true",
       "n": 150,
       "correct": 86,
       "said": {
        "true": 86,
        "false": 64
       }
      },
      {
       "depth": "3",
       "label": "false",
       "n": 150,
       "correct": 90,
       "said": {
        "true": 60,
        "false": 90
       }
      },
      {
       "depth": "2",
       "label": "true",
       "n": 150,
       "correct": 100,
       "said": {
        "true": 100,
        "false": 50
       }
      },
      {
       "depth": "2",
       "label": "false",
       "n": 150,
       "correct": 92,
       "said": {
        "false": 92,
        "true": 58
       }
      },
      {
       "depth": "1",
       "label": "true",
       "n": 150,
       "correct": 117,
       "said": {
        "true": 117,
        "false": 33
       }
      },
      {
       "depth": "1",
       "label": "false",
       "n": 150,
       "correct": 114,
       "said": {
        "false": 114,
        "true": 36
       }
      },
      {
       "depth": "0",
       "label": "true",
       "n": 150,
       "correct": 126,
       "said": {
        "true": 126,
        "false": 24
       }
      },
      {
       "depth": "0",
       "label": "false",
       "n": 150,
       "correct": 143,
       "said": {
        "false": 143,
        "true": 7
       }
      }
     ],
     "calibration": {
      "n": 1800,
      "ece": 0.13511344444444473,
      "mean_confidence": 0.7728912222222226,
      "accuracy": 0.6377777777777778,
      "bins": [
       {
        "lo": 0,
        "hi": 0.5,
        "n": 0,
        "accuracy": null,
        "confidence": null
       },
       {
        "lo": 0.5,
        "hi": 0.6,
        "n": 270,
        "accuracy": 0.4962962962962963,
        "confidence": 0.5493785185185184
       },
       {
        "lo": 0.6,
        "hi": 0.7,
        "n": 310,
        "accuracy": 0.5612903225806452,
        "confidence": 0.6508412903225806
       },
       {
        "lo": 0.7,
        "hi": 0.8,
        "n": 376,
        "accuracy": 0.5771276595744681,
        "confidence": 0.7516630319148939
       },
       {
        "lo": 0.8,
        "hi": 0.9,
        "n": 439,
        "accuracy": 0.6879271070615034,
        "confidence": 0.8518369020501145
       },
       {
        "lo": 0.9,
        "hi": 1,
        "n": 405,
        "accuracy": 0.7925925925925926,
        "confidence": 0.9494555555555559
       }
      ]
     },
     "models": [
      "kev-latest"
     ],
     "file": "answers/kev-9b/proofwriter-cwa.jsonl.gz"
    },
    "laya": {
     "cross": [
      {
       "depth": "5",
       "label": "true",
       "n": 150,
       "correct": 13,
       "said": {
        "false": 137,
        "true": 13
       }
      },
      {
       "depth": "5",
       "label": "false",
       "n": 150,
       "correct": 125,
       "said": {
        "false": 125,
        "true": 25
       }
      },
      {
       "depth": "4",
       "label": "true",
       "n": 150,
       "correct": 38,
       "said": {
        "true": 38,
        "false": 112
       }
      },
      {
       "depth": "4",
       "label": "false",
       "n": 150,
       "correct": 138,
       "said": {
        "false": 138,
        "true": 12
       }
      },
      {
       "depth": "3",
       "label": "true",
       "n": 150,
       "correct": 29,
       "said": {
        "false": 121,
        "true": 29
       }
      },
      {
       "depth": "3",
       "label": "false",
       "n": 150,
       "correct": 125,
       "said": {
        "true": 25,
        "false": 125
       }
      },
      {
       "depth": "2",
       "label": "true",
       "n": 150,
       "correct": 38,
       "said": {
        "false": 112,
        "true": 38
       }
      },
      {
       "depth": "2",
       "label": "false",
       "n": 150,
       "correct": 126,
       "said": {
        "false": 126,
        "true": 24
       }
      },
      {
       "depth": "1",
       "label": "true",
       "n": 150,
       "correct": 28,
       "said": {
        "false": 122,
        "true": 28
       }
      },
      {
       "depth": "1",
       "label": "false",
       "n": 150,
       "correct": 125,
       "said": {
        "true": 25,
        "false": 125
       }
      },
      {
       "depth": "0",
       "label": "true",
       "n": 150,
       "correct": 77,
       "said": {
        "true": 77,
        "false": 73
       }
      },
      {
       "depth": "0",
       "label": "false",
       "n": 150,
       "correct": 139,
       "said": {
        "false": 139,
        "true": 11
       }
      }
     ],
     "calibration": {
      "n": 1800,
      "ece": 0.3170146666666666,
      "mean_confidence": 0.8728870000000003,
      "accuracy": 0.5561111111111111,
      "bins": [
       {
        "lo": 0,
        "hi": 0.5,
        "n": 0,
        "accuracy": null,
        "confidence": null
       },
       {
        "lo": 0.5,
        "hi": 0.6,
        "n": 99,
        "accuracy": 0.5555555555555556,
        "confidence": 0.5533848484848486
       },
       {
        "lo": 0.6,
        "hi": 0.7,
        "n": 94,
        "accuracy": 0.5319148936170213,
        "confidence": 0.650995744680851
       },
       {
        "lo": 0.7,
        "hi": 0.8,
        "n": 168,
        "accuracy": 0.5357142857142857,
        "confidence": 0.7575339285714285
       },
       {
        "lo": 0.8,
        "hi": 0.9,
        "n": 384,
        "accuracy": 0.5026041666666666,
        "confidence": 0.8577484375000006
       },
       {
        "lo": 0.9,
        "hi": 1,
        "n": 1055,
        "accuracy": 0.581042654028436,
        "confidence": 0.9465182938388622
       }
      ]
     },
     "models": [
      "laya-upstream:0.3.21"
     ],
     "file": "answers/laya/proofwriter-cwa.jsonl.gz"
    },
    "openai-gpt-6-luna-effort-none": {
     "cross": [
      {
       "depth": "5",
       "label": "true",
       "n": 150,
       "correct": 73,
       "said": {
        "false": 77,
        "true": 73
       }
      },
      {
       "depth": "5",
       "label": "false",
       "n": 150,
       "correct": 61,
       "said": {
        "true": 89,
        "false": 61
       }
      },
      {
       "depth": "4",
       "label": "true",
       "n": 150,
       "correct": 68,
       "said": {
        "true": 68,
        "false": 82
       }
      },
      {
       "depth": "4",
       "label": "false",
       "n": 150,
       "correct": 77,
       "said": {
        "false": 77,
        "true": 73
       }
      },
      {
       "depth": "3",
       "label": "true",
       "n": 150,
       "correct": 84,
       "said": {
        "false": 66,
        "true": 84
       }
      },
      {
       "depth": "3",
       "label": "false",
       "n": 150,
       "correct": 87,
       "said": {
        "false": 87,
        "true": 63
       }
      },
      {
       "depth": "2",
       "label": "true",
       "n": 150,
       "correct": 99,
       "said": {
        "true": 99,
        "false": 51
       }
      },
      {
       "depth": "2",
       "label": "false",
       "n": 150,
       "correct": 97,
       "said": {
        "false": 97,
        "true": 53
       }
      },
      {
       "depth": "1",
       "label": "true",
       "n": 150,
       "correct": 124,
       "said": {
        "true": 124,
        "false": 26
       }
      },
      {
       "depth": "1",
       "label": "false",
       "n": 150,
       "correct": 106,
       "said": {
        "false": 106,
        "true": 44
       }
      },
      {
       "depth": "0",
       "label": "true",
       "n": 150,
       "correct": 148,
       "said": {
        "true": 148,
        "false": 2
       }
      },
      {
       "depth": "0",
       "label": "false",
       "n": 150,
       "correct": 146,
       "said": {
        "false": 146,
        "true": 4
       }
      }
     ],
     "calibration": null,
     "models": [
      "gpt-6-luna"
     ],
     "file": "answers/openai-gpt-6-luna-effort-none/proofwriter-cwa.jsonl.gz"
    }
   },
   "latency": {
    "jev": {
     "file": "timing/jev/proofwriter-cwa.jsonl.gz",
     "n": 1800,
     "of": 1800,
     "complete": true,
     "p50": 170.44,
     "p90": 206.71,
     "max": 559.28,
     "total_minutes": 5.270140166666673,
     "concurrency": 1,
     "machine": {
      "cpu": "Apple M1 Max",
      "memory_bytes": 34359738368,
      "model": "MacBookPro18,4",
      "platform": "macOS-26.6.2-arm64-arm-64bit",
      "python": "3.12.2"
     },
     "load_start": null,
     "load_end": null
    },
    "kev-0.8b": {
     "file": "timing/kev-0.8b/proofwriter-cwa.jsonl.gz",
     "n": 1800,
     "of": 1800,
     "complete": true,
     "p50": 120.61,
     "p90": 145.06,
     "max": 280.82,
     "total_minutes": 3.7212803333333326,
     "concurrency": 1,
     "machine": {
      "cpu": "Apple M1 Max",
      "memory_bytes": 34359738368,
      "model": "MacBookPro18,4",
      "platform": "macOS-26.6.2-arm64-arm-64bit",
      "python": "3.12.2"
     },
     "load_start": [
      5.2,
      5.56,
      6.33
     ],
     "load_end": [
      7.16,
      6.37,
      6.51
     ]
    },
    "kev-4b": {
     "file": "timing/kev-4b/proofwriter-cwa.jsonl.gz",
     "n": 1800,
     "of": 1800,
     "complete": true,
     "p50": 802.45,
     "p90": 1044.48,
     "max": 1481.46,
     "total_minutes": 24.99258833333333,
     "concurrency": 1,
     "machine": {
      "cpu": "Apple M1 Max",
      "memory_bytes": 34359738368,
      "model": "MacBookPro18,4",
      "platform": "macOS-26.6.2-arm64-arm-64bit",
      "python": "3.12.2"
     },
     "load_start": [
      11.14,
      8.15,
      7.02
     ],
     "load_end": [
      10.56,
      9.36,
      8.58
     ]
    },
    "laya": {
     "file": "timing/laya/proofwriter-cwa.jsonl.gz",
     "n": 1800,
     "of": 1800,
     "complete": true,
     "p50": 120.52,
     "p90": 398.2,
     "max": 6300.77,
     "total_minutes": 5.688432499999996,
     "concurrency": 1,
     "machine": {
      "cpu": "Apple M1 Max",
      "memory_bytes": 34359738368,
      "model": "MacBookPro18,4",
      "platform": "macOS-26.6.2-arm64-arm-64bit",
      "python": "3.12.2"
     },
     "load_start": [
      7.12,
      8.46,
      8.5
     ],
     "load_end": [
      9.58,
      9.01,
      8.72
     ]
    },
    "openai-gpt-6-luna-effort-none": {
     "file": "timing/openai-gpt-6-luna-effort-none/proofwriter-cwa.jsonl.gz",
     "n": 1800,
     "of": 1800,
     "complete": true,
     "p50": 928.78,
     "p90": 1268.14,
     "max": 10024.71,
     "total_minutes": 29.900497499999968,
     "concurrency": 1,
     "machine": {
      "cpu": "Apple M1 Max",
      "memory_bytes": 34359738368,
      "model": "MacBookPro18,4",
      "platform": "macOS-26.6.2-arm64-arm-64bit",
      "python": "3.12.2"
     },
     "load_start": [
      7.88,
      8.32,
      8.96
     ],
     "load_end": [
      4.64,
      5.36,
      6.57
     ]
    }
   }
  },
  "proofwriter-owa": {
   "studies": {
    "engines": {
     "jev": {
      "file": "studies/proofwriter-owa-jev.jsonl",
      "modified": "2026-10-01T16:36:32.024Z",
      "overall": {
       "n": 1800,
       "correct": 1509,
       "accuracy": 0.8383333333333334,
       "lo": 0.8222222222222222,
       "hi": 0.855,
       "macro_f1": 0.8377907517567103,
       "invalid": 0,
       "best_constant": 0.33611111111111114,
       "chance": 0.3333333333333333,
       "recall": {
        "false": 0.8826446280991735,
        "true": 0.8876033057851239,
        "unknown": 0.7423728813559322
       },
       "confusion": {
        "false": {
         "false": 534,
         "invalid": 0,
         "true": 10,
         "unknown": 61
        },
        "true": {
         "false": 5,
         "invalid": 0,
         "true": 537,
         "unknown": 63
        },
        "unknown": {
         "false": 28,
         "invalid": 0,
         "true": 124,
         "unknown": 438
        }
       },
       "model": [
        "jev-1.13.0"
       ]
      },
      "axes": {
       "depth": [
        {
         "value": "0",
         "n": 300,
         "correct": 293,
         "accuracy": 0.9766666666666667,
         "lo": 0.96,
         "hi": 0.9933333333333333,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 302,
         "correct": 265,
         "accuracy": 0.8774834437086093,
         "lo": 0.8377483443708609,
         "hi": 0.9139072847682119,
         "best_constant": 0.3344370860927152,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 303,
         "correct": 257,
         "accuracy": 0.8481848184818482,
         "lo": 0.801980198019802,
         "hi": 0.8844884488448845,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 303,
         "correct": 242,
         "accuracy": 0.7986798679867987,
         "lo": 0.7524752475247525,
         "hi": 0.8448844884488449,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 303,
         "correct": 218,
         "accuracy": 0.7194719471947195,
         "lo": 0.6666666666666666,
         "hi": 0.7722772277227723,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 289,
         "correct": 234,
         "accuracy": 0.8096885813148789,
         "lo": 0.7647058823529411,
         "hi": 0.8546712802768166,
         "best_constant": 0.3494809688581315,
         "invalid": 0
        }
       ],
       "reference_label": [
        {
         "value": "false",
         "n": 605,
         "correct": 534,
         "accuracy": 0.8826446280991735,
         "lo": 0.8545454545454545,
         "hi": 0.9090909090909091,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "true",
         "n": 605,
         "correct": 537,
         "accuracy": 0.8876033057851239,
         "lo": 0.8611570247933884,
         "hi": 0.9107438016528926,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "unknown",
         "n": 590,
         "correct": 438,
         "accuracy": 0.7423728813559322,
         "lo": 0.7067796610169491,
         "hi": 0.7762711864406779,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "theory_kind": [
        {
         "value": "attribute",
         "n": 1059,
         "correct": 877,
         "accuracy": 0.8281397544853636,
         "lo": 0.8054768649669499,
         "hi": 0.8517469310670444,
         "best_constant": 0.3380547686496695,
         "invalid": 0
        },
        {
         "value": "relation",
         "n": 741,
         "correct": 632,
         "accuracy": 0.8529014844804319,
         "lo": 0.8259109311740891,
         "hi": 0.8771929824561403,
         "best_constant": 0.3441295546558704,
         "invalid": 0
        }
       ],
       "theory_negation": [
        {
         "value": "negation",
         "n": 881,
         "correct": 771,
         "accuracy": 0.8751418842224744,
         "lo": 0.8535754824063564,
         "hi": 0.8967082860385925,
         "best_constant": 0.3507377979568672,
         "invalid": 0
        },
        {
         "value": "no-negation",
         "n": 919,
         "correct": 738,
         "accuracy": 0.8030467899891186,
         "lo": 0.7769314472252449,
         "hi": 0.8313384113166485,
         "best_constant": 0.35473340587595215,
         "invalid": 0
        }
       ],
       "statement_negated": [
        {
         "value": "False",
         "n": 899,
         "correct": 801,
         "accuracy": 0.8909899888765295,
         "lo": 0.8709677419354839,
         "hi": 0.9098998887652948,
         "best_constant": 0.41490545050055616,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 901,
         "correct": 708,
         "accuracy": 0.7857935627081021,
         "lo": 0.7591564927857936,
         "hi": 0.8124306326304107,
         "best_constant": 0.41065482796892344,
         "invalid": 0
        }
       ],
       "strategy": [
        {
         "value": "inv-proof",
         "n": 605,
         "correct": 534,
         "accuracy": 0.8826446280991735,
         "lo": 0.8545454545454545,
         "hi": 0.9090909090909091,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-random",
         "n": 50,
         "correct": 47,
         "accuracy": 0.94,
         "lo": 0.86,
         "hi": 1,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-rconc",
         "n": 249,
         "correct": 125,
         "accuracy": 0.5020080321285141,
         "lo": 0.43775100401606426,
         "hi": 0.5662650602409639,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "proof",
         "n": 605,
         "correct": 537,
         "accuracy": 0.8876033057851239,
         "lo": 0.8611570247933884,
         "hi": 0.9107438016528926,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "random",
         "n": 50,
         "correct": 49,
         "accuracy": 0.98,
         "lo": 0.94,
         "hi": 1,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "rconc",
         "n": 241,
         "correct": 217,
         "accuracy": 0.9004149377593361,
         "lo": 0.8589211618257261,
         "hi": 0.9377593360995851,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "paraphrased": [
        {
         "value": "False",
         "n": 1298,
         "correct": 1117,
         "accuracy": 0.8605546995377504,
         "lo": 0.8420647149460708,
         "hi": 0.8790446841294299,
         "best_constant": 0.34514637904468415,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 502,
         "correct": 392,
         "accuracy": 0.7808764940239044,
         "lo": 0.7450199203187251,
         "hi": 0.8207171314741036,
         "best_constant": 0.35856573705179284,
         "invalid": 0
        }
       ],
       "theory_max_depth": [
        {
         "value": "0",
         "n": 71,
         "correct": 64,
         "accuracy": 0.9014084507042254,
         "lo": 0.8309859154929577,
         "hi": 0.971830985915493,
         "best_constant": 0.6056338028169014,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 145,
         "correct": 128,
         "accuracy": 0.8827586206896552,
         "lo": 0.8275862068965517,
         "hi": 0.9379310344827586,
         "best_constant": 0.43448275862068964,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 206,
         "correct": 178,
         "accuracy": 0.8640776699029126,
         "lo": 0.8155339805825242,
         "hi": 0.912621359223301,
         "best_constant": 0.3932038834951456,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 620,
         "correct": 523,
         "accuracy": 0.8435483870967742,
         "lo": 0.8145161290322581,
         "hi": 0.8725806451612903,
         "best_constant": 0.3387096774193548,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 162,
         "correct": 110,
         "accuracy": 0.6790123456790124,
         "lo": 0.6111111111111112,
         "hi": 0.7530864197530864,
         "best_constant": 0.3765432098765432,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 571,
         "correct": 488,
         "accuracy": 0.8546409807355516,
         "lo": 0.8266199649737302,
         "hi": 0.8844133099824869,
         "best_constant": 0.38353765323992994,
         "invalid": 0
        },
        {
         "value": "6",
         "n": 18,
         "correct": 13,
         "accuracy": 0.7222222222222222,
         "lo": 0.5,
         "hi": 0.9444444444444444,
         "best_constant": 0.4444444444444444,
         "invalid": 0
        },
        {
         "value": "7",
         "n": 5,
         "correct": 3,
         "accuracy": 0.6,
         "lo": 0.2,
         "hi": 1,
         "best_constant": 0.6,
         "invalid": 0
        },
        {
         "value": "8",
         "n": 2,
         "correct": 2,
         "accuracy": 1,
         "lo": 1,
         "hi": 1,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "words_bin": [
        {
         "value": "0-49",
         "n": 247,
         "correct": 230,
         "accuracy": 0.9311740890688259,
         "lo": 0.8987854251012146,
         "hi": 0.9635627530364372,
         "best_constant": 0.3967611336032389,
         "invalid": 0
        },
        {
         "value": "50-79",
         "n": 416,
         "correct": 372,
         "accuracy": 0.8942307692307693,
         "lo": 0.8629807692307693,
         "hi": 0.9230769230769231,
         "best_constant": 0.36778846153846156,
         "invalid": 0
        },
        {
         "value": "80-109",
         "n": 519,
         "correct": 414,
         "accuracy": 0.7976878612716763,
         "lo": 0.7610789980732178,
         "hi": 0.8304431599229287,
         "best_constant": 0.34296724470134876,
         "invalid": 0
        },
        {
         "value": "110+",
         "n": 618,
         "correct": 493,
         "accuracy": 0.7977346278317152,
         "lo": 0.7669902912621359,
         "hi": 0.8300970873786407,
         "best_constant": 0.3705501618122977,
         "invalid": 0
        }
       ],
       "rules_bin": [
        {
         "value": "0-2",
         "n": 204,
         "correct": 188,
         "accuracy": 0.9215686274509803,
         "lo": 0.8823529411764706,
         "hi": 0.9558823529411765,
         "best_constant": 0.3382352941176471,
         "invalid": 0
        },
        {
         "value": "3-5",
         "n": 438,
         "correct": 367,
         "accuracy": 0.8378995433789954,
         "lo": 0.8036529680365296,
         "hi": 0.8698630136986302,
         "best_constant": 0.3835616438356164,
         "invalid": 0
        },
        {
         "value": "6-7",
         "n": 564,
         "correct": 479,
         "accuracy": 0.849290780141844,
         "lo": 0.8191489361702128,
         "hi": 0.8812056737588653,
         "best_constant": 0.3475177304964539,
         "invalid": 0
        },
        {
         "value": "8+",
         "n": 594,
         "correct": 475,
         "accuracy": 0.7996632996632996,
         "lo": 0.7643097643097643,
         "hi": 0.8316498316498316,
         "best_constant": 0.4074074074074074,
         "invalid": 0
        }
       ],
       "facts_bin": [
        {
         "value": "0-3",
         "n": 334,
         "correct": 302,
         "accuracy": 0.9041916167664671,
         "lo": 0.8682634730538922,
         "hi": 0.9341317365269461,
         "best_constant": 0.4281437125748503,
         "invalid": 0
        },
        {
         "value": "4-7",
         "n": 435,
         "correct": 374,
         "accuracy": 0.8597701149425288,
         "lo": 0.825287356321839,
         "hi": 0.8896551724137931,
         "best_constant": 0.36091954022988504,
         "invalid": 0
        },
        {
         "value": "8-12",
         "n": 872,
         "correct": 700,
         "accuracy": 0.8027522935779816,
         "lo": 0.7752293577981652,
         "hi": 0.8291284403669725,
         "best_constant": 0.3577981651376147,
         "invalid": 0
        },
        {
         "value": "13+",
         "n": 159,
         "correct": 133,
         "accuracy": 0.8364779874213837,
         "lo": 0.779874213836478,
         "hi": 0.8930817610062893,
         "best_constant": 0.3836477987421384,
         "invalid": 0
        }
       ],
       "proof_size_bin": [
        {
         "value": "0-1",
         "n": 790,
         "correct": 635,
         "accuracy": 0.8037974683544303,
         "lo": 0.7746835443037975,
         "hi": 0.8278481012658228,
         "best_constant": 0.7468354430379747,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 107,
         "correct": 105,
         "accuracy": 0.9813084112149533,
         "lo": 0.9532710280373832,
         "hi": 1,
         "best_constant": 0.5046728971962616,
         "invalid": 0
        },
        {
         "value": "3-4",
         "n": 296,
         "correct": 270,
         "accuracy": 0.9121621621621622,
         "lo": 0.875,
         "hi": 0.9425675675675675,
         "best_constant": 0.5033783783783784,
         "invalid": 0
        },
        {
         "value": "5+",
         "n": 607,
         "correct": 499,
         "accuracy": 0.8220757825370676,
         "lo": 0.7907742998352554,
         "hi": 0.8517298187808896,
         "best_constant": 0.500823723228995,
         "invalid": 0
        }
       ]
      }
     },
     "kev-0.8b": {
      "file": "studies/proofwriter-owa-kev-0.8b.jsonl",
      "modified": "2026-10-01T16:36:33.362Z",
      "overall": {
       "n": 1800,
       "correct": 964,
       "accuracy": 0.5355555555555556,
       "lo": 0.5138888888888888,
       "hi": 0.5594444444444444,
       "macro_f1": 0.47651847094928895,
       "invalid": 0,
       "best_constant": 0.33611111111111114,
       "chance": 0.3333333333333333,
       "recall": {
        "false": 0.8958677685950414,
        "true": 0.5818181818181818,
        "unknown": 0.11864406779661017
       },
       "confusion": {
        "false": {
         "false": 542,
         "invalid": 0,
         "true": 59,
         "unknown": 4
        },
        "true": {
         "false": 209,
         "invalid": 0,
         "true": 352,
         "unknown": 44
        },
        "unknown": {
         "false": 370,
         "invalid": 0,
         "true": 150,
         "unknown": 70
        }
       },
       "model": [
        "kev-latest"
       ]
      },
      "axes": {
       "depth": [
        {
         "value": "0",
         "n": 300,
         "correct": 211,
         "accuracy": 0.7033333333333334,
         "lo": 0.6533333333333333,
         "hi": 0.7566666666666667,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 302,
         "correct": 163,
         "accuracy": 0.5397350993377483,
         "lo": 0.48344370860927155,
         "hi": 0.5960264900662252,
         "best_constant": 0.3344370860927152,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 303,
         "correct": 142,
         "accuracy": 0.46864686468646866,
         "lo": 0.41254125412541254,
         "hi": 0.528052805280528,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 303,
         "correct": 148,
         "accuracy": 0.4884488448844885,
         "lo": 0.43564356435643564,
         "hi": 0.5445544554455446,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 303,
         "correct": 153,
         "accuracy": 0.504950495049505,
         "lo": 0.44884488448844884,
         "hi": 0.5610561056105611,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 289,
         "correct": 147,
         "accuracy": 0.5086505190311419,
         "lo": 0.4532871972318339,
         "hi": 0.5674740484429066,
         "best_constant": 0.3494809688581315,
         "invalid": 0
        }
       ],
       "reference_label": [
        {
         "value": "false",
         "n": 605,
         "correct": 542,
         "accuracy": 0.8958677685950414,
         "lo": 0.8710743801652893,
         "hi": 0.9173553719008265,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "true",
         "n": 605,
         "correct": 352,
         "accuracy": 0.5818181818181818,
         "lo": 0.540495867768595,
         "hi": 0.6198347107438017,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "unknown",
         "n": 590,
         "correct": 70,
         "accuracy": 0.11864406779661017,
         "lo": 0.09322033898305085,
         "hi": 0.14576271186440679,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "theory_kind": [
        {
         "value": "attribute",
         "n": 1059,
         "correct": 570,
         "accuracy": 0.5382436260623229,
         "lo": 0.5070821529745042,
         "hi": 0.5656279508970727,
         "best_constant": 0.3380547686496695,
         "invalid": 0
        },
        {
         "value": "relation",
         "n": 741,
         "correct": 394,
         "accuracy": 0.5317139001349528,
         "lo": 0.4939271255060729,
         "hi": 0.5668016194331984,
         "best_constant": 0.3441295546558704,
         "invalid": 0
        }
       ],
       "theory_negation": [
        {
         "value": "negation",
         "n": 881,
         "correct": 499,
         "accuracy": 0.5664018161180476,
         "lo": 0.5357548240635641,
         "hi": 0.5959137343927355,
         "best_constant": 0.3507377979568672,
         "invalid": 0
        },
        {
         "value": "no-negation",
         "n": 919,
         "correct": 465,
         "accuracy": 0.5059847660500544,
         "lo": 0.4733405875952122,
         "hi": 0.5408052230685527,
         "best_constant": 0.35473340587595215,
         "invalid": 0
        }
       ],
       "statement_negated": [
        {
         "value": "False",
         "n": 899,
         "correct": 436,
         "accuracy": 0.4849833147942158,
         "lo": 0.45383759733036705,
         "hi": 0.5172413793103449,
         "best_constant": 0.41490545050055616,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 901,
         "correct": 528,
         "accuracy": 0.586015538290788,
         "lo": 0.5549389567147613,
         "hi": 0.6170921198668147,
         "best_constant": 0.41065482796892344,
         "invalid": 0
        }
       ],
       "strategy": [
        {
         "value": "inv-proof",
         "n": 605,
         "correct": 542,
         "accuracy": 0.8958677685950414,
         "lo": 0.8710743801652893,
         "hi": 0.9173553719008265,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-random",
         "n": 50,
         "correct": 11,
         "accuracy": 0.22,
         "lo": 0.12,
         "hi": 0.34,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-rconc",
         "n": 249,
         "correct": 2,
         "accuracy": 0.008032128514056224,
         "lo": 0,
         "hi": 0.020080321285140562,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "proof",
         "n": 605,
         "correct": 352,
         "accuracy": 0.5818181818181818,
         "lo": 0.540495867768595,
         "hi": 0.6198347107438017,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "random",
         "n": 50,
         "correct": 17,
         "accuracy": 0.34,
         "lo": 0.22,
         "hi": 0.46,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "rconc",
         "n": 241,
         "correct": 40,
         "accuracy": 0.16597510373443983,
         "lo": 0.12448132780082988,
         "hi": 0.2157676348547718,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "paraphrased": [
        {
         "value": "False",
         "n": 1298,
         "correct": 698,
         "accuracy": 0.5377503852080123,
         "lo": 0.5115562403697997,
         "hi": 0.5639445300462249,
         "best_constant": 0.34514637904468415,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 502,
         "correct": 266,
         "accuracy": 0.5298804780876494,
         "lo": 0.48804780876494025,
         "hi": 0.5717131474103586,
         "best_constant": 0.35856573705179284,
         "invalid": 0
        }
       ],
       "theory_max_depth": [
        {
         "value": "0",
         "n": 71,
         "correct": 30,
         "accuracy": 0.4225352112676056,
         "lo": 0.30985915492957744,
         "hi": 0.5352112676056338,
         "best_constant": 0.6056338028169014,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 145,
         "correct": 78,
         "accuracy": 0.5379310344827586,
         "lo": 0.46206896551724136,
         "hi": 0.6137931034482759,
         "best_constant": 0.43448275862068964,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 206,
         "correct": 101,
         "accuracy": 0.49029126213592233,
         "lo": 0.4223300970873786,
         "hi": 0.5631067961165048,
         "best_constant": 0.3932038834951456,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 620,
         "correct": 332,
         "accuracy": 0.535483870967742,
         "lo": 0.4967741935483871,
         "hi": 0.5741935483870968,
         "best_constant": 0.3387096774193548,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 162,
         "correct": 83,
         "accuracy": 0.5123456790123457,
         "lo": 0.43209876543209874,
         "hi": 0.5864197530864198,
         "best_constant": 0.3765432098765432,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 571,
         "correct": 331,
         "accuracy": 0.5796847635726795,
         "lo": 0.5376532399299475,
         "hi": 0.6182136602451839,
         "best_constant": 0.38353765323992994,
         "invalid": 0
        },
        {
         "value": "6",
         "n": 18,
         "correct": 7,
         "accuracy": 0.3888888888888889,
         "lo": 0.16666666666666666,
         "hi": 0.6666666666666666,
         "best_constant": 0.4444444444444444,
         "invalid": 0
        },
        {
         "value": "7",
         "n": 5,
         "correct": 1,
         "accuracy": 0.2,
         "lo": 0,
         "hi": 0.6,
         "best_constant": 0.6,
         "invalid": 0
        },
        {
         "value": "8",
         "n": 2,
         "correct": 1,
         "accuracy": 0.5,
         "lo": 0,
         "hi": 1,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "words_bin": [
        {
         "value": "0-49",
         "n": 247,
         "correct": 139,
         "accuracy": 0.562753036437247,
         "lo": 0.4979757085020243,
         "hi": 0.6234817813765182,
         "best_constant": 0.3967611336032389,
         "invalid": 0
        },
        {
         "value": "50-79",
         "n": 416,
         "correct": 227,
         "accuracy": 0.5456730769230769,
         "lo": 0.49759615384615385,
         "hi": 0.59375,
         "best_constant": 0.36778846153846156,
         "invalid": 0
        },
        {
         "value": "80-109",
         "n": 519,
         "correct": 273,
         "accuracy": 0.5260115606936416,
         "lo": 0.48554913294797686,
         "hi": 0.5664739884393064,
         "best_constant": 0.34296724470134876,
         "invalid": 0
        },
        {
         "value": "110+",
         "n": 618,
         "correct": 325,
         "accuracy": 0.5258899676375405,
         "lo": 0.48705501618122976,
         "hi": 0.5647249190938511,
         "best_constant": 0.3705501618122977,
         "invalid": 0
        }
       ],
       "rules_bin": [
        {
         "value": "0-2",
         "n": 204,
         "correct": 120,
         "accuracy": 0.5882352941176471,
         "lo": 0.5196078431372549,
         "hi": 0.6519607843137255,
         "best_constant": 0.3382352941176471,
         "invalid": 0
        },
        {
         "value": "3-5",
         "n": 438,
         "correct": 253,
         "accuracy": 0.5776255707762558,
         "lo": 0.5319634703196348,
         "hi": 0.6232876712328768,
         "best_constant": 0.3835616438356164,
         "invalid": 0
        },
        {
         "value": "6-7",
         "n": 564,
         "correct": 295,
         "accuracy": 0.5230496453900709,
         "lo": 0.48226950354609927,
         "hi": 0.5638297872340425,
         "best_constant": 0.3475177304964539,
         "invalid": 0
        },
        {
         "value": "8+",
         "n": 594,
         "correct": 296,
         "accuracy": 0.4983164983164983,
         "lo": 0.4595959595959596,
         "hi": 0.5387205387205387,
         "best_constant": 0.4074074074074074,
         "invalid": 0
        }
       ],
       "facts_bin": [
        {
         "value": "0-3",
         "n": 334,
         "correct": 175,
         "accuracy": 0.5239520958083832,
         "lo": 0.47005988023952094,
         "hi": 0.5778443113772455,
         "best_constant": 0.4281437125748503,
         "invalid": 0
        },
        {
         "value": "4-7",
         "n": 435,
         "correct": 242,
         "accuracy": 0.5563218390804597,
         "lo": 0.5103448275862069,
         "hi": 0.6022988505747127,
         "best_constant": 0.36091954022988504,
         "invalid": 0
        },
        {
         "value": "8-12",
         "n": 872,
         "correct": 457,
         "accuracy": 0.5240825688073395,
         "lo": 0.4908256880733945,
         "hi": 0.5573394495412844,
         "best_constant": 0.3577981651376147,
         "invalid": 0
        },
        {
         "value": "13+",
         "n": 159,
         "correct": 90,
         "accuracy": 0.5660377358490566,
         "lo": 0.48427672955974843,
         "hi": 0.6415094339622641,
         "best_constant": 0.3836477987421384,
         "invalid": 0
        }
       ],
       "proof_size_bin": [
        {
         "value": "0-1",
         "n": 790,
         "correct": 253,
         "accuracy": 0.320253164556962,
         "lo": 0.28860759493670884,
         "hi": 0.35063291139240504,
         "best_constant": 0.7468354430379747,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 107,
         "correct": 83,
         "accuracy": 0.7757009345794392,
         "lo": 0.6915887850467289,
         "hi": 0.8504672897196262,
         "best_constant": 0.5046728971962616,
         "invalid": 0
        },
        {
         "value": "3-4",
         "n": 296,
         "correct": 207,
         "accuracy": 0.6993243243243243,
         "lo": 0.6486486486486487,
         "hi": 0.7533783783783784,
         "best_constant": 0.5033783783783784,
         "invalid": 0
        },
        {
         "value": "5+",
         "n": 607,
         "correct": 421,
         "accuracy": 0.6935749588138386,
         "lo": 0.6589785831960461,
         "hi": 0.729818780889621,
         "best_constant": 0.500823723228995,
         "invalid": 0
        }
       ]
      }
     },
     "kev-4b": {
      "file": "studies/proofwriter-owa-kev-4b.jsonl",
      "modified": "2026-10-01T16:36:34.693Z",
      "overall": {
       "n": 1800,
       "correct": 964,
       "accuracy": 0.5355555555555556,
       "lo": 0.5138888888888888,
       "hi": 0.56,
       "macro_f1": 0.5318093549140318,
       "invalid": 0,
       "best_constant": 0.33611111111111114,
       "chance": 0.3333333333333333,
       "recall": {
        "false": 0.3586776859504132,
        "true": 0.46611570247933887,
        "unknown": 0.788135593220339
       },
       "confusion": {
        "false": {
         "false": 217,
         "invalid": 0,
         "true": 45,
         "unknown": 343
        },
        "true": {
         "false": 12,
         "invalid": 0,
         "true": 282,
         "unknown": 311
        },
        "unknown": {
         "false": 36,
         "invalid": 0,
         "true": 89,
         "unknown": 465
        }
       },
       "model": [
        "kev-latest"
       ]
      },
      "axes": {
       "depth": [
        {
         "value": "0",
         "n": 300,
         "correct": 263,
         "accuracy": 0.8766666666666667,
         "lo": 0.84,
         "hi": 0.9133333333333333,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 302,
         "correct": 190,
         "accuracy": 0.6291390728476821,
         "lo": 0.5728476821192053,
         "hi": 0.6821192052980133,
         "best_constant": 0.3344370860927152,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 303,
         "correct": 158,
         "accuracy": 0.5214521452145214,
         "lo": 0.46534653465346537,
         "hi": 0.5775577557755776,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 303,
         "correct": 135,
         "accuracy": 0.44554455445544555,
         "lo": 0.3927392739273927,
         "hi": 0.49834983498349833,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 303,
         "correct": 113,
         "accuracy": 0.37293729372937295,
         "lo": 0.31683168316831684,
         "hi": 0.42244224422442245,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 289,
         "correct": 105,
         "accuracy": 0.3633217993079585,
         "lo": 0.31141868512110726,
         "hi": 0.4186851211072664,
         "best_constant": 0.3494809688581315,
         "invalid": 0
        }
       ],
       "reference_label": [
        {
         "value": "false",
         "n": 605,
         "correct": 217,
         "accuracy": 0.3586776859504132,
         "lo": 0.31900826446280994,
         "hi": 0.3950413223140496,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "true",
         "n": 605,
         "correct": 282,
         "accuracy": 0.46611570247933887,
         "lo": 0.428099173553719,
         "hi": 0.5057851239669422,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "unknown",
         "n": 590,
         "correct": 465,
         "accuracy": 0.788135593220339,
         "lo": 0.7542372881355932,
         "hi": 0.8186440677966101,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "theory_kind": [
        {
         "value": "attribute",
         "n": 1059,
         "correct": 576,
         "accuracy": 0.5439093484419264,
         "lo": 0.5136921624173749,
         "hi": 0.5722379603399433,
         "best_constant": 0.3380547686496695,
         "invalid": 0
        },
        {
         "value": "relation",
         "n": 741,
         "correct": 388,
         "accuracy": 0.5236167341430499,
         "lo": 0.4898785425101215,
         "hi": 0.5587044534412956,
         "best_constant": 0.3441295546558704,
         "invalid": 0
        }
       ],
       "theory_negation": [
        {
         "value": "negation",
         "n": 881,
         "correct": 505,
         "accuracy": 0.5732122587968218,
         "lo": 0.5402951191827469,
         "hi": 0.6072644721906924,
         "best_constant": 0.3507377979568672,
         "invalid": 0
        },
        {
         "value": "no-negation",
         "n": 919,
         "correct": 459,
         "accuracy": 0.499455930359086,
         "lo": 0.4689880304678999,
         "hi": 0.5310119695321001,
         "best_constant": 0.35473340587595215,
         "invalid": 0
        }
       ],
       "statement_negated": [
        {
         "value": "False",
         "n": 899,
         "correct": 482,
         "accuracy": 0.5361512791991101,
         "lo": 0.5072302558398221,
         "hi": 0.5661846496106785,
         "best_constant": 0.41490545050055616,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 901,
         "correct": 482,
         "accuracy": 0.5349611542730299,
         "lo": 0.5016648168701443,
         "hi": 0.5682574916759157,
         "best_constant": 0.41065482796892344,
         "invalid": 0
        }
       ],
       "strategy": [
        {
         "value": "inv-proof",
         "n": 605,
         "correct": 217,
         "accuracy": 0.3586776859504132,
         "lo": 0.31900826446280994,
         "hi": 0.3950413223140496,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-random",
         "n": 50,
         "correct": 41,
         "accuracy": 0.82,
         "lo": 0.7,
         "hi": 0.92,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-rconc",
         "n": 249,
         "correct": 180,
         "accuracy": 0.7228915662650602,
         "lo": 0.6626506024096386,
         "hi": 0.7751004016064257,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "proof",
         "n": 605,
         "correct": 282,
         "accuracy": 0.46611570247933887,
         "lo": 0.428099173553719,
         "hi": 0.5057851239669422,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "random",
         "n": 50,
         "correct": 44,
         "accuracy": 0.88,
         "lo": 0.78,
         "hi": 0.96,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "rconc",
         "n": 241,
         "correct": 200,
         "accuracy": 0.8298755186721992,
         "lo": 0.7800829875518672,
         "hi": 0.8755186721991701,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "paraphrased": [
        {
         "value": "False",
         "n": 1298,
         "correct": 725,
         "accuracy": 0.5585516178736518,
         "lo": 0.5315870570107858,
         "hi": 0.5862865947611711,
         "best_constant": 0.34514637904468415,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 502,
         "correct": 239,
         "accuracy": 0.4760956175298805,
         "lo": 0.4342629482071713,
         "hi": 0.5199203187250996,
         "best_constant": 0.35856573705179284,
         "invalid": 0
        }
       ],
       "theory_max_depth": [
        {
         "value": "0",
         "n": 71,
         "correct": 61,
         "accuracy": 0.8591549295774648,
         "lo": 0.7746478873239436,
         "hi": 0.9295774647887324,
         "best_constant": 0.6056338028169014,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 145,
         "correct": 111,
         "accuracy": 0.7655172413793103,
         "lo": 0.696551724137931,
         "hi": 0.8275862068965517,
         "best_constant": 0.43448275862068964,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 206,
         "correct": 144,
         "accuracy": 0.6990291262135923,
         "lo": 0.6359223300970874,
         "hi": 0.7572815533980582,
         "best_constant": 0.3932038834951456,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 620,
         "correct": 357,
         "accuracy": 0.5758064516129032,
         "lo": 0.5370967741935484,
         "hi": 0.6129032258064516,
         "best_constant": 0.3387096774193548,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 162,
         "correct": 63,
         "accuracy": 0.3888888888888889,
         "lo": 0.32098765432098764,
         "hi": 0.46296296296296297,
         "best_constant": 0.3765432098765432,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 571,
         "correct": 217,
         "accuracy": 0.38003502626970226,
         "lo": 0.3362521891418564,
         "hi": 0.4220665499124343,
         "best_constant": 0.38353765323992994,
         "invalid": 0
        },
        {
         "value": "6",
         "n": 18,
         "correct": 6,
         "accuracy": 0.3333333333333333,
         "lo": 0.1111111111111111,
         "hi": 0.5555555555555556,
         "best_constant": 0.4444444444444444,
         "invalid": 0
        },
        {
         "value": "7",
         "n": 5,
         "correct": 3,
         "accuracy": 0.6,
         "lo": 0.2,
         "hi": 1,
         "best_constant": 0.6,
         "invalid": 0
        },
        {
         "value": "8",
         "n": 2,
         "correct": 2,
         "accuracy": 1,
         "lo": 1,
         "hi": 1,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "words_bin": [
        {
         "value": "0-49",
         "n": 247,
         "correct": 187,
         "accuracy": 0.757085020242915,
         "lo": 0.7044534412955465,
         "hi": 0.8097165991902834,
         "best_constant": 0.3967611336032389,
         "invalid": 0
        },
        {
         "value": "50-79",
         "n": 416,
         "correct": 258,
         "accuracy": 0.6201923076923077,
         "lo": 0.5697115384615384,
         "hi": 0.6634615384615384,
         "best_constant": 0.36778846153846156,
         "invalid": 0
        },
        {
         "value": "80-109",
         "n": 519,
         "correct": 261,
         "accuracy": 0.5028901734104047,
         "lo": 0.45664739884393063,
         "hi": 0.5472061657032755,
         "best_constant": 0.34296724470134876,
         "invalid": 0
        },
        {
         "value": "110+",
         "n": 618,
         "correct": 258,
         "accuracy": 0.4174757281553398,
         "lo": 0.3802588996763754,
         "hi": 0.4563106796116505,
         "best_constant": 0.3705501618122977,
         "invalid": 0
        }
       ],
       "rules_bin": [
        {
         "value": "0-2",
         "n": 204,
         "correct": 161,
         "accuracy": 0.7892156862745098,
         "lo": 0.7352941176470589,
         "hi": 0.8431372549019608,
         "best_constant": 0.3382352941176471,
         "invalid": 0
        },
        {
         "value": "3-5",
         "n": 438,
         "correct": 201,
         "accuracy": 0.4589041095890411,
         "lo": 0.408675799086758,
         "hi": 0.502283105022831,
         "best_constant": 0.3835616438356164,
         "invalid": 0
        },
        {
         "value": "6-7",
         "n": 564,
         "correct": 279,
         "accuracy": 0.4946808510638298,
         "lo": 0.45567375886524825,
         "hi": 0.5372340425531915,
         "best_constant": 0.3475177304964539,
         "invalid": 0
        },
        {
         "value": "8+",
         "n": 594,
         "correct": 323,
         "accuracy": 0.5437710437710438,
         "lo": 0.5084175084175084,
         "hi": 0.5858585858585859,
         "best_constant": 0.4074074074074074,
         "invalid": 0
        }
       ],
       "facts_bin": [
        {
         "value": "0-3",
         "n": 334,
         "correct": 233,
         "accuracy": 0.6976047904191617,
         "lo": 0.6526946107784432,
         "hi": 0.7455089820359282,
         "best_constant": 0.4281437125748503,
         "invalid": 0
        },
        {
         "value": "4-7",
         "n": 435,
         "correct": 258,
         "accuracy": 0.593103448275862,
         "lo": 0.5471264367816092,
         "hi": 0.6413793103448275,
         "best_constant": 0.36091954022988504,
         "invalid": 0
        },
        {
         "value": "8-12",
         "n": 872,
         "correct": 409,
         "accuracy": 0.4690366972477064,
         "lo": 0.43577981651376146,
         "hi": 0.4988532110091743,
         "best_constant": 0.3577981651376147,
         "invalid": 0
        },
        {
         "value": "13+",
         "n": 159,
         "correct": 64,
         "accuracy": 0.4025157232704403,
         "lo": 0.3270440251572327,
         "hi": 0.48427672955974843,
         "best_constant": 0.3836477987421384,
         "invalid": 0
        }
       ],
       "proof_size_bin": [
        {
         "value": "0-1",
         "n": 790,
         "correct": 643,
         "accuracy": 0.8139240506329114,
         "lo": 0.7886075949367088,
         "hi": 0.8417721518987342,
         "best_constant": 0.7468354430379747,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 107,
         "correct": 74,
         "accuracy": 0.6915887850467289,
         "lo": 0.5981308411214953,
         "hi": 0.7757009345794392,
         "best_constant": 0.5046728971962616,
         "invalid": 0
        },
        {
         "value": "3-4",
         "n": 296,
         "correct": 124,
         "accuracy": 0.4189189189189189,
         "lo": 0.36824324324324326,
         "hi": 0.47297297297297297,
         "best_constant": 0.5033783783783784,
         "invalid": 0
        },
        {
         "value": "5+",
         "n": 607,
         "correct": 123,
         "accuracy": 0.2026359143327842,
         "lo": 0.17298187808896212,
         "hi": 0.23393739703459637,
         "best_constant": 0.500823723228995,
         "invalid": 0
        }
       ]
      }
     },
     "kev-9b": {
      "file": "studies/proofwriter-owa-kev-9b.jsonl",
      "modified": "2026-10-01T16:36:36.025Z",
      "overall": {
       "n": 1800,
       "correct": 1054,
       "accuracy": 0.5855555555555556,
       "lo": 0.5633333333333334,
       "hi": 0.6094444444444445,
       "macro_f1": 0.5879600159985933,
       "invalid": 0,
       "best_constant": 0.33611111111111114,
       "chance": 0.3333333333333333,
       "recall": {
        "false": 0.4859504132231405,
        "true": 0.5917355371900826,
        "unknown": 0.6813559322033899
       },
       "confusion": {
        "false": {
         "false": 294,
         "invalid": 0,
         "true": 74,
         "unknown": 237
        },
        "true": {
         "false": 37,
         "invalid": 0,
         "true": 358,
         "unknown": 210
        },
        "unknown": {
         "false": 57,
         "invalid": 0,
         "true": 131,
         "unknown": 402
        }
       },
       "model": [
        "kev-latest"
       ]
      },
      "axes": {
       "depth": [
        {
         "value": "0",
         "n": 300,
         "correct": 268,
         "accuracy": 0.8933333333333333,
         "lo": 0.8566666666666667,
         "hi": 0.9266666666666666,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 302,
         "correct": 223,
         "accuracy": 0.7384105960264901,
         "lo": 0.6887417218543046,
         "hi": 0.7880794701986755,
         "best_constant": 0.3344370860927152,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 303,
         "correct": 177,
         "accuracy": 0.5841584158415841,
         "lo": 0.5313531353135313,
         "hi": 0.636963696369637,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 303,
         "correct": 150,
         "accuracy": 0.49504950495049505,
         "lo": 0.44224422442244227,
         "hi": 0.5544554455445545,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 303,
         "correct": 117,
         "accuracy": 0.38613861386138615,
         "lo": 0.3333333333333333,
         "hi": 0.43564356435643564,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 289,
         "correct": 119,
         "accuracy": 0.4117647058823529,
         "lo": 0.35294117647058826,
         "hi": 0.4671280276816609,
         "best_constant": 0.3494809688581315,
         "invalid": 0
        }
       ],
       "reference_label": [
        {
         "value": "false",
         "n": 605,
         "correct": 294,
         "accuracy": 0.4859504132231405,
         "lo": 0.4462809917355372,
         "hi": 0.5256198347107438,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "true",
         "n": 605,
         "correct": 358,
         "accuracy": 0.5917355371900826,
         "lo": 0.5537190082644629,
         "hi": 0.631404958677686,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "unknown",
         "n": 590,
         "correct": 402,
         "accuracy": 0.6813559322033899,
         "lo": 0.6457627118644068,
         "hi": 0.7169491525423729,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "theory_kind": [
        {
         "value": "attribute",
         "n": 1059,
         "correct": 628,
         "accuracy": 0.5930122757318225,
         "lo": 0.5637393767705382,
         "hi": 0.623229461756374,
         "best_constant": 0.3380547686496695,
         "invalid": 0
        },
        {
         "value": "relation",
         "n": 741,
         "correct": 426,
         "accuracy": 0.5748987854251012,
         "lo": 0.5384615384615384,
         "hi": 0.6099865047233468,
         "best_constant": 0.3441295546558704,
         "invalid": 0
        }
       ],
       "theory_negation": [
        {
         "value": "negation",
         "n": 881,
         "correct": 532,
         "accuracy": 0.6038592508513053,
         "lo": 0.5709421112372304,
         "hi": 0.637911464245176,
         "best_constant": 0.3507377979568672,
         "invalid": 0
        },
        {
         "value": "no-negation",
         "n": 919,
         "correct": 522,
         "accuracy": 0.5680087051142546,
         "lo": 0.5353645266594124,
         "hi": 0.5984766050054406,
         "best_constant": 0.35473340587595215,
         "invalid": 0
        }
       ],
       "statement_negated": [
        {
         "value": "False",
         "n": 899,
         "correct": 537,
         "accuracy": 0.5973303670745272,
         "lo": 0.5639599555061179,
         "hi": 0.6295884315906563,
         "best_constant": 0.41490545050055616,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 901,
         "correct": 517,
         "accuracy": 0.5738068812430632,
         "lo": 0.5382907880133185,
         "hi": 0.6037735849056604,
         "best_constant": 0.41065482796892344,
         "invalid": 0
        }
       ],
       "strategy": [
        {
         "value": "inv-proof",
         "n": 605,
         "correct": 294,
         "accuracy": 0.4859504132231405,
         "lo": 0.4462809917355372,
         "hi": 0.5256198347107438,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-random",
         "n": 50,
         "correct": 38,
         "accuracy": 0.76,
         "lo": 0.64,
         "hi": 0.86,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-rconc",
         "n": 249,
         "correct": 156,
         "accuracy": 0.6265060240963856,
         "lo": 0.5622489959839357,
         "hi": 0.6827309236947792,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "proof",
         "n": 605,
         "correct": 358,
         "accuracy": 0.5917355371900826,
         "lo": 0.5537190082644629,
         "hi": 0.631404958677686,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "random",
         "n": 50,
         "correct": 39,
         "accuracy": 0.78,
         "lo": 0.66,
         "hi": 0.88,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "rconc",
         "n": 241,
         "correct": 169,
         "accuracy": 0.7012448132780082,
         "lo": 0.6473029045643154,
         "hi": 0.7593360995850622,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "paraphrased": [
        {
         "value": "False",
         "n": 1298,
         "correct": 774,
         "accuracy": 0.5963020030816641,
         "lo": 0.5701078582434514,
         "hi": 0.6217257318952234,
         "best_constant": 0.34514637904468415,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 502,
         "correct": 280,
         "accuracy": 0.5577689243027888,
         "lo": 0.5179282868525896,
         "hi": 0.599601593625498,
         "best_constant": 0.35856573705179284,
         "invalid": 0
        }
       ],
       "theory_max_depth": [
        {
         "value": "0",
         "n": 71,
         "correct": 62,
         "accuracy": 0.8732394366197183,
         "lo": 0.7887323943661971,
         "hi": 0.9436619718309859,
         "best_constant": 0.6056338028169014,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 145,
         "correct": 118,
         "accuracy": 0.8137931034482758,
         "lo": 0.7448275862068966,
         "hi": 0.8758620689655172,
         "best_constant": 0.43448275862068964,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 206,
         "correct": 142,
         "accuracy": 0.6893203883495146,
         "lo": 0.6213592233009708,
         "hi": 0.7475728155339806,
         "best_constant": 0.3932038834951456,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 620,
         "correct": 399,
         "accuracy": 0.6435483870967742,
         "lo": 0.603225806451613,
         "hi": 0.6806451612903226,
         "best_constant": 0.3387096774193548,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 162,
         "correct": 69,
         "accuracy": 0.42592592592592593,
         "lo": 0.35185185185185186,
         "hi": 0.5,
         "best_constant": 0.3765432098765432,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 571,
         "correct": 253,
         "accuracy": 0.44308231173380036,
         "lo": 0.4010507880910683,
         "hi": 0.4816112084063047,
         "best_constant": 0.38353765323992994,
         "invalid": 0
        },
        {
         "value": "6",
         "n": 18,
         "correct": 8,
         "accuracy": 0.4444444444444444,
         "lo": 0.2222222222222222,
         "hi": 0.6666666666666666,
         "best_constant": 0.4444444444444444,
         "invalid": 0
        },
        {
         "value": "7",
         "n": 5,
         "correct": 1,
         "accuracy": 0.2,
         "lo": 0,
         "hi": 0.6,
         "best_constant": 0.6,
         "invalid": 0
        },
        {
         "value": "8",
         "n": 2,
         "correct": 2,
         "accuracy": 1,
         "lo": 1,
         "hi": 1,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "words_bin": [
        {
         "value": "0-49",
         "n": 247,
         "correct": 192,
         "accuracy": 0.7773279352226721,
         "lo": 0.7246963562753036,
         "hi": 0.8259109311740891,
         "best_constant": 0.3967611336032389,
         "invalid": 0
        },
        {
         "value": "50-79",
         "n": 416,
         "correct": 263,
         "accuracy": 0.6322115384615384,
         "lo": 0.5865384615384616,
         "hi": 0.6802884615384616,
         "best_constant": 0.36778846153846156,
         "invalid": 0
        },
        {
         "value": "80-109",
         "n": 519,
         "correct": 303,
         "accuracy": 0.5838150289017341,
         "lo": 0.5414258188824663,
         "hi": 0.626204238921002,
         "best_constant": 0.34296724470134876,
         "invalid": 0
        },
        {
         "value": "110+",
         "n": 618,
         "correct": 296,
         "accuracy": 0.47896440129449835,
         "lo": 0.4401294498381877,
         "hi": 0.5194174757281553,
         "best_constant": 0.3705501618122977,
         "invalid": 0
        }
       ],
       "rules_bin": [
        {
         "value": "0-2",
         "n": 204,
         "correct": 161,
         "accuracy": 0.7892156862745098,
         "lo": 0.7352941176470589,
         "hi": 0.8480392156862745,
         "best_constant": 0.3382352941176471,
         "invalid": 0
        },
        {
         "value": "3-5",
         "n": 438,
         "correct": 226,
         "accuracy": 0.5159817351598174,
         "lo": 0.4657534246575342,
         "hi": 0.5616438356164384,
         "best_constant": 0.3835616438356164,
         "invalid": 0
        },
        {
         "value": "6-7",
         "n": 564,
         "correct": 321,
         "accuracy": 0.5691489361702128,
         "lo": 0.5301418439716312,
         "hi": 0.6081560283687943,
         "best_constant": 0.3475177304964539,
         "invalid": 0
        },
        {
         "value": "8+",
         "n": 594,
         "correct": 346,
         "accuracy": 0.5824915824915825,
         "lo": 0.5454545454545454,
         "hi": 0.6212121212121212,
         "best_constant": 0.4074074074074074,
         "invalid": 0
        }
       ],
       "facts_bin": [
        {
         "value": "0-3",
         "n": 334,
         "correct": 240,
         "accuracy": 0.718562874251497,
         "lo": 0.6676646706586826,
         "hi": 0.7664670658682635,
         "best_constant": 0.4281437125748503,
         "invalid": 0
        },
        {
         "value": "4-7",
         "n": 435,
         "correct": 269,
         "accuracy": 0.6183908045977011,
         "lo": 0.5747126436781609,
         "hi": 0.664367816091954,
         "best_constant": 0.36091954022988504,
         "invalid": 0
        },
        {
         "value": "8-12",
         "n": 872,
         "correct": 471,
         "accuracy": 0.5401376146788991,
         "lo": 0.5080275229357798,
         "hi": 0.5711009174311926,
         "best_constant": 0.3577981651376147,
         "invalid": 0
        },
        {
         "value": "13+",
         "n": 159,
         "correct": 74,
         "accuracy": 0.46540880503144655,
         "lo": 0.389937106918239,
         "hi": 0.5408805031446541,
         "best_constant": 0.3836477987421384,
         "invalid": 0
        }
       ],
       "proof_size_bin": [
        {
         "value": "0-1",
         "n": 790,
         "correct": 593,
         "accuracy": 0.7506329113924051,
         "lo": 0.720253164556962,
         "hi": 0.7810126582278482,
         "best_constant": 0.7468354430379747,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 107,
         "correct": 90,
         "accuracy": 0.8411214953271028,
         "lo": 0.7663551401869159,
         "hi": 0.9065420560747663,
         "best_constant": 0.5046728971962616,
         "invalid": 0
        },
        {
         "value": "3-4",
         "n": 296,
         "correct": 178,
         "accuracy": 0.6013513513513513,
         "lo": 0.543918918918919,
         "hi": 0.6554054054054054,
         "best_constant": 0.5033783783783784,
         "invalid": 0
        },
        {
         "value": "5+",
         "n": 607,
         "correct": 193,
         "accuracy": 0.31795716639209226,
         "lo": 0.2833607907742998,
         "hi": 0.3542009884678748,
         "best_constant": 0.500823723228995,
         "invalid": 0
        }
       ]
      }
     },
     "laya": {
      "file": "studies/proofwriter-owa-laya.jsonl",
      "modified": "2026-10-01T16:36:37.381Z",
      "overall": {
       "n": 1800,
       "correct": 764,
       "accuracy": 0.42444444444444446,
       "lo": 0.4027777777777778,
       "hi": 0.445,
       "macro_f1": 0.33666800340747693,
       "invalid": 0,
       "best_constant": 0.33611111111111114,
       "chance": 0.3333333333333333,
       "recall": {
        "false": 0.8512396694214877,
        "true": 0.40165289256198344,
        "unknown": 0.010169491525423728
       },
       "confusion": {
        "false": {
         "false": 515,
         "invalid": 0,
         "true": 89,
         "unknown": 1
        },
        "true": {
         "false": 354,
         "invalid": 0,
         "true": 243,
         "unknown": 8
        },
        "unknown": {
         "false": 457,
         "invalid": 0,
         "true": 127,
         "unknown": 6
        }
       },
       "model": [
        "laya-upstream:0.3.21"
       ]
      },
      "axes": {
       "depth": [
        {
         "value": "0",
         "n": 300,
         "correct": 188,
         "accuracy": 0.6266666666666667,
         "lo": 0.57,
         "hi": 0.6833333333333333,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 302,
         "correct": 127,
         "accuracy": 0.4205298013245033,
         "lo": 0.3675496688741722,
         "hi": 0.4768211920529801,
         "best_constant": 0.3344370860927152,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 303,
         "correct": 125,
         "accuracy": 0.41254125412541254,
         "lo": 0.35973597359735976,
         "hi": 0.46864686468646866,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 303,
         "correct": 116,
         "accuracy": 0.38283828382838286,
         "lo": 0.33003300330033003,
         "hi": 0.44554455445544555,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 303,
         "correct": 109,
         "accuracy": 0.35973597359735976,
         "lo": 0.3102310231023102,
         "hi": 0.4158415841584158,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 289,
         "correct": 99,
         "accuracy": 0.34256055363321797,
         "lo": 0.2906574394463668,
         "hi": 0.39792387543252594,
         "best_constant": 0.3494809688581315,
         "invalid": 0
        }
       ],
       "reference_label": [
        {
         "value": "false",
         "n": 605,
         "correct": 515,
         "accuracy": 0.8512396694214877,
         "lo": 0.8214876033057851,
         "hi": 0.8793388429752066,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "true",
         "n": 605,
         "correct": 243,
         "accuracy": 0.40165289256198344,
         "lo": 0.36198347107438017,
         "hi": 0.4413223140495868,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "unknown",
         "n": 590,
         "correct": 6,
         "accuracy": 0.010169491525423728,
         "lo": 0.003389830508474576,
         "hi": 0.01864406779661017,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "theory_kind": [
        {
         "value": "attribute",
         "n": 1059,
         "correct": 436,
         "accuracy": 0.4117091595845137,
         "lo": 0.380547686496695,
         "hi": 0.4409820585457979,
         "best_constant": 0.3380547686496695,
         "invalid": 0
        },
        {
         "value": "relation",
         "n": 741,
         "correct": 328,
         "accuracy": 0.4426450742240216,
         "lo": 0.4089068825910931,
         "hi": 0.4777327935222672,
         "best_constant": 0.3441295546558704,
         "invalid": 0
        }
       ],
       "theory_negation": [
        {
         "value": "negation",
         "n": 881,
         "correct": 407,
         "accuracy": 0.4619750283768445,
         "lo": 0.4301929625425653,
         "hi": 0.4948921679909194,
         "best_constant": 0.3507377979568672,
         "invalid": 0
        },
        {
         "value": "no-negation",
         "n": 919,
         "correct": 357,
         "accuracy": 0.38846572361262244,
         "lo": 0.35473340587595215,
         "hi": 0.41893362350380847,
         "best_constant": 0.35473340587595215,
         "invalid": 0
        }
       ],
       "statement_negated": [
        {
         "value": "False",
         "n": 899,
         "correct": 348,
         "accuracy": 0.3870967741935484,
         "lo": 0.3548387096774194,
         "hi": 0.42046718576195774,
         "best_constant": 0.41490545050055616,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 901,
         "correct": 416,
         "accuracy": 0.46170921198668147,
         "lo": 0.4306326304106548,
         "hi": 0.49500554938956715,
         "best_constant": 0.41065482796892344,
         "invalid": 0
        }
       ],
       "strategy": [
        {
         "value": "inv-proof",
         "n": 605,
         "correct": 515,
         "accuracy": 0.8512396694214877,
         "lo": 0.8214876033057851,
         "hi": 0.8793388429752066,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-random",
         "n": 50,
         "correct": 1,
         "accuracy": 0.02,
         "lo": 0,
         "hi": 0.06,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-rconc",
         "n": 249,
         "correct": 2,
         "accuracy": 0.008032128514056224,
         "lo": 0,
         "hi": 0.020080321285140562,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "proof",
         "n": 605,
         "correct": 243,
         "accuracy": 0.40165289256198344,
         "lo": 0.36198347107438017,
         "hi": 0.4413223140495868,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "random",
         "n": 50,
         "correct": 1,
         "accuracy": 0.02,
         "lo": 0,
         "hi": 0.06,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "rconc",
         "n": 241,
         "correct": 2,
         "accuracy": 0.008298755186721992,
         "lo": 0,
         "hi": 0.02074688796680498,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "paraphrased": [
        {
         "value": "False",
         "n": 1298,
         "correct": 541,
         "accuracy": 0.4167950693374422,
         "lo": 0.3921417565485362,
         "hi": 0.44298921417565484,
         "best_constant": 0.34514637904468415,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 502,
         "correct": 223,
         "accuracy": 0.4442231075697211,
         "lo": 0.40239043824701193,
         "hi": 0.48804780876494025,
         "best_constant": 0.35856573705179284,
         "invalid": 0
        }
       ],
       "theory_max_depth": [
        {
         "value": "0",
         "n": 71,
         "correct": 28,
         "accuracy": 0.39436619718309857,
         "lo": 0.28169014084507044,
         "hi": 0.5211267605633803,
         "best_constant": 0.6056338028169014,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 145,
         "correct": 70,
         "accuracy": 0.4827586206896552,
         "lo": 0.4,
         "hi": 0.5655172413793104,
         "best_constant": 0.43448275862068964,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 206,
         "correct": 89,
         "accuracy": 0.4320388349514563,
         "lo": 0.3640776699029126,
         "hi": 0.5,
         "best_constant": 0.3932038834951456,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 620,
         "correct": 263,
         "accuracy": 0.4241935483870968,
         "lo": 0.38387096774193546,
         "hi": 0.4645161290322581,
         "best_constant": 0.3387096774193548,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 162,
         "correct": 70,
         "accuracy": 0.43209876543209874,
         "lo": 0.35802469135802467,
         "hi": 0.5123456790123457,
         "best_constant": 0.3765432098765432,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 571,
         "correct": 238,
         "accuracy": 0.4168126094570928,
         "lo": 0.37828371278458844,
         "hi": 0.4553415061295972,
         "best_constant": 0.38353765323992994,
         "invalid": 0
        },
        {
         "value": "6",
         "n": 18,
         "correct": 4,
         "accuracy": 0.2222222222222222,
         "lo": 0.05555555555555555,
         "hi": 0.4444444444444444,
         "best_constant": 0.4444444444444444,
         "invalid": 0
        },
        {
         "value": "7",
         "n": 5,
         "correct": 2,
         "accuracy": 0.4,
         "lo": 0,
         "hi": 0.8,
         "best_constant": 0.6,
         "invalid": 0
        },
        {
         "value": "8",
         "n": 2,
         "correct": 0,
         "accuracy": 0,
         "lo": 0,
         "hi": 0,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "words_bin": [
        {
         "value": "0-49",
         "n": 247,
         "correct": 118,
         "accuracy": 0.4777327935222672,
         "lo": 0.42105263157894735,
         "hi": 0.5425101214574899,
         "best_constant": 0.3967611336032389,
         "invalid": 0
        },
        {
         "value": "50-79",
         "n": 416,
         "correct": 168,
         "accuracy": 0.40384615384615385,
         "lo": 0.3557692307692308,
         "hi": 0.4543269230769231,
         "best_constant": 0.36778846153846156,
         "invalid": 0
        },
        {
         "value": "80-109",
         "n": 519,
         "correct": 209,
         "accuracy": 0.4026974951830443,
         "lo": 0.3603082851637765,
         "hi": 0.44508670520231214,
         "best_constant": 0.34296724470134876,
         "invalid": 0
        },
        {
         "value": "110+",
         "n": 618,
         "correct": 269,
         "accuracy": 0.43527508090614886,
         "lo": 0.3964401294498382,
         "hi": 0.47249190938511326,
         "best_constant": 0.3705501618122977,
         "invalid": 0
        }
       ],
       "rules_bin": [
        {
         "value": "0-2",
         "n": 204,
         "correct": 110,
         "accuracy": 0.5392156862745098,
         "lo": 0.47058823529411764,
         "hi": 0.6078431372549019,
         "best_constant": 0.3382352941176471,
         "invalid": 0
        },
        {
         "value": "3-5",
         "n": 438,
         "correct": 202,
         "accuracy": 0.4611872146118721,
         "lo": 0.4155251141552511,
         "hi": 0.5091324200913242,
         "best_constant": 0.3835616438356164,
         "invalid": 0
        },
        {
         "value": "6-7",
         "n": 564,
         "correct": 240,
         "accuracy": 0.425531914893617,
         "lo": 0.3829787234042553,
         "hi": 0.46631205673758863,
         "best_constant": 0.3475177304964539,
         "invalid": 0
        },
        {
         "value": "8+",
         "n": 594,
         "correct": 212,
         "accuracy": 0.3569023569023569,
         "lo": 0.3181818181818182,
         "hi": 0.398989898989899,
         "best_constant": 0.4074074074074074,
         "invalid": 0
        }
       ],
       "facts_bin": [
        {
         "value": "0-3",
         "n": 334,
         "correct": 142,
         "accuracy": 0.4251497005988024,
         "lo": 0.3712574850299401,
         "hi": 0.47904191616766467,
         "best_constant": 0.4281437125748503,
         "invalid": 0
        },
        {
         "value": "4-7",
         "n": 435,
         "correct": 181,
         "accuracy": 0.4160919540229885,
         "lo": 0.367816091954023,
         "hi": 0.46436781609195404,
         "best_constant": 0.36091954022988504,
         "invalid": 0
        },
        {
         "value": "8-12",
         "n": 872,
         "correct": 371,
         "accuracy": 0.42545871559633025,
         "lo": 0.3944954128440367,
         "hi": 0.45642201834862384,
         "best_constant": 0.3577981651376147,
         "invalid": 0
        },
        {
         "value": "13+",
         "n": 159,
         "correct": 70,
         "accuracy": 0.44025157232704404,
         "lo": 0.36477987421383645,
         "hi": 0.5157232704402516,
         "best_constant": 0.3836477987421384,
         "invalid": 0
        }
       ],
       "proof_size_bin": [
        {
         "value": "0-1",
         "n": 790,
         "correct": 192,
         "accuracy": 0.2430379746835443,
         "lo": 0.21518987341772153,
         "hi": 0.2708860759493671,
         "best_constant": 0.7468354430379747,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 107,
         "correct": 68,
         "accuracy": 0.6355140186915887,
         "lo": 0.5420560747663551,
         "hi": 0.7289719626168224,
         "best_constant": 0.5046728971962616,
         "invalid": 0
        },
        {
         "value": "3-4",
         "n": 296,
         "correct": 185,
         "accuracy": 0.625,
         "lo": 0.5743243243243243,
         "hi": 0.6824324324324325,
         "best_constant": 0.5033783783783784,
         "invalid": 0
        },
        {
         "value": "5+",
         "n": 607,
         "correct": 319,
         "accuracy": 0.5255354200988468,
         "lo": 0.4876441515650741,
         "hi": 0.5634266886326195,
         "best_constant": 0.500823723228995,
         "invalid": 0
        }
       ]
      }
     },
     "openai-gpt-6-luna-effort-none": {
      "file": "studies/proofwriter-owa-openai-gpt-6-luna-effort-none.jsonl",
      "modified": "2026-10-01T16:36:38.695Z",
      "overall": {
       "n": 1800,
       "correct": 1153,
       "accuracy": 0.6405555555555555,
       "lo": 0.6172222222222222,
       "hi": 0.6622222222222223,
       "macro_f1": 0.6472498370671778,
       "invalid": 0,
       "best_constant": 0.33611111111111114,
       "chance": 0.3333333333333333,
       "recall": {
        "false": 0.509090909090909,
        "true": 0.5834710743801653,
        "unknown": 0.8338983050847457
       },
       "confusion": {
        "false": {
         "false": 308,
         "invalid": 0,
         "true": 7,
         "unknown": 290
        },
        "true": {
         "false": 3,
         "invalid": 0,
         "true": 353,
         "unknown": 249
        },
        "unknown": {
         "false": 53,
         "invalid": 0,
         "true": 45,
         "unknown": 492
        }
       },
       "model": [
        "gpt-6-luna"
       ]
      },
      "axes": {
       "depth": [
        {
         "value": "0",
         "n": 300,
         "correct": 280,
         "accuracy": 0.9333333333333333,
         "lo": 0.9033333333333333,
         "hi": 0.96,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 302,
         "correct": 235,
         "accuracy": 0.7781456953642384,
         "lo": 0.7317880794701986,
         "hi": 0.8278145695364238,
         "best_constant": 0.3344370860927152,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 303,
         "correct": 183,
         "accuracy": 0.6039603960396039,
         "lo": 0.5445544554455446,
         "hi": 0.6600660066006601,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 303,
         "correct": 187,
         "accuracy": 0.6171617161716172,
         "lo": 0.5676567656765676,
         "hi": 0.6765676567656765,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 303,
         "correct": 135,
         "accuracy": 0.44554455445544555,
         "lo": 0.38613861386138615,
         "hi": 0.49504950495049505,
         "best_constant": 0.3333333333333333,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 289,
         "correct": 133,
         "accuracy": 0.4602076124567474,
         "lo": 0.4013840830449827,
         "hi": 0.5155709342560554,
         "best_constant": 0.3494809688581315,
         "invalid": 0
        }
       ],
       "reference_label": [
        {
         "value": "false",
         "n": 605,
         "correct": 308,
         "accuracy": 0.509090909090909,
         "lo": 0.4677685950413223,
         "hi": 0.5487603305785124,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "true",
         "n": 605,
         "correct": 353,
         "accuracy": 0.5834710743801653,
         "lo": 0.540495867768595,
         "hi": 0.6214876033057851,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "unknown",
         "n": 590,
         "correct": 492,
         "accuracy": 0.8338983050847457,
         "lo": 0.8050847457627118,
         "hi": 0.8661016949152542,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "theory_kind": [
        {
         "value": "attribute",
         "n": 1059,
         "correct": 662,
         "accuracy": 0.6251180358829084,
         "lo": 0.5977337110481586,
         "hi": 0.6534466477809254,
         "best_constant": 0.3380547686496695,
         "invalid": 0
        },
        {
         "value": "relation",
         "n": 741,
         "correct": 491,
         "accuracy": 0.6626180836707153,
         "lo": 0.6288798920377868,
         "hi": 0.6963562753036437,
         "best_constant": 0.3441295546558704,
         "invalid": 0
        }
       ],
       "theory_negation": [
        {
         "value": "negation",
         "n": 881,
         "correct": 591,
         "accuracy": 0.6708286038592508,
         "lo": 0.6390465380249716,
         "hi": 0.70261066969353,
         "best_constant": 0.3507377979568672,
         "invalid": 0
        },
        {
         "value": "no-negation",
         "n": 919,
         "correct": 562,
         "accuracy": 0.6115342763873776,
         "lo": 0.5810663764961915,
         "hi": 0.6409140369967355,
         "best_constant": 0.35473340587595215,
         "invalid": 0
        }
       ],
       "statement_negated": [
        {
         "value": "False",
         "n": 899,
         "correct": 560,
         "accuracy": 0.6229143492769744,
         "lo": 0.5917686318131257,
         "hi": 0.6529477196885428,
         "best_constant": 0.41490545050055616,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 901,
         "correct": 593,
         "accuracy": 0.658157602663707,
         "lo": 0.6293007769145395,
         "hi": 0.6881243063263041,
         "best_constant": 0.41065482796892344,
         "invalid": 0
        }
       ],
       "strategy": [
        {
         "value": "inv-proof",
         "n": 605,
         "correct": 308,
         "accuracy": 0.509090909090909,
         "lo": 0.4677685950413223,
         "hi": 0.5487603305785124,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-random",
         "n": 50,
         "correct": 50,
         "accuracy": 1,
         "lo": 1,
         "hi": 1,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "inv-rconc",
         "n": 249,
         "correct": 195,
         "accuracy": 0.7831325301204819,
         "lo": 0.7309236947791165,
         "hi": 0.8353413654618473,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "proof",
         "n": 605,
         "correct": 353,
         "accuracy": 0.5834710743801653,
         "lo": 0.540495867768595,
         "hi": 0.6214876033057851,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "random",
         "n": 50,
         "correct": 49,
         "accuracy": 0.98,
         "lo": 0.94,
         "hi": 1,
         "best_constant": null,
         "invalid": 0
        },
        {
         "value": "rconc",
         "n": 241,
         "correct": 198,
         "accuracy": 0.8215767634854771,
         "lo": 0.7717842323651453,
         "hi": 0.8713692946058091,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "paraphrased": [
        {
         "value": "False",
         "n": 1298,
         "correct": 878,
         "accuracy": 0.6764252696456087,
         "lo": 0.6479198767334361,
         "hi": 0.7018489984591679,
         "best_constant": 0.34514637904468415,
         "invalid": 0
        },
        {
         "value": "True",
         "n": 502,
         "correct": 275,
         "accuracy": 0.547808764940239,
         "lo": 0.5039840637450199,
         "hi": 0.5936254980079682,
         "best_constant": 0.35856573705179284,
         "invalid": 0
        }
       ],
       "theory_max_depth": [
        {
         "value": "0",
         "n": 71,
         "correct": 62,
         "accuracy": 0.8732394366197183,
         "lo": 0.7887323943661971,
         "hi": 0.9436619718309859,
         "best_constant": 0.6056338028169014,
         "invalid": 0
        },
        {
         "value": "1",
         "n": 145,
         "correct": 123,
         "accuracy": 0.8482758620689655,
         "lo": 0.7931034482758621,
         "hi": 0.903448275862069,
         "best_constant": 0.43448275862068964,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 206,
         "correct": 169,
         "accuracy": 0.8203883495145631,
         "lo": 0.7669902912621359,
         "hi": 0.8689320388349514,
         "best_constant": 0.3932038834951456,
         "invalid": 0
        },
        {
         "value": "3",
         "n": 620,
         "correct": 424,
         "accuracy": 0.6838709677419355,
         "lo": 0.6483870967741936,
         "hi": 0.7225806451612903,
         "best_constant": 0.3387096774193548,
         "invalid": 0
        },
        {
         "value": "4",
         "n": 162,
         "correct": 66,
         "accuracy": 0.4074074074074074,
         "lo": 0.3271604938271605,
         "hi": 0.47530864197530864,
         "best_constant": 0.3765432098765432,
         "invalid": 0
        },
        {
         "value": "5",
         "n": 571,
         "correct": 296,
         "accuracy": 0.5183887915936952,
         "lo": 0.4763572679509632,
         "hi": 0.5586690017513135,
         "best_constant": 0.38353765323992994,
         "invalid": 0
        },
        {
         "value": "6",
         "n": 18,
         "correct": 10,
         "accuracy": 0.5555555555555556,
         "lo": 0.3333333333333333,
         "hi": 0.7777777777777778,
         "best_constant": 0.4444444444444444,
         "invalid": 0
        },
        {
         "value": "7",
         "n": 5,
         "correct": 1,
         "accuracy": 0.2,
         "lo": 0,
         "hi": 0.6,
         "best_constant": 0.6,
         "invalid": 0
        },
        {
         "value": "8",
         "n": 2,
         "correct": 2,
         "accuracy": 1,
         "lo": 1,
         "hi": 1,
         "best_constant": null,
         "invalid": 0
        }
       ],
       "words_bin": [
        {
         "value": "0-49",
         "n": 247,
         "correct": 226,
         "accuracy": 0.9149797570850202,
         "lo": 0.8744939271255061,
         "hi": 0.9473684210526315,
         "best_constant": 0.3967611336032389,
         "invalid": 0
        },
        {
         "value": "50-79",
         "n": 416,
         "correct": 303,
         "accuracy": 0.7283653846153846,
         "lo": 0.6875,
         "hi": 0.7692307692307693,
         "best_constant": 0.36778846153846156,
         "invalid": 0
        },
        {
         "value": "80-109",
         "n": 519,
         "correct": 299,
         "accuracy": 0.5761078998073218,
         "lo": 0.535645472061657,
         "hi": 0.6184971098265896,
         "best_constant": 0.34296724470134876,
         "invalid": 0
        },
        {
         "value": "110+",
         "n": 618,
         "correct": 325,
         "accuracy": 0.5258899676375405,
         "lo": 0.4886731391585761,
         "hi": 0.5631067961165048,
         "best_constant": 0.3705501618122977,
         "invalid": 0
        }
       ],
       "rules_bin": [
        {
         "value": "0-2",
         "n": 204,
         "correct": 184,
         "accuracy": 0.9019607843137255,
         "lo": 0.8578431372549019,
         "hi": 0.9411764705882353,
         "best_constant": 0.3382352941176471,
         "invalid": 0
        },
        {
         "value": "3-5",
         "n": 438,
         "correct": 259,
         "accuracy": 0.591324200913242,
         "lo": 0.545662100456621,
         "hi": 0.636986301369863,
         "best_constant": 0.3835616438356164,
         "invalid": 0
        },
        {
         "value": "6-7",
         "n": 564,
         "correct": 346,
         "accuracy": 0.6134751773049646,
         "lo": 0.5780141843971631,
         "hi": 0.6542553191489362,
         "best_constant": 0.3475177304964539,
         "invalid": 0
        },
        {
         "value": "8+",
         "n": 594,
         "correct": 364,
         "accuracy": 0.6127946127946128,
         "lo": 0.5723905723905723,
         "hi": 0.6531986531986532,
         "best_constant": 0.4074074074074074,
         "invalid": 0
        }
       ],
       "facts_bin": [
        {
         "value": "0-3",
         "n": 334,
         "correct": 272,
         "accuracy": 0.8143712574850299,
         "lo": 0.7724550898203593,
         "hi": 0.8532934131736527,
         "best_constant": 0.4281437125748503,
         "invalid": 0
        },
        {
         "value": "4-7",
         "n": 435,
         "correct": 291,
         "accuracy": 0.6689655172413793,
         "lo": 0.6229885057471264,
         "hi": 0.7126436781609196,
         "best_constant": 0.36091954022988504,
         "invalid": 0
        },
        {
         "value": "8-12",
         "n": 872,
         "correct": 500,
         "accuracy": 0.573394495412844,
         "lo": 0.5401376146788991,
         "hi": 0.606651376146789,
         "best_constant": 0.3577981651376147,
         "invalid": 0
        },
        {
         "value": "13+",
         "n": 159,
         "correct": 90,
         "accuracy": 0.5660377358490566,
         "lo": 0.49056603773584906,
         "hi": 0.6477987421383647,
         "best_constant": 0.3836477987421384,
         "invalid": 0
        }
       ],
       "proof_size_bin": [
        {
         "value": "0-1",
         "n": 790,
         "correct": 673,
         "accuracy": 0.8518987341772152,
         "lo": 0.8278481012658228,
         "hi": 0.8746835443037975,
         "best_constant": 0.7468354430379747,
         "invalid": 0
        },
        {
         "value": "2",
         "n": 107,
         "correct": 87,
         "accuracy": 0.8130841121495327,
         "lo": 0.7383177570093458,
         "hi": 0.8878504672897196,
         "best_constant": 0.5046728971962616,
         "invalid": 0
        },
        {
         "value": "3-4",
         "n": 296,
         "correct": 189,
         "accuracy": 0.6385135135135135,
         "lo": 0.5844594594594594,
         "hi": 0.6925675675675675,
         "best_constant": 0.5033783783783784,
         "invalid": 0
        },
        {
         "value": "5+",
         "n": 607,
         "correct": 204,
         "accuracy": 0.33607907742998355,
         "lo": 0.2981878088962109,
         "hi": 0.37397034596375617,
         "best_constant": 0.500823723228995,
         "invalid": 0
        }
       ]
      }
     }
    },
    "pairs": {
     "jev|kev-0.8b": {
      "a": "jev",
      "b": "kev-0.8b",
      "file": "studies/proofwriter-owa-jev-vs-kev-0.8b.jsonl",
      "modified": "2026-10-01T16:36:39.079Z",
      "rows": [
       {
        "axis": "overall",
        "ci_high": 0.32666666666666666,
        "ci_low": 0.27666666666666667,
        "diff": 0.30277777777777776,
        "n": 1800,
        "value": "all"
       },
       {
        "axis": "depth",
        "ci_high": 0.32666666666666666,
        "ci_low": 0.21666666666666667,
        "diff": 0.2733333333333333,
        "n": 300,
        "value": "0"
       },
       {
        "axis": "depth",
        "ci_high": 0.40066225165562913,
        "ci_low": 0.27483443708609273,
        "diff": 0.33774834437086093,
        "n": 302,
        "value": "1"
       },
       {
        "axis": "depth",
        "ci_high": 0.44554455445544555,
        "ci_low": 0.3102310231023102,
        "diff": 0.3795379537953795,
        "n": 303,
        "value": "2"
       },
       {
        "axis": "depth",
        "ci_high": 0.3696369636963696,
        "ci_low": 0.23432343234323433,
        "diff": 0.3102310231023102,
        "n": 303,
        "value": "3"
       },
       {
        "axis": "depth",
        "ci_high": 0.27722772277227725,
        "ci_low": 0.15181518151815182,
        "diff": 0.2145214521452145,
        "n": 303,
        "value": "4"
       },
       {
        "axis": "depth",
        "ci_high": 0.35986159169550175,
        "ci_low": 0.23529411764705882,
        "diff": 0.30103806228373703,
        "n": 289,
        "value": "5"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.021487603305785124,
        "ci_low": -0.04628099173553719,
        "diff": -0.013223140495867768,
        "n": 605,
        "value": "false"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.35206611570247937,
        "ci_low": 0.2611570247933884,
        "diff": 0.30578512396694213,
        "n": 605,
        "value": "true"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.6627118644067796,
        "ci_low": 0.5830508474576271,
        "diff": 0.6237288135593221,
        "n": 590,
        "value": "unknown"
       }
      ]
     },
     "jev|kev-4b": {
      "a": "jev",
      "b": "kev-4b",
      "file": "studies/proofwriter-owa-jev-vs-kev-4b.jsonl",
      "modified": "2026-10-01T16:36:39.472Z",
      "rows": [
       {
        "axis": "overall",
        "ci_high": 0.32944444444444443,
        "ci_low": 0.2738888888888889,
        "diff": 0.30277777777777776,
        "n": 1800,
        "value": "all"
       },
       {
        "axis": "depth",
        "ci_high": 0.13666666666666666,
        "ci_low": 0.06333333333333334,
        "diff": 0.1,
        "n": 300,
        "value": "0"
       },
       {
        "axis": "depth",
        "ci_high": 0.3079470198675497,
        "ci_low": 0.19205298013245034,
        "diff": 0.24834437086092714,
        "n": 302,
        "value": "1"
       },
       {
        "axis": "depth",
        "ci_high": 0.38943894389438943,
        "ci_low": 0.25412541254125415,
        "diff": 0.32673267326732675,
        "n": 303,
        "value": "2"
       },
       {
        "axis": "depth",
        "ci_high": 0.4158415841584158,
        "ci_low": 0.28052805280528054,
        "diff": 0.35313531353135313,
        "n": 303,
        "value": "3"
       },
       {
        "axis": "depth",
        "ci_high": 0.42244224422442245,
        "ci_low": 0.27722772277227725,
        "diff": 0.3465346534653465,
        "n": 303,
        "value": "4"
       },
       {
        "axis": "depth",
        "ci_high": 0.5190311418685121,
        "ci_low": 0.370242214532872,
        "diff": 0.4463667820069204,
        "n": 289,
        "value": "5"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.5652892561983471,
        "ci_low": 0.4793388429752066,
        "diff": 0.5239669421487604,
        "n": 605,
        "value": "false"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.4644628099173554,
        "ci_low": 0.38016528925619836,
        "diff": 0.4214876033057851,
        "n": 605,
        "value": "true"
       },
       {
        "axis": "reference_label",
        "ci_high": -0.001694915254237288,
        "ci_low": -0.08813559322033898,
        "diff": -0.04576271186440678,
        "n": 590,
        "value": "unknown"
       }
      ]
     },
     "jev|kev-9b": {
      "a": "jev",
      "b": "kev-9b",
      "file": "studies/proofwriter-owa-jev-vs-kev-9b.jsonl",
      "modified": "2026-10-01T16:36:39.887Z",
      "rows": [
       {
        "axis": "overall",
        "ci_high": 0.2777777777777778,
        "ci_low": 0.22666666666666666,
        "diff": 0.25277777777777777,
        "n": 1800,
        "value": "all"
       },
       {
        "axis": "depth",
        "ci_high": 0.11666666666666667,
        "ci_low": 0.05333333333333334,
        "diff": 0.08333333333333333,
        "n": 300,
        "value": "0"
       },
       {
        "axis": "depth",
        "ci_high": 0.18874172185430463,
        "ci_low": 0.08940397350993377,
        "diff": 0.1390728476821192,
        "n": 302,
        "value": "1"
       },
       {
        "axis": "depth",
        "ci_high": 0.3234323432343234,
        "ci_low": 0.20132013201320131,
        "diff": 0.264026402640264,
        "n": 303,
        "value": "2"
       },
       {
        "axis": "depth",
        "ci_high": 0.36633663366336633,
        "ci_low": 0.2376237623762376,
        "diff": 0.30363036303630364,
        "n": 303,
        "value": "3"
       },
       {
        "axis": "depth",
        "ci_high": 0.40264026402640263,
        "ci_low": 0.26732673267326734,
        "diff": 0.3333333333333333,
        "n": 303,
        "value": "4"
       },
       {
        "axis": "depth",
        "ci_high": 0.47750865051903113,
        "ci_low": 0.328719723183391,
        "diff": 0.39792387543252594,
        "n": 289,
        "value": "5"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.43636363636363634,
        "ci_low": 0.35537190082644626,
        "diff": 0.39669421487603307,
        "n": 605,
        "value": "false"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.3371900826446281,
        "ci_low": 0.256198347107438,
        "diff": 0.2958677685950413,
        "n": 605,
        "value": "true"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.10508474576271186,
        "ci_low": 0.015254237288135594,
        "diff": 0.061016949152542375,
        "n": 590,
        "value": "unknown"
       }
      ]
     },
     "jev|laya": {
      "a": "jev",
      "b": "laya",
      "file": "studies/proofwriter-owa-jev-vs-laya.jsonl",
      "modified": "2026-10-01T16:36:40.369Z",
      "rows": [
       {
        "axis": "overall",
        "ci_high": 0.44,
        "ci_low": 0.3877777777777778,
        "diff": 0.41388888888888886,
        "n": 1800,
        "value": "all"
       },
       {
        "axis": "depth",
        "ci_high": 0.4066666666666667,
        "ci_low": 0.29333333333333333,
        "diff": 0.35,
        "n": 300,
        "value": "0"
       },
       {
        "axis": "depth",
        "ci_high": 0.5198675496688742,
        "ci_low": 0.39403973509933776,
        "diff": 0.45695364238410596,
        "n": 302,
        "value": "1"
       },
       {
        "axis": "depth",
        "ci_high": 0.5016501650165016,
        "ci_low": 0.37293729372937295,
        "diff": 0.43564356435643564,
        "n": 303,
        "value": "2"
       },
       {
        "axis": "depth",
        "ci_high": 0.47854785478547857,
        "ci_low": 0.3432343234323432,
        "diff": 0.4158415841584158,
        "n": 303,
        "value": "3"
       },
       {
        "axis": "depth",
        "ci_high": 0.42574257425742573,
        "ci_low": 0.2838283828382838,
        "diff": 0.35973597359735976,
        "n": 303,
        "value": "4"
       },
       {
        "axis": "depth",
        "ci_high": 0.5294117647058824,
        "ci_low": 0.4013840830449827,
        "diff": 0.4671280276816609,
        "n": 289,
        "value": "5"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.07272727272727272,
        "ci_low": -0.009917355371900827,
        "diff": 0.03140495867768595,
        "n": 605,
        "value": "false"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.5272727272727272,
        "ci_low": 0.4396694214876033,
        "diff": 0.4859504132231405,
        "n": 605,
        "value": "true"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.7661016949152543,
        "ci_low": 0.6966101694915254,
        "diff": 0.7322033898305085,
        "n": 590,
        "value": "unknown"
       }
      ]
     },
     "jev|openai-gpt-6-luna-effort-none": {
      "a": "jev",
      "b": "openai-gpt-6-luna-effort-none",
      "file": "studies/proofwriter-owa-jev-vs-openai-gpt-6-luna-effort-none.jsonl",
      "modified": "2026-10-01T16:36:40.750Z",
      "rows": [
       {
        "axis": "overall",
        "ci_high": 0.22444444444444445,
        "ci_low": 0.17222222222222222,
        "diff": 0.19777777777777777,
        "n": 1800,
        "value": "all"
       },
       {
        "axis": "depth",
        "ci_high": 0.07666666666666666,
        "ci_low": 0.01,
        "diff": 0.043333333333333335,
        "n": 300,
        "value": "0"
       },
       {
        "axis": "depth",
        "ci_high": 0.152317880794702,
        "ci_low": 0.04966887417218543,
        "diff": 0.09933774834437085,
        "n": 302,
        "value": "1"
       },
       {
        "axis": "depth",
        "ci_high": 0.30363036303630364,
        "ci_low": 0.18151815181518152,
        "diff": 0.24422442244224424,
        "n": 303,
        "value": "2"
       },
       {
        "axis": "depth",
        "ci_high": 0.2376237623762376,
        "ci_low": 0.10891089108910891,
        "diff": 0.18151815181518152,
        "n": 303,
        "value": "3"
       },
       {
        "axis": "depth",
        "ci_high": 0.35313531353135313,
        "ci_low": 0.19801980198019803,
        "diff": 0.2739273927392739,
        "n": 303,
        "value": "4"
       },
       {
        "axis": "depth",
        "ci_high": 0.4290657439446367,
        "ci_low": 0.27335640138408307,
        "diff": 0.3494809688581315,
        "n": 289,
        "value": "5"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.41652892561983473,
        "ci_low": 0.3305785123966942,
        "diff": 0.37355371900826445,
        "n": 605,
        "value": "false"
       },
       {
        "axis": "reference_label",
        "ci_high": 0.34545454545454546,
        "ci_low": 0.26611570247933886,
        "diff": 0.30413223140495865,
        "n": 605,
        "value": "true"
       },
       {
        "axis": "reference_label",
        "ci_high": -0.05084745762711865,
        "ci_low": -0.13389830508474576,
        "diff": -0.09152542372881356,
        "n": 590,
        "value": "unknown"
       }
      ]
     }
    },
    "retest": {
     "jev": {
      "file": "studies/retest/proofwriter-owa.jsonl",
      "depth": [
       {
        "value": "0",
        "n": 300,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "1",
        "n": 302,
        "changed": 3,
        "rate": 0.009933774834437087
       },
       {
        "value": "2",
        "n": 303,
        "changed": 7,
        "rate": 0.0231023102310231
       },
       {
        "value": "3",
        "n": 303,
        "changed": 13,
        "rate": 0.0429042904290429
       },
       {
        "value": "4",
        "n": 303,
        "changed": 17,
        "rate": 0.056105610561056105
       },
       {
        "value": "5",
        "n": 289,
        "changed": 10,
        "rate": 0.03460207612456748
       }
      ],
      "n": 1800,
      "agreement": 0.9722222222222222,
      "ac1": 0.9583992926030823,
      "ac1_lo": 0.9475447569504353,
      "ac1_hi": 0.969222241182328,
      "kappa": 0.9582011444526649,
      "changed": 50,
      "accuracy_run1": 0.8383333333333334,
      "accuracy_run2": 0.8416666666666667,
      "prob_shift_mean": 0.02296666666666667,
      "prob_shift_max": 0.36
     },
     "kev-0.8b": {
      "file": "studies/retest/proofwriter-owa.jsonl",
      "depth": [
       {
        "value": "0",
        "n": 300,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "1",
        "n": 302,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "2",
        "n": 303,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "3",
        "n": 303,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "4",
        "n": 303,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "5",
        "n": 289,
        "changed": 0,
        "rate": 0
       }
      ],
      "n": 1800,
      "agreement": 1,
      "ac1": 1,
      "ac1_lo": 1,
      "ac1_hi": 1,
      "kappa": 1,
      "changed": 0,
      "accuracy_run1": 0.5355555555555556,
      "accuracy_run2": 0.5355555555555556,
      "prob_shift_mean": 0,
      "prob_shift_max": 0
     },
     "kev-4b": {
      "file": "studies/retest/proofwriter-owa.jsonl",
      "depth": [
       {
        "value": "0",
        "n": 300,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "1",
        "n": 302,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "2",
        "n": 303,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "3",
        "n": 303,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "4",
        "n": 303,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "5",
        "n": 289,
        "changed": 0,
        "rate": 0
       }
      ],
      "n": 1800,
      "agreement": 1,
      "ac1": 1,
      "ac1_lo": 1,
      "ac1_hi": 1,
      "kappa": 1,
      "changed": 0,
      "accuracy_run1": 0.5355555555555556,
      "accuracy_run2": 0.5355555555555556,
      "prob_shift_mean": 0,
      "prob_shift_max": 0
     },
     "laya": {
      "file": "studies/retest/proofwriter-owa.jsonl",
      "depth": [
       {
        "value": "0",
        "n": 300,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "1",
        "n": 302,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "2",
        "n": 303,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "3",
        "n": 303,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "4",
        "n": 303,
        "changed": 0,
        "rate": 0
       },
       {
        "value": "5",
        "n": 289,
        "changed": 0,
        "rate": 0
       }
      ],
      "n": 1800,
      "agreement": 1,
      "ac1": 1,
      "ac1_lo": 1,
      "ac1_hi": 1,
      "kappa": 1,
      "changed": 0,
      "accuracy_run1": 0.42444444444444446,
      "accuracy_run2": 0.42444444444444446,
      "prob_shift_mean": 0,
      "prob_shift_max": 0
     },
     "openai-gpt-6-luna-effort-none": {
      "file": "studies/retest/proofwriter-owa.jsonl",
      "depth": [
       {
        "value": "0",
        "n": 300,
        "changed": 11,
        "rate": 0.03666666666666667
       },
       {
        "value": "1",
        "n": 302,
        "changed": 20,
        "rate": 0.06622516556291391
       },
       {
        "value": "2",
        "n": 303,
        "changed": 39,
        "rate": 0.12871287128712872
       },
       {
        "value": "3",
        "n": 303,
        "changed": 39,
        "rate": 0.12871287128712872
       },
       {
        "value": "4",
        "n": 303,
        "changed": 37,
        "rate": 0.12211221122112212
       },
       {
        "value": "5",
        "n": 289,
        "changed": 39,
        "rate": 0.13494809688581316
       }
      ],
      "n": 1800,
      "agreement": 0.8972222222222223,
      "ac1": 0.8554453320183055,
      "ac1_lo": 0.8366412361247366,
      "ac1_hi": 0.8760274329372628,
      "kappa": 0.8221914661515025,
      "changed": 185,
      "accuracy_run1": 0.6405555555555555,
      "accuracy_run2": 0.6277777777777778,
      "prob_shift_mean": null,
      "prob_shift_max": null
     }
    }
   },
   "status": {
    "jev": {
     "status": "complete",
     "scored": 1800,
     "answered": 1800,
     "of": 1800
    },
    "laya": {
     "status": "complete",
     "scored": 1800,
     "answered": 1800,
     "of": 1800
    },
    "kev-0.8b": {
     "status": "complete",
     "scored": 1800,
     "answered": 1800,
     "of": 1800
    },
    "kev-4b": {
     "status": "complete",
     "scored": 1800,
     "answered": 1800,
     "of": 1800
    },
    "kev-9b": {
     "status": "complete",
     "scored": 1800,
     "answered": 1800,
     "of": 1800
    },
    "kev-27b": {
     "status": "pending",
     "scored": 0,
     "answered": 0,
     "of": 1800
    },
    "openai-gpt-6-luna-effort-none": {
     "status": "complete",
     "scored": 1800,
     "answered": 1800,
     "of": 1800
    }
   },
   "derived": {
    "jev": {
     "cross": [
      {
       "depth": "5",
       "label": "true",
       "n": 101,
       "correct": 94,
       "said": {
        "true": 94,
        "unknown": 7
       }
      },
      {
       "depth": "5",
       "label": "false",
       "n": 101,
       "correct": 93,
       "said": {
        "false": 93,
        "unknown": 6,
        "true": 2
       }
      },
      {
       "depth": "5",
       "label": "unknown",
       "n": 87,
       "correct": 47,
       "said": {
        "unknown": 47,
        "true": 31,
        "false": 9
       }
      },
      {
       "depth": "4",
       "label": "true",
       "n": 101,
       "correct": 81,
       "said": {
        "true": 81,
        "unknown": 18,
        "false": 2
       }
      },
      {
       "depth": "4",
       "label": "false",
       "n": 101,
       "correct": 80,
       "said": {
        "true": 5,
        "false": 80,
        "unknown": 16
       }
      },
      {
       "depth": "4",
       "label": "unknown",
       "n": 101,
       "correct": 57,
       "said": {
        "true": 37,
        "unknown": 57,
        "false": 7
       }
      },
      {
       "depth": "3",
       "label": "true",
       "n": 101,
       "correct": 87,
       "said": {
        "true": 87,
        "unknown": 12,
        "false": 2
       }
      },
      {
       "depth": "3",
       "label": "false",
       "n": 101,
       "correct": 81,
       "said": {
        "unknown": 20,
        "false": 81
       }
      },
      {
       "depth": "3",
       "label": "unknown",
       "n": 101,
       "correct": 74,
       "said": {
        "unknown": 74,
        "true": 20,
        "false": 7
       }
      },
      {
       "depth": "2",
       "label": "true",
       "n": 101,
       "correct": 87,
       "said": {
        "true": 87,
        "unknown": 13,
        "false": 1
       }
      },
      {
       "depth": "2",
       "label": "false",
       "n": 101,
       "correct": 88,
       "said": {
        "false": 88,
        "unknown": 11,
        "true": 2
       }
      },
      {
       "depth": "2",
       "label": "unknown",
       "n": 101,
       "correct": 82,
       "said": {
        "unknown": 82,
        "true": 16,
        "false": 3
       }
      },
      {
       "depth": "1",
       "label": "true",
       "n": 101,
       "correct": 90,
       "said": {
        "true": 90,
        "unknown": 11
       }
      },
      {
       "depth": "1",
       "label": "false",
       "n": 101,
       "correct": 93,
       "said": {
        "false": 93,
        "unknown": 7,
        "true": 1
       }
      },
      {
       "depth": "1",
       "label": "unknown",
       "n": 100,
       "correct": 82,
       "said": {
        "unknown": 82,
        "true": 17,
        "false": 1
       }
      },
      {
       "depth": "0",
       "label": "true",
       "n": 100,
       "correct": 98,
       "said": {
        "true": 98,
        "unknown": 2
       }
      },
      {
       "depth": "0",
       "label": "false",
       "n": 100,
       "correct": 99,
       "said": {
        "false": 99,
        "unknown": 1
       }
      },
      {
       "depth": "0",
       "label": "unknown",
       "n": 100,
       "correct": 96,
       "said": {
        "unknown": 96,
        "true": 3,
        "false": 1
       }
      }
     ],
     "calibration": {
      "n": 1800,
      "ece": 0.041422222222225524,
      "mean_confidence": 0.8775666666666714,
      "accuracy": 0.8383333333333334,
      "bins": [
       {
        "lo": 0,
        "hi": 0.5,
        "n": 31,
        "accuracy": 0.5161290322580645,
        "confidence": 0.45258064516129054
       },
       {
        "lo": 0.5,
        "hi": 0.6,
        "n": 127,
        "accuracy": 0.5039370078740157,
        "confidence": 0.5432283464566929
       },
       {
        "lo": 0.6,
        "hi": 0.7,
        "n": 146,
        "accuracy": 0.5753424657534246,
        "confidence": 0.6460273972602738
       },
       {
        "lo": 0.7,
        "hi": 0.8,
        "n": 152,
        "accuracy": 0.631578947368421,
        "confidence": 0.7461842105263162
       },
       {
        "lo": 0.8,
        "hi": 0.9,
        "n": 197,
        "accuracy": 0.7817258883248731,
        "confidence": 0.8472588832487313
       },
       {
        "lo": 0.9,
        "hi": 1,
        "n": 1147,
        "accuracy": 0.9546643417611159,
        "confidence": 0.9781604184830042
       }
      ]
     },
     "models": [
      "jev-1.13.0"
     ],
     "file": "answers/jev/proofwriter-owa.jsonl.gz"
    },
    "kev-0.8b": {
     "cross": [
      {
       "depth": "5",
       "label": "true",
       "n": 101,
       "correct": 53,
       "said": {
        "true": 53,
        "false": 42,
        "unknown": 6
       }
      },
      {
       "depth": "5",
       "label": "false",
       "n": 101,
       "correct": 90,
       "said": {
        "true": 10,
        "false": 90,
        "unknown": 1
       }
      },
      {
       "depth": "5",
       "label": "unknown",
       "n": 87,
       "correct": 4,
       "said": {
        "false": 63,
        "unknown": 4,
        "true": 20
       }
      },
      {
       "depth": "4",
       "label": "true",
       "n": 101,
       "correct": 50,
       "said": {
        "true": 50,
        "false": 42,
        "unknown": 9
       }
      },
      {
       "depth": "4",
       "label": "false",
       "n": 101,
       "correct": 93,
       "said": {
        "false": 93,
        "true": 8
       }
      },
      {
       "depth": "4",
       "label": "unknown",
       "n": 101,
       "correct": 10,
       "said": {
        "true": 31,
        "false": 60,
        "unknown": 10
       }
      },
      {
       "depth": "3",
       "label": "true",
       "n": 101,
       "correct": 48,
       "said": {
        "false": 42,
        "true": 48,
        "unknown": 11
       }
      },
      {
       "depth": "3",
       "label": "false",
       "n": 101,
       "correct": 90,
       "said": {
        "false": 90,
        "true": 10,
        "unknown": 1
       }
      },
      {
       "depth": "3",
       "label": "unknown",
       "n": 101,
       "correct": 10,
       "said": {
        "true": 27,
        "false": 64,
        "unknown": 10
       }
      },
      {
       "depth": "2",
       "label": "true",
       "n": 101,
       "correct": 48,
       "said": {
        "unknown": 12,
        "false": 41,
        "true": 48
       }
      },
      {
       "depth": "2",
       "label": "false",
       "n": 101,
       "correct": 88,
       "said": {
        "false": 88,
        "true": 11,
        "unknown": 2
       }
      },
      {
       "depth": "2",
       "label": "unknown",
       "n": 101,
       "correct": 6,
       "said": {
        "true": 31,
        "false": 64,
        "unknown": 6
       }
      },
      {
       "depth": "1",
       "label": "true",
       "n": 101,
       "correct": 62,
       "said": {
        "true": 62,
        "false": 36,
        "unknown": 3
       }
      },
      {
       "depth": "1",
       "label": "false",
       "n": 101,
       "correct": 89,
       "said": {
        "false": 89,
        "true": 12
       }
      },
      {
       "depth": "1",
       "label": "unknown",
       "n": 100,
       "correct": 12,
       "said": {
        "false": 66,
        "unknown": 12,
        "true": 22
       }
      },
      {
       "depth": "0",
       "label": "true",
       "n": 100,
       "correct": 91,
       "said": {
        "true": 91,
        "false": 6,
        "unknown": 3
       }
      },
      {
       "depth": "0",
       "label": "false",
       "n": 100,
       "correct": 92,
       "said": {
        "false": 92,
        "true": 8
       }
      },
      {
       "depth": "0",
       "label": "unknown",
       "n": 100,
       "correct": 28,
       "said": {
        "true": 19,
        "false": 53,
        "unknown": 28
       }
      }
     ],
     "calibration": {
      "n": 1800,
      "ece": 0.11435405555555564,
      "mean_confidence": 0.6499096111111097,
      "accuracy": 0.5355555555555556,
      "bins": [
       {
        "lo": 0,
        "hi": 0.5,
        "n": 441,
        "accuracy": 0.36507936507936506,
        "confidence": 0.42512698412698474
       },
       {
        "lo": 0.5,
        "hi": 0.6,
        "n": 269,
        "accuracy": 0.42379182156133827,
        "confidence": 0.5495531598513013
       },
       {
        "lo": 0.6,
        "hi": 0.7,
        "n": 309,
        "accuracy": 0.5307443365695793,
        "confidence": 0.6515304207119738
       },
       {
        "lo": 0.7,
        "hi": 0.8,
        "n": 325,
        "accuracy": 0.6030769230769231,
        "confidence": 0.7458713846153842
       },
       {
        "lo": 0.8,
        "hi": 0.9,
        "n": 387,
        "accuracy": 0.7260981912144703,
        "confidence": 0.8466457364341087
       },
       {
        "lo": 0.9,
        "hi": 1,
        "n": 69,
        "accuracy": 0.6956521739130435,
        "confidence": 0.9151231884057971
       }
      ]
     },
     "models": [
      "kev-latest"
     ],
     "file": "answers/kev-0.8b/proofwriter-owa.jsonl.gz"
    },
    "kev-4b": {
     "cross": [
      {
       "depth": "5",
       "label": "true",
       "n": 101,
       "correct": 22,
       "said": {
        "unknown": 77,
        "true": 22,
        "false": 2
       }
      },
      {
       "depth": "5",
       "label": "false",
       "n": 101,
       "correct": 12,
       "said": {
        "unknown": 78,
        "false": 12,
        "true": 11
       }
      },
      {
       "depth": "5",
       "label": "unknown",
       "n": 87,
       "correct": 71,
       "said": {
        "unknown": 71,
        "true": 12,
        "false": 4
       }
      },
      {
       "depth": "4",
       "label": "true",
       "n": 101,
       "correct": 20,
       "said": {
        "unknown": 78,
        "true": 20,
        "false": 3
       }
      },
      {
       "depth": "4",
       "label": "false",
       "n": 101,
       "correct": 16,
       "said": {
        "false": 16,
        "unknown": 74,
        "true": 11
       }
      },
      {
       "depth": "4",
       "label": "unknown",
       "n": 101,
       "correct": 77,
       "said": {
        "unknown": 77,
        "true": 17,
        "false": 7
       }
      },
      {
       "depth": "3",
       "label": "true",
       "n": 101,
       "correct": 35,
       "said": {
        "unknown": 65,
        "true": 35,
        "false": 1
       }
      },
      {
       "depth": "3",
       "label": "false",
       "n": 101,
       "correct": 27,
       "said": {
        "unknown": 67,
        "false": 27,
        "true": 7
       }
      },
      {
       "depth": "3",
       "label": "unknown",
       "n": 101,
       "correct": 73,
       "said": {
        "unknown": 73,
        "false": 7,
        "true": 21
       }
      },
      {
       "depth": "2",
       "label": "true",
       "n": 101,
       "correct": 44,
       "said": {
        "unknown": 52,
        "false": 5,
        "true": 44
       }
      },
      {
       "depth": "2",
       "label": "false",
       "n": 101,
       "correct": 32,
       "said": {
        "true": 9,
        "unknown": 60,
        "false": 32
       }
      },
      {
       "depth": "2",
       "label": "unknown",
       "n": 101,
       "correct": 82,
       "said": {
        "unknown": 82,
        "true": 15,
        "false": 4
       }
      },
      {
       "depth": "1",
       "label": "true",
       "n": 101,
       "correct": 65,
       "said": {
        "true": 65,
        "unknown": 35,
        "false": 1
       }
      },
      {
       "depth": "1",
       "label": "false",
       "n": 101,
       "correct": 48,
       "said": {
        "false": 48,
        "unknown": 47,
        "true": 6
       }
      },
      {
       "depth": "1",
       "label": "unknown",
       "n": 100,
       "correct": 77,
       "said": {
        "unknown": 77,
        "true": 14,
        "false": 9
       }
      },
      {
       "depth": "0",
       "label": "true",
       "n": 100,
       "correct": 96,
       "said": {
        "true": 96,
        "unknown": 4
       }
      },
      {
       "depth": "0",
       "label": "false",
       "n": 100,
       "correct": 82,
       "said": {
        "false": 82,
        "unknown": 17,
        "true": 1
       }
      },
      {
       "depth": "0",
       "label": "unknown",
       "n": 100,
       "correct": 85,
       "said": {
        "true": 10,
        "unknown": 85,
        "false": 5
       }
      }
     ],
     "calibration": {
      "n": 1800,
      "ece": 0.16362722222222226,
      "mean_confidence": 0.6991827777777776,
      "accuracy": 0.5355555555555556,
      "bins": [
       {
        "lo": 0,
        "hi": 0.5,
        "n": 221,
        "accuracy": 0.334841628959276,
        "confidence": 0.46293257918552017
       },
       {
        "lo": 0.5,
        "hi": 0.6,
        "n": 358,
        "accuracy": 0.39106145251396646,
        "confidence": 0.5477902234636874
       },
       {
        "lo": 0.6,
        "hi": 0.7,
        "n": 361,
        "accuracy": 0.40443213296398894,
        "confidence": 0.6503094182825482
       },
       {
        "lo": 0.7,
        "hi": 0.8,
        "n": 306,
        "accuracy": 0.5555555555555556,
        "confidence": 0.7501562091503273
       },
       {
        "lo": 0.8,
        "hi": 0.9,
        "n": 280,
        "accuracy": 0.6428571428571429,
        "confidence": 0.8462789285714282
       },
       {
        "lo": 0.9,
        "hi": 1,
        "n": 274,
        "accuracy": 0.927007299270073,
        "confidence": 0.9446875912408762
       }
      ]
     },
     "models": [
      "kev-latest"
     ],
     "file": "answers/kev-4b/proofwriter-owa.jsonl.gz"
    },
    "kev-9b": {
     "cross": [
      {
       "depth": "5",
       "label": "true",
       "n": 101,
       "correct": 36,
       "said": {
        "true": 36,
        "unknown": 59,
        "false": 6
       }
      },
      {
       "depth": "5",
       "label": "false",
       "n": 101,
       "correct": 23,
       "said": {
        "true": 17,
        "false": 23,
        "unknown": 61
       }
      },
      {
       "depth": "5",
       "label": "unknown",
       "n": 87,
       "correct": 60,
       "said": {
        "unknown": 60,
        "true": 22,
        "false": 5
       }
      },
      {
       "depth": "4",
       "label": "true",
       "n": 101,
       "correct": 30,
       "said": {
        "unknown": 60,
        "true": 30,
        "false": 11
       }
      },
      {
       "depth": "4",
       "label": "false",
       "n": 101,
       "correct": 19,
       "said": {
        "unknown": 63,
        "false": 19,
        "true": 19
       }
      },
      {
       "depth": "4",
       "label": "unknown",
       "n": 101,
       "correct": 68,
       "said": {
        "unknown": 68,
        "true": 22,
        "false": 11
       }
      },
      {
       "depth": "3",
       "label": "true",
       "n": 101,
       "correct": 54,
       "said": {
        "false": 10,
        "true": 54,
        "unknown": 37
       }
      },
      {
       "depth": "3",
       "label": "false",
       "n": 101,
       "correct": 40,
       "said": {
        "unknown": 49,
        "true": 12,
        "false": 40
       }
      },
      {
       "depth": "3",
       "label": "unknown",
       "n": 101,
       "correct": 56,
       "said": {
        "unknown": 56,
        "false": 13,
        "true": 32
       }
      },
      {
       "depth": "2",
       "label": "true",
       "n": 101,
       "correct": 62,
       "said": {
        "true": 62,
        "unknown": 35,
        "false": 4
       }
      },
      {
       "depth": "2",
       "label": "false",
       "n": 101,
       "correct": 44,
       "said": {
        "true": 13,
        "unknown": 44,
        "false": 44
       }
      },
      {
       "depth": "2",
       "label": "unknown",
       "n": 101,
       "correct": 71,
       "said": {
        "unknown": 71,
        "true": 21,
        "false": 9
       }
      },
      {
       "depth": "1",
       "label": "true",
       "n": 101,
       "correct": 79,
       "said": {
        "true": 79,
        "unknown": 16,
        "false": 6
       }
      },
      {
       "depth": "1",
       "label": "false",
       "n": 101,
       "correct": 74,
       "said": {
        "false": 74,
        "true": 11,
        "unknown": 16
       }
      },
      {
       "depth": "1",
       "label": "unknown",
       "n": 100,
       "correct": 70,
       "said": {
        "unknown": 70,
        "true": 21,
        "false": 9
       }
      },
      {
       "depth": "0",
       "label": "true",
       "n": 100,
       "correct": 97,
       "said": {
        "true": 97,
        "unknown": 3
       }
      },
      {
       "depth": "0",
       "label": "false",
       "n": 100,
       "correct": 94,
       "said": {
        "false": 94,
        "true": 2,
        "unknown": 4
       }
      },
      {
       "depth": "0",
       "label": "unknown",
       "n": 100,
       "correct": 77,
       "said": {
        "true": 13,
        "false": 10,
        "unknown": 77
       }
      }
     ],
     "calibration": {
      "n": 1800,
      "ece": 0.1579106666666666,
      "mean_confidence": 0.7434662222222227,
      "accuracy": 0.5855555555555556,
      "bins": [
       {
        "lo": 0,
        "hi": 0.5,
        "n": 206,
        "accuracy": 0.3737864077669903,
        "confidence": 0.4535553398058252
       },
       {
        "lo": 0.5,
        "hi": 0.6,
        "n": 276,
        "accuracy": 0.3804347826086957,
        "confidence": 0.5509826086956517
       },
       {
        "lo": 0.6,
        "hi": 0.7,
        "n": 264,
        "accuracy": 0.4696969696969697,
        "confidence": 0.6509590909090908
       },
       {
        "lo": 0.7,
        "hi": 0.8,
        "n": 279,
        "accuracy": 0.5125448028673835,
        "confidence": 0.7505035842293906
       },
       {
        "lo": 0.8,
        "hi": 0.9,
        "n": 295,
        "accuracy": 0.6169491525423729,
        "confidence": 0.8528284745762711
       },
       {
        "lo": 0.9,
        "hi": 1,
        "n": 480,
        "accuracy": 0.88125,
        "confidence": 0.9581406250000002
       }
      ]
     },
     "models": [
      "kev-latest"
     ],
     "file": "answers/kev-9b/proofwriter-owa.jsonl.gz"
    },
    "laya": {
     "cross": [
      {
       "depth": "5",
       "label": "true",
       "n": 101,
       "correct": 17,
       "said": {
        "false": 84,
        "true": 17
       }
      },
      {
       "depth": "5",
       "label": "false",
       "n": 101,
       "correct": 82,
       "said": {
        "false": 82,
        "true": 19
       }
      },
      {
       "depth": "5",
       "label": "unknown",
       "n": 87,
       "correct": 0,
       "said": {
        "false": 74,
        "true": 13
       }
      },
      {
       "depth": "4",
       "label": "true",
       "n": 101,
       "correct": 29,
       "said": {
        "false": 71,
        "true": 29,
        "unknown": 1
       }
      },
      {
       "depth": "4",
       "label": "false",
       "n": 101,
       "correct": 80,
       "said": {
        "true": 21,
        "false": 80
       }
      },
      {
       "depth": "4",
       "label": "unknown",
       "n": 101,
       "correct": 0,
       "said": {
        "false": 73,
        "true": 28
       }
      },
      {
       "depth": "3",
       "label": "true",
       "n": 101,
       "correct": 33,
       "said": {
        "false": 64,
        "true": 33,
        "unknown": 4
       }
      },
      {
       "depth": "3",
       "label": "false",
       "n": 101,
       "correct": 82,
       "said": {
        "false": 82,
        "true": 19
       }
      },
      {
       "depth": "3",
       "label": "unknown",
       "n": 101,
       "correct": 1,
       "said": {
        "true": 26,
        "false": 74,
        "unknown": 1
       }
      },
      {
       "depth": "2",
       "label": "true",
       "n": 101,
       "correct": 38,
       "said": {
        "false": 62,
        "true": 38,
        "unknown": 1
       }
      },
      {
       "depth": "2",
       "label": "false",
       "n": 101,
       "correct": 85,
       "said": {
        "false": 85,
        "true": 15,
        "unknown": 1
       }
      },
      {
       "depth": "2",
       "label": "unknown",
       "n": 101,
       "correct": 2,
       "said": {
        "false": 80,
        "true": 19,
        "unknown": 2
       }
      },
      {
       "depth": "1",
       "label": "true",
       "n": 101,
       "correct": 37,
       "said": {
        "true": 37,
        "false": 63,
        "unknown": 1
       }
      },
      {
       "depth": "1",
       "label": "false",
       "n": 101,
       "correct": 89,
       "said": {
        "false": 89,
        "true": 12
       }
      },
      {
       "depth": "1",
       "label": "unknown",
       "n": 100,
       "correct": 1,
       "said": {
        "false": 76,
        "true": 23,
        "unknown": 1
       }
      },
      {
       "depth": "0",
       "label": "true",
       "n": 100,
       "correct": 89,
       "said": {
        "false": 10,
        "true": 89,
        "unknown": 1
       }
      },
      {
       "depth": "0",
       "label": "false",
       "n": 100,
       "correct": 97,
       "said": {
        "false": 97,
        "true": 3
       }
      },
      {
       "depth": "0",
       "label": "unknown",
       "n": 100,
       "correct": 2,
       "said": {
        "false": 80,
        "unknown": 2,
        "true": 18
       }
      }
     ],
     "calibration": {
      "n": 1800,
      "ece": 0.3229931111111112,
      "mean_confidence": 0.7474375555555544,
      "accuracy": 0.42444444444444446,
      "bins": [
       {
        "lo": 0,
        "hi": 0.5,
        "n": 217,
        "accuracy": 0.3456221198156682,
        "confidence": 0.43621198156682006
       },
       {
        "lo": 0.5,
        "hi": 0.6,
        "n": 180,
        "accuracy": 0.2833333333333333,
        "confidence": 0.5545227777777775
       },
       {
        "lo": 0.6,
        "hi": 0.7,
        "n": 257,
        "accuracy": 0.3463035019455253,
        "confidence": 0.6526715953307398
       },
       {
        "lo": 0.7,
        "hi": 0.8,
        "n": 315,
        "accuracy": 0.40634920634920635,
        "confidence": 0.7534244444444443
       },
       {
        "lo": 0.8,
        "hi": 0.9,
        "n": 418,
        "accuracy": 0.4258373205741627,
        "confidence": 0.850402870813397
       },
       {
        "lo": 0.9,
        "hi": 1,
        "n": 413,
        "accuracy": 0.5883777239709443,
        "confidence": 0.9452343825665862
       }
      ]
     },
     "models": [
      "laya-upstream:0.3.21"
     ],
     "file": "answers/laya/proofwriter-owa.jsonl.gz"
    },
    "openai-gpt-6-luna-effort-none": {
     "cross": [
      {
       "depth": "5",
       "label": "true",
       "n": 101,
       "correct": 36,
       "said": {
        "unknown": 65,
        "true": 36
       }
      },
      {
       "depth": "5",
       "label": "false",
       "n": 101,
       "correct": 35,
       "said": {
        "unknown": 66,
        "false": 35
       }
      },
      {
       "depth": "5",
       "label": "unknown",
       "n": 87,
       "correct": 62,
       "said": {
        "true": 13,
        "unknown": 62,
        "false": 12
       }
      },
      {
       "depth": "4",
       "label": "true",
       "n": 101,
       "correct": 33,
       "said": {
        "unknown": 67,
        "true": 33,
        "false": 1
       }
      },
      {
       "depth": "4",
       "label": "false",
       "n": 101,
       "correct": 27,
       "said": {
        "false": 27,
        "unknown": 73,
        "true": 1
       }
      },
      {
       "depth": "4",
       "label": "unknown",
       "n": 101,
       "correct": 75,
       "said": {
        "unknown": 75,
        "true": 9,
        "false": 17
       }
      },
      {
       "depth": "3",
       "label": "true",
       "n": 101,
       "correct": 53,
       "said": {
        "unknown": 47,
        "true": 53,
        "false": 1
       }
      },
      {
       "depth": "3",
       "label": "false",
       "n": 101,
       "correct": 53,
       "said": {
        "unknown": 48,
        "false": 53
       }
      },
      {
       "depth": "3",
       "label": "unknown",
       "n": 101,
       "correct": 81,
       "said": {
        "unknown": 81,
        "false": 12,
        "true": 8
       }
      },
      {
       "depth": "2",
       "label": "true",
       "n": 101,
       "correct": 57,
       "said": {
        "true": 57,
        "unknown": 43,
        "false": 1
       }
      },
      {
       "depth": "2",
       "label": "false",
       "n": 101,
       "correct": 46,
       "said": {
        "false": 46,
        "unknown": 53,
        "true": 2
       }
      },
      {
       "depth": "2",
       "label": "unknown",
       "n": 101,
       "correct": 80,
       "said": {
        "unknown": 80,
        "false": 10,
        "true": 11
       }
      },
      {
       "depth": "1",
       "label": "true",
       "n": 101,
       "correct": 78,
       "said": {
        "true": 78,
        "unknown": 23
       }
      },
      {
       "depth": "1",
       "label": "false",
       "n": 101,
       "correct": 62,
       "said": {
        "false": 62,
        "unknown": 37,
        "true": 2
       }
      },
      {
       "depth": "1",
       "label": "unknown",
       "n": 100,
       "correct": 95,
       "said": {
        "unknown": 95,
        "true": 3,
        "false": 2
       }
      },
      {
       "depth": "0",
       "label": "true",
       "n": 100,
       "correct": 96,
       "said": {
        "true": 96,
        "unknown": 4
       }
      },
      {
       "depth": "0",
       "label": "false",
       "n": 100,
       "correct": 85,
       "said": {
        "false": 85,
        "unknown": 13,
        "true": 2
       }
      },
      {
       "depth": "0",
       "label": "unknown",
       "n": 100,
       "correct": 99,
       "said": {
        "unknown": 99,
        "true": 1
       }
      }
     ],
     "calibration": null,
     "models": [
      "gpt-6-luna"
     ],
     "file": "answers/openai-gpt-6-luna-effort-none/proofwriter-owa.jsonl.gz"
    }
   },
   "latency": {
    "jev": {
     "file": "timing/jev/proofwriter-owa.jsonl.gz",
     "n": 1800,
     "of": 1800,
     "complete": true,
     "p50": 189.08,
     "p90": 241.14,
     "max": 1259.07,
     "total_minutes": 6.186231666666665,
     "concurrency": 1,
     "machine": {
      "cpu": "Apple M1 Max",
      "memory_bytes": 34359738368,
      "model": "MacBookPro18,4",
      "platform": "macOS-26.6.2-arm64-arm-64bit",
      "python": "3.12.2"
     },
     "load_start": null,
     "load_end": null
    },
    "kev-0.8b": {
     "file": "timing/kev-0.8b/proofwriter-owa.jsonl.gz",
     "n": 1800,
     "of": 1800,
     "complete": true,
     "p50": 122.9,
     "p90": 151.91,
     "max": 379.36,
     "total_minutes": 3.7705958333333323,
     "concurrency": 1,
     "machine": {
      "cpu": "Apple M1 Max",
      "memory_bytes": 34359738368,
      "model": "MacBookPro18,4",
      "platform": "macOS-26.6.2-arm64-arm-64bit",
      "python": "3.12.2"
     },
     "load_start": [
      5.84,
      5.5,
      6.55
     ],
     "load_end": [
      5.2,
      5.56,
      6.33
     ]
    },
    "kev-4b": {
     "file": "timing/kev-4b/proofwriter-owa.jsonl.gz",
     "n": 1800,
     "of": 1800,
     "complete": true,
     "p50": 897.17,
     "p90": 1100.09,
     "max": 1431.46,
     "total_minutes": 27.127203666666652,
     "concurrency": 1,
     "machine": {
      "cpu": "Apple M1 Max",
      "memory_bytes": 34359738368,
      "model": "MacBookPro18,4",
      "platform": "macOS-26.6.2-arm64-arm-64bit",
      "python": "3.12.2"
     },
     "load_start": [
      7.26,
      6.97,
      7.22
     ],
     "load_end": [
      11.14,
      8.15,
      7.02
     ]
    },
    "laya": {
     "file": "timing/laya/proofwriter-owa.jsonl.gz",
     "n": 1800,
     "of": 1800,
     "complete": true,
     "p50": 140.32,
     "p90": 478.99,
     "max": 9460.03,
     "total_minutes": 6.67057816666666,
     "concurrency": 1,
     "machine": {
      "cpu": "Apple M1 Max",
      "memory_bytes": 34359738368,
      "model": "MacBookPro18,4",
      "platform": "macOS-26.6.2-arm64-arm-64bit",
      "python": "3.12.2"
     },
     "load_start": [
      10.03,
      9.27,
      8.55
     ],
     "load_end": [
      7.12,
      8.46,
      8.5
     ]
    },
    "openai-gpt-6-luna-effort-none": {
     "file": "timing/openai-gpt-6-luna-effort-none/proofwriter-owa.jsonl.gz",
     "n": 1800,
     "of": 1800,
     "complete": true,
     "p50": 824.52,
     "p90": 1165.42,
     "max": 5835.67,
     "total_minutes": 26.87960416666667,
     "concurrency": 1,
     "machine": {
      "cpu": "Apple M1 Max",
      "memory_bytes": 34359738368,
      "model": "MacBookPro18,4",
      "platform": "macOS-26.6.2-arm64-arm-64bit",
      "python": "3.12.2"
     },
     "load_start": [
      15.51,
      10.17,
      9.47
     ],
     "load_end": [
      7.88,
      8.32,
      8.96
     ]
    }
   }
  }
 },
 "manifests": [
  {
   "file": "answers/kev-4b/proofwriter-cwa.runs.jsonl",
   "tree": "answers",
   "engine": "kev-4b",
   "task": "proofwriter-cwa",
   "started_at": "2026-10-01T12:19:47.609+00:00",
   "finished_at": "2026-10-01T12:58:01.877+00:00",
   "requested": 1800,
   "answered": 1800,
   "failed": 0,
   "concurrency": 1,
   "machine": {
    "cpu": "Apple M1 Max",
    "memory_bytes": 34359738368,
    "model": "MacBookPro18,4",
    "platform": "macOS-26.6.2-arm64-arm-64bit",
    "python": "3.12.2"
   },
   "memory": null,
   "code": {
    "commit": "eb7cc28f4dba1c6ddc5f254e3bd22bfb7acd1b89",
    "dirty": false
   },
   "load_start": null,
   "load_end": null,
   "busy_processes": null
  },
  {
   "file": "answers/kev-9b/proofwriter-cwa.runs.jsonl",
   "tree": "answers",
   "engine": "kev-9b",
   "task": "proofwriter-cwa",
   "started_at": "2026-10-01T15:34:38.304+00:00",
   "finished_at": "2026-10-01T16:18:47.165+00:00",
   "requested": 1800,
   "answered": 1800,
   "failed": 0,
   "concurrency": 1,
   "machine": {
    "cpu": "Apple M1 Max",
    "memory_bytes": 34359738368,
    "model": "MacBookPro18,4",
    "platform": "macOS-26.6.2-arm64-arm-64bit",
    "python": "3.12.2"
   },
   "memory": {
    "phys_footprint_mb": 17408,
    "phys_footprint_peak_mb": 18432,
    "pid": 22803
   },
   "code": {
    "commit": "91c174320f6023562d657f08ced966064fc83ca1",
    "dirty": false
   },
   "load_start": [
    12.73,
    13.39,
    11.19
   ],
   "load_end": [
    5.12,
    6.96,
    7.86
   ],
   "busy_processes": 4
  },
  {
   "file": "answers/kev-9b/proofwriter-owa.runs.jsonl",
   "tree": "answers",
   "engine": "kev-9b",
   "task": "proofwriter-owa",
   "started_at": "2026-10-01T14:43:30.066+00:00",
   "finished_at": "2026-10-01T14:44:03.153+00:00",
   "requested": 20,
   "answered": 20,
   "failed": 0,
   "concurrency": 1,
   "machine": {
    "cpu": "Apple M1 Max",
    "memory_bytes": 34359738368,
    "model": "MacBookPro18,4",
    "platform": "macOS-26.6.2-arm64-arm-64bit",
    "python": "3.12.2"
   },
   "memory": {
    "phys_footprint_mb": 17408,
    "phys_footprint_peak_mb": 18432,
    "pid": 22803
   },
   "code": {
    "commit": "f98f4a4a8660ff54b5396147c3e19e730c795922",
    "dirty": true
   },
   "load_start": [
    7.95,
    7.82,
    8.05
   ],
   "load_end": [
    8.34,
    7.93,
    8.08
   ],
   "busy_processes": 3
  },
  {
   "file": "answers/kev-9b/proofwriter-owa.runs.jsonl",
   "tree": "answers",
   "engine": "kev-9b",
   "task": "proofwriter-owa",
   "started_at": "2026-10-01T14:44:14.148+00:00",
   "finished_at": "2026-10-01T15:34:37.453+00:00",
   "requested": 1780,
   "answered": 1780,
   "failed": 0,
   "concurrency": 1,
   "machine": {
    "cpu": "Apple M1 Max",
    "memory_bytes": 34359738368,
    "model": "MacBookPro18,4",
    "platform": "macOS-26.6.2-arm64-arm-64bit",
    "python": "3.12.2"
   },
   "memory": {
    "phys_footprint_mb": 17408,
    "phys_footprint_peak_mb": 18432,
    "pid": 22803
   },
   "code": {
    "commit": "91c174320f6023562d657f08ced966064fc83ca1",
    "dirty": false
   },
   "load_start": [
    7.72,
    7.82,
    8.04
   ],
   "load_end": [
    12.73,
    13.39,
    11.19
   ],
   "busy_processes": 3
  },
  {
   "file": "timing/jev/proofwriter-cwa.runs.jsonl",
   "tree": "timing",
   "engine": "jev",
   "task": "proofwriter-cwa",
   "started_at": "2026-10-01T12:04:40.726+00:00",
   "finished_at": "2026-10-01T12:09:58.147+00:00",
   "requested": 1800,
   "answered": 1800,
   "failed": 0,
   "concurrency": 1,
   "machine": {
    "cpu": "Apple M1 Max",
    "memory_bytes": 34359738368,
    "model": "MacBookPro18,4",
    "platform": "macOS-26.6.2-arm64-arm-64bit",
    "python": "3.12.2"
   },
   "memory": null,
   "code": {
    "commit": "372aba648aa16593bee7fb427648f2c7ccfe9251",
    "dirty": false
   },
   "load_start": null,
   "load_end": null,
   "busy_processes": null
  },
  {
   "file": "timing/jev/proofwriter-owa.runs.jsonl",
   "tree": "timing",
   "engine": "jev",
   "task": "proofwriter-owa",
   "started_at": "2026-10-01T11:58:28.241+00:00",
   "finished_at": "2026-10-01T12:04:40.391+00:00",
   "requested": 1800,
   "answered": 1800,
   "failed": 0,
   "concurrency": 1,
   "machine": {
    "cpu": "Apple M1 Max",
    "memory_bytes": 34359738368,
    "model": "MacBookPro18,4",
    "platform": "macOS-26.6.2-arm64-arm-64bit",
    "python": "3.12.2"
   },
   "memory": null,
   "code": {
    "commit": "372aba648aa16593bee7fb427648f2c7ccfe9251",
    "dirty": false
   },
   "load_start": null,
   "load_end": null,
   "busy_processes": null
  },
  {
   "file": "timing/kev-0.8b/proofwriter-cwa.runs.jsonl",
   "tree": "timing",
   "engine": "kev-0.8b",
   "task": "proofwriter-cwa",
   "started_at": "2026-10-01T16:32:21.054+00:00",
   "finished_at": "2026-10-01T16:36:05.108+00:00",
   "requested": 1800,
   "answered": 1800,
   "failed": 0,
   "concurrency": 1,
   "machine": {
    "cpu": "Apple M1 Max",
    "memory_bytes": 34359738368,
    "model": "MacBookPro18,4",
    "platform": "macOS-26.6.2-arm64-arm-64bit",
    "python": "3.12.2"
   },
   "memory": {
    "phys_footprint_mb": 2162,
    "phys_footprint_peak_mb": 3316,
    "pid": 57457
   },
   "code": {
    "commit": "91c174320f6023562d657f08ced966064fc83ca1",
    "dirty": false
   },
   "load_start": [
    5.2,
    5.56,
    6.33
   ],
   "load_end": [
    7.16,
    6.37,
    6.51
   ],
   "busy_processes": 1
  },
  {
   "file": "timing/kev-0.8b/proofwriter-owa.runs.jsonl",
   "tree": "timing",
   "engine": "kev-0.8b",
   "task": "proofwriter-owa",
   "started_at": "2026-10-01T16:28:33.656+00:00",
   "finished_at": "2026-10-01T16:32:20.682+00:00",
   "requested": 1800,
   "answered": 1800,
   "failed": 0,
   "concurrency": 1,
   "machine": {
    "cpu": "Apple M1 Max",
    "memory_bytes": 34359738368,
    "model": "MacBookPro18,4",
    "platform": "macOS-26.6.2-arm64-arm-64bit",
    "python": "3.12.2"
   },
   "memory": {
    "phys_footprint_mb": 2176,
    "phys_footprint_peak_mb": 3316,
    "pid": 57457
   },
   "code": {
    "commit": "91c174320f6023562d657f08ced966064fc83ca1",
    "dirty": false
   },
   "load_start": [
    5.84,
    5.5,
    6.55
   ],
   "load_end": [
    5.2,
    5.56,
    6.33
   ],
   "busy_processes": 2
  },
  {
   "file": "timing/kev-4b/proofwriter-cwa.runs.jsonl",
   "tree": "timing",
   "engine": "kev-4b",
   "task": "proofwriter-cwa",
   "started_at": "2026-10-01T13:56:30.221+00:00",
   "finished_at": "2026-10-01T14:21:30.818+00:00",
   "requested": 1800,
   "answered": 1800,
   "failed": 0,
   "concurrency": 1,
   "machine": {
    "cpu": "Apple M1 Max",
    "memory_bytes": 34359738368,
    "model": "MacBookPro18,4",
    "platform": "macOS-26.6.2-arm64-arm-64bit",
    "python": "3.12.2"
   },
   "memory": {
    "phys_footprint_mb": 8819,
    "phys_footprint_peak_mb": 17408,
    "pid": 1575
   },
   "code": {
    "commit": "a19776a55601ca5ac97297fb35f410c263cf491f",
    "dirty": false
   },
   "load_start": [
    11.14,
    8.15,
    7.02
   ],
   "load_end": [
    10.56,
    9.36,
    8.58
   ],
   "busy_processes": 1
  },
  {
   "file": "timing/kev-4b/proofwriter-owa.runs.jsonl",
   "tree": "timing",
   "engine": "kev-4b",
   "task": "proofwriter-owa",
   "started_at": "2026-10-01T13:29:21.362+00:00",
   "finished_at": "2026-10-01T13:56:29.799+00:00",
   "requested": 1800,
   "answered": 1800,
   "failed": 0,
   "concurrency": 1,
   "machine": {
    "cpu": "Apple M1 Max",
    "memory_bytes": 34359738368,
    "model": "MacBookPro18,4",
    "platform": "macOS-26.6.2-arm64-arm-64bit",
    "python": "3.12.2"
   },
   "memory": {
    "phys_footprint_mb": 8819,
    "phys_footprint_peak_mb": 17408,
    "pid": 1575
   },
   "code": {
    "commit": "a19776a55601ca5ac97297fb35f410c263cf491f",
    "dirty": false
   },
   "load_start": [
    7.26,
    6.97,
    7.22
   ],
   "load_end": [
    11.14,
    8.15,
    7.02
   ],
   "busy_processes": 4
  },
  {
   "file": "timing/laya/proofwriter-cwa.runs.jsonl",
   "tree": "timing",
   "engine": "laya",
   "task": "proofwriter-cwa",
   "started_at": "2026-10-01T14:28:20.463+00:00",
   "finished_at": "2026-10-01T14:34:02.758+00:00",
   "requested": 1800,
   "answered": 1800,
   "failed": 0,
   "concurrency": 1,
   "machine": {
    "cpu": "Apple M1 Max",
    "memory_bytes": 34359738368,
    "model": "MacBookPro18,4",
    "platform": "macOS-26.6.2-arm64-arm-64bit",
    "python": "3.12.2"
   },
   "memory": {
    "phys_footprint_mb": 10240,
    "phys_footprint_peak_mb": 10240,
    "pid": 17800
   },
   "code": {
    "commit": "de9e832161b6534aaa9516b094fe021cd51412cb",
    "dirty": false
   },
   "load_start": [
    7.12,
    8.46,
    8.5
   ],
   "load_end": [
    9.58,
    9.01,
    8.72
   ],
   "busy_processes": 4
  },
  {
   "file": "timing/laya/proofwriter-owa.runs.jsonl",
   "tree": "timing",
   "engine": "laya",
   "task": "proofwriter-owa",
   "started_at": "2026-10-01T14:21:36.507+00:00",
   "finished_at": "2026-10-01T14:28:17.793+00:00",
   "requested": 1800,
   "answered": 1800,
   "failed": 0,
   "concurrency": 1,
   "machine": {
    "cpu": "Apple M1 Max",
    "memory_bytes": 34359738368,
    "model": "MacBookPro18,4",
    "platform": "macOS-26.6.2-arm64-arm-64bit",
    "python": "3.12.2"
   },
   "memory": {
    "phys_footprint_mb": 10240,
    "phys_footprint_peak_mb": 10240,
    "pid": 16278
   },
   "code": {
    "commit": "de9e832161b6534aaa9516b094fe021cd51412cb",
    "dirty": false
   },
   "load_start": [
    10.03,
    9.27,
    8.55
   ],
   "load_end": [
    7.12,
    8.46,
    8.5
   ],
   "busy_processes": 5
  },
  {
   "file": "timing/openai-gpt-6-luna-effort-none/proofwriter-cwa.runs.jsonl",
   "tree": "timing",
   "engine": "openai-gpt-6-luna-effort-none",
   "task": "proofwriter-cwa",
   "started_at": "2026-10-01T15:57:54.689+00:00",
   "finished_at": "2026-10-01T16:27:50.481+00:00",
   "requested": 1800,
   "answered": 1800,
   "failed": 0,
   "concurrency": 1,
   "machine": {
    "cpu": "Apple M1 Max",
    "memory_bytes": 34359738368,
    "model": "MacBookPro18,4",
    "platform": "macOS-26.6.2-arm64-arm-64bit",
    "python": "3.12.2"
   },
   "memory": null,
   "code": {
    "commit": "91c174320f6023562d657f08ced966064fc83ca1",
    "dirty": false
   },
   "load_start": [
    7.88,
    8.32,
    8.96
   ],
   "load_end": [
    4.64,
    5.36,
    6.57
   ],
   "busy_processes": 2
  },
  {
   "file": "timing/openai-gpt-6-luna-effort-none/proofwriter-owa.runs.jsonl",
   "tree": "timing",
   "engine": "openai-gpt-6-luna-effort-none",
   "task": "proofwriter-owa",
   "started_at": "2026-10-01T15:30:59.200+00:00",
   "finished_at": "2026-10-01T15:57:54.205+00:00",
   "requested": 1800,
   "answered": 1800,
   "failed": 0,
   "concurrency": 1,
   "machine": {
    "cpu": "Apple M1 Max",
    "memory_bytes": 34359738368,
    "model": "MacBookPro18,4",
    "platform": "macOS-26.6.2-arm64-arm-64bit",
    "python": "3.12.2"
   },
   "memory": null,
   "code": {
    "commit": "91c174320f6023562d657f08ced966064fc83ca1",
    "dirty": false
   },
   "load_start": [
    15.51,
    10.17,
    9.47
   ],
   "load_end": [
    7.88,
    8.32,
    8.96
   ],
   "busy_processes": 8
  }
 ],
 "prereg": {
  "file": "docs/preregistration.md",
  "modified": "2026-10-01T14:57:01.393Z",
  "intro": "Written before any engine has answered an item. Scored against these predictions word for word.",
  "design": [
   "**Data:** ProofWriter V2020.12.3, official test splits (`depth-0,1,2,3,5` and `NatLang`), archive sha256",
   "**Sample:** `hd build --n 1800 --seed 0` (1,800 items per semantics, difficulty-diverse, nested by seed).",
   "**Tasks:** `proofwriter-owa` (true / false / unknown) and `proofwriter-cwa` (true / false). One request per",
   "**Engine:** Jev (`typesafe-sdk`), model version recorded per row. A version change is a new engine.",
   "**Primary metric:** accuracy by proof depth (0 to 5) with a 95% percentile bootstrap interval (seed 0, 1,000",
   "**Floors:** chance (1 / number of options) and the best constant guesser on the same items."
  ],
  "sections": [
   {
    "heading": "Predictions",
    "predictions": [
     {
      "n": 1,
      "text": "Jev's accuracy falls as proof depth rises, on both tasks. Depth 0 is clearly above depth 5, with intervals that do not overlap on OWA."
     },
     {
      "n": 2,
      "text": "On OWA, Jev's recall on `unknown` is the lowest of the three classes at depth 3 and above."
     },
     {
      "n": 3,
      "text": "By depth 5, Jev is within 10 points of its own best-constant floor on at least one task (that is, it has largely stopped using the rules)."
     },
     {
      "n": 4,
      "text": "Negation in the theory (`theory_negation = negation`) lowers accuracy relative to no negation."
     },
     {
      "n": 5,
      "text": "Paraphrased rules (`NatLang`) are no easier than templated ones."
     },
     {
      "n": 6,
      "text": "No claim about the other axes (length, rule count, proof size, strategy) beyond reporting them; they correlate with depth, so those breakdowns are descriptive."
     }
    ],
    "text": "1. Jev's accuracy falls as proof depth rises, on both tasks. Depth 0 is clearly above depth 5, with intervals\n   that do not overlap on OWA.\n2. On OWA, Jev's recall on `unknown` is the lowest of the three classes at depth 3 and above.\n3. By depth 5, Jev is within 10 points of its own best-constant floor on at least one task (that is, it has\n   largely stopped using the rules).\n4. Negation in the theory (`theory_negation = negation`) lowers accuracy relative to no negation.\n5. Paraphrased rules (`NatLang`) are no easier than templated ones.\n6. No claim about the other axes (length, rule count, proof size, strategy) beyond reporting them; they correlate\n   with depth, so those breakdowns are descriptive."
   },
   {
    "heading": "What would count against us",
    "predictions": [],
    "text": "Accuracy flat across depth (prediction 1 false), or `unknown` recall not the lowest (prediction 2 false), is\nreported as such."
   },
   {
    "heading": "Caveats stated in advance",
    "predictions": [],
    "text": "ProofWriter is synthetic and templated, and models may have seen it in training. Interval widths at 100 to 150\nitems per depth are about plus or minus 8 to 10 points per class-balanced stratum, so only large differences are\nresolvable at the depth-by-label level. Breakdowns are descriptive, not causal."
   },
   {
    "heading": "Amendment 1: Kev (written before Kev answered any item)",
    "predictions": [
     {
      "n": 1,
      "text": "Kev's accuracy falls with proof depth on both tasks."
     },
     {
      "n": 2,
      "text": "Kev is below Jev overall on both tasks (paired interval excludes zero)."
     },
     {
      "n": 3,
      "text": "On OWA, Kev's `unknown` recall is its lowest class recall."
     }
    ],
    "text": "- **Engine:** `kev-0.8b`, Kev's pointer head on Qwen3.5-0.8B-Base, served locally through the same `/v1/systemone`\n  contract and the same typed question as Jev. Pinned identity, identical to the Biased-Decisions Kev study:\n  checkpoint `jaredpalmer/kev-0.8b@54f4f8777356cd5bbbb6c6919c657f26e6f2f6d8`, base\n  `Qwen/Qwen3.5-0.8B-Base@dc7cdfe2ee4154fa7e30f5b51ca41bfa40174e68`, server commit\n  `c9c1f855505336ac32092a5f68305d397f7fcc3e`, MLX bfloat16, prefix cache and date facts off, stored temperature\n  2.406. The server's `/v1/models` response is saved in `answers/kev-0.8b/provenance.json`.\n- **Protocol:** one request at a time; a 20-item OWA timing pilot, then the full 1,800 items per task. Same\n  sample, metrics and floors as above. Paired differences are reported as Jev minus Kev on the same items.\n- **Predictions:**\n  1. Kev's accuracy falls with proof depth on both tasks.\n  2. Kev is below Jev overall on both tasks (paired interval excludes zero).\n  3. On OWA, Kev's `unknown` recall is its lowest class recall."
   },
   {
    "heading": "Amendment 2: LLM classifiers (written before any LLM answered an item)",
    "predictions": [
     {
      "n": 1,
      "text": "Luna's accuracy falls with proof depth on both tasks."
     },
     {
      "n": 2,
      "text": "On OWA, Luna's `unknown` recall is its lowest class recall."
     },
     {
      "n": 3,
      "text": "With reasoning off, Luna does not beat Jev at depth 5 on OWA (the paired interval does not exclude zero in Luna's favour)."
     },
     {
      "n": 4,
      "text": "Paraphrased rules lower Luna's accuracy less than they lower Jev's."
     }
    ],
    "text": "- **Engine:** `openai-gpt-6-luna-effort-none`: OpenAI `gpt-6-luna` through Chat Completions with\n  `reasoning_effort: none`, its lowest setting, so it answers directly as a classifier. The vendor's returned\n  model id is recorded per row.\n- **Protocol:** one request per item. The prompt is the item text, then the task's question and each option with\n  its description (the same wire question every engine gets), asking for `{\"answer\": <option>}`. The reply is\n  constrained by a strict JSON schema with the options as an enum. Any reply that is not exactly one option is\n  scored as invalid and kept raw in the record; it is never re-asked. Same sample, metrics and floors as above;\n  paired differences are Jev minus Luna on the same items.\n- **Spend:** list price $0.10 input and $0.50 output per million tokens; the dry run prices both tasks at about\n  $0.15 in total.\n- **Claude engines** (`hard_decisions/engines/llm.py`) are implemented but not run: there is no Anthropic key.\n- **Predictions:**\n  1. Luna's accuracy falls with proof depth on both tasks.\n  2. On OWA, Luna's `unknown` recall is its lowest class recall.\n  3. With reasoning off, Luna does not beat Jev at depth 5 on OWA (the paired interval does not exclude zero in\n     Luna's favour).\n  4. Paraphrased rules lower Luna's accuracy less than they lower Jev's."
   },
   {
    "heading": "Amendment 3: Kev-4B and Kev-9B (written before either answered an item)",
    "predictions": [
     {
      "n": 1,
      "text": "Accuracy rises with size: Kev-4B is above Kev-0.8B overall on both tasks (paired interval excludes zero)."
     },
     {
      "n": 2,
      "text": "Kev-4B and Kev-9B stay below Jev overall on both tasks."
     },
     {
      "n": 3,
      "text": "Kev-9B is not reliably above Kev-4B (paired interval includes zero) on at least one task."
     }
    ],
    "text": "- **Engines:** `kev-4b` (`jaredpalmer/kev-4b@139fdd94f1b6a6ad80cc15e08fcb99cac885a101` on\n  `Qwen/Qwen3.5-4B-Base@1001bb4d826a52d1f399e183466143f4da7b741b`) and, if memory allows, `kev-9b`\n  (`jaredpalmer/kev-9b@b5d8c18e44c60888d138b65cb6507ff0a5a448a0` on\n  `Qwen/Qwen3.5-9B-Base@68c46c4b3498877f3ef123c856ecfde50c39f404`). Same server commit, MLX bfloat16 and settings as\n  `kev-0.8b`; each checkpoint's own stored temperature (4B: 2.406). Weights in a gitignored cache; the server's\n  `/v1/models` response is saved in `answers/<engine>/provenance.json`. Kev-27B (51 GB) does not fit this machine\n  and is not run.\n- **Protocol:** identical to every other engine: same items, same question, one request at a time, a 20-item OWA\n  pilot, then 1,800 items per task.\n- **Predictions:**\n  1. Accuracy rises with size: Kev-4B is above Kev-0.8B overall on both tasks (paired interval excludes zero).\n  2. Kev-4B and Kev-9B stay below Jev overall on both tasks.\n  3. Kev-9B is not reliably above Kev-4B (paired interval includes zero) on at least one task."
   },
   {
    "heading": "Amendment 4: test-retest repeatability (written before any rerun except Jev's)",
    "predictions": [
     {
      "n": 1,
      "text": "Kev and Laya agree with themselves on at least 99.5% of items on both tasks (local, fixed weights)."
     },
     {
      "n": 2,
      "text": "Jev's AC1 is at least 0.90 on both tasks (observed: 0.958 on both, before this amendment)."
     },
     {
      "n": 3,
      "text": "Where answers change, they change more at depth 3 and above than at depth 0 to 2, for every engine with at least 20 changes."
     }
    ],
    "text": "- **Design:** every engine answers the full sample a second time under the original protocol, one request at a\n  time, into `timing/<engine>/<task>.jsonl.gz`. That rerun also provides the latency data and is never scored.\n  Jev's rerun was collected for timing before this metric was chosen; every other rerun comes after this\n  amendment is committed.\n- **Metrics, per engine and task:** percent agreement (headline); Gwet's AC1 with a 95% percentile bootstrap\n  interval over items (seed 0, 1,000 resamples), the primary chance-corrected measure, chosen because several\n  engines answer one option far more often than the others and Cohen's kappa understates agreement then (the\n  prevalence paradox); Cohen's kappa as a secondary figure; share of answers that change, by proof depth; accuracy in\n  each run; for engines that return probabilities, the mean and maximum absolute change in run 1's chosen\n  option's probability.\n- **Predictions:**\n  1. Kev and Laya agree with themselves on at least 99.5% of items on both tasks (local, fixed weights).\n  2. Jev's AC1 is at least 0.90 on both tasks (observed: 0.958 on both, before this amendment).\n  3. Where answers change, they change more at depth 3 and above than at depth 0 to 2, for every engine with at\n     least 20 changes."
   },
   {
    "heading": "Amendment 6: can GPT-6 Luna's log-probabilities serve as a confidence? (written before the probe ran)",
    "predictions": [
     {
      "n": 1,
      "text": "Luna's ECE exceeds 0.15 on both tasks."
     },
     {
      "n": 2,
      "text": "At least half of Luna's wrong answers are stated at 95% or more, on both tasks."
     },
     {
      "n": 3,
      "text": "Luna's AUROC is lower than Jev's on both tasks."
     },
     {
      "n": 4,
      "text": "On most responses, Luna discloses fewer than all of the task's options."
     }
    ],
    "text": "- **Why:** OpenAI's Decisions API is built on a version of GPT-6 Luna. A decision model is only useful with a\n  confidence you can threshold on. Our Luna arm asked for the answer only, like every engine, so it has none. This\n  probe asks whether Luna's log-probabilities could supply one. It is outside the scored benchmark and changes no\n  benchmark number.\n- **A spot check before this amendment** (12 open-world items at depth 3 to 5, 2026-10-01) found Luna returns the\n  chosen answer's log-probability and at most one or two alternatives with reasoning off, refuses `logprobs` with\n  reasoning on, and gave 5 of 12 wrong answers at 97.8% or more. It motivated the probe and is not part of its\n  results.\n- **Design:** every one of the 3,600 benchmark items, sent with exactly the benchmark's GPT-6 Luna request (same\n  prompt, strict JSON schema, `reasoning_effort: none`, default sampling) plus `logprobs: true, top_logprobs: 5`.\n  Every request and response is kept verbatim in `probes/luna-logprobs/` (`tools/probe_luna_logprobs.py`). The\n  answer's probability is the summed probability of the tokens that spell it.\n- **Metrics** (`tools/analyze_luna_logprobs.py`): a reliability table in six probability bands; expected\n  calibration error (ECE); the share of wrong answers stated at 95% or more and at 99% or more; AUROC of the stated\n  probability for separating right from wrong answers, computed the same way for Jev, Kev and Laya on the same\n  items from their recorded probabilities; how many alternatives Luna discloses; and how often the returned answer\n  is not the most probable disclosed option.\n- **Predictions:**\n  1. Luna's ECE exceeds 0.15 on both tasks.\n  2. At least half of Luna's wrong answers are stated at 95% or more, on both tasks.\n  3. Luna's AUROC is lower than Jev's on both tasks.\n  4. On most responses, Luna discloses fewer than all of the task's options."
   }
  ]
 },
 "probe": {
  "file": "probes/luna-logprobs/analysis.json",
  "modified": "2026-10-01T15:05:10.752Z",
  "tasks": {
   "proofwriter-cwa": {
    "accuracy": 0.6483333333333333,
    "auroc": 0.6789129984527102,
    "bands": [
     {
      "accuracy": 0.49230769230769234,
      "from": 0,
      "mean_stated": 0.2911070032314609,
      "n": 65,
      "to": 0.5
     },
     {
      "accuracy": 0.527027027027027,
      "from": 0.5,
      "mean_stated": 0.6815618963560237,
      "n": 74,
      "to": 0.8
     },
     {
      "accuracy": 0.54,
      "from": 0.8,
      "mean_stated": 0.8612726000977343,
      "n": 50,
      "to": 0.9
     },
     {
      "accuracy": 0.5428571428571428,
      "from": 0.9,
      "mean_stated": 0.9275551275285234,
      "n": 70,
      "to": 0.95
     },
     {
      "accuracy": 0.5167785234899329,
      "from": 0.95,
      "mean_stated": 0.974773335054445,
      "n": 149,
      "to": 0.99
     },
     {
      "accuracy": 0.6810218978102189,
      "from": 0.99,
      "mean_stated": 0.9993562927506754,
      "n": 1370,
      "to": 1
     }
    ],
    "by_depth": {
     "0": {
      "accuracy": 0.98,
      "auroc": 0.9424603174603174,
      "bands": [
       {
        "accuracy": 0.6666666666666666,
        "from": 0,
        "mean_stated": 0.1644404773999861,
        "n": 3,
        "to": 0.5
       },
       {
        "accuracy": 1,
        "from": 0.5,
        "mean_stated": 0.7788542609133176,
        "n": 1,
        "to": 0.8
       },
       {
        "accuracy": null,
        "from": 0.8,
        "mean_stated": null,
        "n": 0,
        "to": 0.9
       },
       {
        "accuracy": 0.75,
        "from": 0.9,
        "mean_stated": 0.9313843096862391,
        "n": 4,
        "to": 0.95
       },
       {
        "accuracy": null,
        "from": 0.95,
        "mean_stated": null,
        "n": 0,
        "to": 0.99
       },
       {
        "accuracy": 0.9858156028368794,
        "from": 0.99,
        "mean_stated": 0.9999392782975487,
        "n": 282,
        "to": 1
       }
      ],
      "ece": 0.021454126751801392,
      "mean_stated": 0.989935425196041,
      "n": 300,
      "wrong": 6,
      "wrong_at_95": 0.6666666666666666,
      "wrong_at_99": 0.6666666666666666
     },
     "1": {
      "accuracy": 0.7766666666666666,
      "auroc": 0.751393248350522,
      "bands": [
       {
        "accuracy": 0.16666666666666666,
        "from": 0,
        "mean_stated": 0.24237824838760755,
        "n": 6,
        "to": 0.5
       },
       {
        "accuracy": 0.45454545454545453,
        "from": 0.5,
        "mean_stated": 0.6619650162182332,
        "n": 11,
        "to": 0.8
       },
       {
        "accuracy": 1,
        "from": 0.8,
        "mean_stated": 0.8823908941605126,
        "n": 6,
        "to": 0.9
       },
       {
        "accuracy": 0.8,
        "from": 0.9,
        "mean_stated": 0.9219880228748595,
        "n": 10,
        "to": 0.95
       },
       {
        "accuracy": 0.7391304347826086,
        "from": 0.95,
        "mean_stated": 0.9740503513889824,
        "n": 23,
        "to": 0.99
       },
       {
        "accuracy": 0.8008298755186722,
        "from": 0.99,
        "mean_stated": 0.9994110589071132,
        "n": 241,
        "to": 1
       }
      ],
      "ece": 0.19307547603690867,
      "mean_stated": 0.9650378166170412,
      "n": 300,
      "wrong": 67,
      "wrong_at_95": 0.8059701492537313,
      "wrong_at_99": 0.7164179104477612
     },
     "2": {
      "accuracy": 0.6433333333333333,
      "auroc": 0.638080480364147,
      "bands": [
       {
        "accuracy": 0.5,
        "from": 0,
        "mean_stated": 0.3467506479471365,
        "n": 14,
        "to": 0.5
       },
       {
        "accuracy": 0.5,
        "from": 0.5,
        "mean_stated": 0.6632102286657326,
        "n": 10,
        "to": 0.8
       },
       {
        "accuracy": 0.6666666666666666,
        "from": 0.8,
        "mean_stated": 0.8649607608136565,
        "n": 12,
        "to": 0.9
       },
       {
        "accuracy": 0.5,
        "from": 0.9,
        "mean_stated": 0.9242110449667916,
        "n": 16,
        "to": 0.95
       },
       {
        "accuracy": 0.52,
        "from": 0.95,
        "mean_stated": 0.9777012962747692,
        "n": 25,
        "to": 0.99
       },
       {
        "accuracy": 0.6801801801801802,
        "from": 0.99,
        "mean_stated": 0.999312308902172,
        "n": 222,
        "to": 1
       }
      ],
      "ece": 0.3174478801592712,
      "mean_stated": 0.946477953350019,
      "n": 300,
      "wrong": 107,
      "wrong_at_95": 0.7757009345794392,
      "wrong_at_99": 0.6635514018691588
     },
     "3": {
      "accuracy": 0.5733333333333334,
      "auroc": 0.5633402979651163,
      "bands": [
       {
        "accuracy": 0.5555555555555556,
        "from": 0,
        "mean_stated": 0.3400159380573663,
        "n": 9,
        "to": 0.5
       },
       {
        "accuracy": 0.7142857142857143,
        "from": 0.5,
        "mean_stated": 0.6792316325768597,
        "n": 14,
        "to": 0.8
       },
       {
        "accuracy": 0.4166666666666667,
        "from": 0.8,
        "mean_stated": 0.866581600563351,
        "n": 12,
        "to": 0.9
       },
       {
        "accuracy": 0.6,
        "from": 0.9,
        "mean_stated": 0.9268193251116719,
        "n": 10,
        "to": 0.95
       },
       {
        "accuracy": 0.4864864864864865,
        "from": 0.95,
        "mean_stated": 0.972933772858035,
        "n": 37,
        "to": 0.99
       },
       {
        "accuracy": 0.5852534562211982,
        "from": 0.99,
        "mean_stated": 0.9990451329240364,
        "n": 217,
        "to": 1
       }
      ],
      "ece": 0.39629709866515933,
      "mean_stated": 0.9534263533714566,
      "n": 300,
      "wrong": 128,
      "wrong_at_95": 0.8515625,
      "wrong_at_99": 0.703125
     },
     "4": {
      "accuracy": 0.44,
      "auroc": 0.5562770562770563,
      "bands": [
       {
        "accuracy": 0.5263157894736842,
        "from": 0,
        "mean_stated": 0.2760520158630455,
        "n": 19,
        "to": 0.5
       },
       {
        "accuracy": 0.4583333333333333,
        "from": 0.5,
        "mean_stated": 0.6877090176411231,
        "n": 24,
        "to": 0.8
       },
       {
        "accuracy": 0.2,
        "from": 0.8,
        "mean_stated": 0.8538965087880882,
        "n": 10,
        "to": 0.9
       },
       {
        "accuracy": 0.42857142857142855,
        "from": 0.9,
        "mean_stated": 0.9280511600438276,
        "n": 14,
        "to": 0.95
       },
       {
        "accuracy": 0.35714285714285715,
        "from": 0.95,
        "mean_stated": 0.9758526861164062,
        "n": 28,
        "to": 0.99
       },
       {
        "accuracy": 0.4482758620689655,
        "from": 0.99,
        "mean_stated": 0.9989689124443156,
        "n": 203,
        "to": 1
       }
      ],
      "ece": 0.5096875796264633,
      "mean_stated": 0.9179875270671459,
      "n": 300,
      "wrong": 168,
      "wrong_at_95": 0.7738095238095238,
      "wrong_at_99": 0.6666666666666666
     },
     "5": {
      "accuracy": 0.4766666666666667,
      "auroc": 0.5023384259053049,
      "bands": [
       {
        "accuracy": 0.5,
        "from": 0,
        "mean_stated": 0.2724802480246628,
        "n": 14,
        "to": 0.5
       },
       {
        "accuracy": 0.5,
        "from": 0.5,
        "mean_stated": 0.6949106660651106,
        "n": 14,
        "to": 0.8
       },
       {
        "accuracy": 0.6,
        "from": 0.8,
        "mean_stated": 0.8451811215518669,
        "n": 10,
        "to": 0.9
       },
       {
        "accuracy": 0.4375,
        "from": 0.9,
        "mean_stated": 0.9334472030190073,
        "n": 16,
        "to": 0.95
       },
       {
        "accuracy": 0.5277777777777778,
        "from": 0.95,
        "mean_stated": 0.9742531007580502,
        "n": 36,
        "to": 0.99
       },
       {
        "accuracy": 0.45365853658536587,
        "from": 0.99,
        "mean_stated": 0.9992505548532111,
        "n": 205,
        "to": 1
       }
      ],
      "ece": 0.4807348922952905,
      "mean_stated": 0.939499779022668,
      "n": 300,
      "wrong": 157,
      "wrong_at_95": 0.8280254777070064,
      "wrong_at_99": 0.7197452229299363
     }
    },
    "ece": 0.3177030344687343,
    "mean_disclosed_mass": 0.9990123341586026,
    "mean_stated": 0.9520608091040621,
    "n": 1800,
    "options_disclosed": {
     "1": 1455,
     "2": 345
    },
    "other_engines_same_items": {
     "jev": {
      "accuracy": 0.8927777777777778,
      "auroc": 0.8575500320811476,
      "bands": [
       {
        "accuracy": null,
        "from": 0,
        "mean_stated": null,
        "n": 0,
        "to": 0.5
       },
       {
        "accuracy": 0.6164874551971327,
        "from": 0.5,
        "mean_stated": 0.65584229390681,
        "n": 279,
        "to": 0.8
       },
       {
        "accuracy": 0.7834394904458599,
        "from": 0.8,
        "mean_stated": 0.8438853503184714,
        "n": 157,
        "to": 0.9
       },
       {
        "accuracy": 0.8613138686131386,
        "from": 0.9,
        "mean_stated": 0.9211678832116789,
        "n": 137,
        "to": 0.95
       },
       {
        "accuracy": 0.9288389513108615,
        "from": 0.95,
        "mean_stated": 0.9684269662921348,
        "n": 267,
        "to": 0.99
       },
       {
        "accuracy": 0.9854166666666667,
        "from": 0.99,
        "mean_stated": 0.9975625,
        "n": 960,
        "to": 1
       }
      ],
      "ece": 0.02827777777777775,
      "n": 1800,
      "wrong_at_95": 0.17098445595854922,
      "wrong_at_99": 0.07253886010362694
     },
     "kev-0.8b": {
      "accuracy": 0.5577777777777778,
      "auroc": 0.5727072864321608,
      "bands": [
       {
        "accuracy": null,
        "from": 0,
        "mean_stated": null,
        "n": 0,
        "to": 0.5
       },
       {
        "accuracy": 0.5200655200655201,
        "from": 0.5,
        "mean_stated": 0.6559042588042588,
        "n": 1221,
        "to": 0.8
       },
       {
        "accuracy": 0.6446280991735537,
        "from": 0.8,
        "mean_stated": 0.848842699724518,
        "n": 363,
        "to": 0.9
       },
       {
        "accuracy": 0.6,
        "from": 0.9,
        "mean_stated": 0.924230303030303,
        "n": 165,
        "to": 0.95
       },
       {
        "accuracy": 0.7058823529411765,
        "from": 0.95,
        "mean_stated": 0.9627803921568627,
        "n": 51,
        "to": 0.99
       },
       {
        "accuracy": null,
        "from": 0.99,
        "mean_stated": null,
        "n": 0,
        "to": 1
       }
      ],
      "ece": 0.17032711111111112,
      "n": 1800,
      "wrong_at_95": 0.018844221105527637,
      "wrong_at_99": 0
     },
     "kev-4b": {
      "accuracy": 0.5877777777777777,
      "auroc": 0.5906232325651307,
      "bands": [
       {
        "accuracy": null,
        "from": 0,
        "mean_stated": null,
        "n": 0,
        "to": 0.5
       },
       {
        "accuracy": 0.5623342175066313,
        "from": 0.5,
        "mean_stated": 0.6259690981432361,
        "n": 1508,
        "to": 0.8
       },
       {
        "accuracy": 0.6936170212765957,
        "from": 0.8,
        "mean_stated": 0.8453782978723404,
        "n": 235,
        "to": 0.9
       },
       {
        "accuracy": 0.8,
        "from": 0.9,
        "mean_stated": 0.921086,
        "n": 50,
        "to": 0.95
       },
       {
        "accuracy": 1,
        "from": 0.95,
        "mean_stated": 0.9529857142857142,
        "n": 7,
        "to": 0.99
       },
       {
        "accuracy": null,
        "from": 0.99,
        "mean_stated": null,
        "n": 0,
        "to": 1
       }
      ],
      "ece": 0.07667150000000003,
      "n": 1800,
      "wrong_at_95": 0,
      "wrong_at_99": 0
     },
     "laya": {
      "accuracy": 0.5561111111111111,
      "auroc": 0.5602557642607705,
      "bands": [
       {
        "accuracy": null,
        "from": 0,
        "mean_stated": null,
        "n": 0,
        "to": 0.5
       },
       {
        "accuracy": 0.5401662049861495,
        "from": 0.5,
        "mean_stated": 0.6738072022160665,
        "n": 361,
        "to": 0.8
       },
       {
        "accuracy": 0.5026041666666666,
        "from": 0.8,
        "mean_stated": 0.8577484375000001,
        "n": 384,
        "to": 0.9
       },
       {
        "accuracy": 0.5157894736842106,
        "from": 0.9,
        "mean_stated": 0.9285298245614034,
        "n": 570,
        "to": 0.95
       },
       {
        "accuracy": 0.6560509554140127,
        "from": 0.95,
        "mean_stated": 0.9668768577494692,
        "n": 471,
        "to": 0.99
       },
       {
        "accuracy": 0.7142857142857143,
        "from": 0.99,
        "mean_stated": 0.9939857142857144,
        "n": 14,
        "to": 1
       }
      ],
      "ece": 0.3167758888888889,
      "n": 1800,
      "wrong_at_95": 0.20775969962453067,
      "wrong_at_99": 0.0050062578222778474
     }
    },
    "responses": 1800,
    "returned_not_most_probable": 65,
    "unparsed": 0,
    "wrong": 633,
    "wrong_at_95": 0.8056872037914692,
    "wrong_at_99": 0.6919431279620853,
    "other_engines_by_depth": {
     "jev": {
      "0": {
       "n": 300,
       "accuracy": 0.9933333333333333,
       "auroc": 0.9781879194630873
      },
      "1": {
       "n": 300,
       "accuracy": 0.9366666666666666,
       "auroc": 0.8722607229818318
      },
      "2": {
       "n": 300,
       "accuracy": 0.92,
       "auroc": 0.8260114734299517
      },
      "3": {
       "n": 300,
       "accuracy": 0.85,
       "auroc": 0.7983006535947712
      },
      "4": {
       "n": 300,
       "accuracy": 0.7633333333333333,
       "auroc": 0.7958361522848884
      },
      "5": {
       "n": 300,
       "accuracy": 0.8933333333333333,
       "auroc": 0.8509211753731343
      }
     },
     "kev-4b": {
      "0": {
       "n": 300,
       "accuracy": 0.8566666666666667,
       "auroc": 0.8330467830965523
      },
      "1": {
       "n": 300,
       "accuracy": 0.6833333333333333,
       "auroc": 0.6284980744544287
      },
      "2": {
       "n": 300,
       "accuracy": 0.5433333333333333,
       "auroc": 0.5855537145671936
      },
      "3": {
       "n": 300,
       "accuracy": 0.56,
       "auroc": 0.48579545454545453
      },
      "4": {
       "n": 300,
       "accuracy": 0.45,
       "auroc": 0.5046913580246913
      },
      "5": {
       "n": 300,
       "accuracy": 0.43333333333333335,
       "auroc": 0.44941176470588234
      }
     },
     "kev-0.8b": {
      "0": {
       "n": 300,
       "accuracy": 0.68,
       "auroc": 0.6609477124183006
      },
      "1": {
       "n": 300,
       "accuracy": 0.5966666666666667,
       "auroc": 0.5362205087954199
      },
      "2": {
       "n": 300,
       "accuracy": 0.5133333333333333,
       "auroc": 0.5479896815513254
      },
      "3": {
       "n": 300,
       "accuracy": 0.53,
       "auroc": 0.5361969757794728
      },
      "4": {
       "n": 300,
       "accuracy": 0.51,
       "auroc": 0.5212974078520297
      },
      "5": {
       "n": 300,
       "accuracy": 0.5166666666666667,
       "auroc": 0.5970189098998888
      }
     },
     "laya": {
      "0": {
       "n": 300,
       "accuracy": 0.72,
       "auroc": 0.6732253086419753
      },
      "1": {
       "n": 300,
       "accuracy": 0.51,
       "auroc": 0.47201102663287536
      },
      "2": {
       "n": 300,
       "accuracy": 0.5466666666666666,
       "auroc": 0.5242333213773315
      },
      "3": {
       "n": 300,
       "accuracy": 0.5133333333333333,
       "auroc": 0.5608210282867817
      },
      "4": {
       "n": 300,
       "accuracy": 0.5866666666666667,
       "auroc": 0.5472186583577713
      },
      "5": {
       "n": 300,
       "accuracy": 0.46,
       "auroc": 0.5489130434782609
      }
     },
     "kev-9b": {
      "0": {
       "n": 300,
       "accuracy": 0.8966666666666666,
       "auroc": 0.8365511452212495
      },
      "1": {
       "n": 300,
       "accuracy": 0.77,
       "auroc": 0.709893970763536
      },
      "2": {
       "n": 300,
       "accuracy": 0.64,
       "auroc": 0.6211902006172839
      },
      "3": {
       "n": 300,
       "accuracy": 0.5866666666666667,
       "auroc": 0.5174807551319648
      },
      "4": {
       "n": 300,
       "accuracy": 0.5033333333333333,
       "auroc": 0.532912573892173
      },
      "5": {
       "n": 300,
       "accuracy": 0.43,
       "auroc": 0.5149372138356226
      }
     }
    },
    "options": [
     "true",
     "false"
    ]
   },
   "proofwriter-owa": {
    "accuracy": 0.6333333333333333,
    "auroc": 0.6617271398192451,
    "bands": [
     {
      "accuracy": 0.40540540540540543,
      "from": 0,
      "mean_stated": 0.3204549342717272,
      "n": 74,
      "to": 0.5
     },
     {
      "accuracy": 0.5212765957446809,
      "from": 0.5,
      "mean_stated": 0.675894688108905,
      "n": 94,
      "to": 0.8
     },
     {
      "accuracy": 0.43333333333333335,
      "from": 0.8,
      "mean_stated": 0.8610550870186143,
      "n": 60,
      "to": 0.9
     },
     {
      "accuracy": 0.4868421052631579,
      "from": 0.9,
      "mean_stated": 0.9283623482280031,
      "n": 76,
      "to": 0.95
     },
     {
      "accuracy": 0.5083798882681564,
      "from": 0.95,
      "mean_stated": 0.9746242066577316,
      "n": 179,
      "to": 0.99
     },
     {
      "accuracy": 0.6874039938556068,
      "from": 0.99,
      "mean_stated": 0.9993341274589197,
      "n": 1302,
      "to": 1
     }
    ],
    "by_depth": {
     "0": {
      "accuracy": 0.9333333333333333,
      "auroc": 0.8611607142857143,
      "bands": [
       {
        "accuracy": 0.5,
        "from": 0,
        "mean_stated": 0.21452480720551884,
        "n": 4,
        "to": 0.5
       },
       {
        "accuracy": 0.25,
        "from": 0.5,
        "mean_stated": 0.6385397980211203,
        "n": 4,
        "to": 0.8
       },
       {
        "accuracy": 0,
        "from": 0.8,
        "mean_stated": 0.8685499554025249,
        "n": 1,
        "to": 0.9
       },
       {
        "accuracy": 0.6666666666666666,
        "from": 0.9,
        "mean_stated": 0.9240947957343653,
        "n": 3,
        "to": 0.95
       },
       {
        "accuracy": 0.8,
        "from": 0.95,
        "mean_stated": 0.9709908145256211,
        "n": 5,
        "to": 0.99
       },
       {
        "accuracy": 0.9568345323741008,
        "from": 0.99,
        "mean_stated": 0.9999279545956745,
        "n": 278,
        "to": 1
       }
      ],
      "ece": 0.05723939918697878,
      "mean_stated": 0.9829601242908683,
      "n": 300,
      "wrong": 20,
      "wrong_at_95": 0.65,
      "wrong_at_99": 0.6
     },
     "1": {
      "accuracy": 0.7649006622516556,
      "auroc": 0.6820315834400341,
      "bands": [
       {
        "accuracy": 0.5,
        "from": 0,
        "mean_stated": 0.33793432615300395,
        "n": 6,
        "to": 0.5
       },
       {
        "accuracy": 0.5833333333333334,
        "from": 0.5,
        "mean_stated": 0.6953082550497577,
        "n": 12,
        "to": 0.8
       },
       {
        "accuracy": 0.3333333333333333,
        "from": 0.8,
        "mean_stated": 0.8813521254242772,
        "n": 3,
        "to": 0.9
       },
       {
        "accuracy": 0.5384615384615384,
        "from": 0.9,
        "mean_stated": 0.9265788543341587,
        "n": 13,
        "to": 0.95
       },
       {
        "accuracy": 0.5769230769230769,
        "from": 0.95,
        "mean_stated": 0.9744141952916651,
        "n": 26,
        "to": 0.99
       },
       {
        "accuracy": 0.819327731092437,
        "from": 0.99,
        "mean_stated": 0.999560451219551,
        "n": 238,
        "to": 1
       }
      ],
      "ece": 0.20607891077527282,
      "mean_stated": 0.9678511861814126,
      "n": 302,
      "wrong": 71,
      "wrong_at_95": 0.7746478873239436,
      "wrong_at_99": 0.6197183098591549
     },
     "2": {
      "accuracy": 0.6237623762376238,
      "auroc": 0.6283532906339924,
      "bands": [
       {
        "accuracy": 0.5294117647058824,
        "from": 0,
        "mean_stated": 0.35983591497267026,
        "n": 17,
        "to": 0.5
       },
       {
        "accuracy": 0.55,
        "from": 0.5,
        "mean_stated": 0.6747446415278046,
        "n": 20,
        "to": 0.8
       },
       {
        "accuracy": 0.4,
        "from": 0.8,
        "mean_stated": 0.8548303828942938,
        "n": 10,
        "to": 0.9
       },
       {
        "accuracy": 0.5789473684210527,
        "from": 0.9,
        "mean_stated": 0.9283797891477439,
        "n": 19,
        "to": 0.95
       },
       {
        "accuracy": 0.46511627906976744,
        "from": 0.95,
        "mean_stated": 0.9707716799471972,
        "n": 43,
        "to": 0.99
       },
       {
        "accuracy": 0.6875,
        "from": 0.99,
        "mean_stated": 0.9992975818354711,
        "n": 192,
        "to": 1
       }
      ],
      "ece": 0.32400501666307163,
      "mean_stated": 0.9287391049088798,
      "n": 303,
      "wrong": 114,
      "wrong_at_95": 0.7280701754385965,
      "wrong_at_99": 0.5263157894736842
     },
     "3": {
      "accuracy": 0.5808580858085809,
      "auroc": 0.5570866141732284,
      "bands": [
       {
        "accuracy": 0.2222222222222222,
        "from": 0,
        "mean_stated": 0.27759873047870953,
        "n": 18,
        "to": 0.5
       },
       {
        "accuracy": 0.5833333333333334,
        "from": 0.5,
        "mean_stated": 0.6670185819777775,
        "n": 24,
        "to": 0.8
       },
       {
        "accuracy": 0.6363636363636364,
        "from": 0.8,
        "mean_stated": 0.8622921548801176,
        "n": 11,
        "to": 0.9
       },
       {
        "accuracy": 0.5454545454545454,
        "from": 0.9,
        "mean_stated": 0.9312732224554594,
        "n": 11,
        "to": 0.95
       },
       {
        "accuracy": 0.5405405405405406,
        "from": 0.95,
        "mean_stated": 0.9777180988722612,
        "n": 37,
        "to": 0.99
       },
       {
        "accuracy": 0.6188118811881188,
        "from": 0.99,
        "mean_stated": 0.998837143433206,
        "n": 202,
        "to": 1
       }
      ],
      "ece": 0.3388617653417691,
      "mean_stated": 0.91971985115035,
      "n": 303,
      "wrong": 127,
      "wrong_at_95": 0.7401574803149606,
      "wrong_at_99": 0.6062992125984252
     },
     "4": {
      "accuracy": 0.4389438943894389,
      "auroc": 0.520499778858912,
      "bands": [
       {
        "accuracy": 0.4,
        "from": 0,
        "mean_stated": 0.29442262004326714,
        "n": 15,
        "to": 0.5
       },
       {
        "accuracy": 0.46153846153846156,
        "from": 0.5,
        "mean_stated": 0.668569481722285,
        "n": 13,
        "to": 0.8
       },
       {
        "accuracy": 0.38095238095238093,
        "from": 0.8,
        "mean_stated": 0.853961958494807,
        "n": 21,
        "to": 0.9
       },
       {
        "accuracy": 0.47058823529411764,
        "from": 0.9,
        "mean_stated": 0.9302853434655406,
        "n": 17,
        "to": 0.95
       },
       {
        "accuracy": 0.4,
        "from": 0.95,
        "mean_stated": 0.9752803678267641,
        "n": 40,
        "to": 0.99
       },
       {
        "accuracy": 0.4489795918367347,
        "from": 0.99,
        "mean_stated": 0.9990088460475032,
        "n": 196,
        "to": 1
       }
      ],
      "ece": 0.50442265500801,
      "mean_stated": 0.9329133560509226,
      "n": 303,
      "wrong": 170,
      "wrong_at_95": 0.7764705882352941,
      "wrong_at_99": 0.6352941176470588
     },
     "5": {
      "accuracy": 0.4532871972318339,
      "auroc": 0.5213064064160788,
      "bands": [
       {
        "accuracy": 0.42857142857142855,
        "from": 0,
        "mean_stated": 0.3784023533261816,
        "n": 14,
        "to": 0.5
       },
       {
        "accuracy": 0.47619047619047616,
        "from": 0.5,
        "mean_stated": 0.687690493768716,
        "n": 21,
        "to": 0.8
       },
       {
        "accuracy": 0.42857142857142855,
        "from": 0.8,
        "mean_stated": 0.8702843020304519,
        "n": 14,
        "to": 0.9
       },
       {
        "accuracy": 0.23076923076923078,
        "from": 0.9,
        "mean_stated": 0.9261274378499773,
        "n": 13,
        "to": 0.95
       },
       {
        "accuracy": 0.5714285714285714,
        "from": 0.95,
        "mean_stated": 0.9763586868731732,
        "n": 28,
        "to": 0.99
       },
       {
        "accuracy": 0.45408163265306123,
        "from": 0.99,
        "mean_stated": 0.9990903197573173,
        "n": 196,
        "to": 1
       }
      ],
      "ece": 0.4793330458406075,
      "mean_stated": 0.9346800262463053,
      "n": 289,
      "wrong": 158,
      "wrong_at_95": 0.7658227848101266,
      "wrong_at_99": 0.689873417721519
     }
    },
    "ece": 0.3164611367643454,
    "mean_disclosed_mass": 0.9984866054239474,
    "mean_stated": 0.9444763520381142,
    "n": 1800,
    "options_disclosed": {
     "1": 1388,
     "2": 376,
     "3": 36
    },
    "other_engines_same_items": {
     "jev": {
      "accuracy": 0.8383333333333334,
      "auroc": 0.8485888790965547,
      "bands": [
       {
        "accuracy": 0.5161290322580645,
        "from": 0,
        "mean_stated": 0.4525806451612903,
        "n": 31,
        "to": 0.5
       },
       {
        "accuracy": 0.5741176470588235,
        "from": 0.5,
        "mean_stated": 0.651129411764706,
        "n": 425,
        "to": 0.8
       },
       {
        "accuracy": 0.7817258883248731,
        "from": 0.8,
        "mean_stated": 0.847258883248731,
        "n": 197,
        "to": 0.9
       },
       {
        "accuracy": 0.8351063829787234,
        "from": 0.9,
        "mean_stated": 0.9207446808510638,
        "n": 188,
        "to": 0.95
       },
       {
        "accuracy": 0.9325396825396826,
        "from": 0.95,
        "mean_stated": 0.9659126984126984,
        "n": 252,
        "to": 0.99
       },
       {
        "accuracy": 0.9943422913719944,
        "from": 0.99,
        "mean_stated": 0.9977934936350779,
        "n": 707,
        "to": 1
       }
      ],
      "ece": 0.04142222222222225,
      "n": 1800,
      "wrong_at_95": 0.07216494845360824,
      "wrong_at_99": 0.013745704467353952
     },
     "kev-0.8b": {
      "accuracy": 0.5355555555555556,
      "auroc": 0.6594538555460701,
      "bands": [
       {
        "accuracy": 0.36507936507936506,
        "from": 0,
        "mean_stated": 0.42512698412698413,
        "n": 441,
        "to": 0.5
       },
       {
        "accuracy": 0.5249169435215947,
        "from": 0.5,
        "mean_stated": 0.6551062015503876,
        "n": 903,
        "to": 0.8
       },
       {
        "accuracy": 0.7260981912144703,
        "from": 0.8,
        "mean_stated": 0.8466457364341086,
        "n": 387,
        "to": 0.9
       },
       {
        "accuracy": 0.6956521739130435,
        "from": 0.9,
        "mean_stated": 0.9151231884057971,
        "n": 69,
        "to": 0.95
       },
       {
        "accuracy": null,
        "from": 0.95,
        "mean_stated": null,
        "n": 0,
        "to": 0.99
       },
       {
        "accuracy": null,
        "from": 0.99,
        "mean_stated": null,
        "n": 0,
        "to": 1
       }
      ],
      "ece": 0.11435405555555556,
      "n": 1800,
      "wrong_at_95": 0,
      "wrong_at_99": 0
     },
     "kev-4b": {
      "accuracy": 0.5355555555555556,
      "auroc": 0.710525074946892,
      "bands": [
       {
        "accuracy": 0.334841628959276,
        "from": 0,
        "mean_stated": 0.46293257918552033,
        "n": 221,
        "to": 0.5
       },
       {
        "accuracy": 0.4448780487804878,
        "from": 0.5,
        "mean_stated": 0.6443106341463415,
        "n": 1025,
        "to": 0.8
       },
       {
        "accuracy": 0.6428571428571429,
        "from": 0.8,
        "mean_stated": 0.8462789285714286,
        "n": 280,
        "to": 0.9
       },
       {
        "accuracy": 0.8954248366013072,
        "from": 0.9,
        "mean_stated": 0.9250607843137255,
        "n": 153,
        "to": 0.95
       },
       {
        "accuracy": 0.9642857142857143,
        "from": 0.95,
        "mean_stated": 0.9676160714285714,
        "n": 112,
        "to": 0.99
       },
       {
        "accuracy": 1,
        "from": 0.99,
        "mean_stated": 0.9930111111111112,
        "n": 9,
        "to": 1
       }
      ],
      "ece": 0.1636971111111111,
      "n": 1800,
      "wrong_at_95": 0.004784688995215311,
      "wrong_at_99": 0
     },
     "laya": {
      "accuracy": 0.42444444444444446,
      "auroc": 0.6115003840789182,
      "bands": [
       {
        "accuracy": 0.3456221198156682,
        "from": 0,
        "mean_stated": 0.4362119815668203,
        "n": 217,
        "to": 0.5
       },
       {
        "accuracy": 0.35638297872340424,
        "from": 0.5,
        "mean_stated": 0.6713821808510638,
        "n": 752,
        "to": 0.8
       },
       {
        "accuracy": 0.4258373205741627,
        "from": 0.8,
        "mean_stated": 0.850402870813397,
        "n": 418,
        "to": 0.9
       },
       {
        "accuracy": 0.5411255411255411,
        "from": 0.9,
        "mean_stated": 0.9256359307359308,
        "n": 231,
        "to": 0.95
       },
       {
        "accuracy": 0.6337209302325582,
        "from": 0.95,
        "mean_stated": 0.968725,
        "n": 172,
        "to": 0.99
       },
       {
        "accuracy": 0.9,
        "from": 0.99,
        "mean_stated": 0.9939199999999999,
        "n": 10,
        "to": 1
       }
      ],
      "ece": 0.32299311111111106,
      "n": 1800,
      "wrong_at_95": 0.06177606177606178,
      "wrong_at_99": 0.0009652509652509653
     }
    },
    "responses": 1800,
    "returned_not_most_probable": 72,
    "unparsed": 0,
    "wrong": 660,
    "wrong_at_95": 0.7545454545454545,
    "wrong_at_99": 0.6212121212121212,
    "other_engines_by_depth": {
     "jev": {
      "0": {
       "n": 300,
       "accuracy": 0.9766666666666667,
       "auroc": 0.9527059970745978
      },
      "1": {
       "n": 302,
       "accuracy": 0.8774834437086093,
       "auroc": 0.8526262111167772
      },
      "2": {
       "n": 303,
       "accuracy": 0.8481848184818482,
       "auroc": 0.8449923870749451
      },
      "3": {
       "n": 303,
       "accuracy": 0.7986798679867987,
       "auroc": 0.8521541796504538
      },
      "4": {
       "n": 303,
       "accuracy": 0.7194719471947195,
       "auroc": 0.726902320561252
      },
      "5": {
       "n": 289,
       "accuracy": 0.8096885813148789,
       "auroc": 0.8281662781662782
      }
     },
     "kev-4b": {
      "0": {
       "n": 300,
       "accuracy": 0.8766666666666667,
       "auroc": 0.8495015928476004
      },
      "1": {
       "n": 302,
       "accuracy": 0.6291390728476821,
       "auroc": 0.7262218045112782
      },
      "2": {
       "n": 303,
       "accuracy": 0.5214521452145214,
       "auroc": 0.6427542557835006
      },
      "3": {
       "n": 303,
       "accuracy": 0.44554455445544555,
       "auroc": 0.6419753086419753
      },
      "4": {
       "n": 303,
       "accuracy": 0.37293729372937295,
       "auroc": 0.558011178388449
      },
      "5": {
       "n": 289,
       "accuracy": 0.3633217993079585,
       "auroc": 0.6083074534161491
      }
     },
     "kev-0.8b": {
      "0": {
       "n": 300,
       "accuracy": 0.7033333333333334,
       "auroc": 0.7151339261941531
      },
      "1": {
       "n": 302,
       "accuracy": 0.5397350993377483,
       "auroc": 0.5843227258683851
      },
      "2": {
       "n": 303,
       "accuracy": 0.46864686468646866,
       "auroc": 0.6474061761875601
      },
      "3": {
       "n": 303,
       "accuracy": 0.4884488448844885,
       "auroc": 0.664843068875327
      },
      "4": {
       "n": 303,
       "accuracy": 0.504950495049505,
       "auroc": 0.6379302832244008
      },
      "5": {
       "n": 289,
       "accuracy": 0.5086505190311419,
       "auroc": 0.684871131551212
      }
     },
     "laya": {
      "0": {
       "n": 300,
       "accuracy": 0.6266666666666667,
       "auroc": 0.7851443768996961
      },
      "1": {
       "n": 302,
       "accuracy": 0.4205298013245033,
       "auroc": 0.6015073115860518
      },
      "2": {
       "n": 303,
       "accuracy": 0.41254125412541254,
       "auroc": 0.5371685393258427
      },
      "3": {
       "n": 303,
       "accuracy": 0.38283828382838286,
       "auroc": 0.5966946339664393
      },
      "4": {
       "n": 303,
       "accuracy": 0.35973597359735976,
       "auroc": 0.47751347772628394
      },
      "5": {
       "n": 289,
       "accuracy": 0.34256055363321797,
       "auroc": 0.5352206273258905
      }
     },
     "kev-9b": {
      "0": {
       "n": 300,
       "accuracy": 0.8933333333333333,
       "auroc": 0.8563432835820896
      },
      "1": {
       "n": 302,
       "accuracy": 0.7384105960264901,
       "auroc": 0.7510926945563944
      },
      "2": {
       "n": 303,
       "accuracy": 0.5841584158415841,
       "auroc": 0.6998699668191194
      },
      "3": {
       "n": 303,
       "accuracy": 0.49504950495049505,
       "auroc": 0.6400217864923747
      },
      "4": {
       "n": 303,
       "accuracy": 0.38613861386138615,
       "auroc": 0.5485938792390406
      },
      "5": {
       "n": 289,
       "accuracy": 0.4117647058823529,
       "auroc": 0.5655709342560553
      }
     }
    },
    "options": [
     "true",
     "false",
     "unknown"
    ]
   }
  },
  "overall": {
   "accuracy": 0.6408333333333334,
   "auroc": 0.6698948457416833,
   "bands": [
    {
     "accuracy": 0.4460431654676259,
     "from": 0,
     "mean_stated": 0.30673108162699836,
     "n": 139,
     "to": 0.5
    },
    {
     "accuracy": 0.5238095238095238,
     "from": 0.5,
     "mean_stated": 0.678390958408231,
     "n": 168,
     "to": 0.8
    },
    {
     "accuracy": 0.4818181818181818,
     "from": 0.8,
     "mean_stated": 0.8611539566000325,
     "n": 110,
     "to": 0.9
    },
    {
     "accuracy": 0.5136986301369864,
     "from": 0.9,
     "mean_stated": 0.927975324604965,
     "n": 146,
     "to": 0.95
    },
    {
     "accuracy": 0.5121951219512195,
     "from": 0.95,
     "mean_stated": 0.9746919509598972,
     "n": 328,
     "to": 0.99
    },
    {
     "accuracy": 0.6841317365269461,
     "from": 0.99,
     "mean_stated": 0.9993454921481807,
     "n": 2672,
     "to": 1
    }
   ],
   "ece": 0.3170820856165399,
   "mean_stated": 0.9482685805710882,
   "n": 3600,
   "wrong": 1293,
   "wrong_at_95": 0.7795823665893271,
   "wrong_at_99": 0.6558391337973705
  },
  "examples": [
   {
    "correct": false,
    "gold": "false",
    "id": "AttNonegNatLang-OWA-479-Q8",
    "request": {
     "logprobs": true,
     "max_completion_tokens": 1024,
     "messages": [
      {
       "content": "Alan is young, round, and kind, but that doesn't mean he isn't also rough and cold at times, as well. That guy Fred sure is nice. That guy Harry sure is nice. Young round people who are green are usually blue. When green, young and round fits a person, you'll see that rough will also fit. It's a certainty that any green, big and kind individual is going to be nice. Kind people with rough skin are usually red because it's wind burn. If someone shows that they are red, then they are also showing that they are green. Is a kind person if feeling cold and looking red then they are usually young.\n\nStatement: Alan is not blue.\n\nUsing only the facts and rules in the text, is the statement true, false, or unknown? A statement is true if it can be derived from the facts and rules, and false if its negation can be derived. If neither the statement nor its negation can be derived, it is unknown. A rule applies only when all of its conditions are established; 'not' in a condition requires the negation to be stated or derived.\n\nOptions:\n- true: The statement follows from the facts and rules.\n- false: The negation of the statement follows from the facts and rules.\n- unknown: Neither the statement nor its negation follows from the facts and rules.\n\nReply with a JSON object {\"answer\": <option>} where <option> is exactly one of: true, false, unknown.",
       "role": "user"
      }
     ],
     "model": "gpt-6-luna",
     "reasoning_effort": "none",
     "response_format": {
      "json_schema": {
       "name": "decision",
       "schema": {
        "additionalProperties": false,
        "properties": {
         "answer": {
          "enum": [
           "true",
           "false",
           "unknown"
          ],
          "type": "string"
         }
        },
        "required": [
         "answer"
        ],
        "type": "object"
       },
       "strict": true
      },
      "type": "json_schema"
     },
     "top_logprobs": 5
    },
    "response": {
     "choices": [
      {
       "finish_reason": "stop",
       "index": 0,
       "logprobs": {
        "content": [
         {
          "bytes": [
           123,
           34
          ],
          "logprob": 0,
          "token": "{\"",
          "top_logprobs": [
           {
            "bytes": [
             123,
             34
            ],
            "logprob": 0,
            "token": "{\""
           }
          ]
         },
         {
          "bytes": [
           97,
           110,
           115,
           119,
           101,
           114
          ],
          "logprob": 0,
          "token": "answer",
          "top_logprobs": [
           {
            "bytes": [
             97,
             110,
             115,
             119,
             101,
             114
            ],
            "logprob": 0,
            "token": "answer"
           }
          ]
         },
         {
          "bytes": [
           34,
           58,
           34
          ],
          "logprob": 0,
          "token": "\":\"",
          "top_logprobs": [
           {
            "bytes": [
             34,
             58,
             34
            ],
            "logprob": 0,
            "token": "\":\""
           }
          ]
         },
         {
          "bytes": [
           117,
           110,
           107,
           110,
           111,
           119,
           110
          ],
          "logprob": -0.001819610595703125,
          "token": "unknown",
          "top_logprobs": [
           {
            "bytes": [
             117,
             110,
             107,
             110,
             111,
             119,
             110
            ],
            "logprob": -0.001819610595703125,
            "token": "unknown"
           }
          ]
         },
         {
          "bytes": [
           34,
           125
          ],
          "logprob": 0,
          "token": "\"}",
          "top_logprobs": [
           {
            "bytes": [
             34,
             125
            ],
            "logprob": 0,
            "token": "\"}"
           }
          ]
         }
        ],
        "refusal": null
       },
       "message": {
        "annotations": [],
        "audio": null,
        "content": "{\"answer\":\"unknown\"}",
        "function_call": null,
        "refusal": null,
        "role": "assistant",
        "tool_calls": null
       }
      }
     ],
     "created": 1790866630,
     "id": "chatcmpl-EUCVa4ydOIAtLD1d5VSwqwdcWrIyc",
     "model": "gpt-6-luna",
     "moderation": null,
     "object": "chat.completion",
     "service_tier": "default",
     "system_fingerprint": null,
     "usage": {
      "completion_tokens": 11,
      "completion_tokens_details": {
       "accepted_prediction_tokens": 0,
       "audio_tokens": 0,
       "reasoning_tokens": 0,
       "rejected_prediction_tokens": 0
      },
      "prompt_tokens": 340,
      "prompt_tokens_details": {
       "audio_tokens": 0,
       "cache_write_tokens": 0,
       "cached_tokens": 0
      },
      "total_tokens": 351
     }
    },
    "task": "proofwriter-owa",
    "answer": "unknown",
    "stated": 0.9981820438919969,
    "answer_tokens": [
     "unknown"
    ],
    "alternatives": [
     {
      "token": "unknown",
      "p": 0.9981820438919969
     }
    ],
    "depth": 3
   },
   {
    "correct": true,
    "gold": "unknown",
    "id": "AttNoneg-OWA-D5-924-Q18",
    "request": {
     "logprobs": true,
     "max_completion_tokens": 1024,
     "messages": [
      {
       "content": "Anne is red. Anne is white. Bob is furry. Bob is red. Fiona is big. Fiona is furry. Fiona is green. Gary is cold. Gary is furry. Gary is green. Green, furry people are nice. All green, nice people are big. Big, green people are furry. All furry people are nice. All big, green people are furry. Big, red people are cold. If someone is nice and red then they are green. Big, cold people are white. If someone is green and cold then they are big.\n\nStatement: Anne is furry.\n\nUsing only the facts and rules in the text, is the statement true, false, or unknown? A statement is true if it can be derived from the facts and rules, and false if its negation can be derived. If neither the statement nor its negation can be derived, it is unknown. A rule applies only when all of its conditions are established; 'not' in a condition requires the negation to be stated or derived.\n\nOptions:\n- true: The statement follows from the facts and rules.\n- false: The negation of the statement follows from the facts and rules.\n- unknown: Neither the statement nor its negation follows from the facts and rules.\n\nReply with a JSON object {\"answer\": <option>} where <option> is exactly one of: true, false, unknown.",
       "role": "user"
      }
     ],
     "model": "gpt-6-luna",
     "reasoning_effort": "none",
     "response_format": {
      "json_schema": {
       "name": "decision",
       "schema": {
        "additionalProperties": false,
        "properties": {
         "answer": {
          "enum": [
           "true",
           "false",
           "unknown"
          ],
          "type": "string"
         }
        },
        "required": [
         "answer"
        ],
        "type": "object"
       },
       "strict": true
      },
      "type": "json_schema"
     },
     "top_logprobs": 5
    },
    "response": {
     "choices": [
      {
       "finish_reason": "stop",
       "index": 0,
       "logprobs": {
        "content": [
         {
          "bytes": [
           123,
           34
          ],
          "logprob": 0,
          "token": "{\"",
          "top_logprobs": [
           {
            "bytes": [
             123,
             34
            ],
            "logprob": 0,
            "token": "{\""
           }
          ]
         },
         {
          "bytes": [
           97,
           110,
           115,
           119,
           101,
           114
          ],
          "logprob": 0,
          "token": "answer",
          "top_logprobs": [
           {
            "bytes": [
             97,
             110,
             115,
             119,
             101,
             114
            ],
            "logprob": 0,
            "token": "answer"
           }
          ]
         },
         {
          "bytes": [
           34,
           58,
           34
          ],
          "logprob": 0,
          "token": "\":\"",
          "top_logprobs": [
           {
            "bytes": [
             34,
             58,
             34
            ],
            "logprob": 0,
            "token": "\":\""
           }
          ]
         },
         {
          "bytes": [
           117,
           110,
           107,
           110,
           111,
           119,
           110
          ],
          "logprob": -0.000003814697265625,
          "token": "unknown",
          "top_logprobs": [
           {
            "bytes": [
             117,
             110,
             107,
             110,
             111,
             119,
             110
            ],
            "logprob": -0.000003814697265625,
            "token": "unknown"
           }
          ]
         },
         {
          "bytes": [
           34,
           125
          ],
          "logprob": -0.000003814697265625,
          "token": "\"}",
          "top_logprobs": [
           {
            "bytes": [
             34,
             125
            ],
            "logprob": -0.000003814697265625,
            "token": "\"}"
           }
          ]
         }
        ],
        "refusal": null
       },
       "message": {
        "annotations": [],
        "audio": null,
        "content": "{\"answer\":\"unknown\"}",
        "function_call": null,
        "refusal": null,
        "role": "assistant",
        "tool_calls": null
       }
      }
     ],
     "created": 1790866630,
     "id": "chatcmpl-EUCVa0u5eg0DyNLwbbvk2m0sTjVW2",
     "model": "gpt-6-luna",
     "moderation": null,
     "object": "chat.completion",
     "service_tier": "default",
     "system_fingerprint": null,
     "usage": {
      "completion_tokens": 11,
      "completion_tokens_details": {
       "accepted_prediction_tokens": 0,
       "audio_tokens": 0,
       "reasoning_tokens": 0,
       "rejected_prediction_tokens": 0
      },
      "prompt_tokens": 320,
      "prompt_tokens_details": {
       "audio_tokens": 0,
       "cache_write_tokens": 0,
       "cached_tokens": 0
      },
      "total_tokens": 331
     }
    },
    "task": "proofwriter-owa",
    "answer": "unknown",
    "stated": 0.9999961853100103,
    "answer_tokens": [
     "unknown"
    ],
    "alternatives": [
     {
      "token": "unknown",
      "p": 0.9999961853100103
     }
    ],
    "depth": 4
   },
   {
    "correct": false,
    "gold": "false",
    "id": "AttNeg-CWA-D5-136-Q12",
    "request": {
     "logprobs": true,
     "max_completion_tokens": 1024,
     "messages": [
      {
       "content": "Anne is nice. Anne is round. Dave is big. Dave is quiet. Dave is round. Erin is green. Erin is quiet. Harry is big. Harry is quiet. Harry is round. All furry, round people are nice. Furry people are nice. All nice people are rough. All green, nice people are furry. Round, big people are furry. If Erin is furry then Erin is big. Round, rough people are big. Quiet people are green. Furry people are quiet.\n\nStatement: Anne is not green.\n\nUsing only the facts and rules in the text, is the statement true or false? Assume that anything which cannot be derived from the facts and rules is false: a positive statement is true only if it can be derived, and a statement with 'not' is true when the unnegated statement cannot be derived. A rule applies when all of its conditions hold; 'not' in a condition holds when the unnegated condition cannot be derived.\n\nOptions:\n- true: The statement holds under the closed-world assumption.\n- false: The statement does not hold under the closed-world assumption.\n\nReply with a JSON object {\"answer\": <option>} where <option> is exactly one of: true, false.",
       "role": "user"
      }
     ],
     "model": "gpt-6-luna",
     "reasoning_effort": "none",
     "response_format": {
      "json_schema": {
       "name": "decision",
       "schema": {
        "additionalProperties": false,
        "properties": {
         "answer": {
          "enum": [
           "true",
           "false"
          ],
          "type": "string"
         }
        },
        "required": [
         "answer"
        ],
        "type": "object"
       },
       "strict": true
      },
      "type": "json_schema"
     },
     "top_logprobs": 5
    },
    "response": {
     "choices": [
      {
       "finish_reason": "stop",
       "index": 0,
       "logprobs": {
        "content": [
         {
          "bytes": [
           123,
           34
          ],
          "logprob": 0,
          "token": "{\"",
          "top_logprobs": [
           {
            "bytes": [
             123,
             34
            ],
            "logprob": 0,
            "token": "{\""
           }
          ]
         },
         {
          "bytes": [
           97,
           110,
           115,
           119,
           101,
           114
          ],
          "logprob": 0,
          "token": "answer",
          "top_logprobs": [
           {
            "bytes": [
             97,
             110,
             115,
             119,
             101,
             114
            ],
            "logprob": 0,
            "token": "answer"
           }
          ]
         },
         {
          "bytes": [
           34,
           58,
           34
          ],
          "logprob": 0,
          "token": "\":\"",
          "top_logprobs": [
           {
            "bytes": [
             34,
             58,
             34
            ],
            "logprob": 0,
            "token": "\":\""
           }
          ]
         },
         {
          "bytes": [
           116,
           114,
           117,
           101
          ],
          "logprob": -0.020458221435546875,
          "token": "true",
          "top_logprobs": [
           {
            "bytes": [
             116,
             114,
             117,
             101
            ],
            "logprob": -0.020458221435546875,
            "token": "true"
           },
           {
            "bytes": [
             102,
             97,
             108,
             115,
             101
            ],
            "logprob": -3.8995132446289062,
            "token": "false"
           }
          ]
         },
         {
          "bytes": [
           34,
           125
          ],
          "logprob": 0,
          "token": "\"}",
          "top_logprobs": [
           {
            "bytes": [
             34,
             125
            ],
            "logprob": 0,
            "token": "\"}"
           }
          ]
         }
        ],
        "refusal": null
       },
       "message": {
        "annotations": [],
        "audio": null,
        "content": "{\"answer\":\"true\"}",
        "function_call": null,
        "refusal": null,
        "role": "assistant",
        "tool_calls": null
       }
      }
     ],
     "created": 1790866853,
     "id": "chatcmpl-EUCZBRXdq8PlqrFEmrh4Fb8W3ZfP9",
     "model": "gpt-6-luna",
     "moderation": null,
     "object": "chat.completion",
     "service_tier": "default",
     "system_fingerprint": null,
     "usage": {
      "completion_tokens": 11,
      "completion_tokens_details": {
       "accepted_prediction_tokens": 0,
       "audio_tokens": 0,
       "reasoning_tokens": 0,
       "rejected_prediction_tokens": 0
      },
      "prompt_tokens": 289,
      "prompt_tokens_details": {
       "audio_tokens": 0,
       "cache_write_tokens": 0,
       "cached_tokens": 0
      },
      "total_tokens": 300
     }
    },
    "task": "proofwriter-cwa",
    "answer": "true",
    "stated": 0.9797496281524662,
    "answer_tokens": [
     "true"
    ],
    "alternatives": [
     {
      "token": "true",
      "p": 0.9797496281524662
     },
     {
      "token": "false",
      "p": 0.020251766703277007
     }
    ],
    "depth": 5
   },
   {
    "correct": true,
    "gold": "true",
    "id": "RelNeg-CWA-D5-871-Q9",
    "request": {
     "logprobs": true,
     "max_completion_tokens": 1024,
     "messages": [
      {
       "content": "The bald eagle visits the squirrel. The cow is rough. The cow is round. The cow likes the bald eagle. The cow likes the tiger. The cow needs the bald eagle. The cow needs the squirrel. The cow needs the tiger. The squirrel likes the bald eagle. The squirrel likes the cow. The squirrel visits the cow. The tiger is red. The tiger needs the squirrel. The tiger visits the squirrel. If something visits the bald eagle then the bald eagle is kind. If something is red and cold then it likes the tiger. If the cow visits the bald eagle and the cow is red then the bald eagle visits the squirrel. If something visits the cow and it is cold then it visits the bald eagle. If something needs the tiger then it is cold. If something likes the bald eagle and the bald eagle is kind then the bald eagle is cold. If something likes the squirrel and it visits the bald eagle then the squirrel needs the tiger. If something needs the cow then it is red. If something is cold then it visits the cow.\n\nStatement: The bald eagle is kind.\n\nUsing only the facts and rules in the text, is the statement true or false? Assume that anything which cannot be derived from the facts and rules is false: a positive statement is true only if it can be derived, and a statement with 'not' is true when the unnegated statement cannot be derived. A rule applies when all of its conditions hold; 'not' in a condition holds when the unnegated condition cannot be derived.\n\nOptions:\n- true: The statement holds under the closed-world assumption.\n- false: The statement does not hold under the closed-world assumption.\n\nReply with a JSON object {\"answer\": <option>} where <option> is exactly one of: true, false.",
       "role": "user"
      }
     ],
     "model": "gpt-6-luna",
     "reasoning_effort": "none",
     "response_format": {
      "json_schema": {
       "name": "decision",
       "schema": {
        "additionalProperties": false,
        "properties": {
         "answer": {
          "enum": [
           "true",
           "false"
          ],
          "type": "string"
         }
        },
        "required": [
         "answer"
        ],
        "type": "object"
       },
       "strict": true
      },
      "type": "json_schema"
     },
     "top_logprobs": 5
    },
    "response": {
     "choices": [
      {
       "finish_reason": "stop",
       "index": 0,
       "logprobs": {
        "content": [
         {
          "bytes": [
           123,
           34
          ],
          "logprob": 0,
          "token": "{\"",
          "top_logprobs": [
           {
            "bytes": [
             123,
             34
            ],
            "logprob": 0,
            "token": "{\""
           }
          ]
         },
         {
          "bytes": [
           97,
           110,
           115,
           119,
           101,
           114
          ],
          "logprob": 0,
          "token": "answer",
          "top_logprobs": [
           {
            "bytes": [
             97,
             110,
             115,
             119,
             101,
             114
            ],
            "logprob": 0,
            "token": "answer"
           }
          ]
         },
         {
          "bytes": [
           34,
           58,
           34
          ],
          "logprob": 0,
          "token": "\":\"",
          "top_logprobs": [
           {
            "bytes": [
             34,
             58,
             34
            ],
            "logprob": 0,
            "token": "\":\""
           }
          ]
         },
         {
          "bytes": [
           116,
           114,
           117,
           101
          ],
          "logprob": 0,
          "token": "true",
          "top_logprobs": [
           {
            "bytes": [
             116,
             114,
             117,
             101
            ],
            "logprob": 0,
            "token": "true"
           }
          ]
         },
         {
          "bytes": [
           34,
           125
          ],
          "logprob": 0.000003814697265625,
          "token": "\"}",
          "top_logprobs": [
           {
            "bytes": [
             34,
             125
            ],
            "logprob": 0.000003814697265625,
            "token": "\"}"
           }
          ]
         }
        ],
        "refusal": null
       },
       "message": {
        "annotations": [],
        "audio": null,
        "content": "{\"answer\":\"true\"}",
        "function_call": null,
        "refusal": null,
        "role": "assistant",
        "tool_calls": null
       }
      }
     ],
     "created": 1790866853,
     "id": "chatcmpl-EUCZBPPeC8fEP4HSEKY2kob2qBT4A",
     "model": "gpt-6-luna",
     "moderation": null,
     "object": "chat.completion",
     "service_tier": "default",
     "system_fingerprint": null,
     "usage": {
      "completion_tokens": 11,
      "completion_tokens_details": {
       "accepted_prediction_tokens": 0,
       "audio_tokens": 0,
       "reasoning_tokens": 0,
       "rejected_prediction_tokens": 0
      },
      "prompt_tokens": 401,
      "prompt_tokens_details": {
       "audio_tokens": 0,
       "cache_write_tokens": 0,
       "cached_tokens": 0
      },
      "total_tokens": 412
     }
    },
    "task": "proofwriter-cwa",
    "answer": "true",
    "stated": 1,
    "answer_tokens": [
     "true"
    ],
    "alternatives": [
     {
      "token": "true",
      "p": 1
     }
    ],
    "depth": 4
   }
  ],
  "runs": [
   {
    "concurrency": 8,
    "failed": 0,
    "finished_at": "2026-10-01T15:05:08.285+00:00",
    "requested": 1800,
    "started_at": "2026-10-01T15:00:52.468+00:00",
    "task": "proofwriter-cwa",
    "top_logprobs": 5
   },
   {
    "concurrency": 8,
    "failed": 0,
    "finished_at": "2026-10-01T15:00:51.157+00:00",
    "requested": 1800,
    "started_at": "2026-10-01T14:57:09.185+00:00",
    "task": "proofwriter-owa",
    "top_logprobs": 5
   }
  ]
 },
 "predictions": [
  {
   "heading": "Predictions",
   "title": "Jev",
   "model": "jev",
   "id": "jev",
   "predictions": [
    {
     "n": 1,
     "text": "Jev's accuracy falls as proof depth rises, on both tasks. Depth 0 is clearly above depth 5, with intervals that do not overlap on OWA.",
     "status": "partly held",
     "checks": [
      "Rule: depth 0 clearly above depth 5 on each task (95% intervals do not overlap), and no deeper level clearly above a shallower one.",
      "OWA: depth 0 97.7% [96.0–99.3], depth 5 81.0% [76.5–85.5], intervals apart.",
      "CWA: depth 0 99.3% [98.3–100.0], depth 5 89.3% [85.7–92.3], intervals apart; but depth 5 (89.3%) is clearly above depth 4 (76.3%)."
     ]
    },
    {
     "n": 2,
     "text": "On OWA, Jev's recall on `unknown` is the lowest of the three classes at depth 3 and above.",
     "status": "held",
     "checks": [
      "Rule: at each of depths 3, 4 and 5 on OWA, Jev's recall on unknown is the lowest of the three classes. (Recall by depth and class is counted from the saved answers.)",
      "Depth 3: true 86.1%, false 80.2%, unknown 73.3%; lowest is unknown.",
      "Depth 4: true 80.2%, false 79.2%, unknown 56.4%; lowest is unknown.",
      "Depth 5: true 93.1%, false 92.1%, unknown 54.0%; lowest is unknown."
     ]
    },
    {
     "n": 3,
     "text": "By depth 5, Jev is within 10 points of its own best-constant floor on at least one task (that is, it has largely stopped using the rules).",
     "status": "failed",
     "checks": [
      "Rule: at depth 5, Jev's accuracy is within 10 points of the best constant guess on at least one task.",
      "OWA: 81.0% against a best constant of 34.9%, +46.0 points above it.",
      "CWA: 89.3% against a best constant of 50.0%, +39.3 points above it."
     ]
    },
    {
     "n": 4,
     "text": "Negation in the theory (`theory_negation = negation`) lowers accuracy relative to no negation.",
     "status": "failed",
     "checks": [
      "Rule: accuracy with negation in the theory is below accuracy without it.",
      "OWA: with negation 87.5% [85.4–89.7], without 80.3% [77.7–83.1]: clearly higher with negation, the opposite of the prediction.",
      "CWA: with negation 93.7% [92.1–95.3], without 85.2% [82.8–87.5]: clearly higher with negation, the opposite of the prediction."
     ]
    },
    {
     "n": 5,
     "text": "Paraphrased rules (`NatLang`) are no easier than templated ones.",
     "status": "held",
     "checks": [
      "Rule: paraphrased rules are not clearly easier (their interval is not wholly above the templated one).",
      "OWA: paraphrased 78.1%, templated 86.1%.",
      "CWA: paraphrased 76.7%, templated 94.0%."
     ]
    },
    {
     "n": 6,
     "text": "No claim about the other axes (length, rule count, proof size, strategy) beyond reporting them; they correlate with depth, so those breakdowns are descriptive.",
     "status": "no claim",
     "checks": [
      "Reported, not predicted: see the breakdown pages."
     ]
    }
   ]
  },
  {
   "heading": "Amendment 1: Kev (written before Kev answered any item)",
   "title": "Kev-0.8B",
   "model": "kev-0.8b",
   "id": "kev-0-8b",
   "predictions": [
    {
     "n": 1,
     "text": "Kev's accuracy falls with proof depth on both tasks.",
     "status": "held",
     "checks": [
      "Rule: depth 0 clearly above depth 5 on each task (95% intervals do not overlap), and no deeper level clearly above a shallower one.",
      "OWA: depth 0 70.3% [65.3–75.7], depth 5 50.9% [45.3–56.7], intervals apart.",
      "CWA: depth 0 68.0% [62.7–73.3], depth 5 51.7% [46.0–57.3], intervals apart."
     ]
    },
    {
     "n": 2,
     "text": "Kev is below Jev overall on both tasks (paired interval excludes zero).",
     "status": "held",
     "checks": [
      "Rule: the paired interval for Jev minus Kev-0.8B lies above zero on both tasks.",
      "OWA: Jev minus Kev-0.8B +30.3 points [+27.7 to +32.7] on 1,800 items.",
      "CWA: Jev minus Kev-0.8B +33.5 points [+30.7 to +36.3] on 1,800 items."
     ]
    },
    {
     "n": 3,
     "text": "On OWA, Kev's `unknown` recall is its lowest class recall.",
     "status": "held",
     "checks": [
      "Rule: on OWA, recall on unknown is the lowest of the three classes.",
      "Kev-0.8B recall on OWA: false 89.6%, true 58.2%, unknown 11.9%; lowest is unknown."
     ]
    }
   ]
  },
  {
   "heading": "Amendment 2: LLM classifiers (written before any LLM answered an item)",
   "title": "GPT-6 Luna",
   "model": "openai-gpt-6-luna-effort-none",
   "id": "gpt-6-luna",
   "predictions": [
    {
     "n": 1,
     "text": "Luna's accuracy falls with proof depth on both tasks.",
     "status": "held",
     "checks": [
      "Rule: depth 0 clearly above depth 5 on each task (95% intervals do not overlap), and no deeper level clearly above a shallower one.",
      "OWA: depth 0 93.3% [90.3–96.0], depth 5 46.0% [40.1–51.6], intervals apart.",
      "CWA: depth 0 98.0% [96.3–99.3], depth 5 44.7% [38.7–50.3], intervals apart."
     ]
    },
    {
     "n": 2,
     "text": "On OWA, Luna's `unknown` recall is its lowest class recall.",
     "status": "failed",
     "checks": [
      "Rule: on OWA, recall on unknown is the lowest of the three classes.",
      "GPT-6 Luna recall on OWA: false 50.9%, true 58.3%, unknown 83.4%; lowest is false."
     ]
    },
    {
     "n": 3,
     "text": "With reasoning off, Luna does not beat Jev at depth 5 on OWA (the paired interval does not exclude zero in Luna's favour).",
     "status": "held",
     "checks": [
      "Rule: at depth 5 on OWA, the paired interval for Jev minus Luna does not lie wholly below zero.",
      "Depth 5, OWA: Jev minus Luna +34.9 points [+27.3 to +42.9] on 289 items."
     ]
    },
    {
     "n": 4,
     "text": "Paraphrased rules lower Luna's accuracy less than they lower Jev's.",
     "status": "mixed",
     "checks": [
      "Rule: on each task, Luna's accuracy drop from templated to paraphrased rules is smaller than Jev's. (No interval was preregistered for this difference, so it is a point comparison.)",
      "OWA: Luna drops +12.9 points, Jev drops +8.0: failed.",
      "CWA: Luna drops +8.4 points, Jev drops +17.3: held."
     ]
    }
   ]
  },
  {
   "heading": "Amendment 3: Kev-4B and Kev-9B (written before either answered an item)",
   "title": "Kev-4B and Kev-9B",
   "model": "kev-4b",
   "id": "kev-4b-and-kev-9b",
   "predictions": [
    {
     "n": 1,
     "text": "Accuracy rises with size: Kev-4B is above Kev-0.8B overall on both tasks (paired interval excludes zero).",
     "status": "failed",
     "checks": [
      "Rule: Kev-4B is above Kev-0.8B overall on both tasks, with a paired interval that excludes zero.",
      "OWA: Kev-4B 53.6%, Kev-0.8B 53.6%.",
      "CWA: Kev-4B 58.8%, Kev-0.8B 55.8%.",
      "Kev-4B is not ahead on OWA, so its interval cannot exclude zero in its favour there."
     ]
    },
    {
     "n": 2,
     "text": "Kev-4B and Kev-9B stay below Jev overall on both tasks.",
     "status": "held",
     "checks": [
      "Rule: the paired interval for Jev minus each larger Kev lies above zero on both tasks.",
      "OWA: Jev minus Kev-4B +30.3 points [+27.4 to +32.9] on 1,800 items.",
      "CWA: Jev minus Kev-4B +30.5 points [+27.9 to +33.1] on 1,800 items.",
      "OWA: Jev minus Kev-9B +25.3 points [+22.7 to +27.8] on 1,800 items.",
      "CWA: Jev minus Kev-9B +25.5 points [+23.0 to +27.8] on 1,800 items."
     ]
    },
    {
     "n": 3,
     "text": "Kev-9B is not reliably above Kev-4B (paired interval includes zero) on at least one task.",
     "status": "pending",
     "checks": [
      "Kev-9B and Kev-4B are both scored, but the harness has not computed their paired interval."
     ]
    }
   ]
  },
  {
   "heading": "Amendment 4: test-retest repeatability (written before any rerun except Jev's)",
   "title": "Repeatability",
   "model": null,
   "id": "repeatability",
   "predictions": [
    {
     "n": 1,
     "text": "Kev and Laya agree with themselves on at least 99.5% of items on both tasks (local, fixed weights).",
     "status": "pending",
     "checks": [
      "Rule: every Kev and Laya rerun agrees with its first run on at least 99.5% of items, on both tasks.",
      "Kev-0.8B on OWA: 100.0% agreement.",
      "Kev-0.8B on CWA: 100.0% agreement.",
      "Kev-4B on OWA: 100.0% agreement.",
      "Kev-4B on CWA: 100.0% agreement.",
      "Laya on OWA: 100.0% agreement.",
      "Laya on CWA: 100.0% agreement.",
      "No scored rerun yet for: Kev-9B on OWA, Kev-9B on CWA."
     ]
    },
    {
     "n": 2,
     "text": "Jev's AC1 is at least 0.90 on both tasks (observed: 0.958 on both, before this amendment).",
     "status": "held",
     "checks": [
      "Rule: Jev's AC1 is at least 0.90 on both tasks. (The amendment notes this was already observed when it was written.)",
      "OWA: AC1 0.958 [0.948–0.969], agreement 97.2%.",
      "CWA: AC1 0.958 [0.944–0.971], agreement 97.9%."
     ]
    },
    {
     "n": 3,
     "text": "Where answers change, they change more at depth 3 and above than at depth 0 to 2, for every engine with at least 20 changes.",
     "status": "held",
     "checks": [
      "Rule: for every model with at least 20 changed answers, the share that changes is higher at depths 3 and above than at depths 0 to 2.",
      "Jev on OWA: 4.5% of answers changed at depths 3 to 5 (40 of 895) against 1.1% at depths 0 to 2 (10 of 905).",
      "Kev-0.8B on OWA: 0 changes, under the 20 the rule needs.",
      "Kev-4B on OWA: 0 changes, under the 20 the rule needs.",
      "Laya on OWA: 0 changes, under the 20 the rule needs.",
      "GPT-6 Luna on OWA: 12.8% of answers changed at depths 3 to 5 (115 of 895) against 7.7% at depths 0 to 2 (70 of 905).",
      "Jev on CWA: 3.3% of answers changed at depths 3 to 5 (30 of 900) against 0.9% at depths 0 to 2 (8 of 900).",
      "Kev-0.8B on CWA: 0 changes, under the 20 the rule needs.",
      "Kev-4B on CWA: 0 changes, under the 20 the rule needs.",
      "Laya on CWA: 0 changes, under the 20 the rule needs.",
      "GPT-6 Luna on CWA: 12.2% of answers changed at depths 3 to 5 (110 of 900) against 6.6% at depths 0 to 2 (59 of 900)."
     ]
    }
   ]
  },
  {
   "heading": "Amendment 6: can GPT-6 Luna's log-probabilities serve as a confidence? (written before the probe ran)",
   "title": "Luna confidence probe",
   "model": "openai-gpt-6-luna-effort-none",
   "id": "luna-confidence-probe",
   "predictions": [
    {
     "n": 1,
     "text": "Luna's ECE exceeds 0.15 on both tasks.",
     "status": "held",
     "checks": [
      "Rule: Luna's expected calibration error is above 0.15 on each task.",
      "OWA: ECE 0.316.",
      "CWA: ECE 0.318."
     ]
    },
    {
     "n": 2,
     "text": "At least half of Luna's wrong answers are stated at 95% or more, on both tasks.",
     "status": "held",
     "checks": [
      "Rule: at least half of Luna's wrong answers are stated at 95% or more, on each task.",
      "OWA: 75.5% of 660 wrong answers.",
      "CWA: 80.6% of 633 wrong answers."
     ]
    },
    {
     "n": 3,
     "text": "Luna's AUROC is lower than Jev's on both tasks.",
     "status": "held",
     "checks": [
      "Rule: Luna's AUROC is below Jev's on each task, on the same items.",
      "OWA: Luna 0.662, Jev 0.849.",
      "CWA: Luna 0.679, Jev 0.858."
     ]
    },
    {
     "n": 4,
     "text": "On most responses, Luna discloses fewer than all of the task's options.",
     "status": "held",
     "checks": [
      "Rule: on more than half of the responses, fewer than all of the task's options are disclosed.",
      "OWA: 1,764 of 1,800 responses disclose fewer than all 3 options.",
      "CWA: 1,455 of 1,800 responses disclose fewer than all 2 options."
     ]
    }
   ]
  }
 ]
}