{
  "suite": "typed",
  "limited": false,
  "logical_cases": 400,
  "decisions": 2000,
  "per_repeat": {
    "0": {
      "attempted": 2000,
      "valid": 2000,
      "correct": 1086,
      "accuracy": 0.543,
      "raw_top1_accuracy": 0.543,
      "valid_accuracy": 0.543,
      "ece_hard": 0.42018999728712986,
      "ece_hard_15": 0.42018999728712986,
      "accepted": 1855,
      "correct_accepted": 1033,
      "accepted_accuracy": 0.5568733153638814,
      "accepted_correct_all": 0.5165,
      "wrong_accepted": 822,
      "coverage": 0.9275,
      "soft_accuracy": 0.4699415881398428,
      "brier_soft": 0.6467756970606577,
      "brier_hard": 0.8516392648761149,
      "nll_hard": 3.497461006045127,
      "kl": 3.546941788027439,
      "tvd": 0.5200121418773491,
      "score_mae": 0.5240398880424766,
      "within_one": 0.89,
      "case_errors": 0,
      "case_latency_ms": {
        "p50": 322.64065100000005,
        "p95": 370.9781752
      },
      "by_workflow": {
        "agent_trace_observability": {
          "attempted": 500,
          "valid": 500,
          "correct": 239,
          "accuracy": 0.478,
          "raw_top1_accuracy": 0.478,
          "valid_accuracy": 0.478,
          "ece_hard": 0.4742327626288438,
          "ece_hard_15": 0.47568923914128086,
          "accepted": 454,
          "correct_accepted": 222,
          "accepted_accuracy": 0.4889867841409692,
          "accepted_correct_all": 0.444,
          "wrong_accepted": 232,
          "coverage": 0.908,
          "soft_accuracy": 0.4216433615247765,
          "brier_soft": 0.6378783915246882,
          "brier_hard": 0.961005000660257,
          "nll_hard": 3.917906234716017,
          "kl": 3.50663929270365,
          "tvd": 0.5601420999434151,
          "score_mae": 0.4962178059642531,
          "within_one": 0.91
        },
        "customer_service": {
          "attempted": 500,
          "valid": 500,
          "correct": 328,
          "accuracy": 0.656,
          "raw_top1_accuracy": 0.656,
          "valid_accuracy": 0.656,
          "ece_hard": 0.30840378110740246,
          "ece_hard_15": 0.30840378110740246,
          "accepted": 465,
          "correct_accepted": 315,
          "accepted_accuracy": 0.6774193548387096,
          "accepted_correct_all": 0.63,
          "wrong_accepted": 150,
          "coverage": 0.93,
          "soft_accuracy": 0.5779399563736318,
          "brier_soft": 0.44559377361513,
          "brier_hard": 0.6299733525964054,
          "nll_hard": 2.624234950420785,
          "kl": 2.8248823869756534,
          "tvd": 0.4115462482937375,
          "score_mae": 0.40054217848284873,
          "within_one": 0.95
        },
        "invoice_processing": {
          "attempted": 500,
          "valid": 500,
          "correct": 222,
          "accuracy": 0.444,
          "raw_top1_accuracy": 0.444,
          "valid_accuracy": 0.444,
          "ece_hard": 0.5161950673081589,
          "ece_hard_15": 0.5204818825749065,
          "accepted": 461,
          "correct_accepted": 209,
          "accepted_accuracy": 0.45336225596529284,
          "accepted_correct_all": 0.418,
          "wrong_accepted": 252,
          "coverage": 0.922,
          "soft_accuracy": 0.3432591182513516,
          "brier_soft": 1.056969743530116,
          "brier_hard": 1.0416117406372296,
          "nll_hard": 4.248122809663714,
          "kl": 4.582999287784135,
          "tvd": 0.6529245967504047,
          "score_mae": 0.6285405639081698,
          "within_one": 0.84
        },
        "security_incidents": {
          "attempted": 500,
          "valid": 500,
          "correct": 297,
          "accuracy": 0.594,
          "raw_top1_accuracy": 0.594,
          "valid_accuracy": 0.594,
          "ece_hard": 0.38192837810411445,
          "ece_hard_15": 0.38192837810411445,
          "accepted": 475,
          "correct_accepted": 287,
          "accepted_accuracy": 0.6042105263157894,
          "accepted_correct_all": 0.574,
          "wrong_accepted": 188,
          "coverage": 0.95,
          "soft_accuracy": 0.5369239164096113,
          "brier_soft": 0.44666087957269646,
          "brier_hard": 0.7739669656105677,
          "nll_hard": 3.199580029379991,
          "kl": 3.2732461846463186,
          "tvd": 0.45543562252183906,
          "score_mae": 0.5708590038146347,
          "within_one": 0.86
        }
      },
      "by_type": {
        "binary": {
          "attempted": 600,
          "valid": 600,
          "correct": 304,
          "accuracy": 0.5066666666666667,
          "raw_top1_accuracy": 0.5066666666666667,
          "valid_accuracy": 0.5066666666666667,
          "ece_hard": 0.4907666187925155,
          "ece_hard_15": 0.4907666187925155,
          "accepted": 599,
          "correct_accepted": 304,
          "accepted_accuracy": 0.5075125208681135,
          "accepted_correct_all": 0.5066666666666667,
          "wrong_accepted": 295,
          "coverage": 0.9983333333333333,
          "soft_accuracy": 0.47105527784959206,
          "brier_soft": 0.7526465589645519,
          "brier_hard": 0.9774066739492673,
          "nll_hard": 5.4731136650699534,
          "kl": 3.94551256496018,
          "tvd": 0.5279605087925154,
          "score_mae": null,
          "within_one": null
        },
        "choice": {
          "attempted": 600,
          "valid": 600,
          "correct": 337,
          "accuracy": 0.5616666666666666,
          "raw_top1_accuracy": 0.5616666666666666,
          "valid_accuracy": 0.5616666666666666,
          "ece_hard": 0.410422244617808,
          "ece_hard_15": 0.4120539323588499,
          "accepted": 569,
          "correct_accepted": 328,
          "accepted_accuracy": 0.5764499121265377,
          "accepted_correct_all": 0.5466666666666666,
          "wrong_accepted": 241,
          "coverage": 0.9483333333333334,
          "soft_accuracy": 0.46882789843009365,
          "brier_soft": 0.5409048351567634,
          "brier_hard": 0.8252222601385263,
          "nll_hard": 3.1004560946420243,
          "kl": 3.1483710110946985,
          "tvd": 0.5120637749621828,
          "score_mae": null,
          "within_one": null
        },
        "ordinal": {
          "attempted": 800,
          "valid": 800,
          "correct": 445,
          "accuracy": 0.55625,
          "raw_top1_accuracy": 0.55625,
          "valid_accuracy": 0.55625,
          "ece_hard": 0.3745833456600821,
          "ece_hard_15": 0.3745833456600821,
          "accepted": 687,
          "correct_accepted": 401,
          "accepted_accuracy": 0.5836972343522562,
          "accepted_correct_all": 0.50125,
          "wrong_accepted": 286,
          "coverage": 0.85875,
          "soft_accuracy": null,
          "brier_soft": null,
          "brier_hard": 0.7771264616244423,
          "nll_hard": 2.313475195328833,
          "kl": null,
          "tvd": null,
          "score_mae": 0.5240398880424766,
          "within_one": 0.89
        }
      },
      "probe_checks": [],
      "probe_failures": 0
    }
  },
  "errors": [],
  "cache": {
    "prompts": {
      "hits": 0,
      "misses": 0,
      "insertions": 0,
      "evictions": 0,
      "skipped": 0
    },
    "candidates": {
      "hits": 0,
      "misses": 0,
      "insertions": 0,
      "evictions": 0,
      "skipped": 0
    }
  },
  "profile": {
    "prepare_ms": 5313.547508,
    "native_ms": 112069.377266,
    "score_ms": 3246.888807
  },
  "unique_batch_elapsed_ms": 120921.260182,
  "logical_input_tokens": 661805,
  "reused_prefix_tokens": 0,
  "evaluated_input_tokens": 661805,
  "scope": "Local L2S1 inference, no official JevBench composite score. Per-case latency is full batch completion, not divided by questions/batch width. Repeat 0 alone is the dataset quality result; later repeats are cache diagnostics. Hard ECE uses ten equal-width bins against argmax labels; ece_hard_15 uses fifteen bins. Bin edges go to the upper bin and probability 1 to the last. Confidence is max candidate probability, not entropy confidence or soft-target calibration. Hard Brier is summed over labels; hard NLL uses natural logs with probability floor 1e-12. Probability metrics average valid labeled decisions. Accepted correctness scores the native selection. Raw accuracy, coverage and accepted_correct_all use the planned denominator. Soft metrics cover binary/choice only, as upstream; ordinal MAE uses the probability-weighted level. AUROC uses tie-aware ranks. No Platt fitting or evaluation-set training. Probe thresholds are upstream heuristics, not logical guarantees; stability also changes rubric wording. Overconfidence uses L2S1 entropy confidence and is not assumed identical to Laya confidence.",
  "boundary_and_other_ms": 291.4466010000033
}
