{
  "suite": "typed",
  "limited": false,
  "logical_cases": 400,
  "decisions": 2000,
  "per_repeat": {
    "0": {
      "attempted": 2000,
      "valid": 2000,
      "correct": 625,
      "accuracy": 0.3125,
      "raw_top1_accuracy": 0.3125,
      "valid_accuracy": 0.3125,
      "ece_hard": 0.49122578356378166,
      "ece_hard_15": 0.49122578356378166,
      "accepted": 1112,
      "correct_accepted": 382,
      "accepted_accuracy": 0.3435251798561151,
      "accepted_correct_all": 0.191,
      "wrong_accepted": 730,
      "coverage": 0.556,
      "soft_accuracy": 0.36557797049241014,
      "brier_soft": 0.6502835864474757,
      "brier_hard": 1.115154911377793,
      "nll_hard": 3.3378542799889592,
      "kl": 2.280329899353666,
      "tvd": 0.5474679274931101,
      "score_mae": 0.942260512337909,
      "within_one": 0.60125,
      "case_errors": 0,
      "case_latency_ms": {
        "p50": 173.45825200000002,
        "p95": 204.21810405
      },
      "by_workflow": {
        "agent_trace_observability": {
          "attempted": 500,
          "valid": 500,
          "correct": 170,
          "accuracy": 0.34,
          "raw_top1_accuracy": 0.34,
          "valid_accuracy": 0.34,
          "ece_hard": 0.4162539228337721,
          "ece_hard_15": 0.4162539228337721,
          "accepted": 201,
          "correct_accepted": 104,
          "accepted_accuracy": 0.5174129353233831,
          "accepted_correct_all": 0.208,
          "wrong_accepted": 97,
          "coverage": 0.402,
          "soft_accuracy": 0.366733706872998,
          "brier_soft": 0.5527779009803165,
          "brier_hard": 0.9940006177340426,
          "nll_hard": 2.907857810112426,
          "kl": 1.9226320617681274,
          "tvd": 0.5434906519054364,
          "score_mae": 0.8273675541450701,
          "within_one": 0.635
        },
        "customer_service": {
          "attempted": 500,
          "valid": 500,
          "correct": 155,
          "accuracy": 0.31,
          "raw_top1_accuracy": 0.31,
          "valid_accuracy": 0.31,
          "ece_hard": 0.5938924941203059,
          "ece_hard_15": 0.5949591699394648,
          "accepted": 403,
          "correct_accepted": 112,
          "accepted_accuracy": 0.27791563275434245,
          "accepted_correct_all": 0.224,
          "wrong_accepted": 291,
          "coverage": 0.806,
          "soft_accuracy": 0.3106593113564554,
          "brier_soft": 0.8841866119036442,
          "brier_hard": 1.2435609877543936,
          "nll_hard": 4.371463067944412,
          "kl": 3.9096373072925115,
          "tvd": 0.6224232548598155,
          "score_mae": 0.7971103348638741,
          "within_one": 0.66
        },
        "invoice_processing": {
          "attempted": 500,
          "valid": 500,
          "correct": 186,
          "accuracy": 0.372,
          "raw_top1_accuracy": 0.372,
          "valid_accuracy": 0.372,
          "ece_hard": 0.40605181924758316,
          "ece_hard_15": 0.40605181924758316,
          "accepted": 268,
          "correct_accepted": 112,
          "accepted_accuracy": 0.417910447761194,
          "accepted_correct_all": 0.224,
          "wrong_accepted": 156,
          "coverage": 0.536,
          "soft_accuracy": 0.4343651018181026,
          "brier_soft": 0.6215623648269162,
          "brier_hard": 0.9745847776850229,
          "nll_hard": 2.376615929885815,
          "kl": 1.6507644953057514,
          "tvd": 0.5051933531309917,
          "score_mae": 0.9910947610539871,
          "within_one": 0.67
        },
        "security_incidents": {
          "attempted": 500,
          "valid": 500,
          "correct": 114,
          "accuracy": 0.228,
          "raw_top1_accuracy": 0.228,
          "valid_accuracy": 0.228,
          "ece_hard": 0.5488997007626499,
          "ece_hard_15": 0.5488997007626499,
          "accepted": 240,
          "correct_accepted": 54,
          "accepted_accuracy": 0.225,
          "accepted_correct_all": 0.108,
          "wrong_accepted": 186,
          "coverage": 0.48,
          "soft_accuracy": 0.35055376192208443,
          "brier_soft": 0.5426074680790262,
          "brier_hard": 1.2484732623377128,
          "nll_hard": 3.695480312013184,
          "kl": 1.6382857330482716,
          "tvd": 0.5187644500761969,
          "score_mae": 1.1534693992887046,
          "within_one": 0.44
        }
      },
      "by_type": {
        "binary": {
          "attempted": 600,
          "valid": 600,
          "correct": 264,
          "accuracy": 0.44,
          "raw_top1_accuracy": 0.44,
          "valid_accuracy": 0.44,
          "ece_hard": 0.3903806370448735,
          "ece_hard_15": 0.3903806370448735,
          "accepted": 374,
          "correct_accepted": 158,
          "accepted_accuracy": 0.42245989304812837,
          "accepted_correct_all": 0.2633333333333333,
          "wrong_accepted": 216,
          "coverage": 0.6233333333333333,
          "soft_accuracy": 0.47745312504946097,
          "brier_soft": 0.5230357443047486,
          "brier_hard": 0.8723281618824539,
          "nll_hard": 1.6506052531067084,
          "kl": 1.0948148081010385,
          "tvd": 0.42767884013673835,
          "score_mae": null,
          "within_one": null
        },
        "choice": {
          "attempted": 600,
          "valid": 600,
          "correct": 161,
          "accuracy": 0.2683333333333333,
          "raw_top1_accuracy": 0.2683333333333333,
          "valid_accuracy": 0.2683333333333333,
          "ece_hard": 0.5465692275466153,
          "ece_hard_15": 0.5465692275466153,
          "accepted": 369,
          "correct_accepted": 125,
          "accepted_accuracy": 0.33875338753387535,
          "accepted_correct_all": 0.20833333333333334,
          "wrong_accepted": 244,
          "coverage": 0.615,
          "soft_accuracy": 0.2537028159353592,
          "brier_soft": 0.7775314285902029,
          "brier_hard": 1.249013908116172,
          "nll_hard": 4.555815485580148,
          "kl": 3.4658449906062923,
          "tvd": 0.667257014849482,
          "score_mae": null,
          "within_one": null
        },
        "ordinal": {
          "attempted": 800,
          "valid": 800,
          "correct": 200,
          "accuracy": 0.25,
          "raw_top1_accuracy": 0.25,
          "valid_accuracy": 0.25,
          "ece_hard": 0.5253520604658375,
          "ece_hard_15": 0.5253520604658375,
          "accepted": 369,
          "correct_accepted": 99,
          "accepted_accuracy": 0.2682926829268293,
          "accepted_correct_all": 0.12375,
          "wrong_accepted": 270,
          "coverage": 0.46125,
          "soft_accuracy": null,
          "brier_soft": null,
          "brier_hard": 1.196880725945513,
          "nll_hard": 3.6898201459572566,
          "kl": null,
          "tvd": null,
          "score_mae": 0.942260512337909,
          "within_one": 0.60125
        }
      },
      "probe_checks": [],
      "probe_failures": 0
    }
  },
  "errors": [],
  "cache": {
    "prompts": {
      "hits": 0,
      "misses": 0,
      "insertions": 0,
      "evictions": 0,
      "skipped": 0
    },
    "candidates": {
      "hits": 0,
      "misses": 0,
      "insertions": 0,
      "evictions": 0,
      "skipped": 0
    }
  },
  "profile": {
    "prepare_ms": 3516.319198,
    "native_ms": 60279.416419,
    "score_ms": 1865.715596
  },
  "unique_batch_elapsed_ms": 66356.364011,
  "logical_input_tokens": 662280,
  "reused_prefix_tokens": 0,
  "evaluated_input_tokens": 662280,
  "scope": "Local L2S1 inference, no official JevBench composite score. Per-case latency is full batch completion, not divided by questions/batch width. Repeat 0 alone is the dataset quality result; later repeats are cache diagnostics. Hard ECE uses ten equal-width bins against argmax labels; ece_hard_15 uses fifteen bins. Bin edges go to the upper bin and probability 1 to the last. Confidence is max candidate probability, not entropy confidence or soft-target calibration. Hard Brier is summed over labels; hard NLL uses natural logs with probability floor 1e-12. Probability metrics average valid labeled decisions. Accepted correctness scores the native selection. Raw accuracy, coverage and accepted_correct_all use the planned denominator. Soft metrics cover binary/choice only, as upstream; ordinal MAE uses the probability-weighted level. AUROC uses tie-aware ranks. No Platt fitting or evaluation-set training. Probe thresholds are upstream heuristics, not logical guarantees; stability also changes rubric wording. Overconfidence uses L2S1 entropy confidence and is not assumed identical to Laya confidence.",
  "boundary_and_other_ms": 694.9127979999903
}
