{"finding": "HUGE\u2212Big v3: +1.78 percentage points, 95% interval [+0.01, +3.70] on the full operational public suite. Adjusting for three tier comparisons gives [-0.49, +3.93] pp, which includes zero: a decisive HUGE advantage over Big is not established.", "common_n": 4655, "models": [{"name": "Smol", "full": 0.3885317100375453, "common": 0.3889055073949782, "truncated_core": 15, "unsupported_core": 0, "latency": {"p50_ms": 16.88216599998782, "p95_ms": 18.517645500001123}, "permutation": {"n": 320, "paired_valid_predictions": 320, "semantic_flip_count": 0, "semantic_flip_rate": 0.0, "metrics": {"n": 320, "accuracy": 0.384375, "macro_f1": 0.12168226039193782, "successful_supported_predictions": 320, "failures_or_unsupported": 0, "calibration": {"n": 320, "nll": 1.059501683108747, "brier": 0.6255237290148342, "ece_10_bins": 0.07666907534003259, "bins": [{"lower": 0.1, "n": 31, "accuracy": 0.12903225806451613, "mean_confidence": 0.15692059263106314}, {"lower": 0.2, "n": 6, "accuracy": 0.16666666666666666, "mean_confidence": 0.23558077216148376}, {"lower": 0.3, "n": 57, "accuracy": 0.3157894736842105, "mean_confidence": 0.37566835001895305}, {"lower": 0.4, "n": 56, "accuracy": 0.2857142857142857, "mean_confidence": 0.44307045532124384}, {"lower": 0.5, "n": 141, "accuracy": 0.5106382978723404, "mean_confidence": 0.5413181308313464}, {"lower": 0.6, "n": 25, "accuracy": 0.36, "mean_confidence": 0.6241083168983459}, {"lower": 0.7, "n": 4, "accuracy": 0.75, "mean_confidence": 0.7243811786174774}], "confidence_basis": "probability assigned to returned decision"}}, "missing_reference_count": 0, "note": "Flip compares selected stable option IDs on each permutation row with its reference row; failures are excluded from flip-rate denominator and remain incorrect in accuracy."}, "predictions_sha256": "d509e6914add7ff53a4c22e3dc14993c32d493ee99bfb450d3a4becd06214838"}, {"name": "Big", "full": 0.7614490278669213, "common": 0.7611601813443893, "truncated_core": 3, "unsupported_core": 5, "latency": {"p50_ms": 70.34716850000677, "p95_ms": 82.7861887499921}, "permutation": {"n": 320, "paired_valid_predictions": 320, "semantic_flip_count": 31, "semantic_flip_rate": 0.096875, "metrics": {"n": 320, "accuracy": 0.771875, "macro_f1": 0.4738826196938666, "successful_supported_predictions": 320, "failures_or_unsupported": 0, "calibration": {"n": 320, "nll": 0.6448529973804664, "brier": 0.33475511621201265, "ece_10_bins": 0.062161904407445484, "bins": [{"lower": 0.1, "n": 7, "accuracy": 0.0, "mean_confidence": 0.168374946620973}, {"lower": 0.2, "n": 7, "accuracy": 0.14285714285714285, "mean_confidence": 0.22723231677205008}, {"lower": 0.3, "n": 10, "accuracy": 0.4, "mean_confidence": 0.35295258985684635}, {"lower": 0.4, "n": 5, "accuracy": 0.4, "mean_confidence": 0.44763205095463954}, {"lower": 0.5, "n": 24, "accuracy": 0.625, "mean_confidence": 0.5541452672272809}, {"lower": 0.6, "n": 17, "accuracy": 0.7647058823529411, "mean_confidence": 0.6608453550190693}, {"lower": 0.7, "n": 37, "accuracy": 0.7837837837837838, "mean_confidence": 0.7452427060891775}, {"lower": 0.8, "n": 68, "accuracy": 0.7352941176470589, "mean_confidence": 0.8585095779523613}, {"lower": 0.9, "n": 145, "accuracy": 0.9172413793103448, "mean_confidence": 0.9458145550603074}], "confidence_basis": "probability assigned to returned decision"}}, "missing_reference_count": 0, "note": "Flip compares selected stable option IDs on each permutation row with its reference row; failures are excluded from flip-rate denominator and remain incorrect in accuracy."}, "predictions_sha256": "bc5007c09129f81fdc8461cc7efe2a749566cd2baf1d8f3f4e148570ac3cec46"}, {"name": "HUGE", "full": 0.7792167483474552, "common": 0.778725671411366, "truncated_core": 0, "unsupported_core": 0, "latency": {"p50_ms": 90.8845460000407, "p95_ms": 102.12233349999877}, "permutation": {"n": 320, "paired_valid_predictions": 320, "semantic_flip_count": 42, "semantic_flip_rate": 0.13125, "metrics": {"n": 320, "accuracy": 0.79375, "macro_f1": 0.5307427683427683, "successful_supported_predictions": 320, "failures_or_unsupported": 0, "calibration": {"n": 320, "nll": 0.751473189747461, "brier": 0.34360826227135455, "ece_10_bins": 0.1233512708451599, "bins": [{"lower": 0.2, "n": 4, "accuracy": 0.25, "mean_confidence": 0.2278776504099369}, {"lower": 0.3, "n": 5, "accuracy": 0.2, "mean_confidence": 0.37006560564041135}, {"lower": 0.4, "n": 10, "accuracy": 0.4, "mean_confidence": 0.47035863995552063}, {"lower": 0.5, "n": 16, "accuracy": 0.75, "mean_confidence": 0.5409834794700146}, {"lower": 0.6, "n": 13, "accuracy": 0.5384615384615384, "mean_confidence": 0.6539576191168565}, {"lower": 0.7, "n": 10, "accuracy": 0.9, "mean_confidence": 0.7672818243503571}, {"lower": 0.8, "n": 30, "accuracy": 0.7333333333333333, "mean_confidence": 0.8577997306982676}, {"lower": 0.9, "n": 232, "accuracy": 0.853448275862069, "mean_confidence": 0.9738065335771133}], "confidence_basis": "probability assigned to returned decision"}}, "missing_reference_count": 0, "note": "Flip compares selected stable option IDs on each permutation row with its reference row; failures are excluded from flip-rate denominator and remain incorrect in accuracy."}, "predictions_sha256": "b35766b78a86bfba93025f9bbf0dd93f895a00294fe526840533b44afabd91cc"}], "enormous": {"name": "ENORMOUS", "full": 0.8306862906689559, "common": null, "truncated_core": 0, "unsupported_core": 0, "latency": null}, "enormous_finding": "ENORMOUS (27B) was scored later on the four-system common-evidence rows: ENORMOUS\u2212HUGE +5.27 pp, 95% interval [+3.49, +7.14] on 4,518 rows."}