{"status":"measured_results_readout","register":"measurement_attestation_only","official_format_lm_eval":{"model":"csoai/sov33-unified","harness":"lm-eval-harness 0.4.12 (official format, reproducible; hf+GGUF validated lane)","results":{"arc_easy":{"acc":0.2534,"ci95":0.0089,"n":"full split"},"arc_challenge":{"acc":0.221,"ci95":0.0121,"n":"full split"},"hellaswag":{"acc":0.259,"ci95":0.0044,"n":"full split"}},"note":"General-knowledge trivia is not the target domain; see frozen_split_governance for the measured specialization. Prior llama-server numbers quarantined 2026-08-03 as instrument fault."},"frozen_split_governance":{"harness":"aiact-frozen-split-harness (published, recomputable)","n_scenarios":170,"results":[{"id":"grok-4.5","score":0.368,"ci95":[0.331,0.408]},{"id":"opus-4.1","score":0.323,"ci95":null},{"id":"sov33-unified","score":0.2508,"ci95":[0.221,0.283],"note":"identical-170 re-run 2026-08-03; statistical tie with sonnet-4.5 and gpt-5 on governance scenarios"},{"id":"sonnet-4.5","score":0.243,"ci95":null},{"id":"gpt-5","score":0.243,"ci95":null},{"id":"llama-4-maverick","score":0.22,"ci95":null},{"id":"gemini-2.5-pro","score":0.217,"ci95":null},{"id":"mistral-large","score":0.209,"ci95":null},{"id":"qwen3-235b","score":0.199,"ci95":null},{"id":"sov34-1.5b-lora","score":0.1975,"ci95":[0.169,0.226],"note":"dual-gate candidate 2026-08-03: generality gate PASS (arc_easy 0.7504, matches own base), frozen-split gate FAIL vs 0.2508 — no successor claims; published with failures included"},{"id":"qwen2.5-1.5b-base","score":0.1767,"ci95":[0.151,0.205],"note":"control: sov34 base weights"}],"excluded":[{"id":"deepseek-v3.2","reason":"94/170 unparseable under strict parser — instrument/format fault, score withheld rather than folded in as zero"},{"id":"kimi-k2.6","reason":"instrument fault: thinking-only responses exhausted token budget (142/170 empty) — excluded, not scored"}]},"govbench_15d_composite":{"dimensions":15,"note":"score_band is a descriptive range of the measured composite — measurement, not certification. Scores from other dimension counts are not comparable.","results":[{"id":"sov33-dist-c3","composite":57,"score_band":"bronze"},{"id":"sov33-evolved","composite":57,"score_band":"bronze"},{"id":"sov33-dist-c2","composite":54.6,"score_band":"bronze"},{"id":"sov33-dist-c1","composite":49.2,"score_band":"below band threshold"},{"id":"qwen2.5-0.5b","composite":43.3,"score_band":"below band threshold"},{"id":"sovereign-v4","composite":42.9,"score_band":"below band threshold"}]},"not_measured":["mmlu","gsm8k","aime","ifeval","bbh"],"artifacts":{"harness_dataset":"https://huggingface.co/datasets/csoai/aiact-frozen-split-harness","model":"https://huggingface.co/csoai/sov33-unified","doi":"10.5281/zenodo.21755657"},"timestamp":"2026-08-05T12:15:04.810Z","sigil_algo":"HMAC-SHA256","sigil":"f9dffb9647033e0efa27e1ffe60a8829fa38aa0f7f4c2b82d803774a45a0edb1","note":"Every score above is from a completed archived run with a published harness. Unmeasured benchmarks are listed explicitly and never estimated."}