{ "schema": "semif-phase1-summary-v1", "date": "2026-09-16", "claim": "Direct option logits are the strongest tested open general-decision baseline; a native reranker remains a retrieval control and is not a drop-in universal decision operator.", "models": { "direct_logits": { "source": "Qwen/Qwen3.5-4B", "revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a" }, "reranker": { "source": "Qwen/Qwen3-Reranker-4B", "revision": "22e683669bc0f0bd69640a1354a6d0aebcfeede5" } }, "semantic_quality": { "authored_144_mean_family_balanced_accuracy": { "direct_logits": 0.8132381608, "reranker": 0.6250733042 }, "wanli_256_balanced_accuracy": { "direct_logits": 0.6365253078, "reranker": 0.5219790242 }, "typesafe_public_102_equal_case_modal_agreement": { "direct_logits": 0.8453205128, "reranker": 0.5595421245, "published_jev": 0.8831410256 }, "typesafe_public_102_tv_distance": { "direct_logits": 0.1770346368, "reranker": 0.4440789631, "published_jev": 0.1268250717 }, "every_judge_grid_36_accuracy": { "direct_logits": 0.8055555556, "reranker": 0.6944444444 }, "every_action_firewall_10_accuracy": { "direct_logits": 0.7, "reranker": 0.7 }, "every_code_rag_recall_at_1": { "direct_logits": 1.0, "reranker": 1.0 }, "every_company_brain_recall_at_1": { "direct_logits": 0.9285714286, "reranker": 0.9285714286 } }, "paired_reranker_minus_direct": { "authored_balanced_accuracy": { "difference": -0.1881648566, "bootstrap_95": [ -0.2557270628, -0.1200396825 ] }, "wanli_balanced_accuracy": { "difference": -0.1145462836, "bootstrap_95": [ -0.1844227325, -0.0444315396 ] } }, "perturbations_36": { "direct_logits": { "base_balanced_accuracy": 0.7227513228, "option_reversal": { "balanced_accuracy": 0.8132275132, "argmax_flips": 10 }, "criterion_wrapper": { "balanced_accuracy": 0.7058201058, "argmax_flips": 9 }, "irrelevant_context": { "balanced_accuracy": 0.8206349206, "argmax_flips": 4 }, "missing_evidence_confident_non_insufficient_at_0_8": 1 }, "reranker": { "base_balanced_accuracy": 0.5301587302, "option_reversal": { "balanced_accuracy": 0.4984126984, "argmax_flips": 2 }, "criterion_wrapper": { "balanced_accuracy": 0.6465608466, "argmax_flips": 9 }, "irrelevant_context": { "balanced_accuracy": 0.5634920635, "argmax_flips": 13 }, "missing_evidence_confident_non_insufficient_at_0_8": 1 } }, "shape777": { "fixture_sha256": "8dcf414b12fc2684e3c4ca5f3ebfd3f525f5346fec4a9bc67eb65138101f55f1", "hardware": "NVIDIA GeForce RTX 3090", "timing_scope": "Warm-loaded BF16 model; prompt construction/tokenization, transfers, forward and CPU readout included; model load and output write excluded.", "direct": [ { "mode": "fresh_batch1", "wall_seconds": 333.063398068, "judgments_per_second": 2.3328891872, "state_p50_seconds": 8.985686894, "peak_cuda_bytes": 8894241280, "argmax_flips_vs_fresh": 0 }, { "mode": "serial_prefix", "wall_seconds": 72.263639052, "judgments_per_second": 10.7522954863, "state_p50_seconds": 1.932188766, "peak_cuda_bytes": 8991294976, "argmax_flips_vs_fresh": 5 }, { "mode": "parallel_suffix", "wall_seconds": 38.792141172, "judgments_per_second": 20.0298301802, "state_p50_seconds": 1.047665117, "peak_cuda_bytes": 11611157504, "argmax_flips_vs_fresh": 6 } ], "reranker": [ { "pair_batch_size": 1, "wall_seconds": 417.30154415, "judgments_per_second": 1.861962916, "state_p50_seconds": 11.284789991, "peak_cuda_bytes": 8192491520, "argmax_flips_vs_batch1": 0 }, { "pair_batch_size": 4, "wall_seconds": 441.661915525, "judgments_per_second": 1.759264208, "state_p50_seconds": 11.94427757, "peak_cuda_bytes": 8619127296, "argmax_flips_vs_batch1": 51 }, { "pair_batch_size": 8, "wall_seconds": 435.194096286, "judgments_per_second": 1.7854102494, "state_p50_seconds": 11.756876173, "peak_cuda_bytes": 9185073152, "argmax_flips_vs_batch1": 54 } ] }, "boundaries": [ "No live Jev endpoint was run.", "The TypeSafe comparison is 102 public rows across 20 cases, not the reported 711-row aggregate.", "The shape777 fixture matches the 37x21 count only; inputs, serving stack and hardware are not equivalent to Jev.", "Conditional option scores are not established as calibrated confidence." ], "decision_vs_compact_generation21": { "model": "Qwen/Qwen3.5-4B", "revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", "hardware": "NVIDIA GeForce RTX 3090", "workload": "Same frozen model, exact state, criteria, hardware and process. Direct path returns two-option distributions; generation emits only an ordered yes/no JSON array.", "direct_parallel": { "runs_seconds": [ 1.036716750008054, 1.019257096981164, 1.023019488027785 ], "median_seconds": 1.023019488027785, "generated_output_tokens": 0, "result": "21 two-option probability distributions" }, "compact_generation": { "runs_seconds": [ 5.286041187006049, 5.331690029008314, 5.413413154950831 ], "time_to_first_token_seconds": [ 0.48786429001484066, 0.4888657660339959, 0.4883691109716892 ], "output_tokens": [ 111, 111, 111 ], "median_seconds": 5.331690029008314, "median_output_tokens": 111, "all_runs_valid_complete_arrays": true, "agreement_with_direct_argmax": 0.8571428571428571 }, "wall_time_ratio_generation_over_direct": 5.211718927551365 }, "verbose_generation_reference21": { "model": "Qwen/Qwen3.5-4B", "revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", "hardware": "NVIDIA GeForce RTX 3090", "workload": "Same frozen BF16 model, owned state, and 21 binary criteria; direct two-option distributions versus one autoregressively generated JSON object.", "direct_parallel": { "runs_seconds": [ 1.06307857, 1.065514043, 1.070278497 ], "median_seconds": 1.065514043, "generated_output_tokens": 0, "result": "21 two-option probability distributions" }, "naive_generation": { "input_tokens": 2927, "time_to_first_token_seconds": 0.642879314, "total_seconds": 18.229238557, "output_tokens": 405, "tokens_per_second_after_first": 22.9726836662, "valid_json": true, "complete_21_key_schema": true }, "wall_time_ratio_generation_over_direct": 17.1083982207, "status": "verbose reference; superseded as headline by compact ordered-array baseline" } }