{ "best_depth": 1, "model_types": [ "qwen3_5" ], "artifact": "Qwen3.8-27B-JANG_2D", "quantization_mode": "affine", "quantization_bits": 2, "note": "best_depth=1, MEASURED against a WARM KV CACHE. 48.49 tok/s vs 48.49 at depth 1 (1.00x). Depth 3 = 39.11 tok/s, a net LOSS. Each extra verified token adds ~20% to the target decode step, so deeper speculation stops paying quickly. output_equivalent NOT asserted.", "reason": "stamped by stamp_qwen38_27b; depth sweep + warm-cache wall-clock, 2026-08-24", "measured_depth1_acceptance_pct": 85.89, "measured_best_depth": 1, "measured_depth_curve": [ { "depth": 1, "per_position_pct": 85.89, "chained_pct": 85.89, "expected_tokens_per_cycle": 1.859 }, { "depth": 2, "per_position_pct": 46.37, "chained_pct": 45.56, "expected_tokens_per_cycle": 2.315 }, { "depth": 3, "per_position_pct": 21.77, "chained_pct": 16.13, "expected_tokens_per_cycle": 2.476 } ], "validated": true, "baseline_tok_s": 48.49, "best_tok_s": 48.49, "speedup_vs_baseline": 1.0, "measured_tok_s_by_depth": { "1": 48.49, "2": 46.13, "3": 39.11 }, "blocked": false }