"""RQ3 confirmatory-harness pins — structure, determinism, gate logic, Holm-7 wiring. These exercise the RQ3 confirmatory analyzer's mechanics on a SMALL SYNTHETIC battery (so the test owns its inputs and does not depend on the sealed live record): (a) the battery loader recovers per-arm retention/latency and the per-run mean-gap signal; (b) the three frozen gates read the right CI bound (perf: lower ≥ +10pp; latency: upper ≤ 100ms; P2: upper ≤ 0.60) and RQ3-P3 is their logical AND; (c) the run-level multi-arm bootstrap is byte-deterministic (same seed → identical CI) and BCa-well-formed; (d) the authoritative Holm is over EXACTLY the frozen size-7 family with the RQ2-P3 slot carrying the mechanism-corrected primary p, and the step-down/survivor logic matches stats.holm_bonferroni. """ import json from cmd_chat.sor.analysis import rq3_confirm as rc from cmd_chat.sor.analysis import stats def _synthetic_battery(tmp_path): """A 3-arm battery where agent clearly beats baselines on retention and is faster, with distinct rebuild-gap signals — so the gate directions are unambiguous.""" def cell(strat, ret, lat): return {"strategy": strat, "runs": len(ret), "throughput_retention": ret, "added_latency_ms": lat} cells = { "a": cell("agent", [1.0] * 8, [100.0] * 8), "s": cell("static", [0.5] * 8, [400.0] * 8), "r": cell("random", [0.6] * 8, [300.0] * 8), } runs = [] # frozen classifier scores SHORTER gaps (more churn) as positive; give the agent # shorter gaps than the baselines so it separates upward (AUC -> 1), the fingerprint # direction the ≤0.60 ceiling is meant to catch. for strat, gap in (("agent", 1.0), ("static", 3.0), ("random", 3.0)): for i in range(8): runs.append({"cell_id": f"RQ3/topo=1house/bridge=off/selector={strat}/churn=kp30s20", "rebuild_gaps": [gap, gap, gap], "run_index": i}) doc = {"schema": "sor-rq3-battery-results/1", "cells": cells, "runs": runs} p = tmp_path / "battery.json" p.write_text(json.dumps(doc)) return p # --- loader ----------------------------------------------------------------- # def test_loader_recovers_per_arm_dvs_and_gap_signal(tmp_path): arms = rc.load_battery(_synthetic_battery(tmp_path)) assert set(arms) == {"agent", "static", "random"} assert arms["agent"]["retention"] == [1.0] * 8 assert arms["static"]["latency"] == [400.0] * 8 # per-run MEAN gap: three equal gaps -> that value per run, 8 runs assert arms["agent"]["gap_signal"] == [1.0] * 8 assert arms["random"]["gap_signal"] == [3.0] * 8 # --- gate directions + RQ3-P3 AND ------------------------------------------- # def test_gates_read_correct_ci_bounds_and_join(tmp_path): arms = rc.load_battery(_synthetic_battery(tmp_path)) res = rc.analyze_rq3(arms, n_resamples=500) perf, lat, p2, joint = (res["RQ3-P1-perf"], res["RQ3-P1-latency"], res["RQ3-P2"], res["RQ3-P3-joint"]) # agent retention 1.0 vs max(0.5,0.6)=0.6 -> margin +0.4, CI lower well above +0.1 assert perf["point"] > 0.1 and perf["holds"] is True and perf["decision"] == "perf-gain" # agent latency 100 vs min(400,300)=300 -> -200ms, upper bound << 100ms assert lat["point"] < 0 and lat["holds"] is True assert lat["min_latency_baseline_arm"] == "random" # agent gap 1.0 vs baseline 3.0 -> fully separable upward AUC=1.0 -> upper > 0.60 -> fails assert p2["point"] == 1.0 and p2["holds"] is False # P3 = (perf ∧ latency) ∧ P2 ; P2 fails here -> H0 assert joint["p1_holds"] is True and joint["p2_holds"] is False assert joint["confirm"] is False and joint["decision"] == "H0" # --- determinism + BCa well-formed ------------------------------------------ # def test_analysis_is_byte_deterministic(tmp_path): b = _synthetic_battery(tmp_path) a1 = rc.analyze_rq3(rc.load_battery(b), n_resamples=800) a2 = rc.analyze_rq3(rc.load_battery(b), n_resamples=800) assert a1 == a2 for t in ("RQ3-P1-perf", "RQ3-P1-latency", "RQ3-P2"): assert a1[t]["ci_lo"] <= a1[t]["point"] <= a1[t]["ci_hi"] assert 0.0 <= a1[t]["p_for_holm"] <= 1.0 # --- multi-arm bootstrap matches the frozen two-sample rule ------------------ # def test_multi_arm_bootstrap_matches_two_sample_diff_for_two_means(tmp_path): a = [1.0, 2.0, 3.0, 4.0, 5.0] b = [0.5, 1.5, 2.5, 3.5] ci_gen, _ = rc._multi_arm_bootstrap( {"x": a, "y": b}, lambda d: stats.mean(d["x"]) - stats.mean(d["y"]), null=0.0, seed=123, n_resamples=1000) ci_ref = stats.two_sample_diff_ci(a, b, stats.mean, seed=123, n_resamples=1000) # same seed + same independent-arm resampling rule -> identical point + interval assert abs(ci_gen.point - ci_ref.point) < 1e-12 assert abs(ci_gen.lo - ci_ref.lo) < 1e-9 and abs(ci_gen.hi - ci_ref.hi) < 1e-9 # --- authoritative Holm-7 --------------------------------------------------- # def test_holm7_family_is_exactly_the_frozen_seven(tmp_path): assert rc.FAMILY_SIZE == 7 assert set(rc.FROZEN_FAMILY) == { "RQ1-P1", "RQ1-P2", "RQ2-P1", "RQ2-P3", "RQ3-P1-perf", "RQ3-P1-latency", "RQ3-P2"} def test_holm7_survivor_logic_with_three_zeros(tmp_path): # three p=0 priors + non-significant RQ3 tests: only the three zeros survive Holm-7. rq3 = { "RQ3-P1-perf": {"p_for_holm": 0.18}, "RQ3-P1-latency": {"p_for_holm": 0.50}, "RQ3-P2": {"p_for_holm": 0.17}, } # stub the two sealed prior records lead = tmp_path / "lead.json" lead.write_text(json.dumps({"confirmatory": { "RQ1-P1": {"p_for_holm": 0.0}, "RQ1-P2": {"p_for_holm": 0.0912}, "RQ2-P1": {"p_for_holm": 0.0}}})) rq2p3 = tmp_path / "rq2p3.json" rq2p3.write_text(json.dumps({"results": {"H1_pooled_spearman": {"p_for_holm": 0.0}}})) holm = rc.authoritative_holm7(rq3, lead, rq2p3) assert holm["family_size"] == 7 assert holm["survivors"] == ["RQ1-P1", "RQ2-P1", "RQ2-P3"] assert set(holm["non_survivors"]) == {"RQ1-P2", "RQ3-P1-perf", "RQ3-P1-latency", "RQ3-P2"} # RQ2-P3 slot must carry the mechanism-corrected primary p, not the lead degenerate 1.0 assert "mechanism-corrected" in holm["p_sources"]["RQ2-P3"] # RQ1-P2 at rank 4 gets multiplier 4 -> 0.0912*4 = 0.3648, not rejected row = {r["name"]: r for r in holm["rows"]} assert row["RQ1-P2"]["multiplier"] == 4 and abs(row["RQ1-P2"]["holm_p"] - 0.3648) < 1e-9