"""R7 — pluggable SelectorPolicy seam + local-OSS (Ollama) agent arm. Covers the RQ3 agent-backend build detail: * the policy seam resolves static/random/agent(heuristic) identically to the prior inline strategies (no behavioural regression); * the Ollama agent backend picks a valid rebuild circuit, is decision-cached (replay-deterministic), falls back to the local heuristic on any model/parse failure (never breaks a run), and records model+digest provenance; * the Claude Code arm is EXPLORATORY-only — it refuses to run and makes no paid call. The live Ollama cell is skipped where no local server/model is present, so the suite stays green offline. This is a *measurement* decision path (a local call to localhost), not relay traffic — containment is untouched. """ import json import urllib.request import pytest from cmd_chat.sor.agent_selector import ( DEFAULT_MODEL, ClaudeExploratoryPolicy, OllamaAgentPolicy, ) from cmd_chat.sor.churn import churn_schedule from cmd_chat.sor.selector import ( HeuristicAgentPolicy, RandomPolicy, SelectorContext, StaticPolicy, make_policy, run_selection, ) NODES = [f"n{i}" for i in range(8)] def _ollama_ready() -> bool: try: with urllib.request.urlopen("http://localhost:11434/api/tags", timeout=5) as r: tags = json.loads(r.read().decode()) except Exception: return False names = {m.get("name") for m in tags.get("models", [])} return DEFAULT_MODEL in names live_ollama = pytest.mark.skipif( not _ollama_ready(), reason=f"requires a local Ollama server with {DEFAULT_MODEL}", ) # --------------------------------------------------------------------------- # # Policy seam. # --------------------------------------------------------------------------- # def test_make_policy_resolves_builtins(): assert isinstance(make_policy("static"), StaticPolicy) assert isinstance(make_policy("random"), RandomPolicy) assert isinstance(make_policy("agent", "heuristic"), HeuristicAgentPolicy) assert isinstance(make_policy("agent", "ollama"), OllamaAgentPolicy) def test_make_policy_rejects_unknown(): with pytest.raises(ValueError): make_policy("nope") with pytest.raises(ValueError): make_policy("agent", "frontier-gpt") def test_heuristic_prefers_least_churned(): ctx = SelectorContext(seed=1, hops=3, kill_counts={"n0": 5, "n1": 5, "n2": 0}) chosen = HeuristicAgentPolicy().choose(NODES, ctx) # Nodes with 0 kills (n2..n7) outrank the twice-killed n0/n1. assert "n0" not in chosen and "n1" not in chosen assert len(chosen) == 3 and len(set(chosen)) == 3 # --------------------------------------------------------------------------- # # Ollama agent backend — fallback path is testable without any server. # --------------------------------------------------------------------------- # def test_agent_falls_back_deterministically_when_model_unreachable(): # Point at a dead port -> every model query fails -> deterministic heuristic. pol = OllamaAgentPolicy(host="http://127.0.0.1:1", timeout=0.5) ctx = SelectorContext(seed=7, hops=3, kill_counts={"n0": 9}) chosen = pol.choose(NODES, ctx) assert chosen == HeuristicAgentPolicy().choose(NODES, ctx) assert pol.decisions()[-1]["source"] == "fallback" # And it is cached: a second identical call is served from cache. again = pol.choose(NODES, ctx) assert again == chosen assert pol.decisions()[-1]["source"] == "cache" def test_agent_validity_guard_rejects_out_of_pool(): pol = OllamaAgentPolicy() assert pol._valid(["n0", "n1", "n2"], NODES, 3) is True assert pol._valid(["n0", "n0", "n2"], NODES, 3) is False # dup assert pol._valid(["n0", "n1"], NODES, 3) is False # wrong length assert pol._valid(["n0", "n1", "zzz"], NODES, 3) is False # not in pool assert pol._valid("nope", NODES, 3) is False def test_agent_provenance_shape(): pol = OllamaAgentPolicy(host="http://127.0.0.1:1", timeout=0.5) ctx = SelectorContext(seed=1, hops=3) pol.choose(NODES, ctx) prov = pol.provenance() assert prov["policy"] == "agent-ollama" assert prov["model"] == DEFAULT_MODEL assert prov["options"]["temperature"] == 0 assert prov["decisions"]["total"] == 1 # --------------------------------------------------------------------------- # # Claude arm — EXPLORATORY only, never spends. # --------------------------------------------------------------------------- # def test_claude_arm_is_exploratory_and_refuses(): pol = ClaudeExploratoryPolicy() with pytest.raises(NotImplementedError): pol.choose(NODES, SelectorContext(seed=1, hops=3)) prov = pol.provenance() assert prov["paid_calls"] is False and prov["wired"] is False def test_run_selection_agent_backend_via_string(): # agent+heuristic through run_selection rebuilds every dropped circuit and # records policy provenance. seed = 0xBADC0DE sched = churn_schedule(seed, NODES, steps=60, kill_prob_pct=35) res = run_selection(seed, NODES, hops=3, schedule=sched, strategy="agent", agent_backend="heuristic") assert res.strategy == "agent" assert res.every_drop_rebuilt assert res.policy_provenance["policy"] == "agent-heuristic" def test_run_selection_writes_selector_sidecar(tmp_path): seed = 42 sched = churn_schedule(seed, NODES, steps=20) res = run_selection(seed, NODES, hops=3, schedule=sched, strategy="agent", agent_backend="ollama", host="http://127.0.0.1:1", timeout=0.5, run_dir=tmp_path) sidecar = json.loads((tmp_path / "selector.json").read_text()) assert sidecar["provenance"]["policy"] == "agent-ollama" assert "decisions" in sidecar # per-state audit log present assert res.policy_provenance["model"] == DEFAULT_MODEL # --------------------------------------------------------------------------- # # Live cell — real local model (skipped if Ollama absent). # --------------------------------------------------------------------------- # @live_ollama def test_live_ollama_picks_valid_circuit_and_pins_digest(): pol = OllamaAgentPolicy() ctx = SelectorContext(seed=123, hops=3, kill_counts={"n0": 3, "n1": 1}) chosen = pol.choose(NODES, ctx) assert chosen is not None and len(chosen) == 3 and len(set(chosen)) == 3 assert all(c in set(NODES) for c in chosen) # Cached replay is byte-identical regardless of model nondeterminism. assert pol.choose(NODES, ctx) == chosen # The exact model weights are pinned for provenance. assert pol.provenance()["model_digest"] @live_ollama def test_live_ollama_run_selection_rebuilds_every_drop(tmp_path): seed = 2024 sched = churn_schedule(seed, NODES, steps=40, kill_prob_pct=35) res = run_selection(seed, NODES, hops=3, schedule=sched, strategy="agent", agent_backend="ollama", run_dir=tmp_path) assert res.every_drop_rebuilt assert (tmp_path / "selector.json").exists()