From 8f5636c203e834e84ddc220d5fc044c23a14a61b Mon Sep 17 00:00:00 2001 From: Evgeny Kiriyak <224408464+evkir@users.noreply.github.com> Date: Fri, 28 Aug 2026 21:01:31 +0200 Subject: [PATCH] Cover the live L2 path against a real server Coverage reported eight lines missing in detector_eval: the --l2 branch and the block that writes a recording. They are the only branches of the command that talk to a model, so nothing reached them, and mocking them would have verified a stand-in for the code rather than the code. A socket that speaks ollama's shape reaches them instead. The flag, the recording wrapper, the default transport and the file written all run as they do live, and the test asserts the property the committed artifact's gate depends on: the live run, the bytes it wrote and a replayed run agree. That required the address to become an option, which is not a concession to the test. ollama does not have to sit on this host, and hard-coding that it does was a defect the missing coverage happened to surface. --- cyberai/cli/detector_eval.py | 11 ++- tests/unit/test_detector_eval_l2_live.py | 94 ++++++++++++++++++++++++ 2 files changed, 104 insertions(+), 1 deletion(-) create mode 100644 tests/unit/test_detector_eval_l2_live.py diff --git a/cyberai/cli/detector_eval.py b/cyberai/cli/detector_eval.py index 16e7620..eaf6adc 100644 --- a/cyberai/cli/detector_eval.py +++ b/cyberai/cli/detector_eval.py @@ -30,6 +30,7 @@ ) from cyberai.core.security.guard import DEFAULT_THRESHOLD from cyberai.core.security.llm_classifier import ( + DEFAULT_BASE_URL, DEFAULT_MODEL, LLMClassifier, RecordMismatch, @@ -146,6 +147,13 @@ def detector() -> None: show_default=True, help="Model the L2 layer asks. Local only, by design.", ) +@click.option( + "--l2-url", + default=DEFAULT_BASE_URL, + show_default=True, + help="Where the local model answers. ollama does not have to sit on this " + "host, and a test needs somewhere real to point at.", +) @click.option( "--l2-record", type=click.Path(dir_okay=False, writable=True, path_type=Path), @@ -172,6 +180,7 @@ def detector_eval( report: Path | None, use_l2: bool, l2_model: str, + l2_url: str, l2_record: Path | None, l2_replay: Path | None, ) -> None: @@ -193,7 +202,7 @@ def detector_eval( scorer = combined_scorer(classifier) layers = f"L1+L2 ({recording_model(l2_replay)})" elif use_l2: - classifier = LLMClassifier(model=l2_model) + classifier = LLMClassifier(model=l2_model, base_url=l2_url) if l2_record is not None: classifier.transport = recording_transport(classifier.transport, captured) scorer = combined_scorer(classifier) diff --git a/tests/unit/test_detector_eval_l2_live.py b/tests/unit/test_detector_eval_l2_live.py new file mode 100644 index 0000000..d3ed475 --- /dev/null +++ b/tests/unit/test_detector_eval_l2_live.py @@ -0,0 +1,94 @@ +"""The live L2 path, driven against a real server rather than a stand-in. + +--l2 and --l2-record are the only branches of this command that talk to a +model, so no other test reaches them and coverage reported them missing. +Pointing the command at a socket that speaks ollama's shape exercises the +flag, the recording wrapper and the file it writes through the same code a +real run uses -- the default transport included. + +The address had to become an option for that, which is not a concession to +the test: ollama does not have to sit on this host, and hard-coding that it +does was a defect the test happened to surface. +""" + +import http.server +import json +import pathlib +import threading + +import pytest +from click.testing import CliRunner + +from cyberai.cli.detector_eval import detector +from cyberai.core.security.eval_corpus import load_corpus +from cyberai.core.security.llm_classifier import _fingerprint + +CORPUS = str(pathlib.Path(__file__).resolve().parents[2] / "tests" / "corpus") + + +class _Ollama(http.server.BaseHTTPRequestHandler): + """Answers 'injection' for everything, which is enough to be recorded.""" + + def do_POST(self): + self.rfile.read(int(self.headers.get("content-length", 0))) + body = json.dumps( + {"message": {"content": json.dumps({"verdict": "injection", "reason": "r"})}} + ).encode() + self.send_response(200) + self.send_header("content-type", "application/json") + self.send_header("content-length", str(len(body))) + self.end_headers() + self.wfile.write(body) + + def log_message(self, *args): + pass + + +@pytest.fixture +def served(): + server = http.server.HTTPServer(("127.0.0.1", 0), _Ollama) + threading.Thread(target=server.serve_forever, daemon=True).start() + yield f"http://127.0.0.1:{server.server_port}" + server.shutdown() + + +@pytest.mark.unit +def test_the_live_flag_scores_through_a_real_request(served): + result = CliRunner().invoke( + detector, ["eval", "--corpus", CORPUS, "--l2", "--l2-url", served, "--json"] + ) + assert result.exit_code == 0, result.output + payload = json.loads(result.output) + assert payload["layers"].startswith("L1+L2 (") + assert payload["overall"]["true_positive"] == 49 + assert payload["blind_subclasses"] == [] + + +@pytest.mark.unit +def test_a_recorded_run_writes_what_a_later_run_can_replay(tmp_path, served): + """A recording is only worth writing if it loads again and scores the same. + + Asserted end to end rather than on the file's shape: the live run, the + bytes it wrote and the replayed run all have to agree, which is the + property the committed artifact's gate depends on. + """ + recording = tmp_path / "nested" / "verdicts.json" + live = CliRunner().invoke( + detector, + ["eval", "--corpus", CORPUS, "--l2", "--l2-url", served, "--l2-record", str(recording)], + ) + assert live.exit_code == 0, live.output + assert "verdicts recorded" in live.output + + body = json.loads(recording.read_text(encoding="utf-8")) + assert set(body["verdicts"]) == {_fingerprint(s.text) for s in load_corpus(CORPUS)} + + replayed = CliRunner().invoke( + detector, ["eval", "--corpus", CORPUS, "--l2-replay", str(recording), "--json"] + ) + assert replayed.exit_code == 0, replayed.output + + live_again = CliRunner().invoke( + detector, ["eval", "--corpus", CORPUS, "--l2", "--l2-url", served, "--json"] + ) + assert json.loads(replayed.output)["overall"] == json.loads(live_again.output)["overall"]