"""Tests for reflect_auto.py — answer validation, batch persistence and the notification threshold.""" import asyncio import json import subprocess import sys from datetime import datetime from pathlib import Path from types import SimpleNamespace from typing import Any import pytest sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "scripts")) import reflect_auto TODAY = "2026-09-08" WHEN = datetime(2026, 9, 8) def _stats(**overrides): """The per-run statistics dict that feeds both the report and state.json.""" stats = {"at": "2026-09-08 03:30", "sessions": 4, "batches": 1, "open": 0, "watch": 0, "repeat_per_100": 0.0} stats.update(overrides) return stats def _raw_finding(**overrides): finding = { "pattern": "retry-without-diagnosis", "severity": "medium", "diagnosis": "After an error the same call is repeated with identical arguments.", "evidence": [{"session": "websocket_abc", "when": "2026-07-11", "excerpt": "web_fetch → ERROR ×4"}], "occurrences": 1, "sessions_affected": 1, "proposal": "Add a hard STOP gate.", } finding.update(overrides) return finding def _answer(*findings) -> str: return "```json\n" + json.dumps({"findings": list(findings)}) + "\n```" def _parsed(answer: str, session_count: int = 10) -> list[dict]: """Only the findings — notes about what was salvaged are asserted where they matter.""" return reflect_auto.parse_findings(answer, session_count)[0] def _filed(**overrides): """A record as it would already sit in findings.jsonl.""" record = { "id": "f0001", "status": "watch", "created": "2026-09-01", "pattern": "retry-without-diagnosis", "severity": "medium", "diagnosis": "After an error the same call is repeated.", "evidence": [{"session": "websocket_old"}], "occurrences": 1, "sessions_affected": 1, "proposal": "Add a hard STOP gate.", } record.update(overrides) return record class TestParseFindings: def test_accepts_a_well_formed_answer(self): parsed = _parsed(_answer(_raw_finding())) assert parsed[0]["pattern"] == "retry-without-diagnosis" assert parsed[0]["patch"] is None def test_accepts_an_empty_findings_list(self): assert _parsed(_answer()) == [] def test_accepts_bare_json_without_a_fence(self): assert _parsed(json.dumps({"findings": []})) == [] def test_ignores_narration_around_the_block(self): answer = "Here are the findings:\n" + _answer(_raw_finding()) + "\nDone." assert len(_parsed(answer)) == 1 def test_rejects_an_answer_without_json(self): with pytest.raises(reflect_auto.FindingsError): reflect_auto.parse_findings("I found three problems but I am not sending JSON.", 10) def test_unescaped_quote_error_points_at_the_offending_text(self): """The real failure seen on the first live run: a Czech „…" closing with ASCII ". A bare "no parseable json" message gives the retry nothing to work with, so the error must name the line, the column and the surrounding text. """ broken = '```json\n{"findings": [{"pattern": "x", "diagnosis": "runtime said („blocked") and then"}]}\n```' with pytest.raises(reflect_auto.FindingsError) as raised: reflect_auto.parse_findings(broken, 10) message = str(raised.value) assert "line 1 column" in message assert "blocked" in message assert "unescaped double quote" in message def test_error_is_located_inside_the_block_not_the_whole_reply(self): """Column numbers measured across the fence would be meaningless to the model.""" broken = 'Here is the result:\n\n```json\n{"findings": [{"diagnosis": "a "b" c"}]}\n```' with pytest.raises(reflect_auto.FindingsError) as raised: reflect_auto.parse_findings(broken, 10) assert "line 1 column" in str(raised.value) def test_last_block_wins_when_the_model_shows_its_work(self): first = json.dumps({"findings": [_raw_finding(pattern="draft-version")]}) second = json.dumps({"findings": [_raw_finding(pattern="final-version")]}) answer = f"First draft:\n```json\n{first}\n```\nCorrected:\n```json\n{second}\n```" assert _parsed(answer)[0]["pattern"] == "final-version" @pytest.mark.parametrize( "override", [ {"pattern": "Retry Without Diagnosis"}, {"pattern": "ab"}, {"severity": "critical"}, {"diagnosis": ""}, {"occurrences": 0}, {"occurrences": "sedm"}, {"evidence": []}, {"evidence": [{"when": "2026-07-11"}]}, ], ) def test_malformed_finding_is_dropped_and_the_others_survive(self, override): """Re-asking costs a ~420k token turn — one bad record must not throw the batch away.""" parsed, problems = reflect_auto.parse_findings( _answer(_raw_finding(**override), _raw_finding(pattern="something-else")), 10 ) assert [finding["pattern"] for finding in parsed] == ["something-else"] assert any("dropped" in problem for problem in problems) def test_an_answer_with_nothing_usable_still_raises(self): """Nothing salvageable means the turn was wasted — that is worth one retry.""" with pytest.raises(reflect_auto.FindingsError): _parsed(_answer(_raw_finding(severity="critical"))) def test_unknown_fields_are_ignored_not_fatal(self): parsed, problems = reflect_auto.parse_findings(_answer(_raw_finding(confidence=0.9)), 10) assert "confidence" not in parsed[0] assert any("confidence" in problem for problem in problems) def test_extra_findings_are_trimmed_not_the_batch(self): many = [_raw_finding(pattern=f"pattern-{index}") for index in range(reflect_auto.MAX_FINDINGS_PER_BATCH + 1)] parsed, problems = reflect_auto.parse_findings(_answer(*many), 10) assert len(parsed) == reflect_auto.MAX_FINDINGS_PER_BATCH assert any("kept the first" in problem for problem in problems) def test_overlong_diagnosis_is_truncated(self): parsed = _parsed(_answer(_raw_finding(diagnosis="x" * (reflect_auto.MAX_DIAGNOSIS_CHARS + 50)))) assert len(parsed[0]["diagnosis"]) == reflect_auto.MAX_DIAGNOSIS_CHARS assert parsed[0]["diagnosis"].endswith("…") class TestParsePatch: def _patch(self, **overrides): patch = {"file": "skills/note/SKILL.md", "old_text": "try again", "new_text": "STOP and diagnose"} patch.update(overrides) return patch def test_accepts_a_complete_patch(self): parsed = _parsed(_answer(_raw_finding(patch=self._patch()))) assert parsed[0]["patch"]["file"] == "skills/note/SKILL.md" @pytest.mark.parametrize( "override", [ {"old_text": ""}, {"new_text": "try again"}, {"file": "/etc/passwd"}, {"file": "../../../etc/passwd"}, ], ) def test_unsafe_or_empty_patch_is_dropped_but_the_finding_survives(self, override): """The diagnosis and the proposal are still worth reviewing without a patch.""" parsed, problems = reflect_auto.parse_findings(_answer(_raw_finding(patch=self._patch(**override))), 10) assert len(parsed) == 1 assert parsed[0]["patch"] is None assert any("kept the finding without it" in problem for problem in problems) def test_incomplete_patch_is_dropped_but_the_finding_survives(self): parsed = _parsed(_answer(_raw_finding(patch={"file": "a.md", "old_text": "x"}))) assert parsed[0]["patch"] is None class TestCounts: """The counts drive the threshold and the ranking, and nothing but this checks them.""" def test_more_sessions_than_occurrences_is_impossible(self): """Seen live: a filed finding claimed 4 occurrences across 5 sessions.""" parsed, problems = reflect_auto.parse_findings(_answer(_raw_finding(occurrences=4, sessions_affected=5)), 10) assert parsed[0]["sessions_affected"] == 4 assert any("clamped" in problem for problem in problems) def test_more_sessions_than_the_slice_held_is_impossible(self): parsed, _ = reflect_auto.parse_findings(_answer(_raw_finding(occurrences=9, sessions_affected=6)), 3) assert parsed[0]["sessions_affected"] == 3 def test_a_coherent_count_is_left_alone(self): parsed, problems = reflect_auto.parse_findings(_answer(_raw_finding(occurrences=7, sessions_affected=4)), 10) assert (parsed[0]["occurrences"], parsed[0]["sessions_affected"]) == (7, 4) assert problems == [] class TestEvidenceFolding: """`occurrences` sums across runs, so the examples behind it have to survive the fold.""" def _evidence(self, session: str) -> list[dict]: return [{"session": session, "when": "2026-09-02", "excerpt": "web_fetch → ERROR"}] def test_the_fold_keeps_the_older_evidence(self): existing = [_filed(evidence=self._evidence("websocket_old"))] raw = _raw_finding(evidence=self._evidence("websocket_new")) merged = reflect_auto.merge_findings(existing, _parsed(_answer(raw)), TODAY) sessions = [item["session"] for item in merged[0].evidence] assert sessions == ["websocket_new", "websocket_old"], "newest first, older behind it" def test_the_same_evidence_twice_is_kept_once(self): existing = [_filed(evidence=self._evidence("websocket_same"))] raw = _raw_finding(evidence=self._evidence("websocket_same")) merged = reflect_auto.merge_findings(existing, _parsed(_answer(raw)), TODAY) assert len(merged[0].evidence) == 1 def test_the_evidence_list_is_capped(self): older = [{"session": f"websocket_{i}", "excerpt": str(i)} for i in range(10)] raw = _raw_finding(evidence=self._evidence("websocket_new")) merged = reflect_auto.merge_findings([_filed(evidence=older)], _parsed(_answer(raw)), TODAY) assert len(merged[0].evidence) == reflect_auto.MAX_EVIDENCE def test_a_regression_does_not_mix_evidence_from_before_the_fix(self): existing = [_filed(status="applied", evidence=self._evidence("websocket_old"))] raw = _raw_finding(evidence=self._evidence("websocket_new")) merged = reflect_auto.merge_findings(existing, _parsed(_answer(raw)), TODAY) assert [item["session"] for item in merged[0].evidence] == ["websocket_new"] class TestThreshold: def test_first_sighting_stays_silent(self): """A single occurrence is noise, not a pattern — it must not reach Telegram.""" merged = reflect_auto.merge_findings([], _parsed(_answer(_raw_finding())), TODAY) assert merged[0].status == reflect_auto.STATUS_WATCH def test_repeat_within_one_batch_opens_immediately(self): raw = _raw_finding(occurrences=7, sessions_affected=4) merged = reflect_auto.merge_findings([], _parsed(_answer(raw)), TODAY) assert merged[0].status == reflect_auto.STATUS_OPEN def test_many_occurrences_in_one_session_stay_silent(self): """One session looping seven times is still one session — not yet a habit.""" raw = _raw_finding(occurrences=7, sessions_affected=1) merged = reflect_auto.merge_findings([], _parsed(_answer(raw)), TODAY) assert merged[0].status == reflect_auto.STATUS_WATCH def test_second_sighting_promotes_a_watched_pattern(self): merged = reflect_auto.merge_findings([_filed()], _parsed(_answer(_raw_finding())), TODAY) assert merged[0].status == reflect_auto.STATUS_OPEN assert merged[0].occurrences == 2 assert merged[0].sessions_affected == 2 def test_promotion_records_the_superseded_record(self): merged = reflect_auto.merge_findings([_filed()], _parsed(_answer(_raw_finding())), TODAY) assert merged[0].history == ("2026-09-01:f0001",) def test_reappearing_after_apply_is_a_regression(self): """A fixed pattern coming back must open at once, however few the occurrences.""" merged = reflect_auto.merge_findings([_filed(status="applied")], _parsed(_answer(_raw_finding())), TODAY) assert merged[0].status == reflect_auto.STATUS_OPEN assert merged[0].regression_of == "f0001" assert merged[0].occurrences == 1 def test_evidence_from_before_the_fix_is_not_a_regression(self): """The window reaches back over sessions that predate the fix; they say nothing about it.""" applied = _filed(status="applied", applied={"at": "2026-09-01 10:00", "sha": "abc", "file": "SOUL.md"}) raw = _raw_finding(occurrences=9, sessions_affected=5, evidence=[{"session": "ws", "when": "2026-08-31"}]) merged = reflect_auto.merge_findings([applied], _parsed(_answer(raw)), TODAY) assert merged[0].regression_of is None assert merged[0].status == reflect_auto.STATUS_WATCH, "a finding about the past must not open" def test_evidence_from_after_the_fix_is_a_regression(self): applied = _filed(status="applied", applied={"at": "2026-09-01 10:00", "sha": "abc", "file": "SOUL.md"}) raw = _raw_finding(evidence=[{"session": "ws", "when": "2026-09-02 13:53"}]) merged = reflect_auto.merge_findings([applied], _parsed(_answer(raw)), TODAY) assert merged[0].regression_of == "f0001" assert merged[0].status == reflect_auto.STATUS_OPEN def test_rejected_pattern_never_opens_again(self): """Rejection is a decision, not a deferral — a repeat must not start nagging again.""" raw = _raw_finding(occurrences=9, sessions_affected=5) merged = reflect_auto.merge_findings([_filed(status="rejected")], _parsed(_answer(raw)), TODAY) assert merged[0].status == reflect_auto.STATUS_WATCH def test_rejection_survives_the_run_after_next(self): """The watch record filed after a rejection is newer — it must not supersede the decision.""" store = [_filed(status="rejected")] raw = _raw_finding(occurrences=9, sessions_affected=5) first = reflect_auto.merge_findings(store, _parsed(_answer(raw)), "2026-09-02") second = reflect_auto.merge_findings(reflect_auto.supersede(store, first), _parsed(_answer(raw)), "2026-09-03") assert second[0].status == reflect_auto.STATUS_WATCH def test_rejected_pattern_does_not_inherit_counts(self): merged = reflect_auto.merge_findings([_filed(status="rejected")], _parsed(_answer(_raw_finding())), TODAY) assert merged[0].occurrences == 1 def test_regression_flag_survives_a_second_sighting(self): """A pattern that came back after a fix stays flagged until it is dealt with.""" applied = [_filed(status="applied")] first = reflect_auto.merge_findings(applied, _parsed(_answer(_raw_finding())), "2026-09-02") second = reflect_auto.merge_findings( reflect_auto.supersede(applied, first), _parsed(_answer(_raw_finding())), "2026-09-03" ) assert second[0].regression_of == "f0001" def test_ids_are_unique_against_existing_records(self): merged = reflect_auto.merge_findings([_filed()], _parsed(_answer(_raw_finding(pattern="other-thing"))), TODAY) assert merged[0].id != "f0001" class TestLastSeen: """`created` is when a record was filed; only the evidence says when the pattern last occurred.""" def _merge(self, *evidence: dict) -> reflect_auto.Finding: raw = _raw_finding(evidence=list(evidence)) return reflect_auto.merge_findings([], _parsed(_answer(raw)), TODAY)[0] def test_the_newest_evidence_date_wins(self): finding = self._merge( {"session": "ws_a", "when": "2026-08-20"}, {"session": "ws_b", "when": "2026-08-31 13:53"}, {"session": "ws_c", "when": "2026-08-25"}, ) assert finding.last_seen == "2026-08-31", "mixed shapes, only the leading date counts" def test_undated_evidence_falls_back_to_created(self): finding = self._merge({"session": "ws_a", "excerpt": "no when at all"}) assert finding.last_seen == TODAY def test_non_date_when_is_ignored(self): finding = self._merge({"session": "ws_a", "when": "yesterday"}, {"session": "ws_b", "when": "turn 4"}) assert finding.last_seen == TODAY def test_folded_evidence_from_an_earlier_run_counts_too(self): existing = [_filed(evidence=[{"session": "ws_old", "when": "2026-09-05"}])] raw = _raw_finding(evidence=[{"session": "ws_new", "when": "2026-08-01"}]) merged = reflect_auto.merge_findings(existing, _parsed(_answer(raw)), TODAY) assert merged[0].last_seen == "2026-09-05" def test_last_seen_is_stored_for_the_review_to_sort_by(self): assert self._merge({"session": "ws_a", "when": "2026-08-20"}).to_json()["last_seen"] == "2026-08-20" class TestSupersede: def test_folded_record_is_replaced_not_duplicated(self): existing = [_filed()] merged = reflect_auto.merge_findings(existing, _parsed(_answer(_raw_finding())), TODAY) result = reflect_auto.supersede(existing, merged) assert len(result) == 1 assert result[0]["status"] == reflect_auto.STATUS_OPEN def test_unrelated_records_survive(self): existing = [_filed(id="f0001", pattern="other-thing", status="applied")] merged = reflect_auto.merge_findings(existing, _parsed(_answer(_raw_finding())), TODAY) result = reflect_auto.supersede(existing, merged) assert {record["pattern"] for record in result} == {"other-thing", "retry-without-diagnosis"} def test_applied_history_is_kept_for_regression_tracking(self): existing = [_filed(status="applied")] merged = reflect_auto.merge_findings(existing, _parsed(_answer(_raw_finding())), TODAY) result = reflect_auto.supersede(existing, merged) assert len(result) == 2 class TestRendering: def test_known_patterns_feed_the_id_vocabulary_back(self): text = reflect_auto._known_patterns([_filed()]) assert "`retry-without-diagnosis`" in text assert "[watch]" in text def test_known_patterns_handles_an_empty_store(self): assert "none yet" in reflect_auto._known_patterns([]) def test_report_marks_regressions(self): merged = reflect_auto.merge_findings([_filed(status="applied")], _parsed(_answer(_raw_finding())), TODAY) report = reflect_auto.render_report(merged, _stats(), None, WHEN) assert "REGRESSION" in report def test_report_renders_a_patch_as_a_diff(self): raw = _raw_finding(patch={"file": "a.md", "old_text": "try again", "new_text": "STOP"}) merged = reflect_auto.merge_findings([], _parsed(_answer(raw)), TODAY) report = reflect_auto.render_report(merged, _stats(), None, WHEN) assert "```diff" in report assert "- try again" in report assert "+ STOP" in report def test_report_shows_how_old_a_finding_is(self): """`occurrences` is cumulative, so the count alone cannot say whether this is still live.""" merged = reflect_auto.merge_findings([_filed()], _parsed(_answer(_raw_finding())), TODAY) report = reflect_auto.render_report(merged, _stats(), None, WHEN) assert "first seen 2026-09-01, last seen 2026-07-11" in report, "the evidence date, not the refile date" def test_a_brand_new_finding_reports_one_date(self): raw = _raw_finding(evidence=[{"session": "websocket_abc", "excerpt": "no date here"}]) merged = reflect_auto.merge_findings([], _parsed(_answer(raw)), TODAY) report = reflect_auto.render_report(merged, _stats(), None, WHEN) assert f"first seen {TODAY}, last seen {TODAY}" in report def test_report_marks_a_finding_the_window_no_longer_reaches(self): """Nothing in the window backs it any more, so no future run will refresh it either.""" raw = _raw_finding(occurrences=7, sessions_affected=4) merged = reflect_auto.merge_findings([], _parsed(_answer(raw)), TODAY) report = reflect_auto.render_report(merged, _stats(window_from="2026-08-18T00:00:00"), None, WHEN) assert "STALE" in report def test_a_finding_inside_the_window_is_not_stale(self): raw = _raw_finding(evidence=[{"session": "websocket_abc", "when": "2026-09-01", "excerpt": "x"}]) merged = reflect_auto.merge_findings([], _parsed(_answer(raw)), TODAY) report = reflect_auto.render_report(merged, _stats(window_from="2026-08-18T00:00:00"), None, WHEN) assert "STALE" not in report def test_empty_report_says_so(self): report = reflect_auto.render_report([], _stats(), None, WHEN) assert "Nothing to report" in report def test_report_shows_the_known_pattern_rate(self): """The one number proving the skill pays off — without it results/ is dead weight.""" report = reflect_auto.render_report([], _stats(repeat_per_100=3.2), 5.1, WHEN) assert "Known patterns: 3.2 occurrences / 100 sessions (previous run 5.1)." in report def test_first_run_has_no_trend_to_show(self): assert "(minule" not in reflect_auto.render_report([], _stats(), None, WHEN) class TestPromptContract: def test_prompt_forbids_writing(self): """The read-only instruction is one of the two guards on the analysis run.""" prompt = reflect_auto.build_prompt("digest", "none", 3) assert "Write nothing" in prompt def test_prompt_carries_digest_and_known_patterns(self): prompt = reflect_auto.build_prompt("SESSION-DIGEST-HERE", "PATTERN-LIST-HERE", 3) assert "SESSION-DIGEST-HERE" in prompt assert "PATTERN-LIST-HERE" in prompt def test_prompt_caps_file_reads(self): """Every extra tool iteration re-prefills the whole digest — the cap is what keeps a batch cheap.""" assert "Read at most 2 files" in reflect_auto.build_prompt("digest", "none", 3) class FakeBot: """Stands in for Nanobot: hands out canned replies and records what it was asked.""" def __init__(self, replies: list[Any]): self._replies = list(replies) self.calls: list[tuple[str, str]] = [] async def run(self, message: str, *, session_key: str): self.calls.append((message, session_key)) reply = self._replies.pop(0) if len(self._replies) > 1 else self._replies[0] if isinstance(reply, Exception): raise reply return reply def _reply(content: str = "", stop_reason: str | None = None, error: str | None = None) -> SimpleNamespace: return SimpleNamespace(content=content, stop_reason=stop_reason, error=error) def _llm_failure() -> SimpleNamespace: """What nanobot hands back when the provider gives up: the error text as the reply.""" return _reply("Error calling LLM: timed out after 300s", stop_reason="error", error="timed out after 300s") def _session_records(started: str, turns: int = 3) -> list[dict]: """A conversation long enough to clear reflect_distill.MIN_MESSAGES.""" records: list[dict] = [{"_type": "metadata", "key": "websocket:x", "created_at": started}] for i in range(turns): records.append({"role": "user", "content": f"dotaz {i}"}) records.append({"role": "assistant", "content": f"answer {i}"}) return records EARLIER = "2026-07-11T14:02:03" LATER = "2026-08-20T09:15:00" @pytest.fixture def workspace(tmp_path): """A workspace that is a real git repo with two analysable sessions, like the server's.""" sessions = tmp_path / "sessions" sessions.mkdir() for name, started in (("websocket_first", EARLIER), ("websocket_second", LATER)): records = _session_records(started) (sessions / f"{name}.jsonl").write_text( "\n".join(json.dumps(r, ensure_ascii=False) for r in records), encoding="utf-8" ) (tmp_path / "reflect").mkdir() for args in ( ["init", "-q"], ["config", "user.email", "test@example.com"], ["config", "user.name", "test"], ["add", "-A"], ["commit", "-q", "-m", "init"], ): subprocess.run(["git", *args], cwd=tmp_path, check=True, capture_output=True) return tmp_path def _args(**overrides): """One session per batch (budget_chars=1), so the two fixture sessions make two batches.""" args = { "dry_run": False, "all": False, "budget_chars": 1, "max_batches": 0, "deadline_minutes": 20, "window_days": 0, } args.update(overrides) return SimpleNamespace(**args) def _state(workspace: Path) -> dict: return json.loads((workspace / "reflect" / "state.json").read_text(encoding="utf-8")) def _findings(workspace: Path) -> list[dict]: path = workspace / "reflect" / "findings.jsonl" return [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines() if line.strip()] def _seed_findings(workspace: Path, *records: dict) -> None: path = workspace / "reflect" / "findings.jsonl" path.write_text("\n".join(json.dumps(r, ensure_ascii=False) for r in records) + "\n", encoding="utf-8") def _run(workspace: Path, bot, monkeypatch, **arg_overrides) -> tuple[str, int]: monkeypatch.setattr(reflect_auto, "_open_bot", lambda preset: bot) return asyncio.run(reflect_auto._run(workspace, _args(**arg_overrides), WHEN)) class TestRunPersistence: """The 2026-09-02 failure: a 30min timeout threw away two finished batches and kept the cursor.""" def test_every_batch_is_filed_as_it_finishes(self, workspace, monkeypatch): bot = FakeBot( [ _reply(_answer(_raw_finding())), _reply(_answer(_raw_finding(pattern="tool-call-leaked-as-text"))), ] ) _run(workspace, bot, monkeypatch) assert {record["pattern"] for record in _findings(workspace)} == { "retry-without-diagnosis", "tool-call-leaked-as-text", } assert _state(workspace)["cursor"] == LATER assert [run["batches"] for run in _state(workspace)["runs"]] == [2], "one run entry, updated in place" def test_the_same_pattern_in_two_batches_folds_into_one_record(self, workspace, monkeypatch): """Per-batch filing must not turn cross-batch dedup into duplicate records.""" bot = FakeBot([_reply(_answer(_raw_finding()))]) _run(workspace, bot, monkeypatch) records = _findings(workspace) assert len(records) == 1 assert records[0]["occurrences"] == 2 assert records[0]["status"] == "open", "the second sighting is what promotes a watched pattern" def test_the_report_counts_a_repeated_pattern_once(self, workspace, monkeypatch): """The report and the Telegram line have to say what the store holds, not how often it was refiled.""" bot = FakeBot([_reply(_answer(_raw_finding()))]) message, _ = _run(workspace, bot, monkeypatch) report = (workspace / "results" / f"{WHEN:%Y-%m-%d}_reflect.md").read_text(encoding="utf-8") assert len([line for line in report.splitlines() if line.startswith("## ")]) == 1 assert "Findings: 1 (1 to review, 0 watched)" in report assert _state(workspace)["runs"][-1]["open"] == 1 assert "1 findings to review" in message def test_a_batch_that_dies_does_not_take_the_finished_one_with_it(self, workspace, monkeypatch): bot = FakeBot([_reply(_answer(_raw_finding())), RuntimeError("provider down")]) with pytest.raises(RuntimeError): _run(workspace, bot, monkeypatch) assert len(_findings(workspace)) == 1 assert _state(workspace)["cursor"] == EARLIER assert _state(workspace)["runs"][-1]["batches"] == 1 def test_the_cursor_never_moves_backwards(self, workspace, monkeypatch): """An `--all` walk starts at the oldest session and must not undo the nightly progress.""" reflect_auto._save_state(workspace, {"cursor": "2026-12-31T00:00:00", "runs": []}) bot = FakeBot([_reply(_answer(_raw_finding()))]) _run(workspace, bot, monkeypatch, all=True) assert _state(workspace)["cursor"] == "2026-12-31T00:00:00" def test_the_deadline_stops_the_run_and_says_so(self, workspace, monkeypatch): _seed_findings(workspace, _filed()) bot = FakeBot([_reply(_answer(_raw_finding()))]) message, _ = _run(workspace, bot, monkeypatch, deadline_minutes=0) assert len(bot.calls) == 1 assert _state(workspace)["cursor"] == EARLIER assert "Analysed 1/2 batches." in message def test_a_finished_run_reports_no_partial_progress(self, workspace, monkeypatch): _seed_findings(workspace, _filed()) bot = FakeBot([_reply(_answer(_raw_finding()))]) message, _ = _run(workspace, bot, monkeypatch) assert "batches" not in message def test_an_unfinished_run_speaks_up_even_with_no_findings(self, workspace, monkeypatch): """Silence here is how a starved cursor went unnoticed for three nights.""" bot = FakeBot([_reply(_answer(_raw_finding()))]) message, _ = _run(workspace, bot, monkeypatch, deadline_minutes=0) assert "1 batches left" in message def test_a_finished_run_with_no_findings_stays_quiet(self, workspace, monkeypatch): bot = FakeBot([_reply(_answer())]) message, _ = _run(workspace, bot, monkeypatch) assert message == "" class TestWindow: """Findings must describe recent behaviour; a months-old backlog must not starve the run.""" def test_the_window_skips_everything_older(self, workspace, monkeypatch): bot = FakeBot([_reply(_answer(_raw_finding()))]) _run(workspace, bot, monkeypatch, window_days=21) assert len(bot.calls) == 1, "only the session inside the window is analysable" assert "websocket_second" in bot.calls[0][0] assert "websocket_first" not in bot.calls[0][0] assert _state(workspace)["cursor"] == LATER, "the skipped backlog is skipped for good" def test_the_cursor_still_wins_over_the_window(self, workspace, monkeypatch): """The window is a ceiling, not a rewind — nothing already counted may be re-read.""" reflect_auto._save_state(workspace, {"cursor": LATER, "runs": []}) bot = FakeBot([_reply(_answer(_raw_finding()))]) message, sessions = _run(workspace, bot, monkeypatch, window_days=365) assert bot.calls == [] assert (message, sessions) == ("", 0) def test_the_report_says_what_the_window_covered(self, workspace, monkeypatch): bot = FakeBot([_reply(_answer(_raw_finding()))]) _run(workspace, bot, monkeypatch, window_days=21) report = (workspace / "results" / f"{WHEN:%Y-%m-%d}_reflect.md").read_text(encoding="utf-8") assert "Window: from 2026-08-18, batches 1/1." in report def test_a_live_all_run_is_refused(self, workspace): """`--all` re-reads counted sessions and merge_findings would sum their occurrences in.""" with pytest.raises(SystemExit): reflect_auto.main(["--all", "--workspace", str(workspace)]) def test_a_dry_all_run_is_allowed(self, workspace): assert reflect_auto.main(["--all", "--dry-run", "--workspace", str(workspace)]) == 0 assert list((workspace / "tmp").glob("reflect-batch.*.md")) class TestLlmErrorHandling: """A dead provider is not a malformed answer; conflating the two burned a validator attempt.""" def test_a_provider_failure_re_asks_the_original_question(self, workspace, monkeypatch): bot = FakeBot([_llm_failure(), _reply(_answer(_raw_finding()))]) parsed = asyncio.run(reflect_auto._resolve_findings(bot, "reflect:test-0", "PROMPT", workspace, 4)) assert len(parsed) == 1 first, second = bot.calls assert second[0] == "PROMPT", "the retry must re-ask, not blame the model for the error text" assert second[1] != first[1], "the failed turn poisoned that session; retry needs a fresh key" def test_a_dead_provider_is_reported_as_such(self, workspace, monkeypatch): bot = FakeBot([_llm_failure()]) with pytest.raises(reflect_auto.ReflectError, match="model unavailable"): asyncio.run(reflect_auto._resolve_findings(bot, "reflect:test-0", "PROMPT", workspace, 4)) assert len(bot.calls) == reflect_auto.MAX_LLM_ERROR_RETRIES + 1 def test_a_malformed_answer_still_gets_validator_feedback(self, workspace, monkeypatch): bot = FakeBot([_reply("no json here"), _reply(_answer(_raw_finding()))]) parsed = asyncio.run(reflect_auto._resolve_findings(bot, "reflect:test-0", "PROMPT", workspace, 4)) assert len(parsed) == 1 assert "rejected by the validator" in bot.calls[1][0]