Files
nanobot-runtime/skills/reflect/tests/test_reflect_distill.py
2026-09-02 10:36:37 +02:00

242 lines
11 KiB
Python

"""Tests for reflect_distill.py — mechanical distillation of session logs."""
import json
import sys
from pathlib import Path
import pytest
sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "scripts"))
import reflect_distill
def _write_session(directory: Path, name: str, records: list[dict]) -> Path:
path = directory / f"{name}.jsonl"
path.write_text("\n".join(json.dumps(r, ensure_ascii=False) for r in records), encoding="utf-8")
return path
def _conversation(turns: int = 3) -> list[dict]:
"""A session long enough to clear MIN_MESSAGES."""
records: list[dict] = [{"_type": "metadata", "key": "websocket:x", "created_at": "2026-07-11T14:02:03"}]
for i in range(turns):
records.append({"role": "user", "content": f"dotaz {i}"})
records.append({"role": "assistant", "content": f"answer {i}"})
return records
class TestNoiseFilter:
@pytest.mark.parametrize(
"name",
[
"reflect_20260908-033000",
"compact-memory-auto_20260715-054147",
"detach_abc",
"dream_xyz",
"cron_e81cda77",
"cli_kimi-ollama-test",
"wiki-compile",
"note-compile",
],
)
def test_machinery_prefixes_are_noise(self, name):
assert reflect_distill.is_noise(name)
@pytest.mark.parametrize("name", ["wiki-capture-test2", "test-note-search-zzz"])
def test_prefix_variants_are_noise(self, name):
"""Exact matching used to let `…test2` through; the filter matches prefixes."""
assert reflect_distill.is_noise(name)
@pytest.mark.parametrize("name", ["websocket_e5a6aac3-2c0a", "telegram_8826147089"])
def test_real_conversations_are_kept(self, name):
assert not reflect_distill.is_noise(name)
def test_base64_session_names_are_decoded(self):
"""Older sessions are stored base64-encoded; `ZHJlYW06…` is `dream:…`."""
assert reflect_distill.is_noise("ZHJlYW06MjAyNjA4MzEtMTA0NjM2")
assert not reflect_distill.is_noise("d2Vic29ja2V0OjUyZDBmMzM4")
def test_reflect_excludes_its_own_sessions(self):
"""Without this the skill would analyse its own runs on the next pass."""
assert reflect_distill.is_noise("reflect_20260908-033000")
class TestDistillSession:
def test_short_sessions_are_skipped(self, tmp_path):
path = _write_session(tmp_path, "websocket_short", [{"role": "user", "content": "ahoj"}])
assert reflect_distill.distill_session(path) is None
def test_noise_sessions_are_skipped(self, tmp_path):
path = _write_session(tmp_path, "dream_nightly", _conversation())
assert reflect_distill.distill_session(path) is None
def test_header_carries_key_time_and_count(self, tmp_path):
path = _write_session(tmp_path, "websocket_abc", _conversation())
digest = reflect_distill.distill_session(path)
assert digest.text.startswith(reflect_distill.SESSION_MARKER)
assert "websocket_abc" in digest.text
assert "2026-07-11 14:02" in digest.text
assert digest.message_count == 6
def test_header_is_not_a_markdown_heading(self, tmp_path):
"""Assistant prose is full of `###`, so a heading would not read as a boundary."""
records = _conversation()
records.append({"role": "assistant", "content": "### Summary\ndone"})
path = _write_session(tmp_path, "websocket_abc", records)
digest = reflect_distill.distill_session(path)
assert not digest.text.startswith("#")
assert digest.text.count(reflect_distill.SESSION_MARKER) == 1
def test_user_and_assistant_prose_is_kept_whole(self, tmp_path):
prose = "this is a long answer " * 40
records = _conversation()
records.append({"role": "assistant", "content": prose})
path = _write_session(tmp_path, "websocket_abc", records)
assert prose.strip() in reflect_distill.distill_session(path).text
def test_slash_commands_are_dropped(self, tmp_path):
records = _conversation()
records.append({"role": "user", "content": "/model", "_command": True})
path = _write_session(tmp_path, "websocket_abc", records)
assert "/model" not in reflect_distill.distill_session(path).text
def test_decoded_key_is_used_as_name(self, tmp_path):
path = _write_session(tmp_path, "d2Vic29ja2V0OjUyZDBmMzM4", _conversation())
assert reflect_distill.distill_session(path).name == "websocket:52d0f338"
class TestToolRendering:
def _digest_with_tool(self, tmp_path, call, result):
records = _conversation()
records.append({"role": "assistant", "content": "", "tool_calls": [call]})
records.append(result)
path = _write_session(tmp_path, "websocket_abc", records)
return reflect_distill.distill_session(path).text
def test_tool_result_body_is_replaced_by_metadata(self, tmp_path):
"""Tool results are 89% of the corpus and carry almost no diagnostic value."""
body = "x" * 5000
text = self._digest_with_tool(
tmp_path,
{"id": "c1", "function": {"name": "read_file", "arguments": '{"path": "/tmp/a.md"}'}},
{"role": "tool", "tool_call_id": "c1", "name": "read_file", "content": body},
)
assert body not in text
assert "read_file: ok, 5.0 kB" in text
def test_errors_are_flagged_with_their_message(self, tmp_path):
text = self._digest_with_tool(
tmp_path,
{"id": "c1", "function": {"name": "web_fetch", "arguments": '{"url": "https://x.dev"}'}},
{"role": "tool", "tool_call_id": "c1", "name": "web_fetch", "content": "Error: 403 Forbidden"},
)
assert "ERROR Error: 403 Forbidden" in text
def test_workspace_path_prefix_is_stripped(self, tmp_path):
text = self._digest_with_tool(
tmp_path,
{
"id": "c1",
"function": {
"name": "read_file",
"arguments": '{"path": "/home/nanobot/.nanobot/workspace/skills/note/SKILL.md"}',
},
},
{"role": "tool", "tool_call_id": "c1", "name": "read_file", "content": "ok"},
)
assert "path=skills/note/SKILL.md" in text
def test_malformed_arguments_still_render(self, tmp_path):
"""A truncated argument blob is itself a signal worth seeing, not a crash."""
text = self._digest_with_tool(
tmp_path,
{"id": "c1", "function": {"name": "exec", "arguments": '{"command": "ls'}},
{"role": "tool", "tool_call_id": "c1", "name": "exec", "content": "ok"},
)
assert "exec(" in text
def test_long_urls_keep_their_distinguishing_tail(self, tmp_path):
"""Front-truncation made different fetches look identical and hid real retries."""
base = "https://raw.githubusercontent.com/some-owner/some-repo/main/packages/core/src/"
first = self._digest_with_tool(
tmp_path,
{"id": "c1", "function": {"name": "web_fetch", "arguments": json.dumps({"url": base + "alpha.ts"})}},
{"role": "tool", "tool_call_id": "c1", "name": "web_fetch", "content": "ok"},
)
second = self._digest_with_tool(
tmp_path,
{"id": "c1", "function": {"name": "web_fetch", "arguments": json.dumps({"url": base + "omega.ts"})}},
{"role": "tool", "tool_call_id": "c1", "name": "web_fetch", "content": "ok"},
)
assert "alpha.ts" in first
assert "omega.ts" in second
assert first != second
def test_repeated_identical_calls_render_identically(self, tmp_path):
"""The LLM detects retry loops by seeing the same line twice — so it must match."""
call = {"id": "c1", "function": {"name": "web_fetch", "arguments": '{"url": "https://x.dev/a"}'}}
records = _conversation()
for _ in range(2):
records.append({"role": "assistant", "content": "", "tool_calls": [call]})
records.append({"role": "tool", "tool_call_id": "c1", "name": "web_fetch", "content": "Error: 403"})
path = _write_session(tmp_path, "websocket_abc", records)
lines = [line for line in reflect_distill.distill_session(path).text.splitlines() if "→ web_fetch" in line]
assert len(lines) == 2
assert lines[0] == lines[1]
def test_reasoning_is_shortened_not_dropped(self, tmp_path):
records = _conversation()
records.append({"role": "assistant", "content": "ok", "reasoning_content": "reasoning " * 200})
path = _write_session(tmp_path, "websocket_abc", records)
text = reflect_distill.distill_session(path).text
assert "~ reasoning" in text
assert len(text) < 2000
class TestBatching:
def test_batches_respect_the_character_budget(self, tmp_path):
for i in range(6):
_write_session(tmp_path, f"websocket_{i}", _conversation(turns=8))
paths = reflect_distill.collect_sessions(tmp_path)
batches = list(reflect_distill.iter_batches(paths, budget_chars=400))
assert len(batches) > 1
assert all(batch for batch in batches)
def test_oversized_session_is_not_split_or_dropped(self, tmp_path):
"""Truncating a huge session would hide exactly the runaway loops worth finding."""
records = _conversation()
records.append({"role": "assistant", "content": "y" * 5000})
_write_session(tmp_path, "websocket_big", records)
batches = list(reflect_distill.iter_batches(reflect_distill.collect_sessions(tmp_path), budget_chars=100))
assert len(batches) == 1
assert len(batches[0][0]) > 5000
def test_collect_sessions_skips_noise(self, tmp_path):
_write_session(tmp_path, "websocket_keep", _conversation())
_write_session(tmp_path, "dream_drop", _conversation())
names = [p.stem for p in reflect_distill.collect_sessions(tmp_path)]
assert names == ["websocket_keep"]
def test_cursor_written_by_a_run_excludes_those_sessions_next_time(self, tmp_path):
"""The round trip reflect_auto actually makes: digest.started becomes the next `since`.
A display-formatted cursor (`2026-07-11 14:02`) compares wrong against the raw ISO in
the log, because `T` sorts above a space — every session of that day came back.
"""
_write_session(tmp_path, "websocket_done", _conversation())
digests = [
d
for batch in reflect_distill.iter_batches(reflect_distill.collect_sessions(tmp_path), 10**6)
for d in batch
]
cursor = max(digest.started for digest in digests)
assert reflect_distill.collect_sessions(tmp_path, since=cursor) == []
def test_since_excludes_already_processed_sessions(self, tmp_path):
_write_session(tmp_path, "websocket_old", _conversation())
new = _conversation()
new[0]["created_at"] = "2026-08-20T09:00:00"
_write_session(tmp_path, "websocket_new", new)
names = [p.stem for p in reflect_distill.collect_sessions(tmp_path, since="2026-08-01T00:00:00")]
assert names == ["websocket_new"]