feat: validate encoder identity and expose retrieval modes

This commit is contained in:
emil28092005
2026-09-16 04:26:06 +03:00
parent abffddaa5e
commit 424aba1d5f
11 changed files with 74 additions and 7 deletions
+15
View File
@@ -119,3 +119,18 @@ def test_training_updates_weights_and_can_resume(tiny_model, tmp_path):
after.save(output / "last")
with pytest.raises(ValueError, match="weights and optimizer state"):
train(data, output, config, "cpu", output / "last")
def test_fingerprint_detects_tokenizer_change(tiny_model):
original = Encoder(str(tiny_model), max_length=32, query_length=16)
path = tiny_model / "tokenizer.json"
config = json.loads(path.read_text())
vocab = config["model"]["vocab"]
vocab["read"], vocab["write"] = vocab["write"], vocab["read"]
path.write_text(json.dumps(config))
changed = Encoder(str(tiny_model), max_length=32, query_length=16)
assert changed.fingerprint != original.fingerprint
assert (
changed.tokenize(["read file"])["input_ids"].tolist()
!= original.tokenize(["read file"])["input_ids"].tolist()
)
+17
View File
@@ -263,3 +263,20 @@ def test_source_that_grows_after_indexing_is_bounded(repository, tmp_path):
(repository / "numbers.py").write_text("x" * 1_000_001)
with pytest.raises(ValueError, match="file-size limit"):
scout.read(symbol.id)
def test_long_function_tail_has_a_searchable_fragment(repository, tmp_path):
lines = ["def long_function():"] + [f" value_{i} = {i}" for i in range(90)]
lines.append(" return unique_tail_marker")
(repository / "long.py").write_text("\n".join(lines) + "\n")
path = tmp_path / "index.sqlite"
build_index(repository, path)
scout = Scout(Index(path))
result = scout.search("unique_tail_marker", mode="lexical", top_k=1)
hit = result["results"][0]
assert hit["kind"] == "fragment"
assert "unique_tail_marker" in hit["content"]
assert hit["start_line"] > 48
parent = scout.read(hit["parent_id"])
assert parent["name"] == "long_function"
assert parent["start_line"] == 1