Implement corpus indexing and MCP search
This commit is contained in:
@@ -0,0 +1,106 @@
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from corpus_mcp.database import (
|
||||
CorpusError,
|
||||
build_database,
|
||||
read_document_excerpt,
|
||||
search_database,
|
||||
)
|
||||
|
||||
|
||||
def test_plain_search_returns_highlighted_excerpt(corpus: tuple[Path, Path]) -> None:
|
||||
_, database = corpus
|
||||
|
||||
result = search_database(database, "quick search")
|
||||
|
||||
assert result["syntax"] == "plain"
|
||||
assert [match["path"] for match in result["matches"]] == ["guide.md"]
|
||||
excerpt = result["matches"][0]["excerpts"][0]
|
||||
assert "<match>quick</match>" in excerpt["text"]
|
||||
assert "<match>search</match>" in excerpt["text"]
|
||||
|
||||
|
||||
def test_fts5_search_supports_phrases_and_rejects_invalid_syntax(
|
||||
corpus: tuple[Path, Path],
|
||||
) -> None:
|
||||
_, database = corpus
|
||||
|
||||
result = search_database(database, '"full text"', syntax="fts5")
|
||||
assert result["matches"][0]["path"] == "guide.md"
|
||||
|
||||
with pytest.raises(CorpusError, match="Invalid FTS5 query"):
|
||||
search_database(database, '"unterminated', syntax="fts5")
|
||||
|
||||
|
||||
def test_search_returns_separate_excerpts_from_a_large_document(tmp_path: Path) -> None:
|
||||
source = tmp_path / "source"
|
||||
source.mkdir()
|
||||
content = "needle first\n\n" + ("filler " * 800) + "\n\nneedle second"
|
||||
(source / "large.md").write_text(content, encoding="utf-8")
|
||||
database = tmp_path / "corpus.sqlite"
|
||||
build_database(source, database)
|
||||
|
||||
result = search_database(database, "needle", excerpts_per_document=2)
|
||||
|
||||
excerpts = result["matches"][0]["excerpts"]
|
||||
assert len(excerpts) == 2
|
||||
assert excerpts[0]["chunk_end_char"] <= len(content)
|
||||
assert excerpts[1]["chunk_end_char"] <= len(content)
|
||||
|
||||
|
||||
def test_large_document_does_not_hide_other_matching_documents(
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
source = tmp_path / "source"
|
||||
source.mkdir()
|
||||
(source / "large.txt").write_text(("common term " * 200_000), encoding="utf-8")
|
||||
(source / "small.txt").write_text("common term", encoding="utf-8")
|
||||
database = tmp_path / "corpus.sqlite"
|
||||
build_database(source, database)
|
||||
|
||||
result = search_database(database, "common", limit=2)
|
||||
|
||||
assert {match["path"] for match in result["matches"]} == {
|
||||
"large.txt",
|
||||
"small.txt",
|
||||
}
|
||||
|
||||
|
||||
def test_read_document_excerpt_is_bounded_and_navigable(
|
||||
corpus: tuple[Path, Path],
|
||||
) -> None:
|
||||
_, database = corpus
|
||||
|
||||
first = read_document_excerpt(database, "guide.md", max_chars=12)
|
||||
assert first["next_offset"] is not None
|
||||
second = read_document_excerpt(
|
||||
database, "guide.md", offset=first["next_offset"], max_chars=12
|
||||
)
|
||||
|
||||
assert len(first["text"]) == 12
|
||||
assert first["previous_offset"] is None
|
||||
assert second["start_char"] == 12
|
||||
assert second["previous_offset"] == 0
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("kwargs", "message"),
|
||||
[
|
||||
({"offset": -1}, "offset must not be negative"),
|
||||
({"max_chars": 20_001}, "max_chars must be between"),
|
||||
],
|
||||
)
|
||||
def test_read_document_excerpt_validates_bounds(
|
||||
corpus: tuple[Path, Path], kwargs: dict, message: str
|
||||
) -> None:
|
||||
_, database = corpus
|
||||
with pytest.raises(CorpusError, match=message):
|
||||
read_document_excerpt(database, "guide.md", **kwargs)
|
||||
|
||||
|
||||
def test_search_validates_empty_query(corpus: tuple[Path, Path]) -> None:
|
||||
_, database = corpus
|
||||
with pytest.raises(CorpusError, match="must not be empty"):
|
||||
search_database(database, " ")
|
||||
Reference in New Issue
Block a user