107 lines
3.3 KiB
Python
107 lines
3.3 KiB
Python
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
from corpus_mcp.database import (
|
|
CorpusError,
|
|
build_database,
|
|
read_document_excerpt,
|
|
search_database,
|
|
)
|
|
|
|
|
|
def test_plain_search_returns_highlighted_excerpt(corpus: tuple[Path, Path]) -> None:
|
|
_, database = corpus
|
|
|
|
result = search_database(database, "quick search")
|
|
|
|
assert result["syntax"] == "plain"
|
|
assert [match["path"] for match in result["matches"]] == ["guide.md"]
|
|
excerpt = result["matches"][0]["excerpts"][0]
|
|
assert "<match>quick</match>" in excerpt["text"]
|
|
assert "<match>search</match>" in excerpt["text"]
|
|
|
|
|
|
def test_fts5_search_supports_phrases_and_rejects_invalid_syntax(
|
|
corpus: tuple[Path, Path],
|
|
) -> None:
|
|
_, database = corpus
|
|
|
|
result = search_database(database, '"full text"', syntax="fts5")
|
|
assert result["matches"][0]["path"] == "guide.md"
|
|
|
|
with pytest.raises(CorpusError, match="Invalid FTS5 query"):
|
|
search_database(database, '"unterminated', syntax="fts5")
|
|
|
|
|
|
def test_search_returns_separate_excerpts_from_a_large_document(tmp_path: Path) -> None:
|
|
source = tmp_path / "source"
|
|
source.mkdir()
|
|
content = "needle first\n\n" + ("filler " * 800) + "\n\nneedle second"
|
|
(source / "large.md").write_text(content, encoding="utf-8")
|
|
database = tmp_path / "corpus.sqlite"
|
|
build_database(source, database)
|
|
|
|
result = search_database(database, "needle", excerpts_per_document=2)
|
|
|
|
excerpts = result["matches"][0]["excerpts"]
|
|
assert len(excerpts) == 2
|
|
assert excerpts[0]["chunk_end_char"] <= len(content)
|
|
assert excerpts[1]["chunk_end_char"] <= len(content)
|
|
|
|
|
|
def test_large_document_does_not_hide_other_matching_documents(
|
|
tmp_path: Path,
|
|
) -> None:
|
|
source = tmp_path / "source"
|
|
source.mkdir()
|
|
(source / "large.txt").write_text(("common term " * 200_000), encoding="utf-8")
|
|
(source / "small.txt").write_text("common term", encoding="utf-8")
|
|
database = tmp_path / "corpus.sqlite"
|
|
build_database(source, database)
|
|
|
|
result = search_database(database, "common", limit=2)
|
|
|
|
assert {match["path"] for match in result["matches"]} == {
|
|
"large.txt",
|
|
"small.txt",
|
|
}
|
|
|
|
|
|
def test_read_document_excerpt_is_bounded_and_navigable(
|
|
corpus: tuple[Path, Path],
|
|
) -> None:
|
|
_, database = corpus
|
|
|
|
first = read_document_excerpt(database, "guide.md", max_chars=12)
|
|
assert first["next_offset"] is not None
|
|
second = read_document_excerpt(
|
|
database, "guide.md", offset=first["next_offset"], max_chars=12
|
|
)
|
|
|
|
assert len(first["text"]) == 12
|
|
assert first["previous_offset"] is None
|
|
assert second["start_char"] == 12
|
|
assert second["previous_offset"] == 0
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("kwargs", "message"),
|
|
[
|
|
({"offset": -1}, "offset must not be negative"),
|
|
({"max_chars": 20_001}, "max_chars must be between"),
|
|
],
|
|
)
|
|
def test_read_document_excerpt_validates_bounds(
|
|
corpus: tuple[Path, Path], kwargs: dict, message: str
|
|
) -> None:
|
|
_, database = corpus
|
|
with pytest.raises(CorpusError, match=message):
|
|
read_document_excerpt(database, "guide.md", **kwargs)
|
|
|
|
|
|
def test_search_validates_empty_query(corpus: tuple[Path, Path]) -> None:
|
|
_, database = corpus
|
|
with pytest.raises(CorpusError, match="must not be empty"):
|
|
search_database(database, " ")
|