mirror of
https://github.com/basicmachines-co/basic-memory
synced 2026-06-21 13:47:35 +00:00
fix(core): repair FTS half of hybrid search for natural-language queries
Hybrid search was silently running vector-only on natural-language queries — the FTS branch contributed zero candidates. Two causes in the SQLite (and parallel Postgres) FTS query preparation: 1. Sentence punctuation forced phrase matching. A question like "When did Melanie paint a sunrise?" reached FTS5 as the exact phrase '"When did Melanie paint a sunrise?"*', which matches no document. The FTS5 tokenizer ignores this punctuation in the index, so stripping it from word edges loses nothing — but leaving it disabled the entire FTS contribution. _prepare_single_term now strips ?!.,;: from word edges of multi-word queries (interior characters — hyphens, slashes in permalinks/paths — untouched). 2. No relaxation when strict all-terms-AND matched nothing. Questions rarely have every word in one document, so even after (1) the strict AND returned zero rows. The hybrid path now retries once with an OR-joined, stopword-filtered, content-term query when the strict query is empty. bm25/ts_rank still rank multi-term matches first, and fusion with the vector branch keeps relaxed lexical candidates from dominating precision. The relaxation is gated behind a new allow_relaxed=False parameter on SearchRepositoryBase.search; only _search_hybrid opts in. Strict FTS behavior (search_type=text, title, permalink, link resolution) is unchanged — the service layer keeps its own conservative fallback. No config flag, default-safe. Discovered via the benchmark harness: two different fusion algorithms produced byte-identical rankings across 1,986 queries (impossible with two live sources), and instrumentation confirmed fts=0 on 40/40 sampled LoCoMo queries. Benchmark impact (corrected LoCoMo, 1,986 queries, same index, retrieval metrics — every category improves, no regression): recall@5 0.745 -> 0.823 (+7.9) MRR 0.618 -> 0.718 (+10.0) headline r5 0.734 -> 0.801, MRR 0.621 -> 0.706 Largest gains on open_domain (+0.10 r5) and adversarial (+0.12 r5); smallest on temporal (+0.003 r5 / +0.02 MRR). Tests: punctuation no longer phrase-quotes; relaxation builds the expected OR query and respects boolean/quoted/short-query intent; the hybrid opt-in surfaces a partial-overlap document while the default strict path still returns empty. Parallel coverage for Postgres. Full SQLite unit suite green (2968 passed); ty + ruff clean. Signed-off-by: Drew Cain <groksrc@gmail.com>
This commit is contained in:
@@ -18,6 +18,7 @@ from basic_memory.repository.search_index_row import SearchIndexRow
|
||||
from basic_memory.repository.search_repository_base import (
|
||||
SearchRepositoryBase,
|
||||
VectorChunkState,
|
||||
relaxed_query_words,
|
||||
)
|
||||
from basic_memory.repository.metadata_filters import parse_metadata_filters
|
||||
from basic_memory.repository.semantic_errors import SemanticDependenciesMissingError
|
||||
@@ -176,6 +177,14 @@ class PostgresSearchRepository(SearchRepositoryBase):
|
||||
# For non-Boolean queries, prepare single term
|
||||
return self._prepare_single_term(term, is_prefix)
|
||||
|
||||
@staticmethod
|
||||
def _relaxed_tsquery_text(search_text: Optional[str]) -> Optional[str]:
|
||||
"""OR-relaxed tsquery expression for a failed strict query, or None."""
|
||||
words = relaxed_query_words(search_text)
|
||||
if not words:
|
||||
return None
|
||||
return " | ".join(f"{word}:*" for word in words)
|
||||
|
||||
def _prepare_boolean_query(self, query: str) -> str:
|
||||
"""Convert Boolean query to tsquery format.
|
||||
|
||||
@@ -234,6 +243,14 @@ class PostgresSearchRepository(SearchRepositoryBase):
|
||||
for char in special_chars:
|
||||
cleaned_term = cleaned_term.replace(char, " ")
|
||||
|
||||
# Sentence punctuation carries no lexical signal in tsquery either;
|
||||
# strip it from word edges so question-form queries produce clean
|
||||
# lexemes (parity with the SQLite FTS5 preparation).
|
||||
if " " in cleaned_term:
|
||||
cleaned_term = " ".join(
|
||||
word.strip("?!.,;") for word in cleaned_term.split() if word.strip("?!.,;")
|
||||
)
|
||||
|
||||
# Handle multi-word queries
|
||||
if " " in cleaned_term:
|
||||
words = [w for w in cleaned_term.split() if w.strip()]
|
||||
@@ -908,6 +925,7 @@ class PostgresSearchRepository(SearchRepositoryBase):
|
||||
min_similarity: Optional[float] = None,
|
||||
limit: int = 10,
|
||||
offset: int = 0,
|
||||
allow_relaxed: bool = False,
|
||||
) -> List[SearchIndexRow]:
|
||||
"""Search across all indexed content using PostgreSQL tsvector."""
|
||||
# --- Dispatch vector / hybrid modes (shared logic) ---
|
||||
@@ -982,6 +1000,20 @@ class PostgresSearchRepository(SearchRepositoryBase):
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
result = await session.execute(text(sql), params)
|
||||
rows = result.fetchall()
|
||||
# Trigger: multi-word natural-language query matched nothing
|
||||
# under the default all-terms-AND tsquery semantics.
|
||||
# Why: questions rarely have every word in one document;
|
||||
# without relaxation the FTS half of hybrid search contributes
|
||||
# zero candidates (parity with the SQLite path).
|
||||
# Outcome: one retry with OR-joined prefix lexemes; ts_rank
|
||||
# still ranks multi-term matches first.
|
||||
relaxed = (
|
||||
self._relaxed_tsquery_text(search_text) if allow_relaxed and not rows else None
|
||||
)
|
||||
if relaxed and params.get("text"):
|
||||
params["text"] = relaxed
|
||||
result = await session.execute(text(sql), params)
|
||||
rows = result.fetchall()
|
||||
except Exception as e:
|
||||
if self._is_tsquery_syntax_error(e):
|
||||
logger.warning(f"tsquery syntax error for search term: {search_text}, error: {e}")
|
||||
|
||||
@@ -40,6 +40,36 @@ BULLET_PATTERN = re.compile(r"^[\-\*]\s+")
|
||||
OVERSIZED_ENTITY_VECTOR_SHARD_SIZE = 256
|
||||
_SQLITE_MAX_PREPARE_WINDOW = 8
|
||||
|
||||
# Interrogative/function words contribute lexical noise when a strict
|
||||
# full-text query is relaxed: "when OR did OR a" matches loud wrong documents
|
||||
# that displace genuine results from the ranking window.
|
||||
RELAXATION_STOPWORDS = frozenset(
|
||||
"a an and are as at be but by did do does for from had has have how i in is it of on "
|
||||
"or that the their they this to was we were what when where which who whom whose why "
|
||||
"will with you your".split()
|
||||
)
|
||||
|
||||
|
||||
def relaxed_query_words(search_text: Optional[str]) -> Optional[list[str]]:
|
||||
"""Content-bearing words for OR-relaxing a strict full-text query.
|
||||
|
||||
Returns None when relaxation must not apply: empty input, quoted phrases,
|
||||
or explicit boolean queries (user intent is not second-guessed).
|
||||
"""
|
||||
if not search_text:
|
||||
return None
|
||||
stripped = search_text.strip()
|
||||
if '"' in stripped or any(op in f" {stripped} " for op in (" AND ", " OR ", " NOT ")):
|
||||
return None
|
||||
words = [word.strip("?!.,;:") for word in stripped.split()]
|
||||
words = [
|
||||
word
|
||||
for word in words
|
||||
if word and word.isalnum() and word.lower() not in RELAXATION_STOPWORDS
|
||||
]
|
||||
return words or None
|
||||
|
||||
|
||||
# Entity, observation, and relation rows in search_index carry ids from independent
|
||||
# auto-increment sequences, so a bare id is ambiguous across row types. Every map in
|
||||
# the vector/hybrid retrieval path must key rows by (type, id) to avoid collisions.
|
||||
@@ -229,6 +259,7 @@ class SearchRepositoryBase(ABC):
|
||||
min_similarity: Optional[float] = None,
|
||||
limit: int = 10,
|
||||
offset: int = 0,
|
||||
allow_relaxed: bool = False,
|
||||
) -> List[SearchIndexRow]:
|
||||
"""Search across all indexed content.
|
||||
|
||||
@@ -2174,6 +2205,9 @@ class SearchRepositoryBase(ABC):
|
||||
query_start = time.perf_counter()
|
||||
candidate_limit = max(self._semantic_vector_k, (limit + offset) * 10)
|
||||
fts_start = time.perf_counter()
|
||||
# allow_relaxed: question-form queries rarely AND-match, and a dead FTS
|
||||
# branch silently degrades hybrid to vector-only ranking. Fusion plus
|
||||
# bm25 keep relaxed lexical candidates from dominating precision.
|
||||
fts_results = await self.search(
|
||||
search_text=search_text,
|
||||
permalink=permalink,
|
||||
@@ -2187,6 +2221,7 @@ class SearchRepositoryBase(ABC):
|
||||
retrieval_mode=SearchRetrievalMode.FTS,
|
||||
limit=candidate_limit,
|
||||
offset=0,
|
||||
allow_relaxed=True,
|
||||
)
|
||||
fts_ms = (time.perf_counter() - fts_start) * 1000
|
||||
vector_start = time.perf_counter()
|
||||
|
||||
@@ -23,7 +23,10 @@ from basic_memory.models.search import (
|
||||
from basic_memory.repository.embedding_provider import EmbeddingProvider
|
||||
from basic_memory.repository.embedding_provider_factory import create_embedding_provider
|
||||
from basic_memory.repository.search_index_row import SearchIndexRow
|
||||
from basic_memory.repository.search_repository_base import SearchRepositoryBase
|
||||
from basic_memory.repository.search_repository_base import (
|
||||
SearchRepositoryBase,
|
||||
relaxed_query_words,
|
||||
)
|
||||
from basic_memory.repository.metadata_filters import parse_metadata_filters, build_sqlite_json_path
|
||||
from basic_memory.repository.semantic_errors import SemanticDependenciesMissingError
|
||||
from basic_memory.schemas.search import SearchItemType, SearchRetrievalMode
|
||||
@@ -255,6 +258,19 @@ class SQLiteSearchRepository(SearchRepositoryBase):
|
||||
if "*" in term and all(c.isalnum() or c in "*_-" for c in term):
|
||||
return term
|
||||
|
||||
# Natural-language queries arrive with sentence punctuation that FTS5
|
||||
# treats as syntax ("When did Melanie paint a sunrise?"). The tokenizer
|
||||
# ignores this punctuation in the INDEX, so stripping it from word
|
||||
# edges loses nothing — but leaving it forces the whole question into
|
||||
# an exact-phrase match that returns zero rows, silently disabling the
|
||||
# FTS half of hybrid search. Interior characters (hyphens, slashes —
|
||||
# permalinks and paths) are untouched.
|
||||
if " " in term:
|
||||
words = [word.strip("?!.,;:") for word in term.split()]
|
||||
term = " ".join(word for word in words if word)
|
||||
if not term:
|
||||
return ""
|
||||
|
||||
# Characters that can cause FTS5 syntax errors when used as operators
|
||||
# We're more conservative here - only quote when we detect problematic patterns
|
||||
problematic_chars = [
|
||||
@@ -351,6 +367,14 @@ class SQLiteSearchRepository(SearchRepositoryBase):
|
||||
# For non-Boolean queries, use the single term preparation logic
|
||||
return self._prepare_single_term(term, is_prefix)
|
||||
|
||||
@staticmethod
|
||||
def _relaxed_fts_text(search_text: Optional[str]) -> Optional[str]:
|
||||
"""OR-relaxed FTS5 expression for a failed strict query, or None."""
|
||||
words = relaxed_query_words(search_text)
|
||||
if not words:
|
||||
return None
|
||||
return " OR ".join(f"{word}*" for word in words)
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# sqlite-vec extension loading (SQLite-specific)
|
||||
# ------------------------------------------------------------------
|
||||
@@ -953,8 +977,15 @@ class SQLiteSearchRepository(SearchRepositoryBase):
|
||||
min_similarity: Optional[float] = None,
|
||||
limit: int = 10,
|
||||
offset: int = 0,
|
||||
allow_relaxed: bool = False,
|
||||
) -> List[SearchIndexRow]:
|
||||
"""Search across all indexed content using SQLite FTS5."""
|
||||
"""Search across all indexed content using SQLite FTS5.
|
||||
|
||||
``allow_relaxed=True`` retries a zero-result strict multi-word query
|
||||
with OR-joined content terms. Only the hybrid path opts in: its FTS
|
||||
branch otherwise contributes nothing for question-form queries.
|
||||
Service-level FTS searches keep their own conservative fallback.
|
||||
"""
|
||||
# --- Dispatch vector / hybrid modes (shared logic) ---
|
||||
dispatched = await self._dispatch_retrieval_mode(
|
||||
search_text=search_text,
|
||||
@@ -1021,6 +1052,21 @@ class SQLiteSearchRepository(SearchRepositoryBase):
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
result = await session.execute(text(sql), params)
|
||||
rows = result.fetchall()
|
||||
# Trigger: multi-word natural-language query matched nothing
|
||||
# under the default all-terms-AND semantics.
|
||||
# Why: questions ("when did X do Y") rarely have every word in
|
||||
# one document; without relaxation the FTS half of hybrid
|
||||
# search contributes zero candidates and ranking degrades to
|
||||
# vector-only.
|
||||
# Outcome: one retry with OR-joined prefix terms; bm25 still
|
||||
# ranks multi-term matches first.
|
||||
relaxed = (
|
||||
self._relaxed_fts_text(search_text) if allow_relaxed and not rows else None
|
||||
)
|
||||
if relaxed and params.get("text"):
|
||||
params["text"] = relaxed
|
||||
result = await session.execute(text(sql), params)
|
||||
rows = result.fetchall()
|
||||
except Exception as e:
|
||||
# Handle FTS5 syntax errors and provide user-friendly feedback
|
||||
if self._is_fts5_syntax_error(e): # pragma: no cover
|
||||
|
||||
Reference in New Issue
Block a user