perf: cache PreparedQuery per stream, skip double-normalize in dedupe (#282)
Scoring hot path (_normalize_score_dedupe) re-tokenized the same ranking_query ~240x per stream: once per item for local_relevance, plus ~5x per item across snippet windows. Query tokens are immutable within a stream, so compute them once as relevance.PreparedQuery and thread through signals.annotate_stream and snippet.extract_best_snippet. dedupe._PreparedText called normalize_text twice: once in __init__ and again via get_ngrams. Factor out _ngrams_of_normalized so the prepared path skips the redundant pass while get_ngrams keeps its public contract. Behavior unchanged.
This commit is contained in:
@@ -39,11 +39,14 @@ def normalize_text(text: str) -> str:
|
||||
return re.sub(r"\s+", " ", text).strip()
|
||||
|
||||
|
||||
def _ngrams_of_normalized(norm: str, n: int = 3) -> set[str]:
|
||||
if len(norm) < n:
|
||||
return {norm} if norm else set()
|
||||
return {norm[index:index + n] for index in range(len(norm) - n + 1)}
|
||||
|
||||
|
||||
def get_ngrams(text: str, n: int = 3) -> set[str]:
|
||||
text = normalize_text(text)
|
||||
if len(text) < n:
|
||||
return {text} if text else set()
|
||||
return {text[index:index + n] for index in range(len(text) - n + 1)}
|
||||
return _ngrams_of_normalized(normalize_text(text), n)
|
||||
|
||||
|
||||
def jaccard_similarity(left: set[str], right: set[str]) -> float:
|
||||
@@ -90,7 +93,7 @@ class _PreparedText:
|
||||
|
||||
def __init__(self, raw: str) -> None:
|
||||
norm = normalize_text(raw)
|
||||
self.ngrams = get_ngrams(norm) if norm else set()
|
||||
self.ngrams = _ngrams_of_normalized(norm)
|
||||
self.tokens = _tokenize(norm)
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user