Files
last30days-skill/skills/last30days/scripts/lib/snippet.py
T
Ilia Alshanetsky e6b89f2644 perf: cache PreparedQuery per stream, skip double-normalize in dedupe (#282)
Scoring hot path (_normalize_score_dedupe) re-tokenized the same
ranking_query ~240x per stream: once per item for local_relevance,
plus ~5x per item across snippet windows. Query tokens are immutable
within a stream, so compute them once as relevance.PreparedQuery and
thread through signals.annotate_stream and snippet.extract_best_snippet.

dedupe._PreparedText called normalize_text twice: once in __init__ and
again via get_ngrams. Factor out _ngrams_of_normalized so the prepared
path skips the redundant pass while get_ngrams keeps its public contract.

Behavior unchanged.
2026-04-25 14:16:57 -07:00

52 lines
1.5 KiB
Python

"""Best-window extraction for rerankable evidence snippets."""
from __future__ import annotations
from . import relevance, schema
def _truncate_words(text: str, max_words: int) -> str:
words = text.split()
if len(words) <= max_words:
return text.strip()
return " ".join(words[:max_words]).strip() + "..."
def _windows(words: list[str], size: int, overlap: int) -> list[str]:
if not words:
return []
if len(words) <= size:
return [" ".join(words)]
step = max(1, size - overlap)
return [
" ".join(words[start:start + size])
for start in range(0, len(words), step)
]
def extract_best_snippet(
item: schema.SourceItem,
ranking_query: "str | relevance.PreparedQuery",
max_words: int = 120,
) -> str:
"""Prefer existing snippets, else extract the best matching evidence window."""
preferred = item.snippet.strip()
if preferred:
return _truncate_words(preferred, max_words)
body = item.body.strip()
if not body:
return _truncate_words(item.title, max_words)
words = body.split()
candidates = _windows(words, size=min(max_words, 110), overlap=30)
if not candidates:
return _truncate_words(body, max_words)
prepared_query = ranking_query if isinstance(ranking_query, relevance.PreparedQuery) else relevance.PreparedQuery(ranking_query)
best = max(
candidates,
key=lambda candidate: relevance.token_overlap_relevance(prepared_query, candidate),
)
return _truncate_words(best, max_words)