primer.common.text

Minimal text normalization shared by BM25, the toy embedder, and RAG.

Real systems use a proper analyzer (Lucene/Elasticsearch: lowercasing, stemming, stopwords) for keyword search, and a subword tokenizer (BPE, see primer.ml.tokenization) for neural models. This is intentionally simple so the retrieval code stays readable.

Further reading:

on GitHub
 1"""
 2Minimal text normalization shared by BM25, the toy embedder, and RAG.
 3
 4Real systems use a proper analyzer (Lucene/Elasticsearch: lowercasing,
 5stemming, stopwords) for keyword search, and a subword tokenizer (BPE, see
 6`primer.ml.tokenization`) for neural models. This is intentionally simple
 7so the retrieval code stays readable.
 8
 9Further reading:
10- Elasticsearch analyzers: https://www.elastic.co/guide/en/elasticsearch/reference/current/analysis.html
11"""
12
13from __future__ import annotations
14
15import re
16
17# Small English stopword list. Stopwords carry little meaning for retrieval,
18# and dropping them keeps BM25 and the toy embedder focused on content words.
19STOPWORDS = frozenset(
20    """
21    a an and are as at be by can do does for from has have how i if in is it
22    its me my of on or our so that the their them this to was we what when
23    where which who why will with you your
24    """.split()
25)
26
27# Keep tokens like "err-4012", "q3", "vpn", "2fa": letters/digits joined by
28# single hyphens. Everything else (punctuation, whitespace) is a separator.
29_TOKEN_RE = re.compile(r"[a-z0-9]+(?:-[a-z0-9]+)*")
30
31
32def tokenize(text: str, drop_stopwords: bool = True) -> list[str]:
33    """Lowercase and split `text` into word tokens.
34
35    >>> tokenize("How do I reset my password? Error ERR-4012!")
36    ['reset', 'password', 'error', 'err-4012']
37    """
38    toks = _TOKEN_RE.findall(text.lower())
39    if drop_stopwords:
40        toks = [t for t in toks if t not in STOPWORDS]
41    return toks
STOPWORDS = frozenset({'does', 'who', 'was', 'do', 'for', 'what', 'why', 'can', 'that', 'me', 'my', 'an', 'of', 'when', 'and', 'as', 'them', 'by', 'is', 'to', 'have', 'with', 'will', 'if', 'this', 'we', 'so', 'be', 'their', 'or', 'you', 'in', 'your', 'our', 'a', 'are', 'has', 'how', 'i', 'on', 'its', 'at', 'it', 'the', 'from', 'which', 'where'})
def tokenize(text: str, drop_stopwords: bool = True) -> list[str]: on GitHub
33def tokenize(text: str, drop_stopwords: bool = True) -> list[str]:
34    """Lowercase and split `text` into word tokens.
35
36    >>> tokenize("How do I reset my password? Error ERR-4012!")
37    ['reset', 'password', 'error', 'err-4012']
38    """
39    toks = _TOKEN_RE.findall(text.lower())
40    if drop_stopwords:
41        toks = [t for t in toks if t not in STOPWORDS]
42    return toks

Lowercase and split text into word tokens.

>>> tokenize("How do I reset my password? Error ERR-4012!")
['reset', 'password', 'error', 'err-4012']