-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathpreprocessing.py
More file actions
129 lines (109 loc) · 4.5 KB
/
Copy pathpreprocessing.py
File metadata and controls
129 lines (109 loc) · 4.5 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
"""
Preprocessing Module for Nova IR System.
Handles text tokenization, lowercasing, stop-word removal, and Porter stemming using NLTK.
"""
import re
import string
from typing import List, Set
import nltk
from nltk.corpus import stopwords
from nltk.stem import PorterStemmer
# Initialize stemmer
_stemmer = PorterStemmer()
# English stop words fallback set
_FALLBACK_STOPS = {
"i", "me", "my", "myself", "we", "our", "ours", "ourselves", "you", "your", "yours",
"yourself", "yourselves", "he", "him", "his", "himself", "she", "her", "hers",
"herself", "it", "its", "itself", "they", "them", "their", "theirs", "themselves",
"what", "which", "who", "whom", "this", "that", "these", "those", "am", "is", "are",
"was", "were", "be", "been", "being", "have", "has", "had", "having", "do", "does",
"did", "doing", "a", "an", "the", "and", "but", "if", "or", "because", "as", "until",
"while", "of", "at", "by", "for", "with", "about", "against", "between", "into",
"through", "during", "before", "after", "above", "below", "to", "from", "up", "down",
"in", "out", "on", "off", "over", "under", "again", "further", "then", "once", "here",
"there", "when", "where", "why", "how", "all", "any", "both", "each", "few", "more",
"most", "other", "some", "such", "no", "nor", "not", "only", "own", "same", "so",
"than", "too", "very", "s", "t", "can", "will", "just", "don", "should", "now"
}
try:
_stop_words: Set[str] = set(stopwords.words("english"))
except Exception:
try:
nltk.download("stopwords", quiet=True)
_stop_words = set(stopwords.words("english"))
except Exception:
_stop_words = _FALLBACK_STOPS
def tokenize_and_clean(text: str) -> List[str]:
"""
Tokenizes text, strips punctuation, and converts to lowercase.
"""
if not text:
return []
# Replace punctuation with whitespace and keep alphanumeric tokens
cleaned = re.sub(r'[^a-zA-Z0-9\s]', ' ', text.lower())
tokens = [t.strip() for t in cleaned.split() if t.strip()]
return tokens
def preprocess_tokens(tokens: List[str], remove_stops: bool = True, stem: bool = True) -> List[str]:
"""
Applies stop-word removal and Porter stemming to a token list.
"""
processed = []
for token in tokens:
if remove_stops and token in _stop_words:
continue
if len(token) <= 1 and not token.isdigit():
continue
if stem:
token = _stemmer.stem(token)
processed.append(token)
return processed
def preprocess_text(text: str, remove_stops: bool = True, stem: bool = True) -> List[str]:
"""
Full preprocessing pipeline: Text -> Tokens -> Filter Stops -> Stemmed Tokens.
"""
tokens = tokenize_and_clean(text)
return preprocess_tokens(tokens, remove_stops=remove_stops, stem=stem)
def preprocess_text_to_str(text: str) -> str:
"""
Returns a space-separated string of preprocessed tokens (useful for scikit-learn).
"""
return " ".join(preprocess_text(text))
def get_query_stem_map(query: str) -> dict:
"""
Returns a dictionary mapping original word stems to their matched words
to assist with search result keyword highlighting.
"""
raw_tokens = tokenize_and_clean(query)
stem_map = {}
for token in raw_tokens:
if token not in _stop_words:
st = _stemmer.stem(token)
if st not in stem_map:
stem_map[st] = set()
stem_map[st].add(token)
return stem_map
def highlight_matched_keywords(text: str, query_stems: dict, max_len: int = 320) -> str:
"""
Creates an informative snippet highlighting words matching the query stems in <mark> tags.
"""
if not text:
return ""
words = text.split()
highlighted_words = []
for word in words:
# Strip punctuation for stem checking
clean_w = re.sub(r'[^a-zA-Z0-9]', '', word.lower())
if clean_w:
w_stem = _stemmer.stem(clean_w)
if w_stem in query_stems:
highlighted_words.append(f"<mark style='background-color: #FEF08A; color: #854D0E; font-weight: 600; padding: 1px 4px; border-radius: 3px;'>{word}</mark>")
else:
highlighted_words.append(word)
else:
highlighted_words.append(word)
snippet = " ".join(highlighted_words)
if len(words) > 50 and len(snippet) > max_len:
# Trim around first match or default prefix
trimmed = " ".join(highlighted_words[:50]) + "..."
return trimmed
return snippet