""" Model-free sparse vector encoder for Qdrant hybrid search. This encoder intentionally avoids external model dependencies. It converts Korean/English/numeric tokens into stable integer dimensions and assigns simple field-aware weights. The goal is exact keyword recall for terms such as "희망드림" while dense embeddings keep handling semantic similarity. """ from __future__ import annotations import math import re import zlib from collections import Counter from typing import Any, Dict, Iterable, List, Tuple _TOKEN_RE = re.compile(r"[0-9a-zA-Z가-힣][0-9a-zA-Z가-힣+\-_.]*") _MAX_DIM = 2_000_000_000 STOPWORDS = { "은", "는", "이", "가", "을", "를", "의", "에", "에서", "으로", "로", "와", "과", "도", "만", "및", "또는", "그리고", "안내", "문의", "방법", } def _token_id(token: str) -> int: # Qdrant sparse indices are unsigned integer dimensions. crc32 is stable # across processes, unlike Python's built-in hash(). return zlib.crc32(token.encode("utf-8")) % _MAX_DIM def tokenize(text: Any) -> List[str]: raw = str(text or "").lower() tokens: List[str] = [] for match in _TOKEN_RE.finditer(raw): token = match.group(0).strip("._-+") if len(token) < 2: continue if token in STOPWORDS: continue tokens.append(token) return tokens def _weighted_counts(parts: Iterable[Tuple[Any, float]]) -> Counter: counts: Counter = Counter() for text, weight in parts: for token in tokenize(text): counts[token] += weight return counts def _to_sparse_vector(counts: Counter, *, max_terms: int) -> Dict[str, List[float]]: if not counts: return {"indices": [], "values": []} # Field-aware TF weight. Without corpus-wide IDF, rare proper nouns still # get strong exact-match behavior because they occupy unique dimensions. scored = [ (token, 1.0 + math.log(float(count))) for token, count in counts.items() if count > 0 ] scored.sort(key=lambda item: item[1], reverse=True) by_index: Dict[int, float] = {} for token, value in scored[:max_terms]: idx = _token_id(token) by_index[idx] = by_index.get(idx, 0.0) + float(value) ordered = sorted(by_index.items()) return { "indices": [idx for idx, _ in ordered], "values": [round(value, 6) for _, value in ordered], } def encode_document(meta: Dict[str, Any], *, max_terms: int = 256) -> Dict[str, List[float]]: """Encode FAQ payload into a model-free sparse vector.""" q = meta.get("q") or meta.get("question") or "" a = meta.get("a") or meta.get("answer") or "" parts = [ (q, 3.0), (a, 1.0), (meta.get("category"), 1.4), (meta.get("source"), 1.2), ] return _to_sparse_vector(_weighted_counts(parts), max_terms=max_terms) def encode_query(query: str, *, max_terms: int = 64) -> Dict[str, List[float]]: """Encode user query into a sparse vector using the same token space.""" return _to_sparse_vector(_weighted_counts([(query, 1.0)]), max_terms=max_terms)