4b86b2a660
- server-dev start/stop/deploy 및 Gitea push 자동 배포 - local-dev 로컬 개발 환경 Co-authored-by: Cursor <cursoragent@cursor.com>
116 lines
3.1 KiB
Python
116 lines
3.1 KiB
Python
"""
|
|
Model-free sparse vector encoder for Qdrant hybrid search.
|
|
|
|
This encoder intentionally avoids external model dependencies. It converts
|
|
Korean/English/numeric tokens into stable integer dimensions and assigns
|
|
simple field-aware weights. The goal is exact keyword recall for terms such as
|
|
"희망드림" while dense embeddings keep handling semantic similarity.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import math
|
|
import re
|
|
import zlib
|
|
from collections import Counter
|
|
from typing import Any, Dict, Iterable, List, Tuple
|
|
|
|
|
|
_TOKEN_RE = re.compile(r"[0-9a-zA-Z가-힣][0-9a-zA-Z가-힣+\-_.]*")
|
|
_MAX_DIM = 2_000_000_000
|
|
|
|
STOPWORDS = {
|
|
"은",
|
|
"는",
|
|
"이",
|
|
"가",
|
|
"을",
|
|
"를",
|
|
"의",
|
|
"에",
|
|
"에서",
|
|
"으로",
|
|
"로",
|
|
"와",
|
|
"과",
|
|
"도",
|
|
"만",
|
|
"및",
|
|
"또는",
|
|
"그리고",
|
|
"안내",
|
|
"문의",
|
|
"방법",
|
|
}
|
|
|
|
|
|
def _token_id(token: str) -> int:
|
|
# Qdrant sparse indices are unsigned integer dimensions. crc32 is stable
|
|
# across processes, unlike Python's built-in hash().
|
|
return zlib.crc32(token.encode("utf-8")) % _MAX_DIM
|
|
|
|
|
|
def tokenize(text: Any) -> List[str]:
|
|
raw = str(text or "").lower()
|
|
tokens: List[str] = []
|
|
for match in _TOKEN_RE.finditer(raw):
|
|
token = match.group(0).strip("._-+")
|
|
if len(token) < 2:
|
|
continue
|
|
if token in STOPWORDS:
|
|
continue
|
|
tokens.append(token)
|
|
return tokens
|
|
|
|
|
|
def _weighted_counts(parts: Iterable[Tuple[Any, float]]) -> Counter:
|
|
counts: Counter = Counter()
|
|
for text, weight in parts:
|
|
for token in tokenize(text):
|
|
counts[token] += weight
|
|
return counts
|
|
|
|
|
|
def _to_sparse_vector(counts: Counter, *, max_terms: int) -> Dict[str, List[float]]:
|
|
if not counts:
|
|
return {"indices": [], "values": []}
|
|
|
|
# Field-aware TF weight. Without corpus-wide IDF, rare proper nouns still
|
|
# get strong exact-match behavior because they occupy unique dimensions.
|
|
scored = [
|
|
(token, 1.0 + math.log(float(count)))
|
|
for token, count in counts.items()
|
|
if count > 0
|
|
]
|
|
scored.sort(key=lambda item: item[1], reverse=True)
|
|
|
|
by_index: Dict[int, float] = {}
|
|
for token, value in scored[:max_terms]:
|
|
idx = _token_id(token)
|
|
by_index[idx] = by_index.get(idx, 0.0) + float(value)
|
|
|
|
ordered = sorted(by_index.items())
|
|
return {
|
|
"indices": [idx for idx, _ in ordered],
|
|
"values": [round(value, 6) for _, value in ordered],
|
|
}
|
|
|
|
|
|
def encode_document(meta: Dict[str, Any], *, max_terms: int = 256) -> Dict[str, List[float]]:
|
|
"""Encode FAQ payload into a model-free sparse vector."""
|
|
q = meta.get("q") or meta.get("question") or ""
|
|
a = meta.get("a") or meta.get("answer") or ""
|
|
parts = [
|
|
(q, 3.0),
|
|
(a, 1.0),
|
|
(meta.get("category"), 1.4),
|
|
(meta.get("source"), 1.2),
|
|
]
|
|
return _to_sparse_vector(_weighted_counts(parts), max_terms=max_terms)
|
|
|
|
|
|
def encode_query(query: str, *, max_terms: int = 64) -> Dict[str, List[float]]:
|
|
"""Encode user query into a sparse vector using the same token space."""
|
|
return _to_sparse_vector(_weighted_counts([(query, 1.0)]), max_terms=max_terms)
|
|
|