Files
exAichatbot_agent/exAiChatBot-chatbot2.0-agent/scripts/sparse_encoder.py
T
Macbook 4b86b2a660 Agent 2.0 exdev 서버 배포 스택
- server-dev start/stop/deploy 및 Gitea push 자동 배포
- local-dev 로컬 개발 환경

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-21 22:57:30 +09:00

116 lines
3.1 KiB
Python

"""
Model-free sparse vector encoder for Qdrant hybrid search.
This encoder intentionally avoids external model dependencies. It converts
Korean/English/numeric tokens into stable integer dimensions and assigns
simple field-aware weights. The goal is exact keyword recall for terms such as
"희망드림" while dense embeddings keep handling semantic similarity.
"""
from __future__ import annotations
import math
import re
import zlib
from collections import Counter
from typing import Any, Dict, Iterable, List, Tuple
_TOKEN_RE = re.compile(r"[0-9a-zA-Z가-힣][0-9a-zA-Z가-힣+\-_.]*")
_MAX_DIM = 2_000_000_000
STOPWORDS = {
"은",
"는",
"이",
"가",
"을",
"를",
"의",
"에",
"에서",
"으로",
"로",
"와",
"과",
"도",
"만",
"및",
"또는",
"그리고",
"안내",
"문의",
"방법",
}
def _token_id(token: str) -> int:
# Qdrant sparse indices are unsigned integer dimensions. crc32 is stable
# across processes, unlike Python's built-in hash().
return zlib.crc32(token.encode("utf-8")) % _MAX_DIM
def tokenize(text: Any) -> List[str]:
raw = str(text or "").lower()
tokens: List[str] = []
for match in _TOKEN_RE.finditer(raw):
token = match.group(0).strip("._-+")
if len(token) < 2:
continue
if token in STOPWORDS:
continue
tokens.append(token)
return tokens
def _weighted_counts(parts: Iterable[Tuple[Any, float]]) -> Counter:
counts: Counter = Counter()
for text, weight in parts:
for token in tokenize(text):
counts[token] += weight
return counts
def _to_sparse_vector(counts: Counter, *, max_terms: int) -> Dict[str, List[float]]:
if not counts:
return {"indices": [], "values": []}
# Field-aware TF weight. Without corpus-wide IDF, rare proper nouns still
# get strong exact-match behavior because they occupy unique dimensions.
scored = [
(token, 1.0 + math.log(float(count)))
for token, count in counts.items()
if count > 0
]
scored.sort(key=lambda item: item[1], reverse=True)
by_index: Dict[int, float] = {}
for token, value in scored[:max_terms]:
idx = _token_id(token)
by_index[idx] = by_index.get(idx, 0.0) + float(value)
ordered = sorted(by_index.items())
return {
"indices": [idx for idx, _ in ordered],
"values": [round(value, 6) for _, value in ordered],
}
def encode_document(meta: Dict[str, Any], *, max_terms: int = 256) -> Dict[str, List[float]]:
"""Encode FAQ payload into a model-free sparse vector."""
q = meta.get("q") or meta.get("question") or ""
a = meta.get("a") or meta.get("answer") or ""
parts = [
(q, 3.0),
(a, 1.0),
(meta.get("category"), 1.4),
(meta.get("source"), 1.2),
]
return _to_sparse_vector(_weighted_counts(parts), max_terms=max_terms)
def encode_query(query: str, *, max_terms: int = 64) -> Dict[str, List[float]]:
"""Encode user query into a sparse vector using the same token space."""
return _to_sparse_vector(_weighted_counts([(query, 1.0)]), max_terms=max_terms)