Agent 2.0 exdev 서버 배포 스택
- server-dev start/stop/deploy 및 Gitea push 자동 배포 - local-dev 로컬 개발 환경 Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
@@ -0,0 +1,115 @@
|
||||
"""
|
||||
Model-free sparse vector encoder for Qdrant hybrid search.
|
||||
|
||||
This encoder intentionally avoids external model dependencies. It converts
|
||||
Korean/English/numeric tokens into stable integer dimensions and assigns
|
||||
simple field-aware weights. The goal is exact keyword recall for terms such as
|
||||
"희망드림" while dense embeddings keep handling semantic similarity.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
import re
|
||||
import zlib
|
||||
from collections import Counter
|
||||
from typing import Any, Dict, Iterable, List, Tuple
|
||||
|
||||
|
||||
_TOKEN_RE = re.compile(r"[0-9a-zA-Z가-힣][0-9a-zA-Z가-힣+\-_.]*")
|
||||
_MAX_DIM = 2_000_000_000
|
||||
|
||||
STOPWORDS = {
|
||||
"은",
|
||||
"는",
|
||||
"이",
|
||||
"가",
|
||||
"을",
|
||||
"를",
|
||||
"의",
|
||||
"에",
|
||||
"에서",
|
||||
"으로",
|
||||
"로",
|
||||
"와",
|
||||
"과",
|
||||
"도",
|
||||
"만",
|
||||
"및",
|
||||
"또는",
|
||||
"그리고",
|
||||
"안내",
|
||||
"문의",
|
||||
"방법",
|
||||
}
|
||||
|
||||
|
||||
def _token_id(token: str) -> int:
|
||||
# Qdrant sparse indices are unsigned integer dimensions. crc32 is stable
|
||||
# across processes, unlike Python's built-in hash().
|
||||
return zlib.crc32(token.encode("utf-8")) % _MAX_DIM
|
||||
|
||||
|
||||
def tokenize(text: Any) -> List[str]:
|
||||
raw = str(text or "").lower()
|
||||
tokens: List[str] = []
|
||||
for match in _TOKEN_RE.finditer(raw):
|
||||
token = match.group(0).strip("._-+")
|
||||
if len(token) < 2:
|
||||
continue
|
||||
if token in STOPWORDS:
|
||||
continue
|
||||
tokens.append(token)
|
||||
return tokens
|
||||
|
||||
|
||||
def _weighted_counts(parts: Iterable[Tuple[Any, float]]) -> Counter:
|
||||
counts: Counter = Counter()
|
||||
for text, weight in parts:
|
||||
for token in tokenize(text):
|
||||
counts[token] += weight
|
||||
return counts
|
||||
|
||||
|
||||
def _to_sparse_vector(counts: Counter, *, max_terms: int) -> Dict[str, List[float]]:
|
||||
if not counts:
|
||||
return {"indices": [], "values": []}
|
||||
|
||||
# Field-aware TF weight. Without corpus-wide IDF, rare proper nouns still
|
||||
# get strong exact-match behavior because they occupy unique dimensions.
|
||||
scored = [
|
||||
(token, 1.0 + math.log(float(count)))
|
||||
for token, count in counts.items()
|
||||
if count > 0
|
||||
]
|
||||
scored.sort(key=lambda item: item[1], reverse=True)
|
||||
|
||||
by_index: Dict[int, float] = {}
|
||||
for token, value in scored[:max_terms]:
|
||||
idx = _token_id(token)
|
||||
by_index[idx] = by_index.get(idx, 0.0) + float(value)
|
||||
|
||||
ordered = sorted(by_index.items())
|
||||
return {
|
||||
"indices": [idx for idx, _ in ordered],
|
||||
"values": [round(value, 6) for _, value in ordered],
|
||||
}
|
||||
|
||||
|
||||
def encode_document(meta: Dict[str, Any], *, max_terms: int = 256) -> Dict[str, List[float]]:
|
||||
"""Encode FAQ payload into a model-free sparse vector."""
|
||||
q = meta.get("q") or meta.get("question") or ""
|
||||
a = meta.get("a") or meta.get("answer") or ""
|
||||
parts = [
|
||||
(q, 3.0),
|
||||
(a, 1.0),
|
||||
(meta.get("category"), 1.4),
|
||||
(meta.get("source"), 1.2),
|
||||
]
|
||||
return _to_sparse_vector(_weighted_counts(parts), max_terms=max_terms)
|
||||
|
||||
|
||||
def encode_query(query: str, *, max_terms: int = 64) -> Dict[str, List[float]]:
|
||||
"""Encode user query into a sparse vector using the same token space."""
|
||||
return _to_sparse_vector(_weighted_counts([(query, 1.0)]), max_terms=max_terms)
|
||||
|
||||
Reference in New Issue
Block a user