Files
exAichatbot_agent/exAiChatBot-chatbot2.0-agent/scripts/preprocess_qa.py
T
Macbook 4b86b2a660 Agent 2.0 exdev 서버 배포 스택
- server-dev start/stop/deploy 및 Gitea push 자동 배포
- local-dev 로컬 개발 환경

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-21 22:57:30 +09:00

78 lines
3.0 KiB
Python

"""
preprocess_qa.py
────────────────────────────────────────────
원본 QA(긴 질문) → LLM 한 문장 요약(q_short) 추가
실행:
python preprocess_qa.py
산출물:
data/qa.jsonl # q, a, q_short
"""
import json, re, pathlib, os
from api_clients import SGLangClient
# ────────────────────────────────────────────
# 0. 경로 설정
BASE_DIR = pathlib.Path(__file__).resolve().parent.parent # rag-demo/
SRC = BASE_DIR / "data" / "qa_raw.jsonl" # 긴 질문·답
DST = BASE_DIR / "data" / "qa.jsonl" # 요약 포함
# ────────────────────────────────────────────
# 1. 후처리: <think>·마크다운·개행 제거 → 첫 문장만
def postprocess(text: str) -> str:
text = re.sub(r"<think>.*?</think>", " ", text, flags=re.S)
text = re.sub(r"[\n\r]+", " ", text)
text = re.sub(r"\s+", " ", text).strip()
# 질문형(물음표 포함) 추출
q_match = re.search(r"([^?]+[?])", text)
if q_match:
return q_match.group(1).strip()
# 물음표가 없으면 설명문 잘라내고 마지막에 물음표 추가
text = re.split(r"[.]", text, maxsplit=1)[0].strip()
if not text.endswith("?"):
text += "?"
return text
# ────────────────────────────────────────────
# 2. 요약용 LLM 클라이언트 (외부 SGLang API)
llm_client = SGLangClient()
SYSTEM_PROMPT = (
"당신은 사용자의 긴 질문을 **같은 의미의 질문 한 문장**으로 바꿔주는 도우미다.\n"
"규칙:\n"
"1) 반드시 질문형 어미로 끝나는 문장(물음표 포함)을 출력하라.\n"
"2) 절대 정의·답변·설명을 포함하지 마라.\n"
"3) 문장 하나만 출력하고 다른 문구를 덧붙이지 마라."
)
def summarize(text: str) -> str:
"""SGLang API를 사용한 질문 요약"""
messages = [
{"role": "system", "content": SYSTEM_PROMPT},
{"role": "user", "content": text},
]
raw = llm_client.chat_completion(
messages=messages,
max_tokens=64,
temperature=0.1 # deterministic에 가깝게
)
return postprocess(raw)
# ────────────────────────────────────────────
# 3. 원본 QA 읽어 요약 추가
assert SRC.exists(), f"{SRC} not found"
DST.parent.mkdir(parents=True, exist_ok=True)
with SRC.open(encoding="utf-8") as fin, DST.open("w", encoding="utf-8") as fout:
for line in fin:
if not line.strip():
continue
obj = json.loads(line)
obj["q_short"] = summarize(obj["q"])
fout.write(json.dumps(obj, ensure_ascii=False) + "\n")
print("✅ 요약 추가 완료 →", DST)