""" rag-demo/scripts/run_service_qa.py (리팩토링 버전) ──────────────────────────────────────────────── FastAPI 서버: 질문 → 임베딩 → 벡터 검색 → 재랭킹 → 답변 (모든 AI 모델은 외부 API 사용 + MongoDB 대화 이력) """ import json import os import re from pathlib import Path from datetime import datetime, timezone from typing import Optional, Dict, Any from fastapi import FastAPI from pydantic import BaseModel, Field, model_validator # 외부 API 클라이언트 및 벡터 스토어 from api_clients import TEIEmbeddingClient, TEIRerankerClient, SGLangClient from vector_store import get_vector_store from chat_history import get_chat_history_manager # 리팩토링된 핸들러 from handlers import ( Config, QueryRewriter, SearchHandler, PromptBuilder, LLMHandler, ResponseHandler, IntentDetector, GreetingHandler, EmotionDetector, EmotionHandler, SuggestionHandler ) app = FastAPI() # ─────────────────────────────────────────── # 1) 초기화 # ─────────────────────────────────────────── print("[Service] 초기화 시작...") # API 클라이언트 embed_client = TEIEmbeddingClient() rerank_client = TEIRerankerClient() llm_client = SGLangClient() # 벡터 스토어 DATA_DIR = Path("/app/data") vector_store = get_vector_store() vector_store.load(str(DATA_DIR)) print(f"[Service] 벡터 스토어 로드 완료: {vector_store.count()}개") # MongoDB 대화 이력 try: chat_manager = get_chat_history_manager() CHAT_HISTORY_ENABLED = True print("[Service] ✅ 대화 이력 기능 활성화") except Exception as e: print(f"[Service] ❌ 대화 이력 기능 비활성화: {e}") chat_manager = None CHAT_HISTORY_ENABLED = False # 설정 로드 config = Config.from_env(chat_history_enabled=CHAT_HISTORY_ENABLED) config.print_summary() # 핸들러 초기화 query_rewriter = QueryRewriter(llm_client, embed_client) search_handler = SearchHandler(vector_store, embed_client, rerank_client, config) prompt_builder = PromptBuilder() llm_handler = LLMHandler(llm_client, config) response_handler = ResponseHandler(DATA_DIR, chat_manager, config) intent_detector = IntentDetector() greeting_handler = GreetingHandler() emotion_detector = EmotionDetector() emotion_handler = EmotionHandler() suggestion_handler = SuggestionHandler( low_confidence_threshold=config.low_confidence_threshold, high_confidence_threshold=config.high_confidence_threshold, ) print("[Service] 초기화 완료!\n") LLM_PROMPT_LOG_ENABLED = os.getenv("LLM_PROMPT_LOG_ENABLED", "true").lower() == "true" LLM_PROMPT_LOG_MAX_CHARS = int(os.getenv("LLM_PROMPT_LOG_MAX_CHARS", "12000")) # ─────────────────────────────────────────── # 2) 요청/응답 스키마 # ─────────────────────────────────────────── class QueryRequest(BaseModel): query: str = Field(..., min_length=1, max_length=500, description="사용자 질문") bot_id: Optional[str] = Field(None, max_length=100, description="봇 ID", alias="botId") domain_data: Optional[dict] = Field(None, description="chatbotApi 도메인 서비스 DB 조회 결과", alias="domainData") intent_type: Optional[str] = Field(None, description="의도 타입 (FARE_SEARCH 등)", alias="intentType") class Config: populate_by_name = True class IntentAnalysisRequest(BaseModel): query: str = Field(..., min_length=1, max_length=500, description="사용자 질문") bot_id: Optional[str] = Field(None, max_length=100, description="봇 ID", alias="botId") pending_intent_type: Optional[str] = Field(None, description="이전 턴에서 대기 중인 의도", alias="pendingIntentType") pending_params: Optional[Dict[str, Any]] = Field(None, description="이전 턴에서 수집된 파라미터", alias="pendingParams") intent_definitions: Optional[list[Dict[str, Any]]] = Field(None, description="chatbotApi가 전달한 intent 정의", alias="intentDefinitions") class Config: populate_by_name = True class AgentChatRequest(BaseModel): query: Optional[str] = Field(None, min_length=1, max_length=500) question: Optional[str] = Field(None, min_length=1, max_length=500) bot_id: Optional[str] = Field(None, max_length=100, alias="botId") pending_intent_type: Optional[str] = Field(None, alias="pendingIntentType") pending_params: Optional[Dict[str, Any]] = Field(None, alias="pendingParams") class Config: populate_by_name = True @model_validator(mode="after") def resolve_query(self): resolved = (self.query or self.question or "").strip() if not resolved: raise ValueError("query or question is required") self.query = resolved return self # Agent tool loop (Phase 2+) from agent.pending_store import AgentPendingStore from agent.tool_executor import ToolExecutor from agent.agent_service import AgentService pending_store = AgentPendingStore() tool_executor = ToolExecutor( search_handler=search_handler, config=config, ) agent_service = AgentService( llm_client=llm_client, tool_executor=tool_executor, config=config, pending_store=pending_store, prompt_builder=prompt_builder, llm_handler=llm_handler, intent_detector=intent_detector, greeting_handler=greeting_handler, emotion_detector=emotion_detector, emotion_handler=emotion_handler, suggestion_handler=suggestion_handler, chat_manager=chat_manager if CHAT_HISTORY_ENABLED else None, response_handler=response_handler, query_rewriter=query_rewriter, ) print("[Service] ✅ Agent tool-calling 모듈 초기화") # ─────────────────────────────────────────── # 3) 헬퍼 함수 # ─────────────────────────────────────────── def handle_no_match_with_rewriting( original_query: str, bot_id: Optional[str], ts: str ) -> Optional[dict]: """검색 실패 시 Query Rewriting 시도 Returns: 재검색 성공 시 검색 결과, 실패 시 None """ if not (config.query_rewrite_enabled and CHAT_HISTORY_ENABLED and bot_id): return None # 대화 이력 조회 try: history = chat_manager.get_recent_history( bot_id=bot_id, hours=config.chat_history_hours, limit=config.chat_history_limit ) if not history: return None # 질문 재작성 rewritten_query = query_rewriter.rewrite_query(original_query, history, ts) if not rewritten_query: return None # 재임베딩 & 재검색 print(f"[ask] {ts} 🔄 재임베딩 중... (threshold={config.threshold_rewrite})") query_vec = search_handler.embed_query(rewritten_query, ts) if query_vec is None: return None print(f"[ask] {ts} 🔍 재검색 중...") search_results = search_handler.search(query_vec, config.threshold_rewrite, ts) if search_results: top_scores = [r["score"] for r in search_results[:5]] print(f"[ask] {ts} ✅ 재검색 성공! {len(search_results)}개 발견 (top 5: {top_scores})") print(f"[ask] {ts} 📝 적용: '{original_query}' → '{rewritten_query}'") return { "results": search_results, "rewritten_query": rewritten_query } print(f"[ask] {ts} ⚠️ 재검색도 실패 (threshold={config.threshold_rewrite} 미달)") return None except Exception as e: print(f"[ask] {ts} ❌ Query Rewriting 프로세스 실패: {e}") return None def handle_no_match_guidance( original_query: str, bot_id: Optional[str], ts: str ) -> dict: """검색 실패 시 LLM으로 짧은 답변 불가/역할 안내를 생성한다.""" print(f"[ask] {ts} Query Rewriting도 실패 → guidance LLM 답변 생성") conversation_context = "" if CHAT_HISTORY_ENABLED and bot_id: try: conversation_context = chat_manager.get_context_for_llm( bot_id=bot_id, hours=config.chat_history_hours, max_conversations=config.chat_history_limit ) if conversation_context: print(f"[ask] {ts} 🔄 대화 이력 포함 (guidance)") except Exception as e: print(f"[ask] {ts} 대화 이력 조회 실패: {e}") try: system_prompt, user_prompt = prompt_builder.build_guidance_prompt( original_query, conversation_context ) guidance_response = llm_handler.generate_answer( system_prompt, user_prompt, ts ) except Exception as e: print(f"[ask] {ts} ❌ guidance LLM 실패, 기본 안내 사용: {e}") guidance_response = ( "문의하신 내용은 현재 정확한 정보를 확인하기 어렵습니다.\n" "정확한 확인이 필요한 경우 한국도로공사 콜센터(1588-2504)로 문의해 주세요." ) # MongoDB 저장 response_handler.save_to_mongodb( bot_id=bot_id, user_query=original_query, ai_response=guidance_response, matched_questions=[], scores=[], metadata={ "type": "no_match", "reason": "threshold_not_met", "threshold": config.threshold, "answer_confidence": "low", "num_references": 0, "statusMsg": "no_match", }, ts=ts ) return response_handler.build_no_match_response( answer=guidance_response, bot_id=bot_id, top_k=config.top_k ) def has_usable_domain_data(domain_data: Optional[dict]) -> bool: """LLM 답변 근거로 사용할 수 있는 도메인 데이터인지 확인""" if not domain_data: return False status = domain_data.get("status") if status is False: return False if isinstance(status, str) and status.lower() == "false": return False return True def build_domain_data_fallback_answer(domain_data: Optional[dict]) -> str: """LLM 실패 시 domainData만으로 최소 응답 생성""" if not domain_data: return llm_handler.generate_default_guidance() summary = domain_data.get("llmSummary") if summary is not None and str(summary).strip(): return str(summary).strip() status_msg = domain_data.get("statusMsg") if status_msg is not None and str(status_msg).strip(): return str(status_msg).strip() lines = [] for key, value in domain_data.items(): if key in ("status", "errorMsg") or value is None: continue lines.append(f"- {key}: {value}") if lines: return "조회된 정보는 다음과 같습니다.\n" + "\n".join(lines) return llm_handler.generate_default_guidance() def log_llm_messages(messages: list[Dict[str, Any]], ts: str, intent_type: Optional[str], domain_data_present: bool): """실제 LLM 호출에 전달되는 messages를 로그로 출력한다.""" if not LLM_PROMPT_LOG_ENABLED: return print( f"[ask] {ts} ===== LLM messages 시작 " f"(intentType={intent_type}, domainData={domain_data_present}, count={len(messages)}) =====" ) for idx, message in enumerate(messages): role = message.get("role") raw_content = message.get("content", "") content = str(raw_content) original_length = len(content) if original_length > LLM_PROMPT_LOG_MAX_CHARS: content = ( content[:LLM_PROMPT_LOG_MAX_CHARS] + f"\n...(이하 로그 생략, totalChars={original_length})" ) print(f"[ask] {ts} [LLM message {idx}] role={role}, chars={original_length}") print(content) print(f"[ask] {ts} ===== LLM messages 끝 =====") def _extract_json_object(text: str) -> Dict[str, Any]: """LLM 응답에서 JSON object만 추출한다.""" cleaned = llm_handler._remove_think_tags(text).strip() cleaned = re.sub(r"^```(?:json)?\s*", "", cleaned) cleaned = re.sub(r"\s*```$", "", cleaned) try: return json.loads(cleaned) except json.JSONDecodeError: match = re.search(r"\{.*\}", cleaned, flags=re.DOTALL) if not match: raise return json.loads(match.group(0)) def _fallback_intent_analysis(query: str) -> Dict[str, Any]: """LLM 분석 실패 시 최소한의 안전한 라우팅만 수행한다.""" text = query.strip() params: Dict[str, Any] = {} if any(keyword in text for keyword in ["통행요금", "통행료", "요금"]): if "에서" in text and "까지" in text: before, after = text.split("에서", 1) destination = after.split("까지", 1)[0].strip() params["fromIc"] = before.strip() or None params["toIc"] = destination or None return { "intentType": "FARE_SEARCH", "confidence": 0.65, "params": params, "routeType": "domain", "needsClarification": False, "clarificationQuestion": None, "reason": "keyword_fallback" } return { "intentType": None, "confidence": 0.0, "params": {}, "routeType": "rag", "needsClarification": False, "clarificationQuestion": None, "reason": "llm_failed_fallback_to_rag" } def _build_intent_router_prompt(intent_definitions: Optional[list[Dict[str, Any]]]) -> str: intents = intent_definitions or [] lines = [ "당신은 한국도로공사 카카오 챗봇의 intent router입니다.", "사용자 발화를 아래 JSON object 하나로만 분류하세요. 설명, markdown, 코드블록은 금지합니다.", "", "허용 intentType:" ] for definition in intents: intent_type = definition.get("intentType") params = definition.get("params") or [] param_descriptions = [] for param in params: if isinstance(param, dict): name = param.get("name") required_label = "required" if param.get("required") else "optional" description = param.get("description") or "" examples = param.get("examples") or [] example_text = f" 예: {', '.join(map(str, examples))}" if examples else "" param_descriptions.append(f"{name}({required_label}): {description}{example_text}") else: param_descriptions.append(str(param)) lines.append(f"- {intent_type}: {definition.get('description', '')}. params: {'; '.join(param_descriptions)}") lines.extend([ "", "규칙:", "- 특정 도메인 DB/API 조회 intent가 아니면 intentType은 null로 둡니다.", "- 미납 조회 방법, 미납 납부 방법, 미납 확인 경로처럼 방법/절차/위치를 묻는 안내성 질문은 차량번호가 없으면 FARE_UNPAID로 분류하지 말고 intentType을 null로 둡니다.", "- \"여기\", \"거기\", \"저기\", \"현재 위치\", \"내 위치\"는 실제 IC명으로 확정하지 말고 해당 param을 null로 둡니다.", "- 알 수 없는 필수 파라미터는 추측하지 말고 null로 둡니다.", "- 이전 pendingIntentType이 있고 사용자가 누락 파라미터만 짧게 답한 경우, pending intent의 param으로 해석하세요.", "- 차량번호 파라미터(carNo)는 공백과 하이픈을 제거한 전체 차량번호로 반환하세요. 예: 12가 3456→12가3456, 123가-4567→123가4567. 끝자리만 있는 부분 차량번호는 carNo로 확정하지 말고 null로 둡니다.", "- IC/영업소명 파라미터(fromIc, toIc, icName)는 IC, 영업소, 톨게이트, 요금소 같은 접미사와 공백을 제거한 짧은 한글명으로 반환하세요. 예: 판교IC→판교, 서울 영업소→서울, 신갈 톨게이트→신갈.", "- 휴게소명 파라미터(restAreaName)는 휴게소 접미사와 공백을 제거한 짧은 한글명으로 반환하세요. 예: 죽전휴게소→죽전, 죽전 휴게소→죽전, 망향휴게소→망향.", "", "반드시 다음 schema를 지키세요:", "{", " \"intentType\": string|null,", " \"confidence\": number,", " \"params\": object,", " \"reason\": string", "}" ]) return "\n".join(lines) @app.post("/intent/analyze") def analyze_intent(req: IntentAnalysisRequest): """LLM 기반 intent/slot 분석 전용 엔드포인트.""" ts = datetime.now(timezone.utc).isoformat() print(f"\n[intent] {ts} 분석 요청: {req.query}") system_prompt = _build_intent_router_prompt(req.intent_definitions) user_prompt = { "query": req.query, "pendingIntentType": req.pending_intent_type, "pendingParams": req.pending_params or {} } try: raw = llm_handler.generate_answer_from_messages( messages=[ {"role": "system", "content": system_prompt}, {"role": "user", "content": json.dumps(user_prompt, ensure_ascii=False)} ], ts=ts, max_tokens=512, temperature=0.0 ) result = _extract_json_object(raw) result.setdefault("params", {}) result.setdefault("confidence", 0.0) result.setdefault("routeType", "rag" if not result.get("intentType") else "domain") result.setdefault("needsClarification", False) result.setdefault("clarificationQuestion", None) result.setdefault("reason", "llm") print(f"[intent] {ts} 분석 결과: {result}") return result except Exception as e: print(f"[intent] {ts} ❌ LLM intent 분석 실패: {e}") result = _fallback_intent_analysis(req.query) print(f"[intent] {ts} 폴백 분석 결과: {result}") return result # ─────────────────────────────────────────── # 4) /ask 엔드포인트 # ─────────────────────────────────────────── @app.post("/ask") def ask(q: QueryRequest): """질문 처리 메인 엔드포인트""" ts = datetime.now(timezone.utc).isoformat() print(f"\n[ask] {ts} 질문: {q.query}") # ─── ⓪ 의도 감지 ───────────────────────────── intent = intent_detector.detect(q.query, ts) # 특별 의도 처리 (인사, 종료, 불만) if intent.is_special(): print(f"[ask] {ts} 특별 의도 감지: {intent.name}") # 특별 응답 생성 response = greeting_handler.generate_response( intent_name=intent.name, query=q.query, matched_keywords=intent.matched_keywords, bot_id=q.bot_id ) # MongoDB 저장 (대화 이력 기록) if CHAT_HISTORY_ENABLED: response_handler.save_to_mongodb( bot_id=q.bot_id, user_query=q.query, ai_response=response["answer"], matched_questions=[], scores=[], metadata={ "type": "special_intent", "intent": intent.name, "confidence": intent.confidence, "matched_keywords": intent.matched_keywords }, ts=ts ) print(f"[ask] {ts} 특별 응답 반환: {intent.name}") return response original_query = q.query rewritten_query = None domain_data_available = has_usable_domain_data(q.domain_data) # ─── ① 임베딩 ───────────────────────────── query_vec = search_handler.embed_query(q.query, ts) if query_vec is None: if domain_data_available: print(f"[ask] {ts} 임베딩 실패, domainData 기반 LLM 답변으로 진행") search_results = [] else: return { "answer": "죄송합니다. 일시적인 오류가 발생했습니다. (임베딩 실패)", "matched_question": None, "score": None, } else: # ─── ② 검색 ─────────────────────────────── search_results = search_handler.search(query_vec, config.threshold, ts) # 검색 실패 → domainData가 있으면 즉시 LLM, 없으면 Query Rewriting 시도 if not search_results: response_handler.log_failed_query(original_query, ts) print(f"[ask] {ts} threshold 미달: q={original_query}") if domain_data_available: print(f"[ask] {ts} threshold 미달이지만 domainData 존재 → 재작성/재검색 생략, domainData 기반 LLM 답변으로 진행") search_results = [] else: # Query Rewriting 시도. 임베딩 실패로 검색을 못 한 경우에는 재작성도 건너뛴다. rewrite_result = None if query_vec is not None: rewrite_result = handle_no_match_with_rewriting(original_query, q.bot_id, ts) else: print(f"[ask] {ts} 임베딩 실패로 Query Rewriting 건너뜀") if rewrite_result: search_results = rewrite_result["results"] rewritten_query = rewrite_result["rewritten_query"] else: # 재검색도 실패 → 질문 유도 return handle_no_match_guidance(original_query, q.bot_id, ts) # ─── ③ 재랭킹 ───────────────────────────── candidates = [r["meta"] for r in search_results] faiss_scores = [r["score"] for r in search_results[:config.top_n_for_llm]] # Reranking 시 재작성 질문 사용 (있으면) query_for_rerank = rewritten_query if rewritten_query else q.query top_n_results, top_scores, rerank_used, rerank_info = search_handler.rerank( query_for_rerank, candidates, ts ) # ─── ③-1 낮은 신뢰도 시 Full Context Query Rewriting ────── if ( config.query_rewrite_enabled and top_scores and top_scores[0] < config.low_confidence_threshold and not rewritten_query ): print( f"[ask] {ts} 낮은 신뢰도 감지 ({top_scores[0]:.2f} < {config.low_confidence_threshold}) " f"→ Full Context Rewriting 시도" ) if CHAT_HISTORY_ENABLED and q.bot_id: try: # 대화 이력 조회 (질문 + 답변) history_data = chat_manager.get_recent_history( bot_id=q.bot_id, hours=config.chat_history_hours, limit=config.chat_history_limit ) if history_data: # Full Context Rewriting (질문 + 답변 활용) full_context_query = query_rewriter.rewrite_query_with_full_context( original_query=original_query, history=history_data, ts=ts ) if full_context_query: # 재임베딩 & 재검색 print(f"[ask] {ts} 🔄 Full Context 재임베딩 중...") query_vec_full = search_handler.embed_query(full_context_query, ts) if query_vec_full is not None: print(f"[ask] {ts} 🔍 Full Context 재검색 중...") search_results_full = search_handler.search( query_vec_full, config.threshold_rewrite, # 더 관대한 threshold ts ) if search_results_full and len(search_results_full) > 0: prior_rerank_score = top_scores[0] if top_scores else None candidates_full = [r["meta"] for r in search_results_full] # 재리랭킹 후 리랭커 점수끼리만 비교 (Qdrant 코사인과 혼용 금지) ( top_n_full, top_scores_full, rerank_used_full, rerank_info_full, ) = search_handler.rerank( full_context_query, candidates_full, ts ) full_rerank_score = ( top_scores_full[0] if top_scores_full else None ) if full_rerank_score is not None and ( prior_rerank_score is None or full_rerank_score > prior_rerank_score ): if prior_rerank_score is not None: print( f"[ask] {ts} ✅ Full Context 재검색 성공! " f"리랭커 점수 향상: " f"{prior_rerank_score:.2f} → {full_rerank_score:.2f}" ) else: print( f"[ask] {ts} ✅ Full Context 재검색 성공! " f"리랭커 점수: {full_rerank_score:.2f}" ) top_n_results = top_n_full top_scores = top_scores_full rerank_used = rerank_used_full rerank_info = rerank_info_full rewritten_query = full_context_query else: prior_label = ( f"{prior_rerank_score:.2f}" if prior_rerank_score is not None else "None" ) full_label = ( f"{full_rerank_score:.2f}" if full_rerank_score is not None else "None" ) print( f"[ask] {ts} ⚠️ Full Context 재검색 리랭커 점수 미개선 " f"({full_label} ≤ {prior_label})" ) else: print(f"[ask] {ts} ⚠️ Full Context 재검색 결과 없음") except Exception as e: print(f"[ask] {ts} ❌ Full Context Rewriting 실패: {e}") # ─── ④ LLM 답변 생성 (Messages Format) ───── # 감정 분석 emotion = emotion_detector.detect(original_query, ts) # 대화 이력 조회 (messages format) conversation_messages = [] if config.chat_history_always_include and CHAT_HISTORY_ENABLED and q.bot_id: try: conversation_messages = chat_manager.get_messages_for_llm( bot_id=q.bot_id, hours=config.chat_history_hours, max_conversations=config.chat_history_limit ) if conversation_messages: print(f"[ask] {ts} 🔄 대화 이력 포함 ({len(conversation_messages)//2}개 대화, bot_id={q.bot_id})") except Exception as e: print(f"[ask] {ts} 대화 이력 조회 실패: {e}") # 감정별 추가 지시사항 emotion_instruction = emotion_handler.get_emotion_instruction(emotion.primary) # Messages 프롬프트 생성 (표준 chat completion format + 감정 정보) messages = prompt_builder.build_answer_prompt_messages( original_query=original_query, rewritten_query=rewritten_query, references=top_n_results, scores=top_scores, conversation_history=conversation_messages, emotion_instruction=emotion_instruction, emotion_name=emotion.primary, domain_data=q.domain_data ) log_llm_messages(messages, ts, q.intent_type, domain_data_available) # LLM 답변 생성 (Messages Format) # LLM이 감정 정보를 받아 맥락에 맞게 공감하며 답변 생성 try: answer = llm_handler.generate_answer_from_messages(messages, ts) # ❌ 정해진 공감 메시지 제거 (LLM이 직접 맥락 파악하여 공감) # answer = emotion_handler.enhance_answer_with_empathy(...) # 낮은 신뢰도 시 대안 질문 제안 추가 answer = suggestion_handler.enhance_answer_with_suggestions( answer=answer, top_results=top_n_results, top_scores=top_scores, max_suggestions=3 ) except Exception as e: print(f"[ask] {ts} ❌ LLM 실패, 폴백 답변 사용: {e}") if top_n_results: answer = llm_handler.generate_fallback_answer(top_n_results[0]) else: answer = build_domain_data_fallback_answer(q.domain_data) # ─── ⑤ 저장 & 로깅 ───────────────────────── # 신뢰도 레벨 계산 confidence_level = suggestion_handler.get_confidence_level(top_scores[0] if top_scores else None) # MongoDB 저장 references = response_handler.build_references(top_n_results, top_scores) faq_urls = response_handler.extract_faq_urls(top_n_results) metadata = { "rerank_used": rerank_used, "faiss_top_k": config.top_k, "num_references": len(top_n_results), "references": references, "faq_urls": faq_urls, "emotion": emotion.primary, "emotion_intensity": emotion.intensity, "emotion_confidence": emotion.confidence, "answer_confidence": confidence_level, "top_score": top_scores[0] if top_scores else None, "domain_data_present": domain_data_available, "domain_data_only": domain_data_available and not top_n_results } if rewritten_query: metadata["query_rewritten"] = True metadata["original_query"] = original_query metadata["rewritten_query"] = rewritten_query response_handler.save_to_mongodb( bot_id=q.bot_id, user_query=original_query, ai_response=answer, matched_questions=[r["q"] for r in top_n_results], scores=top_scores, metadata=metadata, ts=ts ) # 로깅 response_handler.log_success( original_query, rewritten_query, top_n_results, top_scores, answer, ts ) response_handler.print_console_log( original_query, rewritten_query, top_n_results, top_scores, answer, ts ) # ─── ⑥ 응답 반환 ─────────────────────────── return response_handler.build_response( answer=answer, matched_questions=[r["q"] for r in top_n_results], scores=top_scores, bot_id=q.bot_id, references=references, faq_urls=faq_urls, rerank_info={ "used": rerank_used, "faiss_top_k": config.top_k, "rerank_top_n": len(top_n_results), "faiss_scores": [round(s, 4) for s in faiss_scores], "rerank_scores": top_scores, "detail": rerank_info if rerank_info else "Reranking not used" }, faiss_scores=faiss_scores ) @app.post("/agent/chat") def agent_chat(request: AgentChatRequest): """Tool-calling agent — RAG + domain tools(chatbotApi) 통합 응답.""" ts = datetime.now(timezone.utc).isoformat() print(f"\n[agent] {ts} 질문: {request.query} botId={request.bot_id}") result = agent_service.chat( request.query, bot_id=request.bot_id, pending_intent_type=request.pending_intent_type, pending_params=request.pending_params, ) print(f"[agent] {ts} routeType={result.get('routeType')} intent={result.get('intentType')}") return result @app.get("/health") def healthz(): """헬스체크 엔드포인트""" from api_clients import check_api_health api_health = check_api_health() vector_count = vector_store.count() all_ok = all(api_health.values()) and vector_count > 0 return { "status": "ok" if all_ok else "degraded", "vector_count": vector_count, "api_services": api_health } if __name__ == "__main__": import uvicorn uvicorn.run(app, host="0.0.0.0", port=28012)