Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
139 changes: 105 additions & 34 deletions sidecar/local_ai_core/inference/parsers/conversational_logic.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,18 @@
# Component: conversational_logic.py
class ConversationalLogic(BaseDelegate):
@staticmethod
def _looks_user_role_confusion_answer(text: str) -> bool:
value = str(text or "").strip()
if not value:
return False
lowered = value.lower()
if lowered.startswith("오늘은") and lowered.endswith("?"):
if re.search(r"(추천해\s*줄래요|알려\s*줄래요|정해\s*줄래요|해\s*줄래요|해줄래요)\??$", value):
return True
if re.search(r"(추천해\s*줄래요|알려\s*줄래요|정해\s*줄래요|해\s*줄래요|해줄래요)\??$", value):
return True
return False
@staticmethod
def _is_action_request_query(query: str) -> bool:
lowered = str(query or "").strip().lower()
if not lowered:
Expand Down Expand Up @@ -309,7 +321,7 @@ def _generate_conversation_candidate(

# Model-native fallback: if we have a non-empty primary answer and no critical leak,
# prefer returning it over forcing quality-guard retry loops.
if answer and not leak_blocked:
if answer and not leak_blocked and not hard_issues:
return _ConversationCandidateResult(
answer=answer,
rewrite_used=rewrite_used,
Expand All @@ -326,40 +338,44 @@ def _generate_conversation_candidate(
),
)

last_resort = self._generate_last_resort_direct_answer(
engine=engine,
query=query,
response_language=response_language,
profile=profile,
mlx_model_path=mlx_model_path,
llama_model_path=llama_model_path,
max_tokens=max_tokens,
)
if last_resort and self._looks_conversational_answer(
last_resort,
response_language=response_language,
query=query,
):
final_answer = last_resort
question_count = self._question_sentence_count(final_answer)
recommendation_shape = (
"three_options"
if is_recommendation_query and self._looks_three_option_shape(final_answer)
else None
)
return _ConversationCandidateResult(
answer=final_answer,
rewrite_used=rewrite_used,
quality_repair_reason="|".join(
[item for item in [repair_reason, "last_resort_direct"] if item][:2]
),
repair_triggered=True,
repair_success=True,
leak_blocked=leak_blocked,
direct_first_applied=True,
question_count_after_postprocess=question_count,
recommendation_shape=recommendation_shape,
last_resort_enabled = str(
os.getenv("LOCAL_AI_CONVERSATION_LAST_RESORT_ENABLED", "0") or "0"
).strip().lower() in {"1", "true", "yes", "on"}
if last_resort_enabled:
last_resort = self._generate_last_resort_direct_answer(
engine=engine,
query=query,
response_language=response_language,
profile=profile,
mlx_model_path=mlx_model_path,
llama_model_path=llama_model_path,
max_tokens=max_tokens,
)
if last_resort and self._looks_conversational_answer(
last_resort,
response_language=response_language,
query=query,
):
final_answer = last_resort
question_count = self._question_sentence_count(final_answer)
recommendation_shape = (
"three_options"
if is_recommendation_query and self._looks_three_option_shape(final_answer)
else None
)
return _ConversationCandidateResult(
answer=final_answer,
rewrite_used=rewrite_used,
quality_repair_reason="|".join(
[item for item in [repair_reason, "last_resort_direct"] if item][:2]
),
repair_triggered=True,
repair_success=True,
leak_blocked=leak_blocked,
direct_first_applied=True,
question_count_after_postprocess=question_count,
recommendation_shape=recommendation_shape,
)

clarification_enabled = str(
os.getenv("LOCAL_AI_CONVERSATION_CLARIFICATION_FALLBACK_ENABLED", "0") or "0"
Expand Down Expand Up @@ -413,10 +429,17 @@ def _conversation_hard_issues(self, issues: list[str]) -> list[str]:
hard = {
"meta_leak",
"context_leak",
"clarification_template_leak",
"meta_only_ack",
"role_confusion",
"leading_fragment",
"pathological_repetition",
"query_echo_hard",
"stale_user_echo",
"language_mismatch",
"avoidable_clarification",
"continuation_artifact",
"comma_loop_artifact",
}
return [item for item in issues if item in hard]

Expand Down Expand Up @@ -597,10 +620,15 @@ def _conversation_quality_issues(self, *, query: str, answer: str, response_lang
if not cleaned:
return ["empty"]
issues: list[str] = []
lowered_clean = cleaned.lower()
if self._looks_instructional_meta_response(cleaned):
issues.append("meta_leak")
if self._contains_context_leak_phrase(cleaned):
issues.append("context_leak")
if "continuation:" in lowered_clean or ":pleaseprovidethetextyouwouldlikemetocontinue." in lowered_clean:
issues.append("continuation_artifact")
if re.search(r"(?:,\s*){5,}", cleaned):
issues.append("comma_loop_artifact")
if self._has_duplicate_sentences(cleaned):
issues.append("duplicate_sentence")
if self._has_pathological_repetition(cleaned):
Expand All @@ -614,6 +642,12 @@ def _conversation_quality_issues(self, *, query: str, answer: str, response_lang
issues.append("leading_fragment")
if self._looks_clarification_template_leak(cleaned):
issues.append("clarification_template_leak")
if self._looks_avoidable_clarification_answer(query=query, answer=cleaned):
issues.append("avoidable_clarification")
if re.match(r"^\s*한\s*번에\s*(?:바로\s*)?(?:본문만\s*)?(?:출력|정리|답변)\s*(?:할게요|해볼게요|드릴게요|해드릴게요)\.?\s*$", cleaned):
issues.append("meta_only_ack")
if self._looks_user_role_confusion_answer(cleaned):
issues.append("role_confusion")

if response_language == "ko":
ko_chars = len(re.findall(r"[가-힣]", cleaned))
Expand All @@ -630,11 +664,42 @@ def _conversation_quality_issues(self, *, query: str, answer: str, response_lang
issues.append("informal_tone")
return issues

@staticmethod
def _looks_avoidable_clarification_answer(*, query: str, answer: str) -> bool:
q = str(query or "").strip().lower()
a = str(answer or "").strip().lower()
if not q or not a:
return False
direct_task_request = bool(
any(token in q for token in ("정리해줘", "정리", "요약", "뽑아줘", "추천해줘", "알려줘"))
and any(token in q for token in ("3개", "세 개", "두 개", "한 줄", "짧게", "간단히"))
)
if not direct_task_request:
return False
clarification_markers = (
"알려주시면",
"말씀해주시면",
"구체적으로",
"어떤",
"무엇을",
"어떤 종류",
"더 알려",
"provide",
"tell me",
"which one",
)
if any(marker in a for marker in clarification_markers):
has_answer_shape = bool(re.search(r"(?:^|\n)\s*(?:1[.)]|-|\*)\s+", answer))
return not has_answer_shape
return False

@staticmethod
def _looks_leading_fragment(text: str) -> bool:
value = str(text or "").strip()
if not value:
return False
if re.match(r"^(?:으로는|로는|에는|에서는|와는|과는)\s+", value):
return True
return bool(
re.match(
r"^(?:께(?:서는|요)?|을|를|이|가|은|는|도)\s*(?:붙여드리겠습니다|도와드리겠습니다|안내해드리겠습니다|질문하신|오늘|집중)",
Expand All @@ -658,6 +723,10 @@ def _looks_clarification_template_leak(text: str) -> bool:
return True
if "please provide" in lowered and "1." in lowered and "2." in lowered:
return True
if "한 번에 바로 본문만 출력할게요" in value or "바로 본문만 출력할게요" in value:
return True
if re.match(r"^\s*한\s*번에\s*.*(?:답변|정리).*(?:드릴게요|해드릴게요)\.?\s*$", value):
return True
return False

def _korean_quality_issues(self, *, query: str, answer: str, response_language: str) -> list[str]:
Expand All @@ -672,6 +741,8 @@ def _korean_quality_issues(self, *, query: str, answer: str, response_language:
def _minimal_safe_conversation_answer(self, *, query: str, response_language: str) -> str:
lowered = (query or "").strip().lower()
if response_language == "ko":
if re.search(r"(내일|오늘).*(할\s*일|할일).*(3개|세\s*개)", query):
return "1. 내일 가장 중요한 일 1개를 먼저 끝내기.\n2. 25분 집중 + 5분 휴식으로 두 번 진행하기.\n3. 마감 10분 전에 결과 점검하고 정리하기."
if any(token in lowered for token in ("뭐 먹", "메뉴", "점심", "저녁", "아침", "먹을")):
return "지금은 속이 편한 메뉴 한 가지(국밥, 죽, 비빔밥 중 하나)로 고르는 게 가장 무난해요."
if any(token in lowered for token in ("몇 시", "몇시", "자야", "수면", "잠")):
Expand Down
77 changes: 73 additions & 4 deletions sidecar/local_ai_core/inference/parsers/result_sanitizer.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,52 @@

# Component: result_sanitizer.py
class ResultSanitizer(BaseDelegate):
@staticmethod
def _strip_query_echo_prefix(text: str, query: str) -> str:
value = str(text or "").strip()
q = str(query or "").strip()
if not value or not q:
return value
q_compact = re.sub(r"\s+", "", q)
if len(q_compact) < 6:
return value
# Remove exact query echo at the beginning, but keep any useful suffix
# that follows on the same line.
patterns = [
rf"^\s*{re.escape(q)}\s*[::,\-]?\s*",
rf"^\s*{re.escape(q_compact)}\s*[::,\-]?\s*",
]
for pattern in patterns:
updated = re.sub(pattern, "", value)
if updated != value and updated.strip():
return updated.strip()
# Handle first-line echo without dropping the whole first line.
lines = value.split("\n")
if lines:
first = lines[0].strip()
first_compact = re.sub(r"\s+", "", first)
if first_compact.startswith(q_compact):
suffix = first_compact[len(q_compact):].strip(" ::,-")
if suffix:
lines[0] = suffix
return "\n".join([ln for ln in lines if str(ln).strip()]).strip()
if len(lines) > 1:
return "\n".join([ln for ln in lines[1:] if str(ln).strip()]).strip()
return value

@staticmethod
def _collapse_punctuation_loops(text: str) -> str:
value = str(text or "").strip()
if not value:
return ""
# Collapse runaway comma sequences like ",,,,,,," into a single comma.
value = re.sub(r"(?:,\s*){3,}", ", ", value)
# Normalize repeated terminal punctuation.
value = re.sub(r"([.!?])(?:\s*\1){2,}", r"\1", value)
# If content ends with a dangling comma, finish as a sentence for UX.
value = re.sub(r",\s*$", ".", value)
value = re.sub(r"\s{2,}", " ", value).strip()
return value

@staticmethod
def _normalize_korean_leading_address(text: str) -> str:
Expand All @@ -24,6 +70,15 @@ def _normalize_korean_leading_address(text: str) -> str:
# model artifact and degrades conversational tone.
value = re.sub(r"^\s*께서는\s+", "", value).strip()
value = re.sub(r"^\s*당신은\s+", "", value).strip()
# Repair common broken Korean leading fragments.
if re.match(r"^\s*로는\s+", value):
value = re.sub(r"^\s*로는\s+", "제 기준으로는 ", value).strip()
elif re.match(r"^\s*으로는\s+", value):
value = re.sub(r"^\s*으로는\s+", "제 기준으로는 ", value).strip()
elif re.match(r"^\s*에는\s+", value):
value = re.sub(r"^\s*에는\s+", "이 경우에는 ", value).strip()
elif re.match(r"^\s*에서는\s+", value):
value = re.sub(r"^\s*에서는\s+", "이 상황에서는 ", value).strip()
return value
@staticmethod
def _heuristic_korean_spacing(text: str) -> str:
Expand Down Expand Up @@ -195,6 +250,17 @@ def _normalize_three_option_recommendation(self, answer: str, *, response_langua
return "\n".join(lines).strip()

def _postprocess_conversational_answer(self, answer: str, *, query: str, response_language: str) -> str:
def _strip_meta_preamble(value: str) -> str:
text = str(value or "").strip()
if not text:
return ""
text = re.sub(
r"^\s*한\s*번에\s*(?:바로\s*)?(?:본문만\s*출력|실행\s*가능한\s*답변|딱\s*맞는\s*답변?)\s*을?\s*(?:드릴게요|해드릴게요)\.?\s*",
"",
text,
).strip()
return text

raw_pass_mode = str(os.getenv("LOCAL_AI_CONVERSATION_RAW_PASS_ENABLED", "0") or "0").strip().lower() in {
"1", "true", "yes", "on"
}
Expand Down Expand Up @@ -228,13 +294,11 @@ def _postprocess_conversational_answer(self, answer: str, *, query: str, respons
text = re.sub(r"(?im)^\s*(?:user|assistant|사용자|어시스턴트)\s*[::]\s*", "", text).strip()
# Guard: if the answer is a near-verbatim copy of recent history,
# find and remove the echoed prefix (up to 120 chars).
if "\n" in text:
first_line = text.split("\n")[0].strip()
if len(first_line) >= 8 and self._is_hard_query_echo(first_line, query):
text = "\n".join(text.split("\n")[1:]).strip()
text = self._strip_query_echo_prefix(text, query)
# Keep model-native wording; only reject obvious hard prompt echo.
if len(re.sub(r"\s+", "", query or "")) >= 10 and self._is_hard_query_echo(text, query):
return ""
text = _strip_meta_preamble(text)
return self._normalize_korean_leading_address(text)

light_postprocess = str(os.getenv("LOCAL_AI_CONVERSATION_LIGHT_POSTPROCESS", "0") or "0").strip().lower() in {
Expand Down Expand Up @@ -299,6 +363,7 @@ def _postprocess_conversational_answer(self, answer: str, *, query: str, respons
if ko_chars >= 24 and len(re.findall(r"\s+", fallback)) <= 1 and len(fallback) >= 40:
fallback = self._heuristic_korean_spacing(fallback)
return self._normalize_korean_leading_address(fallback[:320].strip())
text = _strip_meta_preamble(text)
return self._normalize_korean_leading_address(text)

minimal_mode = str(os.getenv("LOCAL_AI_MINIMAL_POSTPROCESS", "0") or "0").strip().lower() in {
Expand Down Expand Up @@ -371,6 +436,7 @@ def _postprocess_conversational_answer(self, answer: str, *, query: str, respons
if en_words >= 8 and ko_chars < 6:
return ""
text = re.sub(r"\s{2,}", " ", text).strip()
text = _strip_meta_preamble(text)
return self._normalize_korean_leading_address(text)

text = (answer or "").strip()
Expand All @@ -386,6 +452,7 @@ def _postprocess_conversational_answer(self, answer: str, *, query: str, respons
return text
text = re.sub(r"\.\s*입니다\.$", ".", text)
text = re.sub(r"\s{2,}", " ", text).strip()
text = self._collapse_punctuation_loops(text)
text = re.sub(r"(?im)^\s*(?:최종\s*답변|final\s*answer)\s*[::]\s*", "", text).strip()
text = re.sub(r"(?i)\bokay,\s*let\s*me\s*process\s*this\.?\s*", "", text).strip()
text = re.sub(r"(?i)\bthat'?s\s*straightforward\.?\s*", "", text).strip()
Expand All @@ -404,6 +471,8 @@ def _postprocess_conversational_answer(self, answer: str, *, query: str, respons
text = re.sub(r"최대한\s*짧고\s*명확하게\s*답하세요\.?\s*", "", text).strip()
text = self._dedupe_conversation_sentences(text)
text = self._limit_question_sentences(text, max_questions=1)
text = self._collapse_punctuation_loops(text)
text = _strip_meta_preamble(text)
if not text:
return ""

Expand Down
10 changes: 5 additions & 5 deletions sidecar/local_ai_core/memory/service_mixins.py
Original file line number Diff line number Diff line change
Expand Up @@ -1142,13 +1142,13 @@ def _should_drop_assistant_digest_text(cls, *, assistant_text: str, user_query:
return True
if cls._looks_like_instruction_leak(cleaned):
return True
if cls._has_duplicate_sentence(cleaned):
return True
# Prefer preserving assistant turns in digest unless corruption is severe.
if cls._is_high_repetition_text(cleaned):
return True
if cls._looks_like_open_loop(cleaned):
return True
if cls._token_overlap(cleaned, user_query) >= 0.82 and len(cleaned) <= 220:
# Keep conversationally open assistant turns in digest so
# short follow-ups (e.g., "방금 답변 한 줄로") can summarize
# the actual immediately previous response instead of stale context.
if cls._token_overlap(cleaned, user_query) >= 0.92 and len(cleaned) <= 120:
return True
if cls._contains_context_leak_phrase(cleaned):
return True
Expand Down
Loading
Loading