chore(agent): uncommitted changes from task

This commit is contained in:
Alberto-Audrix committed 2026-09-23 16:13:31 +07:00
1 parent 3e7a2a03e7
commit 6dda9a13f4
2 files changed
+114 -5

No files matched your search

@@ -519,6 +519,79 @@ def _normalize_string_newlines(fragment: str) -> str:
return "".join(out)
def _strip_json_noise(fragment: str) -> str:
"""Remove JS-style comments and trailing commas OUTSIDE string literals.
String-state-aware (same scanner idea as `_normalize_string_newlines`):
`"… // …"` and `"…, }…"` inside JSON string values survive — the old blind
regexes corrupted valid JSON into the silent one-liner fallback.
"""
out: list[str] = []
in_str = False
i = 0
n = len(fragment)
while i < n:
ch = fragment[i]
if in_str:
if ch == "\\" and i + 1 < n:
out.append(ch)
out.append(fragment[i + 1])
i += 2
continue
out.append(ch)
if ch == '"':
in_str = False
i += 1
continue
if ch == '"':
in_str = True
out.append(ch)
i += 1
continue
if (
ch == "/"
and i + 1 < n
and fragment[i + 1] == "/"
and (i == 0 or fragment[i - 1] != ":") # keep "http://"
):
while i < n and fragment[i] != "\n":
i += 1 # drop comment, keep the newline
continue
if ch == "/" and i + 1 < n and fragment[i + 1] == "*":
end = fragment.find("*/", i + 2)
if end != -1:
i = end + 2
continue
# unclosed block comment — keep as-is (legacy regex was a no-op)
if ch == ",":
j = i + 1
while j < n:
cj = fragment[j]
if cj in " \t\r\n":
j += 1
elif (
cj == "/"
and j + 1 < n
and fragment[j + 1] == "/"
and (j == 0 or fragment[j - 1] != ":")
):
while j < n and fragment[j] != "\n":
j += 1
elif cj == "/" and j + 1 < n and fragment[j + 1] == "*":
end = fragment.find("*/", j + 2)
if end == -1:
break
j = end + 2
else:
break
if j < n and fragment[j] in "}]":
i += 1 # drop only the comma; keep following whitespace
continue
out.append(ch)
i += 1
return "".join(out)
def parse_llm_json(text: str) -> dict[str, str] | None:
if not text:
return None
@@ -527,10 +600,8 @@ def parse_llm_json(text: str) -> dict[str, str] | None:
cleaned = re.sub(r"^```(?:json)?\s*", "", text.strip(), flags=re.I)
cleaned = re.sub(r"\s*```\s*$", "", cleaned)
# Tolerate small-model JSON noise (FE parser parity): // and /* */ comments,
# trailing commas. Lookbehind keeps "http://" URLs intact.
cleaned = re.sub(r"(?<!:)//[^\n]*", "", cleaned)
cleaned = re.sub(r"/\*.*?\*/", "", cleaned, flags=re.S)
cleaned = re.sub(r",(\s*[}\]])", r"\1", cleaned)
# trailing commas — string-aware, see _strip_json_noise.
cleaned = _strip_json_noise(cleaned)
first = cleaned.find("{")
last = cleaned.rfind("}")
if first < 0 or last <= first:
@@ -543,7 +614,13 @@ def parse_llm_json(text: str) -> dict[str, str] | None:
return None
parts = re.split(r"(?<=[.!?])\s+", prose, maxsplit=1)
kesimpulan = parts[0].strip()
insight = parts[1].strip() if len(parts) > 1 else kesimpulan
# No sentence split → don't store the same text twice (FE shows both
# fields); reuse the JSON-path message for a missing recommendation.
insight = (
parts[1].strip()
if len(parts) > 1
else "Model tidak mengembalikan rekomendasi eksplisit."
)
return {"kesimpulan": kesimpulan, "insight": insight}
fragment = cleaned[first : last + 1]
try:
@@ -122,6 +122,38 @@ class ParseLlmJsonTests(SimpleTestCase):
self.assertIsNotNone(out)
self.assertIn("FCR 1.70", out["kesimpulan"])
def test_line_comment_inside_string_value_survives(self):
"""`//` inside a JSON string must not be stripped as a comment (review MEDIUM)."""
raw = (
'{"kesimpulan": "Pakan // revisi besok dikirim ke kandang 1.", '
'"insight": "Manajemen pakan perlu penyesuaian rasio harian."}'
)
out = parse_llm_json(raw)
self.assertIsNotNone(out)
self.assertIn("// revisi besok", out["kesimpulan"])
def test_trailing_comma_lookalike_inside_string_value_survives(self):
"""`, }` inside a JSON string must not be eaten by trailing-comma cleanup (review LOW)."""
raw = (
'{"kesimpulan": "Kandang 2, } blok selatan aman dari insiden.", '
'"insight": "Tidak ada anomali lingkungan pada periode berjalan ini."}'
)
out = parse_llm_json(raw)
self.assertIsNotNone(out)
self.assertIn(", }", out["kesimpulan"])
def test_prose_without_sentence_split_not_duplicated(self):
"""No `.!?` split: prose stored once as kesimpulan, insight must differ (review LOW)."""
raw = (
"FCR membaik ke 0,18 dan lebih baik dari standar CP 707 berkat "
"manajemen pakan yang konsisten ditinjau setiap hari tanpa kendala "
"berarti di kandang blok selatan periode berjalan ini"
)
out = parse_llm_json(raw)
self.assertIsNotNone(out)
self.assertIn("FCR membaik", out["kesimpulan"])
self.assertNotEqual(out["insight"], out["kesimpulan"])
class TopicWhitelistTests(SimpleTestCase):
def test_unknown_topic_rejected_before_any_work(self):