chore(agent): uncommitted changes from task
This commit is contained in:
1 parent
3e7a2a03e7
commit
6dda9a13f4
2 files changed
+114
-5
No files matched your search
@@ -519,6 +519,79 @@ def _normalize_string_newlines(fragment: str) -> str:
|
||||
return "".join(out)
|
||||
|
||||
|
||||
def _strip_json_noise(fragment: str) -> str:
|
||||
"""Remove JS-style comments and trailing commas OUTSIDE string literals.
|
||||
|
||||
String-state-aware (same scanner idea as `_normalize_string_newlines`):
|
||||
`"… // …"` and `"…, }…"` inside JSON string values survive — the old blind
|
||||
regexes corrupted valid JSON into the silent one-liner fallback.
|
||||
"""
|
||||
out: list[str] = []
|
||||
in_str = False
|
||||
i = 0
|
||||
n = len(fragment)
|
||||
while i < n:
|
||||
ch = fragment[i]
|
||||
if in_str:
|
||||
if ch == "\\" and i + 1 < n:
|
||||
out.append(ch)
|
||||
out.append(fragment[i + 1])
|
||||
i += 2
|
||||
continue
|
||||
out.append(ch)
|
||||
if ch == '"':
|
||||
in_str = False
|
||||
i += 1
|
||||
continue
|
||||
if ch == '"':
|
||||
in_str = True
|
||||
out.append(ch)
|
||||
i += 1
|
||||
continue
|
||||
if (
|
||||
ch == "/"
|
||||
and i + 1 < n
|
||||
and fragment[i + 1] == "/"
|
||||
and (i == 0 or fragment[i - 1] != ":") # keep "http://"
|
||||
):
|
||||
while i < n and fragment[i] != "\n":
|
||||
i += 1 # drop comment, keep the newline
|
||||
continue
|
||||
if ch == "/" and i + 1 < n and fragment[i + 1] == "*":
|
||||
end = fragment.find("*/", i + 2)
|
||||
if end != -1:
|
||||
i = end + 2
|
||||
continue
|
||||
# unclosed block comment — keep as-is (legacy regex was a no-op)
|
||||
if ch == ",":
|
||||
j = i + 1
|
||||
while j < n:
|
||||
cj = fragment[j]
|
||||
if cj in " \t\r\n":
|
||||
j += 1
|
||||
elif (
|
||||
cj == "/"
|
||||
and j + 1 < n
|
||||
and fragment[j + 1] == "/"
|
||||
and (j == 0 or fragment[j - 1] != ":")
|
||||
):
|
||||
while j < n and fragment[j] != "\n":
|
||||
j += 1
|
||||
elif cj == "/" and j + 1 < n and fragment[j + 1] == "*":
|
||||
end = fragment.find("*/", j + 2)
|
||||
if end == -1:
|
||||
break
|
||||
j = end + 2
|
||||
else:
|
||||
break
|
||||
if j < n and fragment[j] in "}]":
|
||||
i += 1 # drop only the comma; keep following whitespace
|
||||
continue
|
||||
out.append(ch)
|
||||
i += 1
|
||||
return "".join(out)
|
||||
|
||||
|
||||
def parse_llm_json(text: str) -> dict[str, str] | None:
|
||||
if not text:
|
||||
return None
|
||||
@@ -527,10 +600,8 @@ def parse_llm_json(text: str) -> dict[str, str] | None:
|
||||
cleaned = re.sub(r"^```(?:json)?\s*", "", text.strip(), flags=re.I)
|
||||
cleaned = re.sub(r"\s*```\s*$", "", cleaned)
|
||||
# Tolerate small-model JSON noise (FE parser parity): // and /* */ comments,
|
||||
# trailing commas. Lookbehind keeps "http://" URLs intact.
|
||||
cleaned = re.sub(r"(?<!:)//[^\n]*", "", cleaned)
|
||||
cleaned = re.sub(r"/\*.*?\*/", "", cleaned, flags=re.S)
|
||||
cleaned = re.sub(r",(\s*[}\]])", r"\1", cleaned)
|
||||
# trailing commas — string-aware, see _strip_json_noise.
|
||||
cleaned = _strip_json_noise(cleaned)
|
||||
first = cleaned.find("{")
|
||||
last = cleaned.rfind("}")
|
||||
if first < 0 or last <= first:
|
||||
@@ -543,7 +614,13 @@ def parse_llm_json(text: str) -> dict[str, str] | None:
|
||||
return None
|
||||
parts = re.split(r"(?<=[.!?])\s+", prose, maxsplit=1)
|
||||
kesimpulan = parts[0].strip()
|
||||
insight = parts[1].strip() if len(parts) > 1 else kesimpulan
|
||||
# No sentence split → don't store the same text twice (FE shows both
|
||||
# fields); reuse the JSON-path message for a missing recommendation.
|
||||
insight = (
|
||||
parts[1].strip()
|
||||
if len(parts) > 1
|
||||
else "Model tidak mengembalikan rekomendasi eksplisit."
|
||||
)
|
||||
return {"kesimpulan": kesimpulan, "insight": insight}
|
||||
fragment = cleaned[first : last + 1]
|
||||
try:
|
||||
|
||||
@@ -122,6 +122,38 @@ class ParseLlmJsonTests(SimpleTestCase):
|
||||
self.assertIsNotNone(out)
|
||||
self.assertIn("FCR 1.70", out["kesimpulan"])
|
||||
|
||||
def test_line_comment_inside_string_value_survives(self):
|
||||
"""`//` inside a JSON string must not be stripped as a comment (review MEDIUM)."""
|
||||
raw = (
|
||||
'{"kesimpulan": "Pakan // revisi besok dikirim ke kandang 1.", '
|
||||
'"insight": "Manajemen pakan perlu penyesuaian rasio harian."}'
|
||||
)
|
||||
out = parse_llm_json(raw)
|
||||
self.assertIsNotNone(out)
|
||||
self.assertIn("// revisi besok", out["kesimpulan"])
|
||||
|
||||
def test_trailing_comma_lookalike_inside_string_value_survives(self):
|
||||
"""`, }` inside a JSON string must not be eaten by trailing-comma cleanup (review LOW)."""
|
||||
raw = (
|
||||
'{"kesimpulan": "Kandang 2, } blok selatan aman dari insiden.", '
|
||||
'"insight": "Tidak ada anomali lingkungan pada periode berjalan ini."}'
|
||||
)
|
||||
out = parse_llm_json(raw)
|
||||
self.assertIsNotNone(out)
|
||||
self.assertIn(", }", out["kesimpulan"])
|
||||
|
||||
def test_prose_without_sentence_split_not_duplicated(self):
|
||||
"""No `.!?` split: prose stored once as kesimpulan, insight must differ (review LOW)."""
|
||||
raw = (
|
||||
"FCR membaik ke 0,18 dan lebih baik dari standar CP 707 berkat "
|
||||
"manajemen pakan yang konsisten ditinjau setiap hari tanpa kendala "
|
||||
"berarti di kandang blok selatan periode berjalan ini"
|
||||
)
|
||||
out = parse_llm_json(raw)
|
||||
self.assertIsNotNone(out)
|
||||
self.assertIn("FCR membaik", out["kesimpulan"])
|
||||
self.assertNotEqual(out["insight"], out["kesimpulan"])
|
||||
|
||||
|
||||
class TopicWhitelistTests(SimpleTestCase):
|
||||
def test_unknown_topic_rejected_before_any_work(self):
|
||||
|
||||
Reference in new issue
Block a user