chore(agent): uncommitted changes from task
This commit is contained in:
1 parent
3e7a2a03e7
commit
6dda9a13f4
2 files changed
+114
-5
No files matched your search
@@ -519,6 +519,79 @@ def _normalize_string_newlines(fragment: str) -> str:
|
|||||||
return "".join(out)
|
return "".join(out)
|
||||||
|
|
||||||
|
|
||||||
|
def _strip_json_noise(fragment: str) -> str:
|
||||||
|
"""Remove JS-style comments and trailing commas OUTSIDE string literals.
|
||||||
|
|
||||||
|
String-state-aware (same scanner idea as `_normalize_string_newlines`):
|
||||||
|
`"… // …"` and `"…, }…"` inside JSON string values survive — the old blind
|
||||||
|
regexes corrupted valid JSON into the silent one-liner fallback.
|
||||||
|
"""
|
||||||
|
out: list[str] = []
|
||||||
|
in_str = False
|
||||||
|
i = 0
|
||||||
|
n = len(fragment)
|
||||||
|
while i < n:
|
||||||
|
ch = fragment[i]
|
||||||
|
if in_str:
|
||||||
|
if ch == "\\" and i + 1 < n:
|
||||||
|
out.append(ch)
|
||||||
|
out.append(fragment[i + 1])
|
||||||
|
i += 2
|
||||||
|
continue
|
||||||
|
out.append(ch)
|
||||||
|
if ch == '"':
|
||||||
|
in_str = False
|
||||||
|
i += 1
|
||||||
|
continue
|
||||||
|
if ch == '"':
|
||||||
|
in_str = True
|
||||||
|
out.append(ch)
|
||||||
|
i += 1
|
||||||
|
continue
|
||||||
|
if (
|
||||||
|
ch == "/"
|
||||||
|
and i + 1 < n
|
||||||
|
and fragment[i + 1] == "/"
|
||||||
|
and (i == 0 or fragment[i - 1] != ":") # keep "http://"
|
||||||
|
):
|
||||||
|
while i < n and fragment[i] != "\n":
|
||||||
|
i += 1 # drop comment, keep the newline
|
||||||
|
continue
|
||||||
|
if ch == "/" and i + 1 < n and fragment[i + 1] == "*":
|
||||||
|
end = fragment.find("*/", i + 2)
|
||||||
|
if end != -1:
|
||||||
|
i = end + 2
|
||||||
|
continue
|
||||||
|
# unclosed block comment — keep as-is (legacy regex was a no-op)
|
||||||
|
if ch == ",":
|
||||||
|
j = i + 1
|
||||||
|
while j < n:
|
||||||
|
cj = fragment[j]
|
||||||
|
if cj in " \t\r\n":
|
||||||
|
j += 1
|
||||||
|
elif (
|
||||||
|
cj == "/"
|
||||||
|
and j + 1 < n
|
||||||
|
and fragment[j + 1] == "/"
|
||||||
|
and (j == 0 or fragment[j - 1] != ":")
|
||||||
|
):
|
||||||
|
while j < n and fragment[j] != "\n":
|
||||||
|
j += 1
|
||||||
|
elif cj == "/" and j + 1 < n and fragment[j + 1] == "*":
|
||||||
|
end = fragment.find("*/", j + 2)
|
||||||
|
if end == -1:
|
||||||
|
break
|
||||||
|
j = end + 2
|
||||||
|
else:
|
||||||
|
break
|
||||||
|
if j < n and fragment[j] in "}]":
|
||||||
|
i += 1 # drop only the comma; keep following whitespace
|
||||||
|
continue
|
||||||
|
out.append(ch)
|
||||||
|
i += 1
|
||||||
|
return "".join(out)
|
||||||
|
|
||||||
|
|
||||||
def parse_llm_json(text: str) -> dict[str, str] | None:
|
def parse_llm_json(text: str) -> dict[str, str] | None:
|
||||||
if not text:
|
if not text:
|
||||||
return None
|
return None
|
||||||
@@ -527,10 +600,8 @@ def parse_llm_json(text: str) -> dict[str, str] | None:
|
|||||||
cleaned = re.sub(r"^```(?:json)?\s*", "", text.strip(), flags=re.I)
|
cleaned = re.sub(r"^```(?:json)?\s*", "", text.strip(), flags=re.I)
|
||||||
cleaned = re.sub(r"\s*```\s*$", "", cleaned)
|
cleaned = re.sub(r"\s*```\s*$", "", cleaned)
|
||||||
# Tolerate small-model JSON noise (FE parser parity): // and /* */ comments,
|
# Tolerate small-model JSON noise (FE parser parity): // and /* */ comments,
|
||||||
# trailing commas. Lookbehind keeps "http://" URLs intact.
|
# trailing commas — string-aware, see _strip_json_noise.
|
||||||
cleaned = re.sub(r"(?<!:)//[^\n]*", "", cleaned)
|
cleaned = _strip_json_noise(cleaned)
|
||||||
cleaned = re.sub(r"/\*.*?\*/", "", cleaned, flags=re.S)
|
|
||||||
cleaned = re.sub(r",(\s*[}\]])", r"\1", cleaned)
|
|
||||||
first = cleaned.find("{")
|
first = cleaned.find("{")
|
||||||
last = cleaned.rfind("}")
|
last = cleaned.rfind("}")
|
||||||
if first < 0 or last <= first:
|
if first < 0 or last <= first:
|
||||||
@@ -543,7 +614,13 @@ def parse_llm_json(text: str) -> dict[str, str] | None:
|
|||||||
return None
|
return None
|
||||||
parts = re.split(r"(?<=[.!?])\s+", prose, maxsplit=1)
|
parts = re.split(r"(?<=[.!?])\s+", prose, maxsplit=1)
|
||||||
kesimpulan = parts[0].strip()
|
kesimpulan = parts[0].strip()
|
||||||
insight = parts[1].strip() if len(parts) > 1 else kesimpulan
|
# No sentence split → don't store the same text twice (FE shows both
|
||||||
|
# fields); reuse the JSON-path message for a missing recommendation.
|
||||||
|
insight = (
|
||||||
|
parts[1].strip()
|
||||||
|
if len(parts) > 1
|
||||||
|
else "Model tidak mengembalikan rekomendasi eksplisit."
|
||||||
|
)
|
||||||
return {"kesimpulan": kesimpulan, "insight": insight}
|
return {"kesimpulan": kesimpulan, "insight": insight}
|
||||||
fragment = cleaned[first : last + 1]
|
fragment = cleaned[first : last + 1]
|
||||||
try:
|
try:
|
||||||
|
|||||||
@@ -122,6 +122,38 @@ class ParseLlmJsonTests(SimpleTestCase):
|
|||||||
self.assertIsNotNone(out)
|
self.assertIsNotNone(out)
|
||||||
self.assertIn("FCR 1.70", out["kesimpulan"])
|
self.assertIn("FCR 1.70", out["kesimpulan"])
|
||||||
|
|
||||||
|
def test_line_comment_inside_string_value_survives(self):
|
||||||
|
"""`//` inside a JSON string must not be stripped as a comment (review MEDIUM)."""
|
||||||
|
raw = (
|
||||||
|
'{"kesimpulan": "Pakan // revisi besok dikirim ke kandang 1.", '
|
||||||
|
'"insight": "Manajemen pakan perlu penyesuaian rasio harian."}'
|
||||||
|
)
|
||||||
|
out = parse_llm_json(raw)
|
||||||
|
self.assertIsNotNone(out)
|
||||||
|
self.assertIn("// revisi besok", out["kesimpulan"])
|
||||||
|
|
||||||
|
def test_trailing_comma_lookalike_inside_string_value_survives(self):
|
||||||
|
"""`, }` inside a JSON string must not be eaten by trailing-comma cleanup (review LOW)."""
|
||||||
|
raw = (
|
||||||
|
'{"kesimpulan": "Kandang 2, } blok selatan aman dari insiden.", '
|
||||||
|
'"insight": "Tidak ada anomali lingkungan pada periode berjalan ini."}'
|
||||||
|
)
|
||||||
|
out = parse_llm_json(raw)
|
||||||
|
self.assertIsNotNone(out)
|
||||||
|
self.assertIn(", }", out["kesimpulan"])
|
||||||
|
|
||||||
|
def test_prose_without_sentence_split_not_duplicated(self):
|
||||||
|
"""No `.!?` split: prose stored once as kesimpulan, insight must differ (review LOW)."""
|
||||||
|
raw = (
|
||||||
|
"FCR membaik ke 0,18 dan lebih baik dari standar CP 707 berkat "
|
||||||
|
"manajemen pakan yang konsisten ditinjau setiap hari tanpa kendala "
|
||||||
|
"berarti di kandang blok selatan periode berjalan ini"
|
||||||
|
)
|
||||||
|
out = parse_llm_json(raw)
|
||||||
|
self.assertIsNotNone(out)
|
||||||
|
self.assertIn("FCR membaik", out["kesimpulan"])
|
||||||
|
self.assertNotEqual(out["insight"], out["kesimpulan"])
|
||||||
|
|
||||||
|
|
||||||
class TopicWhitelistTests(SimpleTestCase):
|
class TopicWhitelistTests(SimpleTestCase):
|
||||||
def test_unknown_topic_rejected_before_any_work(self):
|
def test_unknown_topic_rejected_before_any_work(self):
|
||||||
|
|||||||
Reference in new issue
Block a user