Files
dashboard-cpsp/scripts/extract_iot_cycle.py
T
2026-08-26 13:46:05 +07:00

138 lines
4.9 KiB
Python

"""Extract IoT panel rows for Sukawarna cycle 2026-05-22 .. 2026-07-09."""
import json
import urllib.request
from collections import defaultdict
from datetime import datetime, timedelta, timezone
BASE = "https://dashboard.cpsp.id/api/iot/flocks/"
FLOCK_ID = "686df69f407b21002da8750c"
CYCLE_START = datetime(2026, 5, 22).date()
CYCLE_END = datetime(2026, 7, 9).date()
TZ = timezone(timedelta(hours=7))
def parse_dt(value: str) -> datetime:
return datetime.fromisoformat(value.replace("Z", "+00:00"))
def fetch_page(page: int) -> dict:
with urllib.request.urlopen(f"{BASE}?page={page}", timeout=60) as resp:
return json.loads(resp.read())
def to_panel_fields(d: dict) -> dict:
"""Map API payload.data -> iot_panel columns (no experience_temperature)."""
hum = d.get("humidity") or 0
temp = d.get("actualTemperature") or 0
return {
"wind_speed": float(d.get("wind") or 0),
"humidity": float(hum / 10 if hum > 100 else hum),
"water_total": float(d.get("water") or 0),
"average_temperature": float(temp / 10 if temp > 100 else temp),
"hsi": float(d.get("HSI") or 0),
"cycle_day": int(d.get("day") or 0),
}
def main() -> None:
p1 = fetch_page(1)
total = p1["count"]
page_size = len(p1["results"])
last_page = (total + page_size - 1) // page_size
print(f"Scanning {last_page} pages for flock {FLOCK_ID}")
print(f"Cycle window: {CYCLE_START} .. {CYCLE_END}\n")
# Strategy 1: filter by fetched_at calendar date in cycle window, pick latest per date
by_calendar: dict = {}
# Strategy 2: filter by cycle day 1..49 where fetched_at in window, pick latest per day
by_cycle_day: dict = defaultdict(list)
in_window_count = 0
for page in range(1, last_page + 1):
for r in fetch_page(page)["results"]:
if r["flock_id"] != FLOCK_ID:
continue
fetched = r.get("fetched_at")
if not fetched:
continue
cal = parse_dt(fetched).astimezone(TZ).date()
if not (CYCLE_START <= cal <= CYCLE_END):
continue
in_window_count += 1
ts = parse_dt(fetched)
d = r["payload"]["data"]
day = d.get("day")
prev = by_calendar.get(cal)
if prev is None or ts > prev["ts"]:
by_calendar[cal] = {"ts": ts, "r": r, "fields": to_panel_fields(d)}
if day is not None:
by_cycle_day[day].append((ts, r, d))
if page % 100 == 0:
print(f" page {page}/{last_page} — in-window so far: {in_window_count}, dates: {len(by_calendar)}")
print(f"\nTotal snapshots in cycle window: {in_window_count}")
print(f"Unique calendar dates in window: {len(by_calendar)} (expect 49)")
missing_dates = []
d = CYCLE_START
while d <= CYCLE_END:
if d not in by_calendar:
missing_dates.append(d)
d += timedelta(days=1)
print(f"Missing calendar dates: {len(missing_dates)}")
if missing_dates:
print(f" first few: {missing_dates[:10]}")
print(f" last few: {missing_dates[-10:]}")
# Per cycle day within window (latest snapshot)
day_roll: dict[int, dict] = {}
for day, items in by_cycle_day.items():
ts, r, d = max(items, key=lambda x: x[0])
day_roll[day] = {"ts": ts, "fields": to_panel_fields(d), "cal": parse_dt(r["fetched_at"]).astimezone(TZ).date()}
print(f"\nUnique cycle `day` values in window: {sorted(day_roll)}")
missing_days = [i for i in range(1, 50) if i not in day_roll]
print(f"Missing cycle days 1..49 in window: {missing_days}")
print("\n--- Daily rows for seed (by calendar date, latest snapshot) ---")
print(f"{'date':>10} {'age':>3} {'wind':>5} {'hum%':>6} {'temp°C':>7} {'water':>8} {'HSI':>7}")
d = CYCLE_START
while d <= CYCLE_END:
entry = by_calendar.get(d)
if entry:
f = entry["fields"]
print(
f"{d} {f['cycle_day']:3d} {f['wind_speed']:5.0f} {f['humidity']:6.1f} "
f"{f['average_temperature']:7.1f} {f['water_total']:8.0f} {f['hsi']:7.2f}"
)
else:
age = (d - CYCLE_START).days + 1
print(f"{d} {age:3d} — no API data —")
d += timedelta(days=1)
out = Path(__file__).resolve().parent / "iot_panel_cycle_extract.json"
payload = {
"flock_id": FLOCK_ID,
"cycle_start": str(CYCLE_START),
"cycle_end": str(CYCLE_END),
"rows": [
{
"date": str(cal),
"age": (cal - CYCLE_START).days + 1,
**by_calendar[cal]["fields"],
}
for cal in sorted(by_calendar)
],
}
out.write_text(json.dumps(payload, indent=2), encoding="utf-8")
print(f"\nWrote {len(payload['rows'])} rows to {out}")
from pathlib import Path
if __name__ == "__main__":
main()