Files
dashboard-cpsp/scripts/extract_iot_10min.py
T

97 lines
3.5 KiB
Python

"""Extract real 10-minute IoT snapshots for the Sukawarna Kandang 2 cycle window.
The public IoT API (dashboard.cpsp.id/api/iot/flocks/) polls the coop every 10
minutes, so a full cycle day holds up to 144 rows. This script scans the API's
history for flock 686df69f407b21002da8750c (Kandang 2 Lantai 1) inside the
seeded cycle window 2026-05-22 .. 2026-07-09 and writes every snapshot to
``iot_panel_10min_extract.json``. The Django seed command reads this cache so
Kandang 2 Lantai 1 is seeded from real sensor readings; the other floors and
Kandang 1 are generated from it.
"""
import json
import sys
import urllib.request
from datetime import datetime, timedelta, timezone
from pathlib import Path
SCRIPT_DIR = Path(__file__).resolve().parent
sys.path.insert(0, str(SCRIPT_DIR))
from experience_temperature import LANTAI_1_IOT_FLOCK_ID, panel_fields_from_payload
BASE = "https://dashboard.cpsp.id/api/iot/flocks/"
FLOCK_ID = LANTAI_1_IOT_FLOCK_ID
CYCLE_START = datetime(2026, 5, 22).date()
CYCLE_END = datetime(2026, 7, 9).date()
TZ = timezone(timedelta(hours=7))
OUT = SCRIPT_DIR / "iot_panel_10min_extract.json"
def parse_dt(value: str) -> datetime:
return datetime.fromisoformat(value.replace("Z", "+00:00"))
def fetch_page(page: int) -> dict:
with urllib.request.urlopen(f"{BASE}?page={page}", timeout=60) as resp:
return json.loads(resp.read())
def main() -> None:
p1 = fetch_page(1)
last_page = (p1["count"] + len(p1["results"]) - 1) // len(p1["results"])
hi = min(last_page, 320)
print(f"Scanning pages 1..{hi} (of {last_page}) for flock {FLOCK_ID}", flush=True)
# key: (date, slot_index) -> mapped row (keep the latest snapshot per 10-min slot)
rows_by_slot: dict[tuple[str, int], dict] = {}
in_window = 0
for page in range(1, hi + 1):
for r in fetch_page(page)["results"]:
if r["flock_id"] != FLOCK_ID:
continue
fetched = r.get("fetched_at")
if not fetched:
continue
ts = parse_dt(fetched).astimezone(TZ)
cal = ts.date()
if not (CYCLE_START <= cal <= CYCLE_END):
continue
in_window += 1
d = r["payload"]["data"]
age = (cal - CYCLE_START).days + 1
fields = panel_fields_from_payload(d, age)
slot = (ts.hour * 60 + ts.minute) // 10
key = (str(cal), slot)
rows_by_slot[key] = {
"timestamp": ts.isoformat(),
"date": str(cal),
"age": age,
**fields,
}
if page % 50 == 0:
print(f" page {page}/{hi} — in-window so far: {in_window}", flush=True)
rows = [rows_by_slot[k] for k in sorted(rows_by_slot, key=lambda k: (k[0], k[1]))]
out = {
"flock_id": FLOCK_ID,
"cycle_start": str(CYCLE_START),
"cycle_end": str(CYCLE_END),
"count": len(rows),
"rows": rows,
}
OUT.write_text(json.dumps(out, indent=2), encoding="utf-8")
per_date: dict[str, int] = {}
for row in rows:
per_date[row["date"]] = per_date.get(row["date"], 0) + 1
print(f"\nSnapshots in window: {in_window}; unique slots written: {len(rows)}")
print(f"Unique dates: {len(per_date)} (expect 49)")
counts = sorted(set(per_date.values()))
print(f"Rows-per-date distribution: {counts}")
low = [d for d, c in per_date.items() if c < 144]
print(f"Dates with fewer than 144 rows ({len(low)}): {low[:20]}")
print(f"Wrote {OUT}")
if __name__ == "__main__":
main()