"""When each recording actually happened, and which working day it belongs to. The archive's folders are wrong about both. `2026-08-13/batch001.mp4` was recorded at 00:07, which under a 06:00-to-06:00 shift belongs to the working day of 2026-08-12 — and `2026-08-07/batch4.mp4` was recorded the previous evening entirely. Sampling the archive, roughly a quarter of the files land on a different working day once their real timestamp is read (REQ-160…163). Nothing on disk is touched. The archive is mounted read-only and is the user's own data; this builds an index beside it instead, and every page groups and orders by the index rather than by the folder name. The original path stays the file's identity, so results already recorded against it survive. """ import os import time from typing import List, Optional from backend import db, jobs, library, projects, video_clock CUTOFF_HOUR = 6 """A working day runs 06:00 to 06:00 (REQ-161).""" MIN_CONFIDENCE = 0.10 MIN_AGREEING = 2 """Below either of these a reading is kept but flagged: it is a guess, not a measurement, and one misread digit is what puts a recording on the wrong day.""" class ArchiveIndexError(Exception): pass def _trusted(confidence: float, agreeing: int) -> bool: return confidence >= MIN_CONFIDENCE and agreeing >= MIN_AGREEING def store(project_id: int, video_rel: str, started_at: Optional[str], confidence: float = 0.0, agreeing: int = 0, source: str = "ocr", error: str = "") -> None: """`started_at` is wall-clock text, 'YYYY-MM-DD HH:MM:SS'. Never an epoch. The overlay has no timezone, so converting it to one makes the answer depend on which timezone the process happens to run in — the backend container is UTC and the browser is not. """ working = "" if started_at: working = video_clock.working_day(_as_datetime(started_at), CUTOFF_HOUR) with db.cursor() as cur: cur.execute( """INSERT INTO video_clock (project_id, video_rel, folder_date, started_at, working_day, confidence, agreeing, source, error, read_at) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?) ON CONFLICT(project_id, video_rel) DO UPDATE SET started_at = excluded.started_at, working_day = excluded.working_day, confidence = excluded.confidence, agreeing = excluded.agreeing, source = excluded.source, error = excluded.error, read_at = excluded.read_at""", (project_id, video_rel, video_rel.split("/")[0], started_at, working, confidence, agreeing, source, error, time.time()), ) def _as_datetime(text: str): import datetime return datetime.datetime.strptime(text, "%Y-%m-%d %H:%M:%S") def set_manual(project_id: int, video_rel: str, started_at: Optional[str]) -> dict: """A hand-entered start time. Outranks any reading and is never overwritten by a later scan — the whole point is that it is the one the user verified.""" store(project_id, video_rel, started_at, confidence=1.0, agreeing=99, source="manual" if started_at is not None else "none") return index(project_id).get(video_rel, {}) def _sidecar(project: dict, rel: str) -> Optional[dict]: """Waktu asli yang ditulis perekam di sebelah videonya (REQ-170). File baru datang dari MediaMTX, jadi waktunya sudah pasti dari server dan tidak perlu dibaca OCR sama sekali. Kalau ada sidecar, ia menang atas pembacaan overlay: sumbernya server, bukan tebakan dari piksel. """ import json try: path = library.resolve(project["video_root"], rel) except Exception: return None sidecar = os.path.splitext(path)[0] + ".json" if not os.path.isfile(sidecar): return None try: with open(sidecar, encoding="utf-8") as handle: payload = json.load(handle) started = payload.get("started_at") if not started: return None _as_datetime(started) # tolak isi yang tidak berbentuk waktu return {"started_at": started, "working_day": video_clock.working_day(_as_datetime(started), CUTOFF_HOUR), "source": payload.get("source") or "sidecar", "trusted": True, "confidence": 1.0, "agreeing": 99, "folder_date": rel.split("/")[0], "error": ""} except (OSError, ValueError, KeyError): return None def index(project_id: int) -> dict: """Everything known about the archive's timestamps, keyed by video path.""" with db.cursor() as cur: cur.execute("SELECT * FROM video_clock WHERE project_id = ?", (project_id,)) rows = [dict(row) for row in cur.fetchall()] out = {} for row in rows: row["trusted"] = bool(row["source"] == "manual" or _trusted(row["confidence"], row["agreeing"])) out[row["video_rel"]] = row return out def assign_batch_numbers(rows: List[dict]) -> List[dict]: """Number the recordings 1..N inside each working day, by real start time. Rows without a known start keep their folder grouping and sort last within it: an unreadable recording must not silently take position 1 and push everything else along. """ known = [r for r in rows if r.get("started_at")] unknown = [r for r in rows if not r.get("started_at")] known.sort(key=lambda r: (r["working_day"], r["started_at"])) counters: dict = {} for row in known: day = row["working_day"] counters[day] = counters.get(day, 0) + 1 row["batch_no"] = counters[day] # Dipanggil dari dua tempat dengan nama kunci berbeda: tabel Counting # Accuracy memakai `video_rel`, daftar arsip memakai `rel`. def path_of(row): return row.get("video_rel") or row.get("rel") or "" for row in unknown: row["working_day"] = row.get("working_day") or path_of(row).split("/")[0] row["batch_no"] = None unknown.sort(key=path_of) return known + unknown def cycles(project_id: int) -> List[dict]: """The archive as a list of cycles, newest first (REQ-165). A cycle is one 06:00-to-05:59 shift, so it always covers two calendar dates and is named after the one it starts on. Recordings whose start time is not known yet fall back to their folder name, so nothing disappears from the archive just because its overlay could not be read. """ project = projects.get(project_id) if project is None: raise ArchiveIndexError("No such project") known = index(project_id) buckets: dict = {} for day in library.list_dates(project["video_root"]): for name in _video_names(project, day["date"]): rel = f"{day['date']}/{name}" timing = _sidecar(project, rel) or known.get(rel) or {} cycle = timing.get("working_day") or day["date"] bucket = buckets.setdefault(cycle, {"cycle": cycle, "video_count": 0, "flagged": 0, "first_start": None}) bucket["video_count"] += 1 if not timing.get("started_at") or not timing.get("trusted"): bucket["flagged"] += 1 start = timing.get("started_at") if start and (bucket["first_start"] is None or start < bucket["first_start"]): bucket["first_start"] = start return sorted(buckets.values(), key=lambda b: b["cycle"], reverse=True) def _video_names(project: dict, date: str) -> List[str]: """Filenames only — `library.list_videos` runs ffprobe on every file, which is far too much work just to count what is in a cycle.""" import os as _os from backend import video as video_module folder = _os.path.join(library._effective_root(project["video_root"]), date) try: return [f for f in _os.listdir(folder) if f.lower().endswith(video_module.VIDEO_EXTS)] except OSError: return [] def cycle_videos(project_id: int, cycle: str) -> List[dict]: """Every recording in one cycle, in the order it was actually made.""" project = projects.get(project_id) if project is None: raise ArchiveIndexError("No such project") known = index(project_id) # A cycle normally draws from two folders, but a moved recording can come # from any of them. Ask the index which folders actually contribute rather # than running ffprobe across the whole archive to find out. folders = {row["folder_date"] for row in known.values() if row.get("working_day") == cycle} folders.add(cycle) # Berkas baru belum tentu ada di indeks; sidecar-nya bisa memindahkannya ke # siklus ini dari folder tanggal sebelah. for day in library.list_dates(project["video_root"]): if day["date"] in folders: continue for name in _video_names(project, day["date"]): side = _sidecar(project, f"{day['date']}/{name}") if side and side["working_day"] == cycle: folders.add(day["date"]) break rows = [] for day in library.list_dates(project["video_root"]): if day["date"] not in folders: continue for video in library.list_videos(project["video_root"], day["date"], project_id): rel = video["rel"] timing = _sidecar(project, rel) or known.get(rel) or {} if (timing.get("working_day") or day["date"]) != cycle: continue rows.append({ **video, "folder_date": day["date"], "working_day": timing.get("working_day") or "", "started_at": timing.get("started_at"), "clock_trusted": bool(timing.get("trusted")), "clock_error": timing.get("error") or "", "moved": bool(timing.get("working_day") and timing["working_day"] != day["date"]), "truck_hits": timing.get("truck_hits"), "truck_samples": timing.get("truck_samples"), }) return assign_batch_numbers(rows) TRUCK_SAMPLES = 12 """Frames sampled per recording for the truck check (REQ-166). Enough to answer "is there a truck in this recording at all", which is the assumption the whole batch numbering rests on: recording starts when a truck arrives and stops when it leaves, so one file is one batch. Reading every frame to time the truck's arrival precisely would cost hours of GPU for an answer the recording trigger already gives. """ def store_truck(project_id: int, video_rel: str, hits: int, samples: int, model_path: str) -> None: """One statement, one cursor. This used to UPDATE and then call `store()` for the missing-row case — which opened a second connection while the first still held a write transaction, and SQLite answered "database is locked" eight recordings into the scan. """ label = os.path.basename(os.path.dirname(model_path)) with db.cursor() as cur: cur.execute( """INSERT INTO video_clock (project_id, video_rel, folder_date, truck_hits, truck_samples, truck_model, truck_checked_at) VALUES (?, ?, ?, ?, ?, ?, ?) ON CONFLICT(project_id, video_rel) DO UPDATE SET truck_hits = excluded.truck_hits, truck_samples = excluded.truck_samples, truck_model = excluded.truck_model, truck_checked_at = excluded.truck_checked_at""", (project_id, video_rel, video_rel.split("/")[0], hits, samples, label, time.time()), ) def count_truck_frames(path: str, model, conf: float = 0.45, samples: int = TRUCK_SAMPLES) -> tuple: """How many of `samples` evenly spaced frames show a truck.""" import cv2 capture = cv2.VideoCapture(path) if not capture.isOpened(): raise ArchiveIndexError(f"Could not open {path}") total = int(capture.get(cv2.CAP_PROP_FRAME_COUNT) or 0) hits = taken = 0 try: for index in range(samples): # Spread across the middle 90%: the very first and last frames of a # trigger-started recording can catch the truck half out of shot. position = int(total * (0.05 + 0.9 * index / max(1, samples - 1))) capture.set(cv2.CAP_PROP_POS_FRAMES, position) ok, frame = capture.read() if not ok or frame is None: continue taken += 1 result = model.predict(frame, conf=conf, verbose=False)[0] names = model.names if any(names[int(box.cls[0])] == "truck" for box in result.boxes): hits += 1 finally: capture.release() return hits, taken def queue_truck_scan(project_id: int, model_path: str, rescan: bool = False) -> dict: project = projects.get(project_id) if project is None: raise ArchiveIndexError("No such project") if not os.path.isfile(model_path): raise ArchiveIndexError(f"Model not found: {model_path}") return jobs.create( "truck-scan", params={"model_path": model_path, "rescan": rescan}, project_id=project_id, message="checking each recording for a truck", ).to_dict() @jobs.handler("truck-scan") def _run_truck_scan(job) -> None: import numpy as np from ultralytics import YOLO project = projects.get(job.project_id) model_path = job.params["model_path"] rescan = bool(job.params.get("rescan")) known = index(job.project_id) todo = [] for day in library.list_dates(project["video_root"]): for name in _video_names(project, day["date"]): rel = f"{day['date']}/{name}" row = known.get(rel) or {} if not rescan and row.get("truck_samples"): continue todo.append(rel) job.progress(0, len(todo)) job.log(f"Checking {len(todo)} recording(s) for a truck, " f"{TRUCK_SAMPLES} frames each, with {os.path.basename(model_path)}") model = YOLO(model_path) model(np.zeros((720, 1280, 3), dtype=np.uint8), imgsz=640, verbose=False) empty = broken = 0 for position, rel in enumerate(todo): if job.cancelled: job.log(f"Cancelled after {position} recording(s)") return try: path = library.resolve(project["video_root"], rel) hits, taken = count_truck_frames(path, model) except Exception as exc: store_truck(job.project_id, rel, 0, 0, model_path) job.log(f"{rel}: {exc}") broken += 1 job.progress(position + 1, len(todo)) continue store_truck(job.project_id, rel, hits, taken, model_path) if taken and hits == 0: empty += 1 job.log(f"{rel}: no truck in any of {taken} sampled frames — " "this recording may not be a batch") job.progress(position + 1, len(todo), rel) job.log(f"Done. {empty} recording(s) with no truck, {broken} unreadable") def queue_scan(project_id: int, rescan: bool = False) -> dict: project = projects.get(project_id) if project is None: raise ArchiveIndexError("No such project") return jobs.create( "clock-scan", params={"rescan": rescan}, project_id=project_id, message="reading timestamps from the archive", ).to_dict() @jobs.handler("clock-scan") def _run_scan(job) -> None: project = projects.get(job.project_id) rescan = bool(job.params.get("rescan")) existing = index(job.project_id) todo = [] for day in library.list_dates(project["video_root"]): for video in library.list_videos(project["video_root"], day["date"]): rel = video["rel"] known = existing.get(rel) # A hand-entered time is never re-read; a rescan redoes the rest. if known and (known["source"] == "manual" or (not rescan and known["started_at"])): continue todo.append(rel) job.progress(0, len(todo)) job.log(f"Reading the timestamp overlay from {len(todo)} recording(s)") read = flagged = failed = 0 for position, rel in enumerate(todo): if job.cancelled: job.log(f"Cancelled after {position} recording(s)") return side = _sidecar(project, rel) if side is not None: store(job.project_id, rel, side["started_at"], confidence=1.0, agreeing=99, source=side["source"]) read += 1 job.progress(position + 1, len(todo), rel) continue try: path = library.resolve(project["video_root"], rel) result = video_clock.read_video_start(path) except Exception as exc: store(job.project_id, rel, None, error=f"{type(exc).__name__}: {exc}") job.log(f"{rel}: {exc}") failed += 1 job.progress(position + 1, len(todo)) continue if result["start"] is None: store(job.project_id, rel, None, error=result.get("error", "unreadable")) failed += 1 else: confidence, agreeing = result["confidence"], result.get("agreeing", 1) store(job.project_id, rel, result["start"].strftime("%Y-%m-%d %H:%M:%S"), confidence=confidence, agreeing=agreeing) if _trusted(confidence, agreeing): read += 1 else: flagged += 1 job.log(f"{rel}: {result['start']} — low confidence " f"({confidence}, {agreeing} frame(s) agreed), needs review") job.progress(position + 1, len(todo), f"{rel}") job.log(f"Read {read}, flagged {flagged} for review, {failed} unreadable")