This commit includes major additions and updates to the frontend and backend architectures, introducing new dataset management, live counting features, batch processing, and triage logic. Includes new UI pages, components, and API routes.
449 lines
18 KiB
Python
449 lines
18 KiB
Python
"""When each recording actually happened, and which working day it belongs to.
|
|
|
|
The archive's folders are wrong about both. `2026-08-13/batch001.mp4` was
|
|
recorded at 00:07, which under a 06:00-to-06:00 shift belongs to the working day
|
|
of 2026-08-12 — and `2026-08-07/batch4.mp4` was recorded the previous evening
|
|
entirely. Sampling the archive, roughly a quarter of the files land on a
|
|
different working day once their real timestamp is read (REQ-160…163).
|
|
|
|
Nothing on disk is touched. The archive is mounted read-only and is the user's
|
|
own data; this builds an index beside it instead, and every page groups and
|
|
orders by the index rather than by the folder name. The original path stays the
|
|
file's identity, so results already recorded against it survive.
|
|
"""
|
|
|
|
import os
|
|
import time
|
|
from typing import List, Optional
|
|
|
|
from backend import db, jobs, library, projects, video_clock
|
|
|
|
CUTOFF_HOUR = 6
|
|
"""A working day runs 06:00 to 06:00 (REQ-161)."""
|
|
|
|
MIN_CONFIDENCE = 0.10
|
|
MIN_AGREEING = 2
|
|
"""Below either of these a reading is kept but flagged: it is a guess, not a
|
|
measurement, and one misread digit is what puts a recording on the wrong day."""
|
|
|
|
|
|
class ArchiveIndexError(Exception):
|
|
pass
|
|
|
|
|
|
def _trusted(confidence: float, agreeing: int) -> bool:
|
|
return confidence >= MIN_CONFIDENCE and agreeing >= MIN_AGREEING
|
|
|
|
|
|
def store(project_id: int, video_rel: str, started_at: Optional[str],
|
|
confidence: float = 0.0, agreeing: int = 0, source: str = "ocr",
|
|
error: str = "") -> None:
|
|
"""`started_at` is wall-clock text, 'YYYY-MM-DD HH:MM:SS'.
|
|
|
|
Never an epoch. The overlay has no timezone, so converting it to one makes
|
|
the answer depend on which timezone the process happens to run in — the
|
|
backend container is UTC and the browser is not.
|
|
"""
|
|
working = ""
|
|
if started_at:
|
|
working = video_clock.working_day(_as_datetime(started_at), CUTOFF_HOUR)
|
|
with db.cursor() as cur:
|
|
cur.execute(
|
|
"""INSERT INTO video_clock (project_id, video_rel, folder_date, started_at,
|
|
working_day, confidence, agreeing, source, error,
|
|
read_at)
|
|
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
ON CONFLICT(project_id, video_rel) DO UPDATE SET
|
|
started_at = excluded.started_at, working_day = excluded.working_day,
|
|
confidence = excluded.confidence, agreeing = excluded.agreeing,
|
|
source = excluded.source, error = excluded.error,
|
|
read_at = excluded.read_at""",
|
|
(project_id, video_rel, video_rel.split("/")[0], started_at, working,
|
|
confidence, agreeing, source, error, time.time()),
|
|
)
|
|
|
|
|
|
def _as_datetime(text: str):
|
|
import datetime
|
|
|
|
return datetime.datetime.strptime(text, "%Y-%m-%d %H:%M:%S")
|
|
|
|
|
|
def set_manual(project_id: int, video_rel: str, started_at: Optional[str]) -> dict:
|
|
"""A hand-entered start time. Outranks any reading and is never overwritten
|
|
by a later scan — the whole point is that it is the one the user verified."""
|
|
store(project_id, video_rel, started_at, confidence=1.0, agreeing=99,
|
|
source="manual" if started_at is not None else "none")
|
|
return index(project_id).get(video_rel, {})
|
|
|
|
|
|
def _sidecar(project: dict, rel: str) -> Optional[dict]:
|
|
"""Waktu asli yang ditulis perekam di sebelah videonya (REQ-170).
|
|
|
|
File baru datang dari MediaMTX, jadi waktunya sudah pasti dari server dan
|
|
tidak perlu dibaca OCR sama sekali. Kalau ada sidecar, ia menang atas
|
|
pembacaan overlay: sumbernya server, bukan tebakan dari piksel.
|
|
"""
|
|
import json
|
|
|
|
try:
|
|
path = library.resolve(project["video_root"], rel)
|
|
except Exception:
|
|
return None
|
|
sidecar = os.path.splitext(path)[0] + ".json"
|
|
if not os.path.isfile(sidecar):
|
|
return None
|
|
try:
|
|
with open(sidecar, encoding="utf-8") as handle:
|
|
payload = json.load(handle)
|
|
started = payload.get("started_at")
|
|
if not started:
|
|
return None
|
|
_as_datetime(started) # tolak isi yang tidak berbentuk waktu
|
|
return {"started_at": started,
|
|
"working_day": video_clock.working_day(_as_datetime(started), CUTOFF_HOUR),
|
|
"source": payload.get("source") or "sidecar",
|
|
"trusted": True, "confidence": 1.0, "agreeing": 99,
|
|
"folder_date": rel.split("/")[0], "error": ""}
|
|
except (OSError, ValueError, KeyError):
|
|
return None
|
|
|
|
|
|
def index(project_id: int) -> dict:
|
|
"""Everything known about the archive's timestamps, keyed by video path."""
|
|
with db.cursor() as cur:
|
|
cur.execute("SELECT * FROM video_clock WHERE project_id = ?", (project_id,))
|
|
rows = [dict(row) for row in cur.fetchall()]
|
|
|
|
out = {}
|
|
for row in rows:
|
|
row["trusted"] = bool(row["source"] == "manual"
|
|
or _trusted(row["confidence"], row["agreeing"]))
|
|
out[row["video_rel"]] = row
|
|
return out
|
|
|
|
|
|
def assign_batch_numbers(rows: List[dict]) -> List[dict]:
|
|
"""Number the recordings 1..N inside each working day, by real start time.
|
|
|
|
Rows without a known start keep their folder grouping and sort last within
|
|
it: an unreadable recording must not silently take position 1 and push
|
|
everything else along.
|
|
"""
|
|
known = [r for r in rows if r.get("started_at")]
|
|
unknown = [r for r in rows if not r.get("started_at")]
|
|
|
|
known.sort(key=lambda r: (r["working_day"], r["started_at"]))
|
|
counters: dict = {}
|
|
for row in known:
|
|
day = row["working_day"]
|
|
counters[day] = counters.get(day, 0) + 1
|
|
row["batch_no"] = counters[day]
|
|
|
|
# Dipanggil dari dua tempat dengan nama kunci berbeda: tabel Counting
|
|
# Accuracy memakai `video_rel`, daftar arsip memakai `rel`.
|
|
def path_of(row):
|
|
return row.get("video_rel") or row.get("rel") or ""
|
|
|
|
for row in unknown:
|
|
row["working_day"] = row.get("working_day") or path_of(row).split("/")[0]
|
|
row["batch_no"] = None
|
|
unknown.sort(key=path_of)
|
|
return known + unknown
|
|
|
|
|
|
def cycles(project_id: int) -> List[dict]:
|
|
"""The archive as a list of cycles, newest first (REQ-165).
|
|
|
|
A cycle is one 06:00-to-05:59 shift, so it always covers two calendar dates
|
|
and is named after the one it starts on. Recordings whose start time is not
|
|
known yet fall back to their folder name, so nothing disappears from the
|
|
archive just because its overlay could not be read.
|
|
"""
|
|
project = projects.get(project_id)
|
|
if project is None:
|
|
raise ArchiveIndexError("No such project")
|
|
known = index(project_id)
|
|
|
|
buckets: dict = {}
|
|
for day in library.list_dates(project["video_root"]):
|
|
for name in _video_names(project, day["date"]):
|
|
rel = f"{day['date']}/{name}"
|
|
timing = _sidecar(project, rel) or known.get(rel) or {}
|
|
cycle = timing.get("working_day") or day["date"]
|
|
bucket = buckets.setdefault(cycle, {"cycle": cycle, "video_count": 0,
|
|
"flagged": 0, "first_start": None})
|
|
bucket["video_count"] += 1
|
|
if not timing.get("started_at") or not timing.get("trusted"):
|
|
bucket["flagged"] += 1
|
|
start = timing.get("started_at")
|
|
if start and (bucket["first_start"] is None or start < bucket["first_start"]):
|
|
bucket["first_start"] = start
|
|
|
|
return sorted(buckets.values(), key=lambda b: b["cycle"], reverse=True)
|
|
|
|
|
|
def _video_names(project: dict, date: str) -> List[str]:
|
|
"""Filenames only — `library.list_videos` runs ffprobe on every file, which
|
|
is far too much work just to count what is in a cycle."""
|
|
import os as _os
|
|
|
|
from backend import video as video_module
|
|
|
|
folder = _os.path.join(library._effective_root(project["video_root"]), date)
|
|
try:
|
|
return [f for f in _os.listdir(folder)
|
|
if f.lower().endswith(video_module.VIDEO_EXTS)]
|
|
except OSError:
|
|
return []
|
|
|
|
|
|
def cycle_videos(project_id: int, cycle: str) -> List[dict]:
|
|
"""Every recording in one cycle, in the order it was actually made."""
|
|
project = projects.get(project_id)
|
|
if project is None:
|
|
raise ArchiveIndexError("No such project")
|
|
known = index(project_id)
|
|
|
|
# A cycle normally draws from two folders, but a moved recording can come
|
|
# from any of them. Ask the index which folders actually contribute rather
|
|
# than running ffprobe across the whole archive to find out.
|
|
folders = {row["folder_date"] for row in known.values()
|
|
if row.get("working_day") == cycle}
|
|
folders.add(cycle)
|
|
# Berkas baru belum tentu ada di indeks; sidecar-nya bisa memindahkannya ke
|
|
# siklus ini dari folder tanggal sebelah.
|
|
for day in library.list_dates(project["video_root"]):
|
|
if day["date"] in folders:
|
|
continue
|
|
for name in _video_names(project, day["date"]):
|
|
side = _sidecar(project, f"{day['date']}/{name}")
|
|
if side and side["working_day"] == cycle:
|
|
folders.add(day["date"])
|
|
break
|
|
|
|
rows = []
|
|
for day in library.list_dates(project["video_root"]):
|
|
if day["date"] not in folders:
|
|
continue
|
|
for video in library.list_videos(project["video_root"], day["date"], project_id):
|
|
rel = video["rel"]
|
|
timing = _sidecar(project, rel) or known.get(rel) or {}
|
|
if (timing.get("working_day") or day["date"]) != cycle:
|
|
continue
|
|
rows.append({
|
|
**video,
|
|
"folder_date": day["date"],
|
|
"working_day": timing.get("working_day") or "",
|
|
"started_at": timing.get("started_at"),
|
|
"clock_trusted": bool(timing.get("trusted")),
|
|
"clock_error": timing.get("error") or "",
|
|
"moved": bool(timing.get("working_day")
|
|
and timing["working_day"] != day["date"]),
|
|
"truck_hits": timing.get("truck_hits"),
|
|
"truck_samples": timing.get("truck_samples"),
|
|
})
|
|
return assign_batch_numbers(rows)
|
|
|
|
|
|
TRUCK_SAMPLES = 12
|
|
"""Frames sampled per recording for the truck check (REQ-166).
|
|
|
|
Enough to answer "is there a truck in this recording at all", which is the
|
|
assumption the whole batch numbering rests on: recording starts when a truck
|
|
arrives and stops when it leaves, so one file is one batch. Reading every frame
|
|
to time the truck's arrival precisely would cost hours of GPU for an answer the
|
|
recording trigger already gives.
|
|
"""
|
|
|
|
|
|
def store_truck(project_id: int, video_rel: str, hits: int, samples: int,
|
|
model_path: str) -> None:
|
|
"""One statement, one cursor.
|
|
|
|
This used to UPDATE and then call `store()` for the missing-row case — which
|
|
opened a second connection while the first still held a write transaction,
|
|
and SQLite answered "database is locked" eight recordings into the scan.
|
|
"""
|
|
label = os.path.basename(os.path.dirname(model_path))
|
|
with db.cursor() as cur:
|
|
cur.execute(
|
|
"""INSERT INTO video_clock (project_id, video_rel, folder_date,
|
|
truck_hits, truck_samples, truck_model,
|
|
truck_checked_at)
|
|
VALUES (?, ?, ?, ?, ?, ?, ?)
|
|
ON CONFLICT(project_id, video_rel) DO UPDATE SET
|
|
truck_hits = excluded.truck_hits,
|
|
truck_samples = excluded.truck_samples,
|
|
truck_model = excluded.truck_model,
|
|
truck_checked_at = excluded.truck_checked_at""",
|
|
(project_id, video_rel, video_rel.split("/")[0], hits, samples,
|
|
label, time.time()),
|
|
)
|
|
|
|
|
|
def count_truck_frames(path: str, model, conf: float = 0.45,
|
|
samples: int = TRUCK_SAMPLES) -> tuple:
|
|
"""How many of `samples` evenly spaced frames show a truck."""
|
|
import cv2
|
|
|
|
capture = cv2.VideoCapture(path)
|
|
if not capture.isOpened():
|
|
raise ArchiveIndexError(f"Could not open {path}")
|
|
total = int(capture.get(cv2.CAP_PROP_FRAME_COUNT) or 0)
|
|
hits = taken = 0
|
|
try:
|
|
for index in range(samples):
|
|
# Spread across the middle 90%: the very first and last frames of a
|
|
# trigger-started recording can catch the truck half out of shot.
|
|
position = int(total * (0.05 + 0.9 * index / max(1, samples - 1)))
|
|
capture.set(cv2.CAP_PROP_POS_FRAMES, position)
|
|
ok, frame = capture.read()
|
|
if not ok or frame is None:
|
|
continue
|
|
taken += 1
|
|
result = model.predict(frame, conf=conf, verbose=False)[0]
|
|
names = model.names
|
|
if any(names[int(box.cls[0])] == "truck" for box in result.boxes):
|
|
hits += 1
|
|
finally:
|
|
capture.release()
|
|
return hits, taken
|
|
|
|
|
|
def queue_truck_scan(project_id: int, model_path: str, rescan: bool = False) -> dict:
|
|
project = projects.get(project_id)
|
|
if project is None:
|
|
raise ArchiveIndexError("No such project")
|
|
if not os.path.isfile(model_path):
|
|
raise ArchiveIndexError(f"Model not found: {model_path}")
|
|
return jobs.create(
|
|
"truck-scan",
|
|
params={"model_path": model_path, "rescan": rescan},
|
|
project_id=project_id,
|
|
message="checking each recording for a truck",
|
|
).to_dict()
|
|
|
|
|
|
@jobs.handler("truck-scan")
|
|
def _run_truck_scan(job) -> None:
|
|
import numpy as np
|
|
from ultralytics import YOLO
|
|
|
|
project = projects.get(job.project_id)
|
|
model_path = job.params["model_path"]
|
|
rescan = bool(job.params.get("rescan"))
|
|
known = index(job.project_id)
|
|
|
|
todo = []
|
|
for day in library.list_dates(project["video_root"]):
|
|
for name in _video_names(project, day["date"]):
|
|
rel = f"{day['date']}/{name}"
|
|
row = known.get(rel) or {}
|
|
if not rescan and row.get("truck_samples"):
|
|
continue
|
|
todo.append(rel)
|
|
|
|
job.progress(0, len(todo))
|
|
job.log(f"Checking {len(todo)} recording(s) for a truck, "
|
|
f"{TRUCK_SAMPLES} frames each, with {os.path.basename(model_path)}")
|
|
|
|
model = YOLO(model_path)
|
|
model(np.zeros((720, 1280, 3), dtype=np.uint8), imgsz=640, verbose=False)
|
|
|
|
empty = broken = 0
|
|
for position, rel in enumerate(todo):
|
|
if job.cancelled:
|
|
job.log(f"Cancelled after {position} recording(s)")
|
|
return
|
|
try:
|
|
path = library.resolve(project["video_root"], rel)
|
|
hits, taken = count_truck_frames(path, model)
|
|
except Exception as exc:
|
|
store_truck(job.project_id, rel, 0, 0, model_path)
|
|
job.log(f"{rel}: {exc}")
|
|
broken += 1
|
|
job.progress(position + 1, len(todo))
|
|
continue
|
|
|
|
store_truck(job.project_id, rel, hits, taken, model_path)
|
|
if taken and hits == 0:
|
|
empty += 1
|
|
job.log(f"{rel}: no truck in any of {taken} sampled frames — "
|
|
"this recording may not be a batch")
|
|
job.progress(position + 1, len(todo), rel)
|
|
|
|
job.log(f"Done. {empty} recording(s) with no truck, {broken} unreadable")
|
|
|
|
|
|
def queue_scan(project_id: int, rescan: bool = False) -> dict:
|
|
project = projects.get(project_id)
|
|
if project is None:
|
|
raise ArchiveIndexError("No such project")
|
|
return jobs.create(
|
|
"clock-scan",
|
|
params={"rescan": rescan},
|
|
project_id=project_id,
|
|
message="reading timestamps from the archive",
|
|
).to_dict()
|
|
|
|
|
|
@jobs.handler("clock-scan")
|
|
def _run_scan(job) -> None:
|
|
project = projects.get(job.project_id)
|
|
rescan = bool(job.params.get("rescan"))
|
|
existing = index(job.project_id)
|
|
|
|
todo = []
|
|
for day in library.list_dates(project["video_root"]):
|
|
for video in library.list_videos(project["video_root"], day["date"]):
|
|
rel = video["rel"]
|
|
known = existing.get(rel)
|
|
# A hand-entered time is never re-read; a rescan redoes the rest.
|
|
if known and (known["source"] == "manual"
|
|
or (not rescan and known["started_at"])):
|
|
continue
|
|
todo.append(rel)
|
|
|
|
job.progress(0, len(todo))
|
|
job.log(f"Reading the timestamp overlay from {len(todo)} recording(s)")
|
|
read = flagged = failed = 0
|
|
|
|
for position, rel in enumerate(todo):
|
|
if job.cancelled:
|
|
job.log(f"Cancelled after {position} recording(s)")
|
|
return
|
|
side = _sidecar(project, rel)
|
|
if side is not None:
|
|
store(job.project_id, rel, side["started_at"], confidence=1.0,
|
|
agreeing=99, source=side["source"])
|
|
read += 1
|
|
job.progress(position + 1, len(todo), rel)
|
|
continue
|
|
try:
|
|
path = library.resolve(project["video_root"], rel)
|
|
result = video_clock.read_video_start(path)
|
|
except Exception as exc:
|
|
store(job.project_id, rel, None, error=f"{type(exc).__name__}: {exc}")
|
|
job.log(f"{rel}: {exc}")
|
|
failed += 1
|
|
job.progress(position + 1, len(todo))
|
|
continue
|
|
|
|
if result["start"] is None:
|
|
store(job.project_id, rel, None, error=result.get("error", "unreadable"))
|
|
failed += 1
|
|
else:
|
|
confidence, agreeing = result["confidence"], result.get("agreeing", 1)
|
|
store(job.project_id, rel, result["start"].strftime("%Y-%m-%d %H:%M:%S"),
|
|
confidence=confidence, agreeing=agreeing)
|
|
if _trusted(confidence, agreeing):
|
|
read += 1
|
|
else:
|
|
flagged += 1
|
|
job.log(f"{rel}: {result['start']} — low confidence "
|
|
f"({confidence}, {agreeing} frame(s) agreed), needs review")
|
|
job.progress(position + 1, len(todo), f"{rel}")
|
|
|
|
job.log(f"Read {read}, flagged {flagged} for review, {failed} unreadable")
|