- REQ-181: class_params {name: {threshold?, iou_threshold?, min_box_frac?}}
on /preview and /autolabel, per-class override table in both modals;
empty overrides take the unchanged global path
- REQ-180: x button on each review sidebar class row clears that class on
the current frame only via bulk-delete, no confirmation
- includes REQ-178 empty date-folder cycle fix (archive_index.py)
456 lines
18 KiB
Python
456 lines
18 KiB
Python
"""When each recording actually happened, and which working day it belongs to.
|
|
|
|
The archive's folders are wrong about both. `2026-08-13/batch001.mp4` was
|
|
recorded at 00:07, which under a 06:00-to-06:00 shift belongs to the working day
|
|
of 2026-08-12 — and `2026-08-07/batch4.mp4` was recorded the previous evening
|
|
entirely. Sampling the archive, roughly a quarter of the files land on a
|
|
different working day once their real timestamp is read (REQ-160…163).
|
|
|
|
Nothing on disk is touched. The archive is mounted read-only and is the user's
|
|
own data; this builds an index beside it instead, and every page groups and
|
|
orders by the index rather than by the folder name. The original path stays the
|
|
file's identity, so results already recorded against it survive.
|
|
"""
|
|
|
|
import os
|
|
import time
|
|
from typing import List, Optional
|
|
|
|
from backend import db, jobs, library, projects, video_clock
|
|
|
|
CUTOFF_HOUR = 6
|
|
"""A working day runs 06:00 to 06:00 (REQ-161)."""
|
|
|
|
MIN_CONFIDENCE = 0.10
|
|
MIN_AGREEING = 2
|
|
"""Below either of these a reading is kept but flagged: it is a guess, not a
|
|
measurement, and one misread digit is what puts a recording on the wrong day."""
|
|
|
|
|
|
class ArchiveIndexError(Exception):
|
|
pass
|
|
|
|
|
|
def _trusted(confidence: float, agreeing: int) -> bool:
|
|
return confidence >= MIN_CONFIDENCE and agreeing >= MIN_AGREEING
|
|
|
|
|
|
def store(project_id: int, video_rel: str, started_at: Optional[str],
|
|
confidence: float = 0.0, agreeing: int = 0, source: str = "ocr",
|
|
error: str = "") -> None:
|
|
"""`started_at` is wall-clock text, 'YYYY-MM-DD HH:MM:SS'.
|
|
|
|
Never an epoch. The overlay has no timezone, so converting it to one makes
|
|
the answer depend on which timezone the process happens to run in — the
|
|
backend container is UTC and the browser is not.
|
|
"""
|
|
working = ""
|
|
if started_at:
|
|
working = video_clock.working_day(_as_datetime(started_at), CUTOFF_HOUR)
|
|
with db.cursor() as cur:
|
|
cur.execute(
|
|
"""INSERT INTO video_clock (project_id, video_rel, folder_date, started_at,
|
|
working_day, confidence, agreeing, source, error,
|
|
read_at)
|
|
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
ON CONFLICT(project_id, video_rel) DO UPDATE SET
|
|
started_at = excluded.started_at, working_day = excluded.working_day,
|
|
confidence = excluded.confidence, agreeing = excluded.agreeing,
|
|
source = excluded.source, error = excluded.error,
|
|
read_at = excluded.read_at""",
|
|
(project_id, video_rel, video_rel.split("/")[0], started_at, working,
|
|
confidence, agreeing, source, error, time.time()),
|
|
)
|
|
|
|
|
|
def _as_datetime(text: str):
|
|
import datetime
|
|
|
|
return datetime.datetime.strptime(text, "%Y-%m-%d %H:%M:%S")
|
|
|
|
|
|
def set_manual(project_id: int, video_rel: str, started_at: Optional[str]) -> dict:
|
|
"""A hand-entered start time. Outranks any reading and is never overwritten
|
|
by a later scan — the whole point is that it is the one the user verified."""
|
|
store(project_id, video_rel, started_at, confidence=1.0, agreeing=99,
|
|
source="manual" if started_at is not None else "none")
|
|
return index(project_id).get(video_rel, {})
|
|
|
|
|
|
def _sidecar(project: dict, rel: str) -> Optional[dict]:
|
|
"""Waktu asli yang ditulis perekam di sebelah videonya (REQ-170).
|
|
|
|
File baru datang dari MediaMTX, jadi waktunya sudah pasti dari server dan
|
|
tidak perlu dibaca OCR sama sekali. Kalau ada sidecar, ia menang atas
|
|
pembacaan overlay: sumbernya server, bukan tebakan dari piksel.
|
|
"""
|
|
import json
|
|
|
|
try:
|
|
path = library.resolve(project["video_root"], rel)
|
|
except Exception:
|
|
return None
|
|
sidecar = os.path.splitext(path)[0] + ".json"
|
|
if not os.path.isfile(sidecar):
|
|
return None
|
|
try:
|
|
with open(sidecar, encoding="utf-8") as handle:
|
|
payload = json.load(handle)
|
|
started = payload.get("started_at")
|
|
if not started:
|
|
return None
|
|
_as_datetime(started) # tolak isi yang tidak berbentuk waktu
|
|
return {"started_at": started,
|
|
"working_day": video_clock.working_day(_as_datetime(started), CUTOFF_HOUR),
|
|
"source": payload.get("source") or "sidecar",
|
|
"trusted": True, "confidence": 1.0, "agreeing": 99,
|
|
"folder_date": rel.split("/")[0], "error": ""}
|
|
except (OSError, ValueError, KeyError):
|
|
return None
|
|
|
|
|
|
def index(project_id: int) -> dict:
|
|
"""Everything known about the archive's timestamps, keyed by video path."""
|
|
with db.cursor() as cur:
|
|
cur.execute("SELECT * FROM video_clock WHERE project_id = ?", (project_id,))
|
|
rows = [dict(row) for row in cur.fetchall()]
|
|
|
|
out = {}
|
|
for row in rows:
|
|
row["trusted"] = bool(row["source"] == "manual"
|
|
or _trusted(row["confidence"], row["agreeing"]))
|
|
out[row["video_rel"]] = row
|
|
return out
|
|
|
|
|
|
def assign_batch_numbers(rows: List[dict]) -> List[dict]:
|
|
"""Number the recordings 1..N inside each working day, by real start time.
|
|
|
|
Rows without a known start keep their folder grouping and sort last within
|
|
it: an unreadable recording must not silently take position 1 and push
|
|
everything else along.
|
|
"""
|
|
known = [r for r in rows if r.get("started_at")]
|
|
unknown = [r for r in rows if not r.get("started_at")]
|
|
|
|
known.sort(key=lambda r: (r["working_day"], r["started_at"]))
|
|
counters: dict = {}
|
|
for row in known:
|
|
day = row["working_day"]
|
|
counters[day] = counters.get(day, 0) + 1
|
|
row["batch_no"] = counters[day]
|
|
|
|
# Dipanggil dari dua tempat dengan nama kunci berbeda: tabel Counting
|
|
# Accuracy memakai `video_rel`, daftar arsip memakai `rel`.
|
|
def path_of(row):
|
|
return row.get("video_rel") or row.get("rel") or ""
|
|
|
|
for row in unknown:
|
|
row["working_day"] = row.get("working_day") or path_of(row).split("/")[0]
|
|
row["batch_no"] = None
|
|
unknown.sort(key=path_of)
|
|
return known + unknown
|
|
|
|
|
|
def cycles(project_id: int) -> List[dict]:
|
|
"""The archive as a list of cycles, newest first (REQ-165).
|
|
|
|
A cycle is one 06:00-to-05:59 shift, so it always covers two calendar dates
|
|
and is named after the one it starts on. Recordings whose start time is not
|
|
known yet fall back to their folder name, so nothing disappears from the
|
|
archive just because its overlay could not be read.
|
|
"""
|
|
project = projects.get(project_id)
|
|
if project is None:
|
|
raise ArchiveIndexError("No such project")
|
|
known = index(project_id)
|
|
|
|
buckets: dict = {}
|
|
for day in library.list_dates(project["video_root"]):
|
|
names = _video_names(project, day["date"])
|
|
if not names:
|
|
# Folder still holds no recording: it must still appear in the
|
|
# list, or a folder just created with "Folder baru" is invisible.
|
|
buckets.setdefault(day["date"], {"cycle": day["date"], "video_count": 0,
|
|
"flagged": 0, "first_start": None})
|
|
continue
|
|
for name in names:
|
|
rel = f"{day['date']}/{name}"
|
|
timing = _sidecar(project, rel) or known.get(rel) or {}
|
|
cycle = timing.get("working_day") or day["date"]
|
|
bucket = buckets.setdefault(cycle, {"cycle": cycle, "video_count": 0,
|
|
"flagged": 0, "first_start": None})
|
|
bucket["video_count"] += 1
|
|
if not timing.get("started_at") or not timing.get("trusted"):
|
|
bucket["flagged"] += 1
|
|
start = timing.get("started_at")
|
|
if start and (bucket["first_start"] is None or start < bucket["first_start"]):
|
|
bucket["first_start"] = start
|
|
|
|
return sorted(buckets.values(), key=lambda b: b["cycle"], reverse=True)
|
|
|
|
|
|
def _video_names(project: dict, date: str) -> List[str]:
|
|
"""Filenames only — `library.list_videos` runs ffprobe on every file, which
|
|
is far too much work just to count what is in a cycle."""
|
|
import os as _os
|
|
|
|
from backend import video as video_module
|
|
|
|
folder = _os.path.join(library._effective_root(project["video_root"]), date)
|
|
try:
|
|
return [f for f in _os.listdir(folder)
|
|
if f.lower().endswith(video_module.VIDEO_EXTS)]
|
|
except OSError:
|
|
return []
|
|
|
|
|
|
def cycle_videos(project_id: int, cycle: str) -> List[dict]:
|
|
"""Every recording in one cycle, in the order it was actually made."""
|
|
project = projects.get(project_id)
|
|
if project is None:
|
|
raise ArchiveIndexError("No such project")
|
|
known = index(project_id)
|
|
|
|
# A cycle normally draws from two folders, but a moved recording can come
|
|
# from any of them. Ask the index which folders actually contribute rather
|
|
# than running ffprobe across the whole archive to find out.
|
|
folders = {row["folder_date"] for row in known.values()
|
|
if row.get("working_day") == cycle}
|
|
folders.add(cycle)
|
|
# Berkas baru belum tentu ada di indeks; sidecar-nya bisa memindahkannya ke
|
|
# siklus ini dari folder tanggal sebelah.
|
|
for day in library.list_dates(project["video_root"]):
|
|
if day["date"] in folders:
|
|
continue
|
|
for name in _video_names(project, day["date"]):
|
|
side = _sidecar(project, f"{day['date']}/{name}")
|
|
if side and side["working_day"] == cycle:
|
|
folders.add(day["date"])
|
|
break
|
|
|
|
rows = []
|
|
for day in library.list_dates(project["video_root"]):
|
|
if day["date"] not in folders:
|
|
continue
|
|
for video in library.list_videos(project["video_root"], day["date"], project_id):
|
|
rel = video["rel"]
|
|
timing = _sidecar(project, rel) or known.get(rel) or {}
|
|
if (timing.get("working_day") or day["date"]) != cycle:
|
|
continue
|
|
rows.append({
|
|
**video,
|
|
"folder_date": day["date"],
|
|
"working_day": timing.get("working_day") or "",
|
|
"started_at": timing.get("started_at"),
|
|
"clock_trusted": bool(timing.get("trusted")),
|
|
"clock_error": timing.get("error") or "",
|
|
"moved": bool(timing.get("working_day")
|
|
and timing["working_day"] != day["date"]),
|
|
"truck_hits": timing.get("truck_hits"),
|
|
"truck_samples": timing.get("truck_samples"),
|
|
})
|
|
return assign_batch_numbers(rows)
|
|
|
|
|
|
TRUCK_SAMPLES = 12
|
|
"""Frames sampled per recording for the truck check (REQ-166).
|
|
|
|
Enough to answer "is there a truck in this recording at all", which is the
|
|
assumption the whole batch numbering rests on: recording starts when a truck
|
|
arrives and stops when it leaves, so one file is one batch. Reading every frame
|
|
to time the truck's arrival precisely would cost hours of GPU for an answer the
|
|
recording trigger already gives.
|
|
"""
|
|
|
|
|
|
def store_truck(project_id: int, video_rel: str, hits: int, samples: int,
|
|
model_path: str) -> None:
|
|
"""One statement, one cursor.
|
|
|
|
This used to UPDATE and then call `store()` for the missing-row case — which
|
|
opened a second connection while the first still held a write transaction,
|
|
and SQLite answered "database is locked" eight recordings into the scan.
|
|
"""
|
|
label = os.path.basename(os.path.dirname(model_path))
|
|
with db.cursor() as cur:
|
|
cur.execute(
|
|
"""INSERT INTO video_clock (project_id, video_rel, folder_date,
|
|
truck_hits, truck_samples, truck_model,
|
|
truck_checked_at)
|
|
VALUES (?, ?, ?, ?, ?, ?, ?)
|
|
ON CONFLICT(project_id, video_rel) DO UPDATE SET
|
|
truck_hits = excluded.truck_hits,
|
|
truck_samples = excluded.truck_samples,
|
|
truck_model = excluded.truck_model,
|
|
truck_checked_at = excluded.truck_checked_at""",
|
|
(project_id, video_rel, video_rel.split("/")[0], hits, samples,
|
|
label, time.time()),
|
|
)
|
|
|
|
|
|
def count_truck_frames(path: str, model, conf: float = 0.45,
|
|
samples: int = TRUCK_SAMPLES) -> tuple:
|
|
"""How many of `samples` evenly spaced frames show a truck."""
|
|
import cv2
|
|
|
|
capture = cv2.VideoCapture(path)
|
|
if not capture.isOpened():
|
|
raise ArchiveIndexError(f"Could not open {path}")
|
|
total = int(capture.get(cv2.CAP_PROP_FRAME_COUNT) or 0)
|
|
hits = taken = 0
|
|
try:
|
|
for index in range(samples):
|
|
# Spread across the middle 90%: the very first and last frames of a
|
|
# trigger-started recording can catch the truck half out of shot.
|
|
position = int(total * (0.05 + 0.9 * index / max(1, samples - 1)))
|
|
capture.set(cv2.CAP_PROP_POS_FRAMES, position)
|
|
ok, frame = capture.read()
|
|
if not ok or frame is None:
|
|
continue
|
|
taken += 1
|
|
result = model.predict(frame, conf=conf, verbose=False)[0]
|
|
names = model.names
|
|
if any(names[int(box.cls[0])] == "truck" for box in result.boxes):
|
|
hits += 1
|
|
finally:
|
|
capture.release()
|
|
return hits, taken
|
|
|
|
|
|
def queue_truck_scan(project_id: int, model_path: str, rescan: bool = False) -> dict:
|
|
project = projects.get(project_id)
|
|
if project is None:
|
|
raise ArchiveIndexError("No such project")
|
|
if not os.path.isfile(model_path):
|
|
raise ArchiveIndexError(f"Model not found: {model_path}")
|
|
return jobs.create(
|
|
"truck-scan",
|
|
params={"model_path": model_path, "rescan": rescan},
|
|
project_id=project_id,
|
|
message="checking each recording for a truck",
|
|
).to_dict()
|
|
|
|
|
|
@jobs.handler("truck-scan")
|
|
def _run_truck_scan(job) -> None:
|
|
import numpy as np
|
|
from ultralytics import YOLO
|
|
|
|
project = projects.get(job.project_id)
|
|
model_path = job.params["model_path"]
|
|
rescan = bool(job.params.get("rescan"))
|
|
known = index(job.project_id)
|
|
|
|
todo = []
|
|
for day in library.list_dates(project["video_root"]):
|
|
for name in _video_names(project, day["date"]):
|
|
rel = f"{day['date']}/{name}"
|
|
row = known.get(rel) or {}
|
|
if not rescan and row.get("truck_samples"):
|
|
continue
|
|
todo.append(rel)
|
|
|
|
job.progress(0, len(todo))
|
|
job.log(f"Checking {len(todo)} recording(s) for a truck, "
|
|
f"{TRUCK_SAMPLES} frames each, with {os.path.basename(model_path)}")
|
|
|
|
model = YOLO(model_path)
|
|
model(np.zeros((720, 1280, 3), dtype=np.uint8), imgsz=640, verbose=False)
|
|
|
|
empty = broken = 0
|
|
for position, rel in enumerate(todo):
|
|
if job.cancelled:
|
|
job.log(f"Cancelled after {position} recording(s)")
|
|
return
|
|
try:
|
|
path = library.resolve(project["video_root"], rel)
|
|
hits, taken = count_truck_frames(path, model)
|
|
except Exception as exc:
|
|
store_truck(job.project_id, rel, 0, 0, model_path)
|
|
job.log(f"{rel}: {exc}")
|
|
broken += 1
|
|
job.progress(position + 1, len(todo))
|
|
continue
|
|
|
|
store_truck(job.project_id, rel, hits, taken, model_path)
|
|
if taken and hits == 0:
|
|
empty += 1
|
|
job.log(f"{rel}: no truck in any of {taken} sampled frames — "
|
|
"this recording may not be a batch")
|
|
job.progress(position + 1, len(todo), rel)
|
|
|
|
job.log(f"Done. {empty} recording(s) with no truck, {broken} unreadable")
|
|
|
|
|
|
def queue_scan(project_id: int, rescan: bool = False) -> dict:
|
|
project = projects.get(project_id)
|
|
if project is None:
|
|
raise ArchiveIndexError("No such project")
|
|
return jobs.create(
|
|
"clock-scan",
|
|
params={"rescan": rescan},
|
|
project_id=project_id,
|
|
message="reading timestamps from the archive",
|
|
).to_dict()
|
|
|
|
|
|
@jobs.handler("clock-scan")
|
|
def _run_scan(job) -> None:
|
|
project = projects.get(job.project_id)
|
|
rescan = bool(job.params.get("rescan"))
|
|
existing = index(job.project_id)
|
|
|
|
todo = []
|
|
for day in library.list_dates(project["video_root"]):
|
|
for video in library.list_videos(project["video_root"], day["date"]):
|
|
rel = video["rel"]
|
|
known = existing.get(rel)
|
|
# A hand-entered time is never re-read; a rescan redoes the rest.
|
|
if known and (known["source"] == "manual"
|
|
or (not rescan and known["started_at"])):
|
|
continue
|
|
todo.append(rel)
|
|
|
|
job.progress(0, len(todo))
|
|
job.log(f"Reading the timestamp overlay from {len(todo)} recording(s)")
|
|
read = flagged = failed = 0
|
|
|
|
for position, rel in enumerate(todo):
|
|
if job.cancelled:
|
|
job.log(f"Cancelled after {position} recording(s)")
|
|
return
|
|
side = _sidecar(project, rel)
|
|
if side is not None:
|
|
store(job.project_id, rel, side["started_at"], confidence=1.0,
|
|
agreeing=99, source=side["source"])
|
|
read += 1
|
|
job.progress(position + 1, len(todo), rel)
|
|
continue
|
|
try:
|
|
path = library.resolve(project["video_root"], rel)
|
|
result = video_clock.read_video_start(path)
|
|
except Exception as exc:
|
|
store(job.project_id, rel, None, error=f"{type(exc).__name__}: {exc}")
|
|
job.log(f"{rel}: {exc}")
|
|
failed += 1
|
|
job.progress(position + 1, len(todo))
|
|
continue
|
|
|
|
if result["start"] is None:
|
|
store(job.project_id, rel, None, error=result.get("error", "unreadable"))
|
|
failed += 1
|
|
else:
|
|
confidence, agreeing = result["confidence"], result.get("agreeing", 1)
|
|
store(job.project_id, rel, result["start"].strftime("%Y-%m-%d %H:%M:%S"),
|
|
confidence=confidence, agreeing=agreeing)
|
|
if _trusted(confidence, agreeing):
|
|
read += 1
|
|
else:
|
|
flagged += 1
|
|
job.log(f"{rel}: {result['start']} — low confidence "
|
|
f"({confidence}, {agreeing} frame(s) agreed), needs review")
|
|
job.progress(position + 1, len(todo), f"{rel}")
|
|
|
|
job.log(f"Read {read}, flagged {flagged} for review, {failed} unreadable")
|