Files
feedmill-auto-label/backend/archive_index.py
T
asus 5c7c122105 feat: add counting bench, triage, and dataset modules
This commit includes major additions and updates to the frontend and backend architectures, introducing new dataset management, live counting features, batch processing, and triage logic. Includes new UI pages, components, and API routes.
2026-08-14 16:28:52 +07:00

449 lines
18 KiB
Python

"""When each recording actually happened, and which working day it belongs to.
The archive's folders are wrong about both. `2026-08-13/batch001.mp4` was
recorded at 00:07, which under a 06:00-to-06:00 shift belongs to the working day
of 2026-08-12 — and `2026-08-07/batch4.mp4` was recorded the previous evening
entirely. Sampling the archive, roughly a quarter of the files land on a
different working day once their real timestamp is read (REQ-160…163).
Nothing on disk is touched. The archive is mounted read-only and is the user's
own data; this builds an index beside it instead, and every page groups and
orders by the index rather than by the folder name. The original path stays the
file's identity, so results already recorded against it survive.
"""
import os
import time
from typing import List, Optional
from backend import db, jobs, library, projects, video_clock
CUTOFF_HOUR = 6
"""A working day runs 06:00 to 06:00 (REQ-161)."""
MIN_CONFIDENCE = 0.10
MIN_AGREEING = 2
"""Below either of these a reading is kept but flagged: it is a guess, not a
measurement, and one misread digit is what puts a recording on the wrong day."""
class ArchiveIndexError(Exception):
pass
def _trusted(confidence: float, agreeing: int) -> bool:
return confidence >= MIN_CONFIDENCE and agreeing >= MIN_AGREEING
def store(project_id: int, video_rel: str, started_at: Optional[str],
confidence: float = 0.0, agreeing: int = 0, source: str = "ocr",
error: str = "") -> None:
"""`started_at` is wall-clock text, 'YYYY-MM-DD HH:MM:SS'.
Never an epoch. The overlay has no timezone, so converting it to one makes
the answer depend on which timezone the process happens to run in — the
backend container is UTC and the browser is not.
"""
working = ""
if started_at:
working = video_clock.working_day(_as_datetime(started_at), CUTOFF_HOUR)
with db.cursor() as cur:
cur.execute(
"""INSERT INTO video_clock (project_id, video_rel, folder_date, started_at,
working_day, confidence, agreeing, source, error,
read_at)
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
ON CONFLICT(project_id, video_rel) DO UPDATE SET
started_at = excluded.started_at, working_day = excluded.working_day,
confidence = excluded.confidence, agreeing = excluded.agreeing,
source = excluded.source, error = excluded.error,
read_at = excluded.read_at""",
(project_id, video_rel, video_rel.split("/")[0], started_at, working,
confidence, agreeing, source, error, time.time()),
)
def _as_datetime(text: str):
import datetime
return datetime.datetime.strptime(text, "%Y-%m-%d %H:%M:%S")
def set_manual(project_id: int, video_rel: str, started_at: Optional[str]) -> dict:
"""A hand-entered start time. Outranks any reading and is never overwritten
by a later scan — the whole point is that it is the one the user verified."""
store(project_id, video_rel, started_at, confidence=1.0, agreeing=99,
source="manual" if started_at is not None else "none")
return index(project_id).get(video_rel, {})
def _sidecar(project: dict, rel: str) -> Optional[dict]:
"""Waktu asli yang ditulis perekam di sebelah videonya (REQ-170).
File baru datang dari MediaMTX, jadi waktunya sudah pasti dari server dan
tidak perlu dibaca OCR sama sekali. Kalau ada sidecar, ia menang atas
pembacaan overlay: sumbernya server, bukan tebakan dari piksel.
"""
import json
try:
path = library.resolve(project["video_root"], rel)
except Exception:
return None
sidecar = os.path.splitext(path)[0] + ".json"
if not os.path.isfile(sidecar):
return None
try:
with open(sidecar, encoding="utf-8") as handle:
payload = json.load(handle)
started = payload.get("started_at")
if not started:
return None
_as_datetime(started) # tolak isi yang tidak berbentuk waktu
return {"started_at": started,
"working_day": video_clock.working_day(_as_datetime(started), CUTOFF_HOUR),
"source": payload.get("source") or "sidecar",
"trusted": True, "confidence": 1.0, "agreeing": 99,
"folder_date": rel.split("/")[0], "error": ""}
except (OSError, ValueError, KeyError):
return None
def index(project_id: int) -> dict:
"""Everything known about the archive's timestamps, keyed by video path."""
with db.cursor() as cur:
cur.execute("SELECT * FROM video_clock WHERE project_id = ?", (project_id,))
rows = [dict(row) for row in cur.fetchall()]
out = {}
for row in rows:
row["trusted"] = bool(row["source"] == "manual"
or _trusted(row["confidence"], row["agreeing"]))
out[row["video_rel"]] = row
return out
def assign_batch_numbers(rows: List[dict]) -> List[dict]:
"""Number the recordings 1..N inside each working day, by real start time.
Rows without a known start keep their folder grouping and sort last within
it: an unreadable recording must not silently take position 1 and push
everything else along.
"""
known = [r for r in rows if r.get("started_at")]
unknown = [r for r in rows if not r.get("started_at")]
known.sort(key=lambda r: (r["working_day"], r["started_at"]))
counters: dict = {}
for row in known:
day = row["working_day"]
counters[day] = counters.get(day, 0) + 1
row["batch_no"] = counters[day]
# Dipanggil dari dua tempat dengan nama kunci berbeda: tabel Counting
# Accuracy memakai `video_rel`, daftar arsip memakai `rel`.
def path_of(row):
return row.get("video_rel") or row.get("rel") or ""
for row in unknown:
row["working_day"] = row.get("working_day") or path_of(row).split("/")[0]
row["batch_no"] = None
unknown.sort(key=path_of)
return known + unknown
def cycles(project_id: int) -> List[dict]:
"""The archive as a list of cycles, newest first (REQ-165).
A cycle is one 06:00-to-05:59 shift, so it always covers two calendar dates
and is named after the one it starts on. Recordings whose start time is not
known yet fall back to their folder name, so nothing disappears from the
archive just because its overlay could not be read.
"""
project = projects.get(project_id)
if project is None:
raise ArchiveIndexError("No such project")
known = index(project_id)
buckets: dict = {}
for day in library.list_dates(project["video_root"]):
for name in _video_names(project, day["date"]):
rel = f"{day['date']}/{name}"
timing = _sidecar(project, rel) or known.get(rel) or {}
cycle = timing.get("working_day") or day["date"]
bucket = buckets.setdefault(cycle, {"cycle": cycle, "video_count": 0,
"flagged": 0, "first_start": None})
bucket["video_count"] += 1
if not timing.get("started_at") or not timing.get("trusted"):
bucket["flagged"] += 1
start = timing.get("started_at")
if start and (bucket["first_start"] is None or start < bucket["first_start"]):
bucket["first_start"] = start
return sorted(buckets.values(), key=lambda b: b["cycle"], reverse=True)
def _video_names(project: dict, date: str) -> List[str]:
"""Filenames only — `library.list_videos` runs ffprobe on every file, which
is far too much work just to count what is in a cycle."""
import os as _os
from backend import video as video_module
folder = _os.path.join(library._effective_root(project["video_root"]), date)
try:
return [f for f in _os.listdir(folder)
if f.lower().endswith(video_module.VIDEO_EXTS)]
except OSError:
return []
def cycle_videos(project_id: int, cycle: str) -> List[dict]:
"""Every recording in one cycle, in the order it was actually made."""
project = projects.get(project_id)
if project is None:
raise ArchiveIndexError("No such project")
known = index(project_id)
# A cycle normally draws from two folders, but a moved recording can come
# from any of them. Ask the index which folders actually contribute rather
# than running ffprobe across the whole archive to find out.
folders = {row["folder_date"] for row in known.values()
if row.get("working_day") == cycle}
folders.add(cycle)
# Berkas baru belum tentu ada di indeks; sidecar-nya bisa memindahkannya ke
# siklus ini dari folder tanggal sebelah.
for day in library.list_dates(project["video_root"]):
if day["date"] in folders:
continue
for name in _video_names(project, day["date"]):
side = _sidecar(project, f"{day['date']}/{name}")
if side and side["working_day"] == cycle:
folders.add(day["date"])
break
rows = []
for day in library.list_dates(project["video_root"]):
if day["date"] not in folders:
continue
for video in library.list_videos(project["video_root"], day["date"], project_id):
rel = video["rel"]
timing = _sidecar(project, rel) or known.get(rel) or {}
if (timing.get("working_day") or day["date"]) != cycle:
continue
rows.append({
**video,
"folder_date": day["date"],
"working_day": timing.get("working_day") or "",
"started_at": timing.get("started_at"),
"clock_trusted": bool(timing.get("trusted")),
"clock_error": timing.get("error") or "",
"moved": bool(timing.get("working_day")
and timing["working_day"] != day["date"]),
"truck_hits": timing.get("truck_hits"),
"truck_samples": timing.get("truck_samples"),
})
return assign_batch_numbers(rows)
TRUCK_SAMPLES = 12
"""Frames sampled per recording for the truck check (REQ-166).
Enough to answer "is there a truck in this recording at all", which is the
assumption the whole batch numbering rests on: recording starts when a truck
arrives and stops when it leaves, so one file is one batch. Reading every frame
to time the truck's arrival precisely would cost hours of GPU for an answer the
recording trigger already gives.
"""
def store_truck(project_id: int, video_rel: str, hits: int, samples: int,
model_path: str) -> None:
"""One statement, one cursor.
This used to UPDATE and then call `store()` for the missing-row case — which
opened a second connection while the first still held a write transaction,
and SQLite answered "database is locked" eight recordings into the scan.
"""
label = os.path.basename(os.path.dirname(model_path))
with db.cursor() as cur:
cur.execute(
"""INSERT INTO video_clock (project_id, video_rel, folder_date,
truck_hits, truck_samples, truck_model,
truck_checked_at)
VALUES (?, ?, ?, ?, ?, ?, ?)
ON CONFLICT(project_id, video_rel) DO UPDATE SET
truck_hits = excluded.truck_hits,
truck_samples = excluded.truck_samples,
truck_model = excluded.truck_model,
truck_checked_at = excluded.truck_checked_at""",
(project_id, video_rel, video_rel.split("/")[0], hits, samples,
label, time.time()),
)
def count_truck_frames(path: str, model, conf: float = 0.45,
samples: int = TRUCK_SAMPLES) -> tuple:
"""How many of `samples` evenly spaced frames show a truck."""
import cv2
capture = cv2.VideoCapture(path)
if not capture.isOpened():
raise ArchiveIndexError(f"Could not open {path}")
total = int(capture.get(cv2.CAP_PROP_FRAME_COUNT) or 0)
hits = taken = 0
try:
for index in range(samples):
# Spread across the middle 90%: the very first and last frames of a
# trigger-started recording can catch the truck half out of shot.
position = int(total * (0.05 + 0.9 * index / max(1, samples - 1)))
capture.set(cv2.CAP_PROP_POS_FRAMES, position)
ok, frame = capture.read()
if not ok or frame is None:
continue
taken += 1
result = model.predict(frame, conf=conf, verbose=False)[0]
names = model.names
if any(names[int(box.cls[0])] == "truck" for box in result.boxes):
hits += 1
finally:
capture.release()
return hits, taken
def queue_truck_scan(project_id: int, model_path: str, rescan: bool = False) -> dict:
project = projects.get(project_id)
if project is None:
raise ArchiveIndexError("No such project")
if not os.path.isfile(model_path):
raise ArchiveIndexError(f"Model not found: {model_path}")
return jobs.create(
"truck-scan",
params={"model_path": model_path, "rescan": rescan},
project_id=project_id,
message="checking each recording for a truck",
).to_dict()
@jobs.handler("truck-scan")
def _run_truck_scan(job) -> None:
import numpy as np
from ultralytics import YOLO
project = projects.get(job.project_id)
model_path = job.params["model_path"]
rescan = bool(job.params.get("rescan"))
known = index(job.project_id)
todo = []
for day in library.list_dates(project["video_root"]):
for name in _video_names(project, day["date"]):
rel = f"{day['date']}/{name}"
row = known.get(rel) or {}
if not rescan and row.get("truck_samples"):
continue
todo.append(rel)
job.progress(0, len(todo))
job.log(f"Checking {len(todo)} recording(s) for a truck, "
f"{TRUCK_SAMPLES} frames each, with {os.path.basename(model_path)}")
model = YOLO(model_path)
model(np.zeros((720, 1280, 3), dtype=np.uint8), imgsz=640, verbose=False)
empty = broken = 0
for position, rel in enumerate(todo):
if job.cancelled:
job.log(f"Cancelled after {position} recording(s)")
return
try:
path = library.resolve(project["video_root"], rel)
hits, taken = count_truck_frames(path, model)
except Exception as exc:
store_truck(job.project_id, rel, 0, 0, model_path)
job.log(f"{rel}: {exc}")
broken += 1
job.progress(position + 1, len(todo))
continue
store_truck(job.project_id, rel, hits, taken, model_path)
if taken and hits == 0:
empty += 1
job.log(f"{rel}: no truck in any of {taken} sampled frames — "
"this recording may not be a batch")
job.progress(position + 1, len(todo), rel)
job.log(f"Done. {empty} recording(s) with no truck, {broken} unreadable")
def queue_scan(project_id: int, rescan: bool = False) -> dict:
project = projects.get(project_id)
if project is None:
raise ArchiveIndexError("No such project")
return jobs.create(
"clock-scan",
params={"rescan": rescan},
project_id=project_id,
message="reading timestamps from the archive",
).to_dict()
@jobs.handler("clock-scan")
def _run_scan(job) -> None:
project = projects.get(job.project_id)
rescan = bool(job.params.get("rescan"))
existing = index(job.project_id)
todo = []
for day in library.list_dates(project["video_root"]):
for video in library.list_videos(project["video_root"], day["date"]):
rel = video["rel"]
known = existing.get(rel)
# A hand-entered time is never re-read; a rescan redoes the rest.
if known and (known["source"] == "manual"
or (not rescan and known["started_at"])):
continue
todo.append(rel)
job.progress(0, len(todo))
job.log(f"Reading the timestamp overlay from {len(todo)} recording(s)")
read = flagged = failed = 0
for position, rel in enumerate(todo):
if job.cancelled:
job.log(f"Cancelled after {position} recording(s)")
return
side = _sidecar(project, rel)
if side is not None:
store(job.project_id, rel, side["started_at"], confidence=1.0,
agreeing=99, source=side["source"])
read += 1
job.progress(position + 1, len(todo), rel)
continue
try:
path = library.resolve(project["video_root"], rel)
result = video_clock.read_video_start(path)
except Exception as exc:
store(job.project_id, rel, None, error=f"{type(exc).__name__}: {exc}")
job.log(f"{rel}: {exc}")
failed += 1
job.progress(position + 1, len(todo))
continue
if result["start"] is None:
store(job.project_id, rel, None, error=result.get("error", "unreadable"))
failed += 1
else:
confidence, agreeing = result["confidence"], result.get("agreeing", 1)
store(job.project_id, rel, result["start"].strftime("%Y-%m-%d %H:%M:%S"),
confidence=confidence, agreeing=agreeing)
if _trusted(confidence, agreeing):
read += 1
else:
flagged += 1
job.log(f"{rel}: {result['start']} — low confidence "
f"({confidence}, {agreeing} frame(s) agreed), needs review")
job.progress(position + 1, len(todo), f"{rel}")
job.log(f"Read {read}, flagged {flagged} for review, {failed} unreadable")