Files
reTraining/backend/library.py
T
asus 5c7c122105 feat: add counting bench, triage, and dataset modules
This commit includes major additions and updates to the frontend and backend architectures, introducing new dataset management, live counting features, batch processing, and triage logic. Includes new UI pages, components, and API routes.
2026-08-14 16:28:52 +07:00

181 lines
6.5 KiB
Python

"""The video archive, read as `<video_root>/<date>/<batch>.<ext>` (REQ-010…012).
Read-only by construction: this module only ever lists and stats files, never
writes into the user's recordings (REQ-074).
"""
import os
import re
from typing import List, Optional
from backend import config, db, projects, video
class LibraryError(Exception):
pass
def _safe_join(video_root: str, *parts: str) -> str:
"""Join under the archive root, refusing anything that escapes it."""
root = os.path.realpath(video_root)
target = os.path.realpath(os.path.join(root, *parts))
if target != root and not target.startswith(root + os.sep):
raise LibraryError("Path is outside the video archive")
return target
def batch_label(filename: str) -> str:
"""`batch-4.mp4` -> `batch-4`. The filename is the batch's identity."""
return os.path.splitext(filename)[0]
def _batch_sort_key(filename: str):
"""Sort batch-2 before batch-10, and keep unnumbered names after them."""
numbers = re.findall(r"\d+", batch_label(filename))
return (0, int(numbers[0])) if numbers else (1, 0), filename.lower()
def _effective_root(video_root: str) -> str:
"""Return video_root if it exists, otherwise fall back to config.VIDEO_ROOT."""
if os.path.isdir(video_root):
return video_root
if os.path.isdir(config.VIDEO_ROOT):
return config.VIDEO_ROOT
return video_root
def list_dates(video_root: str) -> List[dict]:
"""Date folders, newest name first, with how many videos each holds."""
video_root = _effective_root(video_root)
if not os.path.isdir(video_root):
raise LibraryError(f"Video archive folder not found: {video_root}")
dates = []
for name in sorted(os.listdir(video_root), reverse=True):
path = os.path.join(video_root, name)
if not os.path.isdir(path) or name.startswith("."):
continue
try:
count = sum(1 for f in os.listdir(path) if f.lower().endswith(video.VIDEO_EXTS))
except OSError:
continue
dates.append({"date": name, "video_count": count})
return dates
def list_videos(video_root: str, date: str, project_id: Optional[int] = None) -> List[dict]:
"""Videos in one date folder, with metadata and how often each was used."""
video_root = _effective_root(video_root)
folder = _safe_join(video_root, date)
if not os.path.isdir(folder):
raise LibraryError(f"No such date in the archive: {date}")
project = projects.get(project_id) if project_id is not None else None
used = _usage(project_id)
videos = []
for filename in sorted(
(f for f in os.listdir(folder) if f.lower().endswith(video.VIDEO_EXTS)),
key=_batch_sort_key,
):
path = os.path.join(folder, filename)
rel = f"{date}/{filename}"
entry = {
"rel": rel,
"filename": filename,
"batch_label": batch_label(filename),
"used_count": used.get(os.path.realpath(path), 0),
}
try:
entry.update(video.probe(path))
if project is not None:
ensure_video_preview(project, rel)
except video.VideoError as exc:
# A file ffprobe cannot read still belongs in the list, flagged —
# hiding it would look like the archive is missing recordings.
entry.update({"duration": 0.0, "width": 0, "height": 0, "fps": 0.0,
"error": str(exc)})
videos.append(entry)
return videos
import threading
_conversion_queue = set()
_conversion_lock = threading.Lock()
#: Serialises preview transcodes. Each one saturates several cores on its own.
_conversion_slot = threading.Semaphore(1)
def ensure_video_preview(project: dict, rel: str) -> None:
"""Asynchronously convert video to H.264 if it's not natively web-supported."""
rel_key = rel.replace("/", "_")
base, _ = os.path.splitext(rel_key)
preview_filename = f"{base}.mp4"
preview_dir = os.path.join(config.project_dir(project["slug"]), "previews")
preview_path = os.path.join(preview_dir, preview_filename)
if os.path.isfile(preview_path):
return
try:
full_path = resolve(project["video_root"], rel)
info = video.probe(full_path)
if info.get("codec_name") == "h264" and full_path.lower().endswith(".mp4"):
return
except Exception:
return
with _conversion_lock:
if preview_path in _conversion_queue:
return
_conversion_queue.add(preview_path)
def _worker():
# One conversion at a time. The queue above only stops the *same* file
# being converted twice; it never bounded how many ran at once, so
# opening a date folder with 28 videos started 28 simultaneous x264
# encodes. That pinned every core, drove load average past 250, and
# starved everything else in the process — inference included.
with _conversion_slot:
try:
if os.path.isfile(preview_path):
return
os.makedirs(preview_dir, exist_ok=True)
# A preview only has to be watchable in a browser, so it is not
# worth `-preset medium -crf 18`: veryfast/23 encodes several
# times faster for a difference nobody scrubbing footage sees.
video.convert_to_h264(full_path, output_path=preview_path,
crf=23, preset="veryfast")
except Exception as exc:
print(f"[PREVIEW CONVERSION ERROR] {rel}: {exc}")
finally:
with _conversion_lock:
_conversion_queue.discard(preview_path)
threading.Thread(target=_worker, daemon=True).start()
def _usage(project_id: Optional[int]) -> dict:
"""How many batches already came out of each video path (REQ-012)."""
if project_id is None:
return {}
with db.cursor() as cur:
cur.execute(
"SELECT video_path, COUNT(*) FROM batches WHERE project_id = ? GROUP BY video_path",
(project_id,),
)
return {os.path.realpath(row[0]): row[1] for row in cur.fetchall()}
def resolve(video_root: str, rel: str) -> str:
"""Turn a `<date>/<file>` reference into an absolute path inside the archive."""
video_root = _effective_root(video_root)
path = _safe_join(video_root, rel)
if not os.path.isfile(path):
raise LibraryError(f"No such video: {rel}")
if not path.lower().endswith(video.VIDEO_EXTS):
raise LibraryError("That file is not a video")
return path