Files
reTraining/backend/library.py
T

220 lines
7.9 KiB
Python

"""The video archive, read as `<video_root>/<date>/<batch>.<ext>` (REQ-010…012).
Read-only by construction: this module only ever lists and stats files, never
writes into the user's recordings (REQ-074).
"""
import os
import re
from typing import List, Optional
from backend import config, db, projects, video
class LibraryError(Exception):
pass
def _safe_join(video_root: str, *parts: str) -> str:
"""Join under the archive root, refusing anything that escapes it."""
root = os.path.realpath(video_root)
target = os.path.realpath(os.path.join(root, *parts))
if target != root and not target.startswith(root + os.sep):
raise LibraryError("Path is outside the video archive")
return target
def batch_label(filename: str) -> str:
"""`batch-4.mp4` -> `batch-4`. The filename is the batch's identity."""
return os.path.splitext(filename)[0]
def _batch_sort_key(filename: str):
"""Sort batch-2 before batch-10, and keep unnumbered names after them."""
numbers = re.findall(r"\d+", batch_label(filename))
return (0, int(numbers[0])) if numbers else (1, 0), filename.lower()
def _effective_root(video_root: str) -> str:
"""Return video_root if it exists, otherwise fall back to config.VIDEO_ROOT."""
if os.path.isdir(video_root):
return video_root
if os.path.isdir(config.VIDEO_ROOT):
return config.VIDEO_ROOT
return video_root
def list_dates(video_root: str) -> List[dict]:
"""Date folders, newest name first, with how many videos each holds."""
video_root = _effective_root(video_root)
if not os.path.isdir(video_root):
raise LibraryError(f"Video archive folder not found: {video_root}")
dates = []
for name in sorted(os.listdir(video_root), reverse=True):
path = os.path.join(video_root, name)
if not os.path.isdir(path) or name.startswith("."):
continue
try:
count = sum(1 for f in os.listdir(path) if f.lower().endswith(video.VIDEO_EXTS))
except OSError:
continue
dates.append({"date": name, "video_count": count})
return dates
def list_videos(video_root: str, date: str, project_id: Optional[int] = None) -> List[dict]:
"""Videos in one date folder, with metadata and how often each was used."""
video_root = _effective_root(video_root)
folder = _safe_join(video_root, date)
if not os.path.isdir(folder):
raise LibraryError(f"No such date in the archive: {date}")
project = projects.get(project_id) if project_id is not None else None
used = _usage(project_id)
videos = []
for filename in sorted(
(f for f in os.listdir(folder) if f.lower().endswith(video.VIDEO_EXTS)),
key=_batch_sort_key,
):
path = os.path.join(folder, filename)
rel = f"{date}/{filename}"
entry = {
"rel": rel,
"filename": filename,
"batch_label": batch_label(filename),
"used_count": used.get(os.path.realpath(path), 0),
}
try:
entry.update(video.probe(path))
if project is not None:
ensure_video_preview(project, rel)
except video.VideoError as exc:
# A file ffprobe cannot read still belongs in the list, flagged —
# hiding it would look like the archive is missing recordings.
entry.update({"duration": 0.0, "width": 0, "height": 0, "fps": 0.0,
"error": str(exc)})
videos.append(entry)
return videos
class LibraryConflict(LibraryError):
pass
def create_date(video_root: str, date: str) -> str:
"""Make a YYYY-MM-DD folder; never touch an existing one (REQ-171)."""
if not re.fullmatch(r"\d{4}-\d{2}-\d{2}", date):
raise LibraryError("Folder name must be YYYY-MM-DD")
folder = _safe_join(_effective_root(video_root), date)
if os.path.exists(folder):
raise LibraryConflict(f"Folder already exists: {date}")
os.makedirs(folder)
return folder
def upload_video(video_root: str, date: str, filename: str, chunks) -> str:
"""Stream `chunks` into <date>/<filename>; atomic via .part + rename."""
name = os.path.basename(filename or "")
if not name or name != filename or not name.lower().endswith(video.VIDEO_EXTS):
raise LibraryError(f"Unsupported file: {filename}")
root = _effective_root(video_root)
if not os.path.isdir(_safe_join(root, date)):
raise LibraryError(f"No such date in the archive: {date}")
dest = _safe_join(root, date, name)
if os.path.exists(dest):
raise LibraryConflict(f"Already exists: {date}/{name}")
part = dest + ".part"
try:
with open(part, "wb") as handle:
for chunk in chunks:
handle.write(chunk)
os.replace(part, dest)
except BaseException:
if os.path.exists(part):
os.unlink(part)
raise
return f"{date}/{name}"
import threading
_conversion_queue = set()
_conversion_lock = threading.Lock()
#: Serialises preview transcodes. Each one saturates several cores on its own.
_conversion_slot = threading.Semaphore(1)
def ensure_video_preview(project: dict, rel: str) -> None:
"""Asynchronously convert video to H.264 if it's not natively web-supported."""
rel_key = rel.replace("/", "_")
base, _ = os.path.splitext(rel_key)
preview_filename = f"{base}.mp4"
preview_dir = os.path.join(config.project_dir(project["slug"]), "previews")
preview_path = os.path.join(preview_dir, preview_filename)
if os.path.isfile(preview_path):
return
try:
full_path = resolve(project["video_root"], rel)
info = video.probe(full_path)
if info.get("codec_name") == "h264" and full_path.lower().endswith(".mp4"):
return
except Exception:
return
with _conversion_lock:
if preview_path in _conversion_queue:
return
_conversion_queue.add(preview_path)
def _worker():
# One conversion at a time. The queue above only stops the *same* file
# being converted twice; it never bounded how many ran at once, so
# opening a date folder with 28 videos started 28 simultaneous x264
# encodes. That pinned every core, drove load average past 250, and
# starved everything else in the process — inference included.
with _conversion_slot:
try:
if os.path.isfile(preview_path):
return
os.makedirs(preview_dir, exist_ok=True)
# A preview only has to be watchable in a browser, so it is not
# worth `-preset medium -crf 18`: veryfast/23 encodes several
# times faster for a difference nobody scrubbing footage sees.
video.convert_to_h264(full_path, output_path=preview_path,
crf=23, preset="veryfast")
except Exception as exc:
print(f"[PREVIEW CONVERSION ERROR] {rel}: {exc}")
finally:
with _conversion_lock:
_conversion_queue.discard(preview_path)
threading.Thread(target=_worker, daemon=True).start()
def _usage(project_id: Optional[int]) -> dict:
"""How many batches already came out of each video path (REQ-012)."""
if project_id is None:
return {}
with db.cursor() as cur:
cur.execute(
"SELECT video_path, COUNT(*) FROM batches WHERE project_id = ? GROUP BY video_path",
(project_id,),
)
return {os.path.realpath(row[0]): row[1] for row in cur.fetchall()}
def resolve(video_root: str, rel: str) -> str:
"""Turn a `<date>/<file>` reference into an absolute path inside the archive."""
video_root = _effective_root(video_root)
path = _safe_join(video_root, rel)
if not os.path.isfile(path):
raise LibraryError(f"No such video: {rel}")
if not path.lower().endswith(video.VIDEO_EXTS):
raise LibraryError("That file is not a video")
return path