Files
reTraining/backend/video_clock.py
T
asus 5c7c122105 feat: add counting bench, triage, and dataset modules
This commit includes major additions and updates to the frontend and backend architectures, introducing new dataset management, live counting features, batch processing, and triage logic. Includes new UI pages, components, and API routes.
2026-08-14 16:28:52 +07:00

250 lines
9.1 KiB
Python

"""Read the burned-in timestamp from a recording's overlay (REQ-160).
The archive's folder names do not say when a recording was made. Neither does
its mtime: `2026-08-13/batch003.mp4` is 38 minutes long but the next file's
mtime is 13 minutes later, because those are file *copy* times, not recording
times. The only trustworthy clock is the one the camera burns into the top-right
of every frame, in a fixed font at a fixed position:
2026-08-13 08:27:28
So this reads it. Not a general OCR — 12 glyphs (0-9, '-', ':') in one typeface
at one size, which a nearest-template match handles exactly and without adding
an OCR dependency to the image.
Isolating the text uses the one thing that distinguishes it from the wall and
the sacks behind it: it is bright *and* outlined in black. A plain brightness
threshold picks up a lit wall and merges glyphs together; requiring a dark pixel
within a few px of every bright one does not.
One caveat this module deliberately does not paper over: video time is not real
time. Measured on batch003, 1,940 seconds of video covers 783 seconds of wall
clock — the camera records at roughly 10 fps and stores at 25. So a file's
duration says nothing about when it ended, and only the *start* timestamp is
trusted here.
"""
import os
from typing import Optional
# The overlay's box in a 1280x720 frame. Glyphs are found inside it by column
# runs rather than at fixed offsets: '1', '-' and ':' are narrower than a digit,
# so a slot grid measured off one timestamp clips the wider glyphs of the next.
# Where the overlay sits in a 1280x720 frame. Nearly every recording puts it in
# the same rows; the one 1080p file in the archive lands ~10 px lower once
# scaled down, so a second band is tried for it rather than detecting the band
# per frame — detection was flakier than the two fixed guesses it replaced.
CANDIDATE_BANDS = ((28, 60), (38, 74), (20, 54))
CROP_LEFT, CROP_RIGHT = 950, 1268
SCALE = 3
GLYPH_SIZE = (24, 32)
GLYPH_COUNT = 18 # "YYYY-MM-DDHH:MM:SS" without the space
MIN_RUN_WIDTH = 4
TEMPLATE_PATH = os.path.join(os.path.dirname(__file__), "assets", "clock_glyphs.npz")
_templates = None
_keys = None
class ClockError(Exception):
pass
def _load():
global _templates, _keys
if _templates is None:
import numpy as np
data = np.load(TEMPLATE_PATH)
_keys = list(data.keys())
_templates = np.stack([data[k] for k in _keys])
return _templates, _keys
def _mask(frame, band=CANDIDATE_BANDS[0]):
"""The overlay's glyphs in one candidate band, isolated from the scene.
Bright alone is not enough — a lit wall clears any brightness threshold and
merges the glyphs into one blob. What separates the text is that every
stroke is outlined in black, so a bright pixel only counts when a dark one
sits within a few pixels of it.
"""
import cv2
import numpy as np
if frame.shape[0] != 720 or frame.shape[1] != 1280:
frame = cv2.resize(frame, (1280, 720))
grey = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)
roi = grey[band[0]:band[1], CROP_LEFT:CROP_RIGHT]
roi = cv2.resize(roi, None, fx=SCALE, fy=SCALE, interpolation=cv2.INTER_CUBIC)
bright = (roi > 195).astype(np.uint8)
dark = (roi < 90).astype(np.uint8)
near_dark = cv2.dilate(dark, np.ones((13, 13), np.uint8))
return ((bright & near_dark) * 255).astype(np.uint8)
def _glyphs(binary):
"""Cut the strip into one tight image per glyph, left to right."""
import cv2
import numpy as np
columns = binary.sum(axis=0) // 255
runs, start = [], None
for index, value in enumerate(columns):
if value > 0 and start is None:
start = index
elif value == 0 and start is not None:
if index - start >= MIN_RUN_WIDTH:
runs.append((start, index))
start = None
if start is not None:
runs.append((start, len(columns)))
out = []
for left, right in runs:
column = binary[:, left:right]
rows = np.where(column.sum(axis=1) > 0)[0]
if len(rows) == 0:
continue
tight = column[rows[0]:rows[-1] + 1, :]
out.append(cv2.resize(tight, GLYPH_SIZE, interpolation=cv2.INTER_AREA).astype(np.float32))
return out
def _decode(patches) -> tuple:
import numpy as np
templates, keys = _load()
chars, worst = [], 0.0
for patch in patches:
distances = ((templates - patch) ** 2).sum(axis=(1, 2))
order = np.argsort(distances)
best, runner_up = distances[order[0]], distances[order[1]]
chars.append(keys[int(order[0])])
worst = max(worst, best / max(1.0, runner_up))
return "".join(chars), round(1.0 - min(1.0, worst), 3)
def read_frame(frame) -> tuple:
"""Decode one frame's overlay. Returns (text, confidence).
Confidence is the worst per-glyph separation across the strip — the distance
to the best template over the distance to the runner-up. A glyph that matches
its own template several times better than any other is safe; one that barely
wins is what a wrong digit looks like. Every candidate band is tried and the
best *parseable* reading wins, so a misplaced band scores itself out rather
than silently recording the wrong hour.
"""
best = ("", 0.0)
for band in CANDIDATE_BANDS:
patches = _glyphs(_mask(frame, band))
if len(patches) != GLYPH_COUNT:
continue
text, confidence = _decode(patches)
if parse(text) is not None and confidence > best[1]:
best = (text, confidence)
return best
def _shaped(text: str) -> bool:
if len(text) != 18:
return False
for got, want in zip(text, "dddd-dd-dddd:dd:dd"):
if want == "d" and not got.isdigit():
return False
if want != "d" and got != want:
return False
return True
MIN_YEAR, MAX_YEAR = 2015, 2100
def parse(text: str):
"""The decoded string as a datetime, or None if it is not a real one.
The year range matters: a single misread digit turned 2026 into 7026 on one
recording, and `strptime` accepts that happily. A timestamp outside these
bounds is a decoding failure, not a recording from the far future.
"""
import datetime
if not _shaped(text):
return None
try:
stamp = datetime.datetime.strptime(text, "%Y-%m-%d%H:%M:%S")
except ValueError:
return None
return stamp if MIN_YEAR <= stamp.year <= MAX_YEAR else None
def read_video_start(path: str, probe_seconds=(2, 8, 20, 45)) -> dict:
"""When the recording in `path` started, read off its own overlay.
Several frames are sampled rather than one. A single frame can be caught
mid-transition or behind a passing sack, and a lone unverifiable reading is
exactly the kind of thing that would silently reassign a video to the wrong
working day. A reading is accepted only when two frames agree, after
subtracting the video-time offset between them.
"""
import cv2
import datetime
capture = cv2.VideoCapture(path)
if not capture.isOpened():
raise ClockError(f"Could not open {path}")
fps = capture.get(cv2.CAP_PROP_FPS) or 25.0
readings = []
try:
for offset in probe_seconds:
capture.set(cv2.CAP_PROP_POS_FRAMES, int(offset * fps))
ok, frame = capture.read()
if not ok or frame is None:
continue
text, confidence = read_frame(frame)
stamp = parse(text)
if stamp is not None and confidence > 0.0:
readings.append({"at": offset, "stamp": stamp,
"text": text, "confidence": confidence})
finally:
capture.release()
if not readings:
return {"start": None, "confidence": 0.0, "readings": [],
"error": "no readable timestamp overlay"}
# Video time runs slower than the wall clock on these recordings, so two
# readings cannot be checked by assuming a 1:1 offset. What they must agree
# on is the ordering and a sane elapsed span.
first = readings[0]
agreed = [r for r in readings[1:]
if 0 <= (r["stamp"] - first["stamp"]).total_seconds() <= r["at"] * 2]
span = (readings[-1]["stamp"] - first["stamp"]).total_seconds()
return {
"start": first["stamp"],
"confidence": round(min(r["confidence"] for r in readings), 3),
"agreeing": len(agreed) + 1,
"readings": [{"at": r["at"], "text": r["text"]} for r in readings],
"rate": round(span / max(1, readings[-1]["at"] - first["at"]), 2) if span else None,
"error": "" if agreed else "only one frame produced a usable reading",
}
def working_day(stamp, cutoff_hour: int = 6) -> str:
"""The working day a recording belongs to (REQ-161).
A shift runs 06:00 to 06:00, so anything before the cutoff belongs to the
previous calendar day. `predict.py` has the same idea in `get_counting_date`,
with a different cutoff.
"""
import datetime
if stamp is None:
return ""
day = stamp.date()
if stamp.hour < cutoff_hour:
day = day - datetime.timedelta(days=1)
return day.isoformat()