feat: sync live-count, exemplar annotation modules, and update .gitignore

This commit is contained in:
ervanfahriaw committed 2026-08-24 14:57:42 +07:00
1 parent b6624eeff9
commit ac95674c07
39 files changed
+3958 -423

No files matched your search

+5
View File
@@ -73,6 +73,11 @@ weights/
*.sqlite *.sqlite
*.sqlite3 *.sqlite3
# Asset exceptions
!backend/assets/
!backend/assets/**
# Training Logs & Caches # Training Logs & Caches
*.log *.log
*.tfevents* *.tfevents*
+14 -3
View File
@@ -36,6 +36,11 @@ class AutolabelRequest(BaseModel):
append: bool = False append: bool = False
custom_model_path: Optional[str] = None custom_model_path: Optional[str] = None
class Exemplar(BaseModel):
box: list[float] # [cx, cy, w, h], normalized 0..1
positive: bool = True
class PreviewRequest(BaseModel): class PreviewRequest(BaseModel):
frame_id: int frame_id: int
engine: str engine: str
@@ -44,6 +49,10 @@ class PreviewRequest(BaseModel):
min_box_frac: float = 0.0 min_box_frac: float = 0.0
target_class_names: Optional[list[str]] = None target_class_names: Optional[list[str]] = None
custom_model_path: Optional[str] = None custom_model_path: Optional[str] = None
# Drawn box exemplars (REQ-172): normalized cxcywh, positive or negative.
# Preview-only — `autolabel.start` deliberately has no equivalent.
exemplars: Optional[list[Exemplar]] = None
exemplar_class_name: Optional[str] = None
@router.post("/api/projects/{project_id}/batches") @router.post("/api/projects/{project_id}/batches")
@@ -162,14 +171,14 @@ async def autolabel_with_model(
@router.post("/api/batches/{batch_id}/preview") @router.post("/api/batches/{batch_id}/preview")
def preview_autolabel(batch_id: int, request: PreviewRequest) -> dict: def preview_autolabel(batch_id: int, request: PreviewRequest) -> dict:
from backend import autolabel, jobs from backend import jobs, preview
if not jobs.gpu_lock.acquire(timeout=20): if not jobs.gpu_lock.acquire(timeout=20):
busy = jobs.running_types() busy = jobs.running_types()
kind = busy[0] if busy else "background" kind = busy[0] if busy else "background"
raise HTTPException(409, f"The GPU is busy with a {kind} job — wait for it to finish") raise HTTPException(409, f"The GPU is busy with a {kind} job — wait for it to finish")
try: try:
shapes = autolabel.preview_frame( shapes = preview.preview_frame(
batch_id=batch_id, batch_id=batch_id,
frame_id=request.frame_id, frame_id=request.frame_id,
engine=request.engine, engine=request.engine,
@@ -177,7 +186,9 @@ def preview_autolabel(batch_id: int, request: PreviewRequest) -> dict:
iou_threshold=request.iou_threshold, iou_threshold=request.iou_threshold,
min_box_frac=request.min_box_frac, min_box_frac=request.min_box_frac,
target_class_names=request.target_class_names, target_class_names=request.target_class_names,
custom_model_path=request.custom_model_path custom_model_path=request.custom_model_path,
exemplars=[e.model_dump() for e in request.exemplars or []],
exemplar_class_name=request.exemplar_class_name,
) )
return {"shapes": shapes} return {"shapes": shapes}
except Exception as exc: except Exception as exc:
+30 -6
View File
@@ -15,9 +15,9 @@ router = APIRouter(tags=["live-count"])
class StartRequest(BaseModel): class StartRequest(BaseModel):
# Either a raw source (RTSP URL or absolute path) or an archive-relative # Either a live stream — which must be a WebRTC (WHEP) URL, REQ-176 — or an
# path like "2026-08-13/batch001.mp4", which the backend resolves — the # archive-relative path like "2026-08-13/batch001.mp4", which the backend
# frontend never needs to know where the archive is mounted. # resolves so the frontend never needs to know where the archive is mounted.
source: str = "" source: str = ""
source_rel: Optional[str] = None source_rel: Optional[str] = None
model_path: Optional[str] = None model_path: Optional[str] = None
@@ -60,14 +60,22 @@ def available_models(project_id: int) -> dict:
def start(project_id: int, request: StartRequest) -> dict: def start(project_id: int, request: StartRequest) -> dict:
project = project_or_404(project_id) project = project_or_404(project_id)
source = request.source source, whep = request.source, ""
if request.source_rel: if request.source_rel:
try: try:
source = library.resolve(project["video_root"], request.source_rel) source = library.resolve(project["video_root"], request.source_rel)
except library.LibraryError as exc: except library.LibraryError as exc:
raise HTTPException(400, str(exc)) raise HTTPException(400, str(exc))
elif source:
# A live source is a WebRTC URL and nothing else. The browser watches it
# over WebRTC; the counter decodes the RTSP leg of the same MediaMTX
# path, derived here (REQ-176).
try:
whep, source = live_count.whep_url(source), live_count.whep_to_rtsp(source)
except live_count.LiveCountError as exc:
raise HTTPException(400, str(exc))
if not source: if not source:
raise HTTPException(400, "Pick a video or enter a stream URL") raise HTTPException(400, "Pick a video or enter a WebRTC stream URL")
path = request.model_path path = request.model_path
if not path and request.model_version_id is not None: if not path and request.model_version_id is not None:
@@ -88,6 +96,7 @@ def start(project_id: int, request: StartRequest) -> dict:
unload_confirm_frames=request.unload_confirm_frames, unload_confirm_frames=request.unload_confirm_frames,
min_area_scale=request.min_area_scale, min_area_scale=request.min_area_scale,
spatial_dedup=request.spatial_dedup, spatial_dedup=request.spatial_dedup,
whep=whep,
) )
except live_count.LiveCountError as exc: except live_count.LiveCountError as exc:
raise HTTPException(400, str(exc)) raise HTTPException(400, str(exc))
@@ -118,9 +127,24 @@ def status() -> dict:
return live_count.status() return live_count.status()
@router.get("/api/live-count/overlay")
def overlay() -> dict:
"""Geometry only — boxes, line and counts for the frame just processed.
What the WebRTC preview draws over the video, instead of the server
encoding a JPEG per frame for it (REQ-177).
"""
return live_count.overlay()
@router.get("/api/live-count/stream") @router.get("/api/live-count/stream")
def stream(): def stream():
"""MJPEG of the annotated frames. Ends when the session does.""" """MJPEG of the annotated frames — the archive-file preview. Ends with the session."""
if live_count.status().get("preview") == "webrtc":
# Nothing is encoding JPEGs for this session; without this the generator
# would sit on a worker thread for ten seconds producing nothing.
raise HTTPException(409, "This session is watched over WebRTC, not MJPEG")
def frames(): def frames():
blank_streak = 0 blank_streak = 0
while True: while True:
+37
View File
@@ -40,6 +40,25 @@ class AssistRequest(BaseModel):
threshold: float = 0.5 threshold: float = 0.5
class PoolExemplar(BaseModel):
# Normalized xyxy against the frame, as drawn on the review canvas.
box: List[float]
positive: bool = True
class ExemplarLabelRequest(BaseModel):
# The whole frame-local pool, newest last (REQ-173/174).
exemplars: List[PoolExemplar]
class_id: int = 0
# The filter panel (REQ-175). Defaults mirror `exemplar.DEFAULTS`.
threshold: float = 0.5
iou_threshold: float = 0.8
min_box_frac: float = 0.002
max_detections: int = 100
# Off by default: a drag previews, only Apply writes.
apply: bool = False
@router.get("/api/frames/{frame_id}/annotations") @router.get("/api/frames/{frame_id}/annotations")
def list_annotations(frame_id: int) -> dict: def list_annotations(frame_id: int) -> dict:
target = review_store.frame(frame_id) target = review_store.frame(frame_id)
@@ -99,6 +118,24 @@ def assist(frame_id: int, request: AssistRequest) -> dict:
raise HTTPException(400, str(exc)) raise HTTPException(400, str(exc))
@router.post("/api/frames/{frame_id}/exemplar-label")
def exemplar_label(frame_id: int, request: ExemplarLabelRequest) -> dict:
from backend import exemplar as exemplar_store
try:
return exemplar_store.label(
frame_id, request.class_id,
[item.model_dump() for item in request.exemplars],
threshold=request.threshold,
iou_threshold=request.iou_threshold,
min_box_frac=request.min_box_frac,
max_detections=request.max_detections,
apply=request.apply,
)
except review_store.ReviewError as exc:
raise HTTPException(400, str(exc))
@router.post("/api/frames/{frame_id}/status") @router.post("/api/frames/{frame_id}/status")
def set_status(frame_id: int, request: StatusRequest) -> dict: def set_status(frame_id: int, request: StatusRequest) -> dict:
try: try:
Binary file not shown.
-122
View File
@@ -259,125 +259,3 @@ def _reset_reviewed(batch_id: int) -> None:
"AND review_status = 'approved'", "AND review_status = 'approved'",
(batch_id,), (batch_id,),
) )
def preview_frame(
batch_id: int,
frame_id: int,
engine: str,
threshold: float = DEFAULT_THRESHOLD,
iou_threshold: float = DEFAULT_IOU,
min_box_frac: float = 0.0,
target_class_names: Optional[List[str]] = None,
custom_model_path: Optional[str] = None
) -> List[dict]:
batch = batches.get(batch_id)
if not batch:
raise ValueError("No such batch")
project = projects.get(batch["project_id"])
frame = next((f for f in batches.frames(batch_id) if f["id"] == frame_id), None)
if not frame:
raise ValueError("Frame not found")
directory = batches.frames_dir(batch["project_slug"], batch_id)
frame_file = os.path.join(directory, frame["filename"])
fw = max(1, frame.get("width") or 1)
fh = max(1, frame.get("height") or 1)
yolo_model = None
sam3_target_classes = []
if engine == "sam3" and not custom_model_path:
allowed_classes_set = {c.strip().lower() for c in target_class_names} if target_class_names else None
if allowed_classes_set:
sam3_target_classes = [c for c in project["classes"] if c["name"].strip().lower() in allowed_classes_set or c["prompt"].strip().lower() in allowed_classes_set]
else:
sam3_target_classes = [c for c in project["classes"]]
prompts = [c["prompt"] for c in sam3_target_classes]
if prompts:
from backend.sam3_engine import get_engine
get_engine()
else:
from ultralytics import YOLO
if custom_model_path and os.path.isfile(custom_model_path):
m_path = custom_model_path
else:
m_path = projects.training_start_point(project)
with db.cursor() as cur:
cur.execute("SELECT weights_path FROM model_versions WHERE project_id = ? ORDER BY version DESC LIMIT 1", (project["id"],))
row = cur.fetchone()
if row and os.path.isfile(row[0]):
m_path = row[0]
yolo_model = YOLO(m_path)
name_to_class_id = {item["name"].strip().lower(): item["class_id"] for item in project["classes"]}
allowed_classes_set = {c.strip().lower() for c in target_class_names} if target_class_names else None
all_raw_detections = []
if yolo_model is not None:
results = yolo_model.predict(frame_file, conf=threshold, verbose=False)
if results and len(results) > 0:
model_names = results[0].names
for box in results[0].boxes:
cls_idx = int(box.cls[0].item())
raw_cls_name = str(model_names.get(cls_idx, cls_idx)).strip().lower()
target_class_id = name_to_class_id.get(raw_cls_name)
if target_class_id is None:
for item in project["classes"]:
if item["class_id"] == cls_idx:
target_class_id = item["class_id"]
break
if target_class_id is None and 0 <= cls_idx < len(project["classes"]):
target_class_id = project["classes"][cls_idx]["class_id"]
if target_class_id is None:
continue
target_cls_obj = next((c for c in project["classes"] if c["class_id"] == target_class_id), None)
proj_cls_name = target_cls_obj["name"].strip().lower() if target_cls_obj else ""
if allowed_classes_set is not None:
if (raw_cls_name not in allowed_classes_set and
proj_cls_name not in allowed_classes_set and
str(target_class_id) not in allowed_classes_set):
continue
score = float(box.conf[0].item())
xyxyn = box.xyxyn[0].tolist()
all_raw_detections.append(labeling.Detection(
class_id=target_class_id,
class_name=proj_cls_name or raw_cls_name,
box=[xyxyn[0]*fw, xyxyn[1]*fh, xyxyn[2]*fw, xyxyn[3]*fh],
score=score,
mask=None
))
if engine == "sam3" and sam3_target_classes:
prompts = [(c.get("prompt") or c["name"]).strip() for c in sam3_target_classes]
res = labeling.label_image(
frame_file, frame["filename"], prompts, threshold,
iou_threshold=iou_threshold, min_box_frac=min_box_frac
)
if not res.error and res.detections:
for det in res.detections:
if 0 <= det.class_id < len(sam3_target_classes):
real_cls = sam3_target_classes[det.class_id]
det.class_id = real_cls["class_id"]
det.class_name = real_cls["name"]
all_raw_detections.append(det)
kept = labeling.deduplicate(all_raw_detections, iou_threshold=iou_threshold)
items = []
for det in kept:
if project["label_type"] == "bbox" or det.mask is None:
geom = review.bbox(det.box[0]/fw, det.box[1]/fh, det.box[2]/fw, det.box[3]/fh)
items.append({"class_id": det.class_id, "geometry": geom, "score": det.score})
else:
for geometry in _geometries(det, fw, fh, project["label_type"]):
items.append({"class_id": det.class_id, "geometry": geometry, "score": det.score})
return items
+270
View File
@@ -0,0 +1,270 @@
"""Exemplar-driven manual labeling in the review editor (REQ-173/174/175).
A drag on the review canvas is not just a rectangle: it is a visual prompt.
The drawn box joins a frame-local pool, the pool is replayed against SAM3
together with the class's text prompt, and the whole class is re-detected on
that frame from the result.
The pool lives in the editor, not in the database, and is sent whole on every
call. That keeps this module stateless and matches REQ-172's reasoning: SAM3's
geometric prompts pool features from *this* image, so a pool only means
anything for as long as the user is looking at the frame it was drawn on.
Split out of `review.py` because that file is already at the 400-line limit.
"""
import json
import time
from typing import List, Optional
from backend import db, projects, review
# A detection this close to a box the user drew is the same object: the user's
# own shape wins, so the detection is dropped rather than stacked on top of it.
DUPLICATE_IOU = 0.6
# A detection overlapping a negative box by this much is what the user pointed
# at when they said "not this" (REQ-174). Lower than DUPLICATE_IOU because a
# negative is drawn roughly, around something the user wants gone.
NEGATIVE_IOU = 0.3
# Below this, no detection is really "inside" a drawn box, so a polygon project
# keeps the rectangle rather than snapping to an unrelated mask.
SNAP_IOU = 0.1
# What the filter panel opens with (REQ-175). Measured on a dense `sack` frame:
# NMS at 0.8 only removes near-duplicates and a 0.002 area floor only removes
# specks, where the aggressive-looking values delete real, touching objects.
DEFAULTS = {
"threshold": 0.5,
"iou_threshold": 0.8,
"min_box_frac": 0.002,
"max_detections": 100,
}
def _class_prompt(project_id: int, class_id: int) -> str:
project = projects.get(project_id)
for item in project["classes"]:
if item["class_id"] == class_id:
return (item.get("prompt") or item["name"]).strip()
raise review.ReviewError(f"Class {class_id} does not exist in this project")
def _rect(points: List[float], label_type: str) -> dict:
"""The drawn rectangle as a storable shape for this project."""
x0, y0, x1, y1 = points
if label_type == "bbox":
return review.bbox(x0, y0, x1, y1)
return review.polygon([(x0, y0), (x1, y0), (x1, y1), (x0, y1)])
def _cxcywh(points: List[float]) -> List[float]:
x0, y0, x1, y1 = points
return [(x0 + x1) / 2, (y0 + y1) / 2, x1 - x0, y1 - y0]
def _replace_class(frame_id: int, class_id: int, items: List[dict]) -> None:
"""Swap every shape of one class on one frame for a fresh set.
"Replace everything, re-add drawn": the user's exemplar shapes are part of
`items`, so they come back verbatim in the same transaction.
"""
now = time.time()
with db.cursor() as cur:
cur.execute("DELETE FROM annotations WHERE frame_id = ? AND class_id = ?",
(frame_id, class_id))
cur.executemany(
"""INSERT INTO annotations (frame_id, class_id, geometry, score, source, created_at)
VALUES (?, ?, ?, ?, ?, ?)""",
[(frame_id, class_id, json.dumps(item["geometry"]), item.get("score", 1.0),
item.get("source", "auto"), now) for item in items],
)
def _append_drawn(frame_id: int, class_id: int, drawn: List[dict]) -> int:
"""Add drawn shapes the frame does not already carry, leaving the rest alone.
The pool is re-sent whole on every call, so most of it is usually already
stored; only what is genuinely new gets inserted.
"""
from backend.labeling import _iou
existing = [review.to_box(row["geometry"]) for row in review.listing(frame_id)
if row["class_id"] == class_id]
fresh = [item for item in drawn
if not any(_iou(review.to_box(item["geometry"]), box) >= 0.9
for box in existing)]
for item in fresh:
review.add(frame_id, class_id, item["geometry"], source="manual")
return len(fresh)
def _drop_negative_overlaps(frame_id: int, class_id: int,
negatives: List[List[float]]) -> int:
"""Delete shapes of this class the user shift-dragged over (REQ-174).
Used on the path where SAM3 never runs; the re-detect path filters the
detections instead, which has the same effect on what ends up stored.
"""
from backend.labeling import _iou
doomed = [row["id"] for row in review.listing(frame_id)
if row["class_id"] == class_id
and any(_iou(review.to_box(row["geometry"]), box) >= NEGATIVE_IOU
for box in negatives)]
return review.delete_many(doomed)
def label(frame_id: int, class_id: int, exemplars: List[dict],
threshold: float = 0.5, iou_threshold: float = 0.8,
min_box_frac: float = 0.002, max_detections: int = 100,
apply: bool = False) -> dict:
"""Detect one class on one frame from the frame's exemplar pool.
`exemplars` is the whole pool, newest last, each
`{"box": [x0, y0, x1, y1], "positive": bool}` normalized to the frame.
Positive boxes are both prompts and labels; negative boxes are prompts and
deletions, never labels.
Nothing is written unless `apply` is set (REQ-175): a drag previews, the
filter panel re-previews, and only Apply touches the frame. Apply re-runs
rather than trusting shapes sent back from the browser — SAM3 is
deterministic for a given pool and threshold, so the second pass reproduces
what was previewed.
"""
from backend import batches, jobs
from backend.labeling import _iou, deduplicate
from PIL import Image
target = review.frame(frame_id)
if target is None:
raise review.ReviewError("No such frame")
label_type = target["label_type"]
prompt = _class_prompt(target["project_id"], class_id)
positives, negatives = [], []
for item in exemplars:
box = review.validate({"type": "bbox", "points": item["box"]}, "bbox")["points"]
(positives if item.get("positive", True) else negatives).append(box)
if not positives and not negatives:
raise review.ReviewError("No exemplars to run")
# The GPU lock is shared with background jobs. Shorter than `review.assist`'s
# 20s on purpose: this fires from a mouse gesture, so a long stall would feel
# like a hung editor — and the fallback keeps the drawing rather than failing.
if not jobs.gpu_lock.acquire(timeout=5):
busy = jobs.running_types()
kind = busy[0] if busy else "background"
drawn = [{"geometry": _rect(box, label_type), "score": 1.0, "source": "manual"}
for box in positives]
if apply:
# Never the replace path here: with no detections to put back, it
# would wipe the class and leave only the drawings. Applying a
# detection-less run just files the drawings and honours the
# negatives.
_append_drawn(frame_id, class_id, drawn)
_drop_negative_overlaps(frame_id, class_id, negatives)
return _result(frame_id, drawn, apply, redetected=False,
message=f"The GPU is busy with a {kind} job — this is your drawing "
"only, nothing was detected")
try:
from backend.sam3_engine import get_engine
path = batches.frame_path(frame_id)
with Image.open(path) as handle:
image = handle.convert("RGB")
width, height = image.size
engine = get_engine()
state = engine.open_state(image)
found = engine.apply_prompts(
state, threshold=threshold, text=prompt,
exemplars=[{"box": _cxcywh(box), "positive": True} for box in positives]
+ [{"box": _cxcywh(box), "positive": False} for box in negatives],
)
finally:
jobs.gpu_lock.release()
# The panel's filters, in the order the batch job applies them (REQ-175):
# area floor, then NMS, then the cap on how many survive.
if min_box_frac > 0:
floor = width * height * min_box_frac
found = [d for d in found
if (d.box[2] - d.box[0]) * (d.box[3] - d.box[1]) >= floor]
found = deduplicate(found, iou_threshold)
found.sort(key=lambda d: d.score, reverse=True)
if max_detections > 0:
found = found[:max_detections]
detections = [(_norm_box(d.box, width, height), d) for d in found]
items: List[dict] = []
# The user's own boxes first, so the duplicate check below measures against
# what they drew rather than the other way round.
for box in positives:
geometry = _rect(box, label_type)
if label_type != "bbox":
snapped = _snap(box, detections, width, height)
if snapped is not None:
geometry = snapped
items.append({"geometry": geometry, "score": 1.0, "source": "manual"})
for norm, detection in detections:
if any(_iou(norm, box) >= NEGATIVE_IOU for box in negatives):
continue
if any(_iou(norm, box) >= DUPLICATE_IOU for box in positives):
continue
for geometry in _detection_shapes(detection, norm, width, height, label_type):
items.append({"geometry": geometry, "score": detection.score, "source": "auto"})
if apply:
_replace_class(frame_id, class_id, items)
return _result(frame_id, items, apply, redetected=True, message=None)
def _result(frame_id: int, items: List[dict], applied: bool,
redetected: bool, message: Optional[str]) -> dict:
"""A preview carries the shapes; an apply also carries the frame as stored."""
return {
"shapes": items,
"applied": applied,
"redetected": redetected,
"message": message,
"annotations": review.listing(frame_id) if applied else None,
}
def _norm_box(box: List[float], width: int, height: int) -> List[float]:
return [box[0] / width, box[1] / height, box[2] / width, box[3] / height]
def _snap(drawn: List[float], detections, width: int, height: int) -> Optional[dict]:
"""The mask polygon of whatever SAM3 found inside a drawn box.
A rectangle is a bad polygon label, so in a polygon project the drag is a
prompt for the shape rather than the shape itself (REQ-173).
"""
from backend.labeling import _iou
best = None
best_iou = SNAP_IOU
for norm, detection in detections:
if detection.mask is None:
continue
overlap = _iou(norm, drawn)
if overlap >= best_iou:
best, best_iou = detection, overlap
if best is None:
return None
points = review.mask_to_polygons(best.mask)
if not points or len(points[0]) < 3:
return None
return review.polygon([(x / width, y / height) for x, y in points[0]])
def _detection_shapes(detection, norm: List[float], width: int, height: int,
label_type: str) -> List[dict]:
if label_type == "bbox" or detection.mask is None:
return [review.bbox(*norm)]
return [review.polygon([(x / width, y / height) for x, y in points])
for points in review.mask_to_polygons(detection.mask)
if len(points) >= 3]
+11 -1
View File
@@ -68,8 +68,13 @@ def label_image(
threshold: float, threshold: float,
iou_threshold: float = 0.8, iou_threshold: float = 0.8,
min_box_frac: float = 0.0, min_box_frac: float = 0.0,
exemplar_index: int = -1,
exemplars: Optional[List[dict]] = None,
) -> ImageResult: ) -> ImageResult:
"""Detect every prompt in one image and return the surviving instances.""" """Detect every prompt in one image and return the surviving instances.
When `exemplars` are given, the prompt at `exemplar_index` also carries them
as drawn box exemplars (REQ-172); every other prompt runs on text alone."""
try: try:
image = Image.open(image_path).convert("RGB") image = Image.open(image_path).convert("RGB")
except Exception as exc: # unreadable/corrupt frame: report, don't abort the job except Exception as exc: # unreadable/corrupt frame: report, don't abort the job
@@ -77,6 +82,11 @@ def label_image(
width, height = image.size width, height = image.size
try: try:
if exemplars and 0 <= exemplar_index < len(prompts):
detections = get_engine().detect_with_exemplars(
image, prompts, threshold, exemplar_index, exemplars
)
else:
detections = get_engine().detect(image, prompts, threshold) detections = get_engine().detect(image, prompts, threshold)
except Exception as exc: except Exception as exc:
return ImageResult(image_path, rel_path, width, height, error=str(exc)) return ImageResult(image_path, rel_path, width, height, error=str(exc))
+53 -127
View File
@@ -1,4 +1,4 @@
"""Live counting test bench: point a trained model at an RTSP stream and watch it count. """Live counting test bench: point a trained model at a live stream and watch it count.
This is a **test harness**, not the production counter. It reuses the real This is a **test harness**, not the production counter. It reuses the real
pipeline pieces from `algoritma-batch` — ByteTrack, the bbox stabiliser and the pipeline pieces from `algoritma-batch` — ByteTrack, the bbox stabiliser and the
@@ -22,6 +22,10 @@ import cv2
import numpy as np import numpy as np
from backend import config, jobs from backend import config, jobs
from backend import live_render
from backend.live_source import ( # re-exported: callers catch live_count.LiveCountError
LiveCountError, _is_stream, _open, whep_to_rtsp, whep_url,
)
# `algoritma-batch/src` is copied to /app/src in the image; in a source checkout # `algoritma-batch/src` is copied to /app/src in the image; in a source checkout
# it still lives under algoritma-batch/. Both are made importable as `src.*`. # it still lives under algoritma-batch/. Both are made importable as `src.*`.
@@ -31,10 +35,6 @@ for candidate in ("/app", os.path.join(_REPO, "algoritma-batch")):
sys.path.insert(0, candidate) sys.path.insert(0, candidate)
class LiveCountError(Exception):
pass
class Session: class Session:
"""One running counter. Owns a capture thread and the latest rendered frame.""" """One running counter. Owns a capture thread and the latest rendered frame."""
@@ -43,8 +43,11 @@ class Session:
dedup_radius: float, margin: int, imgsz: int, dedup_radius: float, margin: int, imgsz: int,
entry_travel_min: float, handoff_radius: float, entry_travel_min: float, handoff_radius: float,
unload_confirm_frames: int, min_area_scale: float, unload_confirm_frames: int, min_area_scale: float,
spatial_dedup: bool): spatial_dedup: bool, whep_url: str = ""):
self.source = source self.source = source
# When the browser watches the camera over WebRTC it never asks for the
# MJPEG, so encoding a JPEG per frame would be pure waste (REQ-177).
self.whep_url = whep_url
self.model_path = model_path self.model_path = model_path
self.line_y = line_y self.line_y = line_y
self.line_x_start = line_x_start self.line_x_start = line_x_start
@@ -75,6 +78,7 @@ class Session:
self.events: list[dict] = [] self.events: list[dict] = []
self._jpeg: Optional[bytes] = None self._jpeg: Optional[bytes] = None
self._overlay: dict = {}
self._counter = None # set once the worker builds it self._counter = None # set once the worker builds it
self._lock = threading.Lock() self._lock = threading.Lock()
self._thread = threading.Thread(target=self._run, name="live-count", daemon=True) self._thread = threading.Thread(target=self._run, name="live-count", daemon=True)
@@ -92,6 +96,10 @@ class Session:
with self._lock: with self._lock:
return self._jpeg return self._jpeg
def overlay(self) -> dict:
with self._lock:
return dict(self._overlay)
def move_line(self, line_y: Optional[int] = None, line_x_start: Optional[int] = None, def move_line(self, line_y: Optional[int] = None, line_x_start: Optional[int] = None,
line_x_end: Optional[int] = None) -> dict: line_x_end: Optional[int] = None) -> dict:
"""Reposition the counting line without restarting. """Reposition the counting line without restarting.
@@ -119,6 +127,8 @@ class Session:
return { return {
"running": self._thread.is_alive(), "running": self._thread.is_alive(),
"source": self.source, "source": self.source,
"whep_url": self.whep_url,
"preview": "webrtc" if self.whep_url else "mjpeg",
"model_path": self.model_path, "model_path": self.model_path,
"error": self.error, "error": self.error,
"frames": self.frames, "frames": self.frames,
@@ -227,7 +237,11 @@ class Session:
self.fps = since / (now - tick) self.fps = since / (now - tick)
tick, since = now, 0 tick, since = now, 0
self._render(frame, inside, outside, counter) self._set_overlay(inside, outside, counter)
if not self.whep_url:
jpeg = live_render.render(self, frame, inside, outside, counter)
with self._lock:
self._jpeg = jpeg
except Exception as exc: # surfaced in status(), not swallowed except Exception as exc: # surfaced in status(), not swallowed
self.error = f"{type(exc).__name__}: {exc}" self.error = f"{type(exc).__name__}: {exc}"
finally: finally:
@@ -263,59 +277,33 @@ class Session:
if not self.error: if not self.error:
self.error = f"trace write failed: {exc}" self.error = f"trace write failed: {exc}"
def _render(self, frame, detections, ignored, counter) -> None: def _set_overlay(self, detections, ignored, counter) -> None:
height, width = frame.shape[:2] """The same drawing `_render` burns into the JPEG, as geometry.
# Shade what the region excludes. Without this the neighbouring truck's
# sacks simply vanish from the overlay, and "are they being ignored?"
# looks identical to "is the model missing them?".
if self.line_x_start > 0 or self.line_x_end < width:
shade = frame.copy()
if self.line_x_start > 0:
cv2.rectangle(shade, (0, 0), (self.line_x_start, height), (0, 0, 0), -1)
if self.line_x_end < width:
cv2.rectangle(shade, (self.line_x_end, 0), (width, height), (0, 0, 0), -1)
cv2.addWeighted(shade, 0.55, frame, 0.45, 0, frame)
# Ignored detections stay visible, in grey, so the region can be judged.
for det in ignored:
x1, y1, x2, y2 = (int(v) for v in det.bbox)
cv2.rectangle(frame, (x1, y1), (x2, y2), (130, 130, 130), 1)
for edge in (self.line_x_start, self.line_x_end):
if 0 < edge < width:
cv2.line(frame, (edge, 0), (edge, height), (255, 0, 255), 2)
cv2.putText(frame, "IGNORED", (max(4, self.line_x_start - 92), height - 14),
cv2.FONT_HERSHEY_SIMPLEX, 0.5, (255, 0, 255), 1, cv2.LINE_AA)
cv2.putText(frame, "IGNORED", (min(width - 88, self.line_x_end + 8), height - 14),
cv2.FONT_HERSHEY_SIMPLEX, 0.5, (255, 0, 255), 1, cv2.LINE_AA)
The browser draws it on a canvas over the WebRTC video, so the frames
themselves never pass through this process. Coordinates are in the
1280x720 space the pipeline works in; the canvas scales them.
"""
boxes = []
for det in detections: for det in detections:
x1, y1, x2, y2 = (int(v) for v in det.bbox) state = "counted" if counter.counted_tracks.get(det.track_id) else "tracked"
counted = bool(counter.counted_tracks.get(det.track_id)) boxes.append({"id": det.track_id, "b": [int(v) for v in det.bbox],
colour = (74, 222, 128) if counted else (248, 191, 113) "c": round(float(det.confidence), 2), "s": state})
cv2.rectangle(frame, (x1, y1), (x2, y2), colour, 2) for det in ignored:
cv2.putText(frame, f"#{det.track_id} {det.confidence:.2f}", (x1, max(14, y1 - 6)), boxes.append({"id": det.track_id, "b": [int(v) for v in det.bbox],
cv2.FONT_HERSHEY_SIMPLEX, 0.45, colour, 1, cv2.LINE_AA) "c": round(float(det.confidence), 2), "s": "ignored"})
payload = {
cv2.line(frame, (self.line_x_start, self.line_y), (self.line_x_end, self.line_y), "frame": self.frames,
(0, 255, 255), 2) "line": {"y": self.line_y, "x_start": self.line_x_start,
for edge in (self.line_y - self.margin, self.line_y + self.margin): "x_end": self.line_x_end, "margin": self.margin},
cv2.line(frame, (self.line_x_start, edge), (self.line_x_end, edge), "boxes": boxes,
(0, 160, 160), 1) "loading": self.loading,
"unloading": self.unloading,
panel = f"IN {self.loading} OUT {self.unloading} NET {self.loading - self.unloading}" "net": self.loading - self.unloading,
cv2.rectangle(frame, (12, 12), (12 + 9 * len(panel) + 20, 84), (0, 0, 0), -1) "fps": round(self.fps, 1),
cv2.putText(frame, panel, (24, 46), cv2.FONT_HERSHEY_SIMPLEX, 0.8, }
(74, 222, 128), 2, cv2.LINE_AA)
cv2.putText(frame, f"{self.fps:.1f} fps {self.tracked} tracked {self.ignored} ignored",
(24, 72), cv2.FONT_HERSHEY_SIMPLEX, 0.55, (200, 200, 200), 1, cv2.LINE_AA)
ok, buffer = cv2.imencode(".jpg", frame, [cv2.IMWRITE_JPEG_QUALITY, 75])
if ok:
with self._lock: with self._lock:
self._jpeg = buffer.tobytes() self._overlay = payload
def _too_small(bbox, scale: float) -> bool: def _too_small(bbox, scale: float) -> bool:
"""`predict.py`'s perspective curve: a sack at the top of the frame is """`predict.py`'s perspective curve: a sack at the top of the frame is
@@ -339,74 +327,6 @@ def _too_small(bbox, scale: float) -> bool:
return (x2 - x1) * (y2 - y1) < minimum * scale return (x2 - x1) * (y2 - y1) < minimum * scale
def _is_stream(source: str) -> bool:
return str(source).startswith(("rtsp://", "rtmp://", "http://", "https://"))
class _ThreadedStream:
"""Decode in a background thread and always hand out the newest frame.
A plain VideoCapture.read() on RTSP is blocking, and decoding 1080p costs
more than inference does — measured here at 6.7 fps end-to-end against 200
fps for the model itself. Worse, reading slower than the camera sends builds
a backlog, so the picture drifts further behind real time the longer it
runs. Dropping stale frames keeps latency flat, which is what a counting
test needs to mean anything. `predict.py` does the same thing.
"""
def __init__(self, source: str):
self._capture = cv2.VideoCapture(source)
try:
self._capture.set(cv2.CAP_PROP_BUFFERSIZE, 1)
except Exception:
pass
self._frame = None
self._lock = threading.Lock()
self._running = True
self._thread = threading.Thread(target=self._pump, daemon=True)
if self._capture.isOpened():
self._thread.start()
def _pump(self) -> None:
while self._running:
ok, frame = self._capture.read()
if not ok:
time.sleep(0.01)
continue
with self._lock:
self._frame = frame
def isOpened(self) -> bool:
return self._capture.isOpened()
def read(self):
with self._lock:
if self._frame is None:
return False, None
frame, self._frame = self._frame, None
return True, frame
def release(self) -> None:
self._running = False
# The thread is only started when the capture opened, so a failed
# source would otherwise raise "cannot join thread before it is
# started" here — inside the caller's finally, skipping the GPU lock
# release and wedging every later session on "GPU busy".
if self._thread.is_alive():
self._thread.join(timeout=2)
self._capture.release()
def _open(source: str):
if _is_stream(source):
os.environ.setdefault(
"OPENCV_FFMPEG_CAPTURE_OPTIONS",
"rtsp_transport;tcp|buffer_size;20480000|max_delay;500000",
)
return _ThreadedStream(source)
return cv2.VideoCapture(source)
# ---- module-level single session ---------------------------------------- # ---- module-level single session ----------------------------------------
_session: Optional[Session] = None _session: Optional[Session] = None
@@ -417,7 +337,8 @@ def start(source: str, model_path: str, line_y: int, line_x_start: int, line_x_e
conf: float = 0.35, dedup_radius: float = 60.0, margin: int = 5, conf: float = 0.35, dedup_radius: float = 60.0, margin: int = 5,
imgsz: int = 640, entry_travel_min: float = 60.0, imgsz: int = 640, entry_travel_min: float = 60.0,
handoff_radius: float = 100.0, unload_confirm_frames: int = 3, handoff_radius: float = 100.0, unload_confirm_frames: int = 3,
min_area_scale: float = 1.0, spatial_dedup: bool = False) -> dict: min_area_scale: float = 1.0, spatial_dedup: bool = False,
whep: str = "") -> dict:
global _session global _session
with _guard: with _guard:
if _session is not None and _session.status()["running"]: if _session is not None and _session.status()["running"]:
@@ -429,7 +350,7 @@ def start(source: str, model_path: str, line_y: int, line_x_start: int, line_x_e
_session = Session(source, model_path, line_y, line_x_start, line_x_end, _session = Session(source, model_path, line_y, line_x_start, line_x_end,
conf, dedup_radius, margin, imgsz, entry_travel_min, conf, dedup_radius, margin, imgsz, entry_travel_min,
handoff_radius, unload_confirm_frames, min_area_scale, handoff_radius, unload_confirm_frames, min_area_scale,
spatial_dedup) spatial_dedup, whep)
_session.start() _session.start()
time.sleep(0.4) # let an immediate failure surface in the response time.sleep(0.4) # let an immediate failure surface in the response
return _session.status() return _session.status()
@@ -459,9 +380,14 @@ def status() -> dict:
"fps": 0.0, "tracked": 0, "ignored": 0, "too_small": 0, "traced": 0, "fps": 0.0, "tracked": 0, "ignored": 0, "too_small": 0, "traced": 0,
"trace_path": "", "elapsed": 0.0, "events": [], "trace_path": "", "elapsed": 0.0, "events": [],
"error": "", "source": "", "model_path": "", "error": "", "source": "", "model_path": "",
"whep_url": "", "preview": "mjpeg",
"line": {"y": 0, "x_start": 0, "x_end": 1280}} "line": {"y": 0, "x_start": 0, "x_end": 1280}}
return _session.status() return _session.status()
def snapshot() -> Optional[bytes]: def snapshot() -> Optional[bytes]:
return _session.snapshot() if _session is not None else None return _session.snapshot() if _session is not None else None
def overlay() -> dict:
return _session.overlay() if _session is not None else {}
+60
View File
@@ -0,0 +1,60 @@
"""Burning the counting overlay into the frame, as an MJPEG.
Split out of `live_count.py` to keep it inside the 400-line limit. This is the
fallback preview, used for archive files; a WebRTC session draws the same
geometry on a canvas in the browser instead (REQ-177).
"""
import cv2
def render(session, frame, detections, ignored, counter):
height, width = frame.shape[:2]
# Shade what the region excludes. Without this the neighbouring truck's
# sacks simply vanish from the overlay, and "are they being ignored?"
# looks identical to "is the model missing them?".
if session.line_x_start > 0 or session.line_x_end < width:
shade = frame.copy()
if session.line_x_start > 0:
cv2.rectangle(shade, (0, 0), (session.line_x_start, height), (0, 0, 0), -1)
if session.line_x_end < width:
cv2.rectangle(shade, (session.line_x_end, 0), (width, height), (0, 0, 0), -1)
cv2.addWeighted(shade, 0.55, frame, 0.45, 0, frame)
# Ignored detections stay visible, in grey, so the region can be judged.
for det in ignored:
x1, y1, x2, y2 = (int(v) for v in det.bbox)
cv2.rectangle(frame, (x1, y1), (x2, y2), (130, 130, 130), 1)
for edge in (session.line_x_start, session.line_x_end):
if 0 < edge < width:
cv2.line(frame, (edge, 0), (edge, height), (255, 0, 255), 2)
cv2.putText(frame, "IGNORED", (max(4, session.line_x_start - 92), height - 14),
cv2.FONT_HERSHEY_SIMPLEX, 0.5, (255, 0, 255), 1, cv2.LINE_AA)
cv2.putText(frame, "IGNORED", (min(width - 88, session.line_x_end + 8), height - 14),
cv2.FONT_HERSHEY_SIMPLEX, 0.5, (255, 0, 255), 1, cv2.LINE_AA)
for det in detections:
x1, y1, x2, y2 = (int(v) for v in det.bbox)
counted = bool(counter.counted_tracks.get(det.track_id))
colour = (74, 222, 128) if counted else (248, 191, 113)
cv2.rectangle(frame, (x1, y1), (x2, y2), colour, 2)
cv2.putText(frame, f"#{det.track_id} {det.confidence:.2f}", (x1, max(14, y1 - 6)),
cv2.FONT_HERSHEY_SIMPLEX, 0.45, colour, 1, cv2.LINE_AA)
cv2.line(frame, (session.line_x_start, session.line_y), (session.line_x_end, session.line_y),
(0, 255, 255), 2)
for edge in (session.line_y - session.margin, session.line_y + session.margin):
cv2.line(frame, (session.line_x_start, edge), (session.line_x_end, edge),
(0, 160, 160), 1)
panel = f"IN {session.loading} OUT {session.unloading} NET {session.loading - session.unloading}"
cv2.rectangle(frame, (12, 12), (12 + 9 * len(panel) + 20, 84), (0, 0, 0), -1)
cv2.putText(frame, panel, (24, 46), cv2.FONT_HERSHEY_SIMPLEX, 0.8,
(74, 222, 128), 2, cv2.LINE_AA)
cv2.putText(frame, f"{session.fps:.1f} fps {session.tracked} tracked {session.ignored} ignored",
(24, 72), cv2.FONT_HERSHEY_SIMPLEX, 0.55, (200, 200, 200), 1, cv2.LINE_AA)
ok, buffer = cv2.imencode(".jpg", frame, [cv2.IMWRITE_JPEG_QUALITY, 75])
return buffer.tobytes() if ok else None
+131
View File
@@ -0,0 +1,131 @@
"""Opening a live source, and the URL algebra around it.
Split out of `live_count.py` so that file stays inside the 400-line limit: this
is the transport layer (what a "source" is, how it is opened, how a WebRTC URL
maps to the RTSP leg of the same stream), not the counting logic.
"""
import os
import threading
import time
import cv2
class LiveCountError(Exception):
pass
def _is_stream(source: str) -> bool:
return str(source).startswith(("rtsp://", "rtmp://", "http://", "https://"))
# MediaMTX fans one camera out to several protocols on one host: WHEP for
# browsers, RTSP for decoders. The ports are the server's, not ours, so they
# come from the environment rather than the code (REQ-176).
WHEP_PATH = os.environ.get("MEDIAMTX_WHEP_PATH", "/whep")
RTSP_PORT = os.environ.get("MEDIAMTX_RTSP_PORT", "8554")
def whep_url(source: str) -> str:
"""The URL a browser POSTs its SDP offer to, for a WebRTC stream URL."""
trimmed = source.rstrip("/")
return trimmed if trimmed.endswith(WHEP_PATH) else trimmed + WHEP_PATH
def whep_to_rtsp(source: str) -> str:
"""`http://host:8889/cam` -> `rtsp://host:8554/cam`.
The browser and the counter watch the same camera, but not over the same
protocol: WebRTC is what makes the *preview* cheap, while pulling it into
Python would add ICE and a jitter buffer on top of exactly the same H.264
decode RTSP already does. So the AI counts from the RTSP leg of the same
MediaMTX path — one ingest, two consumers.
"""
from urllib.parse import urlparse
parsed = urlparse(source)
if parsed.scheme not in ("http", "https") or not parsed.hostname:
raise LiveCountError(
"A live source must be a WebRTC (WHEP) URL, e.g. http://host:8889/cam")
path = parsed.path.rstrip("/")
if path.endswith(WHEP_PATH):
path = path[: -len(WHEP_PATH)]
if not path.strip("/"):
raise LiveCountError(f"No stream path in {source} — expected e.g. http://host:8889/cam")
return f"rtsp://{parsed.hostname}:{RTSP_PORT}{path}"
class _ThreadedStream:
"""Decode in a background thread and always hand out the newest frame.
A plain VideoCapture.read() on RTSP is blocking, and getting the frames
there costs far more than looking at them does — measured on this camera at
8.3 fps for decode alone against 205 fps for tracking and inference. Worse, reading slower than the camera sends builds
a backlog, so the picture drifts further behind real time the longer it
runs. Dropping stale frames keeps latency flat, which is what a counting
test needs to mean anything. `predict.py` does the same thing.
"""
def __init__(self, source: str):
self._capture = cv2.VideoCapture(source)
try:
self._capture.set(cv2.CAP_PROP_BUFFERSIZE, 1)
except Exception:
pass
self._frame = None
self._lock = threading.Lock()
self._running = True
self._thread = threading.Thread(target=self._pump, daemon=True)
if self._capture.isOpened():
self._thread.start()
def _pump(self) -> None:
while self._running:
ok, frame = self._capture.read()
if not ok:
time.sleep(0.01)
continue
with self._lock:
self._frame = frame
def isOpened(self) -> bool:
return self._capture.isOpened()
def read(self):
with self._lock:
if self._frame is None:
return False, None
frame, self._frame = self._frame, None
return True, frame
def release(self) -> None:
self._running = False
# The thread is only started when the capture opened, so a failed
# source would otherwise raise "cannot join thread before it is
# started" here — inside the caller's finally, skipping the GPU lock
# release and wedging every later session on "GPU busy".
if self._thread.is_alive():
self._thread.join(timeout=2)
self._capture.release()
# TCP, and it is worth writing down why, because the obvious reasoning gives the
# wrong answer here. The link to the streaming server is a ZeroTier VPN with 15%
# packet loss and a 41-104 ms round trip, which is exactly the case where UDP is
# supposed to win — and with the ffmpeg CLI it does, 16 fps against 6.8. Through
# OpenCV it loses badly: measured over 30 s on this camera, tcp 6.4 fps, udp with
# a socket buffer 4.4, bare udp 2.4. OpenCV drops what it cannot reassemble
# rather than showing it, so the loss becomes missing frames. Left overridable
# because on a clean LAN the answer flips back.
RTSP_TRANSPORT = os.environ.get("RTSP_TRANSPORT", "tcp")
def _open(source: str):
if _is_stream(source):
os.environ.setdefault(
"OPENCV_FFMPEG_CAPTURE_OPTIONS",
f"rtsp_transport;{RTSP_TRANSPORT}|buffer_size;1048576|max_delay;200000",
)
return _ThreadedStream(source)
return cv2.VideoCapture(source)
+151
View File
@@ -0,0 +1,151 @@
"""The auto-annotate preview: one frame, run now, nothing written (REQ-171/172).
Split out of `autolabel.py` to keep that file under the 400-line limit. The job
path and this path share `labeling.label_image`, so what the preview shows is
what a batch run would write — with one deliberate exception: drawn box
exemplars (REQ-172) only ever apply here. SAM3's geometric prompts pool features
from the current image, so replaying them on another frame would ask about
whatever happens to sit at those coordinates there.
"""
import os
from typing import List, Optional
from backend import batches, db, labeling, projects, review
from backend.autolabel import DEFAULT_IOU, DEFAULT_THRESHOLD, _geometries
def preview_frame(
batch_id: int,
frame_id: int,
engine: str,
threshold: float = DEFAULT_THRESHOLD,
iou_threshold: float = DEFAULT_IOU,
min_box_frac: float = 0.0,
target_class_names: Optional[List[str]] = None,
custom_model_path: Optional[str] = None,
exemplars: Optional[List[dict]] = None,
exemplar_class_name: Optional[str] = None,
) -> List[dict]:
batch = batches.get(batch_id)
if not batch:
raise ValueError("No such batch")
project = projects.get(batch["project_id"])
frame = next((f for f in batches.frames(batch_id) if f["id"] == frame_id), None)
if not frame:
raise ValueError("Frame not found")
directory = batches.frames_dir(batch["project_slug"], batch_id)
frame_file = os.path.join(directory, frame["filename"])
fw = max(1, frame.get("width") or 1)
fh = max(1, frame.get("height") or 1)
yolo_model = None
sam3_target_classes = []
if engine == "sam3" and not custom_model_path:
allowed_classes_set = {c.strip().lower() for c in target_class_names} if target_class_names else None
if allowed_classes_set:
sam3_target_classes = [c for c in project["classes"] if c["name"].strip().lower() in allowed_classes_set or c["prompt"].strip().lower() in allowed_classes_set]
else:
sam3_target_classes = [c for c in project["classes"]]
prompts = [c["prompt"] for c in sam3_target_classes]
if prompts:
from backend.sam3_engine import get_engine
get_engine()
else:
from ultralytics import YOLO
if custom_model_path and os.path.isfile(custom_model_path):
m_path = custom_model_path
else:
m_path = projects.training_start_point(project)
with db.cursor() as cur:
cur.execute("SELECT weights_path FROM model_versions WHERE project_id = ? ORDER BY version DESC LIMIT 1", (project["id"],))
row = cur.fetchone()
if row and os.path.isfile(row[0]):
m_path = row[0]
yolo_model = YOLO(m_path)
name_to_class_id = {item["name"].strip().lower(): item["class_id"] for item in project["classes"]}
allowed_classes_set = {c.strip().lower() for c in target_class_names} if target_class_names else None
all_raw_detections = []
if yolo_model is not None:
results = yolo_model.predict(frame_file, conf=threshold, verbose=False)
if results and len(results) > 0:
model_names = results[0].names
for box in results[0].boxes:
cls_idx = int(box.cls[0].item())
raw_cls_name = str(model_names.get(cls_idx, cls_idx)).strip().lower()
target_class_id = name_to_class_id.get(raw_cls_name)
if target_class_id is None:
for item in project["classes"]:
if item["class_id"] == cls_idx:
target_class_id = item["class_id"]
break
if target_class_id is None and 0 <= cls_idx < len(project["classes"]):
target_class_id = project["classes"][cls_idx]["class_id"]
if target_class_id is None:
continue
target_cls_obj = next((c for c in project["classes"] if c["class_id"] == target_class_id), None)
proj_cls_name = target_cls_obj["name"].strip().lower() if target_cls_obj else ""
if allowed_classes_set is not None:
if (raw_cls_name not in allowed_classes_set and
proj_cls_name not in allowed_classes_set and
str(target_class_id) not in allowed_classes_set):
continue
score = float(box.conf[0].item())
xyxyn = box.xyxyn[0].tolist()
all_raw_detections.append(labeling.Detection(
class_id=target_class_id,
class_name=proj_cls_name or raw_cls_name,
box=[xyxyn[0]*fw, xyxyn[1]*fh, xyxyn[2]*fw, xyxyn[3]*fh],
score=score,
mask=None
))
if engine == "sam3" and sam3_target_classes:
prompts = [(c.get("prompt") or c["name"]).strip() for c in sam3_target_classes]
# Exemplars belong to exactly one class — the chip that was active when
# they were drawn. An unknown name means no exemplar class, so the run
# falls back to plain text rather than silently attaching the boxes to
# whichever class happens to be first.
exemplar_index = -1
if exemplars and exemplar_class_name:
wanted = exemplar_class_name.strip().lower()
exemplar_index = next(
(i for i, c in enumerate(sam3_target_classes)
if c["name"].strip().lower() == wanted),
-1,
)
res = labeling.label_image(
frame_file, frame["filename"], prompts, threshold,
iou_threshold=iou_threshold, min_box_frac=min_box_frac,
exemplar_index=exemplar_index, exemplars=exemplars
)
if not res.error and res.detections:
for det in res.detections:
if 0 <= det.class_id < len(sam3_target_classes):
real_cls = sam3_target_classes[det.class_id]
det.class_id = real_cls["class_id"]
det.class_name = real_cls["name"]
all_raw_detections.append(det)
kept = labeling.deduplicate(all_raw_detections, iou_threshold=iou_threshold)
items = []
for det in kept:
if project["label_type"] == "bbox" or det.mask is None:
geom = review.bbox(det.box[0]/fw, det.box[1]/fh, det.box[2]/fw, det.box[3]/fh)
items.append({"class_id": det.class_id, "geometry": geom, "score": det.score})
else:
for geometry in _geometries(det, fw, fh, project["label_type"]):
items.append({"class_id": det.class_id, "geometry": geometry, "score": det.score})
return items
+35
View File
@@ -103,6 +103,41 @@ class Sam3Engine:
def detect_with_exemplars(
self,
image: Image.Image,
prompts: List[str],
threshold: float,
exemplar_index: int,
exemplars: List[dict],
) -> List[Detection]:
"""`detect()`, but one prompt also carries drawn box exemplars (REQ-172).
Still one `set_image` for the whole call. The prompt set is reset before
every class because `state["geometric_prompt"]` survives `set_text_prompt`
— without the reset, one class's boxes would leak into the next class.
"""
processor = Sam3Processor(self.model, device=self.device)
processor.confidence_threshold = threshold
detections: List[Detection] = []
with torch.autocast(self.device, dtype=self.autocast_dtype):
state = processor.set_image(image)
for class_id, prompt in enumerate(prompts):
processor.reset_all_prompts(state)
output = processor.set_text_prompt(prompt=prompt, state=state)
if class_id == exemplar_index:
for exemplar in exemplars:
output = processor.add_geometric_prompt(
box=exemplar["box"],
label=bool(exemplar.get("positive", True)),
state=state,
)
detections.extend(self._collect(output, class_id, prompt))
del state
return detections
# ---- interactive / exemplar prompting ------------------------------ # ---- interactive / exemplar prompting ------------------------------
BIN
View File
Binary file not shown.
+115
View File
@@ -123,9 +123,14 @@ box.
| `batches.py` | batch lifecycle | **new** | | `batches.py` | batch lifecycle | **new** |
| `review.py` | annotation CRUD, frame status, click-assist | **new** | | `review.py` | annotation CRUD, frame status, click-assist | **new** |
| `autolabel.py` | the SAM3 job over a whole batch | **new** | | `autolabel.py` | the SAM3 job over a whole batch | **new** |
| `preview.py` | one-frame preview for the auto-annotate modal (REQ-171,172) | **new** |
| `exemplar.py` | exemplar-driven labeling in the review editor (REQ-173,174) | **new** |
| `dataset.py` | merge into the master dataset, stable split | **new** | | `dataset.py` | merge into the master dataset, stable split | **new** |
| `evaluate.py` | validate base vs new model | **new** | | `evaluate.py` | validate base vs new model | **new** |
| `hardware.py` | VRAM detection → training defaults | **new** | | `hardware.py` | VRAM detection → training defaults | **new** |
| `live_count.py` | the live counting session: capture → track → count | **new** |
| `live_source.py` | what a source is, how it opens, WHEP↔RTSP (REQ-176) | **new** |
| `live_render.py` | the MJPEG overlay, archive-file preview only (REQ-177) | **new** |
| `api/` | the FastAPI routes, one module per domain | **new** | | `api/` | the FastAPI routes, one module per domain | **new** |
Removed: `uploads.py`, `static/index.html`, and the old flow's endpoints. Removed: `uploads.py`, `static/index.html`, and the old flow's endpoints.
@@ -167,6 +172,9 @@ POST /api/projects/{id}/batches # {rel, start_sec, end_sec, fps}
GET /api/batches/{id} # status + review progress (REQ-045) GET /api/batches/{id} # status + review progress (REQ-045)
GET /api/batches/{id}/frames # frames + statuses GET /api/batches/{id}/frames # frames + statuses
POST /api/batches/{id}/autolabel # {threshold} → job (REQ-030,032,034) POST /api/batches/{id}/autolabel # {threshold} → job (REQ-030,032,034)
POST /api/batches/{id}/preview # one frame, run now, nothing written;
# +exemplars[] {box:[cx,cy,w,h], positive}
# +exemplar_class_name (REQ-171,172)
DELETE /api/batches/{id}/classes/{class_id}/annotations # clear all shapes of class in batch (REQ-046) DELETE /api/batches/{id}/classes/{class_id}/annotations # clear all shapes of class in batch (REQ-046)
POST /api/batches/{ids}/approve # one or many, comma-separated → one merge job (REQ-131) POST /api/batches/{ids}/approve # one or many, comma-separated → one merge job (REQ-131)
GET /api/batches/{ids}/triage/summary # one or many, comma-separated (REQ-130) GET /api/batches/{ids}/triage/summary # one or many, comma-separated (REQ-130)
@@ -180,6 +188,7 @@ POST /api/frames/{id}/annotations # add a manual shape (REQ-042)
PATCH /api/annotations/{id} # move/resize/reclass PATCH /api/annotations/{id} # move/resize/reclass
DELETE /api/annotations/{id} DELETE /api/annotations/{id}
POST /api/frames/{id}/assist # click/box → SAM3 shape (REQ-043) POST /api/frames/{id}/assist # click/box → SAM3 shape (REQ-043)
POST /api/frames/{id}/exemplar-label # drawn pool → re-detect one class (REQ-173,174)
POST /api/frames/{id}/status # approved | rejected | pending (REQ-041) POST /api/frames/{id}/status # approved | rejected | pending (REQ-041)
POST /api/projects/{id}/train # → train job (REQ-060,061,062) POST /api/projects/{id}/train # → train job (REQ-060,061,062)
@@ -187,11 +196,36 @@ GET /api/projects/{id}/models # versions + metrics (REQ-063,06
GET /api/models/{id}/weights # download best.pt GET /api/models/{id}/weights # download best.pt
POST /api/models/{id}/promote # make it the project's base model (REQ-064) POST /api/models/{id}/promote # make it the project's base model (REQ-064)
GET /api/projects/{id}/live-count/models # weights this project can count with
POST /api/projects/{id}/live-count/start # {source|source_rel, model_path, dials}
# source must be a WHEP URL (REQ-176)
POST /api/live-count/stop
PATCH /api/live-count/line # move the line mid-session
GET /api/live-count/status # counts + preview: "webrtc" | "mjpeg"
GET /api/live-count/overlay # boxes/line/counts for the canvas (REQ-177)
GET /api/live-count/stream # MJPEG; 409 on a WebRTC session (REQ-177)
GET /api/jobs?project_id=… REQ-070,071 GET /api/jobs?project_id=… REQ-070,071
GET /api/jobs/{id} GET /api/jobs/{id}
POST /api/jobs/{id}/cancel POST /api/jobs/{id}/cancel
``` ```
### Live counting (REQ-176, REQ-177)
One camera, one ingest on the streaming server, two consumers:
```
camera ──▶ MediaMTX ──┬── WHEP :8889/cam/whep ──▶ browser <video> + <canvas> overlay
└── RTSP :8554/cam ──▶ backend decode → YOLO → counter
└─▶ GET /api/live-count/overlay
```
The user types only the WHEP URL; `live_source.whep_to_rtsp` derives the RTSP one, with the
ports coming from `MEDIAMTX_RTSP_PORT` / `MEDIAMTX_WHEP_PATH`. Because the browser plays the
camera directly, a live session encodes no JPEG at all — `live_render.py` runs only for
archive files, which have no WebRTC leg. The canvas draws exactly what `live_render.py`
would have burned in, so both previews describe the same session.
## Job flows ## Job flows
**extract (REQ-020…023).** `ffmpeg -ss <start> -to <end> -i <video> -vf fps=<n> -q:v 2 **extract (REQ-020…023).** `ffmpeg -ss <start> -to <end> -i <video> -vf fps=<n> -q:v 2
@@ -277,6 +311,87 @@ Pages:
The canvas editor is hand-written; the normalized-coordinate conventions already exist in The canvas editor is hand-written; the normalized-coordinate conventions already exist in
`sessions.py` (`annotations_payload`, `detections_payload`) as a reference. `sessions.py` (`annotations_payload`, `detections_payload`) as a reference.
**Auto-annotate modal (REQ-171, REQ-172).** `AutoAnnotateModal.jsx` splits into
`PreviewShapes.jsx` (the result overlay, shared with the mass modal),
`ClassPromptPanel.jsx` (class chips + the editable SAM3 prompt) and `ExemplarCanvas.jsx`
(the drag-to-draw layer). All three overlays and the `<img>` share one shrink-wrapped
`position: relative` wrapper — they are sized to it, so nothing else may sit inside it or
every box shifts off the pixels it describes.
With SAM3 one selected chip is *active*: it owns the prompt field and any exemplars drawn on
the frame, so its chip is a pair of buttons — the name activates, the `×` deselects.
Exemplars are normalized `[cx, cy, w, h]`, sent only to `/preview`, and dropped whenever the
frame or the active class changes. A redraw is debounced 250 ms and re-runs the whole prompt
set from empty, which is also how undo works — SAM3 can only append geometric prompts.
**Exemplar-driven labeling in review (REQ-173, REQ-174, REQ-175).** In `draw` mode a drag on
`AnnotationCanvas` is an exemplar, not a rectangle: `onExemplar(box, positive)` where
`positive` is `!event.shiftKey`. `hooks/useExemplarPool.js` keeps the pool in a ref as well
as state — the ref is what gets sent, so a drag that lands mid-flight is never lost — and posts
the **whole pool** to `/exemplar-label` 400 ms after the last drag. One pass runs at a time;
a drag arriving during a pass sets `rerunWanted` so exactly one rerun follows instead of a
queue. The pool is cleared by a frame change or a class change, and `Undo example` re-runs
with the shortened pool.
The run is a **dry run by default** (REQ-175). `ExemplarFilterPanel.jsx` is passed to
`ReviewSidebar` and rendered above the class list — not over the frame, which is where its
proposals are drawn. It is mounted only while a run is undecided (`pool.active`): the first
drag opens it, Apply and Discard close it, and the pool outlives it. Its four sliders
re-preview on a 250 ms debounce. While a preview is up the canvas **hides the stored shapes
of the class under review** — the run replaces them wholesale, and leaving them on screen made
a rejected detection look like it had never gone; other classes stay, dimmed
(`svg.previewing`). Proposals draw dashed on top, green for the boxes the user drew and
class-colored for what SAM3 found.
`Apply` re-runs with `apply: true` rather than posting the previewed geometry back — SAM3 is
deterministic for a pool and a threshold, and the browser should not be the authority on what
gets stored. `Discard` truncates the pool to `appliedRef`, the length it had at the last
successful apply, so a rejected run leaves neither shapes nor prompts behind. A successful
apply drops the negatives it sent and keeps the positives (REQ-174); drags that landed while
that request was in flight are not part of it and stay at the end of the pool, with a rerun
queued for them.
`.canvas-wrap`'s overlay rules are scoped to its direct child (`> svg`): they set
`position: absolute; width: 100%`, which any nested SVG — an icon in a panel, say — would
otherwise inherit and stretch across the whole frame.
Request/response:
```
POST /api/frames/{id}/exemplar-label
{ "exemplars": [{"box": [x0,y0,x1,y1], "positive": true}, …], # normalized xyxy
"class_id": 0,
"threshold": 0.5, "iou_threshold": 0.8, # the panel (REQ-175)
"min_box_frac": 0.002, "max_detections": 100,
"apply": false }
→ { "shapes": [{"geometry": …, "score": …, "source": "manual"|"auto"}, …],
"applied": false, "redetected": true, "message": null,
"annotations": null } # the frame, on apply only
```
`backend/exemplar.py` converts each box to SAM3's normalized `[cx, cy, w, h]`, runs one
`open_state` + `apply_prompts` with the class's stored `prompt` as text plus every exemplar,
then rewrites the frame **for that class only**:
- positives are stored first as `source='manual'` — the literal rectangle in a `bbox`
project, the mask polygon of whatever SAM3 found inside it (IoU ≥ `SNAP_IOU`) in a
`polygon` one;
- detections overlapping a negative by ≥ `NEGATIVE_IOU` (0.3) are dropped, and ones
overlapping a positive by ≥ `DUPLICATE_IOU` (0.6) are dropped as the user's own shape
already covers them;
- what remains is written as `source='auto'`, so a later batch re-run replaces it (REQ-034)
while the drawn shapes survive.
The panel's filters run before any of that, in the order the batch job uses them: area floor,
then NMS (`labeling.deduplicate`), then the cap on how many survive.
The delete-and-reinsert happens in one `db.cursor()` transaction, so the frame is never
briefly empty. Other classes on the frame are never touched. The GPU lock is taken with a
**5 s** timeout — shorter than `assist`'s 20 s because this fires from a mouse gesture; on
timeout the response carries `redetected: false`, a message the panel shows, and the drawn
boxes alone as the preview. Applying that run **appends** the drawn boxes and honours the
negatives instead of taking the replace path — with no detections to put back, replacing
would wipe the class and leave only the drawings.
## Docker (REQ-072) ## Docker (REQ-072)
- `Dockerfile` — python 3.12, `ffmpeg`, `uv`, CUDA torch, `uv pip install -e sam3/`. The - `Dockerfile` — python 3.12, `ffmpeg`, `uv`, CUDA torch, `uv pip install -e sam3/`. The
+102
View File
@@ -100,6 +100,19 @@ changes.
found nothing on (REQ-033) writes no annotations, so a resume re-does it; that is accepted found nothing on (REQ-033) writes no annotations, so a resume re-does it; that is accepted
rather than tracked. rather than tracked.
- **REQ-171** — In the auto-annotate modal the SAM3 text prompt of each selected class is
editable in place, next to the live preview. Saving it writes
`project_classes.prompt` — the same field the Projects page edits — so the batch job and
every later run send that text. The preview is the tuning surface; the stored prompt is the
artifact it produces.
- **REQ-172** — On the previewed frame the user can drag **positive** and **negative** box
exemplars (shift-drag for negative). They are appended to the active class's text prompt,
re-run immediately, and can be undone or cleared. Exemplars are a **tuning aid only**: they
are never written as annotations and never carried into the batch job, because SAM3's
geometric prompts pool features from the current image — replaying them on another frame
would ask about whatever happens to sit at those coordinates there. They belong to exactly
one class, so a new frame or a new active class discards them.
## E. Review & correction ## E. Review & correction
- **REQ-040** — The user reviews frames one at a time, with fast navigation (left/right - **REQ-040** — The user reviews frames one at a time, with fast navigation (left/right
@@ -117,6 +130,53 @@ changes.
- **REQ-046** — The user can delete/clear all annotations of a specific class across all frames in - **REQ-046** — The user can delete/clear all annotations of a specific class across all frames in
the current batch from the Review editor. the current batch from the Review editor.
- **REQ-173** — In the review editor a plain drag on the canvas is an **exemplar-driven
label**, not just a rectangle. It proposes, in one action — and REQ-175's Apply is what
makes any of it real — that (a) the drawn shape becomes a `manual` annotation of the
active class — snapped to a SAM3 polygon first when the batch's
`label_type` is `polygon`, since a rectangle is a bad polygon label — (b) appends the box
to the frame's positive exemplar pool for that class, and (c) re-runs SAM3 over the whole
frame with the class's text prompt plus the pooled exemplars, deleting every existing shape
of that class on the frame and writing the detections in its place, then re-inserting the
pooled exemplar shapes verbatim so the user's own drawings always survive. The pool is
**frame-local and ephemeral** for the same reason as REQ-172 — SAM3's geometric prompts pool
features from the current image — so leaving the frame or switching the active class clears
it; the annotations it produced persist like any other. If the GPU lock (REQ-070) is not
free, the run comes back with the drawn shapes alone and says so, so the user can still
file them (REQ-175) and labeling is never blocked by a background job.
- **REQ-174** — **Shift**-drag in the review editor adds a **negative** exemplar. It is never
stored as an annotation; it deletes any existing shape of the active class that overlaps it,
and it is sent as a negative box in the REQ-173 re-detect. It is the "not this, and not
things like this" gesture, so it doubles as a delete. A negative is **spent on Apply**: the
frame it was applied to no longer carries what it rejected, so the drawing is dropped from
the pool while the positives stay on as prompts.
- **REQ-175** — An exemplar drag **previews**; it never writes on its own. The run's result
is drawn over the frame as proposals and a small panel floats on the canvas with the four
filters that decide what survives — confidence, NMS overlap, minimum box size, maximum
shapes — each re-running the preview as it moves. **Apply** writes the previewed set,
**Discard** rewinds the pool to whatever is already on the frame and leaves it untouched.
The panel is scoped to this gesture: its values are not stored, not shared with the
auto-annotate modal, and reset with the frame. Defaults are confidence `0.5`, NMS `0.8`,
min box `0.002`, max `100` — deliberately permissive, because on a dense frame an
aggressive NMS or area floor deletes real, touching objects rather than duplicates.
## E4. Live counting preview
- **REQ-176** — A **live** source on the Live Count page is a **WebRTC (WHEP) URL** and
nothing else; an RTSP URL is rejected with a message saying so. The backend derives the
RTSP leg of the same streaming-server path from it (`http://host:8889/cam` →
`rtsp://host:8554/cam`) and counts from that: WebRTC is what makes the browser preview
cheap, but pulling it into Python would add ICE and a jitter buffer on top of the identical
H.264 decode. One ingest on the streaming server, two consumers. The ports are read from
the environment (`MEDIAMTX_RTSP_PORT`, `MEDIAMTX_WHEP_PATH`), never hardcoded. Archive
files are unaffected — they are still opened as files.
- **REQ-177** — A live session is **watched over WebRTC**, played straight from the streaming
server by the browser: the frames never pass through this app and it encodes no JPEG for
them. What the model saw — boxes, ids, confidences, the counting line and its band, the
ignored region, the running totals — is served as geometry from
`GET /api/live-count/overlay` and drawn on a canvas over the video. The MJPEG endpoint
remains the preview for **archive files** only, and refuses a WebRTC session.
## F. Master dataset ## F. Master dataset
- **REQ-050** — Approving a batch **merges** its approved frames and their labels into the - **REQ-050** — Approving a batch **merges** its approved frames and their labels into the
@@ -164,6 +224,31 @@ changes.
and the reason it did or did not count, so a miss can be attributed to the model, the and the reason it did or did not count, so a miss can be attributed to the model, the
tracker, or the counter. tracker, or the counter.
- **REQ-145** — Counting algorithms are **pluggable**. Each registers under a stable id
(`line_cross`, `possession`) and the session constructs one by id. The `Counter` protocol
in `src/interfaces.py` is the contract, corrected to match reality: `update()` returns the
frame's count events, not `None`. Adding an algorithm must not require editing
`live_count.py` or `counting_bench.py`.
- **REQ-146** — Each algorithm **declares its own parameters** — name, type, default, range —
and an endpoint serves that declaration, mirroring `live-count/models`. The frontend renders
its controls from the declaration and hardcodes no per-algorithm parameter list. The start
request carries `algorithm` plus an opaque `params` object validated against the
declaration, replacing today's flat line-specific fields.
- **REQ-147** — Geometry is generalised from a line to a **named shape set**. `line_cross`
declares one horizontal segment; `possession` declares a bed polygon and an approach zone.
The editor's drag channel (`move_line`) becomes shape-agnostic, so any algorithm's geometry
is adjustable live without a new endpoint.
- **REQ-148** — The **possession counter**: every sack track carries an `owner_id`, the person
track it currently overlaps, or none when at rest. A count fires on an ownership change that
crosses the bed boundary — person-outside to bed, or person-outside to person-inside.
Ownership is sticky with hysteresis, so occlusion by the carrier's back and the unowned
mid-air phase of a thrown sack do not break it. This requires a `person` class alongside
`sack` from the detector.
- **REQ-149** — Every count run records **which algorithm and parameter set** produced it, and
accuracy is comparable per algorithm against the same ground truth. Switching algorithms
adds results, it never invalidates stored ones — so `count_runs` is keyed by
`(project, video, algorithm)`, not by video alone.
## F4. Counting accuracy bench ## F4. Counting accuracy bench
- **REQ-150** — A page lists every archive video as a row: date, batch, length, and the - **REQ-150** — A page lists every archive video as a row: date, batch, length, and the
@@ -179,6 +264,23 @@ changes.
which is what makes counting a 30-minute video practical. A run records the parameters and which is what makes counting a 30-minute video practical. A run records the parameters and
model it used. model it used.
- **REQ-154** — Ground truth can be **imported in bulk** from the operations sheet
(`./GT.xlsx`, `DATA MUAT PAKAN PER LINE`). The camera watches **Line 1**; Line 2 is
recorded for completeness but never scored. Each sheet is one working day; a row is one
truck with a `BAG` count, a `DUS` count and a plate.
- **REQ-155** — `BAG` (sacks) and `DUS` (boxes) are **separate commodities**, counted and
scored separately. A box already resting in the truck bed is a legitimate object of a
different class, not a detection fault.
- **REQ-156** — An import never silently guesses. Recordings are aligned to sheet rows by
start time against row order, the proposed pairing is **shown for human confirmation**
before anything is written, and each imported value records that it came from the sheet
rather than from a hand count. A recording that merged two trucks
(`BATCH_MERGE_THRESHOLD_SECONDS`) is flagged, not paired.
- **REQ-157** — Sheet values are **order quantities, not hand counts** — 67% of them are
exactly 160 or 180 — so they score aggregate accuracy across many trucks and never
adjudicate a single video. Per-event truth for algorithm comparison comes from a
hand-counted clip, held separately.
## F5. Real recording times and working days ## F5. Real recording times and working days
- **REQ-160** — Each recording's start time is read from the timestamp the camera burns into - **REQ-160** — Each recording's start time is read from the timestamp the camera burns into
+167
View File
@@ -1040,6 +1040,173 @@ two will disagree.
`2026-08-06/batch4`, `2026-08-06/batch9`, `2026-08-14/batch016` — likely truncated) and 11 were `2026-08-06/batch4`, `2026-08-06/batch9`, `2026-08-14/batch016` — likely truncated) and 11 were
read with low confidence. Both are flagged amber in the table and accept a hand-typed time. read with low confidence. Both are flagged amber in the table and accept a hand-typed time.
## Task — Ground truth import from the ops sheet (REQ-154…157)
1. Parse `docs/GT.xlsx` into rows → verify: 6 sheets (10–15 Aug 2026), Line 1 only, stopping
at the first blank plate so the inline totals row is not read as a truck. Expected Line 1
bag totals: 4780 / 4322 / 4365 / 5800 / 5645 / 9155. `[TODO]`
2. `ground_truth_bag` / `ground_truth_dus` + `gt_source` on `count_runs` (REQ-155, REQ-156) →
verify: migration runs on the live DB, existing hand-typed values survive as
`gt_source='manual'`. `[TODO]`
3. Alignment preview with human confirmation (REQ-156) → verify: a dry run on 14 Aug proposes
26 recordings against 32 Line-1 trucks, flags the shortfall, and writes nothing until
confirmed. `[TODO]`
4. Bench scores bag and box separately (REQ-155) → verify: the accuracy row shows both, and
totals only over rows that have a ground truth. `[TODO]`
## Task — Pluggable counting algorithms (REQ-145…149)
1. Fix the `Counter` protocol and register `line_cross` behind it (REQ-145) → verify: a live
session on a known clip returns **the same counts as before** the refactor — this step
changes no behaviour. `[TODO]`
2. Parameter declaration endpoint + generic frontend controls (REQ-146) → verify: the
live-count panel renders `line_cross`'s dials from the declaration alone, with no
algorithm-specific code in the page. `[TODO]`
3. Shape-agnostic geometry channel (REQ-147) → verify: dragging the line still works; a
two-shape stub algorithm is adjustable through the same endpoint. `[TODO]`
4. `count_runs` keyed by `(project, video, algorithm)` (REQ-149) → verify: the same video
counted by two algorithms yields two rows and two accuracy figures. `[TODO]`
5. The possession counter (REQ-148) → verify: on the hand-counted clip it beats `line_cross`
on sacks that are occluded by the carrier and on sacks thrown in by the sender. **Blocked**
until the detector emits a `person` class and one clip has per-event truth. `[TODO]`
## Task — Exemplar prompting in the auto-annotate modal (REQ-171, REQ-172) `[DONE]`
1. `Sam3Engine.detect_with_exemplars` — one `set_image`, prompts looped over it, boxes
appended to one prompt only → verify: a negative box owned by `sack` sitting on a truck
leaves the truck detections untouched, while the same box owned by `truck` suppresses
them. `[DONE]` — on frame 86031 of batch 594: text-only `{truck: 5}`, owned-by-sack
`{truck: 5}`, owned-by-truck `{}`. The `reset_all_prompts` before each prompt is what
stops the leak; `state["geometric_prompt"]` survives `set_text_prompt` otherwise.
2. `exemplars` + `exemplar_class_name` through `labeling.label_image` → `preview.py` →
`POST /api/batches/{id}/preview` → verify: an unknown class name falls back to plain text
rather than attaching the boxes to whichever class is first. `[DONE]` — 17 shapes for
both text-only and `exemplar_class_name: "nonexistent"`.
3. `preview_frame` moved out of `autolabel.py` into `preview.py` → verify: `autolabel.py` is
back under the 400-line limit and the job path still imports. `[DONE]` — 261 and 151
lines; container starts and registers the `autolabel` handler.
4. Editable class prompt in the modal, saved to `project_classes.prompt` (REQ-171) →
verify: a PATCH round-trips and the Projects page shows the new text. `[DONE]` — class 2
`box → cardboard box → box` via the existing `PATCH /api/projects/{id}`; no new endpoint.
5. `ExemplarCanvas.jsx` drag/shift-drag/undo/clear with 250 ms debounced re-run, and the
modal split into `PreviewShapes.jsx` + `ClassPromptPanel.jsx` to stay under 400 lines →
verify: `npm run build` clean, every file under the limit. `[DONE]` — 398 / 126 / 137 /
64 lines, build green, both containers redeployed.
**Deliberately not built:** exemplars in the batch job. SAM3's geometric prompts pool
features from the current image, so a box drawn on frame 1 asks about whatever sits at those
coordinates on frame 400. The batch job stays text-only; the exemplars exist to find the text
that works.
## Task — Exemplar-driven labeling in the review editor (REQ-173, REQ-174) `[DONE]`
1. `backend/exemplar.py` — pool → one SAM3 pass (class prompt + boxes) → rewrite that class
on that frame → verify: on frame 55446 (batch 426, `sack`), one positive drawn from an
existing box gives 52 class-0 shapes, exactly 1 of them `manual` with the drawn geometry,
and the frame's class-1 shapes are untouched. `[DONE]` — verified; warm pass 0.4 s, first
pass 7.6 s (model load).
2. Negative exemplars delete what they cover (REQ-174) → verify: shift-drag over one of the
detections and no `auto` shape overlapping it by ≥ 0.3 IoU comes back, while the drawn
positive survives. `[DONE]` — max IoU with the negative afterwards 0.078, manual shape
still present.
3. GPU-busy fallback → verify: hold `jobs.gpu_lock`, drag, and the drawn shape is still
stored with `redetected: false` and a legible message. `[DONE]` — "Saved your shape — the
GPU is busy with a background job…", 58 shapes vs 57 before, no exception.
4. `POST /api/frames/{id}/exemplar-label` + `AnnotationCanvas` drag/shift-drag with the pool
drawn as dashed ghosts, 400 ms debounce, undo/clear, and the busy message under the canvas
→ verify: `vite build` clean and every touched file under 400 lines. `[DONE]` — build
green; `exemplar.py` 211, `api/review.py` 141, canvas 287, `useExemplarPool.js` 81. The
pool logic went into that hook rather than into `ReviewPage.jsx`, which was already over
the limit before this task (620 lines) and ends it at 628.
## Task — Filter panel and preview for exemplar runs (REQ-175) `[DONE]`
1. `exemplar.label(..., apply=False)` — dry run by default, returning `shapes` instead of
writing → verify: two previews in a row leave the row count untouched. `[DONE]` — frame
55446 stayed at 57 rows across a default preview (52 shapes) and a filtered one (20).
2. The four filters, applied in the batch job's order (area floor → NMS → cap) → verify:
each one visibly bites on a dense frame. `[DONE]` — from 52 shapes: NMS 0.05 → 32,
min box 0.05 → 1, cap 5 → 5, confidence 0.9 → 15.
3. `apply: true` writes exactly what was previewed → verify: the applied frame matches the
preview count and leaves other classes alone. `[DONE]` — 20 previewed, 20 class-0 shapes
stored (1 of them the drawn `manual` box), the frame's 2 class-1 shapes untouched.
4. `ExemplarFilterPanel.jsx` floating in the canvas corner, sliders re-previewing on 250 ms,
Apply/Discard/Undo/Reset, Enter and Esc bound → verify: `vite build` clean, files under
the limit. `[DONE]` — panel 108, hook 125, canvas 314 lines; build green; both containers
rebuilt and the live endpoint returns `applied: false` for a drag.
5. The class under review hides while its preview is up → verify: a negative exemplar's
effect is visible instead of being masked by the stored box underneath it. `[DONE]` —
frame 55446: 51 detections with one positive, 50 with a negative added; before this the
removed box stayed on screen at 35% opacity and the run looked inert.
**Deliberately not built:** saving the filter values. They describe one frame's run, and the
auto-annotate modal already owns the batch-wide numbers — sharing them would let a tweak made
while reviewing one frame silently change what the next batch job does.
**Deliberately not built:** persisting the pool. It is a prompt about *this* image, so it
dies with the frame, exactly as in REQ-172. What persists is the annotations it produced.
## Task 32 — WebRTC preview for the live counting page (REQ-176, REQ-177) `[DONE]`
The live view cost far more than it should: the backend re-encoded every annotated frame to
JPEG and pushed it over MJPEG, on top of decoding the camera. The camera already reaches the
browser cheaply over WebRTC, so the frames stop travelling through this app entirely.
1. A live source must be a WHEP URL; the RTSP leg is derived → verify: **[DONE]**
`POST .../live-count/start` with `rtsp://192.168.192.96:8554/cam` →
`400 "A live source must be a WebRTC (WHEP) URL…"`; with
`http://192.168.192.96:8889/cam` → `200`, `source: "rtsp://192.168.192.96:8554/cam"`,
`whep_url: "http://192.168.192.96:8889/cam/whep"`, `preview: "webrtc"`.
2. The AI counts from that stream → verify: **[DONE]** 185 frames in 49 s off the live
camera, `error: ""`. That rate is the link's, not the model's — see below.
3. No JPEG is encoded for a WebRTC session → verify: **[DONE]** `GET /api/live-count/stream`
downloaded 0 bytes during a running WebRTC session, and now answers `409`.
4. The overlay feed carries what the model saw, and tracks the line live → verify:
**[DONE]** `GET /api/live-count/overlay` returned 27 boxes with ids and confidences;
after `PATCH /api/live-count/line {"line_y":300}` the feed reported `line.y: 300`.
5. The 400-line limit holds → verify: **[DONE]** `live_count.py` was already 467 lines, so
the transport layer went to `live_source.py` (120) and the MJPEG overlay to
`live_render.py` (60), leaving it at 393. On the frontend the preview moved to
`LiveVideoPanel.jsx` and the slider table to `liveCountFields.js`, leaving
`LiveCountPage.jsx` at 383. `npm run build` passes.
**Not verified here:** the WHEP handshake in a real browser. The endpoint was confirmed live
(`POST http://192.168.192.96:8889/cam/whep` answers, rejecting a deliberately malformed SDP
with `400`), but the negotiation itself needs a browser, not curl.
### Where the live FPS actually goes — measured, 2026-08-19
The live session runs at 4-6 fps and it is not the model. Measured in the backend container
against `rtsp://192.168.192.96:8554/cam`:
| Stage | Rate |
|---|---|
| ByteTrack + YOLO inference | **205 fps** |
| `cv2.resize` to 1280x720 | 5348 fps |
| Decode from RTSP | **6.4 fps** |
The camera is 704x576 HEVC at 350 kbit/s — nothing about it is expensive. The link is: the
route to the streaming server is a ZeroTier VPN measuring **15% packet loss** and a 41-104 ms
round trip. The comment in `live_source.py` claiming the cost was "decoding 1080p on the CPU"
was simply wrong and has been corrected; so has the hint on the page.
Transport was changed to UDP and changed back, because the measurement contradicts the
theory. Through the **ffmpeg CLI**, UDP wins as expected — 16 fps at 1.00x realtime against
TCP's 6.8 fps at 0.52x. Through **OpenCV** it loses: tcp 6.4 fps, udp+socket buffer 4.4, bare
udp 2.4, and a live session on UDP showed 18-second stalls waiting for a keyframe. OpenCV
drops what it cannot reassemble instead of showing it, so the loss lands as missing frames.
`RTSP_TRANSPORT` is left as an env override, defaulting to `tcp`.
**Not fixable in this repo.** Inference has ~50x the headroom the link delivers, so nothing
in the app is worth optimising. The lever is where the counter runs: next to MediaMTX it
would count at the camera's full rate. Worth checking whether the ZeroTier path is relayed
rather than direct (`zerotier-cli peers` — a `RELAY` row explains both the loss and the RTT).
**Deliberately not built:** an aiortc/WHEP client in the backend. It would be "WebRTC only"
end to end, but the decode cost is identical to RTSP and it adds ICE and keyframe-loss
failure modes to the counting path. The saving was always on the browser side.
## Known open points ## Known open points
- *Not closed by any task, by choice:* **any rebuild kills the running job.** Task 14's resume - *Not closed by any task, by choice:* **any rebuild kills the running job.** Task 14's resume
+1550
View File
File diff suppressed because it is too large. Load diff
+5
View File
@@ -117,6 +117,9 @@ export const api = {
body: { annotation_ids: annotationIds, class_id: classId }, body: { annotation_ids: annotationIds, class_id: classId },
}), }),
assist: (frameId, body) => request(`/frames/${frameId}/assist`, { method: 'POST', body }), assist: (frameId, body) => request(`/frames/${frameId}/assist`, { method: 'POST', body }),
// The whole frame-local exemplar pool, re-sent on every drag (REQ-173/174).
exemplarLabel: (frameId, body) =>
request(`/frames/${frameId}/exemplar-label`, { method: 'POST', body }),
setFrameStatus: (frameId, status) => setFrameStatus: (frameId, status) =>
request(`/frames/${frameId}/status`, { method: 'POST', body: { status } }), request(`/frames/${frameId}/status`, { method: 'POST', body: { status } }),
@@ -150,6 +153,8 @@ export const api = {
liveCountStop: () => request('/live-count/stop', { method: 'POST' }), liveCountStop: () => request('/live-count/stop', { method: 'POST' }),
liveCountMoveLine: (body) => request('/live-count/line', { method: 'PATCH', body }), liveCountMoveLine: (body) => request('/live-count/line', { method: 'PATCH', body }),
liveCountStatus: () => request('/live-count/status'), liveCountStatus: () => request('/live-count/status'),
// Polled far faster than status: this is what the WebRTC preview draws.
liveCountOverlay: () => request('/live-count/overlay'),
// `key` busts the browser cache so a restarted session gets a fresh connection. // `key` busts the browser cache so a restarted session gets a fresh connection.
liveCountStreamUrl: (key = 0) => `/api/live-count/stream?k=${key}`, liveCountStreamUrl: (key = 0) => `/api/live-count/stream?k=${key}`,
+63 -5
View File
@@ -437,7 +437,7 @@ main.page {
user-select: none; user-select: none;
} }
.canvas-wrap svg { .canvas-wrap > svg {
position: absolute; position: absolute;
top: 0; top: 0;
left: 0; left: 0;
@@ -447,7 +447,7 @@ main.page {
touch-action: none; touch-action: none;
} }
.canvas-wrap svg.assist { cursor: copy; } .canvas-wrap > svg.assist { cursor: copy; }
.canvas-wrap .shape rect, .canvas-wrap .shape rect,
.canvas-wrap .shape polygon { .canvas-wrap .shape polygon {
@@ -478,9 +478,9 @@ main.page {
/* Select mode: the cursor promises a marquee, and a shape is a target to tick /* Select mode: the cursor promises a marquee, and a shape is a target to tick
rather than something to drag (REQ-045a). */ rather than something to drag (REQ-045a). */
.canvas-wrap svg.selecting { cursor: cell; } .canvas-wrap > svg.selecting { cursor: cell; }
.canvas-wrap svg.selecting .shape rect, .canvas-wrap > svg.selecting .shape rect,
.canvas-wrap svg.selecting .shape polygon { cursor: pointer; } .canvas-wrap > svg.selecting .shape polygon { cursor: pointer; }
.canvas-wrap .shape.marked rect, .canvas-wrap .shape.marked rect,
.canvas-wrap .shape.marked polygon { .canvas-wrap .shape.marked polygon {
@@ -532,6 +532,22 @@ main.page {
.canvas-wrap .handle-ne, .canvas-wrap .handle-sw { cursor: nesw-resize; } .canvas-wrap .handle-ne, .canvas-wrap .handle-sw { cursor: nesw-resize; }
.canvas-wrap .handle-vertex, .canvas-wrap .handle-midpoint { cursor: pointer; } .canvas-wrap .handle-vertex, .canvas-wrap .handle-midpoint { cursor: pointer; }
/* The frame-local exemplar pool (REQ-173/174): a prompt, not a shape, so it is
drawn behind everything, dashed, and never a pointer target. */
.canvas-wrap .exemplar {
fill: none;
stroke-width: 1.5;
stroke-dasharray: 2 4;
opacity: 0.85;
pointer-events: none;
vector-effect: non-scaling-stroke;
}
.canvas-wrap .exemplar.negative {
fill: rgba(248, 113, 113, 0.1);
stroke-dasharray: 6 3;
}
.canvas-wrap .draft { .canvas-wrap .draft {
fill: rgba(255, 255, 255, 0.08); fill: rgba(255, 255, 255, 0.08);
stroke-width: 2; stroke-width: 2;
@@ -724,3 +740,45 @@ kbd {
.topbar .crumbs { display: none; } .topbar .crumbs { display: none; }
main.page { padding: 16px 12px 40px; } main.page { padding: 16px 12px 40px; }
} }
/* The exemplar filter popup (REQ-175). It sits at the top of the sidebar, not
over the frame: the proposals it governs are drawn on the canvas, and a card
floating on top of them hid the thing being judged. The accent border is what
marks it as a live decision rather than another standing panel. */
.exemplar-panel {
padding: 10px 12px;
border-radius: var(--radius);
border: 1px solid rgba(56, 189, 248, 0.45);
background: rgba(17, 24, 39, 0.92);
box-shadow: 0 4px 16px rgba(0, 0, 0, 0.35);
line-height: 1.45;
}
.exemplar-panel input[type="range"] {
accent-color: var(--accent);
display: block;
cursor: pointer;
}
.exemplar-panel .btn { cursor: pointer; transition: background 160ms ease, border-color 160ms ease; }
/* Preview shapes are proposals, not labels: dashed and unclickable. The class
under review is hidden while they show (see AnnotationCanvas); what stays
visible belongs to other classes, dimmed so the two never read as one set. */
.canvas-wrap > svg.previewing .shape { opacity: 0.4; }
.canvas-wrap .preview-shape {
fill: rgba(56, 189, 248, 0.1);
stroke-width: 2;
stroke-dasharray: 4 3;
pointer-events: none;
vector-effect: non-scaling-stroke;
}
.canvas-wrap .preview-shape.drawn {
fill: rgba(52, 211, 153, 0.16);
stroke-dasharray: none;
}
@media (prefers-reduced-motion: reduce) {
.exemplar-panel .btn { transition: none; }
}
+56 -4
View File
@@ -39,7 +39,8 @@ function overlaps(geometry, [mx0, my0, mx1, my1]) {
export default function AnnotationCanvas({ export default function AnnotationCanvas({
frame, imageUrl, annotations, selectedId, activeClass, assistMode, classes, frame, imageUrl, annotations, selectedId, activeClass, assistMode, classes,
mode = 'draw', selectedIds, onSelect, onCreate, onUpdate, onAssist, onMarquee, mode = 'draw', selectedIds, exemplars = [], preview = null,
onSelect, onCreate, onUpdate, onAssist, onExemplar, onMarquee,
}) { }) {
const wrapRef = useRef(null) const wrapRef = useRef(null)
const svgRef = useRef(null) const svgRef = useRef(null)
@@ -173,7 +174,10 @@ export default function AnnotationCanvas({
return return
} }
if (x1 - x0 >= MIN_SIZE && y1 - y0 >= MIN_SIZE) { if (x1 - x0 >= MIN_SIZE && y1 - y0 >= MIN_SIZE) {
// Draw mode is exemplar mode (REQ-173): the drag is both the label and
// the visual prompt, and Shift makes it a negative one (REQ-174).
if (assistMode) onAssist([x0, y0, x1, y1]) if (assistMode) onAssist([x0, y0, x1, y1])
else if (onExemplar) onExemplar([x0, y0, x1, y1], !additive.current)
else onCreate({ type: 'bbox', points: [x0, y0, x1, y1] }) else onCreate({ type: 'bbox', points: [x0, y0, x1, y1] })
} }
return return
@@ -209,12 +213,19 @@ export default function AnnotationCanvas({
ref={svgRef} ref={svgRef}
viewBox={`0 0 ${width} ${height}`} viewBox={`0 0 ${width} ${height}`}
preserveAspectRatio="none" preserveAspectRatio="none"
className={[assistMode ? 'assist' : '', selecting ? 'selecting' : ''].filter(Boolean).join(' ') || undefined} className={[assistMode ? 'assist' : '', selecting ? 'selecting' : '',
preview ? 'previewing' : ''].filter(Boolean).join(' ') || undefined}
onPointerDown={startDraw} onPointerDown={startDraw}
onPointerMove={onPointerMove} onPointerMove={onPointerMove}
onPointerUp={onPointerUp} onPointerUp={onPointerUp}
> >
{annotations.map((annotation) => ( {/* A preview replaces this class on the frame wholesale, so its stored
shapes step aside for the proposals — left on screen they read as
part of the result and a rejected box looks like it never went.
Other classes are untouched by the run and stay as they are. */}
{annotations
.filter((annotation) => !(preview && annotation.class_id === activeClass))
.map((annotation) => (
<Shape <Shape
key={annotation.id} key={annotation.id}
annotation={annotation} annotation={annotation}
@@ -234,6 +245,44 @@ export default function AnnotationCanvas({
/> />
))} ))}
{preview?.map((shape, i) => {
const drawn = shape.source === 'manual'
const stroke = drawn ? '#34d399' : classColor(activeClass)
if (shape.geometry.type === 'bbox') {
const [x0, y0, x1, y1] = shape.geometry.points
return (
<rect
key={`preview-${i}`}
className={`preview-shape${drawn ? ' drawn' : ''}`}
x={x0 * width} y={y0 * height}
width={(x1 - x0) * width} height={(y1 - y0) * height}
stroke={stroke}
/>
)
}
return (
<polygon
key={`preview-${i}`}
className={`preview-shape${drawn ? ' drawn' : ''}`}
points={shape.geometry.points.map(([x, y]) => `${x * width},${y * height}`).join(' ')}
stroke={stroke}
/>
)
})}
{!selecting && exemplars.map((item, i) => {
const [x0, y0, x1, y1] = item.box
return (
<rect
key={`exemplar-${i}`}
className={`exemplar${item.positive ? '' : ' negative'}`}
x={x0 * width} y={y0 * height}
width={(x1 - x0) * width} height={(y1 - y0) * height}
stroke={item.positive ? classColor(activeClass) : '#f87171'}
/>
)
})}
{draft && (() => { {draft && (() => {
const [x0, y0, x1, y1] = normalise(draft) const [x0, y0, x1, y1] = normalise(draft)
return ( return (
@@ -241,7 +290,10 @@ export default function AnnotationCanvas({
className={selecting ? 'draft marquee' : assistMode ? 'draft assist' : 'draft'} className={selecting ? 'draft marquee' : assistMode ? 'draft assist' : 'draft'}
x={x0 * width} y={y0 * height} x={x0 * width} y={y0 * height}
width={(x1 - x0) * width} height={(y1 - y0) * height} width={(x1 - x0) * width} height={(y1 - y0) * height}
stroke={selecting ? '#38bdf8' : assistMode ? 'var(--accent)' : classColor(activeClass)} stroke={selecting ? '#38bdf8'
: assistMode ? 'var(--accent)'
: additive.current ? '#f87171'
: classColor(activeClass)}
/> />
) )
})()} })()}
+165 -102
View File
@@ -1,5 +1,8 @@
import React, { useState, useEffect, useRef } from 'react' import React, { useState, useEffect, useRef } from 'react'
import { api } from '../api' import { api } from '../api'
import ExemplarCanvas from './ExemplarCanvas'
import { PreviewShapes } from './PreviewShapes'
import ClassPromptPanel from './ClassPromptPanel'
function useDebounce(value, delay) { function useDebounce(value, delay) {
const [debouncedValue, setDebouncedValue] = useState(value) const [debouncedValue, setDebouncedValue] = useState(value)
@@ -14,69 +17,6 @@ function useDebounce(value, delay) {
return debouncedValue return debouncedValue
} }
// Shared with MassAutoAnnotateModal: draws detection boxes/polygons over a
// frame in a 0..10000 viewBox.
export function PreviewShapes({ shapes, project }) {
return shapes.map((shape, i) => {
if (!shape.geometry || !shape.geometry.points) return null
const classObj = project.classes.find(c => c.class_id === shape.class_id)
const className = classObj?.name || 'Unknown'
const color = ['#38bdf8', '#34d399', '#f472b6', '#a78bfa', '#fbbf24'][shape.class_id % 5] || '#fff'
let minX = 1, minY = 1, maxX = 0, maxY = 0
if (shape.geometry.type === 'bbox') {
const [left, top, right, bottom] = shape.geometry.points
minX = left; minY = top; maxX = right; maxY = bottom;
} else {
shape.geometry.points.forEach(pt => {
if (pt[0] < minX) minX = pt[0]
if (pt[1] < minY) minY = pt[1]
if (pt[0] > maxX) maxX = pt[0]
if (pt[1] > maxY) maxY = pt[1]
})
}
const x0 = minX * 10000
const y0 = minY * 10000
const bw = (maxX - minX) * 10000
const bh = (maxY - minY) * 10000
return (
<g key={i}>
{shape.geometry.type === 'polygon' && (
<polygon
points={shape.geometry.points.map(pt => `${pt[0] * 10000},${pt[1] * 10000}`).join(' ')}
fill={color}
fillOpacity={0.35}
stroke={color}
strokeWidth="10"
/>
)}
<rect
x={x0}
y={y0}
width={bw}
height={bh}
fill="none"
stroke={color}
strokeWidth="20"
strokeDasharray="40 20"
/>
<text
x={x0}
y={y0 > 300 ? y0 - 100 : y0 + 300}
fill={color}
fontSize="240"
fontWeight="bold"
style={{ textShadow: '10px 10px 10px #000, -10px -10px 10px #000, 10px -10px 10px #000, -10px 10px 10px #000' }}
>
{className} {shape.score ? `${(shape.score * 100).toFixed(1)}%` : ''}
</text>
</g>
)
})
}
export default function AutoAnnotateModal({ export default function AutoAnnotateModal({
batch, batch,
project, project,
@@ -103,11 +43,30 @@ export default function AutoAnnotateModal({
return [] return []
}) })
// SAM3 prompt tuning (REQ-171): the class prompt is edited here, next to the
// live preview, and saved back onto the project so the batch job uses it.
const [classPrompts, setClassPrompts] = useState(() =>
Object.fromEntries(project.classes.map(c => [c.class_id, c.prompt || c.name]))
)
const [promptDraft, setPromptDraft] = useState('')
const [promptSaving, setPromptSaving] = useState(false)
const [promptError, setPromptError] = useState('')
// Exemplars (REQ-172): tuning aid for the previewed frame only. They belong to
// the active class and are never written or carried into the batch job.
const [activeClassName, setActiveClassName] = useState(null)
const [exemplars, setExemplars] = useState([])
// Bumped on every add/undo/clear. Debouncing the count rather than the array
// is what lets "clear all" re-run too — an empty array on its own is
// indistinguishable from the initial state.
const [exemplarRev, setExemplarRev] = useState(0)
// Preview state // Preview state
const [frames, setFrames] = useState([]) const [frames, setFrames] = useState([])
const [frameIndex, setFrameIndex] = useState(0) const [frameIndex, setFrameIndex] = useState(0)
const [previewShapes, setPreviewShapes] = useState([]) const [previewShapes, setPreviewShapes] = useState([])
const [isLoadingPreview, setIsLoadingPreview] = useState(false) const [isLoadingPreview, setIsLoadingPreview] = useState(false)
const [previewError, setPreviewError] = useState('')
const [isSubmitting, setIsSubmitting] = useState(false) const [isSubmitting, setIsSubmitting] = useState(false)
// Fetch frames on mount // Fetch frames on mount
@@ -130,6 +89,7 @@ export default function AutoAnnotateModal({
let isMounted = true let isMounted = true
setIsLoadingPreview(true) setIsLoadingPreview(true)
setPreviewError('')
api.preview(batch.id, { api.preview(batch.id, {
frame_id: frame.id, frame_id: frame.id,
engine, engine,
@@ -137,24 +97,93 @@ export default function AutoAnnotateModal({
iou_threshold: iouThreshold, iou_threshold: iouThreshold,
min_box_frac: minBoxFrac, min_box_frac: minBoxFrac,
target_class_names: selectedClasses, target_class_names: selectedClasses,
custom_model_path: customModelStagedPath custom_model_path: customModelStagedPath,
exemplars: engine === 'sam3' ? exemplars : [],
exemplar_class_name: engine === 'sam3' ? activeClassName : null
}).then(res => { }).then(res => {
if (isMounted && res.shapes) { if (isMounted && res.shapes) {
setPreviewShapes(res.shapes) setPreviewShapes(res.shapes)
} }
}).catch(err => { }).catch(err => {
console.error("Preview failed:", err) // The GPU lock (REQ-065) answers 409 here while a batch job holds it, and
// that is the one failure the user can act on — so it goes on screen.
setPreviewError(err.message || 'Preview failed')
setPreviewShapes([]) setPreviewShapes([])
}).finally(() => { }).finally(() => {
setIsLoadingPreview(false) setIsLoadingPreview(false)
}) })
} }
const activeClass = project.classes.find(c => c.name === activeClassName) || null
const sam3Tuning = engine === 'sam3'
// Clear shapes when frame changes // Clear shapes when frame changes
useEffect(() => { useEffect(() => {
setPreviewShapes([]) setPreviewShapes([])
}, [frameIndex]) }, [frameIndex])
// The active class owns the exemplars, so keep it pointing at something the
// user actually ticked.
useEffect(() => {
if (!sam3Tuning) return
if (!activeClassName || !selectedClasses.includes(activeClassName)) {
setActiveClassName(selectedClasses[0] || null)
}
}, [sam3Tuning, selectedClasses, activeClassName])
// Exemplars are pooled from one image and belong to one class, so both a new
// frame and a new class invalidate them (REQ-172).
useEffect(() => {
setExemplars([])
setExemplarRev(0)
}, [frameIndex, activeClassName])
useEffect(() => {
if (activeClass) {
setPromptDraft(classPrompts[activeClass.class_id] ?? activeClass.name)
setPromptError('')
}
}, [activeClassName])
// Auto re-run: drawing a box is the question, the redrawn preview is the
// answer, so waiting for a button press in between defeats the point.
// set_image is already cached for this frame, so only the grounding head runs.
const debouncedRev = useDebounce(exemplarRev, 250)
useEffect(() => {
if (debouncedRev > 0) handlePreview()
}, [debouncedRev])
const addExemplar = (exemplar) => {
setExemplars(prev => [...prev, exemplar])
setExemplarRev(rev => rev + 1)
}
const undoExemplar = () => {
setExemplars(prev => prev.slice(0, -1))
setExemplarRev(rev => rev + 1)
}
const clearExemplars = () => {
setExemplars([])
setExemplarRev(rev => rev + 1)
}
const savePrompt = async () => {
if (!activeClass) return
const text = promptDraft.trim()
if (!text || text === classPrompts[activeClass.class_id]) return
setPromptSaving(true)
setPromptError('')
try {
await api.patchProject(project.id, { prompts: { [activeClass.class_id]: text } })
setClassPrompts({ ...classPrompts, [activeClass.class_id]: text })
} catch (err) {
setPromptError(err.message || 'Could not save the prompt')
} finally {
setPromptSaving(false)
}
}
const handleStart = async () => { const handleStart = async () => {
setIsSubmitting(true) setIsSubmitting(true)
try { try {
@@ -195,31 +224,30 @@ export default function AutoAnnotateModal({
<div style={{ position: 'relative', width: '100%', background: '#000', borderRadius: 6, overflow: 'hidden', display: 'flex', justifyContent: 'center', alignItems: 'center' }}> <div style={{ position: 'relative', width: '100%', background: '#000', borderRadius: 6, overflow: 'hidden', display: 'flex', justifyContent: 'center', alignItems: 'center' }}>
{currentFrame ? ( {currentFrame ? (
<div className="relative inline-block" style={{ width: '100%', textAlign: 'center' }}> /* This wrapper must hug the image and nothing else: both overlays
* are sized to it, so anything else inside would shift every box
* off the pixels it describes. */
<div style={{ position: 'relative', display: 'inline-block', lineHeight: 0, maxWidth: '100%' }}>
<img <img
src={api.frameUrl(currentFrame.id)} src={api.frameUrl(currentFrame.id)}
alt="Preview Frame" alt="Preview Frame"
style={{ maxWidth: '100%', maxHeight: '42vh', objectFit: 'contain', display: 'block', margin: '0 auto' }} style={{ maxWidth: '100%', maxHeight: '42vh', display: 'block' }}
/> />
<svg viewBox="0 0 10000 10000" preserveAspectRatio="none" style={{ position: 'absolute', top: 0, left: 0, width: '100%', height: '100%', pointerEvents: 'none' }}> <svg viewBox="0 0 10000 10000" preserveAspectRatio="none" style={{ position: 'absolute', top: 0, left: 0, width: '100%', height: '100%', pointerEvents: 'none' }}>
<PreviewShapes shapes={previewShapes} project={project} /> <PreviewShapes shapes={previewShapes} project={project} />
</svg> </svg>
{sam3Tuning && activeClass && (
<ExemplarCanvas
exemplars={exemplars}
onAdd={addExemplar}
disabled={isLoadingPreview}
/>
)}
{isLoadingPreview && ( {isLoadingPreview && (
<div style={{ position: 'absolute', top: 8, right: 8, background: 'rgba(0,0,0,0.65)', padding: '4px 8px', borderRadius: 4, color: '#38bdf8', fontSize: '0.8rem', backdropFilter: 'blur(4px)' }}> <div style={{ position: 'absolute', top: 8, right: 8, background: 'rgba(0,0,0,0.65)', padding: '4px 8px', borderRadius: 4, color: '#38bdf8', fontSize: '0.8rem', backdropFilter: 'blur(4px)' }}>
Inferring... Inferring...
</div> </div>
)} )}
<div className="flex justify-between items-center" style={{ padding: '8px 12px' }}>
<button
type="button"
onClick={handlePreview}
disabled={isLoadingPreview || frames.length === 0}
className="px-4 py-1.5 bg-indigo-50 text-indigo-700 font-medium rounded-md hover:bg-indigo-100 disabled:opacity-50"
style={{ fontSize: '0.85rem' }}
>
Run Preview
</button>
</div>
</div> </div>
) : ( ) : (
<div style={{ display: 'flex', alignItems: 'center', justifyContent: 'center', minHeight: 200, color: '#71717a' }}> <div style={{ display: 'flex', alignItems: 'center', justifyContent: 'center', minHeight: 200, color: '#71717a' }}>
@@ -228,6 +256,50 @@ export default function AutoAnnotateModal({
)} )}
</div> </div>
{previewError && (
<div style={{ marginTop: 8, padding: '6px 10px', borderRadius: 4, fontSize: '0.8rem', background: 'rgba(244, 63, 94, 0.12)', border: '1px solid rgba(244, 63, 94, 0.4)', color: '#fda4af' }}>
{previewError}
</div>
)}
<div className="row" style={{ gap: 8, marginTop: 10, flexWrap: 'wrap' }}>
<button
type="button"
className="btn"
onClick={handlePreview}
disabled={isLoadingPreview || frames.length === 0}
style={{ fontSize: '0.82rem' }}
>
Run Preview
</button>
{sam3Tuning && activeClass && (
<>
<button
type="button"
className="btn btn-ghost"
onClick={undoExemplar}
disabled={exemplars.length === 0}
style={{ fontSize: '0.82rem' }}
>
Undo box
</button>
<button
type="button"
className="btn btn-ghost"
onClick={clearExemplars}
disabled={exemplars.length === 0}
style={{ fontSize: '0.82rem' }}
>
Clear boxes
</button>
<span className="hint" style={{ fontSize: '0.76rem', marginLeft: 'auto' }}>
Drag = example of <strong style={{ color: '#34d399' }}>{activeClass.name}</strong>
{' · '}Shift-drag = not this
</span>
</>
)}
</div>
<div style={{ marginTop: 10 }}> <div style={{ marginTop: 10 }}>
<div className="row" style={{ justifyContent: 'space-between', marginBottom: 4 }}> <div className="row" style={{ justifyContent: 'space-between', marginBottom: 4 }}>
<span className="hint" style={{ fontSize: '0.8rem' }}>Preview Frame ({frameIndex + 1} / {frames.length}):</span> <span className="hint" style={{ fontSize: '0.8rem' }}>Preview Frame ({frameIndex + 1} / {frames.length}):</span>
@@ -291,30 +363,21 @@ export default function AutoAnnotateModal({
/> />
</div> </div>
<div style={{ flex: 1, minHeight: 60, overflowY: 'auto' }}> <ClassPromptPanel
<span className="hint" style={{ fontSize: '0.8rem' }}>Target Classes:</span> classes={project.classes}
<div style={{ display: 'flex', flexWrap: 'wrap', gap: 5, marginTop: 6 }}> selectedClasses={selectedClasses}
{project.classes.map(c => { onSelectedChange={setSelectedClasses}
const isSelected = selectedClasses.includes(c.name) sam3Tuning={sam3Tuning}
return ( activeClassName={activeClassName}
<button onActivate={setActiveClassName}
key={c.name} activeClass={activeClass}
className={`class-chip ${isSelected ? 'active' : ''}`} promptDraft={promptDraft}
style={{ fontSize: '0.75rem', padding: '2px 8px', border: isSelected ? '1px solid #c084fc' : '1px solid #3f3f46', cursor: 'pointer' }} onPromptDraftChange={setPromptDraft}
onClick={() => { onPromptSave={savePrompt}
if (isSelected) { promptSaving={promptSaving}
setSelectedClasses(selectedClasses.filter(n => n !== c.name)) promptError={promptError}
} else { savedPrompt={activeClass ? classPrompts[activeClass.class_id] : ''}
setSelectedClasses([...selectedClasses, c.name]) />
}
}}
>
{isSelected ? '✓ ' : ''}{c.name}
</button>
)
})}
</div>
</div>
<div className="row" style={{ justifyContent: 'flex-end', gap: 10, marginTop: 16, paddingTop: 10, borderTop: '1px solid rgba(255,255,255,0.08)' }}> <div className="row" style={{ justifyContent: 'flex-end', gap: 10, marginTop: 16, paddingTop: 10, borderTop: '1px solid rgba(255,255,255,0.08)' }}>
<button className="btn btn-ghost" onClick={onClose} disabled={isSubmitting}>Cancel</button> <button className="btn btn-ghost" onClick={onClose} disabled={isSubmitting}>Cancel</button>
@@ -0,0 +1,137 @@
import React from 'react'
/* Target-class picker for the auto-annotate modal, plus the SAM3 prompt editor
* (REQ-171).
*
* The prompt lives on the project class, not on the job, so what is tuned here
* against the live preview is exactly what a batch run will send. Editing it is
* a real write to `project_classes.prompt` — the same field the Projects page
* edits, just reachable from where the feedback is.
*
* With SAM3 a chip carries a second meaning: one selected chip is *active*, and
* it owns the prompt shown below and any exemplars drawn on the frame. Because
* a click then has to mean "activate" rather than "deselect", each chip is a
* pair of buttons — the name activates, the × deselects.
*/
export default function ClassPromptPanel({
classes,
selectedClasses,
onSelectedChange,
sam3Tuning,
activeClassName,
onActivate,
activeClass,
promptDraft,
onPromptDraftChange,
onPromptSave,
promptSaving,
promptError,
savedPrompt,
}) {
const toggle = (name) => {
if (selectedClasses.includes(name)) {
onSelectedChange(selectedClasses.filter(n => n !== name))
} else {
onSelectedChange([...selectedClasses, name])
}
}
const dirty = activeClass && promptDraft.trim() && promptDraft.trim() !== savedPrompt
return (
<div style={{ flex: 1, minHeight: 60, overflowY: 'auto' }}>
<span className="hint" style={{ fontSize: '0.8rem' }}>Target Classes:</span>
<div style={{ display: 'flex', flexWrap: 'wrap', gap: 5, marginTop: 6 }}>
{classes.map(c => {
const isSelected = selectedClasses.includes(c.name)
const isActive = sam3Tuning && isSelected && c.name === activeClassName
if (!sam3Tuning) {
return (
<button
key={c.name}
className={`class-chip ${isSelected ? 'active' : ''}`}
style={{ fontSize: '0.75rem', padding: '2px 8px', border: isSelected ? '1px solid #c084fc' : '1px solid #3f3f46' }}
onClick={() => toggle(c.name)}
>
{isSelected ? '✓ ' : ''}{c.name}
</button>
)
}
return (
<span
key={c.name}
style={{
display: 'inline-flex', alignItems: 'stretch', borderRadius: 4,
overflow: 'hidden',
border: isActive ? '1px solid #34d399'
: isSelected ? '1px solid #c084fc' : '1px solid #3f3f46',
boxShadow: isActive ? '0 0 0 1px rgba(52, 211, 153, 0.45)' : 'none',
transition: 'border-color 180ms ease, box-shadow 180ms ease',
}}
>
<button
className={`class-chip ${isSelected ? 'active' : ''}`}
style={{ fontSize: '0.75rem', padding: '2px 8px', border: 'none', borderRadius: 0 }}
aria-pressed={isActive}
title={isSelected ? `Tune "${c.name}"` : `Include "${c.name}"`}
onClick={() => {
if (!isSelected) onSelectedChange([...selectedClasses, c.name])
onActivate(c.name)
}}
>
{isSelected ? '✓ ' : ''}{c.name}
</button>
{isSelected && (
<button
className="class-chip"
style={{ fontSize: '0.75rem', padding: '2px 6px', border: 'none', borderLeft: '1px solid #3f3f46', borderRadius: 0, color: '#a1a1aa' }}
title={`Exclude "${c.name}" from this run`}
aria-label={`Exclude ${c.name}`}
onClick={() => toggle(c.name)}
>
×
</button>
)}
</span>
)
})}
</div>
{sam3Tuning && activeClass && (
<div style={{ marginTop: 12 }}>
<label htmlFor="sam3-prompt" className="hint" style={{ fontSize: '0.8rem' }}>
SAM3 prompt for <strong style={{ color: '#34d399' }}>{activeClass.name}</strong>:
</label>
<div className="row" style={{ gap: 6, marginTop: 4 }}>
<input
id="sam3-prompt"
value={promptDraft}
onChange={(e) => onPromptDraftChange(e.target.value)}
onKeyDown={(e) => { if (e.key === 'Enter') { e.preventDefault(); onPromptSave() } }}
placeholder={activeClass.name}
style={{ fontSize: '0.82rem' }}
/>
<button
type="button"
className="btn"
onClick={onPromptSave}
disabled={!dirty || promptSaving}
style={{ fontSize: '0.8rem', whiteSpace: 'nowrap' }}
>
{promptSaving ? 'Saving…' : 'Save'}
</button>
</div>
<p className="hint" style={{ fontSize: '0.74rem', marginTop: 4 }}>
{promptError
? <span style={{ color: '#fda4af' }}>{promptError}</span>
: dirty
? 'Unsaved — the batch run still uses the stored prompt until you save.'
: 'Saved on the class. This is the text the batch run sends.'}
</p>
</div>
)}
</div>
)
}
+126
View File
@@ -0,0 +1,126 @@
import React, { useEffect, useRef, useState } from 'react'
/* Drag-to-draw box exemplars over the auto-annotate preview frame (REQ-172).
*
* Coordinates are normalized 0..1 against this overlay, which is sized to the
* image itself — so the numbers sent to SAM3 are resolution-independent and
* survive the preview being scaled to whatever the window allows.
*
* Boxes are stored in the format the backend wants: [cx, cy, w, h].
*/
const POSITIVE = '#34d399'
const NEGATIVE = '#f43f5e'
// Under this the drag was really a click. Zero-area boxes make SAM3's ROI pool
// return nothing useful, so they are dropped rather than sent.
const MIN_SIDE = 0.005
function toCorners(box) {
const [cx, cy, w, h] = box
return { x0: cx - w / 2, y0: cy - h / 2, w, h }
}
export default function ExemplarCanvas({ exemplars, onAdd, disabled = false }) {
const svgRef = useRef(null)
const [draft, setDraft] = useState(null) // { x0, y0, x1, y1, positive }
// Escape abandons the box being dragged. Without it a drag started by mistake
// can only be finished, and every finished box costs a GPU round trip.
useEffect(() => {
if (!draft) return undefined
const onKey = (event) => { if (event.key === 'Escape') setDraft(null) }
window.addEventListener('keydown', onKey)
return () => window.removeEventListener('keydown', onKey)
}, [draft])
const pointAt = (event) => {
const rect = svgRef.current.getBoundingClientRect()
return {
x: Math.min(1, Math.max(0, (event.clientX - rect.left) / rect.width)),
y: Math.min(1, Math.max(0, (event.clientY - rect.top) / rect.height)),
}
}
const handleDown = (event) => {
if (disabled || event.button !== 0) return
const { x, y } = pointAt(event)
event.currentTarget.setPointerCapture(event.pointerId)
// Shift marks the box as a counter-example: "not this one".
setDraft({ x0: x, y0: y, x1: x, y1: y, positive: !event.shiftKey })
}
const handleMove = (event) => {
if (!draft) return
const { x, y } = pointAt(event)
setDraft({ ...draft, x1: x, y1: y })
}
const handleUp = () => {
if (!draft) return
const w = Math.abs(draft.x1 - draft.x0)
const h = Math.abs(draft.y1 - draft.y0)
setDraft(null)
if (w < MIN_SIDE || h < MIN_SIDE) return
onAdd({
box: [(draft.x0 + draft.x1) / 2, (draft.y0 + draft.y1) / 2, w, h],
positive: draft.positive,
})
}
const draftRect = draft && {
x0: Math.min(draft.x0, draft.x1),
y0: Math.min(draft.y0, draft.y1),
w: Math.abs(draft.x1 - draft.x0),
h: Math.abs(draft.y1 - draft.y0),
}
return (
<svg
ref={svgRef}
viewBox="0 0 10000 10000"
preserveAspectRatio="none"
onPointerDown={handleDown}
onPointerMove={handleMove}
onPointerUp={handleUp}
onPointerCancel={() => setDraft(null)}
style={{
position: 'absolute', top: 0, left: 0, width: '100%', height: '100%',
cursor: disabled ? 'default' : 'crosshair',
touchAction: 'none',
}}
>
{exemplars.map((exemplar, index) => {
const { x0, y0, w, h } = toCorners(exemplar.box)
const color = exemplar.positive ? POSITIVE : NEGATIVE
return (
<g key={index}>
<rect
x={x0 * 10000} y={y0 * 10000} width={w * 10000} height={h * 10000}
fill={color} fillOpacity={0.12}
stroke={color} strokeWidth="30"
strokeDasharray={exemplar.positive ? undefined : '90 60'}
/>
<text
x={x0 * 10000 + 60} y={y0 * 10000 + 300}
fill={color} fontSize="260" fontWeight="bold"
style={{ textShadow: '0 0 40px #000, 0 0 40px #000' }}
>
{exemplar.positive ? '+' : '−'}{index + 1}
</text>
</g>
)
})}
{draftRect && (
<rect
x={draftRect.x0 * 10000} y={draftRect.y0 * 10000}
width={draftRect.w * 10000} height={draftRect.h * 10000}
fill="none"
stroke={draft.positive ? POSITIVE : NEGATIVE}
strokeWidth="30" strokeDasharray="60 40"
/>
)}
</svg>
)
}
@@ -0,0 +1,125 @@
import { useEffect, useRef } from 'react'
import { CheckIcon, XIcon } from './Icons'
/* The filter popup for an exemplar run (REQ-175).
*
* It lives at the top of the sidebar rather than over the frame — the shapes it
* governs are drawn on the canvas, and a card on top of them covered the thing
* being judged. It is mounted only while a run is undecided: the first drag
* opens it, Apply and Discard close it. Nothing here is stored — the run it
* governs is one frame's, and the next frame starts from the defaults again. */
const SLIDERS = [
{ key: 'threshold', label: 'Confidence', min: 0.05, max: 0.95, step: 0.05,
color: '#38bdf8', format: (v) => v.toFixed(2),
hint: 'SAM3 score cutoff. Lower finds more, and more junk.' },
{ key: 'iou_threshold', label: 'Overlap (NMS)', min: 0.1, max: 1, step: 0.05,
color: '#c084fc', format: (v) => v.toFixed(2),
hint: 'Two boxes overlapping this much are one object — the weaker goes. Lower deletes more.' },
{ key: 'min_box_frac', label: 'Min box size', min: 0, max: 0.05, step: 0.001,
color: '#34d399', format: (v) => `${(v * 100).toFixed(1)}% of frame`,
hint: 'Boxes smaller than this are specks. Higher deletes more.' },
{ key: 'max_detections', label: 'Max shapes', min: 5, max: 300, step: 5,
color: '#fbbf24', format: (v) => String(v),
hint: 'Keep only the highest-scoring N.' },
]
export default function ExemplarFilterPanel({
filters, preview, busy, className, negatives = 0, replacing = 0,
onFilter, onReset, onApply, onDiscard, onUndo,
}) {
const rootRef = useRef(null)
// The sidebar scrolls, so a run opened while it is scrolled down would ask
// for a decision the user cannot see.
useEffect(() => {
rootRef.current?.scrollIntoView({ block: 'nearest' })
}, [])
// Enter applies, Escape discards — the panel is modal in intent even though
// it never blocks the canvas underneath.
useEffect(() => {
function onKey(event) {
if (event.target?.matches?.('input, textarea, select')) {
if (event.key !== 'Enter' && event.key !== 'Escape') return
}
if (event.key === 'Enter') { event.preventDefault(); event.stopPropagation(); onApply() }
if (event.key === 'Escape') { event.preventDefault(); event.stopPropagation(); onDiscard() }
}
window.addEventListener('keydown', onKey, true)
return () => window.removeEventListener('keydown', onKey, true)
}, [onApply, onDiscard])
const drawn = preview?.shapes.filter((shape) => shape.source === 'manual').length ?? 0
const found = (preview?.shapes.length ?? 0) - drawn
return (
<div className="exemplar-panel" role="dialog" aria-label="Auto-label filters" ref={rootRef}>
<div className="row" style={{ justifyContent: 'space-between', marginBottom: 8 }}>
<strong style={{ fontSize: '0.82rem' }}>Auto-label “{className}”</strong>
<button type="button" className="btn" style={{ padding: '1px 6px', fontSize: '0.74rem' }}
onClick={onUndo} title="Drop the last example you drew">Undo</button>
</div>
<p className="muted" style={{ fontSize: '0.78rem', margin: '0 0 4px' }}>
{busy
? 'asking SAM3…'
: preview
? `${found} found + ${drawn} drawn`
+ (negatives ? ` · ${negatives} rejected` : '')
: 'Drag an example on the frame'}
</p>
{preview && !busy && (
<p className="muted" style={{ fontSize: '0.72rem', margin: '0 0 10px' }}>
Apply replaces the {replacing} “{className}” shape{replacing === 1 ? '' : 's'} on
this frame — nothing is saved yet.
</p>
)}
{preview?.message && (
<p className="error-banner" style={{ fontSize: '0.75rem', padding: '4px 8px', marginBottom: 10 }}>
{preview.message}
</p>
)}
{SLIDERS.map((slider) => (
<div key={slider.key} style={{ marginBottom: 10 }}>
<div className="row" style={{ justifyContent: 'space-between', marginBottom: 2 }}>
<label htmlFor={`exf-${slider.key}`} className="hint" style={{ fontSize: '0.76rem' }}>
{slider.label}
</label>
<strong style={{ color: slider.color, fontSize: '0.78rem' }}>
{slider.format(filters[slider.key])}
</strong>
</div>
<input
id={`exf-${slider.key}`}
type="range"
min={slider.min} max={slider.max} step={slider.step}
value={filters[slider.key]}
title={slider.hint}
onChange={(event) => onFilter(slider.key, parseFloat(event.target.value))}
style={{ width: '100%', cursor: 'pointer' }}
/>
</div>
))}
<div className="row" style={{ gap: 6, marginTop: 12, flexWrap: 'wrap' }}>
<button type="button" className="btn" style={{ padding: '2px 8px', fontSize: '0.76rem' }}
onClick={onReset} title="Back to the defaults">
Reset
</button>
<span className="spacer" style={{ flex: 1 }} />
<button type="button" className="btn" style={{ padding: '2px 8px', fontSize: '0.76rem' }}
onClick={onDiscard} title="Throw the run away, leave the frame as it is [Esc]">
<XIcon size={12} /> Discard
</button>
<button type="button" className="btn btn-primary" style={{ padding: '2px 8px', fontSize: '0.76rem' }}
onClick={onApply} disabled={busy || !preview}
title="Write these shapes to the frame [Enter]">
<CheckIcon size={12} /> Apply
</button>
</div>
</div>
)
}
+224
View File
@@ -0,0 +1,224 @@
import React, { useEffect, useRef, useState } from 'react'
import { api } from '../api'
/* The live-count preview.
*
* A WebRTC session shows the camera itself — the browser pulls it straight from
* the streaming server over WHEP, so the frames never pass through this app and
* the backend never encodes a JPEG for them (REQ-177). What the model saw is
* drawn on a canvas on top, from a small JSON feed. An archive file has no
* WebRTC leg, so it keeps the MJPEG the backend renders.
*
* Coordinates arrive in the 1280x720 space the pipeline works in; the canvas is
* sized to that, and CSS stretches it over the video. That is exact whatever the
* camera's real resolution is — the backend stretches each frame to 1280x720 per
* axis, and stretching the canvas back over the picture is the inverse of it — but
* only while the picture *fills* the element box. Do not give the video an
* `aspect-ratio` or an `object-fit` that letterboxes it: this camera is 704x576,
* so a forced 16:9 would pillarbox the picture and leave every box offset. */
const W = 1280
const H = 720
const OVERLAY_HZ = 10
const COLOURS = {
counted: '#4ade80',
tracked: '#f8bf71',
ignored: '#828282',
}
export default function LiveVideoPanel({ running, whepUrl, streamKey, placing, onPlace }) {
return (
<div className="panel" style={{ padding: 0, overflow: 'hidden', background: '#000', minHeight: 320 }}>
{!running ? (
<p className="empty" style={{ padding: 60, textAlign: 'center' }}>
Not running. Set a source and press Start.
</p>
) : whepUrl ? (
<WebrtcPreview whepUrl={whepUrl} placing={placing} onPlace={onPlace} />
) : (
<img
key={streamKey}
src={api.liveCountStreamUrl(streamKey)}
alt="Live counting"
onClick={onPlace}
title={`Click to place: ${placing.replace('line_', '').replace('_', ' ')}`}
style={{ width: '100%', display: 'block', cursor: 'crosshair' }}
/>
)}
</div>
)
}
function WebrtcPreview({ whepUrl, placing, onPlace }) {
const videoRef = useRef(null)
const canvasRef = useRef(null)
// A failed handshake used to be invisible: the canvas kept drawing boxes over
// a black rectangle, which looks exactly like a broken renderer rather than a
// stream that never arrived. The state is on screen now.
const [link, setLink] = useState('connecting')
useEffect(() => {
let pc = null
let resourceUrl = ''
let cancelled = false
async function connect() {
pc = new RTCPeerConnection()
pc.addTransceiver('video', { direction: 'recvonly' })
pc.ontrack = (event) => {
if (videoRef.current) videoRef.current.srcObject = event.streams[0]
}
pc.oniceconnectionstatechange = () => {
if (cancelled) return
if (pc.iceConnectionState === 'connected') setLink('live')
else if (['failed', 'disconnected', 'closed'].includes(pc.iceConnectionState)) {
setLink(`WebRTC ${pc.iceConnectionState} — is ${whepUrl} reachable from this browser?`)
}
}
await pc.setLocalDescription(await pc.createOffer())
// MediaMTX accepts a complete offer; gathering first avoids trickling
// candidates over a second request.
await new Promise((resolve) => {
if (pc.iceGatheringState === 'complete') return resolve()
pc.onicegatheringstatechange = () => pc.iceGatheringState === 'complete' && resolve()
setTimeout(resolve, 2000)
})
if (cancelled) return
const response = await fetch(whepUrl, {
method: 'POST',
headers: { 'Content-Type': 'application/sdp' },
body: pc.localDescription.sdp,
})
if (!response.ok) throw new Error(`WHEP ${response.status} from ${whepUrl}`)
resourceUrl = response.headers.get('Location') || ''
const answer = await response.text()
if (cancelled) return
await pc.setRemoteDescription({ type: 'answer', sdp: answer })
}
connect().catch((exc) => {
if (!cancelled) setLink(exc.message)
})
return () => {
cancelled = true
// Tell the server the viewer left, so it stops sending to a dead peer.
if (resourceUrl) {
fetch(new URL(resourceUrl, whepUrl).href, { method: 'DELETE' }).catch(() => {})
}
if (pc) pc.close()
}
}, [whepUrl])
// Poll the geometry and repaint. Independent of the video's frame rate: the
// boxes are a few hundred bytes, the video is never touched.
useEffect(() => {
let timer = null
let stopped = false
async function tick() {
try {
draw(canvasRef.current, await api.liveCountOverlay())
} catch {
/* a dropped poll is corrected by the next one */
}
if (!stopped) timer = setTimeout(tick, 1000 / OVERLAY_HZ)
}
tick()
return () => {
stopped = true
clearTimeout(timer)
}
}, [])
return (
<div
onClick={onPlace}
title={`Click to place: ${placing.replace('line_', '').replace('_', ' ')}`}
style={{ position: 'relative', width: '100%', cursor: 'crosshair', lineHeight: 0 }}
>
<video
ref={videoRef}
autoPlay
muted
playsInline
style={{ width: '100%', display: 'block', background: '#000' }}
/>
<canvas
ref={canvasRef}
width={W}
height={H}
style={{ position: 'absolute', inset: 0, width: '100%', height: '100%' }}
/>
{link !== 'live' && (
<p style={{
position: 'absolute', inset: 'auto 12px 12px 12px', margin: 0, lineHeight: 1.4,
fontSize: '0.76rem', color: link === 'connecting' ? '#a1a1aa' : '#f87171',
}}>
{link === 'connecting' ? `Connecting to ${whepUrl}…` : link}
</p>
)}
</div>
)
}
/* Mirrors what `backend/live_render.py` burns into the MJPEG, so the two
* previews say the same thing about the same session. */
function draw(canvas, data) {
if (!canvas) return
const ctx = canvas.getContext('2d')
ctx.clearRect(0, 0, W, H)
if (!data || !data.line) return
const { y, x_start: xs, x_end: xe, margin } = data.line
// Shade what the region excludes, so "ignored" never looks like "missed".
ctx.fillStyle = 'rgba(0, 0, 0, 0.55)'
if (xs > 0) ctx.fillRect(0, 0, xs, H)
if (xe < W) ctx.fillRect(xe, 0, W - xe, H)
ctx.lineWidth = 2
ctx.font = '16px ui-monospace, monospace'
for (const box of data.boxes || []) {
const [x1, y1, x2, y2] = box.b
ctx.strokeStyle = COLOURS[box.s] || COLOURS.tracked
ctx.lineWidth = box.s === 'ignored' ? 1 : 2
ctx.strokeRect(x1, y1, x2 - x1, y2 - y1)
if (box.s !== 'ignored') {
ctx.fillStyle = ctx.strokeStyle
ctx.fillText(`#${box.id} ${box.c}`, x1, Math.max(14, y1 - 5))
}
}
ctx.lineWidth = 2
ctx.strokeStyle = '#ff00ff'
for (const edge of [xs, xe]) {
if (edge > 0 && edge < W) {
ctx.beginPath()
ctx.moveTo(edge, 0)
ctx.lineTo(edge, H)
ctx.stroke()
}
}
ctx.strokeStyle = '#ffff00'
ctx.beginPath()
ctx.moveTo(xs, y)
ctx.lineTo(xe, y)
ctx.stroke()
ctx.strokeStyle = '#00a0a0'
ctx.lineWidth = 1
for (const edge of [y - margin, y + margin]) {
ctx.beginPath()
ctx.moveTo(xs, edge)
ctx.lineTo(xe, edge)
ctx.stroke()
}
const panel = `IN ${data.loading} OUT ${data.unloading} NET ${data.net}`
ctx.fillStyle = 'rgba(0, 0, 0, 0.85)'
ctx.fillRect(12, 12, 9 * panel.length + 30, 46)
ctx.fillStyle = COLOURS.counted
ctx.font = 'bold 24px ui-monospace, monospace'
ctx.fillText(panel, 24, 44)
}
@@ -1,6 +1,6 @@
import React, { useEffect, useState } from 'react' import React, { useEffect, useState } from 'react'
import { api } from '../api' import { api } from '../api'
import { PreviewShapes } from './AutoAnnotateModal' import { PreviewShapes } from './PreviewShapes'
const ENGINES = [ const ENGINES = [
{ id: 'sam3', label: 'SAM3' }, { id: 'sam3', label: 'SAM3' },
+64
View File
@@ -0,0 +1,64 @@
import React from 'react'
// Shared with MassAutoAnnotateModal: draws detection boxes/polygons over a
// frame in a 0..10000 viewBox.
export function PreviewShapes({ shapes, project }) {
return shapes.map((shape, i) => {
if (!shape.geometry || !shape.geometry.points) return null
const classObj = project.classes.find(c => c.class_id === shape.class_id)
const className = classObj?.name || 'Unknown'
const color = ['#38bdf8', '#34d399', '#f472b6', '#a78bfa', '#fbbf24'][shape.class_id % 5] || '#fff'
let minX = 1, minY = 1, maxX = 0, maxY = 0
if (shape.geometry.type === 'bbox') {
const [left, top, right, bottom] = shape.geometry.points
minX = left; minY = top; maxX = right; maxY = bottom;
} else {
shape.geometry.points.forEach(pt => {
if (pt[0] < minX) minX = pt[0]
if (pt[1] < minY) minY = pt[1]
if (pt[0] > maxX) maxX = pt[0]
if (pt[1] > maxY) maxY = pt[1]
})
}
const x0 = minX * 10000
const y0 = minY * 10000
const bw = (maxX - minX) * 10000
const bh = (maxY - minY) * 10000
return (
<g key={i}>
{shape.geometry.type === 'polygon' && (
<polygon
points={shape.geometry.points.map(pt => `${pt[0] * 10000},${pt[1] * 10000}`).join(' ')}
fill={color}
fillOpacity={0.35}
stroke={color}
strokeWidth="10"
/>
)}
<rect
x={x0}
y={y0}
width={bw}
height={bh}
fill="none"
stroke={color}
strokeWidth="20"
strokeDasharray="40 20"
/>
<text
x={x0}
y={y0 > 300 ? y0 - 100 : y0 + 300}
fill={color}
fontSize="240"
fontWeight="bold"
style={{ textShadow: '10px 10px 10px #000, -10px -10px 10px #000, 10px -10px 10px #000, -10px 10px 10px #000' }}
>
{className} {shape.score ? `${(shape.score * 100).toFixed(1)}%` : ''}
</text>
</g>
)
})
}
@@ -4,6 +4,7 @@ import { TrashIcon } from './Icons'
import ShortcutsPanel from './ShortcutsPanel' import ShortcutsPanel from './ShortcutsPanel'
export default function ReviewSidebar({ export default function ReviewSidebar({
exemplarPanel,
classesList, classesList,
activeClass, activeClass,
reclass, reclass,
@@ -18,6 +19,7 @@ export default function ReviewSidebar({
}) { }) {
return ( return (
<aside className="review-side stack"> <aside className="review-side stack">
{exemplarPanel}
<div className="panel side-panel"> <div className="panel side-panel">
<h2>Classes</h2> <h2>Classes</h2>
<div className="class-list"> <div className="class-list">
+9 -1
View File
@@ -5,7 +5,15 @@ export default function ShortcutsPanel() {
<dl className="shortcuts"> <dl className="shortcuts">
<div> <div>
<dt><kbd>Drag</kbd></dt> <dt><kbd>Drag</kbd></dt>
<dd>Add box / resize / move</dd> <dd>Example of this class → preview a re-detect</dd>
</div>
<div>
<dt><kbd>Enter</kbd> <kbd>Esc</kbd></dt>
<dd>Apply / discard the preview</dd>
</div>
<div>
<dt><kbd>Shift</kbd>+<kbd>Drag</kbd></dt>
<dd>Not this → drop it and re-detect</dd>
</div> </div>
<div> <div>
<dt><kbd>V</kbd></dt> <dt><kbd>V</kbd></dt>
@@ -0,0 +1,31 @@
/* The tuning dials on the live-count page, and what each one means.
* Data only — split out of `LiveCountPage.jsx` to keep it under 400 lines. */
// `live` fields can be moved during a session — placing a counting line means
// watching the stream while you move it, and a restart would throw the counts away.
export const FIELDS = [
{ key: 'line_y', label: 'Counting line Y', min: 0, max: 720, step: 1, live: true,
hint: 'Sacks are counted as they cross this line. Click the video to place it.' },
{ key: 'line_x_start', label: 'Line start X', min: 0, max: 1280, step: 1, live: true,
hint: 'Ignore anything left of this.' },
{ key: 'line_x_end', label: 'Line end X', min: 0, max: 1280, step: 1, live: true,
hint: 'Ignore anything right of this.' },
{ key: 'margin', label: 'Band margin (px)', min: 0, max: 120, step: 1,
hint: 'Dead band around the line, so jitter alone never counts.' },
{ key: 'entry_travel_min', label: 'Entry travel min (px)', min: 0, max: 200, step: 1,
hint: 'A track must move this far from where it first appeared before it can count. '
+ 'Raise it to kill ghost boxes that blink into existence next to the line.' },
{ key: 'handoff_radius', label: 'Hand-off radius (px)', min: 0, max: 300, step: 5,
hint: 'When a track dies, a new track born this close to where it was heading inherits '
+ 'its history — this is what stops an ID switch at the line losing the count. '
+ 'The most sensitive dial here: too large and unrelated sacks adopt each other. '
+ 'Calibrate against a clip with a known count.' },
{ key: 'unload_confirm_frames', label: 'Unload confirm (frames)', min: 1, max: 15, step: 1,
hint: 'Frames a sack must stay above the band before it counts as unloaded. Stops a '
+ 'worker repositioning a sack from cancelling a real count.' },
{ key: 'min_area_scale', label: 'Min area scale', min: 0, max: 2, step: 0.1,
hint: 'Perspective-aware size gate: boxes too small for their depth are fragments, not '
+ 'sacks. 0 turns it off.' },
{ key: 'conf', label: 'Confidence', min: 0.05, max: 0.95, step: 0.05,
hint: 'Detector threshold.' },
]
+153
View File
@@ -0,0 +1,153 @@
import { useCallback, useEffect, useRef, useState } from 'react'
import { api } from '../api'
/* The review editor's exemplar pool and its preview (REQ-173/174/175).
*
* A drag is both a label and a visual prompt. It never writes: the pool is
* posted as a dry run, the result comes back as preview shapes, the filter
* panel re-previews as its sliders move, and only Apply commits. The pool
* lives in a ref as well as in state — the ref is what gets sent, so a drag
* that lands while a pass is in flight is never lost — and it dies with the
* frame, because SAM3's geometric prompts pool features from *this* image. */
const DEBOUNCE_MS = 400
const SLIDER_DEBOUNCE_MS = 250
export const FILTER_DEFAULTS = {
threshold: 0.5,
iou_threshold: 0.8,
min_box_frac: 0.002,
max_detections: 100,
}
export default function useExemplarPool({ frameId, classId, onApplied, onError, onBusy }) {
const [exemplars, setExemplars] = useState([])
const [preview, setPreview] = useState(null) // { shapes, message, redetected }
// The panel is open from the first drag until the run is applied or thrown
// away — not for as long as the pool exists, which outlives it (REQ-175).
const [active, setActive] = useState(false)
const [filters, setFilters] = useState(FILTER_DEFAULTS)
const poolRef = useRef([])
const filterRef = useRef(FILTER_DEFAULTS)
// How much of the pool is already on the frame: Discard rewinds to here.
const appliedRef = useRef(0)
const timer = useRef(null)
const runRef = useRef(null)
const inFlight = useRef(false)
const rerunWanted = useRef(false)
// A pool only means anything on the image and for the class it was drawn
// for, so both changes discard it.
useEffect(() => {
clearTimeout(timer.current)
poolRef.current = []
appliedRef.current = 0
setExemplars([])
setPreview(null)
setActive(false)
}, [frameId, classId])
// One SAM3 pass per pool, not per drag: a burst of quick drags or slider
// moves coalesces into a single run, and a change arriving mid-pass queues
// exactly one rerun rather than stacking.
const run = useCallback(async ({ apply = false } = {}) => {
if (!frameId || !poolRef.current.length) return
if (inFlight.current) { rerunWanted.current = true; return }
inFlight.current = true
onBusy?.(true)
// What was actually sent: a drag landing mid-flight must not be counted as
// applied, and must not be lost either — it re-previews below.
const sent = poolRef.current
try {
const payload = await api.exemplarLabel(frameId, {
exemplars: sent, class_id: classId, apply, ...filterRef.current,
})
if (apply) {
// A negative is spent once it has been applied: the frame no longer
// carries what it rejected, so the drawing goes and only the positives
// stay behind as prompts. Drags that landed mid-flight are not part of
// this run and keep their place at the end of the pool.
const extra = poolRef.current.slice(sent.length)
const kept = sent.filter((item) => item.positive)
poolRef.current = [...kept, ...extra]
appliedRef.current = kept.length
setExemplars(poolRef.current)
setPreview(null)
setActive(false)
onApplied(payload.annotations)
} else {
setPreview({ shapes: payload.shapes, message: payload.message,
redetected: payload.redetected })
}
} catch (exc) {
onError(exc.message)
} finally {
inFlight.current = false
onBusy?.(false)
// A drag that arrived mid-pass re-opens the panel: its result was not in
// what came back.
if (rerunWanted.current) {
rerunWanted.current = false
setActive(true)
runRef.current?.()
}
}
}, [frameId, classId, onApplied, onError, onBusy])
runRef.current = run
const schedule = useCallback((delay) => {
clearTimeout(timer.current)
timer.current = setTimeout(() => runRef.current?.(), delay)
}, [])
const add = useCallback((box, positive) => {
if (!frameId) return
poolRef.current = [...poolRef.current, { box, positive }]
setExemplars(poolRef.current)
setActive(true)
// A drag during a pass cannot be in that pass's result, so it always earns
// a rerun — including during an apply, which would otherwise swallow it.
if (inFlight.current) rerunWanted.current = true
schedule(DEBOUNCE_MS)
}, [frameId, schedule])
const setFilter = useCallback((key, value) => {
filterRef.current = { ...filterRef.current, [key]: value }
setFilters(filterRef.current)
if (poolRef.current.length) schedule(SLIDER_DEBOUNCE_MS)
}, [schedule])
const resetFilters = useCallback(() => {
filterRef.current = FILTER_DEFAULTS
setFilters(FILTER_DEFAULTS)
if (poolRef.current.length) schedule(SLIDER_DEBOUNCE_MS)
}, [schedule])
const apply = useCallback(() => {
clearTimeout(timer.current)
runRef.current?.({ apply: true })
}, [])
// Discard rewinds the pool to what is already on the frame, so the drags
// since the last Apply are undone in one go and nothing was ever written.
const discard = useCallback(() => {
clearTimeout(timer.current)
poolRef.current = poolRef.current.slice(0, appliedRef.current)
setExemplars(poolRef.current)
setPreview(null)
setActive(false)
}, [])
const undo = useCallback(() => {
poolRef.current = poolRef.current.slice(0, -1)
appliedRef.current = Math.min(appliedRef.current, poolRef.current.length)
setExemplars(poolRef.current)
if (poolRef.current.length) schedule(SLIDER_DEBOUNCE_MS)
else { setPreview(null); setActive(false) }
}, [schedule])
useEffect(() => () => clearTimeout(timer.current), [])
return { exemplars, preview, active, filters,
add, setFilter, resetFilters, apply, discard, undo }
}
+24 -51
View File
@@ -1,52 +1,31 @@
import React, { useCallback, useEffect, useRef, useState } from 'react' import React, { useCallback, useEffect, useRef, useState } from 'react'
import { api } from '../api' import { api } from '../api'
import { AlertIcon } from '../components/Icons' import { AlertIcon } from '../components/Icons'
import LiveVideoPanel from '../components/LiveVideoPanel'
import { FIELDS } from '../components/liveCountFields'
/* Live counting test bench. /* Live counting test bench.
* *
* Point a trained model at an RTSP camera (or a local file) and watch it count. * Point a trained model at the WebRTC camera feed (or a local file) and watch it
* count. A live session is watched over WebRTC straight from the streaming
* server, so the preview costs this app nothing (REQ-176, REQ-177).
* It runs the same tracker, stabiliser and line-cross counter the production * It runs the same tracker, stabiliser and line-cross counter the production
* script uses, so a number here means the same thing there. What it leaves out * script uses, so a number here means the same thing there. What it leaves out
* is the batch lifecycle and its database — this answers "does the model count * is the batch lifecycle and its database — this answers "does the model count
* correctly", not "how many sacks today". */ * correctly", not "how many sacks today". */
// `live` fields can be moved during a session — placing a counting line means
// watching the stream while you move it, and a restart would throw the counts away.
const FIELDS = [
{ key: 'line_y', label: 'Counting line Y', min: 0, max: 720, step: 1, live: true,
hint: 'Sacks are counted as they cross this line. Click the video to place it.' },
{ key: 'line_x_start', label: 'Line start X', min: 0, max: 1280, step: 1, live: true,
hint: 'Ignore anything left of this.' },
{ key: 'line_x_end', label: 'Line end X', min: 0, max: 1280, step: 1, live: true,
hint: 'Ignore anything right of this.' },
{ key: 'margin', label: 'Band margin (px)', min: 0, max: 120, step: 1,
hint: 'Dead band around the line, so jitter alone never counts.' },
{ key: 'entry_travel_min', label: 'Entry travel min (px)', min: 0, max: 200, step: 1,
hint: 'A track must move this far from where it first appeared before it can count. '
+ 'Raise it to kill ghost boxes that blink into existence next to the line.' },
{ key: 'handoff_radius', label: 'Hand-off radius (px)', min: 0, max: 300, step: 5,
hint: 'When a track dies, a new track born this close to where it was heading inherits '
+ 'its history — this is what stops an ID switch at the line losing the count. '
+ 'The most sensitive dial here: too large and unrelated sacks adopt each other. '
+ 'Calibrate against a clip with a known count.' },
{ key: 'unload_confirm_frames', label: 'Unload confirm (frames)', min: 1, max: 15, step: 1,
hint: 'Frames a sack must stay above the band before it counts as unloaded. Stops a '
+ 'worker repositioning a sack from cancelling a real count.' },
{ key: 'min_area_scale', label: 'Min area scale', min: 0, max: 2, step: 0.1,
hint: 'Perspective-aware size gate: boxes too small for their depth are fragments, not '
+ 'sacks. 0 turns it off.' },
{ key: 'conf', label: 'Confidence', min: 0.05, max: 0.95, step: 0.05,
hint: 'Detector threshold.' },
]
export default function LiveCountPage({ projectId, onProject }) { export default function LiveCountPage({ projectId, onProject }) {
const [models, setModels] = useState([]) const [models, setModels] = useState([])
const [modelPath, setModelPath] = useState('') const [modelPath, setModelPath] = useState('')
// Archive video by default: a local file decodes at ~100 fps, an RTSP camera // Archive video by default: a local file decodes at ~100 fps, the camera at
// at ~6 because OpenCV decodes 1080p on the CPU. Testing counting accuracy is // whatever the link delivers. Tracking and inference run at 205 fps, so a live
// far quicker against a file. // session is bound by how fast frames arrive, never by the model — testing
// counting accuracy is far quicker against a file.
const [mode, setMode] = useState('file') const [mode, setMode] = useState('file')
const [source, setSource] = useState('rtsp://192.168.192.96:8554/cam') // The stream server is deployment-specific, so it is an env override with a
// sane default rather than a constant (REQ-176).
const [source, setSource] = useState(
import.meta.env.VITE_WHEP_URL || 'http://192.168.192.96:8889/cam')
const [dates, setDates] = useState([]) const [dates, setDates] = useState([])
const [date, setDate] = useState('') const [date, setDate] = useState('')
const [videos, setVideos] = useState([]) const [videos, setVideos] = useState([])
@@ -194,7 +173,7 @@ export default function LiveCountPage({ projectId, onProject }) {
<div> <div>
<label className="hint" style={{ fontSize: '0.8rem' }}>Source</label> <label className="hint" style={{ fontSize: '0.8rem' }}>Source</label>
<div style={{ display: 'flex', gap: 6, margin: '6px 0 8px' }}> <div style={{ display: 'flex', gap: 6, margin: '6px 0 8px' }}>
{[['file', 'Archive video'], ['stream', 'RTSP stream']].map(([id, label]) => ( {[['file', 'Archive video'], ['stream', 'WebRTC stream']].map(([id, label]) => (
<button <button
key={id} key={id}
type="button" type="button"
@@ -252,7 +231,10 @@ export default function LiveCountPage({ projectId, onProject }) {
style={{ width: '100%', fontSize: '0.8rem' }} style={{ width: '100%', fontSize: '0.8rem' }}
/> />
<p className="hint" style={{ fontSize: '0.72rem', margin: '4px 0 0' }}> <p className="hint" style={{ fontSize: '0.72rem', margin: '4px 0 0' }}>
Real time, but capped near 6 fps — OpenCV decodes this camera's 1080p on the CPU. WebRTC (WHEP) URL only, e.g. <span className="mono">http://host:8889/cam</span>.
The browser plays this feed directly; the counter reads the RTSP leg of the
same stream. Its FPS is the network's, not the model's — inference alone
runs at ~205 fps.
</p> </p>
</> </>
)} )}
@@ -365,22 +347,13 @@ export default function LiveCountPage({ projectId, onProject }) {
</div> </div>
)} )}
<div className="panel" style={{ padding: 0, overflow: 'hidden', background: '#000', minHeight: 320 }}> <LiveVideoPanel
{running ? ( running={running}
<img whepUrl={status.whep_url || ''}
key={streamKey} streamKey={streamKey}
src={api.liveCountStreamUrl(streamKey)} placing={placing}
alt="Live counting" onPlace={placeOnClick}
onClick={placeOnClick}
title={`Click to place: ${placing.replace('line_', '').replace('_', ' ')}`}
style={{ width: '100%', display: 'block', cursor: 'crosshair' }}
/> />
) : (
<p className="empty" style={{ padding: 60, textAlign: 'center' }}>
Not running. Set a source and press Start.
</p>
)}
</div>
{(status.events?.length ?? 0) > 0 && ( {(status.events?.length ?? 0) > 0 && (
<div className="panel" style={{ padding: 14 }}> <div className="panel" style={{ padding: 14 }}>
+42
View File
@@ -6,6 +6,8 @@ import Filmstrip from '../components/Filmstrip'
import { AlertIcon, CheckIcon, XIcon } from '../components/Icons' import { AlertIcon, CheckIcon, XIcon } from '../components/Icons'
import QuickReclassBar from '../components/QuickReclassBar' import QuickReclassBar from '../components/QuickReclassBar'
import ReviewSidebar from '../components/ReviewSidebar' import ReviewSidebar from '../components/ReviewSidebar'
import ExemplarFilterPanel from '../components/ExemplarFilterPanel'
import useExemplarPool from '../hooks/useExemplarPool'
export default function ReviewPage({ batchId: rawBatchId, projectId, onProject }) { export default function ReviewPage({ batchId: rawBatchId, projectId, onProject }) {
const [batch, setBatch] = useState(null) const [batch, setBatch] = useState(null)
@@ -112,6 +114,17 @@ export default function ReviewPage({ batchId: rawBatchId, projectId, onProject }
} catch (exc) { setError(exc.message) } } catch (exc) { setError(exc.message) }
} }
// The exemplar pool (REQ-173/174/175) owns the drag gesture in draw mode.
const onExemplarApplied = useCallback((rows) => {
setAnnotations(rows)
setSelectedId(null)
if (frame) patchFrameLocally(frame.id, { annotation_count: rows.length })
}, [frame])
const pool = useExemplarPool({
frameId: frame?.id, classId: activeClass,
onApplied: onExemplarApplied, onError: setError, onBusy: setBusy,
})
async function assist(box) { async function assist(box) {
setBusy(true); setError('') setBusy(true); setError('')
try { try {
@@ -359,6 +372,8 @@ export default function ReviewPage({ batchId: rawBatchId, projectId, onProject }
const reviewed = (batch.review?.approved ?? 0) + (batch.review?.rejected ?? 0) const reviewed = (batch.review?.approved ?? 0) + (batch.review?.rejected ?? 0)
const classesList = project.classes ?? [] const classesList = project.classes ?? []
const activeClassName =
classesList.find((item) => item.class_id === activeClass)?.name ?? 'this class'
async function approveAllFrames() { async function approveAllFrames() {
if (!window.confirm(`Mark all ${batch.review?.pending ?? 0} pending frames as approved?`)) return if (!window.confirm(`Mark all ${batch.review?.pending ?? 0} pending frames as approved?`)) return
@@ -432,10 +447,13 @@ export default function ReviewPage({ batchId: rawBatchId, projectId, onProject }
classes={classesList} classes={classesList}
mode={mode} mode={mode}
selectedIds={markedIds} selectedIds={markedIds}
exemplars={pool.exemplars}
preview={pool.preview?.shapes ?? null}
onSelect={setSelectedId} onSelect={setSelectedId}
onCreate={createShape} onCreate={createShape}
onUpdate={updateShape} onUpdate={updateShape}
onAssist={assist} onAssist={assist}
onExemplar={pool.add}
onMarquee={onMarquee} onMarquee={onMarquee}
/> />
)} )}
@@ -476,6 +494,15 @@ export default function ReviewPage({ batchId: rawBatchId, projectId, onProject }
</span> </span>
</> </>
)} )}
{mode === 'draw' && (
<span className="muted" style={{ fontSize: '0.78rem' }}>
{pool.exemplars.length
? `${pool.exemplars.filter((e) => e.positive).length} example(s), `
+ `${pool.exemplars.filter((e) => !e.positive).length} negative — `
+ 'tune the filters, then Apply'
: 'Drag an example of this class · Shift-drag = not this'}
</span>
)}
</div> </div>
{mode === 'select' && markedIds.length > 0 && ( {mode === 'select' && markedIds.length > 0 && (
@@ -556,6 +583,21 @@ export default function ReviewPage({ batchId: rawBatchId, projectId, onProject }
</div> </div>
<ReviewSidebar <ReviewSidebar
exemplarPanel={mode === 'draw' && pool.active && (
<ExemplarFilterPanel
filters={pool.filters}
preview={pool.preview}
busy={busy}
className={activeClassName}
negatives={pool.exemplars.filter((item) => !item.positive).length}
replacing={annotations.filter((row) => row.class_id === activeClass).length}
onFilter={pool.setFilter}
onReset={pool.resetFilters}
onApply={pool.apply}
onDiscard={pool.discard}
onUndo={pool.undo}
/>
)}
classesList={classesList} classesList={classesList}
activeClass={activeClass} activeClass={activeClass}
reclass={reclass} reclass={reclass}
Executable → Regular
View File
File mode changed.
Executable → Regular
View File
File mode changed.
Executable → Regular
View File
File mode changed.
Executable → Regular
View File
File mode changed.
Executable → Regular
View File
File mode changed.