feat: sync live-count, exemplar annotation modules, and update .gitignore
This commit is contained in:
1 parent
b6624eeff9
commit
ac95674c07
39 files changed
+3979
-444
No files matched your search
@@ -73,6 +73,11 @@ weights/
|
||||
*.sqlite
|
||||
*.sqlite3
|
||||
|
||||
# Asset exceptions
|
||||
!backend/assets/
|
||||
!backend/assets/**
|
||||
|
||||
|
||||
# Training Logs & Caches
|
||||
*.log
|
||||
*.tfevents*
|
||||
|
||||
+14
-3
@@ -36,6 +36,11 @@ class AutolabelRequest(BaseModel):
|
||||
append: bool = False
|
||||
custom_model_path: Optional[str] = None
|
||||
|
||||
class Exemplar(BaseModel):
|
||||
box: list[float] # [cx, cy, w, h], normalized 0..1
|
||||
positive: bool = True
|
||||
|
||||
|
||||
class PreviewRequest(BaseModel):
|
||||
frame_id: int
|
||||
engine: str
|
||||
@@ -44,6 +49,10 @@ class PreviewRequest(BaseModel):
|
||||
min_box_frac: float = 0.0
|
||||
target_class_names: Optional[list[str]] = None
|
||||
custom_model_path: Optional[str] = None
|
||||
# Drawn box exemplars (REQ-172): normalized cxcywh, positive or negative.
|
||||
# Preview-only — `autolabel.start` deliberately has no equivalent.
|
||||
exemplars: Optional[list[Exemplar]] = None
|
||||
exemplar_class_name: Optional[str] = None
|
||||
|
||||
|
||||
@router.post("/api/projects/{project_id}/batches")
|
||||
@@ -162,14 +171,14 @@ async def autolabel_with_model(
|
||||
|
||||
@router.post("/api/batches/{batch_id}/preview")
|
||||
def preview_autolabel(batch_id: int, request: PreviewRequest) -> dict:
|
||||
from backend import autolabel, jobs
|
||||
from backend import jobs, preview
|
||||
|
||||
if not jobs.gpu_lock.acquire(timeout=20):
|
||||
busy = jobs.running_types()
|
||||
kind = busy[0] if busy else "background"
|
||||
raise HTTPException(409, f"The GPU is busy with a {kind} job — wait for it to finish")
|
||||
try:
|
||||
shapes = autolabel.preview_frame(
|
||||
shapes = preview.preview_frame(
|
||||
batch_id=batch_id,
|
||||
frame_id=request.frame_id,
|
||||
engine=request.engine,
|
||||
@@ -177,7 +186,9 @@ def preview_autolabel(batch_id: int, request: PreviewRequest) -> dict:
|
||||
iou_threshold=request.iou_threshold,
|
||||
min_box_frac=request.min_box_frac,
|
||||
target_class_names=request.target_class_names,
|
||||
custom_model_path=request.custom_model_path
|
||||
custom_model_path=request.custom_model_path,
|
||||
exemplars=[e.model_dump() for e in request.exemplars or []],
|
||||
exemplar_class_name=request.exemplar_class_name,
|
||||
)
|
||||
return {"shapes": shapes}
|
||||
except Exception as exc:
|
||||
|
||||
@@ -15,9 +15,9 @@ router = APIRouter(tags=["live-count"])
|
||||
|
||||
|
||||
class StartRequest(BaseModel):
|
||||
# Either a raw source (RTSP URL or absolute path) or an archive-relative
|
||||
# path like "2026-08-13/batch001.mp4", which the backend resolves — the
|
||||
# frontend never needs to know where the archive is mounted.
|
||||
# Either a live stream — which must be a WebRTC (WHEP) URL, REQ-176 — or an
|
||||
# archive-relative path like "2026-08-13/batch001.mp4", which the backend
|
||||
# resolves so the frontend never needs to know where the archive is mounted.
|
||||
source: str = ""
|
||||
source_rel: Optional[str] = None
|
||||
model_path: Optional[str] = None
|
||||
@@ -60,14 +60,22 @@ def available_models(project_id: int) -> dict:
|
||||
def start(project_id: int, request: StartRequest) -> dict:
|
||||
project = project_or_404(project_id)
|
||||
|
||||
source = request.source
|
||||
source, whep = request.source, ""
|
||||
if request.source_rel:
|
||||
try:
|
||||
source = library.resolve(project["video_root"], request.source_rel)
|
||||
except library.LibraryError as exc:
|
||||
raise HTTPException(400, str(exc))
|
||||
elif source:
|
||||
# A live source is a WebRTC URL and nothing else. The browser watches it
|
||||
# over WebRTC; the counter decodes the RTSP leg of the same MediaMTX
|
||||
# path, derived here (REQ-176).
|
||||
try:
|
||||
whep, source = live_count.whep_url(source), live_count.whep_to_rtsp(source)
|
||||
except live_count.LiveCountError as exc:
|
||||
raise HTTPException(400, str(exc))
|
||||
if not source:
|
||||
raise HTTPException(400, "Pick a video or enter a stream URL")
|
||||
raise HTTPException(400, "Pick a video or enter a WebRTC stream URL")
|
||||
|
||||
path = request.model_path
|
||||
if not path and request.model_version_id is not None:
|
||||
@@ -88,6 +96,7 @@ def start(project_id: int, request: StartRequest) -> dict:
|
||||
unload_confirm_frames=request.unload_confirm_frames,
|
||||
min_area_scale=request.min_area_scale,
|
||||
spatial_dedup=request.spatial_dedup,
|
||||
whep=whep,
|
||||
)
|
||||
except live_count.LiveCountError as exc:
|
||||
raise HTTPException(400, str(exc))
|
||||
@@ -118,9 +127,24 @@ def status() -> dict:
|
||||
return live_count.status()
|
||||
|
||||
|
||||
@router.get("/api/live-count/overlay")
|
||||
def overlay() -> dict:
|
||||
"""Geometry only — boxes, line and counts for the frame just processed.
|
||||
|
||||
What the WebRTC preview draws over the video, instead of the server
|
||||
encoding a JPEG per frame for it (REQ-177).
|
||||
"""
|
||||
return live_count.overlay()
|
||||
|
||||
|
||||
@router.get("/api/live-count/stream")
|
||||
def stream():
|
||||
"""MJPEG of the annotated frames. Ends when the session does."""
|
||||
"""MJPEG of the annotated frames — the archive-file preview. Ends with the session."""
|
||||
if live_count.status().get("preview") == "webrtc":
|
||||
# Nothing is encoding JPEGs for this session; without this the generator
|
||||
# would sit on a worker thread for ten seconds producing nothing.
|
||||
raise HTTPException(409, "This session is watched over WebRTC, not MJPEG")
|
||||
|
||||
def frames():
|
||||
blank_streak = 0
|
||||
while True:
|
||||
|
||||
@@ -40,6 +40,25 @@ class AssistRequest(BaseModel):
|
||||
threshold: float = 0.5
|
||||
|
||||
|
||||
class PoolExemplar(BaseModel):
|
||||
# Normalized xyxy against the frame, as drawn on the review canvas.
|
||||
box: List[float]
|
||||
positive: bool = True
|
||||
|
||||
|
||||
class ExemplarLabelRequest(BaseModel):
|
||||
# The whole frame-local pool, newest last (REQ-173/174).
|
||||
exemplars: List[PoolExemplar]
|
||||
class_id: int = 0
|
||||
# The filter panel (REQ-175). Defaults mirror `exemplar.DEFAULTS`.
|
||||
threshold: float = 0.5
|
||||
iou_threshold: float = 0.8
|
||||
min_box_frac: float = 0.002
|
||||
max_detections: int = 100
|
||||
# Off by default: a drag previews, only Apply writes.
|
||||
apply: bool = False
|
||||
|
||||
|
||||
@router.get("/api/frames/{frame_id}/annotations")
|
||||
def list_annotations(frame_id: int) -> dict:
|
||||
target = review_store.frame(frame_id)
|
||||
@@ -99,6 +118,24 @@ def assist(frame_id: int, request: AssistRequest) -> dict:
|
||||
raise HTTPException(400, str(exc))
|
||||
|
||||
|
||||
@router.post("/api/frames/{frame_id}/exemplar-label")
|
||||
def exemplar_label(frame_id: int, request: ExemplarLabelRequest) -> dict:
|
||||
from backend import exemplar as exemplar_store
|
||||
|
||||
try:
|
||||
return exemplar_store.label(
|
||||
frame_id, request.class_id,
|
||||
[item.model_dump() for item in request.exemplars],
|
||||
threshold=request.threshold,
|
||||
iou_threshold=request.iou_threshold,
|
||||
min_box_frac=request.min_box_frac,
|
||||
max_detections=request.max_detections,
|
||||
apply=request.apply,
|
||||
)
|
||||
except review_store.ReviewError as exc:
|
||||
raise HTTPException(400, str(exc))
|
||||
|
||||
|
||||
@router.post("/api/frames/{frame_id}/status")
|
||||
def set_status(frame_id: int, request: StatusRequest) -> dict:
|
||||
try:
|
||||
|
||||
Binary file not shown.
@@ -259,125 +259,3 @@ def _reset_reviewed(batch_id: int) -> None:
|
||||
"AND review_status = 'approved'",
|
||||
(batch_id,),
|
||||
)
|
||||
|
||||
|
||||
def preview_frame(
|
||||
batch_id: int,
|
||||
frame_id: int,
|
||||
engine: str,
|
||||
threshold: float = DEFAULT_THRESHOLD,
|
||||
iou_threshold: float = DEFAULT_IOU,
|
||||
min_box_frac: float = 0.0,
|
||||
target_class_names: Optional[List[str]] = None,
|
||||
custom_model_path: Optional[str] = None
|
||||
) -> List[dict]:
|
||||
batch = batches.get(batch_id)
|
||||
if not batch:
|
||||
raise ValueError("No such batch")
|
||||
project = projects.get(batch["project_id"])
|
||||
|
||||
frame = next((f for f in batches.frames(batch_id) if f["id"] == frame_id), None)
|
||||
if not frame:
|
||||
raise ValueError("Frame not found")
|
||||
|
||||
directory = batches.frames_dir(batch["project_slug"], batch_id)
|
||||
frame_file = os.path.join(directory, frame["filename"])
|
||||
fw = max(1, frame.get("width") or 1)
|
||||
fh = max(1, frame.get("height") or 1)
|
||||
|
||||
yolo_model = None
|
||||
sam3_target_classes = []
|
||||
|
||||
if engine == "sam3" and not custom_model_path:
|
||||
allowed_classes_set = {c.strip().lower() for c in target_class_names} if target_class_names else None
|
||||
if allowed_classes_set:
|
||||
sam3_target_classes = [c for c in project["classes"] if c["name"].strip().lower() in allowed_classes_set or c["prompt"].strip().lower() in allowed_classes_set]
|
||||
else:
|
||||
sam3_target_classes = [c for c in project["classes"]]
|
||||
|
||||
prompts = [c["prompt"] for c in sam3_target_classes]
|
||||
if prompts:
|
||||
from backend.sam3_engine import get_engine
|
||||
get_engine()
|
||||
else:
|
||||
from ultralytics import YOLO
|
||||
if custom_model_path and os.path.isfile(custom_model_path):
|
||||
m_path = custom_model_path
|
||||
else:
|
||||
m_path = projects.training_start_point(project)
|
||||
with db.cursor() as cur:
|
||||
cur.execute("SELECT weights_path FROM model_versions WHERE project_id = ? ORDER BY version DESC LIMIT 1", (project["id"],))
|
||||
row = cur.fetchone()
|
||||
if row and os.path.isfile(row[0]):
|
||||
m_path = row[0]
|
||||
yolo_model = YOLO(m_path)
|
||||
|
||||
name_to_class_id = {item["name"].strip().lower(): item["class_id"] for item in project["classes"]}
|
||||
allowed_classes_set = {c.strip().lower() for c in target_class_names} if target_class_names else None
|
||||
|
||||
all_raw_detections = []
|
||||
|
||||
if yolo_model is not None:
|
||||
results = yolo_model.predict(frame_file, conf=threshold, verbose=False)
|
||||
if results and len(results) > 0:
|
||||
model_names = results[0].names
|
||||
for box in results[0].boxes:
|
||||
cls_idx = int(box.cls[0].item())
|
||||
raw_cls_name = str(model_names.get(cls_idx, cls_idx)).strip().lower()
|
||||
|
||||
target_class_id = name_to_class_id.get(raw_cls_name)
|
||||
if target_class_id is None:
|
||||
for item in project["classes"]:
|
||||
if item["class_id"] == cls_idx:
|
||||
target_class_id = item["class_id"]
|
||||
break
|
||||
if target_class_id is None and 0 <= cls_idx < len(project["classes"]):
|
||||
target_class_id = project["classes"][cls_idx]["class_id"]
|
||||
|
||||
if target_class_id is None:
|
||||
continue
|
||||
|
||||
target_cls_obj = next((c for c in project["classes"] if c["class_id"] == target_class_id), None)
|
||||
proj_cls_name = target_cls_obj["name"].strip().lower() if target_cls_obj else ""
|
||||
|
||||
if allowed_classes_set is not None:
|
||||
if (raw_cls_name not in allowed_classes_set and
|
||||
proj_cls_name not in allowed_classes_set and
|
||||
str(target_class_id) not in allowed_classes_set):
|
||||
continue
|
||||
|
||||
score = float(box.conf[0].item())
|
||||
xyxyn = box.xyxyn[0].tolist()
|
||||
all_raw_detections.append(labeling.Detection(
|
||||
class_id=target_class_id,
|
||||
class_name=proj_cls_name or raw_cls_name,
|
||||
box=[xyxyn[0]*fw, xyxyn[1]*fh, xyxyn[2]*fw, xyxyn[3]*fh],
|
||||
score=score,
|
||||
mask=None
|
||||
))
|
||||
|
||||
if engine == "sam3" and sam3_target_classes:
|
||||
prompts = [(c.get("prompt") or c["name"]).strip() for c in sam3_target_classes]
|
||||
res = labeling.label_image(
|
||||
frame_file, frame["filename"], prompts, threshold,
|
||||
iou_threshold=iou_threshold, min_box_frac=min_box_frac
|
||||
)
|
||||
if not res.error and res.detections:
|
||||
for det in res.detections:
|
||||
if 0 <= det.class_id < len(sam3_target_classes):
|
||||
real_cls = sam3_target_classes[det.class_id]
|
||||
det.class_id = real_cls["class_id"]
|
||||
det.class_name = real_cls["name"]
|
||||
all_raw_detections.append(det)
|
||||
|
||||
kept = labeling.deduplicate(all_raw_detections, iou_threshold=iou_threshold)
|
||||
items = []
|
||||
for det in kept:
|
||||
if project["label_type"] == "bbox" or det.mask is None:
|
||||
geom = review.bbox(det.box[0]/fw, det.box[1]/fh, det.box[2]/fw, det.box[3]/fh)
|
||||
items.append({"class_id": det.class_id, "geometry": geom, "score": det.score})
|
||||
else:
|
||||
for geometry in _geometries(det, fw, fh, project["label_type"]):
|
||||
items.append({"class_id": det.class_id, "geometry": geometry, "score": det.score})
|
||||
|
||||
return items
|
||||
@@ -0,0 +1,270 @@
|
||||
"""Exemplar-driven manual labeling in the review editor (REQ-173/174/175).
|
||||
|
||||
A drag on the review canvas is not just a rectangle: it is a visual prompt.
|
||||
The drawn box joins a frame-local pool, the pool is replayed against SAM3
|
||||
together with the class's text prompt, and the whole class is re-detected on
|
||||
that frame from the result.
|
||||
|
||||
The pool lives in the editor, not in the database, and is sent whole on every
|
||||
call. That keeps this module stateless and matches REQ-172's reasoning: SAM3's
|
||||
geometric prompts pool features from *this* image, so a pool only means
|
||||
anything for as long as the user is looking at the frame it was drawn on.
|
||||
|
||||
Split out of `review.py` because that file is already at the 400-line limit.
|
||||
"""
|
||||
|
||||
import json
|
||||
import time
|
||||
from typing import List, Optional
|
||||
|
||||
from backend import db, projects, review
|
||||
|
||||
# A detection this close to a box the user drew is the same object: the user's
|
||||
# own shape wins, so the detection is dropped rather than stacked on top of it.
|
||||
DUPLICATE_IOU = 0.6
|
||||
# A detection overlapping a negative box by this much is what the user pointed
|
||||
# at when they said "not this" (REQ-174). Lower than DUPLICATE_IOU because a
|
||||
# negative is drawn roughly, around something the user wants gone.
|
||||
NEGATIVE_IOU = 0.3
|
||||
# Below this, no detection is really "inside" a drawn box, so a polygon project
|
||||
# keeps the rectangle rather than snapping to an unrelated mask.
|
||||
SNAP_IOU = 0.1
|
||||
|
||||
# What the filter panel opens with (REQ-175). Measured on a dense `sack` frame:
|
||||
# NMS at 0.8 only removes near-duplicates and a 0.002 area floor only removes
|
||||
# specks, where the aggressive-looking values delete real, touching objects.
|
||||
DEFAULTS = {
|
||||
"threshold": 0.5,
|
||||
"iou_threshold": 0.8,
|
||||
"min_box_frac": 0.002,
|
||||
"max_detections": 100,
|
||||
}
|
||||
|
||||
|
||||
def _class_prompt(project_id: int, class_id: int) -> str:
|
||||
project = projects.get(project_id)
|
||||
for item in project["classes"]:
|
||||
if item["class_id"] == class_id:
|
||||
return (item.get("prompt") or item["name"]).strip()
|
||||
raise review.ReviewError(f"Class {class_id} does not exist in this project")
|
||||
|
||||
|
||||
def _rect(points: List[float], label_type: str) -> dict:
|
||||
"""The drawn rectangle as a storable shape for this project."""
|
||||
x0, y0, x1, y1 = points
|
||||
if label_type == "bbox":
|
||||
return review.bbox(x0, y0, x1, y1)
|
||||
return review.polygon([(x0, y0), (x1, y0), (x1, y1), (x0, y1)])
|
||||
|
||||
|
||||
def _cxcywh(points: List[float]) -> List[float]:
|
||||
x0, y0, x1, y1 = points
|
||||
return [(x0 + x1) / 2, (y0 + y1) / 2, x1 - x0, y1 - y0]
|
||||
|
||||
|
||||
def _replace_class(frame_id: int, class_id: int, items: List[dict]) -> None:
|
||||
"""Swap every shape of one class on one frame for a fresh set.
|
||||
|
||||
"Replace everything, re-add drawn": the user's exemplar shapes are part of
|
||||
`items`, so they come back verbatim in the same transaction.
|
||||
"""
|
||||
now = time.time()
|
||||
with db.cursor() as cur:
|
||||
cur.execute("DELETE FROM annotations WHERE frame_id = ? AND class_id = ?",
|
||||
(frame_id, class_id))
|
||||
cur.executemany(
|
||||
"""INSERT INTO annotations (frame_id, class_id, geometry, score, source, created_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?)""",
|
||||
[(frame_id, class_id, json.dumps(item["geometry"]), item.get("score", 1.0),
|
||||
item.get("source", "auto"), now) for item in items],
|
||||
)
|
||||
|
||||
|
||||
def _append_drawn(frame_id: int, class_id: int, drawn: List[dict]) -> int:
|
||||
"""Add drawn shapes the frame does not already carry, leaving the rest alone.
|
||||
|
||||
The pool is re-sent whole on every call, so most of it is usually already
|
||||
stored; only what is genuinely new gets inserted.
|
||||
"""
|
||||
from backend.labeling import _iou
|
||||
|
||||
existing = [review.to_box(row["geometry"]) for row in review.listing(frame_id)
|
||||
if row["class_id"] == class_id]
|
||||
fresh = [item for item in drawn
|
||||
if not any(_iou(review.to_box(item["geometry"]), box) >= 0.9
|
||||
for box in existing)]
|
||||
for item in fresh:
|
||||
review.add(frame_id, class_id, item["geometry"], source="manual")
|
||||
return len(fresh)
|
||||
|
||||
|
||||
def _drop_negative_overlaps(frame_id: int, class_id: int,
|
||||
negatives: List[List[float]]) -> int:
|
||||
"""Delete shapes of this class the user shift-dragged over (REQ-174).
|
||||
|
||||
Used on the path where SAM3 never runs; the re-detect path filters the
|
||||
detections instead, which has the same effect on what ends up stored.
|
||||
"""
|
||||
from backend.labeling import _iou
|
||||
|
||||
doomed = [row["id"] for row in review.listing(frame_id)
|
||||
if row["class_id"] == class_id
|
||||
and any(_iou(review.to_box(row["geometry"]), box) >= NEGATIVE_IOU
|
||||
for box in negatives)]
|
||||
return review.delete_many(doomed)
|
||||
|
||||
|
||||
def label(frame_id: int, class_id: int, exemplars: List[dict],
|
||||
threshold: float = 0.5, iou_threshold: float = 0.8,
|
||||
min_box_frac: float = 0.002, max_detections: int = 100,
|
||||
apply: bool = False) -> dict:
|
||||
"""Detect one class on one frame from the frame's exemplar pool.
|
||||
|
||||
`exemplars` is the whole pool, newest last, each
|
||||
`{"box": [x0, y0, x1, y1], "positive": bool}` normalized to the frame.
|
||||
Positive boxes are both prompts and labels; negative boxes are prompts and
|
||||
deletions, never labels.
|
||||
|
||||
Nothing is written unless `apply` is set (REQ-175): a drag previews, the
|
||||
filter panel re-previews, and only Apply touches the frame. Apply re-runs
|
||||
rather than trusting shapes sent back from the browser — SAM3 is
|
||||
deterministic for a given pool and threshold, so the second pass reproduces
|
||||
what was previewed.
|
||||
"""
|
||||
from backend import batches, jobs
|
||||
from backend.labeling import _iou, deduplicate
|
||||
from PIL import Image
|
||||
|
||||
target = review.frame(frame_id)
|
||||
if target is None:
|
||||
raise review.ReviewError("No such frame")
|
||||
label_type = target["label_type"]
|
||||
prompt = _class_prompt(target["project_id"], class_id)
|
||||
|
||||
positives, negatives = [], []
|
||||
for item in exemplars:
|
||||
box = review.validate({"type": "bbox", "points": item["box"]}, "bbox")["points"]
|
||||
(positives if item.get("positive", True) else negatives).append(box)
|
||||
if not positives and not negatives:
|
||||
raise review.ReviewError("No exemplars to run")
|
||||
|
||||
# The GPU lock is shared with background jobs. Shorter than `review.assist`'s
|
||||
# 20s on purpose: this fires from a mouse gesture, so a long stall would feel
|
||||
# like a hung editor — and the fallback keeps the drawing rather than failing.
|
||||
if not jobs.gpu_lock.acquire(timeout=5):
|
||||
busy = jobs.running_types()
|
||||
kind = busy[0] if busy else "background"
|
||||
drawn = [{"geometry": _rect(box, label_type), "score": 1.0, "source": "manual"}
|
||||
for box in positives]
|
||||
if apply:
|
||||
# Never the replace path here: with no detections to put back, it
|
||||
# would wipe the class and leave only the drawings. Applying a
|
||||
# detection-less run just files the drawings and honours the
|
||||
# negatives.
|
||||
_append_drawn(frame_id, class_id, drawn)
|
||||
_drop_negative_overlaps(frame_id, class_id, negatives)
|
||||
return _result(frame_id, drawn, apply, redetected=False,
|
||||
message=f"The GPU is busy with a {kind} job — this is your drawing "
|
||||
"only, nothing was detected")
|
||||
|
||||
try:
|
||||
from backend.sam3_engine import get_engine
|
||||
|
||||
path = batches.frame_path(frame_id)
|
||||
with Image.open(path) as handle:
|
||||
image = handle.convert("RGB")
|
||||
width, height = image.size
|
||||
engine = get_engine()
|
||||
state = engine.open_state(image)
|
||||
found = engine.apply_prompts(
|
||||
state, threshold=threshold, text=prompt,
|
||||
exemplars=[{"box": _cxcywh(box), "positive": True} for box in positives]
|
||||
+ [{"box": _cxcywh(box), "positive": False} for box in negatives],
|
||||
)
|
||||
finally:
|
||||
jobs.gpu_lock.release()
|
||||
|
||||
# The panel's filters, in the order the batch job applies them (REQ-175):
|
||||
# area floor, then NMS, then the cap on how many survive.
|
||||
if min_box_frac > 0:
|
||||
floor = width * height * min_box_frac
|
||||
found = [d for d in found
|
||||
if (d.box[2] - d.box[0]) * (d.box[3] - d.box[1]) >= floor]
|
||||
found = deduplicate(found, iou_threshold)
|
||||
found.sort(key=lambda d: d.score, reverse=True)
|
||||
if max_detections > 0:
|
||||
found = found[:max_detections]
|
||||
|
||||
detections = [(_norm_box(d.box, width, height), d) for d in found]
|
||||
items: List[dict] = []
|
||||
|
||||
# The user's own boxes first, so the duplicate check below measures against
|
||||
# what they drew rather than the other way round.
|
||||
for box in positives:
|
||||
geometry = _rect(box, label_type)
|
||||
if label_type != "bbox":
|
||||
snapped = _snap(box, detections, width, height)
|
||||
if snapped is not None:
|
||||
geometry = snapped
|
||||
items.append({"geometry": geometry, "score": 1.0, "source": "manual"})
|
||||
|
||||
for norm, detection in detections:
|
||||
if any(_iou(norm, box) >= NEGATIVE_IOU for box in negatives):
|
||||
continue
|
||||
if any(_iou(norm, box) >= DUPLICATE_IOU for box in positives):
|
||||
continue
|
||||
for geometry in _detection_shapes(detection, norm, width, height, label_type):
|
||||
items.append({"geometry": geometry, "score": detection.score, "source": "auto"})
|
||||
|
||||
if apply:
|
||||
_replace_class(frame_id, class_id, items)
|
||||
return _result(frame_id, items, apply, redetected=True, message=None)
|
||||
|
||||
|
||||
def _result(frame_id: int, items: List[dict], applied: bool,
|
||||
redetected: bool, message: Optional[str]) -> dict:
|
||||
"""A preview carries the shapes; an apply also carries the frame as stored."""
|
||||
return {
|
||||
"shapes": items,
|
||||
"applied": applied,
|
||||
"redetected": redetected,
|
||||
"message": message,
|
||||
"annotations": review.listing(frame_id) if applied else None,
|
||||
}
|
||||
|
||||
|
||||
def _norm_box(box: List[float], width: int, height: int) -> List[float]:
|
||||
return [box[0] / width, box[1] / height, box[2] / width, box[3] / height]
|
||||
|
||||
|
||||
def _snap(drawn: List[float], detections, width: int, height: int) -> Optional[dict]:
|
||||
"""The mask polygon of whatever SAM3 found inside a drawn box.
|
||||
|
||||
A rectangle is a bad polygon label, so in a polygon project the drag is a
|
||||
prompt for the shape rather than the shape itself (REQ-173).
|
||||
"""
|
||||
from backend.labeling import _iou
|
||||
|
||||
best = None
|
||||
best_iou = SNAP_IOU
|
||||
for norm, detection in detections:
|
||||
if detection.mask is None:
|
||||
continue
|
||||
overlap = _iou(norm, drawn)
|
||||
if overlap >= best_iou:
|
||||
best, best_iou = detection, overlap
|
||||
if best is None:
|
||||
return None
|
||||
points = review.mask_to_polygons(best.mask)
|
||||
if not points or len(points[0]) < 3:
|
||||
return None
|
||||
return review.polygon([(x / width, y / height) for x, y in points[0]])
|
||||
|
||||
|
||||
def _detection_shapes(detection, norm: List[float], width: int, height: int,
|
||||
label_type: str) -> List[dict]:
|
||||
if label_type == "bbox" or detection.mask is None:
|
||||
return [review.bbox(*norm)]
|
||||
return [review.polygon([(x / width, y / height) for x, y in points])
|
||||
for points in review.mask_to_polygons(detection.mask)
|
||||
if len(points) >= 3]
|
||||
+12
-2
@@ -68,8 +68,13 @@ def label_image(
|
||||
threshold: float,
|
||||
iou_threshold: float = 0.8,
|
||||
min_box_frac: float = 0.0,
|
||||
exemplar_index: int = -1,
|
||||
exemplars: Optional[List[dict]] = None,
|
||||
) -> ImageResult:
|
||||
"""Detect every prompt in one image and return the surviving instances."""
|
||||
"""Detect every prompt in one image and return the surviving instances.
|
||||
|
||||
When `exemplars` are given, the prompt at `exemplar_index` also carries them
|
||||
as drawn box exemplars (REQ-172); every other prompt runs on text alone."""
|
||||
try:
|
||||
image = Image.open(image_path).convert("RGB")
|
||||
except Exception as exc: # unreadable/corrupt frame: report, don't abort the job
|
||||
@@ -77,7 +82,12 @@ def label_image(
|
||||
|
||||
width, height = image.size
|
||||
try:
|
||||
detections = get_engine().detect(image, prompts, threshold)
|
||||
if exemplars and 0 <= exemplar_index < len(prompts):
|
||||
detections = get_engine().detect_with_exemplars(
|
||||
image, prompts, threshold, exemplar_index, exemplars
|
||||
)
|
||||
else:
|
||||
detections = get_engine().detect(image, prompts, threshold)
|
||||
except Exception as exc:
|
||||
return ImageResult(image_path, rel_path, width, height, error=str(exc))
|
||||
|
||||
|
||||
+54
-128
@@ -1,4 +1,4 @@
|
||||
"""Live counting test bench: point a trained model at an RTSP stream and watch it count.
|
||||
"""Live counting test bench: point a trained model at a live stream and watch it count.
|
||||
|
||||
This is a **test harness**, not the production counter. It reuses the real
|
||||
pipeline pieces from `algoritma-batch` — ByteTrack, the bbox stabiliser and the
|
||||
@@ -22,6 +22,10 @@ import cv2
|
||||
import numpy as np
|
||||
|
||||
from backend import config, jobs
|
||||
from backend import live_render
|
||||
from backend.live_source import ( # re-exported: callers catch live_count.LiveCountError
|
||||
LiveCountError, _is_stream, _open, whep_to_rtsp, whep_url,
|
||||
)
|
||||
|
||||
# `algoritma-batch/src` is copied to /app/src in the image; in a source checkout
|
||||
# it still lives under algoritma-batch/. Both are made importable as `src.*`.
|
||||
@@ -31,10 +35,6 @@ for candidate in ("/app", os.path.join(_REPO, "algoritma-batch")):
|
||||
sys.path.insert(0, candidate)
|
||||
|
||||
|
||||
class LiveCountError(Exception):
|
||||
pass
|
||||
|
||||
|
||||
class Session:
|
||||
"""One running counter. Owns a capture thread and the latest rendered frame."""
|
||||
|
||||
@@ -43,8 +43,11 @@ class Session:
|
||||
dedup_radius: float, margin: int, imgsz: int,
|
||||
entry_travel_min: float, handoff_radius: float,
|
||||
unload_confirm_frames: int, min_area_scale: float,
|
||||
spatial_dedup: bool):
|
||||
spatial_dedup: bool, whep_url: str = ""):
|
||||
self.source = source
|
||||
# When the browser watches the camera over WebRTC it never asks for the
|
||||
# MJPEG, so encoding a JPEG per frame would be pure waste (REQ-177).
|
||||
self.whep_url = whep_url
|
||||
self.model_path = model_path
|
||||
self.line_y = line_y
|
||||
self.line_x_start = line_x_start
|
||||
@@ -75,6 +78,7 @@ class Session:
|
||||
self.events: list[dict] = []
|
||||
|
||||
self._jpeg: Optional[bytes] = None
|
||||
self._overlay: dict = {}
|
||||
self._counter = None # set once the worker builds it
|
||||
self._lock = threading.Lock()
|
||||
self._thread = threading.Thread(target=self._run, name="live-count", daemon=True)
|
||||
@@ -92,6 +96,10 @@ class Session:
|
||||
with self._lock:
|
||||
return self._jpeg
|
||||
|
||||
def overlay(self) -> dict:
|
||||
with self._lock:
|
||||
return dict(self._overlay)
|
||||
|
||||
def move_line(self, line_y: Optional[int] = None, line_x_start: Optional[int] = None,
|
||||
line_x_end: Optional[int] = None) -> dict:
|
||||
"""Reposition the counting line without restarting.
|
||||
@@ -119,6 +127,8 @@ class Session:
|
||||
return {
|
||||
"running": self._thread.is_alive(),
|
||||
"source": self.source,
|
||||
"whep_url": self.whep_url,
|
||||
"preview": "webrtc" if self.whep_url else "mjpeg",
|
||||
"model_path": self.model_path,
|
||||
"error": self.error,
|
||||
"frames": self.frames,
|
||||
@@ -227,7 +237,11 @@ class Session:
|
||||
self.fps = since / (now - tick)
|
||||
tick, since = now, 0
|
||||
|
||||
self._render(frame, inside, outside, counter)
|
||||
self._set_overlay(inside, outside, counter)
|
||||
if not self.whep_url:
|
||||
jpeg = live_render.render(self, frame, inside, outside, counter)
|
||||
with self._lock:
|
||||
self._jpeg = jpeg
|
||||
except Exception as exc: # surfaced in status(), not swallowed
|
||||
self.error = f"{type(exc).__name__}: {exc}"
|
||||
finally:
|
||||
@@ -263,59 +277,33 @@ class Session:
|
||||
if not self.error:
|
||||
self.error = f"trace write failed: {exc}"
|
||||
|
||||
def _render(self, frame, detections, ignored, counter) -> None:
|
||||
height, width = frame.shape[:2]
|
||||
|
||||
# Shade what the region excludes. Without this the neighbouring truck's
|
||||
# sacks simply vanish from the overlay, and "are they being ignored?"
|
||||
# looks identical to "is the model missing them?".
|
||||
if self.line_x_start > 0 or self.line_x_end < width:
|
||||
shade = frame.copy()
|
||||
if self.line_x_start > 0:
|
||||
cv2.rectangle(shade, (0, 0), (self.line_x_start, height), (0, 0, 0), -1)
|
||||
if self.line_x_end < width:
|
||||
cv2.rectangle(shade, (self.line_x_end, 0), (width, height), (0, 0, 0), -1)
|
||||
cv2.addWeighted(shade, 0.55, frame, 0.45, 0, frame)
|
||||
|
||||
# Ignored detections stay visible, in grey, so the region can be judged.
|
||||
for det in ignored:
|
||||
x1, y1, x2, y2 = (int(v) for v in det.bbox)
|
||||
cv2.rectangle(frame, (x1, y1), (x2, y2), (130, 130, 130), 1)
|
||||
|
||||
for edge in (self.line_x_start, self.line_x_end):
|
||||
if 0 < edge < width:
|
||||
cv2.line(frame, (edge, 0), (edge, height), (255, 0, 255), 2)
|
||||
cv2.putText(frame, "IGNORED", (max(4, self.line_x_start - 92), height - 14),
|
||||
cv2.FONT_HERSHEY_SIMPLEX, 0.5, (255, 0, 255), 1, cv2.LINE_AA)
|
||||
cv2.putText(frame, "IGNORED", (min(width - 88, self.line_x_end + 8), height - 14),
|
||||
cv2.FONT_HERSHEY_SIMPLEX, 0.5, (255, 0, 255), 1, cv2.LINE_AA)
|
||||
def _set_overlay(self, detections, ignored, counter) -> None:
|
||||
"""The same drawing `_render` burns into the JPEG, as geometry.
|
||||
|
||||
The browser draws it on a canvas over the WebRTC video, so the frames
|
||||
themselves never pass through this process. Coordinates are in the
|
||||
1280x720 space the pipeline works in; the canvas scales them.
|
||||
"""
|
||||
boxes = []
|
||||
for det in detections:
|
||||
x1, y1, x2, y2 = (int(v) for v in det.bbox)
|
||||
counted = bool(counter.counted_tracks.get(det.track_id))
|
||||
colour = (74, 222, 128) if counted else (248, 191, 113)
|
||||
cv2.rectangle(frame, (x1, y1), (x2, y2), colour, 2)
|
||||
cv2.putText(frame, f"#{det.track_id} {det.confidence:.2f}", (x1, max(14, y1 - 6)),
|
||||
cv2.FONT_HERSHEY_SIMPLEX, 0.45, colour, 1, cv2.LINE_AA)
|
||||
|
||||
cv2.line(frame, (self.line_x_start, self.line_y), (self.line_x_end, self.line_y),
|
||||
(0, 255, 255), 2)
|
||||
for edge in (self.line_y - self.margin, self.line_y + self.margin):
|
||||
cv2.line(frame, (self.line_x_start, edge), (self.line_x_end, edge),
|
||||
(0, 160, 160), 1)
|
||||
|
||||
panel = f"IN {self.loading} OUT {self.unloading} NET {self.loading - self.unloading}"
|
||||
cv2.rectangle(frame, (12, 12), (12 + 9 * len(panel) + 20, 84), (0, 0, 0), -1)
|
||||
cv2.putText(frame, panel, (24, 46), cv2.FONT_HERSHEY_SIMPLEX, 0.8,
|
||||
(74, 222, 128), 2, cv2.LINE_AA)
|
||||
cv2.putText(frame, f"{self.fps:.1f} fps {self.tracked} tracked {self.ignored} ignored",
|
||||
(24, 72), cv2.FONT_HERSHEY_SIMPLEX, 0.55, (200, 200, 200), 1, cv2.LINE_AA)
|
||||
|
||||
ok, buffer = cv2.imencode(".jpg", frame, [cv2.IMWRITE_JPEG_QUALITY, 75])
|
||||
if ok:
|
||||
with self._lock:
|
||||
self._jpeg = buffer.tobytes()
|
||||
|
||||
state = "counted" if counter.counted_tracks.get(det.track_id) else "tracked"
|
||||
boxes.append({"id": det.track_id, "b": [int(v) for v in det.bbox],
|
||||
"c": round(float(det.confidence), 2), "s": state})
|
||||
for det in ignored:
|
||||
boxes.append({"id": det.track_id, "b": [int(v) for v in det.bbox],
|
||||
"c": round(float(det.confidence), 2), "s": "ignored"})
|
||||
payload = {
|
||||
"frame": self.frames,
|
||||
"line": {"y": self.line_y, "x_start": self.line_x_start,
|
||||
"x_end": self.line_x_end, "margin": self.margin},
|
||||
"boxes": boxes,
|
||||
"loading": self.loading,
|
||||
"unloading": self.unloading,
|
||||
"net": self.loading - self.unloading,
|
||||
"fps": round(self.fps, 1),
|
||||
}
|
||||
with self._lock:
|
||||
self._overlay = payload
|
||||
|
||||
def _too_small(bbox, scale: float) -> bool:
|
||||
"""`predict.py`'s perspective curve: a sack at the top of the frame is
|
||||
@@ -339,74 +327,6 @@ def _too_small(bbox, scale: float) -> bool:
|
||||
return (x2 - x1) * (y2 - y1) < minimum * scale
|
||||
|
||||
|
||||
def _is_stream(source: str) -> bool:
|
||||
return str(source).startswith(("rtsp://", "rtmp://", "http://", "https://"))
|
||||
|
||||
|
||||
class _ThreadedStream:
|
||||
"""Decode in a background thread and always hand out the newest frame.
|
||||
|
||||
A plain VideoCapture.read() on RTSP is blocking, and decoding 1080p costs
|
||||
more than inference does — measured here at 6.7 fps end-to-end against 200
|
||||
fps for the model itself. Worse, reading slower than the camera sends builds
|
||||
a backlog, so the picture drifts further behind real time the longer it
|
||||
runs. Dropping stale frames keeps latency flat, which is what a counting
|
||||
test needs to mean anything. `predict.py` does the same thing.
|
||||
"""
|
||||
|
||||
def __init__(self, source: str):
|
||||
self._capture = cv2.VideoCapture(source)
|
||||
try:
|
||||
self._capture.set(cv2.CAP_PROP_BUFFERSIZE, 1)
|
||||
except Exception:
|
||||
pass
|
||||
self._frame = None
|
||||
self._lock = threading.Lock()
|
||||
self._running = True
|
||||
self._thread = threading.Thread(target=self._pump, daemon=True)
|
||||
if self._capture.isOpened():
|
||||
self._thread.start()
|
||||
|
||||
def _pump(self) -> None:
|
||||
while self._running:
|
||||
ok, frame = self._capture.read()
|
||||
if not ok:
|
||||
time.sleep(0.01)
|
||||
continue
|
||||
with self._lock:
|
||||
self._frame = frame
|
||||
|
||||
def isOpened(self) -> bool:
|
||||
return self._capture.isOpened()
|
||||
|
||||
def read(self):
|
||||
with self._lock:
|
||||
if self._frame is None:
|
||||
return False, None
|
||||
frame, self._frame = self._frame, None
|
||||
return True, frame
|
||||
|
||||
def release(self) -> None:
|
||||
self._running = False
|
||||
# The thread is only started when the capture opened, so a failed
|
||||
# source would otherwise raise "cannot join thread before it is
|
||||
# started" here — inside the caller's finally, skipping the GPU lock
|
||||
# release and wedging every later session on "GPU busy".
|
||||
if self._thread.is_alive():
|
||||
self._thread.join(timeout=2)
|
||||
self._capture.release()
|
||||
|
||||
|
||||
def _open(source: str):
|
||||
if _is_stream(source):
|
||||
os.environ.setdefault(
|
||||
"OPENCV_FFMPEG_CAPTURE_OPTIONS",
|
||||
"rtsp_transport;tcp|buffer_size;20480000|max_delay;500000",
|
||||
)
|
||||
return _ThreadedStream(source)
|
||||
return cv2.VideoCapture(source)
|
||||
|
||||
|
||||
# ---- module-level single session ----------------------------------------
|
||||
|
||||
_session: Optional[Session] = None
|
||||
@@ -417,7 +337,8 @@ def start(source: str, model_path: str, line_y: int, line_x_start: int, line_x_e
|
||||
conf: float = 0.35, dedup_radius: float = 60.0, margin: int = 5,
|
||||
imgsz: int = 640, entry_travel_min: float = 60.0,
|
||||
handoff_radius: float = 100.0, unload_confirm_frames: int = 3,
|
||||
min_area_scale: float = 1.0, spatial_dedup: bool = False) -> dict:
|
||||
min_area_scale: float = 1.0, spatial_dedup: bool = False,
|
||||
whep: str = "") -> dict:
|
||||
global _session
|
||||
with _guard:
|
||||
if _session is not None and _session.status()["running"]:
|
||||
@@ -429,7 +350,7 @@ def start(source: str, model_path: str, line_y: int, line_x_start: int, line_x_e
|
||||
_session = Session(source, model_path, line_y, line_x_start, line_x_end,
|
||||
conf, dedup_radius, margin, imgsz, entry_travel_min,
|
||||
handoff_radius, unload_confirm_frames, min_area_scale,
|
||||
spatial_dedup)
|
||||
spatial_dedup, whep)
|
||||
_session.start()
|
||||
time.sleep(0.4) # let an immediate failure surface in the response
|
||||
return _session.status()
|
||||
@@ -459,9 +380,14 @@ def status() -> dict:
|
||||
"fps": 0.0, "tracked": 0, "ignored": 0, "too_small": 0, "traced": 0,
|
||||
"trace_path": "", "elapsed": 0.0, "events": [],
|
||||
"error": "", "source": "", "model_path": "",
|
||||
"whep_url": "", "preview": "mjpeg",
|
||||
"line": {"y": 0, "x_start": 0, "x_end": 1280}}
|
||||
return _session.status()
|
||||
|
||||
|
||||
def snapshot() -> Optional[bytes]:
|
||||
return _session.snapshot() if _session is not None else None
|
||||
|
||||
|
||||
def overlay() -> dict:
|
||||
return _session.overlay() if _session is not None else {}
|
||||
@@ -0,0 +1,60 @@
|
||||
"""Burning the counting overlay into the frame, as an MJPEG.
|
||||
|
||||
Split out of `live_count.py` to keep it inside the 400-line limit. This is the
|
||||
fallback preview, used for archive files; a WebRTC session draws the same
|
||||
geometry on a canvas in the browser instead (REQ-177).
|
||||
"""
|
||||
|
||||
import cv2
|
||||
|
||||
|
||||
def render(session, frame, detections, ignored, counter):
|
||||
height, width = frame.shape[:2]
|
||||
|
||||
# Shade what the region excludes. Without this the neighbouring truck's
|
||||
# sacks simply vanish from the overlay, and "are they being ignored?"
|
||||
# looks identical to "is the model missing them?".
|
||||
if session.line_x_start > 0 or session.line_x_end < width:
|
||||
shade = frame.copy()
|
||||
if session.line_x_start > 0:
|
||||
cv2.rectangle(shade, (0, 0), (session.line_x_start, height), (0, 0, 0), -1)
|
||||
if session.line_x_end < width:
|
||||
cv2.rectangle(shade, (session.line_x_end, 0), (width, height), (0, 0, 0), -1)
|
||||
cv2.addWeighted(shade, 0.55, frame, 0.45, 0, frame)
|
||||
|
||||
# Ignored detections stay visible, in grey, so the region can be judged.
|
||||
for det in ignored:
|
||||
x1, y1, x2, y2 = (int(v) for v in det.bbox)
|
||||
cv2.rectangle(frame, (x1, y1), (x2, y2), (130, 130, 130), 1)
|
||||
|
||||
for edge in (session.line_x_start, session.line_x_end):
|
||||
if 0 < edge < width:
|
||||
cv2.line(frame, (edge, 0), (edge, height), (255, 0, 255), 2)
|
||||
cv2.putText(frame, "IGNORED", (max(4, session.line_x_start - 92), height - 14),
|
||||
cv2.FONT_HERSHEY_SIMPLEX, 0.5, (255, 0, 255), 1, cv2.LINE_AA)
|
||||
cv2.putText(frame, "IGNORED", (min(width - 88, session.line_x_end + 8), height - 14),
|
||||
cv2.FONT_HERSHEY_SIMPLEX, 0.5, (255, 0, 255), 1, cv2.LINE_AA)
|
||||
|
||||
for det in detections:
|
||||
x1, y1, x2, y2 = (int(v) for v in det.bbox)
|
||||
counted = bool(counter.counted_tracks.get(det.track_id))
|
||||
colour = (74, 222, 128) if counted else (248, 191, 113)
|
||||
cv2.rectangle(frame, (x1, y1), (x2, y2), colour, 2)
|
||||
cv2.putText(frame, f"#{det.track_id} {det.confidence:.2f}", (x1, max(14, y1 - 6)),
|
||||
cv2.FONT_HERSHEY_SIMPLEX, 0.45, colour, 1, cv2.LINE_AA)
|
||||
|
||||
cv2.line(frame, (session.line_x_start, session.line_y), (session.line_x_end, session.line_y),
|
||||
(0, 255, 255), 2)
|
||||
for edge in (session.line_y - session.margin, session.line_y + session.margin):
|
||||
cv2.line(frame, (session.line_x_start, edge), (session.line_x_end, edge),
|
||||
(0, 160, 160), 1)
|
||||
|
||||
panel = f"IN {session.loading} OUT {session.unloading} NET {session.loading - session.unloading}"
|
||||
cv2.rectangle(frame, (12, 12), (12 + 9 * len(panel) + 20, 84), (0, 0, 0), -1)
|
||||
cv2.putText(frame, panel, (24, 46), cv2.FONT_HERSHEY_SIMPLEX, 0.8,
|
||||
(74, 222, 128), 2, cv2.LINE_AA)
|
||||
cv2.putText(frame, f"{session.fps:.1f} fps {session.tracked} tracked {session.ignored} ignored",
|
||||
(24, 72), cv2.FONT_HERSHEY_SIMPLEX, 0.55, (200, 200, 200), 1, cv2.LINE_AA)
|
||||
|
||||
ok, buffer = cv2.imencode(".jpg", frame, [cv2.IMWRITE_JPEG_QUALITY, 75])
|
||||
return buffer.tobytes() if ok else None
|
||||
@@ -0,0 +1,131 @@
|
||||
"""Opening a live source, and the URL algebra around it.
|
||||
|
||||
Split out of `live_count.py` so that file stays inside the 400-line limit: this
|
||||
is the transport layer (what a "source" is, how it is opened, how a WebRTC URL
|
||||
maps to the RTSP leg of the same stream), not the counting logic.
|
||||
"""
|
||||
|
||||
import os
|
||||
import threading
|
||||
import time
|
||||
|
||||
import cv2
|
||||
|
||||
|
||||
class LiveCountError(Exception):
|
||||
pass
|
||||
|
||||
|
||||
def _is_stream(source: str) -> bool:
|
||||
return str(source).startswith(("rtsp://", "rtmp://", "http://", "https://"))
|
||||
|
||||
|
||||
# MediaMTX fans one camera out to several protocols on one host: WHEP for
|
||||
# browsers, RTSP for decoders. The ports are the server's, not ours, so they
|
||||
# come from the environment rather than the code (REQ-176).
|
||||
WHEP_PATH = os.environ.get("MEDIAMTX_WHEP_PATH", "/whep")
|
||||
RTSP_PORT = os.environ.get("MEDIAMTX_RTSP_PORT", "8554")
|
||||
|
||||
|
||||
def whep_url(source: str) -> str:
|
||||
"""The URL a browser POSTs its SDP offer to, for a WebRTC stream URL."""
|
||||
trimmed = source.rstrip("/")
|
||||
return trimmed if trimmed.endswith(WHEP_PATH) else trimmed + WHEP_PATH
|
||||
|
||||
|
||||
def whep_to_rtsp(source: str) -> str:
|
||||
"""`http://host:8889/cam` -> `rtsp://host:8554/cam`.
|
||||
|
||||
The browser and the counter watch the same camera, but not over the same
|
||||
protocol: WebRTC is what makes the *preview* cheap, while pulling it into
|
||||
Python would add ICE and a jitter buffer on top of exactly the same H.264
|
||||
decode RTSP already does. So the AI counts from the RTSP leg of the same
|
||||
MediaMTX path — one ingest, two consumers.
|
||||
"""
|
||||
from urllib.parse import urlparse
|
||||
|
||||
parsed = urlparse(source)
|
||||
if parsed.scheme not in ("http", "https") or not parsed.hostname:
|
||||
raise LiveCountError(
|
||||
"A live source must be a WebRTC (WHEP) URL, e.g. http://host:8889/cam")
|
||||
path = parsed.path.rstrip("/")
|
||||
if path.endswith(WHEP_PATH):
|
||||
path = path[: -len(WHEP_PATH)]
|
||||
if not path.strip("/"):
|
||||
raise LiveCountError(f"No stream path in {source} — expected e.g. http://host:8889/cam")
|
||||
return f"rtsp://{parsed.hostname}:{RTSP_PORT}{path}"
|
||||
|
||||
|
||||
class _ThreadedStream:
|
||||
"""Decode in a background thread and always hand out the newest frame.
|
||||
|
||||
A plain VideoCapture.read() on RTSP is blocking, and getting the frames
|
||||
there costs far more than looking at them does — measured on this camera at
|
||||
8.3 fps for decode alone against 205 fps for tracking and inference. Worse, reading slower than the camera sends builds
|
||||
a backlog, so the picture drifts further behind real time the longer it
|
||||
runs. Dropping stale frames keeps latency flat, which is what a counting
|
||||
test needs to mean anything. `predict.py` does the same thing.
|
||||
"""
|
||||
|
||||
def __init__(self, source: str):
|
||||
self._capture = cv2.VideoCapture(source)
|
||||
try:
|
||||
self._capture.set(cv2.CAP_PROP_BUFFERSIZE, 1)
|
||||
except Exception:
|
||||
pass
|
||||
self._frame = None
|
||||
self._lock = threading.Lock()
|
||||
self._running = True
|
||||
self._thread = threading.Thread(target=self._pump, daemon=True)
|
||||
if self._capture.isOpened():
|
||||
self._thread.start()
|
||||
|
||||
def _pump(self) -> None:
|
||||
while self._running:
|
||||
ok, frame = self._capture.read()
|
||||
if not ok:
|
||||
time.sleep(0.01)
|
||||
continue
|
||||
with self._lock:
|
||||
self._frame = frame
|
||||
|
||||
def isOpened(self) -> bool:
|
||||
return self._capture.isOpened()
|
||||
|
||||
def read(self):
|
||||
with self._lock:
|
||||
if self._frame is None:
|
||||
return False, None
|
||||
frame, self._frame = self._frame, None
|
||||
return True, frame
|
||||
|
||||
def release(self) -> None:
|
||||
self._running = False
|
||||
# The thread is only started when the capture opened, so a failed
|
||||
# source would otherwise raise "cannot join thread before it is
|
||||
# started" here — inside the caller's finally, skipping the GPU lock
|
||||
# release and wedging every later session on "GPU busy".
|
||||
if self._thread.is_alive():
|
||||
self._thread.join(timeout=2)
|
||||
self._capture.release()
|
||||
|
||||
|
||||
# TCP, and it is worth writing down why, because the obvious reasoning gives the
|
||||
# wrong answer here. The link to the streaming server is a ZeroTier VPN with 15%
|
||||
# packet loss and a 41-104 ms round trip, which is exactly the case where UDP is
|
||||
# supposed to win — and with the ffmpeg CLI it does, 16 fps against 6.8. Through
|
||||
# OpenCV it loses badly: measured over 30 s on this camera, tcp 6.4 fps, udp with
|
||||
# a socket buffer 4.4, bare udp 2.4. OpenCV drops what it cannot reassemble
|
||||
# rather than showing it, so the loss becomes missing frames. Left overridable
|
||||
# because on a clean LAN the answer flips back.
|
||||
RTSP_TRANSPORT = os.environ.get("RTSP_TRANSPORT", "tcp")
|
||||
|
||||
|
||||
def _open(source: str):
|
||||
if _is_stream(source):
|
||||
os.environ.setdefault(
|
||||
"OPENCV_FFMPEG_CAPTURE_OPTIONS",
|
||||
f"rtsp_transport;{RTSP_TRANSPORT}|buffer_size;1048576|max_delay;200000",
|
||||
)
|
||||
return _ThreadedStream(source)
|
||||
return cv2.VideoCapture(source)
|
||||
@@ -0,0 +1,151 @@
|
||||
"""The auto-annotate preview: one frame, run now, nothing written (REQ-171/172).
|
||||
|
||||
Split out of `autolabel.py` to keep that file under the 400-line limit. The job
|
||||
path and this path share `labeling.label_image`, so what the preview shows is
|
||||
what a batch run would write — with one deliberate exception: drawn box
|
||||
exemplars (REQ-172) only ever apply here. SAM3's geometric prompts pool features
|
||||
from the current image, so replaying them on another frame would ask about
|
||||
whatever happens to sit at those coordinates there.
|
||||
"""
|
||||
|
||||
import os
|
||||
from typing import List, Optional
|
||||
|
||||
from backend import batches, db, labeling, projects, review
|
||||
from backend.autolabel import DEFAULT_IOU, DEFAULT_THRESHOLD, _geometries
|
||||
|
||||
def preview_frame(
|
||||
batch_id: int,
|
||||
frame_id: int,
|
||||
engine: str,
|
||||
threshold: float = DEFAULT_THRESHOLD,
|
||||
iou_threshold: float = DEFAULT_IOU,
|
||||
min_box_frac: float = 0.0,
|
||||
target_class_names: Optional[List[str]] = None,
|
||||
custom_model_path: Optional[str] = None,
|
||||
exemplars: Optional[List[dict]] = None,
|
||||
exemplar_class_name: Optional[str] = None,
|
||||
) -> List[dict]:
|
||||
batch = batches.get(batch_id)
|
||||
if not batch:
|
||||
raise ValueError("No such batch")
|
||||
project = projects.get(batch["project_id"])
|
||||
|
||||
frame = next((f for f in batches.frames(batch_id) if f["id"] == frame_id), None)
|
||||
if not frame:
|
||||
raise ValueError("Frame not found")
|
||||
|
||||
directory = batches.frames_dir(batch["project_slug"], batch_id)
|
||||
frame_file = os.path.join(directory, frame["filename"])
|
||||
fw = max(1, frame.get("width") or 1)
|
||||
fh = max(1, frame.get("height") or 1)
|
||||
|
||||
yolo_model = None
|
||||
sam3_target_classes = []
|
||||
|
||||
if engine == "sam3" and not custom_model_path:
|
||||
allowed_classes_set = {c.strip().lower() for c in target_class_names} if target_class_names else None
|
||||
if allowed_classes_set:
|
||||
sam3_target_classes = [c for c in project["classes"] if c["name"].strip().lower() in allowed_classes_set or c["prompt"].strip().lower() in allowed_classes_set]
|
||||
else:
|
||||
sam3_target_classes = [c for c in project["classes"]]
|
||||
|
||||
prompts = [c["prompt"] for c in sam3_target_classes]
|
||||
if prompts:
|
||||
from backend.sam3_engine import get_engine
|
||||
get_engine()
|
||||
else:
|
||||
from ultralytics import YOLO
|
||||
if custom_model_path and os.path.isfile(custom_model_path):
|
||||
m_path = custom_model_path
|
||||
else:
|
||||
m_path = projects.training_start_point(project)
|
||||
with db.cursor() as cur:
|
||||
cur.execute("SELECT weights_path FROM model_versions WHERE project_id = ? ORDER BY version DESC LIMIT 1", (project["id"],))
|
||||
row = cur.fetchone()
|
||||
if row and os.path.isfile(row[0]):
|
||||
m_path = row[0]
|
||||
yolo_model = YOLO(m_path)
|
||||
|
||||
name_to_class_id = {item["name"].strip().lower(): item["class_id"] for item in project["classes"]}
|
||||
allowed_classes_set = {c.strip().lower() for c in target_class_names} if target_class_names else None
|
||||
|
||||
all_raw_detections = []
|
||||
|
||||
if yolo_model is not None:
|
||||
results = yolo_model.predict(frame_file, conf=threshold, verbose=False)
|
||||
if results and len(results) > 0:
|
||||
model_names = results[0].names
|
||||
for box in results[0].boxes:
|
||||
cls_idx = int(box.cls[0].item())
|
||||
raw_cls_name = str(model_names.get(cls_idx, cls_idx)).strip().lower()
|
||||
|
||||
target_class_id = name_to_class_id.get(raw_cls_name)
|
||||
if target_class_id is None:
|
||||
for item in project["classes"]:
|
||||
if item["class_id"] == cls_idx:
|
||||
target_class_id = item["class_id"]
|
||||
break
|
||||
if target_class_id is None and 0 <= cls_idx < len(project["classes"]):
|
||||
target_class_id = project["classes"][cls_idx]["class_id"]
|
||||
|
||||
if target_class_id is None:
|
||||
continue
|
||||
|
||||
target_cls_obj = next((c for c in project["classes"] if c["class_id"] == target_class_id), None)
|
||||
proj_cls_name = target_cls_obj["name"].strip().lower() if target_cls_obj else ""
|
||||
|
||||
if allowed_classes_set is not None:
|
||||
if (raw_cls_name not in allowed_classes_set and
|
||||
proj_cls_name not in allowed_classes_set and
|
||||
str(target_class_id) not in allowed_classes_set):
|
||||
continue
|
||||
|
||||
score = float(box.conf[0].item())
|
||||
xyxyn = box.xyxyn[0].tolist()
|
||||
all_raw_detections.append(labeling.Detection(
|
||||
class_id=target_class_id,
|
||||
class_name=proj_cls_name or raw_cls_name,
|
||||
box=[xyxyn[0]*fw, xyxyn[1]*fh, xyxyn[2]*fw, xyxyn[3]*fh],
|
||||
score=score,
|
||||
mask=None
|
||||
))
|
||||
|
||||
if engine == "sam3" and sam3_target_classes:
|
||||
prompts = [(c.get("prompt") or c["name"]).strip() for c in sam3_target_classes]
|
||||
# Exemplars belong to exactly one class — the chip that was active when
|
||||
# they were drawn. An unknown name means no exemplar class, so the run
|
||||
# falls back to plain text rather than silently attaching the boxes to
|
||||
# whichever class happens to be first.
|
||||
exemplar_index = -1
|
||||
if exemplars and exemplar_class_name:
|
||||
wanted = exemplar_class_name.strip().lower()
|
||||
exemplar_index = next(
|
||||
(i for i, c in enumerate(sam3_target_classes)
|
||||
if c["name"].strip().lower() == wanted),
|
||||
-1,
|
||||
)
|
||||
res = labeling.label_image(
|
||||
frame_file, frame["filename"], prompts, threshold,
|
||||
iou_threshold=iou_threshold, min_box_frac=min_box_frac,
|
||||
exemplar_index=exemplar_index, exemplars=exemplars
|
||||
)
|
||||
if not res.error and res.detections:
|
||||
for det in res.detections:
|
||||
if 0 <= det.class_id < len(sam3_target_classes):
|
||||
real_cls = sam3_target_classes[det.class_id]
|
||||
det.class_id = real_cls["class_id"]
|
||||
det.class_name = real_cls["name"]
|
||||
all_raw_detections.append(det)
|
||||
|
||||
kept = labeling.deduplicate(all_raw_detections, iou_threshold=iou_threshold)
|
||||
items = []
|
||||
for det in kept:
|
||||
if project["label_type"] == "bbox" or det.mask is None:
|
||||
geom = review.bbox(det.box[0]/fw, det.box[1]/fh, det.box[2]/fw, det.box[3]/fh)
|
||||
items.append({"class_id": det.class_id, "geometry": geom, "score": det.score})
|
||||
else:
|
||||
for geometry in _geometries(det, fw, fh, project["label_type"]):
|
||||
items.append({"class_id": det.class_id, "geometry": geometry, "score": det.score})
|
||||
|
||||
return items
|
||||
@@ -103,6 +103,41 @@ class Sam3Engine:
|
||||
|
||||
|
||||
|
||||
def detect_with_exemplars(
|
||||
self,
|
||||
image: Image.Image,
|
||||
prompts: List[str],
|
||||
threshold: float,
|
||||
exemplar_index: int,
|
||||
exemplars: List[dict],
|
||||
) -> List[Detection]:
|
||||
"""`detect()`, but one prompt also carries drawn box exemplars (REQ-172).
|
||||
|
||||
Still one `set_image` for the whole call. The prompt set is reset before
|
||||
every class because `state["geometric_prompt"]` survives `set_text_prompt`
|
||||
— without the reset, one class's boxes would leak into the next class.
|
||||
"""
|
||||
processor = Sam3Processor(self.model, device=self.device)
|
||||
processor.confidence_threshold = threshold
|
||||
|
||||
detections: List[Detection] = []
|
||||
with torch.autocast(self.device, dtype=self.autocast_dtype):
|
||||
state = processor.set_image(image)
|
||||
for class_id, prompt in enumerate(prompts):
|
||||
processor.reset_all_prompts(state)
|
||||
output = processor.set_text_prompt(prompt=prompt, state=state)
|
||||
if class_id == exemplar_index:
|
||||
for exemplar in exemplars:
|
||||
output = processor.add_geometric_prompt(
|
||||
box=exemplar["box"],
|
||||
label=bool(exemplar.get("positive", True)),
|
||||
state=state,
|
||||
)
|
||||
detections.extend(self._collect(output, class_id, prompt))
|
||||
del state
|
||||
return detections
|
||||
|
||||
|
||||
|
||||
# ---- interactive / exemplar prompting ------------------------------
|
||||
|
||||
|
||||
Binary file not shown.
+115
@@ -123,9 +123,14 @@ box.
|
||||
| `batches.py` | batch lifecycle | **new** |
|
||||
| `review.py` | annotation CRUD, frame status, click-assist | **new** |
|
||||
| `autolabel.py` | the SAM3 job over a whole batch | **new** |
|
||||
| `preview.py` | one-frame preview for the auto-annotate modal (REQ-171,172) | **new** |
|
||||
| `exemplar.py` | exemplar-driven labeling in the review editor (REQ-173,174) | **new** |
|
||||
| `dataset.py` | merge into the master dataset, stable split | **new** |
|
||||
| `evaluate.py` | validate base vs new model | **new** |
|
||||
| `hardware.py` | VRAM detection → training defaults | **new** |
|
||||
| `live_count.py` | the live counting session: capture → track → count | **new** |
|
||||
| `live_source.py` | what a source is, how it opens, WHEP↔RTSP (REQ-176) | **new** |
|
||||
| `live_render.py` | the MJPEG overlay, archive-file preview only (REQ-177) | **new** |
|
||||
| `api/` | the FastAPI routes, one module per domain | **new** |
|
||||
|
||||
Removed: `uploads.py`, `static/index.html`, and the old flow's endpoints.
|
||||
@@ -167,6 +172,9 @@ POST /api/projects/{id}/batches # {rel, start_sec, end_sec, fps}
|
||||
GET /api/batches/{id} # status + review progress (REQ-045)
|
||||
GET /api/batches/{id}/frames # frames + statuses
|
||||
POST /api/batches/{id}/autolabel # {threshold} → job (REQ-030,032,034)
|
||||
POST /api/batches/{id}/preview # one frame, run now, nothing written;
|
||||
# +exemplars[] {box:[cx,cy,w,h], positive}
|
||||
# +exemplar_class_name (REQ-171,172)
|
||||
DELETE /api/batches/{id}/classes/{class_id}/annotations # clear all shapes of class in batch (REQ-046)
|
||||
POST /api/batches/{ids}/approve # one or many, comma-separated → one merge job (REQ-131)
|
||||
GET /api/batches/{ids}/triage/summary # one or many, comma-separated (REQ-130)
|
||||
@@ -180,6 +188,7 @@ POST /api/frames/{id}/annotations # add a manual shape (REQ-042)
|
||||
PATCH /api/annotations/{id} # move/resize/reclass
|
||||
DELETE /api/annotations/{id}
|
||||
POST /api/frames/{id}/assist # click/box → SAM3 shape (REQ-043)
|
||||
POST /api/frames/{id}/exemplar-label # drawn pool → re-detect one class (REQ-173,174)
|
||||
POST /api/frames/{id}/status # approved | rejected | pending (REQ-041)
|
||||
|
||||
POST /api/projects/{id}/train # → train job (REQ-060,061,062)
|
||||
@@ -187,11 +196,36 @@ GET /api/projects/{id}/models # versions + metrics (REQ-063,06
|
||||
GET /api/models/{id}/weights # download best.pt
|
||||
POST /api/models/{id}/promote # make it the project's base model (REQ-064)
|
||||
|
||||
GET /api/projects/{id}/live-count/models # weights this project can count with
|
||||
POST /api/projects/{id}/live-count/start # {source|source_rel, model_path, dials}
|
||||
# source must be a WHEP URL (REQ-176)
|
||||
POST /api/live-count/stop
|
||||
PATCH /api/live-count/line # move the line mid-session
|
||||
GET /api/live-count/status # counts + preview: "webrtc" | "mjpeg"
|
||||
GET /api/live-count/overlay # boxes/line/counts for the canvas (REQ-177)
|
||||
GET /api/live-count/stream # MJPEG; 409 on a WebRTC session (REQ-177)
|
||||
|
||||
GET /api/jobs?project_id=… REQ-070,071
|
||||
GET /api/jobs/{id}
|
||||
POST /api/jobs/{id}/cancel
|
||||
```
|
||||
|
||||
### Live counting (REQ-176, REQ-177)
|
||||
|
||||
One camera, one ingest on the streaming server, two consumers:
|
||||
|
||||
```
|
||||
camera ──▶ MediaMTX ──┬── WHEP :8889/cam/whep ──▶ browser <video> + <canvas> overlay
|
||||
└── RTSP :8554/cam ──▶ backend decode → YOLO → counter
|
||||
└─▶ GET /api/live-count/overlay
|
||||
```
|
||||
|
||||
The user types only the WHEP URL; `live_source.whep_to_rtsp` derives the RTSP one, with the
|
||||
ports coming from `MEDIAMTX_RTSP_PORT` / `MEDIAMTX_WHEP_PATH`. Because the browser plays the
|
||||
camera directly, a live session encodes no JPEG at all — `live_render.py` runs only for
|
||||
archive files, which have no WebRTC leg. The canvas draws exactly what `live_render.py`
|
||||
would have burned in, so both previews describe the same session.
|
||||
|
||||
## Job flows
|
||||
|
||||
**extract (REQ-020…023).** `ffmpeg -ss <start> -to <end> -i <video> -vf fps=<n> -q:v 2
|
||||
@@ -277,6 +311,87 @@ Pages:
|
||||
The canvas editor is hand-written; the normalized-coordinate conventions already exist in
|
||||
`sessions.py` (`annotations_payload`, `detections_payload`) as a reference.
|
||||
|
||||
**Auto-annotate modal (REQ-171, REQ-172).** `AutoAnnotateModal.jsx` splits into
|
||||
`PreviewShapes.jsx` (the result overlay, shared with the mass modal),
|
||||
`ClassPromptPanel.jsx` (class chips + the editable SAM3 prompt) and `ExemplarCanvas.jsx`
|
||||
(the drag-to-draw layer). All three overlays and the `<img>` share one shrink-wrapped
|
||||
`position: relative` wrapper — they are sized to it, so nothing else may sit inside it or
|
||||
every box shifts off the pixels it describes.
|
||||
|
||||
With SAM3 one selected chip is *active*: it owns the prompt field and any exemplars drawn on
|
||||
the frame, so its chip is a pair of buttons — the name activates, the `×` deselects.
|
||||
Exemplars are normalized `[cx, cy, w, h]`, sent only to `/preview`, and dropped whenever the
|
||||
frame or the active class changes. A redraw is debounced 250 ms and re-runs the whole prompt
|
||||
set from empty, which is also how undo works — SAM3 can only append geometric prompts.
|
||||
|
||||
**Exemplar-driven labeling in review (REQ-173, REQ-174, REQ-175).** In `draw` mode a drag on
|
||||
`AnnotationCanvas` is an exemplar, not a rectangle: `onExemplar(box, positive)` where
|
||||
`positive` is `!event.shiftKey`. `hooks/useExemplarPool.js` keeps the pool in a ref as well
|
||||
as state — the ref is what gets sent, so a drag that lands mid-flight is never lost — and posts
|
||||
the **whole pool** to `/exemplar-label` 400 ms after the last drag. One pass runs at a time;
|
||||
a drag arriving during a pass sets `rerunWanted` so exactly one rerun follows instead of a
|
||||
queue. The pool is cleared by a frame change or a class change, and `Undo example` re-runs
|
||||
with the shortened pool.
|
||||
|
||||
The run is a **dry run by default** (REQ-175). `ExemplarFilterPanel.jsx` is passed to
|
||||
`ReviewSidebar` and rendered above the class list — not over the frame, which is where its
|
||||
proposals are drawn. It is mounted only while a run is undecided (`pool.active`): the first
|
||||
drag opens it, Apply and Discard close it, and the pool outlives it. Its four sliders
|
||||
re-preview on a 250 ms debounce. While a preview is up the canvas **hides the stored shapes
|
||||
of the class under review** — the run replaces them wholesale, and leaving them on screen made
|
||||
a rejected detection look like it had never gone; other classes stay, dimmed
|
||||
(`svg.previewing`). Proposals draw dashed on top, green for the boxes the user drew and
|
||||
class-colored for what SAM3 found.
|
||||
`Apply` re-runs with `apply: true` rather than posting the previewed geometry back — SAM3 is
|
||||
deterministic for a pool and a threshold, and the browser should not be the authority on what
|
||||
gets stored. `Discard` truncates the pool to `appliedRef`, the length it had at the last
|
||||
successful apply, so a rejected run leaves neither shapes nor prompts behind. A successful
|
||||
apply drops the negatives it sent and keeps the positives (REQ-174); drags that landed while
|
||||
that request was in flight are not part of it and stay at the end of the pool, with a rerun
|
||||
queued for them.
|
||||
|
||||
`.canvas-wrap`'s overlay rules are scoped to its direct child (`> svg`): they set
|
||||
`position: absolute; width: 100%`, which any nested SVG — an icon in a panel, say — would
|
||||
otherwise inherit and stretch across the whole frame.
|
||||
|
||||
Request/response:
|
||||
|
||||
```
|
||||
POST /api/frames/{id}/exemplar-label
|
||||
{ "exemplars": [{"box": [x0,y0,x1,y1], "positive": true}, …], # normalized xyxy
|
||||
"class_id": 0,
|
||||
"threshold": 0.5, "iou_threshold": 0.8, # the panel (REQ-175)
|
||||
"min_box_frac": 0.002, "max_detections": 100,
|
||||
"apply": false }
|
||||
→ { "shapes": [{"geometry": …, "score": …, "source": "manual"|"auto"}, …],
|
||||
"applied": false, "redetected": true, "message": null,
|
||||
"annotations": null } # the frame, on apply only
|
||||
```
|
||||
|
||||
`backend/exemplar.py` converts each box to SAM3's normalized `[cx, cy, w, h]`, runs one
|
||||
`open_state` + `apply_prompts` with the class's stored `prompt` as text plus every exemplar,
|
||||
then rewrites the frame **for that class only**:
|
||||
|
||||
- positives are stored first as `source='manual'` — the literal rectangle in a `bbox`
|
||||
project, the mask polygon of whatever SAM3 found inside it (IoU ≥ `SNAP_IOU`) in a
|
||||
`polygon` one;
|
||||
- detections overlapping a negative by ≥ `NEGATIVE_IOU` (0.3) are dropped, and ones
|
||||
overlapping a positive by ≥ `DUPLICATE_IOU` (0.6) are dropped as the user's own shape
|
||||
already covers them;
|
||||
- what remains is written as `source='auto'`, so a later batch re-run replaces it (REQ-034)
|
||||
while the drawn shapes survive.
|
||||
|
||||
The panel's filters run before any of that, in the order the batch job uses them: area floor,
|
||||
then NMS (`labeling.deduplicate`), then the cap on how many survive.
|
||||
|
||||
The delete-and-reinsert happens in one `db.cursor()` transaction, so the frame is never
|
||||
briefly empty. Other classes on the frame are never touched. The GPU lock is taken with a
|
||||
**5 s** timeout — shorter than `assist`'s 20 s because this fires from a mouse gesture; on
|
||||
timeout the response carries `redetected: false`, a message the panel shows, and the drawn
|
||||
boxes alone as the preview. Applying that run **appends** the drawn boxes and honours the
|
||||
negatives instead of taking the replace path — with no detections to put back, replacing
|
||||
would wipe the class and leave only the drawings.
|
||||
|
||||
## Docker (REQ-072)
|
||||
|
||||
- `Dockerfile` — python 3.12, `ffmpeg`, `uv`, CUDA torch, `uv pip install -e sam3/`. The
|
||||
|
||||
@@ -100,6 +100,19 @@ changes.
|
||||
found nothing on (REQ-033) writes no annotations, so a resume re-does it; that is accepted
|
||||
rather than tracked.
|
||||
|
||||
- **REQ-171** — In the auto-annotate modal the SAM3 text prompt of each selected class is
|
||||
editable in place, next to the live preview. Saving it writes
|
||||
`project_classes.prompt` — the same field the Projects page edits — so the batch job and
|
||||
every later run send that text. The preview is the tuning surface; the stored prompt is the
|
||||
artifact it produces.
|
||||
- **REQ-172** — On the previewed frame the user can drag **positive** and **negative** box
|
||||
exemplars (shift-drag for negative). They are appended to the active class's text prompt,
|
||||
re-run immediately, and can be undone or cleared. Exemplars are a **tuning aid only**: they
|
||||
are never written as annotations and never carried into the batch job, because SAM3's
|
||||
geometric prompts pool features from the current image — replaying them on another frame
|
||||
would ask about whatever happens to sit at those coordinates there. They belong to exactly
|
||||
one class, so a new frame or a new active class discards them.
|
||||
|
||||
## E. Review & correction
|
||||
|
||||
- **REQ-040** — The user reviews frames one at a time, with fast navigation (left/right
|
||||
@@ -117,6 +130,53 @@ changes.
|
||||
- **REQ-046** — The user can delete/clear all annotations of a specific class across all frames in
|
||||
the current batch from the Review editor.
|
||||
|
||||
- **REQ-173** — In the review editor a plain drag on the canvas is an **exemplar-driven
|
||||
label**, not just a rectangle. It proposes, in one action — and REQ-175's Apply is what
|
||||
makes any of it real — that (a) the drawn shape becomes a `manual` annotation of the
|
||||
active class — snapped to a SAM3 polygon first when the batch's
|
||||
`label_type` is `polygon`, since a rectangle is a bad polygon label — (b) appends the box
|
||||
to the frame's positive exemplar pool for that class, and (c) re-runs SAM3 over the whole
|
||||
frame with the class's text prompt plus the pooled exemplars, deleting every existing shape
|
||||
of that class on the frame and writing the detections in its place, then re-inserting the
|
||||
pooled exemplar shapes verbatim so the user's own drawings always survive. The pool is
|
||||
**frame-local and ephemeral** for the same reason as REQ-172 — SAM3's geometric prompts pool
|
||||
features from the current image — so leaving the frame or switching the active class clears
|
||||
it; the annotations it produced persist like any other. If the GPU lock (REQ-070) is not
|
||||
free, the run comes back with the drawn shapes alone and says so, so the user can still
|
||||
file them (REQ-175) and labeling is never blocked by a background job.
|
||||
- **REQ-174** — **Shift**-drag in the review editor adds a **negative** exemplar. It is never
|
||||
stored as an annotation; it deletes any existing shape of the active class that overlaps it,
|
||||
and it is sent as a negative box in the REQ-173 re-detect. It is the "not this, and not
|
||||
things like this" gesture, so it doubles as a delete. A negative is **spent on Apply**: the
|
||||
frame it was applied to no longer carries what it rejected, so the drawing is dropped from
|
||||
the pool while the positives stay on as prompts.
|
||||
- **REQ-175** — An exemplar drag **previews**; it never writes on its own. The run's result
|
||||
is drawn over the frame as proposals and a small panel floats on the canvas with the four
|
||||
filters that decide what survives — confidence, NMS overlap, minimum box size, maximum
|
||||
shapes — each re-running the preview as it moves. **Apply** writes the previewed set,
|
||||
**Discard** rewinds the pool to whatever is already on the frame and leaves it untouched.
|
||||
The panel is scoped to this gesture: its values are not stored, not shared with the
|
||||
auto-annotate modal, and reset with the frame. Defaults are confidence `0.5`, NMS `0.8`,
|
||||
min box `0.002`, max `100` — deliberately permissive, because on a dense frame an
|
||||
aggressive NMS or area floor deletes real, touching objects rather than duplicates.
|
||||
|
||||
## E4. Live counting preview
|
||||
|
||||
- **REQ-176** — A **live** source on the Live Count page is a **WebRTC (WHEP) URL** and
|
||||
nothing else; an RTSP URL is rejected with a message saying so. The backend derives the
|
||||
RTSP leg of the same streaming-server path from it (`http://host:8889/cam` →
|
||||
`rtsp://host:8554/cam`) and counts from that: WebRTC is what makes the browser preview
|
||||
cheap, but pulling it into Python would add ICE and a jitter buffer on top of the identical
|
||||
H.264 decode. One ingest on the streaming server, two consumers. The ports are read from
|
||||
the environment (`MEDIAMTX_RTSP_PORT`, `MEDIAMTX_WHEP_PATH`), never hardcoded. Archive
|
||||
files are unaffected — they are still opened as files.
|
||||
- **REQ-177** — A live session is **watched over WebRTC**, played straight from the streaming
|
||||
server by the browser: the frames never pass through this app and it encodes no JPEG for
|
||||
them. What the model saw — boxes, ids, confidences, the counting line and its band, the
|
||||
ignored region, the running totals — is served as geometry from
|
||||
`GET /api/live-count/overlay` and drawn on a canvas over the video. The MJPEG endpoint
|
||||
remains the preview for **archive files** only, and refuses a WebRTC session.
|
||||
|
||||
## F. Master dataset
|
||||
|
||||
- **REQ-050** — Approving a batch **merges** its approved frames and their labels into the
|
||||
@@ -164,6 +224,31 @@ changes.
|
||||
and the reason it did or did not count, so a miss can be attributed to the model, the
|
||||
tracker, or the counter.
|
||||
|
||||
- **REQ-145** — Counting algorithms are **pluggable**. Each registers under a stable id
|
||||
(`line_cross`, `possession`) and the session constructs one by id. The `Counter` protocol
|
||||
in `src/interfaces.py` is the contract, corrected to match reality: `update()` returns the
|
||||
frame's count events, not `None`. Adding an algorithm must not require editing
|
||||
`live_count.py` or `counting_bench.py`.
|
||||
- **REQ-146** — Each algorithm **declares its own parameters** — name, type, default, range —
|
||||
and an endpoint serves that declaration, mirroring `live-count/models`. The frontend renders
|
||||
its controls from the declaration and hardcodes no per-algorithm parameter list. The start
|
||||
request carries `algorithm` plus an opaque `params` object validated against the
|
||||
declaration, replacing today's flat line-specific fields.
|
||||
- **REQ-147** — Geometry is generalised from a line to a **named shape set**. `line_cross`
|
||||
declares one horizontal segment; `possession` declares a bed polygon and an approach zone.
|
||||
The editor's drag channel (`move_line`) becomes shape-agnostic, so any algorithm's geometry
|
||||
is adjustable live without a new endpoint.
|
||||
- **REQ-148** — The **possession counter**: every sack track carries an `owner_id`, the person
|
||||
track it currently overlaps, or none when at rest. A count fires on an ownership change that
|
||||
crosses the bed boundary — person-outside to bed, or person-outside to person-inside.
|
||||
Ownership is sticky with hysteresis, so occlusion by the carrier's back and the unowned
|
||||
mid-air phase of a thrown sack do not break it. This requires a `person` class alongside
|
||||
`sack` from the detector.
|
||||
- **REQ-149** — Every count run records **which algorithm and parameter set** produced it, and
|
||||
accuracy is comparable per algorithm against the same ground truth. Switching algorithms
|
||||
adds results, it never invalidates stored ones — so `count_runs` is keyed by
|
||||
`(project, video, algorithm)`, not by video alone.
|
||||
|
||||
## F4. Counting accuracy bench
|
||||
|
||||
- **REQ-150** — A page lists every archive video as a row: date, batch, length, and the
|
||||
@@ -179,6 +264,23 @@ changes.
|
||||
which is what makes counting a 30-minute video practical. A run records the parameters and
|
||||
model it used.
|
||||
|
||||
- **REQ-154** — Ground truth can be **imported in bulk** from the operations sheet
|
||||
(`./GT.xlsx`, `DATA MUAT PAKAN PER LINE`). The camera watches **Line 1**; Line 2 is
|
||||
recorded for completeness but never scored. Each sheet is one working day; a row is one
|
||||
truck with a `BAG` count, a `DUS` count and a plate.
|
||||
- **REQ-155** — `BAG` (sacks) and `DUS` (boxes) are **separate commodities**, counted and
|
||||
scored separately. A box already resting in the truck bed is a legitimate object of a
|
||||
different class, not a detection fault.
|
||||
- **REQ-156** — An import never silently guesses. Recordings are aligned to sheet rows by
|
||||
start time against row order, the proposed pairing is **shown for human confirmation**
|
||||
before anything is written, and each imported value records that it came from the sheet
|
||||
rather than from a hand count. A recording that merged two trucks
|
||||
(`BATCH_MERGE_THRESHOLD_SECONDS`) is flagged, not paired.
|
||||
- **REQ-157** — Sheet values are **order quantities, not hand counts** — 67% of them are
|
||||
exactly 160 or 180 — so they score aggregate accuracy across many trucks and never
|
||||
adjudicate a single video. Per-event truth for algorithm comparison comes from a
|
||||
hand-counted clip, held separately.
|
||||
|
||||
## F5. Real recording times and working days
|
||||
|
||||
- **REQ-160** — Each recording's start time is read from the timestamp the camera burns into
|
||||
|
||||
+167
@@ -1040,6 +1040,173 @@ two will disagree.
|
||||
`2026-08-06/batch4`, `2026-08-06/batch9`, `2026-08-14/batch016` — likely truncated) and 11 were
|
||||
read with low confidence. Both are flagged amber in the table and accept a hand-typed time.
|
||||
|
||||
## Task — Ground truth import from the ops sheet (REQ-154…157)
|
||||
|
||||
1. Parse `docs/GT.xlsx` into rows → verify: 6 sheets (10–15 Aug 2026), Line 1 only, stopping
|
||||
at the first blank plate so the inline totals row is not read as a truck. Expected Line 1
|
||||
bag totals: 4780 / 4322 / 4365 / 5800 / 5645 / 9155. `[TODO]`
|
||||
2. `ground_truth_bag` / `ground_truth_dus` + `gt_source` on `count_runs` (REQ-155, REQ-156) →
|
||||
verify: migration runs on the live DB, existing hand-typed values survive as
|
||||
`gt_source='manual'`. `[TODO]`
|
||||
3. Alignment preview with human confirmation (REQ-156) → verify: a dry run on 14 Aug proposes
|
||||
26 recordings against 32 Line-1 trucks, flags the shortfall, and writes nothing until
|
||||
confirmed. `[TODO]`
|
||||
4. Bench scores bag and box separately (REQ-155) → verify: the accuracy row shows both, and
|
||||
totals only over rows that have a ground truth. `[TODO]`
|
||||
|
||||
## Task — Pluggable counting algorithms (REQ-145…149)
|
||||
|
||||
1. Fix the `Counter` protocol and register `line_cross` behind it (REQ-145) → verify: a live
|
||||
session on a known clip returns **the same counts as before** the refactor — this step
|
||||
changes no behaviour. `[TODO]`
|
||||
2. Parameter declaration endpoint + generic frontend controls (REQ-146) → verify: the
|
||||
live-count panel renders `line_cross`'s dials from the declaration alone, with no
|
||||
algorithm-specific code in the page. `[TODO]`
|
||||
3. Shape-agnostic geometry channel (REQ-147) → verify: dragging the line still works; a
|
||||
two-shape stub algorithm is adjustable through the same endpoint. `[TODO]`
|
||||
4. `count_runs` keyed by `(project, video, algorithm)` (REQ-149) → verify: the same video
|
||||
counted by two algorithms yields two rows and two accuracy figures. `[TODO]`
|
||||
5. The possession counter (REQ-148) → verify: on the hand-counted clip it beats `line_cross`
|
||||
on sacks that are occluded by the carrier and on sacks thrown in by the sender. **Blocked**
|
||||
until the detector emits a `person` class and one clip has per-event truth. `[TODO]`
|
||||
|
||||
## Task — Exemplar prompting in the auto-annotate modal (REQ-171, REQ-172) `[DONE]`
|
||||
|
||||
1. `Sam3Engine.detect_with_exemplars` — one `set_image`, prompts looped over it, boxes
|
||||
appended to one prompt only → verify: a negative box owned by `sack` sitting on a truck
|
||||
leaves the truck detections untouched, while the same box owned by `truck` suppresses
|
||||
them. `[DONE]` — on frame 86031 of batch 594: text-only `{truck: 5}`, owned-by-sack
|
||||
`{truck: 5}`, owned-by-truck `{}`. The `reset_all_prompts` before each prompt is what
|
||||
stops the leak; `state["geometric_prompt"]` survives `set_text_prompt` otherwise.
|
||||
2. `exemplars` + `exemplar_class_name` through `labeling.label_image` → `preview.py` →
|
||||
`POST /api/batches/{id}/preview` → verify: an unknown class name falls back to plain text
|
||||
rather than attaching the boxes to whichever class is first. `[DONE]` — 17 shapes for
|
||||
both text-only and `exemplar_class_name: "nonexistent"`.
|
||||
3. `preview_frame` moved out of `autolabel.py` into `preview.py` → verify: `autolabel.py` is
|
||||
back under the 400-line limit and the job path still imports. `[DONE]` — 261 and 151
|
||||
lines; container starts and registers the `autolabel` handler.
|
||||
4. Editable class prompt in the modal, saved to `project_classes.prompt` (REQ-171) →
|
||||
verify: a PATCH round-trips and the Projects page shows the new text. `[DONE]` — class 2
|
||||
`box → cardboard box → box` via the existing `PATCH /api/projects/{id}`; no new endpoint.
|
||||
5. `ExemplarCanvas.jsx` drag/shift-drag/undo/clear with 250 ms debounced re-run, and the
|
||||
modal split into `PreviewShapes.jsx` + `ClassPromptPanel.jsx` to stay under 400 lines →
|
||||
verify: `npm run build` clean, every file under the limit. `[DONE]` — 398 / 126 / 137 /
|
||||
64 lines, build green, both containers redeployed.
|
||||
|
||||
**Deliberately not built:** exemplars in the batch job. SAM3's geometric prompts pool
|
||||
features from the current image, so a box drawn on frame 1 asks about whatever sits at those
|
||||
coordinates on frame 400. The batch job stays text-only; the exemplars exist to find the text
|
||||
that works.
|
||||
|
||||
## Task — Exemplar-driven labeling in the review editor (REQ-173, REQ-174) `[DONE]`
|
||||
|
||||
1. `backend/exemplar.py` — pool → one SAM3 pass (class prompt + boxes) → rewrite that class
|
||||
on that frame → verify: on frame 55446 (batch 426, `sack`), one positive drawn from an
|
||||
existing box gives 52 class-0 shapes, exactly 1 of them `manual` with the drawn geometry,
|
||||
and the frame's class-1 shapes are untouched. `[DONE]` — verified; warm pass 0.4 s, first
|
||||
pass 7.6 s (model load).
|
||||
2. Negative exemplars delete what they cover (REQ-174) → verify: shift-drag over one of the
|
||||
detections and no `auto` shape overlapping it by ≥ 0.3 IoU comes back, while the drawn
|
||||
positive survives. `[DONE]` — max IoU with the negative afterwards 0.078, manual shape
|
||||
still present.
|
||||
3. GPU-busy fallback → verify: hold `jobs.gpu_lock`, drag, and the drawn shape is still
|
||||
stored with `redetected: false` and a legible message. `[DONE]` — "Saved your shape — the
|
||||
GPU is busy with a background job…", 58 shapes vs 57 before, no exception.
|
||||
4. `POST /api/frames/{id}/exemplar-label` + `AnnotationCanvas` drag/shift-drag with the pool
|
||||
drawn as dashed ghosts, 400 ms debounce, undo/clear, and the busy message under the canvas
|
||||
→ verify: `vite build` clean and every touched file under 400 lines. `[DONE]` — build
|
||||
green; `exemplar.py` 211, `api/review.py` 141, canvas 287, `useExemplarPool.js` 81. The
|
||||
pool logic went into that hook rather than into `ReviewPage.jsx`, which was already over
|
||||
the limit before this task (620 lines) and ends it at 628.
|
||||
|
||||
## Task — Filter panel and preview for exemplar runs (REQ-175) `[DONE]`
|
||||
|
||||
1. `exemplar.label(..., apply=False)` — dry run by default, returning `shapes` instead of
|
||||
writing → verify: two previews in a row leave the row count untouched. `[DONE]` — frame
|
||||
55446 stayed at 57 rows across a default preview (52 shapes) and a filtered one (20).
|
||||
2. The four filters, applied in the batch job's order (area floor → NMS → cap) → verify:
|
||||
each one visibly bites on a dense frame. `[DONE]` — from 52 shapes: NMS 0.05 → 32,
|
||||
min box 0.05 → 1, cap 5 → 5, confidence 0.9 → 15.
|
||||
3. `apply: true` writes exactly what was previewed → verify: the applied frame matches the
|
||||
preview count and leaves other classes alone. `[DONE]` — 20 previewed, 20 class-0 shapes
|
||||
stored (1 of them the drawn `manual` box), the frame's 2 class-1 shapes untouched.
|
||||
4. `ExemplarFilterPanel.jsx` floating in the canvas corner, sliders re-previewing on 250 ms,
|
||||
Apply/Discard/Undo/Reset, Enter and Esc bound → verify: `vite build` clean, files under
|
||||
the limit. `[DONE]` — panel 108, hook 125, canvas 314 lines; build green; both containers
|
||||
rebuilt and the live endpoint returns `applied: false` for a drag.
|
||||
|
||||
5. The class under review hides while its preview is up → verify: a negative exemplar's
|
||||
effect is visible instead of being masked by the stored box underneath it. `[DONE]` —
|
||||
frame 55446: 51 detections with one positive, 50 with a negative added; before this the
|
||||
removed box stayed on screen at 35% opacity and the run looked inert.
|
||||
|
||||
**Deliberately not built:** saving the filter values. They describe one frame's run, and the
|
||||
auto-annotate modal already owns the batch-wide numbers — sharing them would let a tweak made
|
||||
while reviewing one frame silently change what the next batch job does.
|
||||
|
||||
**Deliberately not built:** persisting the pool. It is a prompt about *this* image, so it
|
||||
dies with the frame, exactly as in REQ-172. What persists is the annotations it produced.
|
||||
|
||||
## Task 32 — WebRTC preview for the live counting page (REQ-176, REQ-177) `[DONE]`
|
||||
|
||||
The live view cost far more than it should: the backend re-encoded every annotated frame to
|
||||
JPEG and pushed it over MJPEG, on top of decoding the camera. The camera already reaches the
|
||||
browser cheaply over WebRTC, so the frames stop travelling through this app entirely.
|
||||
|
||||
1. A live source must be a WHEP URL; the RTSP leg is derived → verify: **[DONE]**
|
||||
`POST .../live-count/start` with `rtsp://192.168.192.96:8554/cam` →
|
||||
`400 "A live source must be a WebRTC (WHEP) URL…"`; with
|
||||
`http://192.168.192.96:8889/cam` → `200`, `source: "rtsp://192.168.192.96:8554/cam"`,
|
||||
`whep_url: "http://192.168.192.96:8889/cam/whep"`, `preview: "webrtc"`.
|
||||
2. The AI counts from that stream → verify: **[DONE]** 185 frames in 49 s off the live
|
||||
camera, `error: ""`. That rate is the link's, not the model's — see below.
|
||||
3. No JPEG is encoded for a WebRTC session → verify: **[DONE]** `GET /api/live-count/stream`
|
||||
downloaded 0 bytes during a running WebRTC session, and now answers `409`.
|
||||
4. The overlay feed carries what the model saw, and tracks the line live → verify:
|
||||
**[DONE]** `GET /api/live-count/overlay` returned 27 boxes with ids and confidences;
|
||||
after `PATCH /api/live-count/line {"line_y":300}` the feed reported `line.y: 300`.
|
||||
5. The 400-line limit holds → verify: **[DONE]** `live_count.py` was already 467 lines, so
|
||||
the transport layer went to `live_source.py` (120) and the MJPEG overlay to
|
||||
`live_render.py` (60), leaving it at 393. On the frontend the preview moved to
|
||||
`LiveVideoPanel.jsx` and the slider table to `liveCountFields.js`, leaving
|
||||
`LiveCountPage.jsx` at 383. `npm run build` passes.
|
||||
|
||||
**Not verified here:** the WHEP handshake in a real browser. The endpoint was confirmed live
|
||||
(`POST http://192.168.192.96:8889/cam/whep` answers, rejecting a deliberately malformed SDP
|
||||
with `400`), but the negotiation itself needs a browser, not curl.
|
||||
|
||||
### Where the live FPS actually goes — measured, 2026-08-19
|
||||
|
||||
The live session runs at 4-6 fps and it is not the model. Measured in the backend container
|
||||
against `rtsp://192.168.192.96:8554/cam`:
|
||||
|
||||
| Stage | Rate |
|
||||
|---|---|
|
||||
| ByteTrack + YOLO inference | **205 fps** |
|
||||
| `cv2.resize` to 1280x720 | 5348 fps |
|
||||
| Decode from RTSP | **6.4 fps** |
|
||||
|
||||
The camera is 704x576 HEVC at 350 kbit/s — nothing about it is expensive. The link is: the
|
||||
route to the streaming server is a ZeroTier VPN measuring **15% packet loss** and a 41-104 ms
|
||||
round trip. The comment in `live_source.py` claiming the cost was "decoding 1080p on the CPU"
|
||||
was simply wrong and has been corrected; so has the hint on the page.
|
||||
|
||||
Transport was changed to UDP and changed back, because the measurement contradicts the
|
||||
theory. Through the **ffmpeg CLI**, UDP wins as expected — 16 fps at 1.00x realtime against
|
||||
TCP's 6.8 fps at 0.52x. Through **OpenCV** it loses: tcp 6.4 fps, udp+socket buffer 4.4, bare
|
||||
udp 2.4, and a live session on UDP showed 18-second stalls waiting for a keyframe. OpenCV
|
||||
drops what it cannot reassemble instead of showing it, so the loss lands as missing frames.
|
||||
`RTSP_TRANSPORT` is left as an env override, defaulting to `tcp`.
|
||||
|
||||
**Not fixable in this repo.** Inference has ~50x the headroom the link delivers, so nothing
|
||||
in the app is worth optimising. The lever is where the counter runs: next to MediaMTX it
|
||||
would count at the camera's full rate. Worth checking whether the ZeroTier path is relayed
|
||||
rather than direct (`zerotier-cli peers` — a `RELAY` row explains both the loss and the RTT).
|
||||
|
||||
**Deliberately not built:** an aiortc/WHEP client in the backend. It would be "WebRTC only"
|
||||
end to end, but the decode cost is identical to RTSP and it adds ICE and keyframe-loss
|
||||
failure modes to the counting path. The saving was always on the browser side.
|
||||
|
||||
## Known open points
|
||||
|
||||
- *Not closed by any task, by choice:* **any rebuild kills the running job.** Task 14's resume
|
||||
|
||||
+1550
File diff suppressed because it is too large.
Load diff
@@ -117,6 +117,9 @@ export const api = {
|
||||
body: { annotation_ids: annotationIds, class_id: classId },
|
||||
}),
|
||||
assist: (frameId, body) => request(`/frames/${frameId}/assist`, { method: 'POST', body }),
|
||||
// The whole frame-local exemplar pool, re-sent on every drag (REQ-173/174).
|
||||
exemplarLabel: (frameId, body) =>
|
||||
request(`/frames/${frameId}/exemplar-label`, { method: 'POST', body }),
|
||||
setFrameStatus: (frameId, status) =>
|
||||
request(`/frames/${frameId}/status`, { method: 'POST', body: { status } }),
|
||||
|
||||
@@ -150,6 +153,8 @@ export const api = {
|
||||
liveCountStop: () => request('/live-count/stop', { method: 'POST' }),
|
||||
liveCountMoveLine: (body) => request('/live-count/line', { method: 'PATCH', body }),
|
||||
liveCountStatus: () => request('/live-count/status'),
|
||||
// Polled far faster than status: this is what the WebRTC preview draws.
|
||||
liveCountOverlay: () => request('/live-count/overlay'),
|
||||
// `key` busts the browser cache so a restarted session gets a fresh connection.
|
||||
liveCountStreamUrl: (key = 0) => `/api/live-count/stream?k=${key}`,
|
||||
|
||||
|
||||
+63
-5
@@ -437,7 +437,7 @@ main.page {
|
||||
user-select: none;
|
||||
}
|
||||
|
||||
.canvas-wrap svg {
|
||||
.canvas-wrap > svg {
|
||||
position: absolute;
|
||||
top: 0;
|
||||
left: 0;
|
||||
@@ -447,7 +447,7 @@ main.page {
|
||||
touch-action: none;
|
||||
}
|
||||
|
||||
.canvas-wrap svg.assist { cursor: copy; }
|
||||
.canvas-wrap > svg.assist { cursor: copy; }
|
||||
|
||||
.canvas-wrap .shape rect,
|
||||
.canvas-wrap .shape polygon {
|
||||
@@ -478,9 +478,9 @@ main.page {
|
||||
|
||||
/* Select mode: the cursor promises a marquee, and a shape is a target to tick
|
||||
rather than something to drag (REQ-045a). */
|
||||
.canvas-wrap svg.selecting { cursor: cell; }
|
||||
.canvas-wrap svg.selecting .shape rect,
|
||||
.canvas-wrap svg.selecting .shape polygon { cursor: pointer; }
|
||||
.canvas-wrap > svg.selecting { cursor: cell; }
|
||||
.canvas-wrap > svg.selecting .shape rect,
|
||||
.canvas-wrap > svg.selecting .shape polygon { cursor: pointer; }
|
||||
|
||||
.canvas-wrap .shape.marked rect,
|
||||
.canvas-wrap .shape.marked polygon {
|
||||
@@ -532,6 +532,22 @@ main.page {
|
||||
.canvas-wrap .handle-ne, .canvas-wrap .handle-sw { cursor: nesw-resize; }
|
||||
.canvas-wrap .handle-vertex, .canvas-wrap .handle-midpoint { cursor: pointer; }
|
||||
|
||||
/* The frame-local exemplar pool (REQ-173/174): a prompt, not a shape, so it is
|
||||
drawn behind everything, dashed, and never a pointer target. */
|
||||
.canvas-wrap .exemplar {
|
||||
fill: none;
|
||||
stroke-width: 1.5;
|
||||
stroke-dasharray: 2 4;
|
||||
opacity: 0.85;
|
||||
pointer-events: none;
|
||||
vector-effect: non-scaling-stroke;
|
||||
}
|
||||
|
||||
.canvas-wrap .exemplar.negative {
|
||||
fill: rgba(248, 113, 113, 0.1);
|
||||
stroke-dasharray: 6 3;
|
||||
}
|
||||
|
||||
.canvas-wrap .draft {
|
||||
fill: rgba(255, 255, 255, 0.08);
|
||||
stroke-width: 2;
|
||||
@@ -724,3 +740,45 @@ kbd {
|
||||
.topbar .crumbs { display: none; }
|
||||
main.page { padding: 16px 12px 40px; }
|
||||
}
|
||||
|
||||
/* The exemplar filter popup (REQ-175). It sits at the top of the sidebar, not
|
||||
over the frame: the proposals it governs are drawn on the canvas, and a card
|
||||
floating on top of them hid the thing being judged. The accent border is what
|
||||
marks it as a live decision rather than another standing panel. */
|
||||
.exemplar-panel {
|
||||
padding: 10px 12px;
|
||||
border-radius: var(--radius);
|
||||
border: 1px solid rgba(56, 189, 248, 0.45);
|
||||
background: rgba(17, 24, 39, 0.92);
|
||||
box-shadow: 0 4px 16px rgba(0, 0, 0, 0.35);
|
||||
line-height: 1.45;
|
||||
}
|
||||
|
||||
.exemplar-panel input[type="range"] {
|
||||
accent-color: var(--accent);
|
||||
display: block;
|
||||
cursor: pointer;
|
||||
}
|
||||
.exemplar-panel .btn { cursor: pointer; transition: background 160ms ease, border-color 160ms ease; }
|
||||
|
||||
/* Preview shapes are proposals, not labels: dashed and unclickable. The class
|
||||
under review is hidden while they show (see AnnotationCanvas); what stays
|
||||
visible belongs to other classes, dimmed so the two never read as one set. */
|
||||
.canvas-wrap > svg.previewing .shape { opacity: 0.4; }
|
||||
|
||||
.canvas-wrap .preview-shape {
|
||||
fill: rgba(56, 189, 248, 0.1);
|
||||
stroke-width: 2;
|
||||
stroke-dasharray: 4 3;
|
||||
pointer-events: none;
|
||||
vector-effect: non-scaling-stroke;
|
||||
}
|
||||
|
||||
.canvas-wrap .preview-shape.drawn {
|
||||
fill: rgba(52, 211, 153, 0.16);
|
||||
stroke-dasharray: none;
|
||||
}
|
||||
|
||||
@media (prefers-reduced-motion: reduce) {
|
||||
.exemplar-panel .btn { transition: none; }
|
||||
}
|
||||
@@ -39,7 +39,8 @@ function overlaps(geometry, [mx0, my0, mx1, my1]) {
|
||||
|
||||
export default function AnnotationCanvas({
|
||||
frame, imageUrl, annotations, selectedId, activeClass, assistMode, classes,
|
||||
mode = 'draw', selectedIds, onSelect, onCreate, onUpdate, onAssist, onMarquee,
|
||||
mode = 'draw', selectedIds, exemplars = [], preview = null,
|
||||
onSelect, onCreate, onUpdate, onAssist, onExemplar, onMarquee,
|
||||
}) {
|
||||
const wrapRef = useRef(null)
|
||||
const svgRef = useRef(null)
|
||||
@@ -173,7 +174,10 @@ export default function AnnotationCanvas({
|
||||
return
|
||||
}
|
||||
if (x1 - x0 >= MIN_SIZE && y1 - y0 >= MIN_SIZE) {
|
||||
// Draw mode is exemplar mode (REQ-173): the drag is both the label and
|
||||
// the visual prompt, and Shift makes it a negative one (REQ-174).
|
||||
if (assistMode) onAssist([x0, y0, x1, y1])
|
||||
else if (onExemplar) onExemplar([x0, y0, x1, y1], !additive.current)
|
||||
else onCreate({ type: 'bbox', points: [x0, y0, x1, y1] })
|
||||
}
|
||||
return
|
||||
@@ -209,30 +213,75 @@ export default function AnnotationCanvas({
|
||||
ref={svgRef}
|
||||
viewBox={`0 0 ${width} ${height}`}
|
||||
preserveAspectRatio="none"
|
||||
className={[assistMode ? 'assist' : '', selecting ? 'selecting' : ''].filter(Boolean).join(' ') || undefined}
|
||||
className={[assistMode ? 'assist' : '', selecting ? 'selecting' : '',
|
||||
preview ? 'previewing' : ''].filter(Boolean).join(' ') || undefined}
|
||||
onPointerDown={startDraw}
|
||||
onPointerMove={onPointerMove}
|
||||
onPointerUp={onPointerUp}
|
||||
>
|
||||
{annotations.map((annotation) => (
|
||||
<Shape
|
||||
key={annotation.id}
|
||||
annotation={annotation}
|
||||
width={width}
|
||||
height={height}
|
||||
scale={scale}
|
||||
handle={handle}
|
||||
selected={annotation.id === selectedId}
|
||||
marked={marked.has(annotation.id)}
|
||||
readOnly={selecting}
|
||||
classes={classes}
|
||||
onStartMove={startMove}
|
||||
onStartResize={startResize}
|
||||
onStartVertex={startVertex}
|
||||
onStartMidpoint={startMidpoint}
|
||||
onDeleteVertex={deleteVertex}
|
||||
/>
|
||||
))}
|
||||
{/* A preview replaces this class on the frame wholesale, so its stored
|
||||
shapes step aside for the proposals — left on screen they read as
|
||||
part of the result and a rejected box looks like it never went.
|
||||
Other classes are untouched by the run and stay as they are. */}
|
||||
{annotations
|
||||
.filter((annotation) => !(preview && annotation.class_id === activeClass))
|
||||
.map((annotation) => (
|
||||
<Shape
|
||||
key={annotation.id}
|
||||
annotation={annotation}
|
||||
width={width}
|
||||
height={height}
|
||||
scale={scale}
|
||||
handle={handle}
|
||||
selected={annotation.id === selectedId}
|
||||
marked={marked.has(annotation.id)}
|
||||
readOnly={selecting}
|
||||
classes={classes}
|
||||
onStartMove={startMove}
|
||||
onStartResize={startResize}
|
||||
onStartVertex={startVertex}
|
||||
onStartMidpoint={startMidpoint}
|
||||
onDeleteVertex={deleteVertex}
|
||||
/>
|
||||
))}
|
||||
|
||||
{preview?.map((shape, i) => {
|
||||
const drawn = shape.source === 'manual'
|
||||
const stroke = drawn ? '#34d399' : classColor(activeClass)
|
||||
if (shape.geometry.type === 'bbox') {
|
||||
const [x0, y0, x1, y1] = shape.geometry.points
|
||||
return (
|
||||
<rect
|
||||
key={`preview-${i}`}
|
||||
className={`preview-shape${drawn ? ' drawn' : ''}`}
|
||||
x={x0 * width} y={y0 * height}
|
||||
width={(x1 - x0) * width} height={(y1 - y0) * height}
|
||||
stroke={stroke}
|
||||
/>
|
||||
)
|
||||
}
|
||||
return (
|
||||
<polygon
|
||||
key={`preview-${i}`}
|
||||
className={`preview-shape${drawn ? ' drawn' : ''}`}
|
||||
points={shape.geometry.points.map(([x, y]) => `${x * width},${y * height}`).join(' ')}
|
||||
stroke={stroke}
|
||||
/>
|
||||
)
|
||||
})}
|
||||
|
||||
{!selecting && exemplars.map((item, i) => {
|
||||
const [x0, y0, x1, y1] = item.box
|
||||
return (
|
||||
<rect
|
||||
key={`exemplar-${i}`}
|
||||
className={`exemplar${item.positive ? '' : ' negative'}`}
|
||||
x={x0 * width} y={y0 * height}
|
||||
width={(x1 - x0) * width} height={(y1 - y0) * height}
|
||||
stroke={item.positive ? classColor(activeClass) : '#f87171'}
|
||||
/>
|
||||
)
|
||||
})}
|
||||
|
||||
{draft && (() => {
|
||||
const [x0, y0, x1, y1] = normalise(draft)
|
||||
@@ -241,7 +290,10 @@ export default function AnnotationCanvas({
|
||||
className={selecting ? 'draft marquee' : assistMode ? 'draft assist' : 'draft'}
|
||||
x={x0 * width} y={y0 * height}
|
||||
width={(x1 - x0) * width} height={(y1 - y0) * height}
|
||||
stroke={selecting ? '#38bdf8' : assistMode ? 'var(--accent)' : classColor(activeClass)}
|
||||
stroke={selecting ? '#38bdf8'
|
||||
: assistMode ? 'var(--accent)'
|
||||
: additive.current ? '#f87171'
|
||||
: classColor(activeClass)}
|
||||
/>
|
||||
)
|
||||
})()}
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
import React, { useState, useEffect, useRef } from 'react'
|
||||
import { api } from '../api'
|
||||
import ExemplarCanvas from './ExemplarCanvas'
|
||||
import { PreviewShapes } from './PreviewShapes'
|
||||
import ClassPromptPanel from './ClassPromptPanel'
|
||||
|
||||
function useDebounce(value, delay) {
|
||||
const [debouncedValue, setDebouncedValue] = useState(value)
|
||||
@@ -14,69 +17,6 @@ function useDebounce(value, delay) {
|
||||
return debouncedValue
|
||||
}
|
||||
|
||||
// Shared with MassAutoAnnotateModal: draws detection boxes/polygons over a
|
||||
// frame in a 0..10000 viewBox.
|
||||
export function PreviewShapes({ shapes, project }) {
|
||||
return shapes.map((shape, i) => {
|
||||
if (!shape.geometry || !shape.geometry.points) return null
|
||||
const classObj = project.classes.find(c => c.class_id === shape.class_id)
|
||||
const className = classObj?.name || 'Unknown'
|
||||
const color = ['#38bdf8', '#34d399', '#f472b6', '#a78bfa', '#fbbf24'][shape.class_id % 5] || '#fff'
|
||||
|
||||
let minX = 1, minY = 1, maxX = 0, maxY = 0
|
||||
if (shape.geometry.type === 'bbox') {
|
||||
const [left, top, right, bottom] = shape.geometry.points
|
||||
minX = left; minY = top; maxX = right; maxY = bottom;
|
||||
} else {
|
||||
shape.geometry.points.forEach(pt => {
|
||||
if (pt[0] < minX) minX = pt[0]
|
||||
if (pt[1] < minY) minY = pt[1]
|
||||
if (pt[0] > maxX) maxX = pt[0]
|
||||
if (pt[1] > maxY) maxY = pt[1]
|
||||
})
|
||||
}
|
||||
|
||||
const x0 = minX * 10000
|
||||
const y0 = minY * 10000
|
||||
const bw = (maxX - minX) * 10000
|
||||
const bh = (maxY - minY) * 10000
|
||||
|
||||
return (
|
||||
<g key={i}>
|
||||
{shape.geometry.type === 'polygon' && (
|
||||
<polygon
|
||||
points={shape.geometry.points.map(pt => `${pt[0] * 10000},${pt[1] * 10000}`).join(' ')}
|
||||
fill={color}
|
||||
fillOpacity={0.35}
|
||||
stroke={color}
|
||||
strokeWidth="10"
|
||||
/>
|
||||
)}
|
||||
<rect
|
||||
x={x0}
|
||||
y={y0}
|
||||
width={bw}
|
||||
height={bh}
|
||||
fill="none"
|
||||
stroke={color}
|
||||
strokeWidth="20"
|
||||
strokeDasharray="40 20"
|
||||
/>
|
||||
<text
|
||||
x={x0}
|
||||
y={y0 > 300 ? y0 - 100 : y0 + 300}
|
||||
fill={color}
|
||||
fontSize="240"
|
||||
fontWeight="bold"
|
||||
style={{ textShadow: '10px 10px 10px #000, -10px -10px 10px #000, 10px -10px 10px #000, -10px 10px 10px #000' }}
|
||||
>
|
||||
{className} {shape.score ? `${(shape.score * 100).toFixed(1)}%` : ''}
|
||||
</text>
|
||||
</g>
|
||||
)
|
||||
})
|
||||
}
|
||||
|
||||
export default function AutoAnnotateModal({
|
||||
batch,
|
||||
project,
|
||||
@@ -103,11 +43,30 @@ export default function AutoAnnotateModal({
|
||||
return []
|
||||
})
|
||||
|
||||
// SAM3 prompt tuning (REQ-171): the class prompt is edited here, next to the
|
||||
// live preview, and saved back onto the project so the batch job uses it.
|
||||
const [classPrompts, setClassPrompts] = useState(() =>
|
||||
Object.fromEntries(project.classes.map(c => [c.class_id, c.prompt || c.name]))
|
||||
)
|
||||
const [promptDraft, setPromptDraft] = useState('')
|
||||
const [promptSaving, setPromptSaving] = useState(false)
|
||||
const [promptError, setPromptError] = useState('')
|
||||
|
||||
// Exemplars (REQ-172): tuning aid for the previewed frame only. They belong to
|
||||
// the active class and are never written or carried into the batch job.
|
||||
const [activeClassName, setActiveClassName] = useState(null)
|
||||
const [exemplars, setExemplars] = useState([])
|
||||
// Bumped on every add/undo/clear. Debouncing the count rather than the array
|
||||
// is what lets "clear all" re-run too — an empty array on its own is
|
||||
// indistinguishable from the initial state.
|
||||
const [exemplarRev, setExemplarRev] = useState(0)
|
||||
|
||||
// Preview state
|
||||
const [frames, setFrames] = useState([])
|
||||
const [frameIndex, setFrameIndex] = useState(0)
|
||||
const [previewShapes, setPreviewShapes] = useState([])
|
||||
const [isLoadingPreview, setIsLoadingPreview] = useState(false)
|
||||
const [previewError, setPreviewError] = useState('')
|
||||
const [isSubmitting, setIsSubmitting] = useState(false)
|
||||
|
||||
// Fetch frames on mount
|
||||
@@ -130,6 +89,7 @@ export default function AutoAnnotateModal({
|
||||
|
||||
let isMounted = true
|
||||
setIsLoadingPreview(true)
|
||||
setPreviewError('')
|
||||
api.preview(batch.id, {
|
||||
frame_id: frame.id,
|
||||
engine,
|
||||
@@ -137,24 +97,93 @@ export default function AutoAnnotateModal({
|
||||
iou_threshold: iouThreshold,
|
||||
min_box_frac: minBoxFrac,
|
||||
target_class_names: selectedClasses,
|
||||
custom_model_path: customModelStagedPath
|
||||
custom_model_path: customModelStagedPath,
|
||||
exemplars: engine === 'sam3' ? exemplars : [],
|
||||
exemplar_class_name: engine === 'sam3' ? activeClassName : null
|
||||
}).then(res => {
|
||||
if (isMounted && res.shapes) {
|
||||
setPreviewShapes(res.shapes)
|
||||
}
|
||||
}).catch(err => {
|
||||
console.error("Preview failed:", err)
|
||||
// The GPU lock (REQ-065) answers 409 here while a batch job holds it, and
|
||||
// that is the one failure the user can act on — so it goes on screen.
|
||||
setPreviewError(err.message || 'Preview failed')
|
||||
setPreviewShapes([])
|
||||
}).finally(() => {
|
||||
setIsLoadingPreview(false)
|
||||
})
|
||||
}
|
||||
|
||||
const activeClass = project.classes.find(c => c.name === activeClassName) || null
|
||||
const sam3Tuning = engine === 'sam3'
|
||||
|
||||
// Clear shapes when frame changes
|
||||
useEffect(() => {
|
||||
setPreviewShapes([])
|
||||
}, [frameIndex])
|
||||
|
||||
// The active class owns the exemplars, so keep it pointing at something the
|
||||
// user actually ticked.
|
||||
useEffect(() => {
|
||||
if (!sam3Tuning) return
|
||||
if (!activeClassName || !selectedClasses.includes(activeClassName)) {
|
||||
setActiveClassName(selectedClasses[0] || null)
|
||||
}
|
||||
}, [sam3Tuning, selectedClasses, activeClassName])
|
||||
|
||||
// Exemplars are pooled from one image and belong to one class, so both a new
|
||||
// frame and a new class invalidate them (REQ-172).
|
||||
useEffect(() => {
|
||||
setExemplars([])
|
||||
setExemplarRev(0)
|
||||
}, [frameIndex, activeClassName])
|
||||
|
||||
useEffect(() => {
|
||||
if (activeClass) {
|
||||
setPromptDraft(classPrompts[activeClass.class_id] ?? activeClass.name)
|
||||
setPromptError('')
|
||||
}
|
||||
}, [activeClassName])
|
||||
|
||||
// Auto re-run: drawing a box is the question, the redrawn preview is the
|
||||
// answer, so waiting for a button press in between defeats the point.
|
||||
// set_image is already cached for this frame, so only the grounding head runs.
|
||||
const debouncedRev = useDebounce(exemplarRev, 250)
|
||||
useEffect(() => {
|
||||
if (debouncedRev > 0) handlePreview()
|
||||
}, [debouncedRev])
|
||||
|
||||
const addExemplar = (exemplar) => {
|
||||
setExemplars(prev => [...prev, exemplar])
|
||||
setExemplarRev(rev => rev + 1)
|
||||
}
|
||||
|
||||
const undoExemplar = () => {
|
||||
setExemplars(prev => prev.slice(0, -1))
|
||||
setExemplarRev(rev => rev + 1)
|
||||
}
|
||||
|
||||
const clearExemplars = () => {
|
||||
setExemplars([])
|
||||
setExemplarRev(rev => rev + 1)
|
||||
}
|
||||
|
||||
const savePrompt = async () => {
|
||||
if (!activeClass) return
|
||||
const text = promptDraft.trim()
|
||||
if (!text || text === classPrompts[activeClass.class_id]) return
|
||||
setPromptSaving(true)
|
||||
setPromptError('')
|
||||
try {
|
||||
await api.patchProject(project.id, { prompts: { [activeClass.class_id]: text } })
|
||||
setClassPrompts({ ...classPrompts, [activeClass.class_id]: text })
|
||||
} catch (err) {
|
||||
setPromptError(err.message || 'Could not save the prompt')
|
||||
} finally {
|
||||
setPromptSaving(false)
|
||||
}
|
||||
}
|
||||
|
||||
const handleStart = async () => {
|
||||
setIsSubmitting(true)
|
||||
try {
|
||||
@@ -195,31 +224,30 @@ export default function AutoAnnotateModal({
|
||||
|
||||
<div style={{ position: 'relative', width: '100%', background: '#000', borderRadius: 6, overflow: 'hidden', display: 'flex', justifyContent: 'center', alignItems: 'center' }}>
|
||||
{currentFrame ? (
|
||||
<div className="relative inline-block" style={{ width: '100%', textAlign: 'center' }}>
|
||||
/* This wrapper must hug the image and nothing else: both overlays
|
||||
* are sized to it, so anything else inside would shift every box
|
||||
* off the pixels it describes. */
|
||||
<div style={{ position: 'relative', display: 'inline-block', lineHeight: 0, maxWidth: '100%' }}>
|
||||
<img
|
||||
src={api.frameUrl(currentFrame.id)}
|
||||
alt="Preview Frame"
|
||||
style={{ maxWidth: '100%', maxHeight: '42vh', objectFit: 'contain', display: 'block', margin: '0 auto' }}
|
||||
style={{ maxWidth: '100%', maxHeight: '42vh', display: 'block' }}
|
||||
/>
|
||||
<svg viewBox="0 0 10000 10000" preserveAspectRatio="none" style={{ position: 'absolute', top: 0, left: 0, width: '100%', height: '100%', pointerEvents: 'none' }}>
|
||||
<PreviewShapes shapes={previewShapes} project={project} />
|
||||
</svg>
|
||||
{sam3Tuning && activeClass && (
|
||||
<ExemplarCanvas
|
||||
exemplars={exemplars}
|
||||
onAdd={addExemplar}
|
||||
disabled={isLoadingPreview}
|
||||
/>
|
||||
)}
|
||||
{isLoadingPreview && (
|
||||
<div style={{ position: 'absolute', top: 8, right: 8, background: 'rgba(0,0,0,0.65)', padding: '4px 8px', borderRadius: 4, color: '#38bdf8', fontSize: '0.8rem', backdropFilter: 'blur(4px)' }}>
|
||||
Inferring...
|
||||
</div>
|
||||
)}
|
||||
<div className="flex justify-between items-center" style={{ padding: '8px 12px' }}>
|
||||
<button
|
||||
type="button"
|
||||
onClick={handlePreview}
|
||||
disabled={isLoadingPreview || frames.length === 0}
|
||||
className="px-4 py-1.5 bg-indigo-50 text-indigo-700 font-medium rounded-md hover:bg-indigo-100 disabled:opacity-50"
|
||||
style={{ fontSize: '0.85rem' }}
|
||||
>
|
||||
Run Preview
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
) : (
|
||||
<div style={{ display: 'flex', alignItems: 'center', justifyContent: 'center', minHeight: 200, color: '#71717a' }}>
|
||||
@@ -227,6 +255,50 @@ export default function AutoAnnotateModal({
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{previewError && (
|
||||
<div style={{ marginTop: 8, padding: '6px 10px', borderRadius: 4, fontSize: '0.8rem', background: 'rgba(244, 63, 94, 0.12)', border: '1px solid rgba(244, 63, 94, 0.4)', color: '#fda4af' }}>
|
||||
{previewError}
|
||||
</div>
|
||||
)}
|
||||
|
||||
<div className="row" style={{ gap: 8, marginTop: 10, flexWrap: 'wrap' }}>
|
||||
<button
|
||||
type="button"
|
||||
className="btn"
|
||||
onClick={handlePreview}
|
||||
disabled={isLoadingPreview || frames.length === 0}
|
||||
style={{ fontSize: '0.82rem' }}
|
||||
>
|
||||
Run Preview
|
||||
</button>
|
||||
{sam3Tuning && activeClass && (
|
||||
<>
|
||||
<button
|
||||
type="button"
|
||||
className="btn btn-ghost"
|
||||
onClick={undoExemplar}
|
||||
disabled={exemplars.length === 0}
|
||||
style={{ fontSize: '0.82rem' }}
|
||||
>
|
||||
Undo box
|
||||
</button>
|
||||
<button
|
||||
type="button"
|
||||
className="btn btn-ghost"
|
||||
onClick={clearExemplars}
|
||||
disabled={exemplars.length === 0}
|
||||
style={{ fontSize: '0.82rem' }}
|
||||
>
|
||||
Clear boxes
|
||||
</button>
|
||||
<span className="hint" style={{ fontSize: '0.76rem', marginLeft: 'auto' }}>
|
||||
Drag = example of <strong style={{ color: '#34d399' }}>{activeClass.name}</strong>
|
||||
{' · '}Shift-drag = not this
|
||||
</span>
|
||||
</>
|
||||
)}
|
||||
</div>
|
||||
|
||||
<div style={{ marginTop: 10 }}>
|
||||
<div className="row" style={{ justifyContent: 'space-between', marginBottom: 4 }}>
|
||||
@@ -291,30 +363,21 @@ export default function AutoAnnotateModal({
|
||||
/>
|
||||
</div>
|
||||
|
||||
<div style={{ flex: 1, minHeight: 60, overflowY: 'auto' }}>
|
||||
<span className="hint" style={{ fontSize: '0.8rem' }}>Target Classes:</span>
|
||||
<div style={{ display: 'flex', flexWrap: 'wrap', gap: 5, marginTop: 6 }}>
|
||||
{project.classes.map(c => {
|
||||
const isSelected = selectedClasses.includes(c.name)
|
||||
return (
|
||||
<button
|
||||
key={c.name}
|
||||
className={`class-chip ${isSelected ? 'active' : ''}`}
|
||||
style={{ fontSize: '0.75rem', padding: '2px 8px', border: isSelected ? '1px solid #c084fc' : '1px solid #3f3f46', cursor: 'pointer' }}
|
||||
onClick={() => {
|
||||
if (isSelected) {
|
||||
setSelectedClasses(selectedClasses.filter(n => n !== c.name))
|
||||
} else {
|
||||
setSelectedClasses([...selectedClasses, c.name])
|
||||
}
|
||||
}}
|
||||
>
|
||||
{isSelected ? '✓ ' : ''}{c.name}
|
||||
</button>
|
||||
)
|
||||
})}
|
||||
</div>
|
||||
</div>
|
||||
<ClassPromptPanel
|
||||
classes={project.classes}
|
||||
selectedClasses={selectedClasses}
|
||||
onSelectedChange={setSelectedClasses}
|
||||
sam3Tuning={sam3Tuning}
|
||||
activeClassName={activeClassName}
|
||||
onActivate={setActiveClassName}
|
||||
activeClass={activeClass}
|
||||
promptDraft={promptDraft}
|
||||
onPromptDraftChange={setPromptDraft}
|
||||
onPromptSave={savePrompt}
|
||||
promptSaving={promptSaving}
|
||||
promptError={promptError}
|
||||
savedPrompt={activeClass ? classPrompts[activeClass.class_id] : ''}
|
||||
/>
|
||||
|
||||
<div className="row" style={{ justifyContent: 'flex-end', gap: 10, marginTop: 16, paddingTop: 10, borderTop: '1px solid rgba(255,255,255,0.08)' }}>
|
||||
<button className="btn btn-ghost" onClick={onClose} disabled={isSubmitting}>Cancel</button>
|
||||
|
||||
@@ -0,0 +1,137 @@
|
||||
import React from 'react'
|
||||
|
||||
/* Target-class picker for the auto-annotate modal, plus the SAM3 prompt editor
|
||||
* (REQ-171).
|
||||
*
|
||||
* The prompt lives on the project class, not on the job, so what is tuned here
|
||||
* against the live preview is exactly what a batch run will send. Editing it is
|
||||
* a real write to `project_classes.prompt` — the same field the Projects page
|
||||
* edits, just reachable from where the feedback is.
|
||||
*
|
||||
* With SAM3 a chip carries a second meaning: one selected chip is *active*, and
|
||||
* it owns the prompt shown below and any exemplars drawn on the frame. Because
|
||||
* a click then has to mean "activate" rather than "deselect", each chip is a
|
||||
* pair of buttons — the name activates, the × deselects.
|
||||
*/
|
||||
export default function ClassPromptPanel({
|
||||
classes,
|
||||
selectedClasses,
|
||||
onSelectedChange,
|
||||
sam3Tuning,
|
||||
activeClassName,
|
||||
onActivate,
|
||||
activeClass,
|
||||
promptDraft,
|
||||
onPromptDraftChange,
|
||||
onPromptSave,
|
||||
promptSaving,
|
||||
promptError,
|
||||
savedPrompt,
|
||||
}) {
|
||||
const toggle = (name) => {
|
||||
if (selectedClasses.includes(name)) {
|
||||
onSelectedChange(selectedClasses.filter(n => n !== name))
|
||||
} else {
|
||||
onSelectedChange([...selectedClasses, name])
|
||||
}
|
||||
}
|
||||
|
||||
const dirty = activeClass && promptDraft.trim() && promptDraft.trim() !== savedPrompt
|
||||
|
||||
return (
|
||||
<div style={{ flex: 1, minHeight: 60, overflowY: 'auto' }}>
|
||||
<span className="hint" style={{ fontSize: '0.8rem' }}>Target Classes:</span>
|
||||
<div style={{ display: 'flex', flexWrap: 'wrap', gap: 5, marginTop: 6 }}>
|
||||
{classes.map(c => {
|
||||
const isSelected = selectedClasses.includes(c.name)
|
||||
const isActive = sam3Tuning && isSelected && c.name === activeClassName
|
||||
|
||||
if (!sam3Tuning) {
|
||||
return (
|
||||
<button
|
||||
key={c.name}
|
||||
className={`class-chip ${isSelected ? 'active' : ''}`}
|
||||
style={{ fontSize: '0.75rem', padding: '2px 8px', border: isSelected ? '1px solid #c084fc' : '1px solid #3f3f46' }}
|
||||
onClick={() => toggle(c.name)}
|
||||
>
|
||||
{isSelected ? '✓ ' : ''}{c.name}
|
||||
</button>
|
||||
)
|
||||
}
|
||||
|
||||
return (
|
||||
<span
|
||||
key={c.name}
|
||||
style={{
|
||||
display: 'inline-flex', alignItems: 'stretch', borderRadius: 4,
|
||||
overflow: 'hidden',
|
||||
border: isActive ? '1px solid #34d399'
|
||||
: isSelected ? '1px solid #c084fc' : '1px solid #3f3f46',
|
||||
boxShadow: isActive ? '0 0 0 1px rgba(52, 211, 153, 0.45)' : 'none',
|
||||
transition: 'border-color 180ms ease, box-shadow 180ms ease',
|
||||
}}
|
||||
>
|
||||
<button
|
||||
className={`class-chip ${isSelected ? 'active' : ''}`}
|
||||
style={{ fontSize: '0.75rem', padding: '2px 8px', border: 'none', borderRadius: 0 }}
|
||||
aria-pressed={isActive}
|
||||
title={isSelected ? `Tune "${c.name}"` : `Include "${c.name}"`}
|
||||
onClick={() => {
|
||||
if (!isSelected) onSelectedChange([...selectedClasses, c.name])
|
||||
onActivate(c.name)
|
||||
}}
|
||||
>
|
||||
{isSelected ? '✓ ' : ''}{c.name}
|
||||
</button>
|
||||
{isSelected && (
|
||||
<button
|
||||
className="class-chip"
|
||||
style={{ fontSize: '0.75rem', padding: '2px 6px', border: 'none', borderLeft: '1px solid #3f3f46', borderRadius: 0, color: '#a1a1aa' }}
|
||||
title={`Exclude "${c.name}" from this run`}
|
||||
aria-label={`Exclude ${c.name}`}
|
||||
onClick={() => toggle(c.name)}
|
||||
>
|
||||
×
|
||||
</button>
|
||||
)}
|
||||
</span>
|
||||
)
|
||||
})}
|
||||
</div>
|
||||
|
||||
{sam3Tuning && activeClass && (
|
||||
<div style={{ marginTop: 12 }}>
|
||||
<label htmlFor="sam3-prompt" className="hint" style={{ fontSize: '0.8rem' }}>
|
||||
SAM3 prompt for <strong style={{ color: '#34d399' }}>{activeClass.name}</strong>:
|
||||
</label>
|
||||
<div className="row" style={{ gap: 6, marginTop: 4 }}>
|
||||
<input
|
||||
id="sam3-prompt"
|
||||
value={promptDraft}
|
||||
onChange={(e) => onPromptDraftChange(e.target.value)}
|
||||
onKeyDown={(e) => { if (e.key === 'Enter') { e.preventDefault(); onPromptSave() } }}
|
||||
placeholder={activeClass.name}
|
||||
style={{ fontSize: '0.82rem' }}
|
||||
/>
|
||||
<button
|
||||
type="button"
|
||||
className="btn"
|
||||
onClick={onPromptSave}
|
||||
disabled={!dirty || promptSaving}
|
||||
style={{ fontSize: '0.8rem', whiteSpace: 'nowrap' }}
|
||||
>
|
||||
{promptSaving ? 'Saving…' : 'Save'}
|
||||
</button>
|
||||
</div>
|
||||
<p className="hint" style={{ fontSize: '0.74rem', marginTop: 4 }}>
|
||||
{promptError
|
||||
? <span style={{ color: '#fda4af' }}>{promptError}</span>
|
||||
: dirty
|
||||
? 'Unsaved — the batch run still uses the stored prompt until you save.'
|
||||
: 'Saved on the class. This is the text the batch run sends.'}
|
||||
</p>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
@@ -0,0 +1,126 @@
|
||||
import React, { useEffect, useRef, useState } from 'react'
|
||||
|
||||
/* Drag-to-draw box exemplars over the auto-annotate preview frame (REQ-172).
|
||||
*
|
||||
* Coordinates are normalized 0..1 against this overlay, which is sized to the
|
||||
* image itself — so the numbers sent to SAM3 are resolution-independent and
|
||||
* survive the preview being scaled to whatever the window allows.
|
||||
*
|
||||
* Boxes are stored in the format the backend wants: [cx, cy, w, h].
|
||||
*/
|
||||
|
||||
const POSITIVE = '#34d399'
|
||||
const NEGATIVE = '#f43f5e'
|
||||
|
||||
// Under this the drag was really a click. Zero-area boxes make SAM3's ROI pool
|
||||
// return nothing useful, so they are dropped rather than sent.
|
||||
const MIN_SIDE = 0.005
|
||||
|
||||
function toCorners(box) {
|
||||
const [cx, cy, w, h] = box
|
||||
return { x0: cx - w / 2, y0: cy - h / 2, w, h }
|
||||
}
|
||||
|
||||
export default function ExemplarCanvas({ exemplars, onAdd, disabled = false }) {
|
||||
const svgRef = useRef(null)
|
||||
const [draft, setDraft] = useState(null) // { x0, y0, x1, y1, positive }
|
||||
|
||||
// Escape abandons the box being dragged. Without it a drag started by mistake
|
||||
// can only be finished, and every finished box costs a GPU round trip.
|
||||
useEffect(() => {
|
||||
if (!draft) return undefined
|
||||
const onKey = (event) => { if (event.key === 'Escape') setDraft(null) }
|
||||
window.addEventListener('keydown', onKey)
|
||||
return () => window.removeEventListener('keydown', onKey)
|
||||
}, [draft])
|
||||
|
||||
const pointAt = (event) => {
|
||||
const rect = svgRef.current.getBoundingClientRect()
|
||||
return {
|
||||
x: Math.min(1, Math.max(0, (event.clientX - rect.left) / rect.width)),
|
||||
y: Math.min(1, Math.max(0, (event.clientY - rect.top) / rect.height)),
|
||||
}
|
||||
}
|
||||
|
||||
const handleDown = (event) => {
|
||||
if (disabled || event.button !== 0) return
|
||||
const { x, y } = pointAt(event)
|
||||
event.currentTarget.setPointerCapture(event.pointerId)
|
||||
// Shift marks the box as a counter-example: "not this one".
|
||||
setDraft({ x0: x, y0: y, x1: x, y1: y, positive: !event.shiftKey })
|
||||
}
|
||||
|
||||
const handleMove = (event) => {
|
||||
if (!draft) return
|
||||
const { x, y } = pointAt(event)
|
||||
setDraft({ ...draft, x1: x, y1: y })
|
||||
}
|
||||
|
||||
const handleUp = () => {
|
||||
if (!draft) return
|
||||
const w = Math.abs(draft.x1 - draft.x0)
|
||||
const h = Math.abs(draft.y1 - draft.y0)
|
||||
setDraft(null)
|
||||
if (w < MIN_SIDE || h < MIN_SIDE) return
|
||||
onAdd({
|
||||
box: [(draft.x0 + draft.x1) / 2, (draft.y0 + draft.y1) / 2, w, h],
|
||||
positive: draft.positive,
|
||||
})
|
||||
}
|
||||
|
||||
const draftRect = draft && {
|
||||
x0: Math.min(draft.x0, draft.x1),
|
||||
y0: Math.min(draft.y0, draft.y1),
|
||||
w: Math.abs(draft.x1 - draft.x0),
|
||||
h: Math.abs(draft.y1 - draft.y0),
|
||||
}
|
||||
|
||||
return (
|
||||
<svg
|
||||
ref={svgRef}
|
||||
viewBox="0 0 10000 10000"
|
||||
preserveAspectRatio="none"
|
||||
onPointerDown={handleDown}
|
||||
onPointerMove={handleMove}
|
||||
onPointerUp={handleUp}
|
||||
onPointerCancel={() => setDraft(null)}
|
||||
style={{
|
||||
position: 'absolute', top: 0, left: 0, width: '100%', height: '100%',
|
||||
cursor: disabled ? 'default' : 'crosshair',
|
||||
touchAction: 'none',
|
||||
}}
|
||||
>
|
||||
{exemplars.map((exemplar, index) => {
|
||||
const { x0, y0, w, h } = toCorners(exemplar.box)
|
||||
const color = exemplar.positive ? POSITIVE : NEGATIVE
|
||||
return (
|
||||
<g key={index}>
|
||||
<rect
|
||||
x={x0 * 10000} y={y0 * 10000} width={w * 10000} height={h * 10000}
|
||||
fill={color} fillOpacity={0.12}
|
||||
stroke={color} strokeWidth="30"
|
||||
strokeDasharray={exemplar.positive ? undefined : '90 60'}
|
||||
/>
|
||||
<text
|
||||
x={x0 * 10000 + 60} y={y0 * 10000 + 300}
|
||||
fill={color} fontSize="260" fontWeight="bold"
|
||||
style={{ textShadow: '0 0 40px #000, 0 0 40px #000' }}
|
||||
>
|
||||
{exemplar.positive ? '+' : '−'}{index + 1}
|
||||
</text>
|
||||
</g>
|
||||
)
|
||||
})}
|
||||
|
||||
{draftRect && (
|
||||
<rect
|
||||
x={draftRect.x0 * 10000} y={draftRect.y0 * 10000}
|
||||
width={draftRect.w * 10000} height={draftRect.h * 10000}
|
||||
fill="none"
|
||||
stroke={draft.positive ? POSITIVE : NEGATIVE}
|
||||
strokeWidth="30" strokeDasharray="60 40"
|
||||
/>
|
||||
)}
|
||||
</svg>
|
||||
)
|
||||
}
|
||||
@@ -0,0 +1,125 @@
|
||||
import { useEffect, useRef } from 'react'
|
||||
import { CheckIcon, XIcon } from './Icons'
|
||||
|
||||
/* The filter popup for an exemplar run (REQ-175).
|
||||
*
|
||||
* It lives at the top of the sidebar rather than over the frame — the shapes it
|
||||
* governs are drawn on the canvas, and a card on top of them covered the thing
|
||||
* being judged. It is mounted only while a run is undecided: the first drag
|
||||
* opens it, Apply and Discard close it. Nothing here is stored — the run it
|
||||
* governs is one frame's, and the next frame starts from the defaults again. */
|
||||
|
||||
const SLIDERS = [
|
||||
{ key: 'threshold', label: 'Confidence', min: 0.05, max: 0.95, step: 0.05,
|
||||
color: '#38bdf8', format: (v) => v.toFixed(2),
|
||||
hint: 'SAM3 score cutoff. Lower finds more, and more junk.' },
|
||||
{ key: 'iou_threshold', label: 'Overlap (NMS)', min: 0.1, max: 1, step: 0.05,
|
||||
color: '#c084fc', format: (v) => v.toFixed(2),
|
||||
hint: 'Two boxes overlapping this much are one object — the weaker goes. Lower deletes more.' },
|
||||
{ key: 'min_box_frac', label: 'Min box size', min: 0, max: 0.05, step: 0.001,
|
||||
color: '#34d399', format: (v) => `${(v * 100).toFixed(1)}% of frame`,
|
||||
hint: 'Boxes smaller than this are specks. Higher deletes more.' },
|
||||
{ key: 'max_detections', label: 'Max shapes', min: 5, max: 300, step: 5,
|
||||
color: '#fbbf24', format: (v) => String(v),
|
||||
hint: 'Keep only the highest-scoring N.' },
|
||||
]
|
||||
|
||||
export default function ExemplarFilterPanel({
|
||||
filters, preview, busy, className, negatives = 0, replacing = 0,
|
||||
onFilter, onReset, onApply, onDiscard, onUndo,
|
||||
}) {
|
||||
const rootRef = useRef(null)
|
||||
|
||||
// The sidebar scrolls, so a run opened while it is scrolled down would ask
|
||||
// for a decision the user cannot see.
|
||||
useEffect(() => {
|
||||
rootRef.current?.scrollIntoView({ block: 'nearest' })
|
||||
}, [])
|
||||
|
||||
// Enter applies, Escape discards — the panel is modal in intent even though
|
||||
// it never blocks the canvas underneath.
|
||||
useEffect(() => {
|
||||
function onKey(event) {
|
||||
if (event.target?.matches?.('input, textarea, select')) {
|
||||
if (event.key !== 'Enter' && event.key !== 'Escape') return
|
||||
}
|
||||
if (event.key === 'Enter') { event.preventDefault(); event.stopPropagation(); onApply() }
|
||||
if (event.key === 'Escape') { event.preventDefault(); event.stopPropagation(); onDiscard() }
|
||||
}
|
||||
window.addEventListener('keydown', onKey, true)
|
||||
return () => window.removeEventListener('keydown', onKey, true)
|
||||
}, [onApply, onDiscard])
|
||||
|
||||
const drawn = preview?.shapes.filter((shape) => shape.source === 'manual').length ?? 0
|
||||
const found = (preview?.shapes.length ?? 0) - drawn
|
||||
|
||||
return (
|
||||
<div className="exemplar-panel" role="dialog" aria-label="Auto-label filters" ref={rootRef}>
|
||||
<div className="row" style={{ justifyContent: 'space-between', marginBottom: 8 }}>
|
||||
<strong style={{ fontSize: '0.82rem' }}>Auto-label “{className}”</strong>
|
||||
<button type="button" className="btn" style={{ padding: '1px 6px', fontSize: '0.74rem' }}
|
||||
onClick={onUndo} title="Drop the last example you drew">Undo</button>
|
||||
</div>
|
||||
|
||||
<p className="muted" style={{ fontSize: '0.78rem', margin: '0 0 4px' }}>
|
||||
{busy
|
||||
? 'asking SAM3…'
|
||||
: preview
|
||||
? `${found} found + ${drawn} drawn`
|
||||
+ (negatives ? ` · ${negatives} rejected` : '')
|
||||
: 'Drag an example on the frame'}
|
||||
</p>
|
||||
{preview && !busy && (
|
||||
<p className="muted" style={{ fontSize: '0.72rem', margin: '0 0 10px' }}>
|
||||
Apply replaces the {replacing} “{className}” shape{replacing === 1 ? '' : 's'} on
|
||||
this frame — nothing is saved yet.
|
||||
</p>
|
||||
)}
|
||||
|
||||
{preview?.message && (
|
||||
<p className="error-banner" style={{ fontSize: '0.75rem', padding: '4px 8px', marginBottom: 10 }}>
|
||||
{preview.message}
|
||||
</p>
|
||||
)}
|
||||
|
||||
{SLIDERS.map((slider) => (
|
||||
<div key={slider.key} style={{ marginBottom: 10 }}>
|
||||
<div className="row" style={{ justifyContent: 'space-between', marginBottom: 2 }}>
|
||||
<label htmlFor={`exf-${slider.key}`} className="hint" style={{ fontSize: '0.76rem' }}>
|
||||
{slider.label}
|
||||
</label>
|
||||
<strong style={{ color: slider.color, fontSize: '0.78rem' }}>
|
||||
{slider.format(filters[slider.key])}
|
||||
</strong>
|
||||
</div>
|
||||
<input
|
||||
id={`exf-${slider.key}`}
|
||||
type="range"
|
||||
min={slider.min} max={slider.max} step={slider.step}
|
||||
value={filters[slider.key]}
|
||||
title={slider.hint}
|
||||
onChange={(event) => onFilter(slider.key, parseFloat(event.target.value))}
|
||||
style={{ width: '100%', cursor: 'pointer' }}
|
||||
/>
|
||||
</div>
|
||||
))}
|
||||
|
||||
<div className="row" style={{ gap: 6, marginTop: 12, flexWrap: 'wrap' }}>
|
||||
<button type="button" className="btn" style={{ padding: '2px 8px', fontSize: '0.76rem' }}
|
||||
onClick={onReset} title="Back to the defaults">
|
||||
Reset
|
||||
</button>
|
||||
<span className="spacer" style={{ flex: 1 }} />
|
||||
<button type="button" className="btn" style={{ padding: '2px 8px', fontSize: '0.76rem' }}
|
||||
onClick={onDiscard} title="Throw the run away, leave the frame as it is [Esc]">
|
||||
<XIcon size={12} /> Discard
|
||||
</button>
|
||||
<button type="button" className="btn btn-primary" style={{ padding: '2px 8px', fontSize: '0.76rem' }}
|
||||
onClick={onApply} disabled={busy || !preview}
|
||||
title="Write these shapes to the frame [Enter]">
|
||||
<CheckIcon size={12} /> Apply
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
@@ -0,0 +1,224 @@
|
||||
import React, { useEffect, useRef, useState } from 'react'
|
||||
import { api } from '../api'
|
||||
|
||||
/* The live-count preview.
|
||||
*
|
||||
* A WebRTC session shows the camera itself — the browser pulls it straight from
|
||||
* the streaming server over WHEP, so the frames never pass through this app and
|
||||
* the backend never encodes a JPEG for them (REQ-177). What the model saw is
|
||||
* drawn on a canvas on top, from a small JSON feed. An archive file has no
|
||||
* WebRTC leg, so it keeps the MJPEG the backend renders.
|
||||
*
|
||||
* Coordinates arrive in the 1280x720 space the pipeline works in; the canvas is
|
||||
* sized to that, and CSS stretches it over the video. That is exact whatever the
|
||||
* camera's real resolution is — the backend stretches each frame to 1280x720 per
|
||||
* axis, and stretching the canvas back over the picture is the inverse of it — but
|
||||
* only while the picture *fills* the element box. Do not give the video an
|
||||
* `aspect-ratio` or an `object-fit` that letterboxes it: this camera is 704x576,
|
||||
* so a forced 16:9 would pillarbox the picture and leave every box offset. */
|
||||
|
||||
const W = 1280
|
||||
const H = 720
|
||||
const OVERLAY_HZ = 10
|
||||
|
||||
const COLOURS = {
|
||||
counted: '#4ade80',
|
||||
tracked: '#f8bf71',
|
||||
ignored: '#828282',
|
||||
}
|
||||
|
||||
export default function LiveVideoPanel({ running, whepUrl, streamKey, placing, onPlace }) {
|
||||
return (
|
||||
<div className="panel" style={{ padding: 0, overflow: 'hidden', background: '#000', minHeight: 320 }}>
|
||||
{!running ? (
|
||||
<p className="empty" style={{ padding: 60, textAlign: 'center' }}>
|
||||
Not running. Set a source and press Start.
|
||||
</p>
|
||||
) : whepUrl ? (
|
||||
<WebrtcPreview whepUrl={whepUrl} placing={placing} onPlace={onPlace} />
|
||||
) : (
|
||||
<img
|
||||
key={streamKey}
|
||||
src={api.liveCountStreamUrl(streamKey)}
|
||||
alt="Live counting"
|
||||
onClick={onPlace}
|
||||
title={`Click to place: ${placing.replace('line_', '').replace('_', ' ')}`}
|
||||
style={{ width: '100%', display: 'block', cursor: 'crosshair' }}
|
||||
/>
|
||||
)}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
function WebrtcPreview({ whepUrl, placing, onPlace }) {
|
||||
const videoRef = useRef(null)
|
||||
const canvasRef = useRef(null)
|
||||
// A failed handshake used to be invisible: the canvas kept drawing boxes over
|
||||
// a black rectangle, which looks exactly like a broken renderer rather than a
|
||||
// stream that never arrived. The state is on screen now.
|
||||
const [link, setLink] = useState('connecting')
|
||||
|
||||
useEffect(() => {
|
||||
let pc = null
|
||||
let resourceUrl = ''
|
||||
let cancelled = false
|
||||
|
||||
async function connect() {
|
||||
pc = new RTCPeerConnection()
|
||||
pc.addTransceiver('video', { direction: 'recvonly' })
|
||||
pc.ontrack = (event) => {
|
||||
if (videoRef.current) videoRef.current.srcObject = event.streams[0]
|
||||
}
|
||||
pc.oniceconnectionstatechange = () => {
|
||||
if (cancelled) return
|
||||
if (pc.iceConnectionState === 'connected') setLink('live')
|
||||
else if (['failed', 'disconnected', 'closed'].includes(pc.iceConnectionState)) {
|
||||
setLink(`WebRTC ${pc.iceConnectionState} — is ${whepUrl} reachable from this browser?`)
|
||||
}
|
||||
}
|
||||
await pc.setLocalDescription(await pc.createOffer())
|
||||
// MediaMTX accepts a complete offer; gathering first avoids trickling
|
||||
// candidates over a second request.
|
||||
await new Promise((resolve) => {
|
||||
if (pc.iceGatheringState === 'complete') return resolve()
|
||||
pc.onicegatheringstatechange = () => pc.iceGatheringState === 'complete' && resolve()
|
||||
setTimeout(resolve, 2000)
|
||||
})
|
||||
if (cancelled) return
|
||||
const response = await fetch(whepUrl, {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/sdp' },
|
||||
body: pc.localDescription.sdp,
|
||||
})
|
||||
if (!response.ok) throw new Error(`WHEP ${response.status} from ${whepUrl}`)
|
||||
resourceUrl = response.headers.get('Location') || ''
|
||||
const answer = await response.text()
|
||||
if (cancelled) return
|
||||
await pc.setRemoteDescription({ type: 'answer', sdp: answer })
|
||||
}
|
||||
|
||||
connect().catch((exc) => {
|
||||
if (!cancelled) setLink(exc.message)
|
||||
})
|
||||
|
||||
return () => {
|
||||
cancelled = true
|
||||
// Tell the server the viewer left, so it stops sending to a dead peer.
|
||||
if (resourceUrl) {
|
||||
fetch(new URL(resourceUrl, whepUrl).href, { method: 'DELETE' }).catch(() => {})
|
||||
}
|
||||
if (pc) pc.close()
|
||||
}
|
||||
}, [whepUrl])
|
||||
|
||||
// Poll the geometry and repaint. Independent of the video's frame rate: the
|
||||
// boxes are a few hundred bytes, the video is never touched.
|
||||
useEffect(() => {
|
||||
let timer = null
|
||||
let stopped = false
|
||||
|
||||
async function tick() {
|
||||
try {
|
||||
draw(canvasRef.current, await api.liveCountOverlay())
|
||||
} catch {
|
||||
/* a dropped poll is corrected by the next one */
|
||||
}
|
||||
if (!stopped) timer = setTimeout(tick, 1000 / OVERLAY_HZ)
|
||||
}
|
||||
tick()
|
||||
return () => {
|
||||
stopped = true
|
||||
clearTimeout(timer)
|
||||
}
|
||||
}, [])
|
||||
|
||||
return (
|
||||
<div
|
||||
onClick={onPlace}
|
||||
title={`Click to place: ${placing.replace('line_', '').replace('_', ' ')}`}
|
||||
style={{ position: 'relative', width: '100%', cursor: 'crosshair', lineHeight: 0 }}
|
||||
>
|
||||
<video
|
||||
ref={videoRef}
|
||||
autoPlay
|
||||
muted
|
||||
playsInline
|
||||
style={{ width: '100%', display: 'block', background: '#000' }}
|
||||
/>
|
||||
<canvas
|
||||
ref={canvasRef}
|
||||
width={W}
|
||||
height={H}
|
||||
style={{ position: 'absolute', inset: 0, width: '100%', height: '100%' }}
|
||||
/>
|
||||
{link !== 'live' && (
|
||||
<p style={{
|
||||
position: 'absolute', inset: 'auto 12px 12px 12px', margin: 0, lineHeight: 1.4,
|
||||
fontSize: '0.76rem', color: link === 'connecting' ? '#a1a1aa' : '#f87171',
|
||||
}}>
|
||||
{link === 'connecting' ? `Connecting to ${whepUrl}…` : link}
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
/* Mirrors what `backend/live_render.py` burns into the MJPEG, so the two
|
||||
* previews say the same thing about the same session. */
|
||||
function draw(canvas, data) {
|
||||
if (!canvas) return
|
||||
const ctx = canvas.getContext('2d')
|
||||
ctx.clearRect(0, 0, W, H)
|
||||
if (!data || !data.line) return
|
||||
const { y, x_start: xs, x_end: xe, margin } = data.line
|
||||
|
||||
// Shade what the region excludes, so "ignored" never looks like "missed".
|
||||
ctx.fillStyle = 'rgba(0, 0, 0, 0.55)'
|
||||
if (xs > 0) ctx.fillRect(0, 0, xs, H)
|
||||
if (xe < W) ctx.fillRect(xe, 0, W - xe, H)
|
||||
|
||||
ctx.lineWidth = 2
|
||||
ctx.font = '16px ui-monospace, monospace'
|
||||
for (const box of data.boxes || []) {
|
||||
const [x1, y1, x2, y2] = box.b
|
||||
ctx.strokeStyle = COLOURS[box.s] || COLOURS.tracked
|
||||
ctx.lineWidth = box.s === 'ignored' ? 1 : 2
|
||||
ctx.strokeRect(x1, y1, x2 - x1, y2 - y1)
|
||||
if (box.s !== 'ignored') {
|
||||
ctx.fillStyle = ctx.strokeStyle
|
||||
ctx.fillText(`#${box.id} ${box.c}`, x1, Math.max(14, y1 - 5))
|
||||
}
|
||||
}
|
||||
|
||||
ctx.lineWidth = 2
|
||||
ctx.strokeStyle = '#ff00ff'
|
||||
for (const edge of [xs, xe]) {
|
||||
if (edge > 0 && edge < W) {
|
||||
ctx.beginPath()
|
||||
ctx.moveTo(edge, 0)
|
||||
ctx.lineTo(edge, H)
|
||||
ctx.stroke()
|
||||
}
|
||||
}
|
||||
|
||||
ctx.strokeStyle = '#ffff00'
|
||||
ctx.beginPath()
|
||||
ctx.moveTo(xs, y)
|
||||
ctx.lineTo(xe, y)
|
||||
ctx.stroke()
|
||||
ctx.strokeStyle = '#00a0a0'
|
||||
ctx.lineWidth = 1
|
||||
for (const edge of [y - margin, y + margin]) {
|
||||
ctx.beginPath()
|
||||
ctx.moveTo(xs, edge)
|
||||
ctx.lineTo(xe, edge)
|
||||
ctx.stroke()
|
||||
}
|
||||
|
||||
const panel = `IN ${data.loading} OUT ${data.unloading} NET ${data.net}`
|
||||
ctx.fillStyle = 'rgba(0, 0, 0, 0.85)'
|
||||
ctx.fillRect(12, 12, 9 * panel.length + 30, 46)
|
||||
ctx.fillStyle = COLOURS.counted
|
||||
ctx.font = 'bold 24px ui-monospace, monospace'
|
||||
ctx.fillText(panel, 24, 44)
|
||||
}
|
||||
@@ -1,6 +1,6 @@
|
||||
import React, { useEffect, useState } from 'react'
|
||||
import { api } from '../api'
|
||||
import { PreviewShapes } from './AutoAnnotateModal'
|
||||
import { PreviewShapes } from './PreviewShapes'
|
||||
|
||||
const ENGINES = [
|
||||
{ id: 'sam3', label: 'SAM3' },
|
||||
|
||||
@@ -0,0 +1,64 @@
|
||||
import React from 'react'
|
||||
|
||||
// Shared with MassAutoAnnotateModal: draws detection boxes/polygons over a
|
||||
// frame in a 0..10000 viewBox.
|
||||
export function PreviewShapes({ shapes, project }) {
|
||||
return shapes.map((shape, i) => {
|
||||
if (!shape.geometry || !shape.geometry.points) return null
|
||||
const classObj = project.classes.find(c => c.class_id === shape.class_id)
|
||||
const className = classObj?.name || 'Unknown'
|
||||
const color = ['#38bdf8', '#34d399', '#f472b6', '#a78bfa', '#fbbf24'][shape.class_id % 5] || '#fff'
|
||||
|
||||
let minX = 1, minY = 1, maxX = 0, maxY = 0
|
||||
if (shape.geometry.type === 'bbox') {
|
||||
const [left, top, right, bottom] = shape.geometry.points
|
||||
minX = left; minY = top; maxX = right; maxY = bottom;
|
||||
} else {
|
||||
shape.geometry.points.forEach(pt => {
|
||||
if (pt[0] < minX) minX = pt[0]
|
||||
if (pt[1] < minY) minY = pt[1]
|
||||
if (pt[0] > maxX) maxX = pt[0]
|
||||
if (pt[1] > maxY) maxY = pt[1]
|
||||
})
|
||||
}
|
||||
|
||||
const x0 = minX * 10000
|
||||
const y0 = minY * 10000
|
||||
const bw = (maxX - minX) * 10000
|
||||
const bh = (maxY - minY) * 10000
|
||||
|
||||
return (
|
||||
<g key={i}>
|
||||
{shape.geometry.type === 'polygon' && (
|
||||
<polygon
|
||||
points={shape.geometry.points.map(pt => `${pt[0] * 10000},${pt[1] * 10000}`).join(' ')}
|
||||
fill={color}
|
||||
fillOpacity={0.35}
|
||||
stroke={color}
|
||||
strokeWidth="10"
|
||||
/>
|
||||
)}
|
||||
<rect
|
||||
x={x0}
|
||||
y={y0}
|
||||
width={bw}
|
||||
height={bh}
|
||||
fill="none"
|
||||
stroke={color}
|
||||
strokeWidth="20"
|
||||
strokeDasharray="40 20"
|
||||
/>
|
||||
<text
|
||||
x={x0}
|
||||
y={y0 > 300 ? y0 - 100 : y0 + 300}
|
||||
fill={color}
|
||||
fontSize="240"
|
||||
fontWeight="bold"
|
||||
style={{ textShadow: '10px 10px 10px #000, -10px -10px 10px #000, 10px -10px 10px #000, -10px 10px 10px #000' }}
|
||||
>
|
||||
{className} {shape.score ? `${(shape.score * 100).toFixed(1)}%` : ''}
|
||||
</text>
|
||||
</g>
|
||||
)
|
||||
})
|
||||
}
|
||||
@@ -4,6 +4,7 @@ import { TrashIcon } from './Icons'
|
||||
import ShortcutsPanel from './ShortcutsPanel'
|
||||
|
||||
export default function ReviewSidebar({
|
||||
exemplarPanel,
|
||||
classesList,
|
||||
activeClass,
|
||||
reclass,
|
||||
@@ -18,6 +19,7 @@ export default function ReviewSidebar({
|
||||
}) {
|
||||
return (
|
||||
<aside className="review-side stack">
|
||||
{exemplarPanel}
|
||||
<div className="panel side-panel">
|
||||
<h2>Classes</h2>
|
||||
<div className="class-list">
|
||||
|
||||
@@ -5,7 +5,15 @@ export default function ShortcutsPanel() {
|
||||
<dl className="shortcuts">
|
||||
<div>
|
||||
<dt><kbd>Drag</kbd></dt>
|
||||
<dd>Add box / resize / move</dd>
|
||||
<dd>Example of this class → preview a re-detect</dd>
|
||||
</div>
|
||||
<div>
|
||||
<dt><kbd>Enter</kbd> <kbd>Esc</kbd></dt>
|
||||
<dd>Apply / discard the preview</dd>
|
||||
</div>
|
||||
<div>
|
||||
<dt><kbd>Shift</kbd>+<kbd>Drag</kbd></dt>
|
||||
<dd>Not this → drop it and re-detect</dd>
|
||||
</div>
|
||||
<div>
|
||||
<dt><kbd>V</kbd></dt>
|
||||
|
||||
@@ -0,0 +1,31 @@
|
||||
/* The tuning dials on the live-count page, and what each one means.
|
||||
* Data only — split out of `LiveCountPage.jsx` to keep it under 400 lines. */
|
||||
|
||||
// `live` fields can be moved during a session — placing a counting line means
|
||||
// watching the stream while you move it, and a restart would throw the counts away.
|
||||
export const FIELDS = [
|
||||
{ key: 'line_y', label: 'Counting line Y', min: 0, max: 720, step: 1, live: true,
|
||||
hint: 'Sacks are counted as they cross this line. Click the video to place it.' },
|
||||
{ key: 'line_x_start', label: 'Line start X', min: 0, max: 1280, step: 1, live: true,
|
||||
hint: 'Ignore anything left of this.' },
|
||||
{ key: 'line_x_end', label: 'Line end X', min: 0, max: 1280, step: 1, live: true,
|
||||
hint: 'Ignore anything right of this.' },
|
||||
{ key: 'margin', label: 'Band margin (px)', min: 0, max: 120, step: 1,
|
||||
hint: 'Dead band around the line, so jitter alone never counts.' },
|
||||
{ key: 'entry_travel_min', label: 'Entry travel min (px)', min: 0, max: 200, step: 1,
|
||||
hint: 'A track must move this far from where it first appeared before it can count. '
|
||||
+ 'Raise it to kill ghost boxes that blink into existence next to the line.' },
|
||||
{ key: 'handoff_radius', label: 'Hand-off radius (px)', min: 0, max: 300, step: 5,
|
||||
hint: 'When a track dies, a new track born this close to where it was heading inherits '
|
||||
+ 'its history — this is what stops an ID switch at the line losing the count. '
|
||||
+ 'The most sensitive dial here: too large and unrelated sacks adopt each other. '
|
||||
+ 'Calibrate against a clip with a known count.' },
|
||||
{ key: 'unload_confirm_frames', label: 'Unload confirm (frames)', min: 1, max: 15, step: 1,
|
||||
hint: 'Frames a sack must stay above the band before it counts as unloaded. Stops a '
|
||||
+ 'worker repositioning a sack from cancelling a real count.' },
|
||||
{ key: 'min_area_scale', label: 'Min area scale', min: 0, max: 2, step: 0.1,
|
||||
hint: 'Perspective-aware size gate: boxes too small for their depth are fragments, not '
|
||||
+ 'sacks. 0 turns it off.' },
|
||||
{ key: 'conf', label: 'Confidence', min: 0.05, max: 0.95, step: 0.05,
|
||||
hint: 'Detector threshold.' },
|
||||
]
|
||||
@@ -0,0 +1,153 @@
|
||||
import { useCallback, useEffect, useRef, useState } from 'react'
|
||||
import { api } from '../api'
|
||||
|
||||
/* The review editor's exemplar pool and its preview (REQ-173/174/175).
|
||||
*
|
||||
* A drag is both a label and a visual prompt. It never writes: the pool is
|
||||
* posted as a dry run, the result comes back as preview shapes, the filter
|
||||
* panel re-previews as its sliders move, and only Apply commits. The pool
|
||||
* lives in a ref as well as in state — the ref is what gets sent, so a drag
|
||||
* that lands while a pass is in flight is never lost — and it dies with the
|
||||
* frame, because SAM3's geometric prompts pool features from *this* image. */
|
||||
|
||||
const DEBOUNCE_MS = 400
|
||||
const SLIDER_DEBOUNCE_MS = 250
|
||||
|
||||
export const FILTER_DEFAULTS = {
|
||||
threshold: 0.5,
|
||||
iou_threshold: 0.8,
|
||||
min_box_frac: 0.002,
|
||||
max_detections: 100,
|
||||
}
|
||||
|
||||
export default function useExemplarPool({ frameId, classId, onApplied, onError, onBusy }) {
|
||||
const [exemplars, setExemplars] = useState([])
|
||||
const [preview, setPreview] = useState(null) // { shapes, message, redetected }
|
||||
// The panel is open from the first drag until the run is applied or thrown
|
||||
// away — not for as long as the pool exists, which outlives it (REQ-175).
|
||||
const [active, setActive] = useState(false)
|
||||
const [filters, setFilters] = useState(FILTER_DEFAULTS)
|
||||
const poolRef = useRef([])
|
||||
const filterRef = useRef(FILTER_DEFAULTS)
|
||||
// How much of the pool is already on the frame: Discard rewinds to here.
|
||||
const appliedRef = useRef(0)
|
||||
const timer = useRef(null)
|
||||
const runRef = useRef(null)
|
||||
const inFlight = useRef(false)
|
||||
const rerunWanted = useRef(false)
|
||||
|
||||
// A pool only means anything on the image and for the class it was drawn
|
||||
// for, so both changes discard it.
|
||||
useEffect(() => {
|
||||
clearTimeout(timer.current)
|
||||
poolRef.current = []
|
||||
appliedRef.current = 0
|
||||
setExemplars([])
|
||||
setPreview(null)
|
||||
setActive(false)
|
||||
}, [frameId, classId])
|
||||
|
||||
// One SAM3 pass per pool, not per drag: a burst of quick drags or slider
|
||||
// moves coalesces into a single run, and a change arriving mid-pass queues
|
||||
// exactly one rerun rather than stacking.
|
||||
const run = useCallback(async ({ apply = false } = {}) => {
|
||||
if (!frameId || !poolRef.current.length) return
|
||||
if (inFlight.current) { rerunWanted.current = true; return }
|
||||
inFlight.current = true
|
||||
onBusy?.(true)
|
||||
// What was actually sent: a drag landing mid-flight must not be counted as
|
||||
// applied, and must not be lost either — it re-previews below.
|
||||
const sent = poolRef.current
|
||||
try {
|
||||
const payload = await api.exemplarLabel(frameId, {
|
||||
exemplars: sent, class_id: classId, apply, ...filterRef.current,
|
||||
})
|
||||
if (apply) {
|
||||
// A negative is spent once it has been applied: the frame no longer
|
||||
// carries what it rejected, so the drawing goes and only the positives
|
||||
// stay behind as prompts. Drags that landed mid-flight are not part of
|
||||
// this run and keep their place at the end of the pool.
|
||||
const extra = poolRef.current.slice(sent.length)
|
||||
const kept = sent.filter((item) => item.positive)
|
||||
poolRef.current = [...kept, ...extra]
|
||||
appliedRef.current = kept.length
|
||||
setExemplars(poolRef.current)
|
||||
setPreview(null)
|
||||
setActive(false)
|
||||
onApplied(payload.annotations)
|
||||
} else {
|
||||
setPreview({ shapes: payload.shapes, message: payload.message,
|
||||
redetected: payload.redetected })
|
||||
}
|
||||
} catch (exc) {
|
||||
onError(exc.message)
|
||||
} finally {
|
||||
inFlight.current = false
|
||||
onBusy?.(false)
|
||||
// A drag that arrived mid-pass re-opens the panel: its result was not in
|
||||
// what came back.
|
||||
if (rerunWanted.current) {
|
||||
rerunWanted.current = false
|
||||
setActive(true)
|
||||
runRef.current?.()
|
||||
}
|
||||
}
|
||||
}, [frameId, classId, onApplied, onError, onBusy])
|
||||
runRef.current = run
|
||||
|
||||
const schedule = useCallback((delay) => {
|
||||
clearTimeout(timer.current)
|
||||
timer.current = setTimeout(() => runRef.current?.(), delay)
|
||||
}, [])
|
||||
|
||||
const add = useCallback((box, positive) => {
|
||||
if (!frameId) return
|
||||
poolRef.current = [...poolRef.current, { box, positive }]
|
||||
setExemplars(poolRef.current)
|
||||
setActive(true)
|
||||
// A drag during a pass cannot be in that pass's result, so it always earns
|
||||
// a rerun — including during an apply, which would otherwise swallow it.
|
||||
if (inFlight.current) rerunWanted.current = true
|
||||
schedule(DEBOUNCE_MS)
|
||||
}, [frameId, schedule])
|
||||
|
||||
const setFilter = useCallback((key, value) => {
|
||||
filterRef.current = { ...filterRef.current, [key]: value }
|
||||
setFilters(filterRef.current)
|
||||
if (poolRef.current.length) schedule(SLIDER_DEBOUNCE_MS)
|
||||
}, [schedule])
|
||||
|
||||
const resetFilters = useCallback(() => {
|
||||
filterRef.current = FILTER_DEFAULTS
|
||||
setFilters(FILTER_DEFAULTS)
|
||||
if (poolRef.current.length) schedule(SLIDER_DEBOUNCE_MS)
|
||||
}, [schedule])
|
||||
|
||||
const apply = useCallback(() => {
|
||||
clearTimeout(timer.current)
|
||||
runRef.current?.({ apply: true })
|
||||
}, [])
|
||||
|
||||
// Discard rewinds the pool to what is already on the frame, so the drags
|
||||
// since the last Apply are undone in one go and nothing was ever written.
|
||||
const discard = useCallback(() => {
|
||||
clearTimeout(timer.current)
|
||||
poolRef.current = poolRef.current.slice(0, appliedRef.current)
|
||||
setExemplars(poolRef.current)
|
||||
setPreview(null)
|
||||
setActive(false)
|
||||
}, [])
|
||||
|
||||
const undo = useCallback(() => {
|
||||
poolRef.current = poolRef.current.slice(0, -1)
|
||||
appliedRef.current = Math.min(appliedRef.current, poolRef.current.length)
|
||||
setExemplars(poolRef.current)
|
||||
if (poolRef.current.length) schedule(SLIDER_DEBOUNCE_MS)
|
||||
else { setPreview(null); setActive(false) }
|
||||
}, [schedule])
|
||||
|
||||
useEffect(() => () => clearTimeout(timer.current), [])
|
||||
|
||||
return { exemplars, preview, active, filters,
|
||||
add, setFilter, resetFilters, apply, discard, undo }
|
||||
}
|
||||
@@ -1,52 +1,31 @@
|
||||
import React, { useCallback, useEffect, useRef, useState } from 'react'
|
||||
import { api } from '../api'
|
||||
import { AlertIcon } from '../components/Icons'
|
||||
import LiveVideoPanel from '../components/LiveVideoPanel'
|
||||
import { FIELDS } from '../components/liveCountFields'
|
||||
|
||||
/* Live counting test bench.
|
||||
*
|
||||
* Point a trained model at an RTSP camera (or a local file) and watch it count.
|
||||
* Point a trained model at the WebRTC camera feed (or a local file) and watch it
|
||||
* count. A live session is watched over WebRTC straight from the streaming
|
||||
* server, so the preview costs this app nothing (REQ-176, REQ-177).
|
||||
* It runs the same tracker, stabiliser and line-cross counter the production
|
||||
* script uses, so a number here means the same thing there. What it leaves out
|
||||
* is the batch lifecycle and its database — this answers "does the model count
|
||||
* correctly", not "how many sacks today". */
|
||||
|
||||
// `live` fields can be moved during a session — placing a counting line means
|
||||
// watching the stream while you move it, and a restart would throw the counts away.
|
||||
const FIELDS = [
|
||||
{ key: 'line_y', label: 'Counting line Y', min: 0, max: 720, step: 1, live: true,
|
||||
hint: 'Sacks are counted as they cross this line. Click the video to place it.' },
|
||||
{ key: 'line_x_start', label: 'Line start X', min: 0, max: 1280, step: 1, live: true,
|
||||
hint: 'Ignore anything left of this.' },
|
||||
{ key: 'line_x_end', label: 'Line end X', min: 0, max: 1280, step: 1, live: true,
|
||||
hint: 'Ignore anything right of this.' },
|
||||
{ key: 'margin', label: 'Band margin (px)', min: 0, max: 120, step: 1,
|
||||
hint: 'Dead band around the line, so jitter alone never counts.' },
|
||||
{ key: 'entry_travel_min', label: 'Entry travel min (px)', min: 0, max: 200, step: 1,
|
||||
hint: 'A track must move this far from where it first appeared before it can count. '
|
||||
+ 'Raise it to kill ghost boxes that blink into existence next to the line.' },
|
||||
{ key: 'handoff_radius', label: 'Hand-off radius (px)', min: 0, max: 300, step: 5,
|
||||
hint: 'When a track dies, a new track born this close to where it was heading inherits '
|
||||
+ 'its history — this is what stops an ID switch at the line losing the count. '
|
||||
+ 'The most sensitive dial here: too large and unrelated sacks adopt each other. '
|
||||
+ 'Calibrate against a clip with a known count.' },
|
||||
{ key: 'unload_confirm_frames', label: 'Unload confirm (frames)', min: 1, max: 15, step: 1,
|
||||
hint: 'Frames a sack must stay above the band before it counts as unloaded. Stops a '
|
||||
+ 'worker repositioning a sack from cancelling a real count.' },
|
||||
{ key: 'min_area_scale', label: 'Min area scale', min: 0, max: 2, step: 0.1,
|
||||
hint: 'Perspective-aware size gate: boxes too small for their depth are fragments, not '
|
||||
+ 'sacks. 0 turns it off.' },
|
||||
{ key: 'conf', label: 'Confidence', min: 0.05, max: 0.95, step: 0.05,
|
||||
hint: 'Detector threshold.' },
|
||||
]
|
||||
|
||||
export default function LiveCountPage({ projectId, onProject }) {
|
||||
const [models, setModels] = useState([])
|
||||
const [modelPath, setModelPath] = useState('')
|
||||
// Archive video by default: a local file decodes at ~100 fps, an RTSP camera
|
||||
// at ~6 because OpenCV decodes 1080p on the CPU. Testing counting accuracy is
|
||||
// far quicker against a file.
|
||||
// Archive video by default: a local file decodes at ~100 fps, the camera at
|
||||
// whatever the link delivers. Tracking and inference run at 205 fps, so a live
|
||||
// session is bound by how fast frames arrive, never by the model — testing
|
||||
// counting accuracy is far quicker against a file.
|
||||
const [mode, setMode] = useState('file')
|
||||
const [source, setSource] = useState('rtsp://192.168.192.96:8554/cam')
|
||||
// The stream server is deployment-specific, so it is an env override with a
|
||||
// sane default rather than a constant (REQ-176).
|
||||
const [source, setSource] = useState(
|
||||
import.meta.env.VITE_WHEP_URL || 'http://192.168.192.96:8889/cam')
|
||||
const [dates, setDates] = useState([])
|
||||
const [date, setDate] = useState('')
|
||||
const [videos, setVideos] = useState([])
|
||||
@@ -194,7 +173,7 @@ export default function LiveCountPage({ projectId, onProject }) {
|
||||
<div>
|
||||
<label className="hint" style={{ fontSize: '0.8rem' }}>Source</label>
|
||||
<div style={{ display: 'flex', gap: 6, margin: '6px 0 8px' }}>
|
||||
{[['file', 'Archive video'], ['stream', 'RTSP stream']].map(([id, label]) => (
|
||||
{[['file', 'Archive video'], ['stream', 'WebRTC stream']].map(([id, label]) => (
|
||||
<button
|
||||
key={id}
|
||||
type="button"
|
||||
@@ -252,7 +231,10 @@ export default function LiveCountPage({ projectId, onProject }) {
|
||||
style={{ width: '100%', fontSize: '0.8rem' }}
|
||||
/>
|
||||
<p className="hint" style={{ fontSize: '0.72rem', margin: '4px 0 0' }}>
|
||||
Real time, but capped near 6 fps — OpenCV decodes this camera's 1080p on the CPU.
|
||||
WebRTC (WHEP) URL only, e.g. <span className="mono">http://host:8889/cam</span>.
|
||||
The browser plays this feed directly; the counter reads the RTSP leg of the
|
||||
same stream. Its FPS is the network's, not the model's — inference alone
|
||||
runs at ~205 fps.
|
||||
</p>
|
||||
</>
|
||||
)}
|
||||
@@ -365,22 +347,13 @@ export default function LiveCountPage({ projectId, onProject }) {
|
||||
</div>
|
||||
)}
|
||||
|
||||
<div className="panel" style={{ padding: 0, overflow: 'hidden', background: '#000', minHeight: 320 }}>
|
||||
{running ? (
|
||||
<img
|
||||
key={streamKey}
|
||||
src={api.liveCountStreamUrl(streamKey)}
|
||||
alt="Live counting"
|
||||
onClick={placeOnClick}
|
||||
title={`Click to place: ${placing.replace('line_', '').replace('_', ' ')}`}
|
||||
style={{ width: '100%', display: 'block', cursor: 'crosshair' }}
|
||||
/>
|
||||
) : (
|
||||
<p className="empty" style={{ padding: 60, textAlign: 'center' }}>
|
||||
Not running. Set a source and press Start.
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
<LiveVideoPanel
|
||||
running={running}
|
||||
whepUrl={status.whep_url || ''}
|
||||
streamKey={streamKey}
|
||||
placing={placing}
|
||||
onPlace={placeOnClick}
|
||||
/>
|
||||
|
||||
{(status.events?.length ?? 0) > 0 && (
|
||||
<div className="panel" style={{ padding: 14 }}>
|
||||
|
||||
@@ -6,6 +6,8 @@ import Filmstrip from '../components/Filmstrip'
|
||||
import { AlertIcon, CheckIcon, XIcon } from '../components/Icons'
|
||||
import QuickReclassBar from '../components/QuickReclassBar'
|
||||
import ReviewSidebar from '../components/ReviewSidebar'
|
||||
import ExemplarFilterPanel from '../components/ExemplarFilterPanel'
|
||||
import useExemplarPool from '../hooks/useExemplarPool'
|
||||
|
||||
export default function ReviewPage({ batchId: rawBatchId, projectId, onProject }) {
|
||||
const [batch, setBatch] = useState(null)
|
||||
@@ -112,6 +114,17 @@ export default function ReviewPage({ batchId: rawBatchId, projectId, onProject }
|
||||
} catch (exc) { setError(exc.message) }
|
||||
}
|
||||
|
||||
// The exemplar pool (REQ-173/174/175) owns the drag gesture in draw mode.
|
||||
const onExemplarApplied = useCallback((rows) => {
|
||||
setAnnotations(rows)
|
||||
setSelectedId(null)
|
||||
if (frame) patchFrameLocally(frame.id, { annotation_count: rows.length })
|
||||
}, [frame])
|
||||
const pool = useExemplarPool({
|
||||
frameId: frame?.id, classId: activeClass,
|
||||
onApplied: onExemplarApplied, onError: setError, onBusy: setBusy,
|
||||
})
|
||||
|
||||
async function assist(box) {
|
||||
setBusy(true); setError('')
|
||||
try {
|
||||
@@ -359,6 +372,8 @@ export default function ReviewPage({ batchId: rawBatchId, projectId, onProject }
|
||||
|
||||
const reviewed = (batch.review?.approved ?? 0) + (batch.review?.rejected ?? 0)
|
||||
const classesList = project.classes ?? []
|
||||
const activeClassName =
|
||||
classesList.find((item) => item.class_id === activeClass)?.name ?? 'this class'
|
||||
|
||||
async function approveAllFrames() {
|
||||
if (!window.confirm(`Mark all ${batch.review?.pending ?? 0} pending frames as approved?`)) return
|
||||
@@ -432,10 +447,13 @@ export default function ReviewPage({ batchId: rawBatchId, projectId, onProject }
|
||||
classes={classesList}
|
||||
mode={mode}
|
||||
selectedIds={markedIds}
|
||||
exemplars={pool.exemplars}
|
||||
preview={pool.preview?.shapes ?? null}
|
||||
onSelect={setSelectedId}
|
||||
onCreate={createShape}
|
||||
onUpdate={updateShape}
|
||||
onAssist={assist}
|
||||
onExemplar={pool.add}
|
||||
onMarquee={onMarquee}
|
||||
/>
|
||||
)}
|
||||
@@ -476,6 +494,15 @@ export default function ReviewPage({ batchId: rawBatchId, projectId, onProject }
|
||||
</span>
|
||||
</>
|
||||
)}
|
||||
{mode === 'draw' && (
|
||||
<span className="muted" style={{ fontSize: '0.78rem' }}>
|
||||
{pool.exemplars.length
|
||||
? `${pool.exemplars.filter((e) => e.positive).length} example(s), `
|
||||
+ `${pool.exemplars.filter((e) => !e.positive).length} negative — `
|
||||
+ 'tune the filters, then Apply'
|
||||
: 'Drag an example of this class · Shift-drag = not this'}
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{mode === 'select' && markedIds.length > 0 && (
|
||||
@@ -556,6 +583,21 @@ export default function ReviewPage({ batchId: rawBatchId, projectId, onProject }
|
||||
</div>
|
||||
|
||||
<ReviewSidebar
|
||||
exemplarPanel={mode === 'draw' && pool.active && (
|
||||
<ExemplarFilterPanel
|
||||
filters={pool.filters}
|
||||
preview={pool.preview}
|
||||
busy={busy}
|
||||
className={activeClassName}
|
||||
negatives={pool.exemplars.filter((item) => !item.positive).length}
|
||||
replacing={annotations.filter((row) => row.class_id === activeClass).length}
|
||||
onFilter={pool.setFilter}
|
||||
onReset={pool.resetFilters}
|
||||
onApply={pool.apply}
|
||||
onDiscard={pool.discard}
|
||||
onUndo={pool.undo}
|
||||
/>
|
||||
)}
|
||||
classesList={classesList}
|
||||
activeClass={activeClass}
|
||||
reclass={reclass}
|
||||
|
||||
Executable → Regular
File mode changed.
Executable → Regular
File mode changed.
Executable → Regular
File mode changed.
Executable → Regular
File mode changed.
Reference in new issue
Block a user