feat: add counting bench, triage, and dataset modules

This commit includes major additions and updates to the frontend and backend architectures, introducing new dataset management, live counting features, batch processing, and triage logic. Includes new UI pages, components, and API routes.
This commit is contained in:
asus committed 2026-08-14 16:28:52 +07:00
1 parent 8285400254
commit 5c7c122105
80 files changed
+20074 -1412

No files matched your search

+98 -10
View File
@@ -12,7 +12,8 @@ import shutil
import time
from typing import Optional
from backend import config, dataset, db, evaluate, hardware, jobs, projects
from backend import (augment, base_dataset, config, dataset, datasets, db, evaluate,
hardware, jobs, projects)
PRETRAINED = {"bbox": "yolo11n.pt", "polygon": "yolo11n-seg.pt"}
@@ -25,20 +26,84 @@ def models_dir(project_slug: str) -> str:
return os.path.join(config.project_dir(project_slug), "models")
def start(project_id: int, epochs: int = 50, overrides: Optional[dict] = None, batch_ids: Optional[list] = None, class_ids: Optional[list] = None) -> dict:
#: Rough bytes one decoded 640px training image occupies in the RAM cache,
#: measured against the run that OOMed: 10.2 GB across 15,774 images.
_BYTES_PER_CACHED_IMAGE = 700_000
def _cache_mode(train_images: int, job) -> object:
"""Pick Ultralytics' `cache` argument for the memory this host actually has.
RAM caching is a large speedup and worth taking when it fits. It is only
taken with three times the headroom the raw estimate asks for: the run that
died had ~10 GB of cache on a 30 GB host and still lost, because the
dataloader workers fork after the cache is built and their copy-on-write
pages are what turn "just fits" into a kill. Two-times headroom would have
green-lit exactly the run that failed.
"""
needed = train_images * _BYTES_PER_CACHED_IMAGE
try:
import psutil
available = psutil.virtual_memory().available
except Exception:
available = 0
if available == 0:
job.log(f"Image cache: disk (cannot read free memory; {train_images} images)")
return "disk"
if needed * 3 <= available:
job.log(f"Image cache: RAM (~{needed / 1e9:.1f} GB of "
f"{available / 1e9:.1f} GB free)")
return "ram"
job.log(f"Image cache: disk (RAM cache would need ~{needed / 1e9:.1f} GB, "
f"only {available / 1e9:.1f} GB free)")
return "disk"
def start(project_id: int, epochs: int = 50, overrides: Optional[dict] = None,
batch_ids: Optional[list] = None, class_ids: Optional[list] = None,
dataset_ids: Optional[list] = None,
base_dataset_ids: Optional[list] = None) -> dict:
project = projects.get(project_id)
if project is None:
raise TrainingError("No such project")
counts = dataset.summary(project_id)["splits"]
bases = list(base_dataset_ids or [])
# No dataset picked means "everything this project has", which is what the
# single-dataset app always did. Base datasets are opt-in, so an empty pick
# never silently drags them in.
chosen = list(dataset_ids or [])
if not chosen and not bases:
chosen = [item["id"] for item in datasets.listing(project_id)]
if not chosen and not bases:
raise TrainingError(
"This project has no dataset yet — approve and merge a batch before training"
)
items = datasets.combined_items(project_id, chosen) if chosen else []
base_train = sum(base_dataset.get(bid)["image_count"] for bid in bases
if base_dataset.get(bid) is not None)
counts = {"train": sum(1 for i in items if i["split"] == "train") + base_train,
"val": sum(1 for i in items if i["split"] == "val")}
if counts["train"] == 0:
raise TrainingError(
"The master dataset is empty — approve and merge a batch before training"
"The chosen dataset(s) hold no training images — merge a batch before training"
)
# A base dataset is train-only, so it can never supply the val split that
# REQ-063's base-vs-new comparison is measured on.
if counts["val"] == 0:
raise TrainingError(
"Nothing to validate on — a base dataset only contributes training images, "
"so pick at least one of this project's own datasets too"
)
settings = hardware.resolve(overrides, epochs)
job = jobs.create(
"train",
params={"project_id": project_id, "settings": settings, "batch_ids": batch_ids, "class_ids": class_ids},
params={"project_id": project_id, "settings": settings, "batch_ids": batch_ids,
"class_ids": class_ids, "dataset_ids": chosen,
"base_dataset_ids": bases},
project_id=project_id,
message=f"{counts['train']} train / {counts['val']} val",
)
@@ -103,7 +168,10 @@ def _run_train(job) -> None:
settings = job.params["settings"]
batch_ids = job.params.get("batch_ids")
class_ids = job.params.get("class_ids")
data_yaml = dataset.write_data_yaml(project, batch_ids=batch_ids, selected_class_ids=class_ids)
dataset_ids = job.params.get("dataset_ids") or []
data_yaml = dataset.write_data_yaml(project, dataset_ids, batch_ids=batch_ids,
selected_class_ids=class_ids, require_val=True,
base_dataset_ids=job.params.get("base_dataset_ids") or [])
# SAM3 and a training run must not hold VRAM at the same time (REQ-065).
from backend.sam3_engine import release_engine
@@ -130,6 +198,8 @@ def _run_train(job) -> None:
epoch = getattr(trainer, 'epoch', 0) + 1
total = getattr(trainer, 'epochs', settings["epochs"])
job.progress(epoch, total, f"epoch {epoch}/{total}")
if job.cancelled:
trainer.stop_training = True
model.add_callback("on_fit_epoch_end", on_epoch)
job.progress(0, settings["epochs"])
@@ -138,6 +208,22 @@ def _run_train(job) -> None:
if torch.cuda.is_available():
torch.backends.cudnn.benchmark = True
# REQ-110: explicit rather than inherited. An untouched project gets MEDIUM,
# which is Ultralytics' own default set, so this changes nothing by itself.
augmentation = augment.get(project["id"])
job.log(f"Augmentation: {augmentation['preset']} — "
+ ", ".join(f"{k}={v:g}" for k, v in sorted(augmentation["settings"].items())))
# `cache="ram"` used to be hardcoded. It holds the whole training set in
# memory, which was invisible at a few hundred images and fatal at fifteen
# thousand: the run below died mid-epoch with no traceback, killed by the
# host OOM killer, because 10 GB of cache plus per-worker copies did not fit
# in 30 GB. RAM caching is now earned, not assumed (CLAUDE.md §9).
train_list = os.path.join(os.path.dirname(data_yaml), "selected_train.txt")
with open(train_list, encoding="utf-8") as handle:
train_images = sum(1 for line in handle if line.strip())
cache_mode = _cache_mode(train_images, job)
keep_run_dir = False
try:
model.train(
@@ -145,9 +231,10 @@ def _run_train(job) -> None:
epochs=settings["epochs"],
imgsz=settings["imgsz"],
batch=settings["batch"],
**augmentation["settings"],
device=settings["device"],
workers=settings.get("workers", 8),
cache="ram",
cache=cache_mode,
project=os.path.join(out_dir, "runs"),
name="train",
exist_ok=True,
@@ -175,10 +262,11 @@ def _run_train(job) -> None:
cur.execute(
"""INSERT INTO model_versions (project_id, version, weights_path,
parent_model_path, metrics, base_metrics,
created_at)
VALUES (?, ?, ?, ?, ?, ?, ?)""",
created_at, augment)
VALUES (?, ?, ?, ?, ?, ?, ?, ?)""",
(project["id"], version, weights, project["base_model_path"],
json.dumps(comparison["new"]), json.dumps(comparison["base"]), time.time()),
json.dumps(comparison["new"]), json.dumps(comparison["base"]), time.time(),
json.dumps(augmentation["settings"])),
)
new = comparison["new"]