"""The master dataset: approved frames merged in, batch after batch (REQ-050…054). The one rule that matters here is the stable val split. A frame's membership is recorded once in `dataset_items` and never revised, so an image that was in `val` for the last comparison is still in `val` for the next one. Without that, a rising mAP could just mean an easier val set. Label files are plain YOLO: detect class_id cx cy w h (normalized) segment class_id x1 y1 x2 y2 … (normalized polygon) """ import hashlib import os import shutil import time from typing import List, Optional from backend import batches, config, datasets, db, jobs, projects, review, triage DatasetError = datasets.DatasetError def dataset_dir(project_slug: str, dataset_id: int) -> str: return datasets.dataset_root(project_slug, dataset_id) def runs_dir(project_slug: str) -> str: """Where a training run's assembled view lives. It sits outside any one dataset because a run may combine several, and the combined view belongs to the run, not to any of its sources. """ return os.path.join(config.project_dir(project_slug), "runs") def approve(batch_ids, dataset_id: Optional[int] = None, dataset_name: str = "") -> dict: """Sign a selection off and queue one merge into one named dataset (REQ-131). The rules in force right now are frozen onto the target dataset (REQ-132): the merge runs under them, and a later rule edit cannot rewrite what this dataset claims to be. Without `dataset_id` a new dataset is created, so merging the same batches again never collides with the earlier result — it produces a second dataset holding them as they look now. """ ids = triage.as_ids(batch_ids) if not ids: raise DatasetError("Pick at least one batch to merge") selected = [] for batch_id in ids: batch = batches.get(batch_id) if batch is None: raise DatasetError("No such batch") if batch["review"]["approved"] == 0: raise DatasetError( f"No frame in {batch['date_label']}/{batch['batch_label']} is approved " "— there is nothing to merge") selected.append(batch) if len({batch["project_id"] for batch in selected}) > 1: raise DatasetError("Those batches are not all in the same project") project_id = selected[0]["project_id"] # Frames that are not approved — rejected or never looked at — are simply # left behind. Only what the user signed off on enters the dataset, so a # partly-reviewed batch can be merged for the part that is done. with db.cursor() as cur: placeholders = ",".join("?" for _ in ids) cur.execute( f"""SELECT 1 FROM jobs WHERE batch_id IN ({placeholders}) AND type = 'merge' AND status IN ('queued', 'running')""", ids, ) if cur.fetchone() is not None: raise DatasetError("A merge for one of these batches is already queued") resolver = triage.Resolver(project_id) if dataset_id is None: target = datasets.create(project_id, name=dataset_name, rule_version=resolver.version(), rules=resolver.rules) dataset_id = target["id"] else: target = datasets.get(dataset_id) if target is None: raise DatasetError("No such dataset") if all(_unmerged_approved(batch["id"], dataset_id) == 0 for batch in selected): raise DatasetError( f"Every approved frame of this selection is already in \u201c{target['name']}\u201d") # An existing dataset keeps the rules it was cut under; a second merge # into it must not re-cut the frames already there under new ones. if not target["rules"]: datasets.snapshot_rules(dataset_id, resolver.rules, resolver.version()) for batch in selected: batches.set_status(batch["id"], "approved") labels = ", ".join(f"{b['date_label']}/{b['batch_label']}" for b in selected) job = jobs.create( "merge", params={"batch_ids": ids, "dataset_id": dataset_id}, project_id=project_id, batch_id=ids[0], message=labels, ) return job.to_dict() def _unmerged_approved(batch_id: int, dataset_id: int) -> int: """Approved frames of this batch not yet in *this* dataset.""" with db.cursor() as cur: cur.execute( """SELECT COUNT(*) FROM frames f LEFT JOIN dataset_items d ON d.frame_id = f.id AND d.dataset_id = ? WHERE f.batch_id = ? AND f.review_status = 'approved' AND d.id IS NULL""", (dataset_id, batch_id), ) return cur.fetchone()[0] def _label_line(class_id: int, geometry: dict, label_type: str) -> str: if label_type == "bbox": x0, y0, x1, y1 = review.to_box(geometry) return (f"{class_id} {(x0 + x1) / 2:.6f} {(y0 + y1) / 2:.6f} " f"{x1 - x0:.6f} {y1 - y0:.6f}") points = geometry["points"] if geometry["type"] == "bbox": x0, y0, x1, y1 = geometry["points"] points = [[x0, y0], [x1, y0], [x1, y1], [x0, y1]] coords = " ".join(f"{value:.6f}" for point in points for value in point) return f"{class_id} {coords}" def split_for(project_id: int, batch_id: int, stem: str, val_every: int) -> str: """Which split a frame belongs to, derived from its identity rather than from how many rows happen to precede it. A positional every-Nth rule makes membership depend on insertion history, so deleting or re-merging a batch silently reshuffles every later frame — and a frame that was in `val` for the last comparison could land in `train` for the next one. Hashing the identity makes the stable-val-split invariant true by construction: the same frame always lands in the same split, whatever else happened to the dataset. Rows already in `dataset_items` keep the split they were recorded with; nothing recomputes them. """ if val_every <= 0: return "train" digest = hashlib.sha1(f"{project_id}/{batch_id}/{stem}".encode("utf-8")).hexdigest() return "val" if int(digest[:8], 16) % val_every == 0 else "train" def resync(dataset_id: int) -> dict: """Rewrite one dataset's labels from the current annotations and rules. This used to run automatically before every training run, which quietly undid the triage applied at merge: a dataset merged under "reclass small boxes" had its labels rebuilt from the raw annotations on the next run, so the files stopped matching the `rule_version` stamped on them. It is now a deliberate act, and it re-stamps that version so the dataset never claims a rule set it is not in. Only frames a human signed off on are written. A merged frame whose batch was auto-annotated again drops back to `pending`, and rewriting its label from fresh model output would push predictions nobody checked into the dataset. """ target = datasets.get(dataset_id) if target is None: raise DatasetError("No such dataset") project = projects.get(target["project_id"]) resolver = triage.Resolver(project["id"]) root = dataset_dir(project["slug"], dataset_id) with db.cursor() as cur: cur.execute( """SELECT d.frame_id, d.label_rel FROM dataset_items d JOIN frames f ON f.id = d.frame_id WHERE d.dataset_id = ? AND f.review_status = 'approved'""", (dataset_id,), ) items = cur.fetchall() written = 0 emptied = 0 for frame_id, label_rel in items: annotations = review.listing(frame_id) resolved = resolver.resolve_shapes(annotations) if resolved is None: # Every shape was dropped. The image stays in the dataset but an # empty label would claim it is empty, so the file is left as it was # and the count is reported (REQ-104). emptied += 1 continue lines = [_label_line(item["class_id"], item["geometry"], project["label_type"]) for item in resolved] path = os.path.join(root, label_rel) os.makedirs(os.path.dirname(path), exist_ok=True) _write_atomic(path, "\n".join(lines) + ("\n" if lines else "")) written += 1 # Resync is the one deliberate way an existing dataset adopts today's rules, # so the snapshot moves with the labels (REQ-132). datasets.snapshot_rules(dataset_id, resolver.rules, resolver.version()) return {"labels_written": written, "frames_left_alone": emptied, "rule_version": resolver.version()} def _write_atomic(path: str, text: str) -> None: """Write via temp file + rename, so a training run never reads a half-written label file or a truncated data.yaml.""" tmp = f"{path}.tmp" with open(tmp, "w", encoding="utf-8") as handle: handle.write(text) os.replace(tmp, path) def _build_selected_tree(run_root: str, rows: list, class_map: Optional[dict]) -> tuple: """Materialise the run's view of the chosen datasets under `runs/selected/`. Labels are copied from what each dataset holds on disk — not re-derived from the live annotations. A dataset is the snapshot of a batch as it was merged, under the triage rules recorded in its `rule_version`; re-resolving here would train on today's rules while the dataset claims yesterday's, and two runs over the same dataset could then disagree. Change the rules and merge again into a new dataset, or resync this one on purpose. The only thing this does apply is a per-run class filter, which renumbers ids into a contiguous 0..k-1 space. That contradicts `project_classes`, so it cannot be written back into the dataset's own label files. """ selected_root = os.path.join(run_root, "selected") if os.path.isdir(selected_root): shutil.rmtree(selected_root) listed = {"train": [], "val": []} excluded = 0 for row in rows: split, source_image, source_label = row["split"], row["source_image"], row["source_label"] lines = _read_label(source_label) if class_map is not None: kept = [] for line in lines: head, _, rest = line.partition(" ") try: current = int(head) except ValueError: continue if current in class_map: kept.append(f"{class_map[current]} {rest}") # A frame that had shapes but none of the chosen classes is not a # negative sample of those classes — it is a frame full of things the # run was told to ignore, and an empty label would teach exactly that. if lines and not kept: excluded += 1 continue lines = kept stem = os.path.basename(source_image) image_dst = os.path.join(selected_root, "images", split, stem) label_dst = os.path.join(selected_root, "labels", split, os.path.splitext(stem)[0] + ".txt") os.makedirs(os.path.dirname(image_dst), exist_ok=True) os.makedirs(os.path.dirname(label_dst), exist_ok=True) if not os.path.exists(image_dst): os.symlink(source_image, image_dst) _write_atomic(label_dst, "\n".join(lines) + ("\n" if lines else "")) listed[split].append(image_dst) listed["excluded"] = excluded return selected_root, listed def _read_label(path: str) -> List[str]: if not os.path.isfile(path): return [] with open(path, encoding="utf-8") as handle: return [line for line in handle.read().splitlines() if line.strip()] def write_data_yaml(project: dict, dataset_ids: List[int], batch_ids: list = None, selected_class_ids: Optional[List[int]] = None, require_val: bool = False, base_dataset_ids: Optional[List[int]] = None) -> str: """Assemble the chosen datasets into one data.yaml for a run (REQ-051, REQ-110). Always via the `selected/` tree of symlinks, even for a single dataset with no filters. The alternative — pointing YOLO at a dataset folder directly — only works while a run uses exactly one dataset, and it puts a per-run class renumbering into the shared label files. One assembly path is easier to trust than two that diverge the moment a second dataset is picked. """ if not dataset_ids and not base_dataset_ids: raise DatasetError("Pick at least one dataset to train on") run_root = runs_dir(project["slug"]) os.makedirs(run_root, exist_ok=True) target_classes = project["classes"] class_map = None if selected_class_ids is not None and len(selected_class_ids) > 0: target_classes = [c for c in project["classes"] if c["class_id"] in selected_class_ids] class_map = {cid: idx for idx, cid in enumerate(sorted(selected_class_ids))} names = ", ".join(f"'{item['name']}'" for item in target_classes) items = datasets.combined_items(project["id"], dataset_ids) if batch_ids: keep = _frames_of_batches(set(batch_ids)) items = [item for item in items if item["frame_id"] in keep] rows = [] for item in items: root = dataset_dir(project["slug"], item["dataset_id"]) rows.append({ "frame_id": item["frame_id"], "split": item["split"], "source_image": os.path.join(root, item["image_rel"]), "source_label": os.path.join(root, item["label_rel"]), }) # Base datasets are appended, never merged into the dedupe above: they carry # no frame_id, and they are always train-only (REQ-122). if base_dataset_ids: from backend import base_dataset rows.extend(base_dataset.rows(project["id"], base_dataset_ids, project["slug"])) selected_root, listed = _build_selected_tree(run_root, rows, class_map) if require_val: _require_val(len(listed["val"]), "the selected dataset(s)") train_txt = os.path.join(run_root, "selected_train.txt") val_txt = os.path.join(run_root, "selected_val.txt") _write_atomic(train_txt, "\n".join(listed["train"]) + "\n") _write_atomic(val_txt, "\n".join(listed["val"]) + "\n") path = os.path.join(run_root, "selected_data.yaml") _write_atomic(path, f"path: {selected_root}\n" f"train: {train_txt}\n" f"val: {val_txt}\n\n" f"nc: {len(target_classes)}\n" f"names: [{names}]\n") return path def _frames_of_batches(batch_ids: set) -> set: with db.cursor() as cur: cur.execute( f"SELECT id FROM frames WHERE batch_id IN ({','.join('?' for _ in batch_ids)})", list(batch_ids), ) return {row[0] for row in cur.fetchall()} def _require_val(count: int, subject: str) -> None: """Refuse to build a dataset with an empty val split. Falling back to the training images produces a base-vs-new mAP measured on data the model was fitted to — a number that looks fine and means nothing. For a system whose whole purpose is answering "did retraining help?", this has to fail loudly. """ if count == 0: raise DatasetError( f"There are no validation images in {subject}, so a base-vs-new comparison " "would be measured on the training images. Merge more frames, or lower the " "project's val_every." ) def summary(project_id: int) -> dict: """Counts only. The per-shape size analytics that used to live here walked every annotation in the project on every page load — Data Prep already serves that, per batch, from `triage`.""" with db.cursor() as cur: cur.execute( "SELECT split, COUNT(*) FROM dataset_items WHERE project_id = ? GROUP BY split", (project_id,), ) splits = {"train": 0, "val": 0} for split, count in cur.fetchall(): splits[split] = count cur.execute( """SELECT b.id, b.date_label, b.batch_label, b.merged_at, COUNT(d.id) AS images FROM batches b LEFT JOIN frames f ON f.batch_id = b.id LEFT JOIN dataset_items d ON d.frame_id = f.id WHERE b.project_id = ? AND b.status = 'merged' GROUP BY b.id ORDER BY b.merged_at""", (project_id,), ) merged = [dict(row) for row in cur.fetchall()] return { "splits": splits, "total": splits["train"] + splits["val"], "batches": merged, } def drop_class_from_labels(project: dict, class_id: int) -> dict: """Rewrite every label file on disk after a class is deleted (REQ-007). Two edits per file: lines of the deleted class go, and every id above it comes down by one. Skipping this would leave `2` in old files meaning a class that is now `1` — labels that quietly name the wrong thing are worse than labels that are missing. """ with db.cursor() as cur: cur.execute("SELECT label_rel, dataset_id FROM dataset_items WHERE project_id = ?", (project["id"],)) label_files = [(row[0], row[1]) for row in cur.fetchall()] rewritten = 0 dropped = 0 for rel, dataset_id in label_files: path = os.path.join(dataset_dir(project["slug"], dataset_id), rel) if not os.path.isfile(path): continue with open(path, encoding="utf-8") as handle: lines = handle.read().splitlines() kept, touched = [], False for line in lines: if not line.strip(): continue head, _, rest = line.partition(" ") try: current = int(head) except ValueError: kept.append(line) continue if current == class_id: dropped += 1 touched = True continue if current > class_id: current -= 1 touched = True kept.append(f"{current} {rest}") if touched: # An emptied file stays as an empty file: the image is still a valid # negative sample (REQ-033), it just has nothing on it any more. with open(path, "w", encoding="utf-8") as handle: handle.write("\n".join(kept) + ("\n" if kept else "")) rewritten += 1 return {"label_files_rewritten": rewritten, "dataset_lines_removed": dropped} def zip_path(project: dict, dataset_id: int) -> str: """Zip one dataset for download (REQ-054).""" root = dataset_dir(project["slug"], dataset_id) if not os.path.isdir(os.path.join(root, "images")): raise DatasetError("This dataset is still empty") archive = os.path.join(config.project_dir(project["slug"]), f"dataset-{dataset_id}") return shutil.make_archive(archive, "zip", root) @jobs.handler("merge") def _run_merge(job) -> None: ids = job.params.get("batch_ids") or [job.params["batch_id"]] selected = [batches.get(bid) for bid in ids] if any(batch is None for batch in selected): raise DatasetError("A batch disappeared before the merge started") project = projects.get(selected[0]["project_id"]) dataset_id = job.params["dataset_id"] target = datasets.get(dataset_id) if target is None: raise DatasetError("The target dataset disappeared before the merge started") root = dataset_dir(project["slug"], dataset_id) for split in ("train", "val"): os.makedirs(os.path.join(root, "images", split), exist_ok=True) os.makedirs(os.path.join(root, "labels", split), exist_ok=True) work = [] for batch in selected: frames = [f for f in batches.frames(batch["id"]) if f["review_status"] == "approved"] work.extend((batch, frame) for frame in frames) job.progress(0, len(work)) job.log(f"Merging {len(work)} approved frame(s) from {len(selected)} batch(es) " f"into \u201c{target['name']}\u201d") # Triage gates the merge (REQ-104), under the rules frozen onto this dataset # when it was created (REQ-132) — not under whatever the project says now. resolver = triage.Resolver(project["id"], frozen=target["rules"]) gating = bool(resolver.rules or resolver.overrides) if gating: job.log(f"Applying {len(resolver.rules)} triage rule(s), version {resolver.version()}") added = {"train": 0, "val": 0} skipped = 0 triaged_out = 0 cancelled = False for index, (batch, frame) in enumerate(work): if job.cancelled: job.log(f"Cancelled after {index} frame(s)") cancelled = True break annotations = review.listing(frame["id"]) if gating: resolved = resolver.resolve_shapes(annotations) if resolved is None: triaged_out += 1 job.progress(index + 1, len(work)) continue annotations = resolved with db.cursor() as cur: cur.execute("SELECT 1 FROM dataset_items WHERE dataset_id = ? AND frame_id = ?", (dataset_id, frame["id"])) if cur.fetchone() is not None: skipped += 1 job.progress(index + 1, len(work)) continue stem = f"{batch['id']}__{os.path.splitext(frame['filename'])[0]}" # A frame's split is decided once for the whole project and every # later dataset inherits it. Letting each dataset re-decide would put # the same image in `val` for one run and `train` for the next, so a # base-vs-new mAP would be measured on images the new model had been # fitted to. The hash agrees with itself, but rows merged before the # hash existed carry a positional split — those have to be honoured, # not recomputed. cur.execute( "SELECT split FROM dataset_items WHERE frame_id = ? LIMIT 1", (frame["id"],), ) previous = cur.fetchone() split = previous[0] if previous else split_for( project["id"], batch["id"], stem, project["val_every"]) image_rel = f"images/{split}/{stem}.jpg" label_rel = f"labels/{split}/{stem}.txt" shutil.copyfile( os.path.join(batches.frames_dir(project["slug"], batch["id"]), frame["filename"]), os.path.join(root, image_rel)) lines = [_label_line(item["class_id"], item["geometry"], project["label_type"]) for item in annotations] # An approved frame with nothing on it is a negative sample, and an # empty .txt is how YOLO spells that (REQ-033). with open(os.path.join(root, label_rel), "w", encoding="utf-8") as handle: handle.write("\n".join(lines) + ("\n" if lines else "")) cur.execute( """INSERT INTO dataset_items (project_id, dataset_id, frame_id, split, image_rel, label_rel, added_at) VALUES (?, ?, ?, ?, ?, ?, ?)""", (project["id"], dataset_id, frame["id"], split, image_rel, label_rel, time.time()), ) added[split] += 1 job.progress(index + 1, len(work)) if cancelled: # Leaving them 'merged' would be a lie: the frames after the break point # have no dataset_items rows and no files, and approve() refuses to # re-merge a merged batch, so they could never be added. The per-frame # dataset_items guard already makes re-running the merge idempotent. job.log("Batches left approved — re-approve them to finish the merge") return with db.cursor() as cur: cur.executemany( "UPDATE batches SET status = 'merged', merged_at = ? WHERE id = ?", [(time.time(), bid) for bid in ids], ) totals = datasets.get(dataset_id)["splits"] job.log(f"Added {added['train']} train / {added['val']} val" + (f", skipped {skipped} already in this dataset" if skipped else "") + (f", held back {triaged_out} by triage" if triaged_out else "")) job.log(f"\u201c{target['name']}\u201d now holds " f"{totals['train']} train / {totals['val']} val")