This commit includes major additions and updates to the frontend and backend architectures, introducing new dataset management, live counting features, batch processing, and triage logic. Includes new UI pages, components, and API routes.
578 lines
24 KiB
Python
578 lines
24 KiB
Python
"""The master dataset: approved frames merged in, batch after batch (REQ-050…054).
|
|
|
|
The one rule that matters here is the stable val split. A frame's membership is
|
|
recorded once in `dataset_items` and never revised, so an image that was in
|
|
`val` for the last comparison is still in `val` for the next one. Without that,
|
|
a rising mAP could just mean an easier val set.
|
|
|
|
Label files are plain YOLO:
|
|
|
|
detect class_id cx cy w h (normalized)
|
|
segment class_id x1 y1 x2 y2 … (normalized polygon)
|
|
"""
|
|
|
|
import hashlib
|
|
import os
|
|
import shutil
|
|
import time
|
|
from typing import List, Optional
|
|
|
|
from backend import batches, config, datasets, db, jobs, projects, review, triage
|
|
|
|
DatasetError = datasets.DatasetError
|
|
|
|
|
|
def dataset_dir(project_slug: str, dataset_id: int) -> str:
|
|
return datasets.dataset_root(project_slug, dataset_id)
|
|
|
|
|
|
def runs_dir(project_slug: str) -> str:
|
|
"""Where a training run's assembled view lives.
|
|
|
|
It sits outside any one dataset because a run may combine several, and the
|
|
combined view belongs to the run, not to any of its sources.
|
|
"""
|
|
return os.path.join(config.project_dir(project_slug), "runs")
|
|
|
|
|
|
def approve(batch_ids, dataset_id: Optional[int] = None,
|
|
dataset_name: str = "") -> dict:
|
|
"""Sign a selection off and queue one merge into one named dataset (REQ-131).
|
|
|
|
The rules in force right now are frozen onto the target dataset (REQ-132):
|
|
the merge runs under them, and a later rule edit cannot rewrite what this
|
|
dataset claims to be.
|
|
|
|
Without `dataset_id` a new dataset is created, so merging the same batches
|
|
again never collides with the earlier result — it produces a second dataset
|
|
holding them as they look now.
|
|
"""
|
|
ids = triage.as_ids(batch_ids)
|
|
if not ids:
|
|
raise DatasetError("Pick at least one batch to merge")
|
|
selected = []
|
|
for batch_id in ids:
|
|
batch = batches.get(batch_id)
|
|
if batch is None:
|
|
raise DatasetError("No such batch")
|
|
if batch["review"]["approved"] == 0:
|
|
raise DatasetError(
|
|
f"No frame in {batch['date_label']}/{batch['batch_label']} is approved "
|
|
"— there is nothing to merge")
|
|
selected.append(batch)
|
|
if len({batch["project_id"] for batch in selected}) > 1:
|
|
raise DatasetError("Those batches are not all in the same project")
|
|
project_id = selected[0]["project_id"]
|
|
|
|
# Frames that are not approved — rejected or never looked at — are simply
|
|
# left behind. Only what the user signed off on enters the dataset, so a
|
|
# partly-reviewed batch can be merged for the part that is done.
|
|
with db.cursor() as cur:
|
|
placeholders = ",".join("?" for _ in ids)
|
|
cur.execute(
|
|
f"""SELECT 1 FROM jobs WHERE batch_id IN ({placeholders}) AND type = 'merge'
|
|
AND status IN ('queued', 'running')""",
|
|
ids,
|
|
)
|
|
if cur.fetchone() is not None:
|
|
raise DatasetError("A merge for one of these batches is already queued")
|
|
|
|
resolver = triage.Resolver(project_id)
|
|
if dataset_id is None:
|
|
target = datasets.create(project_id, name=dataset_name,
|
|
rule_version=resolver.version(), rules=resolver.rules)
|
|
dataset_id = target["id"]
|
|
else:
|
|
target = datasets.get(dataset_id)
|
|
if target is None:
|
|
raise DatasetError("No such dataset")
|
|
if all(_unmerged_approved(batch["id"], dataset_id) == 0 for batch in selected):
|
|
raise DatasetError(
|
|
f"Every approved frame of this selection is already in \u201c{target['name']}\u201d")
|
|
# An existing dataset keeps the rules it was cut under; a second merge
|
|
# into it must not re-cut the frames already there under new ones.
|
|
if not target["rules"]:
|
|
datasets.snapshot_rules(dataset_id, resolver.rules, resolver.version())
|
|
|
|
for batch in selected:
|
|
batches.set_status(batch["id"], "approved")
|
|
labels = ", ".join(f"{b['date_label']}/{b['batch_label']}" for b in selected)
|
|
job = jobs.create(
|
|
"merge",
|
|
params={"batch_ids": ids, "dataset_id": dataset_id},
|
|
project_id=project_id,
|
|
batch_id=ids[0],
|
|
message=labels,
|
|
)
|
|
return job.to_dict()
|
|
|
|
|
|
def _unmerged_approved(batch_id: int, dataset_id: int) -> int:
|
|
"""Approved frames of this batch not yet in *this* dataset."""
|
|
with db.cursor() as cur:
|
|
cur.execute(
|
|
"""SELECT COUNT(*) FROM frames f
|
|
LEFT JOIN dataset_items d
|
|
ON d.frame_id = f.id AND d.dataset_id = ?
|
|
WHERE f.batch_id = ? AND f.review_status = 'approved' AND d.id IS NULL""",
|
|
(dataset_id, batch_id),
|
|
)
|
|
return cur.fetchone()[0]
|
|
|
|
|
|
def _label_line(class_id: int, geometry: dict, label_type: str) -> str:
|
|
if label_type == "bbox":
|
|
x0, y0, x1, y1 = review.to_box(geometry)
|
|
return (f"{class_id} {(x0 + x1) / 2:.6f} {(y0 + y1) / 2:.6f} "
|
|
f"{x1 - x0:.6f} {y1 - y0:.6f}")
|
|
points = geometry["points"]
|
|
if geometry["type"] == "bbox":
|
|
x0, y0, x1, y1 = geometry["points"]
|
|
points = [[x0, y0], [x1, y0], [x1, y1], [x0, y1]]
|
|
coords = " ".join(f"{value:.6f}" for point in points for value in point)
|
|
return f"{class_id} {coords}"
|
|
|
|
|
|
def split_for(project_id: int, batch_id: int, stem: str, val_every: int) -> str:
|
|
"""Which split a frame belongs to, derived from its identity rather than from
|
|
how many rows happen to precede it.
|
|
|
|
A positional every-Nth rule makes membership depend on insertion history, so
|
|
deleting or re-merging a batch silently reshuffles every later frame — and a
|
|
frame that was in `val` for the last comparison could land in `train` for the
|
|
next one. Hashing the identity makes the stable-val-split invariant true by
|
|
construction: the same frame always lands in the same split, whatever else
|
|
happened to the dataset. Rows already in `dataset_items` keep the split they
|
|
were recorded with; nothing recomputes them.
|
|
"""
|
|
if val_every <= 0:
|
|
return "train"
|
|
digest = hashlib.sha1(f"{project_id}/{batch_id}/{stem}".encode("utf-8")).hexdigest()
|
|
return "val" if int(digest[:8], 16) % val_every == 0 else "train"
|
|
|
|
|
|
def resync(dataset_id: int) -> dict:
|
|
"""Rewrite one dataset's labels from the current annotations and rules.
|
|
|
|
This used to run automatically before every training run, which quietly
|
|
undid the triage applied at merge: a dataset merged under "reclass small
|
|
boxes" had its labels rebuilt from the raw annotations on the next run, so
|
|
the files stopped matching the `rule_version` stamped on them. It is now a
|
|
deliberate act, and it re-stamps that version so the dataset never claims a
|
|
rule set it is not in.
|
|
|
|
Only frames a human signed off on are written. A merged frame whose batch was
|
|
auto-annotated again drops back to `pending`, and rewriting its label from
|
|
fresh model output would push predictions nobody checked into the dataset.
|
|
"""
|
|
target = datasets.get(dataset_id)
|
|
if target is None:
|
|
raise DatasetError("No such dataset")
|
|
project = projects.get(target["project_id"])
|
|
resolver = triage.Resolver(project["id"])
|
|
root = dataset_dir(project["slug"], dataset_id)
|
|
|
|
with db.cursor() as cur:
|
|
cur.execute(
|
|
"""SELECT d.frame_id, d.label_rel FROM dataset_items d
|
|
JOIN frames f ON f.id = d.frame_id
|
|
WHERE d.dataset_id = ? AND f.review_status = 'approved'""",
|
|
(dataset_id,),
|
|
)
|
|
items = cur.fetchall()
|
|
|
|
written = 0
|
|
emptied = 0
|
|
for frame_id, label_rel in items:
|
|
annotations = review.listing(frame_id)
|
|
resolved = resolver.resolve_shapes(annotations)
|
|
if resolved is None:
|
|
# Every shape was dropped. The image stays in the dataset but an
|
|
# empty label would claim it is empty, so the file is left as it was
|
|
# and the count is reported (REQ-104).
|
|
emptied += 1
|
|
continue
|
|
lines = [_label_line(item["class_id"], item["geometry"], project["label_type"])
|
|
for item in resolved]
|
|
path = os.path.join(root, label_rel)
|
|
os.makedirs(os.path.dirname(path), exist_ok=True)
|
|
_write_atomic(path, "\n".join(lines) + ("\n" if lines else ""))
|
|
written += 1
|
|
|
|
# Resync is the one deliberate way an existing dataset adopts today's rules,
|
|
# so the snapshot moves with the labels (REQ-132).
|
|
datasets.snapshot_rules(dataset_id, resolver.rules, resolver.version())
|
|
|
|
return {"labels_written": written, "frames_left_alone": emptied,
|
|
"rule_version": resolver.version()}
|
|
|
|
|
|
def _write_atomic(path: str, text: str) -> None:
|
|
"""Write via temp file + rename, so a training run never reads a half-written
|
|
label file or a truncated data.yaml."""
|
|
tmp = f"{path}.tmp"
|
|
with open(tmp, "w", encoding="utf-8") as handle:
|
|
handle.write(text)
|
|
os.replace(tmp, path)
|
|
|
|
|
|
def _build_selected_tree(run_root: str, rows: list, class_map: Optional[dict]) -> tuple:
|
|
"""Materialise the run's view of the chosen datasets under `runs/selected/`.
|
|
|
|
Labels are copied from what each dataset holds on disk — not re-derived from
|
|
the live annotations. A dataset is the snapshot of a batch as it was merged,
|
|
under the triage rules recorded in its `rule_version`; re-resolving here
|
|
would train on today's rules while the dataset claims yesterday's, and two
|
|
runs over the same dataset could then disagree. Change the rules and merge
|
|
again into a new dataset, or resync this one on purpose.
|
|
|
|
The only thing this does apply is a per-run class filter, which renumbers ids
|
|
into a contiguous 0..k-1 space. That contradicts `project_classes`, so it
|
|
cannot be written back into the dataset's own label files.
|
|
"""
|
|
selected_root = os.path.join(run_root, "selected")
|
|
if os.path.isdir(selected_root):
|
|
shutil.rmtree(selected_root)
|
|
listed = {"train": [], "val": []}
|
|
excluded = 0
|
|
for row in rows:
|
|
split, source_image, source_label = row["split"], row["source_image"], row["source_label"]
|
|
lines = _read_label(source_label)
|
|
|
|
if class_map is not None:
|
|
kept = []
|
|
for line in lines:
|
|
head, _, rest = line.partition(" ")
|
|
try:
|
|
current = int(head)
|
|
except ValueError:
|
|
continue
|
|
if current in class_map:
|
|
kept.append(f"{class_map[current]} {rest}")
|
|
# A frame that had shapes but none of the chosen classes is not a
|
|
# negative sample of those classes — it is a frame full of things the
|
|
# run was told to ignore, and an empty label would teach exactly that.
|
|
if lines and not kept:
|
|
excluded += 1
|
|
continue
|
|
lines = kept
|
|
|
|
stem = os.path.basename(source_image)
|
|
image_dst = os.path.join(selected_root, "images", split, stem)
|
|
label_dst = os.path.join(selected_root, "labels", split,
|
|
os.path.splitext(stem)[0] + ".txt")
|
|
os.makedirs(os.path.dirname(image_dst), exist_ok=True)
|
|
os.makedirs(os.path.dirname(label_dst), exist_ok=True)
|
|
if not os.path.exists(image_dst):
|
|
os.symlink(source_image, image_dst)
|
|
_write_atomic(label_dst, "\n".join(lines) + ("\n" if lines else ""))
|
|
listed[split].append(image_dst)
|
|
listed["excluded"] = excluded
|
|
return selected_root, listed
|
|
|
|
|
|
def _read_label(path: str) -> List[str]:
|
|
if not os.path.isfile(path):
|
|
return []
|
|
with open(path, encoding="utf-8") as handle:
|
|
return [line for line in handle.read().splitlines() if line.strip()]
|
|
|
|
|
|
def write_data_yaml(project: dict, dataset_ids: List[int], batch_ids: list = None,
|
|
selected_class_ids: Optional[List[int]] = None,
|
|
require_val: bool = False,
|
|
base_dataset_ids: Optional[List[int]] = None) -> str:
|
|
"""Assemble the chosen datasets into one data.yaml for a run (REQ-051, REQ-110).
|
|
|
|
Always via the `selected/` tree of symlinks, even for a single dataset with
|
|
no filters. The alternative — pointing YOLO at a dataset folder directly —
|
|
only works while a run uses exactly one dataset, and it puts a per-run class
|
|
renumbering into the shared label files. One assembly path is easier to
|
|
trust than two that diverge the moment a second dataset is picked.
|
|
"""
|
|
if not dataset_ids and not base_dataset_ids:
|
|
raise DatasetError("Pick at least one dataset to train on")
|
|
run_root = runs_dir(project["slug"])
|
|
os.makedirs(run_root, exist_ok=True)
|
|
|
|
target_classes = project["classes"]
|
|
class_map = None
|
|
if selected_class_ids is not None and len(selected_class_ids) > 0:
|
|
target_classes = [c for c in project["classes"] if c["class_id"] in selected_class_ids]
|
|
class_map = {cid: idx for idx, cid in enumerate(sorted(selected_class_ids))}
|
|
|
|
names = ", ".join(f"'{item['name']}'" for item in target_classes)
|
|
|
|
items = datasets.combined_items(project["id"], dataset_ids)
|
|
if batch_ids:
|
|
keep = _frames_of_batches(set(batch_ids))
|
|
items = [item for item in items if item["frame_id"] in keep]
|
|
|
|
rows = []
|
|
for item in items:
|
|
root = dataset_dir(project["slug"], item["dataset_id"])
|
|
rows.append({
|
|
"frame_id": item["frame_id"],
|
|
"split": item["split"],
|
|
"source_image": os.path.join(root, item["image_rel"]),
|
|
"source_label": os.path.join(root, item["label_rel"]),
|
|
})
|
|
|
|
# Base datasets are appended, never merged into the dedupe above: they carry
|
|
# no frame_id, and they are always train-only (REQ-122).
|
|
if base_dataset_ids:
|
|
from backend import base_dataset
|
|
rows.extend(base_dataset.rows(project["id"], base_dataset_ids, project["slug"]))
|
|
|
|
selected_root, listed = _build_selected_tree(run_root, rows, class_map)
|
|
if require_val:
|
|
_require_val(len(listed["val"]), "the selected dataset(s)")
|
|
|
|
train_txt = os.path.join(run_root, "selected_train.txt")
|
|
val_txt = os.path.join(run_root, "selected_val.txt")
|
|
_write_atomic(train_txt, "\n".join(listed["train"]) + "\n")
|
|
_write_atomic(val_txt, "\n".join(listed["val"]) + "\n")
|
|
|
|
path = os.path.join(run_root, "selected_data.yaml")
|
|
_write_atomic(path,
|
|
f"path: {selected_root}\n"
|
|
f"train: {train_txt}\n"
|
|
f"val: {val_txt}\n\n"
|
|
f"nc: {len(target_classes)}\n"
|
|
f"names: [{names}]\n")
|
|
return path
|
|
|
|
|
|
def _frames_of_batches(batch_ids: set) -> set:
|
|
with db.cursor() as cur:
|
|
cur.execute(
|
|
f"SELECT id FROM frames WHERE batch_id IN ({','.join('?' for _ in batch_ids)})",
|
|
list(batch_ids),
|
|
)
|
|
return {row[0] for row in cur.fetchall()}
|
|
|
|
|
|
def _require_val(count: int, subject: str) -> None:
|
|
"""Refuse to build a dataset with an empty val split.
|
|
|
|
Falling back to the training images produces a base-vs-new mAP measured on
|
|
data the model was fitted to — a number that looks fine and means nothing.
|
|
For a system whose whole purpose is answering "did retraining help?", this
|
|
has to fail loudly.
|
|
"""
|
|
if count == 0:
|
|
raise DatasetError(
|
|
f"There are no validation images in {subject}, so a base-vs-new comparison "
|
|
"would be measured on the training images. Merge more frames, or lower the "
|
|
"project's val_every."
|
|
)
|
|
|
|
|
|
def summary(project_id: int) -> dict:
|
|
"""Counts only. The per-shape size analytics that used to live here walked
|
|
every annotation in the project on every page load — Data Prep already
|
|
serves that, per batch, from `triage`."""
|
|
with db.cursor() as cur:
|
|
cur.execute(
|
|
"SELECT split, COUNT(*) FROM dataset_items WHERE project_id = ? GROUP BY split",
|
|
(project_id,),
|
|
)
|
|
splits = {"train": 0, "val": 0}
|
|
for split, count in cur.fetchall():
|
|
splits[split] = count
|
|
cur.execute(
|
|
"""SELECT b.id, b.date_label, b.batch_label, b.merged_at,
|
|
COUNT(d.id) AS images
|
|
FROM batches b
|
|
LEFT JOIN frames f ON f.batch_id = b.id
|
|
LEFT JOIN dataset_items d ON d.frame_id = f.id
|
|
WHERE b.project_id = ? AND b.status = 'merged'
|
|
GROUP BY b.id ORDER BY b.merged_at""",
|
|
(project_id,),
|
|
)
|
|
merged = [dict(row) for row in cur.fetchall()]
|
|
|
|
return {
|
|
"splits": splits,
|
|
"total": splits["train"] + splits["val"],
|
|
"batches": merged,
|
|
}
|
|
|
|
|
|
def drop_class_from_labels(project: dict, class_id: int) -> dict:
|
|
"""Rewrite every label file on disk after a class is deleted (REQ-007).
|
|
|
|
Two edits per file: lines of the deleted class go, and every id above it
|
|
comes down by one. Skipping this would leave `2` in old files meaning a
|
|
class that is now `1` — labels that quietly name the wrong thing are worse
|
|
than labels that are missing.
|
|
"""
|
|
with db.cursor() as cur:
|
|
cur.execute("SELECT label_rel, dataset_id FROM dataset_items WHERE project_id = ?",
|
|
(project["id"],))
|
|
label_files = [(row[0], row[1]) for row in cur.fetchall()]
|
|
|
|
rewritten = 0
|
|
dropped = 0
|
|
for rel, dataset_id in label_files:
|
|
path = os.path.join(dataset_dir(project["slug"], dataset_id), rel)
|
|
if not os.path.isfile(path):
|
|
continue
|
|
with open(path, encoding="utf-8") as handle:
|
|
lines = handle.read().splitlines()
|
|
|
|
kept, touched = [], False
|
|
for line in lines:
|
|
if not line.strip():
|
|
continue
|
|
head, _, rest = line.partition(" ")
|
|
try:
|
|
current = int(head)
|
|
except ValueError:
|
|
kept.append(line)
|
|
continue
|
|
if current == class_id:
|
|
dropped += 1
|
|
touched = True
|
|
continue
|
|
if current > class_id:
|
|
current -= 1
|
|
touched = True
|
|
kept.append(f"{current} {rest}")
|
|
|
|
if touched:
|
|
# An emptied file stays as an empty file: the image is still a valid
|
|
# negative sample (REQ-033), it just has nothing on it any more.
|
|
with open(path, "w", encoding="utf-8") as handle:
|
|
handle.write("\n".join(kept) + ("\n" if kept else ""))
|
|
rewritten += 1
|
|
|
|
return {"label_files_rewritten": rewritten, "dataset_lines_removed": dropped}
|
|
|
|
|
|
def zip_path(project: dict, dataset_id: int) -> str:
|
|
"""Zip one dataset for download (REQ-054)."""
|
|
root = dataset_dir(project["slug"], dataset_id)
|
|
if not os.path.isdir(os.path.join(root, "images")):
|
|
raise DatasetError("This dataset is still empty")
|
|
archive = os.path.join(config.project_dir(project["slug"]), f"dataset-{dataset_id}")
|
|
return shutil.make_archive(archive, "zip", root)
|
|
|
|
|
|
@jobs.handler("merge")
|
|
def _run_merge(job) -> None:
|
|
ids = job.params.get("batch_ids") or [job.params["batch_id"]]
|
|
selected = [batches.get(bid) for bid in ids]
|
|
if any(batch is None for batch in selected):
|
|
raise DatasetError("A batch disappeared before the merge started")
|
|
project = projects.get(selected[0]["project_id"])
|
|
dataset_id = job.params["dataset_id"]
|
|
target = datasets.get(dataset_id)
|
|
if target is None:
|
|
raise DatasetError("The target dataset disappeared before the merge started")
|
|
root = dataset_dir(project["slug"], dataset_id)
|
|
for split in ("train", "val"):
|
|
os.makedirs(os.path.join(root, "images", split), exist_ok=True)
|
|
os.makedirs(os.path.join(root, "labels", split), exist_ok=True)
|
|
|
|
work = []
|
|
for batch in selected:
|
|
frames = [f for f in batches.frames(batch["id"]) if f["review_status"] == "approved"]
|
|
work.extend((batch, frame) for frame in frames)
|
|
job.progress(0, len(work))
|
|
job.log(f"Merging {len(work)} approved frame(s) from {len(selected)} batch(es) "
|
|
f"into \u201c{target['name']}\u201d")
|
|
|
|
# Triage gates the merge (REQ-104), under the rules frozen onto this dataset
|
|
# when it was created (REQ-132) — not under whatever the project says now.
|
|
resolver = triage.Resolver(project["id"], frozen=target["rules"])
|
|
gating = bool(resolver.rules or resolver.overrides)
|
|
if gating:
|
|
job.log(f"Applying {len(resolver.rules)} triage rule(s), version {resolver.version()}")
|
|
|
|
added = {"train": 0, "val": 0}
|
|
skipped = 0
|
|
triaged_out = 0
|
|
cancelled = False
|
|
for index, (batch, frame) in enumerate(work):
|
|
if job.cancelled:
|
|
job.log(f"Cancelled after {index} frame(s)")
|
|
cancelled = True
|
|
break
|
|
|
|
annotations = review.listing(frame["id"])
|
|
if gating:
|
|
resolved = resolver.resolve_shapes(annotations)
|
|
if resolved is None:
|
|
triaged_out += 1
|
|
job.progress(index + 1, len(work))
|
|
continue
|
|
annotations = resolved
|
|
|
|
with db.cursor() as cur:
|
|
cur.execute("SELECT 1 FROM dataset_items WHERE dataset_id = ? AND frame_id = ?",
|
|
(dataset_id, frame["id"]))
|
|
if cur.fetchone() is not None:
|
|
skipped += 1
|
|
job.progress(index + 1, len(work))
|
|
continue
|
|
|
|
stem = f"{batch['id']}__{os.path.splitext(frame['filename'])[0]}"
|
|
# A frame's split is decided once for the whole project and every
|
|
# later dataset inherits it. Letting each dataset re-decide would put
|
|
# the same image in `val` for one run and `train` for the next, so a
|
|
# base-vs-new mAP would be measured on images the new model had been
|
|
# fitted to. The hash agrees with itself, but rows merged before the
|
|
# hash existed carry a positional split — those have to be honoured,
|
|
# not recomputed.
|
|
cur.execute(
|
|
"SELECT split FROM dataset_items WHERE frame_id = ? LIMIT 1",
|
|
(frame["id"],),
|
|
)
|
|
previous = cur.fetchone()
|
|
split = previous[0] if previous else split_for(
|
|
project["id"], batch["id"], stem, project["val_every"])
|
|
image_rel = f"images/{split}/{stem}.jpg"
|
|
label_rel = f"labels/{split}/{stem}.txt"
|
|
shutil.copyfile(
|
|
os.path.join(batches.frames_dir(project["slug"], batch["id"]), frame["filename"]),
|
|
os.path.join(root, image_rel))
|
|
|
|
lines = [_label_line(item["class_id"], item["geometry"], project["label_type"])
|
|
for item in annotations]
|
|
# An approved frame with nothing on it is a negative sample, and an
|
|
# empty .txt is how YOLO spells that (REQ-033).
|
|
with open(os.path.join(root, label_rel), "w", encoding="utf-8") as handle:
|
|
handle.write("\n".join(lines) + ("\n" if lines else ""))
|
|
|
|
cur.execute(
|
|
"""INSERT INTO dataset_items (project_id, dataset_id, frame_id, split,
|
|
image_rel, label_rel, added_at)
|
|
VALUES (?, ?, ?, ?, ?, ?, ?)""",
|
|
(project["id"], dataset_id, frame["id"], split, image_rel, label_rel,
|
|
time.time()),
|
|
)
|
|
added[split] += 1
|
|
job.progress(index + 1, len(work))
|
|
|
|
if cancelled:
|
|
# Leaving them 'merged' would be a lie: the frames after the break point
|
|
# have no dataset_items rows and no files, and approve() refuses to
|
|
# re-merge a merged batch, so they could never be added. The per-frame
|
|
# dataset_items guard already makes re-running the merge idempotent.
|
|
job.log("Batches left approved — re-approve them to finish the merge")
|
|
return
|
|
|
|
with db.cursor() as cur:
|
|
cur.executemany(
|
|
"UPDATE batches SET status = 'merged', merged_at = ? WHERE id = ?",
|
|
[(time.time(), bid) for bid in ids],
|
|
)
|
|
|
|
totals = datasets.get(dataset_id)["splits"]
|
|
job.log(f"Added {added['train']} train / {added['val']} val"
|
|
+ (f", skipped {skipped} already in this dataset" if skipped else "")
|
|
+ (f", held back {triaged_out} by triage" if triaged_out else ""))
|
|
job.log(f"\u201c{target['name']}\u201d now holds "
|
|
f"{totals['train']} train / {totals['val']} val")
|