"""Named dataset routes: list, rename, delete, download (REQ-110…113).""" import os from typing import List, Optional from fastapi import APIRouter, HTTPException from fastapi.responses import FileResponse from pydantic import BaseModel from backend import dataset, datasets from backend.api.common import project_or_404 router = APIRouter(tags=["datasets"]) class CreateRequest(BaseModel): name: str = "" note: str = "" class RenameRequest(BaseModel): name: Optional[str] = None note: Optional[str] = None class CombineRequest(BaseModel): dataset_ids: List[int] @router.get("/api/projects/{project_id}/datasets") def list_datasets(project_id: int) -> dict: project_or_404(project_id) return {"datasets": datasets.listing(project_id)} @router.post("/api/projects/{project_id}/datasets") def create_dataset(project_id: int, body: CreateRequest) -> dict: project_or_404(project_id) return datasets.create(project_id, name=body.name, note=body.note) @router.get("/api/datasets/{dataset_id}") def get_dataset(dataset_id: int) -> dict: found = datasets.get(dataset_id) if found is None: raise HTTPException(404, "No such dataset") return found @router.patch("/api/datasets/{dataset_id}") def rename_dataset(dataset_id: int, body: RenameRequest) -> dict: if datasets.get(dataset_id) is None: raise HTTPException(404, "No such dataset") return datasets.rename(dataset_id, name=body.name, note=body.note) @router.delete("/api/datasets/{dataset_id}") def delete_dataset(dataset_id: int) -> dict: if not datasets.delete(dataset_id): raise HTTPException(404, "No such dataset") return {"deleted": True} @router.post("/api/projects/{project_id}/datasets/combine-preview") def combine_preview(project_id: int, body: CombineRequest) -> dict: """What a run over these datasets would actually see. The totals of two datasets do not add up when they share frames, and being handed 4,000 images after picking two datasets of 3,000 is the kind of surprise that makes people distrust the numbers. """ project_or_404(project_id) items = datasets.combined_items(project_id, body.dataset_ids) report = datasets.overlap_report(project_id, body.dataset_ids, len(items)) report["splits"] = { "train": sum(1 for item in items if item["split"] == "train"), "val": sum(1 for item in items if item["split"] == "val"), } return report @router.post("/api/datasets/{dataset_id}/resync") def resync_dataset(dataset_id: int) -> dict: try: return dataset.resync(dataset_id) except datasets.DatasetError as exc: raise HTTPException(400, str(exc)) @router.get("/api/datasets/{dataset_id}/download") def download_dataset(dataset_id: int): found = datasets.get(dataset_id) if found is None: raise HTTPException(404, "No such dataset") project = project_or_404(found["project_id"]) try: path = dataset.zip_path(project, dataset_id) except datasets.DatasetError as exc: raise HTTPException(400, str(exc)) return FileResponse(path, media_type="application/zip", filename=os.path.basename(path))