Files
reTraining/algoritma-batch/migrate_cutoff_0600.py
T
asus 5c7c122105 feat: add counting bench, triage, and dataset modules
This commit includes major additions and updates to the frontend and backend architectures, introducing new dataset management, live counting features, batch processing, and triage logic. Includes new UI pages, components, and API routes.
2026-08-14 16:28:52 +07:00

153 lines
5.7 KiB
Python

"""Re-file the Jetson's stored batches under the 06:00 counting day.
`predict.py` used to turn the counting day over at 20:00, which filed the whole
day shift under the previous date — a batch that started at 08:27 on the 13th was
stored as the 12th. The archive's cycles run 06:00 to 06:00, so the two disagreed.
The default is now 06:00; this brings rows written before that change into line.
For every batch it recomputes `counting_date` from `start_time`, renumbers
`batch_number` 1..N within each counting day (per camera and object label,
ordered by start time), and rebuilds `daily_summaries` from the result.
Run it on the Jetson, against its own database:
python migrate_cutoff_0600.py --db /opt/jetson-counter/jetson_counter.db --dry-run
python migrate_cutoff_0600.py --db /opt/jetson-counter/jetson_counter.db
Nothing is written without a backup, and `--dry-run` writes nothing at all.
Stop `predict.py` first: it holds an active batch in memory and would write it
back under the old numbering.
"""
import argparse
import datetime
import os
import shutil
import sqlite3
import sys
def counting_date(start_time: str, cutoff_hour: int) -> str:
"""The counting day a batch belongs to, from when it started."""
stamp = datetime.datetime.fromisoformat(start_time)
day = stamp.date()
if stamp.hour < cutoff_hour:
day = day - datetime.timedelta(days=1)
return day.isoformat()
def plan(connection, cutoff_hour: int):
"""What each row should become. Ordered by start time inside each day."""
rows = connection.execute(
"""SELECT id, counting_date, batch_number, camera_name, object_label,
count, start_time
FROM batches ORDER BY start_time"""
).fetchall()
counters: dict = {}
changes = []
for row in rows:
try:
new_date = counting_date(row["start_time"], cutoff_hour)
except (TypeError, ValueError):
# A row whose start_time cannot be parsed is left exactly as it is;
# guessing its day would be worse than leaving it visibly odd.
changes.append({"row": row, "new_date": row["counting_date"],
"new_number": row["batch_number"], "skipped": True})
continue
key = (new_date, row["camera_name"], row["object_label"])
counters[key] = counters.get(key, 0) + 1
changes.append({"row": row, "new_date": new_date,
"new_number": counters[key], "skipped": False})
return changes
def apply(connection, changes) -> None:
"""Rewrite the table.
`batches` has UNIQUE(counting_date, batch_number, camera_name, object_label),
so renumbering in place collides with rows that have not moved yet. The
numbers are parked in a negative range first, which cannot collide with any
real batch number, and then written to their final values.
"""
cursor = connection.cursor()
for offset, change in enumerate(changes, start=1):
cursor.execute("UPDATE batches SET batch_number = ? WHERE id = ?",
(-offset, change["row"]["id"]))
for change in changes:
cursor.execute(
"UPDATE batches SET counting_date = ?, batch_number = ? WHERE id = ?",
(change["new_date"], change["new_number"], change["row"]["id"]),
)
cursor.execute("DELETE FROM daily_summaries")
cursor.execute(
"""INSERT INTO daily_summaries
(counting_date, camera_name, object_label, total_count, total_batches)
SELECT counting_date, camera_name, object_label, SUM(count), COUNT(id)
FROM batches
GROUP BY counting_date, camera_name, object_label"""
)
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--db", required=True, help="path to jetson_counter.db")
parser.add_argument("--cutoff-hour", type=int, default=6)
parser.add_argument("--dry-run", action="store_true")
args = parser.parse_args()
if not os.path.isfile(args.db):
print(f"No database at {args.db}")
return 1
connection = sqlite3.connect(args.db)
connection.row_factory = sqlite3.Row
changes = plan(connection, args.cutoff_hour)
if not changes:
print("No batches stored — nothing to do.")
return 0
moved = [c for c in changes
if c["new_date"] != c["row"]["counting_date"]
or c["new_number"] != c["row"]["batch_number"]]
skipped = [c for c in changes if c["skipped"]]
print(f"{len(changes)} batch(es) stored, {len(moved)} would change, "
f"{len(skipped)} unparseable and left alone.\n")
for change in moved[:20]:
row = change["row"]
print(f" {row['start_time'][:19]} "
f"{row['counting_date']} #{row['batch_number']:<4} -> "
f"{change['new_date']} #{change['new_number']}")
if len(moved) > 20:
print(f" … and {len(moved) - 20} more")
if args.dry_run:
print("\nDry run — nothing written.")
return 0
if not moved:
print("\nAlready consistent with the 06:00 cutoff.")
return 0
backup = f"{args.db}.before-0600-{datetime.datetime.now():%Y%m%d-%H%M%S}"
shutil.copyfile(args.db, backup)
print(f"\nBackup written to {backup}")
try:
with connection:
apply(connection, changes)
except Exception as exc:
print(f"FAILED, database left untouched by the transaction: {exc}")
print(f"The backup at {backup} is still the pre-migration state.")
return 1
days = connection.execute(
"SELECT COUNT(DISTINCT counting_date) FROM batches").fetchone()[0]
print(f"Done. {len(moved)} batch(es) re-filed across {days} counting day(s).")
return 0
if __name__ == "__main__":
sys.exit(main())