#pragma once #include #include #include #include #include #include #include #include #include #include #include #include #include #include "chicken_counter/config.hpp" #include "chicken_counter/types.hpp" namespace cc { template struct TRTDeleter { void operator()(T* p) const { delete p; } }; template using TRT_ptr = std::unique_ptr>; struct CudaStreamDeleter { void operator()(cudaStream_t* s) const { cudaStreamDestroy(*s); delete s; } }; using Cuda_stream_ptr = std::unique_ptr; struct CudaHostDeleter { template void operator()(T* p) const { cudaFreeHost(p); } }; template using Cuda_host_ptr = std::unique_ptr; inline void trt_check(cudaError_t e, const char* m = "") { if (e != cudaSuccess) throw std::runtime_error(std::string("CUDA:") + cudaGetErrorString(e) + " " + m); } // --------------------------------------------------------------------------- // TensorRT engine (build from ONNX, cache to disk) // --------------------------------------------------------------------------- class TensorRTEngine { public: TensorRTEngine(const std::string& onnx_path, const std::string& cache_path = "") { if (!cache_path.empty()) { std::ifstream fc(cache_path, std::ios::binary | std::ios::ate); if (fc) { size_t sz = fc.tellg(); fc.seekg(0); std::vector data(sz); fc.read(data.data(), sz); runtime.reset(nvinfer1::createInferRuntime(logger)); engine.reset(runtime->deserializeCudaEngine(data.data(), sz)); if (engine) { std::cerr << "[trt] loaded cache: " << cache_path << "\n"; init_io(); return; } } } std::cerr << "[trt] building from " << onnx_path << " ...\n"; auto builder = TRT_ptr(nvinfer1::createInferBuilder(logger)); auto network = TRT_ptr( builder->createNetworkV2(1U << static_cast( nvinfer1::NetworkDefinitionCreationFlag::kEXPLICIT_BATCH))); auto parser = TRT_ptr(nvonnxparser::createParser(*network, logger)); if (!parser->parseFromFile(onnx_path.c_str(), static_cast(nvinfer1::ILogger::Severity::kWARNING))) throw std::runtime_error("ONNX parse failed"); auto config = TRT_ptr(builder->createBuilderConfig()); config->setMemoryPoolLimit(nvinfer1::MemoryPoolType::kWORKSPACE, 256ULL << 20); if (builder->platformHasFastFp16()) config->setFlag(nvinfer1::BuilderFlag::kFP16); // set optimisation profile if input has dynamic dims int nb = network->getNbInputs(); if (nb > 0) { auto in = network->getInput(0); auto prof = builder->createOptimizationProfile(); nvinfer1::Dims minD = in->getDimensions(), optD = minD, maxD = minD; for (int d = 0; d < minD.nbDims; ++d) { if (minD.d[d] < 0) { minD.d[d] = 1; optD.d[d] = 1; maxD.d[d] = 1; if (d == 2) { optD.d[d] = 640; maxD.d[d] = 640; } if (d == 3) { optD.d[d] = 640; maxD.d[d] = 640; } } } prof->setDimensions(in->getName(), nvinfer1::OptProfileSelector::kMIN, minD); prof->setDimensions(in->getName(), nvinfer1::OptProfileSelector::kOPT, optD); prof->setDimensions(in->getName(), nvinfer1::OptProfileSelector::kMAX, maxD); config->addOptimizationProfile(prof); } auto plan = TRT_ptr( builder->buildSerializedNetwork(*network, *config)); if (!plan) throw std::runtime_error("buildSerializedNetwork failed"); runtime.reset(nvinfer1::createInferRuntime(logger)); engine.reset(runtime->deserializeCudaEngine(plan->data(), plan->size())); if (!cache_path.empty()) { std::ofstream out(cache_path, std::ios::binary); out.write(static_cast(plan->data()), plan->size()); std::cerr << "[trt] cached: " << cache_path << " (" << plan->size() << " B)\n"; } init_io(); } ~TensorRTEngine() { for (auto& kv : buffers) cudaFree(kv.second); } void run(const float* input, float* output) { trt_check(cudaMemcpyAsync(buffers[input_name], input, input_bytes, cudaMemcpyHostToDevice, *stream)); context->enqueueV3(*stream); trt_check(cudaMemcpyAsync(output, buffers[output_name], output_bytes, cudaMemcpyDeviceToHost, *stream)); cudaStreamSynchronize(*stream); } std::string input_name, output_name; size_t input_bytes = 0, output_bytes = 0; int output_num_classes = 1, output_num_cells = 8400; bool output_layout_packed = true; // true = [1,N,8400], false = [1,8400,N] private: struct Logger : nvinfer1::ILogger { void log(Severity sev, const char* msg) noexcept override { if (sev <= Severity::kWARNING) std::cerr << "[trt] " << msg << std::endl; } }; Logger logger; TRT_ptr runtime; TRT_ptr engine; TRT_ptr context; std::unordered_map buffers; Cuda_stream_ptr stream; void init_io() { context.reset(engine->createExecutionContext()); if (!context) throw std::runtime_error("createExecutionContext failed"); int nb = engine->getNbIOTensors(); for (int i = 0; i < nb; ++i) { auto name = engine->getIOTensorName(i); auto mode = engine->getTensorIOMode(name); auto shape = engine->getTensorShape(name); size_t bytes = 1; for (int d = 0; d < shape.nbDims; ++d) bytes *= shape.d[d]; bytes *= sizeof(float); void* ptr = nullptr; cudaError_t e = cudaMalloc(&ptr, bytes); if (e != cudaSuccess) throw std::runtime_error(std::string("cudaMalloc:") + cudaGetErrorString(e)); if (!context->setTensorAddress(name, ptr)) throw std::runtime_error(std::string("setTensorAddress: ") + name); buffers[name] = ptr; if (mode == nvinfer1::TensorIOMode::kINPUT) { input_name = name; input_bytes = bytes; } else { output_name = name; output_bytes = bytes; // cache output layout once int d1 = (shape.nbDims >= 2) ? shape.d[1] : 1; int d2 = (shape.nbDims >= 3) ? shape.d[2] : 1; if (d2 > d1) { output_layout_packed = true; // [1, N, cells] output_num_classes = std::max(1, d1 - 4); output_num_cells = d2; } else { output_layout_packed = false; // [1, cells, N] output_num_classes = std::max(1, d2 - 4); output_num_cells = d1; } } } stream.reset(new cudaStream_t{}); trt_check(cudaStreamCreate(stream.get())); std::cerr << "[trt] ready: in=" << input_name << " (" << input_bytes << "B) out=" << output_name << " (" << output_bytes << "B) cls=" << output_num_classes << " cells=" << output_num_cells << "\n"; } }; // --------------------------------------------------------------------------- // DetectionTracker (TensorRT inference + SORT tracking) // --------------------------------------------------------------------------- class DetectionTracker { public: DetectionTracker(const CameraConfig& config) : config(config), imgsz(config.detection.imgsz), conf_thresh(config.detection.conf), iou_thresh(config.detection.iou), verbose(config.performance.verbose), track_buffer(config.tracker.track_buffer) { std::string path = config.detection.model_path; size_t dot = path.rfind('.'); std::string kind = (dot != std::string::npos) ? path.substr(dot) : ""; if (kind == ".engine" || kind == ".onnx") { std::string onnx = (kind == ".engine") ? path.substr(0, path.size() - 7) + ".onnx" : path; use_trt = true; trt = std::make_unique(onnx, path + ".cache"); } else { throw std::runtime_error("Unsupported model format: " + kind); } // alloc pinned output buffer (reused every inference) int out_floats = static_cast(trt->output_bytes / sizeof(float)); cudaMallocHost(&output_buf, trt->output_bytes); output_buf_size = out_floats; std::cerr << "[model] ready\n"; } ~DetectionTracker() { if (output_buf) cudaFreeHost(output_buf); } void reset_tracking() { active_tracks.clear(); last_frame_id = 0; } std::vector infer(const cv::Mat& frame, const cv::Rect& crop_rect = {}) { ++last_frame_id; cv::Mat source; int off_x = 0, off_y = 0; if (!crop_rect.empty() && crop_rect.width > 0 && crop_rect.height > 0) { source = frame(crop_rect); off_x = crop_rect.x; off_y = crop_rect.y; } else { source = frame; } float scale; int pad_x, pad_y; cv::Mat blob = preprocess(source, scale, pad_x, pad_y); // inference (blob.data is already NCHW float, use directly) trt->run(reinterpret_cast(blob.data), output_buf); // decode auto dets = decode(scale, pad_x, pad_y, source.cols, source.rows); auto tracks = associate(dets, off_x, off_y); if (verbose && last_frame_id % 30 == 0) std::cerr << "[track] f=" << last_frame_id << " det=" << dets.size() << " trk=" << tracks.size() << "\n"; return tracks; } CameraConfig config; private: bool use_trt = false; std::unique_ptr trt; int imgsz; float conf_thresh, iou_thresh; bool verbose; int track_buffer, last_frame_id = 0; float* output_buf = nullptr; int output_buf_size = 0; // --- Kalman track --- struct KalmanTrack { int id; cv::KalmanFilter kf; cv::Rect2f bbox; int hits = 0, time_since_update = 0; KalmanTrack(int tid, const cv::Rect2f& b) : id(tid), bbox(b) { kf.init(7, 4, 0, CV_32F); kf.transitionMatrix = (cv::Mat_(7, 7) << 1,0,0,0,1,0,0, 0,1,0,0,0,1,0, 0,0,1,0,0,0,1, 0,0,0,1,0,0,0, 0,0,0,0,1,0,0, 0,0,0,0,0,1,0, 0,0,0,0,0,0,1); cv::setIdentity(kf.measurementMatrix); cv::setIdentity(kf.processNoiseCov, cv::Scalar::all(1e-2)); cv::setIdentity(kf.measurementNoiseCov, cv::Scalar::all(1e-1)); cv::setIdentity(kf.errorCovPost, cv::Scalar::all(1)); kf.statePost.at(0) = b.x + b.width/2; kf.statePost.at(1) = b.y + b.height/2; kf.statePost.at(2) = b.area(); kf.statePost.at(3) = b.width / b.height; } cv::Rect2f predict() { cv::Mat p = kf.predict(); float w = std::sqrt(std::max(1.0f, p.at(2) * p.at(3))); float h = std::max(1.0f, p.at(2) / w); bbox = cv::Rect2f(p.at(0) - w/2, p.at(1) - h/2, w, h); return bbox; } void update(const cv::Rect2f& b) { float cx = b.x + b.width/2, cy = b.y + b.height/2; kf.correct((cv::Mat_(4, 1) << cx, cy, b.area(), b.width / b.height)); bbox = b; hits++; time_since_update = 0; } }; std::vector active_tracks; int next_track_id = 1; // --- preprocess --- cv::Mat preprocess(const cv::Mat& img, float& scale, int& pad_x, int& pad_y) { int w = img.cols, h = img.rows; scale = static_cast(imgsz) / std::max(w, h); int nw = static_cast(w * scale), nh = static_cast(h * scale); pad_x = (imgsz - nw) / 2; pad_y = (imgsz - nh) / 2; cv::Mat r, p; cv::resize(img, r, {nw, nh}); cv::copyMakeBorder(r, p, pad_y, imgsz - nh - pad_y, pad_x, imgsz - nw - pad_x, cv::BORDER_CONSTANT, {114, 114, 114}); return cv::dnn::blobFromImage(p, 1.0/255.0, {imgsz, imgsz}, cv::Scalar(), true, false); } // --- decode (uses cache layout) --- std::vector decode(float scale, int pad_x, int pad_y, int ow, int oh) { std::vector iboxes; iboxes.reserve(32); std::vector scores; scores.reserve(32); int nc = trt->output_num_classes; int cells = trt->output_num_cells; int stride = nc + 4; // total channels per cell const float* d = output_buf; if (trt->output_layout_packed) { // layout [1, N, cells] — each channel is contiguous stride apart for (int i = 0; i < cells; ++i) { float maxc = 0; for (int c = 0; c < nc; ++c) { float v = d[(4 + c) * cells + i]; if (v > maxc) maxc = v; } if (maxc < conf_thresh) continue; float cx = d[i], cy = d[cells + i], w = d[2*cells + i], h = d[3*cells + i]; float x = (cx - pad_x) / scale, y = (cy - pad_y) / scale; float bw = w / scale, bh = h / scale; int x1 = std::max(0, std::min(static_cast(x - bw/2), ow)); int y1 = std::max(0, std::min(static_cast(y - bh/2), oh)); int x2 = std::max(0, std::min(static_cast(x + bw/2), ow)); int y2 = std::max(0, std::min(static_cast(y + bh/2), oh)); if (x2 > x1 && y2 > y1) { iboxes.push_back({x1, y1, x2 - x1, y2 - y1}); scores.push_back(maxc); } } } else { // layout [1, cells, N] — contiguous per row for (int i = 0; i < cells; ++i) { const float* row = d + i * stride; float maxc = 0; for (int c = 0; c < nc; ++c) { float v = row[4 + c]; if (v > maxc) maxc = v; } if (maxc < conf_thresh) continue; float cx = row[0], cy = row[1], w = row[2], h = row[3]; float x = (cx - pad_x) / scale, y = (cy - pad_y) / scale; int x1 = std::max(0, std::min(static_cast(x - w/scale/2), ow)); int y1 = std::max(0, std::min(static_cast(y - h/scale/2), oh)); int x2 = std::max(0, std::min(static_cast(x + w/scale/2), ow)); int y2 = std::max(0, std::min(static_cast(y + h/scale/2), oh)); if (x2 > x1 && y2 > y1) { iboxes.push_back({x1, y1, x2 - x1, y2 - y1}); scores.push_back(maxc); } } } std::vector idx; cv::dnn::NMSBoxes(iboxes, scores, conf_thresh, iou_thresh, idx); std::vector out; out.reserve(idx.size()); for (int i : idx) out.push_back(iboxes[i]); return out; } // --- SORT association --- std::vector associate(const std::vector& dets, int ox, int oy) { for (auto& t : active_tracks) { t.predict(); t.time_since_update++; } int nd = static_cast(dets.size()), nt = static_cast(active_tracks.size()); if (nd == 0) goto cleanup; { std::vector> iou(nt, std::vector(nd)); for (int t = 0; t < nt; ++t) for (int d = 0; d < nd; ++d) iou[t][d] = 1.0 - _iou(active_tracks[t].bbox, dets[d]); std::vector order(nd); for (int i = 0; i < nd; ++i) order[i] = i; std::sort(order.begin(), order.end(), [&](int a, int b){ return dets[a].area() > dets[b].area(); }); std::vector used(nd, false); std::vector match(nt, -1); for (int d : order) { int best = -1; double best_cost = 0.3; for (int t = 0; t < nt; ++t) { if (match[t] >= 0) continue; if (iou[t][d] < best_cost) { best_cost = iou[t][d]; best = t; } } if (best >= 0) { match[best] = d; used[d] = true; } } for (int t = 0; t < nt; ++t) if (match[t] >= 0) active_tracks[t].update(dets[match[t]]); for (int d = 0; d < nd; ++d) if (!used[d]) { KalmanTrack tk(++next_track_id, dets[d]); tk.hits = 1; active_tracks.push_back(tk); } } cleanup: active_tracks.erase(std::remove_if(active_tracks.begin(), active_tracks.end(), [this](const KalmanTrack& t){ return t.time_since_update > track_buffer; }), active_tracks.end()); std::vector out; for (const auto& t : active_tracks) { if (t.hits < 3) continue; auto& b = t.bbox; int x1 = static_cast(b.x) + ox, y1 = static_cast(b.y) + oy; int x2 = static_cast(b.x + b.width) + ox, y2 = static_cast(b.y + b.height) + oy; out.push_back({t.id, 0, 0.9f, x1, y1, x2, y2, (x1+x2)/2, (y1+y2)/2, {}}); } return out; } static double _iou(const cv::Rect2f& a, const cv::Rect2f& b) { float ix1 = std::max(a.x, b.x), iy1 = std::max(a.y, b.y); float ix2 = std::min(a.x + a.width, b.x + b.width); float iy2 = std::min(a.y + a.height, b.y + b.height); if (ix2 <= ix1 || iy2 <= iy1) return 0.0; float I = (ix2 - ix1) * (iy2 - iy1); float U = a.area() + b.area() - I; return U > 0 ? I / U : 0.0; } }; } // namespace cc