forked from zakaria/chicken-counting-sukawarna-det
421 lines
18 KiB
C++
421 lines
18 KiB
C++
#pragma once
|
|
|
|
#include <algorithm>
|
|
#include <cmath>
|
|
#include <fstream>
|
|
#include <iostream>
|
|
#include <memory>
|
|
#include <vector>
|
|
|
|
#include <opencv2/core.hpp>
|
|
#include <opencv2/dnn.hpp>
|
|
#include <opencv2/imgproc.hpp>
|
|
#include <opencv2/video/tracking.hpp>
|
|
|
|
#include <NvInfer.h>
|
|
#include <NvOnnxParser.h>
|
|
#include <cuda_runtime.h>
|
|
|
|
#include "chicken_counter/config.hpp"
|
|
#include "chicken_counter/types.hpp"
|
|
|
|
namespace cc {
|
|
|
|
template <typename T> struct TRTDeleter { void operator()(T* p) const { delete p; } };
|
|
template <typename T> using TRT_ptr = std::unique_ptr<T, TRTDeleter<T>>;
|
|
|
|
struct CudaStreamDeleter { void operator()(cudaStream_t* s) const { cudaStreamDestroy(*s); delete s; } };
|
|
using Cuda_stream_ptr = std::unique_ptr<cudaStream_t, CudaStreamDeleter>;
|
|
|
|
struct CudaHostDeleter { template <typename T> void operator()(T* p) const { cudaFreeHost(p); } };
|
|
template <typename T> using Cuda_host_ptr = std::unique_ptr<T, CudaHostDeleter>;
|
|
|
|
inline void trt_check(cudaError_t e, const char* m = "") {
|
|
if (e != cudaSuccess) throw std::runtime_error(std::string("CUDA:") + cudaGetErrorString(e) + " " + m);
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// TensorRT engine (build from ONNX, cache to disk)
|
|
// ---------------------------------------------------------------------------
|
|
class TensorRTEngine {
|
|
public:
|
|
TensorRTEngine(const std::string& onnx_path, const std::string& cache_path = "") {
|
|
if (!cache_path.empty()) {
|
|
std::ifstream fc(cache_path, std::ios::binary | std::ios::ate);
|
|
if (fc) {
|
|
size_t sz = fc.tellg(); fc.seekg(0);
|
|
std::vector<char> data(sz);
|
|
fc.read(data.data(), sz);
|
|
runtime.reset(nvinfer1::createInferRuntime(logger));
|
|
engine.reset(runtime->deserializeCudaEngine(data.data(), sz));
|
|
if (engine) {
|
|
std::cerr << "[trt] loaded cache: " << cache_path << "\n";
|
|
init_io();
|
|
return;
|
|
}
|
|
}
|
|
}
|
|
|
|
std::cerr << "[trt] building from " << onnx_path << " ...\n";
|
|
auto builder = TRT_ptr<nvinfer1::IBuilder>(nvinfer1::createInferBuilder(logger));
|
|
auto network = TRT_ptr<nvinfer1::INetworkDefinition>(
|
|
builder->createNetworkV2(1U << static_cast<uint32_t>(
|
|
nvinfer1::NetworkDefinitionCreationFlag::kEXPLICIT_BATCH)));
|
|
auto parser = TRT_ptr<nvonnxparser::IParser>(nvonnxparser::createParser(*network, logger));
|
|
if (!parser->parseFromFile(onnx_path.c_str(),
|
|
static_cast<int>(nvinfer1::ILogger::Severity::kWARNING)))
|
|
throw std::runtime_error("ONNX parse failed");
|
|
|
|
auto config = TRT_ptr<nvinfer1::IBuilderConfig>(builder->createBuilderConfig());
|
|
config->setMemoryPoolLimit(nvinfer1::MemoryPoolType::kWORKSPACE, 256ULL << 20);
|
|
if (builder->platformHasFastFp16()) config->setFlag(nvinfer1::BuilderFlag::kFP16);
|
|
|
|
// set optimisation profile if input has dynamic dims
|
|
int nb = network->getNbInputs();
|
|
if (nb > 0) {
|
|
auto in = network->getInput(0);
|
|
auto prof = builder->createOptimizationProfile();
|
|
nvinfer1::Dims minD = in->getDimensions(), optD = minD, maxD = minD;
|
|
for (int d = 0; d < minD.nbDims; ++d) {
|
|
if (minD.d[d] < 0) {
|
|
minD.d[d] = 1; optD.d[d] = 1; maxD.d[d] = 1;
|
|
if (d == 2) { optD.d[d] = 640; maxD.d[d] = 640; }
|
|
if (d == 3) { optD.d[d] = 640; maxD.d[d] = 640; }
|
|
}
|
|
}
|
|
prof->setDimensions(in->getName(), nvinfer1::OptProfileSelector::kMIN, minD);
|
|
prof->setDimensions(in->getName(), nvinfer1::OptProfileSelector::kOPT, optD);
|
|
prof->setDimensions(in->getName(), nvinfer1::OptProfileSelector::kMAX, maxD);
|
|
config->addOptimizationProfile(prof);
|
|
}
|
|
|
|
auto plan = TRT_ptr<nvinfer1::IHostMemory>(
|
|
builder->buildSerializedNetwork(*network, *config));
|
|
if (!plan) throw std::runtime_error("buildSerializedNetwork failed");
|
|
|
|
runtime.reset(nvinfer1::createInferRuntime(logger));
|
|
engine.reset(runtime->deserializeCudaEngine(plan->data(), plan->size()));
|
|
|
|
if (!cache_path.empty()) {
|
|
std::ofstream out(cache_path, std::ios::binary);
|
|
out.write(static_cast<const char*>(plan->data()), plan->size());
|
|
std::cerr << "[trt] cached: " << cache_path << " (" << plan->size() << " B)\n";
|
|
}
|
|
init_io();
|
|
}
|
|
|
|
~TensorRTEngine() {
|
|
for (auto& kv : buffers) cudaFree(kv.second);
|
|
}
|
|
|
|
void run(const float* input, float* output) {
|
|
trt_check(cudaMemcpyAsync(buffers[input_name], input, input_bytes,
|
|
cudaMemcpyHostToDevice, *stream));
|
|
context->enqueueV3(*stream);
|
|
trt_check(cudaMemcpyAsync(output, buffers[output_name], output_bytes,
|
|
cudaMemcpyDeviceToHost, *stream));
|
|
cudaStreamSynchronize(*stream);
|
|
}
|
|
|
|
std::string input_name, output_name;
|
|
size_t input_bytes = 0, output_bytes = 0;
|
|
int output_num_classes = 1, output_num_cells = 8400;
|
|
bool output_layout_packed = true; // true = [1,N,8400], false = [1,8400,N]
|
|
|
|
private:
|
|
struct Logger : nvinfer1::ILogger {
|
|
void log(Severity sev, const char* msg) noexcept override {
|
|
if (sev <= Severity::kWARNING) std::cerr << "[trt] " << msg << std::endl;
|
|
}
|
|
};
|
|
Logger logger;
|
|
TRT_ptr<nvinfer1::IRuntime> runtime;
|
|
TRT_ptr<nvinfer1::ICudaEngine> engine;
|
|
TRT_ptr<nvinfer1::IExecutionContext> context;
|
|
std::unordered_map<std::string, void*> buffers;
|
|
Cuda_stream_ptr stream;
|
|
|
|
void init_io() {
|
|
context.reset(engine->createExecutionContext());
|
|
if (!context) throw std::runtime_error("createExecutionContext failed");
|
|
|
|
int nb = engine->getNbIOTensors();
|
|
for (int i = 0; i < nb; ++i) {
|
|
auto name = engine->getIOTensorName(i);
|
|
auto mode = engine->getTensorIOMode(name);
|
|
auto shape = engine->getTensorShape(name);
|
|
size_t bytes = 1;
|
|
for (int d = 0; d < shape.nbDims; ++d) bytes *= shape.d[d];
|
|
bytes *= sizeof(float);
|
|
void* ptr = nullptr;
|
|
cudaError_t e = cudaMalloc(&ptr, bytes);
|
|
if (e != cudaSuccess) throw std::runtime_error(std::string("cudaMalloc:") + cudaGetErrorString(e));
|
|
if (!context->setTensorAddress(name, ptr))
|
|
throw std::runtime_error(std::string("setTensorAddress: ") + name);
|
|
buffers[name] = ptr;
|
|
|
|
if (mode == nvinfer1::TensorIOMode::kINPUT) {
|
|
input_name = name; input_bytes = bytes;
|
|
} else {
|
|
output_name = name; output_bytes = bytes;
|
|
// cache output layout once
|
|
int d1 = (shape.nbDims >= 2) ? shape.d[1] : 1;
|
|
int d2 = (shape.nbDims >= 3) ? shape.d[2] : 1;
|
|
if (d2 > d1) {
|
|
output_layout_packed = true; // [1, N, cells]
|
|
output_num_classes = std::max(1, d1 - 4);
|
|
output_num_cells = d2;
|
|
} else {
|
|
output_layout_packed = false; // [1, cells, N]
|
|
output_num_classes = std::max(1, d2 - 4);
|
|
output_num_cells = d1;
|
|
}
|
|
}
|
|
}
|
|
stream.reset(new cudaStream_t{});
|
|
trt_check(cudaStreamCreate(stream.get()));
|
|
std::cerr << "[trt] ready: in=" << input_name << " (" << input_bytes
|
|
<< "B) out=" << output_name << " (" << output_bytes
|
|
<< "B) cls=" << output_num_classes << " cells=" << output_num_cells << "\n";
|
|
}
|
|
};
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// DetectionTracker (TensorRT inference + SORT tracking)
|
|
// ---------------------------------------------------------------------------
|
|
class DetectionTracker {
|
|
public:
|
|
DetectionTracker(const CameraConfig& config)
|
|
: config(config), imgsz(config.detection.imgsz),
|
|
conf_thresh(config.detection.conf),
|
|
iou_thresh(config.detection.iou),
|
|
verbose(config.performance.verbose),
|
|
track_buffer(config.tracker.track_buffer)
|
|
{
|
|
std::string path = config.detection.model_path;
|
|
size_t dot = path.rfind('.');
|
|
std::string kind = (dot != std::string::npos) ? path.substr(dot) : "";
|
|
|
|
if (kind == ".engine" || kind == ".onnx") {
|
|
std::string onnx = (kind == ".engine")
|
|
? path.substr(0, path.size() - 7) + ".onnx" : path;
|
|
use_trt = true;
|
|
trt = std::make_unique<TensorRTEngine>(onnx, path + ".cache");
|
|
} else {
|
|
throw std::runtime_error("Unsupported model format: " + kind);
|
|
}
|
|
|
|
// alloc pinned output buffer (reused every inference)
|
|
int out_floats = static_cast<int>(trt->output_bytes / sizeof(float));
|
|
cudaMallocHost(&output_buf, trt->output_bytes);
|
|
output_buf_size = out_floats;
|
|
|
|
std::cerr << "[model] ready\n";
|
|
}
|
|
|
|
~DetectionTracker() {
|
|
if (output_buf) cudaFreeHost(output_buf);
|
|
}
|
|
|
|
void reset_tracking() {
|
|
active_tracks.clear();
|
|
last_frame_id = 0;
|
|
}
|
|
|
|
std::vector<TrackObservation> infer(const cv::Mat& frame,
|
|
const cv::Rect& crop_rect = {}) {
|
|
++last_frame_id;
|
|
cv::Mat source;
|
|
int off_x = 0, off_y = 0;
|
|
if (!crop_rect.empty() && crop_rect.width > 0 && crop_rect.height > 0) {
|
|
source = frame(crop_rect);
|
|
off_x = crop_rect.x; off_y = crop_rect.y;
|
|
} else {
|
|
source = frame;
|
|
}
|
|
|
|
float scale; int pad_x, pad_y;
|
|
cv::Mat blob = preprocess(source, scale, pad_x, pad_y);
|
|
|
|
// inference (blob.data is already NCHW float, use directly)
|
|
trt->run(reinterpret_cast<float*>(blob.data), output_buf);
|
|
|
|
// decode
|
|
auto dets = decode(scale, pad_x, pad_y, source.cols, source.rows);
|
|
auto tracks = associate(dets, off_x, off_y);
|
|
|
|
if (verbose && last_frame_id % 30 == 0)
|
|
std::cerr << "[track] f=" << last_frame_id << " det=" << dets.size()
|
|
<< " trk=" << tracks.size() << "\n";
|
|
return tracks;
|
|
}
|
|
|
|
CameraConfig config;
|
|
|
|
private:
|
|
bool use_trt = false;
|
|
std::unique_ptr<TensorRTEngine> trt;
|
|
int imgsz;
|
|
float conf_thresh, iou_thresh;
|
|
bool verbose;
|
|
int track_buffer, last_frame_id = 0;
|
|
|
|
float* output_buf = nullptr;
|
|
int output_buf_size = 0;
|
|
|
|
// --- Kalman track ---
|
|
struct KalmanTrack {
|
|
int id; cv::KalmanFilter kf; cv::Rect2f bbox;
|
|
int hits = 0, time_since_update = 0;
|
|
KalmanTrack(int tid, const cv::Rect2f& b) : id(tid), bbox(b) {
|
|
kf.init(7, 4, 0, CV_32F);
|
|
kf.transitionMatrix = (cv::Mat_<float>(7, 7) <<
|
|
1,0,0,0,1,0,0, 0,1,0,0,0,1,0, 0,0,1,0,0,0,1, 0,0,0,1,0,0,0,
|
|
0,0,0,0,1,0,0, 0,0,0,0,0,1,0, 0,0,0,0,0,0,1);
|
|
cv::setIdentity(kf.measurementMatrix);
|
|
cv::setIdentity(kf.processNoiseCov, cv::Scalar::all(1e-2));
|
|
cv::setIdentity(kf.measurementNoiseCov, cv::Scalar::all(1e-1));
|
|
cv::setIdentity(kf.errorCovPost, cv::Scalar::all(1));
|
|
kf.statePost.at<float>(0) = b.x + b.width/2;
|
|
kf.statePost.at<float>(1) = b.y + b.height/2;
|
|
kf.statePost.at<float>(2) = b.area();
|
|
kf.statePost.at<float>(3) = b.width / b.height;
|
|
}
|
|
cv::Rect2f predict() {
|
|
cv::Mat p = kf.predict();
|
|
float w = std::sqrt(std::max(1.0f, p.at<float>(2) * p.at<float>(3)));
|
|
float h = std::max(1.0f, p.at<float>(2) / w);
|
|
bbox = cv::Rect2f(p.at<float>(0) - w/2, p.at<float>(1) - h/2, w, h);
|
|
return bbox;
|
|
}
|
|
void update(const cv::Rect2f& b) {
|
|
float cx = b.x + b.width/2, cy = b.y + b.height/2;
|
|
kf.correct((cv::Mat_<float>(4, 1) << cx, cy, b.area(), b.width / b.height));
|
|
bbox = b; hits++; time_since_update = 0;
|
|
}
|
|
};
|
|
std::vector<KalmanTrack> active_tracks;
|
|
int next_track_id = 1;
|
|
|
|
// --- preprocess ---
|
|
cv::Mat preprocess(const cv::Mat& img, float& scale, int& pad_x, int& pad_y) {
|
|
int w = img.cols, h = img.rows;
|
|
scale = static_cast<float>(imgsz) / std::max(w, h);
|
|
int nw = static_cast<int>(w * scale), nh = static_cast<int>(h * scale);
|
|
pad_x = (imgsz - nw) / 2;
|
|
pad_y = (imgsz - nh) / 2;
|
|
cv::Mat r, p;
|
|
cv::resize(img, r, {nw, nh});
|
|
cv::copyMakeBorder(r, p, pad_y, imgsz - nh - pad_y, pad_x, imgsz - nw - pad_x,
|
|
cv::BORDER_CONSTANT, {114, 114, 114});
|
|
return cv::dnn::blobFromImage(p, 1.0/255.0, {imgsz, imgsz}, cv::Scalar(), true, false);
|
|
}
|
|
|
|
// --- decode (uses cache layout) ---
|
|
std::vector<cv::Rect2f> decode(float scale, int pad_x, int pad_y, int ow, int oh) {
|
|
std::vector<cv::Rect> iboxes; iboxes.reserve(32);
|
|
std::vector<float> scores; scores.reserve(32);
|
|
|
|
int nc = trt->output_num_classes;
|
|
int cells = trt->output_num_cells;
|
|
int stride = nc + 4; // total channels per cell
|
|
const float* d = output_buf;
|
|
|
|
if (trt->output_layout_packed) {
|
|
// layout [1, N, cells] — each channel is contiguous stride apart
|
|
for (int i = 0; i < cells; ++i) {
|
|
float maxc = 0;
|
|
for (int c = 0; c < nc; ++c) {
|
|
float v = d[(4 + c) * cells + i];
|
|
if (v > maxc) maxc = v;
|
|
}
|
|
if (maxc < conf_thresh) continue;
|
|
float cx = d[i], cy = d[cells + i], w = d[2*cells + i], h = d[3*cells + i];
|
|
float x = (cx - pad_x) / scale, y = (cy - pad_y) / scale;
|
|
float bw = w / scale, bh = h / scale;
|
|
int x1 = std::max(0, std::min(static_cast<int>(x - bw/2), ow));
|
|
int y1 = std::max(0, std::min(static_cast<int>(y - bh/2), oh));
|
|
int x2 = std::max(0, std::min(static_cast<int>(x + bw/2), ow));
|
|
int y2 = std::max(0, std::min(static_cast<int>(y + bh/2), oh));
|
|
if (x2 > x1 && y2 > y1) { iboxes.push_back({x1, y1, x2 - x1, y2 - y1}); scores.push_back(maxc); }
|
|
}
|
|
} else {
|
|
// layout [1, cells, N] — contiguous per row
|
|
for (int i = 0; i < cells; ++i) {
|
|
const float* row = d + i * stride;
|
|
float maxc = 0;
|
|
for (int c = 0; c < nc; ++c) { float v = row[4 + c]; if (v > maxc) maxc = v; }
|
|
if (maxc < conf_thresh) continue;
|
|
float cx = row[0], cy = row[1], w = row[2], h = row[3];
|
|
float x = (cx - pad_x) / scale, y = (cy - pad_y) / scale;
|
|
int x1 = std::max(0, std::min(static_cast<int>(x - w/scale/2), ow));
|
|
int y1 = std::max(0, std::min(static_cast<int>(y - h/scale/2), oh));
|
|
int x2 = std::max(0, std::min(static_cast<int>(x + w/scale/2), ow));
|
|
int y2 = std::max(0, std::min(static_cast<int>(y + h/scale/2), oh));
|
|
if (x2 > x1 && y2 > y1) { iboxes.push_back({x1, y1, x2 - x1, y2 - y1}); scores.push_back(maxc); }
|
|
}
|
|
}
|
|
|
|
std::vector<int> idx; cv::dnn::NMSBoxes(iboxes, scores, conf_thresh, iou_thresh, idx);
|
|
std::vector<cv::Rect2f> out; out.reserve(idx.size());
|
|
for (int i : idx) out.push_back(iboxes[i]);
|
|
return out;
|
|
}
|
|
|
|
// --- SORT association ---
|
|
std::vector<TrackObservation> associate(const std::vector<cv::Rect2f>& dets, int ox, int oy) {
|
|
for (auto& t : active_tracks) { t.predict(); t.time_since_update++; }
|
|
int nd = static_cast<int>(dets.size()), nt = static_cast<int>(active_tracks.size());
|
|
if (nd == 0) goto cleanup;
|
|
|
|
{
|
|
std::vector<std::vector<double>> iou(nt, std::vector<double>(nd));
|
|
for (int t = 0; t < nt; ++t)
|
|
for (int d = 0; d < nd; ++d)
|
|
iou[t][d] = 1.0 - _iou(active_tracks[t].bbox, dets[d]);
|
|
std::vector<int> order(nd); for (int i = 0; i < nd; ++i) order[i] = i;
|
|
std::sort(order.begin(), order.end(), [&](int a, int b){ return dets[a].area() > dets[b].area(); });
|
|
std::vector<bool> used(nd, false); std::vector<int> match(nt, -1);
|
|
for (int d : order) {
|
|
int best = -1; double best_cost = 0.3;
|
|
for (int t = 0; t < nt; ++t) {
|
|
if (match[t] >= 0) continue;
|
|
if (iou[t][d] < best_cost) { best_cost = iou[t][d]; best = t; }
|
|
}
|
|
if (best >= 0) { match[best] = d; used[d] = true; }
|
|
}
|
|
for (int t = 0; t < nt; ++t) if (match[t] >= 0) active_tracks[t].update(dets[match[t]]);
|
|
for (int d = 0; d < nd; ++d) if (!used[d]) {
|
|
KalmanTrack tk(++next_track_id, dets[d]); tk.hits = 1; active_tracks.push_back(tk);
|
|
}
|
|
}
|
|
|
|
cleanup:
|
|
active_tracks.erase(std::remove_if(active_tracks.begin(), active_tracks.end(),
|
|
[this](const KalmanTrack& t){ return t.time_since_update > track_buffer; }),
|
|
active_tracks.end());
|
|
|
|
std::vector<TrackObservation> out;
|
|
for (const auto& t : active_tracks) {
|
|
if (t.hits < 3) continue;
|
|
auto& b = t.bbox;
|
|
int x1 = static_cast<int>(b.x) + ox, y1 = static_cast<int>(b.y) + oy;
|
|
int x2 = static_cast<int>(b.x + b.width) + ox, y2 = static_cast<int>(b.y + b.height) + oy;
|
|
out.push_back({t.id, 0, 0.9f, x1, y1, x2, y2, (x1+x2)/2, (y1+y2)/2, {}});
|
|
}
|
|
return out;
|
|
}
|
|
|
|
static double _iou(const cv::Rect2f& a, const cv::Rect2f& b) {
|
|
float ix1 = std::max(a.x, b.x), iy1 = std::max(a.y, b.y);
|
|
float ix2 = std::min(a.x + a.width, b.x + b.width);
|
|
float iy2 = std::min(a.y + a.height, b.y + b.height);
|
|
if (ix2 <= ix1 || iy2 <= iy1) return 0.0;
|
|
float I = (ix2 - ix1) * (iy2 - iy1);
|
|
float U = a.area() + b.area() - I;
|
|
return U > 0 ? I / U : 0.0;
|
|
}
|
|
};
|
|
|
|
} // namespace cc
|