"""Nuclio handler for CVAT automatic annotation using OpenVINO 2025 IR (.xml/.bin). This file combines YOLOv9 inference logic with Nuclio serverless handler structure. It loads an OpenVINO Intermediate Representation (IR) model consisting of a ``.xml`` file (network topology) and a ``.bin`` file (weights). Adjust ``MODEL_XML`` and ``MODEL_BIN`` if your files are located elsewhere. """ import base64 import json import os from pathlib import Path import cv2 import numpy as np import openvino as ov from openvino.preprocess import PrePostProcessor from openvino.preprocess import ColorFormat from openvino import Layout, Type #MODEL_DIR = "chicken-detection-model-v26n-300e-best-2026-05-02-NEW_openvino_model" #MODEL_NAME = "chicken-detection-model-v26n-300e-best-2026-05-02-NEW" #MODEL_XML = os.getenv("MODEL_XML", f"/models/{MODEL_DIR}/{MODEL_NAME}.xml") #MODEL_BIN = os.getenv("MODEL_BIN", f"/models/{MODEL_DIR}/{MODEL_NAME}.bin") MODEL_XML = os.getenv("MODEL_XML","/models/best.xml") MODEL_BIN = os.getenv("MODEL_BIN", "/models/best.bin") classes_file = os.getenv("CLASSES_FILE", "/opt/nuclio/classes.json") try: with open(classes_file, 'r') as f: classes_data = json.load(f) coconame = [cls['name'] for cls in sorted(classes_data, key=lambda x: x['id'])] except (FileNotFoundError, json.JSONDecodeError): coconame = ["chicken", "not-chicken", "half-chicken"] class Yolov9: def __init__( self, xml_model_path=MODEL_XML, bin_model_path=MODEL_BIN, conf=0.1, nms=0.4 ): # Step 1. Initialize OpenVINO Runtime core core = ov.Core() # Step 2. Read a model if bin_model_path: model = core.read_model( str(Path(xml_model_path)), str(Path(bin_model_path)) ) else: model = core.read_model(str(Path(xml_model_path))) self.input_shape = model.input(0).shape # NCHW: [1, C, H, W] _, _, self.input_height, self.input_width = self.input_shape # Step 3. Initialize Preprocessing for the model ppp = PrePostProcessor(model) # Specify input image format ppp.input().tensor().set_element_type(Type.u8).set_layout( Layout("NHWC") ).set_color_format(ColorFormat.BGR) # Specify preprocess pipeline to input image without resizing ppp.input().preprocess().convert_element_type(Type.f32).convert_color( ColorFormat.RGB ).scale([255.0, 255.0, 255.0]) # Specify model's input layout ppp.input().model().set_layout(Layout("NCHW")) # Specify output results format ppp.output().tensor().set_element_type(Type.f32) # Embed above steps in the graph model = ppp.build() self.compiled_model = core.compile_model(model, "CPU") self.conf_thresh = conf self.nms_thresh = nms self.colors = [] # Create random colors np.random.seed(42) # Setting seed for reproducibility for i in range(len(coconame)): color = tuple(np.random.randint(100, 256, size=3)) self.colors.append(color) def resize_and_pad(self, image): old_h, old_w = image.shape[:2] ratio = min(self.input_width / old_w, self.input_height / old_h) new_w = int(round(old_w * ratio)) new_h = int(round(old_h * ratio)) image = cv2.resize(image, (new_w, new_h)) dw = self.input_width - new_w dh = self.input_height - new_h top = dh // 2 bottom = dh - top left = dw // 2 right = dw - left color = [114, 114, 114] new_im = cv2.copyMakeBorder( image, top, bottom, left, right, cv2.BORDER_CONSTANT, value=color ) return new_im, ratio, (left, top) def predict(self, img): input_tensor = np.expand_dims(img, 0) infer_request = self.compiled_model.create_infer_request() infer_request.infer({0: input_tensor}) output = infer_request.get_output_tensor() detections = output.data[0] # [300, 6] end2end: [x1, y1, x2, y2, conf, class_id] boxes = [] class_ids = [] confidences = [] for detection in detections: x1, y1, x2, y2, confidence, class_id = detection if confidence > self.conf_thresh: confidences.append(float(confidence)) class_ids.append(int(class_id)) boxes.append(np.array([x1, y1, x2 - x1, y2 - y1])) if not boxes: return [] indexes = cv2.dnn.NMSBoxes( boxes, confidences, self.conf_thresh, self.nms_thresh ) results = [] for i in indexes: j = i.item() box = boxes[j] results.append( { "class_index": class_ids[j], "confidence": confidences[j], "box": np.array([box[0], box[1], box[0] + box[2], box[1] + box[3]]), } ) return results def draw(self, img, detections, ratio, pad_left, pad_top): for detection in detections: box = detection["box"] classId = detection["class_index"] confidence = detection["confidence"] xmin = (box[0] - pad_left) / ratio ymin = (box[1] - pad_top) / ratio xmax = (box[2] - pad_left) / ratio ymax = (box[3] - pad_top) / ratio # Drawing detection box cv2.rectangle( img, (int(xmin), int(ymin)), (int(xmax), int(ymax)), tuple(map(int, self.colors[classId])), 3, ) # Detection box text class_string = coconame[classId] + " " + str(confidence)[:4] text_size, _ = cv2.getTextSize(class_string, cv2.FONT_HERSHEY_DUPLEX, 1, 2) text_rect = (xmin, ymin - 40, text_size[0] + 10, text_size[1] + 20) cv2.rectangle( img, (int(text_rect[0]), int(text_rect[1])), (int(text_rect[0] + text_rect[2]), int(text_rect[1] + text_rect[3])), tuple(map(int, self.colors[classId])), cv2.FILLED, ) cv2.putText( img, class_string, (int(xmin + 5), int(ymin - 10)), cv2.FONT_HERSHEY_DUPLEX, 1, (0, 0, 0), 2, cv2.LINE_AA, ) def init_context(context): """Nuclio init_context – called once per container. Loads the IR model and compiles it for the CPU. """ context.logger.info("Init context ----> 0%") model = Yolov9(MODEL_XML, MODEL_BIN, conf=0.1, nms=0.4) context.user_data.model = model context.logger.info("Init context ----> 100%") def handler(context, event): """Nuclio handler – called for every request. Expects a JSON body with a base‑64 encoded image under the key ``"image"``. Returns a CVAT‑compatible JSON with detected objects. """ context.logger.info("Run OpenVINO YOLOv9 model") # Parse request body try: data = event.body image_b64 = data["image"] except Exception as exc: context.logger.error(f"Invalid request body: {exc}") return context.Response( body=json.dumps({"error": "Invalid request body"}), status_code=400, content_type="application/json", ) # Decode image image_bytes = base64.b64decode(image_b64) image = cv2.imdecode(np.frombuffer(image_bytes, np.uint8), cv2.IMREAD_COLOR) if image is None: context.logger.error("Failed to decode image") return context.Response( body=json.dumps({"error": "Failed to decode image"}), status_code=400, content_type="application/json", ) # Get model from context model = context.user_data.model print("Prepare Model") # Preprocess: resize and pad img_resized, ratio, (pad_left, pad_top) = model.resize_and_pad(image) # Inference detections = model.predict(img_resized) # Convert detections to CVAT-compatible format shapes = [] for detection in detections: class_id = detection["class_index"] confidence = float(detection["confidence"]) box = detection["box"] x1 = (box[0] - pad_left) / ratio y1 = (box[1] - pad_top) / ratio x2 = (box[2] - pad_left) / ratio y2 = (box[3] - pad_top) / ratio # Convert to pixel coordinates x_min_px = int(max(0, x1)) y_min_px = int(max(0, y1)) x_max_px = int(min(image.shape[1], x2)) y_max_px = int(min(image.shape[0], y2)) label = coconame[class_id] if class_id < len(coconame) else "unknown" shapes.append( { "label": label, "points": [x_min_px, y_min_px, x_max_px, y_max_px], "type": "rectangle", "confidence": str(confidence), } ) context.logger.info(f"Detected {len(shapes)} objects") return context.Response( body=json.dumps(shapes), headers={}, content_type="application/json", status_code=200, )