First Commit

This commit is contained in:
proitlab committed 2026-08-01 18:30:02 +07:00
1 parent 382f0e4cbe
commit fa88919f59
7 files changed
+2062

No files matched your search

+295
View File
@@ -0,0 +1,295 @@
"""Nuclio handler for CVAT automatic annotation using OpenVINO 2025 IR (.xml/.bin).
This file combines YOLOv9 inference logic with Nuclio serverless handler structure.
It loads an OpenVINO Intermediate Representation (IR) model consisting of a
``.xml`` file (network topology) and a ``.bin`` file (weights).
Adjust ``MODEL_XML`` and ``MODEL_BIN`` if your files are located elsewhere.
"""
import base64
import json
import os
from pathlib import Path
import cv2
import numpy as np
import openvino as ov
from openvino.preprocess import PrePostProcessor
from openvino.preprocess import ColorFormat
from openvino import Layout, Type
# Paths to the IR model files – change if your model is in a different location.
MODEL_XML = os.getenv("MODEL_XML","/models/best.xml")
MODEL_BIN = os.getenv("MODEL_BIN", "/models/best.bin")
# Read class names from JSON file
classes_file = os.getenv("CLASSES_FILE", "/opt/nuclio/classes.json")
with open(classes_file, 'r') as f:
classes_data = json.load(f)
coconame = [cls['name'] for cls in sorted(classes_data, key=lambda x: x['id'])]
class Yolov9:
def __init__(
self, xml_model_path=MODEL_XML, bin_model_path=MODEL_BIN, conf=0.1, nms=0.4
):
# Step 1. Initialize OpenVINO Runtime core
core = ov.Core()
# Step 2. Read a model
if bin_model_path:
model = core.read_model(
str(Path(xml_model_path)), str(Path(bin_model_path))
)
else:
model = core.read_model(str(Path(xml_model_path)))
# Step 3. Initialize Preprocessing for the model
ppp = PrePostProcessor(model)
# Specify input image format
ppp.input().tensor().set_element_type(Type.u8).set_layout(
Layout("NHWC")
).set_color_format(ColorFormat.BGR)
# Specify preprocess pipeline to input image without resizing
ppp.input().preprocess().convert_element_type(Type.f32).convert_color(
ColorFormat.RGB
).scale([255.0, 255.0, 255.0])
# Specify model's input layout
ppp.input().model().set_layout(Layout("NCHW"))
# Specify output results format
ppp.output().tensor().set_element_type(Type.f32)
# Embed above steps in the graph
model = ppp.build()
self.compiled_model = core.compile_model(model, "CPU")
#self.input_shape = self.compiled_model.input(0).shape
#_, _, self.input_height, self.input_width = self.input_shape
self.input_width = 640
self.input_height = 640
self.conf_thresh = conf
self.nms_thresh = nms
self.colors = []
# Create random colors
np.random.seed(42) # Setting seed for reproducibility
for i in range(len(coconame)):
color = tuple(np.random.randint(100, 256, size=3))
self.colors.append(color)
def resize_and_pad(self, image):
old_h, old_w = image.shape[:2]
ratio = min(self.input_width / old_w, self.input_height / old_h)
new_w = int(old_w * ratio)
new_h = int(old_h * ratio)
image = cv2.resize(image, (new_w, new_h))
delta_w = self.input_width - new_w
delta_h = self.input_height - new_h
color = [100, 100, 100]
new_im = cv2.copyMakeBorder(
image, 0, delta_h, 0, delta_w, cv2.BORDER_CONSTANT, value=color
)
return new_im, delta_w, delta_h
def predict(self, img):
# Step 4. Create tensor from image
input_tensor = np.expand_dims(img, 0)
# Step 5. Create an infer request for model inference
infer_request = self.compiled_model.create_infer_request()
infer_request.infer({0: input_tensor})
# Step 6. Retrieve inference results
output = infer_request.get_output_tensor()
detections = output.data[0].T
# Step 7. Postprocessing including NMS
boxes = []
class_ids = []
confidences = []
for prediction in detections:
classes_scores = prediction[4:]
_, _, _, max_indx = cv2.minMaxLoc(classes_scores)
class_id = max_indx[1]
if classes_scores[class_id] > self.conf_thresh:
confidences.append(classes_scores[class_id])
class_ids.append(class_id)
x, y, w, h = (
prediction[0].item(),
prediction[1].item(),
prediction[2].item(),
prediction[3].item(),
)
xmin = x - (w / 2)
ymin = y - (h / 2)
box = np.array([xmin, ymin, w, h])
boxes.append(box)
indexes = cv2.dnn.NMSBoxes(
boxes, confidences, self.conf_thresh, self.nms_thresh
)
results = []
for i in indexes:
j = i.item()
results.append(
{
"class_index": class_ids[j],
"confidence": confidences[j],
"box": boxes[j],
}
)
return results
def draw(self, img, detections, dw, dh):
# Step 8. Print results and save Figure with detections
for detection in detections:
box = detection["box"]
classId = detection["class_index"]
confidence = detection["confidence"]
rx = img.shape[1] / (self.input_width - dw)
ry = img.shape[0] / (self.input_height - dh)
box[0] = rx * box[0]
box[1] = ry * box[1]
box[2] = rx * box[2]
box[3] = ry * box[3]
xmax = box[0] + box[2]
ymax = box[1] + box[3]
# Drawing detection box
cv2.rectangle(
img,
(int(box[0]), int(box[1])),
(int(xmax), int(ymax)),
tuple(map(int, self.colors[classId])),
3,
)
# Detection box text
class_string = coconame[classId] + " " + str(confidence)[:4]
text_size, _ = cv2.getTextSize(class_string, cv2.FONT_HERSHEY_DUPLEX, 1, 2)
text_rect = (box[0], box[1] - 40, text_size[0] + 10, text_size[1] + 20)
cv2.rectangle(
img,
(int(text_rect[0]), int(text_rect[1])),
(int(text_rect[0] + text_rect[2]), int(text_rect[1] + text_rect[3])),
tuple(map(int, self.colors[classId])),
cv2.FILLED,
)
cv2.putText(
img,
class_string,
(int(box[0] + 5), int(box[1] - 10)),
cv2.FONT_HERSHEY_DUPLEX,
1,
(0, 0, 0),
2,
cv2.LINE_AA,
)
def init_context(context):
"""Nuclio init_context – called once per container.
Loads the IR model and compiles it for the CPU.
"""
context.logger.info("Init context ----> 0%")
model = Yolov9(MODEL_XML, MODEL_BIN, conf=0.1, nms=0.4)
context.user_data.model = model
context.logger.info("Init context ----> 100%")
def handler(context, event):
"""Nuclio handler – called for every request.
Expects a JSON body with a base‑64 encoded image under the key ``"image"``.
Returns a CVAT‑compatible JSON with detected objects.
"""
context.logger.info("Run OpenVINO YOLOv9 model")
# Parse request body
try:
data = event.body
image_b64 = data["image"]
except Exception as exc:
context.logger.error(f"Invalid request body: {exc}")
return context.Response(
body=json.dumps({"error": "Invalid request body"}),
status_code=400,
content_type="application/json",
)
# Decode image
image_bytes = base64.b64decode(image_b64)
image = cv2.imdecode(np.frombuffer(image_bytes, np.uint8), cv2.IMREAD_COLOR)
if image is None:
context.logger.error("Failed to decode image")
return context.Response(
body=json.dumps({"error": "Failed to decode image"}),
status_code=400,
content_type="application/json",
)
# Get model from context
model = context.user_data.model
print("Prepare Model")
# Preprocess: resize and pad
img_resized, dw, dh = model.resize_and_pad(image)
#print("Resize Image")
# Inference
detections = model.predict(img_resized)
#print("Detecion")
# Convert detections to CVAT-compatible format
shapes = []
for detection in detections:
class_id = detection["class_index"]
confidence = float(detection["confidence"])
box = detection["box"]
# Scale box coordinates back to original image size
rx = image.shape[1] / (model.input_width - dw)
ry = image.shape[0] / (model.input_height - dh)
xmin = box[0] * rx
ymin = box[1] * ry
xmax = (box[0] + box[2]) * rx
ymax = (box[1] + box[3]) * ry
# Convert to pixel coordinates
x_min_px = int(max(0, xmin))
y_min_px = int(max(0, ymin))
x_max_px = int(min(image.shape[1], xmax))
y_max_px = int(min(image.shape[0], ymax))
label = coconame[class_id] if class_id < len(coconame) else "unknown"
shapes.append(
{
"label": label,
"points": [x_min_px, y_min_px, x_max_px, y_max_px],
"type": "rectangle",
"confidence": str(confidence),
}
)
context.logger.info(f"Detected {len(shapes)} objects")
return context.Response(
body=json.dumps(shapes),
headers={},
content_type="application/json",
status_code=200,
)
+275
View File
@@ -0,0 +1,275 @@
"""Nuclio handler for CVAT automatic annotation using OpenVINO 2025 IR (.xml/.bin).
This file combines YOLOv9 inference logic with Nuclio serverless handler structure.
It loads an OpenVINO Intermediate Representation (IR) model consisting of a
``.xml`` file (network topology) and a ``.bin`` file (weights).
Adjust ``MODEL_XML`` and ``MODEL_BIN`` if your files are located elsewhere.
"""
import base64
import json
import os
from pathlib import Path
import cv2
import numpy as np
import openvino as ov
from openvino.preprocess import PrePostProcessor
from openvino.preprocess import ColorFormat
from openvino import Layout, Type
#MODEL_DIR = "chicken-detection-model-v26n-300e-best-2026-05-02-NEW_openvino_model"
#MODEL_NAME = "chicken-detection-model-v26n-300e-best-2026-05-02-NEW"
#MODEL_XML = os.getenv("MODEL_XML", f"/models/{MODEL_DIR}/{MODEL_NAME}.xml")
#MODEL_BIN = os.getenv("MODEL_BIN", f"/models/{MODEL_DIR}/{MODEL_NAME}.bin")
MODEL_XML = os.getenv("MODEL_XML","/models/best.xml")
MODEL_BIN = os.getenv("MODEL_BIN", "/models/best.bin")
classes_file = os.getenv("CLASSES_FILE", "/opt/nuclio/classes.json")
try:
with open(classes_file, 'r') as f:
classes_data = json.load(f)
coconame = [cls['name'] for cls in sorted(classes_data, key=lambda x: x['id'])]
except (FileNotFoundError, json.JSONDecodeError):
coconame = ["chicken", "not-chicken", "half-chicken"]
class Yolov9:
def __init__(
self, xml_model_path=MODEL_XML, bin_model_path=MODEL_BIN, conf=0.1, nms=0.4
):
# Step 1. Initialize OpenVINO Runtime core
core = ov.Core()
# Step 2. Read a model
if bin_model_path:
model = core.read_model(
str(Path(xml_model_path)), str(Path(bin_model_path))
)
else:
model = core.read_model(str(Path(xml_model_path)))
self.input_shape = model.input(0).shape # NCHW: [1, C, H, W]
_, _, self.input_height, self.input_width = self.input_shape
# Step 3. Initialize Preprocessing for the model
ppp = PrePostProcessor(model)
# Specify input image format
ppp.input().tensor().set_element_type(Type.u8).set_layout(
Layout("NHWC")
).set_color_format(ColorFormat.BGR)
# Specify preprocess pipeline to input image without resizing
ppp.input().preprocess().convert_element_type(Type.f32).convert_color(
ColorFormat.RGB
).scale([255.0, 255.0, 255.0])
# Specify model's input layout
ppp.input().model().set_layout(Layout("NCHW"))
# Specify output results format
ppp.output().tensor().set_element_type(Type.f32)
# Embed above steps in the graph
model = ppp.build()
self.compiled_model = core.compile_model(model, "CPU")
self.conf_thresh = conf
self.nms_thresh = nms
self.colors = []
# Create random colors
np.random.seed(42) # Setting seed for reproducibility
for i in range(len(coconame)):
color = tuple(np.random.randint(100, 256, size=3))
self.colors.append(color)
def resize_and_pad(self, image):
old_h, old_w = image.shape[:2]
ratio = min(self.input_width / old_w, self.input_height / old_h)
new_w = int(round(old_w * ratio))
new_h = int(round(old_h * ratio))
image = cv2.resize(image, (new_w, new_h))
dw = self.input_width - new_w
dh = self.input_height - new_h
top = dh // 2
bottom = dh - top
left = dw // 2
right = dw - left
color = [114, 114, 114]
new_im = cv2.copyMakeBorder(
image, top, bottom, left, right, cv2.BORDER_CONSTANT, value=color
)
return new_im, ratio, (left, top)
def predict(self, img):
input_tensor = np.expand_dims(img, 0)
infer_request = self.compiled_model.create_infer_request()
infer_request.infer({0: input_tensor})
output = infer_request.get_output_tensor()
detections = output.data[0] # [300, 6] end2end: [x1, y1, x2, y2, conf, class_id]
boxes = []
class_ids = []
confidences = []
for detection in detections:
x1, y1, x2, y2, confidence, class_id = detection
if confidence > self.conf_thresh:
confidences.append(float(confidence))
class_ids.append(int(class_id))
boxes.append(np.array([x1, y1, x2 - x1, y2 - y1]))
if not boxes:
return []
indexes = cv2.dnn.NMSBoxes(
boxes, confidences, self.conf_thresh, self.nms_thresh
)
results = []
for i in indexes:
j = i.item()
box = boxes[j]
results.append(
{
"class_index": class_ids[j],
"confidence": confidences[j],
"box": np.array([box[0], box[1], box[0] + box[2], box[1] + box[3]]),
}
)
return results
def draw(self, img, detections, ratio, pad_left, pad_top):
for detection in detections:
box = detection["box"]
classId = detection["class_index"]
confidence = detection["confidence"]
xmin = (box[0] - pad_left) / ratio
ymin = (box[1] - pad_top) / ratio
xmax = (box[2] - pad_left) / ratio
ymax = (box[3] - pad_top) / ratio
# Drawing detection box
cv2.rectangle(
img,
(int(xmin), int(ymin)),
(int(xmax), int(ymax)),
tuple(map(int, self.colors[classId])),
3,
)
# Detection box text
class_string = coconame[classId] + " " + str(confidence)[:4]
text_size, _ = cv2.getTextSize(class_string, cv2.FONT_HERSHEY_DUPLEX, 1, 2)
text_rect = (xmin, ymin - 40, text_size[0] + 10, text_size[1] + 20)
cv2.rectangle(
img,
(int(text_rect[0]), int(text_rect[1])),
(int(text_rect[0] + text_rect[2]), int(text_rect[1] + text_rect[3])),
tuple(map(int, self.colors[classId])),
cv2.FILLED,
)
cv2.putText(
img,
class_string,
(int(xmin + 5), int(ymin - 10)),
cv2.FONT_HERSHEY_DUPLEX,
1,
(0, 0, 0),
2,
cv2.LINE_AA,
)
def init_context(context):
"""Nuclio init_context – called once per container.
Loads the IR model and compiles it for the CPU.
"""
context.logger.info("Init context ----> 0%")
model = Yolov9(MODEL_XML, MODEL_BIN, conf=0.1, nms=0.4)
context.user_data.model = model
context.logger.info("Init context ----> 100%")
def handler(context, event):
"""Nuclio handler – called for every request.
Expects a JSON body with a base‑64 encoded image under the key ``"image"``.
Returns a CVAT‑compatible JSON with detected objects.
"""
context.logger.info("Run OpenVINO YOLOv9 model")
# Parse request body
try:
data = event.body
image_b64 = data["image"]
except Exception as exc:
context.logger.error(f"Invalid request body: {exc}")
return context.Response(
body=json.dumps({"error": "Invalid request body"}),
status_code=400,
content_type="application/json",
)
# Decode image
image_bytes = base64.b64decode(image_b64)
image = cv2.imdecode(np.frombuffer(image_bytes, np.uint8), cv2.IMREAD_COLOR)
if image is None:
context.logger.error("Failed to decode image")
return context.Response(
body=json.dumps({"error": "Failed to decode image"}),
status_code=400,
content_type="application/json",
)
# Get model from context
model = context.user_data.model
print("Prepare Model")
# Preprocess: resize and pad
img_resized, ratio, (pad_left, pad_top) = model.resize_and_pad(image)
# Inference
detections = model.predict(img_resized)
# Convert detections to CVAT-compatible format
shapes = []
for detection in detections:
class_id = detection["class_index"]
confidence = float(detection["confidence"])
box = detection["box"]
x1 = (box[0] - pad_left) / ratio
y1 = (box[1] - pad_top) / ratio
x2 = (box[2] - pad_left) / ratio
y2 = (box[3] - pad_top) / ratio
# Convert to pixel coordinates
x_min_px = int(max(0, x1))
y_min_px = int(max(0, y1))
x_max_px = int(min(image.shape[1], x2))
y_max_px = int(min(image.shape[0], y2))
label = coconame[class_id] if class_id < len(coconame) else "unknown"
shapes.append(
{
"label": label,
"points": [x_min_px, y_min_px, x_max_px, y_max_px],
"type": "rectangle",
"confidence": str(confidence),
}
)
context.logger.info(f"Detected {len(shapes)} objects")
return context.Response(
body=json.dumps(shapes),
headers={},
content_type="application/json",
status_code=200,
)
+273
View File
@@ -0,0 +1,273 @@
"""Nuclio handler for CVAT automatic annotation using OpenVINO 2025 IR (.xml/.bin).
This file combines YOLOv9 inference logic with Nuclio serverless handler structure.
It loads an OpenVINO Intermediate Representation (IR) model consisting of a
``.xml`` file (network topology) and a ``.bin`` file (weights).
Adjust ``MODEL_XML`` and ``MODEL_BIN`` if your files are located elsewhere.
"""
import base64
import json
import os
from pathlib import Path
import cv2
import numpy as np
import openvino as ov
from openvino.preprocess import PrePostProcessor
from openvino.preprocess import ColorFormat
from openvino import Layout, Type
MODEL_DIR = "chicken-detection-model-v26n-300e-best-2026-05-02-NEW_openvino_model"
MODEL_NAME = "chicken-detection-model-v26n-300e-best-2026-05-02-NEW"
MODEL_XML = os.getenv("MODEL_XML", f"/models/{MODEL_DIR}/{MODEL_NAME}.xml")
MODEL_BIN = os.getenv("MODEL_BIN", f"/models/{MODEL_DIR}/{MODEL_NAME}.bin")
classes_file = os.getenv("CLASSES_FILE", "/opt/nuclio/classes.json")
try:
with open(classes_file, 'r') as f:
classes_data = json.load(f)
coconame = [cls['name'] for cls in sorted(classes_data, key=lambda x: x['id'])]
except (FileNotFoundError, json.JSONDecodeError):
coconame = ["chicken", "not-chicken", "half-chicken"]
class Yolov9:
def __init__(
self, xml_model_path=MODEL_XML, bin_model_path=MODEL_BIN, conf=0.1, nms=0.4
):
# Step 1. Initialize OpenVINO Runtime core
core = ov.Core()
# Step 2. Read a model
if bin_model_path:
model = core.read_model(
str(Path(xml_model_path)), str(Path(bin_model_path))
)
else:
model = core.read_model(str(Path(xml_model_path)))
self.input_shape = model.input(0).shape # NCHW: [1, C, H, W]
_, _, self.input_height, self.input_width = self.input_shape
# Step 3. Initialize Preprocessing for the model
ppp = PrePostProcessor(model)
# Specify input image format
ppp.input().tensor().set_element_type(Type.u8).set_layout(
Layout("NHWC")
).set_color_format(ColorFormat.BGR)
# Specify preprocess pipeline to input image without resizing
ppp.input().preprocess().convert_element_type(Type.f32).convert_color(
ColorFormat.RGB
).scale([255.0, 255.0, 255.0])
# Specify model's input layout
ppp.input().model().set_layout(Layout("NCHW"))
# Specify output results format
ppp.output().tensor().set_element_type(Type.f32)
# Embed above steps in the graph
model = ppp.build()
self.compiled_model = core.compile_model(model, "CPU")
self.conf_thresh = conf
self.nms_thresh = nms
self.colors = []
# Create random colors
np.random.seed(42) # Setting seed for reproducibility
for i in range(len(coconame)):
color = tuple(np.random.randint(100, 256, size=3))
self.colors.append(color)
def resize_and_pad(self, image):
old_h, old_w = image.shape[:2]
ratio = min(self.input_width / old_w, self.input_height / old_h)
new_w = int(round(old_w * ratio))
new_h = int(round(old_h * ratio))
image = cv2.resize(image, (new_w, new_h))
dw = self.input_width - new_w
dh = self.input_height - new_h
top = dh // 2
bottom = dh - top
left = dw // 2
right = dw - left
color = [114, 114, 114]
new_im = cv2.copyMakeBorder(
image, top, bottom, left, right, cv2.BORDER_CONSTANT, value=color
)
return new_im, ratio, (left, top)
def predict(self, img):
input_tensor = np.expand_dims(img, 0)
infer_request = self.compiled_model.create_infer_request()
infer_request.infer({0: input_tensor})
output = infer_request.get_output_tensor()
detections = output.data[0] # [300, 6] end2end: [x1, y1, x2, y2, conf, class_id]
boxes = []
class_ids = []
confidences = []
for detection in detections:
x1, y1, x2, y2, confidence, class_id = detection
if confidence > self.conf_thresh:
confidences.append(float(confidence))
class_ids.append(int(class_id))
boxes.append(np.array([x1, y1, x2 - x1, y2 - y1]))
if not boxes:
return []
indexes = cv2.dnn.NMSBoxes(
boxes, confidences, self.conf_thresh, self.nms_thresh
)
results = []
for i in indexes:
j = i.item()
box = boxes[j]
results.append(
{
"class_index": class_ids[j],
"confidence": confidences[j],
"box": np.array([box[0], box[1], box[0] + box[2], box[1] + box[3]]),
}
)
return results
def draw(self, img, detections, ratio, pad_left, pad_top):
for detection in detections:
box = detection["box"]
classId = detection["class_index"]
confidence = detection["confidence"]
xmin = (box[0] - pad_left) / ratio
ymin = (box[1] - pad_top) / ratio
xmax = (box[2] - pad_left) / ratio
ymax = (box[3] - pad_top) / ratio
# Drawing detection box
cv2.rectangle(
img,
(int(xmin), int(ymin)),
(int(xmax), int(ymax)),
tuple(map(int, self.colors[classId])),
3,
)
# Detection box text
class_string = coconame[classId] + " " + str(confidence)[:4]
text_size, _ = cv2.getTextSize(class_string, cv2.FONT_HERSHEY_DUPLEX, 1, 2)
text_rect = (xmin, ymin - 40, text_size[0] + 10, text_size[1] + 20)
cv2.rectangle(
img,
(int(text_rect[0]), int(text_rect[1])),
(int(text_rect[0] + text_rect[2]), int(text_rect[1] + text_rect[3])),
tuple(map(int, self.colors[classId])),
cv2.FILLED,
)
cv2.putText(
img,
class_string,
(int(xmin + 5), int(ymin - 10)),
cv2.FONT_HERSHEY_DUPLEX,
1,
(0, 0, 0),
2,
cv2.LINE_AA,
)
def init_context(context):
"""Nuclio init_context – called once per container.
Loads the IR model and compiles it for the CPU.
"""
context.logger.info("Init context ----> 0%")
model = Yolov9(MODEL_XML, MODEL_BIN, conf=0.1, nms=0.4)
context.user_data.model = model
context.logger.info("Init context ----> 100%")
def handler(context, event):
"""Nuclio handler – called for every request.
Expects a JSON body with a base‑64 encoded image under the key ``"image"``.
Returns a CVAT‑compatible JSON with detected objects.
"""
context.logger.info("Run OpenVINO YOLOv9 model")
# Parse request body
try:
data = event.body
image_b64 = data["image"]
except Exception as exc:
context.logger.error(f"Invalid request body: {exc}")
return context.Response(
body=json.dumps({"error": "Invalid request body"}),
status_code=400,
content_type="application/json",
)
# Decode image
image_bytes = base64.b64decode(image_b64)
image = cv2.imdecode(np.frombuffer(image_bytes, np.uint8), cv2.IMREAD_COLOR)
if image is None:
context.logger.error("Failed to decode image")
return context.Response(
body=json.dumps({"error": "Failed to decode image"}),
status_code=400,
content_type="application/json",
)
# Get model from context
model = context.user_data.model
print("Prepare Model")
# Preprocess: resize and pad
img_resized, ratio, (pad_left, pad_top) = model.resize_and_pad(image)
# Inference
detections = model.predict(img_resized)
# Convert detections to CVAT-compatible format
shapes = []
for detection in detections:
class_id = detection["class_index"]
confidence = float(detection["confidence"])
box = detection["box"]
x1 = (box[0] - pad_left) / ratio
y1 = (box[1] - pad_top) / ratio
x2 = (box[2] - pad_left) / ratio
y2 = (box[3] - pad_top) / ratio
# Convert to pixel coordinates
x_min_px = int(max(0, x1))
y_min_px = int(max(0, y1))
x_max_px = int(min(image.shape[1], x2))
y_max_px = int(min(image.shape[0], y2))
label = coconame[class_id] if class_id < len(coconame) else "unknown"
shapes.append(
{
"label": label,
"points": [x_min_px, y_min_px, x_max_px, y_max_px],
"type": "rectangle",
"confidence": str(confidence),
}
)
context.logger.info(f"Detected {len(shapes)} objects")
return context.Response(
body=json.dumps(shapes),
headers={},
content_type="application/json",
status_code=200,
)
+295
View File
@@ -0,0 +1,295 @@
"""Nuclio handler for CVAT automatic annotation using OpenVINO 2025 IR (.xml/.bin).
This file combines YOLOv9 inference logic with Nuclio serverless handler structure.
It loads an OpenVINO Intermediate Representation (IR) model consisting of a
``.xml`` file (network topology) and a ``.bin`` file (weights).
Adjust ``MODEL_XML`` and ``MODEL_BIN`` if your files are located elsewhere.
"""
import base64
import json
import os
from pathlib import Path
import cv2
import numpy as np
import openvino as ov
from openvino.preprocess import PrePostProcessor
from openvino.preprocess import ColorFormat
from openvino import Layout, Type
# Paths to the IR model files – change if your model is in a different location.
MODEL_XML = os.getenv("MODEL_XML","/models/best.xml")
MODEL_BIN = os.getenv("MODEL_BIN", "/models/best.bin")
# Read class names from JSON file
classes_file = os.getenv("CLASSES_FILE", "/opt/nuclio/classes.json")
with open(classes_file, 'r') as f:
classes_data = json.load(f)
coconame = [cls['name'] for cls in sorted(classes_data, key=lambda x: x['id'])]
class Yolov9:
def __init__(
self, xml_model_path=MODEL_XML, bin_model_path=MODEL_BIN, conf=0.1, nms=0.4
):
# Step 1. Initialize OpenVINO Runtime core
core = ov.Core()
# Step 2. Read a model
if bin_model_path:
model = core.read_model(
str(Path(xml_model_path)), str(Path(bin_model_path))
)
else:
model = core.read_model(str(Path(xml_model_path)))
# Step 3. Initialize Preprocessing for the model
ppp = PrePostProcessor(model)
# Specify input image format
ppp.input().tensor().set_element_type(Type.u8).set_layout(
Layout("NHWC")
).set_color_format(ColorFormat.BGR)
# Specify preprocess pipeline to input image without resizing
ppp.input().preprocess().convert_element_type(Type.f32).convert_color(
ColorFormat.RGB
).scale([255.0, 255.0, 255.0])
# Specify model's input layout
ppp.input().model().set_layout(Layout("NCHW"))
# Specify output results format
ppp.output().tensor().set_element_type(Type.f32)
# Embed above steps in the graph
model = ppp.build()
self.compiled_model = core.compile_model(model, "CPU")
#self.input_shape = self.compiled_model.input(0).shape
#_, _, self.input_height, self.input_width = self.input_shape
self.input_width = 320
self.input_height = 320
self.conf_thresh = conf
self.nms_thresh = nms
self.colors = []
# Create random colors
np.random.seed(42) # Setting seed for reproducibility
for i in range(len(coconame)):
color = tuple(np.random.randint(100, 256, size=3))
self.colors.append(color)
def resize_and_pad(self, image):
old_h, old_w = image.shape[:2]
ratio = min(self.input_width / old_w, self.input_height / old_h)
new_w = int(old_w * ratio)
new_h = int(old_h * ratio)
image = cv2.resize(image, (new_w, new_h))
delta_w = self.input_width - new_w
delta_h = self.input_height - new_h
color = [100, 100, 100]
new_im = cv2.copyMakeBorder(
image, 0, delta_h, 0, delta_w, cv2.BORDER_CONSTANT, value=color
)
return new_im, delta_w, delta_h
def predict(self, img):
# Step 4. Create tensor from image
input_tensor = np.expand_dims(img, 0)
# Step 5. Create an infer request for model inference
infer_request = self.compiled_model.create_infer_request()
infer_request.infer({0: input_tensor})
# Step 6. Retrieve inference results
output = infer_request.get_output_tensor()
detections = output.data[0].T
# Step 7. Postprocessing including NMS
boxes = []
class_ids = []
confidences = []
for prediction in detections:
classes_scores = prediction[4:]
_, _, _, max_indx = cv2.minMaxLoc(classes_scores)
class_id = max_indx[1]
if classes_scores[class_id] > self.conf_thresh:
confidences.append(classes_scores[class_id])
class_ids.append(class_id)
x, y, w, h = (
prediction[0].item(),
prediction[1].item(),
prediction[2].item(),
prediction[3].item(),
)
xmin = x - (w / 2)
ymin = y - (h / 2)
box = np.array([xmin, ymin, w, h])
boxes.append(box)
indexes = cv2.dnn.NMSBoxes(
boxes, confidences, self.conf_thresh, self.nms_thresh
)
results = []
for i in indexes:
j = i.item()
results.append(
{
"class_index": class_ids[j],
"confidence": confidences[j],
"box": boxes[j],
}
)
return results
def draw(self, img, detections, dw, dh):
# Step 8. Print results and save Figure with detections
for detection in detections:
box = detection["box"]
classId = detection["class_index"]
confidence = detection["confidence"]
rx = img.shape[1] / (self.input_width - dw)
ry = img.shape[0] / (self.input_height - dh)
box[0] = rx * box[0]
box[1] = ry * box[1]
box[2] = rx * box[2]
box[3] = ry * box[3]
xmax = box[0] + box[2]
ymax = box[1] + box[3]
# Drawing detection box
cv2.rectangle(
img,
(int(box[0]), int(box[1])),
(int(xmax), int(ymax)),
tuple(map(int, self.colors[classId])),
3,
)
# Detection box text
class_string = coconame[classId] + " " + str(confidence)[:4]
text_size, _ = cv2.getTextSize(class_string, cv2.FONT_HERSHEY_DUPLEX, 1, 2)
text_rect = (box[0], box[1] - 40, text_size[0] + 10, text_size[1] + 20)
cv2.rectangle(
img,
(int(text_rect[0]), int(text_rect[1])),
(int(text_rect[0] + text_rect[2]), int(text_rect[1] + text_rect[3])),
tuple(map(int, self.colors[classId])),
cv2.FILLED,
)
cv2.putText(
img,
class_string,
(int(box[0] + 5), int(box[1] - 10)),
cv2.FONT_HERSHEY_DUPLEX,
1,
(0, 0, 0),
2,
cv2.LINE_AA,
)
def init_context(context):
"""Nuclio init_context – called once per container.
Loads the IR model and compiles it for the CPU.
"""
context.logger.info("Init context ----> 0%")
model = Yolov9(MODEL_XML, MODEL_BIN, conf=0.1, nms=0.4)
context.user_data.model = model
context.logger.info("Init context ----> 100%")
def handler(context, event):
"""Nuclio handler – called for every request.
Expects a JSON body with a base‑64 encoded image under the key ``"image"``.
Returns a CVAT‑compatible JSON with detected objects.
"""
context.logger.info("Run OpenVINO YOLOv9 model")
# Parse request body
try:
data = event.body
image_b64 = data["image"]
except Exception as exc:
context.logger.error(f"Invalid request body: {exc}")
return context.Response(
body=json.dumps({"error": "Invalid request body"}),
status_code=400,
content_type="application/json",
)
# Decode image
image_bytes = base64.b64decode(image_b64)
image = cv2.imdecode(np.frombuffer(image_bytes, np.uint8), cv2.IMREAD_COLOR)
if image is None:
context.logger.error("Failed to decode image")
return context.Response(
body=json.dumps({"error": "Failed to decode image"}),
status_code=400,
content_type="application/json",
)
# Get model from context
model = context.user_data.model
print("Prepare Model")
# Preprocess: resize and pad
img_resized, dw, dh = model.resize_and_pad(image)
#print("Resize Image")
# Inference
detections = model.predict(img_resized)
#print("Detecion")
# Convert detections to CVAT-compatible format
shapes = []
for detection in detections:
class_id = detection["class_index"]
confidence = float(detection["confidence"])
box = detection["box"]
# Scale box coordinates back to original image size
rx = image.shape[1] / (model.input_width - dw)
ry = image.shape[0] / (model.input_height - dh)
xmin = box[0] * rx
ymin = box[1] * ry
xmax = (box[0] + box[2]) * rx
ymax = (box[1] + box[3]) * ry
# Convert to pixel coordinates
x_min_px = int(max(0, xmin))
y_min_px = int(max(0, ymin))
x_max_px = int(min(image.shape[1], xmax))
y_max_px = int(min(image.shape[0], ymax))
label = coconame[class_id] if class_id < len(coconame) else "unknown"
shapes.append(
{
"label": label,
"points": [x_min_px, y_min_px, x_max_px, y_max_px],
"type": "rectangle",
"confidence": str(confidence),
}
)
context.logger.info(f"Detected {len(shapes)} objects")
return context.Response(
body=json.dumps(shapes),
headers={},
content_type="application/json",
status_code=200,
)
+338
View File
@@ -0,0 +1,338 @@
"""Nuclio handler for Ultralytics YOLOv8 pose models exported to OpenVINO (CVAT skeleton output)."""
from __future__ import annotations
import base64
import io
import json
import os
from pathlib import Path
from typing import Any
import cv2
import numpy as np
from openvino import Core
from PIL import Image
MODELS_DIR = Path(os.environ.get("MODELS_DIR", "/models"))
CLASSES_PATH = Path(os.environ.get("CLASSES_PATH", "/opt/nuclio/classes.json"))
DEFAULT_IMGSZ = int(os.environ.get("MODEL_IMGSZ", "640"))
DEFAULT_NUM_KEYPOINTS = int(os.environ.get("NUM_KEYPOINTS", "9"))
DEFAULT_CONF_THRESHOLD = float(os.environ.get("CONF_THRESHOLD", "0.25"))
DEFAULT_IOU_THRESHOLD = float(os.environ.get("IOU_THRESHOLD", "0.45"))
def load_classes(path: Path) -> dict[int, str]:
with path.open(encoding="utf-8") as handle:
payload = json.load(handle)
class_map = payload.get("class", payload)
return {int(class_id): str(name) for class_id, name in class_map.items()}
def find_model_xml(models_dir: Path) -> Path:
candidates = [
models_dir / "best.xml",
models_dir / "best_openvino_model" / "best.xml",
*sorted(models_dir.glob("*.xml")),
*sorted(models_dir.glob("**/*.xml")),
]
for candidate in candidates:
if candidate.is_file():
return candidate
raise FileNotFoundError(f"No OpenVINO XML model found under {models_dir}")
def letterbox(
image: np.ndarray,
new_shape: tuple[int, int] = (640, 640),
color: tuple[int, int, int] = (114, 114, 114),
) -> tuple[np.ndarray, float, tuple[float, float]]:
height, width = image.shape[:2]
target_height, target_width = new_shape
scale = min(target_height / height, target_width / width)
new_unpad_width = int(round(width * scale))
new_unpad_height = int(round(height * scale))
resized = cv2.resize(image, (new_unpad_width, new_unpad_height), interpolation=cv2.INTER_LINEAR)
pad_width = target_width - new_unpad_width
pad_height = target_height - new_unpad_height
pad_left = pad_width / 2
pad_top = pad_height / 2
padded = cv2.copyMakeBorder(
resized,
int(round(pad_top - 0.1)),
int(round(pad_height - pad_top)),
int(round(pad_left - 0.1)),
int(round(pad_width - pad_left)),
cv2.BORDER_CONSTANT,
value=color,
)
return padded, scale, (pad_left, pad_top)
def preprocess_image(image_bgr: np.ndarray, imgsz: int) -> tuple[np.ndarray, float, tuple[float, float]]:
letterboxed, scale, pad = letterbox(image_bgr, new_shape=(imgsz, imgsz))
rgb = letterboxed[:, :, ::-1].transpose(2, 0, 1)
tensor = np.expand_dims(rgb, axis=0).astype(np.float32) / 255.0
return tensor, scale, pad
def xywh_to_xyxy(boxes: np.ndarray) -> np.ndarray:
converted = np.empty_like(boxes)
converted[:, 0] = boxes[:, 0] - boxes[:, 2] / 2
converted[:, 1] = boxes[:, 1] - boxes[:, 3] / 2
converted[:, 2] = boxes[:, 0] + boxes[:, 2] / 2
converted[:, 3] = boxes[:, 1] + boxes[:, 3] / 2
return converted
def box_iou(box: np.ndarray, boxes: np.ndarray) -> np.ndarray:
inter_x1 = np.maximum(box[0], boxes[:, 0])
inter_y1 = np.maximum(box[1], boxes[:, 1])
inter_x2 = np.minimum(box[2], boxes[:, 2])
inter_y2 = np.minimum(box[3], boxes[:, 3])
inter_area = np.maximum(0.0, inter_x2 - inter_x1) * np.maximum(0.0, inter_y2 - inter_y1)
box_area = (box[2] - box[0]) * (box[3] - box[1])
boxes_area = (boxes[:, 2] - boxes[:, 0]) * (boxes[:, 3] - boxes[:, 1])
return inter_area / (box_area + boxes_area - inter_area + 1e-6)
def non_max_suppression(
boxes: np.ndarray,
scores: np.ndarray,
iou_threshold: float,
max_detections: int = 300,
) -> list[int]:
order = scores.argsort()[::-1]
keep: list[int] = []
while order.size > 0 and len(keep) < max_detections:
current = int(order[0])
keep.append(current)
if order.size == 1:
break
remaining = order[1:]
ious = box_iou(boxes[current], boxes[remaining])
order = remaining[ious <= iou_threshold]
return keep
def scale_boxes(
input_shape: tuple[int, int],
boxes: np.ndarray,
image_shape: tuple[int, int],
scale: float,
pad: tuple[float, float],
) -> np.ndarray:
boxes = boxes.copy()
pad_x, pad_y = pad
boxes[:, [0, 2]] -= pad_x
boxes[:, [1, 3]] -= pad_y
boxes[:, :4] /= scale
boxes[:, [0, 2]] = boxes[:, [0, 2]].clip(0, image_shape[1])
boxes[:, [1, 3]] = boxes[:, [1, 3]].clip(0, image_shape[0])
return boxes
def scale_keypoints(
keypoints: np.ndarray,
scale: float,
pad: tuple[float, float],
image_shape: tuple[int, int],
) -> np.ndarray:
scaled = keypoints.copy()
pad_x, pad_y = pad
scaled[..., 0] -= pad_x
scaled[..., 1] -= pad_y
scaled[..., :2] /= scale
scaled[..., 0] = scaled[..., 0].clip(0, image_shape[1])
scaled[..., 1] = scaled[..., 1].clip(0, image_shape[0])
return scaled
def parse_pose_output(
output: np.ndarray,
num_classes: int,
num_keypoints: int,
) -> np.ndarray:
if output.ndim == 3:
predictions = output[0].T
elif output.ndim == 2:
predictions = output
else:
raise ValueError(f"Unexpected model output rank: {output.ndim}")
expected_channels = 4 + num_classes + num_keypoints * 3
if predictions.shape[1] != expected_channels:
raise ValueError(
f"Expected {expected_channels} output channels for "
f"{num_classes} classes and {num_keypoints} keypoints, "
f"got {predictions.shape[1]}"
)
return predictions
def decode_detections(
predictions: np.ndarray,
class_names: dict[int, str],
num_keypoints: int,
input_shape: tuple[int, int],
image_shape: tuple[int, int],
scale: float,
pad: tuple[float, float],
conf_threshold: float,
iou_threshold: float,
) -> list[dict[str, Any]]:
boxes_xywh = predictions[:, :4]
class_scores = predictions[:, 4 : 4 + len(class_names)]
keypoints = predictions[:, 4 + len(class_names) :].reshape(-1, num_keypoints, 3)
class_ids = np.argmax(class_scores, axis=1)
confidences = class_scores[np.arange(class_scores.shape[0]), class_ids]
mask = confidences >= conf_threshold
if not np.any(mask):
return []
boxes_xywh = boxes_xywh[mask]
boxes_xyxy = xywh_to_xyxy(boxes_xywh)
keypoints = keypoints[mask]
class_ids = class_ids[mask]
confidences = confidences[mask]
keep = non_max_suppression(boxes_xyxy, confidences, iou_threshold=iou_threshold)
if not keep:
return []
boxes_xyxy = boxes_xyxy[keep]
keypoints = keypoints[keep]
class_ids = class_ids[keep]
confidences = confidences[keep]
image_hw = (image_shape[0], image_shape[1])
boxes_xyxy = scale_boxes(input_shape, boxes_xyxy, image_hw, scale, pad)
keypoints = scale_keypoints(keypoints, scale, pad, image_hw)
detections: list[dict[str, Any]] = []
for box, kpts, class_id, confidence in zip(boxes_xyxy, keypoints, class_ids, confidences):
detections.append(
{
"class_id": int(class_id),
"label": class_names[int(class_id)],
"confidence": float(confidence),
"box": box,
"keypoints": kpts,
}
)
return detections
def build_skeleton_results(
detections: list[dict[str, Any]],
num_keypoints: int,
conf_threshold: float,
) -> list[dict[str, Any]]:
results: list[dict[str, Any]] = []
sublabel_names = [str(index) for index in range(num_keypoints)]
for detection in detections:
elements = []
for index, sublabel_name in enumerate(sublabel_names):
x_coord, y_coord, kpt_conf = detection["keypoints"][index]
elements.append(
{
"label": sublabel_name,
"type": "points",
"outside": 0 if float(kpt_conf) >= conf_threshold else 1,
"points": [float(x_coord), float(y_coord)],
"confidence": str(float(kpt_conf)),
}
)
if all(element["outside"] for element in elements):
continue
results.append(
{
"confidence": str(detection["confidence"]),
"label": detection["label"],
"type": "skeleton",
"elements": elements,
}
)
return results
def init_context(context) -> None:
context.logger.info("Initializing OpenVINO YOLO pose handler")
if not CLASSES_PATH.is_file():
raise FileNotFoundError(f"Classes file not found: {CLASSES_PATH}")
class_names = load_classes(CLASSES_PATH)
model_xml = find_model_xml(MODELS_DIR)
context.logger.info(f"Loading OpenVINO model from {model_xml}")
core = Core()
compiled_model = core.compile_model(model_xml, "CPU")
input_layer = compiled_model.input(0)
input_shape = tuple(input_layer.shape)
if len(input_shape) == 4:
imgsz = int(input_shape[2])
else:
imgsz = DEFAULT_IMGSZ
context.user_data.compiled_model = compiled_model
context.user_data.class_names = class_names
context.user_data.num_keypoints = DEFAULT_NUM_KEYPOINTS
context.user_data.imgsz = imgsz
context.logger.info(
f"Ready: classes={class_names}, imgsz={imgsz}, keypoints={DEFAULT_NUM_KEYPOINTS}"
)
def handler(context, event):
try:
payload = event.body
if isinstance(payload, (bytes, bytearray)):
payload = json.loads(payload.decode("utf-8"))
if isinstance(payload, str):
payload = json.loads(payload)
image_b64 = payload["image"]
threshold = float(payload.get("threshold", DEFAULT_CONF_THRESHOLD))
image_bytes = base64.b64decode(image_b64)
image_rgb = np.array(Image.open(io.BytesIO(image_bytes)).convert("RGB"))
image_bgr = image_rgb[:, :, ::-1]
tensor, scale, pad = preprocess_image(image_bgr, context.user_data.imgsz)
outputs = context.user_data.compiled_model([tensor])
raw_output = next(iter(outputs.values()))
predictions = parse_pose_output(
np.array(raw_output),
num_classes=len(context.user_data.class_names),
num_keypoints=context.user_data.num_keypoints,
)
detections = decode_detections(
predictions=predictions,
class_names=context.user_data.class_names,
num_keypoints=context.user_data.num_keypoints,
input_shape=(context.user_data.imgsz, context.user_data.imgsz),
image_shape=image_bgr.shape,
scale=scale,
pad=pad,
conf_threshold=threshold,
iou_threshold=DEFAULT_IOU_THRESHOLD,
)
results = build_skeleton_results(
detections,
num_keypoints=context.user_data.num_keypoints,
conf_threshold=threshold,
)
context.logger.info(f"Returning {len(results)} skeleton detections")
return context.Response(
body=json.dumps(results),
headers={},
content_type="application/json",
status_code=200,
)
except Exception as exc:
context.logger.error(f"Pose handler failed: {exc}", exc_info=True)
raise
+335
View File
@@ -0,0 +1,335 @@
"""Nuclio handler for Ultralytics YOLO-seg models exported to OpenVINO (CVAT mask output)."""
from __future__ import annotations
import base64
import io
import json
import os
from pathlib import Path
from typing import Any
import cv2
import numpy as np
from openvino import Core
from PIL import Image
MODELS_DIR = Path(os.environ.get("MODELS_DIR", "/models"))
CLASSES_PATH = Path(os.environ.get("CLASSES_PATH", "/opt/nuclio/classes.json"))
DEFAULT_IMGSZ = int(os.environ.get("MODEL_IMGSZ", "320"))
DEFAULT_CONF_THRESHOLD = float(os.environ.get("CONF_THRESHOLD", "0.25"))
DEFAULT_MASK_THRESHOLD = float(os.environ.get("MASK_THRESHOLD", "0.0"))
BOX_CHANNELS = 6
def load_classes(path: Path) -> dict[int, str]:
with path.open(encoding="utf-8") as handle:
payload = json.load(handle)
if isinstance(payload, list):
return {int(item["id"]): str(item["name"]) for item in payload}
class_map = payload.get("class", payload)
return {int(class_id): str(name) for class_id, name in class_map.items()}
def find_model_xml(models_dir: Path) -> Path:
candidates = [
models_dir / "best.xml",
models_dir / "best_openvino_model" / "best.xml",
*sorted(models_dir.glob("*.xml")),
*sorted(models_dir.glob("**/*.xml")),
]
for candidate in candidates:
if candidate.is_file():
return candidate
raise FileNotFoundError(
f"No OpenVINO XML model found under {models_dir}. "
"Mount the exported openvino_model directory at /models."
)
def letterbox(
image: np.ndarray,
new_shape: tuple[int, int] = (320, 320),
color: tuple[int, int, int] = (114, 114, 114),
) -> tuple[np.ndarray, float, tuple[float, float]]:
height, width = image.shape[:2]
target_height, target_width = new_shape
scale = min(target_height / height, target_width / width)
new_unpad_width = int(round(width * scale))
new_unpad_height = int(round(height * scale))
resized = cv2.resize(image, (new_unpad_width, new_unpad_height), interpolation=cv2.INTER_LINEAR)
pad_width = target_width - new_unpad_width
pad_height = target_height - new_unpad_height
pad_left = pad_width / 2
pad_top = pad_height / 2
padded = cv2.copyMakeBorder(
resized,
int(round(pad_top - 0.1)),
int(round(pad_height - pad_top)),
int(round(pad_left - 0.1)),
int(round(pad_width - pad_left)),
cv2.BORDER_CONSTANT,
value=color,
)
return padded, scale, (pad_left, pad_top)
def preprocess_image(image_bgr: np.ndarray, imgsz: int) -> tuple[np.ndarray, float, tuple[float, float]]:
letterboxed, scale, pad = letterbox(image_bgr, new_shape=(imgsz, imgsz))
rgb = letterboxed[:, :, ::-1].transpose(2, 0, 1)
tensor = np.expand_dims(rgb, axis=0).astype(np.float32) / 255.0
return tensor, scale, pad
def scale_boxes(
boxes: np.ndarray,
image_shape: tuple[int, int],
scale: float,
pad: tuple[float, float],
) -> np.ndarray:
boxes = boxes.copy()
pad_x, pad_y = pad
boxes[:, [0, 2]] -= pad_x
boxes[:, [1, 3]] -= pad_y
boxes[:, :4] /= scale
boxes[:, [0, 2]] = boxes[:, [0, 2]].clip(0, image_shape[1])
boxes[:, [1, 3]] = boxes[:, [1, 3]].clip(0, image_shape[0])
return boxes
def to_cvat_mask(box: list[int], mask: np.ndarray) -> list[int]:
xtl, ytl, xbr, ybr = box
flattened = mask[ytl : ybr + 1, xtl : xbr + 1].astype(np.uint8).ravel().tolist()
flattened.extend([xtl, ytl, xbr, ybr])
return flattened
def split_seg_outputs(raw_outputs: dict[Any, np.ndarray]) -> tuple[np.ndarray, np.ndarray]:
detections = None
proto = None
for value in raw_outputs.values():
array = np.array(value)
if array.ndim == 3 and array.shape[-1] >= BOX_CHANNELS:
detections = array
elif array.ndim == 4:
proto = array
if detections is None or proto is None:
shapes = {str(key): np.array(value).shape for key, value in raw_outputs.items()}
raise ValueError(
"Expected YOLO-seg outputs: detections (1, N, 6+nm) and proto (1, nm, H, W). "
f"Got shapes: {shapes}. Re-export the segmentation model to OpenVINO."
)
return detections, proto
def decode_end2end_detections(
detections: np.ndarray,
class_names: dict[int, str],
conf_threshold: float,
) -> tuple[np.ndarray, np.ndarray, np.ndarray, np.ndarray]:
predictions = detections[0] if detections.ndim == 3 else detections
if predictions.ndim != 2 or predictions.shape[1] < BOX_CHANNELS:
raise ValueError(
f"Unexpected detections shape {predictions.shape}. "
"End-to-end YOLO-seg expects (N, 6+nm) with [x1,y1,x2,y2,conf,cls,...]."
)
boxes = predictions[:, :4]
confidences = predictions[:, 4]
class_ids = predictions[:, 5].astype(np.int32)
mask_coeffs = predictions[:, BOX_CHANNELS:]
keep = confidences >= conf_threshold
if not np.any(keep):
empty = np.empty((0, 4), dtype=np.float32)
return empty, empty, np.empty((0,), dtype=np.int32), np.empty((0,), dtype=np.float32)
boxes = boxes[keep]
confidences = confidences[keep]
class_ids = class_ids[keep]
mask_coeffs = mask_coeffs[keep]
known = np.array([class_id in class_names for class_id in class_ids], dtype=bool)
if not np.any(known):
empty = np.empty((0, 4), dtype=np.float32)
return empty, empty, np.empty((0,), dtype=np.int32), np.empty((0,), dtype=np.float32)
return boxes[known], mask_coeffs[known], class_ids[known], confidences[known]
def process_masks(
proto: np.ndarray,
mask_coeffs: np.ndarray,
boxes_letterbox: np.ndarray,
imgsz: int,
image_shape: tuple[int, int],
scale: float,
pad: tuple[float, float],
mask_threshold: float,
) -> list[np.ndarray]:
proto_maps = proto[0] if proto.ndim == 4 else proto
mask_dim, proto_h, proto_w = proto_maps.shape
if mask_coeffs.shape[1] != mask_dim:
raise ValueError(
f"Mask coefficient dim {mask_coeffs.shape[1]} does not match proto channels {mask_dim}."
)
flat_proto = proto_maps.reshape(mask_dim, -1)
masks = mask_coeffs @ flat_proto
masks = masks.reshape(-1, proto_h, proto_w)
pad_x, pad_y = pad
image_h, image_w = image_shape
binary_masks: list[np.ndarray] = []
for index, mask in enumerate(masks):
mask_letterbox = cv2.resize(mask, (imgsz, imgsz), interpolation=cv2.INTER_LINEAR)
x1, y1, x2, y2 = boxes_letterbox[index]
x1_i = max(0, int(np.floor(x1)))
y1_i = max(0, int(np.floor(y1)))
x2_i = min(imgsz, int(np.ceil(x2)))
y2_i = min(imgsz, int(np.ceil(y2)))
cropped = np.zeros_like(mask_letterbox)
cropped[y1_i:y2_i, x1_i:x2_i] = mask_letterbox[y1_i:y2_i, x1_i:x2_i]
top = int(round(pad_y - 0.1))
left = int(round(pad_x - 0.1))
bottom = imgsz - int(round(pad_y + 0.1))
right = imgsz - int(round(pad_x + 0.1))
unpadded = cropped[top:bottom, left:right]
if unpadded.size == 0:
binary_masks.append(np.zeros((image_h, image_w), dtype=np.uint8))
continue
resized = cv2.resize(unpadded, (image_w, image_h), interpolation=cv2.INTER_LINEAR)
binary_masks.append((resized > mask_threshold).astype(np.uint8))
return binary_masks
def build_mask_results(
boxes: np.ndarray,
masks: list[np.ndarray],
class_ids: np.ndarray,
confidences: np.ndarray,
class_names: dict[int, str],
) -> list[dict[str, Any]]:
results: list[dict[str, Any]] = []
image_h, image_w = masks[0].shape if masks else (0, 0)
for box, mask, class_id, confidence in zip(boxes, masks, class_ids, confidences):
if int(mask.sum()) == 0:
continue
xtl = max(0, int(np.floor(box[0])))
ytl = max(0, int(np.floor(box[1])))
xbr = min(image_w - 1, int(np.ceil(box[2])))
ybr = min(image_h - 1, int(np.ceil(box[3])))
if xbr <= xtl or ybr <= ytl:
ys, xs = np.where(mask > 0)
if len(xs) == 0:
continue
xtl, xbr = int(xs.min()), int(xs.max())
ytl, ybr = int(ys.min()), int(ys.max())
results.append(
{
"confidence": str(float(confidence)),
"label": class_names[int(class_id)],
"type": "mask",
"mask": to_cvat_mask([xtl, ytl, xbr, ybr], mask),
}
)
return results
def init_context(context) -> None:
context.logger.info("Initializing OpenVINO YOLO segmentation handler")
if not CLASSES_PATH.is_file():
raise FileNotFoundError(
f"Classes file not found: {CLASSES_PATH}. "
"Mount classes.json at /opt/nuclio/classes.json."
)
class_names = load_classes(CLASSES_PATH)
model_xml = find_model_xml(MODELS_DIR)
context.logger.info(f"Loading OpenVINO model from {model_xml}")
core = Core()
compiled_model = core.compile_model(model_xml, "CPU")
input_layer = compiled_model.input(0)
input_shape = tuple(input_layer.shape)
imgsz = int(input_shape[2]) if len(input_shape) == 4 else DEFAULT_IMGSZ
context.user_data.compiled_model = compiled_model
context.user_data.class_names = class_names
context.user_data.imgsz = imgsz
context.logger.info(f"Ready: classes={class_names}, imgsz={imgsz}")
def handler(context, event):
try:
payload = event.body
if isinstance(payload, (bytes, bytearray)):
payload = json.loads(payload.decode("utf-8"))
if isinstance(payload, str):
payload = json.loads(payload)
image_b64 = payload["image"]
threshold = float(payload.get("threshold", DEFAULT_CONF_THRESHOLD))
image_bytes = base64.b64decode(image_b64)
image_rgb = np.array(Image.open(io.BytesIO(image_bytes)).convert("RGB"))
image_bgr = image_rgb[:, :, ::-1]
image_hw = (image_bgr.shape[0], image_bgr.shape[1])
tensor, scale, pad = preprocess_image(image_bgr, context.user_data.imgsz)
outputs = context.user_data.compiled_model([tensor])
detections, proto = split_seg_outputs(outputs)
boxes_letterbox, mask_coeffs, class_ids, confidences = decode_end2end_detections(
detections,
class_names=context.user_data.class_names,
conf_threshold=threshold,
)
if len(confidences) == 0:
return context.Response(
body=json.dumps([]),
headers={},
content_type="application/json",
status_code=200,
)
masks = process_masks(
proto=proto,
mask_coeffs=mask_coeffs,
boxes_letterbox=boxes_letterbox,
imgsz=context.user_data.imgsz,
image_shape=image_hw,
scale=scale,
pad=pad,
mask_threshold=DEFAULT_MASK_THRESHOLD,
)
boxes = scale_boxes(boxes_letterbox, image_hw, scale, pad)
results = build_mask_results(
boxes=boxes,
masks=masks,
class_ids=class_ids,
confidences=confidences,
class_names=context.user_data.class_names,
)
context.logger.info(f"Returning {len(results)} mask detections")
return context.Response(
body=json.dumps(results),
headers={},
content_type="application/json",
status_code=200,
)
except Exception as exc:
context.logger.error(f"Segmentation handler failed: {exc}", exc_info=True)
raise
+251
View File
@@ -0,0 +1,251 @@
"""Nuclio handler for CVAT automatic pose annotation using OpenVINO IR model.
Loads a single-person pose estimation model (e.g. Human Pose Estimation,
HRNet, OpenPose) exported to OpenVINO IR (.xml / .bin). The handler
returns keypoint annotations in CVAT-compatible format.
Environment variables
---------------------
MODEL_XML : path to the .xml file (default: /models/pose_model.xml)
MODEL_BIN : path to the .bin file (default: /models/pose_model.bin)
CONF_THRESHOLD : minimum keypoint confidence (default: 0.3)
KEYPOINTS_FILE : path to JSON file with keypoint labels (default: keypoints.json)
SKELETON_FILE : path to JSON file with skeleton edges (default: skeleton.json)
OUTPUT_TYPE : "skeleton" or "points" – how to return shapes (default: "skeleton")
"""
import base64
import json
import os
from pathlib import Path
import cv2
import numpy as np
import openvino as ov
from openvino.preprocess import PrePostProcessor
from openvino.preprocess import ColorFormat
from openvino import Layout, Type
MODEL_XML = os.getenv("MODEL_XML", "/models/best.xml")
MODEL_BIN = os.getenv("MODEL_BIN", "/models/best.bin")
CONF_THRESHOLD = float(os.getenv("CONF_THRESHOLD", "0.3"))
OUTPUT_TYPE = os.getenv("OUTPUT_TYPE", "skeleton")
_COCO_KEYPOINTS = [
"nose", "left_eye", "right_eye", "left_ear", "right_ear",
"left_shoulder", "right_shoulder", "left_elbow", "right_elbow",
"left_wrist", "right_wrist", "left_hip", "right_hip",
"left_knee", "right_knee", "left_ankle", "right_ankle",
]
_COCO_SKELETON = [
[15, 13], [13, 11], [16, 14], [14, 12], [11, 12],
[5, 11], [6, 12], [5, 6], [5, 7], [6, 8],
[7, 9], [8, 10], [1, 2], [0, 1], [0, 2],
[1, 3], [2, 4], [3, 5], [4, 6],
]
def _load_json_list(env_var, fallback):
path = os.getenv(env_var, "")
if path and os.path.isfile(path):
with open(path, "r") as fh:
return json.load(fh)
return fallback
KEYPOINT_LABELS = _load_json_list("KEYPOINTS_FILE", _COCO_KEYPOINTS)
SKELETON_EDGES = _load_json_list("SKELETON_FILE", _COCO_SKELETON)
class PoseEstimator:
def __init__(self, xml_path=MODEL_XML, bin_path=MODEL_BIN, conf=CONF_THRESHOLD):
core = ov.Core()
if os.path.isfile(bin_path):
model = core.read_model(str(Path(xml_path)), str(Path(bin_path)))
else:
model = core.read_model(str(Path(xml_path)))
ppp = PrePostProcessor(model)
ppp.input().tensor().set_element_type(Type.u8).set_layout(
Layout("NHWC")
).set_color_format(ColorFormat.BGR)
ppp.input().preprocess().convert_element_type(Type.f32).convert_color(
ColorFormat.RGB
).scale([255.0, 255.0, 255.0])
ppp.input().model().set_layout(Layout("NCHW"))
ppp.output().tensor().set_element_type(Type.f32)
model = ppp.build()
self.compiled_model = core.compile_model(model, "CPU")
input_shape = self.compiled_model.input(0).shape
self.input_height = input_shape[2]
self.input_width = input_shape[3]
self.conf = conf
def resize_and_pad(self, image):
old_h, old_w = image.shape[:2]
ratio = min(self.input_width / old_w, self.input_height / old_h)
new_w = int(old_w * ratio)
new_h = int(old_h * ratio)
image = cv2.resize(image, (new_w, new_h))
delta_w = self.input_width - new_w
delta_h = self.input_height - new_h
padded = cv2.copyMakeBorder(
image, 0, delta_h, 0, delta_w, cv2.BORDER_CONSTANT, value=(100, 100, 100)
)
return padded, delta_w, delta_h
def predict(self, img, delta_w, delta_h, orig_shape):
input_tensor = np.expand_dims(img, 0)
infer_request = self.compiled_model.create_infer_request()
infer_request.infer({0: input_tensor})
output = infer_request.get_output_tensor().data
return self._postprocess(output, delta_w, delta_h, orig_shape)
def _postprocess(self, output, delta_w, delta_h, orig_shape):
"""Extract keypoints depending on output tensor shape.
- 4‑D [1, K, H, W] → heatmap‑based; argmax per channel.
- 3‑D [1, K, 3] / [1, K, 2] → direct coordinate regression.
"""
orig_h, orig_w = orig_shape[:2]
rx = orig_w / (self.input_width - delta_w)
ry = orig_h / (self.input_height - delta_h)
if output.ndim == 4:
return self._heatmap_keypoints(output, rx, ry)
return self._direct_keypoints(output, rx, ry)
def _heatmap_keypoints(self, heatmaps, rx, ry):
keypoints = []
for i in range(heatmaps.shape[1]):
hmap = heatmaps[0, i]
_, conf, _, loc = cv2.minMaxLoc(hmap)
label = KEYPOINT_LABELS[i] if i < len(KEYPOINT_LABELS) else f"kp_{i}"
keypoints.append({
"label": label,
"x": float(loc[0] * rx),
"y": float(loc[1] * ry),
"conf": float(conf),
})
return keypoints
def _direct_keypoints(self, output, rx, ry):
keypoints = []
has_conf = output.shape[2] >= 3
for i in range(output.shape[1]):
label = KEYPOINT_LABELS[i] if i < len(KEYPOINT_LABELS) else f"kp_{i}"
x = output[0, i, 0] * rx
y = output[0, i, 1] * ry
conf = float(output[0, i, 2]) if has_conf else 1.0
keypoints.append({
"label": label,
"x": float(x),
"y": float(y),
"conf": conf,
})
return keypoints
def _build_shapes_skeleton(keypoints):
"""Return a single skeleton shape containing all visible keypoints."""
filtered = [kp for kp in keypoints if kp["conf"] >= CONF_THRESHOLD]
if not filtered:
return []
elements = []
for kp in keypoints:
visible = kp["conf"] >= CONF_THRESHOLD
elements.append({
"label": kp["label"],
"type": "points",
"points": [kp["x"], kp["y"]],
"occluded": not visible,
"outside": not visible,
})
return [{
"label": "person",
"type": "skeleton",
"elements": elements,
}]
def _build_shapes_points(keypoints):
"""Return one ``points`` shape per visible keypoint."""
shapes = []
for kp in keypoints:
if kp["conf"] >= CONF_THRESHOLD:
shapes.append({
"label": kp["label"],
"type": "points",
"points": [kp["x"], kp["y"]],
"confidence": str(kp["conf"]),
})
return shapes
def init_context(context):
context.logger.info("Init context ----> 0%")
model = PoseEstimator(MODEL_XML, MODEL_BIN, conf=CONF_THRESHOLD)
context.user_data.model = model
context.logger.info(
"Init context ----> 100%% (input: %dx%d, labels: %d, output: %s)",
model.input_width,
model.input_height,
len(KEYPOINT_LABELS),
OUTPUT_TYPE,
)
def handler(context, event):
context.logger.info("Run OpenVINO Pose Estimation model")
try:
data = event.body
image_b64 = data["image"]
except Exception:
context.logger.error("Invalid request body – missing 'image' key")
return context.Response(
body=json.dumps({"error": "Invalid request body"}),
status_code=400,
content_type="application/json",
)
image_bytes = base64.b64decode(image_b64)
image = cv2.imdecode(np.frombuffer(image_bytes, np.uint8), cv2.IMREAD_COLOR)
if image is None:
context.logger.error("Failed to decode image")
return context.Response(
body=json.dumps({"error": "Failed to decode image"}),
status_code=400,
content_type="application/json",
)
model = context.user_data.model
img_resized, dw, dh = model.resize_and_pad(image)
keypoints = model.predict(img_resized, dw, dh, image.shape)
if OUTPUT_TYPE == "points":
shapes = _build_shapes_points(keypoints)
else:
shapes = _build_shapes_skeleton(keypoints)
visible = sum(1 for kp in keypoints if kp["conf"] >= CONF_THRESHOLD)
context.logger.info(
"Detected %d / %d keypoints", visible, len(keypoints)
)
return context.Response(
body=json.dumps(shapes),
headers={},
content_type="application/json",
status_code=200,
)