restore fused DetectionOutput in the OpenVINO SSD model conversion

This commit is contained in:
Josh Hawkins
2026-07-29 07:03:04 -05:00
parent 1a8062ad34
commit 3a07751a37
+83 -19
View File
@@ -1,10 +1,14 @@
"""Convert the default SSDLite MobileNet v2 model to OpenVINO IR. """Convert the default SSDLite MobileNet v2 model to OpenVINO IR.
Replaces the legacy openvino-dev Model Optimizer conversion. The TensorFlow Replaces the legacy openvino-dev Model Optimizer conversion. The TensorFlow
frontend converts the Object Detection API frozen graph natively; the four TF frontend translates the Object Detection API pre and post processors literally,
outputs are then repacked into the single [1, 1, 100, 7] DetectionOutput-style producing per-class NonMaxSuppression, NonZero ops and map loops with data
tensor that Frigate's OpenVINO detector expects, and the input is flipped to dependent shapes that the GPU plugin handles very badly. Both are cut out the
BGR to match the legacy reverse_input_channels behavior. way ssd_v2_support.json used to do it: the preprocessor is an identity at the
native 300x300 input, and the postprocessor becomes a single fused
DetectionOutput. The result is the [1, 1, 100, 7] tensor that Frigate's
OpenVINO detector expects, with the input flipped to BGR to match the legacy
reverse_input_channels behavior.
""" """
import numpy as np import numpy as np
@@ -12,31 +16,91 @@ import openvino as ov
from openvino import opset8 as ops from openvino import opset8 as ops
from openvino.preprocess import PrePostProcessor from openvino.preprocess import PrePostProcessor
MODEL_DIR = "/models/ssdlite_mobilenet_v2_coco_2018_05_09"
OUTPUT_PATH = "/models/ssdlite_mobilenet_v2.xml"
INPUT_SHAPE = [1, 300, 300, 3]
# faster_rcnn_box_coder divides the deltas by pipeline.config's y/x/height/width
# scales of 10/10/5/5, which DetectionOutput expresses as per-prior variances.
BOX_VARIANCES = np.float32([0.1, 0.1, 0.2, 0.2])
model = ov.convert_model( model = ov.convert_model(
"/models/ssdlite_mobilenet_v2_coco_2018_05_09/frozen_inference_graph.pb", f"{MODEL_DIR}/frozen_inference_graph.pb",
input=[("image_tensor:0", [1, 300, 300, 3])], input=[("image_tensor:0", INPUT_SHAPE)],
) )
# rows of (image_id, class_id, score, xmin, ymin, xmax, ymax) nodes = {op.get_friendly_name(): op for op in model.get_ordered_ops()}
boxes = model.output("detection_boxes:0").get_node().input_value(0) parameter = model.get_parameters()[0]
classes = model.output("detection_classes:0").get_node().input_value(0)
scores = model.output("detection_scores:0").get_node().input_value(0)
# (ymin,xmin,ymax,xmax) -> (xmin,ymin,xmax,ymax) preprocessor = nodes["Preprocessor/map/TensorArrayStack/TensorArrayGatherV3"]
boxes = ops.gather(boxes, [1, 0, 3, 2], 2) box_deltas = nodes["Postprocessor/Reshape_1"].output(0)
classes = ops.unsqueeze(classes, 2) class_scores = nodes["Postprocessor/convert_scores"].output(0)
scores = ops.unsqueeze(scores, 2) anchors_output = nodes["Postprocessor/Reshape"].output(0)
image_id = ops.multiply(scores, np.float32(0.0))
detections = ops.concat([image_id, classes, scores, boxes], 2) # The anchors only depend on the static input shape, so fold them into a
detections = ops.unsqueeze(detections, 1) # constant and drop the generator subgraph with the rest of the postprocessor.
probe = ov.Core().compile_model(
ov.Model([anchors_output, preprocessor.output(0)], [parameter], "probe"), "CPU"
)
probe_input = np.random.default_rng(0).integers(0, 255, INPUT_SHAPE, dtype=np.uint8)
anchors, resized = (out.copy() for out in probe([probe_input]).values())
assert np.allclose(resized, probe_input, atol=1e-3), (
"preprocessor is not an identity at 300x300, it cannot be bypassed"
)
image = ops.convert(parameter, "f32")
for consumer in list(preprocessor.output(0).get_target_inputs()):
consumer.replace_source_output(image.output(0))
# (ymin, xmin, ymax, xmax) -> (xmin, ymin, xmax, ymax)
priors = anchors[:, [1, 0, 3, 2]].astype(np.float32).reshape(-1)
variances = np.tile(BOX_VARIANCES, len(anchors))
proposals = ops.constant(np.stack([priors, variances])[np.newaxis])
# (ty, tx, th, tw) -> (dx, dy, dw, dh) for the CENTER_SIZE decode
box_logits = ops.reshape(ops.gather(box_deltas, [1, 0, 3, 2], 1), [1, -1], False)
class_preds = ops.reshape(class_scores, [1, -1], False)
detections = ops.detection_output(
box_logits,
class_preds,
proposals,
{
"background_label_id": 0,
"top_k": 100,
"keep_top_k": [100],
"nms_threshold": 0.6,
"confidence_threshold": 0.3,
"code_type": "caffe.PriorBoxParameter.CENTER_SIZE",
"share_location": True,
"variance_encoded_in_target": False,
"normalized": True,
"clip_before_nms": False,
"clip_after_nms": True,
"decrease_label_id": False,
},
)
detections.output(0).get_tensor().set_names({"detection_out"}) detections.output(0).get_tensor().set_names({"detection_out"})
model = ov.Model([detections], model.get_parameters(), "ssdlite_mobilenet_v2") model = ov.Model([detections], [parameter], "ssdlite_mobilenet_v2")
ppp = PrePostProcessor(model) ppp = PrePostProcessor(model)
ppp.input().tensor().set_layout(ov.Layout("NHWC")) ppp.input().tensor().set_layout(ov.Layout("NHWC"))
ppp.input().preprocess().reverse_channels() ppp.input().preprocess().reverse_channels()
model = ppp.build() model = ppp.build()
ov.save_model(model, "/models/ssdlite_mobilenet_v2.xml", compress_to_fp16=True) # Fail the build rather than silently ship the dynamically shaped graph again.
op_types = [op.get_type_name() for op in model.get_ordered_ops()]
assert op_types.count("DetectionOutput") == 1, "postprocessor was not fused"
for dynamic_op in ("NonMaxSuppression", "NonZero", "Loop", "TensorIterator"):
assert dynamic_op not in op_types, f"{dynamic_op} left in the graph"
output_shape = model.outputs[0].get_partial_shape()
assert output_shape.is_static and list(output_shape) == [1, 1, 100, 7], (
f"unexpected detector output shape {output_shape}"
)
ov.save_model(model, OUTPUT_PATH, compress_to_fp16=True)