Implement annotated frames for GenAI Review (#24379)
CI / AMD64 Build (push) Canceled after 0s
CI / AMD64 Smoke Test (push) Canceled after 0s
CI / ARM Build (push) Canceled after 0s
CI / Jetson Jetpack 6 (push) Canceled after 0s
CI / AMD64 Extra Build (push) Canceled after 0s
CI / ARM Extra Build (push) Canceled after 0s
CI / Synaptics Build (push) Canceled after 0s
CI / Assemble and push default build (push) Canceled after 0s

* Implement annotated frames mode for GenAI reviews to improve models with lacking temporal understanding

* Updates

* Improve debug sharing

* Do not number objects

* Fix assumptions

* Remove unhelpful content

* Improve object data sent as part of prompt

* Cleanup ollama dumbness

* Bind db

* Fixes

* Cleanup
This commit is contained in:
Nicolas Mowen
2026-09-17 10:28:37 -06:00
committed by GitHub
parent 10a0d5ea37
commit eccd10cd94
19 changed files with 1128 additions and 118 deletions
+42 -2
View File
@@ -105,8 +105,21 @@ class GenAIClient:
debug_save: bool,
activity_context_prompt: str,
response_style: str = "default",
frame_captions: list[str] | None = None,
) -> ReviewMetadata | None:
"""Generate a description for the review item activity."""
"""Generate a description for the review item activity.
`frame_captions` holds one caption per thumbnail for the annotated
frame mode; each is sent directly before its frame.
"""
if frame_captions and len(frame_captions) != len(thumbnails):
logger.warning(
"Got %d frame captions for %d thumbnails, sending plain frames",
len(frame_captions),
len(thumbnails),
)
frame_captions = None
context_prompt = build_review_description_prompt(
review_data,
thumbnails,
@@ -114,6 +127,7 @@ class GenAIClient:
preferred_language,
activity_context_prompt,
response_style,
frame_captions,
)
logger.debug(
@@ -129,9 +143,30 @@ class GenAIClient:
) as f:
f.write(context_prompt)
if frame_captions:
# One file per frame, numbered to match the image it precedes
# (0.txt goes with 0.jpg), so the debug folder replays without
# having to re-derive the mapping.
for index, caption in enumerate(frame_captions):
with open(
os.path.join(
CLIPS_DIR,
"genai-requests",
review_data["id"],
f"{index}.txt",
),
"w",
) as f:
f.write(caption)
response_format = build_review_description_response_format(concerns)
response = self._send(context_prompt, thumbnails, response_format)
response = self._send(
context_prompt,
thumbnails,
response_format,
image_captions=frame_captions,
)
if debug_save and response:
with open(
@@ -269,6 +304,7 @@ class GenAIClient:
images: list[bytes],
response_format: dict | None = None,
enable_thinking: bool = False,
image_captions: list[str] | None = None,
) -> str | None:
"""Submit a request to the provider.
@@ -276,6 +312,10 @@ class GenAIClient:
``supports_toggleable_thinking``. Description-style callers leave it
at the default (off) since synthesis tasks don't benefit from
reasoning traces.
``image_captions`` carries one caption per image, to be placed
immediately before its image so the model can tell the frames apart.
Providers build their request order with ``interleave_images``.
"""
return None
+9 -3
View File
@@ -13,6 +13,7 @@ from google.genai.types import FunctionCallingConfigMode
from frigate.config import GenAIProviderEnum
from frigate.genai import GenAIClient, register_genai_provider
from frigate.genai.utils import interleave_images
logger = logging.getLogger(__name__)
@@ -118,11 +119,16 @@ class GeminiClient(GenAIClient):
images: list[bytes],
response_format: dict | None = None,
enable_thinking: bool = False,
image_captions: list[str] | None = None,
) -> str | None:
"""Submit a request to Gemini."""
contents = [prompt] + [
types.Part.from_bytes(data=img, mime_type="image/jpeg") for img in images
contents: list[Any] = [
part
if isinstance(part, str)
else types.Part.from_bytes(data=part, mime_type="image/jpeg")
for part in interleave_images(prompt, images, image_captions)
]
try:
# Merge runtime_options into generation_config if provided
generation_config_dict: dict[str, Any] = {"candidate_count": 1}
@@ -136,7 +142,7 @@ class GeminiClient(GenAIClient):
response = self.provider.models.generate_content(
model=self.genai_config.model,
contents=contents, # type: ignore[arg-type]
contents=contents,
config=types.GenerateContentConfig(
**generation_config_dict,
),
+10 -10
View File
@@ -14,7 +14,7 @@ from PIL import Image
from frigate.config import GenAIProviderEnum
from frigate.genai import GenAIClient, register_genai_provider
from frigate.genai.utils import parse_tool_calls_from_message
from frigate.genai.utils import interleave_images, parse_tool_calls_from_message
logger = logging.getLogger(__name__)
@@ -333,6 +333,7 @@ class LlamaCppClient(GenAIClient):
images: list[bytes],
response_format: dict | None = None,
enable_thinking: bool = False,
image_captions: list[str] | None = None,
) -> str | None:
"""Submit a request to llama.cpp server."""
if self.provider is None:
@@ -342,18 +343,17 @@ class LlamaCppClient(GenAIClient):
return None
try:
content = [
{
"type": "text",
"text": prompt,
}
]
for image in images:
encoded_image = base64.b64encode(image).decode("utf-8")
content: list[dict[str, Any]] = []
for part in interleave_images(prompt, images, image_captions):
if isinstance(part, str):
content.append({"type": "text", "text": part})
continue
encoded_image = base64.b64encode(part).decode("utf-8")
content.append(
{
"type": "image_url",
"image_url": { # type: ignore[dict-item]
"image_url": {
"url": f"data:image/jpeg;base64,{encoded_image}",
},
}
+88 -52
View File
@@ -14,7 +14,7 @@ from ollama import ResponseError
from frigate.config import GenAIProviderEnum
from frigate.genai import GenAIClient, register_genai_provider
from frigate.genai.utils import parse_tool_calls_from_message
from frigate.genai.utils import interleave_images, parse_tool_calls_from_message
logger = logging.getLogger(__name__)
@@ -50,6 +50,28 @@ def _extract_ollama_stats(response: Any) -> dict[str, Any] | None:
return stats or None
# Ollama replaces each occurrence of this marker in a message, in order, with
# the next image from the message's images list. Without markers it puts every
# image before the text.
IMAGE_PLACEHOLDER = "[img]"
def _flatten_parts(parts: list[str | bytes]) -> tuple[str, list[bytes] | None]:
"""Collapse ordered text and image parts into Ollama's (content, images)
shape, marking where each image goes so the order survives."""
text: list[str] = []
images: list[bytes] = []
for part in parts:
if isinstance(part, bytes):
text.append(IMAGE_PLACEHOLDER)
images.append(part)
elif part:
text.append(part)
return "\n".join(text), (images or None)
def _normalize_multimodal_content(
content: Any,
) -> tuple[str | None, list[bytes] | None]:
@@ -58,13 +80,13 @@ def _normalize_multimodal_content(
The chat API constructs user messages with content as a list of
``{"type": "text"}`` and ``{"type": "image_url"}`` parts when a tool
returns a live frame. Ollama's SDK requires content to be a string and
images to be passed in a separate field, so we extract each.
images to be passed in a separate field, so images are pulled out and
their positions marked with placeholders.
"""
if not isinstance(content, list):
return content, None
text_parts: list[str] = []
images: list[bytes] = []
parts: list[str | bytes] = []
for part in content:
if not isinstance(part, dict):
continue
@@ -72,17 +94,20 @@ def _normalize_multimodal_content(
if part_type == "text":
text = part.get("text")
if text:
text_parts.append(str(text))
parts.append(str(text))
elif part_type == "image_url":
url = (part.get("image_url") or {}).get("url", "")
if isinstance(url, str) and url.startswith("data:"):
try:
encoded = url.split(",", 1)[1]
images.append(base64.b64decode(encoded, validate=True))
parts.append(base64.b64decode(encoded, validate=True))
except (ValueError, IndexError, binascii.Error) as e:
logger.debug("Failed to decode multimodal image url: %s", e)
return ("\n".join(text_parts) if text_parts else None), (images or None)
if not parts:
return None, None
return _flatten_parts(parts)
@register_genai_provider(GenAIProviderEnum.ollama)
@@ -196,58 +221,46 @@ class OllamaClient(GenAIClient):
images: list[bytes],
response_format: dict | None = None,
enable_thinking: bool = False,
image_captions: list[str] | None = None,
) -> str | None:
"""Submit a request to Ollama"""
"""Submit a request to Ollama through the chat API, the same path the
tool-calling chat uses, with image placeholders keeping any captions
next to their frames."""
if self.provider is None:
logger.warning(
"Ollama provider has not been initialized, a description will not be generated. Check your Ollama configuration."
)
return None
content, message_images = _flatten_parts(
interleave_images(prompt, images, image_captions)
)
message: dict[str, Any] = {"role": "user", "content": content}
if message_images:
message["images"] = message_images
request_params = self._build_request_params(
[message], None, None, enable_thinking=enable_thinking
)
if response_format and response_format.get("type") == "json_schema":
schema = response_format.get("json_schema", {}).get("schema")
if schema:
request_params["format"] = self._clean_schema_for_ollama(schema)
logger.debug(
"Ollama chat request: model=%s, prompt_len=%s, image_count=%s, "
"has_format=%s, think=%s",
self.genai_config.model,
len(prompt),
len(images),
"format" in request_params,
request_params.get("think"),
)
try:
ollama_options = {
**self.provider_options,
**self.genai_config.runtime_options,
}
if response_format and response_format.get("type") == "json_schema":
schema = response_format.get("json_schema", {}).get("schema")
if schema:
ollama_options["format"] = self._clean_schema_for_ollama(schema)
if self.supports_toggleable_thinking:
ollama_options["think"] = enable_thinking
logger.debug(
"Ollama generate request: model=%s, prompt_len=%s, image_count=%s, "
"has_format=%s, options=%s",
self.genai_config.model,
len(prompt),
len(images) if images else 0,
"format" in ollama_options,
{k: v for k, v in ollama_options.items() if k != "format"},
)
result = self.provider.generate(
self.genai_config.model,
prompt,
images=images if images else None,
**ollama_options,
)
logger.debug(
"Ollama generate response: done=%s, done_reason=%s, eval_count=%s, "
"prompt_eval_count=%s, response_len=%s",
result.get("done"),
result.get("done_reason"),
result.get("eval_count"),
result.get("prompt_eval_count"),
len(result.get("response", "") or ""),
)
response_text = str(result["response"]).strip()
if not response_text:
logger.warning(
"Ollama returned a blank response for model %s (done_reason=%s, "
"eval_count=%s). Check model output, ensure thinking is disabled.",
self.genai_config.model,
result.get("done_reason"),
result.get("eval_count"),
)
return response_text
response = self.provider.chat(**request_params)
except (
TimeoutException,
ResponseError,
@@ -257,6 +270,27 @@ class OllamaClient(GenAIClient):
logger.warning("Ollama returned an error: %s", str(e))
return None
logger.debug(
"Ollama chat response: done=%s, done_reason=%s, eval_count=%s, "
"prompt_eval_count=%s",
response.get("done"),
response.get("done_reason"),
response.get("eval_count"),
response.get("prompt_eval_count"),
)
response_text = self._message_from_response(response)["content"] or ""
if not response_text:
logger.warning(
"Ollama returned a blank response for model %s (done_reason=%s, "
"eval_count=%s). Check model output, ensure thinking is disabled.",
self.genai_config.model,
response.get("done_reason"),
response.get("eval_count"),
)
return response_text
def list_models(self) -> list[str]:
"""Return available model names from the Ollama server."""
client = self.provider
@@ -306,6 +340,8 @@ class OllamaClient(GenAIClient):
}
if images:
msg_dict["images"] = images
elif msg.get("images"):
msg_dict["images"] = msg["images"]
if msg.get("tool_call_id"):
msg_dict["tool_call_id"] = msg["tool_call_id"]
if msg.get("name"):
+10 -9
View File
@@ -11,6 +11,7 @@ from openai import OpenAI
from frigate.config import GenAIProviderEnum
from frigate.genai import GenAIClient, register_genai_provider
from frigate.genai.utils import interleave_images
logger = logging.getLogger(__name__)
@@ -63,21 +64,21 @@ class OpenAIClient(GenAIClient):
images: list[bytes],
response_format: dict | None = None,
enable_thinking: bool = False,
image_captions: list[str] | None = None,
) -> str | None:
"""Submit a request to OpenAI."""
encoded_images = [base64.b64encode(image).decode("utf-8") for image in images]
messages_content: list[dict] = [
{
"type": "text",
"text": prompt,
}
]
for image in encoded_images:
messages_content: list[dict] = []
for part in interleave_images(prompt, images, image_captions):
if isinstance(part, str):
messages_content.append({"type": "text", "text": part})
continue
encoded = base64.b64encode(part).decode("utf-8")
messages_content.append(
{
"type": "image_url",
"image_url": {
"url": f"data:image/jpeg;base64,{image}",
"url": f"data:image/jpeg;base64,{encoded}",
"detail": "low",
},
}
+15 -2
View File
@@ -59,6 +59,13 @@ def get_review_field_guidelines(response_style: str = "default") -> dict[str, st
}
# Explains the per-frame labels and tracker notes used by the annotated frame
# mode. Neither the notes nor this guidance say whether repeated detections are
# the same subject, since the tracking data cannot tell.
FRAME_ANNOTATION_GUIDANCE = """- Each image below is immediately preceded by a text label giving its frame number and how many seconds into the sequence it was captured. Use these labels to track the order of events and the time between them.
- Some images below are preceded by notes from the camera's object tracker recording what changed at that point: an object being first detected, starting to move, reversing direction, stopping, or no longer being detected. These notes come from tracking data rather than from the images, and they are reliable. Use them to establish how many distinct activities occur and in what order, and describe every one of them."""
def build_review_description_prompt(
review_data: dict[str, Any],
thumbnails: list[bytes],
@@ -66,8 +73,13 @@ def build_review_description_prompt(
preferred_language: str | None,
activity_context_prompt: str,
response_style: str = "default",
frame_captions: list[str] | None = None,
) -> str:
"""Build the prompt for review activity description generation."""
"""Build the prompt for review activity description generation.
When `frame_captions` is set, each caption is sent directly before its
image, so the prompt explains that layout.
"""
def get_concern_prompt() -> str:
if concerns:
@@ -93,6 +105,7 @@ def build_review_description_prompt(
return "\n- (No objects detected)"
fields = get_review_field_guidelines(response_style)
frame_guidance = f"\n{FRAME_ANNOTATION_GUIDANCE}" if frame_captions else ""
return f"""
Your task is to analyze a sequence of images taken in chronological order from a security camera.
@@ -130,7 +143,7 @@ Respond with a JSON object matching the provided schema. Field-specific guidance
## Sequence Details
- Camera: {review_data["camera"]}
- Total frames: {len(thumbnails)} (Frame 1 = earliest, Frame {len(thumbnails)} = latest)
- Total frames: {len(thumbnails)} (Frame 1 = earliest, Frame {len(thumbnails)} = latest){frame_guidance}
- Activity started at {review_data["start"]} and lasted {review_data["duration"]} seconds
- Zones involved: {", ".join(review_data["zones"]) if review_data["zones"] else "None"}
+19
View File
@@ -7,6 +7,25 @@ from typing import Any
logger = logging.getLogger(__name__)
def interleave_images(
prompt: str, images: list[bytes], captions: list[str] | None = None
) -> list[str | bytes]:
"""The prompt, then each image preceded by its caption when one is given.
Providers map the text and image parts onto their own request format, so
every provider sends the same order.
"""
parts: list[str | bytes] = [prompt]
for index, image in enumerate(images):
if captions and index < len(captions):
parts.append(captions[index])
parts.append(image)
return parts
def parse_tool_calls_from_message(
message: dict[str, Any],
) -> list[dict[str, Any]] | None: