mirror of
https://github.com/blakeblackshear/frigate.git
synced 2026-10-08 15:52:48 +03:00
Implement annotated frames for GenAI Review (#24379)
CI / AMD64 Build (push) Canceled after 0s
CI / AMD64 Smoke Test (push) Canceled after 0s
CI / ARM Build (push) Canceled after 0s
CI / Jetson Jetpack 6 (push) Canceled after 0s
CI / AMD64 Extra Build (push) Canceled after 0s
CI / ARM Extra Build (push) Canceled after 0s
CI / Synaptics Build (push) Canceled after 0s
CI / Assemble and push default build (push) Canceled after 0s
CI / AMD64 Build (push) Canceled after 0s
CI / AMD64 Smoke Test (push) Canceled after 0s
CI / ARM Build (push) Canceled after 0s
CI / Jetson Jetpack 6 (push) Canceled after 0s
CI / AMD64 Extra Build (push) Canceled after 0s
CI / ARM Extra Build (push) Canceled after 0s
CI / Synaptics Build (push) Canceled after 0s
CI / Assemble and push default build (push) Canceled after 0s
* Implement annotated frames mode for GenAI reviews to improve models with lacking temporal understanding * Updates * Improve debug sharing * Do not number objects * Fix assumptions * Remove unhelpful content * Improve object data sent as part of prompt * Cleanup ollama dumbness * Bind db * Fixes * Cleanup
This commit is contained in:
@@ -105,8 +105,21 @@ class GenAIClient:
|
||||
debug_save: bool,
|
||||
activity_context_prompt: str,
|
||||
response_style: str = "default",
|
||||
frame_captions: list[str] | None = None,
|
||||
) -> ReviewMetadata | None:
|
||||
"""Generate a description for the review item activity."""
|
||||
"""Generate a description for the review item activity.
|
||||
|
||||
`frame_captions` holds one caption per thumbnail for the annotated
|
||||
frame mode; each is sent directly before its frame.
|
||||
"""
|
||||
if frame_captions and len(frame_captions) != len(thumbnails):
|
||||
logger.warning(
|
||||
"Got %d frame captions for %d thumbnails, sending plain frames",
|
||||
len(frame_captions),
|
||||
len(thumbnails),
|
||||
)
|
||||
frame_captions = None
|
||||
|
||||
context_prompt = build_review_description_prompt(
|
||||
review_data,
|
||||
thumbnails,
|
||||
@@ -114,6 +127,7 @@ class GenAIClient:
|
||||
preferred_language,
|
||||
activity_context_prompt,
|
||||
response_style,
|
||||
frame_captions,
|
||||
)
|
||||
|
||||
logger.debug(
|
||||
@@ -129,9 +143,30 @@ class GenAIClient:
|
||||
) as f:
|
||||
f.write(context_prompt)
|
||||
|
||||
if frame_captions:
|
||||
# One file per frame, numbered to match the image it precedes
|
||||
# (0.txt goes with 0.jpg), so the debug folder replays without
|
||||
# having to re-derive the mapping.
|
||||
for index, caption in enumerate(frame_captions):
|
||||
with open(
|
||||
os.path.join(
|
||||
CLIPS_DIR,
|
||||
"genai-requests",
|
||||
review_data["id"],
|
||||
f"{index}.txt",
|
||||
),
|
||||
"w",
|
||||
) as f:
|
||||
f.write(caption)
|
||||
|
||||
response_format = build_review_description_response_format(concerns)
|
||||
|
||||
response = self._send(context_prompt, thumbnails, response_format)
|
||||
response = self._send(
|
||||
context_prompt,
|
||||
thumbnails,
|
||||
response_format,
|
||||
image_captions=frame_captions,
|
||||
)
|
||||
|
||||
if debug_save and response:
|
||||
with open(
|
||||
@@ -269,6 +304,7 @@ class GenAIClient:
|
||||
images: list[bytes],
|
||||
response_format: dict | None = None,
|
||||
enable_thinking: bool = False,
|
||||
image_captions: list[str] | None = None,
|
||||
) -> str | None:
|
||||
"""Submit a request to the provider.
|
||||
|
||||
@@ -276,6 +312,10 @@ class GenAIClient:
|
||||
``supports_toggleable_thinking``. Description-style callers leave it
|
||||
at the default (off) since synthesis tasks don't benefit from
|
||||
reasoning traces.
|
||||
|
||||
``image_captions`` carries one caption per image, to be placed
|
||||
immediately before its image so the model can tell the frames apart.
|
||||
Providers build their request order with ``interleave_images``.
|
||||
"""
|
||||
return None
|
||||
|
||||
|
||||
@@ -13,6 +13,7 @@ from google.genai.types import FunctionCallingConfigMode
|
||||
|
||||
from frigate.config import GenAIProviderEnum
|
||||
from frigate.genai import GenAIClient, register_genai_provider
|
||||
from frigate.genai.utils import interleave_images
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -118,11 +119,16 @@ class GeminiClient(GenAIClient):
|
||||
images: list[bytes],
|
||||
response_format: dict | None = None,
|
||||
enable_thinking: bool = False,
|
||||
image_captions: list[str] | None = None,
|
||||
) -> str | None:
|
||||
"""Submit a request to Gemini."""
|
||||
contents = [prompt] + [
|
||||
types.Part.from_bytes(data=img, mime_type="image/jpeg") for img in images
|
||||
contents: list[Any] = [
|
||||
part
|
||||
if isinstance(part, str)
|
||||
else types.Part.from_bytes(data=part, mime_type="image/jpeg")
|
||||
for part in interleave_images(prompt, images, image_captions)
|
||||
]
|
||||
|
||||
try:
|
||||
# Merge runtime_options into generation_config if provided
|
||||
generation_config_dict: dict[str, Any] = {"candidate_count": 1}
|
||||
@@ -136,7 +142,7 @@ class GeminiClient(GenAIClient):
|
||||
|
||||
response = self.provider.models.generate_content(
|
||||
model=self.genai_config.model,
|
||||
contents=contents, # type: ignore[arg-type]
|
||||
contents=contents,
|
||||
config=types.GenerateContentConfig(
|
||||
**generation_config_dict,
|
||||
),
|
||||
|
||||
@@ -14,7 +14,7 @@ from PIL import Image
|
||||
|
||||
from frigate.config import GenAIProviderEnum
|
||||
from frigate.genai import GenAIClient, register_genai_provider
|
||||
from frigate.genai.utils import parse_tool_calls_from_message
|
||||
from frigate.genai.utils import interleave_images, parse_tool_calls_from_message
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -333,6 +333,7 @@ class LlamaCppClient(GenAIClient):
|
||||
images: list[bytes],
|
||||
response_format: dict | None = None,
|
||||
enable_thinking: bool = False,
|
||||
image_captions: list[str] | None = None,
|
||||
) -> str | None:
|
||||
"""Submit a request to llama.cpp server."""
|
||||
if self.provider is None:
|
||||
@@ -342,18 +343,17 @@ class LlamaCppClient(GenAIClient):
|
||||
return None
|
||||
|
||||
try:
|
||||
content = [
|
||||
{
|
||||
"type": "text",
|
||||
"text": prompt,
|
||||
}
|
||||
]
|
||||
for image in images:
|
||||
encoded_image = base64.b64encode(image).decode("utf-8")
|
||||
content: list[dict[str, Any]] = []
|
||||
for part in interleave_images(prompt, images, image_captions):
|
||||
if isinstance(part, str):
|
||||
content.append({"type": "text", "text": part})
|
||||
continue
|
||||
|
||||
encoded_image = base64.b64encode(part).decode("utf-8")
|
||||
content.append(
|
||||
{
|
||||
"type": "image_url",
|
||||
"image_url": { # type: ignore[dict-item]
|
||||
"image_url": {
|
||||
"url": f"data:image/jpeg;base64,{encoded_image}",
|
||||
},
|
||||
}
|
||||
|
||||
@@ -14,7 +14,7 @@ from ollama import ResponseError
|
||||
|
||||
from frigate.config import GenAIProviderEnum
|
||||
from frigate.genai import GenAIClient, register_genai_provider
|
||||
from frigate.genai.utils import parse_tool_calls_from_message
|
||||
from frigate.genai.utils import interleave_images, parse_tool_calls_from_message
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -50,6 +50,28 @@ def _extract_ollama_stats(response: Any) -> dict[str, Any] | None:
|
||||
return stats or None
|
||||
|
||||
|
||||
# Ollama replaces each occurrence of this marker in a message, in order, with
|
||||
# the next image from the message's images list. Without markers it puts every
|
||||
# image before the text.
|
||||
IMAGE_PLACEHOLDER = "[img]"
|
||||
|
||||
|
||||
def _flatten_parts(parts: list[str | bytes]) -> tuple[str, list[bytes] | None]:
|
||||
"""Collapse ordered text and image parts into Ollama's (content, images)
|
||||
shape, marking where each image goes so the order survives."""
|
||||
text: list[str] = []
|
||||
images: list[bytes] = []
|
||||
|
||||
for part in parts:
|
||||
if isinstance(part, bytes):
|
||||
text.append(IMAGE_PLACEHOLDER)
|
||||
images.append(part)
|
||||
elif part:
|
||||
text.append(part)
|
||||
|
||||
return "\n".join(text), (images or None)
|
||||
|
||||
|
||||
def _normalize_multimodal_content(
|
||||
content: Any,
|
||||
) -> tuple[str | None, list[bytes] | None]:
|
||||
@@ -58,13 +80,13 @@ def _normalize_multimodal_content(
|
||||
The chat API constructs user messages with content as a list of
|
||||
``{"type": "text"}`` and ``{"type": "image_url"}`` parts when a tool
|
||||
returns a live frame. Ollama's SDK requires content to be a string and
|
||||
images to be passed in a separate field, so we extract each.
|
||||
images to be passed in a separate field, so images are pulled out and
|
||||
their positions marked with placeholders.
|
||||
"""
|
||||
if not isinstance(content, list):
|
||||
return content, None
|
||||
|
||||
text_parts: list[str] = []
|
||||
images: list[bytes] = []
|
||||
parts: list[str | bytes] = []
|
||||
for part in content:
|
||||
if not isinstance(part, dict):
|
||||
continue
|
||||
@@ -72,17 +94,20 @@ def _normalize_multimodal_content(
|
||||
if part_type == "text":
|
||||
text = part.get("text")
|
||||
if text:
|
||||
text_parts.append(str(text))
|
||||
parts.append(str(text))
|
||||
elif part_type == "image_url":
|
||||
url = (part.get("image_url") or {}).get("url", "")
|
||||
if isinstance(url, str) and url.startswith("data:"):
|
||||
try:
|
||||
encoded = url.split(",", 1)[1]
|
||||
images.append(base64.b64decode(encoded, validate=True))
|
||||
parts.append(base64.b64decode(encoded, validate=True))
|
||||
except (ValueError, IndexError, binascii.Error) as e:
|
||||
logger.debug("Failed to decode multimodal image url: %s", e)
|
||||
|
||||
return ("\n".join(text_parts) if text_parts else None), (images or None)
|
||||
if not parts:
|
||||
return None, None
|
||||
|
||||
return _flatten_parts(parts)
|
||||
|
||||
|
||||
@register_genai_provider(GenAIProviderEnum.ollama)
|
||||
@@ -196,58 +221,46 @@ class OllamaClient(GenAIClient):
|
||||
images: list[bytes],
|
||||
response_format: dict | None = None,
|
||||
enable_thinking: bool = False,
|
||||
image_captions: list[str] | None = None,
|
||||
) -> str | None:
|
||||
"""Submit a request to Ollama"""
|
||||
"""Submit a request to Ollama through the chat API, the same path the
|
||||
tool-calling chat uses, with image placeholders keeping any captions
|
||||
next to their frames."""
|
||||
if self.provider is None:
|
||||
logger.warning(
|
||||
"Ollama provider has not been initialized, a description will not be generated. Check your Ollama configuration."
|
||||
)
|
||||
return None
|
||||
|
||||
content, message_images = _flatten_parts(
|
||||
interleave_images(prompt, images, image_captions)
|
||||
)
|
||||
message: dict[str, Any] = {"role": "user", "content": content}
|
||||
|
||||
if message_images:
|
||||
message["images"] = message_images
|
||||
|
||||
request_params = self._build_request_params(
|
||||
[message], None, None, enable_thinking=enable_thinking
|
||||
)
|
||||
|
||||
if response_format and response_format.get("type") == "json_schema":
|
||||
schema = response_format.get("json_schema", {}).get("schema")
|
||||
if schema:
|
||||
request_params["format"] = self._clean_schema_for_ollama(schema)
|
||||
|
||||
logger.debug(
|
||||
"Ollama chat request: model=%s, prompt_len=%s, image_count=%s, "
|
||||
"has_format=%s, think=%s",
|
||||
self.genai_config.model,
|
||||
len(prompt),
|
||||
len(images),
|
||||
"format" in request_params,
|
||||
request_params.get("think"),
|
||||
)
|
||||
|
||||
try:
|
||||
ollama_options = {
|
||||
**self.provider_options,
|
||||
**self.genai_config.runtime_options,
|
||||
}
|
||||
if response_format and response_format.get("type") == "json_schema":
|
||||
schema = response_format.get("json_schema", {}).get("schema")
|
||||
if schema:
|
||||
ollama_options["format"] = self._clean_schema_for_ollama(schema)
|
||||
if self.supports_toggleable_thinking:
|
||||
ollama_options["think"] = enable_thinking
|
||||
logger.debug(
|
||||
"Ollama generate request: model=%s, prompt_len=%s, image_count=%s, "
|
||||
"has_format=%s, options=%s",
|
||||
self.genai_config.model,
|
||||
len(prompt),
|
||||
len(images) if images else 0,
|
||||
"format" in ollama_options,
|
||||
{k: v for k, v in ollama_options.items() if k != "format"},
|
||||
)
|
||||
result = self.provider.generate(
|
||||
self.genai_config.model,
|
||||
prompt,
|
||||
images=images if images else None,
|
||||
**ollama_options,
|
||||
)
|
||||
logger.debug(
|
||||
"Ollama generate response: done=%s, done_reason=%s, eval_count=%s, "
|
||||
"prompt_eval_count=%s, response_len=%s",
|
||||
result.get("done"),
|
||||
result.get("done_reason"),
|
||||
result.get("eval_count"),
|
||||
result.get("prompt_eval_count"),
|
||||
len(result.get("response", "") or ""),
|
||||
)
|
||||
response_text = str(result["response"]).strip()
|
||||
if not response_text:
|
||||
logger.warning(
|
||||
"Ollama returned a blank response for model %s (done_reason=%s, "
|
||||
"eval_count=%s). Check model output, ensure thinking is disabled.",
|
||||
self.genai_config.model,
|
||||
result.get("done_reason"),
|
||||
result.get("eval_count"),
|
||||
)
|
||||
return response_text
|
||||
response = self.provider.chat(**request_params)
|
||||
except (
|
||||
TimeoutException,
|
||||
ResponseError,
|
||||
@@ -257,6 +270,27 @@ class OllamaClient(GenAIClient):
|
||||
logger.warning("Ollama returned an error: %s", str(e))
|
||||
return None
|
||||
|
||||
logger.debug(
|
||||
"Ollama chat response: done=%s, done_reason=%s, eval_count=%s, "
|
||||
"prompt_eval_count=%s",
|
||||
response.get("done"),
|
||||
response.get("done_reason"),
|
||||
response.get("eval_count"),
|
||||
response.get("prompt_eval_count"),
|
||||
)
|
||||
response_text = self._message_from_response(response)["content"] or ""
|
||||
|
||||
if not response_text:
|
||||
logger.warning(
|
||||
"Ollama returned a blank response for model %s (done_reason=%s, "
|
||||
"eval_count=%s). Check model output, ensure thinking is disabled.",
|
||||
self.genai_config.model,
|
||||
response.get("done_reason"),
|
||||
response.get("eval_count"),
|
||||
)
|
||||
|
||||
return response_text
|
||||
|
||||
def list_models(self) -> list[str]:
|
||||
"""Return available model names from the Ollama server."""
|
||||
client = self.provider
|
||||
@@ -306,6 +340,8 @@ class OllamaClient(GenAIClient):
|
||||
}
|
||||
if images:
|
||||
msg_dict["images"] = images
|
||||
elif msg.get("images"):
|
||||
msg_dict["images"] = msg["images"]
|
||||
if msg.get("tool_call_id"):
|
||||
msg_dict["tool_call_id"] = msg["tool_call_id"]
|
||||
if msg.get("name"):
|
||||
|
||||
@@ -11,6 +11,7 @@ from openai import OpenAI
|
||||
|
||||
from frigate.config import GenAIProviderEnum
|
||||
from frigate.genai import GenAIClient, register_genai_provider
|
||||
from frigate.genai.utils import interleave_images
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -63,21 +64,21 @@ class OpenAIClient(GenAIClient):
|
||||
images: list[bytes],
|
||||
response_format: dict | None = None,
|
||||
enable_thinking: bool = False,
|
||||
image_captions: list[str] | None = None,
|
||||
) -> str | None:
|
||||
"""Submit a request to OpenAI."""
|
||||
encoded_images = [base64.b64encode(image).decode("utf-8") for image in images]
|
||||
messages_content: list[dict] = [
|
||||
{
|
||||
"type": "text",
|
||||
"text": prompt,
|
||||
}
|
||||
]
|
||||
for image in encoded_images:
|
||||
messages_content: list[dict] = []
|
||||
for part in interleave_images(prompt, images, image_captions):
|
||||
if isinstance(part, str):
|
||||
messages_content.append({"type": "text", "text": part})
|
||||
continue
|
||||
|
||||
encoded = base64.b64encode(part).decode("utf-8")
|
||||
messages_content.append(
|
||||
{
|
||||
"type": "image_url",
|
||||
"image_url": {
|
||||
"url": f"data:image/jpeg;base64,{image}",
|
||||
"url": f"data:image/jpeg;base64,{encoded}",
|
||||
"detail": "low",
|
||||
},
|
||||
}
|
||||
|
||||
@@ -59,6 +59,13 @@ def get_review_field_guidelines(response_style: str = "default") -> dict[str, st
|
||||
}
|
||||
|
||||
|
||||
# Explains the per-frame labels and tracker notes used by the annotated frame
|
||||
# mode. Neither the notes nor this guidance say whether repeated detections are
|
||||
# the same subject, since the tracking data cannot tell.
|
||||
FRAME_ANNOTATION_GUIDANCE = """- Each image below is immediately preceded by a text label giving its frame number and how many seconds into the sequence it was captured. Use these labels to track the order of events and the time between them.
|
||||
- Some images below are preceded by notes from the camera's object tracker recording what changed at that point: an object being first detected, starting to move, reversing direction, stopping, or no longer being detected. These notes come from tracking data rather than from the images, and they are reliable. Use them to establish how many distinct activities occur and in what order, and describe every one of them."""
|
||||
|
||||
|
||||
def build_review_description_prompt(
|
||||
review_data: dict[str, Any],
|
||||
thumbnails: list[bytes],
|
||||
@@ -66,8 +73,13 @@ def build_review_description_prompt(
|
||||
preferred_language: str | None,
|
||||
activity_context_prompt: str,
|
||||
response_style: str = "default",
|
||||
frame_captions: list[str] | None = None,
|
||||
) -> str:
|
||||
"""Build the prompt for review activity description generation."""
|
||||
"""Build the prompt for review activity description generation.
|
||||
|
||||
When `frame_captions` is set, each caption is sent directly before its
|
||||
image, so the prompt explains that layout.
|
||||
"""
|
||||
|
||||
def get_concern_prompt() -> str:
|
||||
if concerns:
|
||||
@@ -93,6 +105,7 @@ def build_review_description_prompt(
|
||||
return "\n- (No objects detected)"
|
||||
|
||||
fields = get_review_field_guidelines(response_style)
|
||||
frame_guidance = f"\n{FRAME_ANNOTATION_GUIDANCE}" if frame_captions else ""
|
||||
|
||||
return f"""
|
||||
Your task is to analyze a sequence of images taken in chronological order from a security camera.
|
||||
@@ -130,7 +143,7 @@ Respond with a JSON object matching the provided schema. Field-specific guidance
|
||||
## Sequence Details
|
||||
|
||||
- Camera: {review_data["camera"]}
|
||||
- Total frames: {len(thumbnails)} (Frame 1 = earliest, Frame {len(thumbnails)} = latest)
|
||||
- Total frames: {len(thumbnails)} (Frame 1 = earliest, Frame {len(thumbnails)} = latest){frame_guidance}
|
||||
- Activity started at {review_data["start"]} and lasted {review_data["duration"]} seconds
|
||||
- Zones involved: {", ".join(review_data["zones"]) if review_data["zones"] else "None"}
|
||||
|
||||
|
||||
@@ -7,6 +7,25 @@ from typing import Any
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def interleave_images(
|
||||
prompt: str, images: list[bytes], captions: list[str] | None = None
|
||||
) -> list[str | bytes]:
|
||||
"""The prompt, then each image preceded by its caption when one is given.
|
||||
|
||||
Providers map the text and image parts onto their own request format, so
|
||||
every provider sends the same order.
|
||||
"""
|
||||
parts: list[str | bytes] = [prompt]
|
||||
|
||||
for index, image in enumerate(images):
|
||||
if captions and index < len(captions):
|
||||
parts.append(captions[index])
|
||||
|
||||
parts.append(image)
|
||||
|
||||
return parts
|
||||
|
||||
|
||||
def parse_tool_calls_from_message(
|
||||
message: dict[str, Any],
|
||||
) -> list[dict[str, Any]] | None:
|
||||
|
||||
Reference in New Issue
Block a user