mirror of
https://github.com/blakeblackshear/frigate.git
synced 2026-10-12 01:32:48 +03:00
* bump backend and frontend dependencies
Bumps every outdated dependency that passed testing, with the code changes each one needs.
- peewee 4.5 and peewee_migrate 2.3: use SqliteDatabase in place of the removed SqliteExtDatabase, and fix types for peewee's bundled stubs
- mypy 2.4.0: remove unused type ignores and add the annotations mypy 2 needs
- pandas 3.0: convert the motion activity index with as_unit("s"), because pandas 3 infers second resolution for whole-second timestamps and the old // 10**9 returned 1
- transformers 5.19: load Jina tokenizers from the saved directory with trust_remote_code off and use CLIPImageProcessor, since v5 removed CLIPFeatureExtractor and remote code pulled in torch
- tensorflow-cpu 2.21: pin protobuf to 7.36 in the TensorRT amd64 image, because the old 3.20.3 pin was installed over TensorFlow and broke its import
- pydantic 2.13 and google-genai 2.29: regenerate the API spec
- ruff 0.16: set an explicit lint select, since 0.16 widened the default rules
- react-router-dom 7: set useTransitions={false} on BrowserRouter to keep v6 update timing, which useSearchEffect relies on
- typescript 6: drop baseUrl and add node types in tsconfig
- radix: update to the latest release and remove the slot patch, which upstream now includes
- apexcharts 7.8: guard the tooltip formatter against null values
- cryptography 50, joserfc 1.7, pywebpush 2.5, openai 3.26, aiohttp, uvicorn, onvif-zeep-async, scipy, framer-motion 14, react-day-picker 10 and other minor and patch bumps
* silence httpx2 request logging
* redownload an incomplete jina tokenizer
An interrupted tokenizer download left the Hugging Face cache directory without the saved tokenizer files, and since the download is skipped whenever that directory exists, every restart failed to load the tokenizer. The Jina constructors now remove a tokenizer directory that has no tokenizer_config.json so the download runs again.
229 lines
8.5 KiB
Python
229 lines
8.5 KiB
Python
"""JinaV1 Embeddings."""
|
|
|
|
import logging
|
|
import os
|
|
import shutil
|
|
import threading
|
|
|
|
from transformers import AutoTokenizer, CLIPImageProcessor
|
|
from transformers.utils.logging import disable_progress_bar
|
|
|
|
from frigate.comms.inter_process import InterProcessRequestor
|
|
from frigate.const import MODEL_CACHE_DIR, UPDATE_MODEL_STATE
|
|
from frigate.detectors.detection_runners import BaseModelRunner, get_optimized_runner
|
|
|
|
# importing this without pytorch or others causes a warning
|
|
# https://github.com/huggingface/transformers/issues/27214
|
|
# suppressed by setting env TRANSFORMERS_NO_ADVISORY_WARNINGS=1
|
|
from frigate.embeddings.types import EnrichmentModelTypeEnum
|
|
from frigate.types import ModelStatusTypesEnum
|
|
from frigate.util.downloader import ModelDownloader
|
|
|
|
from .base_embedding import BaseEmbedding
|
|
|
|
# disables the progress bar for downloading tokenizers and feature extractors
|
|
disable_progress_bar()
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
class JinaV1TextEmbedding(BaseEmbedding):
|
|
def __init__(
|
|
self,
|
|
model_size: str,
|
|
requestor: InterProcessRequestor,
|
|
device: str = "AUTO",
|
|
):
|
|
HF_ENDPOINT = os.environ.get("HF_ENDPOINT", "https://huggingface.co")
|
|
super().__init__(
|
|
model_name="jinaai/jina-clip-v1",
|
|
model_file="text_model_fp16.onnx",
|
|
download_urls={
|
|
"text_model_fp16.onnx": f"{HF_ENDPOINT}/jinaai/jina-clip-v1/resolve/main/onnx/text_model_fp16.onnx",
|
|
},
|
|
)
|
|
self.tokenizer_file = "tokenizer"
|
|
self.requestor = requestor
|
|
self.model_size = model_size
|
|
self.device = device
|
|
self.download_path = os.path.join(MODEL_CACHE_DIR, self.model_name)
|
|
self.tokenizer = None
|
|
self.feature_extractor = None
|
|
self.runner = None
|
|
self._lock = threading.Lock()
|
|
files_names = list(self.download_urls.keys()) + [self.tokenizer_file]
|
|
|
|
# an interrupted download leaves the hub cache without the saved tokenizer
|
|
tokenizer_path = os.path.join(self.download_path, self.tokenizer_file)
|
|
if os.path.isdir(tokenizer_path) and not os.path.exists(
|
|
os.path.join(tokenizer_path, "tokenizer_config.json")
|
|
):
|
|
shutil.rmtree(tokenizer_path)
|
|
|
|
if not all(
|
|
os.path.exists(os.path.join(self.download_path, n)) for n in files_names
|
|
):
|
|
logger.debug(f"starting model download for {self.model_name}")
|
|
self.downloader = ModelDownloader(
|
|
model_name=self.model_name,
|
|
download_path=self.download_path,
|
|
file_names=files_names,
|
|
download_func=self._download_model,
|
|
)
|
|
self.downloader.ensure_model_files()
|
|
else:
|
|
self.downloader = None
|
|
ModelDownloader.mark_files_state(
|
|
self.requestor,
|
|
self.model_name,
|
|
files_names,
|
|
ModelStatusTypesEnum.downloaded,
|
|
)
|
|
self._load_model_and_utils()
|
|
logger.debug(f"models are already downloaded for {self.model_name}")
|
|
|
|
def _download_model(self, path: str):
|
|
try:
|
|
file_name = os.path.basename(path)
|
|
|
|
if file_name in self.download_urls:
|
|
ModelDownloader.download_from_url(self.download_urls[file_name], path)
|
|
elif file_name == self.tokenizer_file:
|
|
if not os.path.exists(path + "/" + self.model_name):
|
|
logger.info(f"Downloading {self.model_name} tokenizer")
|
|
|
|
tokenizer = AutoTokenizer.from_pretrained(
|
|
self.model_name,
|
|
trust_remote_code=False,
|
|
cache_dir=f"{MODEL_CACHE_DIR}/{self.model_name}/tokenizer",
|
|
clean_up_tokenization_spaces=True,
|
|
)
|
|
tokenizer.save_pretrained(path)
|
|
|
|
self.downloader.requestor.send_data(
|
|
UPDATE_MODEL_STATE,
|
|
{
|
|
"model": f"{self.model_name}-{file_name}",
|
|
"state": ModelStatusTypesEnum.downloaded,
|
|
},
|
|
)
|
|
except Exception:
|
|
self.downloader.requestor.send_data(
|
|
UPDATE_MODEL_STATE,
|
|
{
|
|
"model": f"{self.model_name}-{file_name}",
|
|
"state": ModelStatusTypesEnum.error,
|
|
},
|
|
)
|
|
|
|
def _load_model_and_utils(self):
|
|
if self.runner is None:
|
|
if self.downloader:
|
|
self.downloader.wait_for_download()
|
|
|
|
tokenizer_path = os.path.join(
|
|
f"{MODEL_CACHE_DIR}/{self.model_name}/tokenizer"
|
|
)
|
|
self.tokenizer = AutoTokenizer.from_pretrained(
|
|
tokenizer_path,
|
|
trust_remote_code=False,
|
|
clean_up_tokenization_spaces=True,
|
|
)
|
|
|
|
self.runner = get_optimized_runner(
|
|
os.path.join(self.download_path, self.model_file),
|
|
self.device,
|
|
model_type=EnrichmentModelTypeEnum.jina_v1.value,
|
|
)
|
|
|
|
def _preprocess_inputs(self, raw_inputs):
|
|
with self._lock:
|
|
max_length = max(len(self.tokenizer.encode(text)) for text in raw_inputs)
|
|
return [
|
|
self.tokenizer(
|
|
text,
|
|
padding="max_length",
|
|
truncation=True,
|
|
max_length=max_length,
|
|
return_tensors="np",
|
|
)
|
|
for text in raw_inputs
|
|
]
|
|
|
|
|
|
class JinaV1ImageEmbedding(BaseEmbedding):
|
|
def __init__(
|
|
self,
|
|
model_size: str,
|
|
requestor: InterProcessRequestor,
|
|
device: str = "AUTO",
|
|
):
|
|
model_file = (
|
|
"vision_model_fp16.onnx"
|
|
if model_size == "large"
|
|
else "vision_model_quantized.onnx"
|
|
)
|
|
HF_ENDPOINT = os.environ.get("HF_ENDPOINT", "https://huggingface.co")
|
|
super().__init__(
|
|
model_name="jinaai/jina-clip-v1",
|
|
model_file=model_file,
|
|
download_urls={
|
|
model_file: f"{HF_ENDPOINT}/jinaai/jina-clip-v1/resolve/main/onnx/{model_file}",
|
|
"preprocessor_config.json": f"{HF_ENDPOINT}/jinaai/jina-clip-v1/resolve/main/preprocessor_config.json",
|
|
},
|
|
)
|
|
self.requestor = requestor
|
|
self.model_size = model_size
|
|
self.device = device
|
|
self.download_path = os.path.join(MODEL_CACHE_DIR, self.model_name)
|
|
self.feature_extractor = None
|
|
self.runner: BaseModelRunner | None = None
|
|
self._lock = threading.Lock()
|
|
files_names = list(self.download_urls.keys())
|
|
if not all(
|
|
os.path.exists(os.path.join(self.download_path, n)) for n in files_names
|
|
):
|
|
logger.debug(f"starting model download for {self.model_name}")
|
|
self.downloader = ModelDownloader(
|
|
model_name=self.model_name,
|
|
download_path=self.download_path,
|
|
file_names=files_names,
|
|
download_func=self._download_model,
|
|
)
|
|
self.downloader.ensure_model_files()
|
|
# Avoid lazy loading in worker threads: block until downloads complete
|
|
# and load the model on the main thread during initialization.
|
|
self._load_model_and_utils()
|
|
else:
|
|
self.downloader = None
|
|
ModelDownloader.mark_files_state(
|
|
self.requestor,
|
|
self.model_name,
|
|
files_names,
|
|
ModelStatusTypesEnum.downloaded,
|
|
)
|
|
self._load_model_and_utils()
|
|
logger.debug(f"models are already downloaded for {self.model_name}")
|
|
|
|
def _load_model_and_utils(self):
|
|
if self.runner is None:
|
|
if self.downloader:
|
|
self.downloader.wait_for_download()
|
|
|
|
self.feature_extractor = CLIPImageProcessor.from_pretrained(
|
|
f"{MODEL_CACHE_DIR}/{self.model_name}",
|
|
)
|
|
|
|
self.runner = get_optimized_runner(
|
|
os.path.join(self.download_path, self.model_file),
|
|
self.device,
|
|
model_type=EnrichmentModelTypeEnum.jina_v1.value,
|
|
)
|
|
|
|
def _preprocess_inputs(self, raw_inputs):
|
|
with self._lock:
|
|
processed_images = [self._process_image(img) for img in raw_inputs]
|
|
return [
|
|
self.feature_extractor(images=image, return_tensors="np")
|
|
for image in processed_images
|
|
]
|