Compare commits

...
2 Commits
Author SHA1 Message Date
Nicolas Mowen 45d3c6389e update docs 2026-07-19 17:25:16 -06:00
Nicolas Mowen 761a431c55 Use manual context size if set 2026-07-19 17:22:33 -06:00
3 changed files with 33 additions and 3 deletions
+4 -2
View File
@@ -78,7 +78,7 @@ All llama.cpp native options can be passed through `provider_options`, including
- Set **Provider** to `llamacpp`
- Set **Base URL** to your llama.cpp server address (e.g., `http://localhost:8080`)
- Set **Model** to the name of your model
- Under **Provider Options**, set `context_size` to tell Frigate your context size so it can send the appropriate amount of information
- Optionally, under **Provider Options**, set `context_size` to override the context size Frigate detects from the server
</TabItem>
<TabItem value="yaml">
@@ -89,12 +89,14 @@ genai:
base_url: http://localhost:8080
model: your-model-name
provider_options:
context_size: 16000 # Tell Frigate your context size so it can send the appropriate amount of information.
context_size: 16000 # Optional, overrides the context size reported by the server.
```
</TabItem>
</ConfigTabs>
Frigate queries the llama.cpp server for the model's context size at startup and logs it along with the other detected capabilities. If `context_size` is set in `provider_options`, that value is always used instead, even when the server reports its own.
### Ollama
[Ollama](https://ollama.com/) allows you to self-host large language models and keep everything running locally. It is highly recommended to host this server on a machine with an Nvidia graphics card, or on a Apple silicon Mac for best performance.
+1 -1
View File
@@ -192,7 +192,7 @@ class LlamaCppClient(GenAIClient):
logger.info(
"llama.cpp model '%s' initialized — context: %s, vision: %s, audio: %s, tools: %s, reasoning: %s",
configured_model,
self._context_size or "unknown",
self.get_context_size(),
self._supports_vision,
self._supports_audio,
self._supports_tools,
+28
View File
@@ -491,6 +491,34 @@ class TestLlamaCppProvider(unittest.TestCase):
final = _final_message(self._run_with_lines(client, lines, MULTIMODAL_MESSAGES))
self.assertEqual(final["content"], "ok")
def _validated_client(self, server_context_size, provider_options=None):
"""Build a client as if the server reported the given context size."""
cfg = GenAIConfig(
provider="llamacpp",
model="m",
base_url="http://localhost:9999",
provider_options=provider_options or {},
)
info = {
"context_size": server_context_size,
"supports_vision": False,
"supports_audio": False,
"supports_tools": False,
"supports_reasoning": False,
"media_marker": "<__media__>",
}
cls = PROVIDERS[GenAIProviderEnum.llamacpp]
with patch.object(cls, "_get_model_info", return_value=info):
return cls(cfg, timeout=5)
def test_server_context_size_used_without_override(self):
client = self._validated_client(4096)
self.assertEqual(client.get_context_size(), 4096)
def test_provider_options_context_size_overrides_server(self):
client = self._validated_client(4096, {"context_size": 32768})
self.assertEqual(client.get_context_size(), 32768)
if __name__ == "__main__":
unittest.main()