from __future__ import annotations import httpx from app.config import Settings from backends.openai_compat import OpenAICompatEmbed, OpenAICompatLLM class GemmaLLM(OpenAICompatLLM): """Gemma 4 12B Unified on vLLM — vision extract. Runs on the GPU box, not the Lamp.""" def __init__(self, settings: Settings, *, client: httpx.Client | None = None) -> None: super().__init__( settings, name="gemma4-unified", accepts_images=settings.llm_accepts_images, extra_body={}, client=client, ) class GemmaEmbed(OpenAICompatEmbed): """Same Gemma 4 12B Unified server, /v1/embeddings (omni). Dim 3840.""" def __init__(self, settings: Settings, *, client: httpx.Client | None = None) -> None: super().__init__(settings, name="gemma4-omni-embed", client=client)