feat: add refresh_models support to inference adapters (default: false) (#3719)

# What does this PR do? inference adapters can now configure `refresh_models: bool` to control periodic model listing from their providers BREAKING CHANGE: together inference adapter default changed. previously always refreshed, now follows config. addresses "models: refresh" on #3517 ## Test Plan ci w/ new tests
2025-12-04 02:03:44 +00:00 · 2025-10-07 09:19:56 -04:00 · 2025-10-07 09:19:56 -04:00 · e892a3f7f4
commit e892a3f7f4
parent 8b9af03a1b
31 changed files with 33 additions and 67 deletions
--- a/llama_stack/providers/remote/inference/together/together.py
+++ b/llama_stack/providers/remote/inference/together/together.py
@ -63,9 +63,6 @@ class TogetherInferenceAdapter(OpenAIMixin, NeedsRequestProviderData):
        # Together's /v1/models is not compatible with OpenAI's /v1/models. Together support ticket #13355 -> will not fix, use Together's own client
        return [m.id for m in await self._get_client().models.list()]

-    async def should_refresh_models(self) -> bool:
-        return True
-
    async def openai_embeddings(
        self,
        model: str,