chore: refactor (chat)completions endpoints to use shared params struct (#3761)

# What does this PR do? Converts openai(_chat)_completions params to pydantic BaseModel to reduce code duplication across all providers. ## Test Plan CI --- [//]: # (BEGIN SAPLING FOOTER) Stack created with [Sapling](https://sapling-scm.com). Best reviewed with [ReviewStack](https://reviewstack.dev/llamastack/llama-stack/pull/3761). * #3777 * __->__ #3761
2025-12-04 02:03:44 +00:00 · 2025-10-10 15:46:34 -07:00 · 2025-10-10 15:46:34 -07:00 · 80d58ab519
commit 80d58ab519
parent 6954fe2274
33 changed files with 599 additions and 890 deletions
--- a/llama_stack/providers/inline/agents/meta_reference/responses/streaming.py
+++ b/llama_stack/providers/inline/agents/meta_reference/responses/streaming.py
@ -49,6 +49,7 @@ from llama_stack.apis.inference import (
    OpenAIAssistantMessageParam,
    OpenAIChatCompletion,
    OpenAIChatCompletionChunk,
+    OpenAIChatCompletionRequest,
    OpenAIChatCompletionToolCall,
    OpenAIChoice,
    OpenAIMessageParam,
@ -168,7 +169,7 @@ class StreamingResponseOrchestrator:
                # (some providers don't support non-empty response_format when tools are present)
                response_format = None if self.ctx.response_format.type == "text" else self.ctx.response_format
                logger.debug(f"calling openai_chat_completion with tools: {self.ctx.chat_tools}")
-                completion_result = await self.inference_api.openai_chat_completion(
+                params = OpenAIChatCompletionRequest(
                    model=self.ctx.model,
                    messages=messages,
                    tools=self.ctx.chat_tools,
@ -179,6 +180,7 @@ class StreamingResponseOrchestrator:
                        "include_usage": True,
                    },
                )
+                completion_result = await self.inference_api.openai_chat_completion(params)

                # Process streaming chunks and build complete response
                completion_result_data = None