feat: completing text /chat-completion and /completion tests (#1223)

# What does this PR do? The goal is to have a fairly complete set of provider and e2e tests for /chat-completion and /completion. This is the current list, ``` grep -oE "def test_[a-zA-Z_+]*" llama_stack/providers/tests/inference/test_text_inference.py | cut -d' ' -f2 ``` - test_model_list - test_text_completion_non_streaming - test_text_completion_streaming - test_text_completion_logprobs_non_streaming - test_text_completion_logprobs_streaming - test_text_completion_structured_output - test_text_chat_completion_non_streaming - test_text_chat_completion_structured_output - test_text_chat_completion_streaming - test_text_chat_completion_with_tool_calling - test_text_chat_completion_with_tool_calling_streaming ``` grep -oE "def test_[a-zA-Z_+]*" tests/client-sdk/inference/test_text_inference.py | cut -d' ' -f2 ``` - test_text_completion_non_streaming - test_text_completion_streaming - test_text_completion_log_probs_non_streaming - test_text_completion_log_probs_streaming - test_text_completion_structured_output - test_text_chat_completion_non_streaming - test_text_chat_completion_streaming - test_text_chat_completion_with_tool_calling_and_non_streaming - test_text_chat_completion_with_tool_calling_and_streaming - test_text_chat_completion_with_tool_choice_required - test_text_chat_completion_with_tool_choice_none - test_text_chat_completion_structured_output - test_text_chat_completion_tool_calling_tools_not_in_request ## Test plan == Set up Ollama local server ``` OLLAMA_HOST=127.0.0.1:8321 with-proxy ollama serve OLLAMA_HOST=127.0.0.1:8321 ollama run llama3.2:3b-instruct-fp16 --keepalive 60m ``` == Run a provider test ``` conda activate stack OLLAMA_URL="http://localhost:8321" \ pytest -v -s -k "ollama" --inference-model="llama3.2:3b-instruct-fp16" \ llama_stack/providers/tests/inference/test_text_inference.py::TestInference ``` == Run an e2e test ``` conda activate sherpa with-proxy pip install llama-stack export INFERENCE_MODEL=llama3.2:3b-instruct-fp16 export LLAMA_STACK_PORT=8322 with-proxy llama stack build --template ollama with-proxy llama stack run --env OLLAMA_URL=http://localhost:8321 ollama ``` ``` conda activate stack LLAMA_STACK_PORT=8322 LLAMA_STACK_BASE_URL="http://localhost:8322" \ pytest -v -s --inference-model="llama3.2:3b-instruct-fp16" \ tests/client-sdk/inference/test_text_inference.py ```
2025-12-06 02:30:58 +00:00 · 2025-02-25 11:37:04 -08:00 · 2025-02-25 11:37:04 -08:00 · 3a31611486
commit 3a31611486
parent 9b130f96a7
8 changed files with 479 additions and 223 deletions
--- a/llama_stack/providers/tests/test_cases/test_case.py
+++ b/llama_stack/providers/tests/test_cases/test_case.py
@ -9,7 +9,10 @@ import pathlib


 class TestCase:
-    _apis = ["chat_completion", "completion"]
+    _apis = [
+        "inference/chat_completion",
+        "inference/completion",
+    ]
    _jsonblob = {}

    def __init__(self, name):
@ -17,7 +20,12 @@ class TestCase:
        if self._jsonblob == {}:
            for api in self._apis:
                with open(pathlib.Path(__file__).parent / f"{api}.json", "r") as f:
-                    TestCase._jsonblob.update({f"{api}-{k}": v for k, v in json.load(f).items()})
+                    coloned = api.replace("/", ":")
+                    try:
+                        loaded = json.load(f)
+                    except json.JSONDecodeError as e:
+                        raise ValueError(f"There is a syntax error in {api}.json: {e}") from e
+                    TestCase._jsonblob.update({f"{coloned}:{k}": v for k, v in loaded.items()})

        # loading this test case
        tc = self._jsonblob.get(name)
@ -25,7 +33,6 @@ class TestCase:
            raise ValueError(f"Test case {name} not found")

        # these are the only fields we need
-        self.name = tc.get("name")
        self.data = tc.get("data")

    def __getitem__(self, key):