feat: completing text /chat-completion and /completion tests (#1223)

# What does this PR do? The goal is to have a fairly complete set of provider and e2e tests for /chat-completion and /completion. This is the current list, ``` grep -oE "def test_[a-zA-Z_+]*" llama_stack/providers/tests/inference/test_text_inference.py | cut -d' ' -f2 ``` - test_model_list - test_text_completion_non_streaming - test_text_completion_streaming - test_text_completion_logprobs_non_streaming - test_text_completion_logprobs_streaming - test_text_completion_structured_output - test_text_chat_completion_non_streaming - test_text_chat_completion_structured_output - test_text_chat_completion_streaming - test_text_chat_completion_with_tool_calling - test_text_chat_completion_with_tool_calling_streaming ``` grep -oE "def test_[a-zA-Z_+]*" tests/client-sdk/inference/test_text_inference.py | cut -d' ' -f2 ``` - test_text_completion_non_streaming - test_text_completion_streaming - test_text_completion_log_probs_non_streaming - test_text_completion_log_probs_streaming - test_text_completion_structured_output - test_text_chat_completion_non_streaming - test_text_chat_completion_streaming - test_text_chat_completion_with_tool_calling_and_non_streaming - test_text_chat_completion_with_tool_calling_and_streaming - test_text_chat_completion_with_tool_choice_required - test_text_chat_completion_with_tool_choice_none - test_text_chat_completion_structured_output - test_text_chat_completion_tool_calling_tools_not_in_request ## Test plan == Set up Ollama local server ``` OLLAMA_HOST=127.0.0.1:8321 with-proxy ollama serve OLLAMA_HOST=127.0.0.1:8321 ollama run llama3.2:3b-instruct-fp16 --keepalive 60m ``` == Run a provider test ``` conda activate stack OLLAMA_URL="http://localhost:8321" \ pytest -v -s -k "ollama" --inference-model="llama3.2:3b-instruct-fp16" \ llama_stack/providers/tests/inference/test_text_inference.py::TestInference ``` == Run an e2e test ``` conda activate sherpa with-proxy pip install llama-stack export INFERENCE_MODEL=llama3.2:3b-instruct-fp16 export LLAMA_STACK_PORT=8322 with-proxy llama stack build --template ollama with-proxy llama stack run --env OLLAMA_URL=http://localhost:8321 ollama ``` ``` conda activate stack LLAMA_STACK_PORT=8322 LLAMA_STACK_BASE_URL="http://localhost:8322" \ pytest -v -s --inference-model="llama3.2:3b-instruct-fp16" \ tests/client-sdk/inference/test_text_inference.py ```
2025-02-25 11:37:04 -08:00 · 2025-02-25 11:37:04 -08:00 · 3a31611486
commit 3a31611486
parent 9b130f96a7
8 changed files with 479 additions and 223 deletions
--- a/llama_stack/providers/tests/test_cases/chat_completion.json
+++ b/llama_stack/providers/tests/test_cases/chat_completion.json
@ -1,24 +0,0 @@
-{
-    "01": {
-        "name": "structured output",
-        "data": {
-            "notes": "We include context about Michael Jordan in the prompt so that the test is focused on the funtionality of the model and not on the information embedded in the model. Llama 3.2 3B Instruct tends to think MJ played for 14 seasons.",
-            "messages": [
-              {
-                "role": "system",
-                "content": "You are a helpful assistant. Michael Jordan was born in 1963. He played basketball for the Chicago Bulls for 15 seasons."
-              },
-              {
-                "role": "user",
-                "content": "Please give me information about Michael Jordan."
-              }
-            ],
-            "expected": {
-                "first_name": "Michael",
-                "last_name": "Jordan",
-                "year_of_birth": 1963,
-                "num_seasons_in_nba": 15
-            }
-        }
-    }
-}
--- a/llama_stack/providers/tests/test_cases/completion.json
+++ b/llama_stack/providers/tests/test_cases/completion.json
@ -1,13 +0,0 @@
-{
-    "01": {
-        "name": "structured output",
-        "data": {
-            "user_input": "Michael Jordan was born in 1963. He played basketball for the Chicago Bulls. He retired in 2003.",
-            "expected": {
-                "name": "Michael Jordan",
-                "year_born": "1963",
-                "year_retired": "2003"
-            }
-        }
-    }
-}
--- a/llama_stack/providers/tests/test_cases/inference/chat_completion.json
+++ b/llama_stack/providers/tests/test_cases/inference/chat_completion.json
@ -0,0 +1,171 @@
+{
+  "non_streaming_01": {
+    "data": {
+      "question": "Which planet do humans live on?",
+      "expected": "Earth"
+    }
+  },
+  "non_streaming_02": {
+    "data": {
+      "question": "Which planet has rings around it with a name starting with letter S?",
+      "expected": "Saturn"
+    }
+  },
+  "sample_messages": {
+    "data": {
+      "messages": [
+        {
+          "role": "system",
+          "content": "You are a helpful assistant."
+        },
+        {
+          "role": "user",
+          "content": "What's the weather like today?"
+        }
+      ]
+    }
+  },
+  "streaming_01": {
+    "data": {
+      "question": "What's the name of the Sun in latin?",
+      "expected": "Sol"
+    }
+  },
+  "streaming_02": {
+    "data": {
+      "question": "What is the name of the US captial?",
+      "expected": "Washington"
+    }
+  },
+  "tool_calling": {
+    "data": {
+      "messages": [
+        {"role": "system", "content": "You are a helpful assistant."},
+        {"role": "user", "content": "What's the weather like in San Francisco?"}
+      ],
+      "tools": [
+        {
+          "tool_name": "get_weather",
+          "description": "Get the current weather",
+          "parameters": {
+            "location": {
+              "param_type": "string",
+              "description": "The city and state, e.g. San Francisco, CA"
+            }
+          }
+        }
+      ],
+      "expected": {
+        "location": "San Francisco, CA"
+      }
+    }
+  },
+  "sample_messages_tool_calling": {
+    "data": {
+      "messages": [
+        {
+          "role": "system",
+          "content": "You are a helpful assistant."
+        },
+        {
+          "role": "user",
+          "content": "What's the weather like today?"
+        },
+        {
+          "role": "user",
+          "content": "What's the weather like in San Francisco?"
+        }
+      ],
+      "tools": [
+        {
+          "tool_name": "get_weather",
+          "description": "Get the current weather",
+          "parameters": {
+            "location": {
+                "param_type": "string",
+                "description": "The city and state, e.g. San Francisco, CA",
+                "required": true
+            }
+          }
+        }
+      ],
+      "expected": {
+        "location": "San Francisco"
+      }
+    }
+  },
+  "structured_output": {
+    "data": {
+      "notes": "We include context about Michael Jordan in the prompt so that the test is focused on the funtionality of the model and not on the information embedded in the model. Llama 3.2 3B Instruct tends to think MJ played for 14 seasons.",
+      "messages": [
+        {
+          "role": "system",
+          "content": "You are a helpful assistant. Michael Jordan was born in 1963. He played basketball for the Chicago Bulls for 15 seasons."
+        },
+        {
+          "role": "user",
+          "content": "Please give me information about Michael Jordan."
+        }
+      ],
+      "expected": {
+        "first_name": "Michael",
+        "last_name": "Jordan",
+        "year_of_birth": 1963,
+        "num_seasons_in_nba": 15
+      }
+    }
+  },
+  "tool_calling_tools_absent": {
+    "data": {
+      "messages": [
+        {
+          "role": "system",
+          "content": "You are a helpful assistant."
+        },
+        {
+          "role": "user",
+          "content": "What pods are in the namespace openshift-lightspeed?"
+        },
+        {
+          "role": "assistant",
+          "content": "",
+          "stop_reason": "end_of_turn",
+          "tool_calls": [
+            {
+              "call_id": "1",
+              "tool_name": "get_object_namespace_list",
+              "arguments": {
+                "kind": "pod",
+                "namespace": "openshift-lightspeed"
+              }
+            }
+          ]
+        },
+        {
+          "role": "tool",
+          "call_id": "1",
+          "tool_name": "get_object_namespace_list",
+          "content": "the objects are pod1, pod2, pod3"
+        }
+      ],
+      "tools": [
+        {
+          "tool_name": "get_object_namespace_list",
+          "description": "Get the list of objects in a namespace",
+          "parameters": {
+            "kind": {
+                "param_type": "string",
+                "description": "the type of object",
+                "required": true
+            },
+            "namespace": {
+                "param_type": "string",
+                "description": "the name of the namespace",
+                "required": true
+            }
+          }
+        }
+      ]
+    }
+  }
+}
--- a/llama_stack/providers/tests/test_cases/inference/completion.json
+++ b/llama_stack/providers/tests/test_cases/inference/completion.json
@ -0,0 +1,43 @@
+{
+    "sanity": {
+        "data": {
+            "content": "Complete the sentence using one word: Roses are red, violets are "
+        }
+    },
+    "non_streaming": {
+        "data": {
+            "content": "Micheael Jordan is born in ",
+            "expected": "1963"
+        }
+    },
+    "streaming": {
+        "data": {
+            "content": "Roses are red,"
+        }
+    },
+    "log_probs": {
+        "data": {
+            "content": "Complete the sentence: Micheael Jordan is born in "
+        }
+    },
+    "logprobs_non_streaming": {
+        "data": {
+            "content": "Micheael Jordan is born in "
+        }
+    },
+    "logprobs_streaming": {
+        "data": {
+            "content": "Roses are red,"
+        }
+    },
+    "structured_output": {
+        "data": {
+            "user_input": "Michael Jordan was born in 1963. He played basketball for the Chicago Bulls. He retired in 2003.",
+            "expected": {
+                "name": "Michael Jordan",
+                "year_born": "1963",
+                "year_retired": "2003"
+            }
+        }
+    }
+}
--- a/llama_stack/providers/tests/test_cases/test_case.py
+++ b/llama_stack/providers/tests/test_cases/test_case.py
@ -9,7 +9,10 @@ import pathlib


 class TestCase:
-    _apis = ["chat_completion", "completion"]
+    _apis = [
+        "inference/chat_completion",
+        "inference/completion",
+    ]
    _jsonblob = {}

    def __init__(self, name):
@ -17,7 +20,12 @@ class TestCase:
        if self._jsonblob == {}:
            for api in self._apis:
                with open(pathlib.Path(__file__).parent / f"{api}.json", "r") as f:
-                    TestCase._jsonblob.update({f"{api}-{k}": v for k, v in json.load(f).items()})
+                    coloned = api.replace("/", ":")
+                    try:
+                        loaded = json.load(f)
+                    except json.JSONDecodeError as e:
+                        raise ValueError(f"There is a syntax error in {api}.json: {e}") from e
+                    TestCase._jsonblob.update({f"{coloned}:{k}": v for k, v in loaded.items()})

        # loading this test case
        tc = self._jsonblob.get(name)
@ -25,7 +33,6 @@ class TestCase:
            raise ValueError(f"Test case {name} not found")

        # these are the only fields we need
-        self.name = tc.get("name")
        self.data = tc.get("data")

    def __getitem__(self, key):