feat(tests): introduce a test "suite" concept to encompass dirs, options (#3339)

Our integration tests need to be 'grouped' because each group often needs a specific set of models it works with. We separated vision tests due to this, and we have a separate set of tests which test "Responses" API. This PR makes this system a bit more official so it is very easy to target these groups and apply all testing infrastructure towards all the groups (for example, record-replay) uniformly. There are three suites declared: - base - vision - responses Note that our CI currently runs the "base" and "vision" suites. You can use the `--suite` option when running pytest (or any of the testing scripts or workflows.) For example: ``` OLLAMA_URL=http://localhost:11434 \ pytest -s -v tests/integration/ --stack-config starter --suite vision ```
2025-12-03 09:53:45 +00:00 · 2025-09-05 13:58:49 -07:00 · 2025-09-05 13:58:49 -07:00 · 47b640370e
commit 47b640370e
parent 0c2757a05b
25 changed files with 255 additions and 161 deletions
--- a/tests/integration/suites.py
+++ b/tests/integration/suites.py
@ -0,0 +1,53 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the terms described in the LICENSE file in
+# the root directory of this source tree.
+
+# Central definition of integration test suites. You can use these suites by passing --suite=name to pytest.
+# For example:
+#
+# ```bash
+# pytest tests/integration/ --suite=vision
+# ```
+#
+# Each suite can:
+# - restrict collection to specific roots (dirs or files)
+# - provide default CLI option values (e.g. text_model, embedding_model, etc.)
+
+from pathlib import Path
+
+this_dir = Path(__file__).parent
+default_roots = [
+    str(p)
+    for p in this_dir.glob("*")
+    if p.is_dir()
+    and p.name not in ("__pycache__", "fixtures", "test_cases", "recordings", "responses", "post_training")
+]
+
+SUITE_DEFINITIONS: dict[str, dict] = {
+    "base": {
+        "description": "Base suite that includes most tests but runs them with a text Ollama model",
+        "roots": default_roots,
+        "defaults": {
+            "text_model": "ollama/llama3.2:3b-instruct-fp16",
+            "embedding_model": "sentence-transformers/all-MiniLM-L6-v2",
+        },
+    },
+    "responses": {
+        "description": "Suite that includes only the OpenAI Responses tests; needs a strong tool-calling model",
+        "roots": ["tests/integration/responses"],
+        "defaults": {
+            "text_model": "openai/gpt-4o",
+            "embedding_model": "sentence-transformers/all-MiniLM-L6-v2",
+        },
+    },
+    "vision": {
+        "description": "Suite that includes only the vision tests",
+        "roots": ["tests/integration/inference/test_vision_inference.py"],
+        "defaults": {
+            "vision_model": "ollama/llama3.2-vision:11b",
+            "embedding_model": "sentence-transformers/all-MiniLM-L6-v2",
+        },
+    },
+}