feat(tests): record responses for evals and telemetry tests (#2954)

Continuing with https://github.com/meta-llama/llama-stack/pull/2952 This also includes a "fix" to inference store related tests so that we pull a large number of inference responses from the DB so as to always find the one we just wrote.
2025-12-03 09:53:45 +00:00 · 2025-07-29 15:46:21 -07:00 · 2025-07-29 15:46:21 -07:00 · 0ac503ec0d
commit 0ac503ec0d
parent 81c7d6fa2e
13 changed files with 881 additions and 41 deletions
--- a/tests/integration/inference/test_openai_completion.py
+++ b/tests/integration/inference/test_openai_completion.py
@ -345,7 +345,7 @@ def test_inference_store(compat_client, client_with_models, text_model_id, strea
        response_id = response.id
        content = response.choices[0].message.content

-    responses = client.chat.completions.list()
+    responses = client.chat.completions.list(limit=1000)
    assert response_id in [r.id for r in responses.data]

    retrieved_response = client.chat.completions.retrieve(response_id)
@ -410,7 +410,7 @@ def test_inference_store_tool_calls(compat_client, client_with_models, text_mode
        response_id = response.id
        content = response.choices[0].message.content

-    responses = client.chat.completions.list()
+    responses = client.chat.completions.list(limit=1000)
    assert response_id in [r.id for r in responses.data]

    retrieved_response = client.chat.completions.retrieve(response_id)