scoring fix

2025-12-20 19:02:27 +00:00 · 2024-11-06 18:07:16 -08:00 · 2024-11-06 18:07:16 -08:00 · 56239fce90
commit 56239fce90
parent c5cf9c30be
10 changed files with 104 additions and 15 deletions
--- a/llama_stack/providers/tests/scoring/test_scoring.py
+++ b/llama_stack/providers/tests/scoring/test_scoring.py
@ -7,6 +7,8 @@

 import pytest

+from llama_stack.apis.scoring_functions import *  # noqa: F403
+
 from llama_stack.providers.tests.datasetio.test_datasetio import register_dataset

 # How to run this test:
@ -64,3 +66,51 @@ class TestScoring:
        for x in scoring_functions:
            assert x in response.results
            assert len(response.results[x].score_rows) == 5
+
+    @pytest.mark.asyncio
+    async def test_scoring_score_with_params(self, scoring_stack):
+        scoring_impl, scoring_functions_impl, datasetio_impl, datasets_impl = (
+            scoring_stack
+        )
+        await register_dataset(datasets_impl)
+        response = await datasets_impl.list_datasets()
+        assert len(response) == 1
+
+        # scoring individual rows
+        rows = await datasetio_impl.get_rows_paginated(
+            dataset_id="test_dataset",
+            rows_in_page=3,
+        )
+        assert len(rows.rows) == 3
+
+        params = {
+            "meta-reference::llm_as_judge_8b_correctness": LLMAsJudgeScoringFnParams(
+                judge_model="Llama3.1-405B-Instruct",
+                prompt_template="Output a number response in the following format: Score: <number>, where <number> is the number between 0 and 9.",
+                judge_score_regexes=[r"Score: (\d+)"],
+            )
+        }
+
+        scoring_functions = [
+            "meta-reference::llm_as_judge_8b_correctness",
+        ]
+        response = await scoring_impl.score(
+            input_rows=rows.rows,
+            scoring_functions=scoring_functions,
+            scoring_params=params,
+        )
+        assert len(response.results) == len(scoring_functions)
+        for x in scoring_functions:
+            assert x in response.results
+            assert len(response.results[x].score_rows) == len(rows.rows)
+
+        # score batch
+        response = await scoring_impl.score_batch(
+            dataset_id="test_dataset",
+            scoring_functions=scoring_functions,
+            scoring_params=params,
+        )
+        assert len(response.results) == len(scoring_functions)
+        for x in scoring_functions:
+            assert x in response.results
+            assert len(response.results[x].score_rows) == 5