Initial commit

2025-12-05 18:27:22 +00:00 · 2024-06-25 15:47:57 -07:00 · 2024-06-25 15:47:57 -07:00 · 5d5acc8ed5
commit 5d5acc8ed5
81 changed files with 4458 additions and 0 deletions
--- a/llama_toolchain/reward_scoring/api/init.py
+++ b/llama_toolchain/reward_scoring/api/init.py
@ -0,0 +1,8 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the terms described in the LICENSE file in
+# the root directory of this source tree.
+
+from .datatypes import *  # noqa: F401 F403
+from .endpoints import *  # noqa: F401 F403
--- a/llama_toolchain/reward_scoring/api/datatypes.py
+++ b/llama_toolchain/reward_scoring/api/datatypes.py
@ -0,0 +1,31 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the terms described in the LICENSE file in
+# the root directory of this source tree.
+
+from typing import List
+
+from pydantic import BaseModel
+
+from strong_typing.schema import json_schema_type
+
+from llama_models.llama3_1.api.datatypes import *  # noqa: F403
+
+
+@json_schema_type
+class ScoredMessage(BaseModel):
+    message: Message
+    score: float
+
+
+@json_schema_type
+class DialogGenerations(BaseModel):
+    dialog: List[Message]
+    sampled_generations: List[Message]
+
+
+@json_schema_type
+class ScoredDialogGenerations(BaseModel):
+    dialog: List[Message]
+    scored_generations: List[ScoredMessage]
--- a/llama_toolchain/reward_scoring/api/endpoints.py
+++ b/llama_toolchain/reward_scoring/api/endpoints.py
@ -0,0 +1,33 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the terms described in the LICENSE file in
+# the root directory of this source tree.
+
+from typing import List, Protocol, Union
+from .datatypes import *  # noqa: F403
+
+from pyopenapi import webmethod
+
+
+@json_schema_type
+class RewardScoringRequest(BaseModel):
+    """Request to score a reward function. A list of prompts and a list of responses per prompt."""
+
+    dialog_generations: List[DialogGenerations]
+    model: RewardModel
+
+
+@json_schema_type
+class RewardScoringResponse(BaseModel):
+    """Response from the reward scoring. Batch of (prompt, response, score) tuples that pass the threshold."""
+
+    scored_generations: List[ScoredDialogGenerations]
+
+
+class RewardScoring(Protocol):
+    @webmethod(route="/reward_scoring/score")
+    def post_score(
+        self,
+        request: RewardScoringRequest,
+    ) -> Union[RewardScoringResponse]: ...