mirror of
https://github.com/meta-llama/llama-stack.git
synced 2025-12-03 09:53:45 +00:00
# What does this PR do? <!-- Provide a short summary of what this PR does and why. Link to relevant issues if applicable. --> This PR is responsible for attaching prompts to storage stores in run configs. It allows to specify prompts as stores in different distributions. The need of this functionality was initiated in #3514 > Note, #3514 is divided on three separate PRs. Current PR is the first of three. <!-- If resolving an issue, uncomment and update the line below --> <!-- Closes #[issue-number] --> ## Test Plan <!-- Describe the tests you ran to verify your changes with result summaries. *Provide clear instructions so the plan can be easily re-executed.* --> Manual testing and updated CI unit tests Prerequisites: 1. `uv run --with llama-stack llama stack list-deps starter | xargs -L1 uv pip install` 2. `llama stack run starter ` ``` INFO 2025-10-23 15:36:17,387 llama_stack.cli.stack.run:100 cli: Using run configuration: /Users/ianmiller/llama-stack/llama_stack/distributions/starter/run.yaml INFO 2025-10-23 15:36:17,423 llama_stack.cli.stack.run:157 cli: HTTPS enabled with certificates: Key: None Cert: None INFO 2025-10-23 15:36:17,424 llama_stack.cli.stack.run:159 cli: Listening on ['::', '0.0.0.0']:8321 INFO 2025-10-23 15:36:17,749 llama_stack.core.server.server:521 core::server: Run configuration: INFO 2025-10-23 15:36:17,756 llama_stack.core.server.server:524 core::server: apis: - agents - batches - datasetio - eval - files - inference - post_training - safety - scoring - tool_runtime - vector_io image_name: starter providers: agents: - config: persistence: agent_state: backend: kv_default namespace: agents responses: backend: sql_default max_write_queue_size: 10000 num_writers: 4 table_name: responses provider_id: meta-reference provider_type: inline::meta-reference batches: - config: kvstore: backend: kv_default namespace: batches provider_id: reference provider_type: inline::reference datasetio: - config: kvstore: backend: kv_default namespace: datasetio::huggingface provider_id: huggingface provider_type: remote::huggingface - config: kvstore: backend: kv_default namespace: datasetio::localfs provider_id: localfs provider_type: inline::localfs eval: - config: kvstore: backend: kv_default namespace: eval provider_id: meta-reference provider_type: inline::meta-reference files: - config: metadata_store: backend: sql_default table_name: files_metadata storage_dir: /Users/ianmiller/.llama/distributions/starter/files provider_id: meta-reference-files provider_type: inline::localfs inference: - config: api_key: '********' url: https://api.fireworks.ai/inference/v1 provider_id: fireworks provider_type: remote::fireworks - config: api_key: '********' url: https://api.together.xyz/v1 provider_id: together provider_type: remote::together - config: {} provider_id: bedrock provider_type: remote::bedrock - config: api_key: '********' base_url: https://api.openai.com/v1 provider_id: openai provider_type: remote::openai - config: api_key: '********' provider_id: anthropic provider_type: remote::anthropic - config: api_key: '********' provider_id: gemini provider_type: remote::gemini - config: api_key: '********' url: https://api.groq.com provider_id: groq provider_type: remote::groq - config: api_key: '********' url: https://api.sambanova.ai/v1 provider_id: sambanova provider_type: remote::sambanova - config: {} provider_id: sentence-transformers provider_type: inline::sentence-transformers post_training: - config: checkpoint_format: meta provider_id: torchtune-cpu provider_type: inline::torchtune-cpu safety: - config: excluded_categories: [] provider_id: llama-guard provider_type: inline::llama-guard - config: {} provider_id: code-scanner provider_type: inline::code-scanner scoring: - config: {} provider_id: basic provider_type: inline::basic - config: {} provider_id: llm-as-judge provider_type: inline::llm-as-judge - config: openai_api_key: '********' provider_id: braintrust provider_type: inline::braintrust tool_runtime: - config: api_key: '********' max_results: 3 provider_id: brave-search provider_type: remote::brave-search - config: api_key: '********' max_results: 3 provider_id: tavily-search provider_type: remote::tavily-search - config: {} provider_id: rag-runtime provider_type: inline::rag-runtime - config: {} provider_id: model-context-protocol provider_type: remote::model-context-protocol vector_io: - config: persistence: backend: kv_default namespace: vector_io::faiss provider_id: faiss provider_type: inline::faiss - config: db_path: /Users/ianmiller/.llama/distributions/starter/sqlite_vec.db persistence: backend: kv_default namespace: vector_io::sqlite_vec provider_id: sqlite-vec provider_type: inline::sqlite-vec registered_resources: benchmarks: [] datasets: [] models: [] scoring_fns: [] shields: [] tool_groups: - provider_id: tavily-search toolgroup_id: builtin::websearch - provider_id: rag-runtime toolgroup_id: builtin::rag vector_stores: [] server: port: 8321 storage: backends: kv_default: db_path: /Users/ianmiller/.llama/distributions/starter/kvstore.db type: kv_sqlite sql_default: db_path: /Users/ianmiller/.llama/distributions/starter/sql_store.db type: sql_sqlite stores: conversations: backend: sql_default table_name: openai_conversations inference: backend: sql_default max_write_queue_size: 10000 num_writers: 4 table_name: inference_store metadata: backend: kv_default namespace: registry prompts: backend: kv_default namespace: prompts telemetry: enabled: true vector_stores: default_embedding_model: model_id: nomic-ai/nomic-embed-text-v1.5 provider_id: sentence-transformers default_provider_id: faiss version: 2 INFO 2025-10-23 15:36:20,032 llama_stack.providers.utils.inference.inference_store:74 inference: Write queue disabled for SQLite to avoid concurrency issues WARNING 2025-10-23 15:36:20,422 llama_stack.providers.inline.telemetry.meta_reference.telemetry:84 telemetry: OTEL_EXPORTER_OTLP_ENDPOINT is not set, skipping telemetry INFO 2025-10-23 15:36:22,379 llama_stack.providers.utils.inference.openai_mixin:436 providers::utils: OpenAIInferenceAdapter.list_provider_model_ids() returned 105 models INFO 2025-10-23 15:36:22,703 uvicorn.error:84 uncategorized: Started server process [17328] INFO 2025-10-23 15:36:22,704 uvicorn.error:48 uncategorized: Waiting for application startup. INFO 2025-10-23 15:36:22,706 llama_stack.core.server.server:179 core::server: Starting up Llama Stack server (version: 0.3.0) INFO 2025-10-23 15:36:22,707 llama_stack.core.stack:470 core: starting registry refresh task INFO 2025-10-23 15:36:22,708 uvicorn.error:62 uncategorized: Application startup complete. INFO 2025-10-23 15:36:22,708 uvicorn.error:216 uncategorized: Uvicorn running on http://['::', '0.0.0.0']:8321 (Press CTRL+C to quit) ``` As you can see, prompts are attached to stores in config Testing: 1. Create prompt: ``` curl -X POST http://localhost:8321/v1/prompts \ -H "Content-Type: application/json" \ -d '{ "prompt": "Hello {{name}}! You are working at {{company}}. Your role is {{role}} at {{company}}. Remember, {{name}}, to be {{tone}}.", "variables": ["name", "company", "role", "tone"] }' ``` `{"prompt":"Hello {{name}}! You are working at {{company}}. Your role is {{role}} at {{company}}. Remember, {{name}}, to be {{tone}}.","version":1,"prompt_id":"pmpt_a90e09e67acfe23776f2778c603eb6c17e139dab5f6e163f","variables":["name","company","role","tone"],"is_default":false}% ` 2. Get prompt: `curl -X GET http://localhost:8321/v1/prompts/pmpt_a90e09e67acfe23776f2778c603eb6c17e139dab5f6e163f` `{"prompt":"Hello {{name}}! You are working at {{company}}. Your role is {{role}} at {{company}}. Remember, {{name}}, to be {{tone}}.","version":1,"prompt_id":"pmpt_a90e09e67acfe23776f2778c603eb6c17e139dab5f6e163f","variables":["name","company","role","tone"],"is_default":false}% ` 3. Query sqlite KV storage to check created prompt: ``` sqlite> .mode column sqlite> .headers on sqlite> SELECT * FROM kvstore WHERE key LIKE 'prompts:v1:%'; key value expiration ------------------------------------------------------------ ------------------------------------------------------------ ---------- prompts:v1:pmpt_a90e09e67acfe23776f2778c603eb6c17e139dab5f6e {"prompt_id": "pmpt_a90e09e67acfe23776f2778c603eb6c17e139dab 163f:1 5f6e163f", "prompt": "Hello {{name}}! You are working at {{c ompany}}. Your role is {{role}} at {{company}}. Remember, {{ name}}, to be {{tone}}.", "version": 1, "variables": ["name" , "company", "role", "tone"], "is_default": false} prompts:v1:pmpt_a90e09e67acfe23776f2778c603eb6c17e139dab5f6e 1 163f:default sqlite> ```
232 lines
8.5 KiB
Python
232 lines
8.5 KiB
Python
# Copyright (c) Meta Platforms, Inc. and affiliates.
|
|
# All rights reserved.
|
|
#
|
|
# This source code is licensed under the terms described in the LICENSE file in
|
|
# the root directory of this source tree.
|
|
|
|
import json
|
|
from typing import Any
|
|
|
|
from pydantic import BaseModel
|
|
|
|
from llama_stack.apis.prompts import ListPromptsResponse, Prompt, Prompts
|
|
from llama_stack.core.datatypes import StackRunConfig
|
|
from llama_stack.providers.utils.kvstore import KVStore, kvstore_impl
|
|
|
|
|
|
class PromptServiceConfig(BaseModel):
|
|
"""Configuration for the built-in prompt service.
|
|
|
|
:param run_config: Stack run configuration containing distribution info
|
|
"""
|
|
|
|
run_config: StackRunConfig
|
|
|
|
|
|
async def get_provider_impl(config: PromptServiceConfig, deps: dict[Any, Any]):
|
|
"""Get the prompt service implementation."""
|
|
impl = PromptServiceImpl(config, deps)
|
|
await impl.initialize()
|
|
return impl
|
|
|
|
|
|
class PromptServiceImpl(Prompts):
|
|
"""Built-in prompt service implementation using KVStore."""
|
|
|
|
def __init__(self, config: PromptServiceConfig, deps: dict[Any, Any]):
|
|
self.config = config
|
|
self.deps = deps
|
|
self.kvstore: KVStore
|
|
|
|
async def initialize(self) -> None:
|
|
# Use prompts store reference from run config
|
|
prompts_ref = self.config.run_config.storage.stores.prompts
|
|
if not prompts_ref:
|
|
raise ValueError("storage.stores.prompts must be configured in run config")
|
|
self.kvstore = await kvstore_impl(prompts_ref)
|
|
|
|
def _get_default_key(self, prompt_id: str) -> str:
|
|
"""Get the KVStore key that stores the default version number."""
|
|
return f"prompts:v1:{prompt_id}:default"
|
|
|
|
async def _get_prompt_key(self, prompt_id: str, version: int | None = None) -> str:
|
|
"""Get the KVStore key for prompt data, returning default version if applicable."""
|
|
if version:
|
|
return self._get_version_key(prompt_id, str(version))
|
|
|
|
default_key = self._get_default_key(prompt_id)
|
|
resolved_version = await self.kvstore.get(default_key)
|
|
if resolved_version is None:
|
|
raise ValueError(f"Prompt {prompt_id}:default not found")
|
|
return self._get_version_key(prompt_id, resolved_version)
|
|
|
|
def _get_version_key(self, prompt_id: str, version: str) -> str:
|
|
"""Get the KVStore key for a specific prompt version."""
|
|
return f"prompts:v1:{prompt_id}:{version}"
|
|
|
|
def _get_list_key_prefix(self) -> str:
|
|
"""Get the key prefix for listing prompts."""
|
|
return "prompts:v1:"
|
|
|
|
def _serialize_prompt(self, prompt: Prompt) -> str:
|
|
"""Serialize a prompt to JSON string for storage."""
|
|
return json.dumps(
|
|
{
|
|
"prompt_id": prompt.prompt_id,
|
|
"prompt": prompt.prompt,
|
|
"version": prompt.version,
|
|
"variables": prompt.variables or [],
|
|
"is_default": prompt.is_default,
|
|
}
|
|
)
|
|
|
|
def _deserialize_prompt(self, data: str) -> Prompt:
|
|
"""Deserialize a prompt from JSON string."""
|
|
obj = json.loads(data)
|
|
return Prompt(
|
|
prompt_id=obj["prompt_id"],
|
|
prompt=obj["prompt"],
|
|
version=obj["version"],
|
|
variables=obj.get("variables", []),
|
|
is_default=obj.get("is_default", False),
|
|
)
|
|
|
|
async def list_prompts(self) -> ListPromptsResponse:
|
|
"""List all prompts (default versions only)."""
|
|
prefix = self._get_list_key_prefix()
|
|
keys = await self.kvstore.keys_in_range(prefix, prefix + "\xff")
|
|
|
|
prompts = []
|
|
for key in keys:
|
|
if key.endswith(":default"):
|
|
try:
|
|
default_version = await self.kvstore.get(key)
|
|
if default_version:
|
|
prompt_id = key.replace(prefix, "").replace(":default", "")
|
|
version_key = self._get_version_key(prompt_id, default_version)
|
|
data = await self.kvstore.get(version_key)
|
|
if data:
|
|
prompt = self._deserialize_prompt(data)
|
|
prompts.append(prompt)
|
|
except (json.JSONDecodeError, KeyError):
|
|
continue
|
|
|
|
prompts.sort(key=lambda p: p.prompt_id or "", reverse=True)
|
|
return ListPromptsResponse(data=prompts)
|
|
|
|
async def get_prompt(self, prompt_id: str, version: int | None = None) -> Prompt:
|
|
"""Get a prompt by its identifier and optional version."""
|
|
key = await self._get_prompt_key(prompt_id, version)
|
|
data = await self.kvstore.get(key)
|
|
if data is None:
|
|
raise ValueError(f"Prompt {prompt_id}:{version if version else 'default'} not found")
|
|
return self._deserialize_prompt(data)
|
|
|
|
async def create_prompt(
|
|
self,
|
|
prompt: str,
|
|
variables: list[str] | None = None,
|
|
) -> Prompt:
|
|
"""Create a new prompt."""
|
|
if variables is None:
|
|
variables = []
|
|
|
|
prompt_obj = Prompt(
|
|
prompt_id=Prompt.generate_prompt_id(),
|
|
prompt=prompt,
|
|
version=1,
|
|
variables=variables,
|
|
)
|
|
|
|
version_key = self._get_version_key(prompt_obj.prompt_id, str(prompt_obj.version))
|
|
data = self._serialize_prompt(prompt_obj)
|
|
await self.kvstore.set(version_key, data)
|
|
|
|
default_key = self._get_default_key(prompt_obj.prompt_id)
|
|
await self.kvstore.set(default_key, str(prompt_obj.version))
|
|
|
|
return prompt_obj
|
|
|
|
async def update_prompt(
|
|
self,
|
|
prompt_id: str,
|
|
prompt: str,
|
|
version: int,
|
|
variables: list[str] | None = None,
|
|
set_as_default: bool = True,
|
|
) -> Prompt:
|
|
"""Update an existing prompt (increments version)."""
|
|
if version < 1:
|
|
raise ValueError("Version must be >= 1")
|
|
if variables is None:
|
|
variables = []
|
|
|
|
prompt_versions = await self.list_prompt_versions(prompt_id)
|
|
latest_prompt = max(prompt_versions.data, key=lambda x: int(x.version))
|
|
|
|
if version and latest_prompt.version != version:
|
|
raise ValueError(
|
|
f"'{version}' is not the latest prompt version for prompt_id='{prompt_id}'. Use the latest version '{latest_prompt.version}' in request."
|
|
)
|
|
|
|
current_version = latest_prompt.version if version is None else version
|
|
new_version = current_version + 1
|
|
|
|
updated_prompt = Prompt(prompt_id=prompt_id, prompt=prompt, version=new_version, variables=variables)
|
|
|
|
version_key = self._get_version_key(prompt_id, str(new_version))
|
|
data = self._serialize_prompt(updated_prompt)
|
|
await self.kvstore.set(version_key, data)
|
|
|
|
if set_as_default:
|
|
await self.set_default_version(prompt_id, new_version)
|
|
|
|
return updated_prompt
|
|
|
|
async def delete_prompt(self, prompt_id: str) -> None:
|
|
"""Delete a prompt and all its versions."""
|
|
await self.get_prompt(prompt_id)
|
|
|
|
prefix = f"prompts:v1:{prompt_id}:"
|
|
keys = await self.kvstore.keys_in_range(prefix, prefix + "\xff")
|
|
|
|
for key in keys:
|
|
await self.kvstore.delete(key)
|
|
|
|
async def list_prompt_versions(self, prompt_id: str) -> ListPromptsResponse:
|
|
"""List all versions of a specific prompt."""
|
|
prefix = f"prompts:v1:{prompt_id}:"
|
|
keys = await self.kvstore.keys_in_range(prefix, prefix + "\xff")
|
|
|
|
default_version = None
|
|
prompts = []
|
|
|
|
for key in keys:
|
|
data = await self.kvstore.get(key)
|
|
if key.endswith(":default"):
|
|
default_version = data
|
|
else:
|
|
if data:
|
|
prompt_obj = self._deserialize_prompt(data)
|
|
prompts.append(prompt_obj)
|
|
|
|
if not prompts:
|
|
raise ValueError(f"Prompt {prompt_id} not found")
|
|
|
|
for prompt in prompts:
|
|
prompt.is_default = str(prompt.version) == default_version
|
|
|
|
prompts.sort(key=lambda x: x.version)
|
|
return ListPromptsResponse(data=prompts)
|
|
|
|
async def set_default_version(self, prompt_id: str, version: int) -> Prompt:
|
|
"""Set which version of a prompt should be the default, If not set. the default is the latest."""
|
|
version_key = self._get_version_key(prompt_id, str(version))
|
|
data = await self.kvstore.get(version_key)
|
|
if data is None:
|
|
raise ValueError(f"Prompt {prompt_id} version {version} not found")
|
|
|
|
default_key = self._get_default_key(prompt_id)
|
|
await self.kvstore.set(default_key, str(version))
|
|
|
|
return self._deserialize_prompt(data)
|