feat: inference passthrough provider (#1166)

## What does this PR do? In this PR, we implement a passthrough inference provider that works for any endpoints that respect llama stack inference API definition. ## Test Plan config some endpoint that respect llama stack inference API definition and got the inference results successfully <img width="1268" alt="Screenshot 2025-02-19 at 8 52 51 PM" src="https://github.com/user-attachments/assets/447816e4-ea7a-4365-b90c-386dc7dcf4a1" />
2025-10-12 05:54:38 +00:00 · 2025-02-19 21:47:00 -08:00 · 2025-02-19 21:47:00 -08:00 · 2b995c22eb
commit 2b995c22eb
parent d39f8de619
6 changed files with 364 additions and 0 deletions
--- a/llama_stack/templates/passthrough/build.yaml
+++ b/llama_stack/templates/passthrough/build.yaml
@ -0,0 +1,32 @@
+version: '2'
+distribution_spec:
+  description: Use for running LLM inference with the endpoint that compatible with Llama Stack API
+  providers:
+    inference:
+    - remote::passthrough
+    vector_io:
+    - inline::faiss
+    - remote::chromadb
+    - remote::pgvector
+    safety:
+    - inline::llama-guard
+    agents:
+    - inline::meta-reference
+    telemetry:
+    - inline::meta-reference
+    eval:
+    - inline::meta-reference
+    datasetio:
+    - remote::huggingface
+    - inline::localfs
+    scoring:
+    - inline::basic
+    - inline::llm-as-judge
+    - inline::braintrust
+    tool_runtime:
+    - remote::brave-search
+    - remote::tavily-search
+    - inline::code-interpreter
+    - inline::rag-runtime
+    - remote::model-context-protocol
+image_type: conda