From 39872ca4b411da603a9b170c746a4badefde3d75 Mon Sep 17 00:00:00 2001
From: Xi Yan <xiyan@meta.com>
Date: Tue, 29 Oct 2024 15:00:58 -0700
Subject: [PATCH] distributions

---
 distributions/ollama/README.md                |  20 +--
 distributions/tgi/README.md                   |  35 ++---
 distributions/together/README.md              |  20 +--
 .../getting_started/distributions/index.md    |   3 +
 .../getting_started/distributions/ollama.md   | 129 ++++++++++++++++++
 .../getting_started/distributions/tgi.md      | 112 +++++++++++++++
 .../getting_started/distributions/together.md |  65 +++++++++
 7 files changed, 336 insertions(+), 48 deletions(-)
 create mode 100644 docs/source/getting_started/distributions/ollama.md
 create mode 100644 docs/source/getting_started/distributions/tgi.md
 create mode 100644 docs/source/getting_started/distributions/together.md

diff --git a/distributions/ollama/README.md b/distributions/ollama/README.md
index f969d86ec..e1508f5b0 100644
--- a/distributions/ollama/README.md
+++ b/distributions/ollama/README.md
@@ -7,7 +7,7 @@ The `llamastack/distribution-ollama` distribution consists of the following prov
 | **Provider(s)** 	| remote::ollama 	| meta-reference 	| remote::pgvector, remote::chroma 	| remote::ollama 	| meta-reference 	|
 
 
-### Start a Distribution (Single Node GPU)
+### Docker: Start a Distribution (Single Node GPU)
 
 > [!NOTE]
 > This assumes you have access to GPU to start a Ollama server with access to your GPU.
@@ -38,7 +38,7 @@ To kill the server
 docker compose down
 ```
 
-### Start the Distribution (Single Node CPU)
+### Docker: Start the Distribution (Single Node CPU)
 
 > [!NOTE]
 > This will start an ollama server with CPU only, please see [Ollama Documentations](https://github.com/ollama/ollama) for serving models on CPU only.
@@ -50,7 +50,7 @@ compose.yaml  run.yaml
 $ docker compose up
 ```
 
-### (Alternative) ollama run + llama stack run
+### Conda: ollama run + llama stack run
 
 If you wish to separately spin up a Ollama server, and connect with Llama Stack, you may use the following commands.
 
@@ -69,6 +69,13 @@ ollama run <model_id>
 
 #### Start Llama Stack server pointing to Ollama server
 
+**Via Conda**
+
+```
+llama stack build --template ollama --image-type conda
+llama stack run ./gpu/run.yaml
+```
+
 **Via Docker**
 ```
 docker run --network host -it -p 5000:5000 -v ~/.llama:/root/.llama -v ./gpu/run.yaml:/root/llamastack-run-ollama.yaml --gpus=all llamastack/distribution-ollama --yaml_config /root/llamastack-run-ollama.yaml
@@ -83,13 +90,6 @@ inference:
       url: http://127.0.0.1:14343
 ```
 
-**Via Conda**
-
-```
-llama stack build --template ollama --image-type conda
-llama stack run ./gpu/run.yaml
-```
-
 ### Model Serving
 
 #### Downloading model via Ollama
diff --git a/distributions/tgi/README.md b/distributions/tgi/README.md
index f274f8ff0..aeb693364 100644
--- a/distributions/tgi/README.md
+++ b/distributions/tgi/README.md
@@ -8,17 +8,14 @@ The `llamastack/distribution-tgi` distribution consists of the following provide
 | **Provider(s)** 	| remote::tgi   	| meta-reference 	| meta-reference, remote::pgvector, remote::chroma 	| meta-reference 	| meta-reference 	|
 
 
-### Start the Distribution (Single Node GPU)
+### Docker: Start the Distribution (Single Node GPU)
 
 > [!NOTE]
 > This assumes you have access to GPU to start a TGI server with access to your GPU.
 
 
 ```
-$ cd distributions/tgi/gpu
-$ ls
-compose.yaml  tgi-run.yaml
-$ docker compose up
+$ cd distributions/tgi/gpu && docker compose up
 ```
 
 The script will first start up TGI server, then start up Llama Stack distribution server hooking up to the remote TGI provider for inference. You should be able to see the following outputs --
@@ -37,16 +34,13 @@ To kill the server
 docker compose down
 ```
 
-### Start the Distribution (Single Node CPU)
+### Docker: Start the Distribution (Single Node CPU)
 
 > [!NOTE]
 > This assumes you have an hosted endpoint compatible with TGI server.
 
 ```
-$ cd distributions/tgi/cpu
-$ ls
-compose.yaml  run.yaml
-$ docker compose up
+$ cd distributions/tgi/cpu && docker compose up
 ```
 
 Replace <ENTER_YOUR_TGI_HOSTED_ENDPOINT> in `run.yaml` file with your TGI endpoint.
@@ -58,20 +52,28 @@ inference:
       url: <ENTER_YOUR_TGI_HOSTED_ENDPOINT>
 ```
 
-### (Alternative) TGI server + llama stack run (Single Node GPU)
+### Conda: TGI server + llama stack run
 
 If you wish to separately spin up a TGI server, and connect with Llama Stack, you may use the following commands.
 
-#### (optional) Start TGI server locally
+#### Start TGI server locally
 - Please check the [TGI Getting Started Guide](https://github.com/huggingface/text-generation-inference?tab=readme-ov-file#get-started) to get a TGI endpoint.
 
 ```
 docker run --rm -it -v $HOME/.cache/huggingface:/data -p 5009:5009 --gpus all ghcr.io/huggingface/text-generation-inference:latest --dtype bfloat16 --usage-stats on --sharded false --model-id meta-llama/Llama-3.1-8B-Instruct --port 5009
 ```
 
-
 #### Start Llama Stack server pointing to TGI server
 
+**Via Conda**
+
+```bash
+llama stack build --template tgi --image-type conda
+# -- start a TGI server endpoint
+llama stack run ./gpu/run.yaml
+```
+
+**Via Docker**
 ```
 docker run --network host -it -p 5000:5000 -v ./run.yaml:/root/my-run.yaml --gpus=all llamastack/distribution-tgi --yaml_config /root/my-run.yaml
 ```
@@ -85,13 +87,6 @@ inference:
       url: http://127.0.0.1:5009
 ```
 
-**Via Conda**
-
-```bash
-llama stack build --template tgi --image-type conda
-# -- start a TGI server endpoint
-llama stack run ./gpu/run.yaml
-```
 
 ### Model Serving
 To serve a new model with `tgi`, change the docker command flag `--model-id <model-to-serve>`.
diff --git a/distributions/together/README.md b/distributions/together/README.md
index 378b7c0c7..5f9c90071 100644
--- a/distributions/together/README.md
+++ b/distributions/together/README.md
@@ -11,7 +11,7 @@ The `llamastack/distribution-together` distribution consists of the following pr
 | **Provider(s)** 	| remote::together   	| meta-reference 	| meta-reference, remote::weaviate 	| meta-reference 	| meta-reference 	|
 
 
-### Start the Distribution (Single Node CPU)
+### Docker: Start the Distribution (Single Node CPU)
 
 > [!NOTE]
 > This assumes you have an hosted endpoint at Together with API Key.
@@ -33,23 +33,7 @@ inference:
       api_key: <optional api key>
 ```
 
-### (Alternative) llama stack run (Single Node CPU)
-
-```
-docker run --network host -it -p 5000:5000 -v ./run.yaml:/root/my-run.yaml --gpus=all llamastack/distribution-together --yaml_config /root/my-run.yaml
-```
-
-Make sure in you `run.yaml` file, you inference provider is pointing to the correct Together URL server endpoint. E.g.
-```
-inference:
-  - provider_id: together
-    provider_type: remote::together
-    config:
-      url: https://api.together.xyz/v1
-      api_key: <optional api key>
-```
-
-**Via Conda**
+### Conda llama stack run (Single Node CPU)
 
 ```bash
 llama stack build --template together --image-type conda
diff --git a/docs/source/getting_started/distributions/index.md b/docs/source/getting_started/distributions/index.md
index 05ceb4787..aa0e3cf98 100644
--- a/docs/source/getting_started/distributions/index.md
+++ b/docs/source/getting_started/distributions/index.md
@@ -6,4 +6,7 @@ A Distribution is where APIs and Providers are assembled together to provide a c
 :maxdepth: 1
 
 meta-reference-gpu
+ollama
+tgi
+together
 ```
diff --git a/docs/source/getting_started/distributions/ollama.md b/docs/source/getting_started/distributions/ollama.md
new file mode 100644
index 000000000..e1508f5b0
--- /dev/null
+++ b/docs/source/getting_started/distributions/ollama.md
@@ -0,0 +1,129 @@
+# Ollama Distribution
+
+The `llamastack/distribution-ollama` distribution consists of the following provider configurations.
+
+| **API**         	| **Inference**  	| **Agents**     	| **Memory**                       	| **Safety**     	| **Telemetry**  	|
+|-----------------	|----------------	|----------------	|----------------------------------	|----------------	|----------------	|
+| **Provider(s)** 	| remote::ollama 	| meta-reference 	| remote::pgvector, remote::chroma 	| remote::ollama 	| meta-reference 	|
+
+
+### Docker: Start a Distribution (Single Node GPU)
+
+> [!NOTE]
+> This assumes you have access to GPU to start a Ollama server with access to your GPU.
+
+```
+$ cd distributions/ollama/gpu
+$ ls
+compose.yaml  run.yaml
+$ docker compose up
+```
+
+You will see outputs similar to following ---
+```
+[ollama]               | [GIN] 2024/10/18 - 21:19:41 | 200 |     226.841µs |             ::1 | GET      "/api/ps"
+[ollama]               | [GIN] 2024/10/18 - 21:19:42 | 200 |      60.908µs |             ::1 | GET      "/api/ps"
+INFO:     Started server process [1]
+INFO:     Waiting for application startup.
+INFO:     Application startup complete.
+INFO:     Uvicorn running on http://[::]:5000 (Press CTRL+C to quit)
+[llamastack] | Resolved 12 providers
+[llamastack] |  inner-inference => ollama0
+[llamastack] |  models => __routing_table__
+[llamastack] |  inference => __autorouted__
+```
+
+To kill the server
+```
+docker compose down
+```
+
+### Docker: Start the Distribution (Single Node CPU)
+
+> [!NOTE]
+> This will start an ollama server with CPU only, please see [Ollama Documentations](https://github.com/ollama/ollama) for serving models on CPU only.
+
+```
+$ cd distributions/ollama/cpu
+$ ls
+compose.yaml  run.yaml
+$ docker compose up
+```
+
+### Conda: ollama run + llama stack run
+
+If you wish to separately spin up a Ollama server, and connect with Llama Stack, you may use the following commands.
+
+#### Start Ollama server.
+- Please check the [Ollama Documentations](https://github.com/ollama/ollama) for more details.
+
+**Via Docker**
+```
+docker run -d -v ollama:/root/.ollama -p 11434:11434 --name ollama ollama/ollama
+```
+
+**Via CLI**
+```
+ollama run <model_id>
+```
+
+#### Start Llama Stack server pointing to Ollama server
+
+**Via Conda**
+
+```
+llama stack build --template ollama --image-type conda
+llama stack run ./gpu/run.yaml
+```
+
+**Via Docker**
+```
+docker run --network host -it -p 5000:5000 -v ~/.llama:/root/.llama -v ./gpu/run.yaml:/root/llamastack-run-ollama.yaml --gpus=all llamastack/distribution-ollama --yaml_config /root/llamastack-run-ollama.yaml
+```
+
+Make sure in you `run.yaml` file, you inference provider is pointing to the correct Ollama endpoint. E.g.
+```
+inference:
+  - provider_id: ollama0
+    provider_type: remote::ollama
+    config:
+      url: http://127.0.0.1:14343
+```
+
+### Model Serving
+
+#### Downloading model via Ollama
+
+You can use ollama for managing model downloads.
+
+```
+ollama pull llama3.1:8b-instruct-fp16
+ollama pull llama3.1:70b-instruct-fp16
+```
+
+> [!NOTE]
+> Please check the [OLLAMA_SUPPORTED_MODELS](https://github.com/meta-llama/llama-stack/blob/main/llama_stack/providers/adapters/inference/ollama/ollama.py) for the supported Ollama models.
+
+
+To serve a new model with `ollama`
+```
+ollama run <model_name>
+```
+
+To make sure that the model is being served correctly, run `ollama ps` to get a list of models being served by ollama.
+```
+$ ollama ps
+
+NAME                         ID              SIZE     PROCESSOR    UNTIL
+llama3.1:8b-instruct-fp16    4aacac419454    17 GB    100% GPU     4 minutes from now
+```
+
+To verify that the model served by ollama is correctly connected to Llama Stack server
+```
+$ llama-stack-client models list
++----------------------+----------------------+---------------+-----------------------------------------------+
+| identifier           | llama_model          | provider_id   | metadata                                      |
++======================+======================+===============+===============================================+
+| Llama3.1-8B-Instruct | Llama3.1-8B-Instruct | ollama0       | {'ollama_model': 'llama3.1:8b-instruct-fp16'} |
++----------------------+----------------------+---------------+-----------------------------------------------+
+```
diff --git a/docs/source/getting_started/distributions/tgi.md b/docs/source/getting_started/distributions/tgi.md
new file mode 100644
index 000000000..aeb693364
--- /dev/null
+++ b/docs/source/getting_started/distributions/tgi.md
@@ -0,0 +1,112 @@
+# TGI Distribution
+
+The `llamastack/distribution-tgi` distribution consists of the following provider configurations.
+
+
+| **API**         	| **Inference** 	| **Agents**     	| **Memory**                                       	| **Safety**     	| **Telemetry**  	|
+|-----------------	|---------------	|----------------	|--------------------------------------------------	|----------------	|----------------	|
+| **Provider(s)** 	| remote::tgi   	| meta-reference 	| meta-reference, remote::pgvector, remote::chroma 	| meta-reference 	| meta-reference 	|
+
+
+### Docker: Start the Distribution (Single Node GPU)
+
+> [!NOTE]
+> This assumes you have access to GPU to start a TGI server with access to your GPU.
+
+
+```
+$ cd distributions/tgi/gpu && docker compose up
+```
+
+The script will first start up TGI server, then start up Llama Stack distribution server hooking up to the remote TGI provider for inference. You should be able to see the following outputs --
+```
+[text-generation-inference] | 2024-10-15T18:56:33.810397Z  INFO text_generation_router::server: router/src/server.rs:1813: Using config Some(Llama)
+[text-generation-inference] | 2024-10-15T18:56:33.810448Z  WARN text_generation_router::server: router/src/server.rs:1960: Invalid hostname, defaulting to 0.0.0.0
+[text-generation-inference] | 2024-10-15T18:56:33.864143Z  INFO text_generation_router::server: router/src/server.rs:2353: Connected
+INFO:     Started server process [1]
+INFO:     Waiting for application startup.
+INFO:     Application startup complete.
+INFO:     Uvicorn running on http://[::]:5000 (Press CTRL+C to quit)
+```
+
+To kill the server
+```
+docker compose down
+```
+
+### Docker: Start the Distribution (Single Node CPU)
+
+> [!NOTE]
+> This assumes you have an hosted endpoint compatible with TGI server.
+
+```
+$ cd distributions/tgi/cpu && docker compose up
+```
+
+Replace <ENTER_YOUR_TGI_HOSTED_ENDPOINT> in `run.yaml` file with your TGI endpoint.
+```
+inference:
+  - provider_id: tgi0
+    provider_type: remote::tgi
+    config:
+      url: <ENTER_YOUR_TGI_HOSTED_ENDPOINT>
+```
+
+### Conda: TGI server + llama stack run
+
+If you wish to separately spin up a TGI server, and connect with Llama Stack, you may use the following commands.
+
+#### Start TGI server locally
+- Please check the [TGI Getting Started Guide](https://github.com/huggingface/text-generation-inference?tab=readme-ov-file#get-started) to get a TGI endpoint.
+
+```
+docker run --rm -it -v $HOME/.cache/huggingface:/data -p 5009:5009 --gpus all ghcr.io/huggingface/text-generation-inference:latest --dtype bfloat16 --usage-stats on --sharded false --model-id meta-llama/Llama-3.1-8B-Instruct --port 5009
+```
+
+#### Start Llama Stack server pointing to TGI server
+
+**Via Conda**
+
+```bash
+llama stack build --template tgi --image-type conda
+# -- start a TGI server endpoint
+llama stack run ./gpu/run.yaml
+```
+
+**Via Docker**
+```
+docker run --network host -it -p 5000:5000 -v ./run.yaml:/root/my-run.yaml --gpus=all llamastack/distribution-tgi --yaml_config /root/my-run.yaml
+```
+
+Make sure in you `run.yaml` file, you inference provider is pointing to the correct TGI server endpoint. E.g.
+```
+inference:
+  - provider_id: tgi0
+    provider_type: remote::tgi
+    config:
+      url: http://127.0.0.1:5009
+```
+
+
+### Model Serving
+To serve a new model with `tgi`, change the docker command flag `--model-id <model-to-serve>`.
+
+This can be done by edit the `command` args in `compose.yaml`. E.g. Replace "Llama-3.2-1B-Instruct" with the model you want to serve.
+
+```
+command: ["--dtype", "bfloat16", "--usage-stats", "on", "--sharded", "false", "--model-id", "meta-llama/Llama-3.2-1B-Instruct", "--port", "5009", "--cuda-memory-fraction", "0.3"]
+```
+
+or by changing the docker run command's `--model-id` flag
+```
+docker run --rm -it -v $HOME/.cache/huggingface:/data -p 5009:5009 --gpus all ghcr.io/huggingface/text-generation-inference:latest --dtype bfloat16 --usage-stats on --sharded false --model-id meta-llama/Llama-3.2-1B-Instruct --port 5009
+```
+
+In `run.yaml`, make sure you point the correct server endpoint to the TGI server endpoint serving your model.
+```
+inference:
+  - provider_id: tgi0
+    provider_type: remote::tgi
+    config:
+      url: http://127.0.0.1:5009
+```
diff --git a/docs/source/getting_started/distributions/together.md b/docs/source/getting_started/distributions/together.md
new file mode 100644
index 000000000..5f9c90071
--- /dev/null
+++ b/docs/source/getting_started/distributions/together.md
@@ -0,0 +1,65 @@
+# Together Distribution
+
+### Connect to a Llama Stack Together Endpoint
+- You may connect to a hosted endpoint `https://llama-stack.together.ai`, serving a Llama Stack distribution
+
+The `llamastack/distribution-together` distribution consists of the following provider configurations.
+
+
+| **API**         	| **Inference** 	| **Agents**     	| **Memory**                                       	| **Safety**     	| **Telemetry**  	|
+|-----------------	|---------------	|----------------	|--------------------------------------------------	|----------------	|----------------	|
+| **Provider(s)** 	| remote::together   	| meta-reference 	| meta-reference, remote::weaviate 	| meta-reference 	| meta-reference 	|
+
+
+### Docker: Start the Distribution (Single Node CPU)
+
+> [!NOTE]
+> This assumes you have an hosted endpoint at Together with API Key.
+
+```
+$ cd distributions/together
+$ ls
+compose.yaml  run.yaml
+$ docker compose up
+```
+
+Make sure in you `run.yaml` file, you inference provider is pointing to the correct Together URL server endpoint. E.g.
+```
+inference:
+  - provider_id: together
+    provider_type: remote::together
+    config:
+      url: https://api.together.xyz/v1
+      api_key: <optional api key>
+```
+
+### Conda llama stack run (Single Node CPU)
+
+```bash
+llama stack build --template together --image-type conda
+# -- modify run.yaml to a valid Together server endpoint
+llama stack run ./run.yaml
+```
+
+### Model Serving
+
+Use `llama-stack-client models list` to check the available models served by together.
+
+```
+$ llama-stack-client models list
++------------------------------+------------------------------+---------------+------------+
+| identifier                   | llama_model                  | provider_id   | metadata   |
++==============================+==============================+===============+============+
+| Llama3.1-8B-Instruct         | Llama3.1-8B-Instruct         | together0     | {}         |
++------------------------------+------------------------------+---------------+------------+
+| Llama3.1-70B-Instruct        | Llama3.1-70B-Instruct        | together0     | {}         |
++------------------------------+------------------------------+---------------+------------+
+| Llama3.1-405B-Instruct       | Llama3.1-405B-Instruct       | together0     | {}         |
++------------------------------+------------------------------+---------------+------------+
+| Llama3.2-3B-Instruct         | Llama3.2-3B-Instruct         | together0     | {}         |
++------------------------------+------------------------------+---------------+------------+
+| Llama3.2-11B-Vision-Instruct | Llama3.2-11B-Vision-Instruct | together0     | {}         |
++------------------------------+------------------------------+---------------+------------+
+| Llama3.2-90B-Vision-Instruct | Llama3.2-90B-Vision-Instruct | together0     | {}         |
++------------------------------+------------------------------+---------------+------------+
+```