forked from phoenix-oss/llama-stack-mirror
		
	fix nvidia inference provider (#781)
# What does this PR do? - fixes to nvidia inference provider to account for strategy update - update nvidia templates ## Test Plan ``` llama stack run ./llama_stack/templates/nvidia/run.yaml --port 5000 LLAMA_STACK_BASE_URL="http://localhost:5000" pytest -v tests/client-sdk/inference/test_inference.py --html=report.html --self-contained-html ``` <img width="1288" alt="image" src="https://github.com/user-attachments/assets/d20f9aea-525e-47de-a5be-586e022e0d55" /> **NOTE** - vision inference broken - tool calling broken - /completion broken cc @mattf @cdgamarose-nv for improving NVIDIA inference adapter ## Sources Please link relevant resources if necessary. ## Before submitting - [ ] This PR fixes a typo or improves the docs (you can dismiss the other checks if that's the case). - [ ] Ran pre-commit to handle lint / formatting issues. - [ ] Read the [contributor guideline](https://github.com/meta-llama/llama-stack/blob/main/CONTRIBUTING.md), Pull Request section? - [ ] Updated relevant documentation. - [ ] Wrote necessary unit or integration tests.
This commit is contained in:
		
							parent
							
								
									965644ce68
								
							
						
					
					
						commit
						b76bef169c
					
				
					 5 changed files with 351 additions and 262 deletions
				
			
		|  | @ -89,8 +89,49 @@ metadata_store: | |||
|   db_path: ${env.SQLITE_STORE_DIR:~/.llama/distributions/nvidia}/registry.db | ||||
| models: | ||||
| - metadata: {} | ||||
|   model_id: ${env.INFERENCE_MODEL} | ||||
|   model_id: meta-llama/Llama-3-8B-Instruct | ||||
|   provider_id: nvidia | ||||
|   provider_model_id: meta/llama3-8b-instruct | ||||
|   model_type: llm | ||||
| - metadata: {} | ||||
|   model_id: meta-llama/Llama-3-70B-Instruct | ||||
|   provider_id: nvidia | ||||
|   provider_model_id: meta/llama3-70b-instruct | ||||
|   model_type: llm | ||||
| - metadata: {} | ||||
|   model_id: meta-llama/Llama-3.1-8B-Instruct | ||||
|   provider_id: nvidia | ||||
|   provider_model_id: meta/llama-3.1-8b-instruct | ||||
|   model_type: llm | ||||
| - metadata: {} | ||||
|   model_id: meta-llama/Llama-3.1-70B-Instruct | ||||
|   provider_id: nvidia | ||||
|   provider_model_id: meta/llama-3.1-70b-instruct | ||||
|   model_type: llm | ||||
| - metadata: {} | ||||
|   model_id: meta-llama/Llama-3.1-405B-Instruct-FP8 | ||||
|   provider_id: nvidia | ||||
|   provider_model_id: meta/llama-3.1-405b-instruct | ||||
|   model_type: llm | ||||
| - metadata: {} | ||||
|   model_id: meta-llama/Llama-3.2-1B-Instruct | ||||
|   provider_id: nvidia | ||||
|   provider_model_id: meta/llama-3.2-1b-instruct | ||||
|   model_type: llm | ||||
| - metadata: {} | ||||
|   model_id: meta-llama/Llama-3.2-3B-Instruct | ||||
|   provider_id: nvidia | ||||
|   provider_model_id: meta/llama-3.2-3b-instruct | ||||
|   model_type: llm | ||||
| - metadata: {} | ||||
|   model_id: meta-llama/Llama-3.2-11B-Vision-Instruct | ||||
|   provider_id: nvidia | ||||
|   provider_model_id: meta/llama-3.2-11b-vision-instruct | ||||
|   model_type: llm | ||||
| - metadata: {} | ||||
|   model_id: meta-llama/Llama-3.2-90B-Vision-Instruct | ||||
|   provider_id: nvidia | ||||
|   provider_model_id: meta/llama-3.2-90b-vision-instruct | ||||
|   model_type: llm | ||||
| shields: [] | ||||
| memory_banks: [] | ||||
|  |  | |||
		Loading…
	
	Add table
		Add a link
		
	
		Reference in a new issue