add vertex ai countTokens endpoint

2025-04-25 18:54:30 +00:00 · 2024-08-03 17:34:10 -07:00 · 2024-08-03 17:34:10 -07:00 · c8438715af
commit c8438715af
parent 2d5c57e545
2 changed files with 68 additions and 0 deletions
--- a/litellm/proxy/vertex_ai_endpoints/vertex_endpoints.py
+++ b/litellm/proxy/vertex_ai_endpoints/vertex_endpoints.py
@ -182,6 +182,69 @@ async def vertex_predict_endpoint(
        raise exception_handler(e) from e


+@router.post(
+    "/vertex-ai/publishers/google/models/{model_id:path}:countTokens",
+    dependencies=[Depends(user_api_key_auth)],
+    tags=["Vertex AI endpoints"],
+)
+async def vertex_countTokens_endpoint(
+    request: Request,
+    fastapi_response: Response,
+    model_id: str,
+    user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
+):
+    """
+    this is a pass through endpoint for the Vertex AI API. /countTokens endpoint
+    https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/count-tokens#curl
+
+
+    Example Curl:
+    ```
+    curl http://localhost:4000/vertex-ai/publishers/google/models/gemini-1.5-flash-001:countTokens \
+      -H "Content-Type: application/json" \
+      -H "Authorization: Bearer sk-1234" \
+      -d '{"contents":[{"role": "user", "parts":[{"text": "hi"}]}]}'
+    ```
+
+    it uses the vertex ai credentials on the proxy and forwards to vertex ai api
+    """
+    try:
+        response = await execute_post_vertex_ai_request(
+            request=request,
+            route=f"/publishers/google/models/{model_id}:countTokens",
+        )
+        return response
+    except Exception as e:
+        raise exception_handler(e) from e
+
+
+@router.post(
+    "/vertex-ai/batchPredictionJobs",
+    dependencies=[Depends(user_api_key_auth)],
+    tags=["Vertex AI endpoints"],
+)
+async def vertex_create_batch_prediction_job(
+    request: Request,
+    fastapi_response: Response,
+    user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
+):
+    """
+    this is a pass through endpoint for the Vertex AI API. /batchPredictionJobs endpoint
+
+    Vertex API Reference: https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/batch-prediction-api#syntax
+
+    it uses the vertex ai credentials on the proxy and forwards to vertex ai api
+    """
+    try:
+        response = await execute_post_vertex_ai_request(
+            request=request,
+            route="/batchPredictionJobs",
+        )
+        return response
+    except Exception as e:
+        raise exception_handler(e) from e
+
+
@router.post(
    "/vertex-ai/tuningJobs",
    dependencies=[Depends(user_api_key_auth)],