Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
18 changes: 18 additions & 0 deletions providers/google-vertex/anthropic/claude-haiku-4-5@20251001.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,14 @@ costs:
output_cost_per_token: 0.0000055
output_cost_per_token_batches: 0.00000275
region: us-east5
- cache_creation_input_token_cost: 0.000001375
cache_creation_input_token_cost_per_hour: 0.0000022
cache_read_input_token_cost: 1.1e-7
input_cost_per_token: 0.0000011
input_cost_per_token_batches: 5.5e-7
output_cost_per_token: 0.0000055
output_cost_per_token_batches: 0.00000275
region: us
- cache_creation_input_token_cost: 0.00000125
cache_creation_input_token_cost_per_hour: 0.000002
cache_read_input_token_cost: 1e-7
Expand All @@ -23,6 +31,14 @@ costs:
output_cost_per_token: 0.0000055
output_cost_per_token_batches: 0.00000275
region: europe-west1
- cache_creation_input_token_cost: 0.000001375
cache_creation_input_token_cost_per_hour: 0.0000022
cache_read_input_token_cost: 1.1e-7
input_cost_per_token: 0.0000011
input_cost_per_token_batches: 5.5e-7
output_cost_per_token: 0.0000055
output_cost_per_token_batches: 0.00000275
region: eu
- cache_creation_input_token_cost: 0.000001375
cache_creation_input_token_cost_per_hour: 0.0000022
cache_read_input_token_cost: 1.1e-7
Expand Down Expand Up @@ -60,6 +76,8 @@ removeParams:
sources:
- https://docs.cloud.google.com/vertex-ai/generative-ai/docs/partner-models/claude/haiku-4-5
- https://platform.claude.com/docs/en/about-claude/pricing
- https://platform.claude.com/docs/en/about-claude/models/overview
- https://platform.claude.com/docs/en/build-with-claude/claude-on-vertex-ai
status: active
supportedModes:
- chat
Expand Down
10 changes: 10 additions & 0 deletions providers/google-vertex/anthropic/claude-opus-4-6.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -3,13 +3,17 @@ costs:
cache_creation_input_token_cost_per_hour: 0.000011
cache_read_input_token_cost: 5.5e-7
input_cost_per_token: 0.0000055
input_cost_per_token_batches: 0.00000275
output_cost_per_token: 0.0000275
output_cost_per_token_batches: 0.00001375
region: us-east5
- cache_creation_input_token_cost: 0.000006875
cache_creation_input_token_cost_per_hour: 0.000011
cache_read_input_token_cost: 5.5e-7
input_cost_per_token: 0.0000055
input_cost_per_token_batches: 0.00000275
output_cost_per_token: 0.0000275
output_cost_per_token_batches: 0.00001375
region: us
- cache_creation_input_token_cost: 0.00000625
cache_creation_input_token_cost_per_hour: 0.00001
Expand All @@ -23,19 +27,25 @@ costs:
cache_creation_input_token_cost_per_hour: 0.000011
cache_read_input_token_cost: 5.5e-7
input_cost_per_token: 0.0000055
input_cost_per_token_batches: 0.00000275
output_cost_per_token: 0.0000275
output_cost_per_token_batches: 0.00001375
region: europe-west1
- cache_creation_input_token_cost: 0.000006875
cache_creation_input_token_cost_per_hour: 0.000011
cache_read_input_token_cost: 5.5e-7
input_cost_per_token: 0.0000055
input_cost_per_token_batches: 0.00000275
output_cost_per_token: 0.0000275
output_cost_per_token_batches: 0.00001375
region: eu
- cache_creation_input_token_cost: 0.000006875
cache_creation_input_token_cost_per_hour: 0.000011
cache_read_input_token_cost: 5.5e-7
input_cost_per_token: 0.0000055
input_cost_per_token_batches: 0.00000275
output_cost_per_token: 0.0000275
output_cost_per_token_batches: 0.00001375
region: asia-southeast1
features:
- function_calling
Expand Down
1 change: 1 addition & 0 deletions providers/google-vertex/anthropic/claude-sonnet-4-5.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -156,6 +156,7 @@ sources:
- https://platform.claude.com/docs/en/about-claude/pricing
- https://platform.claude.com/docs/en/build-with-claude/claude-on-vertex-ai
- https://docs.cloud.google.com/vertex-ai/generative-ai/docs/partner-models/use-claude
- https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/partner-models/claude/sonnet-4-5
- https://cloud.google.com/blog/products/ai-machine-learning/multi-region-endpoints-for-claude-available-on-vertex-ai
status: active
supportedModes:
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -11,5 +11,6 @@ sources:
- https://huggingface.co/dandelin/vilt-b32-finetuned-vqa
- https://console.cloud.google.com/vertex-ai/publishers/dandelin/model-garden/vilt-b32-finetuned-vqa
- https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_vilt_vqa.ipynb
status: active
supportedModes:
- unknown
9 changes: 6 additions & 3 deletions providers/google-vertex/deepseek-ai/deepseek-ocr-maas.yaml
Original file line number Diff line number Diff line change
@@ -1,8 +1,10 @@
costs:
- input_cost_per_page: 0.0003
input_cost_per_token: 3e-7
output_cost_per_token: 0.0000012
region: global # docs specify us-central1 but the model is only available via global inference
output_cost_per_token: 1.2e-6
region: global
deprecationDate: "2026-07-21"
isDeprecated: true
limits:
context_window: 8192
max_output_tokens: 8192
Expand All @@ -22,9 +24,10 @@ params:
maxValue: 8192
minValue: 1
provisioning: serverless
retirementDate: "2026-10-21"
sources:
- https://docs.cloud.google.com/vertex-ai/generative-ai/docs/maas/deepseek/deepseek-ocr
- https://docs.cloud.google.com/vertex-ai/generative-ai/docs/maas/deepseek
status: active
status: deprecated
supportedModes:
- ocr
11 changes: 7 additions & 4 deletions providers/google-vertex/deepseek-ai/deepseek-r1-0528-maas.yaml
Original file line number Diff line number Diff line change
@@ -1,14 +1,16 @@
costs:
- input_cost_per_token: 0.00000135
- input_cost_per_token: 1.35e-6
input_cost_per_token_batches: 6.75e-7
output_cost_per_token: 0.0000054
output_cost_per_token_batches: 0.0000027
output_cost_per_token: 5.4e-6
output_cost_per_token_batches: 2.7e-6
region: us-central1
deprecationDate: "2026-07-21"
features:
- function_calling
- structured_output
- system_messages
- tool_choice
isDeprecated: true
limits:
context_window: 163840
max_output_tokens: 32768
Expand All @@ -22,10 +24,11 @@ modalities:
mode: chat
model: deepseek-ai/deepseek-r1-0528-maas
provisioning: serverless
retirementDate: "2026-10-21"
sources:
- https://docs.cloud.google.com/vertex-ai/generative-ai/docs/maas/deepseek
- https://docs.cloud.google.com/vertex-ai/generative-ai/docs/maas/deepseek/r1-0528
status: active
status: deprecated
supportedModes:
- chat
thinking: true
7 changes: 5 additions & 2 deletions providers/google-vertex/deepseek-ai/deepseek-v3-1.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -2,13 +2,15 @@ costs:
- cache_read_input_token_cost: 6e-8
input_cost_per_token: 6e-7
input_cost_per_token_batches: 3e-7
output_cost_per_token: 0.0000017
output_cost_per_token: 1.7e-6
output_cost_per_token_batches: 8.5e-7
region: us-central1
deprecationDate: "2026-07-21"
features:
- function_calling
- structured_output
- prompt_caching
isDeprecated: true
limits:
context_window: 163840
max_output_tokens: 32768
Expand All @@ -21,10 +23,11 @@ modalities:
mode: chat
model: deepseek-ai/deepseek-v3-1
provisioning: provisioned
retirementDate: "2026-10-21"
sources:
- https://docs.cloud.google.com/vertex-ai/generative-ai/docs/maas/deepseek/deepseek-v31
- https://cloud.google.com/vertex-ai/generative-ai/pricing#deepseek-models
status: active
status: deprecated
supportedModes:
- chat
thinking: true
7 changes: 5 additions & 2 deletions providers/google-vertex/deepseek-ai/deepseek-v3-2.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -2,13 +2,15 @@ costs:
- cache_read_input_token_cost: 5.6e-8
input_cost_per_token: 5.6e-7
input_cost_per_token_batches: 2.8e-7
output_cost_per_token: 0.00000168
output_cost_per_token: 1.68e-6
output_cost_per_token_batches: 8.4e-7
region: global
deprecationDate: "2026-07-21"
features:
- function_calling
- structured_output
- prompt_caching
isDeprecated: true
limits:
context_window: 163840
max_output_tokens: 65536
Expand All @@ -26,9 +28,10 @@ params:
maxValue: 65536
minValue: 1
provisioning: provisioned
retirementDate: "2026-10-21"
sources:
- https://docs.cloud.google.com/vertex-ai/generative-ai/docs/maas/deepseek/deepseek-v32
- https://docs.cloud.google.com/vertex-ai/generative-ai/docs/maas/deepseek
status: active
status: deprecated
supportedModes:
- chat
9 changes: 6 additions & 3 deletions providers/google-vertex/deepseek-ai/deepseek-v3.1-maas.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -2,15 +2,17 @@ costs:
- cache_read_input_token_cost: 6e-8
input_cost_per_token: 6e-7
input_cost_per_token_batches: 3e-7
output_cost_per_token: 0.0000017
output_cost_per_token: 1.7e-6
output_cost_per_token_batches: 8.5e-7
region: us-west2 # docs specify us-central1 but the model is available via us-west2 inference
region: us-west2
deprecationDate: "2026-07-21"
features:
- function_calling
- tool_choice
- structured_output
- system_messages
- prompt_caching
isDeprecated: true
limits:
context_window: 163840
max_output_tokens: 32768
Expand All @@ -24,11 +26,12 @@ modalities:
mode: chat
model: deepseek-ai/deepseek-v3.1-maas
provisioning: serverless
retirementDate: "2026-10-21"
sources:
- https://docs.cloud.google.com/vertex-ai/generative-ai/docs/maas/deepseek/deepseek-v31
- https://docs.cloud.google.com/vertex-ai/generative-ai/docs/maas/deepseek
- https://docs.cloud.google.com/vertex-ai/generative-ai/docs/maas/use-open-models
status: active
status: deprecated
supportedModes:
- chat
thinking: true
12 changes: 11 additions & 1 deletion providers/google-vertex/gemini-3-pro-image-preview.yaml
Original file line number Diff line number Diff line change
@@ -1,12 +1,13 @@
costs:
- input_cost_per_image: 0.00112
input_cost_per_token: 0.000002
input_cost_per_token_batches: 0.000001
output_cost_per_image_token: 0.00012
output_cost_per_token: 0.000012
output_cost_per_token_batches: 0.000006
region: global
features:
- system_messages
- structured_output
limits:
context_window: 65536
max_input_tokens: 65536
Expand All @@ -22,6 +23,15 @@ modalities:
- image
mode: image
model: gemini-3-pro-image-preview
params:
- defaultValue: 1.0
key: temperature
maxValue: 2
minValue: 0
- defaultValue: 0.95
key: top_p
maxValue: 1
minValue: 0
provisioning: serverless
removeParams:
- tool_choice
Expand Down
1 change: 1 addition & 0 deletions providers/google-vertex/gemma3.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@ limits:
context_window: 128000
max_input_tokens: 128000
max_output_tokens: 128000
max_tokens: 128000
modalities:
input:
- text
Expand Down
7 changes: 4 additions & 3 deletions providers/google-vertex/google/gemini-2.5-flash-lite.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -129,8 +129,8 @@ features:
limits:
context_window: 1048576
max_input_tokens: 1048576
max_output_tokens: 65535
max_tokens: 65535
max_output_tokens: 65536
max_tokens: 65536
modalities:
input:
- text
Expand All @@ -149,10 +149,11 @@ params:
key: top_p
- defaultValue: 256
key: max_tokens
maxValue: 65535
maxValue: 65536
provisioning: serverless
sources:
- https://docs.cloud.google.com/vertex-ai/generative-ai/docs/models/gemini/2-5-flash-lite
- https://ai.google.dev/gemini-api/docs/models/gemini-2.5-flash-lite
status: active
supportedModes:
- chat
Expand Down
2 changes: 1 addition & 1 deletion providers/google-vertex/google/gemini-2.5-pro-tts.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -91,6 +91,6 @@ sources:
- https://ai.google.dev/gemini-api/docs/models
- https://ai.google.dev/gemini-api/docs/pricing
- https://docs.cloud.google.com/text-to-speech/docs/gemini-tts
status: active
status: preview
Comment thread
architkumar-truefoundry marked this conversation as resolved.
supportedModes:
- text_to_speech
4 changes: 3 additions & 1 deletion providers/google-vertex/google/gemini-3.1-pro-preview.yaml
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
costs:
- cache_read_input_token_cost: 2e-7
- cache_creation_input_token_cost_per_hour: 0.0000045
cache_read_input_token_cost: 2e-7
input_cost_per_token: 0.000002
input_cost_per_token_batches: 0.000001
output_cost_per_token: 0.000012
Expand Down Expand Up @@ -46,6 +47,7 @@ params:
provisioning: serverless
sources:
- https://docs.cloud.google.com/vertex-ai/generative-ai/docs/models/gemini/3-1-pro
- https://ai.google.dev/gemini-api/docs/models/gemini-3.1-pro-preview
status: preview
supportedModes:
- chat
Expand Down
4 changes: 4 additions & 0 deletions providers/google-vertex/google/gemma3n.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -3,18 +3,22 @@ features:
limits:
context_window: 32000
max_input_tokens: 32000
max_output_tokens: 32000
max_tokens: 32000
modalities:
input:
- text
- image
- audio
- video
output:
- text
mode: chat
model: google/gemma3n
provisioning: provisioned
sources:
- https://ai.google.dev/gemma/docs/gemma-3n
- https://ai.google.dev/gemma/docs/gemma-3n/model_card
- https://docs.cloud.google.com/vertex-ai/generative-ai/docs/open-models/use-gemma
status: active
supportedModes:
Expand Down
2 changes: 2 additions & 0 deletions providers/google-vertex/google/gemma4.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,8 @@ modalities:
input:
- text
- image
- audio
- video
- pdf
output:
- text
Expand Down
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
costs:
- input_cost_per_character: 0.000002
- input_cost_per_character: 2e-6
region: "*"
modalities:
input:
Expand All @@ -13,7 +13,7 @@ removeParams:
- max_tokens
- temperature
- top_p
- "n"
- n
Comment thread
architkumar-truefoundry marked this conversation as resolved.
- stop
- stream
- tool_choice
Expand Down
1 change: 1 addition & 0 deletions providers/google-vertex/google/lyria-3-pro-preview.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@ modalities:
- image
output:
- audio
- text
mode: unknown # correct mode is music, but that is not supported yet
model: lyria-3-pro-preview
provisioning: serverless
Expand Down
Loading
Loading