mirror of
https://github.com/netbirdio/netbird.git
synced 2026-08-04 11:35:11 -04:00
283 lines
8.0 KiB
YAML
283 lines
8.0 KiB
YAML
# Default LLM pricing used by NetBird's Agent Network cost metering.
|
|
# GENERATED from the management catalog — do not edit this file in the
|
|
# repository; regenerate with:
|
|
#
|
|
# go generate ./management/internals/modules/agentnetwork/pricing
|
|
#
|
|
# Operators: copy this file to <datadir>/defaults_llm_pricing.yaml (or
|
|
# any path configured via management.json:
|
|
#
|
|
# { "AgentNetwork": { "PricingDefaultsFile": "/path/defaults_llm_pricing.yaml" } }
|
|
#
|
|
# ) and adjust the entries you want to change. Management re-reads the
|
|
# file periodically (mtime poll, every minute): the live table feeds the
|
|
# proxies' cost metering and the dashboard's model-price prefill, so
|
|
# edits apply without a restart. Your file only needs the entries you
|
|
# want to change — but each entry REPLACES the built-in entry for that
|
|
# surface+model whole, so repeat the cache rates you want to keep.
|
|
# Unknown fields and negative or non-finite rates are rejected: at
|
|
# startup that fails boot (for an explicitly configured path); at
|
|
# runtime the previous table is kept and a warning is logged. Deleting
|
|
# the file reverts to the built-in defaults below.
|
|
#
|
|
# Top-level keys are pricing surfaces — the parser shape requests are
|
|
# metered under: "openai" (also Azure, Mistral, and OpenAI-compatible
|
|
# gateways), "anthropic" (also Anthropic-on-Vertex), "bedrock"
|
|
# (normalized ids, e.g. anthropic.claude-sonnet-4-5). Model keys must be
|
|
# the normalized id the proxy meters (version/region suffixes stripped).
|
|
#
|
|
# Values are USD per 1_000 tokens. Optional cache fields:
|
|
# cached_input_per_1k OpenAI shape: rate for cached prompt tokens
|
|
# (a SUBSET of input tokens). Absent -> cached
|
|
# portion bills at input_per_1k.
|
|
# cache_read_per_1k Anthropic shape: rate for cache_read tokens
|
|
# (ADDITIVE to input). Absent -> input rate.
|
|
# cache_creation_per_1k Anthropic shape: rate for cache_creation
|
|
# tokens (ADDITIVE to input). Absent -> input
|
|
# rate.
|
|
|
|
anthropic:
|
|
claude-fable-5:
|
|
input_per_1k: 0.01
|
|
output_per_1k: 0.05
|
|
cache_read_per_1k: 0.001
|
|
cache_creation_per_1k: 0.0125
|
|
claude-haiku-4-5:
|
|
input_per_1k: 0.001
|
|
output_per_1k: 0.005
|
|
cache_read_per_1k: 0.0001
|
|
cache_creation_per_1k: 0.00125
|
|
claude-opus-4-1:
|
|
input_per_1k: 0.015
|
|
output_per_1k: 0.075
|
|
cache_read_per_1k: 0.0015
|
|
cache_creation_per_1k: 0.01875
|
|
claude-opus-4-6:
|
|
input_per_1k: 0.005
|
|
output_per_1k: 0.025
|
|
cache_read_per_1k: 0.0005
|
|
cache_creation_per_1k: 0.00625
|
|
claude-opus-4-7:
|
|
input_per_1k: 0.005
|
|
output_per_1k: 0.025
|
|
cache_read_per_1k: 0.0005
|
|
cache_creation_per_1k: 0.00625
|
|
claude-opus-4-8:
|
|
input_per_1k: 0.005
|
|
output_per_1k: 0.025
|
|
cache_read_per_1k: 0.0005
|
|
cache_creation_per_1k: 0.00625
|
|
claude-opus-5:
|
|
input_per_1k: 0.005
|
|
output_per_1k: 0.025
|
|
cache_read_per_1k: 0.0005
|
|
cache_creation_per_1k: 0.00625
|
|
claude-sonnet-4-5:
|
|
input_per_1k: 0.003
|
|
output_per_1k: 0.015
|
|
cache_read_per_1k: 0.0003
|
|
cache_creation_per_1k: 0.00375
|
|
claude-sonnet-4-6:
|
|
input_per_1k: 0.003
|
|
output_per_1k: 0.015
|
|
cache_read_per_1k: 0.0003
|
|
cache_creation_per_1k: 0.00375
|
|
kimi-k3:
|
|
input_per_1k: 0.003
|
|
output_per_1k: 0.015
|
|
cached_input_per_1k: 0.0003
|
|
cache_read_per_1k: 0.0003
|
|
"kimi-k3[1m]":
|
|
input_per_1k: 0.003
|
|
output_per_1k: 0.015
|
|
cache_read_per_1k: 0.0003
|
|
|
|
bedrock:
|
|
amazon.nova-2-lite:
|
|
input_per_1k: 0.0003
|
|
output_per_1k: 0.0025
|
|
amazon.nova-lite:
|
|
input_per_1k: 0.00006
|
|
output_per_1k: 0.00024
|
|
amazon.nova-micro:
|
|
input_per_1k: 0.000035
|
|
output_per_1k: 0.00014
|
|
amazon.nova-pro:
|
|
input_per_1k: 0.0008
|
|
output_per_1k: 0.0032
|
|
anthropic.claude-haiku-4-5:
|
|
input_per_1k: 0.001
|
|
output_per_1k: 0.005
|
|
cache_read_per_1k: 0.0001
|
|
cache_creation_per_1k: 0.00125
|
|
anthropic.claude-opus-4-1:
|
|
input_per_1k: 0.015
|
|
output_per_1k: 0.075
|
|
cache_read_per_1k: 0.0015
|
|
cache_creation_per_1k: 0.01875
|
|
anthropic.claude-opus-4-6:
|
|
input_per_1k: 0.005
|
|
output_per_1k: 0.025
|
|
cache_read_per_1k: 0.0005
|
|
cache_creation_per_1k: 0.00625
|
|
anthropic.claude-opus-4-7:
|
|
input_per_1k: 0.005
|
|
output_per_1k: 0.025
|
|
cache_read_per_1k: 0.0005
|
|
cache_creation_per_1k: 0.00625
|
|
anthropic.claude-opus-4-8:
|
|
input_per_1k: 0.005
|
|
output_per_1k: 0.025
|
|
cache_read_per_1k: 0.0005
|
|
cache_creation_per_1k: 0.00625
|
|
anthropic.claude-opus-5:
|
|
input_per_1k: 0.005
|
|
output_per_1k: 0.025
|
|
cache_read_per_1k: 0.0005
|
|
cache_creation_per_1k: 0.00625
|
|
anthropic.claude-sonnet-4-5:
|
|
input_per_1k: 0.003
|
|
output_per_1k: 0.015
|
|
cache_read_per_1k: 0.0003
|
|
cache_creation_per_1k: 0.00375
|
|
anthropic.claude-sonnet-4-6:
|
|
input_per_1k: 0.003
|
|
output_per_1k: 0.015
|
|
cache_read_per_1k: 0.0003
|
|
cache_creation_per_1k: 0.00375
|
|
meta.llama3-3-70b-instruct:
|
|
input_per_1k: 0.00072
|
|
output_per_1k: 0.00072
|
|
|
|
openai:
|
|
codestral-2508:
|
|
input_per_1k: 0.0003
|
|
output_per_1k: 0.0009
|
|
codestral-latest:
|
|
input_per_1k: 0.001
|
|
output_per_1k: 0.003
|
|
devstral-medium-latest:
|
|
input_per_1k: 0.0004
|
|
output_per_1k: 0.002
|
|
devstral-small-latest:
|
|
input_per_1k: 0.0001
|
|
output_per_1k: 0.0003
|
|
gpt-3.5-turbo:
|
|
input_per_1k: 0.0005
|
|
output_per_1k: 0.0015
|
|
gpt-35-turbo:
|
|
input_per_1k: 0.0005
|
|
output_per_1k: 0.0015
|
|
gpt-4-turbo:
|
|
input_per_1k: 0.01
|
|
output_per_1k: 0.03
|
|
gpt-4.1:
|
|
input_per_1k: 0.002
|
|
output_per_1k: 0.008
|
|
cached_input_per_1k: 0.0005
|
|
gpt-4.1-mini:
|
|
input_per_1k: 0.0004
|
|
output_per_1k: 0.0016
|
|
cached_input_per_1k: 0.0001
|
|
gpt-4.1-nano:
|
|
input_per_1k: 0.0001
|
|
output_per_1k: 0.0004
|
|
cached_input_per_1k: 0.000025
|
|
gpt-4o:
|
|
input_per_1k: 0.0025
|
|
output_per_1k: 0.01
|
|
cached_input_per_1k: 0.00125
|
|
gpt-4o-mini:
|
|
input_per_1k: 0.00015
|
|
output_per_1k: 0.0006
|
|
cached_input_per_1k: 0.000075
|
|
gpt-5:
|
|
input_per_1k: 0.00125
|
|
output_per_1k: 0.01
|
|
cached_input_per_1k: 0.000125
|
|
gpt-5-mini:
|
|
input_per_1k: 0.00025
|
|
output_per_1k: 0.002
|
|
cached_input_per_1k: 0.000025
|
|
gpt-5-nano:
|
|
input_per_1k: 0.00005
|
|
output_per_1k: 0.0004
|
|
cached_input_per_1k: 0.000005
|
|
gpt-5.3-chat-latest:
|
|
input_per_1k: 0.00175
|
|
output_per_1k: 0.014
|
|
cached_input_per_1k: 0.000175
|
|
gpt-5.3-codex:
|
|
input_per_1k: 0.00175
|
|
output_per_1k: 0.014
|
|
cached_input_per_1k: 0.000175
|
|
gpt-5.4:
|
|
input_per_1k: 0.0025
|
|
output_per_1k: 0.015
|
|
cached_input_per_1k: 0.00025
|
|
gpt-5.4-mini:
|
|
input_per_1k: 0.00075
|
|
output_per_1k: 0.0045
|
|
cached_input_per_1k: 0.000075
|
|
gpt-5.4-nano:
|
|
input_per_1k: 0.0002
|
|
output_per_1k: 0.00125
|
|
cached_input_per_1k: 0.00002
|
|
gpt-5.4-pro:
|
|
input_per_1k: 0.03
|
|
output_per_1k: 0.18
|
|
cached_input_per_1k: 0.003
|
|
gpt-5.5:
|
|
input_per_1k: 0.005
|
|
output_per_1k: 0.03
|
|
cached_input_per_1k: 0.0005
|
|
gpt-5.5-pro:
|
|
input_per_1k: 0.03
|
|
output_per_1k: 0.18
|
|
cached_input_per_1k: 0.003
|
|
kimi-k3:
|
|
input_per_1k: 0.003
|
|
output_per_1k: 0.015
|
|
cached_input_per_1k: 0.0003
|
|
cache_read_per_1k: 0.0003
|
|
magistral-medium-latest:
|
|
input_per_1k: 0.002
|
|
output_per_1k: 0.005
|
|
magistral-small-latest:
|
|
input_per_1k: 0.0005
|
|
output_per_1k: 0.0015
|
|
ministral-3-14b-2512:
|
|
input_per_1k: 0.0002
|
|
output_per_1k: 0.0002
|
|
ministral-3-3b-2512:
|
|
input_per_1k: 0.0001
|
|
output_per_1k: 0.0001
|
|
ministral-8b-latest:
|
|
input_per_1k: 0.00015
|
|
output_per_1k: 0.00015
|
|
mistral-embed:
|
|
input_per_1k: 0.0001
|
|
output_per_1k: 0
|
|
mistral-large-latest:
|
|
input_per_1k: 0.0005
|
|
output_per_1k: 0.0015
|
|
mistral-medium-3-5:
|
|
input_per_1k: 0.0015
|
|
output_per_1k: 0.0075
|
|
mistral-medium-latest:
|
|
input_per_1k: 0.0004
|
|
output_per_1k: 0.002
|
|
mistral-small-latest:
|
|
input_per_1k: 0.00006
|
|
output_per_1k: 0.00018
|
|
o4-mini:
|
|
input_per_1k: 0.0011
|
|
output_per_1k: 0.0044
|
|
cached_input_per_1k: 0.000275
|
|
text-embedding-3-large:
|
|
input_per_1k: 0.00013
|
|
output_per_1k: 0
|
|
text-embedding-3-small:
|
|
input_per_1k: 0.00002
|
|
output_per_1k: 0
|