From fe5c0cb6c3284eff56773b716c765208fb2fcf41 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Thu, 16 Jul 2026 00:58:48 -0700 Subject: [PATCH] feat(pricing): refresh Fireworks snapshot to 2026-07, cover full serverless catalog + cached picker pricing MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Refresh _OFFICIAL_DOCS_PRICING fireworks entries against current docs.fireworks.ai/serverless/pricing: qwen3p6-plus is gone (replaced by qwen3p7-plus); add glm-5p2/5p1, kimi-k2p7-code, deepseek-v4-flash, minimax-m3/m2p7, gpt-oss-120b/20b, and the routers/*-fast tiers with their distinct higher rates. - Picker pricing via get_pricing_for_provider('fireworks'): pure dict transform over the shared models.dev in-memory/disk cache (1h TTL) + _pricing_cache memoization — no new network call on the picker path. - Wire pricing display into the generic api-key-provider setup flow so Fireworks model pickers show $/M columns like OpenRouter/Nous do. - Invariant tests: plugin fallback_models all priced, fast tiers price higher than standard, every row carries cache_read < input. --- agent/usage_pricing.py | 162 ++++++++++++++++++++++++++++-- hermes_cli/model_setup_flows.py | 13 +++ hermes_cli/models.py | 51 ++++++++++ tests/agent/test_usage_pricing.py | 61 ++++++++++- 4 files changed, 274 insertions(+), 13 deletions(-) diff --git a/agent/usage_pricing.py b/agent/usage_pricing.py index 663b20477c22..99af6343de21 100644 --- a/agent/usage_pricing.py +++ b/agent/usage_pricing.py @@ -613,6 +613,7 @@ _OFFICIAL_DOCS_PRICING: Dict[tuple[str, str], PricingEntry] = { # through when configured with provider="fireworks". Fireworks publishes a # cached_input rate per model alongside input/output, which maps to # cache_read_cost_per_million. No separately published cache_write rate. + # Snapshot of https://docs.fireworks.ai/serverless/pricing (Standard tier). ( "fireworks", "kimi-k2p6", @@ -622,7 +623,29 @@ _OFFICIAL_DOCS_PRICING: Dict[tuple[str, str], PricingEntry] = { cache_read_cost_per_million=Decimal("0.16"), source="official_docs_snapshot", source_url="https://docs.fireworks.ai/serverless/pricing", - pricing_version="fireworks-pricing-2026-05", + pricing_version="fireworks-pricing-2026-07", + ), + ( + "fireworks", + "kimi-k2p7-code", + ): PricingEntry( + input_cost_per_million=Decimal("0.95"), + output_cost_per_million=Decimal("4.00"), + cache_read_cost_per_million=Decimal("0.19"), + source="official_docs_snapshot", + source_url="https://docs.fireworks.ai/serverless/pricing", + pricing_version="fireworks-pricing-2026-07", + ), + ( + "fireworks", + "glm-5p2", + ): PricingEntry( + input_cost_per_million=Decimal("1.40"), + output_cost_per_million=Decimal("4.40"), + cache_read_cost_per_million=Decimal("0.14"), + source="official_docs_snapshot", + source_url="https://docs.fireworks.ai/serverless/pricing", + pricing_version="fireworks-pricing-2026-07", ), ( "fireworks", @@ -633,18 +656,141 @@ _OFFICIAL_DOCS_PRICING: Dict[tuple[str, str], PricingEntry] = { cache_read_cost_per_million=Decimal("0.145"), source="official_docs_snapshot", source_url="https://docs.fireworks.ai/serverless/pricing", - pricing_version="fireworks-pricing-2026-05", + pricing_version="fireworks-pricing-2026-07", ), ( "fireworks", - "qwen3p6-plus", + "deepseek-v4-flash", ): PricingEntry( - input_cost_per_million=Decimal("0.50"), - output_cost_per_million=Decimal("3.00"), - cache_read_cost_per_million=Decimal("0.10"), + input_cost_per_million=Decimal("0.14"), + output_cost_per_million=Decimal("0.28"), + cache_read_cost_per_million=Decimal("0.028"), source="official_docs_snapshot", - source_url="https://fireworks.ai/models/fireworks/qwen3p6-plus", - pricing_version="fireworks-pricing-2026-05", + source_url="https://docs.fireworks.ai/serverless/pricing", + pricing_version="fireworks-pricing-2026-07", + ), + ( + "fireworks", + "qwen3p7-plus", + ): PricingEntry( + input_cost_per_million=Decimal("0.40"), + output_cost_per_million=Decimal("1.60"), + cache_read_cost_per_million=Decimal("0.08"), + source="official_docs_snapshot", + source_url="https://docs.fireworks.ai/serverless/pricing", + pricing_version="fireworks-pricing-2026-07", + ), + ( + "fireworks", + "minimax-m3", + ): PricingEntry( + input_cost_per_million=Decimal("0.30"), + output_cost_per_million=Decimal("1.20"), + cache_read_cost_per_million=Decimal("0.06"), + source="official_docs_snapshot", + source_url="https://docs.fireworks.ai/serverless/pricing", + pricing_version="fireworks-pricing-2026-07", + ), + ( + "fireworks", + "gpt-oss-120b", + ): PricingEntry( + input_cost_per_million=Decimal("0.15"), + output_cost_per_million=Decimal("0.60"), + cache_read_cost_per_million=Decimal("0.015"), + source="official_docs_snapshot", + source_url="https://docs.fireworks.ai/serverless/pricing", + pricing_version="fireworks-pricing-2026-07", + ), + ( + "fireworks", + "gpt-oss-20b", + ): PricingEntry( + input_cost_per_million=Decimal("0.07"), + output_cost_per_million=Decimal("0.30"), + cache_read_cost_per_million=Decimal("0.035"), + source="official_docs_snapshot", + source_url="https://docs.fireworks.ai/serverless/pricing", + pricing_version="fireworks-pricing-2026-07", + ), + ( + "fireworks", + "glm-5p1", + ): PricingEntry( + input_cost_per_million=Decimal("1.40"), + output_cost_per_million=Decimal("4.40"), + cache_read_cost_per_million=Decimal("0.26"), + source="official_docs_snapshot", + source_url="https://docs.fireworks.ai/serverless/pricing", + pricing_version="fireworks-pricing-2026-07", + ), + ( + "fireworks", + "minimax-m2p7", + ): PricingEntry( + input_cost_per_million=Decimal("0.30"), + output_cost_per_million=Decimal("1.20"), + cache_read_cost_per_million=Decimal("0.06"), + source="official_docs_snapshot", + source_url="https://docs.fireworks.ai/serverless/pricing", + pricing_version="fireworks-pricing-2026-07", + ), + # Fast/turbo serving tiers — exposed as accounts/fireworks/routers/, + # so rsplit("/", 1) yields these distinct ids with their own (higher) rates. + ( + "fireworks", + "kimi-k2p6-fast", + ): PricingEntry( + input_cost_per_million=Decimal("2.00"), + output_cost_per_million=Decimal("8.00"), + cache_read_cost_per_million=Decimal("0.30"), + source="official_docs_snapshot", + source_url="https://docs.fireworks.ai/serverless/pricing", + pricing_version="fireworks-pricing-2026-07", + ), + ( + "fireworks", + "kimi-k2p6-turbo", + ): PricingEntry( + input_cost_per_million=Decimal("2.00"), + output_cost_per_million=Decimal("8.00"), + cache_read_cost_per_million=Decimal("0.30"), + source="official_docs_snapshot", + source_url="https://docs.fireworks.ai/serverless/pricing", + pricing_version="fireworks-pricing-2026-07", + ), + ( + "fireworks", + "kimi-k2p7-code-fast", + ): PricingEntry( + input_cost_per_million=Decimal("1.90"), + output_cost_per_million=Decimal("8.00"), + cache_read_cost_per_million=Decimal("0.38"), + source="official_docs_snapshot", + source_url="https://docs.fireworks.ai/serverless/pricing", + pricing_version="fireworks-pricing-2026-07", + ), + ( + "fireworks", + "glm-5p2-fast", + ): PricingEntry( + input_cost_per_million=Decimal("2.10"), + output_cost_per_million=Decimal("6.60"), + cache_read_cost_per_million=Decimal("0.21"), + source="official_docs_snapshot", + source_url="https://docs.fireworks.ai/serverless/pricing", + pricing_version="fireworks-pricing-2026-07", + ), + ( + "fireworks", + "glm-5p1-fast", + ): PricingEntry( + input_cost_per_million=Decimal("2.80"), + output_cost_per_million=Decimal("8.80"), + cache_read_cost_per_million=Decimal("0.52"), + source="official_docs_snapshot", + source_url="https://docs.fireworks.ai/serverless/pricing", + pricing_version="fireworks-pricing-2026-07", ), } diff --git a/hermes_cli/model_setup_flows.py b/hermes_cli/model_setup_flows.py index fd7029f56dfe..394360c16073 100644 --- a/hermes_cli/model_setup_flows.py +++ b/hermes_cli/model_setup_flows.py @@ -2857,9 +2857,22 @@ def _model_flow_api_key_provider(config, provider_id, current_model=""): model_list = list(dict.fromkeys(mid for mid in model_list if mid)) if model_list: + # Per-model pricing, when the provider supports it (fireworks via the + # models.dev disk cache, novita/deepinfra via their cached /models + # endpoints). get_pricing_for_provider() is memoized in-process and + # returns {} for providers without pricing — never a blocking fetch + # beyond the catalog lookup that already happened above. + pricing: dict = {} + try: + from hermes_cli.models import get_pricing_for_provider + + pricing = get_pricing_for_provider(provider_id) or {} + except Exception: + pricing = {} selected = _prompt_model_selection( model_list, current_model=current_model, + pricing=pricing, confirm_provider=provider_id, confirm_base_url=effective_base, confirm_api_key=existing_key, diff --git a/hermes_cli/models.py b/hermes_cli/models.py index 37f264c9f878..47552624cfe1 100644 --- a/hermes_cli/models.py +++ b/hermes_cli/models.py @@ -1654,6 +1654,8 @@ def get_pricing_for_provider(provider: str, *, force_refresh: bool = False) -> d return _fetch_novita_pricing(force_refresh=force_refresh) if normalized == "deepinfra": return _fetch_deepinfra_pricing(force_refresh=force_refresh) + if normalized == "fireworks": + return _fireworks_pricing_from_models_dev(force_refresh=force_refresh) if normalized == "nous": api_key, base_url = _resolve_nous_pricing_credentials() if base_url: @@ -1670,6 +1672,55 @@ def get_pricing_for_provider(provider: str, *, force_refresh: bool = False) -> d return {} +def _fireworks_pricing_from_models_dev( + *, + force_refresh: bool = False, +) -> dict[str, dict[str, str]]: + """Derive Fireworks picker pricing from the models.dev registry cache. + + No dedicated network fetch: ``fetch_models_dev()`` already maintains an + in-memory + disk cache (1h TTL) that every picker surface shares, so this + is a pure dict transform on the picker path — no added latency and no + per-render network call. Results are additionally memoized in + ``_pricing_cache`` so repeated menu renders within a process are free. + + models.dev publishes Fireworks costs in USD per 1M tokens; the shared + pricing formatter expects per-token strings, so divide by 1M. + """ + cache_key = "models.dev/fireworks" + if not force_refresh and cache_key in _pricing_cache: + return _pricing_cache[cache_key] + + result: dict[str, dict[str, str]] = {} + try: + from agent.models_dev import _get_provider_models + + models = _get_provider_models("fireworks") or {} + for mid, entry in models.items(): + if not isinstance(entry, dict): + continue + cost = entry.get("cost") + if not isinstance(cost, dict): + continue + inp = cost.get("input") + out = cost.get("output") + if inp is None and out is None: + continue + row: dict[str, str] = { + "prompt": str(float(inp or 0) / 1_000_000), + "completion": str(float(out or 0) / 1_000_000), + } + cache_read = cost.get("cache_read") + if cache_read: + row["input_cache_read"] = str(float(cache_read) / 1_000_000) + result[str(mid)] = row + except Exception: + result = {} + + _pricing_cache[cache_key] = result + return result + + def _fetch_novita_pricing( timeout: float = 8.0, *, diff --git a/tests/agent/test_usage_pricing.py b/tests/agent/test_usage_pricing.py index 7b584ebe0e9b..7523931299e5 100644 --- a/tests/agent/test_usage_pricing.py +++ b/tests/agent/test_usage_pricing.py @@ -351,10 +351,10 @@ def test_fireworks_base_url_host_match_alone_routes_to_pricing(): assert float(entry.output_cost_per_million) == 3.48 -def test_fireworks_qwen3p6_plus_estimate_usage_cost(): - """End-to-end: Fireworks Qwen3.6-Plus sessions report a dollar estimate.""" +def test_fireworks_qwen3p7_plus_estimate_usage_cost(): + """End-to-end: Fireworks Qwen3.7-Plus sessions report a dollar estimate.""" result = estimate_usage_cost( - "accounts/fireworks/models/qwen3p6-plus", + "accounts/fireworks/models/qwen3p7-plus", CanonicalUsage(input_tokens=1_000_000, output_tokens=500_000), provider="fireworks", base_url="https://api.fireworks.ai/inference/v1", @@ -362,5 +362,56 @@ def test_fireworks_qwen3p6_plus_estimate_usage_cost(): assert result.status == "estimated" assert result.amount_usd is not None - # 1M input × $0.50/M + 500K output × $3.00/M = $0.50 + $1.50 = $2.00 - assert float(result.amount_usd) == 2.00 + # 1M input × $0.40/M + 500K output × $1.60/M = $0.40 + $0.80 = $1.20 + assert float(result.amount_usd) == 1.20 + + +def test_fireworks_router_fast_tier_prices_distinctly(): + """Fast serving tiers live under accounts/fireworks/routers/-fast and + bill at higher rates than the standard model — the routing layer's + rsplit("/", 1) must land on the distinct fast-tier entry.""" + standard = get_pricing_entry( + "accounts/fireworks/models/kimi-k2p6", + provider="fireworks", + base_url="https://api.fireworks.ai/inference/v1", + ) + fast = get_pricing_entry( + "accounts/fireworks/routers/kimi-k2p6-fast", + provider="fireworks", + base_url="https://api.fireworks.ai/inference/v1", + ) + assert standard is not None and fast is not None + assert fast.input_cost_per_million > standard.input_cost_per_million + assert fast.output_cost_per_million > standard.output_cost_per_million + + +def test_fireworks_plugin_fallback_models_all_have_pricing(): + """Invariant: every model in the Fireworks provider plugin's + fallback_models (the picker's curated safety net) must resolve to a + pricing entry — otherwise the default picker choices bill as unknown.""" + from providers import get_provider_profile + + profile = get_provider_profile("fireworks") + assert profile is not None + for mid in profile.fallback_models: + entry = get_pricing_entry( + mid, + provider="fireworks", + base_url="https://api.fireworks.ai/inference/v1", + ) + assert entry is not None, f"no pricing entry for fallback model {mid}" + assert entry.input_cost_per_million is not None, mid + + +def test_fireworks_rows_all_carry_cache_read_pricing(): + """Invariant: Fireworks publishes cached-input rates for every serverless + model, and Hermes prompt caching is active on Fireworks sessions — every + snapshot row must carry a cache_read rate cheaper than fresh input.""" + from agent.usage_pricing import _OFFICIAL_DOCS_PRICING + + fw_rows = [k for k in _OFFICIAL_DOCS_PRICING if k[0] == "fireworks"] + assert fw_rows, "expected at least one fireworks pricing row" + for key in fw_rows: + entry = _OFFICIAL_DOCS_PRICING[key] + assert entry.cache_read_cost_per_million is not None, key + assert entry.cache_read_cost_per_million < entry.input_cost_per_million, key