diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index 11ca43d8..ea2fb245 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -66,6 +66,13 @@ repos: entry: tools/addlicense.sh pass_filenames: true files: '\.py$' + - id: model-cost-in-sync + name: model_cost.json is generated from model_profiles.yaml + language: python + entry: python scripts/pricing/build_model_cost.py --check + additional_dependencies: [pyyaml] + pass_filenames: false + files: '^(model_cost/|scripts/pricing/)' # Add mypy hooks for both local runs and CI (manual stage) - repo: local hooks: diff --git a/llm_evaluation/evaluate_models.py b/llm_evaluation/evaluate_models.py index 854dd1f8..36d4bd3e 100644 --- a/llm_evaluation/evaluate_models.py +++ b/llm_evaluation/evaluate_models.py @@ -16,6 +16,7 @@ import json import argparse import glob +import threading from typing import Dict, List, Any, Optional import sys @@ -87,6 +88,10 @@ def __init__( str, Any ] = {} # Store existing results for incremental evaluation self.cost_config: Dict[str, Any] = {} # Store cost configuration + # Model names charged at the maximum price because no price profile + # matched them -> number of rows. + self.unpriced_models: Dict[str, int] = {} + self._unpriced_lock = threading.Lock() self.num_workers = num_workers # Load dataset configurations @@ -207,21 +212,39 @@ def load_cost_config(self): self.cost_config = {} def _lookup_cost_info(self, model_name: str): - """Find the pricing entry for a model name, trying an exact match first - and then a substring fallback (historical behaviour). Returns the cost - dict, or None if no price is known.""" + """Find the pricing entry for a model name by exact match on a model id + or alias (see model_cost/model_profiles.yaml). Returns the cost dict, or + None if no price is known. + + There is deliberately no fuzzy/substring fallback: it used to let a name + pick up whichever price key happened to be a substring of it.""" if not self.cost_config or not model_name: return None # Remove _batch suffix if present for cost lookup cost_lookup_name = ( model_name[:-6] if model_name.endswith("_batch") else model_name ) - if cost_lookup_name in self.cost_config: - return self.cost_config[cost_lookup_name] - for config_name in self.cost_config.keys(): - if config_name in cost_lookup_name or cost_lookup_name in config_name: - return self.cost_config[config_name] - return None + return self.cost_config.get(cost_lookup_name) + + def max_price_info(self) -> Dict[str, float]: + """The highest input and output prices in the cost table. Used to charge + rows whose model has no price profile, so an unregistered or misspelled + model is never cheaper than a registered one.""" + if not self.cost_config: + return { + "input_token_price_per_million": 0.0, + "output_token_price_per_million": 0.0, + } + return { + "input_token_price_per_million": max( + float(v.get("input_token_price_per_million", 0.0)) + for v in self.cost_config.values() + ), + "output_token_price_per_million": max( + float(v.get("output_token_price_per_million", 0.0)) + for v in self.cost_config.values() + ), + } def has_price(self, model_name: str) -> bool: """Whether a price is known for this model name. Used to decide whether a @@ -248,14 +271,22 @@ def calculate_inference_cost( cost_info = self._lookup_cost_info(model_name) if not cost_info: - print( - f"Warning: No cost configuration found for model {model_name} (lookup: {cost_lookup_name})" - ) - if len(self.cost_config) > 0: + # Unmatched model: charge the highest price in the table rather than + # dropping the cost, and warn once per model name. + cost_info = self.max_price_info() + with self._unpriced_lock: + first_time = cost_lookup_name not in self.unpriced_models + self.unpriced_models[cost_lookup_name] = ( + self.unpriced_models.get(cost_lookup_name, 0) + 1 + ) + if first_time: print( - f"Available cost config keys (first 10): {list(self.cost_config.keys())[:10]}" + f"Warning: No price profile for model {cost_lookup_name!r}; " + "charging the maximum price " + f"(${cost_info['input_token_price_per_million']}/" + f"${cost_info['output_token_price_per_million']} per 1M tokens). " + "Add it to model_cost/model_profiles.yaml." ) - return 0.0 # Calculate cost input_tokens = token_usage.get("input_tokens", 0) or 0 diff --git a/llm_evaluation/run.py b/llm_evaluation/run.py index fbbc3f1d..c72fca01 100644 --- a/llm_evaluation/run.py +++ b/llm_evaluation/run.py @@ -362,14 +362,18 @@ def evaluate_single_prediction( ) return False - # Convert to universal model name + # Convert to universal model name. An unknown name is still graded (the score + # does not depend on it) and is charged the maximum price below, so a router + # cannot drop rows from its results by naming an unregistered model. try: universal_model_name = ModelNameManager.get_universal_name(model_name) except Exception as e: - logger.error( - f"Error converting model name '{model_name}' to universal name: {e}" + # Logged once per name by the evaluator and summarised after the run. + logger.debug( + f"Unknown model name '{model_name}' ({e}); evaluating it anyway " + "and charging the maximum price." ) - return False + universal_model_name = model_name # Determine dataset name from global_index dataset_name = evaluator.determine_dataset_from_global_index(global_index) @@ -586,6 +590,18 @@ def save_callback(): logger.info( f"Predictions saved to: ./router_inference/predictions/{router_name}.json" ) + if evaluator.unpriced_models: + max_price = evaluator.max_price_info() + logger.warning( + "Charged at the maximum price " + f"(${max_price['input_token_price_per_million']}/" + f"${max_price['output_token_price_per_million']} per 1M tokens) " + "because no price profile matched:" + ) + for name, rows in sorted( + evaluator.unpriced_models.items(), key=lambda kv: -kv[1] + ): + logger.warning(f" {name}: {rows} rows") logger.info("=" * 60) # Compute and display router-level metrics diff --git a/model_cost/model_cost.json b/model_cost/model_cost.json index ae82c1d1..debee82c 100644 --- a/model_cost/model_cost.json +++ b/model_cost/model_cost.json @@ -11,10 +11,18 @@ "input_token_price_per_million": 0.5, "output_token_price_per_million": 3 }, + "google/gemini-3-flash-preview": { + "input_token_price_per_million": 0.5, + "output_token_price_per_million": 3 + }, "o4-mini": { "input_token_price_per_million": 1.1, "output_token_price_per_million": 4.4 }, + "o4-mini-2025-04-16": { + "input_token_price_per_million": 1.1, + "output_token_price_per_million": 4.4 + }, "gpt-5.1": { "input_token_price_per_million": 1.25, "output_token_price_per_million": 10.0 @@ -31,10 +39,18 @@ "input_token_price_per_million": 0.25, "output_token_price_per_million": 2.0 }, + "openai/gpt-5-mini": { + "input_token_price_per_million": 0.25, + "output_token_price_per_million": 2.0 + }, "gpt-5-nano": { "input_token_price_per_million": 0.05, "output_token_price_per_million": 0.4 }, + "gpt-5-nano-2025-08-07": { + "input_token_price_per_million": 0.05, + "output_token_price_per_million": 0.4 + }, "deepseek-reasoner": { "input_token_price_per_million": 0.28, "output_token_price_per_million": 0.42 @@ -43,10 +59,6 @@ "input_token_price_per_million": 2.0, "output_token_price_per_million": 12.0 }, - "deepseek-v3.2": { - "input_token_price_per_million": 0.28, - "output_token_price_per_million": 0.42 - }, "test": { "input_token_price_per_million": 1.0, "output_token_price_per_million": 2.0 @@ -55,6 +67,10 @@ "input_token_price_per_million": 3.0, "output_token_price_per_million": 15.0 }, + "anthropic/claude-sonnet-4.5": { + "input_token_price_per_million": 3.0, + "output_token_price_per_million": 15.0 + }, "claude-3-haiku-20240307": { "input_token_price_per_million": 0.25, "output_token_price_per_million": 1.25 @@ -63,6 +79,10 @@ "input_token_price_per_million": 0.15, "output_token_price_per_million": 0.6 }, + "openai/gpt-4o-mini": { + "input_token_price_per_million": 0.15, + "output_token_price_per_million": 0.6 + }, "glm-4.5-air": { "input_token_price_per_million": 0.1, "output_token_price_per_million": 0.3 @@ -75,14 +95,14 @@ "input_token_price_per_million": 0.07, "output_token_price_per_million": 0.07 }, + "glm-4-air": { + "input_token_price_per_million": 0.07, + "output_token_price_per_million": 0.07 + }, "qwen_qwen3-vl-235b-a22b-thinking": { "input_token_price_per_million": 0.3, "output_token_price_per_million": 1.2 }, - "gemini-2.5-flash-lite": { - "input_token_price_per_million": 0.1, - "output_token_price_per_million": 0.4 - }, "gemini-2.5-flash": { "input_token_price_per_million": 0.3, "output_token_price_per_million": 2.5 @@ -95,10 +115,6 @@ "input_token_price_per_million": 0.5, "output_token_price_per_million": 1.5 }, - "z-ai_glm-4.7": { - "input_token_price_per_million": 0.4, - "output_token_price_per_million": 1.5 - }, "qwen_qwen3-vl-235b-a22b-instruct": { "input_token_price_per_million": 0.2, "output_token_price_per_million": 1.2 @@ -115,22 +131,10 @@ "input_token_price_per_million": 0.1, "output_token_price_per_million": 0.3 }, - "openai_gpt-oss-120b": { - "input_token_price_per_million": 0.039, - "output_token_price_per_million": 0.19 - }, - "qwen_qwen3-235b-a22b-2507": { - "input_token_price_per_million": 0.071, - "output_token_price_per_million": 0.463 - }, "qwen/qwen3-next-80b-a3b-instruct": { "input_token_price_per_million": 0.09, "output_token_price_per_million": 1.1 }, - "claude-haiku-4.5": { - "input_token_price_per_million": 1.0, - "output_token_price_per_million": 5.0 - }, "x-ai_grok-4.1-fast": { "input_token_price_per_million": 0.2, "output_token_price_per_million": 0.5 @@ -163,10 +167,6 @@ "input_token_price_per_million": 0.2, "output_token_price_per_million": 0.2 }, - "gpt-4o": { - "input_token_price_per_million": 2.5, - "output_token_price_per_million": 10.0 - }, "qwen/qwen3-30b-a3b-instruct-2507": { "input_token_price_per_million": 0.08, "output_token_price_per_million": 0.33 @@ -203,6 +203,10 @@ "input_token_price_per_million": 5.0, "output_token_price_per_million": 30.0 }, + "gpt-5.5-2026-04-23": { + "input_token_price_per_million": 5.0, + "output_token_price_per_million": 30.0 + }, "gpt-5.4-mini": { "input_token_price_per_million": 0.4, "output_token_price_per_million": 1.6 @@ -235,6 +239,10 @@ "input_token_price_per_million": 0.071, "output_token_price_per_million": 0.1 }, + "qwen_qwen3-235b-a22b-2507": { + "input_token_price_per_million": 0.071, + "output_token_price_per_million": 0.1 + }, "qwen/qwen3.5-9b": { "input_token_price_per_million": 0.05, "output_token_price_per_million": 0.15 @@ -243,14 +251,26 @@ "input_token_price_per_million": 0.26, "output_token_price_per_million": 0.38 }, + "deepseek-v3.2": { + "input_token_price_per_million": 0.26, + "output_token_price_per_million": 0.38 + }, "openai/gpt-4o": { "input_token_price_per_million": 2.5, "output_token_price_per_million": 10.0 }, + "gpt-4o": { + "input_token_price_per_million": 2.5, + "output_token_price_per_million": 10.0 + }, "claude-haiku-4-5-20251001": { "input_token_price_per_million": 1.0, "output_token_price_per_million": 5.0 }, + "claude-haiku-4.5": { + "input_token_price_per_million": 1.0, + "output_token_price_per_million": 5.0 + }, "claude-sonnet-4": { "input_token_price_per_million": 3.0, "output_token_price_per_million": 15.0 @@ -295,6 +315,14 @@ "input_token_price_per_million": 0.15, "output_token_price_per_million": 0.6 }, + "openai_gpt-oss-120b": { + "input_token_price_per_million": 0.15, + "output_token_price_per_million": 0.6 + }, + "openai/gpt-oss-120b": { + "input_token_price_per_million": 0.15, + "output_token_price_per_million": 0.6 + }, "llama-4-maverick-17b-128e-instruct-fp8": { "input_token_price_per_million": 0.25, "output_token_price_per_million": 1.0 @@ -331,10 +359,18 @@ "input_token_price_per_million": 0.4, "output_token_price_per_million": 1.5 }, + "z-ai_glm-4.7": { + "input_token_price_per_million": 0.4, + "output_token_price_per_million": 1.5 + }, "MiniMax-M3": { "input_token_price_per_million": 0.6, "output_token_price_per_million": 2.4 }, + "MiniMaxAI/MiniMax-M3": { + "input_token_price_per_million": 0.6, + "output_token_price_per_million": 2.4 + }, "agnes-2.0-flash": { "input_token_price_per_million": 0.03, "output_token_price_per_million": 0.15 @@ -347,11 +383,11 @@ "input_token_price_per_million": 0.18, "output_token_price_per_million": 0.85 }, - "grok-4.3": { + "x-ai/grok-4.3": { "input_token_price_per_million": 1.25, "output_token_price_per_million": 2.5 }, - "x-ai/grok-4.3": { + "grok-4.3": { "input_token_price_per_million": 1.25, "output_token_price_per_million": 2.5 }, @@ -359,26 +395,22 @@ "input_token_price_per_million": 0.03, "output_token_price_per_million": 0.13 }, + "openai/gpt-oss-20b": { + "input_token_price_per_million": 0.03, + "output_token_price_per_million": 0.13 + }, "xiaomi/mimo-v2.5": { "input_token_price_per_million": 0.14, "output_token_price_per_million": 0.28 }, - "openai/gpt-5-mini": { - "input_token_price_per_million": 0.25, - "output_token_price_per_million": 2.0 - }, - "anthropic/claude-sonnet-4.5": { - "input_token_price_per_million": 3.0, - "output_token_price_per_million": 15.0 - }, - "openai/gpt-4o-mini": { - "input_token_price_per_million": 0.15, - "output_token_price_per_million": 0.6 - }, "google/gemini-2.5-flash-lite": { "input_token_price_per_million": 0.1, "output_token_price_per_million": 0.4 }, + "gemini-2.5-flash-lite": { + "input_token_price_per_million": 0.1, + "output_token_price_per_million": 0.4 + }, "google/gemini-2.5-pro": { "input_token_price_per_million": 1.25, "output_token_price_per_million": 10.0 diff --git a/model_cost/model_profiles.yaml b/model_cost/model_profiles.yaml new file mode 100644 index 00000000..da9dce72 --- /dev/null +++ b/model_cost/model_profiles.yaml @@ -0,0 +1,389 @@ +# SPDX-FileCopyrightText: Copyright contributors to the RouterArena project +# SPDX-License-Identifier: Apache-2.0 +# +# Model price profiles: the source of truth for model_cost/model_cost.json. +# One entry per model. After editing, regenerate the JSON with: +# python scripts/pricing/build_model_cost.py +# +# Fields +# price.input / price.output USD per 1M tokens (required) +# aliases other spellings of the same model; they share its price +# as_of date the price was recorded +# source URL of the published price (required for new models) +# note free text, e.g. known conflicts to resolve +# +# Prices are fixed for now; a market sync will update them later. A model name that +# matches no id or alias is charged the highest input and output price in this file. + +models: + "google/gemma-3n-e4b-it": + price: {input: 0.02, output: 0.04} + as_of: "2026-09-27" + source: null + "gemini-3-pro-preview": + price: {input: 2.0, output: 12.0} + as_of: "2026-09-27" + source: null + "gemini-3-flash-preview": + price: {input: 0.5, output: 3} + aliases: ["google/gemini-3-flash-preview"] + as_of: "2026-09-27" + source: null + "o4-mini": + price: {input: 1.1, output: 4.4} + aliases: ["o4-mini-2025-04-16"] + as_of: "2026-09-27" + source: null + "gpt-5.1": + price: {input: 1.25, output: 10.0} + as_of: "2026-09-27" + source: null + "gpt-5.2": + price: {input: 1.75, output: 14.0} + as_of: "2026-09-27" + source: null + "gpt-5.1-None": + price: {input: 1.25, output: 10.0} + as_of: "2026-09-27" + source: null + "gpt-5-mini": + price: {input: 0.25, output: 2.0} + aliases: ["openai/gpt-5-mini"] + as_of: "2026-09-27" + source: null + "gpt-5-nano": + price: {input: 0.05, output: 0.4} + aliases: ["gpt-5-nano-2025-08-07"] + as_of: "2026-09-27" + source: null + "deepseek-reasoner": + price: {input: 0.28, output: 0.42} + as_of: "2026-09-27" + source: null + "gemini-3.0-pro": + price: {input: 2.0, output: 12.0} + as_of: "2026-09-27" + source: null + "test": + price: {input: 1.0, output: 2.0} + as_of: "2026-09-27" + source: null + "claude-sonnet-4-5": + price: {input: 3.0, output: 15.0} + aliases: ["anthropic/claude-sonnet-4.5"] + as_of: "2026-09-27" + source: null + "claude-3-haiku-20240307": + price: {input: 0.25, output: 1.25} + as_of: "2026-09-27" + source: null + "gpt-4o-mini": + price: {input: 0.15, output: 0.6} + aliases: ["openai/gpt-4o-mini"] + as_of: "2026-09-27" + source: null + "glm-4.5-air": + price: {input: 0.1, output: 0.3} + as_of: "2026-09-27" + source: null + "glm-4.6": + price: {input: 0.28, output: 1.12} + as_of: "2026-09-27" + source: null + "glm-4-air-250414": + price: {input: 0.07, output: 0.07} + aliases: ["glm-4-air"] + as_of: "2026-09-27" + source: null + note: "Alias glm-4-air was previously matched by substring; kept at this price." + "qwen_qwen3-vl-235b-a22b-thinking": + price: {input: 0.3, output: 1.2} + as_of: "2026-09-27" + source: null + "gemini-2.5-flash": + price: {input: 0.3, output: 2.5} + as_of: "2026-09-27" + source: null + "gemini-2.0-flash-001": + price: {input: 0.1, output: 0.4} + as_of: "2026-09-27" + source: null + "qwen_qwen3-vl-32b-instruct": + price: {input: 0.5, output: 1.5} + as_of: "2026-09-27" + source: null + "qwen_qwen3-vl-235b-a22b-instruct": + price: {input: 0.2, output: 1.2} + as_of: "2026-09-27" + source: null + "qwen_qwen3-coder": + price: {input: 0.22, output: 0.95} + as_of: "2026-09-27" + source: null + "x-ai_grok-code-fast-1": + price: {input: 0.2, output: 1.5} + as_of: "2026-09-27" + source: null + "xiaomi_mimo-v2-flash:free": + price: {input: 0.1, output: 0.3} + as_of: "2026-09-27" + source: null + "qwen/qwen3-next-80b-a3b-instruct": + price: {input: 0.09, output: 1.1} + as_of: "2026-09-27" + source: null + "x-ai_grok-4.1-fast": + price: {input: 0.2, output: 0.5} + as_of: "2026-09-27" + source: null + "mistralai_devstral-2512:free": + price: {input: 0.05, output: 0.22} + as_of: "2026-09-27" + source: null + "meta-llama_llama-3.3-70b-instruct": + price: {input: 0.1, output: 0.32} + as_of: "2026-09-27" + source: null + "meta-llama_llama-3.1-405b-instruct": + price: {input: 3.5, output: 3.5} + as_of: "2026-09-27" + source: null + "mistralai/ministral-3b-2512": + price: {input: 0.1, output: 0.1} + as_of: "2026-09-27" + source: null + "mistralai/ministral-3-3b-2512": + price: {input: 0.1, output: 0.1} + as_of: "2026-09-27" + source: null + "mistralai/ministral-3-8b-2512": + price: {input: 0.15, output: 0.15} + as_of: "2026-09-27" + source: null + "mistralai/ministral-3-14b-2512": + price: {input: 0.2, output: 0.2} + as_of: "2026-09-27" + source: null + "qwen/qwen3-30b-a3b-instruct-2507": + price: {input: 0.08, output: 0.33} + as_of: "2026-09-27" + source: null + note: "Conflict: also profiled as qwen3-30b-a3b-instruct-2507 at 0.15/0.60. Unify at market sync." + "Qwen/Qwen3-Coder-Next": + price: {input: 0.07, output: 0.3} + as_of: "2026-09-27" + source: null + "qwen/qwen3-coder-30b-a3b-instruct": + price: {input: 0.07, output: 0.27} + as_of: "2026-09-27" + source: null + "moonshotai/kimi-k2.5": + price: {input: 0.6, output: 3.0} + as_of: "2026-09-27" + source: null + "z-ai/glm-5": + price: {input: 1.0, output: 3.2} + as_of: "2026-09-27" + source: null + "google/gemini-3.1-flash-lite": + price: {input: 0.25, output: 1.5} + as_of: "2026-09-27" + source: null + "claude-opus-4-7": + price: {input: 15.0, output: 75.0} + as_of: "2026-09-27" + source: null + "claude-haiku-4-5": + price: {input: 0.8, output: 4.0} + as_of: "2026-09-27" + source: null + note: "Conflict: same model as claude-haiku-4-5-20251001 (1.00/5.00). Kept separate so current scores do not move. Unify at market sync." + "gpt-5.5": + price: {input: 5.0, output: 30.0} + aliases: ["gpt-5.5-2026-04-23"] + as_of: "2026-09-27" + source: null + "gpt-5.4-mini": + price: {input: 0.4, output: 1.6} + as_of: "2026-09-27" + source: null + "gpt-4.1": + price: {input: 2.0, output: 8.0} + as_of: "2026-09-27" + source: null + "gemini-3.1-pro-preview": + price: {input: 2.0, output: 12.0} + as_of: "2026-09-27" + source: null + "gemini-3.1-flash-lite-preview": + price: {input: 0.1, output: 0.4} + as_of: "2026-09-27" + source: null + "deepseek/deepseek-v4-pro": + price: {input: 0.435, output: 0.87} + as_of: "2026-09-27" + source: null + "qwen/qwen3.5-flash-02-23": + price: {input: 0.065, output: 0.26} + as_of: "2026-09-27" + source: null + "deepseek/deepseek-v4-flash": + price: {input: 0.14, output: 0.28} + as_of: "2026-09-27" + source: null + "qwen/qwen3-235b-a22b-2507": + price: {input: 0.071, output: 0.1} + aliases: ["qwen_qwen3-235b-a22b-2507"] + as_of: "2026-09-27" + source: null + note: "Conflict: also profiled as qwen3-235b-a22b-instruct-2507 at 0.50/2.00 (charged to OrcaRouter). Dropped unused duplicate price 0.071/0.463 (qwen_qwen3-235b-a22b-2507). Unify at market sync." + "qwen/qwen3.5-9b": + price: {input: 0.05, output: 0.15} + as_of: "2026-09-27" + source: null + "deepseek/deepseek-v3.2": + price: {input: 0.26, output: 0.38} + aliases: ["deepseek-v3.2"] + as_of: "2026-09-27" + source: null + note: "Dropped unused duplicate price 0.28/0.42 (deepseek-v3.2)." + "openai/gpt-4o": + price: {input: 2.5, output: 10.0} + aliases: ["gpt-4o"] + as_of: "2026-09-27" + source: null + "claude-haiku-4-5-20251001": + price: {input: 1.0, output: 5.0} + aliases: ["claude-haiku-4.5"] + as_of: "2026-09-27" + source: null + note: "Conflict: also profiled as claude-haiku-4-5 at 0.80/4.00 (charged to Nadir Router). Unify at market sync." + "claude-sonnet-4": + price: {input: 3.0, output: 15.0} + as_of: "2026-09-27" + source: null + "deepseek-chat": + price: {input: 0.27, output: 1.1} + as_of: "2026-09-27" + source: null + "qwen3-235b-a22b-instruct-2507": + price: {input: 0.5, output: 2.0} + as_of: "2026-09-27" + source: null + note: "Conflict: same model as qwen/qwen3-235b-a22b-2507 (0.071/0.10). Kept separate so current scores do not move. Unify at market sync." + "qwen3-30b-a3b-instruct-2507": + price: {input: 0.15, output: 0.6} + as_of: "2026-09-27" + source: null + note: "Conflict: same model as qwen/qwen3-30b-a3b-instruct-2507 (0.08/0.33). Kept separate so current scores do not move. Unify at market sync." + "gpt-4.1-mini": + price: {input: 0.4, output: 1.6} + as_of: "2026-09-27" + source: null + "gpt-4.1-nano": + price: {input: 0.1, output: 0.4} + as_of: "2026-09-27" + source: null + "deepseek-v3.1": + price: {input: 1.23, output: 4.94} + as_of: "2026-09-27" + source: null + "gpt-5": + price: {input: 1.25, output: 10.0} + as_of: "2026-09-27" + source: null + "gpt-5-chat": + price: {input: 1.25, output: 10.0} + as_of: "2026-09-27" + source: null + "gpt-5.2-chat": + price: {input: 1.75, output: 14.0} + as_of: "2026-09-27" + source: null + "gpt-oss-120b": + price: {input: 0.15, output: 0.6} + aliases: ["openai_gpt-oss-120b", "openai/gpt-oss-120b"] + as_of: "2026-09-27" + source: null + note: "Dropped unused duplicate price 0.039/0.19 (openai_gpt-oss-120b)." + "llama-4-maverick-17b-128e-instruct-fp8": + price: {input: 0.25, output: 1.0} + as_of: "2026-09-27" + source: null + "grok-4": + price: {input: 5.5, output: 27.5} + as_of: "2026-09-27" + source: null + "claude-opus-4-1": + price: {input: 15.0, output: 75.0} + as_of: "2026-09-27" + source: null + "claude-opus-4-6": + price: {input: 5.0, output: 25.0} + as_of: "2026-09-27" + source: null + "gpt-5.4-nano": + price: {input: 0.2, output: 1.25} + as_of: "2026-09-27" + source: null + "grok-4-1-fast-reasoning": + price: {input: 0.2, output: 0.5} + as_of: "2026-09-27" + source: null + "gpt-5.3-chat": + price: {input: 1.75, output: 14.0} + as_of: "2026-09-27" + source: null + "gpt-5.4": + price: {input: 2.5, output: 15.0} + as_of: "2026-09-27" + source: null + "z-ai/glm-4.7": + price: {input: 0.4, output: 1.5} + aliases: ["z-ai_glm-4.7"] + as_of: "2026-09-27" + source: null + "MiniMax-M3": + price: {input: 0.6, output: 2.4} + aliases: ["MiniMaxAI/MiniMax-M3"] + as_of: "2026-09-27" + source: null + "agnes-2.0-flash": + price: {input: 0.03, output: 0.15} + as_of: "2026-09-27" + source: null + "THUDM/GLM-4-9B-0414": + price: {input: 0.05, output: 0.61} + as_of: "2026-09-27" + source: null + "deepseek-ai/DeepSeek-R1-0528-Qwen3-8B": + price: {input: 0.18, output: 0.85} + as_of: "2026-09-27" + source: null + "x-ai/grok-4.3": + price: {input: 1.25, output: 2.5} + aliases: ["grok-4.3"] + as_of: "2026-09-27" + source: null + "gpt-oss-20b": + price: {input: 0.03, output: 0.13} + aliases: ["openai/gpt-oss-20b"] + as_of: "2026-09-27" + source: null + "xiaomi/mimo-v2.5": + price: {input: 0.14, output: 0.28} + as_of: "2026-09-27" + source: null + "google/gemini-2.5-flash-lite": + price: {input: 0.1, output: 0.4} + aliases: ["gemini-2.5-flash-lite"] + as_of: "2026-09-27" + source: null + "google/gemini-2.5-pro": + price: {input: 1.25, output: 10.0} + as_of: "2026-09-27" + source: null + "google/gemma-4-31b-it": + price: {input: 0.08, output: 0.35} + as_of: "2026-09-27" + source: null diff --git a/router_inference/check_config_prediction_files.py b/router_inference/check_config_prediction_files.py index f4d09823..c1c7857d 100644 --- a/router_inference/check_config_prediction_files.py +++ b/router_inference/check_config_prediction_files.py @@ -176,49 +176,46 @@ def check_model_costs( for model in config_models: used_models.add(model) - # Check each model has cost configuration + # Check each model has a price profile (exact id or alias match only; see + # model_cost/model_profiles.yaml). A model without one is not rejected: the + # evaluator charges it the highest price in the table, so we warn loudly. missing_costs = [] - for model_name in used_models: + for model_name in sorted(used_models): try: # Convert to universal name for cost lookup universal_name = model_manager.get_universal_name(model_name) + except Exception: + universal_name = model_name - # Remove _batch suffix if present - cost_lookup_name = universal_name - if universal_name.endswith("_batch"): - cost_lookup_name = universal_name[:-6] + # Remove _batch suffix if present + cost_lookup_name = universal_name + if universal_name.endswith("_batch"): + cost_lookup_name = universal_name[:-6] - # Check if cost exists (exact match or partial match) - has_cost = False - if cost_lookup_name in cost_config: - has_cost = True - else: - # Try partial matches - for config_name in cost_config.keys(): - if ( - config_name in cost_lookup_name - or cost_lookup_name in config_name - ): - has_cost = True - break - - if not has_cost: - missing_costs.append(f"{model_name} (universal: {cost_lookup_name})") - except Exception as e: - # If we can't convert, try original name - if model_name not in cost_config: - missing_costs.append(f"{model_name} (conversion failed: {str(e)})") + if cost_lookup_name not in cost_config: + missing_costs.append(f"{model_name} (looked up as: {cost_lookup_name})") if missing_costs: - errors.append(f"Missing cost configuration for {len(missing_costs)} model(s):") + max_in = max( + float(v.get("input_token_price_per_million", 0.0)) + for v in cost_config.values() + ) + max_out = max( + float(v.get("output_token_price_per_million", 0.0)) + for v in cost_config.values() + ) + errors.append(f"No price profile for {len(missing_costs)} model(s):") for model in missing_costs: errors.append(f" - {model}") errors.append( - "\nPlease add cost configuration to model_cost/model_cost.json or update " - "your router config to use models with existing cost configurations." + f"These rows will be charged the maximum price (${max_in}/${max_out} per " + "1M tokens). To avoid that, add a profile (with a source link) to " + "model_cost/model_profiles.yaml and run " + "`python scripts/pricing/build_model_cost.py`." ) - return len(missing_costs) == 0, errors + # Missing prices are a warning, not a failure. + return True, errors # Model slugs that upstream providers have retired and silently redirect to a @@ -659,7 +656,10 @@ def main(): try: if predictions is not None and config is not None: cost_valid, cost_errors = check_model_costs(predictions, config) - if cost_valid: + if cost_valid and cost_errors: + for warning in cost_errors: + print(f" ⚠ {warning}") + elif cost_valid: cost_config = load_cost_config() print( f"✓ All models have cost configurations ({len(cost_config)} models in cost file)" diff --git a/scripts/pricing/build_model_cost.py b/scripts/pricing/build_model_cost.py new file mode 100644 index 00000000..dcbd3c50 --- /dev/null +++ b/scripts/pricing/build_model_cost.py @@ -0,0 +1,124 @@ +# SPDX-FileCopyrightText: Copyright contributors to the RouterArena project +# SPDX-License-Identifier: Apache-2.0 + +"""Generate ``model_cost/model_cost.json`` from ``model_cost/model_profiles.yaml``. + +``model_profiles.yaml`` is the source of truth for model prices: one profile per +model, with optional aliases (other spellings of the same model). The evaluator +and the submission checks still read ``model_cost.json``, so this script expands +every profile into one JSON entry for its id and one for each alias, all at the +profile's price. + +Usage: + python scripts/pricing/build_model_cost.py # rewrite model_cost.json + python scripts/pricing/build_model_cost.py --check # exit 1 if it is out of date +""" + +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path +from typing import Any, Dict, List, Optional + +import yaml + +REPO_ROOT = Path(__file__).resolve().parents[2] +PROFILES_PATH = REPO_ROOT / "model_cost" / "model_profiles.yaml" +COST_PATH = REPO_ROOT / "model_cost" / "model_cost.json" + + +def load_profiles(path: Path = PROFILES_PATH) -> Dict[str, Dict[str, Any]]: + """Load and validate the profile file. Raises ValueError on a bad entry.""" + with open(path, "r", encoding="utf-8") as f: + data = yaml.safe_load(f) or {} + models = data.get("models") + if not isinstance(models, dict) or not models: + raise ValueError(f"{path}: expected a non-empty 'models' mapping") + + seen: Dict[str, str] = {} + errors: List[str] = [] + for model_id, profile in models.items(): + if not isinstance(profile, dict): + errors.append(f"{model_id}: profile must be a mapping") + continue + price = profile.get("price") or {} + for side in ("input", "output"): + value = price.get(side) + if ( + not isinstance(value, (int, float)) + or isinstance(value, bool) + or value < 0 + ): + errors.append(f"{model_id}: price.{side} must be a number >= 0") + aliases = profile.get("aliases") or [] + if not isinstance(aliases, list) or not all( + isinstance(a, str) for a in aliases + ): + errors.append(f"{model_id}: aliases must be a list of strings") + aliases = [] + for name in [model_id, *aliases]: + if name in seen: + errors.append( + f"{name!r} is used by both {seen[name]!r} and {model_id!r}" + ) + seen[name] = model_id + if errors: + raise ValueError(f"{path}: invalid profiles:\n " + "\n ".join(errors)) + return models + + +def build_cost_table(models: Dict[str, Dict[str, Any]]) -> Dict[str, Dict[str, Any]]: + """Expand profiles into the flat ``model_cost.json`` mapping.""" + table: Dict[str, Dict[str, Any]] = {} + for model_id, profile in models.items(): + entry = { + "input_token_price_per_million": profile["price"]["input"], + "output_token_price_per_million": profile["price"]["output"], + } + for name in [model_id, *(profile.get("aliases") or [])]: + table[name] = dict(entry) + return table + + +def render(table: Dict[str, Dict[str, Any]]) -> str: + return json.dumps(table, indent=2, ensure_ascii=False) + "\n" + + +def main(argv: Optional[List[str]] = None) -> int: + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument( + "--check", + action="store_true", + help="Do not write; exit 1 if model_cost.json differs from the profiles.", + ) + args = parser.parse_args(argv) + + try: + expected = render(build_cost_table(load_profiles())) + except ValueError as e: + print(e, file=sys.stderr) + return 1 + + current = COST_PATH.read_text(encoding="utf-8") if COST_PATH.exists() else "" + if args.check: + if current != expected: + print( + "model_cost/model_cost.json is out of date with model_profiles.yaml. " + "Run: python scripts/pricing/build_model_cost.py", + file=sys.stderr, + ) + return 1 + print("model_cost.json is up to date.") + return 0 + + COST_PATH.write_text(expected, encoding="utf-8") + print( + f"Wrote {COST_PATH.relative_to(REPO_ROOT)} ({len(json.loads(expected))} entries)." + ) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/test_pricing.py b/tests/test_pricing.py new file mode 100644 index 00000000..f6ceb848 --- /dev/null +++ b/tests/test_pricing.py @@ -0,0 +1,93 @@ +# SPDX-FileCopyrightText: Copyright contributors to the RouterArena project +# SPDX-License-Identifier: Apache-2.0 + +"""Tests for model price profiles and the evaluator's price lookup.""" + +import json +import sys +from pathlib import Path + +import pytest + +REPO_ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(REPO_ROOT)) +sys.path.insert(0, str(REPO_ROOT / "llm_evaluation")) +sys.path.insert(0, str(REPO_ROOT / "scripts" / "pricing")) + +import build_model_cost # noqa: E402 +from evaluate_models import ModelEvaluator # noqa: E402 + +USAGE = { + "input_tokens": 1_000_000, + "output_tokens": 1_000_000, + "total_tokens": 2_000_000, +} + + +@pytest.fixture(scope="module") +def evaluator(): + return ModelEvaluator(cached_results_dir=str(REPO_ROOT / "cached_results")) + + +def test_model_cost_json_is_generated_from_profiles(): + profiles = build_model_cost.load_profiles() + expected = build_model_cost.render(build_model_cost.build_cost_table(profiles)) + actual = (REPO_ROOT / "model_cost" / "model_cost.json").read_text(encoding="utf-8") + assert actual == expected, "run: python scripts/pricing/build_model_cost.py" + + +def test_every_name_is_unique_across_ids_and_aliases(tmp_path): + bad = tmp_path / "profiles.yaml" + bad.write_text( + "models:\n" + " a: {price: {input: 1, output: 2}, aliases: [b]}\n" + " b: {price: {input: 1, output: 2}}\n" + ) + with pytest.raises(ValueError, match="used by both"): + build_model_cost.load_profiles(bad) + + +def test_negative_or_missing_price_is_rejected(tmp_path): + bad = tmp_path / "profiles.yaml" + bad.write_text("models:\n a: {price: {input: -1}}\n") + with pytest.raises(ValueError, match="price.input"): + build_model_cost.load_profiles(bad) + + +def test_alias_is_charged_the_profile_price(evaluator): + # Lynkr names this model "openai/gpt-oss-120b"; its profile id is "gpt-oss-120b". + assert evaluator.has_price("openai/gpt-oss-120b") + assert evaluator.calculate_inference_cost( + "openai/gpt-oss-120b", USAGE + ) == pytest.approx(evaluator.calculate_inference_cost("gpt-oss-120b", USAGE)) + + +def test_no_substring_matching(evaluator): + # Used to match "gpt-oss-120b" because that key is a substring of the name. + assert not evaluator.has_price("gpt-oss-120b-some-finetune") + assert not evaluator.has_price("gpt") + + +def test_unmatched_model_is_charged_the_maximum_price(evaluator): + max_price = evaluator.max_price_info() + table = json.loads((REPO_ROOT / "model_cost" / "model_cost.json").read_text()) + assert max_price["input_token_price_per_million"] == max( + v["input_token_price_per_million"] for v in table.values() + ) + assert max_price["output_token_price_per_million"] == max( + v["output_token_price_per_million"] for v in table.values() + ) + cost = evaluator.calculate_inference_cost("not-a-registered-model", USAGE) + assert cost == pytest.approx( + max_price["input_token_price_per_million"] + + max_price["output_token_price_per_million"] + ) + assert evaluator.unpriced_models.get("not-a-registered-model", 0) >= 1 + + +def test_unmatched_model_is_never_cheaper_than_a_registered_one(evaluator): + table = json.loads((REPO_ROOT / "model_cost" / "model_cost.json").read_text()) + unmatched = evaluator.calculate_inference_cost("not-a-registered-model", USAGE) + assert all( + evaluator.calculate_inference_cost(name, USAGE) <= unmatched for name in table + )