Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions .pre-commit-config.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -66,6 +66,13 @@ repos:
entry: tools/addlicense.sh
pass_filenames: true
files: '\.py$'
- id: model-cost-in-sync
name: model_cost.json is generated from model_profiles.yaml
language: python
entry: python scripts/pricing/build_model_cost.py --check
additional_dependencies: [pyyaml]
pass_filenames: false
files: '^(model_cost/|scripts/pricing/)'
# Add mypy hooks for both local runs and CI (manual stage)
- repo: local
hooks:
Expand Down
61 changes: 46 additions & 15 deletions llm_evaluation/evaluate_models.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,7 @@
import json
import argparse
import glob
import threading
from typing import Dict, List, Any, Optional
import sys

Expand Down Expand Up @@ -87,6 +88,10 @@ def __init__(
str, Any
] = {} # Store existing results for incremental evaluation
self.cost_config: Dict[str, Any] = {} # Store cost configuration
# Model names charged at the maximum price because no price profile
# matched them -> number of rows.
self.unpriced_models: Dict[str, int] = {}
self._unpriced_lock = threading.Lock()
self.num_workers = num_workers

# Load dataset configurations
Expand Down Expand Up @@ -207,21 +212,39 @@ def load_cost_config(self):
self.cost_config = {}

def _lookup_cost_info(self, model_name: str):
"""Find the pricing entry for a model name, trying an exact match first
and then a substring fallback (historical behaviour). Returns the cost
dict, or None if no price is known."""
"""Find the pricing entry for a model name by exact match on a model id
or alias (see model_cost/model_profiles.yaml). Returns the cost dict, or
None if no price is known.

There is deliberately no fuzzy/substring fallback: it used to let a name
pick up whichever price key happened to be a substring of it."""
if not self.cost_config or not model_name:
return None
# Remove _batch suffix if present for cost lookup
cost_lookup_name = (
model_name[:-6] if model_name.endswith("_batch") else model_name
)
if cost_lookup_name in self.cost_config:
return self.cost_config[cost_lookup_name]
for config_name in self.cost_config.keys():
if config_name in cost_lookup_name or cost_lookup_name in config_name:
return self.cost_config[config_name]
return None
return self.cost_config.get(cost_lookup_name)

def max_price_info(self) -> Dict[str, float]:
"""The highest input and output prices in the cost table. Used to charge
rows whose model has no price profile, so an unregistered or misspelled
model is never cheaper than a registered one."""
if not self.cost_config:
return {
"input_token_price_per_million": 0.0,
"output_token_price_per_million": 0.0,
}
return {
"input_token_price_per_million": max(
float(v.get("input_token_price_per_million", 0.0))
for v in self.cost_config.values()
),
"output_token_price_per_million": max(
float(v.get("output_token_price_per_million", 0.0))
for v in self.cost_config.values()
),
}

def has_price(self, model_name: str) -> bool:
"""Whether a price is known for this model name. Used to decide whether a
Expand All @@ -248,14 +271,22 @@ def calculate_inference_cost(
cost_info = self._lookup_cost_info(model_name)

if not cost_info:
print(
f"Warning: No cost configuration found for model {model_name} (lookup: {cost_lookup_name})"
)
if len(self.cost_config) > 0:
# Unmatched model: charge the highest price in the table rather than
# dropping the cost, and warn once per model name.
cost_info = self.max_price_info()
with self._unpriced_lock:
first_time = cost_lookup_name not in self.unpriced_models
self.unpriced_models[cost_lookup_name] = (
self.unpriced_models.get(cost_lookup_name, 0) + 1
)
if first_time:
print(
f"Available cost config keys (first 10): {list(self.cost_config.keys())[:10]}"
f"Warning: No price profile for model {cost_lookup_name!r}; "
"charging the maximum price "
f"(${cost_info['input_token_price_per_million']}/"
f"${cost_info['output_token_price_per_million']} per 1M tokens). "
"Add it to model_cost/model_profiles.yaml."
)
return 0.0

# Calculate cost
input_tokens = token_usage.get("input_tokens", 0) or 0
Expand Down
24 changes: 20 additions & 4 deletions llm_evaluation/run.py
Original file line number Diff line number Diff line change
Expand Up @@ -362,14 +362,18 @@ def evaluate_single_prediction(
)
return False

# Convert to universal model name
# Convert to universal model name. An unknown name is still graded (the score
# does not depend on it) and is charged the maximum price below, so a router
# cannot drop rows from its results by naming an unregistered model.
try:
universal_model_name = ModelNameManager.get_universal_name(model_name)
except Exception as e:
logger.error(
f"Error converting model name '{model_name}' to universal name: {e}"
# Logged once per name by the evaluator and summarised after the run.
logger.debug(
f"Unknown model name '{model_name}' ({e}); evaluating it anyway "
"and charging the maximum price."
)
return False
universal_model_name = model_name

# Determine dataset name from global_index
dataset_name = evaluator.determine_dataset_from_global_index(global_index)
Expand Down Expand Up @@ -586,6 +590,18 @@ def save_callback():
logger.info(
f"Predictions saved to: ./router_inference/predictions/{router_name}.json"
)
if evaluator.unpriced_models:
max_price = evaluator.max_price_info()
logger.warning(
"Charged at the maximum price "
f"(${max_price['input_token_price_per_million']}/"
f"${max_price['output_token_price_per_million']} per 1M tokens) "
"because no price profile matched:"
)
for name, rows in sorted(
evaluator.unpriced_models.items(), key=lambda kv: -kv[1]
):
logger.warning(f" {name}: {rows} rows")
logger.info("=" * 60)

# Compute and display router-level metrics
Expand Down
Loading
Loading