Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
22 changes: 22 additions & 0 deletions eval/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -194,6 +194,28 @@ with another provider flag or `--api-base`.
Atlas requests are not automatically retried, including timeouts, connection
errors and rate limits, to avoid duplicating billable generation requests.

## Cheaper Inference readers

[Cheaper Inference](https://cheaperinference.com) is an OpenAI-compatible LLM gateway.
Each model costs 15–60% less than the list price of its lab.

Select it explicitly with `--cheaperinference`; other readers are unchanged.
Set `CHEAPER_INFERENCE_API_KEY` and use a bare model ID from the
[model list](https://cheaperinference.com/#models):

```bash
export CHEAPER_INFERENCE_API_KEY=ci_live_...
python run_bench.py --task simpleqa --model gpt-5.4-mini \
--cheaperinference --num-examples 1 --max-concurrent 1 --max-tokens 256
```

This text-only example does not use retrieval. For screenshot benchmarks, use
`gpt-5.4-mini` or `gpt-5.4`, which accept image input. Model IDs are forwarded
unchanged to `https://api.cheaperinference.com/v1` and never select the native
Google SDK. `--api-key` overrides `CHEAPER_INFERENCE_API_KEY`, and other providers'
environment keys are never used. Do not combine this option with another provider
flag or `--api-base`.

## MiniMax readers

Set `MINIMAX_API_KEY` and select either registered model ID. The model context length
Expand Down
17 changes: 17 additions & 0 deletions eval/lib/model_config.py
Original file line number Diff line number Diff line change
Expand Up @@ -68,6 +68,23 @@ def get_atlascloud_config(
}


def get_cheaperinference_config(
model_name: str, api_key: str | None = None
) -> Dict[str, Any]:
"""Resolve an explicitly selected Cheaper Inference reader without native routing."""
if not api_key or api_key == "dummy":
api_key = os.getenv("CHEAPER_INFERENCE_API_KEY")
if not api_key or not api_key.strip() or api_key == "dummy":
raise ValueError(
"Cheaper Inference requires --api-key or CHEAPER_INFERENCE_API_KEY."
)
return {
"api_base": "https://api.cheaperinference.com/v1",
"api_key": api_key,
"model": model_name,
}


def get_model_config(model_name: str) -> Dict[str, Any]:
"""
Get model configuration based on model name.
Expand Down
38 changes: 32 additions & 6 deletions eval/run_bench.py
Original file line number Diff line number Diff line change
Expand Up @@ -74,6 +74,7 @@
from lib.model_config import (
ORCAROUTER_API_BASE,
get_atlascloud_config,
get_cheaperinference_config,
get_model_config,
get_output_filename,
)
Expand Down Expand Up @@ -958,18 +959,24 @@ async def run_async(args):
)

# Get model configuration
model_config = (
get_atlascloud_config(args.model, args.api_key)
if args.atlascloud
else get_model_config(args.model)
)
if args.atlascloud:
model_config = get_atlascloud_config(args.model, args.api_key)
elif args.cheaperinference:
model_config = get_cheaperinference_config(args.model, args.api_key)
else:
model_config = get_model_config(args.model)

# Handle OpenRouter API
if args.atlascloud:
api_base = model_config["api_base"]
api_key = model_config["api_key"]
model = model_config["model"]
logger.info(f"Using Atlas Cloud API with model: {model}")
elif args.cheaperinference:
api_base = model_config["api_base"]
api_key = model_config["api_key"]
model = model_config["model"]
logger.info(f"Using Cheaper Inference API with model: {model}")
elif args.open_router:
api_base = "https://openrouter.ai/api/v1"
if args.api_key and args.api_key != "dummy":
Expand Down Expand Up @@ -1144,7 +1151,11 @@ async def run_async(args):
timeout=args.timeout,
enable_thinking=(False if args.no_think else None),
force_openai_compat=(
args.open_router or args.commonstack or args.orcarouter or args.atlascloud
args.open_router
or args.commonstack
or args.orcarouter
or args.atlascloud
or args.cheaperinference
),
use_litellm=args.litellm,
retry_requests=not args.atlascloud,
Expand Down Expand Up @@ -1374,6 +1385,11 @@ def main():
action="store_true",
help="Use Atlas Cloud with an exact catalog model ID. Requires --api-key or ATLASCLOUD_API_KEY.",
)
parser.add_argument(
"--cheaperinference",
action="store_true",
help="Use Cheaper Inference (https://api.cheaperinference.com/v1) with a bare model ID. Requires --api-key or CHEAPER_INFERENCE_API_KEY.",
)
parser.add_argument(
"--litellm",
action="store_true",
Expand Down Expand Up @@ -1932,6 +1948,16 @@ def main():
parser.error("--atlascloud cannot be combined with another provider flag")
if args.atlascloud and args.api_base:
parser.error("--atlascloud uses its fixed endpoint; omit --api-base")
if args.cheaperinference and (
args.open_router
or args.commonstack
or args.orcarouter
or args.atlascloud
or args.litellm
):
parser.error("--cheaperinference cannot be combined with another provider flag")
if args.cheaperinference and args.api_base:
parser.error("--cheaperinference uses its fixed endpoint; omit --api-base")

# Validate mutually exclusive options
mode_count = sum(
Expand Down
95 changes: 95 additions & 0 deletions tests/test_cheaperinference_client.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,95 @@
"""Offline coverage of the optional Cheaper Inference evaluation reader."""

import asyncio
import sys
import types
from pathlib import Path
from runpy import run_path

import pytest

_LIB = Path(__file__).parents[1] / "eval" / "lib"
if "lib" not in sys.modules:
_pkg = types.ModuleType("lib")
_pkg.__path__ = [str(_LIB)]
sys.modules["lib"] = _pkg

from lib.llm import LLMClient

get_config = run_path(str(_LIB / "model_config.py"))["get_cheaperinference_config"]


@pytest.mark.parametrize("model", ["gpt-5.4-mini", "gemini-3.1-pro"])
def test_cheaperinference_preserves_model_and_isolates_credentials(monkeypatch, model):
monkeypatch.setenv("CHEAPER_INFERENCE_API_KEY", "ci-test-key")
monkeypatch.setenv("OPENAI_API_KEY", "other-key")
monkeypatch.setenv("API_KEY", "generic-key")
config = get_config(model)
assert config == {
"model": model,
"api_base": "https://api.cheaperinference.com/v1",
"api_key": "ci-test-key",
}
assert get_config(model, "explicit-key")["api_key"] == "explicit-key"


@pytest.mark.parametrize("key", [None, "", " ", "dummy"])
def test_cheaperinference_requires_own_key(monkeypatch, key):
monkeypatch.delenv("CHEAPER_INFERENCE_API_KEY", raising=False)
if key is not None:
monkeypatch.setenv("CHEAPER_INFERENCE_API_KEY", key)
monkeypatch.setenv("API_KEY", "generic-key")
monkeypatch.setenv("OPENAI_API_KEY", "openai-key")
with pytest.raises(ValueError, match="CHEAPER_INFERENCE_API_KEY"):
get_config("gpt-5.4-mini")


@pytest.mark.parametrize("model", ["gpt-5.4-mini", "gemini-3.1-pro"])
def test_cheaperinference_client_forwards_image_messages(monkeypatch, model):
calls = []
options = {}

async def create(**kwargs):
calls.append(kwargs)
return types.SimpleNamespace(
choices=[types.SimpleNamespace(message=types.SimpleNamespace(content="4"))],
usage=types.SimpleNamespace(
prompt_tokens=3, completion_tokens=1, total_tokens=4
),
)

def openai_client(**kwargs):
options.update(kwargs)
return types.SimpleNamespace(
chat=types.SimpleNamespace(completions=types.SimpleNamespace(create=create))
)

monkeypatch.setitem(
sys.modules, "openai", types.SimpleNamespace(AsyncOpenAI=openai_client)
)
client = LLMClient(
**get_config(model, "ci-test-key"),
force_openai_compat=True,
max_tokens=64,
)
messages = [
{
"role": "user",
"content": [
{
"type": "image_url",
"image_url": {"url": "data:image/png;base64,AAAA"},
},
{"type": "text", "text": "2+2?"},
],
}
]
text, usage = asyncio.run(client.generate(messages))
assert text == "4"
assert usage["total_tokens"] == 4
assert len(calls) == 1
assert calls[0]["model"] == model
assert calls[0]["messages"] == messages
assert options["base_url"] == "https://api.cheaperinference.com/v1"
assert options["api_key"] == "ci-test-key"
assert not client.is_gemini
Loading