## Motivation GPT OSS tool calling issues. ## Changes Fixes those and adds a bunch of evals for tool calling. Fixes GLM5 prefix caching, where CacheList wasn't getting handled properly. Extracts a bunch of the setup functionality of exo bench to a harness that can be reused elsewhere, such as in the tool calling eval. ## Test Plan ### Automated Testing Let's run the evals for all models
18 lines
377 B
TOML
18 lines
377 B
TOML
[project]
|
|
name = "exo-bench"
|
|
version = "0.1.0"
|
|
description = "Benchmarking tool for exo distributed inference"
|
|
requires-python = ">=3.13"
|
|
dependencies = [
|
|
"httpx>=0.27.0",
|
|
"loguru>=0.7.3",
|
|
"transformers>=5.0.0",
|
|
"huggingface-hub>=0.33.4",
|
|
"tiktoken>=0.12.0",
|
|
"jinja2>=3.1.0",
|
|
]
|
|
|
|
[build-system]
|
|
requires = ["hatchling"]
|
|
build-backend = "hatchling.build"
|