← back to Exo
Implement continuous batching (#1642)
131ad0ff36fff426600be3c5608eac852ba1f0da · 2026-03-09 15:04:45 +0000 · rltakashige
## Motivation
Following the changes made in #1632 !
Closes #1020
## Changes
<!-- Describe what you changed in detail -->
## Why It Works
<!-- Explain why your approach solves the problem -->
## Test Plan
### Manual Testing
<!-- Hardware: (e.g., MacBook Pro M1 Max 32GB, Mac Mini M2 16GB,
connected via Thunderbolt 4) -->
<!-- What you did: -->
<!-- - -->
### Automated Testing
<!-- Describe changes to automated tests, or how existing tests cover
this change -->
<!-- - -->
---------
Co-authored-by: Evan Quiney <evanev7@gmail.com>
Files touched
M .mlx_typings/mlx_lm/generate.pyiM .mlx_typings/mlx_lm/models/cache.pyiA bench/eval_configs/models.tomlM bench/exo_bench.pyA bench/exo_eval.pyM bench/harness.pyA bench/parallel_requests.pyM bench/pyproject.tomlA bench/vendor/__init__.pyA bench/vendor/lcb_testing_util.pyM pyproject.tomlM python/parts.nixM src/exo/main.pyM src/exo/master/adapters/chat_completions.pyM src/exo/shared/constants.pyM src/exo/shared/types/api.pyM src/exo/shared/types/tasks.pyM src/exo/shared/types/text_generation.pyA src/exo/worker/engines/mlx/generator/batch_generate.pyM src/exo/worker/engines/mlx/generator/generate.pyA src/exo/worker/engines/mlx/tests/test_batch_generate.pyM src/exo/worker/engines/mlx/utils_mlx.pyM src/exo/worker/runner/bootstrap.pyM src/exo/worker/runner/image_models/runner.pyM src/exo/worker/runner/llm_inference/batch_generator.pyM src/exo/worker/runner/llm_inference/model_output_parsers.pyM src/exo/worker/runner/llm_inference/runner.pyM src/exo/worker/runner/runner_supervisor.pyA src/exo/worker/tests/unittests/test_mlx/test_batch_vs_generate.pyM src/exo/worker/tests/unittests/test_mlx/test_prefix_cache_architectures.pyM src/exo/worker/tests/unittests/test_runner/test_event_ordering.pyM uv.lock
Diff
commit 131ad0ff36fff426600be3c5608eac852ba1f0da
Author: rltakashige <rl.takashige@gmail.com>
Date: Mon Mar 9 15:04:45 2026 +0000
Implement continuous batching (#1642)
## Motivation
Following the changes made in #1632 !
Closes #1020
## Changes
<!-- Describe what you changed in detail -->
## Why It Works
<!-- Explain why your approach solves the problem -->
## Test Plan
### Manual Testing
<!-- Hardware: (e.g., MacBook Pro M1 Max 32GB, Mac Mini M2 16GB,
connected via Thunderbolt 4) -->
<!-- What you did: -->
<!-- - -->
### Automated Testing
<!-- Describe changes to automated tests, or how existing tests cover
this change -->
<!-- - -->
---------
Co-authored-by: Evan Quiney <evanev7@gmail.com>
---
.mlx_typings/mlx_lm/generate.pyi | 71 +-
.mlx_typings/mlx_lm/models/cache.pyi | 8 +
bench/eval_configs/models.toml | 292 +++++
bench/exo_bench.py | 176 ++-
bench/exo_eval.py | 1375 ++++++++++++++++++++
bench/harness.py | 19 +-
bench/parallel_requests.py | 255 ++++
bench/pyproject.toml | 5 +
bench/vendor/__init__.py | 0
bench/vendor/lcb_testing_util.py | 593 +++++++++
pyproject.toml | 3 +-
python/parts.nix | 29 +-
src/exo/main.py | 10 +
src/exo/master/adapters/chat_completions.py | 1 +
src/exo/shared/constants.py | 2 +
src/exo/shared/types/api.py | 1 +
src/exo/shared/types/tasks.py | 3 +
src/exo/shared/types/text_generation.py | 1 +
.../worker/engines/mlx/generator/batch_generate.py | 383 ++++++
src/exo/worker/engines/mlx/generator/generate.py | 25 +-
.../engines/mlx/tests/test_batch_generate.py | 268 ++++
src/exo/worker/engines/mlx/utils_mlx.py | 54 +
src/exo/worker/runner/bootstrap.py | 4 +
src/exo/worker/runner/image_models/runner.py | 9 +-
.../worker/runner/llm_inference/batch_generator.py | 365 ++++--
.../runner/llm_inference/model_output_parsers.py | 94 +-
src/exo/worker/runner/llm_inference/runner.py | 64 +-
src/exo/worker/runner/runner_supervisor.py | 3 +-
.../unittests/test_mlx/test_batch_vs_generate.py | 389 ++++++
.../test_mlx/test_prefix_cache_architectures.py | 2 +
.../unittests/test_runner/test_event_ordering.py | 53 +-
uv.lock | 719 +++++++++-
32 files changed, 5028 insertions(+), 248 deletions(-)
diff --git a/.mlx_typings/mlx_lm/generate.pyi b/.mlx_typings/mlx_lm/generate.pyi
index 9934e6ea..02904d81 100644
--- a/.mlx_typings/mlx_lm/generate.pyi
+++ b/.mlx_typings/mlx_lm/generate.pyi
@@ -254,7 +254,14 @@ class BatchResponse:
texts: List[str]
stats: BatchStats
-
+ caches: Optional[List[List[Any]]]
+
+def _left_pad_prompts(prompts: Any, max_length: Optional[int] = ...) -> mx.array: ...
+def _right_pad_prompts(prompts: Any, max_length: Optional[int] = ...) -> mx.array: ...
+def _make_cache(
+ model: Any, left_padding: Any, max_kv_size: Optional[int]
+) -> List[Any]: ...
+def _merge_caches(caches: Any) -> List[Any]: ...
@dataclass
class Batch:
uids: List[int]
@@ -263,39 +270,71 @@ class Batch:
max_tokens: List[int]
num_tokens: List[int]
cache: List[Any]
- def __len__(self): # -> int:
- ...
- def filter(self, keep_idx: List[int]): # -> None:
- ...
- def extend(self, other): # -> None:
- ...
+ samplers: List[Any]
+ logits_processors: List[Any]
+ tokens: List[mx.array]
+ def __len__(self) -> int: ...
+ def filter(self, keep_idx: List[int]) -> None: ...
+ def extend(self, other: "Batch") -> None: ...
+ def extract_cache(self, idx: int) -> List[Any]: ...
class BatchGenerator:
+ model: Any
+ max_kv_size: Optional[int]
+ prefill_step_size: int
+ unprocessed_prompts: List[Any]
+ active_batch: Optional[Batch]
+ prompt_progress_callback: Callable[[List[Tuple[int, int, int]]], None]
+ _stats: BatchStats
+
@dataclass
class Response:
uid: int
token: int
logprobs: mx.array
finish_reason: Optional[str]
+ prompt_cache: Any
def __init__(
self,
- model,
+ model: Any,
max_tokens: int = ...,
- stop_tokens: Optional[set] = ...,
+ stop_tokens: Optional[set[int]] = ...,
sampler: Optional[Callable[[mx.array], mx.array]] = ...,
+ logits_processors: Optional[
+ List[Callable[[mx.array, mx.array], mx.array]]
+ ] = ...,
completion_batch_size: int = ...,
prefill_batch_size: int = ...,
prefill_step_size: int = ...,
+ prompt_progress_callback: Optional[
+ Callable[[List[Tuple[int, int, int]]], None]
+ ] = ...,
+ max_kv_size: Optional[int] = ...,
) -> None: ...
+ def close(self) -> None: ...
def insert(
- self, prompts, max_tokens: Union[List[int], int, None] = ...
- ): # -> list[Any]:
- ...
- def stats(self): # -> BatchStats:
- ...
- def next(self): # -> list[Any]:
- ...
+ self,
+ prompts: Any,
+ max_tokens: Union[List[int], int, None] = ...,
+ caches: Any = ...,
+ samplers: Optional[List[Any]] = ...,
+ logits_processors: Optional[List[Any]] = ...,
+ ) -> List[int]: ...
+ def remove(
+ self, uids: List[int], return_prompt_caches: bool = ...
+ ) -> Optional[dict[int, List[Any]]]: ...
+ def stats(self) -> BatchStats: ...
+ def next(self) -> List[Response]: ...
+ def _process_prompts(self, prompts: List[Any]) -> Batch: ...
+ def _step(
+ self,
+ input_tokens: mx.array,
+ prompt_cache: List[Any],
+ samplers: Optional[List[Any]],
+ logits_processors: Optional[List[Any]],
+ tokens: List[mx.array],
+ ) -> Tuple[mx.array, List[mx.array]]: ...
def batch_generate(
model,
diff --git a/.mlx_typings/mlx_lm/models/cache.pyi b/.mlx_typings/mlx_lm/models/cache.pyi
index 57f9aa7e..1a934f40 100644
--- a/.mlx_typings/mlx_lm/models/cache.pyi
+++ b/.mlx_typings/mlx_lm/models/cache.pyi
@@ -170,7 +170,15 @@ class KVCache(_BaseCache):
class RotatingKVCache(_BaseCache):
step = ...
+ keys: mx.array | None
+ values: mx.array | None
+ keep: int
+ max_size: int
+ _idx: int
def __init__(self, max_size, keep=...) -> None: ...
+ def _trim(
+ self, trim_size: int, v: mx.array, append: mx.array | None = ...
+ ) -> mx.array: ...
def update_and_fetch(
self, keys, values
): # -> tuple[array | Any, array | Any] | tuple[array | Any, array | Any | None]:
diff --git a/bench/eval_configs/models.toml b/bench/eval_configs/models.toml
new file mode 100644
index 00000000..477e96cd
--- /dev/null
+++ b/bench/eval_configs/models.toml
@@ -0,0 +1,292 @@
+# Model evaluation configurations for exo_eval.
+#
+# Each [[model]] entry uses `patterns` — a list of substrings matched
+# against the model_id. First matching entry wins.
+#
+# Required fields:
+# name, patterns, reasoning
+#
+# Optional per-model overrides (CLI flags take priority over these):
+# temperature, top_p, max_tokens, reasoning_effort
+#
+# Fallback defaults (when no per-model config):
+# reasoning: temperature=1.0, max_tokens=131072, reasoning_effort="high"
+# non-reasoning: temperature=0.0, max_tokens=16384
+#
+# All per-model values below are sourced from official model cards,
+# generation_config.json files, and vendor documentation.
+
+# ─── Qwen3.5 (Feb 2026) ─────────────────────────────────────────────
+# Source: HuggingFace model cards (Qwen/Qwen3.5-*)
+# 35B-A3B thinking general: temp=1.0, top_p=0.95, top_k=20
+# 397B thinking: temp=0.6, top_p=0.95, top_k=20
+# Non-thinking: temp=0.7, top_p=0.8, top_k=20
+# max_tokens: 32768 general, 81920 for complex math/code
+
+[[model]]
+name = "Qwen3.5 2B"
+patterns = ["Qwen3.5-2B"]
+reasoning = true
+temperature = 0.6
+top_p = 0.95
+max_tokens = 81920
+
+[[model]]
+name = "Qwen3.5 9B"
+patterns = ["Qwen3.5-9B"]
+reasoning = true
+temperature = 0.6
+top_p = 0.95
+max_tokens = 81920
+
+[[model]]
+name = "Qwen3.5 27B"
+patterns = ["Qwen3.5-27B"]
+reasoning = true
+temperature = 0.6
+top_p = 0.95
+max_tokens = 81920
+
+[[model]]
+name = "Qwen3.5 35B A3B"
+patterns = ["Qwen3.5-35B-A3B"]
+reasoning = true
+temperature = 1.0
+top_p = 0.95
+max_tokens = 81920
+
+[[model]]
+name = "Qwen3.5 122B A10B"
+patterns = ["Qwen3.5-122B-A10B"]
+reasoning = true
+temperature = 0.6
+top_p = 0.95
+max_tokens = 81920
+
+[[model]]
+name = "Qwen3.5 397B A17B"
+patterns = ["Qwen3.5-397B-A17B"]
+reasoning = true
+temperature = 0.6
+top_p = 0.95
+max_tokens = 81920
+
+# ─── Qwen3 (Apr 2025) ───────────────────────────────────────────────
+# Source: HuggingFace model cards (Qwen/Qwen3-*)
+# Thinking: temp=0.6, top_p=0.95, top_k=20
+# Non-thinking: temp=0.7, top_p=0.8, top_k=20
+# max_tokens: 32768 general, 38912 for complex math/code
+
+[[model]]
+name = "Qwen3 0.6B"
+patterns = ["Qwen3-0.6B"]
+reasoning = true
+temperature = 0.6
+top_p = 0.95
+max_tokens = 38912
+
+[[model]]
+name = "Qwen3 30B A3B"
+patterns = ["Qwen3-30B-A3B"]
+reasoning = true
+temperature = 0.6
+top_p = 0.95
+max_tokens = 38912
+
+[[model]]
+name = "Qwen3 235B A22B"
+patterns = ["Qwen3-235B-A22B"]
+reasoning = true
+temperature = 0.6
+top_p = 0.95
+max_tokens = 38912
+
+[[model]]
+name = "Qwen3 Next 80B Thinking"
+patterns = ["Qwen3-Next-80B-A3B-Thinking"]
+reasoning = true
+temperature = 0.6
+top_p = 0.95
+max_tokens = 38912
+
+[[model]]
+name = "Qwen3 Next 80B Instruct"
+patterns = ["Qwen3-Next-80B-A3B-Instruct"]
+reasoning = false
+temperature = 0.7
+top_p = 0.8
+max_tokens = 16384
+
+[[model]]
+name = "Qwen3 Coder 480B"
+patterns = ["Qwen3-Coder-480B"]
+reasoning = false
+temperature = 0.7
+top_p = 0.8
+max_tokens = 16384
+
+[[model]]
+name = "Qwen3 Coder Next"
+patterns = ["Qwen3-Coder-Next"]
+reasoning = false
+temperature = 0.7
+top_p = 0.8
+max_tokens = 16384
+
+# ─── GPT-OSS (OpenAI) ───────────────────────────────────────────────
+# Source: OpenAI GitHub README + HuggingFace discussion #21
+# temp=1.0, top_p=1.0, NO top_k, NO repetition_penalty
+# reasoning_effort supported: low/medium/high
+
+[[model]]
+name = "GPT-OSS 20B"
+patterns = ["gpt-oss-20b"]
+reasoning = true
+temperature = 1.0
+top_p = 1.0
+
+[[model]]
+name = "GPT-OSS 120B"
+patterns = ["gpt-oss-120b"]
+reasoning = true
+temperature = 1.0
+top_p = 1.0
+
+# ─── DeepSeek ────────────────────────────────────────────────────────
+# Source: https://api-docs.deepseek.com/quick_start/parameter_settings
+# Coding/Math: temp=0.0, General: temp=1.3, Creative: temp=1.5
+# NOTE: DeepSeek API applies nonlinear temp mapping. These are API values.
+# When running model directly: API temp 1.0 = model temp 0.3
+# We use temp=0.0 for eval (coding/math focus).
+
+[[model]]
+name = "DeepSeek V3.1"
+patterns = ["DeepSeek-V3.1"]
+reasoning = true
+temperature = 0.0
+
+# ─── GLM (ZhipuAI / THUDM) ──────────────────────────────────────────
+# Source: HuggingFace model cards + generation_config.json + docs.z.ai
+# GLM 4.5+: temp=1.0, top_p=0.95
+# Reasoning tasks: 131072 max_tokens; coding/SWE tasks: temp=0.7
+
+[[model]]
+name = "GLM-5"
+patterns = ["GLM-5"]
+reasoning = true
+temperature = 1.0
+top_p = 0.95
+max_tokens = 131072
+
+[[model]]
+name = "GLM 4.5 Air"
+patterns = ["GLM-4.5-Air"]
+reasoning = true
+temperature = 1.0
+top_p = 0.95
+
+[[model]]
+name = "GLM 4.7"
+patterns = ["GLM-4.7-"]
+reasoning = true
+temperature = 1.0
+top_p = 0.95
+max_tokens = 131072
+# Note: matches both GLM-4.7 and GLM-4.7-Flash
+
+# ─── Kimi (Moonshot AI) ─────────────────────────────────────────────
+# Source: HuggingFace model cards (moonshotai/Kimi-K2-*)
+# K2-Instruct: temp=0.6
+# K2-Thinking: temp=1.0, max_length=262144
+# K2.5: thinking temp=1.0, top_p=0.95; instant temp=0.6, top_p=0.95
+
+[[model]]
+name = "Kimi K2 Thinking"
+patterns = ["Kimi-K2-Thinking"]
+reasoning = true
+temperature = 1.0
+max_tokens = 131072
+
+[[model]]
+name = "Kimi K2.5"
+patterns = ["Kimi-K2.5"]
+reasoning = true
+temperature = 1.0
+top_p = 0.95
+max_tokens = 131072
+
+[[model]]
+name = "Kimi K2 Instruct"
+patterns = ["Kimi-K2-Instruct"]
+reasoning = false
+temperature = 0.6
+
+# ─── MiniMax ─────────────────────────────────────────────────────────
+# Source: HuggingFace model cards + generation_config.json
+# All models: temp=1.0, top_p=0.95, top_k=40
+
+[[model]]
+name = "MiniMax M2.5"
+patterns = ["MiniMax-M2.5"]
+reasoning = true
+temperature = 1.0
+top_p = 0.95
+
+[[model]]
+name = "MiniMax M2.1"
+patterns = ["MiniMax-M2.1"]
+reasoning = true
+temperature = 1.0
+top_p = 0.95
+
+# ─── Step (StepFun) ─────────────────────────────────────────────────
+# Source: HuggingFace model card (stepfun-ai/Step-3.5-Flash)
+# Reasoning: temp=1.0, top_p=0.95
+# General chat: temp=0.6, top_p=0.95
+# We use reasoning settings for eval.
+
+[[model]]
+name = "Step 3.5 Flash"
+patterns = ["Step-3.5-Flash"]
+reasoning = true
+temperature = 1.0
+top_p = 0.95
+
+# ─── Llama (Meta) ───────────────────────────────────────────────────
+# Source: generation_config.json + meta-llama/llama-models generation.py
+# All variants: temp=0.6, top_p=0.9
+
+[[model]]
+name = "Llama 3.2 1B"
+patterns = ["Llama-3.2-1B"]
+reasoning = false
+temperature = 0.6
+top_p = 0.9
+
+[[model]]
+name = "Llama 3.2 3B"
+patterns = ["Llama-3.2-3B"]
+reasoning = false
+temperature = 0.6
+top_p = 0.9
+
+[[model]]
+name = "Llama 3.1 8B"
+patterns = ["Llama-3.1-8B", "Meta-Llama-3.1-8B"]
+reasoning = false
+temperature = 0.6
+top_p = 0.9
+
+[[model]]
+name = "Llama 3.1 70B"
+patterns = ["Llama-3.1-70B", "Meta-Llama-3.1-70B"]
+reasoning = false
+temperature = 0.6
+top_p = 0.9
+
+[[model]]
+name = "Llama 3.3 70B"
+patterns = ["Llama-3.3-70B", "llama-3.3-70b"]
+reasoning = false
+temperature = 0.6
+top_p = 0.9
diff --git a/bench/exo_bench.py b/bench/exo_bench.py
index 001f5c56..13e59f8d 100644
--- a/bench/exo_bench.py
+++ b/bench/exo_bench.py
@@ -24,6 +24,7 @@ import json
import sys
import time
from collections.abc import Callable
+from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
from statistics import mean
from typing import Any
@@ -252,6 +253,12 @@ def main() -> int:
ap.add_argument(
"--repeat", type=int, default=1, help="Repetitions per (pp,tg) pair."
)
+ ap.add_argument(
+ "--concurrency",
+ nargs="+",
+ default=["1"],
+ help="Concurrency levels (ints). Accepts commas. E.g. --concurrency 1,2,4,8. Default 1.",
+ )
ap.add_argument(
"--warmup",
type=int,
@@ -282,6 +289,10 @@ def main() -> int:
if args.repeat <= 0:
logger.error("--repeat must be >= 1")
return 2
+ concurrency_list = parse_int_list(args.concurrency)
+ if not concurrency_list or any(c <= 0 for c in concurrency_list):
+ logger.error("--concurrency values must be >= 1")
+ return 2
# Log pairing mode
use_combinations = args.all_combinations or len(pp_list) != len(tg_list)
@@ -293,7 +304,9 @@ def main() -> int:
logger.info(f"pp/tg mode: tandem (zip) - {len(pp_list)} pairs")
client = ExoClient(args.host, args.port, timeout_s=args.timeout)
- short_id, full_model_id = resolve_model_short_id(client, args.model)
+ short_id, full_model_id = resolve_model_short_id(
+ client, args.model, force_download=args.force_download
+ )
tokenizer = load_tokenizer_for_bench(full_model_id)
if tokenizer is None:
@@ -392,52 +405,123 @@ def main() -> int:
pp_tg_pairs = list(zip(pp_list, tg_list, strict=True))
for pp, tg in pp_tg_pairs:
- runs: list[dict[str, Any]] = []
- for r in range(args.repeat):
- time.sleep(3)
- try:
- row, actual_pp_tokens = run_one_completion(
- client, full_model_id, pp, tg, prompt_sizer
+ for concurrency in concurrency_list:
+ logger.info(f"--- pp={pp} tg={tg} concurrency={concurrency} ---")
+ runs: list[dict[str, Any]] = []
+ for r in range(args.repeat):
+ time.sleep(3)
+
+ if concurrency <= 1:
+ # Sequential: single request
+ try:
+ row, actual_pp_tokens = run_one_completion(
+ client, full_model_id, pp, tg, prompt_sizer
+ )
+ except Exception as e:
+ logger.error(e)
+ continue
+ row.update(
+ {
+ "model_short_id": short_id,
+ "model_id": full_model_id,
+ "placement_sharding": sharding,
+ "placement_instance_meta": instance_meta,
+ "placement_nodes": n_nodes,
+ "instance_id": instance_id,
+ "pp_tokens": actual_pp_tokens,
+ "tg": tg,
+ "repeat_index": r,
+ "concurrency": 1,
+ **(
+ {"download_duration_s": download_duration_s}
+ if download_duration_s is not None
+ else {}
+ ),
+ }
+ )
+ runs.append(row)
+ all_rows.append(row)
+ else:
+ # Concurrent: fire N requests in parallel
+ # Each thread gets its own ExoClient (separate HTTP connection)
+ batch_results: list[tuple[dict[str, Any], int]] = []
+ batch_errors = 0
+
+ def _run_concurrent(
+ idx: int, *, _pp: int = pp, _tg: int = tg
+ ) -> tuple[dict[str, Any], int]:
+ c = ExoClient(
+ args.host, args.port, timeout_s=args.timeout
+ )
+ return run_one_completion(
+ c, full_model_id, _pp, _tg, prompt_sizer
+ )
+
+ with ThreadPoolExecutor(max_workers=concurrency) as pool:
+ futures = {
+ pool.submit(_run_concurrent, i): i
+ for i in range(concurrency)
+ }
+ for fut in as_completed(futures):
+ try:
+ batch_results.append(fut.result())
+ except Exception as e:
+ logger.error(f"Concurrent request failed: {e}")
+ batch_errors += 1
+
+ for idx, (row, actual_pp_tokens) in enumerate(
+ batch_results
+ ):
+ row.update(
+ {
+ "model_short_id": short_id,
+ "model_id": full_model_id,
+ "placement_sharding": sharding,
+ "placement_instance_meta": instance_meta,
+ "placement_nodes": n_nodes,
+ "instance_id": instance_id,
+ "pp_tokens": actual_pp_tokens,
+ "tg": tg,
+ "repeat_index": r,
+ "concurrency": concurrency,
+ "concurrent_index": idx,
+ **(
+ {"download_duration_s": download_duration_s}
+ if download_duration_s is not None
+ else {}
+ ),
+ }
+ )
+ runs.append(row)
+ all_rows.append(row)
+
+ if batch_results:
+ agg_gen_tps = sum(
+ x["stats"]["generation_tps"]
+ for x, _ in batch_results
+ if x["stats"]["generation_tps"] > 0
+ )
+ logger.info(
+ f"[concurrent {concurrency}x] "
+ f"agg_gen_tps={agg_gen_tps:.2f} "
+ f"errors={batch_errors}"
+ )
+
+ if runs:
+ prompt_tps = mean(x["stats"]["prompt_tps"] for x in runs)
+ gen_tps = mean(x["stats"]["generation_tps"] for x in runs)
+ ptok = mean(x["stats"]["prompt_tokens"] for x in runs)
+ gtok = mean(x["stats"]["generation_tokens"] for x in runs)
+ peak = mean(
+ x["stats"]["peak_memory_usage"]["inBytes"] for x in runs
+ )
+
+ logger.info(
+ f"prompt_tps={prompt_tps:.2f} gen_tps={gen_tps:.2f} "
+ f"prompt_tokens={ptok} gen_tokens={gtok} "
+ f"peak_memory={format_peak_memory(peak)}\n"
)
- except Exception as e:
- logger.error(e)
- continue
- row.update(
- {
- "model_short_id": short_id,
- "model_id": full_model_id,
- "placement_sharding": sharding,
- "placement_instance_meta": instance_meta,
- "placement_nodes": n_nodes,
- "instance_id": instance_id,
- "pp_tokens": actual_pp_tokens,
- "tg": tg,
- "repeat_index": r,
- **(
- {"download_duration_s": download_duration_s}
- if download_duration_s is not None
- else {}
- ),
- }
- )
- runs.append(row)
- all_rows.append(row)
-
- if runs:
- prompt_tps = mean(x["stats"]["prompt_tps"] for x in runs)
- gen_tps = mean(x["stats"]["generation_tps"] for x in runs)
- ptok = mean(x["stats"]["prompt_tokens"] for x in runs)
- gtok = mean(x["stats"]["generation_tokens"] for x in runs)
- peak = mean(
- x["stats"]["peak_memory_usage"]["inBytes"] for x in runs
- )
-
- logger.info(
- f"prompt_tps={prompt_tps:.2f} gen_tps={gen_tps:.2f} "
- f"prompt_tokens={ptok} gen_tokens={gtok} "
- f"peak_memory={format_peak_memory(peak)}\n"
- )
- time.sleep(2)
+ time.sleep(2)
finally:
try:
client.request_json("DELETE", f"/instance/{instance_id}")
diff --git a/bench/exo_eval.py b/bench/exo_eval.py
new file mode 100644
index 00000000..e0770e75
--- /dev/null
+++ b/bench/exo_eval.py
@@ -0,0 +1,1375 @@
+# type: ignore
+#!/usr/bin/env python3
+"""Quality evaluation for exo — matches Artificial Analysis methodology.
+
+Runs LLM benchmarks against exo's OpenAI-compatible API using the same
+prompts, temperature settings, and answer extraction as Artificial Analysis.
+
+Supported benchmarks:
+ gpqa_diamond - Graduate-level science QA (198 questions, 4-choice MC)
+ mmlu_pro - Multi-task language understanding (12K questions, 10-choice MC)
+ aime_2024 - Math olympiad 2024 (30 problems, integer answers)
+ aime_2025 - Math olympiad 2025 (30 problems, integer answers)
+ humaneval - Python code generation (164 problems, pass@1)
+ livecodebench - Competitive programming (880+ problems, pass@1)
+
+Model configs in eval_configs/models.toml auto-detect reasoning/non-reasoning
+settings per model. Override with --reasoning / --no-reasoning.
+
+Usage:
+ uv run python exo_eval.py --model <model-id> --tasks gpqa_diamond
+ uv run python exo_eval.py --model <model-id> --tasks humaneval,livecodebench --limit 50
+ uv run python exo_eval.py --model <model-id> --tasks gpqa_diamond --compare-concurrency 1,4
+
+References:
+ https://artificialanalysis.ai/methodology/intelligence-benchmarking
+"""
+
+from __future__ import annotations
+
+import argparse
+import asyncio
+import contextlib
+import json
+import multiprocessing
+import random
+import re
+import sys
+import time
+import tomllib
+from dataclasses import dataclass
+from pathlib import Path
+from typing import Any
+
+import httpx
+from harness import (
+ ExoClient,
+ ExoHttpError,
+ add_common_instance_args,
+ instance_id_from_instance,
+ nodes_used_in_instance,
+ resolve_model_short_id,
+ run_planning_phase,
+ settle_and_fetch_placements,
+ wait_for_instance_gone,
+ wait_for_instance_ready,
+)
+from loguru import logger
+
+# ---------------------------------------------------------------------------
+# Artificial Analysis constants
+# ---------------------------------------------------------------------------
+
+MAX_RETRIES = 30
+DEFAULT_MAX_TOKENS = 16_384
+REASONING_MAX_TOKENS = 131_072
+TEMPERATURE_NON_REASONING = 0.0
+TEMPERATURE_REASONING = 1.0
+
+# MC answer extraction: 8 fallback regex patterns.
+# All patterns are tried; the match at the latest text position wins
+# (handles models that self-correct during reasoning).
+_MC_PATTERNS: list[re.Pattern[str]] = [
+ re.compile(
+ r"(?i)[\*\_]{0,2}Answer[\*\_]{0,2}\s*:[\s\*\_]{0,2}\s*([A-Z])(?![a-zA-Z0-9])"
+ ),
+ re.compile(r"\\boxed\{[^}]*([A-Z])[^}]*\}"),
+ re.compile(r"(?i)answer is ([a-zA-Z])"),
+ re.compile(r"(?i)answer is \\\(([a-zA-Z])"),
+ re.compile(r"([A-Z])\)\s*[^A-Z]*$"),
+ re.compile(r"([A-Z])\s+is\s+the\s+correct\s+answer"),
+ re.compile(r"([A-Z])\s*$"),
+ re.compile(r"([A-Z])\s*\."),
+]
+
+# Code extraction: last ```python ... ``` block (AA regex)
+_CODE_BLOCK_RE = re.compile(r"```(?:python|Python)?\s*\n(.*?)```", re.DOTALL)
+
+
+# ---------------------------------------------------------------------------
+# Model config loading
+# ---------------------------------------------------------------------------
+
+
+def load_model_config(model_id: str) -> dict[str, Any] | None:
+ """Look up model in eval_configs/models.toml. Returns config dict or None."""
+ config_path = Path(__file__).resolve().parent / "eval_configs" / "models.toml"
+ if not config_path.exists():
+ return None
+ with open(config_path, "rb") as f:
+ data = tomllib.load(f)
+ for entry in data.get("model", []):
+ patterns = entry.get("patterns", [])
+ if any(p in model_id for p in patterns):
+ return entry
+ return None
+
+
+# ---------------------------------------------------------------------------
+# Answer extraction
+# ---------------------------------------------------------------------------
+
+
+def extract_mc_answer(text: str, valid_letters: str = "ABCD") -> str | None:
+ """Extract MC answer. Last match by text position wins."""
+ valid_set = set(valid_letters)
+ best: tuple[int, str] | None = None
+ for pattern in _MC_PATTERNS:
+ for m in pattern.finditer(text):
+ letter = m.group(1).upper()
+ if letter in valid_set:
+ pos = m.start()
+ if best is None or pos >= best[0]:
+ best = (pos, letter)
+ return best[1] if best else None
+
+
+def extract_boxed_answer(text: str) -> str | None:
+ r"""Extract content from the last \boxed{...}."""
+ matches: list[str] = []
+ idx = 0
+ while True:
+ pos = text.find("\\boxed{", idx)
+ if pos < 0:
+ break
+ depth = 0
+ i = pos + len("\\boxed{")
+ start = i
+ while i < len(text):
+ if text[i] == "{":
+ depth += 1
+ elif text[i] == "}":
+ if depth == 0:
+ matches.append(text[start:i])
+ break
+ depth -= 1
+ i += 1
+ idx = i + 1 if i < len(text) else len(text)
+ return matches[-1].strip() if matches else None
+
+
+def extract_code_block(text: str, preserve_indent: bool = False) -> str | None:
+ """Extract the last Python code block from markdown response.
+
+ If preserve_indent is True, only strip trailing whitespace (keeps leading
+ indentation intact — needed for HumanEval function-body completions).
+ """
+ matches = _CODE_BLOCK_RE.findall(text)
+ if matches:
+ raw = matches[-1]
+ return raw.rstrip() if preserve_indent else raw.strip()
+ # Fallback: try raw code after last ```
+ lines = text.split("\n")
+ backtick_lines = [i for i, line in enumerate(lines) if "```" in line]
+ if len(backtick_lines) >= 2:
+ return "\n".join(lines[backtick_lines[-2] + 1 : backtick_lines[-1]])
+ return None
+
+
+def check_aime_answer(extracted: str, gold: int) -> bool:
+ """Check if extracted AIME answer matches gold integer."""
+ try:
+ return int(extracted.strip()) == gold
+ except ValueError:
+ pass
+ try:
+ from math_verify import parse, verify
+
+ return verify(parse(str(gold)), parse(extracted))
+ except Exception:
+ return False
+
+
+# ---------------------------------------------------------------------------
+# Code execution — official evaluation harnesses
+# ---------------------------------------------------------------------------
+
+# LiveCodeBench: vendored from https://github.com/LiveCodeBench/LiveCodeBench
+# run_test() must execute in a child process because reliability_guard()
+# permanently disables OS functions (os.kill, subprocess.Popen, etc.).
+
+
+def _lcb_worker(
+ sample: dict,
+ code: str,
+ timeout: int,
+ result_holder: list[Any],
+ metadata_holder: list[Any],
+) -> None:
+ """Target for multiprocessing.Process — runs vendored LCB run_test."""
+ from vendor.lcb_testing_util import run_test
+
+ try:
+ results, metadata = run_test(sample, test=code, debug=False, timeout=timeout)
+ result_holder.append(results)
+ metadata_holder.append(metadata)
+ except Exception as e:
+ result_holder.append([-4])
+ metadata_holder.append({"error_code": -4, "error_message": str(e)})
+
+
+def run_livecodebench_test(
+ code: str,
+ sample: dict,
+ timeout: int = 6,
+) -> tuple[bool, str]:
+ """Run LCB evaluation in a subprocess. Returns (passed, diagnostic_info)."""
+ manager = multiprocessing.Manager()
+ result_holder = manager.list()
+ metadata_holder = manager.list()
+
+ proc = multiprocessing.Process(
+ target=_lcb_worker,
+ args=(sample, code, timeout, result_holder, metadata_holder),
+ )
+ proc.start()
+
+ # Global timeout: (per-test timeout + 1) * num_tests + 5
+ num_tests = len(json.loads(sample["input_output"]).get("inputs", []))
+ global_timeout = (timeout + 1) * num_tests + 5
+ proc.join(timeout=global_timeout)
+
+ if proc.is_alive():
+ proc.kill()
+ proc.join()
+ return False, "Global timeout exceeded"
+
+ if not result_holder:
+ return False, "No results returned from worker"
+
+ results = list(result_holder[0])
+ metadata = dict(metadata_holder[0]) if metadata_holder else {}
+
+ # LCB convention: True = pass, negative int = failure code
+ all_passed = all(r is True or r == 1 for r in results)
+ if all_passed:
+ return True, ""
+
+ diag = metadata.get("error_message", "")
+ if not diag and "output" in metadata:
+ diag = f"Got {metadata['output']}, expected {metadata.get('expected', '?')}"
+ return False, diag
+
+
+def run_humaneval_test(
+ problem: dict, completion: str, timeout: float = 10.0
+) -> tuple[bool, str]:
+ """Run HumanEval evaluation using the official human_eval package."""
+ from human_eval.execution import check_correctness
+
+ result = check_correctness(problem, completion, timeout)
+ passed = result["passed"]
+ diag = "" if passed else result.get("result", "failed")
+ return passed, diag
+
+
+# ---------------------------------------------------------------------------
+# Benchmark definitions
+# ---------------------------------------------------------------------------
+
+
+@dataclass
+class QuestionResult:
+ question_id: int
+ prompt: str
+ response: str
+ extracted_answer: str | None
+ gold_answer: str
+ correct: bool
+ error: str | None = None
+ prompt_tokens: int = 0
+ completion_tokens: int = 0
+ reasoning_tokens: int = 0
+ elapsed_s: float = 0.0
+
+
+@dataclass
+class BenchmarkConfig:
+ name: str
+ description: str
+ dataset_name: str
+ dataset_config: str | None
+ split: str
+ kind: str # "mc", "math", "code"
+
+
+BENCHMARKS: dict[str, BenchmarkConfig] = {
+ "gpqa_diamond": BenchmarkConfig(
+ name="gpqa_diamond",
+ description="Graduate-level science QA (198 Q, 4-choice MC)",
+ dataset_name="Idavidrein/gpqa",
+ dataset_config="gpqa_diamond",
+ split="train",
+ kind="mc",
+ ),
+ "mmlu_pro": BenchmarkConfig(
+ name="mmlu_pro",
+ description="Multi-task language understanding (12K Q, 10-choice MC)",
+ dataset_name="TIGER-Lab/MMLU-Pro",
+ dataset_config=None,
+ split="test",
+ kind="mc",
+ ),
+ "aime_2024": BenchmarkConfig(
+ name="aime_2024",
+ description="Math olympiad 2024 (30 problems, integer answers)",
+ dataset_name="HuggingFaceH4/aime_2024",
+ dataset_config=None,
+ split="train",
+ kind="math",
+ ),
+ "aime_2025": BenchmarkConfig(
+ name="aime_2025",
+ description="Math olympiad 2025 (30 problems, integer answers)",
+ dataset_name="MathArena/aime_2025",
+ dataset_config=None,
+ split="train",
+ kind="math",
+ ),
+ "humaneval": BenchmarkConfig(
+ name="humaneval",
+ description="Python code generation (164 problems, pass@1)",
+ dataset_name="openai/openai_humaneval",
+ dataset_config=None,
+ split="test",
+ kind="code",
+ ),
+ "livecodebench": BenchmarkConfig(
+ name="livecodebench",
+ description="Competitive programming (880+ problems, pass@1)",
+ dataset_name="livecodebench/code_generation_lite",
+ dataset_config=None,
+ split="test",
+ kind="code",
+ ),
+}
+
+
+# ---------------------------------------------------------------------------
+# Prompt formatters
+# ---------------------------------------------------------------------------
+
+_GPQA_INSTRUCTION = (
+ "Answer the following multiple choice question. "
+ "The last line of your response should be in the following format: "
+ "'Answer: A/B/C/D' (e.g. 'Answer: A')."
+)
+
+_MMLU_PRO_INSTRUCTION = (
+ "Answer the following multiple choice question. "
+ "The last line of your response should be in the following format: "
+ "'Answer: A/B/C/D/E/F/G/H/I/J' (e.g. 'Answer: A')."
+)
+
+_AIME_INSTRUCTION = (
+ "Solve the following math problem step by step. "
+ "Put your answer inside \\boxed{}.\n"
+ "Remember to put your answer inside \\boxed{}."
+)
+
+_HUMANEVAL_INSTRUCTION = (
+ "Complete the following Python function. Return only the function body "
+ "inside a ```python code block. Do not include the function signature."
+)
+
+# LiveCodeBench: AA uses original prompts without custom system prompts
+_LCB_SYSTEM = (
+ "You are an expert Python programmer. You will be given a question "
+ "(problem specification) and will generate a correct Python program "
+ "that matches the specification and passes all tests."
+)
+
+_LCB_WITH_STARTER = (
+ "### Question:\n{question}\n\n"
+ "### Format: You will use the following starter code to write the "
+ "solution to the problem and enclose your code within delimiters.\n"
+ "```python\n{starter_code}\n```\n\n"
+ "### Answer: (use the provided format with backticks)\n"
+)
+
+_LCB_WITHOUT_STARTER = (
+ "### Question:\n{question}\n\n"
+ "### Format: Read the inputs from stdin solve the problem and write "
+ "the answer to stdout (do not directly test on the sample inputs). "
+ "Enclose your code within delimiters as follows. Ensure that when the "
+ "python program runs, it reads the inputs, runs the algorithm and "
+ "writes output to STDOUT.\n"
+ "```python\n# YOUR CODE HERE\n```\n\n"
+ "### Answer: (use the provided format with backticks)\n"
+)
+
+
+def format_gpqa_question(doc: dict, idx: int) -> tuple[str, str]:
+ """Returns (prompt, correct_letter)."""
+ correct = doc["Correct Answer"]
+ choices = [
+ correct,
+ doc["Incorrect Answer 1"],
+ doc["Incorrect Answer 2"],
+ doc["Incorrect Answer 3"],
+ ]
+ rng = random.Random(idx)
+ order = rng.sample(range(4), 4)
+ shuffled = [choices[i] for i in order]
+ correct_letter = "ABCD"[order.index(0)]
+ choices_text = "\n".join(f"{L}) {shuffled[i]}" for i, L in enumerate("ABCD"))
+ return f"{_GPQA_INSTRUCTION}\n\n{doc['Question']}\n\n{choices_text}", correct_letter
+
+
+def format_mmlu_pro_question(doc: dict) -> tuple[str, str]:
+ """Returns (prompt, correct_letter)."""
+ options = doc["options"]
+ letters = "ABCDEFGHIJ"
+ choices_text = "\n".join(f"{letters[i]}) {opt}" for i, opt in enumerate(options))
+ return f"{_MMLU_PRO_INSTRUCTION}\n\n{doc['question']}\n\n{choices_text}", doc[
+ "answer"
+ ]
+
+
+def format_aime_question(doc: dict) -> tuple[str, int]:
+ """Returns (prompt, correct_answer_int)."""
+ return f"{_AIME_INSTRUCTION}\n\n{doc['problem']}", int(doc["answer"])
+
+
+def format_humaneval_question(doc: dict) -> tuple[str, dict]:
+ """Returns (prompt, metadata_for_execution)."""
+ prompt = f"{_HUMANEVAL_INSTRUCTION}\n\n```python\n{doc['prompt']}```"
+ # Pass the full problem dict — check_correctness needs task_id, prompt,
+ # test, entry_point
+ meta = {
+ "problem": {
+ "task_id": doc["task_id"],
+ "prompt": doc["prompt"],
+ "test": doc["test"],
+ "entry_point": doc["entry_point"],
+ },
+ }
+ return prompt, meta
+
+
+def format_livecodebench_question(doc: dict) -> tuple[str, str | None, dict]:
+ """Returns (prompt, system_message, metadata_for_execution)."""
+ starter_code = doc.get("starter_code", "")
+ question_content = doc["question_content"]
+
+ if starter_code and starter_code.strip():
+ user_msg = _LCB_WITH_STARTER.format(
+ question=question_content, starter_code=starter_code
+ )
+ else:
+ user_msg = _LCB_WITHOUT_STARTER.format(question=question_content)
+
+ # Parse test cases
+ public_tests = (
+ json.loads(doc["public_test_cases"])
+ if isinstance(doc["public_test_cases"], str)
+ else doc["public_test_cases"]
+ )
+ private_tests = doc.get("private_test_cases", "[]")
+ if isinstance(private_tests, str):
+ try:
+ private_tests = json.loads(private_tests)
+ except Exception:
+ import base64
+ import pickle
+ import zlib
+
+ private_tests = json.loads(
+ pickle.loads(
+ zlib.decompress(base64.b64decode(private_tests.encode("utf-8")))
+ )
+ )
+
+ all_tests = public_tests + (
+ private_tests if isinstance(private_tests, list) else []
+ )
+ test_inputs = [t["input"] for t in all_tests]
+ test_outputs = [t["output"] for t in all_tests]
+
+ metadata = doc.get("metadata", "{}")
+ if isinstance(metadata, str):
+ metadata = json.loads(metadata)
+ func_name = metadata.get("func_name")
+
+ # Build the sample dict in official LCB format for run_test()
+ input_output: dict[str, Any] = {
+ "inputs": test_inputs,
+ "outputs": test_outputs,
+ }
+ if func_name:
+ input_output["fn_name"] = func_name
+
+ meta = {
+ "sample": {"input_output": json.dumps(input_output)},
+ }
+ return user_msg, _LCB_SYSTEM, meta
+
+
+# ---------------------------------------------------------------------------
+# API client with retries
+# ---------------------------------------------------------------------------
+
+
+@dataclass
+class ApiResult:
+ content: str
+ prompt_tokens: int
+ completion_tokens: int
+ reasoning_tokens: int
+
+
+async def _call_api(
+ client: httpx.AsyncClient,
+ base_url: str,
+ model: str,
+ prompt: str,
+ temperature: float,
+ max_tokens: int,
+ timeout: float | None,
+ system_message: str | None = None,
+ reasoning_effort: str | None = None,
+ top_p: float | None = None,
+) -> ApiResult:
+ messages = []
+ if system_message:
+ messages.append({"role": "system", "content": system_message})
+ messages.append({"role": "user", "content": prompt})
+
+ body: dict[str, Any] = {
+ "model": model,
+ "messages": messages,
+ "temperature": temperature,
+ "max_tokens": max_tokens,
+ }
+ if reasoning_effort is not None:
+ body["reasoning_effort"] = reasoning_effort
+ if top_p is not None:
+ body["top_p"] = top_p
+
+ resp = await client.post(
+ f"{base_url}/v1/chat/completions",
+ json=body,
+ timeout=timeout,
+ )
+ resp.raise_for_status()
+ data = resp.json()
+ content = data["choices"][0]["message"]["content"]
+ if not content or not content.strip():
+ raise ValueError("Empty response from model")
+ usage = data.get("usage", {})
+ details = usage.get("completion_tokens_details", {})
+ return ApiResult(
+ content=content,
+ prompt_tokens=usage.get("prompt_tokens", 0),
+ completion_tokens=usage.get("completion_tokens", 0),
+ reasoning_tokens=details.get("reasoning_tokens", 0) if details else 0,
+ )
+
+
+async def call_with_retries(
+ client: httpx.AsyncClient,
+ base_url: str,
+ model: str,
+ prompt: str,
+ temperature: float,
+ max_tokens: int,
+ timeout: float | None = None,
+ system_message: str | None = None,
+ reasoning_effort: str | None = None,
+ top_p: float | None = None,
+) -> ApiResult | None:
+ for attempt in range(MAX_RETRIES):
+ try:
+ return await _call_api(
+ client,
+ base_url,
+ model,
+ prompt,
+ temperature,
+ max_tokens,
+ timeout,
+ system_message,
+ reasoning_effort,
+ top_p,
+ )
+ except Exception as e:
+ if attempt < MAX_RETRIES - 1:
+ wait = min(2**attempt, 60)
+ logger.warning(
+ f"Attempt {attempt + 1}/{MAX_RETRIES} failed: {e}. Retrying in {wait}s..."
+ )
+ await asyncio.sleep(wait)
+ else:
+ logger.error(f"All {MAX_RETRIES} retries exhausted. Last error: {e}")
+ return None
+
+
+# ---------------------------------------------------------------------------
+# Evaluation runners
+# ---------------------------------------------------------------------------
+
+
+async def evaluate_benchmark(
+ benchmark_name: str,
+ base_url: str,
+ model: str,
+ temperature: float,
+ max_tokens: int,
+ concurrency: int = 1,
+ limit: int | None = None,
+ timeout: float | None = None,
+ reasoning_effort: str | None = None,
+ top_p: float | None = None,
+ difficulty: str | None = None,
+) -> list[QuestionResult]:
+ """Run a benchmark. Returns per-question results."""
+ import datasets
+
+ config = BENCHMARKS[benchmark_name]
+ logger.info(f"Loading dataset {config.dataset_name}...")
+
+ try:
+ if benchmark_name == "livecodebench":
+ ds = datasets.load_dataset(
+ "json",
+ data_files="hf://datasets/livecodebench/code_generation_lite/*.jsonl",
+ split="train",
+ )
+ else:
+ ds = datasets.load_dataset(
+ config.dataset_name,
+ config.dataset_config,
+ split=config.split,
+ )
+ except Exception as e:
+ logger.error(f"Failed to load dataset: {e}")
+ if "gated" in str(e).lower() or "login" in str(e).lower():
+ logger.error("Dataset requires authentication. Run: huggingface-cli login")
+ return []
+
+ if difficulty and "difficulty" in ds.column_names:
+ ds = ds.filter(lambda x: x["difficulty"] == difficulty)
+ logger.info(f"Filtered to {len(ds)} {difficulty} problems")
+
+ total = len(ds)
+ if limit and limit < total:
+ ds = ds.select(range(limit))
+ total = limit
+
+ logger.info(
+ f"Evaluating {benchmark_name}: {total} questions, concurrency={concurrency}, "
+ f"temperature={temperature}, max_tokens={max_tokens}"
+ )
+
+ if config.kind == "code":
+ logger.warning(
+ "Code benchmarks execute model-generated code. Use a sandboxed environment."
+ )
+
+ semaphore = asyncio.Semaphore(concurrency)
+ results: list[QuestionResult | None] = [None] * total
+ completed = 0
+ lock = asyncio.Lock()
+
+ async def process_question(
+ idx: int, doc: dict, http_client: httpx.AsyncClient
+ ) -> None:
+ nonlocal completed
+ system_msg = None
+
+ if benchmark_name == "gpqa_diamond":
+ prompt, gold = format_gpqa_question(doc, idx)
+ valid_letters = "ABCD"
+ elif benchmark_name == "mmlu_pro":
+ prompt, gold = format_mmlu_pro_question(doc)
+ valid_letters = "ABCDEFGHIJ"[: len(doc["options"])]
+ elif benchmark_name.startswith("aime"):
+ prompt, gold_int = format_aime_question(doc)
+ gold = str(gold_int)
+ elif benchmark_name == "humaneval":
+ prompt, exec_meta = format_humaneval_question(doc)
+ gold = "pass"
+ elif benchmark_name == "livecodebench":
+ prompt, system_msg, exec_meta = format_livecodebench_question(doc)
+ gold = "pass"
+ else:
+ raise ValueError(f"Unknown benchmark: {benchmark_name}")
+
+ async with semaphore:
+ t0 = time.monotonic()
+ api_result = await call_with_retries(
+ http_client,
+ base_url,
+ model,
+ prompt,
+ temperature,
+ max_tokens,
+ timeout,
+ system_message=system_msg,
+ reasoning_effort=reasoning_effort,
+ top_p=top_p,
+ )
+ elapsed = time.monotonic() - t0
+
+ if api_result is None:
+ result = QuestionResult(
+ question_id=idx,
+ prompt=prompt,
+ response="",
+ extracted_answer=None,
+ gold_answer=gold,
+ correct=False,
+ error="API failure after retries",
+ elapsed_s=elapsed,
+ )
+ else:
+ response = api_result.content
+ stats = {
+ "prompt_tokens": api_result.prompt_tokens,
+ "completion_tokens": api_result.completion_tokens,
+ "reasoning_tokens": api_result.reasoning_tokens,
+ "elapsed_s": elapsed,
+ }
+
+ if config.kind == "mc":
+ extracted = extract_mc_answer(response, valid_letters)
+ result = QuestionResult(
+ question_id=idx,
+ prompt=prompt,
+ response=response,
+ extracted_answer=extracted,
+ gold_answer=gold,
+ correct=(extracted == gold) if extracted else False,
+ **stats,
+ )
+ elif config.kind == "math":
+ extracted = extract_boxed_answer(response)
+ correct = (
+ check_aime_answer(extracted, int(gold)) if extracted else False
+ )
+ result = QuestionResult(
+ question_id=idx,
+ prompt=prompt,
+ response=response,
+ extracted_answer=extracted,
+ gold_answer=gold,
+ correct=correct,
+ **stats,
+ )
+ elif config.kind == "code":
+ # HumanEval needs preserved indentation (function body completion)
+ keep_indent = benchmark_name == "humaneval"
+ code = extract_code_block(response, preserve_indent=keep_indent)
+ if code is None:
+ result = QuestionResult(
+ question_id=idx,
+ prompt=prompt,
+ response=response,
+ extracted_answer=None,
+ gold_answer=gold,
+ correct=False,
+ error="No code block extracted",
+ **stats,
+ )
+ elif benchmark_name == "humaneval":
+ passed, diag = run_humaneval_test(
+ exec_meta["problem"],
+ code,
+ )
+ result = QuestionResult(
+ question_id=idx,
+ prompt=prompt,
+ response=response,
+ extracted_answer="pass" if passed else "fail",
+ gold_answer=gold,
+ correct=passed,
+ error=diag if not passed else None,
+ **stats,
+ )
+ elif benchmark_name == "livecodebench":
+ passed, diag = run_livecodebench_test(
+ code,
+ exec_meta["sample"],
+ )
+ result = QuestionResult(
+ question_id=idx,
+ prompt=prompt,
+ response=response,
+ extracted_answer="pass" if passed else "fail",
+ gold_answer=gold,
+ correct=passed,
+ error=diag if not passed else None,
+ **stats,
+ )
+ else:
+ result = QuestionResult(
+ question_id=idx,
+ prompt=prompt,
+ response=response,
+ extracted_answer=None,
+ gold_answer=gold,
+ correct=False,
+ error="Unknown code benchmark",
+ **stats,
+ )
+ else:
+ result = QuestionResult(
+ question_id=idx,
+ prompt=prompt,
+ response=response,
+ extracted_answer=None,
+ gold_answer=gold,
+ correct=False,
+ error="Unsupported kind",
+ **stats,
+ )
+
+ results[idx] = result
+
+ async with lock:
+ completed += 1
+ n = completed
+ if n % max(1, total // 20) == 0 or n == total:
+ correct_so_far = sum(1 for r in results if r is not None and r.correct)
+ answered = sum(1 for r in results if r is not None)
+ logger.info(
+ f" [{n}/{total}] {correct_so_far}/{answered} correct "
+ f"({correct_so_far / max(answered, 1):.1%})"
+ )
+
+ async with httpx.AsyncClient() as http_client:
+ tasks = [process_question(i, doc, http_client) for i, doc in enumerate(ds)]
+ await asyncio.gather(*tasks)
+
+ return [r for r in results if r is not None]
+
+
+# ---------------------------------------------------------------------------
+# Results display
+# ---------------------------------------------------------------------------
+
+
+def print_results(
+ benchmark_name: str,
+ results: list[QuestionResult],
+ concurrency: int | None = None,
+) -> dict[str, Any]:
+ total = len(results)
+ correct = sum(r.correct for r in results)
+ errors = sum(1 for r in results if r.error)
+ no_extract = sum(1 for r in results if r.extracted_answer is None and not r.error)
+ accuracy = correct / max(total, 1)
+
+ total_prompt_tokens = sum(r.prompt_tokens for r in results)
+ total_completion_tokens = sum(r.completion_tokens for r in results)
+ total_reasoning_tokens = sum(r.reasoning_tokens for r in results)
+ total_elapsed = sum(r.elapsed_s for r in results)
+ wall_clock = max(r.elapsed_s for r in results) if results else 0.0
+ avg_gen_tps = total_completion_tokens / total_elapsed if total_elapsed > 0 else 0.0
+
+ label = f"[c={concurrency}] " if concurrency is not None else ""
+ print(f"\n{label}{benchmark_name}: {correct}/{total} ({accuracy:.1%})")
+ tok_line = f" tokens: {total_prompt_tokens:,} prompt + {total_completion_tokens:,} completion"
+ if total_reasoning_tokens > 0:
+ tok_line += f" ({total_reasoning_tokens:,} reasoning)"
+ tok_line += (
+ f" | avg gen tps: {avg_gen_tps:.1f}"
+ f" | total time: {total_elapsed:.1f}s wall clock: {wall_clock:.1f}s"
+ )
+ print(tok_line)
+ if errors:
+ print(f" API errors: {errors}")
+ if no_extract:
+ print(f" No answer extracted: {no_extract}")
+
+ return {
+ "benchmark": benchmark_name,
+ "accuracy": accuracy,
+ "correct": correct,
+ "total": total,
+ "errors": errors,
+ "no_extract": no_extract,
+ "total_prompt_tokens": total_prompt_tokens,
+ "total_completion_tokens": total_completion_tokens,
+ "total_reasoning_tokens": total_reasoning_tokens,
+ "total_elapsed_s": total_elapsed,
+ "wall_clock_s": wall_clock,
+ "avg_gen_tps": avg_gen_tps,
+ }
+
+
+def print_comparison(
+ benchmark_name: str,
+ results_by_c: dict[int, list[QuestionResult]],
+) -> None:
+ levels = sorted(results_by_c.keys())
+ print(f"\n{'=' * 70}")
+ print(f"COMPARISON: {benchmark_name}")
+ print(f"{'=' * 70}")
+
+ header = f"{'Concurrency':<15} {'Accuracy':>10} {'Correct':>10} {'Total':>10} {'Comp Tokens':>12} {'Wall Clock':>12} {'Avg Gen TPS':>12}"
+ print(header)
+ print("-" * len(header))
+ for c in levels:
+ r = results_by_c[c]
+ correct = sum(q.correct for q in r)
+ total = len(r)
+ comp_tok = sum(q.completion_tokens for q in r)
+ total_elapsed = sum(q.elapsed_s for q in r)
+ avg_tps = comp_tok / total_elapsed if total_elapsed > 0 else 0.0
+ wall = max(q.elapsed_s for q in r) if r else 0.0
+ print(
+ f"c={c:<13} {correct / max(total, 1):>10.1%} {correct:>10} {total:>10}"
+ f" {comp_tok:>12,} {wall:>11.1f}s {avg_tps:>12.1f}"
+ )
+
+ if len(levels) >= 2:
+ base_results = results_by_c[levels[0]]
+ test_results = results_by_c[levels[-1]]
+ changed = sum(
+ 1
+ for br, tr in zip(base_results, test_results, strict=True)
+ if br.correct != tr.correct
+ )
+ total = min(len(base_results), len(test_results))
+ print(
+ f"\nQuestions with different correctness (c={levels[0]} vs c={levels[-1]}): {changed}/{total}"
+ )
+ if changed == 0:
+ print("Batching produced identical quality.")
+ elif changed <= total * 0.01:
+ print("Negligible quality difference from batching.")
+ else:
+ print(
+ f"WARNING: {changed / max(total, 1) * 100:.1f}% of questions changed."
+ )
+ print()
+
+
+# ---------------------------------------------------------------------------
+# Interactive task picker
+# ---------------------------------------------------------------------------
+
+
+def pick_tasks_interactive() -> list[str]:
+ import termios
+ import tty
+
+ if not sys.stdin.isatty():
+ logger.error("No --tasks specified and stdin is not a terminal.")
+ return []
+
+ items = [(name, cfg.description) for name, cfg in BENCHMARKS.items()]
+ selected: list[bool] = [False] * len(items)
+ cursor = 0
+ total_lines = len(items) + 4
+
+ def write(s: str) -> None:
+ sys.stdout.write(s)
+
+ def render(first: bool = False) -> None:
+ if not first:
+ write(f"\033[{total_lines}A")
+ write("\033[J")
+ write(
+ "\033[1mSelect benchmarks\033[0m (up/down, space toggle, enter confirm, q quit)\r\n\r\n"
+ )
+ for i, (name, desc) in enumerate(items):
+ marker = ">" if i == cursor else " "
+ check = "x" if selected[i] else " "
+ line = f" {marker} [{check}] {name:<17} {desc}"
+ write(f"\033[7m{line}\033[0m\r\n" if i == cursor else f"{line}\r\n")
+ write(f"\r\n {sum(selected)} selected\r\n")
+ sys.stdout.flush()
+
+ fd = sys.stdin.fileno()
+ old = termios.tcgetattr(fd)
+ try:
+ tty.setraw(fd)
+ write("\033[?25l")
+ render(first=True)
+ while True:
+ ch = sys.stdin.read(1)
+ if ch in ("q", "\x03"):
+ write("\033[?25h\033[0m\r\n")
+ return []
+ elif ch in ("\r", "\n"):
+ break
+ elif ch == " ":
+ selected[cursor] = not selected[cursor]
+ elif ch == "\x1b":
+ seq = sys.stdin.read(2)
+ if seq == "[A":
+ cursor = (cursor - 1) % len(items)
+ elif seq == "[B":
+ cursor = (cursor + 1) % len(items)
+ render()
+ finally:
+ termios.tcsetattr(fd, termios.TCSADRAIN, old)
+ write("\033[?25h\033[0m\r\n")
+ sys.stdout.flush()
+
+ chosen = [name for (name, _), sel in zip(items, selected, strict=True) if sel]
+ if chosen:
+ logger.info(f"Selected: {', '.join(chosen)}")
+ return chosen
+
+
+# ---------------------------------------------------------------------------
+# Results persistence
+# ---------------------------------------------------------------------------
+
+
+def save_results(
+ results_dir: str,
+ benchmark_name: str,
+ model: str,
+ concurrency: int,
+ results: list[QuestionResult],
+ scores: dict[str, Any],
+) -> Path:
+ out_dir = Path(results_dir) / model.replace("/", "_") / benchmark_name
+ out_dir.mkdir(parents=True, exist_ok=True)
+ ts = time.strftime("%Y%m%d_%H%M%S")
+ path = out_dir / f"c{concurrency}_{ts}.json"
+
+ data = {
+ "benchmark": benchmark_name,
+ "model": model,
+ "concurrency": concurrency,
+ "scores": scores,
+ "results": [
+ {
+ "question_id": r.question_id,
+ "prompt": r.prompt,
+ "response": r.response,
+ "extracted_answer": r.extracted_answer,
+ "gold_answer": r.gold_answer,
+ "correct": r.correct,
+ "error": r.error,
+ "prompt_tokens": r.prompt_tokens,
+ "completion_tokens": r.completion_tokens,
+ "reasoning_tokens": r.reasoning_tokens,
+ "elapsed_s": round(r.elapsed_s, 2),
+ }
+ for r in results
+ ],
+ }
+ with open(path, "w") as f:
+ json.dump(data, f, indent=2)
+ logger.info(f"Results saved to {path}")
+ return path
+
+
+# ---------------------------------------------------------------------------
+# CLI
+# ---------------------------------------------------------------------------
+
+
+def parse_int_list(values: list[str]) -> list[int]:
+ items: list[int] = []
+ for v in values:
+ for part in v.split(","):
+ if part.strip():
+ items.append(int(part.strip()))
+ return items
+
+
+def main() -> int:
+ ap = argparse.ArgumentParser(
+ prog="exo-eval",
+ description="Quality evaluation for exo — matches Artificial Analysis methodology.",
+ )
+ add_common_instance_args(ap)
+
+ ap.add_argument(
+ "--tasks",
+ default=None,
+ help="Comma-separated benchmark names. Omit for interactive picker.",
+ )
+ ap.add_argument(
+ "--limit",
+ type=int,
+ default=None,
+ help="Max questions per benchmark (for fast iteration).",
+ )
+
+ reasoning_group = ap.add_mutually_exclusive_group()
+ reasoning_group.add_argument(
+ "--reasoning",
+ action="store_true",
+ default=None,
+ help="Force reasoning-model settings (temperature=0.6, max_tokens=65536).",
+ )
+ reasoning_group.add_argument(
+ "--no-reasoning",
+ action="store_true",
+ default=False,
+ help="Force non-reasoning settings (temperature=0, max_tokens=16384).",
+ )
+
+ ap.add_argument(
+ "--temperature", type=float, default=None, help="Override temperature."
+ )
+ ap.add_argument("--top-p", type=float, default=None, help="Override top_p.")
+ ap.add_argument(
+ "--max-tokens", type=int, default=None, help="Override max output tokens."
+ )
+ ap.add_argument(
+ "--num-concurrent",
+ type=int,
+ default=1,
+ help="Concurrent API requests (default: 1).",
+ )
+ ap.add_argument(
+ "--compare-concurrency",
+ nargs="+",
+ default=None,
+ help="Run at multiple concurrency levels and compare. E.g. --compare-concurrency 1,4",
+ )
+ ap.add_argument(
+ "--request-timeout",
+ type=float,
+ default=None,
+ help="Per-request timeout in seconds (default: no timeout).",
+ )
+ ap.add_argument(
+ "--reasoning-effort",
+ default=None,
+ choices=["low", "medium", "high"],
+ help="Override reasoning effort (default: 'high' for reasoning models, none for non-reasoning).",
+ )
+ ap.add_argument(
+ "--difficulty",
+ default=None,
+ choices=["easy", "medium", "hard"],
+ help="Filter by difficulty (livecodebench only). E.g. --difficulty hard",
+ )
+ ap.add_argument(
+ "--results-dir",
+ default="eval_results",
+ help="Directory for result JSON files (default: eval_results).",
+ )
+ ap.add_argument(
+ "--skip-instance-setup",
+ action="store_true",
+ help="Skip exo instance management (assumes model is already running).",
+ )
+
+ args, _ = ap.parse_known_args()
+
+ # Resolve tasks
+ if args.tasks:
+ task_names = [t.strip() for t in args.tasks.split(",") if t.strip()]
+ else:
+ task_names = pick_tasks_interactive()
+ if not task_names:
+ return 0
+
+ for t in task_names:
+ if t not in BENCHMARKS:
+ logger.error(f"Unknown benchmark '{t}'. Available: {', '.join(BENCHMARKS)}")
+ return 1
+
+ # Instance management
+ client = ExoClient(args.host, args.port, timeout_s=args.timeout)
+ instance_id: str | None = None
+
+ if not args.skip_instance_setup:
+ short_id, full_model_id = resolve_model_short_id(
+ client,
+ args.model,
+ force_download=args.force_download,
+ )
+ selected = settle_and_fetch_placements(
+ client,
+ full_model_id,
+ args,
+ settle_timeout=args.settle_timeout,
+ )
+ if not selected:
+ logger.error("No valid placements matched your filters.")
+ return 1
+
+ selected.sort(
+ key=lambda p: (
+ str(p.get("instance_meta", "")),
+ str(p.get("sharding", "")),
+ -nodes_used_in_instance(p["instance"]),
+ ),
+ reverse=True,
+ )
+ preview = selected[0]
+ instance = preview["instance"]
+ instance_id = instance_id_from_instance(instance)
+
+ logger.info(
+ f"PLACEMENT: {preview['sharding']} / {preview['instance_meta']} / "
+ f"nodes={nodes_used_in_instance(instance)}"
+ )
+
+ settle_deadline = (
+ time.monotonic() + args.settle_timeout if args.settle_timeout > 0 else None
+ )
+ download_duration = run_planning_phase(
+ client,
+ full_model_id,
+ preview,
+ args.danger_delete_downloads,
+ args.timeout,
+ settle_deadline,
+ )
+ if download_duration is not None:
+ logger.info(f"Download: {download_duration:.1f}s")
+
+ client.request_json("POST", "/instance", body={"instance": instance})
+ try:
+ wait_for_instance_ready(client, instance_id)
+ except (RuntimeError, TimeoutError) as e:
+ logger.error(f"Failed to initialize: {e}")
+ with contextlib.suppress(ExoHttpError):
+ client.request_json("DELETE", f"/instance/{instance_id}")
+ return 1
+ time.sleep(1)
+ else:
+ full_model_id = args.model
+
+ # Auto-detect reasoning from model config
+ model_config = load_model_config(full_model_id)
+ if args.reasoning:
+ is_reasoning = True
+ elif args.no_reasoning:
+ is_reasoning = False
+ elif model_config is not None:
+ is_reasoning = model_config.get("reasoning", False)
+ logger.info(
+ f"Auto-detected from config: {model_config['name']} → "
+ f"{'reasoning' if is_reasoning else 'non-reasoning'}"
+ )
+ else:
+ is_reasoning = False
+ logger.warning(
+ f"Model '{full_model_id}' not found in eval_configs/models.toml. "
+ f"Defaulting to non-reasoning. Use --reasoning to override."
+ )
+
+ # Resolve temperature, max_tokens, reasoning_effort
+ # Priority: CLI flag > per-model config > global defaults
+ cfg = model_config or {}
+
+ if args.temperature is not None:
+ temperature = args.temperature
+ elif "temperature" in cfg:
+ temperature = float(cfg["temperature"])
+ else:
+ temperature = (
+ TEMPERATURE_REASONING if is_reasoning else TEMPERATURE_NON_REASONING
+ )
+
+ if args.max_tokens is not None:
+ max_tokens = args.max_tokens
+ elif "max_tokens" in cfg:
+ max_tokens = int(cfg["max_tokens"])
+ else:
+ max_tokens = REASONING_MAX_TOKENS if is_reasoning else DEFAULT_MAX_TOKENS
+
+ if args.top_p is not None:
+ top_p: float | None = args.top_p
+ elif "top_p" in cfg:
+ top_p = float(cfg["top_p"])
+ else:
+ top_p = None # let server use its default
+
+ if args.reasoning_effort is not None:
+ reasoning_effort = args.reasoning_effort
+ elif "reasoning_effort" in cfg:
+ reasoning_effort = str(cfg["reasoning_effort"])
+ else:
+ reasoning_effort = "high" if is_reasoning else None
+ base_url = f"http://{args.host}:{args.port}"
+
+ logger.info(f"Model: {full_model_id}")
+ logger.info(
+ f"Settings: temperature={temperature}, max_tokens={max_tokens}, "
+ + (f"top_p={top_p}, " if top_p is not None else "")
+ + f"reasoning={'yes' if is_reasoning else 'no'}"
+ + (f", reasoning_effort={reasoning_effort}" if reasoning_effort else "")
+ )
+
+ try:
+ if args.compare_concurrency:
+ concurrency_levels = parse_int_list(args.compare_concurrency)
+ for task_name in task_names:
+ results_by_c: dict[int, list[QuestionResult]] = {}
+ for c in concurrency_levels:
+ logger.info(f"\n{'=' * 50}")
+ logger.info(f"Running {task_name} at concurrency={c}")
+ results = asyncio.run(
+ evaluate_benchmark(
+ task_name,
+ base_url,
+ full_model_id,
+ temperature,
+ max_tokens,
+ concurrency=c,
+ limit=args.limit,
+ timeout=args.request_timeout,
+ reasoning_effort=reasoning_effort,
+ top_p=top_p,
+ difficulty=args.difficulty,
+ )
+ )
+ if results:
+ scores = print_results(task_name, results, concurrency=c)
+ save_results(
+ args.results_dir,
+ task_name,
+ full_model_id,
+ c,
+ results,
+ scores,
+ )
+ results_by_c[c] = results
+ if len(results_by_c) >= 2:
+ print_comparison(task_name, results_by_c)
+ else:
+ for task_name in task_names:
+ results = asyncio.run(
+ evaluate_benchmark(
+ task_name,
+ base_url,
+ full_model_id,
+ temperature,
+ max_tokens,
+ concurrency=args.num_concurrent,
+ limit=args.limit,
+ timeout=args.request_timeout,
+ reasoning_effort=reasoning_effort,
+ top_p=top_p,
+ difficulty=args.difficulty,
+ )
+ )
+ if results:
+ scores = print_results(task_name, results)
+ save_results(
+ args.results_dir,
+ task_name,
+ full_model_id,
+ args.num_concurrent,
+ results,
+ scores,
+ )
+ finally:
+ if instance_id is not None:
+ try:
+ client.request_json("DELETE", f"/instance/{instance_id}")
+ except ExoHttpError as e:
+ if e.status != 404:
+ raise
+ wait_for_instance_gone(client, instance_id)
+
+ return 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/bench/harness.py b/bench/harness.py
index 0fb01b68..366de660 100644
--- a/bench/harness.py
+++ b/bench/harness.py
@@ -165,7 +165,9 @@ def wait_for_instance_gone(
raise TimeoutError(f"Instance {instance_id} did not get deleted within {timeout=}")
-def resolve_model_short_id(client: ExoClient, model_arg: str) -> tuple[str, str]:
+def resolve_model_short_id(
+ client: ExoClient, model_arg: str, *, force_download: bool = False
+) -> tuple[str, str]:
models = client.request_json("GET", "/models") or {}
data = models.get("data") or []
@@ -181,6 +183,16 @@ def resolve_model_short_id(client: ExoClient, model_arg: str) -> tuple[str, str]
full_id = str(m["hugging_face_id"])
return short_id, full_id
+ if force_download and "/" in model_arg:
+ logger.info(f"Model not in /models, adding from HuggingFace: {model_arg}")
+ result = client.request_json(
+ "POST", "/models/add", body={"model_id": model_arg}
+ )
+ if result:
+ short_id = str(result.get("name") or model_arg.rsplit("/", 1)[-1])
+ full_id = str(result.get("hugging_face_id") or model_arg)
+ return short_id, full_id
+
raise ValueError(f"Model not found in /models: {model_arg}")
@@ -445,6 +457,11 @@ def add_common_instance_args(ap: argparse.ArgumentParser) -> None:
"--port", type=int, default=int(os.environ.get("EXO_PORT", "52415"))
)
ap.add_argument("--model", required=True, help="Model short id or huggingface id")
+ ap.add_argument(
+ "--force-download",
+ action="store_true",
+ help="If model not in /models, add it from HuggingFace via exo and download.",
+ )
ap.add_argument(
"--max-nodes",
type=int,
diff --git a/bench/parallel_requests.py b/bench/parallel_requests.py
new file mode 100644
index 00000000..c23a6c61
--- /dev/null
+++ b/bench/parallel_requests.py
@@ -0,0 +1,255 @@
+# type: ignore
+import argparse
+import asyncio
+import sys
+import termios
+import time
+import tty
+
+import aiohttp
+
+NUM_REQUESTS = 10
+BASE_URL = ""
+
+QUESTIONS = [
+ "What is the capital of Australia?",
+ "How many bones are in the human body?",
+ "What year did World War II end?",
+ "What is the speed of light in meters per second?",
+ "Who wrote Romeo and Juliet?",
+ "What is the chemical formula for water?",
+ "How many planets are in our solar system?",
+ "What is the largest ocean on Earth?",
+ "Who painted the Mona Lisa?",
+ "What is the boiling point of water in Celsius?",
+]
+
+
+def write(s: str) -> None:
+ sys.stdout.write(s)
+
+
+# ---------------------------------------------------------------------------
+# Model picker (same style as exo_eval)
+# ---------------------------------------------------------------------------
+
+
+def fetch_models() -> list[str]:
+ import json
+ import urllib.request
+
+ with urllib.request.urlopen(f"{BASE_URL}/state") as resp:
+ data = json.loads(resp.read())
+ model_ids: set[str] = set()
+ for instance in data.get("instances", {}).values():
+ for variant in instance.values():
+ sa = variant.get("shardAssignments", {})
+ model_id = sa.get("modelId")
+ if model_id:
+ model_ids.add(model_id)
+ return sorted(model_ids)
+
+
+def pick_model() -> str | None:
+ models = fetch_models()
+ if not models:
+ print("No models found.")
+ return None
+
+ cursor = 0
+ total_lines = len(models) + 4
+
+ def render(first: bool = False) -> None:
+ if not first:
+ write(f"\033[{total_lines}A")
+ write("\033[J")
+ write("\033[1mSelect model\033[0m (up/down, enter confirm, q quit)\r\n\r\n")
+ for i, model in enumerate(models):
+ line = f" {'>' if i == cursor else ' '} {model}"
+ write(f"\033[7m{line}\033[0m\r\n" if i == cursor else f"{line}\r\n")
+ write("\r\n")
+ sys.stdout.flush()
+
+ fd = sys.stdin.fileno()
+ old = termios.tcgetattr(fd)
+ try:
+ tty.setraw(fd)
+ write("\033[?25l")
+ render(first=True)
+ while True:
+ ch = sys.stdin.read(1)
+ if ch in ("q", "\x03"):
+ write("\033[?25h\033[0m\r\n")
+ return None
+ elif ch in ("\r", "\n"):
+ break
+ elif ch == "\x1b":
+ seq = sys.stdin.read(2)
+ if seq == "[A":
+ cursor = (cursor - 1) % len(models)
+ elif seq == "[B":
+ cursor = (cursor + 1) % len(models)
+ render()
+ finally:
+ termios.tcsetattr(fd, termios.TCSADRAIN, old)
+ write(f"\033[{total_lines}A\033[J") # clear picker UI
+ write("\033[?25h\033[0m")
+ sys.stdout.flush()
+
+ return models[cursor]
+
+
+# ---------------------------------------------------------------------------
+# Parallel requests
+# ---------------------------------------------------------------------------
+
+statuses: list[str] = []
+times: list[str] = []
+previews: list[str] = []
+tokens: list[str] = []
+full_responses: list[dict | None] = []
+total_lines = 0
+start_time: float = 0
+selected_model: str = ""
+
+
+def render_progress(first: bool = False) -> None:
+ if not first:
+ write(f"\033[{total_lines}A")
+ write("\033[J")
+ elapsed = time.monotonic() - start_time if start_time else 0
+ done = sum(1 for s in statuses if s == "done")
+ write(
+ f"\033[1m{selected_model}\033[0m [{done}/{NUM_REQUESTS}] {elapsed:.1f}s\r\n\r\n"
+ )
+
+ for i in range(NUM_REQUESTS):
+ q = QUESTIONS[i % len(QUESTIONS)]
+ status = statuses[i]
+ if status == "pending":
+ color = "\033[33m" # yellow
+ elif status == "running":
+ color = "\033[36m" # cyan
+ elif status == "done":
+ color = "\033[32m" # green
+ else:
+ color = "\033[31m" # red
+ write(
+ f" {i:>2} {color}{status:<8}\033[0m {times[i]:>6} {tokens[i]:>5}tok {q[:40]:<40} {previews[i][:50]}\r\n"
+ )
+
+ write("\r\n")
+ sys.stdout.flush()
+
+
+async def send_request(
+ session: aiohttp.ClientSession, i: int, lock: asyncio.Lock
+) -> None:
+ payload = {
+ "model": selected_model,
+ "messages": [{"role": "user", "content": QUESTIONS[i % len(QUESTIONS)]}],
+ "max_tokens": 1024,
+ }
+ statuses[i] = "running"
+ async with lock:
+ render_progress()
+ t0 = time.monotonic()
+ try:
+ async with session.post(
+ f"{BASE_URL}/v1/chat/completions", json=payload
+ ) as resp:
+ data = await resp.json()
+ elapsed = time.monotonic() - t0
+ full_responses[i] = data
+ times[i] = f"{elapsed:.1f}s"
+ if resp.status == 200:
+ choice = data["choices"][0]
+ msg = choice["message"]
+ content = msg.get("content", "")
+ previews[i] = content[:50].replace("\n", " ") or "(empty)"
+ if "usage" in data:
+ tokens[i] = str(data["usage"].get("total_tokens", ""))
+ statuses[i] = "done"
+ else:
+ statuses[i] = f"err:{resp.status}"
+ previews[i] = str(data.get("error", {}).get("message", ""))[:50]
+ except Exception as e:
+ elapsed = time.monotonic() - t0
+ times[i] = f"{elapsed:.1f}s"
+ statuses[i] = "error"
+ previews[i] = str(e)[:50]
+ async with lock:
+ render_progress()
+
+
+async def run_requests(print_stdout: bool = False) -> None:
+ global start_time, total_lines, statuses, times, previews, tokens, full_responses
+
+ statuses = ["pending"] * NUM_REQUESTS
+ times = ["-"] * NUM_REQUESTS
+ previews = ["-"] * NUM_REQUESTS
+ tokens = ["-"] * NUM_REQUESTS
+ full_responses = [None] * NUM_REQUESTS
+ total_lines = NUM_REQUESTS + 4
+
+ write("\033[?25l") # hide cursor
+ start_time = time.monotonic()
+ render_progress(first=True)
+ lock = asyncio.Lock()
+ try:
+ async with aiohttp.ClientSession() as session:
+ tasks = [send_request(session, i, lock) for i in range(NUM_REQUESTS)]
+ await asyncio.gather(*tasks)
+ total = time.monotonic() - start_time
+ write(
+ f"\033[1m=== All {NUM_REQUESTS} requests done in {total:.1f}s ===\033[0m\r\n\r\n"
+ )
+
+ if print_stdout:
+ for i in range(NUM_REQUESTS):
+ data = full_responses[i]
+ if not data or "choices" not in data:
+ continue
+ choice = data["choices"][0]
+ msg = choice["message"]
+ q = QUESTIONS[i % len(QUESTIONS)]
+ write(f"\033[1m--- #{i}: {q} ---\033[0m\r\n")
+ if msg.get("reasoning_content"):
+ write(f"\033[2m[Thinking]: {msg['reasoning_content']}\033[0m\r\n")
+ write(f"{msg.get('content', '')}\r\n")
+ if "usage" in data:
+ u = data["usage"]
+ write(
+ f"\033[2m[Usage: prompt={u.get('prompt_tokens')}, "
+ f"completion={u.get('completion_tokens')}, "
+ f"total={u.get('total_tokens')}]\033[0m\r\n"
+ )
+ write("\r\n")
+ finally:
+ write("\033[?25h") # show cursor
+ sys.stdout.flush()
+
+
+def main() -> None:
+ global selected_model, BASE_URL
+ parser = argparse.ArgumentParser(
+ description="Send parallel requests to an exo cluster"
+ )
+ parser.add_argument(
+ "--host", required=True, help="Hostname of the exo node (e.g. s1)"
+ )
+ parser.add_argument("--port", type=int, default=52415, help="Port (default: 52415)")
+ parser.add_argument(
+ "--stdout", action="store_true", help="Print full responses after completion"
+ )
+ args = parser.parse_args()
+ BASE_URL = f"http://{args.host}:{args.port}"
+ model = pick_model()
+ if not model:
+ return
+ selected_model = model
+ asyncio.run(run_requests(print_stdout=args.stdout))
+
+
+if __name__ == "__main__":
+ main()
diff --git a/bench/pyproject.toml b/bench/pyproject.toml
index 9243858f..188b675d 100644
--- a/bench/pyproject.toml
+++ b/bench/pyproject.toml
@@ -11,6 +11,11 @@ dependencies = [
"tiktoken>=0.12.0",
"jinja2>=3.1.0",
"protobuf>=5.29.0",
+ "datasets>=2.0.0",
+ "math-verify>=0.7.0",
+ "lm-eval[api,math]>=0.4.0",
+ "human-eval>=1.0.3",
+ "numpy>=1.24.0",
]
[build-system]
diff --git a/bench/vendor/__init__.py b/bench/vendor/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/bench/vendor/lcb_testing_util.py b/bench/vendor/lcb_testing_util.py
new file mode 100644
index 00000000..9c26e79c
--- /dev/null
+++ b/bench/vendor/lcb_testing_util.py
@@ -0,0 +1,593 @@
+# type: ignore
+# Vendored from LiveCodeBench (https://github.com/LiveCodeBench/LiveCodeBench)
+# File: lcb_runner/evaluation/testing_util.py
+# License: MIT
+# Vendored 2026-03-07 — do not modify without updating from upstream.
+
+import ast
+import faulthandler
+import json
+import platform
+
+# to run the solution files we're using a timing based approach
+import signal
+import sys
+import time
+
+# used for debugging to time steps
+from datetime import datetime
+from decimal import Decimal
+from enum import Enum
+from io import StringIO
+
+# from pyext import RuntimeModule
+from types import ModuleType
+
+# used for testing the code that reads from input
+from unittest.mock import mock_open, patch
+
+import_string = "from string import *\nfrom re import *\nfrom datetime import *\nfrom collections import *\nfrom heapq import *\nfrom bisect import *\nfrom copy import *\nfrom math import *\nfrom random import *\nfrom statistics import *\nfrom itertools import *\nfrom functools import *\nfrom operator import *\nfrom io import *\nfrom sys import *\nfrom json import *\nfrom builtins import *\nfrom typing import *\nimport string\nimport re\nimport datetime\nimport collections\nimport heapq\nimport bisect\nimport copy\nimport math\nimport random\nimport statistics\nimport itertools\nimport functools\nimport operator\nimport io\nimport sys\nimport json\nsys.setrecursionlimit(50000)\n"
+
+
+def truncatefn(s, length=300):
+ if isinstance(s, str):
+ pass
+ else:
+ s = str(s)
+ if len(s) <= length:
+ return s
+
+ return s[: length // 2] + "...(truncated) ..." + s[-length // 2 :]
+
+
+class CODE_TYPE(Enum):
+ call_based = 0
+ standard_input = 1
+
+
+# stuff for setting up signal timer
+class TimeoutException(Exception):
+ pass
+
+
+def timeout_handler(signum, frame):
+ print("timeout occured: alarm went off")
+ raise TimeoutException
+
+
+# used to capture stdout as a list
+# from https://stackoverflow.com/a/16571630/6416660
+# alternative use redirect_stdout() from contextlib
+class Capturing(list):
+ def __enter__(self):
+ self._stdout = sys.stdout
+ sys.stdout = self._stringio = StringIO()
+ # Make closing the StringIO a no-op
+ self._stringio.close = lambda x: 1
+ return self
+
+ def __exit__(self, *args):
+ self.append(self._stringio.getvalue())
+ del self._stringio # free up some memory
+ sys.stdout = self._stdout
+
+
+# Custom mock for sys.stdin that supports buffer attribute
+class MockStdinWithBuffer:
+ def __init__(self, inputs: str):
+ self.inputs = inputs
+ self._stringio = StringIO(inputs)
+ self.buffer = MockBuffer(inputs)
+
+ def read(self, *args):
+ return self.inputs
+
+ def readline(self, *args):
+ return self._stringio.readline(*args)
+
+ def readlines(self, *args):
+ return self.inputs.split("\n")
+
+ def __getattr__(self, name):
+ # Delegate other attributes to StringIO
+ return getattr(self._stringio, name)
+
+
+class MockBuffer:
+ def __init__(self, inputs: str):
+ self.inputs = inputs.encode("utf-8") # Convert to bytes
+
+ def read(self, *args):
+ # Return as byte strings that can be split
+ return self.inputs
+
+ def readline(self, *args):
+ return self.inputs.split(b"\n")[0] + b"\n"
+
+
+def clean_if_name(code: str) -> str:
+ try:
+ astree = ast.parse(code)
+ last_block = astree.body[-1]
+ if isinstance(last_block, ast.If):
+ condition = last_block.test
+ if ast.unparse(condition).strip() == "__name__ == '__main__'":
+ code = (
+ ast.unparse(astree.body[:-1]) + "\n" + ast.unparse(last_block.body) # type: ignore
+ )
+ except:
+ pass
+
+ return code
+
+
+def make_function(code: str) -> str:
+ try:
+ import_stmts = []
+ all_other_stmts = []
+ astree = ast.parse(code)
+ for stmt in astree.body:
+ if isinstance(stmt, (ast.Import, ast.ImportFrom)):
+ import_stmts.append(stmt)
+ else:
+ all_other_stmts.append(stmt)
+
+ function_ast = ast.FunctionDef(
+ name="wrapped_function",
+ args=ast.arguments(
+ posonlyargs=[], args=[], kwonlyargs=[], kw_defaults=[], defaults=[]
+ ),
+ body=all_other_stmts,
+ decorator_list=[],
+ lineno=-1,
+ )
+ main_code = (
+ import_string
+ + "\n"
+ + ast.unparse(import_stmts)
+ + "\n"
+ + ast.unparse(function_ast)
+ )
+ return main_code
+ except Exception:
+ return code
+
+
+def call_method(method, inputs):
+ if isinstance(inputs, list):
+ inputs = "\n".join(inputs)
+
+ inputs_line_iterator = iter(inputs.split("\n"))
+
+ # Create custom stdin mock with buffer support
+ mock_stdin = MockStdinWithBuffer(inputs)
+
+ # sys.setrecursionlimit(10000)
+
+ # @patch('builtins.input', side_effect=inputs.split("\n"))
+ @patch("builtins.open", mock_open(read_data=inputs))
+ @patch("sys.stdin", mock_stdin) # Use our custom mock instead of StringIO
+ @patch("sys.stdin.readline", lambda *args: next(inputs_line_iterator))
+ @patch("sys.stdin.readlines", lambda *args: inputs.split("\n"))
+ @patch("sys.stdin.read", lambda *args: inputs)
+ # @patch('sys.stdout.write', print)
+ def _inner_call_method(_method):
+ try:
+ return _method()
+ except SystemExit:
+ pass
+ finally:
+ pass
+
+ return _inner_call_method(method)
+
+
+def get_function(compiled_sol, fn_name: str): # type: ignore
+ try:
+ assert hasattr(compiled_sol, fn_name)
+ return getattr(compiled_sol, fn_name)
+ except Exception:
+ return
+
+
+def compile_code(code: str, timeout: int):
+ signal.alarm(timeout)
+ try:
+ tmp_sol = ModuleType("tmp_sol", "")
+ exec(code, tmp_sol.__dict__)
+ if "class Solution" in code:
+ # leetcode wraps solutions in `Solution`
+ # this is a hack to check if it is leetcode solution or not
+ # currently livecodebench only supports LeetCode but
+ # else condition allows future extensibility to other platforms
+ compiled_sol = tmp_sol.Solution()
+ else:
+ # do nothing in the other case since function is accesible
+ compiled_sol = tmp_sol
+
+ assert compiled_sol is not None
+ finally:
+ signal.alarm(0)
+
+ return compiled_sol
+
+
+def convert_line_to_decimals(line: str) -> tuple[bool, list[Decimal]]:
+ try:
+ decimal_line = [Decimal(elem) for elem in line.split()]
+ except:
+ return False, []
+ return True, decimal_line
+
+
+def get_stripped_lines(val: str):
+ ## you don't want empty lines to add empty list after splitlines!
+ val = val.strip()
+
+ return [val_line.strip() for val_line in val.split("\n")]
+
+
+def grade_call_based(
+ code: str, all_inputs: list, all_outputs: list, fn_name: str, timeout: int
+):
+ # call-based clean up logic
+ # need to wrap in try-catch logic after to catch the correct errors, but for now this is fine.
+ code = import_string + "\n\n" + code
+ compiled_sol = compile_code(code, timeout)
+
+ if compiled_sol is None:
+ return
+
+ method = get_function(compiled_sol, fn_name)
+
+ if method is None:
+ return
+
+ all_inputs = [
+ [json.loads(line) for line in inputs.split("\n")] for inputs in all_inputs
+ ]
+
+ all_outputs = [json.loads(output) for output in all_outputs]
+
+ total_execution = 0
+ all_results = []
+ for idx, (gt_inp, gt_out) in enumerate(zip(all_inputs, all_outputs)):
+ signal.alarm(timeout)
+ faulthandler.enable()
+ try:
+ # can lock here so time is useful
+ start = time.time()
+ prediction = method(*gt_inp)
+ total_execution += time.time() - start
+ signal.alarm(0)
+
+ # don't penalize model if it produces tuples instead of lists
+ # ground truth sequences are not tuples
+ if isinstance(prediction, tuple):
+ prediction = list(prediction)
+
+ tmp_result = prediction == gt_out
+
+ # handle floating point comparisons
+
+ all_results.append(tmp_result)
+
+ if not tmp_result:
+ return all_results, {
+ "output": truncatefn(prediction),
+ "inputs": truncatefn(gt_inp),
+ "expected": truncatefn(gt_out),
+ "error_code": -2,
+ "error_message": "Wrong Answer",
+ }
+ except Exception as e:
+ signal.alarm(0)
+ if "timeoutexception" in repr(e).lower():
+ all_results.append(-3)
+ return all_results, {
+ "error": repr(e),
+ "error_code": -3,
+ "error_message": "Time Limit Exceeded",
+ "inputs": truncatefn(gt_inp),
+ "expected": truncatefn(gt_out),
+ }
+ else:
+ all_results.append(-4)
+ return all_results, {
+ "error": repr(e),
+ "error_code": -4,
+ "error_message": "Runtime Error",
+ "inputs": truncatefn(gt_inp),
+ "expected": truncatefn(gt_out),
+ }
+
+ finally:
+ signal.alarm(0)
+ faulthandler.disable()
+
+ return all_results, {"execution time": total_execution}
+
+
+def grade_stdio(
+ code: str,
+ all_inputs: list,
+ all_outputs: list,
+ timeout: int,
+):
+ ## runtime doesn't interact well with __name__ == '__main__'
+ code = clean_if_name(code)
+
+ ## we wrap the given code inside another function
+ code = make_function(code)
+
+ compiled_sol = compile_code(code, timeout)
+ if compiled_sol is None:
+ return
+
+ method = get_function(compiled_sol, "wrapped_function")
+
+ if method is None:
+ return
+
+ all_results = []
+ total_execution_time = 0
+ for idx, (gt_inp, gt_out) in enumerate(zip(all_inputs, all_outputs)):
+ signal.alarm(timeout)
+ faulthandler.enable()
+
+ signal.alarm(timeout)
+ with Capturing() as captured_output:
+ try:
+ start = time.time()
+ call_method(method, gt_inp)
+ total_execution_time += time.time() - start
+ # reset the alarm
+ signal.alarm(0)
+ except Exception as e:
+ signal.alarm(0)
+ if "timeoutexception" in repr(e).lower():
+ all_results.append(-3)
+ return all_results, {
+ "error": repr(e),
+ "error_code": -3,
+ "error_message": "Time Limit Exceeded",
+ "inputs": truncatefn(gt_inp),
+ "expected": truncatefn(gt_out),
+ }
+ else:
+ all_results.append(-4)
+ return all_results, {
+ "error": repr(e),
+ "error_code": -4,
+ "error_message": "Runtime Error",
+ "inputs": truncatefn(gt_inp),
+ "expected": truncatefn(gt_out),
+ }
+
+ finally:
+ signal.alarm(0)
+ faulthandler.disable()
+
+ prediction = captured_output[0]
+
+ stripped_prediction_lines = get_stripped_lines(prediction)
+ stripped_gt_out_lines = get_stripped_lines(gt_out)
+
+ ## WA happens in multiple circumstances
+ ## so cache the return to make it clean!
+ WA_send_args = {
+ "output": truncatefn(prediction),
+ "inputs": truncatefn(gt_inp),
+ "expected": truncatefn(gt_out),
+ "error_code": -2,
+ }
+
+ if len(stripped_prediction_lines) != len(stripped_gt_out_lines):
+ all_results.append(-2)
+ WA_send_args["error_message"] = "Wrong answer: mismatched output length"
+ return all_results, WA_send_args
+
+ for output_line_idx, (
+ stripped_prediction_line,
+ stripped_gt_out_line,
+ ) in enumerate(zip(stripped_prediction_lines, stripped_gt_out_lines)):
+ WA_send_args["error_message"] = (
+ f"Wrong answer at {output_line_idx=}: {truncatefn(stripped_prediction_line)} != {truncatefn(stripped_gt_out_line)}"
+ )
+
+ ## CASE 1: exact match
+ if stripped_prediction_line == stripped_gt_out_line:
+ continue
+
+ ## CASE 2: element-wise comparision
+ ## if there are floating elements
+ ## use `decimal` library for good floating point comparision
+ ## otherwise gotcha: np.isclose(50000000000000000, 50000000000000001) = True
+ ## note that we should always be able to convert to decimals
+
+ success, decimal_prediction_line = convert_line_to_decimals(
+ stripped_prediction_line
+ )
+ if not success:
+ all_results.append(-2)
+ return all_results, WA_send_args
+ success, decimal_gtout_line = convert_line_to_decimals(stripped_gt_out_line)
+ if not success:
+ all_results.append(-2)
+ return all_results, WA_send_args
+
+ if decimal_prediction_line == decimal_gtout_line:
+ continue
+
+ all_results.append(-2)
+ return all_results, WA_send_args
+ all_results.append(True)
+
+ return all_results, {"execution time": total_execution_time}
+
+
+def run_test(sample, test=None, debug=False, timeout=6):
+ """
+ if test(generated_code) is not None it'll try to run the code.
+ otherwise it'll just return an input and output pair.
+ """
+ signal.signal(signal.SIGALRM, timeout_handler)
+
+ # Disable functionalities that can make destructive changes to the test.
+ # max memory is set to 4GB
+ reliability_guard()
+
+ if debug:
+ print(f"start = {datetime.now().time()}")
+
+ try:
+ in_outs = json.loads(sample["input_output"])
+ except ValueError as e:
+ raise e
+ in_outs = None
+
+ if in_outs:
+ if in_outs.get("fn_name") is None:
+ which_type = CODE_TYPE.standard_input # Standard input
+ method_name = None
+
+ else:
+ which_type = CODE_TYPE.call_based # Call-based
+ method_name = in_outs["fn_name"]
+
+ if debug:
+ print(f"loaded input_output = {datetime.now().time()}")
+
+ if test is None:
+ assert False, "should not happen: test code is none"
+ return in_outs, {"error": "no test code provided"}
+ elif test is not None:
+ results = []
+ sol = import_string
+ if debug:
+ print(f"loading test code = {datetime.now().time()}")
+
+ if which_type == CODE_TYPE.call_based:
+ signal.alarm(timeout)
+ try:
+ results, metadata = grade_call_based(
+ code=test,
+ all_inputs=in_outs["inputs"],
+ all_outputs=in_outs["outputs"],
+ fn_name=method_name,
+ timeout=timeout,
+ )
+ return results, metadata
+ except Exception as e:
+ return [-4], {
+ "error_code": -4,
+ "error_message": f"Error during testing: {e}",
+ }
+ finally:
+ signal.alarm(0)
+ elif which_type == CODE_TYPE.standard_input:
+ # sol
+ # if code has if __name__ == "__main__": then remove it
+
+ signal.alarm(timeout)
+ try:
+ results, metadata = grade_stdio(
+ code=test,
+ all_inputs=in_outs["inputs"],
+ all_outputs=in_outs["outputs"],
+ timeout=timeout,
+ )
+ return results, metadata
+ except Exception as e:
+ return [-4], {
+ "error_code": -4,
+ "error_message": f"Error during testing: {e}",
+ }
+ finally:
+ signal.alarm(0)
+
+
+def reliability_guard(maximum_memory_bytes=None):
+ """
+ This disables various destructive functions and prevents the generated code
+ from interfering with the test (e.g. fork bomb, killing other processes,
+ removing filesystem files, etc.)
+ WARNING
+ This function is NOT a security sandbox. Untrusted code, including, model-
+ generated code, should not be blindly executed outside of one. See the
+ Codex paper for more information about OpenAI's code sandbox, and proceed
+ with caution.
+ """
+
+ if maximum_memory_bytes is not None:
+ import resource
+
+ resource.setrlimit(
+ resource.RLIMIT_AS, (maximum_memory_bytes, maximum_memory_bytes)
+ )
+ resource.setrlimit(
+ resource.RLIMIT_DATA, (maximum_memory_bytes, maximum_memory_bytes)
+ )
+ if not platform.uname().system == "Darwin":
+ resource.setrlimit(
+ resource.RLIMIT_STACK, (maximum_memory_bytes, maximum_memory_bytes)
+ )
+
+ faulthandler.disable()
+
+ import builtins
+
+ # builtins.exit = None
+ builtins.quit = None
+
+ import os
+
+ os.environ["OMP_NUM_THREADS"] = "1"
+
+ os.kill = None
+ os.system = None
+ os.putenv = None
+ os.remove = None
+ os.removedirs = None
+ os.rmdir = None
+ os.fchdir = None
+ os.setuid = None
+ os.fork = None
+ os.forkpty = None
+ os.killpg = None
+ os.rename = None
+ os.renames = None
+ os.truncate = None
+ os.replace = None
+ os.unlink = None
+ os.fchmod = None
+ os.fchown = None
+ os.chmod = None
+ os.chown = None
+ os.chroot = None
+ os.fchdir = None
+ os.lchflags = None
+ os.lchmod = None
+ os.lchown = None
+ os.getcwd = None
+ os.chdir = None
+
+ import shutil
+
+ shutil.rmtree = None
+ shutil.move = None
+ shutil.chown = None
+
+ import subprocess
+
+ subprocess.Popen = None
+
+ __builtins__["help"] = None
+
+ import sys
+
+ sys.modules["ipdb"] = None
+ sys.modules["joblib"] = None
+ sys.modules["resource"] = None
+ sys.modules["psutil"] = None
+ sys.modules["tkinter"] = None
diff --git a/pyproject.toml b/pyproject.toml
index 86293059..d6143acc 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -62,7 +62,7 @@ members = ["rust/exo_pyo3_bindings", "bench"]
[tool.uv.sources]
exo_pyo3_bindings = { workspace = true }
mlx = { git = "https://github.com/rltakashige/mlx-jaccl-fix-small-recv.git", branch = "address-rdma-gpu-locks", marker = "sys_platform == 'darwin'" }
-mlx-lm = { git = "https://github.com/ml-explore/mlx-lm", rev = "834fac934c4e04de9b3d723e2b9287a2c60cfd4a" }
+mlx-lm = { git = "https://github.com/rltakashige/mlx-lm", branch = "leo/eval-left-padding-in-batched-rotation" }
# Uncomment to use local mlx/mlx-lm development versions:
# mlx = { path = "/Users/Shared/mlx", editable=true }
# mlx-lm = { path = "/Users/Shared/mlx-lm", editable=true }
@@ -125,6 +125,7 @@ extend-exclude = [
"shared/protobufs/**",
"*mlx_typings/**",
"rust/exo_pyo3_bindings/**",
+ "bench/vendor/**",
]
[tool.ruff.lint]
diff --git a/python/parts.nix b/python/parts.nix
index 9ba703a0..d1d1b6d6 100644
--- a/python/parts.nix
+++ b/python/parts.nix
@@ -43,6 +43,27 @@
final.setuptools
];
});
+ # rouge-score and sacrebleu don't declare setuptools as a build dependency
+ rouge-score = prev.rouge-score.overrideAttrs (old: {
+ nativeBuildInputs = (old.nativeBuildInputs or [ ]) ++ [
+ final.setuptools
+ ];
+ });
+ sacrebleu = prev.sacrebleu.overrideAttrs (old: {
+ nativeBuildInputs = (old.nativeBuildInputs or [ ]) ++ [
+ final.setuptools
+ ];
+ });
+ sqlitedict = prev.sqlitedict.overrideAttrs (old: {
+ nativeBuildInputs = (old.nativeBuildInputs or [ ]) ++ [
+ final.setuptools
+ ];
+ });
+ word2number = prev.word2number.overrideAttrs (old: {
+ nativeBuildInputs = (old.nativeBuildInputs or [ ]) ++ [
+ final.setuptools
+ ];
+ });
} // lib.optionalAttrs pkgs.stdenv.hostPlatform.isDarwin {
# Use our pure Nix-built MLX with Metal support (macOS only)
mlx = self'.packages.mlx;
@@ -96,13 +117,16 @@
"lib/python3.13/site-packages/nvidia*"
];
- exoVenv = (pythonSet.mkVirtualEnv "exo-env" workspace.deps.default).overrideAttrs {
+ # Exclude bench deps from main env (bench has its own benchVenv)
+ exoDeps = removeAttrs workspace.deps.default [ "exo-bench" ];
+
+ exoVenv = (pythonSet.mkVirtualEnv "exo-env" exoDeps).overrideAttrs {
venvIgnoreCollisions = venvCollisionPaths;
};
# Virtual environment with dev dependencies for testing
testVenv = (pythonSet.mkVirtualEnv "exo-test-env" (
- workspace.deps.default // {
+ exoDeps // {
exo = [ "dev" ]; # Include pytest, pytest-asyncio, pytest-env
}
)).overrideAttrs {
@@ -158,6 +182,7 @@
exo-test-env = testVenv;
} // {
exo-bench = mkBenchScript "exo-bench" (inputs.self + /bench/exo_bench.py);
+ exo-eval = mkBenchScript "exo-eval" (inputs.self + /bench/exo_eval.py);
exo-eval-tool-calls = mkBenchScript "exo-eval-tool-calls" (inputs.self + /bench/eval_tool_calls.py);
exo-get-all-models-on-cluster = mkSimplePythonScript "exo-get-all-models-on-cluster" (inputs.self + /tests/get_all_models_on_cluster.py);
};
diff --git a/src/exo/main.py b/src/exo/main.py
index 3b19dd3b..2ecb62c2 100644
--- a/src/exo/main.py
+++ b/src/exo/main.py
@@ -270,6 +270,10 @@ def main():
if args.offline:
logger.info("Running in OFFLINE mode — no internet checks, local models only")
+ if args.no_batch:
+ os.environ["EXO_NO_BATCH"] = "1"
+ logger.info("Continuous batching disabled (--no-batch)")
+
# Set FAST_SYNCH override env var for runner subprocesses
if args.fast_synch is True:
os.environ["EXO_FAST_SYNCH"] = "on"
@@ -300,6 +304,7 @@ class Args(CamelCaseModel):
no_worker: bool = False
no_downloads: bool = False
offline: bool = os.getenv("EXO_OFFLINE", "false").lower() == "true"
+ no_batch: bool = False
fast_synch: bool | None = None # None = auto, True = force on, False = force off
@classmethod
@@ -353,6 +358,11 @@ class Args(CamelCaseModel):
default=os.getenv("EXO_OFFLINE", "false").lower() == "true",
help="Run in offline/air-gapped mode: skip internet checks, use only pre-staged local models",
)
+ parser.add_argument(
+ "--no-batch",
+ action="store_true",
+ help="Disable continuous batching, use sequential generation",
+ )
fast_synch_group = parser.add_mutually_exclusive_group()
fast_synch_group.add_argument(
"--fast-synch",
diff --git a/src/exo/master/adapters/chat_completions.py b/src/exo/master/adapters/chat_completions.py
index af13fdb0..c2221d71 100644
--- a/src/exo/master/adapters/chat_completions.py
+++ b/src/exo/master/adapters/chat_completions.py
@@ -104,6 +104,7 @@ def chat_request_to_text_generation(
else None,
logprobs=request.logprobs or False,
top_logprobs=request.top_logprobs,
+ min_p=request.min_p,
repetition_penalty=request.repetition_penalty,
repetition_context_size=request.repetition_context_size,
)
diff --git a/src/exo/shared/constants.py b/src/exo/shared/constants.py
index e1c9ca19..69534298 100644
--- a/src/exo/shared/constants.py
+++ b/src/exo/shared/constants.py
@@ -81,3 +81,5 @@ EXO_ENABLE_IMAGE_MODELS = (
EXO_OFFLINE = os.getenv("EXO_OFFLINE", "false").lower() == "true"
EXO_TRACING_ENABLED = os.getenv("EXO_TRACING_ENABLED", "false").lower() == "true"
+
+EXO_MAX_CONCURRENT_REQUESTS = int(os.getenv("EXO_MAX_CONCURRENT_REQUESTS", "8"))
diff --git a/src/exo/shared/types/api.py b/src/exo/shared/types/api.py
index 2f13d686..2c0736e0 100644
--- a/src/exo/shared/types/api.py
+++ b/src/exo/shared/types/api.py
@@ -201,6 +201,7 @@ class ChatCompletionRequest(BaseModel):
tools: list[dict[str, Any]] | None = None
reasoning_effort: ReasoningEffort | None = None
enable_thinking: bool | None = None
+ min_p: float | None = None
repetition_penalty: float | None = None
repetition_context_size: int | None = None
tool_choice: str | dict[str, Any] | None = None
diff --git a/src/exo/shared/types/tasks.py b/src/exo/shared/types/tasks.py
index 8d866456..e2e6b5e5 100644
--- a/src/exo/shared/types/tasks.py
+++ b/src/exo/shared/types/tasks.py
@@ -18,6 +18,9 @@ class TaskId(Id):
pass
+CANCEL_ALL_TASKS = TaskId("CANCEL_ALL_TASKS")
+
+
class TaskStatus(str, Enum):
Pending = "Pending"
Running = "Running"
diff --git a/src/exo/shared/types/text_generation.py b/src/exo/shared/types/text_generation.py
index 224f51ee..3c2bcb48 100644
--- a/src/exo/shared/types/text_generation.py
+++ b/src/exo/shared/types/text_generation.py
@@ -67,5 +67,6 @@ class TextGenerationTaskParams(BaseModel, frozen=True):
enable_thinking: bool | None = None
logprobs: bool = False
top_logprobs: int | None = None
+ min_p: float | None = None
repetition_penalty: float | None = None
repetition_context_size: int | None = None
diff --git a/src/exo/worker/engines/mlx/generator/batch_generate.py b/src/exo/worker/engines/mlx/generator/batch_generate.py
new file mode 100644
index 00000000..3b72e0a7
--- /dev/null
+++ b/src/exo/worker/engines/mlx/generator/batch_generate.py
@@ -0,0 +1,383 @@
+import time
+from dataclasses import dataclass, field
+from typing import Callable, cast
+
+import mlx.core as mx
+from mlx_lm.generate import (
+ BatchGenerator as MlxBatchGenerator,
+)
+from mlx_lm.models.cache import RotatingKVCache
+from mlx_lm.sample_utils import make_logits_processors, make_sampler
+from mlx_lm.tokenizer_utils import TokenizerWrapper
+
+from exo.shared.types.api import (
+ CompletionTokensDetails,
+ FinishReason,
+ GenerationStats,
+ PromptTokensDetails,
+ TopLogprobItem,
+ Usage,
+)
+from exo.shared.types.memory import Memory
+from exo.shared.types.mlx import KVCacheType, Model
+from exo.shared.types.text_generation import TextGenerationTaskParams
+from exo.shared.types.worker.runner_response import GenerationResponse
+from exo.worker.engines.mlx.cache import (
+ CacheSnapshot,
+ KVPrefixCache,
+ encode_prompt,
+ make_kv_cache,
+)
+from exo.worker.engines.mlx.constants import DEFAULT_TOP_LOGPROBS, MAX_TOKENS
+from exo.worker.engines.mlx.generator.generate import (
+ ban_token_ids,
+ eos_ids_from_tokenizer,
+ extract_top_logprobs,
+ prefill,
+)
+from exo.worker.engines.mlx.utils_mlx import fix_unmatched_think_end_tokens
+from exo.worker.runner.bootstrap import logger
+
+_MIN_PREFIX_HIT_RATIO_TO_UPDATE = 0.5
+
+
+def _stop_sequences(task_params: TextGenerationTaskParams) -> list[str]:
+ if task_params.stop is None:
+ return []
+ if isinstance(task_params.stop, str):
+ return [task_params.stop]
+ return task_params.stop
+
+
+@dataclass
+class _EngineTask:
+ uid: int
+ task_params: TextGenerationTaskParams
+ all_prompt_tokens: mx.array
+ prefix_hit_length: int
+ matched_index: int | None
+ cache_snapshots: list[CacheSnapshot] | None
+ on_generation_token: Callable[[], None] | None = None
+ generated_text_parts: list[str] = field(default_factory=list)
+ potential_stop_sequence_text: str = ""
+ completion_tokens: int = 0
+ generation_start_time: float = 0.0
+ in_thinking: bool = False
+ reasoning_tokens: int = 0
+
+
+@dataclass(eq=False)
+class ExoBatchGenerator:
+ model: Model
+ tokenizer: TokenizerWrapper
+ group: mx.distributed.Group | None
+ kv_prefix_cache: KVPrefixCache | None
+
+ _exo_gen: MlxBatchGenerator = field(init=False)
+ _active_tasks: dict[int, _EngineTask] = field(default_factory=dict, init=False)
+
+ def __post_init__(self) -> None:
+ self._exo_gen = MlxBatchGenerator(
+ model=self.model,
+ stop_tokens=set(eos_ids_from_tokenizer(self.tokenizer)),
+ prefill_step_size=4096,
+ )
+
+ @property
+ def has_work(self) -> bool:
+ return (
+ bool(self._active_tasks)
+ or bool(self._exo_gen.unprocessed_prompts)
+ or self._exo_gen.active_batch is not None
+ )
+
+ def submit(
+ self,
+ task_params: TextGenerationTaskParams,
+ prompt: str,
+ on_prefill_progress: Callable[[int, int], None] | None = None,
+ distributed_prompt_progress_callback: Callable[[], None] | None = None,
+ on_generation_token: Callable[[], None] | None = None,
+ ) -> int:
+ all_prompt_tokens = encode_prompt(self.tokenizer, prompt)
+ all_prompt_tokens = fix_unmatched_think_end_tokens(
+ all_prompt_tokens, self.tokenizer
+ )
+
+ is_bench = task_params.bench
+
+ prefix_hit_length = 0
+ matched_index: int | None = None
+ prompt_tokens = all_prompt_tokens
+
+ if self.kv_prefix_cache is not None and not is_bench:
+ cache, remaining_tokens, matched_index = self.kv_prefix_cache.get_kv_cache(
+ self.model, all_prompt_tokens
+ )
+ prefix_hit_length = len(all_prompt_tokens) - len(remaining_tokens)
+ if prefix_hit_length > 0:
+ logger.info(
+ f"KV cache hit: {prefix_hit_length}/{len(all_prompt_tokens)} tokens "
+ f"cached ({100 * prefix_hit_length / len(all_prompt_tokens):.1f}%)"
+ )
+ prompt_tokens = remaining_tokens
+ else:
+ cache = make_kv_cache(self.model)
+ else:
+ cache = make_kv_cache(self.model)
+
+ seed = task_params.seed if task_params.seed is not None else 42
+ mx.random.seed(seed)
+
+ sampler = make_sampler(
+ temp=task_params.temperature
+ if task_params.temperature is not None
+ else 0.7,
+ top_p=task_params.top_p if task_params.top_p is not None else 1.0,
+ min_p=task_params.min_p if task_params.min_p is not None else 0.05,
+ top_k=task_params.top_k if task_params.top_k is not None else 0,
+ )
+
+ _prefill_tps, _prefill_tokens, cache_snapshots = prefill(
+ self.model,
+ self.tokenizer,
+ sampler,
+ prompt_tokens[:-1],
+ cache,
+ self.group,
+ on_prefill_progress,
+ distributed_prompt_progress_callback,
+ )
+
+ # We need to clamp rotating kv caches to max size so that mlx lm's _merge_caches behaves
+ for c in cache:
+ if (
+ isinstance(c, RotatingKVCache)
+ and c.keys is not None
+ and c.values is not None
+ and c.keys.shape[2] > c.max_size
+ ):
+ trim_size = c.keys.shape[2] - c.max_size
+ c.keys = c._trim(trim_size, c.keys)
+ c.values = c._trim(trim_size, c.values)
+ c._idx = c.max_size
+
+ if not is_bench:
+ self._save_prefix_cache(
+ all_prompt_tokens,
+ list(cache),
+ cache_snapshots,
+ prefix_hit_length,
+ matched_index,
+ )
+
+ last_tokens = prompt_tokens[-2:]
+
+ logits_processors: list[Callable[[mx.array, mx.array], mx.array]] = (
+ make_logits_processors(
+ repetition_penalty=task_params.repetition_penalty,
+ repetition_context_size=task_params.repetition_context_size,
+ )
+ )
+ if is_bench:
+ # Only sample length eos tokens
+ eos_ids = eos_ids_from_tokenizer(self.tokenizer)
+ logits_processors = [ban_token_ids(eos_ids)] + logits_processors
+
+ max_tokens = task_params.max_output_tokens or MAX_TOKENS
+
+ uids = self._exo_gen.insert(
+ prompts=[last_tokens.tolist()],
+ max_tokens=[max_tokens],
+ caches=[list(cache)],
+ samplers=[sampler],
+ logits_processors=[logits_processors],
+ )
+
+ assert len(uids) == 1
+
+ uid = uids[0]
+
+ self._active_tasks[uid] = _EngineTask(
+ uid=uid,
+ task_params=task_params,
+ all_prompt_tokens=all_prompt_tokens,
+ prefix_hit_length=prefix_hit_length,
+ matched_index=matched_index,
+ cache_snapshots=cache_snapshots or None,
+ on_generation_token=on_generation_token,
+ generation_start_time=time.perf_counter(),
+ )
+
+ return uid
+
+ def step(self) -> list[tuple[int, GenerationResponse]]:
+ if not self.has_work:
+ return []
+
+ responses = self._exo_gen.next()
+
+ results: list[tuple[int, GenerationResponse]] = []
+
+ for response in responses:
+ if response.uid not in self._active_tasks:
+ logger.warning(
+ f"response uid {response.uid} was not found - should be active"
+ )
+ continue
+
+ state = self._active_tasks[response.uid]
+ if state.on_generation_token is not None:
+ state.on_generation_token()
+ text = (
+ ""
+ if response.finish_reason == "stop"
+ else self.tokenizer.decode([response.token])
+ )
+ state.completion_tokens += 1
+ state.generated_text_parts.append(text)
+ state.potential_stop_sequence_text += text
+
+ think_start = self.tokenizer.think_start
+ think_end = self.tokenizer.think_end
+ if think_start is not None and text == think_start:
+ state.in_thinking = True
+ elif think_end is not None and text == think_end:
+ state.in_thinking = False
+ if state.in_thinking:
+ state.reasoning_tokens += 1
+
+ finish_reason: FinishReason | None = cast(
+ FinishReason | None, response.finish_reason
+ )
+ task_params = state.task_params
+ stop_sequences = _stop_sequences(task_params)
+ max_stop_len = max((len(s) for s in stop_sequences), default=0)
+
+ if stop_sequences:
+ for stop_seq in stop_sequences:
+ if stop_seq in state.potential_stop_sequence_text:
+ stop_index = state.potential_stop_sequence_text.find(stop_seq)
+ text_before_stop = state.potential_stop_sequence_text[
+ :stop_index
+ ]
+ chunk_start = len(state.potential_stop_sequence_text) - len(
+ text
+ )
+ text = text_before_stop[chunk_start:]
+ finish_reason = "stop"
+ break
+
+ is_done = finish_reason is not None
+
+ logprob: float | None = None
+ top_logprobs: list[TopLogprobItem] | None = None
+ if task_params.logprobs:
+ logprob, top_logprobs = extract_top_logprobs(
+ logprobs=response.logprobs,
+ tokenizer=self.tokenizer,
+ top_logprobs=task_params.top_logprobs or DEFAULT_TOP_LOGPROBS,
+ selected_token=response.token,
+ )
+
+ stats: GenerationStats | None = None
+ usage: Usage | None = None
+ if is_done:
+ generation_elapsed = time.perf_counter() - state.generation_start_time
+ generation_tps = (
+ state.completion_tokens / generation_elapsed
+ if generation_elapsed > 0
+ else 0.0
+ )
+ mlx_stats = self._exo_gen.stats()
+ stats = GenerationStats(
+ prompt_tps=float(mlx_stats.prompt_tps)
+ if mlx_stats.prompt_time > 0
+ else 0.0,
+ generation_tps=float(generation_tps),
+ prompt_tokens=len(state.all_prompt_tokens),
+ generation_tokens=state.completion_tokens,
+ peak_memory_usage=Memory.from_gb(mx.get_peak_memory() / 1e9),
+ )
+ total_prompt_tokens = len(state.all_prompt_tokens)
+ usage = Usage(
+ prompt_tokens=total_prompt_tokens,
+ completion_tokens=state.completion_tokens,
+ total_tokens=total_prompt_tokens + state.completion_tokens,
+ prompt_tokens_details=PromptTokensDetails(
+ cached_tokens=state.prefix_hit_length
+ ),
+ completion_tokens_details=CompletionTokensDetails(
+ reasoning_tokens=state.reasoning_tokens
+ ),
+ )
+
+ results.append(
+ (
+ response.uid,
+ GenerationResponse(
+ text=text,
+ token=response.token,
+ logprob=logprob,
+ top_logprobs=top_logprobs,
+ finish_reason=finish_reason,
+ stats=stats,
+ usage=usage,
+ ),
+ )
+ )
+
+ if is_done:
+ del self._active_tasks[response.uid]
+ elif (
+ max_stop_len > 0
+ and len(state.potential_stop_sequence_text) > max_stop_len
+ ):
+ state.potential_stop_sequence_text = state.potential_stop_sequence_text[
+ -max_stop_len:
+ ]
+
+ return results
+
+ def cancel(self, uids: list[int]) -> None:
+ self._exo_gen.remove(uids)
+ for uid in uids:
+ self._active_tasks.pop(uid, None)
+
+ def close(self) -> None:
+ self._exo_gen.close()
+
+ def _save_prefix_cache(
+ self,
+ all_prompt_tokens: mx.array,
+ cache: KVCacheType,
+ cache_snapshots: list[CacheSnapshot] | None,
+ prefix_hit_length: int,
+ matched_index: int | None,
+ ) -> None:
+ if self.kv_prefix_cache is None:
+ return
+
+ try:
+ hit_ratio = (
+ prefix_hit_length / len(all_prompt_tokens)
+ if len(all_prompt_tokens) > 0
+ else 0.0
+ )
+ if (
+ matched_index is not None
+ and hit_ratio >= _MIN_PREFIX_HIT_RATIO_TO_UPDATE
+ ):
+ self.kv_prefix_cache.update_kv_cache(
+ matched_index,
+ all_prompt_tokens,
+ cache,
+ cache_snapshots,
+ restore_pos=prefix_hit_length,
+ )
+ else:
+ self.kv_prefix_cache.add_kv_cache(
+ all_prompt_tokens, cache, cache_snapshots
+ )
+ except Exception:
+ logger.warning("Failed to save prefix cache", exc_info=True)
diff --git a/src/exo/worker/engines/mlx/generator/generate.py b/src/exo/worker/engines/mlx/generator/generate.py
index e5e89c7b..eda6a7e5 100644
--- a/src/exo/worker/engines/mlx/generator/generate.py
+++ b/src/exo/worker/engines/mlx/generator/generate.py
@@ -309,7 +309,11 @@ def warmup_inference(
model: Model,
tokenizer: TokenizerWrapper,
group: mx.distributed.Group | None,
+ model_id: ModelId,
) -> int:
+ logger.info(f"warming up inference for instance: {model_id}")
+ t = time.monotonic()
+
content = "Prompt to warm up the inference engine. Repeat this."
warmup_prompt = apply_chat_template(
@@ -350,7 +354,25 @@ def warmup_inference(
mx_barrier(group)
- return tokens_generated
+ logger.info(f"warmed up by generating {tokens_generated} tokens")
+ check_for_cancel_every = min(
+ math.ceil(tokens_generated / min(time.monotonic() - t, 0.001)), 100
+ )
+ if group is not None:
+ check_for_cancel_every = int(
+ mx.max(
+ mx.distributed.all_gather(
+ mx.array([check_for_cancel_every]),
+ group=group,
+ )
+ ).item()
+ )
+
+ logger.info(
+ f"runner checking for cancellation every {check_for_cancel_every} tokens"
+ )
+
+ return check_for_cancel_every
def ban_token_ids(token_ids: list[int]) -> Callable[[mx.array, mx.array], mx.array]:
@@ -484,6 +506,7 @@ def mlx_generate(
sampler = make_sampler(
temp=task.temperature if task.temperature is not None else 0.7,
top_p=task.top_p if task.top_p is not None else 1.0,
+ min_p=task.min_p if task.min_p is not None else 0.05,
top_k=task.top_k if task.top_k is not None else 0,
)
diff --git a/src/exo/worker/engines/mlx/tests/test_batch_generate.py b/src/exo/worker/engines/mlx/tests/test_batch_generate.py
new file mode 100644
index 00000000..3d48590c
--- /dev/null
+++ b/src/exo/worker/engines/mlx/tests/test_batch_generate.py
@@ -0,0 +1,268 @@
+# pyright: reportAny=false, reportUnknownVariableType=false
+# pyright: reportUnknownMemberType=false, reportUnknownArgumentType=false
+# pyright: reportUnknownLambdaType=false, reportPrivateUsage=false
+# pyright: reportInvalidCast=false, reportArgumentType=false
+# pyright: reportUnusedImport=false
+"""Test B=1 vs B=2 equivalence for batch generation.
+
+Verifies that running two requests concurrently in a batch (B=2) produces
+identical token selections to running them sequentially (B=1).
+Uses random weights — no model download required.
+"""
+
+from pathlib import Path
+from typing import cast
+
+import mlx.core as mx
+import mlx.nn as nn
+import mlx.utils
+import pytest
+from mlx_lm.generate import _merge_caches
+from mlx_lm.sample_utils import make_sampler
+from mlx_lm.tokenizer_utils import TokenizerWrapper
+from transformers import AutoTokenizer
+
+# Import batch_generate to activate the right-padding BatchKVCache patch
+import exo.worker.engines.mlx.generator.batch_generate # noqa: F401
+from exo.shared.types.mlx import Model
+from exo.worker.engines.mlx.cache import encode_prompt, make_kv_cache
+from exo.worker.engines.mlx.generator.generate import prefill
+
+NUM_STEPS = 20
+
+
+def _init_random(model: nn.Module) -> None:
+ """Initialize all model parameters with random values."""
+ params = model.parameters()
+ new_params = mlx.utils.tree_map(
+ lambda p: mx.random.normal(shape=p.shape, dtype=p.dtype)
+ if isinstance(p, mx.array)
+ else p,
+ params,
+ )
+ model.update(new_params)
+ mx.eval(model.parameters())
+
+
+def _run_b1_vs_b2(
+ model: Model,
+ tokenizer: TokenizerWrapper,
+ tokens_a: mx.array,
+ tokens_b: mx.array,
+) -> tuple[float, int]:
+ """Run B=1 sequential and B=2 batched, return (max_diff, mismatches)."""
+ sampler = make_sampler(temp=0.0)
+
+ # B=1 sequential
+ cache_a1 = make_kv_cache(model)
+ prefill(model, tokenizer, sampler, tokens_a[:-1], cache_a1, None, None, None)
+ merged_a1 = _merge_caches([[c for c in cache_a1]])
+ for c in merged_a1:
+ c.prepare(lengths=[1], right_padding=[0])
+ model(mx.array([[tokens_a[-2].item()]]), cache=merged_a1)
+ mx.eval([c.state for c in merged_a1])
+ for c in merged_a1:
+ c.finalize()
+
+ cache_b1 = make_kv_cache(model)
+ prefill(model, tokenizer, sampler, tokens_b[:-1], cache_b1, None, None, None)
+ merged_b1 = _merge_caches([[c for c in cache_b1]])
+ for c in merged_b1:
+ c.prepare(lengths=[1], right_padding=[0])
+ model(mx.array([[tokens_b[-2].item()]]), cache=merged_b1)
+ mx.eval([c.state for c in merged_b1])
+ for c in merged_b1:
+ c.finalize()
+
+ b1_logits_a: list[mx.array] = []
+ b1_logits_b: list[mx.array] = []
+ next_a, next_b = tokens_a[-1].item(), tokens_b[-1].item()
+ for _ in range(NUM_STEPS):
+ la = model(mx.array([[next_a]]), cache=merged_a1)
+ mx.eval(la)
+ b1_logits_a.append(la[0, -1])
+ next_a = int(mx.argmax(la[0, -1]).item())
+ lb = model(mx.array([[next_b]]), cache=merged_b1)
+ mx.eval(lb)
+ b1_logits_b.append(lb[0, -1])
+ next_b = int(mx.argmax(lb[0, -1]).item())
+
+ # B=2 batched
+ cache_a2 = make_kv_cache(model)
+ cache_b2 = make_kv_cache(model)
+ prefill(model, tokenizer, sampler, tokens_a[:-1], cache_a2, None, None, None)
+ prefill(model, tokenizer, sampler, tokens_b[:-1], cache_b2, None, None, None)
+ merged_b2 = _merge_caches([list(cache_a2), list(cache_b2)])
+ for c in merged_b2:
+ c.prepare(lengths=[1, 1], right_padding=[0, 0])
+ model(
+ mx.array([[tokens_a[-2].item()], [tokens_b[-2].item()]]),
+ cache=merged_b2,
+ )
+ mx.eval([c.state for c in merged_b2])
+ for c in merged_b2:
+ c.finalize()
+
+ b2_logits_a: list[mx.array] = []
+ b2_logits_b: list[mx.array] = []
+ next_a2, next_b2 = tokens_a[-1].item(), tokens_b[-1].item()
+ for _ in range(NUM_STEPS):
+ l2 = model(mx.array([[next_a2], [next_b2]]), cache=merged_b2)
+ mx.eval(l2)
+ b2_logits_a.append(l2[0, -1])
+ b2_logits_b.append(l2[1, -1])
+ next_a2 = int(mx.argmax(l2[0, -1]).item())
+ next_b2 = int(mx.argmax(l2[1, -1]).item())
+
+ # Compare
+ max_diff = 0.0
+ mismatches = 0
+ for step in range(NUM_STEPS):
+ diff_a = float(
+ mx.max(
+ mx.abs(
+ b1_logits_a[step].astype(mx.float32)
+ - b2_logits_a[step].astype(mx.float32)
+ )
+ ).item()
+ )
+ diff_b = float(
+ mx.max(
+ mx.abs(
+ b1_logits_b[step].astype(mx.float32)
+ - b2_logits_b[step].astype(mx.float32)
+ )
+ ).item()
+ )
+ max_diff = max(max_diff, diff_a, diff_b)
+ if int(mx.argmax(b1_logits_a[step]).item()) != int(
+ mx.argmax(b2_logits_a[step]).item()
+ ):
+ mismatches += 1
+ if int(mx.argmax(b1_logits_b[step]).item()) != int(
+ mx.argmax(b2_logits_b[step]).item()
+ ):
+ mismatches += 1
+
+ return max_diff, mismatches
+
+
+def _make_tokenizer() -> TokenizerWrapper:
+ """Load the Qwen tokenizer (tiny download, shared across Qwen models)."""
+ from huggingface_hub import snapshot_download
+
+ model_path = Path(
+ snapshot_download(
+ "mlx-community/Qwen3.5-35B-A3B-4bit",
+ allow_patterns=["tokenizer*", "*.jinja"],
+ )
+ )
+ hf_tokenizer = AutoTokenizer.from_pretrained(model_path)
+ return TokenizerWrapper(hf_tokenizer)
+
+
+@pytest.mark.slow
+def test_batch_b2_llama() -> None:
+ """Llama-style model (KVCache only) must produce bit-exact logits in B=2.
+
+ Right-padded BatchKVCache keeps data at position 0 for all sequences,
+ so flash attention sees identical data layout as B=1 → bit-exact output.
+ """
+ from mlx_lm.models.llama import Model as LlamaModel
+ from mlx_lm.models.llama import ModelArgs
+
+ mx.random.seed(42)
+ args = ModelArgs(
+ model_type="llama",
+ hidden_size=256,
+ num_hidden_layers=4,
+ intermediate_size=512,
+ num_attention_heads=4,
+ num_key_value_heads=2,
+ rms_norm_eps=1e-6,
+ vocab_size=248320,
+ rope_theta=10000.0,
+ tie_word_embeddings=True,
+ )
+ model = LlamaModel(args)
+ _init_random(model)
+
+ tokenizer = _make_tokenizer()
+ tokens_a = encode_prompt(tokenizer, "Write a short essay about AI.")
+ tokens_b = encode_prompt(tokenizer, "Explain evolution briefly.")
+
+ max_diff, mismatches = _run_b1_vs_b2(
+ cast(Model, model), tokenizer, tokens_a, tokens_b
+ )
+ assert mismatches == 0, f"Llama B=2 token mismatches: {mismatches}/{NUM_STEPS * 2}"
+ assert max_diff < 0.002, f"Llama B=2 max logit diff: {max_diff}"
+
+
+@pytest.mark.slow
+def test_batch_b2_qwen35_moe() -> None:
+ """Qwen3.5 MoE model (hybrid SSM+attention+MoE) must produce bit-exact logits in B=2.
+
+ Right-padded BatchKVCache keeps data at position 0 for all sequences,
+ so flash attention sees identical data layout as B=1 → bit-exact output.
+ """
+ from mlx_lm.models.qwen3_5_moe import Model as Qwen35MoeModel
+ from mlx_lm.models.qwen3_5_moe import ModelArgs
+
+ mx.random.seed(42)
+ config = {
+ "model_type": "qwen3_5_moe",
+ "text_config": {
+ "model_type": "qwen3_5_moe_text",
+ "hidden_size": 256,
+ "num_hidden_layers": 8,
+ "intermediate_size": 512,
+ "num_attention_heads": 4,
+ "num_key_value_heads": 2,
+ "rms_norm_eps": 1e-6,
+ "vocab_size": 248320,
+ "head_dim": 64,
+ "max_position_embeddings": 4096,
+ "full_attention_interval": 4,
+ "layer_types": [
+ "linear_attention",
+ "linear_attention",
+ "linear_attention",
+ "full_attention",
+ "linear_attention",
+ "linear_attention",
+ "linear_attention",
+ "full_attention",
+ ],
+ "linear_conv_kernel_dim": 4,
+ "linear_key_head_dim": 64,
+ "linear_num_key_heads": 4,
+ "linear_num_value_heads": 4,
+ "linear_value_head_dim": 64,
+ "mamba_ssm_dtype": "float32",
+ "num_experts": 8,
+ "num_experts_per_tok": 2,
+ "moe_intermediate_size": 256,
+ "shared_expert_intermediate_size": 256,
+ "rope_parameters": {
+ "rope_type": "default",
+ "rope_theta": 10000000,
+ },
+ "attention_bias": False,
+ "attn_output_gate": True,
+ },
+ }
+ args = ModelArgs.from_dict(config)
+ model = Qwen35MoeModel(args)
+ _init_random(model)
+
+ tokenizer = _make_tokenizer()
+ tokens_a = encode_prompt(tokenizer, "Write a short essay about AI.")
+ tokens_b = encode_prompt(tokenizer, "Explain evolution briefly.")
+
+ max_diff, mismatches = _run_b1_vs_b2(
+ cast(Model, model), tokenizer, tokens_a, tokens_b
+ )
+ assert mismatches == 0, (
+ f"Qwen3.5 MoE B=2 token mismatches: {mismatches}/{NUM_STEPS * 2}"
+ )
+ assert max_diff < 0.002, f"Qwen3.5 MoE B=2 max logit diff: {max_diff}"
diff --git a/src/exo/worker/engines/mlx/utils_mlx.py b/src/exo/worker/engines/mlx/utils_mlx.py
index 4d50e8f3..8b22e6b3 100644
--- a/src/exo/worker/engines/mlx/utils_mlx.py
+++ b/src/exo/worker/engines/mlx/utils_mlx.py
@@ -41,6 +41,7 @@ from exo.download.download_utils import build_model_path
from exo.shared.types.common import Host
from exo.shared.types.memory import Memory
from exo.shared.types.mlx import Model
+from exo.shared.types.tasks import TaskId, TextGeneration
from exo.shared.types.text_generation import TextGenerationTaskParams
from exo.shared.types.worker.instances import (
BoundInstance,
@@ -750,3 +751,56 @@ def _parse_kimi_tool_calls(text: str):
return [_parse_single_tool(match) for match in tool_matches] # pyright: ignore[reportAny]
else:
return [_parse_single_tool(text)]
+
+
+def mx_all_gather_tasks(
+ tasks: list[TextGeneration],
+ group: mx.distributed.Group | None,
+) -> tuple[list[TextGeneration], list[TextGeneration]]:
+ def encode_task_id(task_id: TaskId) -> list[int]:
+ utf8_task_id = task_id.encode()
+ return [
+ int.from_bytes(utf8_task_id[i : i + 1]) for i in range(len(utf8_task_id))
+ ]
+
+ def decode_task_id(encoded_task_id: list[int]) -> TaskId:
+ return TaskId(
+ bytes.decode(b"".join((x).to_bytes(length=1) for x in encoded_task_id))
+ )
+
+ uuid_byte_length = 36
+
+ n_tasks = len(tasks)
+ all_counts = cast(
+ list[int],
+ mx.distributed.all_gather(mx.array([n_tasks]), group=group).tolist(),
+ )
+ max_tasks = max(all_counts)
+ world_size: int = 1 if group is None else group.size()
+
+ if max_tasks == 0:
+ return [], []
+
+ padded = [encode_task_id(task.task_id) for task in tasks] + [
+ [0] * uuid_byte_length
+ ] * (max_tasks - n_tasks)
+
+ assert all(len(encoded_task_id) == uuid_byte_length for encoded_task_id in padded)
+
+ gathered = cast(
+ list[list[list[int]]],
+ mx.distributed.all_gather(mx.array(padded), group=group)
+ .reshape(world_size, max_tasks, -1)
+ .tolist(),
+ )
+ all_task_ids: list[list[TaskId]] = [
+ [decode_task_id(encoded_task_id) for encoded_task_id in rank_tasks[:count]]
+ for rank_tasks, count in zip(gathered, all_counts, strict=True)
+ ]
+
+ agreed_ids = set[TaskId].intersection(*(set(tids) for tids in all_task_ids))
+
+ local_tasks = {task.task_id: task for task in tasks}
+ agreed = [local_tasks[tid] for tid in sorted(agreed_ids)]
+ different = [task for task in tasks if task.task_id not in agreed_ids]
+ return agreed, different
diff --git a/src/exo/worker/runner/bootstrap.py b/src/exo/worker/runner/bootstrap.py
index f4e3fccb..35d4250e 100644
--- a/src/exo/worker/runner/bootstrap.py
+++ b/src/exo/worker/runner/bootstrap.py
@@ -1,4 +1,5 @@
import os
+import resource
import loguru
@@ -21,6 +22,9 @@ def entrypoint(
global logger
logger = _logger
+ soft, hard = resource.getrlimit(resource.RLIMIT_NOFILE)
+ resource.setrlimit(resource.RLIMIT_NOFILE, (min(max(soft, 2048), hard), hard))
+
fast_synch_override = os.environ.get("EXO_FAST_SYNCH")
if fast_synch_override != "off":
os.environ["MLX_METAL_FAST_SYNCH"] = "1"
diff --git a/src/exo/worker/runner/image_models/runner.py b/src/exo/worker/runner/image_models/runner.py
index 0d7fa980..385192cd 100644
--- a/src/exo/worker/runner/image_models/runner.py
+++ b/src/exo/worker/runner/image_models/runner.py
@@ -1,5 +1,4 @@
import base64
-import resource
import time
from typing import TYPE_CHECKING, Literal
@@ -21,6 +20,7 @@ from exo.shared.types.events import (
TracesCollected,
)
from exo.shared.types.tasks import (
+ CANCEL_ALL_TASKS,
ConnectToGroup,
ImageEdits,
ImageGeneration,
@@ -195,9 +195,6 @@ class Runner:
self.cancel_receiver = cancel_receiver
self.bound_instance = bound_instance
- soft, hard = resource.getrlimit(resource.RLIMIT_NOFILE)
- resource.setrlimit(resource.RLIMIT_NOFILE, (min(max(soft, 2048), hard), hard))
-
self.instance, self.runner_id, self.shard_metadata = (
bound_instance.instance,
bound_instance.bound_runner_id,
@@ -244,11 +241,11 @@ class Runner:
if task.task_id in self.seen:
logger.warning("repeat task - potential error")
self.seen.add(task.task_id)
- self.cancelled_tasks.discard(TaskId("CANCEL_CURRENT_TASK"))
+ self.cancelled_tasks.discard(CANCEL_ALL_TASKS)
self.send_task_status(task, TaskStatus.Running)
self.handle_task(task)
was_cancelled = (task.task_id in self.cancelled_tasks) or (
- TaskId("CANCEL_CURRENT_TASK") in self.cancelled_tasks
+ CANCEL_ALL_TASKS in self.cancelled_tasks
)
if not was_cancelled:
self.send_task_status(task, TaskStatus.Complete)
diff --git a/src/exo/worker/runner/llm_inference/batch_generator.py b/src/exo/worker/runner/llm_inference/batch_generator.py
index 1ad7b6a6..e0382ca4 100644
--- a/src/exo/worker/runner/llm_inference/batch_generator.py
+++ b/src/exo/worker/runner/llm_inference/batch_generator.py
@@ -1,24 +1,24 @@
import itertools
-import math
import time
from abc import ABC, abstractmethod
from collections import deque
from collections.abc import Generator, Iterable
from dataclasses import dataclass, field
-from typing import cast
import mlx.core as mx
from mlx_lm.tokenizer_utils import TokenizerWrapper
+from exo.shared.constants import EXO_MAX_CONCURRENT_REQUESTS
from exo.shared.types.chunks import ErrorChunk, PrefillProgressChunk
from exo.shared.types.common import ModelId
from exo.shared.types.events import ChunkGenerated, Event
from exo.shared.types.mlx import Model
-from exo.shared.types.tasks import TaskId, TextGeneration
+from exo.shared.types.tasks import CANCEL_ALL_TASKS, TaskId, TextGeneration
from exo.shared.types.text_generation import TextGenerationTaskParams
from exo.shared.types.worker.runner_response import GenerationResponse, ToolCallResponse
from exo.utils.channels import MpReceiver, MpSender
from exo.worker.engines.mlx.cache import KVPrefixCache
+from exo.worker.engines.mlx.generator.batch_generate import ExoBatchGenerator
from exo.worker.engines.mlx.generator.generate import (
PrefillCancelled,
mlx_generate,
@@ -26,6 +26,7 @@ from exo.worker.engines.mlx.generator.generate import (
)
from exo.worker.engines.mlx.utils_mlx import (
apply_chat_template,
+ mx_all_gather_tasks,
mx_any,
)
from exo.worker.runner.bootstrap import logger
@@ -58,6 +59,14 @@ class GeneratorQueue[T]:
class InferenceGenerator(ABC):
+ _cancelled_tasks: set[TaskId]
+
+ def should_cancel(self, task_id: TaskId) -> bool:
+ return (
+ task_id in self._cancelled_tasks
+ or CANCEL_ALL_TASKS in self._cancelled_tasks
+ )
+
@abstractmethod
def warmup(self) -> None: ...
@@ -115,6 +124,8 @@ class SequentialGenerator(InferenceGenerator):
_cancelled_tasks: set[TaskId] = field(default_factory=set, init=False)
_maybe_queue: list[TextGeneration] = field(default_factory=list, init=False)
+ _maybe_cancel: list[TextGeneration] = field(default_factory=list, init=False)
+ _all_tasks: dict[TaskId, TextGeneration] = field(default_factory=dict, init=False)
_queue: deque[TextGeneration] = field(default_factory=deque, init=False)
_active: (
tuple[
@@ -130,37 +141,19 @@ class SequentialGenerator(InferenceGenerator):
) = field(default=None, init=False)
def warmup(self):
- logger.info(f"warming up inference for instance: {self.model_id}")
-
- t = time.monotonic()
- toks = warmup_inference(
+ self.check_for_cancel_every = warmup_inference(
model=self.model,
tokenizer=self.tokenizer,
group=self.group,
- )
- logger.info(f"warmed up by generating {toks} tokens")
- check_for_cancel_every = min(
- math.ceil(toks / min(time.monotonic() - t, 0.001)), 100
- )
- if self.group is not None:
- self.check_for_cancel_every = int(
- mx.max(
- mx.distributed.all_gather(
- mx.array([check_for_cancel_every]),
- group=self.group,
- )
- ).item()
- )
-
- logger.info(
- f"runner checking for cancellation every {check_for_cancel_every} tokens"
+ model_id=self.model_id,
)
def submit(
self,
task: TextGeneration,
) -> None:
- self._cancelled_tasks.discard(TaskId("CANCEL_CURRENT_TASK"))
+ self._cancelled_tasks.discard(CANCEL_ALL_TASKS)
+ self._all_tasks[task.task_id] = task
self._maybe_queue.append(task)
def agree_on_tasks(self) -> None:
@@ -169,6 +162,23 @@ class SequentialGenerator(InferenceGenerator):
self._queue.extend(task for task in self._maybe_queue if task in agreed)
self._maybe_queue = [task for task in self._maybe_queue if task in different]
+ def agree_on_cancellations(self) -> None:
+ """Agree between all ranks about which tasks to cancel."""
+ has_cancel_all = False
+ for task_id in self.cancel_receiver.collect():
+ if task_id == CANCEL_ALL_TASKS:
+ has_cancel_all = True
+ continue
+ if task_id in self._all_tasks:
+ self._maybe_cancel.append(self._all_tasks[task_id])
+
+ if mx_any(has_cancel_all, self.group):
+ self._cancelled_tasks.add(CANCEL_ALL_TASKS)
+
+ agreed, different = mx_all_gather_tasks(self._maybe_cancel, self.group)
+ self._cancelled_tasks.update(task.task_id for task in agreed)
+ self._maybe_cancel = list(different)
+
def step(
self,
) -> Iterable[
@@ -211,15 +221,19 @@ class SequentialGenerator(InferenceGenerator):
self._send_error(task, e)
raise
queue = GeneratorQueue[GenerationResponse]()
- output_generator = apply_all_parsers(
- queue.gen(),
- apply_chat_template(self.tokenizer, task.task_params),
- self.tool_parser,
- self.tokenizer,
- type(self.model),
- self.model_id,
- task.task_params.tools,
- )
+
+ if task.task_params.bench:
+ output_generator = queue.gen()
+ else:
+ output_generator = apply_all_parsers(
+ queue.gen(),
+ apply_chat_template(self.tokenizer, task.task_params),
+ self.tool_parser,
+ self.tokenizer,
+ type(self.model),
+ self.model_id,
+ task.task_params.tools,
+ )
self._active = (task, mlx_gen, queue, output_generator)
def _send_error(self, task: TextGeneration, e: Exception) -> None:
@@ -253,11 +267,8 @@ class SequentialGenerator(InferenceGenerator):
)
def distributed_prompt_progress_callback() -> None:
- self._cancelled_tasks.update(self.cancel_receiver.collect())
- want_to_cancel = (task.task_id in self._cancelled_tasks) or (
- TaskId("CANCEL_CURRENT_TASK") in self._cancelled_tasks
- )
- if mx_any(want_to_cancel, self.group):
+ self.agree_on_cancellations()
+ if self.should_cancel(task.task_id):
raise PrefillCancelled()
self.agree_on_tasks()
@@ -269,11 +280,8 @@ class SequentialGenerator(InferenceGenerator):
tokens_since_cancel_check += 1
if tokens_since_cancel_check >= self.check_for_cancel_every:
tokens_since_cancel_check = 0
- self._cancelled_tasks.update(self.cancel_receiver.collect())
- want_to_cancel = (task.task_id in self._cancelled_tasks) or (
- TaskId("CANCEL_CURRENT_TASK") in self._cancelled_tasks
- )
- if mx_any(want_to_cancel, self.group):
+ self.agree_on_cancellations()
+ if self.should_cancel(task.task_id):
raise PrefillCancelled()
self.agree_on_tasks()
@@ -294,53 +302,228 @@ class SequentialGenerator(InferenceGenerator):
del self.model, self.tokenizer, self.group
-def mx_all_gather_tasks(
- tasks: list[TextGeneration],
- group: mx.distributed.Group | None,
-) -> tuple[list[TextGeneration], list[TextGeneration]]:
- def encode_task_id(task_id: TaskId) -> list[int]:
- utf8_task_id = task_id.encode()
- return [
- int.from_bytes(utf8_task_id[i : i + 1]) for i in range(len(utf8_task_id))
- ]
+@dataclass(eq=False)
+class BatchGenerator(InferenceGenerator):
+ model: Model
+ tokenizer: TokenizerWrapper
+ group: mx.distributed.Group | None
+ kv_prefix_cache: KVPrefixCache | None
+ tool_parser: ToolParser | None
+ model_id: ModelId
+ device_rank: int
+ cancel_receiver: MpReceiver[TaskId]
+ event_sender: MpSender[Event]
+ check_for_cancel_every: int = 50
+
+ _cancelled_tasks: set[TaskId] = field(default_factory=set, init=False)
+ _maybe_queue: list[TextGeneration] = field(default_factory=list, init=False)
+ _maybe_cancel: list[TextGeneration] = field(default_factory=list, init=False)
+ _all_tasks: dict[TaskId, TextGeneration] = field(default_factory=dict, init=False)
+ _queue: deque[TextGeneration] = field(default_factory=deque, init=False)
+ _mlx_gen: ExoBatchGenerator = field(init=False)
+ _active_tasks: dict[
+ int,
+ tuple[
+ TextGeneration,
+ GeneratorQueue[GenerationResponse],
+ Generator[GenerationResponse | ToolCallResponse | None],
+ ],
+ ] = field(default_factory=dict, init=False)
- def decode_task_id(encoded_task_id: list[int]) -> TaskId:
- return TaskId(
- bytes.decode(b"".join((x).to_bytes(length=1) for x in encoded_task_id))
+ def __post_init__(self) -> None:
+ self._mlx_gen = ExoBatchGenerator(
+ model=self.model,
+ tokenizer=self.tokenizer,
+ group=self.group,
+ kv_prefix_cache=self.kv_prefix_cache,
+ )
+
+ def warmup(self):
+ self.check_for_cancel_every = warmup_inference(
+ model=self.model,
+ tokenizer=self.tokenizer,
+ group=self.group,
+ model_id=self.model_id,
)
- uuid_byte_length = 36
-
- n_tasks = len(tasks)
- all_counts = cast(
- list[int],
- mx.distributed.all_gather(mx.array([n_tasks]), group=group).tolist(),
- )
- max_tasks = max(all_counts)
- world_size: int = 1 if group is None else group.size()
-
- if max_tasks == 0:
- return [], []
-
- padded = [encode_task_id(task.task_id) for task in tasks] + [
- [0] * uuid_byte_length
- ] * (max_tasks - n_tasks)
- gathered = cast(
- list[list[list[int]]],
- mx.distributed.all_gather(mx.array(padded), group=group)
- .reshape(world_size, max_tasks, -1)
- .tolist(),
- )
- all_task_ids: list[list[TaskId]] = [
- [decode_task_id(encoded_task_id) for encoded_task_id in rank_tasks[:count]]
- for rank_tasks, count in zip(gathered, all_counts, strict=True)
- ]
-
- agreed_ids: set[TaskId] = set(all_task_ids[0])
- for rank_tasks in all_task_ids[1:]:
- agreed_ids &= set(rank_tasks)
-
- local_tasks = {task.task_id: task for task in tasks}
- agreed = [local_tasks[tid] for tid in sorted(agreed_ids)]
- different = [task for task in tasks if task.task_id not in agreed_ids]
- return agreed, different
+ def submit(
+ self,
+ task: TextGeneration,
+ ) -> None:
+ self._cancelled_tasks.discard(CANCEL_ALL_TASKS)
+ self._all_tasks[task.task_id] = task
+ self._maybe_queue.append(task)
+
+ def agree_on_tasks(self) -> None:
+ """Agree between all ranks about the task ordering (some may have received in different order or not at all)."""
+ agreed, different = mx_all_gather_tasks(self._maybe_queue, self.group)
+ self._queue.extend(task for task in self._maybe_queue if task in agreed)
+ self._maybe_queue = [task for task in self._maybe_queue if task in different]
+
+ def agree_on_cancellations(self) -> None:
+ """Agree between all ranks about which tasks to cancel."""
+ has_cancel_all = False
+ for task_id in self.cancel_receiver.collect():
+ if task_id == CANCEL_ALL_TASKS:
+ has_cancel_all = True
+ continue
+ if task_id in self._all_tasks:
+ self._maybe_cancel.append(self._all_tasks[task_id])
+
+ if mx_any(has_cancel_all, self.group):
+ self._cancelled_tasks.add(CANCEL_ALL_TASKS)
+
+ agreed, different = mx_all_gather_tasks(self._maybe_cancel, self.group)
+ self._cancelled_tasks.update(task.task_id for task in agreed)
+ self._maybe_cancel = list(different)
+
+ def step(
+ self,
+ ) -> Iterable[
+ tuple[TaskId, GenerationResponse | ToolCallResponse | Cancelled | Finished]
+ ]:
+ if not self._queue:
+ self.agree_on_tasks()
+
+ # Submit any queued tasks to the engine
+ while self._queue and len(self._active_tasks) < EXO_MAX_CONCURRENT_REQUESTS:
+ task = self._queue.popleft()
+ try:
+ uid = self._start_task(task)
+ except PrefillCancelled:
+ continue
+ except Exception as e:
+ self._send_error(task, e)
+ raise
+
+ queue = GeneratorQueue[GenerationResponse]()
+ if task.task_params.bench:
+ output_generator = queue.gen()
+ else:
+ output_generator = apply_all_parsers(
+ queue.gen(),
+ apply_chat_template(self.tokenizer, task.task_params),
+ self.tool_parser,
+ self.tokenizer,
+ type(self.model),
+ self.model_id,
+ task.task_params.tools,
+ )
+ self._active_tasks[uid] = (task, queue, output_generator)
+
+ if not self._mlx_gen.has_work:
+ return self._apply_cancellations()
+
+ results = self._mlx_gen.step()
+
+ output: list[
+ tuple[TaskId, GenerationResponse | ToolCallResponse | Cancelled | Finished]
+ ] = []
+ for uid, response in results:
+ if uid not in self._active_tasks:
+ # should we error here?
+ logger.warning(f"{uid=} not found in active tasks")
+ continue
+
+ task, queue, output_generator = self._active_tasks[uid]
+ queue.push(response)
+ parsed = next(output_generator)
+
+ if parsed is not None:
+ output.append((task.task_id, parsed))
+
+ if response.finish_reason is not None:
+ output.append((task.task_id, Finished()))
+ del self._active_tasks[uid]
+
+ return itertools.chain(output, self._apply_cancellations())
+
+ def _apply_cancellations(
+ self,
+ ) -> list[tuple[TaskId, Cancelled]]:
+ if not self._cancelled_tasks:
+ return []
+
+ cancel_all = CANCEL_ALL_TASKS in self._cancelled_tasks
+
+ uids_to_cancel: list[int] = []
+ results: list[tuple[TaskId, Cancelled]] = []
+
+ for uid, (task, _, _) in list(self._active_tasks.items()):
+ if task.task_id in self._cancelled_tasks or cancel_all:
+ uids_to_cancel.append(uid)
+ results.append((task.task_id, Cancelled()))
+ del self._active_tasks[uid]
+
+ if uids_to_cancel:
+ self._mlx_gen.cancel(uids_to_cancel)
+
+ already_cancelled = {tid for tid, _ in results}
+ for tid in self._cancelled_tasks:
+ if tid != CANCEL_ALL_TASKS and tid not in already_cancelled:
+ results.append((tid, Cancelled()))
+
+ self._cancelled_tasks.clear()
+ return results
+
+ def _send_error(self, task: TextGeneration, e: Exception) -> None:
+ if self.device_rank == 0:
+ self.event_sender.send(
+ ChunkGenerated(
+ command_id=task.command_id,
+ chunk=ErrorChunk(
+ model=self.model_id,
+ finish_reason="error",
+ error_message=str(e),
+ ),
+ )
+ )
+
+ def _start_task(self, task: TextGeneration) -> int:
+ _check_for_debug_prompts(task.task_params)
+ prompt = apply_chat_template(self.tokenizer, task.task_params)
+
+ def on_prefill_progress(processed: int, total: int) -> None:
+ if self.device_rank == 0:
+ self.event_sender.send(
+ ChunkGenerated(
+ command_id=task.command_id,
+ chunk=PrefillProgressChunk(
+ model=self.model_id,
+ processed_tokens=processed,
+ total_tokens=total,
+ ),
+ )
+ )
+
+ def distributed_prompt_progress_callback() -> None:
+ self.agree_on_cancellations()
+ if self.should_cancel(task.task_id):
+ raise PrefillCancelled()
+
+ self.agree_on_tasks()
+
+ tokens_since_cancel_check = self.check_for_cancel_every
+
+ def on_generation_token() -> None:
+ nonlocal tokens_since_cancel_check
+ tokens_since_cancel_check += 1
+ if tokens_since_cancel_check >= self.check_for_cancel_every:
+ tokens_since_cancel_check = 0
+ self.agree_on_cancellations()
+ if self.should_cancel(task.task_id):
+ self._cancelled_tasks.add(task.task_id)
+
+ self.agree_on_tasks()
+
+ return self._mlx_gen.submit(
+ task_params=task.task_params,
+ prompt=prompt,
+ on_prefill_progress=on_prefill_progress,
+ distributed_prompt_progress_callback=distributed_prompt_progress_callback,
+ on_generation_token=on_generation_token,
+ )
+
+ def close(self) -> None:
+ self._mlx_gen.close()
+ del self.model, self.tokenizer, self.group
diff --git a/src/exo/worker/runner/llm_inference/model_output_parsers.py b/src/exo/worker/runner/llm_inference/model_output_parsers.py
index bfdd6e4a..cb5570e1 100644
--- a/src/exo/worker/runner/llm_inference/model_output_parsers.py
+++ b/src/exo/worker/runner/llm_inference/model_output_parsers.py
@@ -21,8 +21,7 @@ from exo.worker.engines.mlx.utils_mlx import (
detect_thinking_prompt_suffix,
)
from exo.worker.runner.bootstrap import logger
-
-from .tool_parsers import ToolParser
+from exo.worker.runner.llm_inference.tool_parsers import ToolParser
@cache
@@ -45,7 +44,8 @@ def apply_all_parsers(
if tokenizer.has_thinking:
mlx_generator = parse_thinking_models(
mlx_generator,
- tokenizer,
+ tokenizer.think_start,
+ tokenizer.think_end,
starts_in_thinking=detect_thinking_prompt_suffix(prompt, tokenizer),
)
@@ -300,7 +300,8 @@ def _could_be_dsml_prefix(text: str) -> bool:
def parse_thinking_models(
responses: Generator[GenerationResponse | None],
- tokenizer: TokenizerWrapper,
+ think_start: str | None,
+ think_end: str | None,
starts_in_thinking: bool = True,
) -> Generator[GenerationResponse | None]:
"""Route thinking tokens via is_thinking flag.
@@ -308,25 +309,23 @@ def parse_thinking_models(
Swallows think tag tokens, sets is_thinking on all others.
Always yields tokens with finish_reason to avoid hanging the chunk stream.
"""
- in_thinking = starts_in_thinking
+ is_thinking = starts_in_thinking
for response in responses:
if response is None:
yield None
continue
+ if response.finish_reason is not None:
+ yield response.model_copy(update={"is_thinking": False})
+ continue
- is_think_tag = (
- tokenizer.think_end is not None and response.text == tokenizer.think_end
- ) or (
- tokenizer.think_start is not None and response.text == tokenizer.think_start
- )
-
- if is_think_tag:
- in_thinking = response.text != tokenizer.think_end
- # Never swallow finish_reason — the chunk stream needs it to terminate.
- if response.finish_reason is not None:
- yield response.model_copy(update={"text": "", "is_thinking": False})
+ if response.text == think_start:
+ is_thinking = True
continue
- yield response.model_copy(update={"is_thinking": in_thinking})
+ if response.text == think_end:
+ is_thinking = False
+ continue
+
+ yield response.model_copy(update={"is_thinking": is_thinking})
def parse_tool_calls(
@@ -340,42 +339,41 @@ def parse_tool_calls(
if response is None:
yield None
continue
+
if not in_tool_call and response.text.startswith(tool_parser.start_parsing):
in_tool_call = True
- if in_tool_call:
- tool_call_text_parts.append(response.text)
- if response.text.endswith(tool_parser.end_parsing):
- # parse the actual tool calls from the tool call text
- parsed = tool_parser.parse("".join(tool_call_text_parts).strip(), tools)
- logger.info(f"parsed {tool_call_text_parts=} into {parsed=}")
- if parsed is not None:
- yield ToolCallResponse(
- tool_calls=parsed, usage=response.usage, stats=response.stats
- )
- else:
- logger.warning(
- f"tool call parsing failed for text {''.join(tool_call_text_parts)}"
- )
- response.text = "".join(tool_call_text_parts)
- yield response
+ if not in_tool_call:
+ yield response
+ continue
- in_tool_call = False
- tool_call_text_parts = []
+ tool_call_text_parts.append(response.text)
+ if response.text.endswith(tool_parser.end_parsing):
+ # parse the actual tool calls from the tool call text
+ combined = "".join(tool_call_text_parts)
+ parsed = tool_parser.parse(combined.strip(), tools=tools)
+ logger.info(f"parsed {tool_call_text_parts=} into {parsed=}")
+ in_tool_call = False
+ tool_call_text_parts = []
+
+ if parsed is None:
+ logger.warning(f"tool call parsing failed for text {combined}")
+ yield response.model_copy(update={"text": combined})
continue
- if response.finish_reason is not None:
- logger.info(
- "tool call parsing interrupted, yield partial tool call as text"
- )
- response = response.model_copy(
- update={
- "text": "".join(tool_call_text_parts),
- "token": 0,
- }
- )
- yield response
+ yield ToolCallResponse(
+ tool_calls=parsed, usage=response.usage, stats=response.stats
+ )
+ continue
- else:
- # fallthrough
+ if response.finish_reason is not None:
+ logger.info(
+ "tool call parsing interrupted, yield partial tool call as text"
+ )
+ response = response.model_copy(
+ update={
+ "text": "".join(tool_call_text_parts),
+ "token": 0,
+ }
+ )
yield response
diff --git a/src/exo/worker/runner/llm_inference/runner.py b/src/exo/worker/runner/llm_inference/runner.py
index 9ff744ac..1f9f1e05 100644
--- a/src/exo/worker/runner/llm_inference/runner.py
+++ b/src/exo/worker/runner/llm_inference/runner.py
@@ -1,4 +1,4 @@
-import resource
+import os
import time
from dataclasses import dataclass
from enum import Enum
@@ -59,12 +59,13 @@ from exo.worker.engines.mlx.utils_mlx import (
)
from exo.worker.runner.bootstrap import logger
from exo.worker.runner.llm_inference.batch_generator import (
+ BatchGenerator,
InferenceGenerator,
SequentialGenerator,
)
from .batch_generator import Cancelled, Finished
-from .tool_parsers import ToolParser, make_mlx_parser
+from .tool_parsers import make_mlx_parser
class ExitCode(str, Enum):
@@ -85,9 +86,6 @@ class Runner:
self.cancel_receiver = cancel_receiver
self.bound_instance = bound_instance
- soft, hard = resource.getrlimit(resource.RLIMIT_NOFILE)
- resource.setrlimit(resource.RLIMIT_NOFILE, (min(max(soft, 2048), hard), hard))
-
self.instance, self.runner_id, self.shard_metadata = (
self.bound_instance.instance,
self.bound_instance.bound_runner_id,
@@ -205,18 +203,6 @@ class Runner:
on_layer_loaded=on_layer_loaded,
)
)
- logger.info(
- f"model has_tool_calling={self.generator.tokenizer.has_tool_calling} using tokens {self.generator.tokenizer.tool_call_start}, {self.generator.tokenizer.tool_call_end}"
- )
- tok = self.generator.tokenizer
- if tok.tool_call_start and tok.tool_call_end and tok.tool_parser: # pyright: ignore[reportAny]
- self.generator.tool_parser = make_mlx_parser(
- tok.tool_call_start,
- tok.tool_call_end,
- tok.tool_parser, # pyright: ignore[reportAny]
- )
-
- self.generator.kv_prefix_cache = KVPrefixCache(self.generator.group)
self.generator = self.generator.build()
@@ -302,7 +288,7 @@ class Runner:
)
for task_id in finished:
- del self.active_tasks[task_id]
+ self.active_tasks.pop(task_id, None)
try:
task = self.task_receiver.receive_nowait()
@@ -392,9 +378,7 @@ class Builder:
cancel_receiver: MpReceiver[TaskId]
inference_model: Model | None = None
tokenizer: TokenizerWrapper | None = None
- tool_parser: ToolParser | None = None
group: mx.distributed.Group | None = None
- kv_prefix_cache: KVPrefixCache | None = None
def build(
self,
@@ -403,14 +387,46 @@ class Builder:
assert self.inference_model
assert self.tokenizer
- return SequentialGenerator(
+ tool_parser = None
+ logger.info(
+ f"model has_tool_calling={self.tokenizer.has_tool_calling} using tokens {self.tokenizer.tool_call_start}, {self.tokenizer.tool_call_end}"
+ )
+ if (
+ self.tokenizer.tool_call_start
+ and self.tokenizer.tool_call_end
+ and self.tokenizer.tool_parser # type: ignore
+ ):
+ tool_parser = make_mlx_parser(
+ self.tokenizer.tool_call_start,
+ self.tokenizer.tool_call_end,
+ self.tokenizer.tool_parser, # type: ignore
+ )
+
+ kv_prefix_cache = KVPrefixCache(self.group)
+
+ device_rank = 0 if self.group is None else self.group.rank()
+ if os.environ.get("EXO_NO_BATCH"):
+ logger.info("using SequentialGenerator (batching disabled)")
+ return SequentialGenerator(
+ model=self.inference_model,
+ tokenizer=self.tokenizer,
+ group=self.group,
+ tool_parser=tool_parser,
+ kv_prefix_cache=kv_prefix_cache,
+ model_id=self.model_id,
+ device_rank=device_rank,
+ cancel_receiver=self.cancel_receiver,
+ event_sender=self.event_sender,
+ )
+ logger.info("using BatchGenerator")
+ return BatchGenerator(
model=self.inference_model,
tokenizer=self.tokenizer,
group=self.group,
- tool_parser=self.tool_parser,
- kv_prefix_cache=self.kv_prefix_cache,
+ tool_parser=tool_parser,
+ kv_prefix_cache=kv_prefix_cache,
model_id=self.model_id,
- device_rank=0 if self.group is None else self.group.rank(),
+ device_rank=device_rank,
cancel_receiver=self.cancel_receiver,
event_sender=self.event_sender,
)
diff --git a/src/exo/worker/runner/runner_supervisor.py b/src/exo/worker/runner/runner_supervisor.py
index 8ada028d..b0882f39 100644
--- a/src/exo/worker/runner/runner_supervisor.py
+++ b/src/exo/worker/runner/runner_supervisor.py
@@ -21,6 +21,7 @@ from exo.shared.types.events import (
TaskStatusUpdated,
)
from exo.shared.types.tasks import (
+ CANCEL_ALL_TASKS,
ImageEdits,
ImageGeneration,
Task,
@@ -125,7 +126,7 @@ class RunnerSupervisor:
with contextlib.suppress(ClosedResourceError):
self._event_sender.close()
with contextlib.suppress(ClosedResourceError):
- self._cancel_sender.send(TaskId("CANCEL_CURRENT_TASK"))
+ self._cancel_sender.send(CANCEL_ALL_TASKS)
with contextlib.suppress(ClosedResourceError):
self._cancel_sender.close()
self.runner_process.join(5)
diff --git a/src/exo/worker/tests/unittests/test_mlx/test_batch_vs_generate.py b/src/exo/worker/tests/unittests/test_mlx/test_batch_vs_generate.py
new file mode 100644
index 00000000..5d1a792b
--- /dev/null
+++ b/src/exo/worker/tests/unittests/test_mlx/test_batch_vs_generate.py
@@ -0,0 +1,389 @@
+import copy
+import gc
+import json
+import shutil
+import tempfile
+from pathlib import Path
+from typing import Any, cast
+
+import mlx.core as mx
+import pytest
+from mlx_lm.tokenizer_utils import TokenizerWrapper
+
+from exo.shared.types.common import ModelId
+from exo.shared.types.mlx import KVCacheType, Model
+from exo.shared.types.text_generation import InputMessage, TextGenerationTaskParams
+from exo.worker.engines.mlx.cache import CacheSnapshot, KVPrefixCache, cache_length
+from exo.worker.engines.mlx.generator.batch_generate import ExoBatchGenerator
+from exo.worker.engines.mlx.generator.generate import mlx_generate
+from exo.worker.engines.mlx.utils_mlx import (
+ apply_chat_template,
+ load_tokenizer_for_model_id,
+)
+
+from .test_prefix_cache_architectures import (
+ ARCHITECTURES,
+ ArchSpec,
+ _arch_available, # pyright: ignore[reportPrivateUsage]
+ _build_model, # pyright: ignore[reportPrivateUsage]
+ _copy_tokenizer, # pyright: ignore[reportPrivateUsage]
+ _find_snapshot, # pyright: ignore[reportPrivateUsage]
+ _reduce_config, # pyright: ignore[reportPrivateUsage]
+)
+
+
+def _make_task(
+ content: str = "Hello, what is 2+2?",
+ max_tokens: int = 10,
+ seed: int = 42,
+) -> TextGenerationTaskParams:
+ return TextGenerationTaskParams(
+ model=ModelId("test"),
+ input=[InputMessage(role="user", content=content)],
+ max_output_tokens=max_tokens,
+ temperature=0.7,
+ seed=seed,
+ )
+
+
+# ── Helpers ──────────────────────────────────────────────────────────────── #
+
+
+def _collect_mlx_generate(
+ model: Model,
+ tokenizer: TokenizerWrapper,
+ task: TextGenerationTaskParams,
+ kv_prefix_cache: KVPrefixCache | None,
+) -> list[int]:
+ """Run mlx_generate and collect output token IDs."""
+ prompt = apply_chat_template(tokenizer=tokenizer, task_params=task)
+ tokens: list[int] = []
+ for resp in mlx_generate(
+ model=model,
+ tokenizer=tokenizer,
+ task=task,
+ prompt=prompt,
+ kv_prefix_cache=kv_prefix_cache,
+ group=None,
+ ):
+ tokens.append(resp.token)
+ if resp.finish_reason is not None:
+ break
+ return tokens
+
+
+def _collect_batch_generate(
+ model: Model,
+ tokenizer: TokenizerWrapper,
+ task_params: TextGenerationTaskParams,
+ kv_prefix_cache: KVPrefixCache | None,
+) -> list[int]:
+ """Run ExoBatchGenerator and collect raw output token IDs"""
+ exo_gen = ExoBatchGenerator(
+ model=model,
+ tokenizer=tokenizer,
+ group=None,
+ kv_prefix_cache=kv_prefix_cache,
+ )
+
+ prompt = apply_chat_template(tokenizer=tokenizer, task_params=task_params)
+ exo_gen.submit(task_params=task_params, prompt=prompt)
+
+ tokens: list[int] = []
+ while exo_gen.has_work:
+ results = exo_gen.step()
+ for _uid, response in results:
+ tokens.append(response.token)
+
+ exo_gen.close()
+ return tokens
+
+
+def _assert_state_equal(sa: object, sb: object, label: str) -> None:
+ """Compare two state items, handling both plain arrays and tuples of arrays (CacheList)."""
+ if isinstance(sa, tuple):
+ assert isinstance(sb, tuple), f"{label}: type mismatch"
+ for k, (arr_a, arr_b) in enumerate(
+ zip(
+ cast(tuple[mx.array, ...], sa),
+ cast(tuple[mx.array, ...], sb),
+ strict=True,
+ )
+ ):
+ a_f = mx.array(arr_a).astype(mx.float32)
+ b_f = mx.array(arr_b).astype(mx.float32)
+ if a_f.size == 0:
+ assert b_f.size == 0, f"{label}[{k}]: size mismatch"
+ continue
+ diff = float(mx.max(mx.abs(a_f - b_f)).item())
+ assert diff == 0.0, f"{label}[{k}]: max diff {diff}"
+ else:
+ sa_f = mx.array(cast(mx.array, sa)).astype(mx.float32)
+ sb_f = mx.array(cast(mx.array, sb)).astype(mx.float32)
+ if sa_f.size == 0:
+ assert sb_f.size == 0, f"{label}: size mismatch"
+ return
+ diff = float(mx.max(mx.abs(sa_f - sb_f)).item())
+ assert diff == 0.0, f"{label}: max diff {diff}"
+
+
+def _compare_cache_arrays(
+ cache_a: KVCacheType,
+ cache_b: KVCacheType,
+ label: str = "",
+) -> None:
+ """Assert two KV caches have identical array values."""
+ assert len(cache_a) == len(cache_b), (
+ f"{label}Cache layer count: {len(cache_a)} vs {len(cache_b)}"
+ )
+ for i, (a, b) in enumerate(zip(cache_a, cache_b, strict=True)):
+ assert type(a) is type(b), (
+ f"{label}Layer {i}: type {type(a).__name__} vs {type(b).__name__}"
+ )
+ states_a = a.state
+ states_b = b.state
+ assert len(states_a) == len(states_b), (
+ f"{label}Layer {i}: state count {len(states_a)} vs {len(states_b)}"
+ )
+ for j, (sa, sb) in enumerate(zip(states_a, states_b, strict=True)):
+ if sa is None and sb is None:
+ continue
+ assert sa is not None and sb is not None, (
+ f"{label}Layer {i}, state {j}: one is None"
+ )
+ _assert_state_equal(sa, sb, f"{label}Layer {i}, state {j}")
+
+
+def _safe_state(cache: object) -> list[object]:
+ """Safely access .state on a cache object. Returns [] if uninitialized."""
+ # RotatingKVCache.state crashes when keys is None (uninitialized)
+ if getattr(cache, "keys", _SENTINEL) is None:
+ return []
+ try:
+ return list(cache.state) # type: ignore[union-attr]
+ except (AttributeError, TypeError):
+ return []
+
+
+_SENTINEL = object()
+
+
+def _compare_snapshots(
+ snaps_a: list[CacheSnapshot] | None,
+ snaps_b: list[CacheSnapshot] | None,
+ label: str = "",
+) -> None:
+ """Assert two snapshot lists are identical."""
+ if snaps_a is None:
+ assert snaps_b is None, f"{label}One side has snapshots, other doesn't"
+ return
+ assert snaps_b is not None, f"{label}One side has snapshots, other doesn't"
+ assert len(snaps_a) == len(snaps_b), (
+ f"{label}Snapshot count: {len(snaps_a)} vs {len(snaps_b)}"
+ )
+ for k, (sa, sb) in enumerate(zip(snaps_a, snaps_b, strict=True)):
+ assert sa.token_count == sb.token_count, (
+ f"{label}Snapshot {k} token_count: {sa.token_count} vs {sb.token_count}"
+ )
+ for layer_i, (s1, s2) in enumerate(zip(sa.states, sb.states, strict=True)):
+ if s1 is None and s2 is None:
+ continue
+ assert s1 is not None and s2 is not None, (
+ f"{label}Snapshot {k}, layer {layer_i}: one state is None"
+ )
+ state_a = _safe_state(s1)
+ state_b = _safe_state(s2)
+ if not state_a and not state_b:
+ continue
+ assert len(state_a) == len(state_b), (
+ f"{label}Snapshot {k}, layer {layer_i}: state length mismatch"
+ )
+ for st_j, (arr_a, arr_b) in enumerate(zip(state_a, state_b, strict=True)):
+ if arr_a is None and arr_b is None:
+ continue
+ assert arr_a is not None and arr_b is not None
+ _assert_state_equal(
+ arr_a,
+ arr_b,
+ f"{label}Snapshot {k}, layer {layer_i}, state {st_j}",
+ )
+
+
+# ── Test class ────────────────────────────────────────────────────────────── #
+
+
+@pytest.mark.slow
+class TestBatchVsGenerate:
+ """Verify BatchGenerator matches mlx_generate for output tokens and prefix cache."""
+
+ @pytest.fixture(autouse=True)
+ def _cleanup(self):
+ yield
+ mx.clear_cache()
+ gc.collect()
+
+ @pytest.mark.parametrize(
+ "spec",
+ ARCHITECTURES,
+ ids=[a.name for a in ARCHITECTURES],
+ )
+ def test_same_output_and_cache(self, spec: ArchSpec) -> None:
+ if not _arch_available(spec):
+ pytest.skip(f"Model {spec.hub_name} not cached locally")
+
+ snapshot = _find_snapshot(spec.hub_name)
+ assert snapshot is not None
+
+ tmpdir = Path(tempfile.mkdtemp(prefix=f"exo_batchtest_{spec.name}_"))
+ try:
+ # Build reduced config
+ with open(snapshot / "config.json") as f:
+ cfg = cast(dict[str, Any], json.load(f))
+ reduced = _reduce_config(copy.deepcopy(cfg))
+ (tmpdir / "config.json").write_text(json.dumps(reduced))
+
+ # Copy tokenizer
+ tok_src = snapshot
+ if spec.tokenizer_hub is not None:
+ alt = _find_snapshot(spec.tokenizer_hub)
+ if alt is not None:
+ tok_src = alt
+ _copy_tokenizer(tok_src, tmpdir)
+
+ # Load tokenizer, build model with random weights
+ model_id = ModelId(f"mlx-community/{spec.hub_name}")
+ tokenizer = load_tokenizer_for_model_id(model_id, tmpdir)
+ mx.random.seed(0)
+ model = _build_model(spec.module, reduced)
+
+ task = _make_task()
+
+ # ── Run mlx_generate path ──
+ # Seed is set inside mlx_generate/ExoBatchGenerator.submit from task.seed
+ kv_mlx = KVPrefixCache(None)
+ mlx_tokens = _collect_mlx_generate(model, tokenizer, task, kv_mlx)
+
+ # ── Run batch generator path ──
+ kv_batch = KVPrefixCache(None)
+ batch_tokens = _collect_batch_generate(model, tokenizer, task, kv_batch)
+
+ # ── Compare output tokens ──
+ assert len(mlx_tokens) > 0, "mlx_generate produced no tokens"
+ assert len(batch_tokens) > 0, "BatchGenerator produced no tokens"
+ assert mlx_tokens == batch_tokens, (
+ f"[{spec.name}] Token mismatch:\n"
+ f" mlx_generate: {mlx_tokens}\n"
+ f" BatchGenerator: {batch_tokens}"
+ )
+
+ # ── Compare prefix cache KV arrays ──
+ assert len(kv_mlx.caches) == 1, "mlx_generate didn't save to prefix cache"
+ assert len(kv_batch.caches) == 1, (
+ "BatchGenerator didn't save to prefix cache"
+ )
+
+ _compare_cache_arrays(
+ kv_mlx.caches[0],
+ kv_batch.caches[0],
+ label=f"[{spec.name}] ",
+ )
+
+ # ── Compare cache lengths ──
+ mlx_len = cache_length(kv_mlx.caches[0])
+ batch_len = cache_length(kv_batch.caches[0])
+ assert mlx_len == batch_len, (
+ f"[{spec.name}] Cache length: mlx={mlx_len} vs batch={batch_len}"
+ )
+
+ # ── Compare snapshots ──
+ _compare_snapshots(
+ kv_mlx._snapshots[0], # pyright: ignore[reportPrivateUsage]
+ kv_batch._snapshots[0], # pyright: ignore[reportPrivateUsage]
+ label=f"[{spec.name}] ",
+ )
+
+ finally:
+ shutil.rmtree(tmpdir, ignore_errors=True)
+
+ @pytest.mark.parametrize(
+ "spec",
+ ARCHITECTURES,
+ ids=[a.name for a in ARCHITECTURES],
+ )
+ def test_concurrent_batch_completes(self, spec: ArchSpec) -> None:
+ """Two requests processed concurrently must both complete without
+ crashing and produce non-empty output.
+
+ Note: batch decode logits are NOT bit-exact with sequential because
+ Metal's matmul kernel picks different reduction tiling for B=1 vs B=2
+ when L=1 (decode step). This introduces sub-ULP float16 diffs in
+ gate_proj/down_proj/lm_head which swiglu amplifies by |up_values|.
+ With random weights these accumulate into argmax flips; with trained
+ weights the diffs are absorbed and output matches exactly (verified
+ with real Llama-3.2-1B-Instruct-4bit weights).
+ """
+ if not _arch_available(spec):
+ pytest.skip(f"Model {spec.hub_name} not cached locally")
+
+ snapshot = _find_snapshot(spec.hub_name)
+ assert snapshot is not None
+
+ tmpdir = Path(tempfile.mkdtemp(prefix=f"exo_concurrent_{spec.name}_"))
+ try:
+ with open(snapshot / "config.json") as f:
+ cfg = cast(dict[str, Any], json.load(f))
+ reduced = _reduce_config(copy.deepcopy(cfg))
+ (tmpdir / "config.json").write_text(json.dumps(reduced))
+
+ tok_src = snapshot
+ if spec.tokenizer_hub is not None:
+ alt = _find_snapshot(spec.tokenizer_hub)
+ if alt is not None:
+ tok_src = alt
+ _copy_tokenizer(tok_src, tmpdir)
+
+ model_id = ModelId(f"mlx-community/{spec.hub_name}")
+ tokenizer = load_tokenizer_for_model_id(model_id, tmpdir)
+ mx.random.seed(0)
+ model = _build_model(spec.module, reduced)
+
+ # Two different prompts → different prompt lengths.
+ task_a = _make_task(content="Hello, what is 2+2?", seed=42)
+ task_a = task_a.model_copy(update={"temperature": 0.0})
+ task_b = _make_task(
+ content="Write a short poem about the ocean and the sky.",
+ seed=99,
+ )
+ task_b = task_b.model_copy(update={"temperature": 0.0})
+
+ # ── Concurrent: submit both to one ExoBatchGenerator ──
+ exo_gen = ExoBatchGenerator(
+ model=model,
+ tokenizer=tokenizer,
+ group=None,
+ kv_prefix_cache=None,
+ )
+
+ prompt_a = apply_chat_template(tokenizer=tokenizer, task_params=task_a)
+ prompt_b = apply_chat_template(tokenizer=tokenizer, task_params=task_b)
+ uid_a = exo_gen.submit(task_params=task_a, prompt=prompt_a)
+ uid_b = exo_gen.submit(task_params=task_b, prompt=prompt_b)
+
+ batch_tokens: dict[int, list[int]] = {uid_a: [], uid_b: []}
+ finished: set[int] = set()
+ while exo_gen.has_work:
+ results = exo_gen.step()
+ for uid, response in results:
+ batch_tokens[uid].append(response.token)
+ if response.finish_reason is not None:
+ finished.add(uid)
+
+ exo_gen.close()
+
+ # ── Verify both completed ──
+ assert len(batch_tokens[uid_a]) > 0, "No tokens for task A"
+ assert len(batch_tokens[uid_b]) > 0, "No tokens for task B"
+ assert uid_a in finished, "Task A never finished"
+ assert uid_b in finished, "Task B never finished"
+ finally:
+ shutil.rmtree(tmpdir, ignore_errors=True)
diff --git a/src/exo/worker/tests/unittests/test_mlx/test_prefix_cache_architectures.py b/src/exo/worker/tests/unittests/test_mlx/test_prefix_cache_architectures.py
index 1a3c05fe..944e8290 100644
--- a/src/exo/worker/tests/unittests/test_mlx/test_prefix_cache_architectures.py
+++ b/src/exo/worker/tests/unittests/test_mlx/test_prefix_cache_architectures.py
@@ -183,6 +183,8 @@ ARCHITECTURES: list[ArchSpec] = [
ArchSpec("gpt_oss", "gpt-oss-20b-MXFP4-Q8", "gpt_oss"),
ArchSpec("step3p5", "Step-3.5-Flash-4bit", "step3p5"),
ArchSpec("kimi_k25", "Kimi-K2.5", "kimi_k25"),
+ ArchSpec("qwen3_5", "Qwen3.5-2B-MLX-8bit", "qwen3_5"),
+ ArchSpec("qwen3_5_moe", "Qwen3.5-35B-A3B-4bit", "qwen3_5_moe"),
]
diff --git a/src/exo/worker/tests/unittests/test_runner/test_event_ordering.py b/src/exo/worker/tests/unittests/test_runner/test_event_ordering.py
index 149bdc1a..8c651cde 100644
--- a/src/exo/worker/tests/unittests/test_runner/test_event_ordering.py
+++ b/src/exo/worker/tests/unittests/test_runner/test_event_ordering.py
@@ -134,11 +134,47 @@ def patch_out_mlx(monkeypatch: pytest.MonkeyPatch):
monkeypatch.setattr(
mlx_model_output_parsers, "detect_thinking_prompt_suffix", make_nothin(False)
)
+ monkeypatch.setattr(mlx_batch_generator, "ExoBatchGenerator", FakeExoBatchGenerator)
+
+
+class FakeExoBatchGenerator:
+ def __init__(self, *_args: object, **_kwargs: object) -> None:
+ self._uid_counter = 0
+ self._pending: dict[int, GenerationResponse] = {}
+
+ @property
+ def has_work(self) -> bool:
+ return bool(self._pending)
+
+ def submit(
+ self,
+ task_params: object = None,
+ prompt: object = None,
+ on_prefill_progress: object = None,
+ distributed_prompt_progress_callback: object = None,
+ on_generation_token: object = None,
+ ) -> int:
+ uid = self._uid_counter
+ self._uid_counter += 1
+ self._pending[uid] = GenerationResponse(
+ text="hi",
+ token=0,
+ finish_reason="stop",
+ usage=None,
+ )
+ return uid
+
+ def step(self) -> list[tuple[int, GenerationResponse]]:
+ results = list(self._pending.items())
+ self._pending.clear()
+ return results
- def fake_generate(*_1: object, **_2: object):
- yield GenerationResponse(token=0, text="hi", finish_reason="stop", usage=None)
+ def cancel(self, uids: list[int]) -> None:
+ for uid in uids:
+ self._pending.pop(uid, None)
- monkeypatch.setattr(mlx_batch_generator, "mlx_generate", fake_generate)
+ def close(self) -> None:
+ pass
# Use a fake event_sender to remove test flakiness.
@@ -165,6 +201,17 @@ class MockTokenizer:
tool_call_end = None
has_tool_calling = False
has_thinking = False
+ think_start = None
+ think_end = None
+ eos_token_ids: list[int] = []
+
+ @staticmethod
+ def decode(_tokens: list[int]) -> str:
+ return "hi"
+
+ @staticmethod
+ def encode(_text: str, add_special_tokens: bool = True) -> list[int]:
+ return [0]
class MockGroup:
diff --git a/uv.lock b/uv.lock
index 1f307ee7..0e79ba87 100644
--- a/uv.lock
+++ b/uv.lock
@@ -2,8 +2,10 @@ version = 1
revision = 3
requires-python = ">=3.13"
resolution-markers = [
- "sys_platform == 'darwin'",
- "sys_platform == 'linux'",
+ "python_full_version >= '3.14' and sys_platform == 'darwin'",
+ "python_full_version < '3.14' and sys_platform == 'darwin'",
+ "python_full_version >= '3.14' and sys_platform == 'linux'",
+ "python_full_version < '3.14' and sys_platform == 'linux'",
]
supported-markers = [
"sys_platform == 'darwin'",
@@ -20,6 +22,15 @@ members = [
"exo-pyo3-bindings",
]
+[[package]]
+name = "absl-py"
+version = "2.4.0"
+source = { registry = "https://pypi.org/simple" }
+sdist = { url = "https://files.pythonhosted.org/packages/64/c7/8de93764ad66968d19329a7e0c147a2bb3c7054c554d4a119111b8f9440f/absl_py-2.4.0.tar.gz", hash = "sha256:8c6af82722b35cf71e0f4d1d47dcaebfff286e27110a99fc359349b247dfb5d4", size = 116543, upload-time = "2026-01-28T10:17:05.322Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/18/a6/907a406bb7d359e6a63f99c313846d9eec4f7e6f7437809e03aa00fa3074/absl_py-2.4.0-py3-none-any.whl", hash = "sha256:88476fd881ca8aab94ffa78b7b6c632a782ab3ba1cd19c9bd423abc4fb4cd28d", size = 135750, upload-time = "2026-01-28T10:17:04.19Z" },
+]
+
[[package]]
name = "aiofiles"
version = "25.1.0"
@@ -139,6 +150,15 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/78/b6/6307fbef88d9b5ee7421e68d78a9f162e0da4900bc5f5793f6d3d0e34fb8/annotated_types-0.7.0-py3-none-any.whl", hash = "sha256:1f02e8b43a8fbbc3f3e0d4f0f4bfc8131bcb4eebe8849b8e5c773f3a1c582a53", size = 13643, upload-time = "2024-05-20T21:33:24.1Z" },
]
+[[package]]
+name = "antlr4-python3-runtime"
+version = "4.11.0"
+source = { registry = "https://pypi.org/simple" }
+sdist = { url = "https://files.pythonhosted.org/packages/a4/1b/59dc12d9816edee8be6a166be7c220910ff18cce2ded9877a08b27396a83/antlr4-python3-runtime-4.11.0.tar.gz", hash = "sha256:6a46d4f6033c8d7e53c8975cee5d83b3ba1b48157f6beb09a4965062dacfe2e0", size = 116951, upload-time = "2022-09-04T19:49:27.321Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/37/f3/2cab1ffe441ffcc4605bf638f524fed86db6699a25d04bbecb5bfdde372b/antlr4_python3_runtime-4.11.0-py3-none-any.whl", hash = "sha256:f523f91387283045f129c343b363a8dda3a07eea62721b8eda95f2d8b817b656", size = 144189, upload-time = "2022-09-04T19:49:25.198Z" },
+]
+
[[package]]
name = "anyio"
version = "4.11.0"
@@ -212,6 +232,15 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/e1/5e/b666bacbbc60fbf415ba9988324a132c9a7a0448a9a8f125074671c0f2c3/cffi-2.0.0-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:6c6c373cfc5c83a975506110d17457138c8c63016b563cc9ed6e056a82f13ce4", size = 223437, upload-time = "2025-09-08T23:23:38.945Z" },
]
+[[package]]
+name = "chardet"
+version = "5.2.0"
+source = { registry = "https://pypi.org/simple" }
+sdist = { url = "https://files.pythonhosted.org/packages/f3/0d/f7b6ab21ec75897ed80c17d79b15951a719226b9fababf1e40ea74d69079/chardet-5.2.0.tar.gz", hash = "sha256:1b3b6ff479a8c414bc3fa2c0852995695c4a026dcd6d0633b2dd092ca39c1cf7", size = 2069618, upload-time = "2023-08-01T19:23:02.662Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/38/6f/f5fbc992a329ee4e0f288c1fe0e2ad9485ed064cac731ed2fe47dcc38cbf/chardet-5.2.0-py3-none-any.whl", hash = "sha256:e1cf59446890a00105fe7b7912492ea04b6e6f06d4b742b2c788469e34c82970", size = 199385, upload-time = "2023-08-01T19:23:00.661Z" },
+]
+
[[package]]
name = "charset-normalizer"
version = "3.4.4"
@@ -256,6 +285,15 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/98/78/01c019cdb5d6498122777c1a43056ebb3ebfeef2076d9d026bfe15583b2b/click-8.3.1-py3-none-any.whl", hash = "sha256:981153a64e25f12d547d3426c367a4857371575ee7ad18df2a6183ab0545b2a6", size = 108274, upload-time = "2025-11-15T20:45:41.139Z" },
]
+[[package]]
+name = "colorama"
+version = "0.4.6"
+source = { registry = "https://pypi.org/simple" }
+sdist = { url = "https://files.pythonhosted.org/packages/d8/53/6f443c9a4a8358a93a6792e2acffb9d9d5cb0a5cfd8802644b7b1c9a02e4/colorama-0.4.6.tar.gz", hash = "sha256:08695f5cb7ed6e0531a20572697297273c47b8cae5a63ffc6d6ed5c201be6e44", size = 27697, upload-time = "2022-10-25T02:36:22.414Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/d1/d6/3965ed04c63042e047cb6a3e6ed1a63a35087b6a609aa3a15ed8ac56c221/colorama-0.4.6-py2.py3-none-any.whl", hash = "sha256:4f1d9991f5acc0ca119f9d443620b77f9d6b33703e51011c16baf57afb285fc6", size = 25335, upload-time = "2022-10-25T02:36:20.889Z" },
+]
+
[[package]]
name = "contourpy"
version = "1.3.3"
@@ -352,6 +390,53 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/e7/05/c19819d5e3d95294a6f5947fb9b9629efb316b96de511b418c53d245aae6/cycler-0.12.1-py3-none-any.whl", hash = "sha256:85cef7cff222d8644161529808465972e51340599459b8ac3ccbac5a854e0d30", size = 8321, upload-time = "2023-10-07T05:32:16.783Z" },
]
+[[package]]
+name = "dataproperty"
+version = "1.1.0"
+source = { registry = "https://pypi.org/simple" }
+dependencies = [
+ { name = "mbstrdecoder", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "typepy", extra = ["datetime"], marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+sdist = { url = "https://files.pythonhosted.org/packages/0b/81/8c8b64ae873cb9014815214c07b63b12e3b18835780fb342223cfe3fe7d8/dataproperty-1.1.0.tar.gz", hash = "sha256:b038437a4097d1a1c497695c3586ea34bea67fdd35372b9a50f30bf044d77d04", size = 42574, upload-time = "2024-12-31T14:37:26.033Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/21/c2/e12e95e289e6081a40454199ab213139ef16a528c7c86432de545b05a23a/DataProperty-1.1.0-py3-none-any.whl", hash = "sha256:c61fcb2e2deca35e6d1eb1f251a7f22f0dcde63e80e61f0cc18c19f42abfd25b", size = 27581, upload-time = "2024-12-31T14:37:22.657Z" },
+]
+
+[[package]]
+name = "datasets"
+version = "4.6.1"
+source = { registry = "https://pypi.org/simple" }
+dependencies = [
+ { name = "dill", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "filelock", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "fsspec", extra = ["http"], marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "httpx", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "huggingface-hub", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "multiprocess", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "numpy", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "packaging", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "pandas", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "pyarrow", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "pyyaml", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "requests", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "tqdm", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "xxhash", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+sdist = { url = "https://files.pythonhosted.org/packages/d7/94/eb81c6fe32e9b6ef92223141b5a553aeff2e9456968424a8533cbe88f476/datasets-4.6.1.tar.gz", hash = "sha256:140ce500bc41939ff6ce995702d66b1f4b2ee7f117bb9b07512fab6804d4070a", size = 593865, upload-time = "2026-02-27T23:26:49.482Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/37/f0/99fe6eb530c7ee9ee1faee48059eb8a6437f80c893a496b98a78864e0fc6/datasets-4.6.1-py3-none-any.whl", hash = "sha256:f53228e6dadc9f837037b1bf3051d7d8c054abbb3eb29f1f022926e08090e0da", size = 520667, upload-time = "2026-02-27T23:26:46.855Z" },
+]
+
+[[package]]
+name = "dill"
+version = "0.4.0"
+source = { registry = "https://pypi.org/simple" }
+sdist = { url = "https://files.pythonhosted.org/packages/12/80/630b4b88364e9a8c8c5797f4602d0f76ef820909ee32f0bacb9f90654042/dill-0.4.0.tar.gz", hash = "sha256:0633f1d2df477324f53a895b02c901fb961bdbf65a17122586ea7019292cbcf0", size = 186976, upload-time = "2025-04-16T00:41:48.867Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/50/3d/9373ad9c56321fdab5b41197068e1d8c25883b3fea29dd361f9b55116869/dill-0.4.0-py3-none-any.whl", hash = "sha256:44f54bf6412c2c8464c14e8243eb163690a9800dbe2c367330883b19c7561049", size = 119668, upload-time = "2025-04-16T00:41:47.671Z" },
+]
+
[[package]]
name = "docutils"
version = "0.22.4"
@@ -361,6 +446,28 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/02/10/5da547df7a391dcde17f59520a231527b8571e6f46fc8efb02ccb370ab12/docutils-0.22.4-py3-none-any.whl", hash = "sha256:d0013f540772d1420576855455d050a2180186c91c15779301ac2ccb3eeb68de", size = 633196, upload-time = "2025-12-18T19:00:18.077Z" },
]
+[[package]]
+name = "evaluate"
+version = "0.4.6"
+source = { registry = "https://pypi.org/simple" }
+dependencies = [
+ { name = "datasets", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "dill", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "fsspec", extra = ["http"], marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "huggingface-hub", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "multiprocess", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "numpy", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "packaging", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "pandas", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "requests", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "tqdm", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "xxhash", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+sdist = { url = "https://files.pythonhosted.org/packages/ad/d0/0c17a8e6e8dc7245f22dea860557c32bae50fc4d287ae030cb0e8ab8720f/evaluate-0.4.6.tar.gz", hash = "sha256:e07036ca12b3c24331f83ab787f21cc2dbf3631813a1631e63e40897c69a3f21", size = 65716, upload-time = "2025-09-18T13:06:30.581Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/3e/af/3e990d8d4002bbc9342adb4facd59506e653da93b2417de0fa6027cb86b1/evaluate-0.4.6-py3-none-any.whl", hash = "sha256:bca85bc294f338377b7ac2f861e21c308b11b2a285f510d7d5394d5df437db29", size = 84069, upload-time = "2025-09-18T13:06:29.265Z" },
+]
+
[[package]]
name = "exo"
version = "0.3.68"
@@ -418,7 +525,7 @@ requires-dist = [
{ name = "mflux", specifier = "==0.15.5" },
{ name = "mlx", marker = "sys_platform == 'darwin'", git = "https://github.com/rltakashige/mlx-jaccl-fix-small-recv.git?branch=address-rdma-gpu-locks" },
{ name = "mlx", extras = ["cpu"], marker = "sys_platform == 'linux'", specifier = "==0.30.6" },
- { name = "mlx-lm", git = "https://github.com/ml-explore/mlx-lm?rev=834fac934c4e04de9b3d723e2b9287a2c60cfd4a" },
+ { name = "mlx-lm", git = "https://github.com/rltakashige/mlx-lm?branch=leo%2Feval-left-padding-in-batched-rotation" },
{ name = "msgspec", specifier = ">=0.19.0" },
{ name = "openai-harmony", specifier = ">=0.0.8" },
{ name = "pillow", specifier = ">=11.0,<12.0" },
@@ -447,10 +554,15 @@ name = "exo-bench"
version = "0.1.0"
source = { editable = "bench" }
dependencies = [
+ { name = "datasets", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
{ name = "httpx", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
{ name = "huggingface-hub", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "human-eval", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
{ name = "jinja2", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "lm-eval", extra = ["api", "math"], marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
{ name = "loguru", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "math-verify", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "numpy", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
{ name = "protobuf", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
{ name = "tiktoken", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
{ name = "transformers", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
@@ -458,10 +570,15 @@ dependencies = [
[package.metadata]
requires-dist = [
+ { name = "datasets", specifier = ">=2.0.0" },
{ name = "httpx", specifier = ">=0.27.0" },
{ name = "huggingface-hub", specifier = ">=0.33.4" },
+ { name = "human-eval", specifier = ">=1.0.3" },
{ name = "jinja2", specifier = ">=3.1.0" },
+ { name = "lm-eval", extras = ["api", "math"], specifier = ">=0.4.0" },
{ name = "loguru", specifier = ">=0.7.3" },
+ { name = "math-verify", specifier = ">=0.7.0" },
+ { name = "numpy", specifier = ">=1.24.0" },
{ name = "protobuf", specifier = ">=5.29.0" },
{ name = "tiktoken", specifier = ">=0.12.0" },
{ name = "transformers", specifier = ">=5.0.0" },
@@ -469,7 +586,7 @@ requires-dist = [
[[package]]
name = "exo-pyo3-bindings"
-version = "0.2.0"
+version = "0.2.1"
source = { editable = "rust/exo_pyo3_bindings" }
[package.dev-dependencies]
@@ -512,6 +629,18 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/b5/36/7fb70f04bf00bc646cd5bb45aa9eddb15e19437a28b8fb2b4a5249fac770/filelock-3.20.3-py3-none-any.whl", hash = "sha256:4b0dda527ee31078689fc205ec4f1c1bf7d56cf88b6dc9426c4f230e46c2dce1", size = 16701, upload-time = "2026-01-09T17:55:04.334Z" },
]
+[[package]]
+name = "fire"
+version = "0.7.1"
+source = { registry = "https://pypi.org/simple" }
+dependencies = [
+ { name = "termcolor", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+sdist = { url = "https://files.pythonhosted.org/packages/c0/00/f8d10588d2019d6d6452653def1ee807353b21983db48550318424b5ff18/fire-0.7.1.tar.gz", hash = "sha256:3b208f05c736de98fb343310d090dcc4d8c78b2a89ea4f32b837c586270a9cbf", size = 88720, upload-time = "2025-08-16T20:20:24.175Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/e5/4c/93d0f85318da65923e4b91c1c2ff03d8a458cbefebe3bc612a6693c7906d/fire-0.7.1-py3-none-any.whl", hash = "sha256:e43fd8a5033a9001e7e2973bab96070694b9f12f2e0ecf96d4683971b5ab1882", size = 115945, upload-time = "2025-08-16T20:20:22.87Z" },
+]
+
[[package]]
name = "fonttools"
version = "4.61.1"
@@ -609,6 +738,11 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/01/c9/97cc5aae1648dcb851958a3ddf73ccd7dbe5650d95203ecb4d7720b4cdbf/fsspec-2026.1.0-py3-none-any.whl", hash = "sha256:cb76aa913c2285a3b49bdd5fc55b1d7c708d7208126b60f2eb8194fe1b4cbdcc", size = 201838, upload-time = "2026-01-09T15:21:34.041Z" },
]
+[package.optional-dependencies]
+http = [
+ { name = "aiohttp", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+
[[package]]
name = "h11"
version = "0.16.0"
@@ -718,6 +852,20 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/90/fb/cb8fe5f71d5622427f20bcab9e06a696a5aaf21bfe7bd0a8a0c63c88abf5/huggingface_hub-1.3.1-py3-none-any.whl", hash = "sha256:efbc7f3153cb84e2bb69b62ed90985e21ecc9343d15647a419fc0ee4b85f0ac3", size = 533351, upload-time = "2026-01-09T14:08:14.519Z" },
]
+[[package]]
+name = "human-eval"
+version = "1.0.3"
+source = { registry = "https://pypi.org/simple" }
+dependencies = [
+ { name = "fire", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "numpy", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "tqdm", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+sdist = { url = "https://files.pythonhosted.org/packages/0a/4b/44832f60f8fb14257019084412229a2ed48e34acd0d9d3f8198196d02dd7/human-eval-1.0.3.tar.gz", hash = "sha256:03ee63d3fbbb6fecbc5dfae58381fc655004ad86caafeb141ca2169c63648766", size = 54651, upload-time = "2023-07-24T18:45:58.99Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/c0/af/85d39b927626279710c27c37cc6ea16257b2035a314ad6a3277c6dc15e82/human_eval-1.0.3-py3-none-any.whl", hash = "sha256:b4e2844c8655a2db4780f6092834cb6ab15c130c56ba0516b15028ccc413dbce", size = 52282, upload-time = "2023-07-24T18:45:57.386Z" },
+]
+
[[package]]
name = "hypercorn"
version = "0.18.0"
@@ -826,6 +974,27 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/62/a1/3d680cbfd5f4b8f15abc1d571870c5fc3e594bb582bc3b64ea099db13e56/jinja2-3.1.6-py3-none-any.whl", hash = "sha256:85ece4451f492d0c13c5dd7c13a64681a86afae63a5f347908daf103ce6d2f67", size = 134899, upload-time = "2025-03-05T20:05:00.369Z" },
]
+[[package]]
+name = "joblib"
+version = "1.5.3"
+source = { registry = "https://pypi.org/simple" }
+sdist = { url = "https://files.pythonhosted.org/packages/41/f2/d34e8b3a08a9cc79a50b2208a93dce981fe615b64d5a4d4abee421d898df/joblib-1.5.3.tar.gz", hash = "sha256:8561a3269e6801106863fd0d6d84bb737be9e7631e33aaed3fb9ce5953688da3", size = 331603, upload-time = "2025-12-15T08:41:46.427Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/7b/91/984aca2ec129e2757d1e4e3c81c3fcda9d0f85b74670a094cc443d9ee949/joblib-1.5.3-py3-none-any.whl", hash = "sha256:5fc3c5039fc5ca8c0276333a188bbd59d6b7ab37fe6632daa76bc7f9ec18e713", size = 309071, upload-time = "2025-12-15T08:41:44.973Z" },
+]
+
+[[package]]
+name = "jsonlines"
+version = "4.0.0"
+source = { registry = "https://pypi.org/simple" }
+dependencies = [
+ { name = "attrs", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+sdist = { url = "https://files.pythonhosted.org/packages/35/87/bcda8e46c88d0e34cad2f09ee2d0c7f5957bccdb9791b0b934ec84d84be4/jsonlines-4.0.0.tar.gz", hash = "sha256:0c6d2c09117550c089995247f605ae4cf77dd1533041d366351f6f298822ea74", size = 11359, upload-time = "2023-09-01T12:34:44.187Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/f8/62/d9ba6323b9202dd2fe166beab8a86d29465c41a0288cbe229fac60c1ab8d/jsonlines-4.0.0-py3-none-any.whl", hash = "sha256:185b334ff2ca5a91362993f42e83588a360cf95ce4b71a73548502bda52a7c55", size = 8701, upload-time = "2023-09-01T12:34:42.563Z" },
+]
+
[[package]]
name = "keyring"
version = "25.7.0"
@@ -894,6 +1063,64 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/64/ad/53bd6b22fa1917746096b6240dd0c546020e358506e8503dce57f3cdcd9a/kiwisolver-1.4.10rc0-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:acc08f93b36220a6baa7df3428cb8847b27717db9be4295c0b1571d040c77327", size = 2391902, upload-time = "2025-08-10T20:22:12.421Z" },
]
+[[package]]
+name = "latex2sympy2-extended"
+version = "1.11.0"
+source = { registry = "https://pypi.org/simple" }
+dependencies = [
+ { name = "antlr4-python3-runtime", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "sympy", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+sdist = { url = "https://files.pythonhosted.org/packages/30/75/456da2da05f6380ea96e6ea804ab2c03e41fc3ed80052307fe8efe6ea20e/latex2sympy2_extended-1.11.0.tar.gz", hash = "sha256:9695657c81b50abba2636638638618db59f4663ed2a4a12d62cef74a40e28fec", size = 207023, upload-time = "2026-01-10T01:43:21.319Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/e9/61/f75cd1fa54d8434276126034aed54dd120747de9a8fa013cdd79545ccbeb/latex2sympy2_extended-1.11.0-py3-none-any.whl", hash = "sha256:aebb77d52ce269e25028e4bea89ddb14d242ba36bcf7b636496fb5fd9728d234", size = 209050, upload-time = "2026-01-10T01:43:19.458Z" },
+]
+
+[package.optional-dependencies]
+antlr4-11-0 = [
+ { name = "antlr4-python3-runtime", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+
+[[package]]
+name = "lm-eval"
+version = "0.4.11"
+source = { registry = "https://pypi.org/simple" }
+dependencies = [
+ { name = "datasets", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "dill", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "evaluate", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "jinja2", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "jsonlines", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "more-itertools", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "numpy", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "pytablewriter", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "rouge-score", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "sacrebleu", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "scikit-learn", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "sqlitedict", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "typing-extensions", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "word2number", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "zstandard", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+sdist = { url = "https://files.pythonhosted.org/packages/18/40/ad60ae17f97902ba4ee67914a94c159552f8643fb4598d1d2430a571f0a2/lm_eval-0.4.11.tar.gz", hash = "sha256:a3891d6d0b4ad17892f2ca1046094554ce3ab6cb522759e4a453858e45649916", size = 3246509, upload-time = "2026-02-13T20:22:59.494Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/df/b3/8c2923ad4c4d911307e20ab9c7d7869d3811b5a76fe6dbe53678aafbeb04/lm_eval-0.4.11-py3-none-any.whl", hash = "sha256:9c9945475a715558649a38ffcb90f46e7bd23a849524a5e838f249161b030517", size = 8744826, upload-time = "2026-02-13T20:22:56.317Z" },
+]
+
+[package.optional-dependencies]
+api = [
+ { name = "aiohttp", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "requests", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "tenacity", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "tiktoken", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "tqdm", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+math = [
+ { name = "antlr4-python3-runtime", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "math-verify", extra = ["antlr4-11-0"], marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "sympy", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+
[[package]]
name = "loguru"
version = "0.7.3"
@@ -903,6 +1130,59 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/0c/29/0348de65b8cc732daa3e33e67806420b2ae89bdce2b04af740289c5c6c8c/loguru-0.7.3-py3-none-any.whl", hash = "sha256:31a33c10c8e1e10422bfd431aeb5d351c7cf7fa671e3c4df004162264b28220c", size = 61595, upload-time = "2024-12-06T11:20:54.538Z" },
]
+[[package]]
+name = "lxml"
+version = "6.0.2"
+source = { registry = "https://pypi.org/simple" }
+sdist = { url = "https://files.pythonhosted.org/packages/aa/88/262177de60548e5a2bfc46ad28232c9e9cbde697bd94132aeb80364675cb/lxml-6.0.2.tar.gz", hash = "sha256:cd79f3367bd74b317dda655dc8fcfa304d9eb6e4fb06b7168c5cf27f96e0cd62", size = 4073426, upload-time = "2025-09-22T04:04:59.287Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/53/fd/4e8f0540608977aea078bf6d79f128e0e2c2bba8af1acf775c30baa70460/lxml-6.0.2-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:9b33d21594afab46f37ae58dfadd06636f154923c4e8a4d754b0127554eb2e77", size = 8648494, upload-time = "2025-09-22T04:01:54.242Z" },
+ { url = "https://files.pythonhosted.org/packages/5d/f4/2a94a3d3dfd6c6b433501b8d470a1960a20ecce93245cf2db1706adf6c19/lxml-6.0.2-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:6c8963287d7a4c5c9a432ff487c52e9c5618667179c18a204bdedb27310f022f", size = 4661146, upload-time = "2025-09-22T04:01:56.282Z" },
+ { url = "https://files.pythonhosted.org/packages/25/2e/4efa677fa6b322013035d38016f6ae859d06cac67437ca7dc708a6af7028/lxml-6.0.2-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:1941354d92699fb5ffe6ed7b32f9649e43c2feb4b97205f75866f7d21aa91452", size = 4946932, upload-time = "2025-09-22T04:01:58.989Z" },
+ { url = "https://files.pythonhosted.org/packages/ce/0f/526e78a6d38d109fdbaa5049c62e1d32fdd70c75fb61c4eadf3045d3d124/lxml-6.0.2-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:bb2f6ca0ae2d983ded09357b84af659c954722bbf04dea98030064996d156048", size = 5100060, upload-time = "2025-09-22T04:02:00.812Z" },
+ { url = "https://files.pythonhosted.org/packages/81/76/99de58d81fa702cc0ea7edae4f4640416c2062813a00ff24bd70ac1d9c9b/lxml-6.0.2-cp313-cp313-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:eb2a12d704f180a902d7fa778c6d71f36ceb7b0d317f34cdc76a5d05aa1dd1df", size = 5019000, upload-time = "2025-09-22T04:02:02.671Z" },
+ { url = "https://files.pythonhosted.org/packages/b5/35/9e57d25482bc9a9882cb0037fdb9cc18f4b79d85df94fa9d2a89562f1d25/lxml-6.0.2-cp313-cp313-manylinux_2_26_i686.manylinux_2_28_i686.whl", hash = "sha256:6ec0e3f745021bfed19c456647f0298d60a24c9ff86d9d051f52b509663feeb1", size = 5348496, upload-time = "2025-09-22T04:02:04.904Z" },
+ { url = "https://files.pythonhosted.org/packages/a6/8e/cb99bd0b83ccc3e8f0f528e9aa1f7a9965dfec08c617070c5db8d63a87ce/lxml-6.0.2-cp313-cp313-manylinux_2_26_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:846ae9a12d54e368933b9759052d6206a9e8b250291109c48e350c1f1f49d916", size = 5643779, upload-time = "2025-09-22T04:02:06.689Z" },
+ { url = "https://files.pythonhosted.org/packages/d0/34/9e591954939276bb679b73773836c6684c22e56d05980e31d52a9a8deb18/lxml-6.0.2-cp313-cp313-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:ef9266d2aa545d7374938fb5c484531ef5a2ec7f2d573e62f8ce722c735685fd", size = 5244072, upload-time = "2025-09-22T04:02:08.587Z" },
+ { url = "https://files.pythonhosted.org/packages/8d/27/b29ff065f9aaca443ee377aff699714fcbffb371b4fce5ac4ca759e436d5/lxml-6.0.2-cp313-cp313-manylinux_2_31_armv7l.whl", hash = "sha256:4077b7c79f31755df33b795dc12119cb557a0106bfdab0d2c2d97bd3cf3dffa6", size = 4718675, upload-time = "2025-09-22T04:02:10.783Z" },
+ { url = "https://files.pythonhosted.org/packages/2b/9f/f756f9c2cd27caa1a6ef8c32ae47aadea697f5c2c6d07b0dae133c244fbe/lxml-6.0.2-cp313-cp313-manylinux_2_38_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:a7c5d5e5f1081955358533be077166ee97ed2571d6a66bdba6ec2f609a715d1a", size = 5255171, upload-time = "2025-09-22T04:02:12.631Z" },
+ { url = "https://files.pythonhosted.org/packages/61/46/bb85ea42d2cb1bd8395484fd72f38e3389611aa496ac7772da9205bbda0e/lxml-6.0.2-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:8f8d0cbd0674ee89863a523e6994ac25fd5be9c8486acfc3e5ccea679bad2679", size = 5057175, upload-time = "2025-09-22T04:02:14.718Z" },
+ { url = "https://files.pythonhosted.org/packages/95/0c/443fc476dcc8e41577f0af70458c50fe299a97bb6b7505bb1ae09aa7f9ac/lxml-6.0.2-cp313-cp313-musllinux_1_2_armv7l.whl", hash = "sha256:2cbcbf6d6e924c28f04a43f3b6f6e272312a090f269eff68a2982e13e5d57659", size = 4785688, upload-time = "2025-09-22T04:02:16.957Z" },
+ { url = "https://files.pythonhosted.org/packages/48/78/6ef0b359d45bb9697bc5a626e1992fa5d27aa3f8004b137b2314793b50a0/lxml-6.0.2-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:dfb874cfa53340009af6bdd7e54ebc0d21012a60a4e65d927c2e477112e63484", size = 5660655, upload-time = "2025-09-22T04:02:18.815Z" },
+ { url = "https://files.pythonhosted.org/packages/ff/ea/e1d33808f386bc1339d08c0dcada6e4712d4ed8e93fcad5f057070b7988a/lxml-6.0.2-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:fb8dae0b6b8b7f9e96c26fdd8121522ce5de9bb5538010870bd538683d30e9a2", size = 5247695, upload-time = "2025-09-22T04:02:20.593Z" },
+ { url = "https://files.pythonhosted.org/packages/4f/47/eba75dfd8183673725255247a603b4ad606f4ae657b60c6c145b381697da/lxml-6.0.2-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:358d9adae670b63e95bc59747c72f4dc97c9ec58881d4627fe0120da0f90d314", size = 5269841, upload-time = "2025-09-22T04:02:22.489Z" },
+ { url = "https://files.pythonhosted.org/packages/03/15/d4a377b385ab693ce97b472fe0c77c2b16ec79590e688b3ccc71fba19884/lxml-6.0.2-cp314-cp314-macosx_10_13_universal2.whl", hash = "sha256:b0c732aa23de8f8aec23f4b580d1e52905ef468afb4abeafd3fec77042abb6fe", size = 8659801, upload-time = "2025-09-22T04:02:30.113Z" },
+ { url = "https://files.pythonhosted.org/packages/c8/e8/c128e37589463668794d503afaeb003987373c5f94d667124ffd8078bbd9/lxml-6.0.2-cp314-cp314-macosx_10_13_x86_64.whl", hash = "sha256:4468e3b83e10e0317a89a33d28f7aeba1caa4d1a6fd457d115dd4ffe90c5931d", size = 4659403, upload-time = "2025-09-22T04:02:32.119Z" },
+ { url = "https://files.pythonhosted.org/packages/00/ce/74903904339decdf7da7847bb5741fc98a5451b42fc419a86c0c13d26fe2/lxml-6.0.2-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:abd44571493973bad4598a3be7e1d807ed45aa2adaf7ab92ab7c62609569b17d", size = 4966974, upload-time = "2025-09-22T04:02:34.155Z" },
+ { url = "https://files.pythonhosted.org/packages/1f/d3/131dec79ce61c5567fecf82515bd9bc36395df42501b50f7f7f3bd065df0/lxml-6.0.2-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:370cd78d5855cfbffd57c422851f7d3864e6ae72d0da615fca4dad8c45d375a5", size = 5102953, upload-time = "2025-09-22T04:02:36.054Z" },
+ { url = "https://files.pythonhosted.org/packages/3a/ea/a43ba9bb750d4ffdd885f2cd333572f5bb900cd2408b67fdda07e85978a0/lxml-6.0.2-cp314-cp314-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:901e3b4219fa04ef766885fb40fa516a71662a4c61b80c94d25336b4934b71c0", size = 5055054, upload-time = "2025-09-22T04:02:38.154Z" },
+ { url = "https://files.pythonhosted.org/packages/60/23/6885b451636ae286c34628f70a7ed1fcc759f8d9ad382d132e1c8d3d9bfd/lxml-6.0.2-cp314-cp314-manylinux_2_26_i686.manylinux_2_28_i686.whl", hash = "sha256:a4bf42d2e4cf52c28cc1812d62426b9503cdb0c87a6de81442626aa7d69707ba", size = 5352421, upload-time = "2025-09-22T04:02:40.413Z" },
+ { url = "https://files.pythonhosted.org/packages/48/5b/fc2ddfc94ddbe3eebb8e9af6e3fd65e2feba4967f6a4e9683875c394c2d8/lxml-6.0.2-cp314-cp314-manylinux_2_26_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:b2c7fdaa4d7c3d886a42534adec7cfac73860b89b4e5298752f60aa5984641a0", size = 5673684, upload-time = "2025-09-22T04:02:42.288Z" },
+ { url = "https://files.pythonhosted.org/packages/29/9c/47293c58cc91769130fbf85531280e8cc7868f7fbb6d92f4670071b9cb3e/lxml-6.0.2-cp314-cp314-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:98a5e1660dc7de2200b00d53fa00bcd3c35a3608c305d45a7bbcaf29fa16e83d", size = 5252463, upload-time = "2025-09-22T04:02:44.165Z" },
+ { url = "https://files.pythonhosted.org/packages/9b/da/ba6eceb830c762b48e711ded880d7e3e89fc6c7323e587c36540b6b23c6b/lxml-6.0.2-cp314-cp314-manylinux_2_31_armv7l.whl", hash = "sha256:dc051506c30b609238d79eda75ee9cab3e520570ec8219844a72a46020901e37", size = 4698437, upload-time = "2025-09-22T04:02:46.524Z" },
+ { url = "https://files.pythonhosted.org/packages/a5/24/7be3f82cb7990b89118d944b619e53c656c97dc89c28cfb143fdb7cd6f4d/lxml-6.0.2-cp314-cp314-manylinux_2_38_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:8799481bbdd212470d17513a54d568f44416db01250f49449647b5ab5b5dccb9", size = 5269890, upload-time = "2025-09-22T04:02:48.812Z" },
+ { url = "https://files.pythonhosted.org/packages/1b/bd/dcfb9ea1e16c665efd7538fc5d5c34071276ce9220e234217682e7d2c4a5/lxml-6.0.2-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:9261bb77c2dab42f3ecd9103951aeca2c40277701eb7e912c545c1b16e0e4917", size = 5097185, upload-time = "2025-09-22T04:02:50.746Z" },
+ { url = "https://files.pythonhosted.org/packages/21/04/a60b0ff9314736316f28316b694bccbbabe100f8483ad83852d77fc7468e/lxml-6.0.2-cp314-cp314-musllinux_1_2_armv7l.whl", hash = "sha256:65ac4a01aba353cfa6d5725b95d7aed6356ddc0a3cd734de00124d285b04b64f", size = 4745895, upload-time = "2025-09-22T04:02:52.968Z" },
+ { url = "https://files.pythonhosted.org/packages/d6/bd/7d54bd1846e5a310d9c715921c5faa71cf5c0853372adf78aee70c8d7aa2/lxml-6.0.2-cp314-cp314-musllinux_1_2_ppc64le.whl", hash = "sha256:b22a07cbb82fea98f8a2fd814f3d1811ff9ed76d0fc6abc84eb21527596e7cc8", size = 5695246, upload-time = "2025-09-22T04:02:54.798Z" },
+ { url = "https://files.pythonhosted.org/packages/fd/32/5643d6ab947bc371da21323acb2a6e603cedbe71cb4c99c8254289ab6f4e/lxml-6.0.2-cp314-cp314-musllinux_1_2_riscv64.whl", hash = "sha256:d759cdd7f3e055d6bc8d9bec3ad905227b2e4c785dc16c372eb5b5e83123f48a", size = 5260797, upload-time = "2025-09-22T04:02:57.058Z" },
+ { url = "https://files.pythonhosted.org/packages/33/da/34c1ec4cff1eea7d0b4cd44af8411806ed943141804ac9c5d565302afb78/lxml-6.0.2-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:945da35a48d193d27c188037a05fec5492937f66fb1958c24fc761fb9d40d43c", size = 5277404, upload-time = "2025-09-22T04:02:58.966Z" },
+ { url = "https://files.pythonhosted.org/packages/5e/5c/42c2c4c03554580708fc738d13414801f340c04c3eff90d8d2d227145275/lxml-6.0.2-cp314-cp314t-macosx_10_13_universal2.whl", hash = "sha256:6162a86d86893d63084faaf4ff937b3daea233e3682fb4474db07395794fa80d", size = 8910380, upload-time = "2025-09-22T04:03:01.645Z" },
+ { url = "https://files.pythonhosted.org/packages/bf/4f/12df843e3e10d18d468a7557058f8d3733e8b6e12401f30b1ef29360740f/lxml-6.0.2-cp314-cp314t-macosx_10_13_x86_64.whl", hash = "sha256:414aaa94e974e23a3e92e7ca5b97d10c0cf37b6481f50911032c69eeb3991bba", size = 4775632, upload-time = "2025-09-22T04:03:03.814Z" },
+ { url = "https://files.pythonhosted.org/packages/e4/0c/9dc31e6c2d0d418483cbcb469d1f5a582a1cd00a1f4081953d44051f3c50/lxml-6.0.2-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:48461bd21625458dd01e14e2c38dd0aea69addc3c4f960c30d9f59d7f93be601", size = 4975171, upload-time = "2025-09-22T04:03:05.651Z" },
+ { url = "https://files.pythonhosted.org/packages/e7/2b/9b870c6ca24c841bdd887504808f0417aa9d8d564114689266f19ddf29c8/lxml-6.0.2-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:25fcc59afc57d527cfc78a58f40ab4c9b8fd096a9a3f964d2781ffb6eb33f4ed", size = 5110109, upload-time = "2025-09-22T04:03:07.452Z" },
+ { url = "https://files.pythonhosted.org/packages/bf/0c/4f5f2a4dd319a178912751564471355d9019e220c20d7db3fb8307ed8582/lxml-6.0.2-cp314-cp314t-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:5179c60288204e6ddde3f774a93350177e08876eaf3ab78aa3a3649d43eb7d37", size = 5041061, upload-time = "2025-09-22T04:03:09.297Z" },
+ { url = "https://files.pythonhosted.org/packages/12/64/554eed290365267671fe001a20d72d14f468ae4e6acef1e179b039436967/lxml-6.0.2-cp314-cp314t-manylinux_2_26_i686.manylinux_2_28_i686.whl", hash = "sha256:967aab75434de148ec80597b75062d8123cadf2943fb4281f385141e18b21338", size = 5306233, upload-time = "2025-09-22T04:03:11.651Z" },
+ { url = "https://files.pythonhosted.org/packages/7a/31/1d748aa275e71802ad9722df32a7a35034246b42c0ecdd8235412c3396ef/lxml-6.0.2-cp314-cp314t-manylinux_2_26_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:d100fcc8930d697c6561156c6810ab4a508fb264c8b6779e6e61e2ed5e7558f9", size = 5604739, upload-time = "2025-09-22T04:03:13.592Z" },
+ { url = "https://files.pythonhosted.org/packages/8f/41/2c11916bcac09ed561adccacceaedd2bf0e0b25b297ea92aab99fd03d0fa/lxml-6.0.2-cp314-cp314t-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:2ca59e7e13e5981175b8b3e4ab84d7da57993eeff53c07764dcebda0d0e64ecd", size = 5225119, upload-time = "2025-09-22T04:03:15.408Z" },
+ { url = "https://files.pythonhosted.org/packages/99/05/4e5c2873d8f17aa018e6afde417c80cc5d0c33be4854cce3ef5670c49367/lxml-6.0.2-cp314-cp314t-manylinux_2_31_armv7l.whl", hash = "sha256:957448ac63a42e2e49531b9d6c0fa449a1970dbc32467aaad46f11545be9af1d", size = 4633665, upload-time = "2025-09-22T04:03:17.262Z" },
+ { url = "https://files.pythonhosted.org/packages/0f/c9/dcc2da1bebd6275cdc723b515f93edf548b82f36a5458cca3578bc899332/lxml-6.0.2-cp314-cp314t-manylinux_2_38_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:b7fc49c37f1786284b12af63152fe1d0990722497e2d5817acfe7a877522f9a9", size = 5234997, upload-time = "2025-09-22T04:03:19.14Z" },
+ { url = "https://files.pythonhosted.org/packages/9c/e2/5172e4e7468afca64a37b81dba152fc5d90e30f9c83c7c3213d6a02a5ce4/lxml-6.0.2-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:e19e0643cc936a22e837f79d01a550678da8377d7d801a14487c10c34ee49c7e", size = 5090957, upload-time = "2025-09-22T04:03:21.436Z" },
+ { url = "https://files.pythonhosted.org/packages/a5/b3/15461fd3e5cd4ddcb7938b87fc20b14ab113b92312fc97afe65cd7c85de1/lxml-6.0.2-cp314-cp314t-musllinux_1_2_armv7l.whl", hash = "sha256:1db01e5cf14345628e0cbe71067204db658e2fb8e51e7f33631f5f4735fefd8d", size = 4764372, upload-time = "2025-09-22T04:03:23.27Z" },
+ { url = "https://files.pythonhosted.org/packages/05/33/f310b987c8bf9e61c4dd8e8035c416bd3230098f5e3cfa69fc4232de7059/lxml-6.0.2-cp314-cp314t-musllinux_1_2_ppc64le.whl", hash = "sha256:875c6b5ab39ad5291588aed6925fac99d0097af0dd62f33c7b43736043d4a2ec", size = 5634653, upload-time = "2025-09-22T04:03:25.767Z" },
+ { url = "https://files.pythonhosted.org/packages/70/ff/51c80e75e0bc9382158133bdcf4e339b5886c6ee2418b5199b3f1a61ed6d/lxml-6.0.2-cp314-cp314t-musllinux_1_2_riscv64.whl", hash = "sha256:cdcbed9ad19da81c480dfd6dd161886db6096083c9938ead313d94b30aadf272", size = 5233795, upload-time = "2025-09-22T04:03:27.62Z" },
+ { url = "https://files.pythonhosted.org/packages/56/4d/4856e897df0d588789dd844dbed9d91782c4ef0b327f96ce53c807e13128/lxml-6.0.2-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:80dadc234ebc532e09be1975ff538d154a7fa61ea5031c03d25178855544728f", size = 5257023, upload-time = "2025-09-22T04:03:30.056Z" },
+]
+
[[package]]
name = "macholib"
version = "1.16.4"
@@ -967,6 +1247,23 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/14/c7/ca723101509b518797fedc2fdf79ba57f886b4aca8a7d31857ba3ee8281f/markupsafe-3.0.3-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:5678211cb9333a6468fb8d8be0305520aa073f50d17f089b5b4b477ea6e67fdc", size = 23672, upload-time = "2025-09-27T18:37:25.271Z" },
]
+[[package]]
+name = "math-verify"
+version = "0.9.0"
+source = { registry = "https://pypi.org/simple" }
+dependencies = [
+ { name = "latex2sympy2-extended", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+sdist = { url = "https://files.pythonhosted.org/packages/4f/12/b8d13b581e110ac2f724a2351a8361a70fa36d057eb945d6379e8747c256/math_verify-0.9.0.tar.gz", hash = "sha256:45ac6c61344ba056b9e99a660a4bc8d044ed408f730aed68c60435aa5eec4645", size = 60329, upload-time = "2026-01-10T01:48:33.056Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/62/76/6b4969bccc842b6567f7e6ee015684b9428a9b7fcbdf479e73716f43597f/math_verify-0.9.0-py3-none-any.whl", hash = "sha256:3703e7c4885354027fa84409d762a596a2906d1fd4deb78361876bd905a76194", size = 29967, upload-time = "2026-01-10T01:48:31.674Z" },
+]
+
+[package.optional-dependencies]
+antlr4-11-0 = [
+ { name = "latex2sympy2-extended", extra = ["antlr4-11-0"], marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+
[[package]]
name = "matplotlib"
version = "3.10.8"
@@ -1006,6 +1303,18 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/4d/4b/e7beb6bbd49f6bae727a12b270a2654d13c397576d25bd6786e47033300f/matplotlib-3.10.8-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:595ba4d8fe983b88f0eec8c26a241e16d6376fe1979086232f481f8f3f67494c", size = 9614011, upload-time = "2025-12-10T22:56:33.85Z" },
]
+[[package]]
+name = "mbstrdecoder"
+version = "1.1.4"
+source = { registry = "https://pypi.org/simple" }
+dependencies = [
+ { name = "chardet", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+sdist = { url = "https://files.pythonhosted.org/packages/31/ab/05ae008357c8bdb6245ebf8a101d99f26c096e0ea20800b318153da23796/mbstrdecoder-1.1.4.tar.gz", hash = "sha256:8105ef9cf6b7d7d69fe7fd6b68a2d8f281ca9b365d7a9b670be376b2e6c81b21", size = 14527, upload-time = "2025-01-18T10:07:31.089Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/30/ac/5ce64a1d4cce00390beab88622a290420401f1cabf05caf2fc0995157c21/mbstrdecoder-1.1.4-py3-none-any.whl", hash = "sha256:03dae4ec50ec0d2ff4743e63fdbd5e0022815857494d35224b60775d3d934a8c", size = 7933, upload-time = "2025-01-18T10:07:29.562Z" },
+]
+
[[package]]
name = "mdurl"
version = "0.1.2"
@@ -1053,7 +1362,8 @@ name = "mlx"
version = "0.30.6"
source = { registry = "https://pypi.org/simple" }
resolution-markers = [
- "sys_platform == 'linux'",
+ "python_full_version >= '3.14' and sys_platform == 'linux'",
+ "python_full_version < '3.14' and sys_platform == 'linux'",
]
wheels = [
{ url = "https://files.pythonhosted.org/packages/93/06/280f6f2ba80520a7109730425eda0d966658793aa0d02d8be8d351f75253/mlx-0.30.6-cp313-cp313-manylinux_2_35_aarch64.whl", hash = "sha256:67e6c9e30a9faeacc209917ef5523177cf9b086914b6b5d83ff886e4294b727d", size = 622011, upload-time = "2026-02-06T03:45:28.165Z" },
@@ -1075,7 +1385,8 @@ name = "mlx"
version = "0.30.7.dev20260225+257d5692"
source = { git = "https://github.com/rltakashige/mlx-jaccl-fix-small-recv.git?branch=address-rdma-gpu-locks#257d5692fc7af6bba3b8afaeb63c549b7d1e43d5" }
resolution-markers = [
- "sys_platform == 'darwin'",
+ "python_full_version >= '3.14' and sys_platform == 'darwin'",
+ "python_full_version < '3.14' and sys_platform == 'darwin'",
]
[[package]]
@@ -1104,8 +1415,8 @@ wheels = [
[[package]]
name = "mlx-lm"
-version = "0.30.8"
-source = { git = "https://github.com/ml-explore/mlx-lm?rev=834fac934c4e04de9b3d723e2b9287a2c60cfd4a#834fac934c4e04de9b3d723e2b9287a2c60cfd4a" }
+version = "0.31.0"
+source = { git = "https://github.com/rltakashige/mlx-lm?branch=leo%2Feval-left-padding-in-batched-rotation#5e0c484cfb5c68d281a71409927ee1bb75adaae2" }
dependencies = [
{ name = "jinja2", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
{ name = "mlx", version = "0.30.7.dev20260225+257d5692", source = { git = "https://github.com/rltakashige/mlx-jaccl-fix-small-recv.git?branch=address-rdma-gpu-locks#257d5692fc7af6bba3b8afaeb63c549b7d1e43d5" }, marker = "sys_platform == 'darwin'" },
@@ -1229,6 +1540,23 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/b7/da/7d22601b625e241d4f23ef1ebff8acfc60da633c9e7e7922e24d10f592b3/multidict-6.7.0-py3-none-any.whl", hash = "sha256:394fc5c42a333c9ffc3e421a4c85e08580d990e08b99f6bf35b4132114c5dcb3", size = 12317, upload-time = "2025-10-06T14:52:29.272Z" },
]
+[[package]]
+name = "multiprocess"
+version = "0.70.18"
+source = { registry = "https://pypi.org/simple" }
+dependencies = [
+ { name = "dill", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+sdist = { url = "https://files.pythonhosted.org/packages/72/fd/2ae3826f5be24c6ed87266bc4e59c46ea5b059a103f3d7e7eb76a52aeecb/multiprocess-0.70.18.tar.gz", hash = "sha256:f9597128e6b3e67b23956da07cf3d2e5cba79e2f4e0fba8d7903636663ec6d0d", size = 1798503, upload-time = "2025-04-17T03:11:27.742Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/ba/d8/0cba6cf51a1a31f20471fbc823a716170c73012ddc4fb85d706630ed6e8f/multiprocess-0.70.18-py310-none-any.whl", hash = "sha256:60c194974c31784019c1f459d984e8f33ee48f10fcf42c309ba97b30d9bd53ea", size = 134948, upload-time = "2025-04-17T03:11:20.223Z" },
+ { url = "https://files.pythonhosted.org/packages/4b/88/9039f2fed1012ef584751d4ceff9ab4a51e5ae264898f0b7cbf44340a859/multiprocess-0.70.18-py311-none-any.whl", hash = "sha256:5aa6eef98e691281b3ad923be2832bf1c55dd2c859acd73e5ec53a66aae06a1d", size = 144462, upload-time = "2025-04-17T03:11:21.657Z" },
+ { url = "https://files.pythonhosted.org/packages/bf/b6/5f922792be93b82ec6b5f270bbb1ef031fd0622847070bbcf9da816502cc/multiprocess-0.70.18-py312-none-any.whl", hash = "sha256:9b78f8e5024b573730bfb654783a13800c2c0f2dfc0c25e70b40d184d64adaa2", size = 150287, upload-time = "2025-04-17T03:11:22.69Z" },
+ { url = "https://files.pythonhosted.org/packages/ee/25/7d7e78e750bc1aecfaf0efbf826c69a791d2eeaf29cf20cba93ff4cced78/multiprocess-0.70.18-py313-none-any.whl", hash = "sha256:871743755f43ef57d7910a38433cfe41319e72be1bbd90b79c7a5ac523eb9334", size = 151917, upload-time = "2025-04-17T03:11:24.044Z" },
+ { url = "https://files.pythonhosted.org/packages/3b/c3/ca84c19bd14cdfc21c388fdcebf08b86a7a470ebc9f5c3c084fc2dbc50f7/multiprocess-0.70.18-py38-none-any.whl", hash = "sha256:dbf705e52a154fe5e90fb17b38f02556169557c2dd8bb084f2e06c2784d8279b", size = 132636, upload-time = "2025-04-17T03:11:24.936Z" },
+ { url = "https://files.pythonhosted.org/packages/6c/28/dd72947e59a6a8c856448a5e74da6201cb5502ddff644fbc790e4bd40b9a/multiprocess-0.70.18-py39-none-any.whl", hash = "sha256:e78ca805a72b1b810c690b6b4cc32579eba34f403094bbbae962b7b5bf9dfcb8", size = 133478, upload-time = "2025-04-17T03:11:26.253Z" },
+]
+
[[package]]
name = "networkx"
version = "3.6.1"
@@ -1265,6 +1593,21 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/e2/c8/97a2d5f7a314cce2c5c49f30c6f161b7f3617960ade4bfc2fd1ee092cb20/nh3-0.3.2-cp38-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:91e9b001101fb4500a2aafe3e7c92928d85242d38bf5ac0aba0b7480da0a4cd6", size = 987439, upload-time = "2025-10-30T11:17:40.81Z" },
]
+[[package]]
+name = "nltk"
+version = "3.9.3"
+source = { registry = "https://pypi.org/simple" }
+dependencies = [
+ { name = "click", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "joblib", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "regex", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "tqdm", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+sdist = { url = "https://files.pythonhosted.org/packages/e1/8f/915e1c12df07c70ed779d18ab83d065718a926e70d3ea33eb0cd66ffb7c0/nltk-3.9.3.tar.gz", hash = "sha256:cb5945d6424a98d694c2b9a0264519fab4363711065a46aa0ae7a2195b92e71f", size = 2923673, upload-time = "2026-02-24T12:05:53.833Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/c2/7e/9af5a710a1236e4772de8dfcc6af942a561327bb9f42b5b4a24d0cf100fd/nltk-3.9.3-py3-none-any.whl", hash = "sha256:60b3db6e9995b3dd976b1f0fa7dec22069b2677e759c28eb69b62ddd44870522", size = 1525385, upload-time = "2026-02-24T12:05:46.54Z" },
+]
+
[[package]]
name = "nodejs-wheel-binaries"
version = "25.2.1rc0"
@@ -1536,6 +1879,51 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/40/35/ddf3a6e8fc754fb939e2ea36fde96c28189184d6115afcf60011bb438ae5/packaging-26.0rc1-py3-none-any.whl", hash = "sha256:ecf921b33c620e357b1eed2ac3bc6313b1582874b0282d0773b6797b79cb0786", size = 74021, upload-time = "2026-01-09T17:41:17.134Z" },
]
+[[package]]
+name = "pandas"
+version = "3.0.1"
+source = { registry = "https://pypi.org/simple" }
+dependencies = [
+ { name = "numpy", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "python-dateutil", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+sdist = { url = "https://files.pythonhosted.org/packages/2e/0c/b28ed414f080ee0ad153f848586d61d1878f91689950f037f976ce15f6c8/pandas-3.0.1.tar.gz", hash = "sha256:4186a699674af418f655dbd420ed87f50d56b4cd6603784279d9eef6627823c8", size = 4641901, upload-time = "2026-02-17T22:20:16.434Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/0b/48/aad6ec4f8d007534c091e9a7172b3ec1b1ee6d99a9cbb936b5eab6c6cf58/pandas-3.0.1-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:5272627187b5d9c20e55d27caf5f2cd23e286aba25cadf73c8590e432e2b7262", size = 10317509, upload-time = "2026-02-17T22:18:59.498Z" },
+ { url = "https://files.pythonhosted.org/packages/a8/14/5990826f779f79148ae9d3a2c39593dc04d61d5d90541e71b5749f35af95/pandas-3.0.1-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:661e0f665932af88c7877f31da0dc743fe9c8f2524bdffe23d24fdcb67ef9d56", size = 9860561, upload-time = "2026-02-17T22:19:02.265Z" },
+ { url = "https://files.pythonhosted.org/packages/fa/80/f01ff54664b6d70fed71475543d108a9b7c888e923ad210795bef04ffb7d/pandas-3.0.1-cp313-cp313-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:75e6e292ff898679e47a2199172593d9f6107fd2dd3617c22c2946e97d5df46e", size = 10365506, upload-time = "2026-02-17T22:19:05.017Z" },
+ { url = "https://files.pythonhosted.org/packages/f2/85/ab6d04733a7d6ff32bfc8382bf1b07078228f5d6ebec5266b91bfc5c4ff7/pandas-3.0.1-cp313-cp313-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:1ff8cf1d2896e34343197685f432450ec99a85ba8d90cce2030c5eee2ef98791", size = 10873196, upload-time = "2026-02-17T22:19:07.204Z" },
+ { url = "https://files.pythonhosted.org/packages/48/a9/9301c83d0b47c23ac5deab91c6b39fd98d5b5db4d93b25df8d381451828f/pandas-3.0.1-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:eca8b4510f6763f3d37359c2105df03a7a221a508f30e396a51d0713d462e68a", size = 11370859, upload-time = "2026-02-17T22:19:09.436Z" },
+ { url = "https://files.pythonhosted.org/packages/59/fe/0c1fc5bd2d29c7db2ab372330063ad555fb83e08422829c785f5ec2176ca/pandas-3.0.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:06aff2ad6f0b94a17822cf8b83bbb563b090ed82ff4fe7712db2ce57cd50d9b8", size = 11924584, upload-time = "2026-02-17T22:19:11.562Z" },
+ { url = "https://files.pythonhosted.org/packages/92/fa/423c89086cca1f039cf1253c3ff5b90f157b5b3757314aa635f6bf3e30aa/pandas-3.0.1-cp313-cp313t-macosx_10_13_x86_64.whl", hash = "sha256:d54855f04f8246ed7b6fc96b05d4871591143c46c0b6f4af874764ed0d2d6f06", size = 10752673, upload-time = "2026-02-17T22:19:18.304Z" },
+ { url = "https://files.pythonhosted.org/packages/22/23/b5a08ec1f40020397f0faba72f1e2c11f7596a6169c7b3e800abff0e433f/pandas-3.0.1-cp313-cp313t-macosx_11_0_arm64.whl", hash = "sha256:4e1b677accee34a09e0dc2ce5624e4a58a1870ffe56fc021e9caf7f23cd7668f", size = 10404967, upload-time = "2026-02-17T22:19:20.726Z" },
+ { url = "https://files.pythonhosted.org/packages/5c/81/94841f1bb4afdc2b52a99daa895ac2c61600bb72e26525ecc9543d453ebc/pandas-3.0.1-cp313-cp313t-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:a9cabbdcd03f1b6cd254d6dda8ae09b0252524be1592594c00b7895916cb1324", size = 10320575, upload-time = "2026-02-17T22:19:24.919Z" },
+ { url = "https://files.pythonhosted.org/packages/0a/8b/2ae37d66a5342a83adadfd0cb0b4bf9c3c7925424dd5f40d15d6cfaa35ee/pandas-3.0.1-cp313-cp313t-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:5ae2ab1f166668b41e770650101e7090824fd34d17915dd9cd479f5c5e0065e9", size = 10710921, upload-time = "2026-02-17T22:19:27.181Z" },
+ { url = "https://files.pythonhosted.org/packages/a2/61/772b2e2757855e232b7ccf7cb8079a5711becb3a97f291c953def15a833f/pandas-3.0.1-cp313-cp313t-musllinux_1_2_aarch64.whl", hash = "sha256:6bf0603c2e30e2cafac32807b06435f28741135cb8697eae8b28c7d492fc7d76", size = 11334191, upload-time = "2026-02-17T22:19:29.411Z" },
+ { url = "https://files.pythonhosted.org/packages/1b/08/b16c6df3ef555d8495d1d265a7963b65be166785d28f06a350913a4fac78/pandas-3.0.1-cp313-cp313t-musllinux_1_2_x86_64.whl", hash = "sha256:6c426422973973cae1f4a23e51d4ae85974f44871b24844e4f7de752dd877098", size = 11782256, upload-time = "2026-02-17T22:19:32.34Z" },
+ { url = "https://files.pythonhosted.org/packages/bb/8b/4bb774a998b97e6c2fd62a9e6cfdaae133b636fd1c468f92afb4ae9a447a/pandas-3.0.1-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:99d0f92ed92d3083d140bf6b97774f9f13863924cf3f52a70711f4e7588f9d0a", size = 10322465, upload-time = "2026-02-17T22:19:36.803Z" },
+ { url = "https://files.pythonhosted.org/packages/72/3a/5b39b51c64159f470f1ca3b1c2a87da290657ca022f7cd11442606f607d1/pandas-3.0.1-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:3b66857e983208654294bb6477b8a63dee26b37bdd0eb34d010556e91261784f", size = 9910632, upload-time = "2026-02-17T22:19:39.001Z" },
+ { url = "https://files.pythonhosted.org/packages/4e/f7/b449ffb3f68c11da12fc06fbf6d2fa3a41c41e17d0284d23a79e1c13a7e4/pandas-3.0.1-cp314-cp314-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:56cf59638bf24dc9bdf2154c81e248b3289f9a09a6d04e63608c159022352749", size = 10440535, upload-time = "2026-02-17T22:19:41.157Z" },
+ { url = "https://files.pythonhosted.org/packages/55/77/6ea82043db22cb0f2bbfe7198da3544000ddaadb12d26be36e19b03a2dc5/pandas-3.0.1-cp314-cp314-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:c1a9f55e0f46951874b863d1f3906dcb57df2d9be5c5847ba4dfb55b2c815249", size = 10893940, upload-time = "2026-02-17T22:19:43.493Z" },
+ { url = "https://files.pythonhosted.org/packages/03/30/f1b502a72468c89412c1b882a08f6eed8a4ee9dc033f35f65d0663df6081/pandas-3.0.1-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:1849f0bba9c8a2fb0f691d492b834cc8dadf617e29015c66e989448d58d011ee", size = 11442711, upload-time = "2026-02-17T22:19:46.074Z" },
+ { url = "https://files.pythonhosted.org/packages/0d/f0/ebb6ddd8fc049e98cabac5c2924d14d1dda26a20adb70d41ea2e428d3ec4/pandas-3.0.1-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:c3d288439e11b5325b02ae6e9cc83e6805a62c40c5a6220bea9beb899c073b1c", size = 11963918, upload-time = "2026-02-17T22:19:48.838Z" },
+ { url = "https://files.pythonhosted.org/packages/66/fc/848bb6710bc6061cb0c5badd65b92ff75c81302e0e31e496d00029fe4953/pandas-3.0.1-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:58eeb1b2e0fb322befcf2bbc9ba0af41e616abadb3d3414a6bc7167f6cbfce32", size = 10772664, upload-time = "2026-02-17T22:19:55.806Z" },
+ { url = "https://files.pythonhosted.org/packages/69/5c/866a9bbd0f79263b4b0db6ec1a341be13a1473323f05c122388e0f15b21d/pandas-3.0.1-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:cd9af1276b5ca9e298bd79a26bda32fa9cc87ed095b2a9a60978d2ca058eaf87", size = 10421286, upload-time = "2026-02-17T22:19:58.091Z" },
+ { url = "https://files.pythonhosted.org/packages/51/a4/2058fb84fb1cfbfb2d4a6d485e1940bb4ad5716e539d779852494479c580/pandas-3.0.1-cp314-cp314t-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:94f87a04984d6b63788327cd9f79dda62b7f9043909d2440ceccf709249ca988", size = 10342050, upload-time = "2026-02-17T22:20:01.376Z" },
+ { url = "https://files.pythonhosted.org/packages/22/1b/674e89996cc4be74db3c4eb09240c4bb549865c9c3f5d9b086ff8fcfbf00/pandas-3.0.1-cp314-cp314t-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:85fe4c4df62e1e20f9db6ebfb88c844b092c22cd5324bdcf94bfa2fc1b391221", size = 10740055, upload-time = "2026-02-17T22:20:04.328Z" },
+ { url = "https://files.pythonhosted.org/packages/d0/f8/e954b750764298c22fa4614376531fe63c521ef517e7059a51f062b87dca/pandas-3.0.1-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:331ca75a2f8672c365ae25c0b29e46f5ac0c6551fdace8eec4cd65e4fac271ff", size = 11357632, upload-time = "2026-02-17T22:20:06.647Z" },
+ { url = "https://files.pythonhosted.org/packages/6d/02/c6e04b694ffd68568297abd03588b6d30295265176a5c01b7459d3bc35a3/pandas-3.0.1-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:15860b1fdb1973fffade772fdb931ccf9b2f400a3f5665aef94a00445d7d8dd5", size = 11810974, upload-time = "2026-02-17T22:20:08.946Z" },
+]
+
+[[package]]
+name = "pathvalidate"
+version = "3.3.1"
+source = { registry = "https://pypi.org/simple" }
+sdist = { url = "https://files.pythonhosted.org/packages/fa/2a/52a8da6fe965dea6192eb716b357558e103aea0a1e9a8352ad575a8406ca/pathvalidate-3.3.1.tar.gz", hash = "sha256:b18c07212bfead624345bb8e1d6141cdcf15a39736994ea0b94035ad2b1ba177", size = 63262, upload-time = "2025-06-15T09:07:20.736Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/9a/70/875f4a23bfc4731703a5835487d0d2fb999031bd415e7d17c0ae615c18b7/pathvalidate-3.3.1-py3-none-any.whl", hash = "sha256:5263baab691f8e1af96092fa5137ee17df5bdfbd6cff1fcac4d6ef4bc2e1735f", size = 24305, upload-time = "2025-06-15T09:07:19.117Z" },
+]
+
[[package]]
name = "piexif"
version = "1.1.3"
@@ -1606,6 +1994,15 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/54/20/4d324d65cc6d9205fabedc306948156824eb9f0ee1633355a8f7ec5c66bf/pluggy-1.6.0-py3-none-any.whl", hash = "sha256:e920276dd6813095e9377c0bc5566d94c932c33b27a3e3945d8389c374dd4746", size = 20538, upload-time = "2025-05-15T12:30:06.134Z" },
]
+[[package]]
+name = "portalocker"
+version = "3.2.0"
+source = { registry = "https://pypi.org/simple" }
+sdist = { url = "https://files.pythonhosted.org/packages/5e/77/65b857a69ed876e1951e88aaba60f5ce6120c33703f7cb61a3c894b8c1b6/portalocker-3.2.0.tar.gz", hash = "sha256:1f3002956a54a8c3730586c5c77bf18fae4149e07eaf1c29fc3faf4d5a3f89ac", size = 95644, upload-time = "2025-06-14T13:20:40.03Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/4b/a6/38c8e2f318bf67d338f4d629e93b0b4b9af331f455f0390ea8ce4a099b26/portalocker-3.2.0-py3-none-any.whl", hash = "sha256:3cdc5f565312224bc570c49337bd21428bba0ef363bbcf58b9ef4a9f11779968", size = 22424, upload-time = "2025-06-14T13:20:38.083Z" },
+]
+
[[package]]
name = "priority"
version = "2.0.0"
@@ -1707,6 +2104,38 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/1c/15/dd6fd869753ce82ff64dcbc18356093471a5a5adf4f77ed1f805d473d859/psutil-7.2.1-cp36-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:99a4cd17a5fdd1f3d014396502daa70b5ec21bf4ffe38393e152f8e449757d67", size = 147402, upload-time = "2025-12-29T08:26:39.21Z" },
]
+[[package]]
+name = "pyarrow"
+version = "23.0.1"
+source = { registry = "https://pypi.org/simple" }
+sdist = { url = "https://files.pythonhosted.org/packages/88/22/134986a4cc224d593c1afde5494d18ff629393d74cc2eddb176669f234a4/pyarrow-23.0.1.tar.gz", hash = "sha256:b8c5873e33440b2bc2f4a79d2b47017a89c5a24116c055625e6f2ee50523f019", size = 1167336, upload-time = "2026-02-16T10:14:12.39Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/47/10/2cbe4c6f0fb83d2de37249567373d64327a5e4d8db72f486db42875b08f6/pyarrow-23.0.1-cp313-cp313-macosx_12_0_arm64.whl", hash = "sha256:6b8fda694640b00e8af3c824f99f789e836720aa8c9379fb435d4c4953a756b8", size = 34210066, upload-time = "2026-02-16T10:10:45.487Z" },
+ { url = "https://files.pythonhosted.org/packages/cb/4f/679fa7e84dadbaca7a65f7cdba8d6c83febbd93ca12fa4adf40ba3b6362b/pyarrow-23.0.1-cp313-cp313-macosx_12_0_x86_64.whl", hash = "sha256:8ff51b1addc469b9444b7c6f3548e19dc931b172ab234e995a60aea9f6e6025f", size = 35825526, upload-time = "2026-02-16T10:10:52.266Z" },
+ { url = "https://files.pythonhosted.org/packages/f9/63/d2747d930882c9d661e9398eefc54f15696547b8983aaaf11d4a2e8b5426/pyarrow-23.0.1-cp313-cp313-manylinux_2_28_aarch64.whl", hash = "sha256:71c5be5cbf1e1cb6169d2a0980850bccb558ddc9b747b6206435313c47c37677", size = 44473279, upload-time = "2026-02-16T10:11:01.557Z" },
+ { url = "https://files.pythonhosted.org/packages/b3/93/10a48b5e238de6d562a411af6467e71e7aedbc9b87f8d3a35f1560ae30fb/pyarrow-23.0.1-cp313-cp313-manylinux_2_28_x86_64.whl", hash = "sha256:9b6f4f17b43bc39d56fec96e53fe89d94bac3eb134137964371b45352d40d0c2", size = 47585798, upload-time = "2026-02-16T10:11:09.401Z" },
+ { url = "https://files.pythonhosted.org/packages/5c/20/476943001c54ef078dbf9542280e22741219a184a0632862bca4feccd666/pyarrow-23.0.1-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:9fc13fc6c403d1337acab46a2c4346ca6c9dec5780c3c697cf8abfd5e19b6b37", size = 48179446, upload-time = "2026-02-16T10:11:17.781Z" },
+ { url = "https://files.pythonhosted.org/packages/4b/b6/5dd0c47b335fcd8edba9bfab78ad961bd0fd55ebe53468cc393f45e0be60/pyarrow-23.0.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:5c16ed4f53247fa3ffb12a14d236de4213a4415d127fe9cebed33d51671113e2", size = 50623972, upload-time = "2026-02-16T10:11:26.185Z" },
+ { url = "https://files.pythonhosted.org/packages/a5/8e/38749c4b1303e6ae76b3c80618f84861ae0c55dd3c2273842ea6f8258233/pyarrow-23.0.1-cp313-cp313t-macosx_12_0_arm64.whl", hash = "sha256:29f7f7419a0e30264ea261fdc0e5fe63ce5a6095003db2945d7cd78df391a7e1", size = 34471544, upload-time = "2026-02-16T10:11:32.535Z" },
+ { url = "https://files.pythonhosted.org/packages/a3/73/f237b2bc8c669212f842bcfd842b04fc8d936bfc9d471630569132dc920d/pyarrow-23.0.1-cp313-cp313t-macosx_12_0_x86_64.whl", hash = "sha256:33d648dc25b51fd8055c19e4261e813dfc4d2427f068bcecc8b53d01b81b0500", size = 35949911, upload-time = "2026-02-16T10:11:39.813Z" },
+ { url = "https://files.pythonhosted.org/packages/0c/86/b912195eee0903b5611bf596833def7d146ab2d301afeb4b722c57ffc966/pyarrow-23.0.1-cp313-cp313t-manylinux_2_28_aarch64.whl", hash = "sha256:cd395abf8f91c673dd3589cadc8cc1ee4e8674fa61b2e923c8dd215d9c7d1f41", size = 44520337, upload-time = "2026-02-16T10:11:47.764Z" },
+ { url = "https://files.pythonhosted.org/packages/69/c2/f2a717fb824f62d0be952ea724b4f6f9372a17eed6f704b5c9526f12f2f1/pyarrow-23.0.1-cp313-cp313t-manylinux_2_28_x86_64.whl", hash = "sha256:00be9576d970c31defb5c32eb72ef585bf600ef6d0a82d5eccaae96639cf9d07", size = 47548944, upload-time = "2026-02-16T10:11:56.607Z" },
+ { url = "https://files.pythonhosted.org/packages/84/a7/90007d476b9f0dc308e3bc57b832d004f848fd6c0da601375d20d92d1519/pyarrow-23.0.1-cp313-cp313t-musllinux_1_2_aarch64.whl", hash = "sha256:c2139549494445609f35a5cda4eb94e2c9e4d704ce60a095b342f82460c73a83", size = 48236269, upload-time = "2026-02-16T10:12:04.47Z" },
+ { url = "https://files.pythonhosted.org/packages/b0/3f/b16fab3e77709856eb6ac328ce35f57a6d4a18462c7ca5186ef31b45e0e0/pyarrow-23.0.1-cp313-cp313t-musllinux_1_2_x86_64.whl", hash = "sha256:7044b442f184d84e2351e5084600f0d7343d6117aabcbc1ac78eb1ae11eb4125", size = 50604794, upload-time = "2026-02-16T10:12:11.797Z" },
+ { url = "https://files.pythonhosted.org/packages/8d/1b/6da9a89583ce7b23ac611f183ae4843cd3a6cf54f079549b0e8c14031e73/pyarrow-23.0.1-cp314-cp314-macosx_12_0_arm64.whl", hash = "sha256:5df1161da23636a70838099d4aaa65142777185cc0cdba4037a18cee7d8db9ca", size = 34238755, upload-time = "2026-02-16T10:12:32.819Z" },
+ { url = "https://files.pythonhosted.org/packages/ae/b5/d58a241fbe324dbaeb8df07be6af8752c846192d78d2272e551098f74e88/pyarrow-23.0.1-cp314-cp314-macosx_12_0_x86_64.whl", hash = "sha256:fa8e51cb04b9f8c9c5ace6bab63af9a1f88d35c0d6cbf53e8c17c098552285e1", size = 35847826, upload-time = "2026-02-16T10:12:38.949Z" },
+ { url = "https://files.pythonhosted.org/packages/54/a5/8cbc83f04aba433ca7b331b38f39e000efd9f0c7ce47128670e737542996/pyarrow-23.0.1-cp314-cp314-manylinux_2_28_aarch64.whl", hash = "sha256:0b95a3994f015be13c63148fef8832e8a23938128c185ee951c98908a696e0eb", size = 44536859, upload-time = "2026-02-16T10:12:45.467Z" },
+ { url = "https://files.pythonhosted.org/packages/36/2e/c0f017c405fcdc252dbccafbe05e36b0d0eb1ea9a958f081e01c6972927f/pyarrow-23.0.1-cp314-cp314-manylinux_2_28_x86_64.whl", hash = "sha256:4982d71350b1a6e5cfe1af742c53dfb759b11ce14141870d05d9e540d13bc5d1", size = 47614443, upload-time = "2026-02-16T10:12:55.525Z" },
+ { url = "https://files.pythonhosted.org/packages/af/6b/2314a78057912f5627afa13ba43809d9d653e6630859618b0fd81a4e0759/pyarrow-23.0.1-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:c250248f1fe266db627921c89b47b7c06fee0489ad95b04d50353537d74d6886", size = 48232991, upload-time = "2026-02-16T10:13:04.729Z" },
+ { url = "https://files.pythonhosted.org/packages/40/f2/1bcb1d3be3460832ef3370d621142216e15a2c7c62602a4ea19ec240dd64/pyarrow-23.0.1-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:5f4763b83c11c16e5f4c15601ba6dfa849e20723b46aa2617cb4bffe8768479f", size = 50645077, upload-time = "2026-02-16T10:13:14.147Z" },
+ { url = "https://files.pythonhosted.org/packages/b5/78/07f67434e910a0f7323269be7bfbf58699bd0c1d080b18a1ab49ba943fe8/pyarrow-23.0.1-cp314-cp314t-macosx_12_0_arm64.whl", hash = "sha256:17cd28e906c18af486a499422740298c52d7c6795344ea5002a7720b4eadf16d", size = 34488692, upload-time = "2026-02-16T10:13:21.541Z" },
+ { url = "https://files.pythonhosted.org/packages/50/76/34cf7ae93ece1f740a04910d9f7e80ba166b9b4ab9596a953e9e62b90fe1/pyarrow-23.0.1-cp314-cp314t-macosx_12_0_x86_64.whl", hash = "sha256:76e823d0e86b4fb5e1cf4a58d293036e678b5a4b03539be933d3b31f9406859f", size = 35964383, upload-time = "2026-02-16T10:13:28.63Z" },
+ { url = "https://files.pythonhosted.org/packages/46/90/459b827238936d4244214be7c684e1b366a63f8c78c380807ae25ed92199/pyarrow-23.0.1-cp314-cp314t-manylinux_2_28_aarch64.whl", hash = "sha256:a62e1899e3078bf65943078b3ad2a6ddcacf2373bc06379aac61b1e548a75814", size = 44538119, upload-time = "2026-02-16T10:13:35.506Z" },
+ { url = "https://files.pythonhosted.org/packages/28/a1/93a71ae5881e99d1f9de1d4554a87be37da11cd6b152239fb5bd924fdc64/pyarrow-23.0.1-cp314-cp314t-manylinux_2_28_x86_64.whl", hash = "sha256:df088e8f640c9fae3b1f495b3c64755c4e719091caf250f3a74d095ddf3c836d", size = 47571199, upload-time = "2026-02-16T10:13:42.504Z" },
+ { url = "https://files.pythonhosted.org/packages/88/a3/d2c462d4ef313521eaf2eff04d204ac60775263f1fb08c374b543f79f610/pyarrow-23.0.1-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:46718a220d64677c93bc243af1d44b55998255427588e400677d7192671845c7", size = 48259435, upload-time = "2026-02-16T10:13:49.226Z" },
+ { url = "https://files.pythonhosted.org/packages/cc/f1/11a544b8c3d38a759eb3fbb022039117fd633e9a7b19e4841cc3da091915/pyarrow-23.0.1-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:a09f3876e87f48bc2f13583ab551f0379e5dfb83210391e68ace404181a20690", size = 50629149, upload-time = "2026-02-16T10:13:57.238Z" },
+]
+
[[package]]
name = "pycparser"
version = "2.23"
@@ -1829,6 +2258,24 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/8b/40/2614036cdd416452f5bf98ec037f38a1afb17f327cb8e6b652d4729e0af8/pyparsing-3.3.1-py3-none-any.whl", hash = "sha256:023b5e7e5520ad96642e2c6db4cb683d3970bd640cdf7115049a6e9c3682df82", size = 121793, upload-time = "2025-12-23T03:14:02.103Z" },
]
+[[package]]
+name = "pytablewriter"
+version = "1.2.1"
+source = { registry = "https://pypi.org/simple" }
+dependencies = [
+ { name = "dataproperty", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "mbstrdecoder", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "pathvalidate", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "setuptools", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "tabledata", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "tcolorpy", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "typepy", extra = ["datetime"], marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+sdist = { url = "https://files.pythonhosted.org/packages/f6/a1/617730f290f04d347103ab40bf67d317df6691b14746f6e1ea039fb57062/pytablewriter-1.2.1.tar.gz", hash = "sha256:7bd0f4f397e070e3b8a34edcf1b9257ccbb18305493d8350a5dbc9957fced959", size = 619241, upload-time = "2025-01-01T15:37:00.04Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/21/4c/c199512f01c845dfe5a7840ab3aae6c60463b5dc2a775be72502dfd9170a/pytablewriter-1.2.1-py3-none-any.whl", hash = "sha256:e906ff7ff5151d70a5f66e0f7b75642a7f2dce8d893c265b79cc9cf6bc04ddb4", size = 91083, upload-time = "2025-01-01T15:36:55.63Z" },
+]
+
[[package]]
name = "pytest"
version = "9.0.2"
@@ -1889,6 +2336,15 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/aa/76/03af049af4dcee5d27442f71b6924f01f3efb5d2bd34f23fcd563f2cc5f5/python_multipart-0.0.21-py3-none-any.whl", hash = "sha256:cf7a6713e01c87aa35387f4774e812c4361150938d20d232800f75ffcf266090", size = 24541, upload-time = "2025-12-17T09:24:21.153Z" },
]
+[[package]]
+name = "pytz"
+version = "2026.1.post1"
+source = { registry = "https://pypi.org/simple" }
+sdist = { url = "https://files.pythonhosted.org/packages/56/db/b8721d71d945e6a8ac63c0fc900b2067181dbb50805958d4d4661cf7d277/pytz-2026.1.post1.tar.gz", hash = "sha256:3378dde6a0c3d26719182142c56e60c7f9af7e968076f31aae569d72a0358ee1", size = 321088, upload-time = "2026-03-03T07:47:50.683Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/10/99/781fe0c827be2742bcc775efefccb3b048a3a9c6ce9aec0cbf4a101677e5/pytz-2026.1.post1-py2.py3-none-any.whl", hash = "sha256:f2fd16142fda348286a75e1a524be810bb05d444e5a081f37f7affc635035f7a", size = 510489, upload-time = "2026-03-03T07:47:49.167Z" },
+]
+
[[package]]
name = "pyyaml"
version = "6.0.3"
@@ -2033,6 +2489,18 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/25/7a/b0178788f8dc6cafce37a212c99565fa1fe7872c70c6c9c1e1a372d9d88f/rich-14.2.0-py3-none-any.whl", hash = "sha256:76bc51fe2e57d2b1be1f96c524b890b816e334ab4c1e45888799bfaab0021edd", size = 243393, upload-time = "2025-10-09T14:16:51.245Z" },
]
+[[package]]
+name = "rouge-score"
+version = "0.1.2"
+source = { registry = "https://pypi.org/simple" }
+dependencies = [
+ { name = "absl-py", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "nltk", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "numpy", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "six", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+sdist = { url = "https://files.pythonhosted.org/packages/e2/c5/9136736c37022a6ad27fea38f3111eb8f02fe75d067f9a985cc358653102/rouge_score-0.1.2.tar.gz", hash = "sha256:c7d4da2683e68c9abf0135ef915d63a46643666f848e558a1b9f7ead17ff0f04", size = 17400, upload-time = "2022-07-22T22:46:22.909Z" }
+
[[package]]
name = "ruff"
version = "0.14.11"
@@ -2076,6 +2544,23 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/4f/ef/c9199e4b6336ee5a9f1979c11b5779c5cf9ab6f8386e0b9a96c8ffba7009/rustworkx-0.17.1-cp39-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:48784a673cf8d04f3cd246fa6b53fd1ccc4d83304503463bd561c153517bccc1", size = 2302783, upload-time = "2025-08-13T01:43:42.073Z" },
]
+[[package]]
+name = "sacrebleu"
+version = "2.6.0"
+source = { registry = "https://pypi.org/simple" }
+dependencies = [
+ { name = "colorama", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "lxml", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "numpy", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "portalocker", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "regex", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "tabulate", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+sdist = { url = "https://files.pythonhosted.org/packages/d3/ed/d7acddcff74d690c56fe26a1f7828bdde548262828d0743414ea916c40c1/sacrebleu-2.6.0.tar.gz", hash = "sha256:91499b6cd46138d95154fff1e863c2f9be57e82f0c719d8dd718d0006cf6c566", size = 1893419, upload-time = "2026-01-12T17:17:20.799Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/06/f2/6c90ccf3ad1d09a7d662a405b274f3c93b92df59c8d6a025d26aaf34d302/sacrebleu-2.6.0-py3-none-any.whl", hash = "sha256:3edc1531575cfe4ad04ce53491a9307e234af1c3f805a1f491cbec844229a8a8", size = 100785, upload-time = "2026-01-12T17:17:18.868Z" },
+]
+
[[package]]
name = "safetensors"
version = "0.7.0"
@@ -2096,6 +2581,79 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/4a/d8/0c8a7dc9b41dcac53c4cbf9df2b9c83e0e0097203de8b37a712b345c0be5/safetensors-0.7.0-cp38-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:b0f6d66c1c538d5a94a73aa9ddca8ccc4227e6c9ff555322ea40bdd142391dd4", size = 677368, upload-time = "2025-11-19T15:18:41.627Z" },
]
+[[package]]
+name = "scikit-learn"
+version = "1.8.0"
+source = { registry = "https://pypi.org/simple" }
+dependencies = [
+ { name = "joblib", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "numpy", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "scipy", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "threadpoolctl", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+sdist = { url = "https://files.pythonhosted.org/packages/0e/d4/40988bf3b8e34feec1d0e6a051446b1f66225f8529b9309becaeef62b6c4/scikit_learn-1.8.0.tar.gz", hash = "sha256:9bccbb3b40e3de10351f8f5068e105d0f4083b1a65fa07b6634fbc401a6287fd", size = 7335585, upload-time = "2025-12-10T07:08:53.618Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/03/aa/e22e0768512ce9255eba34775be2e85c2048da73da1193e841707f8f039c/scikit_learn-1.8.0-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:0d6ae97234d5d7079dc0040990a6f7aeb97cb7fa7e8945f1999a429b23569e0a", size = 8513770, upload-time = "2025-12-10T07:08:03.251Z" },
+ { url = "https://files.pythonhosted.org/packages/58/37/31b83b2594105f61a381fc74ca19e8780ee923be2d496fcd8d2e1147bd99/scikit_learn-1.8.0-cp313-cp313-macosx_12_0_arm64.whl", hash = "sha256:edec98c5e7c128328124a029bceb09eda2d526997780fef8d65e9a69eead963e", size = 8044458, upload-time = "2025-12-10T07:08:05.336Z" },
+ { url = "https://files.pythonhosted.org/packages/2d/5a/3f1caed8765f33eabb723596666da4ebbf43d11e96550fb18bdec42b467b/scikit_learn-1.8.0-cp313-cp313-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:74b66d8689d52ed04c271e1329f0c61635bcaf5b926db9b12d58914cdc01fe57", size = 8610341, upload-time = "2025-12-10T07:08:07.732Z" },
+ { url = "https://files.pythonhosted.org/packages/38/cf/06896db3f71c75902a8e9943b444a56e727418f6b4b4a90c98c934f51ed4/scikit_learn-1.8.0-cp313-cp313-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:8fdf95767f989b0cfedb85f7ed8ca215d4be728031f56ff5a519ee1e3276dc2e", size = 8900022, upload-time = "2025-12-10T07:08:09.862Z" },
+ { url = "https://files.pythonhosted.org/packages/d2/7d/a630359fc9dcc95496588c8d8e3245cc8fd81980251079bc09c70d41d951/scikit_learn-1.8.0-cp313-cp313t-macosx_10_13_x86_64.whl", hash = "sha256:7cc267b6108f0a1499a734167282c00c4ebf61328566b55ef262d48e9849c735", size = 8826045, upload-time = "2025-12-10T07:08:15.215Z" },
+ { url = "https://files.pythonhosted.org/packages/cc/56/a0c86f6930cfcd1c7054a2bc417e26960bb88d32444fe7f71d5c2cfae891/scikit_learn-1.8.0-cp313-cp313t-macosx_12_0_arm64.whl", hash = "sha256:fe1c011a640a9f0791146011dfd3c7d9669785f9fed2b2a5f9e207536cf5c2fd", size = 8420324, upload-time = "2025-12-10T07:08:17.561Z" },
+ { url = "https://files.pythonhosted.org/packages/46/1e/05962ea1cebc1cf3876667ecb14c283ef755bf409993c5946ade3b77e303/scikit_learn-1.8.0-cp313-cp313t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:72358cce49465d140cc4e7792015bb1f0296a9742d5622c67e31399b75468b9e", size = 8680651, upload-time = "2025-12-10T07:08:19.952Z" },
+ { url = "https://files.pythonhosted.org/packages/fe/56/a85473cd75f200c9759e3a5f0bcab2d116c92a8a02ee08ccd73b870f8bb4/scikit_learn-1.8.0-cp313-cp313t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:80832434a6cc114f5219211eec13dcbc16c2bac0e31ef64c6d346cde3cf054cb", size = 8925045, upload-time = "2025-12-10T07:08:22.11Z" },
+ { url = "https://files.pythonhosted.org/packages/24/05/1af2c186174cc92dcab2233f327336058c077d38f6fe2aceb08e6ab4d509/scikit_learn-1.8.0-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:c22a2da7a198c28dd1a6e1136f19c830beab7fdca5b3e5c8bba8394f8a5c45b3", size = 8528667, upload-time = "2025-12-10T07:08:27.541Z" },
+ { url = "https://files.pythonhosted.org/packages/a8/25/01c0af38fe969473fb292bba9dc2b8f9b451f3112ff242c647fee3d0dfe7/scikit_learn-1.8.0-cp314-cp314-macosx_12_0_arm64.whl", hash = "sha256:6b595b07a03069a2b1740dc08c2299993850ea81cce4fe19b2421e0c970de6b7", size = 8066524, upload-time = "2025-12-10T07:08:29.822Z" },
+ { url = "https://files.pythonhosted.org/packages/be/ce/a0623350aa0b68647333940ee46fe45086c6060ec604874e38e9ab7d8e6c/scikit_learn-1.8.0-cp314-cp314-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:29ffc74089f3d5e87dfca4c2c8450f88bdc61b0fc6ed5d267f3988f19a1309f6", size = 8657133, upload-time = "2025-12-10T07:08:31.865Z" },
+ { url = "https://files.pythonhosted.org/packages/b8/cb/861b41341d6f1245e6ca80b1c1a8c4dfce43255b03df034429089ca2a2c5/scikit_learn-1.8.0-cp314-cp314-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:fb65db5d7531bccf3a4f6bec3462223bea71384e2cda41da0f10b7c292b9e7c4", size = 8923223, upload-time = "2025-12-10T07:08:34.166Z" },
+ { url = "https://files.pythonhosted.org/packages/2d/d1/ef294ca754826daa043b2a104e59960abfab4cf653891037d19dd5b6f3cf/scikit_learn-1.8.0-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:4511be56637e46c25721e83d1a9cea9614e7badc7040c4d573d75fbe257d6fd7", size = 8848305, upload-time = "2025-12-10T07:08:41.013Z" },
+ { url = "https://files.pythonhosted.org/packages/5b/e2/b1f8b05138ee813b8e1a4149f2f0d289547e60851fd1bb268886915adbda/scikit_learn-1.8.0-cp314-cp314t-macosx_12_0_arm64.whl", hash = "sha256:a69525355a641bf8ef136a7fa447672fb54fe8d60cab5538d9eb7c6438543fb9", size = 8432257, upload-time = "2025-12-10T07:08:42.873Z" },
+ { url = "https://files.pythonhosted.org/packages/26/11/c32b2138a85dcb0c99f6afd13a70a951bfdff8a6ab42d8160522542fb647/scikit_learn-1.8.0-cp314-cp314t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:c2656924ec73e5939c76ac4c8b026fc203b83d8900362eb2599d8aee80e4880f", size = 8678673, upload-time = "2025-12-10T07:08:45.362Z" },
+ { url = "https://files.pythonhosted.org/packages/c7/57/51f2384575bdec454f4fe4e7a919d696c9ebce914590abf3e52d47607ab8/scikit_learn-1.8.0-cp314-cp314t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:15fc3b5d19cc2be65404786857f2e13c70c83dd4782676dd6814e3b89dc8f5b9", size = 8922467, upload-time = "2025-12-10T07:08:47.408Z" },
+]
+
+[[package]]
+name = "scipy"
+version = "1.17.1"
+source = { registry = "https://pypi.org/simple" }
+dependencies = [
+ { name = "numpy", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+sdist = { url = "https://files.pythonhosted.org/packages/7a/97/5a3609c4f8d58b039179648e62dd220f89864f56f7357f5d4f45c29eb2cc/scipy-1.17.1.tar.gz", hash = "sha256:95d8e012d8cb8816c226aef832200b1d45109ed4464303e997c5b13122b297c0", size = 30573822, upload-time = "2026-02-23T00:26:24.851Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/76/27/07ee1b57b65e92645f219b37148a7e7928b82e2b5dbeccecb4dff7c64f0b/scipy-1.17.1-cp313-cp313-macosx_10_14_x86_64.whl", hash = "sha256:5e3c5c011904115f88a39308379c17f91546f77c1667cea98739fe0fccea804c", size = 31590199, upload-time = "2026-02-23T00:19:17.192Z" },
+ { url = "https://files.pythonhosted.org/packages/ec/ae/db19f8ab842e9b724bf5dbb7db29302a91f1e55bc4d04b1025d6d605a2c5/scipy-1.17.1-cp313-cp313-macosx_12_0_arm64.whl", hash = "sha256:6fac755ca3d2c3edcb22f479fceaa241704111414831ddd3bc6056e18516892f", size = 28154001, upload-time = "2026-02-23T00:19:22.241Z" },
+ { url = "https://files.pythonhosted.org/packages/5b/58/3ce96251560107b381cbd6e8413c483bbb1228a6b919fa8652b0d4090e7f/scipy-1.17.1-cp313-cp313-macosx_14_0_arm64.whl", hash = "sha256:7ff200bf9d24f2e4d5dc6ee8c3ac64d739d3a89e2326ba68aaf6c4a2b838fd7d", size = 20325719, upload-time = "2026-02-23T00:19:26.329Z" },
+ { url = "https://files.pythonhosted.org/packages/b2/83/15087d945e0e4d48ce2377498abf5ad171ae013232ae31d06f336e64c999/scipy-1.17.1-cp313-cp313-macosx_14_0_x86_64.whl", hash = "sha256:4b400bdc6f79fa02a4d86640310dde87a21fba0c979efff5248908c6f15fad1b", size = 22683595, upload-time = "2026-02-23T00:19:30.304Z" },
+ { url = "https://files.pythonhosted.org/packages/b4/e0/e58fbde4a1a594c8be8114eb4aac1a55bcd6587047efc18a61eb1f5c0d30/scipy-1.17.1-cp313-cp313-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:2b64ca7d4aee0102a97f3ba22124052b4bd2152522355073580bf4845e2550b6", size = 32896429, upload-time = "2026-02-23T00:19:35.536Z" },
+ { url = "https://files.pythonhosted.org/packages/f5/5f/f17563f28ff03c7b6799c50d01d5d856a1d55f2676f537ca8d28c7f627cd/scipy-1.17.1-cp313-cp313-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:581b2264fc0aa555f3f435a5944da7504ea3a065d7029ad60e7c3d1ae09c5464", size = 35203952, upload-time = "2026-02-23T00:19:42.259Z" },
+ { url = "https://files.pythonhosted.org/packages/8d/a5/9afd17de24f657fdfe4df9a3f1ea049b39aef7c06000c13db1530d81ccca/scipy-1.17.1-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:beeda3d4ae615106d7094f7e7cef6218392e4465cc95d25f900bebabfded0950", size = 34979063, upload-time = "2026-02-23T00:19:47.547Z" },
+ { url = "https://files.pythonhosted.org/packages/8b/13/88b1d2384b424bf7c924f2038c1c409f8d88bb2a8d49d097861dd64a57b2/scipy-1.17.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:6609bc224e9568f65064cfa72edc0f24ee6655b47575954ec6339534b2798369", size = 37598449, upload-time = "2026-02-23T00:19:53.238Z" },
+ { url = "https://files.pythonhosted.org/packages/6f/6b/17787db8b8114933a66f9dcc479a8272e4b4da75fe03b0c282f7b0ade8cd/scipy-1.17.1-cp313-cp313t-macosx_10_14_x86_64.whl", hash = "sha256:d59c30000a16d8edc7e64152e30220bfbd724c9bbb08368c054e24c651314f0a", size = 31936708, upload-time = "2026-02-23T00:19:58.694Z" },
+ { url = "https://files.pythonhosted.org/packages/38/2e/524405c2b6392765ab1e2b722a41d5da33dc5c7b7278184a8ad29b6cb206/scipy-1.17.1-cp313-cp313t-macosx_12_0_arm64.whl", hash = "sha256:010f4333c96c9bb1a4516269e33cb5917b08ef2166d5556ca2fd9f082a9e6ea0", size = 28570135, upload-time = "2026-02-23T00:20:03.934Z" },
+ { url = "https://files.pythonhosted.org/packages/fd/c3/5bd7199f4ea8556c0c8e39f04ccb014ac37d1468e6cfa6a95c6b3562b76e/scipy-1.17.1-cp313-cp313t-macosx_14_0_arm64.whl", hash = "sha256:2ceb2d3e01c5f1d83c4189737a42d9cb2fc38a6eeed225e7515eef71ad301dce", size = 20741977, upload-time = "2026-02-23T00:20:07.935Z" },
+ { url = "https://files.pythonhosted.org/packages/d9/b8/8ccd9b766ad14c78386599708eb745f6b44f08400a5fd0ade7cf89b6fc93/scipy-1.17.1-cp313-cp313t-macosx_14_0_x86_64.whl", hash = "sha256:844e165636711ef41f80b4103ed234181646b98a53c8f05da12ca5ca289134f6", size = 23029601, upload-time = "2026-02-23T00:20:12.161Z" },
+ { url = "https://files.pythonhosted.org/packages/6d/a0/3cb6f4d2fb3e17428ad2880333cac878909ad1a89f678527b5328b93c1d4/scipy-1.17.1-cp313-cp313t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:158dd96d2207e21c966063e1635b1063cd7787b627b6f07305315dd73d9c679e", size = 33019667, upload-time = "2026-02-23T00:20:17.208Z" },
+ { url = "https://files.pythonhosted.org/packages/f3/c3/2d834a5ac7bf3a0c806ad1508efc02dda3c8c61472a56132d7894c312dea/scipy-1.17.1-cp313-cp313t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:74cbb80d93260fe2ffa334efa24cb8f2f0f622a9b9febf8b483c0b865bfb3475", size = 35264159, upload-time = "2026-02-23T00:20:23.087Z" },
+ { url = "https://files.pythonhosted.org/packages/4d/77/d3ed4becfdbd217c52062fafe35a72388d1bd82c2d0ba5ca19d6fcc93e11/scipy-1.17.1-cp313-cp313t-musllinux_1_2_aarch64.whl", hash = "sha256:dbc12c9f3d185f5c737d801da555fb74b3dcfa1a50b66a1a93e09190f41fab50", size = 35102771, upload-time = "2026-02-23T00:20:28.636Z" },
+ { url = "https://files.pythonhosted.org/packages/bd/12/d19da97efde68ca1ee5538bb261d5d2c062f0c055575128f11a2730e3ac1/scipy-1.17.1-cp313-cp313t-musllinux_1_2_x86_64.whl", hash = "sha256:94055a11dfebe37c656e70317e1996dc197e1a15bbcc351bcdd4610e128fe1ca", size = 37665910, upload-time = "2026-02-23T00:20:34.743Z" },
+ { url = "https://files.pythonhosted.org/packages/cf/83/333afb452af6f0fd70414dc04f898647ee1423979ce02efa75c3b0f2c28e/scipy-1.17.1-cp314-cp314-macosx_10_14_x86_64.whl", hash = "sha256:a48a72c77a310327f6a3a920092fa2b8fd03d7deaa60f093038f22d98e096717", size = 31584510, upload-time = "2026-02-23T00:21:01.015Z" },
+ { url = "https://files.pythonhosted.org/packages/ed/a6/d05a85fd51daeb2e4ea71d102f15b34fedca8e931af02594193ae4fd25f7/scipy-1.17.1-cp314-cp314-macosx_12_0_arm64.whl", hash = "sha256:45abad819184f07240d8a696117a7aacd39787af9e0b719d00285549ed19a1e9", size = 28170131, upload-time = "2026-02-23T00:21:05.888Z" },
+ { url = "https://files.pythonhosted.org/packages/db/7b/8624a203326675d7746a254083a187398090a179335b2e4a20e2ddc46e83/scipy-1.17.1-cp314-cp314-macosx_14_0_arm64.whl", hash = "sha256:3fd1fcdab3ea951b610dc4cef356d416d5802991e7e32b5254828d342f7b7e0b", size = 20342032, upload-time = "2026-02-23T00:21:09.904Z" },
+ { url = "https://files.pythonhosted.org/packages/c9/35/2c342897c00775d688d8ff3987aced3426858fd89d5a0e26e020b660b301/scipy-1.17.1-cp314-cp314-macosx_14_0_x86_64.whl", hash = "sha256:7bdf2da170b67fdf10bca777614b1c7d96ae3ca5794fd9587dce41eb2966e866", size = 22678766, upload-time = "2026-02-23T00:21:14.313Z" },
+ { url = "https://files.pythonhosted.org/packages/ef/f2/7cdb8eb308a1a6ae1e19f945913c82c23c0c442a462a46480ce487fdc0ac/scipy-1.17.1-cp314-cp314-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:adb2642e060a6549c343603a3851ba76ef0b74cc8c079a9a58121c7ec9fe2350", size = 32957007, upload-time = "2026-02-23T00:21:19.663Z" },
+ { url = "https://files.pythonhosted.org/packages/0b/2e/7eea398450457ecb54e18e9d10110993fa65561c4f3add5e8eccd2b9cd41/scipy-1.17.1-cp314-cp314-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:eee2cfda04c00a857206a4330f0c5e3e56535494e30ca445eb19ec624ae75118", size = 35221333, upload-time = "2026-02-23T00:21:25.278Z" },
+ { url = "https://files.pythonhosted.org/packages/d9/77/5b8509d03b77f093a0d52e606d3c4f79e8b06d1d38c441dacb1e26cacf46/scipy-1.17.1-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:d2650c1fb97e184d12d8ba010493ee7b322864f7d3d00d3f9bb97d9c21de4068", size = 35042066, upload-time = "2026-02-23T00:21:31.358Z" },
+ { url = "https://files.pythonhosted.org/packages/f9/df/18f80fb99df40b4070328d5ae5c596f2f00fffb50167e31439e932f29e7d/scipy-1.17.1-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:08b900519463543aa604a06bec02461558a6e1cef8fdbb8098f77a48a83c8118", size = 37612763, upload-time = "2026-02-23T00:21:37.247Z" },
+ { url = "https://files.pythonhosted.org/packages/96/ad/f8c414e121f82e02d76f310f16db9899c4fcde36710329502a6b2a3c0392/scipy-1.17.1-cp314-cp314t-macosx_10_14_x86_64.whl", hash = "sha256:1cc682cea2ae55524432f3cdff9e9a3be743d52a7443d0cba9017c23c87ae2f6", size = 31949750, upload-time = "2026-02-23T00:21:42.289Z" },
+ { url = "https://files.pythonhosted.org/packages/7c/b0/c741e8865d61b67c81e255f4f0a832846c064e426636cd7de84e74d209be/scipy-1.17.1-cp314-cp314t-macosx_12_0_arm64.whl", hash = "sha256:2040ad4d1795a0ae89bfc7e8429677f365d45aa9fd5e4587cf1ea737f927b4a1", size = 28585858, upload-time = "2026-02-23T00:21:47.706Z" },
+ { url = "https://files.pythonhosted.org/packages/ed/1b/3985219c6177866628fa7c2595bfd23f193ceebbe472c98a08824b9466ff/scipy-1.17.1-cp314-cp314t-macosx_14_0_arm64.whl", hash = "sha256:131f5aaea57602008f9822e2115029b55d4b5f7c070287699fe45c661d051e39", size = 20757723, upload-time = "2026-02-23T00:21:52.039Z" },
+ { url = "https://files.pythonhosted.org/packages/c0/19/2a04aa25050d656d6f7b9e7b685cc83d6957fb101665bfd9369ca6534563/scipy-1.17.1-cp314-cp314t-macosx_14_0_x86_64.whl", hash = "sha256:9cdc1a2fcfd5c52cfb3045feb399f7b3ce822abdde3a193a6b9a60b3cb5854ca", size = 23043098, upload-time = "2026-02-23T00:21:56.185Z" },
+ { url = "https://files.pythonhosted.org/packages/86/f1/3383beb9b5d0dbddd030335bf8a8b32d4317185efe495374f134d8be6cce/scipy-1.17.1-cp314-cp314t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:6e3dcd57ab780c741fde8dc68619de988b966db759a3c3152e8e9142c26295ad", size = 33030397, upload-time = "2026-02-23T00:22:01.404Z" },
+ { url = "https://files.pythonhosted.org/packages/41/68/8f21e8a65a5a03f25a79165ec9d2b28c00e66dc80546cf5eb803aeeff35b/scipy-1.17.1-cp314-cp314t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:a9956e4d4f4a301ebf6cde39850333a6b6110799d470dbbb1e25326ac447f52a", size = 35281163, upload-time = "2026-02-23T00:22:07.024Z" },
+ { url = "https://files.pythonhosted.org/packages/84/8d/c8a5e19479554007a5632ed7529e665c315ae7492b4f946b0deb39870e39/scipy-1.17.1-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:a4328d245944d09fd639771de275701ccadf5f781ba0ff092ad141e017eccda4", size = 35116291, upload-time = "2026-02-23T00:22:12.585Z" },
+ { url = "https://files.pythonhosted.org/packages/52/52/e57eceff0e342a1f50e274264ed47497b59e6a4e3118808ee58ddda7b74a/scipy-1.17.1-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:a77cbd07b940d326d39a1d1b37817e2ee4d79cb30e7338f3d0cddffae70fcaa2", size = 37682317, upload-time = "2026-02-23T00:22:18.513Z" },
+]
+
[[package]]
name = "secretstorage"
version = "3.5.0"
@@ -2173,6 +2731,12 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/e9/44/75a9c9421471a6c4805dbf2356f7c181a29c1879239abab1ea2cc8f38b40/sniffio-1.3.1-py3-none-any.whl", hash = "sha256:2f6da418d1f1e0fddd844478f41680e794e6051915791a034ff65e5f100525a2", size = 10235, upload-time = "2024-02-25T23:20:01.196Z" },
]
+[[package]]
+name = "sqlitedict"
+version = "2.1.0"
+source = { registry = "https://pypi.org/simple" }
+sdist = { url = "https://files.pythonhosted.org/packages/12/9a/7620d1e9dcb02839ed6d4b14064e609cdd7a8ae1e47289aa0456796dd9ca/sqlitedict-2.1.0.tar.gz", hash = "sha256:03d9cfb96d602996f1d4c2db2856f1224b96a9c431bdd16e78032a72940f9e8c", size = 21846, upload-time = "2022-12-03T13:39:13.102Z" }
+
[[package]]
name = "starlette"
version = "0.50.0"
@@ -2197,6 +2761,64 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/a2/09/77d55d46fd61b4a135c444fc97158ef34a095e5681d0a6c10b75bf356191/sympy-1.14.0-py3-none-any.whl", hash = "sha256:e091cc3e99d2141a0ba2847328f5479b05d94a6635cb96148ccb3f34671bd8f5", size = 6299353, upload-time = "2025-04-27T18:04:59.103Z" },
]
+[[package]]
+name = "tabledata"
+version = "1.3.4"
+source = { registry = "https://pypi.org/simple" }
+dependencies = [
+ { name = "dataproperty", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "typepy", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+sdist = { url = "https://files.pythonhosted.org/packages/b2/35/171c8977162f1163368406deddde4c59673b62bd0cb2f34948a02effb075/tabledata-1.3.4.tar.gz", hash = "sha256:e9649cab129d718f3bff4150083b77f8a78c30f6634a30caf692b10fdc60cb97", size = 25074, upload-time = "2024-12-31T14:12:31.198Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/08/64/fa4160151976ee4b2cf0c1217a99443ffaeb991956feddfeac9eee9952f8/tabledata-1.3.4-py3-none-any.whl", hash = "sha256:1f56e433bfdeb89f4487abfa48c4603a3b07c5d3a3c7e05ff73dd018c24bd0d4", size = 11820, upload-time = "2024-12-31T14:12:28.584Z" },
+]
+
+[[package]]
+name = "tabulate"
+version = "0.10.0"
+source = { registry = "https://pypi.org/simple" }
+sdist = { url = "https://files.pythonhosted.org/packages/46/58/8c37dea7bbf769b20d58e7ace7e5edfe65b849442b00ffcdd56be88697c6/tabulate-0.10.0.tar.gz", hash = "sha256:e2cfde8f79420f6deeffdeda9aaec3b6bc5abce947655d17ac662b126e48a60d", size = 91754, upload-time = "2026-03-04T18:55:34.402Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/99/55/db07de81b5c630da5cbf5c7df646580ca26dfaefa593667fc6f2fe016d2e/tabulate-0.10.0-py3-none-any.whl", hash = "sha256:f0b0622e567335c8fabaaa659f1b33bcb6ddfe2e496071b743aa113f8774f2d3", size = 39814, upload-time = "2026-03-04T18:55:31.284Z" },
+]
+
+[[package]]
+name = "tcolorpy"
+version = "0.1.7"
+source = { registry = "https://pypi.org/simple" }
+sdist = { url = "https://files.pythonhosted.org/packages/80/cc/44f2d81d8f9093aad81c3467a5bf5718d2b5f786e887b6e4adcfc17ec6b9/tcolorpy-0.1.7.tar.gz", hash = "sha256:0fbf6bf238890bbc2e32662aa25736769a29bf6d880328f310c910a327632614", size = 299437, upload-time = "2024-12-29T15:24:23.847Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/05/a2/ed023f2edd1e011b4d99b6727bce8253842d66c3fbf9ed0a26fc09a92571/tcolorpy-0.1.7-py3-none-any.whl", hash = "sha256:26a59d52027e175a37e0aba72efc99dda43f074db71f55b316d3de37d3251378", size = 8096, upload-time = "2024-12-29T15:24:21.33Z" },
+]
+
+[[package]]
+name = "tenacity"
+version = "9.1.4"
+source = { registry = "https://pypi.org/simple" }
+sdist = { url = "https://files.pythonhosted.org/packages/47/c6/ee486fd809e357697ee8a44d3d69222b344920433d3b6666ccd9b374630c/tenacity-9.1.4.tar.gz", hash = "sha256:adb31d4c263f2bd041081ab33b498309a57c77f9acf2db65aadf0898179cf93a", size = 49413, upload-time = "2026-02-07T10:45:33.841Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/d7/c1/eb8f9debc45d3b7918a32ab756658a0904732f75e555402972246b0b8e71/tenacity-9.1.4-py3-none-any.whl", hash = "sha256:6095a360c919085f28c6527de529e76a06ad89b23659fa881ae0649b867a9d55", size = 28926, upload-time = "2026-02-07T10:45:32.24Z" },
+]
+
+[[package]]
+name = "termcolor"
+version = "3.3.0"
+source = { registry = "https://pypi.org/simple" }
+sdist = { url = "https://files.pythonhosted.org/packages/46/79/cf31d7a93a8fdc6aa0fbb665be84426a8c5a557d9240b6239e9e11e35fc5/termcolor-3.3.0.tar.gz", hash = "sha256:348871ca648ec6a9a983a13ab626c0acce02f515b9e1983332b17af7979521c5", size = 14434, upload-time = "2025-12-29T12:55:21.882Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/33/d1/8bb87d21e9aeb323cc03034f5eaf2c8f69841e40e4853c2627edf8111ed3/termcolor-3.3.0-py3-none-any.whl", hash = "sha256:cf642efadaf0a8ebbbf4bc7a31cec2f9b5f21a9f726f4ccbb08192c9c26f43a5", size = 7734, upload-time = "2025-12-29T12:55:20.718Z" },
+]
+
+[[package]]
+name = "threadpoolctl"
+version = "3.6.0"
+source = { registry = "https://pypi.org/simple" }
+sdist = { url = "https://files.pythonhosted.org/packages/b7/4d/08c89e34946fce2aec4fbb45c9016efd5f4d7f24af8e5d93296e935631d8/threadpoolctl-3.6.0.tar.gz", hash = "sha256:8ab8b4aa3491d812b623328249fab5302a68d2d71745c8a4c719a2fcaba9f44e", size = 21274, upload-time = "2025-03-13T13:49:23.031Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/32/d5/f9a850d79b0851d1d4ef6456097579a9005b31fea68726a4ae5f2d82ddd9/threadpoolctl-3.6.0-py3-none-any.whl", hash = "sha256:43a0b8fd5a2928500110039e43a5eed8480b918967083ea48dc3ab9f13c4a7fb", size = 18638, upload-time = "2025-03-13T13:49:21.846Z" },
+]
+
[[package]]
name = "tiktoken"
version = "0.12.0"
@@ -2379,6 +3001,25 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/3a/7a/882d99539b19b1490cac5d77c67338d126e4122c8276bf640e411650c830/twine-6.2.0-py3-none-any.whl", hash = "sha256:418ebf08ccda9a8caaebe414433b0ba5e25eb5e4a927667122fbe8f829f985d8", size = 42727, upload-time = "2025-09-04T15:43:15.994Z" },
]
+[[package]]
+name = "typepy"
+version = "1.3.4"
+source = { registry = "https://pypi.org/simple" }
+dependencies = [
+ { name = "mbstrdecoder", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+sdist = { url = "https://files.pythonhosted.org/packages/79/59/4c39942077d7de285f762a91024dbda731be693591732977358f77d120fb/typepy-1.3.4.tar.gz", hash = "sha256:89c1f66de6c6133209c43a94d23431d320ba03ef5db18f241091ea594035d9de", size = 39558, upload-time = "2024-12-29T09:18:15.774Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/ee/31/e393c3830bdedd01735bd195c85ac3034b6bcaf6c18142bab60a4047ca36/typepy-1.3.4-py3-none-any.whl", hash = "sha256:d5ed3e0c7f49521bff0603dd08cf8d453371cf68d65a29d3d0038552ccc46e2e", size = 31449, upload-time = "2024-12-29T09:18:13.135Z" },
+]
+
+[package.optional-dependencies]
+datetime = [
+ { name = "packaging", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "python-dateutil", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+ { name = "pytz", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
+]
+
[[package]]
name = "typer-slim"
version = "0.21.1"
@@ -2431,6 +3072,12 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/39/08/aaaad47bc4e9dc8c725e68f9d04865dbcb2052843ff09c97b08904852d84/urllib3-2.6.3-py3-none-any.whl", hash = "sha256:bf272323e553dfb2e87d9bfd225ca7b0f467b919d7bbd355436d3fd37cb0acd4", size = 131584, upload-time = "2026-01-07T16:24:42.685Z" },
]
+[[package]]
+name = "word2number"
+version = "1.1"
+source = { registry = "https://pypi.org/simple" }
+sdist = { url = "https://files.pythonhosted.org/packages/4a/29/a31940c848521f0725f0df6b25dca8917f13a2025b0e8fcbe5d0457e45e6/word2number-1.1.zip", hash = "sha256:70e27a5d387f67b04c71fbb7621c05930b19bfd26efd6851e6e0f9969dcde7d0", size = 9723, upload-time = "2017-06-02T15:45:14.488Z" }
+
[[package]]
name = "wsproto"
version = "1.3.2"
@@ -2443,6 +3090,62 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/a4/f5/10b68b7b1544245097b2a1b8238f66f2fc6dcaeb24ba5d917f52bd2eed4f/wsproto-1.3.2-py3-none-any.whl", hash = "sha256:61eea322cdf56e8cc904bd3ad7573359a242ba65688716b0710a5eb12beab584", size = 24405, upload-time = "2025-11-20T18:18:00.454Z" },
]
+[[package]]
+name = "xxhash"
+version = "3.6.0"
+source = { registry = "https://pypi.org/simple" }
+sdist = { url = "https://files.pythonhosted.org/packages/02/84/30869e01909fb37a6cc7e18688ee8bf1e42d57e7e0777636bd47524c43c7/xxhash-3.6.0.tar.gz", hash = "sha256:f0162a78b13a0d7617b2845b90c763339d1f1d82bb04a4b07f4ab535cc5e05d6", size = 85160, upload-time = "2025-10-02T14:37:08.097Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/33/76/35d05267ac82f53ae9b0e554da7c5e281ee61f3cad44c743f0fcd354f211/xxhash-3.6.0-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:599e64ba7f67472481ceb6ee80fa3bd828fd61ba59fb11475572cc5ee52b89ec", size = 32738, upload-time = "2025-10-02T14:34:55.839Z" },
+ { url = "https://files.pythonhosted.org/packages/31/a8/3fbce1cd96534a95e35d5120637bf29b0d7f5d8fa2f6374e31b4156dd419/xxhash-3.6.0-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:7d8b8aaa30fca4f16f0c84a5c8d7ddee0e25250ec2796c973775373257dde8f1", size = 30821, upload-time = "2025-10-02T14:34:57.219Z" },
+ { url = "https://files.pythonhosted.org/packages/0c/ea/d387530ca7ecfa183cb358027f1833297c6ac6098223fd14f9782cd0015c/xxhash-3.6.0-cp313-cp313-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:d597acf8506d6e7101a4a44a5e428977a51c0fadbbfd3c39650cca9253f6e5a6", size = 194127, upload-time = "2025-10-02T14:34:59.21Z" },
+ { url = "https://files.pythonhosted.org/packages/ba/0c/71435dcb99874b09a43b8d7c54071e600a7481e42b3e3ce1eb5226a5711a/xxhash-3.6.0-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:858dc935963a33bc33490128edc1c12b0c14d9c7ebaa4e387a7869ecc4f3e263", size = 212975, upload-time = "2025-10-02T14:35:00.816Z" },
+ { url = "https://files.pythonhosted.org/packages/84/7a/c2b3d071e4bb4a90b7057228a99b10d51744878f4a8a6dd643c8bd897620/xxhash-3.6.0-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:ba284920194615cb8edf73bf52236ce2e1664ccd4a38fdb543506413529cc546", size = 212241, upload-time = "2025-10-02T14:35:02.207Z" },
+ { url = "https://files.pythonhosted.org/packages/81/5f/640b6eac0128e215f177df99eadcd0f1b7c42c274ab6a394a05059694c5a/xxhash-3.6.0-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:4b54219177f6c6674d5378bd862c6aedf64725f70dd29c472eaae154df1a2e89", size = 445471, upload-time = "2025-10-02T14:35:03.61Z" },
+ { url = "https://files.pythonhosted.org/packages/5e/1e/3c3d3ef071b051cc3abbe3721ffb8365033a172613c04af2da89d5548a87/xxhash-3.6.0-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:42c36dd7dbad2f5238950c377fcbf6811b1cdb1c444fab447960030cea60504d", size = 193936, upload-time = "2025-10-02T14:35:05.013Z" },
+ { url = "https://files.pythonhosted.org/packages/2c/bd/4a5f68381939219abfe1c22a9e3a5854a4f6f6f3c4983a87d255f21f2e5d/xxhash-3.6.0-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:f22927652cba98c44639ffdc7aaf35828dccf679b10b31c4ad72a5b530a18eb7", size = 210440, upload-time = "2025-10-02T14:35:06.239Z" },
+ { url = "https://files.pythonhosted.org/packages/eb/37/b80fe3d5cfb9faff01a02121a0f4d565eb7237e9e5fc66e73017e74dcd36/xxhash-3.6.0-cp313-cp313-musllinux_1_2_i686.whl", hash = "sha256:b45fad44d9c5c119e9c6fbf2e1c656a46dc68e280275007bbfd3d572b21426db", size = 197990, upload-time = "2025-10-02T14:35:07.735Z" },
+ { url = "https://files.pythonhosted.org/packages/d7/fd/2c0a00c97b9e18f72e1f240ad4e8f8a90fd9d408289ba9c7c495ed7dc05c/xxhash-3.6.0-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:6f2580ffab1a8b68ef2b901cde7e55fa8da5e4be0977c68f78fc80f3c143de42", size = 210689, upload-time = "2025-10-02T14:35:09.438Z" },
+ { url = "https://files.pythonhosted.org/packages/93/86/5dd8076a926b9a95db3206aba20d89a7fc14dd5aac16e5c4de4b56033140/xxhash-3.6.0-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:40c391dd3cd041ebc3ffe6f2c862f402e306eb571422e0aa918d8070ba31da11", size = 414068, upload-time = "2025-10-02T14:35:11.162Z" },
+ { url = "https://files.pythonhosted.org/packages/af/3c/0bb129170ee8f3650f08e993baee550a09593462a5cddd8e44d0011102b1/xxhash-3.6.0-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:f205badabde7aafd1a31e8ca2a3e5a763107a71c397c4481d6a804eb5063d8bd", size = 191495, upload-time = "2025-10-02T14:35:12.971Z" },
+ { url = "https://files.pythonhosted.org/packages/f3/30/25e5321c8732759e930c555176d37e24ab84365482d257c3b16362235212/xxhash-3.6.0-cp313-cp313t-macosx_10_13_x86_64.whl", hash = "sha256:a42e633d75cdad6d625434e3468126c73f13f7584545a9cf34e883aa1710e702", size = 32956, upload-time = "2025-10-02T14:35:17.413Z" },
+ { url = "https://files.pythonhosted.org/packages/9f/3c/0573299560d7d9f8ab1838f1efc021a280b5ae5ae2e849034ef3dee18810/xxhash-3.6.0-cp313-cp313t-macosx_11_0_arm64.whl", hash = "sha256:568a6d743219e717b07b4e03b0a828ce593833e498c3b64752e0f5df6bfe84db", size = 31072, upload-time = "2025-10-02T14:35:18.844Z" },
+ { url = "https://files.pythonhosted.org/packages/7a/1c/52d83a06e417cd9d4137722693424885cc9878249beb3a7c829e74bf7ce9/xxhash-3.6.0-cp313-cp313t-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:bec91b562d8012dae276af8025a55811b875baace6af510412a5e58e3121bc54", size = 196409, upload-time = "2025-10-02T14:35:20.31Z" },
+ { url = "https://files.pythonhosted.org/packages/e3/8e/c6d158d12a79bbd0b878f8355432075fc82759e356ab5a111463422a239b/xxhash-3.6.0-cp313-cp313t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:78e7f2f4c521c30ad5e786fdd6bae89d47a32672a80195467b5de0480aa97b1f", size = 215736, upload-time = "2025-10-02T14:35:21.616Z" },
+ { url = "https://files.pythonhosted.org/packages/bc/68/c4c80614716345d55071a396cf03d06e34b5f4917a467faf43083c995155/xxhash-3.6.0-cp313-cp313t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:3ed0df1b11a79856df5ffcab572cbd6b9627034c1c748c5566fa79df9048a7c5", size = 214833, upload-time = "2025-10-02T14:35:23.32Z" },
+ { url = "https://files.pythonhosted.org/packages/7e/e9/ae27c8ffec8b953efa84c7c4a6c6802c263d587b9fc0d6e7cea64e08c3af/xxhash-3.6.0-cp313-cp313t-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:0e4edbfc7d420925b0dd5e792478ed393d6e75ff8fc219a6546fb446b6a417b1", size = 448348, upload-time = "2025-10-02T14:35:25.111Z" },
+ { url = "https://files.pythonhosted.org/packages/d7/6b/33e21afb1b5b3f46b74b6bd1913639066af218d704cc0941404ca717fc57/xxhash-3.6.0-cp313-cp313t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:fba27a198363a7ef87f8c0f6b171ec36b674fe9053742c58dd7e3201c1ab30ee", size = 196070, upload-time = "2025-10-02T14:35:26.586Z" },
+ { url = "https://files.pythonhosted.org/packages/96/b6/fcabd337bc5fa624e7203aa0fa7d0c49eed22f72e93229431752bddc83d9/xxhash-3.6.0-cp313-cp313t-musllinux_1_2_aarch64.whl", hash = "sha256:794fe9145fe60191c6532fa95063765529770edcdd67b3d537793e8004cabbfd", size = 212907, upload-time = "2025-10-02T14:35:28.087Z" },
+ { url = "https://files.pythonhosted.org/packages/4b/d3/9ee6160e644d660fcf176c5825e61411c7f62648728f69c79ba237250143/xxhash-3.6.0-cp313-cp313t-musllinux_1_2_i686.whl", hash = "sha256:6105ef7e62b5ac73a837778efc331a591d8442f8ef5c7e102376506cb4ae2729", size = 200839, upload-time = "2025-10-02T14:35:29.857Z" },
+ { url = "https://files.pythonhosted.org/packages/0d/98/e8de5baa5109394baf5118f5e72ab21a86387c4f89b0e77ef3e2f6b0327b/xxhash-3.6.0-cp313-cp313t-musllinux_1_2_ppc64le.whl", hash = "sha256:f01375c0e55395b814a679b3eea205db7919ac2af213f4a6682e01220e5fe292", size = 213304, upload-time = "2025-10-02T14:35:31.222Z" },
+ { url = "https://files.pythonhosted.org/packages/7b/1d/71056535dec5c3177eeb53e38e3d367dd1d16e024e63b1cee208d572a033/xxhash-3.6.0-cp313-cp313t-musllinux_1_2_s390x.whl", hash = "sha256:d706dca2d24d834a4661619dcacf51a75c16d65985718d6a7d73c1eeeb903ddf", size = 416930, upload-time = "2025-10-02T14:35:32.517Z" },
+ { url = "https://files.pythonhosted.org/packages/dc/6c/5cbde9de2cd967c322e651c65c543700b19e7ae3e0aae8ece3469bf9683d/xxhash-3.6.0-cp313-cp313t-musllinux_1_2_x86_64.whl", hash = "sha256:5f059d9faeacd49c0215d66f4056e1326c80503f51a1532ca336a385edadd033", size = 193787, upload-time = "2025-10-02T14:35:33.827Z" },
+ { url = "https://files.pythonhosted.org/packages/7e/5e/0138bc4484ea9b897864d59fce9be9086030825bc778b76cb5a33a906d37/xxhash-3.6.0-cp314-cp314-macosx_10_13_x86_64.whl", hash = "sha256:a40a3d35b204b7cc7643cbcf8c9976d818cb47befcfac8bbefec8038ac363f3e", size = 32754, upload-time = "2025-10-02T14:35:38.245Z" },
+ { url = "https://files.pythonhosted.org/packages/18/d7/5dac2eb2ec75fd771957a13e5dda560efb2176d5203f39502a5fc571f899/xxhash-3.6.0-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:a54844be970d3fc22630b32d515e79a90d0a3ddb2644d8d7402e3c4c8da61405", size = 30846, upload-time = "2025-10-02T14:35:39.6Z" },
+ { url = "https://files.pythonhosted.org/packages/fe/71/8bc5be2bb00deb5682e92e8da955ebe5fa982da13a69da5a40a4c8db12fb/xxhash-3.6.0-cp314-cp314-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:016e9190af8f0a4e3741343777710e3d5717427f175adfdc3e72508f59e2a7f3", size = 194343, upload-time = "2025-10-02T14:35:40.69Z" },
+ { url = "https://files.pythonhosted.org/packages/e7/3b/52badfb2aecec2c377ddf1ae75f55db3ba2d321c5e164f14461c90837ef3/xxhash-3.6.0-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:4f6f72232f849eb9d0141e2ebe2677ece15adfd0fa599bc058aad83c714bb2c6", size = 213074, upload-time = "2025-10-02T14:35:42.29Z" },
+ { url = "https://files.pythonhosted.org/packages/a2/2b/ae46b4e9b92e537fa30d03dbc19cdae57ed407e9c26d163895e968e3de85/xxhash-3.6.0-cp314-cp314-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:63275a8aba7865e44b1813d2177e0f5ea7eadad3dd063a21f7cf9afdc7054063", size = 212388, upload-time = "2025-10-02T14:35:43.929Z" },
+ { url = "https://files.pythonhosted.org/packages/f5/80/49f88d3afc724b4ac7fbd664c8452d6db51b49915be48c6982659e0e7942/xxhash-3.6.0-cp314-cp314-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:3cd01fa2aa00d8b017c97eb46b9a794fbdca53fc14f845f5a328c71254b0abb7", size = 445614, upload-time = "2025-10-02T14:35:45.216Z" },
+ { url = "https://files.pythonhosted.org/packages/ed/ba/603ce3961e339413543d8cd44f21f2c80e2a7c5cfe692a7b1f2cccf58f3c/xxhash-3.6.0-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:0226aa89035b62b6a86d3c68df4d7c1f47a342b8683da2b60cedcddb46c4d95b", size = 194024, upload-time = "2025-10-02T14:35:46.959Z" },
+ { url = "https://files.pythonhosted.org/packages/78/d1/8e225ff7113bf81545cfdcd79eef124a7b7064a0bba53605ff39590b95c2/xxhash-3.6.0-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:c6e193e9f56e4ca4923c61238cdaced324f0feac782544eb4c6d55ad5cc99ddd", size = 210541, upload-time = "2025-10-02T14:35:48.301Z" },
+ { url = "https://files.pythonhosted.org/packages/6f/58/0f89d149f0bad89def1a8dd38feb50ccdeb643d9797ec84707091d4cb494/xxhash-3.6.0-cp314-cp314-musllinux_1_2_i686.whl", hash = "sha256:9176dcaddf4ca963d4deb93866d739a343c01c969231dbe21680e13a5d1a5bf0", size = 198305, upload-time = "2025-10-02T14:35:49.584Z" },
+ { url = "https://files.pythonhosted.org/packages/11/38/5eab81580703c4df93feb5f32ff8fa7fe1e2c51c1f183ee4e48d4bb9d3d7/xxhash-3.6.0-cp314-cp314-musllinux_1_2_ppc64le.whl", hash = "sha256:c1ce4009c97a752e682b897aa99aef84191077a9433eb237774689f14f8ec152", size = 210848, upload-time = "2025-10-02T14:35:50.877Z" },
+ { url = "https://files.pythonhosted.org/packages/5e/6b/953dc4b05c3ce678abca756416e4c130d2382f877a9c30a20d08ee6a77c0/xxhash-3.6.0-cp314-cp314-musllinux_1_2_s390x.whl", hash = "sha256:8cb2f4f679b01513b7adbb9b1b2f0f9cdc31b70007eaf9d59d0878809f385b11", size = 414142, upload-time = "2025-10-02T14:35:52.15Z" },
+ { url = "https://files.pythonhosted.org/packages/08/a9/238ec0d4e81a10eb5026d4a6972677cbc898ba6c8b9dbaec12ae001b1b35/xxhash-3.6.0-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:653a91d7c2ab54a92c19ccf43508b6a555440b9be1bc8be553376778be7f20b5", size = 191547, upload-time = "2025-10-02T14:35:53.547Z" },
+ { url = "https://files.pythonhosted.org/packages/2c/db/0e99732ed7f64182aef4a6fb145e1a295558deec2a746265dcdec12d191e/xxhash-3.6.0-cp314-cp314t-macosx_10_13_x86_64.whl", hash = "sha256:c5294f596a9017ca5a3e3f8884c00b91ab2ad2933cf288f4923c3fd4346cf3d4", size = 32955, upload-time = "2025-10-02T14:35:58.267Z" },
+ { url = "https://files.pythonhosted.org/packages/55/f4/2a7c3c68e564a099becfa44bb3d398810cc0ff6749b0d3cb8ccb93f23c14/xxhash-3.6.0-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:1cf9dcc4ab9cff01dfbba78544297a3a01dafd60f3bde4e2bfd016cf7e4ddc67", size = 31072, upload-time = "2025-10-02T14:35:59.382Z" },
+ { url = "https://files.pythonhosted.org/packages/c6/d9/72a29cddc7250e8a5819dad5d466facb5dc4c802ce120645630149127e73/xxhash-3.6.0-cp314-cp314t-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:01262da8798422d0685f7cef03b2bd3f4f46511b02830861df548d7def4402ad", size = 196579, upload-time = "2025-10-02T14:36:00.838Z" },
+ { url = "https://files.pythonhosted.org/packages/63/93/b21590e1e381040e2ca305a884d89e1c345b347404f7780f07f2cdd47ef4/xxhash-3.6.0-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:51a73fb7cb3a3ead9f7a8b583ffd9b8038e277cdb8cb87cf890e88b3456afa0b", size = 215854, upload-time = "2025-10-02T14:36:02.207Z" },
+ { url = "https://files.pythonhosted.org/packages/ce/b8/edab8a7d4fa14e924b29be877d54155dcbd8b80be85ea00d2be3413a9ed4/xxhash-3.6.0-cp314-cp314t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:b9c6df83594f7df8f7f708ce5ebeacfc69f72c9fbaaababf6cf4758eaada0c9b", size = 214965, upload-time = "2025-10-02T14:36:03.507Z" },
+ { url = "https://files.pythonhosted.org/packages/27/67/dfa980ac7f0d509d54ea0d5a486d2bb4b80c3f1bb22b66e6a05d3efaf6c0/xxhash-3.6.0-cp314-cp314t-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:627f0af069b0ea56f312fd5189001c24578868643203bca1abbc2c52d3a6f3ca", size = 448484, upload-time = "2025-10-02T14:36:04.828Z" },
+ { url = "https://files.pythonhosted.org/packages/8c/63/8ffc2cc97e811c0ca5d00ab36604b3ea6f4254f20b7bc658ca825ce6c954/xxhash-3.6.0-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:aa912c62f842dfd013c5f21a642c9c10cd9f4c4e943e0af83618b4a404d9091a", size = 196162, upload-time = "2025-10-02T14:36:06.182Z" },
+ { url = "https://files.pythonhosted.org/packages/4b/77/07f0e7a3edd11a6097e990f6e5b815b6592459cb16dae990d967693e6ea9/xxhash-3.6.0-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:b465afd7909db30168ab62afe40b2fcf79eedc0b89a6c0ab3123515dc0df8b99", size = 213007, upload-time = "2025-10-02T14:36:07.733Z" },
+ { url = "https://files.pythonhosted.org/packages/ae/d8/bc5fa0d152837117eb0bef6f83f956c509332ce133c91c63ce07ee7c4873/xxhash-3.6.0-cp314-cp314t-musllinux_1_2_i686.whl", hash = "sha256:a881851cf38b0a70e7c4d3ce81fc7afd86fbc2a024f4cfb2a97cf49ce04b75d3", size = 200956, upload-time = "2025-10-02T14:36:09.106Z" },
+ { url = "https://files.pythonhosted.org/packages/26/a5/d749334130de9411783873e9b98ecc46688dad5db64ca6e04b02acc8b473/xxhash-3.6.0-cp314-cp314t-musllinux_1_2_ppc64le.whl", hash = "sha256:9b3222c686a919a0f3253cfc12bb118b8b103506612253b5baeaac10d8027cf6", size = 213401, upload-time = "2025-10-02T14:36:10.585Z" },
+ { url = "https://files.pythonhosted.org/packages/89/72/abed959c956a4bfc72b58c0384bb7940663c678127538634d896b1195c10/xxhash-3.6.0-cp314-cp314t-musllinux_1_2_s390x.whl", hash = "sha256:c5aa639bc113e9286137cec8fadc20e9cd732b2cc385c0b7fa673b84fc1f2a93", size = 417083, upload-time = "2025-10-02T14:36:12.276Z" },
+ { url = "https://files.pythonhosted.org/packages/0c/b3/62fd2b586283b7d7d665fb98e266decadf31f058f1cf6c478741f68af0cb/xxhash-3.6.0-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:5c1343d49ac102799905e115aee590183c3921d475356cb24b4de29a4bc56518", size = 193913, upload-time = "2025-10-02T14:36:14.025Z" },
+]
+
[[package]]
name = "yarl"
version = "1.22.0"
← d0163610 Ciaran/gpt oss tool call finish reason (#1673)
·
back to Exo
·
feat(dashboard): add mobile drawer support for sidebars (#16 a6aa07ed →