← back to Exo
Add sampling defaults (#1947)
fcc3718efb563e82ab3a5963bc457b7f31f591e3 · 2026-04-21 07:45:33 +0100 · rltakashige
## Motivation
Model quality issues
### Manual Testing
TODO
Files touched
M .mlx_typings/mlx_lm/sample_utils.pyiM resources/inference_model_cards/mlx-community--DeepSeek-V3.1-4bit.tomlM resources/inference_model_cards/mlx-community--DeepSeek-V3.1-8bit.tomlM resources/inference_model_cards/mlx-community--DeepSeek-V3.2-4bit.tomlM resources/inference_model_cards/mlx-community--DeepSeek-V3.2-8bit.tomlM resources/inference_model_cards/mlx-community--GLM-4.5-Air-8bit.tomlM resources/inference_model_cards/mlx-community--GLM-4.5-Air-bf16.tomlM resources/inference_model_cards/mlx-community--GLM-4.7-4bit.tomlM resources/inference_model_cards/mlx-community--GLM-4.7-6bit.tomlM resources/inference_model_cards/mlx-community--GLM-4.7-8bit-gs32.tomlM resources/inference_model_cards/mlx-community--GLM-4.7-Flash-4bit.tomlM resources/inference_model_cards/mlx-community--GLM-4.7-Flash-5bit.tomlM resources/inference_model_cards/mlx-community--GLM-4.7-Flash-6bit.tomlM resources/inference_model_cards/mlx-community--GLM-4.7-Flash-8bit.tomlM resources/inference_model_cards/mlx-community--GLM-5-8bit.tomlM resources/inference_model_cards/mlx-community--GLM-5-MXFP4-Q8.tomlM resources/inference_model_cards/mlx-community--GLM-5-bf16.tomlM resources/inference_model_cards/mlx-community--Kimi-K2-Instruct-4bit.tomlM resources/inference_model_cards/mlx-community--Kimi-K2-Thinking.tomlM resources/inference_model_cards/mlx-community--Kimi-K2.5.tomlM resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-70B-Instruct-HF-4bit.tomlM resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-70B-Instruct-HF-8bit.tomlM resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-70B-Instruct-HF-bf16.tomlM resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-Nano-4B-v1.1-4bit.tomlM resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-Nano-4B-v1.1-8bit.tomlM resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-Nano-4B-v1.1-bf16.tomlM resources/inference_model_cards/mlx-community--Llama-3.2-1B-Instruct-4bit.tomlM resources/inference_model_cards/mlx-community--Llama-3.2-3B-Instruct-4bit.tomlM resources/inference_model_cards/mlx-community--Llama-3.2-3B-Instruct-8bit.tomlM resources/inference_model_cards/mlx-community--Llama-3.3-70B-Instruct-4bit.tomlM resources/inference_model_cards/mlx-community--Llama-3.3-70B-Instruct-8bit.tomlM resources/inference_model_cards/mlx-community--Meta-Llama-3.1-70B-Instruct-4bit.tomlM resources/inference_model_cards/mlx-community--Meta-Llama-3.1-8B-Instruct-4bit.tomlM resources/inference_model_cards/mlx-community--Meta-Llama-3.1-8B-Instruct-8bit.tomlM resources/inference_model_cards/mlx-community--Meta-Llama-3.1-8B-Instruct-bf16.tomlM resources/inference_model_cards/mlx-community--MiniMax-M2.1-3bit.tomlM resources/inference_model_cards/mlx-community--MiniMax-M2.1-8bit.tomlM resources/inference_model_cards/mlx-community--MiniMax-M2.5-4bit.tomlM resources/inference_model_cards/mlx-community--MiniMax-M2.5-6bit.tomlM resources/inference_model_cards/mlx-community--MiniMax-M2.5-8bit.tomlM resources/inference_model_cards/mlx-community--MiniMax-M2.7-4bit-mxfp4.tomlM resources/inference_model_cards/mlx-community--MiniMax-M2.7-4bit.tomlM resources/inference_model_cards/mlx-community--MiniMax-M2.7-5bit.tomlM resources/inference_model_cards/mlx-community--MiniMax-M2.7-6bit.tomlM resources/inference_model_cards/mlx-community--MiniMax-M2.7-8bit.tomlM resources/inference_model_cards/mlx-community--MiniMax-M2.7.tomlM resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-4Bit.tomlM resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-5Bit.tomlM resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-6Bit.tomlM resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-8Bit.tomlM resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-BF16.tomlM resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-MXFP4.tomlM resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-NVFP4.tomlM resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-Nano-9B-v2-4bits.tomlM resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-Nano-9B-v2-6bit.tomlM resources/inference_model_cards/mlx-community--Qwen3-0.6B-4bit.tomlM resources/inference_model_cards/mlx-community--Qwen3-0.6B-8bit.tomlM resources/inference_model_cards/mlx-community--Qwen3-235B-A22B-Instruct-2507-4bit.tomlM resources/inference_model_cards/mlx-community--Qwen3-235B-A22B-Instruct-2507-8bit.tomlM resources/inference_model_cards/mlx-community--Qwen3-30B-A3B-4bit.tomlM resources/inference_model_cards/mlx-community--Qwen3-30B-A3B-8bit.tomlM resources/inference_model_cards/mlx-community--Qwen3-Coder-480B-A35B-Instruct-4bit.tomlM resources/inference_model_cards/mlx-community--Qwen3-Coder-480B-A35B-Instruct-8bit.tomlM resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-4bit.tomlM resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-5bit.tomlM resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-6bit.tomlM resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-8bit.tomlM resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-bf16.tomlM resources/inference_model_cards/mlx-community--Qwen3-Next-80B-A3B-Instruct-4bit.tomlM resources/inference_model_cards/mlx-community--Qwen3-Next-80B-A3B-Instruct-8bit.tomlM resources/inference_model_cards/mlx-community--Qwen3-Next-80B-A3B-Thinking-4bit.tomlM resources/inference_model_cards/mlx-community--Qwen3-Next-80B-A3B-Thinking-8bit.tomlM resources/inference_model_cards/mlx-community--Qwen3-VL-4B-Instruct-4bit.tomlM resources/inference_model_cards/mlx-community--Qwen3.5-122B-A10B-4bit.tomlM resources/inference_model_cards/mlx-community--Qwen3.5-122B-A10B-6bit.tomlM resources/inference_model_cards/mlx-community--Qwen3.5-122B-A10B-8bit.tomlM resources/inference_model_cards/mlx-community--Qwen3.5-122B-A10B-bf16.tomlM resources/inference_model_cards/mlx-community--Qwen3.5-27B-4bit.tomlM resources/inference_model_cards/mlx-community--Qwen3.5-27B-8bit.tomlM resources/inference_model_cards/mlx-community--Qwen3.5-2B-MLX-8bit.tomlM resources/inference_model_cards/mlx-community--Qwen3.5-35B-A3B-4bit.tomlM resources/inference_model_cards/mlx-community--Qwen3.5-35B-A3B-8bit.tomlM resources/inference_model_cards/mlx-community--Qwen3.5-397B-A17B-4bit.tomlM resources/inference_model_cards/mlx-community--Qwen3.5-397B-A17B-6bit.tomlM resources/inference_model_cards/mlx-community--Qwen3.5-397B-A17B-8bit.tomlM resources/inference_model_cards/mlx-community--Qwen3.5-9B-4bit.tomlM resources/inference_model_cards/mlx-community--Qwen3.5-9B-8bit.tomlM resources/inference_model_cards/mlx-community--Qwen3.6-35B-A3B-4bit.tomlM resources/inference_model_cards/mlx-community--Qwen3.6-35B-A3B-5bit.tomlM resources/inference_model_cards/mlx-community--Qwen3.6-35B-A3B-8bit.tomlM resources/inference_model_cards/mlx-community--Qwen3.6-35B-A3B-bf16.tomlM resources/inference_model_cards/mlx-community--Step-3.5-Flash-4bit.tomlM resources/inference_model_cards/mlx-community--Step-3.5-Flash-6bit.tomlM resources/inference_model_cards/mlx-community--Step-3.5-Flash-8Bit.tomlM resources/inference_model_cards/mlx-community--gemma-4-26b-a4b-it-4bit.tomlM resources/inference_model_cards/mlx-community--gemma-4-26b-a4b-it-6bit.tomlM resources/inference_model_cards/mlx-community--gemma-4-26b-a4b-it-8bit.tomlM resources/inference_model_cards/mlx-community--gemma-4-26b-a4b-it-bf16.tomlM resources/inference_model_cards/mlx-community--gemma-4-31b-it-4bit.tomlM resources/inference_model_cards/mlx-community--gemma-4-31b-it-6bit.tomlM resources/inference_model_cards/mlx-community--gemma-4-31b-it-8bit.tomlM resources/inference_model_cards/mlx-community--gemma-4-31b-it-bf16.tomlM resources/inference_model_cards/mlx-community--gemma-4-e2b-it-4bit.tomlM resources/inference_model_cards/mlx-community--gemma-4-e2b-it-6bit.tomlM resources/inference_model_cards/mlx-community--gemma-4-e2b-it-8bit.tomlM resources/inference_model_cards/mlx-community--gemma-4-e2b-it-bf16.tomlM resources/inference_model_cards/mlx-community--gemma-4-e4b-it-4bit.tomlM resources/inference_model_cards/mlx-community--gemma-4-e4b-it-6bit.tomlM resources/inference_model_cards/mlx-community--gemma-4-e4b-it-8bit.tomlM resources/inference_model_cards/mlx-community--gemma-4-e4b-it-bf16.tomlM resources/inference_model_cards/mlx-community--gpt-oss-120b-MXFP4-Q8.tomlM resources/inference_model_cards/mlx-community--gpt-oss-20b-MXFP4-Q8.tomlM resources/inference_model_cards/mlx-community--llama-3.3-70b-instruct-fp16.tomlM src/exo/api/main.pyM src/exo/shared/models/model_cards.pyM src/exo/shared/types/text_generation.pyM src/exo/worker/engines/mlx/generator/batch_generate.pyM src/exo/worker/engines/mlx/generator/generate.py
Diff
commit fcc3718efb563e82ab3a5963bc457b7f31f591e3
Author: rltakashige <rl.takashige@gmail.com>
Date: Tue Apr 21 07:45:33 2026 +0100
Add sampling defaults (#1947)
## Motivation
Model quality issues
### Manual Testing
TODO
---
.mlx_typings/mlx_lm/sample_utils.pyi | 4 +++
.../mlx-community--DeepSeek-V3.1-4bit.toml | 6 ++++
.../mlx-community--DeepSeek-V3.1-8bit.toml | 6 ++++
.../mlx-community--DeepSeek-V3.2-4bit.toml | 6 ++++
.../mlx-community--DeepSeek-V3.2-8bit.toml | 6 ++++
.../mlx-community--GLM-4.5-Air-8bit.toml | 5 +++
.../mlx-community--GLM-4.5-Air-bf16.toml | 5 +++
.../mlx-community--GLM-4.7-4bit.toml | 7 ++++
.../mlx-community--GLM-4.7-6bit.toml | 7 ++++
.../mlx-community--GLM-4.7-8bit-gs32.toml | 7 ++++
.../mlx-community--GLM-4.7-Flash-4bit.toml | 7 ++++
.../mlx-community--GLM-4.7-Flash-5bit.toml | 7 ++++
.../mlx-community--GLM-4.7-Flash-6bit.toml | 7 ++++
.../mlx-community--GLM-4.7-Flash-8bit.toml | 7 ++++
.../mlx-community--GLM-5-8bit.toml | 6 ++++
.../mlx-community--GLM-5-MXFP4-Q8.toml | 6 ++++
.../mlx-community--GLM-5-bf16.toml | 6 ++++
.../mlx-community--Kimi-K2-Instruct-4bit.toml | 5 +++
.../mlx-community--Kimi-K2-Thinking.toml | 5 +++
.../mlx-community--Kimi-K2.5.toml | 14 ++++++++
...y--Llama-3.1-Nemotron-70B-Instruct-HF-4bit.toml | 6 ++++
...y--Llama-3.1-Nemotron-70B-Instruct-HF-8bit.toml | 6 ++++
...y--Llama-3.1-Nemotron-70B-Instruct-HF-bf16.toml | 6 ++++
...nity--Llama-3.1-Nemotron-Nano-4B-v1.1-4bit.toml | 9 +++++
...nity--Llama-3.1-Nemotron-Nano-4B-v1.1-8bit.toml | 9 +++++
...nity--Llama-3.1-Nemotron-Nano-4B-v1.1-bf16.toml | 9 +++++
.../mlx-community--Llama-3.2-1B-Instruct-4bit.toml | 6 ++++
.../mlx-community--Llama-3.2-3B-Instruct-4bit.toml | 6 ++++
.../mlx-community--Llama-3.2-3B-Instruct-8bit.toml | 6 ++++
...mlx-community--Llama-3.3-70B-Instruct-4bit.toml | 6 ++++
...mlx-community--Llama-3.3-70B-Instruct-8bit.toml | 6 ++++
...ommunity--Meta-Llama-3.1-70B-Instruct-4bit.toml | 6 ++++
...community--Meta-Llama-3.1-8B-Instruct-4bit.toml | 6 ++++
...community--Meta-Llama-3.1-8B-Instruct-8bit.toml | 6 ++++
...community--Meta-Llama-3.1-8B-Instruct-bf16.toml | 6 ++++
.../mlx-community--MiniMax-M2.1-3bit.toml | 7 ++++
.../mlx-community--MiniMax-M2.1-8bit.toml | 7 ++++
.../mlx-community--MiniMax-M2.5-4bit.toml | 7 ++++
.../mlx-community--MiniMax-M2.5-6bit.toml | 7 ++++
.../mlx-community--MiniMax-M2.5-8bit.toml | 7 ++++
.../mlx-community--MiniMax-M2.7-4bit-mxfp4.toml | 8 +++++
.../mlx-community--MiniMax-M2.7-4bit.toml | 8 +++++
.../mlx-community--MiniMax-M2.7-5bit.toml | 8 +++++
.../mlx-community--MiniMax-M2.7-6bit.toml | 8 +++++
.../mlx-community--MiniMax-M2.7-8bit.toml | 8 +++++
.../mlx-community--MiniMax-M2.7.toml | 8 +++++
...y--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-4Bit.toml | 6 ++++
...y--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-5Bit.toml | 6 ++++
...y--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-6Bit.toml | 6 ++++
...y--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-8Bit.toml | 6 ++++
...y--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-BF16.toml | 6 ++++
...--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-MXFP4.toml | 6 ++++
...nity--NVIDIA-Nemotron-3-Nano-30B-A3B-NVFP4.toml | 6 ++++
...ommunity--NVIDIA-Nemotron-Nano-9B-v2-4bits.toml | 11 ++++++
...community--NVIDIA-Nemotron-Nano-9B-v2-6bit.toml | 11 ++++++
.../mlx-community--Qwen3-0.6B-4bit.toml | 14 ++++++++
.../mlx-community--Qwen3-0.6B-8bit.toml | 14 ++++++++
...munity--Qwen3-235B-A22B-Instruct-2507-4bit.toml | 7 ++++
...munity--Qwen3-235B-A22B-Instruct-2507-8bit.toml | 7 ++++
.../mlx-community--Qwen3-30B-A3B-4bit.toml | 14 ++++++++
.../mlx-community--Qwen3-30B-A3B-8bit.toml | 14 ++++++++
...unity--Qwen3-Coder-480B-A35B-Instruct-4bit.toml | 8 +++++
...unity--Qwen3-Coder-480B-A35B-Instruct-8bit.toml | 8 +++++
.../mlx-community--Qwen3-Coder-Next-4bit.toml | 7 ++++
.../mlx-community--Qwen3-Coder-Next-5bit.toml | 7 ++++
.../mlx-community--Qwen3-Coder-Next-6bit.toml | 7 ++++
.../mlx-community--Qwen3-Coder-Next-8bit.toml | 7 ++++
.../mlx-community--Qwen3-Coder-Next-bf16.toml | 7 ++++
...ommunity--Qwen3-Next-80B-A3B-Instruct-4bit.toml | 7 ++++
...ommunity--Qwen3-Next-80B-A3B-Instruct-8bit.toml | 7 ++++
...ommunity--Qwen3-Next-80B-A3B-Thinking-4bit.toml | 7 ++++
...ommunity--Qwen3-Next-80B-A3B-Thinking-8bit.toml | 7 ++++
.../mlx-community--Qwen3-VL-4B-Instruct-4bit.toml | 9 +++++
.../mlx-community--Qwen3.5-122B-A10B-4bit.toml | 20 +++++++++++
.../mlx-community--Qwen3.5-122B-A10B-6bit.toml | 20 +++++++++++
.../mlx-community--Qwen3.5-122B-A10B-8bit.toml | 20 +++++++++++
.../mlx-community--Qwen3.5-122B-A10B-bf16.toml | 20 +++++++++++
.../mlx-community--Qwen3.5-27B-4bit.toml | 20 +++++++++++
.../mlx-community--Qwen3.5-27B-8bit.toml | 20 +++++++++++
.../mlx-community--Qwen3.5-2B-MLX-8bit.toml | 20 +++++++++++
.../mlx-community--Qwen3.5-35B-A3B-4bit.toml | 20 +++++++++++
.../mlx-community--Qwen3.5-35B-A3B-8bit.toml | 20 +++++++++++
.../mlx-community--Qwen3.5-397B-A17B-4bit.toml | 20 +++++++++++
.../mlx-community--Qwen3.5-397B-A17B-6bit.toml | 20 +++++++++++
.../mlx-community--Qwen3.5-397B-A17B-8bit.toml | 20 +++++++++++
.../mlx-community--Qwen3.5-9B-4bit.toml | 20 +++++++++++
.../mlx-community--Qwen3.5-9B-8bit.toml | 20 +++++++++++
.../mlx-community--Qwen3.6-35B-A3B-4bit.toml | 20 +++++++++++
.../mlx-community--Qwen3.6-35B-A3B-5bit.toml | 20 +++++++++++
.../mlx-community--Qwen3.6-35B-A3B-8bit.toml | 20 +++++++++++
.../mlx-community--Qwen3.6-35B-A3B-bf16.toml | 20 +++++++++++
.../mlx-community--Step-3.5-Flash-4bit.toml | 12 +++++++
.../mlx-community--Step-3.5-Flash-6bit.toml | 12 +++++++
.../mlx-community--Step-3.5-Flash-8Bit.toml | 12 +++++++
.../mlx-community--gemma-4-26b-a4b-it-4bit.toml | 7 ++++
.../mlx-community--gemma-4-26b-a4b-it-6bit.toml | 7 ++++
.../mlx-community--gemma-4-26b-a4b-it-8bit.toml | 7 ++++
.../mlx-community--gemma-4-26b-a4b-it-bf16.toml | 7 ++++
.../mlx-community--gemma-4-31b-it-4bit.toml | 7 ++++
.../mlx-community--gemma-4-31b-it-6bit.toml | 7 ++++
.../mlx-community--gemma-4-31b-it-8bit.toml | 7 ++++
.../mlx-community--gemma-4-31b-it-bf16.toml | 7 ++++
.../mlx-community--gemma-4-e2b-it-4bit.toml | 7 ++++
.../mlx-community--gemma-4-e2b-it-6bit.toml | 7 ++++
.../mlx-community--gemma-4-e2b-it-8bit.toml | 7 ++++
.../mlx-community--gemma-4-e2b-it-bf16.toml | 7 ++++
.../mlx-community--gemma-4-e4b-it-4bit.toml | 7 ++++
.../mlx-community--gemma-4-e4b-it-6bit.toml | 7 ++++
.../mlx-community--gemma-4-e4b-it-8bit.toml | 7 ++++
.../mlx-community--gemma-4-e4b-it-bf16.toml | 7 ++++
.../mlx-community--gpt-oss-120b-MXFP4-Q8.toml | 7 ++++
.../mlx-community--gpt-oss-20b-MXFP4-Q8.toml | 7 ++++
...mlx-community--llama-3.3-70b-instruct-fp16.toml | 6 ++++
src/exo/api/main.py | 1 +
src/exo/shared/models/model_cards.py | 16 +++++++++
src/exo/shared/types/text_generation.py | 39 ++++++++++++++++++++++
.../worker/engines/mlx/generator/batch_generate.py | 6 +++-
src/exo/worker/engines/mlx/generator/generate.py | 6 +++-
118 files changed, 1127 insertions(+), 2 deletions(-)
diff --git a/.mlx_typings/mlx_lm/sample_utils.pyi b/.mlx_typings/mlx_lm/sample_utils.pyi
index c001e3ec..b9f91a0c 100644
--- a/.mlx_typings/mlx_lm/sample_utils.pyi
+++ b/.mlx_typings/mlx_lm/sample_utils.pyi
@@ -48,6 +48,10 @@ def make_logits_processors(
logit_bias: Optional[Dict[int, float]] = ...,
repetition_penalty: Optional[float] = ...,
repetition_context_size: Optional[int] = ...,
+ presence_penalty: Optional[float] = ...,
+ presence_context_size: Optional[int] = ...,
+ frequency_penalty: Optional[float] = ...,
+ frequency_context_size: Optional[int] = ...,
) -> list[Callable[[mx.array, mx.array], mx.array]]:
"""
Make logits processors for use with ``generate_step``.
diff --git a/resources/inference_model_cards/mlx-community--DeepSeek-V3.1-4bit.toml b/resources/inference_model_cards/mlx-community--DeepSeek-V3.1-4bit.toml
index 01166482..5fbea636 100644
--- a/resources/inference_model_cards/mlx-community--DeepSeek-V3.1-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--DeepSeek-V3.1-4bit.toml
@@ -13,3 +13,9 @@ context_length = 131072
[storage_size]
in_bytes = 405874409472
+
+# Source: https://huggingface.co/deepseek-ai/DeepSeek-V3.1/blob/main/generation_config.json
+# Source: https://huggingface.co/deepseek-ai/DeepSeek-V3.1/discussions/19
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
diff --git a/resources/inference_model_cards/mlx-community--DeepSeek-V3.1-8bit.toml b/resources/inference_model_cards/mlx-community--DeepSeek-V3.1-8bit.toml
index 62313d3a..3ebc0cee 100644
--- a/resources/inference_model_cards/mlx-community--DeepSeek-V3.1-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--DeepSeek-V3.1-8bit.toml
@@ -13,3 +13,9 @@ context_length = 131072
[storage_size]
in_bytes = 765577920512
+
+# Source: https://huggingface.co/deepseek-ai/DeepSeek-V3.1/blob/main/generation_config.json
+# Source: https://huggingface.co/deepseek-ai/DeepSeek-V3.1/discussions/19
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
diff --git a/resources/inference_model_cards/mlx-community--DeepSeek-V3.2-4bit.toml b/resources/inference_model_cards/mlx-community--DeepSeek-V3.2-4bit.toml
index 5288181e..025a53df 100644
--- a/resources/inference_model_cards/mlx-community--DeepSeek-V3.2-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--DeepSeek-V3.2-4bit.toml
@@ -13,3 +13,9 @@ context_length = 131072
[storage_size]
in_bytes = 378086226621
+
+# Source: https://huggingface.co/deepseek-ai/DeepSeek-V3.2/blob/main/generation_config.json
+# Source: https://docs.vllm.ai/projects/recipes/en/latest/DeepSeek/DeepSeek-V3_2.html
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
diff --git a/resources/inference_model_cards/mlx-community--DeepSeek-V3.2-8bit.toml b/resources/inference_model_cards/mlx-community--DeepSeek-V3.2-8bit.toml
index 87f81738..a9b064b3 100644
--- a/resources/inference_model_cards/mlx-community--DeepSeek-V3.2-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--DeepSeek-V3.2-8bit.toml
@@ -13,3 +13,9 @@ context_length = 131072
[storage_size]
in_bytes = 755957120916
+
+# Source: https://huggingface.co/deepseek-ai/DeepSeek-V3.2/blob/main/generation_config.json
+# Source: https://docs.vllm.ai/projects/recipes/en/latest/DeepSeek/DeepSeek-V3_2.html
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
diff --git a/resources/inference_model_cards/mlx-community--GLM-4.5-Air-8bit.toml b/resources/inference_model_cards/mlx-community--GLM-4.5-Air-8bit.toml
index 644fe425..35a4529f 100644
--- a/resources/inference_model_cards/mlx-community--GLM-4.5-Air-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--GLM-4.5-Air-8bit.toml
@@ -13,3 +13,8 @@ context_length = 131072
[storage_size]
in_bytes = 122406567936
+
+# Source: https://docs.z.ai/api-reference/llm/chat-completion
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
diff --git a/resources/inference_model_cards/mlx-community--GLM-4.5-Air-bf16.toml b/resources/inference_model_cards/mlx-community--GLM-4.5-Air-bf16.toml
index a7b5c37a..499754f3 100644
--- a/resources/inference_model_cards/mlx-community--GLM-4.5-Air-bf16.toml
+++ b/resources/inference_model_cards/mlx-community--GLM-4.5-Air-bf16.toml
@@ -13,3 +13,8 @@ context_length = 131072
[storage_size]
in_bytes = 229780750336
+
+# Source: https://docs.z.ai/api-reference/llm/chat-completion
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
diff --git a/resources/inference_model_cards/mlx-community--GLM-4.7-4bit.toml b/resources/inference_model_cards/mlx-community--GLM-4.7-4bit.toml
index c8b32a38..88b5b001 100644
--- a/resources/inference_model_cards/mlx-community--GLM-4.7-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--GLM-4.7-4bit.toml
@@ -13,3 +13,10 @@ context_length = 202752
[storage_size]
in_bytes = 198556925568
+
+# Source: https://huggingface.co/zai-org/GLM-4.7
+# Source: https://unsloth.ai/docs/models/glm-4.7-flash
+# Source: https://docs.z.ai/api-reference/llm/chat-completion
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
diff --git a/resources/inference_model_cards/mlx-community--GLM-4.7-6bit.toml b/resources/inference_model_cards/mlx-community--GLM-4.7-6bit.toml
index 7eb301e5..3296b8eb 100644
--- a/resources/inference_model_cards/mlx-community--GLM-4.7-6bit.toml
+++ b/resources/inference_model_cards/mlx-community--GLM-4.7-6bit.toml
@@ -13,3 +13,10 @@ context_length = 202752
[storage_size]
in_bytes = 286737579648
+
+# Source: https://huggingface.co/zai-org/GLM-4.7
+# Source: https://unsloth.ai/docs/models/glm-4.7-flash
+# Source: https://docs.z.ai/api-reference/llm/chat-completion
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
diff --git a/resources/inference_model_cards/mlx-community--GLM-4.7-8bit-gs32.toml b/resources/inference_model_cards/mlx-community--GLM-4.7-8bit-gs32.toml
index 91365de8..705b7871 100644
--- a/resources/inference_model_cards/mlx-community--GLM-4.7-8bit-gs32.toml
+++ b/resources/inference_model_cards/mlx-community--GLM-4.7-8bit-gs32.toml
@@ -13,3 +13,10 @@ context_length = 202752
[storage_size]
in_bytes = 396963397248
+
+# Source: https://huggingface.co/zai-org/GLM-4.7
+# Source: https://unsloth.ai/docs/models/glm-4.7-flash
+# Source: https://docs.z.ai/api-reference/llm/chat-completion
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
diff --git a/resources/inference_model_cards/mlx-community--GLM-4.7-Flash-4bit.toml b/resources/inference_model_cards/mlx-community--GLM-4.7-Flash-4bit.toml
index d0be6d82..fb203c7a 100644
--- a/resources/inference_model_cards/mlx-community--GLM-4.7-Flash-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--GLM-4.7-Flash-4bit.toml
@@ -13,3 +13,10 @@ context_length = 202752
[storage_size]
in_bytes = 19327352832
+
+# Source: https://huggingface.co/zai-org/GLM-4.7-Flash
+# Source: https://unsloth.ai/docs/models/glm-4.7-flash
+# Source: https://docs.z.ai/api-reference/llm/chat-completion
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
diff --git a/resources/inference_model_cards/mlx-community--GLM-4.7-Flash-5bit.toml b/resources/inference_model_cards/mlx-community--GLM-4.7-Flash-5bit.toml
index 9c7136db..d6dbfa16 100644
--- a/resources/inference_model_cards/mlx-community--GLM-4.7-Flash-5bit.toml
+++ b/resources/inference_model_cards/mlx-community--GLM-4.7-Flash-5bit.toml
@@ -13,3 +13,10 @@ context_length = 202752
[storage_size]
in_bytes = 22548578304
+
+# Source: https://huggingface.co/zai-org/GLM-4.7-Flash
+# Source: https://unsloth.ai/docs/models/glm-4.7-flash
+# Source: https://docs.z.ai/api-reference/llm/chat-completion
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
diff --git a/resources/inference_model_cards/mlx-community--GLM-4.7-Flash-6bit.toml b/resources/inference_model_cards/mlx-community--GLM-4.7-Flash-6bit.toml
index cf2ed455..153e7f12 100644
--- a/resources/inference_model_cards/mlx-community--GLM-4.7-Flash-6bit.toml
+++ b/resources/inference_model_cards/mlx-community--GLM-4.7-Flash-6bit.toml
@@ -13,3 +13,10 @@ context_length = 202752
[storage_size]
in_bytes = 26843545600
+
+# Source: https://huggingface.co/zai-org/GLM-4.7-Flash
+# Source: https://unsloth.ai/docs/models/glm-4.7-flash
+# Source: https://docs.z.ai/api-reference/llm/chat-completion
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
diff --git a/resources/inference_model_cards/mlx-community--GLM-4.7-Flash-8bit.toml b/resources/inference_model_cards/mlx-community--GLM-4.7-Flash-8bit.toml
index 879b322c..2eef1c3f 100644
--- a/resources/inference_model_cards/mlx-community--GLM-4.7-Flash-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--GLM-4.7-Flash-8bit.toml
@@ -13,3 +13,10 @@ context_length = 202752
[storage_size]
in_bytes = 34359738368
+
+# Source: https://huggingface.co/zai-org/GLM-4.7-Flash
+# Source: https://unsloth.ai/docs/models/glm-4.7-flash
+# Source: https://docs.z.ai/api-reference/llm/chat-completion
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
diff --git a/resources/inference_model_cards/mlx-community--GLM-5-8bit.toml b/resources/inference_model_cards/mlx-community--GLM-5-8bit.toml
index 37b0fb2a..222fbd7b 100644
--- a/resources/inference_model_cards/mlx-community--GLM-5-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--GLM-5-8bit.toml
@@ -13,3 +13,9 @@ context_length = 202752
[storage_size]
in_bytes = 790517400864
+
+# Source: https://huggingface.co/zai-org/GLM-5
+# Source: https://docs.z.ai/api-reference/llm/chat-completion
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
diff --git a/resources/inference_model_cards/mlx-community--GLM-5-MXFP4-Q8.toml b/resources/inference_model_cards/mlx-community--GLM-5-MXFP4-Q8.toml
index 2874e4e6..fc4100a3 100644
--- a/resources/inference_model_cards/mlx-community--GLM-5-MXFP4-Q8.toml
+++ b/resources/inference_model_cards/mlx-community--GLM-5-MXFP4-Q8.toml
@@ -13,3 +13,9 @@ context_length = 202752
[storage_size]
in_bytes = 405478939008
+
+# Source: https://huggingface.co/zai-org/GLM-5
+# Source: https://docs.z.ai/api-reference/llm/chat-completion
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
diff --git a/resources/inference_model_cards/mlx-community--GLM-5-bf16.toml b/resources/inference_model_cards/mlx-community--GLM-5-bf16.toml
index 1086d86a..9b9549e8 100644
--- a/resources/inference_model_cards/mlx-community--GLM-5-bf16.toml
+++ b/resources/inference_model_cards/mlx-community--GLM-5-bf16.toml
@@ -13,3 +13,9 @@ context_length = 202752
[storage_size]
in_bytes = 1487822475264
+
+# Source: https://huggingface.co/zai-org/GLM-5
+# Source: https://docs.z.ai/api-reference/llm/chat-completion
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
diff --git a/resources/inference_model_cards/mlx-community--Kimi-K2-Instruct-4bit.toml b/resources/inference_model_cards/mlx-community--Kimi-K2-Instruct-4bit.toml
index 878aa2aa..23b2a572 100644
--- a/resources/inference_model_cards/mlx-community--Kimi-K2-Instruct-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Kimi-K2-Instruct-4bit.toml
@@ -13,3 +13,8 @@ context_length = 131072
[storage_size]
in_bytes = 620622774272
+
+# Source: https://huggingface.co/moonshotai/Kimi-K2-Instruct
+# Source: https://platform.kimi.ai/docs/guide/kimi-k2-quickstart
+[sampling_defaults]
+temperature = 0.6
diff --git a/resources/inference_model_cards/mlx-community--Kimi-K2-Thinking.toml b/resources/inference_model_cards/mlx-community--Kimi-K2-Thinking.toml
index 37f4befc..2c314e8e 100644
--- a/resources/inference_model_cards/mlx-community--Kimi-K2-Thinking.toml
+++ b/resources/inference_model_cards/mlx-community--Kimi-K2-Thinking.toml
@@ -13,3 +13,8 @@ context_length = 262144
[storage_size]
in_bytes = 706522120192
+
+# Source: https://huggingface.co/moonshotai/Kimi-K2-Thinking
+# Source: https://platform.kimi.ai/docs/guide/use-kimi-k2-thinking-model
+[sampling_defaults]
+temperature = 1.0
diff --git a/resources/inference_model_cards/mlx-community--Kimi-K2.5.toml b/resources/inference_model_cards/mlx-community--Kimi-K2.5.toml
index efe7d4bb..1008b389 100644
--- a/resources/inference_model_cards/mlx-community--Kimi-K2.5.toml
+++ b/resources/inference_model_cards/mlx-community--Kimi-K2.5.toml
@@ -19,3 +19,17 @@ image_token_id = 163605
model_type = "kimi_vl"
weights_repo = "davehind/Kimi-K2.5-vision"
processor_repo = "moonshotai/Kimi-K2.5"
+
+# Source: https://deepwiki.com/MoonshotAI/Kimi-K2.5/3.7-recommended-parameters
+# Source: https://unsloth.ai/docs/models/kimi-k2.5
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+min_p = 0.01
+
+# Source: https://deepwiki.com/MoonshotAI/Kimi-K2.5/3.7-recommended-parameters
+# Source: https://unsloth.ai/docs/models/kimi-k2.5
+[sampling_defaults.non_thinking]
+temperature = 0.6
+top_p = 0.95
+min_p = 0.01
diff --git a/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-70B-Instruct-HF-4bit.toml b/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-70B-Instruct-HF-4bit.toml
index d98cabdd..f8663f1b 100644
--- a/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-70B-Instruct-HF-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-70B-Instruct-HF-4bit.toml
@@ -12,3 +12,9 @@ context_length = 131072
[storage_size]
in_bytes = 39688355840
+
+# Source: https://huggingface.co/RedHatAI/Llama-3.1-Nemotron-70B-Instruct-HF-FP8-dynamic
+# Source: https://deepinfra.com/nvidia/Llama-3.1-Nemotron-70B-Instruct/api
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.9
diff --git a/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-70B-Instruct-HF-8bit.toml b/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-70B-Instruct-HF-8bit.toml
index 4f4297ab..71a75024 100644
--- a/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-70B-Instruct-HF-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-70B-Instruct-HF-8bit.toml
@@ -12,3 +12,9 @@ context_length = 131072
[storage_size]
in_bytes = 74964549632
+
+# Source: https://huggingface.co/RedHatAI/Llama-3.1-Nemotron-70B-Instruct-HF-FP8-dynamic
+# Source: https://deepinfra.com/nvidia/Llama-3.1-Nemotron-70B-Instruct/api
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.9
diff --git a/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-70B-Instruct-HF-bf16.toml b/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-70B-Instruct-HF-bf16.toml
index 4ef63c17..03185ccc 100644
--- a/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-70B-Instruct-HF-bf16.toml
+++ b/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-70B-Instruct-HF-bf16.toml
@@ -12,3 +12,9 @@ context_length = 131072
[storage_size]
in_bytes = 141107412992
+
+# Source: https://huggingface.co/RedHatAI/Llama-3.1-Nemotron-70B-Instruct-HF-FP8-dynamic
+# Source: https://deepinfra.com/nvidia/Llama-3.1-Nemotron-70B-Instruct/api
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.9
diff --git a/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-Nano-4B-v1.1-4bit.toml b/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-Nano-4B-v1.1-4bit.toml
index 687c5664..8c409ed5 100644
--- a/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-Nano-4B-v1.1-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-Nano-4B-v1.1-4bit.toml
@@ -12,3 +12,12 @@ context_length = 131072
[storage_size]
in_bytes = 2538706944
+
+# Source: https://huggingface.co/nvidia/Llama-3.1-Nemotron-Nano-4B-v1.1
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+
+# Source: https://huggingface.co/nvidia/Llama-3.1-Nemotron-Nano-4B-v1.1
+[sampling_defaults.non_thinking]
+temperature = 0.0
diff --git a/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-Nano-4B-v1.1-8bit.toml b/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-Nano-4B-v1.1-8bit.toml
index dae6fd93..aa0aaffc 100644
--- a/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-Nano-4B-v1.1-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-Nano-4B-v1.1-8bit.toml
@@ -12,3 +12,12 @@ context_length = 131072
[storage_size]
in_bytes = 4794980352
+
+# Source: https://huggingface.co/nvidia/Llama-3.1-Nemotron-Nano-4B-v1.1
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+
+# Source: https://huggingface.co/nvidia/Llama-3.1-Nemotron-Nano-4B-v1.1
+[sampling_defaults.non_thinking]
+temperature = 0.0
diff --git a/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-Nano-4B-v1.1-bf16.toml b/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-Nano-4B-v1.1-bf16.toml
index 0e261646..2266392e 100644
--- a/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-Nano-4B-v1.1-bf16.toml
+++ b/resources/inference_model_cards/mlx-community--Llama-3.1-Nemotron-Nano-4B-v1.1-bf16.toml
@@ -12,3 +12,12 @@ context_length = 131072
[storage_size]
in_bytes = 9025492992
+
+# Source: https://huggingface.co/nvidia/Llama-3.1-Nemotron-Nano-4B-v1.1
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+
+# Source: https://huggingface.co/nvidia/Llama-3.1-Nemotron-Nano-4B-v1.1
+[sampling_defaults.non_thinking]
+temperature = 0.0
diff --git a/resources/inference_model_cards/mlx-community--Llama-3.2-1B-Instruct-4bit.toml b/resources/inference_model_cards/mlx-community--Llama-3.2-1B-Instruct-4bit.toml
index fa30edc3..c10c53f1 100644
--- a/resources/inference_model_cards/mlx-community--Llama-3.2-1B-Instruct-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Llama-3.2-1B-Instruct-4bit.toml
@@ -13,3 +13,9 @@ context_length = 131072
[storage_size]
in_bytes = 729808896
+
+# Source: https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/blob/main/generation_config.json
+# Source: https://huggingface.co/unsloth/Llama-3.2-1B-Instruct/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.9
diff --git a/resources/inference_model_cards/mlx-community--Llama-3.2-3B-Instruct-4bit.toml b/resources/inference_model_cards/mlx-community--Llama-3.2-3B-Instruct-4bit.toml
index 6255db7b..3ceeee79 100644
--- a/resources/inference_model_cards/mlx-community--Llama-3.2-3B-Instruct-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Llama-3.2-3B-Instruct-4bit.toml
@@ -13,3 +13,9 @@ context_length = 131072
[storage_size]
in_bytes = 1863319552
+
+# Source: https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct/blob/main/generation_config.json
+# Source: https://huggingface.co/unsloth/Llama-3.2-3B-Instruct/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.9
diff --git a/resources/inference_model_cards/mlx-community--Llama-3.2-3B-Instruct-8bit.toml b/resources/inference_model_cards/mlx-community--Llama-3.2-3B-Instruct-8bit.toml
index e2de35ec..66c88e2a 100644
--- a/resources/inference_model_cards/mlx-community--Llama-3.2-3B-Instruct-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--Llama-3.2-3B-Instruct-8bit.toml
@@ -13,3 +13,9 @@ context_length = 131072
[storage_size]
in_bytes = 3501195264
+
+# Source: https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct/blob/main/generation_config.json
+# Source: https://huggingface.co/unsloth/Llama-3.2-3B-Instruct/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.9
diff --git a/resources/inference_model_cards/mlx-community--Llama-3.3-70B-Instruct-4bit.toml b/resources/inference_model_cards/mlx-community--Llama-3.3-70B-Instruct-4bit.toml
index ef81e828..73b4e32a 100644
--- a/resources/inference_model_cards/mlx-community--Llama-3.3-70B-Instruct-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Llama-3.3-70B-Instruct-4bit.toml
@@ -13,3 +13,9 @@ context_length = 131072
[storage_size]
in_bytes = 40652242944
+
+# Source: https://huggingface.co/meta-llama/Llama-3.3-70B-Instruct/blob/main/generation_config.json
+# Source: https://huggingface.co/unsloth/Llama-3.3-70B-Instruct/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.9
diff --git a/resources/inference_model_cards/mlx-community--Llama-3.3-70B-Instruct-8bit.toml b/resources/inference_model_cards/mlx-community--Llama-3.3-70B-Instruct-8bit.toml
index fb83f3d0..285156d7 100644
--- a/resources/inference_model_cards/mlx-community--Llama-3.3-70B-Instruct-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--Llama-3.3-70B-Instruct-8bit.toml
@@ -13,3 +13,9 @@ context_length = 131072
[storage_size]
in_bytes = 76799803392
+
+# Source: https://huggingface.co/meta-llama/Llama-3.3-70B-Instruct/blob/main/generation_config.json
+# Source: https://huggingface.co/unsloth/Llama-3.3-70B-Instruct/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.9
diff --git a/resources/inference_model_cards/mlx-community--Meta-Llama-3.1-70B-Instruct-4bit.toml b/resources/inference_model_cards/mlx-community--Meta-Llama-3.1-70B-Instruct-4bit.toml
index 89313087..214a5fa9 100644
--- a/resources/inference_model_cards/mlx-community--Meta-Llama-3.1-70B-Instruct-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Meta-Llama-3.1-70B-Instruct-4bit.toml
@@ -13,3 +13,9 @@ context_length = 131072
[storage_size]
in_bytes = 40652242944
+
+# Source: https://huggingface.co/meta-llama/Meta-Llama-3.1-70B-Instruct/blob/main/generation_config.json
+# Source: https://huggingface.co/unsloth/Meta-Llama-3.1-70B-Instruct/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.9
diff --git a/resources/inference_model_cards/mlx-community--Meta-Llama-3.1-8B-Instruct-4bit.toml b/resources/inference_model_cards/mlx-community--Meta-Llama-3.1-8B-Instruct-4bit.toml
index a9ab123b..551c7052 100644
--- a/resources/inference_model_cards/mlx-community--Meta-Llama-3.1-8B-Instruct-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Meta-Llama-3.1-8B-Instruct-4bit.toml
@@ -13,3 +13,9 @@ context_length = 131072
[storage_size]
in_bytes = 4637851648
+
+# Source: https://huggingface.co/meta-llama/Meta-Llama-3.1-8B-Instruct/blob/main/generation_config.json
+# Source: https://huggingface.co/unsloth/Meta-Llama-3.1-8B-Instruct/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.9
diff --git a/resources/inference_model_cards/mlx-community--Meta-Llama-3.1-8B-Instruct-8bit.toml b/resources/inference_model_cards/mlx-community--Meta-Llama-3.1-8B-Instruct-8bit.toml
index 84cdbaf1..7ac95f17 100644
--- a/resources/inference_model_cards/mlx-community--Meta-Llama-3.1-8B-Instruct-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--Meta-Llama-3.1-8B-Instruct-8bit.toml
@@ -13,3 +13,9 @@ context_length = 131072
[storage_size]
in_bytes = 8954839040
+
+# Source: https://huggingface.co/meta-llama/Meta-Llama-3.1-8B-Instruct/blob/main/generation_config.json
+# Source: https://huggingface.co/unsloth/Meta-Llama-3.1-8B-Instruct/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.9
diff --git a/resources/inference_model_cards/mlx-community--Meta-Llama-3.1-8B-Instruct-bf16.toml b/resources/inference_model_cards/mlx-community--Meta-Llama-3.1-8B-Instruct-bf16.toml
index f2121d64..cea334f8 100644
--- a/resources/inference_model_cards/mlx-community--Meta-Llama-3.1-8B-Instruct-bf16.toml
+++ b/resources/inference_model_cards/mlx-community--Meta-Llama-3.1-8B-Instruct-bf16.toml
@@ -13,3 +13,9 @@ context_length = 131072
[storage_size]
in_bytes = 16882073600
+
+# Source: https://huggingface.co/meta-llama/Meta-Llama-3.1-8B-Instruct/blob/main/generation_config.json
+# Source: https://huggingface.co/unsloth/Meta-Llama-3.1-8B-Instruct/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.9
diff --git a/resources/inference_model_cards/mlx-community--MiniMax-M2.1-3bit.toml b/resources/inference_model_cards/mlx-community--MiniMax-M2.1-3bit.toml
index 0b22f011..268e1da6 100644
--- a/resources/inference_model_cards/mlx-community--MiniMax-M2.1-3bit.toml
+++ b/resources/inference_model_cards/mlx-community--MiniMax-M2.1-3bit.toml
@@ -13,3 +13,10 @@ context_length = 196608
[storage_size]
in_bytes = 100086644736
+
+# Source: https://huggingface.co/MiniMaxAI/MiniMax-M2.1
+# Source: https://github.com/MiniMax-AI/MiniMax-M2.1
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 40
diff --git a/resources/inference_model_cards/mlx-community--MiniMax-M2.1-8bit.toml b/resources/inference_model_cards/mlx-community--MiniMax-M2.1-8bit.toml
index 32e5bc3f..96ad2c18 100644
--- a/resources/inference_model_cards/mlx-community--MiniMax-M2.1-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--MiniMax-M2.1-8bit.toml
@@ -13,3 +13,10 @@ context_length = 196608
[storage_size]
in_bytes = 242986745856
+
+# Source: https://huggingface.co/MiniMaxAI/MiniMax-M2.1
+# Source: https://github.com/MiniMax-AI/MiniMax-M2.1
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 40
diff --git a/resources/inference_model_cards/mlx-community--MiniMax-M2.5-4bit.toml b/resources/inference_model_cards/mlx-community--MiniMax-M2.5-4bit.toml
index c6418512..08648ad7 100644
--- a/resources/inference_model_cards/mlx-community--MiniMax-M2.5-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--MiniMax-M2.5-4bit.toml
@@ -13,3 +13,10 @@ context_length = 196608
[storage_size]
in_bytes = 128666664960
+
+# Source: https://huggingface.co/MiniMaxAI/MiniMax-M2.5
+# Source: https://github.com/MiniMax-AI/MiniMax-M2.5
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 40
diff --git a/resources/inference_model_cards/mlx-community--MiniMax-M2.5-6bit.toml b/resources/inference_model_cards/mlx-community--MiniMax-M2.5-6bit.toml
index 9c366e6e..6b5e2b74 100644
--- a/resources/inference_model_cards/mlx-community--MiniMax-M2.5-6bit.toml
+++ b/resources/inference_model_cards/mlx-community--MiniMax-M2.5-6bit.toml
@@ -13,3 +13,10 @@ context_length = 196608
[storage_size]
in_bytes = 185826705408
+
+# Source: https://huggingface.co/MiniMaxAI/MiniMax-M2.5
+# Source: https://github.com/MiniMax-AI/MiniMax-M2.5
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 40
diff --git a/resources/inference_model_cards/mlx-community--MiniMax-M2.5-8bit.toml b/resources/inference_model_cards/mlx-community--MiniMax-M2.5-8bit.toml
index c6946fa4..7fb1204d 100644
--- a/resources/inference_model_cards/mlx-community--MiniMax-M2.5-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--MiniMax-M2.5-8bit.toml
@@ -13,3 +13,10 @@ context_length = 196608
[storage_size]
in_bytes = 242986745856
+
+# Source: https://huggingface.co/MiniMaxAI/MiniMax-M2.5
+# Source: https://github.com/MiniMax-AI/MiniMax-M2.5
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 40
diff --git a/resources/inference_model_cards/mlx-community--MiniMax-M2.7-4bit-mxfp4.toml b/resources/inference_model_cards/mlx-community--MiniMax-M2.7-4bit-mxfp4.toml
index 7e6ffffd..bc7281f6 100644
--- a/resources/inference_model_cards/mlx-community--MiniMax-M2.7-4bit-mxfp4.toml
+++ b/resources/inference_model_cards/mlx-community--MiniMax-M2.7-4bit-mxfp4.toml
@@ -13,3 +13,11 @@ context_length = 196608
[storage_size]
in_bytes = 121537496794
+
+# Source: https://huggingface.co/MiniMaxAI/MiniMax-M2.7
+# Source: https://github.com/MiniMax-AI/MiniMax-M2.7
+# Source: https://unsloth.ai/docs/models/minimax-m27
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 40
diff --git a/resources/inference_model_cards/mlx-community--MiniMax-M2.7-4bit.toml b/resources/inference_model_cards/mlx-community--MiniMax-M2.7-4bit.toml
index fc86e18e..44c9d0d2 100644
--- a/resources/inference_model_cards/mlx-community--MiniMax-M2.7-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--MiniMax-M2.7-4bit.toml
@@ -13,3 +13,11 @@ context_length = 196608
[storage_size]
in_bytes = 128682598717
+
+# Source: https://huggingface.co/MiniMaxAI/MiniMax-M2.7
+# Source: https://github.com/MiniMax-AI/MiniMax-M2.7
+# Source: https://unsloth.ai/docs/models/minimax-m27
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 40
diff --git a/resources/inference_model_cards/mlx-community--MiniMax-M2.7-5bit.toml b/resources/inference_model_cards/mlx-community--MiniMax-M2.7-5bit.toml
index 3702afdc..a39096ce 100644
--- a/resources/inference_model_cards/mlx-community--MiniMax-M2.7-5bit.toml
+++ b/resources/inference_model_cards/mlx-community--MiniMax-M2.7-5bit.toml
@@ -13,3 +13,11 @@ context_length = 196608
[storage_size]
in_bytes = 157262619651
+
+# Source: https://huggingface.co/MiniMaxAI/MiniMax-M2.7
+# Source: https://github.com/MiniMax-AI/MiniMax-M2.7
+# Source: https://unsloth.ai/docs/models/minimax-m27
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 40
diff --git a/resources/inference_model_cards/mlx-community--MiniMax-M2.7-6bit.toml b/resources/inference_model_cards/mlx-community--MiniMax-M2.7-6bit.toml
index 6213e77d..7026062c 100644
--- a/resources/inference_model_cards/mlx-community--MiniMax-M2.7-6bit.toml
+++ b/resources/inference_model_cards/mlx-community--MiniMax-M2.7-6bit.toml
@@ -13,3 +13,11 @@ context_length = 196608
[storage_size]
in_bytes = 185842639299
+
+# Source: https://huggingface.co/MiniMaxAI/MiniMax-M2.7
+# Source: https://github.com/MiniMax-AI/MiniMax-M2.7
+# Source: https://unsloth.ai/docs/models/minimax-m27
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 40
diff --git a/resources/inference_model_cards/mlx-community--MiniMax-M2.7-8bit.toml b/resources/inference_model_cards/mlx-community--MiniMax-M2.7-8bit.toml
index c964a587..0067fa4e 100644
--- a/resources/inference_model_cards/mlx-community--MiniMax-M2.7-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--MiniMax-M2.7-8bit.toml
@@ -13,3 +13,11 @@ context_length = 196608
[storage_size]
in_bytes = 243002680786
+
+# Source: https://huggingface.co/MiniMaxAI/MiniMax-M2.7
+# Source: https://github.com/MiniMax-AI/MiniMax-M2.7
+# Source: https://unsloth.ai/docs/models/minimax-m27
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 40
diff --git a/resources/inference_model_cards/mlx-community--MiniMax-M2.7.toml b/resources/inference_model_cards/mlx-community--MiniMax-M2.7.toml
index 4d3645c6..3262d55d 100644
--- a/resources/inference_model_cards/mlx-community--MiniMax-M2.7.toml
+++ b/resources/inference_model_cards/mlx-community--MiniMax-M2.7.toml
@@ -13,3 +13,11 @@ context_length = 196608
[storage_size]
in_bytes = 457492783366
+
+# Source: https://huggingface.co/MiniMaxAI/MiniMax-M2.7
+# Source: https://github.com/MiniMax-AI/MiniMax-M2.7
+# Source: https://unsloth.ai/docs/models/minimax-m27
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 40
diff --git a/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-4Bit.toml b/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-4Bit.toml
index 7b79a205..fb032101 100644
--- a/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-4Bit.toml
+++ b/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-4Bit.toml
@@ -12,3 +12,9 @@ context_length = 262144
[storage_size]
in_bytes = 17775342336
+
+# Source: https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B
+# Source: https://unsloth.ai/docs/models/nemotron-3
+[sampling_defaults]
+temperature = 1.0
+top_p = 1.0
diff --git a/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-5Bit.toml b/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-5Bit.toml
index 8a4c7384..3b2b15c0 100644
--- a/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-5Bit.toml
+++ b/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-5Bit.toml
@@ -12,3 +12,9 @@ context_length = 262144
[storage_size]
in_bytes = 21721476864
+
+# Source: https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B
+# Source: https://unsloth.ai/docs/models/nemotron-3
+[sampling_defaults]
+temperature = 1.0
+top_p = 1.0
diff --git a/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-6Bit.toml b/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-6Bit.toml
index 96b7a7fb..c2a193d9 100644
--- a/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-6Bit.toml
+++ b/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-6Bit.toml
@@ -12,3 +12,9 @@ context_length = 262144
[storage_size]
in_bytes = 25667611392
+
+# Source: https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B
+# Source: https://unsloth.ai/docs/models/nemotron-3
+[sampling_defaults]
+temperature = 1.0
+top_p = 1.0
diff --git a/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-8Bit.toml b/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-8Bit.toml
index 1c95fa90..7ba87193 100644
--- a/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-8Bit.toml
+++ b/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-8Bit.toml
@@ -12,3 +12,9 @@ context_length = 262144
[storage_size]
in_bytes = 33559880448
+
+# Source: https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B
+# Source: https://unsloth.ai/docs/models/nemotron-3
+[sampling_defaults]
+temperature = 1.0
+top_p = 1.0
diff --git a/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-BF16.toml b/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-BF16.toml
index 5364e5bc..d7223a6b 100644
--- a/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-BF16.toml
+++ b/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-BF16.toml
@@ -12,3 +12,9 @@ context_length = 262144
[storage_size]
in_bytes = 63155889408
+
+# Source: https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B
+# Source: https://unsloth.ai/docs/models/nemotron-3
+[sampling_defaults]
+temperature = 1.0
+top_p = 1.0
diff --git a/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-MXFP4.toml b/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-MXFP4.toml
index 1c7137da..033cb9fc 100644
--- a/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-MXFP4.toml
+++ b/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-MLX-MXFP4.toml
@@ -12,3 +12,9 @@ context_length = 262144
[storage_size]
in_bytes = 16788808704
+
+# Source: https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B
+# Source: https://unsloth.ai/docs/models/nemotron-3
+[sampling_defaults]
+temperature = 1.0
+top_p = 1.0
diff --git a/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-NVFP4.toml b/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-NVFP4.toml
index 13aae100..2858cd1f 100644
--- a/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-NVFP4.toml
+++ b/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-3-Nano-30B-A3B-NVFP4.toml
@@ -12,3 +12,9 @@ context_length = 262144
[storage_size]
in_bytes = 19323906944
+
+# Source: https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B
+# Source: https://unsloth.ai/docs/models/nemotron-3
+[sampling_defaults]
+temperature = 1.0
+top_p = 1.0
diff --git a/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-Nano-9B-v2-4bits.toml b/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-Nano-9B-v2-4bits.toml
index 649fecd9..a834e1aa 100644
--- a/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-Nano-9B-v2-4bits.toml
+++ b/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-Nano-9B-v2-4bits.toml
@@ -12,3 +12,14 @@ context_length = 131072
[storage_size]
in_bytes = 5002791936
+
+# Source: https://huggingface.co/nvidia/NVIDIA-Nemotron-Nano-9B-v2
+# Source: https://build.nvidia.com/nvidia/nvidia-nemotron-nano-9b-v2/modelcard
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+
+# Source: https://huggingface.co/nvidia/NVIDIA-Nemotron-Nano-9B-v2
+# Source: https://build.nvidia.com/nvidia/nvidia-nemotron-nano-9b-v2/modelcard
+[sampling_defaults.non_thinking]
+temperature = 0.0
diff --git a/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-Nano-9B-v2-6bit.toml b/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-Nano-9B-v2-6bit.toml
index e5c6422b..ec59e5a0 100644
--- a/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-Nano-9B-v2-6bit.toml
+++ b/resources/inference_model_cards/mlx-community--NVIDIA-Nemotron-Nano-9B-v2-6bit.toml
@@ -12,3 +12,14 @@ context_length = 131072
[storage_size]
in_bytes = 7224298496
+
+# Source: https://huggingface.co/nvidia/NVIDIA-Nemotron-Nano-9B-v2
+# Source: https://build.nvidia.com/nvidia/nvidia-nemotron-nano-9b-v2/modelcard
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+
+# Source: https://huggingface.co/nvidia/NVIDIA-Nemotron-Nano-9B-v2
+# Source: https://build.nvidia.com/nvidia/nvidia-nemotron-nano-9b-v2/modelcard
+[sampling_defaults.non_thinking]
+temperature = 0.0
diff --git a/resources/inference_model_cards/mlx-community--Qwen3-0.6B-4bit.toml b/resources/inference_model_cards/mlx-community--Qwen3-0.6B-4bit.toml
index f99033fc..cf44e9c1 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3-0.6B-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3-0.6B-4bit.toml
@@ -13,3 +13,17 @@ context_length = 32768
[storage_size]
in_bytes = 342884352
+
+# Source: https://huggingface.co/Qwen/Qwen3-0.6B#best-practices
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+top_k = 20
+min_p = 0.0
+
+# Source: https://huggingface.co/Qwen/Qwen3-0.6B#best-practices
+[sampling_defaults.non_thinking]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
diff --git a/resources/inference_model_cards/mlx-community--Qwen3-0.6B-8bit.toml b/resources/inference_model_cards/mlx-community--Qwen3-0.6B-8bit.toml
index 8d7e7643..6061fc49 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3-0.6B-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3-0.6B-8bit.toml
@@ -13,3 +13,17 @@ context_length = 32768
[storage_size]
in_bytes = 698351616
+
+# Source: https://huggingface.co/Qwen/Qwen3-0.6B#best-practices
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+top_k = 20
+min_p = 0.0
+
+# Source: https://huggingface.co/Qwen/Qwen3-0.6B#best-practices
+[sampling_defaults.non_thinking]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
diff --git a/resources/inference_model_cards/mlx-community--Qwen3-235B-A22B-Instruct-2507-4bit.toml b/resources/inference_model_cards/mlx-community--Qwen3-235B-A22B-Instruct-2507-4bit.toml
index 6e2709f5..430b985c 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3-235B-A22B-Instruct-2507-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3-235B-A22B-Instruct-2507-4bit.toml
@@ -13,3 +13,10 @@ context_length = 262144
[storage_size]
in_bytes = 141733920768
+
+# Source: https://huggingface.co/Qwen/Qwen3-235B-A22B-Instruct-2507#best-practices
+[sampling_defaults]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
diff --git a/resources/inference_model_cards/mlx-community--Qwen3-235B-A22B-Instruct-2507-8bit.toml b/resources/inference_model_cards/mlx-community--Qwen3-235B-A22B-Instruct-2507-8bit.toml
index 0e41f4ed..8b3df97e 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3-235B-A22B-Instruct-2507-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3-235B-A22B-Instruct-2507-8bit.toml
@@ -13,3 +13,10 @@ context_length = 262144
[storage_size]
in_bytes = 268435456000
+
+# Source: https://huggingface.co/Qwen/Qwen3-235B-A22B-Instruct-2507#best-practices
+[sampling_defaults]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
diff --git a/resources/inference_model_cards/mlx-community--Qwen3-30B-A3B-4bit.toml b/resources/inference_model_cards/mlx-community--Qwen3-30B-A3B-4bit.toml
index 4e736c14..13951adc 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3-30B-A3B-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3-30B-A3B-4bit.toml
@@ -13,3 +13,17 @@ context_length = 32768
[storage_size]
in_bytes = 17612931072
+
+# Source: https://huggingface.co/Qwen/Qwen3-30B-A3B#best-practices
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+top_k = 20
+min_p = 0.0
+
+# Source: https://huggingface.co/Qwen/Qwen3-30B-A3B#best-practices
+[sampling_defaults.non_thinking]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
diff --git a/resources/inference_model_cards/mlx-community--Qwen3-30B-A3B-8bit.toml b/resources/inference_model_cards/mlx-community--Qwen3-30B-A3B-8bit.toml
index 15308a84..4a4db8e2 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3-30B-A3B-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3-30B-A3B-8bit.toml
@@ -13,3 +13,17 @@ context_length = 32768
[storage_size]
in_bytes = 33279705088
+
+# Source: https://huggingface.co/Qwen/Qwen3-30B-A3B#best-practices
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+top_k = 20
+min_p = 0.0
+
+# Source: https://huggingface.co/Qwen/Qwen3-30B-A3B#best-practices
+[sampling_defaults.non_thinking]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
diff --git a/resources/inference_model_cards/mlx-community--Qwen3-Coder-480B-A35B-Instruct-4bit.toml b/resources/inference_model_cards/mlx-community--Qwen3-Coder-480B-A35B-Instruct-4bit.toml
index 95010af8..26236ff8 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3-Coder-480B-A35B-Instruct-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3-Coder-480B-A35B-Instruct-4bit.toml
@@ -13,3 +13,11 @@ context_length = 262144
[storage_size]
in_bytes = 289910292480
+
+# Source: https://huggingface.co/Qwen/Qwen3-Coder-480B-A35B-Instruct#best-practices
+# Source: https://huggingface.co/unsloth/Qwen3-Coder-Next-GGUF
+[sampling_defaults]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+repetition_penalty = 1.05
diff --git a/resources/inference_model_cards/mlx-community--Qwen3-Coder-480B-A35B-Instruct-8bit.toml b/resources/inference_model_cards/mlx-community--Qwen3-Coder-480B-A35B-Instruct-8bit.toml
index 963fb205..4f145936 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3-Coder-480B-A35B-Instruct-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3-Coder-480B-A35B-Instruct-8bit.toml
@@ -13,3 +13,11 @@ context_length = 262144
[storage_size]
in_bytes = 579820584960
+
+# Source: https://huggingface.co/Qwen/Qwen3-Coder-480B-A35B-Instruct#best-practices
+# Source: https://huggingface.co/unsloth/Qwen3-Coder-Next-GGUF
+[sampling_defaults]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+repetition_penalty = 1.05
diff --git a/resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-4bit.toml b/resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-4bit.toml
index e28c1d95..5130447d 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-4bit.toml
@@ -13,3 +13,10 @@ context_length = 262144
[storage_size]
in_bytes = 45644286500
+
+# Source: https://huggingface.co/mlx-community/Qwen3-Coder-Next-4bit/blob/main/generation_config.json
+# Source: https://huggingface.co/unsloth/Qwen3-Coder-Next-GGUF
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 40
diff --git a/resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-5bit.toml b/resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-5bit.toml
index 58c63f78..b33d2427 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-5bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-5bit.toml
@@ -13,3 +13,10 @@ context_length = 262144
[storage_size]
in_bytes = 57657697020
+
+# Source: https://huggingface.co/mlx-community/Qwen3-Coder-Next-4bit/blob/main/generation_config.json
+# Source: https://huggingface.co/unsloth/Qwen3-Coder-Next-GGUF
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 40
diff --git a/resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-6bit.toml b/resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-6bit.toml
index 17f921e5..fc46f271 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-6bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-6bit.toml
@@ -13,3 +13,10 @@ context_length = 262144
[storage_size]
in_bytes = 68899327465
+
+# Source: https://huggingface.co/mlx-community/Qwen3-Coder-Next-4bit/blob/main/generation_config.json
+# Source: https://huggingface.co/unsloth/Qwen3-Coder-Next-GGUF
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 40
diff --git a/resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-8bit.toml b/resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-8bit.toml
index 934894ac..967bfea6 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-8bit.toml
@@ -13,3 +13,10 @@ context_length = 262144
[storage_size]
in_bytes = 89357758772
+
+# Source: https://huggingface.co/mlx-community/Qwen3-Coder-Next-4bit/blob/main/generation_config.json
+# Source: https://huggingface.co/unsloth/Qwen3-Coder-Next-GGUF
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 40
diff --git a/resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-bf16.toml b/resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-bf16.toml
index fd135bb0..e1c5acac 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-bf16.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3-Coder-Next-bf16.toml
@@ -13,3 +13,10 @@ context_length = 262144
[storage_size]
in_bytes = 157548627945
+
+# Source: https://huggingface.co/mlx-community/Qwen3-Coder-Next-4bit/blob/main/generation_config.json
+# Source: https://huggingface.co/unsloth/Qwen3-Coder-Next-GGUF
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 40
diff --git a/resources/inference_model_cards/mlx-community--Qwen3-Next-80B-A3B-Instruct-4bit.toml b/resources/inference_model_cards/mlx-community--Qwen3-Next-80B-A3B-Instruct-4bit.toml
index 7a331b46..7850c667 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3-Next-80B-A3B-Instruct-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3-Next-80B-A3B-Instruct-4bit.toml
@@ -13,3 +13,10 @@ context_length = 262144
[storage_size]
in_bytes = 46976204800
+
+# Source: https://huggingface.co/Qwen/Qwen3-Next-80B-A3B-Instruct#best-practices
+[sampling_defaults]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
diff --git a/resources/inference_model_cards/mlx-community--Qwen3-Next-80B-A3B-Instruct-8bit.toml b/resources/inference_model_cards/mlx-community--Qwen3-Next-80B-A3B-Instruct-8bit.toml
index a54f01de..1120f824 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3-Next-80B-A3B-Instruct-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3-Next-80B-A3B-Instruct-8bit.toml
@@ -13,3 +13,10 @@ context_length = 262144
[storage_size]
in_bytes = 88814387200
+
+# Source: https://huggingface.co/Qwen/Qwen3-Next-80B-A3B-Instruct#best-practices
+[sampling_defaults]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
diff --git a/resources/inference_model_cards/mlx-community--Qwen3-Next-80B-A3B-Thinking-4bit.toml b/resources/inference_model_cards/mlx-community--Qwen3-Next-80B-A3B-Thinking-4bit.toml
index d126f727..0d8f8952 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3-Next-80B-A3B-Thinking-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3-Next-80B-A3B-Thinking-4bit.toml
@@ -13,3 +13,10 @@ context_length = 262144
[storage_size]
in_bytes = 47080074240
+
+# Source: https://huggingface.co/Qwen/Qwen3-Next-80B-A3B-Thinking#best-practices
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+top_k = 20
+min_p = 0.0
diff --git a/resources/inference_model_cards/mlx-community--Qwen3-Next-80B-A3B-Thinking-8bit.toml b/resources/inference_model_cards/mlx-community--Qwen3-Next-80B-A3B-Thinking-8bit.toml
index b008b130..37cc5635 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3-Next-80B-A3B-Thinking-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3-Next-80B-A3B-Thinking-8bit.toml
@@ -13,3 +13,10 @@ context_length = 262144
[storage_size]
in_bytes = 88814387200
+
+# Source: https://huggingface.co/Qwen/Qwen3-Next-80B-A3B-Thinking#best-practices
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+top_k = 20
+min_p = 0.0
diff --git a/resources/inference_model_cards/mlx-community--Qwen3-VL-4B-Instruct-4bit.toml b/resources/inference_model_cards/mlx-community--Qwen3-VL-4B-Instruct-4bit.toml
index 6d34735f..9f28315c 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3-VL-4B-Instruct-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3-VL-4B-Instruct-4bit.toml
@@ -12,3 +12,12 @@ context_length = 262144
[storage_size]
in_bytes = 3340000000
+
+# Source: https://huggingface.co/Qwen/Qwen3-VL-4B-Instruct#generation-hyperparameters
+# Source: https://unsloth.ai/docs/models/qwen3-how-to-run-and-fine-tune/qwen3-vl-how-to-run-and-fine-tune
+[sampling_defaults]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+repetition_penalty = 1.0
+presence_penalty = 1.5
diff --git a/resources/inference_model_cards/mlx-community--Qwen3.5-122B-A10B-4bit.toml b/resources/inference_model_cards/mlx-community--Qwen3.5-122B-A10B-4bit.toml
index 100ab997..de047902 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3.5-122B-A10B-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3.5-122B-A10B-4bit.toml
@@ -13,3 +13,23 @@ context_length = 262144
[storage_size]
in_bytes = 69593314272
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-122B-A10B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 0.0
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-122B-A10B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults.non_thinking]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
diff --git a/resources/inference_model_cards/mlx-community--Qwen3.5-122B-A10B-6bit.toml b/resources/inference_model_cards/mlx-community--Qwen3.5-122B-A10B-6bit.toml
index 99177f48..25caa21d 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3.5-122B-A10B-6bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3.5-122B-A10B-6bit.toml
@@ -13,3 +13,23 @@ context_length = 262144
[storage_size]
in_bytes = 100120675296
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-122B-A10B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 0.0
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-122B-A10B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults.non_thinking]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
diff --git a/resources/inference_model_cards/mlx-community--Qwen3.5-122B-A10B-8bit.toml b/resources/inference_model_cards/mlx-community--Qwen3.5-122B-A10B-8bit.toml
index cff7cf62..fd8e201c 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3.5-122B-A10B-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3.5-122B-A10B-8bit.toml
@@ -13,3 +13,23 @@ context_length = 262144
[storage_size]
in_bytes = 130648036320
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-122B-A10B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 0.0
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-122B-A10B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults.non_thinking]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
diff --git a/resources/inference_model_cards/mlx-community--Qwen3.5-122B-A10B-bf16.toml b/resources/inference_model_cards/mlx-community--Qwen3.5-122B-A10B-bf16.toml
index 2299c8e7..c65812cb 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3.5-122B-A10B-bf16.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3.5-122B-A10B-bf16.toml
@@ -13,3 +13,23 @@ context_length = 262144
[storage_size]
in_bytes = 245125640160
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-122B-A10B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 0.0
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-122B-A10B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults.non_thinking]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
diff --git a/resources/inference_model_cards/mlx-community--Qwen3.5-27B-4bit.toml b/resources/inference_model_cards/mlx-community--Qwen3.5-27B-4bit.toml
index 8c8a4f11..9a086fd2 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3.5-27B-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3.5-27B-4bit.toml
@@ -13,3 +13,23 @@ context_length = 262144
[storage_size]
in_bytes = 16054266848
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-27B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 0.0
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-27B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults.non_thinking]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
diff --git a/resources/inference_model_cards/mlx-community--Qwen3.5-27B-8bit.toml b/resources/inference_model_cards/mlx-community--Qwen3.5-27B-8bit.toml
index 3c7dc6eb..4c5622e4 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3.5-27B-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3.5-27B-8bit.toml
@@ -13,3 +13,23 @@ context_length = 262144
[storage_size]
in_bytes = 29500943328
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-27B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 0.0
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-27B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults.non_thinking]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
diff --git a/resources/inference_model_cards/mlx-community--Qwen3.5-2B-MLX-8bit.toml b/resources/inference_model_cards/mlx-community--Qwen3.5-2B-MLX-8bit.toml
index a62d4495..a7ddeec6 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3.5-2B-MLX-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3.5-2B-MLX-8bit.toml
@@ -13,3 +13,23 @@ context_length = 262144
[storage_size]
in_bytes = 2662787264
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-9B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-9B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults.non_thinking]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
diff --git a/resources/inference_model_cards/mlx-community--Qwen3.5-35B-A3B-4bit.toml b/resources/inference_model_cards/mlx-community--Qwen3.5-35B-A3B-4bit.toml
index dfb5e244..63cb64e2 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3.5-35B-A3B-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3.5-35B-A3B-4bit.toml
@@ -13,3 +13,23 @@ context_length = 262144
[storage_size]
in_bytes = 20391405152
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-35B-A3B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-35B-A3B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults.non_thinking]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
diff --git a/resources/inference_model_cards/mlx-community--Qwen3.5-35B-A3B-8bit.toml b/resources/inference_model_cards/mlx-community--Qwen3.5-35B-A3B-8bit.toml
index 24881e04..55fd0860 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3.5-35B-A3B-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3.5-35B-A3B-8bit.toml
@@ -13,3 +13,23 @@ context_length = 262144
[storage_size]
in_bytes = 37721130592
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-35B-A3B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-35B-A3B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults.non_thinking]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
diff --git a/resources/inference_model_cards/mlx-community--Qwen3.5-397B-A17B-4bit.toml b/resources/inference_model_cards/mlx-community--Qwen3.5-397B-A17B-4bit.toml
index 7823e87b..08cd2709 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3.5-397B-A17B-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3.5-397B-A17B-4bit.toml
@@ -13,3 +13,23 @@ context_length = 262144
[storage_size]
in_bytes = 223860768352
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-397B-A17B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 0.0
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-397B-A17B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults.non_thinking]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
diff --git a/resources/inference_model_cards/mlx-community--Qwen3.5-397B-A17B-6bit.toml b/resources/inference_model_cards/mlx-community--Qwen3.5-397B-A17B-6bit.toml
index 8df64cad..441200a5 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3.5-397B-A17B-6bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3.5-397B-A17B-6bit.toml
@@ -13,3 +13,23 @@ context_length = 262144
[storage_size]
in_bytes = 322946674272
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-397B-A17B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 0.0
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-397B-A17B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults.non_thinking]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
diff --git a/resources/inference_model_cards/mlx-community--Qwen3.5-397B-A17B-8bit.toml b/resources/inference_model_cards/mlx-community--Qwen3.5-397B-A17B-8bit.toml
index 4aa63400..166d9213 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3.5-397B-A17B-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3.5-397B-A17B-8bit.toml
@@ -13,3 +13,23 @@ context_length = 262144
[storage_size]
in_bytes = 422032580192
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-397B-A17B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 0.0
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-397B-A17B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults.non_thinking]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
diff --git a/resources/inference_model_cards/mlx-community--Qwen3.5-9B-4bit.toml b/resources/inference_model_cards/mlx-community--Qwen3.5-9B-4bit.toml
index c8bb95c7..867739d3 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3.5-9B-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3.5-9B-4bit.toml
@@ -13,3 +13,23 @@ context_length = 262144
[storage_size]
in_bytes = 5950062560
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-9B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-9B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults.non_thinking]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
diff --git a/resources/inference_model_cards/mlx-community--Qwen3.5-9B-8bit.toml b/resources/inference_model_cards/mlx-community--Qwen3.5-9B-8bit.toml
index 6b1dfede..04773d42 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3.5-9B-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3.5-9B-8bit.toml
@@ -13,3 +13,23 @@ context_length = 262144
[storage_size]
in_bytes = 10426433504
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-9B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
+
+# Source: https://huggingface.co/Qwen/Qwen3.5-9B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults.non_thinking]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
diff --git a/resources/inference_model_cards/mlx-community--Qwen3.6-35B-A3B-4bit.toml b/resources/inference_model_cards/mlx-community--Qwen3.6-35B-A3B-4bit.toml
index de63adeb..0f329cb1 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3.6-35B-A3B-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3.6-35B-A3B-4bit.toml
@@ -13,3 +13,23 @@ context_length = 262144
[storage_size]
in_bytes = 20401929952
+
+# Source: https://huggingface.co/Qwen/Qwen3.6-35B-A3B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
+
+# Source: https://huggingface.co/Qwen/Qwen3.6-35B-A3B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults.non_thinking]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
diff --git a/resources/inference_model_cards/mlx-community--Qwen3.6-35B-A3B-5bit.toml b/resources/inference_model_cards/mlx-community--Qwen3.6-35B-A3B-5bit.toml
index d0cdaaac..6d7a60db 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3.6-35B-A3B-5bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3.6-35B-A3B-5bit.toml
@@ -13,3 +13,23 @@ context_length = 262144
[storage_size]
in_bytes = 24731729632
+
+# Source: https://huggingface.co/Qwen/Qwen3.6-35B-A3B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
+
+# Source: https://huggingface.co/Qwen/Qwen3.6-35B-A3B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults.non_thinking]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
diff --git a/resources/inference_model_cards/mlx-community--Qwen3.6-35B-A3B-8bit.toml b/resources/inference_model_cards/mlx-community--Qwen3.6-35B-A3B-8bit.toml
index 01b7f671..b0edb2e0 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3.6-35B-A3B-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3.6-35B-A3B-8bit.toml
@@ -13,3 +13,23 @@ context_length = 262144
[storage_size]
in_bytes = 37721128672
+
+# Source: https://huggingface.co/Qwen/Qwen3.6-35B-A3B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
+
+# Source: https://huggingface.co/Qwen/Qwen3.6-35B-A3B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults.non_thinking]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
diff --git a/resources/inference_model_cards/mlx-community--Qwen3.6-35B-A3B-bf16.toml b/resources/inference_model_cards/mlx-community--Qwen3.6-35B-A3B-bf16.toml
index c876c00c..366e1b7b 100644
--- a/resources/inference_model_cards/mlx-community--Qwen3.6-35B-A3B-bf16.toml
+++ b/resources/inference_model_cards/mlx-community--Qwen3.6-35B-A3B-bf16.toml
@@ -13,3 +13,23 @@ context_length = 262144
[storage_size]
in_bytes = 70214363872
+
+# Source: https://huggingface.co/Qwen/Qwen3.6-35B-A3B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
+
+# Source: https://huggingface.co/Qwen/Qwen3.6-35B-A3B#best-practices
+# Source: https://unsloth.ai/docs/models/qwen3.5
+[sampling_defaults.non_thinking]
+temperature = 0.7
+top_p = 0.8
+top_k = 20
+min_p = 0.0
+repetition_penalty = 1.0
+presence_penalty = 1.5
diff --git a/resources/inference_model_cards/mlx-community--Step-3.5-Flash-4bit.toml b/resources/inference_model_cards/mlx-community--Step-3.5-Flash-4bit.toml
index d2971641..b5977723 100644
--- a/resources/inference_model_cards/mlx-community--Step-3.5-Flash-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--Step-3.5-Flash-4bit.toml
@@ -13,3 +13,15 @@ context_length = 262144
[storage_size]
in_bytes = 114572190076
+
+# Source: https://huggingface.co/stepfun-ai/Step-3.5-Flash/discussions/3
+# Source: https://github.com/stepfun-ai/Step-3.5-Flash/blob/main/llama.cpp/docs/step3.5-flash.md
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+
+# Source: https://huggingface.co/stepfun-ai/Step-3.5-Flash/discussions/3
+# Source: https://github.com/stepfun-ai/Step-3.5-Flash/blob/main/llama.cpp/docs/step3.5-flash.md
+[sampling_defaults.thinking]
+temperature = 1.0
+top_p = 0.95
diff --git a/resources/inference_model_cards/mlx-community--Step-3.5-Flash-6bit.toml b/resources/inference_model_cards/mlx-community--Step-3.5-Flash-6bit.toml
index 4adc3bb4..fe45fa37 100644
--- a/resources/inference_model_cards/mlx-community--Step-3.5-Flash-6bit.toml
+++ b/resources/inference_model_cards/mlx-community--Step-3.5-Flash-6bit.toml
@@ -13,3 +13,15 @@ context_length = 262144
[storage_size]
in_bytes = 159039627774
+
+# Source: https://huggingface.co/stepfun-ai/Step-3.5-Flash/discussions/3
+# Source: https://github.com/stepfun-ai/Step-3.5-Flash/blob/main/llama.cpp/docs/step3.5-flash.md
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+
+# Source: https://huggingface.co/stepfun-ai/Step-3.5-Flash/discussions/3
+# Source: https://github.com/stepfun-ai/Step-3.5-Flash/blob/main/llama.cpp/docs/step3.5-flash.md
+[sampling_defaults.thinking]
+temperature = 1.0
+top_p = 0.95
diff --git a/resources/inference_model_cards/mlx-community--Step-3.5-Flash-8Bit.toml b/resources/inference_model_cards/mlx-community--Step-3.5-Flash-8Bit.toml
index 1306637c..64ec31bb 100644
--- a/resources/inference_model_cards/mlx-community--Step-3.5-Flash-8Bit.toml
+++ b/resources/inference_model_cards/mlx-community--Step-3.5-Flash-8Bit.toml
@@ -13,3 +13,15 @@ context_length = 262144
[storage_size]
in_bytes = 209082699847
+
+# Source: https://huggingface.co/stepfun-ai/Step-3.5-Flash/discussions/3
+# Source: https://github.com/stepfun-ai/Step-3.5-Flash/blob/main/llama.cpp/docs/step3.5-flash.md
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.95
+
+# Source: https://huggingface.co/stepfun-ai/Step-3.5-Flash/discussions/3
+# Source: https://github.com/stepfun-ai/Step-3.5-Flash/blob/main/llama.cpp/docs/step3.5-flash.md
+[sampling_defaults.thinking]
+temperature = 1.0
+top_p = 0.95
diff --git a/resources/inference_model_cards/mlx-community--gemma-4-26b-a4b-it-4bit.toml b/resources/inference_model_cards/mlx-community--gemma-4-26b-a4b-it-4bit.toml
index 95c0be7b..51be323e 100644
--- a/resources/inference_model_cards/mlx-community--gemma-4-26b-a4b-it-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--gemma-4-26b-a4b-it-4bit.toml
@@ -13,3 +13,10 @@ context_length = 262144
[storage_size]
in_bytes = 15608614044
+
+# Source: https://ai.google.dev/gemma/docs/core/model_card_4
+# Source: https://huggingface.co/google/gemma-4-26b-a4b-it/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 64
diff --git a/resources/inference_model_cards/mlx-community--gemma-4-26b-a4b-it-6bit.toml b/resources/inference_model_cards/mlx-community--gemma-4-26b-a4b-it-6bit.toml
index 66c45506..c984d44b 100644
--- a/resources/inference_model_cards/mlx-community--gemma-4-26b-a4b-it-6bit.toml
+++ b/resources/inference_model_cards/mlx-community--gemma-4-26b-a4b-it-6bit.toml
@@ -13,3 +13,10 @@ context_length = 262144
[storage_size]
in_bytes = 21781015708
+
+# Source: https://huggingface.co/google/gemma-4-26b-a4b-it/blob/main/generation_config.json
+# Source: https://ai.google.dev/gemma/docs/core/model_card_4
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 64
diff --git a/resources/inference_model_cards/mlx-community--gemma-4-26b-a4b-it-8bit.toml b/resources/inference_model_cards/mlx-community--gemma-4-26b-a4b-it-8bit.toml
index f0cc1514..fe258366 100644
--- a/resources/inference_model_cards/mlx-community--gemma-4-26b-a4b-it-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--gemma-4-26b-a4b-it-8bit.toml
@@ -13,3 +13,10 @@ context_length = 262144
[storage_size]
in_bytes = 27953417372
+
+# Source: https://huggingface.co/google/gemma-4-26b-a4b-it/blob/main/generation_config.json
+# Source: https://ai.google.dev/gemma/docs/core/model_card_4
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 64
diff --git a/resources/inference_model_cards/mlx-community--gemma-4-26b-a4b-it-bf16.toml b/resources/inference_model_cards/mlx-community--gemma-4-26b-a4b-it-bf16.toml
index ece39982..ea4dbbfc 100644
--- a/resources/inference_model_cards/mlx-community--gemma-4-26b-a4b-it-bf16.toml
+++ b/resources/inference_model_cards/mlx-community--gemma-4-26b-a4b-it-bf16.toml
@@ -13,3 +13,10 @@ context_length = 262144
[storage_size]
in_bytes = 51611872412
+
+# Source: https://ai.google.dev/gemma/docs/core/model_card_4
+# Source: https://huggingface.co/google/gemma-4-26b-a4b-it/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 64
diff --git a/resources/inference_model_cards/mlx-community--gemma-4-31b-it-4bit.toml b/resources/inference_model_cards/mlx-community--gemma-4-31b-it-4bit.toml
index d8a40f35..cb8e6358 100644
--- a/resources/inference_model_cards/mlx-community--gemma-4-31b-it-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--gemma-4-31b-it-4bit.toml
@@ -13,3 +13,10 @@ context_length = 262144
[storage_size]
in_bytes = 18411755224
+
+# Source: https://ai.google.dev/gemma/docs/core/model_card_4
+# Source: https://huggingface.co/google/gemma-4-31B-it/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 64
diff --git a/resources/inference_model_cards/mlx-community--gemma-4-31b-it-6bit.toml b/resources/inference_model_cards/mlx-community--gemma-4-31b-it-6bit.toml
index 6222ce70..84562062 100644
--- a/resources/inference_model_cards/mlx-community--gemma-4-31b-it-6bit.toml
+++ b/resources/inference_model_cards/mlx-community--gemma-4-31b-it-6bit.toml
@@ -13,3 +13,10 @@ context_length = 262144
[storage_size]
in_bytes = 26087306968
+
+# Source: https://ai.google.dev/gemma/docs/core/model_card_4
+# Source: https://huggingface.co/google/gemma-4-31B-it/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 64
diff --git a/resources/inference_model_cards/mlx-community--gemma-4-31b-it-8bit.toml b/resources/inference_model_cards/mlx-community--gemma-4-31b-it-8bit.toml
index 86396118..332a9b00 100644
--- a/resources/inference_model_cards/mlx-community--gemma-4-31b-it-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--gemma-4-31b-it-8bit.toml
@@ -13,3 +13,10 @@ context_length = 262144
[storage_size]
in_bytes = 33762858712
+
+# Source: https://ai.google.dev/gemma/docs/core/model_card_4
+# Source: https://huggingface.co/google/gemma-4-31B-it/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 64
diff --git a/resources/inference_model_cards/mlx-community--gemma-4-31b-it-bf16.toml b/resources/inference_model_cards/mlx-community--gemma-4-31b-it-bf16.toml
index 1d4f740c..6fc0a2dc 100644
--- a/resources/inference_model_cards/mlx-community--gemma-4-31b-it-bf16.toml
+++ b/resources/inference_model_cards/mlx-community--gemma-4-31b-it-bf16.toml
@@ -13,3 +13,10 @@ context_length = 262144
[storage_size]
in_bytes = 62546177752
+
+# Source: https://ai.google.dev/gemma/docs/core/model_card_4
+# Source: https://huggingface.co/google/gemma-4-31B-it/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 64
diff --git a/resources/inference_model_cards/mlx-community--gemma-4-e2b-it-4bit.toml b/resources/inference_model_cards/mlx-community--gemma-4-e2b-it-4bit.toml
index 9d8f9932..db641a2b 100644
--- a/resources/inference_model_cards/mlx-community--gemma-4-e2b-it-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--gemma-4-e2b-it-4bit.toml
@@ -13,3 +13,10 @@ context_length = 131072
[storage_size]
in_bytes = 3580765126
+
+# Source: https://ai.google.dev/gemma/docs/core/model_card_4
+# Source: https://huggingface.co/google/gemma-4-e2b-it/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 64
diff --git a/resources/inference_model_cards/mlx-community--gemma-4-e2b-it-6bit.toml b/resources/inference_model_cards/mlx-community--gemma-4-e2b-it-6bit.toml
index c9c6d1a0..4fdf9dda 100644
--- a/resources/inference_model_cards/mlx-community--gemma-4-e2b-it-6bit.toml
+++ b/resources/inference_model_cards/mlx-community--gemma-4-e2b-it-6bit.toml
@@ -13,3 +13,10 @@ context_length = 131072
[storage_size]
in_bytes = 4739998662
+
+# Source: https://ai.google.dev/gemma/docs/core/model_card_4
+# Source: https://huggingface.co/google/gemma-4-e2b-it/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 64
diff --git a/resources/inference_model_cards/mlx-community--gemma-4-e2b-it-8bit.toml b/resources/inference_model_cards/mlx-community--gemma-4-e2b-it-8bit.toml
index 804da907..8834e16f 100644
--- a/resources/inference_model_cards/mlx-community--gemma-4-e2b-it-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--gemma-4-e2b-it-8bit.toml
@@ -13,3 +13,10 @@ context_length = 131072
[storage_size]
in_bytes = 5899232198
+
+# Source: https://ai.google.dev/gemma/docs/core/model_card_4
+# Source: https://huggingface.co/google/gemma-4-e2b-it/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 64
diff --git a/resources/inference_model_cards/mlx-community--gemma-4-e2b-it-bf16.toml b/resources/inference_model_cards/mlx-community--gemma-4-e2b-it-bf16.toml
index 8f2920da..1bad3dd8 100644
--- a/resources/inference_model_cards/mlx-community--gemma-4-e2b-it-bf16.toml
+++ b/resources/inference_model_cards/mlx-community--gemma-4-e2b-it-bf16.toml
@@ -13,3 +13,10 @@ context_length = 131072
[storage_size]
in_bytes = 10246357958
+
+# Source: https://ai.google.dev/gemma/docs/core/model_card_4
+# Source: https://huggingface.co/google/gemma-4-e2b-it/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 64
diff --git a/resources/inference_model_cards/mlx-community--gemma-4-e4b-it-4bit.toml b/resources/inference_model_cards/mlx-community--gemma-4-e4b-it-4bit.toml
index 1122bbeb..0f4cc821 100644
--- a/resources/inference_model_cards/mlx-community--gemma-4-e4b-it-4bit.toml
+++ b/resources/inference_model_cards/mlx-community--gemma-4-e4b-it-4bit.toml
@@ -13,3 +13,10 @@ context_length = 131072
[storage_size]
in_bytes = 5216992212
+
+# Source: https://ai.google.dev/gemma/docs/core/model_card_4
+# Source: https://huggingface.co/google/gemma-4-e4b-it/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 64
diff --git a/resources/inference_model_cards/mlx-community--gemma-4-e4b-it-6bit.toml b/resources/inference_model_cards/mlx-community--gemma-4-e4b-it-6bit.toml
index 6f3b430c..2cdd2b33 100644
--- a/resources/inference_model_cards/mlx-community--gemma-4-e4b-it-6bit.toml
+++ b/resources/inference_model_cards/mlx-community--gemma-4-e4b-it-6bit.toml
@@ -13,3 +13,10 @@ context_length = 131072
[storage_size]
in_bytes = 7090961364
+
+# Source: https://ai.google.dev/gemma/docs/core/model_card_4
+# Source: https://huggingface.co/google/gemma-4-e4b-it/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 64
diff --git a/resources/inference_model_cards/mlx-community--gemma-4-e4b-it-8bit.toml b/resources/inference_model_cards/mlx-community--gemma-4-e4b-it-8bit.toml
index 48e21fef..1be07028 100644
--- a/resources/inference_model_cards/mlx-community--gemma-4-e4b-it-8bit.toml
+++ b/resources/inference_model_cards/mlx-community--gemma-4-e4b-it-8bit.toml
@@ -13,3 +13,10 @@ context_length = 131072
[storage_size]
in_bytes = 8964930516
+
+# Source: https://ai.google.dev/gemma/docs/core/model_card_4
+# Source: https://huggingface.co/google/gemma-4-e4b-it/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 64
diff --git a/resources/inference_model_cards/mlx-community--gemma-4-e4b-it-bf16.toml b/resources/inference_model_cards/mlx-community--gemma-4-e4b-it-bf16.toml
index db87c2ee..f18df6f3 100644
--- a/resources/inference_model_cards/mlx-community--gemma-4-e4b-it-bf16.toml
+++ b/resources/inference_model_cards/mlx-community--gemma-4-e4b-it-bf16.toml
@@ -13,3 +13,10 @@ context_length = 131072
[storage_size]
in_bytes = 15992314836
+
+# Source: https://ai.google.dev/gemma/docs/core/model_card_4
+# Source: https://huggingface.co/google/gemma-4-e4b-it/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 1.0
+top_p = 0.95
+top_k = 64
diff --git a/resources/inference_model_cards/mlx-community--gpt-oss-120b-MXFP4-Q8.toml b/resources/inference_model_cards/mlx-community--gpt-oss-120b-MXFP4-Q8.toml
index 02e8b755..bdc8f2a9 100644
--- a/resources/inference_model_cards/mlx-community--gpt-oss-120b-MXFP4-Q8.toml
+++ b/resources/inference_model_cards/mlx-community--gpt-oss-120b-MXFP4-Q8.toml
@@ -13,3 +13,10 @@ context_length = 131072
[storage_size]
in_bytes = 70652212224
+
+# Source: https://github.com/openai/gpt-oss/blob/main/README.md
+# Source: https://unsloth.ai/docs/models/gpt-oss-how-to-run-and-fine-tune
+[sampling_defaults]
+temperature = 1.0
+top_p = 1.0
+top_k = 0
diff --git a/resources/inference_model_cards/mlx-community--gpt-oss-20b-MXFP4-Q8.toml b/resources/inference_model_cards/mlx-community--gpt-oss-20b-MXFP4-Q8.toml
index a363bfab..a93d2b1d 100644
--- a/resources/inference_model_cards/mlx-community--gpt-oss-20b-MXFP4-Q8.toml
+++ b/resources/inference_model_cards/mlx-community--gpt-oss-20b-MXFP4-Q8.toml
@@ -13,3 +13,10 @@ context_length = 131072
[storage_size]
in_bytes = 12025908224
+
+# Source: https://github.com/openai/gpt-oss/blob/main/README.md
+# Source: https://unsloth.ai/docs/models/gpt-oss-how-to-run-and-fine-tune
+[sampling_defaults]
+temperature = 1.0
+top_p = 1.0
+top_k = 0
diff --git a/resources/inference_model_cards/mlx-community--llama-3.3-70b-instruct-fp16.toml b/resources/inference_model_cards/mlx-community--llama-3.3-70b-instruct-fp16.toml
index 194a88d7..06cf3131 100644
--- a/resources/inference_model_cards/mlx-community--llama-3.3-70b-instruct-fp16.toml
+++ b/resources/inference_model_cards/mlx-community--llama-3.3-70b-instruct-fp16.toml
@@ -13,3 +13,9 @@ context_length = 131072
[storage_size]
in_bytes = 144383672320
+
+# Source: https://huggingface.co/meta-llama/Llama-3.3-70B-Instruct/blob/main/generation_config.json
+# Source: https://huggingface.co/unsloth/Llama-3.3-70B-Instruct/blob/main/generation_config.json
+[sampling_defaults]
+temperature = 0.6
+top_p = 0.9
diff --git a/src/exo/api/main.py b/src/exo/api/main.py
index 8aaa9fa4..400ff9ab 100644
--- a/src/exo/api/main.py
+++ b/src/exo/api/main.py
@@ -742,6 +742,7 @@ class API:
async def _send_text_generation_with_images(
self, task_params: TextGenerationTaskParams
) -> TextGeneration:
+ task_params = task_params.with_card_sampling_defaults()
images = task_params.images
if not images:
command = TextGeneration(task_params=task_params)
diff --git a/src/exo/shared/models/model_cards.py b/src/exo/shared/models/model_cards.py
index 83ac9d49..a4f8d9ab 100644
--- a/src/exo/shared/models/model_cards.py
+++ b/src/exo/shared/models/model_cards.py
@@ -117,6 +117,21 @@ class VisionCardConfig(CamelCaseModel):
processor_repo: str | None = None
+class SamplingValues(CamelCaseModel):
+ temperature: float | None = None
+ top_p: float | None = None
+ top_k: int | None = None
+ min_p: float | None = None
+ repetition_penalty: float | None = None
+ presence_penalty: float | None = None
+ frequency_penalty: float | None = None
+
+
+class SamplingDefaults(SamplingValues):
+ thinking: SamplingValues | None = None
+ non_thinking: SamplingValues | None = None
+
+
class ModelCard(CamelCaseModel):
model_id: ModelId
storage_size: Memory
@@ -135,6 +150,7 @@ class ModelCard(CamelCaseModel):
trust_remote_code: bool = True
is_custom: bool = False
vision: VisionCardConfig | None = None
+ sampling_defaults: SamplingDefaults = Field(default_factory=SamplingDefaults)
@model_validator(mode="after")
def _autodetect_vision(self) -> "ModelCard":
diff --git a/src/exo/shared/types/text_generation.py b/src/exo/shared/types/text_generation.py
index 89ff9a44..dc763248 100644
--- a/src/exo/shared/types/text_generation.py
+++ b/src/exo/shared/types/text_generation.py
@@ -8,6 +8,7 @@ from typing import Annotated, Any, Literal
from pydantic import BaseModel, Field, WrapValidator
+from exo.shared.logging import logger
from exo.shared.types.common import ModelId, TruncatingString
MessageRole = Literal["user", "assistant", "system", "developer", "tool"]
@@ -109,7 +110,45 @@ class TextGenerationTaskParams(BaseModel, frozen=True):
min_p: float | None = None
repetition_penalty: float | None = None
repetition_context_size: int | None = None
+ presence_penalty: float | None = None
+ frequency_penalty: float | None = None
images: list[Base64Image] = Field(default_factory=list)
image_hashes: dict[int, Base64ImageHash] = Field(default_factory=dict)
total_input_chunks: int = 0
image_count: int = 0
+
+ def with_card_sampling_defaults(self) -> "TextGenerationTaskParams":
+ from exo.shared.models.model_cards import get_card
+
+ card = get_card(self.model)
+ if card is None:
+ return self
+
+ flat = card.sampling_defaults
+ if self.enable_thinking is True and flat.thinking is not None:
+ card_values = flat.thinking
+ elif self.enable_thinking is False and flat.non_thinking is not None:
+ card_values = flat.non_thinking
+ else:
+ card_values = flat
+
+ def resolve[T](request: T | None, card_value: T | None) -> T | None:
+ return request if request is not None else card_value
+
+ updates = {
+ "temperature": resolve(self.temperature, card_values.temperature),
+ "top_p": resolve(self.top_p, card_values.top_p),
+ "top_k": resolve(self.top_k, card_values.top_k),
+ "min_p": resolve(self.min_p, card_values.min_p),
+ "repetition_penalty": resolve(
+ self.repetition_penalty, card_values.repetition_penalty
+ ),
+ "presence_penalty": resolve(
+ self.presence_penalty, card_values.presence_penalty
+ ),
+ "frequency_penalty": resolve(
+ self.frequency_penalty, card_values.frequency_penalty
+ ),
+ }
+ logger.debug(f"Using sampling params for {self.model}:\n{updates}")
+ return self.model_copy(update=updates)
diff --git a/src/exo/worker/engines/mlx/generator/batch_generate.py b/src/exo/worker/engines/mlx/generator/batch_generate.py
index 86d063cb..b0b2a537 100644
--- a/src/exo/worker/engines/mlx/generator/batch_generate.py
+++ b/src/exo/worker/engines/mlx/generator/batch_generate.py
@@ -253,7 +253,11 @@ class ExoBatchGenerator:
logits_processors: list[Callable[[mx.array, mx.array], mx.array]] = (
make_logits_processors(
repetition_penalty=task_params.repetition_penalty,
- repetition_context_size=task_params.repetition_context_size,
+ repetition_context_size=task_params.repetition_context_size
+ if task_params.repetition_context_size is not None
+ else 20,
+ presence_penalty=task_params.presence_penalty,
+ frequency_penalty=task_params.frequency_penalty,
)
)
if is_bench:
diff --git a/src/exo/worker/engines/mlx/generator/generate.py b/src/exo/worker/engines/mlx/generator/generate.py
index ba125760..da1e958b 100644
--- a/src/exo/worker/engines/mlx/generator/generate.py
+++ b/src/exo/worker/engines/mlx/generator/generate.py
@@ -593,7 +593,11 @@ def mlx_generate(
logits_processors: list[Callable[[mx.array, mx.array], mx.array]] = (
make_logits_processors(
repetition_penalty=task.repetition_penalty,
- repetition_context_size=task.repetition_context_size,
+ repetition_context_size=task.repetition_context_size
+ if task.repetition_context_size is not None
+ else 20,
+ presence_penalty=task.presence_penalty,
+ frequency_penalty=task.frequency_penalty,
)
)
if is_bench:
← 8ccfd7fc Fix some misc build issues (#1948)
·
back to Exo
·
Handle missing total_size in safetensors index files (#1956) 49670c86 →