Manchego / manchego_config.json
oraculumai's picture
Manchego v3: CUDA graphs off by default
53251b0 verified
Raw History Blame Contribute Delete
4.16 kB
{
"model": "Manchego",
"version": "v3",
"contract": "semif",
"served_name": "manchego-3",
"release_date": "2026-09-30",
"base_model": "Qwen/Qwen3.5-4B@851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a",
"read_by_manchego_serve": "contract, serving, served_name, release_date, model and version (manchego-serve 0.2); every other field documents the model",
"what_it_does": "state + question + closed option set in; a probability for every option out; one forward pass; no generated text",
"question_types": {
"choice": "pick one option",
"noul": "a yes/no condition; the options are the values 'true' and 'false' (rendered true first); P(true) is the answer",
"score": "distribution over ordered levels; expected level = sum(i * p_i)"
},
"contracts": {
"semif": {
"options": "2 to 16",
"codes": "A-P",
"renderer": "contract_semif.py (in oraculumai/Manchego): SemIf's direct-options-v1 prompt, a system message and one JSON user message {evidence, criterion, options}",
"trained": true,
"use_when": "2 to 16 options and a nonempty state (every training row)"
},
"state_first": {
"options": "17 to 255",
"codes": "A-Z, then a fixed list of two-letter codes, each a single token",
"renderer": "contract_v2.py (in oraculumai/Manchego)",
"trained": false,
"use_when": "17 to 255 options, or a state that is empty (\"\", {}, [] or null); v3 never trained on this prompt"
}
},
"served_policy": "contract semif: the SemIf prompt for 2 to 16 options and a nonempty state, the state-first prompt otherwise; one prompt per forward pass",
"chat_template": {
"thinking": "disabled (enable_thinking=False)",
"add_generation_prompt": true,
"messages": "semif: system + user (JSON); state_first: system + user"
},
"readout": "softmax over the offered option-code tokens at the last prompt position",
"card_numbers_temperature": 1.0,
"serving_temperature": {
"map": "temperature_map.json (TMAP-V15, schema manchego-temperature-map/2)",
"temperatures": {
"choice": 1.5,
"noul": 0.2,
"score": 1.0
},
"bound_to_weights_sha256": "2ee838433bfe278a226dc644667ad4a99ece82cc47325c7645a7dae723c1863b",
"applies_to": "the three v3 builds by weights hash, under contract semif: bf16 2ee838433bfe278a226dc644667ad4a99ece82cc47325c7645a7dae723c1863b (where it was fitted), MLX 8-bit 358b025b04001e50a065f8c87929175211264bd6182af74b67bd6caa2f639657 and MLX 4-bit e1bc5538b8dced2a857b4980dba045c2ca01db1c369aa416fc19f0f5c593e782 (applied as-is: fitted on the bf16 path's development records); manchego-serve 0.2 ships it as its default map for these hashes and gives every other model T = 1.0",
"why": "chosen on development rows to maximise an estimate of JevBench v1.5's composite; the noul temperature sharpens yes/no probabilities on purpose because v1.5 counts a yes/no answer with P(yes) between 0.20 and 0.80 as wrong, so under the map P(yes) is not a calibrated probability",
"off": "--temperature-map off: temperature 1.0 for every question type"
},
"confidence": "(K * max_p - 1) / (K - 1), K = number of offered options",
"longest_training_prompt_tokens": 6386,
"longer_prompts": "untested (manchego-serve accepts up to 32,768 tokens)",
"questions_per_prompt": 1,
"languages": [
"en"
],
"batching": "padded batches move probabilities slightly at bf16/8-bit; score one prompt at a time for exact reproducibility",
"over_limit_inputs": "not truncated and not refused by the model; prompts beyond 6,386 tokens are untested",
"serving": {
"cuda_graphs": false,
"fast_host": true
},
"serving_note": "manchego-serve 0.2 reads 'serving' only with the torch backend on CUDA. fast_host (on) is bit-for-bit the reference arithmetic with less host work. cuda_graphs is OFF: on Linux (an A10, the Docker recipe) padding prompts to the graph sizes made every prompt length slower (median 71 vs 54 ms up to 256 tokens, 275 vs 190 ms up to 1,024); on Windows (an RTX 5090) it was faster because per-call overhead dominates there. --cuda-graphs turns it on; it moved probabilities by up to 0.0385 (0.0626 across other bucket sets) with no chosen option changed on 83 invented test questions."
}