cloudyu commited on
Commit
51740a8
·
verified ·
1 Parent(s): e6b6a72

serve_decide.py: System 1 only by default for every model (thinking is opt-in)

Browse files
Files changed (1) hide show
  1. serve_decide.py +3 -3
serve_decide.py CHANGED
@@ -12,8 +12,8 @@ POST /v1/decide
12
  options list of strings (choice only)
13
  strategy choices with more than 16 options: "single" (one pass, labels A-P then Q-Z, AA, ...), "tournament"
14
  (groups of <=16 + a final of 16), "permute" (single pass over 4 option orders, averaged); default per model
15
- thinking "off", "auto" (think only when the leading option is below `threshold`), "on" (always think);
16
- default per model (GEV: "auto"; JEV-27B: "off")
17
  threshold System 1 confidence below which "auto" switches thinking on (default per model)
18
  reasoning controls, as for the base model's own chat API:
19
  chat_template_kwargs passed to the base model's chat template (e.g. Qwen3.8: {"reasoning_effort": "low"})
@@ -65,7 +65,7 @@ PROFILES = {
65
  "qwen": {"prefix": "", "image": "<|vision_start|><|image_pad|><|vision_end|>", "end_think": "</think>",
66
  "after_think": "\n\n", "strategy": "single", "threshold": 0.8, "mix": 0.5, "thinking": "off"},
67
  "gemma": {"prefix": "<bos>", "image": "<|image|>", "end_think": "<channel|>", "after_think": "",
68
- "strategy": "tournament", "threshold": 0.8, "mix": 0.5, "thinking": "auto"},
69
  }
70
  # Read the full distribution: override generation_config defaults (top_k/top_p) that would truncate processed logprobs.
71
  READ = dict(max_tokens=1, temperature=1.0, top_p=1.0, top_k=0, min_p=0.0, repetition_penalty=1.0,
 
12
  options list of strings (choice only)
13
  strategy choices with more than 16 options: "single" (one pass, labels A-P then Q-Z, AA, ...), "tournament"
14
  (groups of <=16 + a final of 16), "permute" (single pass over 4 option orders, averaged); default per model
15
+ thinking "off" (default: System 1 only), "auto" (think only when the leading option is below `threshold`),
16
+ "on" (always think)
17
  threshold System 1 confidence below which "auto" switches thinking on (default per model)
18
  reasoning controls, as for the base model's own chat API:
19
  chat_template_kwargs passed to the base model's chat template (e.g. Qwen3.8: {"reasoning_effort": "low"})
 
65
  "qwen": {"prefix": "", "image": "<|vision_start|><|image_pad|><|vision_end|>", "end_think": "</think>",
66
  "after_think": "\n\n", "strategy": "single", "threshold": 0.8, "mix": 0.5, "thinking": "off"},
67
  "gemma": {"prefix": "<bos>", "image": "<|image|>", "end_think": "<channel|>", "after_think": "",
68
+ "strategy": "tournament", "threshold": 0.8, "mix": 0.5, "thinking": "off"},
69
  }
70
  # Read the full distribution: override generation_config defaults (top_k/top_p) that would truncate processed logprobs.
71
  READ = dict(max_tokens=1, temperature=1.0, top_p=1.0, top_k=0, min_p=0.0, repetition_penalty=1.0,