| 354 | para(fp, c, "Model endpoint aliases include deepseek-v4-flash and deepseek-v4-pro; both serve the loaded GGUF."); |
| 355 | fputc('\n', fp); |
| 356 | } |
| 357 | |
| 358 | static void print_server_thinking(FILE *fp, const help_colors *c) { |
| 359 | title(fp, c, "Server Thinking Defaults"); |
| 360 | para(fp, c, "DeepSeek-compatible chat requests default to high-effort thinking."); |
| 361 | para(fp, c, "reasoning_effort=max or output_config.effort=max requests Think Max."); |
| 362 | para(fp, c, "Think Max requires --ctx >= 393216; smaller contexts use high."); |
| 363 | para(fp, c, "thinking={type:disabled}, think=false, or model=deepseek-chat selects non-thinking mode."); |
| 364 | para(fp, c, "In thinking mode, client sampling knobs are ignored like the official API."); |
| 365 | fputc('\n', fp); |
| 366 | } |
| 367 | |
| 368 | static void print_kv_cache(FILE *fp, const help_colors *c) { |
| 369 | title(fp, c, "Disk KV Cache"); |
| 370 | opt(fp, c, "--kv-disk-dir DIR", "Enable disk KV checkpoints in DIR."); |
| 371 | opt(fp, c, "--kv-disk-space-mb N", "Disk budget. Default when enabled: 4096"); |
| 372 | opt(fp, c, "--kv-cache-min-tokens N", "Do not save/load checkpoints shorter than N. Default: 512"); |
| 373 | opt(fp, c, "--kv-cache-cold-max-tokens N", "Save cold first prompts up to N tokens. 0 disables. Default: 30000"); |
no test coverage detected