# NOTE: the lw/ and wm/ host overlays carry full-file copies of this config — mirror changes there. # see https://github.com/sigoden/aichat/blob/main/config.example.yaml keybindings: vi editor: nvim model: local:Qwen3-Coder-30B-Instruct-UD-Q3_K_XL # Sessions: persist REPL sessions and keep more history before summarizing. # The default compress_threshold (4000) summarizes far too early for 24k+ windows. save_session: true compress_threshold: 16000 # REPL prompts show live context usage (needs max_input_tokens, set per model below) left_prompt: '{color.green}{?session {session}{?role /}}{role}{color.cyan}{?rag @{rag}}{color.reset}> ' right_prompt: '{color.purple}{?session {consume_tokens}/{max_input_tokens} }{color.reset}' # NOTE: temperature/top_p are intentionally unset — the LAN server applies tuned # per-model sampling via --jinja (e.g. GLM 0.6/0.95); a global value would clobber it. # max_input_tokens = real ctx-size (from the router) minus output headroom. clients: # LAN access (fast; only reachable on the home network) - type: openai-compatible name: local api_base: http://192.168.0.204:11343/v1 models: &lan_models # Speeds are benched decode t/s (clean-night sweeps) — see fl/.config/llamacpp/README.md roster. - name: Qwen3-Coder-30B-Instruct-UD-Q3_K_XL max_input_tokens: 30000 # ctx 32768 · ~30 t/s — main agent coder - name: Qwen3-Coder-Next-UD-IQ3_XXS max_input_tokens: 128000 # ctx 131072 · ~16 t/s — long sessions (128k) - name: Qwen3.6-35B-A3B-MTP-UD-IQ3_XXS max_input_tokens: 22000 # ctx 24576 · ~36 t/s — daily driver supports_vision: true - name: Qwen3.6-35B-A3B-Thinking max_input_tokens: 22000 # ctx 24576 · ~39 t/s — hard problems supports_vision: true - name: Qwen3.5-9B-UD-Q6_K_XL max_input_tokens: 30000 # ctx 32768 · ~32 t/s — quick tasks supports_vision: true - name: Qwen3.8-27B-UD-IQ3_XXS max_input_tokens: 22000 # ctx 24576 · ~15 t/s — hybrid reasoner, quiet desktop only supports_vision: true - name: gemma-4-26B-A4B-it-UD-IQ4_XS max_input_tokens: 22000 # ctx 24576 · ~34 t/s — quality generalist, best vision supports_vision: true - name: gemma-4-E4B-it-UD-Q8_K_XL max_input_tokens: 62000 # ctx 65536 · ~57 t/s — fast generalist, long docs supports_vision: true - name: GLM-4.7-Flash-UD-Q4_K_XL max_input_tokens: 22000 # ctx 24576 · ~21 t/s — quality coder - name: gpt-oss-20b max_input_tokens: 62000 # ctx 65536 · ~38 t/s — fast reasoning + tools - name: gpt-oss-20b-low max_input_tokens: 62000 # ctx 65536 · ~37 t/s — snappy answers # Remote access via duskadiy.com (reachable from anywhere; requires an API key — # $DUSKADIY_API_KEY, from the gitignored conf.d/secrets.fish). The LAN endpoint above is keyless. # Same server, same model ids — reuse the list above via a YAML anchor. - type: openai-compatible name: duskadiy api_base: https://llm.duskadiy.com/api/v1 models: *lan_models