[Sync] llm stuff

This commit is contained in:
Coja
2026-09-17 20:58:21 +02:00
parent ed7e34552d
commit d8f6b30e2c
52 changed files with 2233 additions and 237 deletions
+64
View File
@@ -0,0 +1,64 @@
# WM HOST OVERLAY — full-file copy of common/.config/aichat/config.yaml with the default model
# flipped to the duskadiy client: wm is not on the home LAN, so the `local` endpoint
# (192.168.0.204) is unreachable here (kept below for VPN use). Stowed with --override so it
# shadows the common file on wm only. When the common config changes, mirror the change here.
# see https://github.com/sigoden/aichat/blob/main/config.example.yaml
keybindings: vi
editor: nvim
model: duskadiy:Qwen3-Coder-30B-Instruct-UD-Q3_K_XL
# Sessions: persist REPL sessions and keep more history before summarizing.
# The default compress_threshold (4000) summarizes far too early for 24k+ windows.
save_session: true
compress_threshold: 16000
# REPL prompts show live context usage (needs max_input_tokens, set per model below)
left_prompt: '{color.green}{?session {session}{?role /}}{role}{color.cyan}{?rag @{rag}}{color.reset}> '
right_prompt: '{color.purple}{?session {consume_tokens}/{max_input_tokens} }{color.reset}'
# NOTE: temperature/top_p are intentionally unset — the LAN server applies tuned
# per-model sampling via --jinja (e.g. GLM 0.6/0.95); a global value would clobber it.
# max_input_tokens = real ctx-size (from the router) minus output headroom.
clients:
# LAN access (fast; only reachable on the home network)
- type: openai-compatible
name: local
api_base: http://192.168.0.204:11343/v1
models: &lan_models
# Speeds are benched decode t/s (clean-night sweeps) — see fl/.config/llamacpp/README.md roster.
- name: Qwen3-Coder-30B-Instruct-UD-Q3_K_XL
max_input_tokens: 30000 # ctx 32768 · ~30 t/s — main agent coder
- name: Qwen3-Coder-Next-UD-IQ3_XXS
max_input_tokens: 128000 # ctx 131072 · ~16 t/s — long sessions (128k)
- name: Qwen3.6-35B-A3B-MTP-UD-IQ3_XXS
max_input_tokens: 22000 # ctx 24576 · ~36 t/s — daily driver
supports_vision: true
- name: Qwen3.6-35B-A3B-Thinking
max_input_tokens: 22000 # ctx 24576 · ~39 t/s — hard problems
supports_vision: true
- name: Qwen3.5-9B-UD-Q6_K_XL
max_input_tokens: 30000 # ctx 32768 · ~32 t/s — quick tasks
supports_vision: true
- name: Qwen3.8-27B-UD-IQ3_XXS
max_input_tokens: 22000 # ctx 24576 · ~15 t/s — hybrid reasoner, quiet desktop only
supports_vision: true
- name: gemma-4-26B-A4B-it-UD-IQ4_XS
max_input_tokens: 22000 # ctx 24576 · ~34 t/s — quality generalist, best vision
supports_vision: true
- name: gemma-4-E4B-it-UD-Q8_K_XL
max_input_tokens: 62000 # ctx 65536 · ~57 t/s — fast generalist, long docs
supports_vision: true
- name: GLM-4.7-Flash-UD-Q4_K_XL
max_input_tokens: 22000 # ctx 24576 · ~21 t/s — quality coder
- name: gpt-oss-20b
max_input_tokens: 62000 # ctx 65536 · ~38 t/s — fast reasoning + tools
- name: gpt-oss-20b-low
max_input_tokens: 62000 # ctx 65536 · ~37 t/s — snappy answers
# Remote access via duskadiy.com (reachable from anywhere; requires an API key —
# $DUSKADIY_API_KEY, from the gitignored conf.d/secrets.fish). The LAN endpoint above is keyless.
# Same server, same model ids — reuse the list above via a YAML anchor.
- type: openai-compatible
name: duskadiy
api_base: https://llm.duskadiy.com/api/v1
models: *lan_models