Files
dots/fl/.config/llamacpp/config.ini
T
2026-08-13 01:11:41 +02:00

458 lines
31 KiB
INI
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# llama.cpp model config — AMD RX 7600 XT (ROCm) · 16 GB (17.16 GB total)
# ═════════════════════════════════════════════════════════════════════════════
# ⚠⚠ VRAM SAFETY — READ THIS. This GPU ALSO DRIVES THE DISPLAY. If a model asks for
# more VRAM than is free, ROCm does NOT OOM cleanly — it spills into system RAM
# (GTT), which starves the desktop and FREEZES THE WHOLE PC (reboot required).
# → TARGET ≥ ~2.5 GB free (≤ ~14.5 GB used). HARD FLOOR 1.5 GB (bench.py guard stops there).
# Idle desktop uses ~1.3 GB of 17.16 — but a browser/leftover model can hold 1-3 GB more,
# so measure with a CLEAN baseline (bench.py prints "baseline used"; close GPU apps first;
# sleep-idle keeps a model resident up to 5 min after last use).
# → Verify with bench.py → ledger: bench-results.md (latest) + bench-history.md (all runs).
# → If a model is tight: LOWER ctx-size, or RAISE n-cpu-moe (MoE), or use a smaller quant.
# The values below are deliberately CONSERVATIVE after freeze incidents — raise ctx
# only after bench.py confirms the headroom.
# ═════════════════════════════════════════════════════════════════════════════
# Speed notes: decode is memory-bandwidth-bound (~288 GB/s). Prefer MoE (few active
# params). For MoE that doesn't fit, n-cpu-moe offloads expert FFNs to CPU (48 GB RAM) —
# this is FASTER than letting VRAM spill to GTT, and frees VRAM. threads = 6 (5600X cores).
# Runtime: `sudo rocm-smi --setperflevel high`; `watch -n1 rocm-smi --showmeminfo vram`.
# ─────────────────────────────────────────────────────────────────────────────
# Coding models
# ─────────────────────────────────────────────────────────────────────────────
[Qwen3-Coder-Next-UD-IQ3_XXS]
# ⭐ Big-context coder: Qwen3-Coder-Next 80B-A3B (3B active), Qwen3-Next hybrid attn → cheap KV.
# Weights ~30 GB: ALL experts on CPU (n-cpu-moe 99), only attention+KV on GPU → low VRAM, safe.
# Needs a CURRENT llama.cpp build (Qwen3-Next arch). ⬇ DOWNLOAD unsloth/Qwen3-Coder-Next-GGUF:UD-IQ3_XXS.
# Froze the PC before at n-cpu-moe 48 / ctx 131072 (overcommit) — keep it conservative, verify, then raise ctx.
model = /home/anon/software/models/Qwen3-Coder-Next-UD-IQ3_XXS.gguf
ctx-size = 131072 # ✓ VERIFIED @131072/moe99 (2026-07-16): 15.7 t/s, 6.0 GB used / 11.2 free —
# full 128k at a comfortable margin. Settled.
n-gpu-layers = 99
n-cpu-moe = 99 # ALL experts on CPU — settled AGAIN 2026-08-08 after a two-round
# experiment, CLOSED: GPU experts barely move this model.
# moe56 (04:11) = NO-OP: 6.0 used ≈ moe99's 5.8, all numbers flat →
# only ~48 MoE layers exist; moe ≥ 48 ≡ all-on-CPU. (Also: the old
# "moe48/131k froze the PC" lore was ctx-KV overcommit in a pre-
# quantized-KV config, not expert placement.)
# moe40 (04:37) = 8 layers on GPU, 10.2 used / 6.9 free: prefill
# 33→36 (+9%), decode 16.4 (noise) — 4.2 GB for a rounding error.
# The shallow curve says don't chase it: cache-reuse already makes
# growing sessions incremental, and the 6 GB footprint IS the niche.
# MEASURED (07-16): moe99 = 14.3-16.4 t/s vs moe36@65536 = 16.9.
threads = 6
# no-mmap TRIED AND REVERTED 2026-07-18: llama.cpp's log hint suggested it (CPU experts +
# mmap = page-fault churn), but reading 26 GB fully resident exceeds the router's load
# patience → "failed to load". Retry only if the router ever gets a load-timeout knob.
flash-attn = on
cache-type-k = q8_0
cache-type-v = q4_0
batch-size = 2048
ubatch-size = 256 # small compute buffer = less VRAM
cache-reuse = 256
defrag-thold = 0.1
temp = 0.7 # unsloth default 1.0/top-p0.95/top-k40/min-p0.01 — lower = more deterministic
top-p = 0.95
top-k = 40
min-p = 0.01
jinja = on
sleep-idle-seconds = 300
# ─── RETIRED 2026-07-16: [DeepSeek-Coder-V2-Lite-Instruct-Q8_0] ──────────────────────────
# Mid-2024 model, outclassed by GLM-4.7-Flash + both Qwen coders. Bench kept skipping it
# (~15.3 GB fixed footprint; measured 2026-07-03 that ctx/moe knobs barely move it — only fix
# was a Q4_K_M requant, not worth it as the 4th-best coder). Removed from opencode/pi/aichat.
# File can be deleted from the models dir: DeepSeek-Coder-V2-Lite-Instruct-Q5_K_M.gguf
# ─── RETIRED 2026-08-06: [Qwen3-Coder-30B-Instruct-IQ4_XS] ───────────────────────────────
# Lost the 2026-08-06 quant race to UD-Q3_K_XL@moe10 (31.3 vs 23.0 t/s, +36%, three
# consistent sweeps; it also failed twice to load on ~3.2 GB evening baselines at moe20 =
# 14.1-14.3 used). Clients rewired to Qwen3-Coder-30B-Instruct-UD-Q3_K_XL same day — the
# extended quality trial was deliberately skipped (user call; revert path = git, one commit).
# ⚠ OWUI chats/presets saved against this id need re-picking after the restart.
# File can be deleted from the models dir: Qwen3-Coder-30B-A3B-Instruct-IQ4_XS.gguf
# Reference numbers: moe24 = 21.1 t/s, 11.9/5.3 · moe20 = 23.2 (2026-07-16), 22.0-23.0 (2026-08-05/06)
[Qwen3-Coder-30B-Instruct-UD-Q3_K_XL]
# ⭐ MAIN AGENT CODER — took the slot 2026-08-06 after winning the quant race: 31.3 t/s @moe10
# vs IQ4_XS@moe20 = 23.0 (+36%, three consistent sweeps); Q3_K_S dominated (tombstone below).
# Same 32k ctx and sampling as the retired IQ4_XS — clients only needed the id swap.
# ⚠ QUALITY TRIAL SKIPPED (user call, 2026-08-06): Q3-on-MoE damage looks like wrong API
# signatures / off-by-one logic, NOT word salad — if opencode sessions smell off, the
# IQ4_XS preset + client wiring are one `git revert` away (and the .gguf may still be on
# disk). UD protects attn/shared tensors; the experts are still 3-bit.
# RACE NUMBERS: moe24 = 22.2 · moe16 = 25.1 (07-16) / 25.5 / 26.3 · moe10 = 31.3 (15.0 used /
# 2.1 free @3.1 GB baseline; replicate 30.9 @1.8 free under a 13.9 leftover-model baseline).
model = /home/anon/software/models/Qwen3-Coder-30B-A3B-Instruct-UD-Q3_K_XL.gguf
ctx-size = 32768
n-gpu-layers = 99
n-cpu-moe = 12 # ✓ VERIFIED @moe12 (2026-08-07, 2.9 GB EVENING baseline): 27.0 t/s,
# 14.2 used / 3.0 free — loaded fine in exactly the regime that killed
# IQ4_XS@moe20 twice; the insurance works. Cost vs moe10 (31.3 clean race)
# was ~4 t/s, more than the ~1 estimated — evening compute contention
# likely eats 2-3 of that; confirmed: CLEAN (2026-08-08 02:45, 1.9
# baseline) = 29.5 t/s, 13.2 used / 4.0 free — evening cost was indeed
# contention. Settled: moe12 = 29.5 clean / 27.0 evening, always loads.
threads = 6
flash-attn = on
cache-type-k = q8_0
cache-type-v = q4_0
batch-size = 2048
ubatch-size = 1024 # ✓ SETTLED 2026-08-08: prefill 74 → 89-92 t/s (+24%), decode unchanged,
# buffer +0.5 GB. FINAL GATE PASSED (14:43 sweep, 3.0 GB dirty baseline =
# evening-class): loaded and ran at 14.6 used / 2.6 free. If a worse
# evening (3.3+) ever fails the load: revert to 512, or pay one moe
# step (12→13).
cache-reuse = 256
defrag-thold = 0.1
temp = 0.15
top-p = 0.8
top-k = 20
min-p = 0
repeat-penalty = 1.05
jinja = on
sleep-idle-seconds = 300
# ─── RETIRED 2026-08-06: [Qwen3-Coder-30B-Instruct-Q3_K_S] ───────────────────────────────
# Dominated in the quant race: 30.9 t/s @moe8 (15.2 used / 2.0 free, replicate 30.9) = a tie
# with UD-Q3_K_XL@moe10 (31.3) at equal headroom, from a strictly worse quant (plain Q3_K_S,
# no UD-protected attn/shared tensors). No axis where it wins. Was never client-wired.
# File can be deleted from the models dir: Qwen3-Coder-30B-A3B-Instruct-Q3_K_S.gguf
# ─────────────────────────────────────────────────────────────────────────────
# General Qwen models (35B-A3B is fast MoE; the dense 27Bs are inherently VRAM-tight)
# ─────────────────────────────────────────────────────────────────────────────
[Qwen3.6-35B-A3B-MTP-UD-IQ3_XXS]
# ⭐ Fast NON-THINKING daily driver: Qwen3.6-35B-A3B (MoE, 3B active) — the "bigger Qwen3
# Instruct". Compact precise answers, no reasoning preamble. (Alias kept for clients.)
# File is flat + named without "MTP" (unsloth MTP-repo naming). If the load log complains
# there is no MTP head in this file, comment the two spec- lines out (plain repo file).
model = /home/anon/software/models/Qwen3.6-35B-A3B-UD-IQ3_XXS.gguf
mmproj = /home/anon/software/models/mmproj-Qwen3.6-35B-A3B-F16.gguf # vision ✓ VERIFIED 2026-07-18:
# 13.9 used / 3.3 free (2.8 dirty baseline), decode unchanged at 36-37 t/s.
spec-type = draft-mtp
spec-draft-n-max = 2
reasoning-budget = 0 # hard-disable thinking → instant compact answers. Delete to allow /think.
# ⚠ if startup says "option 'reasoning_budget' not recognized" → remove this line.
ctx-size = 24576 # raised from 16384 (bench @16384: 4.8 GB free) — bench to confirm ≥2.5 free
n-gpu-layers = 99
n-cpu-moe = 16 # MEASURED @24: 19.8 t/s (MTP working!), 10.6 used / 6.6 free → pull experts onto GPU
threads = 6
flash-attn = on
cache-type-k = q8_0
cache-type-v = q4_0
batch-size = 2048
ubatch-size = 512
cache-reuse = 256
defrag-thold = 0.1
temp = 0.7
top-p = 0.8
top-k = 20
min-p = 0
repeat-penalty = 1.05
jinja = on
sleep-idle-seconds = 120
[Qwen3.6-35B-A3B-Thinking]
# Same GGUF as above with reasoning ENABLED (no reasoning-budget line) — fills the "thinking
# model" slot at zero disk cost. Qwen thinking-mode sampling: temp 0.6 / top-p 0.95, no
# repeat-penalty. Slower per answer (reasoning preamble) — use for hard problems.
model = /home/anon/software/models/Qwen3.6-35B-A3B-UD-IQ3_XXS.gguf
mmproj = /home/anon/software/models/mmproj-Qwen3.6-35B-A3B-F16.gguf # vision ✓ VERIFIED 2026-07-18:
# 13.9 used / 3.3 free (2.8 dirty baseline), decode unchanged at 36-37 t/s.
spec-type = draft-mtp
spec-draft-n-max = 2
ctx-size = 24576 # thinking eats ctx — same footprint as the non-thinking preset
n-gpu-layers = 99
n-cpu-moe = 16
threads = 6
flash-attn = on
cache-type-k = q8_0
cache-type-v = q4_0
batch-size = 2048
ubatch-size = 512
cache-reuse = 256
defrag-thold = 0.1
temp = 0.6
top-p = 0.95
top-k = 20
min-p = 0
jinja = on
sleep-idle-seconds = 120
# ─── RETIRED 2026-07-03: [Qwen3.6-27B-UD-IQ3_XXS] ────────────────────────────────────────
# The 27B files (plain AND MTP) were deleted from the models dir — preset removed so the
# router doesn't advertise a dead model. To bring it back: re-download (MTP-repo files are
# named WITHOUT "MTP"), restore the preset, re-add the id to opencode/pi/aichat lists.
# ─── RETIRED 2026-07-16: [Qwen3.5-27B-UD-Q3_K_XL] ────────────────────────────────────────
# Dense 27B: 16.2 GB used / 0.9 free even @8192 (under the 1.5 floor) and ~10 t/s ceiling on
# this card. Role covered by Qwen3.6-35B-A3B (MoE, 2× faster). Removed from opencode/pi/aichat.
# File can be deleted from the models dir: Qwen3.5-27B-UD-Q3_K_XL.gguf
[Qwen3.5-9B-UD-Q6_K_XL]
# Small fast Qwen + MTP. Replaces Qwen3.5-9B-Q8_0 (retired 2026-07-16: Q8 was quality-overkill
# and bandwidth-slow at 17.2 t/s — delete Qwen3.5-9B-Q8_0.gguf; also delete the interim
# Qwen3.5-4B-UD-Q6_K_XL.gguf, superseded by this).
# ⬇ FILE MUST COME FROM unsloth/Qwen3.5-9B-MTP-GGUF (8.99 GB — MTP head embedded, filename
# identical to the plain repo!). If the load log complains there is no MTP head in this
# file, it's the plain-repo file — comment the two spec- lines out.
# ✓ VERIFIED (2026-07-16): 32.2 t/s (MTP working — nearly 2× the Q8's 17.2), prefill 83,
# 12.4 used / 4.8 free @32768. TTFT ~3.7 s = the model thinking (reasoning streams first).
model = /home/anon/software/models/Qwen3.5-9B-UD-Q6_K_XL.gguf
mmproj = /home/anon/software/models/Qwen3.5-9B-MTP-mmproj-F16.gguf # vision (from the MTP repo)
spec-type = draft-mtp
spec-draft-n-max = 3 # ⚗→✓ KEPT (04:11): 33.8 t/s vs 32.0-33.1 @n=2 — small real win.
# Revert to 2 if a future sweep drops below 32.
ctx-size = 32768 # raise toward 65536 once bench confirms free ≥2.5 GB
n-gpu-layers = 99
flash-attn = on
cache-type-k = q8_0
cache-type-v = q4_0 # q4 V to save VRAM
batch-size = 2048
ubatch-size = 512
cache-reuse = 256
defrag-thold = 0.1
temp = 0.7
top-p = 0.8
top-k = 20
min-p = 0
repeat-penalty = 1.05
jinja = on
sleep-idle-seconds = 60
# ─── RETIRED 2026-07-16: [Qwen3.6-27B-Uncensored-HauhauCS-Aggressive-IQ3_M] ──────────────
# Dense: 10 t/s, 2.0 GB free (under the 2.5 target) and unfixable — bandwidth-bound, smaller
# quant would cost the quality that justified it. Removed from opencode/pi/aichat.
# File can be deleted from the models dir: Qwen3.6-27B-Uncensored-HauhauCS-Aggressive-IQ3_M.gguf
# ─────────────────────────────────────────────────────────────────────────────
# Gemma models
# ─────────────────────────────────────────────────────────────────────────────
[gemma-4-26B-A4B-it-UD-IQ4_XS]
# MoE (4B active) + MTP + vision. ~13.5 GB weights → tight. Projector on CPU (no-mmproj-offload).
# From unsloth/gemma-4-26B-A4B-it-qat-GGUF: mmproj-F16.gguf + mtp-gemma-4-26B-A4B-it.gguf.
# ⬇ SOURCES — local name ← upstream (PLAIN repo unsloth/gemma-4-26B-A4B-it-GGUF; the qat
# repo has NO IQ4_XS, its only main is qat-UD-Q4_K_XL 14.25 GB — see decision note below):
# gemma-4-26B-A4B-it-UD-IQ4_XS.gguf ← gemma-4-26B-A4B-it-UD-IQ4_XS.gguf (13.6 GB)
# ✓ current: the 2026-08-05 dl has the Jul 17 template
# mmproj-gemma-4-26B-A4B-it-F16.gguf ← mmproj-F16.gguf (1.19 GB)
# mtp-gemma-4-26B-A4B-it.gguf ← MTP/mtp-gemma-4-26B-A4B-it-Q8_0.gguf (462 MB)
# Keep LOCAL names stable; re-fetch the upstream names above when repos update. Pair main
# and drafter from the SAME repo — the two repos' drafters are different extractions.
# ✓ FIXED 2026-08-08 03:15: matched plain-repo sidecars cured the "Invalid input batch."
# 500s — root cause CONFIRMED as the cross-repo drafter mismatch (old qat-repo mtp under
# the new plain main). Re-benched 32.8 t/s @moe13, 12.1 used / 5.0 free, MTP ✓ — same
# speed as pre-update (32.7), template fix now in place. The *.gguf.bak sidecars are
# deletable once vision + opencode tool-calling are smoke-tested.
# (QAT quality upgrade = separate decision: main → qat-UD-Q4_K_XL, +0.7 GB, moe 13→~16, and
# a section/client id rename round. Not part of this fix.)
# ⚠ vision + MTP together is heavy — if it OOMs/freezes, add n-cpu-moe or drop MTP (keep vision).
model = /home/anon/software/models/gemma-4-26B-A4B-it-UD-IQ4_XS.gguf
mmproj = /home/anon/software/models/mmproj-gemma-4-26B-A4B-it-F16.gguf
model-draft = /home/anon/software/models/mtp-gemma-4-26B-A4B-it.gguf
spec-type = draft-mtp
spec-draft-n-max = 2
reasoning-budget = 0 # added 2026-08-09: the Jul-17 official template makes gemma-4 DELIBERATE
# on hard prompts (probe: 700 tok = 100% reasoning_content, content
# EMPTY, finish=length) — that ate pi's 4096 maxTokens cap and returned
# truncated nothing. Same fix as the 35B daily driver. ⚠ if startup says
# option not recognized, remove; if it loads but a hard probe still emits
# reasoning_content, the gemma template ignores the budget → remove and
# raise client maxTokens to 8192 instead.
ctx-size = 24576 # MEASURED @16384/moe16: 29.6 t/s, 10.7 used / 6.5 free → sliding-window KV is
# cheap, take the ctx back (clients already assume 24576)
# n_ctx_train = 262144 (08-08 load log) — huge ctx headroom exists if a
# long-context vision need ever appears; bench stepwise, free ≥ 2.5.
n-gpu-layers = 99
n-cpu-moe = 13 # MEASURED @16: 26.4 t/s, 11.9 used / 5.2 free → pull ~3 expert layers onto GPU for speed
threads = 6
flash-attn = on
jinja = on # OpenAI-style tool calling (opencode build/plan default to this preset)
cache-type-k = q8_0
cache-type-v = q4_0
batch-size = 2048
ubatch-size = 512
cache-reuse = 256
defrag-thold = 0.1
no-mmproj-offload = true # vision encoder on CPU → frees ~1 GB
temp = 1.0
top-p = 0.95
top-k = 64
sleep-idle-seconds = 60
[gemma-4-E4B-it-UD-Q8_K_XL]
# ~5-7 GB — small/fast, lots of headroom, safe at big ctx.
# Vision: the HF repo ships mmproj-{BF16,F16,F32}.gguf — use F16 (990 MB), saved AS the exact
# path below (rename it; if the path doesn't exist the model fails to load / vision silently
# missing depending on router). Verify: startup log shows the mmproj load, and an image request
# answers instead of erroring "image input is not supported".
# NOTE: repo got Google's chat-template update ~2026-05 — if the main gguf predates that, re-download it too.
model = /home/anon/software/models/gemma-4-E4B-it-UD-Q8_K_XL.gguf
mmproj = /home/anon/software/models/mmproj-gemma-4-E4B-it.gguf
model-draft = /home/anon/software/models/mtp-gemma-4-E4B-it.gguf # ✓ MEASURED (2026-07-18): 57.3 t/s —
# +50% over 38.4 without the draft, 10.5 used / 6.6 free. Fastest model on the box.
spec-type = draft-mtp
spec-draft-n-max = 2
reasoning-budget = 0 # added 2026-08-09, same reason as the 26B: gemma-4 deliberates on hard
# prompts since the Jul-17 template (this is what caused the 20-second
# aichat -e "think" preambles on 2026-08-05). Fast models answer NOW.
# ⚠ same caveats as the 26B — remove if not recognized / not honored.
ctx-size = 98304 # ⚗→✓ CLOSED 2026-08-08: 58.2-60.4 t/s, 10.3-10.6 used / 6.5-6.9 free
# @98304, and n_ctx_train = 131072 confirmed (upstream config.json
# max_position_embeddings) — 98304 is inside training range, no RoPE
# risk. 131072 is the hard ceiling if ever needed (~+0.5-1 GB KV).
n-gpu-layers = 99
flash-attn = on
jinja = on # OpenAI-style tool calling (doc-writer agent defaults to this preset)
cache-type-k = q8_0
cache-type-v = q8_0
batch-size = 2048
ubatch-size = 1024
cache-reuse = 256
defrag-thold = 0.1
temp = 1.0
top-p = 0.95
top-k = 64
sleep-idle-seconds = 30
# ─────────────────────────────────────────────────────────────────────────────
# GLM
# ─────────────────────────────────────────────────────────────────────────────
[GLM-4.7-Flash-UD-Q4_K_XL]
# 30B-A3B MoE, primary coder. Upgraded UD-Q3_K_XL (~14 GB) → UD-Q4_K_XL (17.5 GB, unsloth's
# recommended quant) 2026-07-16, benched clean same day → the old GLM-4.7-Flash-UD-Q3_K_XL.gguf
# can be deleted.
# Tuning trajectory: Q3@moe9 = 28.2 t/s → Q4@moe24 = 17.9 (measured 2026-07-16) → now moe18,
# lower stepwise toward 16 while bench keeps free ≥2.5 GB.
model = /home/anon/software/models/GLM-4.7-Flash-UD-Q4_K_XL.gguf
ctx-size = 24576 # MEASURED @16384/moe24 (Q3): 9.5 GB used, 7.6 free
# n_ctx_train = 202752 (08-08 load log) — MLA KV is compact, ctx could
# go far higher if opencode subagent sessions ever outgrow 24k.
n-gpu-layers = 99
n-cpu-moe = 22 # lowered 24→22 on 2026-08-08: both step-down gates passed post-KV-fix —
# clean sweep @24 = 20.5 t/s, 12.0 used / 5.2 free (02:45, 1.9 baseline);
# elevated-baseline load @24 = 18.7, 13.0/4.2 (@2.9). moe22 est ~21.5
# clean / ~19.5-20 evening, ~12.7 used clean / ~14.0 evening → inside the
# coder-proven evening envelope (14.2 loads). ✓ CLEAN VERIFIED (02:58,
# 2.0 baseline): 21.5 t/s, 12.8 used / 4.4 free — est was 21.5, exact.
# Evening observation pending; revert to 24 if an evening bench shows
# free < 2.5. moe20 stays gated (~14.4-14.7 evening = borderline vs the
# coder's 15.2 evening death).
# (08-05/07 "failures" were the mixed-KV rejection below, not VRAM.)
# Trajectory: Q3@moe9 = 28.2 → Q4@moe24 = 17.9 (07-16) → 18.7/20.5 post-KV-fix.
threads = 6
flash-attn = on
cache-type-k = q8_0
cache-type-v = q8_0 # was q4_0 — BROKE 2026-08-08: a llama.cpp upgrade (the one deprecating
# defrag-thold) enforces same-type K/V for this model (deepseek2/MLA
# conversion; MLA KV is compact, mixed types unsupported). Every load
# since failed at context creation regardless of VRAM. Only THIS preset
# was affected — the other k-q8/v-q4 presets are regular-attention archs.
# If q8/q8 still refuses: delete both cache-type lines (f16 default; MLA
# KV is small enough that the cost is minor at 24576 ctx).
batch-size = 2048
ubatch-size = 512
cache-reuse = 256
defrag-thold = 0.1
# MTP: GLM-4.7-Flash HAS an MTP head upstream (num_nextn_predict_layers: 1) but the GGUF
# conversion (deepseek2 arch) drops those tensors and llama.cpp draft-mtp only supports
# Qwen3.5/3.6 + Gemma 4 so far (PRs #22673/#23398). No mtp-*.gguf exists to download —
# recheck when llama.cpp lands GLM MTP support.
temp = 0.6 # unsloth rec: 0.7/top-p 1.0 (tool-calling) or 1.0/0.95 (general); no repeat-penalty
top-p = 0.95
min-p = 0.01
jinja = on
reasoning-preserve = true # ⚗ option ACCEPTED by the build (loaded + ran 20.9 t/s, 04:11 sweep).
# BEHAVIOR flag — judge multi-turn subagent quality in opencode over the
# next days; remove if reasoning bleed ever pollutes final answers.
sleep-idle-seconds = 60
# ─────────────────────────────────────────────────────────────────────────────
# Utility & candidates
# ─────────────────────────────────────────────────────────────────────────────
[Qwen3-Embedding-0.6B]
# Embeddings-only, for RAG/semantic search (OWUI etc. via /v1/embeddings) — NOT a chat model,
# so it is deliberately absent from the opencode/pi/aichat chat lists.
# Qwen3-Embedding requires last-token pooling. ~0.7 GB — tiny, loads fast.
# (Q8_0 is effectively lossless for a 0.6B embedder — the f16 file is redundant, delete it.)
# CPU-ONLY on purpose: a 0.6B embeds fast on 6 cores, and 0 VRAM means it can never stack
# with a chat model into an overcommit (and never suffers/causes eviction pressure).
model = /home/anon/software/models/Qwen3-Embedding-0.6B-Q8_0.gguf
embeddings = true
pooling = last
ctx-size = 8192
n-gpu-layers = 0
flash-attn = on
batch-size = 8192 # embed long chunks in one pass
ubatch-size = 1024
threads = 6
sleep-idle-seconds = 60
[gpt-oss-20b]
# OpenAI gpt-oss-20b — MoE 21B / 3.6B active, reasoning + tool use (enabled 2026-07-16, NOT
# BENCHED yet). OpenAI post-trained it in MXFP4, so there is exactly ONE sensible GGUF:
# ggml-org/gpt-oss-20b-GGUF → gpt-oss-20b-mxfp4.gguf (~12.1 GB). Requants are never better —
# the old gpt-oss-20b-Q3_K_M.gguf can be deleted.
# Harmony chat template → jinja required. Default KV types kept (gpt-oss attention sinks have
# had issues with quantized KV — verify before adding cache-type lines).
model = /home/anon/software/models/gpt-oss-20b-mxfp4.gguf
ctx-size = 65536 # ⚗→✓ VERIFIED @65536 (04:11): 37.3/37.4 t/s unchanged, 14.0 used / 3.2
# free — the fast long-context option is real (37 t/s vs Coder-Next 16).
# 131072 is OFF the table: 32k→65k cost +1.2 GB (KV slope steepens past
# the sliding window) → 131k would land under 2.5 free. Client caps
# raised 32k→64k on 2026-08-08.
n-gpu-layers = 99
n-cpu-moe = 4 # ✓ SETTLED. MEASURED @6: 31.3 t/s · @4 (2026-07-16): 38.1 t/s, 12.7 used /
# 4.5 free — fastest model in the lineup. TTFT (~3-7 s) is the model THINKING
# (harmony reasoning streams as reasoning_content first) — not a slow load.
threads = 6
flash-attn = on
batch-size = 2048
ubatch-size = 512
cache-reuse = 256
defrag-thold = 0.1
temp = 1.0
top-p = 1.0
jinja = on
sleep-idle-seconds = 120
[gpt-oss-20b-low]
# Same GGUF, reasoning effort LOW — zero disk cost (like the 35B-Thinking alias, inverted).
# Cuts the multi-second thinking preamble for light/interactive use; keep [gpt-oss-20b]
# (default = medium effort) for hard problems. ⚠ if startup says the option is not
# recognized, the llama.cpp build predates chat-template-kwargs — comment that line out.
model = /home/anon/software/models/gpt-oss-20b-mxfp4.gguf
chat-template-kwargs = {"reasoning_effort":"low"}
ctx-size = 65536 # ⚗→✓ VERIFIED @65536 (04:11): 37.3/37.4 t/s unchanged, 14.0 used / 3.2
# free — the fast long-context option is real (37 t/s vs Coder-Next 16).
# 131072 is OFF the table: 32k→65k cost +1.2 GB (KV slope steepens past
# the sliding window) → 131k would land under 2.5 free. Client caps
# raised 32k→64k on 2026-08-08.
n-gpu-layers = 99
n-cpu-moe = 4
threads = 6
flash-attn = on
batch-size = 2048
ubatch-size = 512
cache-reuse = 256
defrag-thold = 0.1
temp = 1.0
top-p = 1.0
jinja = on
sleep-idle-seconds = 60
# ─────────────────────────────────────────────────────────────────────────────
# Notes
# ─────────────────────────────────────────────────────────────────────────────
# n-cpu-moe N : offload the first N MoE layers' experts to CPU. Higher = less VRAM (safer) but
# slower. On this display-driving GPU, a bit of offload is FAR better than a freeze.
# Only applies to MoE models (dense models ignore it — reduce ctx instead).
# override-tensor : finer control, e.g. `blk\.(1[2-9]|[2-9][0-9])\.ffn_.*_exps=CPU`.
# After ANY change: reload the server and run bench.py — keep every "VRAM free" ≥ ~2.5 GB.