Upgrade Beta 1 to GSQ-RCO IQ3_S

This commit is contained in:
Mikei386 committed 2026-09-08 11:46:52 +02:00
1 parent 5746ac0e2c
commit f7ff14a1ce
7 files changed
+31 -21

No files matched your search

+5 -4
View File
@@ -235,9 +235,10 @@ services:
- --spec-draft-p-min
- "0.05"
# Beta 1 keeps the complete GSQ-RCO text model and KV cache on the RTX 5080.
# Only the multimodal projector runs on the RTX 3060. The measured hard
# boundary is 196608 tokens; 192K deliberately retains runtime headroom.
# Beta 1 keeps the complete GSQ-RCO IQ3_S text model and KV cache on the RTX
# 5080. Only the multimodal projector runs on the RTX 3060. The larger IQ3_S
# build starts conservatively at 112K until its hard context boundary has
# been measured; the former IQ3_XXS result does not transfer to this file.
llama-beta1:
<<: *llama-common
container_name: mike-ai-llama-beta1
@@ -258,7 +259,7 @@ services:
- --alias
- qwen-beta-1
- --ctx-size
- "${BETA1_CONTEXT:-192000}"
- "${BETA1_CONTEXT:-112000}"
- --flash-attn
- "on"
- --cache-type-k