From 8f103f33ec8ed42c293e12b6bb39e73f736b1946 Mon Sep 17 00:00:00 2001
From: noonghunna <10742901+noonghunna@users.noreply.github.com>
Date: Fri, 8 May 2026 01:02:10 +0000
Subject: [PATCH] fix: verify-full.sh broken pipe + llama-cpp DISABLE_THINKING
env hook
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Two independent fixes for two recently filed issues:
1. verify-full.sh check_patches "Broken pipe" errors (issue #101 by @a-p-l):
The three `echo "$docker_logs" | grep -q ...` checks emitted
"echo: write error: Broken pipe" on stderr after a successful match,
because grep -q closes stdin early and the upstream echo writes to a
closed pipe. Spurious noise that didn't affect functionality but
looked like a real failure. Replaced with bash here-strings
(`grep -q "..." <<< "$docker_logs"`) which feed grep's stdin
directly without the pipe race. Patch as proposed in the issue.
2. llama-cpp compose DISABLE_THINKING env hook (issue #97 by @syangsao):
Refactor llama.cpp compose entrypoint pattern to match PR #99's
vLLM hook style. Adds a bash entrypoint that conditionally appends
`--chat-template-kwargs '{"enable_thinking":false}'` when
DISABLE_THINKING=1 in compose/.env. Forces the model to emit empty
blocks → response goes straight to actual output,
no thinking-content visible in client UI. Resolves opencode's
leftover-think-block cosmetic issue when the client doesn't expose
extra_body for chat_template_kwargs.
Tradeoff: applies to ALL clients on this server instance — Hermes
agents that use thinking lose reasoning capability. Users who run
both opencode and a thinking-using client should run two server
instances (different ports) with DISABLE_THINKING=1 only on the
opencode-facing one.
Also converts `command:` from folded scalar string to YAML list
(one item per flag/value) to support the bash $@ pattern.
Co-Authored-By: Claude Opus 4.7 (1M context)
---
.../llama-cpp/compose/docker-compose.yml | 61 +++++++++++++++----
scripts/verify-full.sh | 10 ++-
2 files changed, 55 insertions(+), 16 deletions(-)
diff --git a/models/qwen3.6-27b/llama-cpp/compose/docker-compose.yml b/models/qwen3.6-27b/llama-cpp/compose/docker-compose.yml
index 23994d9c..48c265e4 100644
--- a/models/qwen3.6-27b/llama-cpp/compose/docker-compose.yml
+++ b/models/qwen3.6-27b/llama-cpp/compose/docker-compose.yml
@@ -46,6 +46,11 @@
# Override to `auto` to get separate `reasoning_content` field
# (Qwen3.6 thinking trace) — useful for clients that render
# reasoning_content (most don't). Issue: club-3090#97.
+# DISABLE_THINKING set to 1 in .env to add `--chat-template-kwargs '{"enable_thinking":false}'`
+# which forces empty blocks in responses. Useful for
+# clients (e.g. opencode) that display content as the response.
+# Tradeoff: applies to ALL clients on this server — Hermes/agents that
+# use thinking lose reasoning capability. Issue: club-3090#97.
# PORT host port (default: 8020)
# CUDA_VISIBLE_DEVICES which GPU to use (default: 0)
#
@@ -68,19 +73,49 @@ services:
- "${PORT:-8020}:8080"
volumes:
- "${MODEL_DIR:-../../../../models-cache}:/models:ro"
- command: >-
- --host 0.0.0.0
- --port 8080
- -m /models/${GGUF_FILE:-qwen3.6-27b/unsloth-q3kxl/Qwen3.6-27B-UD-Q3_K_XL.gguf}
- --mmproj /models/${MMPROJ_FILE:-qwen3.6-27b/mmproj-F16.gguf}
- -c ${CTX_SIZE:-262144}
- -ngl 99
- -fa on
- --cache-type-k ${KV_TYPE:-q4_0}
- --cache-type-v ${KV_TYPE:-q4_0}
- -np 1
- --jinja
- --reasoning-format ${REASONING_FORMAT:-none}
+ entrypoint:
+ - bash
+ - -c
+ - |
+ set -e
+ # DISABLE_THINKING=1 in compose/.env appends --chat-template-kwargs to disable
+ # Qwen3 thinking server-side. Forces the chat template to insert empty
+ # blocks → output goes straight to the response. Useful for
+ # clients (e.g. opencode) that display content as the response.
+ # Tradeoff: applies to ALL clients on this server instance — Hermes/agents that
+ # use thinking lose reasoning capability. See docs/HARDWARE.md and disc club-3090#97.
+ # Note: $$VAR is YAML-escape for $VAR (compose passes literal $ to bash).
+ EXTRA_ARGS=()
+ if [ "$${DISABLE_THINKING:-0}" = "1" ]; then
+ EXTRA_ARGS+=("--chat-template-kwargs" '{"enable_thinking":false}')
+ echo "[entrypoint] DISABLE_THINKING=1 — chat template will produce empty "
+ fi
+ exec llama-server "$$@" "$${EXTRA_ARGS[@]}"
+ - --
+ command:
+ - --host
+ - 0.0.0.0
+ - --port
+ - "8080"
+ - -m
+ - /models/${GGUF_FILE:-qwen3.6-27b/unsloth-q3kxl/Qwen3.6-27B-UD-Q3_K_XL.gguf}
+ - --mmproj
+ - /models/${MMPROJ_FILE:-qwen3.6-27b/mmproj-F16.gguf}
+ - -c
+ - ${CTX_SIZE:-262144}
+ - -ngl
+ - "99"
+ - -fa
+ - "on"
+ - --cache-type-k
+ - ${KV_TYPE:-q4_0}
+ - --cache-type-v
+ - ${KV_TYPE:-q4_0}
+ - -np
+ - "1"
+ - --jinja
+ - --reasoning-format
+ - ${REASONING_FORMAT:-none}
deploy:
resources:
reservations:
diff --git a/scripts/verify-full.sh b/scripts/verify-full.sh
index 2eefc10d..ac2aec72 100755
--- a/scripts/verify-full.sh
+++ b/scripts/verify-full.sh
@@ -153,12 +153,16 @@ check_patches() {
# club-3090#29. We grep -q each anchor in priority order on the full log.
local docker_logs
docker_logs="$(docker logs "${CONTAINER}" 2>&1)"
- if echo "$docker_logs" | grep -q "\[Genesis\] FAILED"; then
+ # Use here-strings instead of pipes — when grep -q matches early it closes
+ # stdin, and the upstream `echo` then writes to a closed pipe → "Broken pipe"
+ # on stderr (issue #101 by @a-p-l). Here-strings feed the variable directly
+ # to grep without the pipe race.
+ if grep -q "\[Genesis\] FAILED" <<< "$docker_logs"; then
fail "Genesis apply_all reported FAILED patch(es)" \
"Inspect: docker logs ${CONTAINER} 2>&1 | grep -E 'Genesis.*FAILED' | head"
- elif echo "$docker_logs" | grep -q "apply_all elapsed"; then
+ elif grep -q "apply_all elapsed" <<< "$docker_logs"; then
pass "Genesis patches applied (apply_all completed clean)"
- elif echo "$docker_logs" | grep -q "\[Genesis\] applied:"; then
+ elif grep -q "\[Genesis\] applied:" <<< "$docker_logs"; then
pass "Genesis patches applied (partial log — apply_all may still be running)"
else
skip "no Genesis marker in logs (container restarted, or Genesis not loaded)"