aboutsummaryrefslogtreecommitdiffstats
path: root/scripts/servers.sh
diff options
context:
space:
mode:
authorelvis <elvis@claros.ar>2026-09-26 20:20:19 -0300
committerelvis <elvis@claros.ar>2026-09-26 20:20:19 -0300
commit8518a63f55153e7f45fd49ad6caff5555f4e374f (patch)
tree636684eea3fa6f35d78282ab95eb687ef49154af /scripts/servers.sh
parent69de76dc9cbedc6092d1e5ce84094a8030de1470 (diff)
downloadasist-p-main.tar.gz
asist-p-main.zip
Translate code, comments, logs and terminal UI to English; add English README; rename scriptsHEADmain
Diffstat (limited to 'scripts/servers.sh')
-rwxr-xr-xscripts/servers.sh84
1 files changed, 84 insertions, 0 deletions
diff --git a/scripts/servers.sh b/scripts/servers.sh
new file mode 100755
index 0000000..0cf662a
--- /dev/null
+++ b/scripts/servers.sh
@@ -0,0 +1,84 @@
+#!/usr/bin/env bash
+# Starts or stops the two servers on their own, without the assistant.
+#
+# Handy while developing: loading the models takes more than a minute, and
+# this way the Rust binary can be restarted as often as needed without paying
+# for it again. The assistant detects they are already listening and reuses them.
+#
+# scripts/servers.sh start | stop | status
+set -euo pipefail
+cd "$(dirname "$0")/.."
+
+LOGS="${ASIST_LOGS:-logs}"
+LLM_PORT="${ASIST_LLM_PORT:-${ASIST_LLM_PUERTO:-8012}}"
+TTS_PORT="${ASIST_TTS_PORT:-${ASIST_TTS_PUERTO:-8013}}"
+mkdir -p "$LOGS"
+
+alive() { curl -sf -m 2 "http://127.0.0.1:$1/health" >/dev/null 2>&1; }
+
+wait_for() { # port name seconds
+ local end=$(( SECONDS + $3 ))
+ while [ $SECONDS -lt $end ]; do
+ alive "$1" && { echo " $2 ready"; return 0; }
+ sleep 1
+ done
+ echo " $2 did NOT answer within $3 s; see $LOGS/$2.log" >&2
+ return 1
+}
+
+start() {
+ if alive "$LLM_PORT"; then
+ echo " llama-server was already up"
+ else
+ echo " starting llama-server…"
+ nohup vendor/llama.cpp/build/bin/llama-server \
+ --model models/Qwen3.5-2B.Q5_K_M.gguf \
+ --mmproj models/mmproj-BF16.gguf \
+ --host 127.0.0.1 --port "$LLM_PORT" \
+ --threads 10 --threads-batch 10 \
+ --batch-size 512 --ubatch-size 256 \
+ --gpu-layers 10 --split-mode layer --tensor-split 1 --main-gpu 0 \
+ --no-mmap --ctx-size 8192 --parallel 2 --cache-ram 6144 \
+ --rope-freq-base 1000000 --rope-freq-scale 0.25 \
+ --jinja --chat-template-file config/qwen35-no-think.jinja \
+ > "$LOGS/llama-server.log" 2>&1 &
+ fi
+
+ if alive "$TTS_PORT"; then
+ echo " tts-server was already up"
+ else
+ echo " starting tts-server…"
+ # --codec-chunk-dur 1.0 is what makes audio come out in blocks as it is
+ # generated instead of all at the end. See docs/RENDIMIENTO.md.
+ nohup vendor/qwentts.cpp/build/tts-server \
+ --model models/qwen-talker-1.7b-base-Q8_0.gguf \
+ --codec models/qwen-tokenizer-12hz-Q8_0.gguf \
+ --host 127.0.0.1 --port "$TTS_PORT" \
+ --lang spanish --codec-chunk-dur 1.0 \
+ > "$LOGS/tts-server.log" 2>&1 &
+ fi
+
+ wait_for "$LLM_PORT" llama-server 240
+ wait_for "$TTS_PORT" tts-server 240
+}
+
+stop() {
+ # SIGTERM, not SIGKILL: they need a chance to release the GPU.
+ pkill -TERM -f 'llama-server .*--port '"$LLM_PORT" 2>/dev/null && echo " llama-server stopped" || true
+ pkill -TERM -f 'tts-server .*--port '"$TTS_PORT" 2>/dev/null && echo " tts-server stopped" || true
+}
+
+status() {
+ alive "$LLM_PORT" && echo " llama-server up :$LLM_PORT" || echo " llama-server down"
+ alive "$TTS_PORT" && echo " tts-server up :$TTS_PORT" || echo " tts-server down"
+ command -v nvidia-smi >/dev/null && \
+ nvidia-smi --query-gpu=memory.used,memory.total --format=csv,noheader | sed 's/^/ GPU: /'
+}
+
+# The Spanish verbs (arrancar/parar/estado) are still accepted.
+case "${1:-status}" in
+ start|arrancar) start ;;
+ stop|parar) stop ;;
+ status|estado) status ;;
+ *) echo "usage: $0 {start|stop|status}" >&2; exit 1 ;;
+esac