Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
92 changes: 75 additions & 17 deletions deploy/setup-jetson.sh
Original file line number Diff line number Diff line change
Expand Up @@ -92,21 +92,73 @@ else
fi
fi

# 5. Check llama.cpp.
echo "[5/6] Checking llama.cpp..."
if [ -f "$GENIEPOD_DIR/bin/llama-server" ]; then
echo " OK: llama-server"
else
echo " NOT FOUND: llama-server"
echo ""
echo " Build and install llama.cpp with CUDA:"
echo " git clone https://github.com/ggml-org/llama.cpp.git"
echo " cd llama.cpp"
echo " cmake -B build -DGGML_CUDA=ON"
echo " cmake --build build -j\$(nproc)"
echo " sudo cp build/bin/llama-server $GENIEPOD_DIR/bin/"
echo ""
fi
# 5. Check the configured LLM backend binary.
# Issue #27: [services.llm].backend selects llama.cpp (default) or
# genie-ai-runtime. The setup script validates whichever binary the
# configured backend needs so opt-in installs do not fall back silently.
read_services_llm_field() {
# Read a key under [services.llm] from geniepod.toml. Strips quotes,
# trailing comments, and surrounding whitespace. Tolerates read failure.
awk -v key="$1" '
/^\[/ { section = substr($0, 2, length($0)-2); next }
section == "services.llm" {
line = $0
if (match(line, "^[[:space:]]*" key "[[:space:]]*=")) {
sub(/^[^=]*=[[:space:]]*/, "", line)
sub(/[[:space:]]*#.*$/, "", line)
gsub(/["'"'"']/, "", line)
gsub(/^[[:space:]]+|[[:space:]]+$/, "", line)
print line
exit
}
}
' "$CONFIG_DIR/geniepod.toml" 2>/dev/null || true
}

LLM_BACKEND_RAW="$(read_services_llm_field backend)"
LLM_BACKEND="$(echo "${LLM_BACKEND_RAW:-llama_cpp}" | tr '-' '_')"

case "$LLM_BACKEND" in
llama_cpp)
echo "[5/6] Checking llama.cpp backend (configured: llama_cpp)..."
if [ -f "$GENIEPOD_DIR/bin/llama-server" ]; then
echo " OK: llama-server"
else
echo " NOT FOUND: llama-server"
echo ""
echo " Build and install llama.cpp with CUDA:"
echo " git clone https://github.com/ggml-org/llama.cpp.git"
echo " cd llama.cpp"
echo " cmake -B build -DGGML_CUDA=ON"
echo " cmake --build build -j\$(nproc)"
echo " sudo cp build/bin/llama-server $GENIEPOD_DIR/bin/"
echo ""
fi
;;
genie_ai_runtime)
echo "[5/6] Checking genie-ai-runtime backend (configured: genie_ai_runtime)..."
if [ -f "$GENIEPOD_DIR/bin/jllm-server" ]; then
echo " OK: jllm-server"
else
echo " NOT FOUND: jllm-server"
echo ""
echo " Build and install genie-ai-runtime (Jetson-tuned LLM runtime):"
echo " git clone https://github.com/GeniePod/genie-ai-runtime.git"
echo " cd genie-ai-runtime"
echo " cmake -B build -DGGML_CUDA=ON"
echo " cmake --build build -j\$(nproc)"
echo " sudo cp build/bin/jllm-server $GENIEPOD_DIR/bin/"
echo ""
echo " Or revert to llama.cpp by unsetting backend in"
echo " $CONFIG_DIR/geniepod.toml under [services.llm]."
echo ""
fi
;;
*)
echo "[5/6] WARN: unrecognized [services.llm].backend = \"$LLM_BACKEND_RAW\""
echo " Expected llama_cpp or genie_ai_runtime. Skipping backend binary check."
;;
esac

if command -v docker > /dev/null 2>&1 && docker compose version > /dev/null 2>&1; then
echo " OK: docker compose"
Expand Down Expand Up @@ -304,9 +356,15 @@ else
echo " WARN: geniepod.target unit not found — services will not auto-start"
fi

# Resolve the configured LLM systemd unit so the genie-ai-runtime opt-in
# (issue #27) enables the right unit instead of the hardcoded genie-llm.
LLM_UNIT_CONFIGURED="$(read_services_llm_field systemd_unit)"
LLM_UNIT_CONFIGURED="${LLM_UNIT_CONFIGURED:-genie-llm.service}"
LLM_UNIT_NAME="${LLM_UNIT_CONFIGURED%.service}"

# Enable core services. genie-audio runs the I2S/AHUB route setup at boot
# (no-op if /opt/geniepod/bin/genie-audio-init is missing, see ConditionPathExists).
for svc in homeassistant genie-audio genie-whisper genie-whisper-warmup genie-llm genie-llm-warmup genie-core genie-governor genie-health genie-api genie-mqtt; do
for svc in homeassistant genie-audio genie-whisper genie-whisper-warmup "$LLM_UNIT_NAME" genie-llm-warmup genie-core genie-governor genie-health genie-api genie-mqtt; do
if sudo systemctl enable "$svc.service" 2>/dev/null; then
echo " Enabled: $svc"
else
Expand All @@ -324,7 +382,7 @@ echo ""
echo "=== Setup complete ==="
echo ""
echo "Start services:"
echo " sudo systemctl start genie-llm # LLM server (wait ~10s for model load)"
echo " sudo systemctl start $LLM_UNIT_NAME # LLM server (wait ~10s for model load)"
echo " sudo systemctl start genie-core # Voice AI + chat API on :3000"
echo " sudo systemctl start genie-api # System dashboard on :3080"
echo " sudo systemctl start genie-governor"
Expand Down
36 changes: 36 additions & 0 deletions deploy/systemd/genie-ai-runtime.service
Original file line number Diff line number Diff line change
@@ -0,0 +1,36 @@
[Unit]
Description=GeniePod LLM Server (genie-ai-runtime / jllm-server)
Documentation=https://github.com/GeniePod/genie-ai-runtime
Documentation=https://github.com/GeniePod/genie-claw/issues/27
After=network.target
ConditionPathExists=/opt/geniepod/bin/jllm-server

[Service]
Type=simple
# Drop page cache before loading model — Jetson NvMap needs contiguous blocks.
ExecStartPre=/bin/sh -c 'sync && echo 3 > /proc/sys/vm/drop_caches'
ExecStart=/opt/geniepod/bin/jllm-server \
--model ${GENIEPOD_LLM_MODEL} \
--host 0.0.0.0 \
--port 8080 \
--ctx-size 2048 \
--n-gpu-layers 999 \
--threads 4 \
--parallel 1
# The jllm-server flag surface intentionally mirrors llama-server so the
# OpenAI-compatible client in genie-core works against either backend without
# code changes. Issue #27 tracks the opt-in rollout; flip
# [services.llm].backend = "genie_ai_runtime" in geniepod.toml after
# installing /opt/geniepod/bin/jllm-server.

Environment=GENIEPOD_LLM_MODEL=/opt/geniepod/models/phi-4-mini-instruct-q4_k_m.gguf
Restart=on-failure
RestartSec=5
TimeoutStartSec=120

# GPU needs full system access
ProtectSystem=no
SupplementaryGroups=video render

[Install]
WantedBy=geniepod.target
21 changes: 21 additions & 0 deletions doc/configuration.md
Original file line number Diff line number Diff line change
Expand Up @@ -201,6 +201,27 @@ Each service block has:
| --- | --- |
| `url` | Health or base URL |
| `systemd_unit` | Associated systemd unit name |
| `backend` | LLM backend selector; only meaningful for `[services.llm]` |

### `[services.llm].backend`

Opt-in selector for the local LLM runtime (issue #27).

| Value | Runtime | Notes |
| --- | --- | --- |
| `llama_cpp` (default) | `llama-server` from `llama.cpp` | Legacy path. Preserved for unchanged configs. |
| `genie_ai_runtime` | `jllm-server` from `genie-ai-runtime` | Jetson-tuned runtime, opt-in until Phase 2 default flip. |

Hyphenated aliases (`"llama-cpp"`, `"genie-ai-runtime"`) are accepted.

When flipping to `genie_ai_runtime`:

- set `systemd_unit = "genie-ai-runtime.service"` on the same block,
- install `/opt/geniepod/bin/jllm-server`,
- re-run `setup-jetson.sh` to validate the binary and enable the right unit.

`genie-core` exposes the active backend name in `/api/health` and at startup,
and `genie-ctl status` reports it.

## `[telegram]`

Expand Down
1 change: 1 addition & 0 deletions doc/deployment-and-ops.md
Original file line number Diff line number Diff line change
Expand Up @@ -26,6 +26,7 @@ The practical reason is simple:
- `deploy/systemd/genie-governor.service`
- `deploy/systemd/genie-health.service`
- `deploy/systemd/genie-llm.service`
- `deploy/systemd/genie-ai-runtime.service`
- `deploy/systemd/genie-mqtt.service`
- `deploy/systemd/genie-audio.service`
- `deploy/systemd/genie-wakeword.service`
Expand Down