Files
Sylpheed/docker/agent/sylph-agent
Sylpheed RE agent ddd220fa94 agent: stop the launcher claiming it cannot push when it can
The `loose` footer announced "cannot push -- no git credentials are mounted"
unconditionally. It was written before the credential mount existed and then
went stale, so with credentials correctly mounted it told the operator their
work would be lost -- the exact failure the mount was added to prevent.

It now reports what is actually true, and keeps the real warning for the case
where the file genuinely is missing.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-28 17:22:02 +02:00

308 lines
15 KiB
Bash
Executable File

#!/usr/bin/env bash
# Host-side launcher for the Sylpheed RE agent container.
#
# Caps the container at HALF the machine's CPUs and memory, computed at run time
# so it stays half on whatever box it lands on.
#
# ./sylph-agent build build (or rebuild) the image
# ./sylph-agent shell interactive shell in the container
# ./sylph-agent agent [prompt] Claude Code, --dangerously-skip-permissions
# ./sylph-agent loose [task] turn it loose: detached, /loop, self-paced
# ./sylph-agent logs [-f] what the loose agent is doing
# ./sylph-agent remote print the Remote Control link (chat from anywhere)
# ./sylph-agent attach attach to the loose agent's session
# ./sylph-agent run <cmd...> one-shot command
# ./sylph-agent stop stop it
#
# Environment:
# SYLPH_PROJECT host project root (default: three levels up from this file)
# SYLPH_CLAUDE_HOME host dir mounted as the agent's ~/.claude
# (default: $HOME/.claude — shares auth AND memory with you)
# SYLPH_VULKAN=sw force software Vulkan (lavapipe) even if /dev/dri exists
# SYLPH_REMOTE=0 do NOT enable Remote Control (default: enabled for `loose`)
# SYLPH_REMOTE_NAME Remote Control session name (default: sylpheed-agent)
# SYLPH_GIT_CREDENTIALS file with `https://<user>:<token>@host` for push-work
# (default: $HOME/.sylph-git-credentials)
# SYLPH_LOOP_INTERVAL fixed loop cadence, e.g. 30m (default: 45m)
# SYLPH_CPUS / SYLPH_MEM_GB override the computed half
set -euo pipefail
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
IMAGE="${SYLPH_IMAGE:-sylpheed-agent:latest}"
NAME="${SYLPH_NAME:-sylpheed-agent}"
PROJECT="${SYLPH_PROJECT:-$(cd "$HERE/../../.." && pwd)}"
# ── Half the box ─────────────────────────────────────────────────────────────
# LC_ALL=C is required, not tidiness: under a locale with a comma decimal
# separator (de_DE and friends) awk prints "6,0" and docker rejects it as
# --cpus with "failed to parse as a rational number".
HOST_CPUS=$(nproc)
HOST_MEM_KB=$(awk '/MemTotal/{print $2}' /proc/meminfo)
CPUS="${SYLPH_CPUS:-$(LC_ALL=C awk -v c="$HOST_CPUS" 'BEGIN{printf "%.1f", c/2}')}"
MEM_GB="${SYLPH_MEM_GB:-$(LC_ALL=C awk -v m="$HOST_MEM_KB" 'BEGIN{printf "%d", m/1048576/2}')}"
[ "$MEM_GB" -lt 2 ] && MEM_GB=2
# /dev/shm holds the emulator's guest memory (gmem.py reads it there). Docker's
# 64 MB default is far too small for a 512 MB console address space, and the
# failure is an obscure mmap error rather than an out-of-space message. tmpfs
# pages count against the memory cap, so take a third of it and no more.
SHM_GB=$(( MEM_GB / 3 )); [ "$SHM_GB" -lt 1 ] && SHM_GB=1
CLAUDE_JSON="${SYLPH_CLAUDE_JSON:-$HOME/.claude.json}"
if [ ! -f "$CLAUDE_JSON" ]; then
echo "==> WARNING: $CLAUDE_JSON does not exist." >&2
echo " Docker would create a DIRECTORY at that path inside the container," >&2
echo " and Claude Code would fail confusingly. Run \`claude\` once on the" >&2
echo " host first, or set SYLPH_CLAUDE_JSON." >&2
fi
usage() { sed -n '2,20p' "${BASH_SOURCE[0]}" | sed 's/^# \{0,1\}//'; exit "${1:-0}"; }
docker_args() {
local -n _out=$1
_out=(
--name "$NAME"
--hostname sylph-agent
# ── the cap ──
--cpus "$CPUS"
--memory "${MEM_GB}g"
--memory-swap "${MEM_GB}g" # no swap escape hatch: a swapping build
# thrashes the whole host, which is the
# failure this cap exists to prevent
--pids-limit 4096
# /dev/shm as an EXEC-capable tmpfs, not --shm-size. Docker's default mounts
# it `noexec`, and xenia maps its JIT code cache out of a shm file at a fixed
# address — so with noexec it dies at startup with "Unable to allocate code
# cache generated code storage / Cannot initalize processor", which reads
# like an address-space clash rather than a mount flag.
--tmpfs "/dev/shm:rw,exec,nosuid,nodev,size=${SHM_GB}g"
# Dynamic RE needs to attach to a live process: without SYS_PTRACE, gdb and
# strace are installed but inert ("Could not attach to process"), and the
# container's whole reason for existing is watching the emulator run.
# Docker's default seccomp profile also blocks calls the JIT and the guest
# memory mapper rely on.
--cap-add SYS_PTRACE
--security-opt seccomp=unconfined
--security-opt apparmor=unconfined
# ── project ──
# Mounted TWICE, at the same path the host uses and at /work. The host path
# is what makes Claude Code's memory carry over: its per-project state key is
# derived from the working directory, so running at /work would give the
# agent an empty `-work` project instead of the accumulated
# `-home-fabi-RE-Project-Sylpheed` one. /work stays because the toolkit
# scripts and every doc here refer to it.
-v "$PROJECT:$PROJECT"
-v "$PROJECT:/work"
-e "PROJECT_DIR=/work"
# ── claude ──
# The state dir is shared read-write: credentials live in
# .claude/.credentials.json, so token refresh needs to write, and this is
# also what carries the project memory across.
-v "${SYLPH_CLAUDE_HOME:-$HOME/.claude}:/sylph-home/re/.claude"
# ~/.claude.json sits BESIDE that directory and holds `hasCompletedOnboarding`.
# Mounted READ-ONLY at a staging path: the entrypoint copies it to
# ~/.claude.json and stamps onboarding as done. Sharing the file directly
# would (a) re-run the first-run theme wizard whenever the container's
# Claude Code version differs from the host's — a silent hang an unattended
# agent never gets past — and (b) let the container rewrite your host config.
-v "${SYLPH_CLAUDE_JSON:-$HOME/.claude.json}:/sylph-home/re/.claude.host.json:ro"
# persistent build caches, so a container restart is not a rebuild
-v sylph-agent-cargo:/sylph-home/re/.cargo
-v sylph-agent-target:/sylph-home/re/target-container
-v sylph-agent-canary-build:/sylph-home/re/canary-build
)
# ── Vulkan SDK ──
# xenia's shader step calls `spirv-opt --canonicalize-ids`, which Ubuntu's
# packaged SPIRV-Tools (v2025.1) does not have — the build then dies ~500
# objects in. The LunarG SDK has it. Mounting the host's copy at the same path
# is cheaper than baking a 200 MB SDK into the image AND guarantees the
# container produces byte-identical shaders to the host build.
SDK="${VULKAN_SDK:-}"
if [ -z "$SDK" ]; then
SDK=$(ls -d "$HOME"/vulkan-sdk/*/x86_64 2>/dev/null | sort -V | tail -1 || true)
fi
if [ -n "$SDK" ] && [ -x "$SDK/bin/spirv-opt" ]; then
_out+=(-v "$SDK:$SDK:ro" -e "VULKAN_SDK=$SDK")
else
echo "==> NOTE: no Vulkan SDK found on the host. Building Canary's shaders" >&2
echo " needs spirv-opt with --canonicalize-ids (LunarG SDK); Ubuntu's" >&2
echo " packaged SPIRV-Tools is too old. Running is unaffected." >&2
fi
# ── git push ──
# Read-only, and only ever used by `push-work`, which refuses anything but an
# auto/* branch and never force-pushes. Without this the agent's work only
# exists inside the container and dies with it.
GITCRED="${SYLPH_GIT_CREDENTIALS:-$HOME/.sylph-git-credentials}"
if [ -f "$GITCRED" ]; then
_out+=(-v "$GITCRED:/sylph-home/re/.git-credentials:ro")
else
echo "==> NOTE: no git credentials at $GITCRED — the agent cannot push," >&2
echo " so its work will be lost if the container is destroyed. Create it" >&2
echo " with a single line and chmod 600:" >&2
echo " https://<user>:<token>@git.mc02.dev" >&2
echo " or point SYLPH_GIT_CREDENTIALS elsewhere." >&2
fi
[ -n "${ANTHROPIC_API_KEY:-}" ] && _out+=(-e "ANTHROPIC_API_KEY=$ANTHROPIC_API_KEY")
[ -n "${SYLPH_VULKAN:-}" ] && _out+=(-e "SYLPH_VULKAN=$SYLPH_VULKAN")
[ -n "${SYLPH_REMOTE:-}" ] && _out+=(-e "SYLPH_REMOTE=$SYLPH_REMOTE")
[ -n "${SYLPH_REMOTE_NAME:-}" ] && _out+=(-e "SYLPH_REMOTE_NAME=$SYLPH_REMOTE_NAME")
[ -n "${SYLPH_ISO:-}" ] && _out+=(-e "SYLPH_ISO=$SYLPH_ISO")
# ── GPU ──
# Three distinct cases, and conflating them is how you end up believing you
# have hardware Vulkan while actually running llvmpipe:
#
# NVIDIA needs the NVIDIA Container Toolkit (`--gpus all`). Passing
# /dev/dri alone does NOT work — Mesa cannot drive an NVIDIA card,
# and the proprietary userspace lives outside the image.
# Mesa (AMD/Intel) works with a plain /dev/dri passthrough plus the
# host's render/video GIDs.
# neither software Vulkan (lavapipe): correct, and slow.
if [ "${SYLPH_VULKAN:-auto}" = "sw" ]; then
_out+=(-e SYLPH_VULKAN=sw)
elif command -v nvidia-smi >/dev/null 2>&1 && nvidia-smi -L >/dev/null 2>&1; then
if docker info --format '{{json .Runtimes}}' 2>/dev/null | grep -q nvidia; then
_out+=(--gpus all)
else
echo "==> NOTE: NVIDIA GPU found but the NVIDIA Container Toolkit is not" >&2
echo " installed, so hardware Vulkan is unavailable and the container" >&2
echo " will use lavapipe (software — correct, slow). To enable it:" >&2
echo " sudo apt install nvidia-container-toolkit \\" >&2
echo " && sudo nvidia-ctk runtime configure --runtime=docker \\" >&2
echo " && sudo systemctl restart docker" >&2
_out+=(-e SYLPH_VULKAN=sw)
fi
elif [ -e /dev/dri/renderD128 ]; then
_out+=(--device /dev/dri)
for g in render video; do
gid=$(getent group "$g" | cut -d: -f3 || true)
[ -n "$gid" ] && _out+=(--group-add "$gid")
done
else
_out+=(-e SYLPH_VULKAN=sw)
fi
}
case "${1:-}" in
build)
shift
echo "==> building $IMAGE (uid $(id -u), gid $(id -g))"
exec docker build -t "$IMAGE" \
--build-arg "AGENT_UID=$(id -u)" --build-arg "AGENT_GID=$(id -g)" \
"$@" "$HERE"
;;
loose)
shift
# "On the loose": detached, self-paced, working the RE backlog until stopped.
#
# -d WITHOUT --rm so the transcript survives the container exiting; that is
# the only record of what an unattended run did. -t because /loop keeps the
# session alive and schedules its own wake-ups — a `-p`/print-mode run would
# answer once and exit, killing the loop on its first iteration.
TASK="${1:-}"
if [ -z "$TASK" ]; then
if [ -f "$HERE/loop-task.md" ]; then
TASK="$(cat "$HERE/loop-task.md")"
else
TASK="Work the RE backlog in Syplheed-Reborn/docs/re/BACKLOG.md."
fi
fi
# A FIXED interval by default, not self-pacing. Self-pacing requires the
# agent to call ScheduleWakeup itself at the end of every turn, and the one
# thing an agent deep in an experiment reliably forgets is the bookkeeping
# after it. With an interval the harness owns the cadence and a forgotten
# wakeup cannot end the run. Set SYLPH_LOOP_INTERVAL= (empty) to self-pace.
INTERVAL="${SYLPH_LOOP_INTERVAL-45m}"
declare -a ARGS; docker_args ARGS
ARGS+=(-e SYLPH_AUTONOMOUS=1 -w "$PROJECT")
echo "==> loose | cpus=$CPUS mem=${MEM_GB}g shm=${SHM_GB}g"
echo "==> project: $PROJECT (mounted at its own path, so memory carries over)"
echo "==> pacing: ${INTERVAL:-self-paced}"
docker rm -f "$NAME" >/dev/null 2>&1 || true
docker run -d -i -t "${ARGS[@]}" "$IMAGE" "/loop ${INTERVAL:+$INTERVAL }$TASK" >/dev/null
echo
echo " running detached as '$NAME'."
echo " ./sylph-agent remote link to chat with it from anywhere"
echo " ./sylph-agent logs -f follow it"
echo " ./sylph-agent attach chat with it locally (Ctrl-P Ctrl-Q to leave it running)"
echo " ./sylph-agent stop stop it"
echo
# Report what is actually true. This line used to claim unconditionally that
# the agent could not push, which was written before credentials were
# supported and then went stale — telling the operator their work was at risk
# when it was not, which is the exact failure the credential mount fixes.
if [ -f "${SYLPH_GIT_CREDENTIALS:-$HOME/.sylph-git-credentials}" ]; then
echo " It commits to auto/* branches and publishes them with push-work,"
echo " which refuses any other branch and never force-pushes. Review with:"
echo " git -C '$PROJECT/Syplheed-Reborn' fetch origin && git log --oneline origin/auto/..."
else
echo " It commits to auto/* branches and CANNOT PUSH — no git credentials"
echo " are mounted, so its work dies with the container. Review it with:"
echo " git -C '$PROJECT/Syplheed-Reborn' log --oneline auto/..."
fi
;;
logs) shift; exec docker logs "$@" "$NAME" ;;
remote)
# Fish the Remote Control link out of the session's own output. Claude Code
# prints it once, when the session registers with your account — which takes
# a minute or two after launch, so this waits rather than answering "not
# found" to a question that is really "not yet".
printf 'waiting for the session to register' >&2
for i in $(seq 1 90); do
url=$(docker logs "$NAME" 2>&1 \
| sed 's/\x1b\[[0-9;]*[a-zA-Z]//g; s/\r//g' \
| grep -oE 'https://claude\.ai/code/[A-Za-z0-9_-]+' | tail -1)
if [ -n "${url:-}" ]; then
printf '\n' >&2
echo "$url"
exit 0
fi
docker ps -q -f "name=$NAME" | grep -q . || {
printf '\n' >&2
echo "container is not running — start it with: ./sylph-agent loose" >&2
exit 1
}
printf '.' >&2
sleep 4
done
printf '\n' >&2
echo "no Remote Control URL after 6 minutes." >&2
echo " Launched with SYLPH_REMOTE=0? Or check: ./sylph-agent logs | tail" >&2
exit 1
;;
shell|agent|run)
mode=$1; shift
declare -a ARGS; docker_args ARGS
echo "==> $mode | cpus=$CPUS mem=${MEM_GB}g shm=${SHM_GB}g (host: ${HOST_CPUS} cpus, $((HOST_MEM_KB/1048576))g)"
echo "==> project: $PROJECT -> /work"
docker rm -f "$NAME" >/dev/null 2>&1 || true
# Allocate a TTY only when stdin actually is one: `docker run -it` fails
# outright ("cannot attach stdin to a TTY-enabled container") under a
# pipeline or a CI runner, which is exactly where `run` gets used.
TTY=(-i); [ -t 0 ] && TTY=(-it)
case "$mode" in
shell) exec docker run --rm "${TTY[@]}" "${ARGS[@]}" "$IMAGE" bash ;;
agent)
# The flag the user asked for. Refused under root, which is why the
# image runs as an unprivileged `agent` user.
ARGS+=(-e SYLPH_AUTONOMOUS=1)
exec docker run --rm "${TTY[@]}" "${ARGS[@]}" "$IMAGE" "$@"
;;
run) exec docker run --rm "${TTY[@]}" "${ARGS[@]}" "$IMAGE" "$@" ;;
esac
;;
stop) exec docker rm -f "$NAME" ;;
doctor)
declare -a ARGS; docker_args ARGS
exec docker run --rm "${ARGS[@]}" "$IMAGE" sylph-doctor
;;
""|-h|--help) usage 0 ;;
*) echo "unknown command: $1" >&2; usage 2 ;;
esac