Two bugs, both mine, both found by starting the thing.
**Volume mount points must exist AND be owned by the agent before USER agent.**
Docker seeds a named volume from whatever the image has at that path, ownership
included, and creates a ROOT-OWNED directory when the path is absent. Either way
the agent cannot write, and the failure surfaced far from its cause: "clone
FAILED", with no permission error anywhere in sight. The port's own Dockerfile
already carried a comment explaining this trap, which I then walked into for
/work and /exchange.
**The clone guard checked for a .git directory, not a usable HEAD.** A clone
interrupted partway -- the container was removed while one ran -- leaves a .git
with no commits, and a presence check then skips the retry forever and hands the
agent an empty repository that looks like a checkout. It now verifies HEAD, and
clones via a temp directory so a partial result never lands in /work at all.
Also: the port launcher's path defaults still assumed the old repo root, so it
mounted no disc; and the stale /reborn notice is gone now that there is one
repository.
Verified running: both agents cloned c58196b, `share` on PATH from /work/tools,
/exchange agent-owned, canary at /canary for the decoder, disc at /disc for the
port.
326 lines
16 KiB
Bash
Executable File
326 lines
16 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# Host-side launcher for the Sylpheed RE agent container.
|
|
#
|
|
# Caps the container at HALF the machine's CPUs and memory, computed at run time
|
|
# so it stays half on whatever box it lands on.
|
|
#
|
|
# ./sylph-agent build build (or rebuild) the image
|
|
# ./sylph-agent shell interactive shell in the container
|
|
# ./sylph-agent agent [prompt] Claude Code, --dangerously-skip-permissions
|
|
# ./sylph-agent loose [task] turn it loose: detached, /loop, self-paced
|
|
# ./sylph-agent logs [-f] what the loose agent is doing
|
|
# ./sylph-agent remote print the Remote Control link (chat from anywhere)
|
|
# ./sylph-agent attach attach to the loose agent's session
|
|
# ./sylph-agent run <cmd...> one-shot command
|
|
# ./sylph-agent stop stop it
|
|
#
|
|
# Environment:
|
|
# SYLPH_PROJECT host project root (default: three levels up from this file)
|
|
# SYLPH_CLAUDE_HOME host dir mounted as the agent's ~/.claude
|
|
# (default: $HOME/.claude — shares auth AND memory with you)
|
|
# SYLPH_VULKAN=sw force software Vulkan (lavapipe) even if /dev/dri exists
|
|
# SYLPH_REMOTE=0 do NOT enable Remote Control (default: enabled for `loose`)
|
|
# SYLPH_REMOTE_NAME Remote Control session name (default: sylpheed-agent)
|
|
# SYLPH_GIT_CREDENTIALS file with `https://<user>:<token>@host` for push-work
|
|
# (default: $HOME/.sylph-git-credentials)
|
|
# SYLPH_LOOP_INTERVAL fixed loop cadence, e.g. 30m (default: 45m)
|
|
# SYLPH_CPUS / SYLPH_MEM_GB override the computed half
|
|
set -euo pipefail
|
|
|
|
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
IMAGE="${SYLPH_IMAGE:-sylpheed-agent:latest}"
|
|
NAME="${SYLPH_NAME:-sylpheed-agent}"
|
|
PROJECT="${SYLPH_PROJECT:-$(cd "$HERE/../../.." && pwd)}"
|
|
|
|
# ── Half the box ─────────────────────────────────────────────────────────────
|
|
# LC_ALL=C is required, not tidiness: under a locale with a comma decimal
|
|
# separator (de_DE and friends) awk prints "6,0" and docker rejects it as
|
|
# --cpus with "failed to parse as a rational number".
|
|
HOST_CPUS=$(nproc)
|
|
HOST_MEM_KB=$(awk '/MemTotal/{print $2}' /proc/meminfo)
|
|
# Fixed, not "half the host": half was right when this was the only agent. There
|
|
# are now two, and a Referee is planned, so the budget is split deliberately
|
|
# instead of each container claiming half of a box it shares. The decoder gets
|
|
# the larger share because it builds and drives the emulator.
|
|
CPUS="${SYLPH_CPUS:-5}"
|
|
MEM_GB="${SYLPH_MEM_GB:-6}"
|
|
[ "$MEM_GB" -lt 2 ] && MEM_GB=2
|
|
# /dev/shm holds the emulator's guest memory (gmem.py reads it there). Docker's
|
|
# 64 MB default is far too small for a 512 MB console address space, and the
|
|
# failure is an obscure mmap error rather than an out-of-space message. tmpfs
|
|
# pages count against the memory cap, so take a third of it and no more.
|
|
SHM_GB=$(( MEM_GB / 3 )); [ "$SHM_GB" -lt 1 ] && SHM_GB=1
|
|
|
|
CLAUDE_JSON="${SYLPH_CLAUDE_JSON:-$HOME/.claude.json}"
|
|
if [ ! -f "$CLAUDE_JSON" ]; then
|
|
echo "==> WARNING: $CLAUDE_JSON does not exist." >&2
|
|
echo " Docker would create a DIRECTORY at that path inside the container," >&2
|
|
echo " and Claude Code would fail confusingly. Run \`claude\` once on the" >&2
|
|
echo " host first, or set SYLPH_CLAUDE_JSON." >&2
|
|
fi
|
|
|
|
usage() { sed -n '2,20p' "${BASH_SOURCE[0]}" | sed 's/^# \{0,1\}//'; exit "${1:-0}"; }
|
|
|
|
docker_args() {
|
|
local -n _out=$1
|
|
_out=(
|
|
--name "$NAME"
|
|
--hostname sylph-agent
|
|
# ── the cap ──
|
|
--cpus "$CPUS"
|
|
--memory "${MEM_GB}g"
|
|
--memory-swap "${MEM_GB}g" # no swap escape hatch: a swapping build
|
|
# thrashes the whole host, which is the
|
|
# failure this cap exists to prevent
|
|
--pids-limit 4096
|
|
# /dev/shm as an EXEC-capable tmpfs, not --shm-size. Docker's default mounts
|
|
# it `noexec`, and xenia maps its JIT code cache out of a shm file at a fixed
|
|
# address — so with noexec it dies at startup with "Unable to allocate code
|
|
# cache generated code storage / Cannot initalize processor", which reads
|
|
# like an address-space clash rather than a mount flag.
|
|
--tmpfs "/dev/shm:rw,exec,nosuid,nodev,size=${SHM_GB}g"
|
|
# Dynamic RE needs to attach to a live process: without SYS_PTRACE, gdb and
|
|
# strace are installed but inert ("Could not attach to process"), and the
|
|
# container's whole reason for existing is watching the emulator run.
|
|
# Docker's default seccomp profile also blocks calls the JIT and the guest
|
|
# memory mapper rely on.
|
|
--cap-add SYS_PTRACE
|
|
--security-opt seccomp=unconfined
|
|
--security-opt apparmor=unconfined
|
|
# ── the repository ──
|
|
# The agent's OWN clone, in its own volume -- not a bind mount of a human's
|
|
# working tree. That arrangement bit this project three times: an agent's
|
|
# `git config --local` captured a human's commits, a credential helper
|
|
# leaked a container-only path onto the host, and a `git add -A` swept an
|
|
# agent's in-flight files into someone else's commit. Separate checkouts
|
|
# make all three impossible rather than merely discouraged.
|
|
#
|
|
# The cost, accepted knowingly: Claude Code keys its per-project memory off
|
|
# the working directory, so moving from the host path to /work starts that
|
|
# memory empty. The corpus in docs/ is the memory that matters and it
|
|
# travels with the clone.
|
|
-v "sylpheed-decoder-repo:/work"
|
|
# Xenia Canary stays a separate repository -- it is a fork tracking upstream
|
|
# and carries our instrumentation. Read-write: building probes into it is
|
|
# real work, not a side effect.
|
|
-v "${SYLPH_CANARY:-$HOME/RE Project Sylpheed/xenia-canary}:/canary"
|
|
# The shared exchange: transient files with provenance, outside git history.
|
|
-v "sylpheed-exchange:/exchange"
|
|
-e "PROJECT_DIR=/work"
|
|
-e "SYLPH_EXCHANGE=/exchange"
|
|
-e "SYLPH_AGENT=decoder"
|
|
-e "SYLPH_REPO_URL=https://git.mc02.dev/fabi/Sylpheed.git"
|
|
-e "XENIA_SRC=/canary"
|
|
# ── claude ──
|
|
# The state dir is shared read-write: credentials live in
|
|
# .claude/.credentials.json, so token refresh needs to write, and this is
|
|
# also what carries the project memory across.
|
|
-v "${SYLPH_CLAUDE_HOME:-$HOME/.claude}:/sylph-home/re/.claude"
|
|
# ~/.claude.json sits BESIDE that directory and holds `hasCompletedOnboarding`.
|
|
# Mounted READ-ONLY at a staging path: the entrypoint copies it to
|
|
# ~/.claude.json and stamps onboarding as done. Sharing the file directly
|
|
# would (a) re-run the first-run theme wizard whenever the container's
|
|
# Claude Code version differs from the host's — a silent hang an unattended
|
|
# agent never gets past — and (b) let the container rewrite your host config.
|
|
-v "${SYLPH_CLAUDE_JSON:-$HOME/.claude.json}:/sylph-home/re/.claude.host.json:ro"
|
|
# persistent build caches, so a container restart is not a rebuild
|
|
-v sylph-agent-cargo:/sylph-home/re/.cargo
|
|
-v sylph-agent-target:/sylph-home/re/target-container
|
|
-v sylph-agent-canary-build:/sylph-home/re/canary-build
|
|
)
|
|
# ── Vulkan SDK ──
|
|
# xenia's shader step calls `spirv-opt --canonicalize-ids`, which Ubuntu's
|
|
# packaged SPIRV-Tools (v2025.1) does not have — the build then dies ~500
|
|
# objects in. The LunarG SDK has it. Mounting the host's copy at the same path
|
|
# is cheaper than baking a 200 MB SDK into the image AND guarantees the
|
|
# container produces byte-identical shaders to the host build.
|
|
SDK="${VULKAN_SDK:-}"
|
|
if [ -z "$SDK" ]; then
|
|
SDK=$(ls -d "$HOME"/vulkan-sdk/*/x86_64 2>/dev/null | sort -V | tail -1 || true)
|
|
fi
|
|
if [ -n "$SDK" ] && [ -x "$SDK/bin/spirv-opt" ]; then
|
|
_out+=(-v "$SDK:$SDK:ro" -e "VULKAN_SDK=$SDK")
|
|
else
|
|
echo "==> NOTE: no Vulkan SDK found on the host. Building Canary's shaders" >&2
|
|
echo " needs spirv-opt with --canonicalize-ids (LunarG SDK); Ubuntu's" >&2
|
|
echo " packaged SPIRV-Tools is too old. Running is unaffected." >&2
|
|
fi
|
|
|
|
# ── git push ──
|
|
# Read-only, and only ever used by `push-work`, which refuses anything but an
|
|
# auto/* branch and never force-pushes. Without this the agent's work only
|
|
# exists inside the container and dies with it.
|
|
GITCRED="${SYLPH_GIT_CREDENTIALS:-$HOME/.sylph-git-credentials}"
|
|
if [ -f "$GITCRED" ]; then
|
|
_out+=(-v "$GITCRED:/sylph-home/re/.git-credentials.host:ro")
|
|
else
|
|
echo "==> NOTE: no git credentials at $GITCRED — the agent cannot push," >&2
|
|
echo " so its work will be lost if the container is destroyed. Create it" >&2
|
|
echo " with a single line and chmod 600:" >&2
|
|
echo " https://<user>:<token>@git.mc02.dev" >&2
|
|
echo " or point SYLPH_GIT_CREDENTIALS elsewhere." >&2
|
|
fi
|
|
|
|
[ -n "${ANTHROPIC_API_KEY:-}" ] && _out+=(-e "ANTHROPIC_API_KEY=$ANTHROPIC_API_KEY")
|
|
[ -n "${SYLPH_VULKAN:-}" ] && _out+=(-e "SYLPH_VULKAN=$SYLPH_VULKAN")
|
|
[ -n "${SYLPH_REMOTE:-}" ] && _out+=(-e "SYLPH_REMOTE=$SYLPH_REMOTE")
|
|
[ -n "${SYLPH_REMOTE_NAME:-}" ] && _out+=(-e "SYLPH_REMOTE_NAME=$SYLPH_REMOTE_NAME")
|
|
[ -n "${SYLPH_ISO:-}" ] && _out+=(-e "SYLPH_ISO=$SYLPH_ISO")
|
|
|
|
# ── GPU ──
|
|
# Three distinct cases, and conflating them is how you end up believing you
|
|
# have hardware Vulkan while actually running llvmpipe:
|
|
#
|
|
# NVIDIA needs the NVIDIA Container Toolkit (`--gpus all`). Passing
|
|
# /dev/dri alone does NOT work — Mesa cannot drive an NVIDIA card,
|
|
# and the proprietary userspace lives outside the image.
|
|
# Mesa (AMD/Intel) works with a plain /dev/dri passthrough plus the
|
|
# host's render/video GIDs.
|
|
# neither software Vulkan (lavapipe): correct, and slow.
|
|
if [ "${SYLPH_VULKAN:-auto}" = "sw" ]; then
|
|
_out+=(-e SYLPH_VULKAN=sw)
|
|
elif command -v nvidia-smi >/dev/null 2>&1 && nvidia-smi -L >/dev/null 2>&1; then
|
|
if docker info --format '{{json .Runtimes}}' 2>/dev/null | grep -q nvidia; then
|
|
_out+=(--gpus all)
|
|
else
|
|
echo "==> NOTE: NVIDIA GPU found but the NVIDIA Container Toolkit is not" >&2
|
|
echo " installed, so hardware Vulkan is unavailable and the container" >&2
|
|
echo " will use lavapipe (software — correct, slow). To enable it:" >&2
|
|
echo " sudo apt install nvidia-container-toolkit \\" >&2
|
|
echo " && sudo nvidia-ctk runtime configure --runtime=docker \\" >&2
|
|
echo " && sudo systemctl restart docker" >&2
|
|
_out+=(-e SYLPH_VULKAN=sw)
|
|
fi
|
|
elif [ -e /dev/dri/renderD128 ]; then
|
|
_out+=(--device /dev/dri)
|
|
for g in render video; do
|
|
gid=$(getent group "$g" | cut -d: -f3 || true)
|
|
[ -n "$gid" ] && _out+=(--group-add "$gid")
|
|
done
|
|
else
|
|
_out+=(-e SYLPH_VULKAN=sw)
|
|
fi
|
|
}
|
|
|
|
case "${1:-}" in
|
|
build)
|
|
shift
|
|
echo "==> building $IMAGE (uid $(id -u), gid $(id -g))"
|
|
exec docker build -t "$IMAGE" \
|
|
--build-arg "AGENT_UID=$(id -u)" --build-arg "AGENT_GID=$(id -g)" \
|
|
"$@" "$HERE"
|
|
;;
|
|
|
|
loose)
|
|
shift
|
|
# "On the loose": detached, self-paced, working the RE backlog until stopped.
|
|
#
|
|
# -d WITHOUT --rm so the transcript survives the container exiting; that is
|
|
# the only record of what an unattended run did. -t because /loop keeps the
|
|
# session alive and schedules its own wake-ups — a `-p`/print-mode run would
|
|
# answer once and exit, killing the loop on its first iteration.
|
|
TASK="${1:-}"
|
|
if [ -z "$TASK" ]; then
|
|
if [ -f "$HERE/../../docs/agents/decoder-loop.md" ]; then
|
|
TASK="$(cat "$HERE/../../docs/agents/decoder-loop.md")"
|
|
else
|
|
TASK="Work the RE backlog in Syplheed-Reborn/docs/re/BACKLOG.md."
|
|
fi
|
|
fi
|
|
# A FIXED interval by default, not self-pacing. Self-pacing requires the
|
|
# agent to call ScheduleWakeup itself at the end of every turn, and the one
|
|
# thing an agent deep in an experiment reliably forgets is the bookkeeping
|
|
# after it. With an interval the harness owns the cadence and a forgotten
|
|
# wakeup cannot end the run. Set SYLPH_LOOP_INTERVAL= (empty) to self-pace.
|
|
INTERVAL="${SYLPH_LOOP_INTERVAL-45m}"
|
|
declare -a ARGS; docker_args ARGS
|
|
ARGS+=(-e SYLPH_AUTONOMOUS=1 -w /work)
|
|
echo "==> loose | cpus=$CPUS mem=${MEM_GB}g shm=${SHM_GB}g"
|
|
echo "==> repo: own clone in volume sylpheed-decoder-repo -> /work"
|
|
echo "==> pacing: ${INTERVAL:-self-paced}"
|
|
docker rm -f "$NAME" >/dev/null 2>&1 || true
|
|
docker run -d -i -t "${ARGS[@]}" "$IMAGE" "/loop ${INTERVAL:+$INTERVAL }$TASK" >/dev/null
|
|
echo
|
|
echo " running detached as '$NAME'."
|
|
echo " ./sylph-agent remote link to chat with it from anywhere"
|
|
echo " ./sylph-agent logs -f follow it"
|
|
echo " ./sylph-agent attach chat with it locally (Ctrl-P Ctrl-Q to leave it running)"
|
|
echo " ./sylph-agent stop stop it"
|
|
echo
|
|
# Report what is actually true. This line used to claim unconditionally that
|
|
# the agent could not push, which was written before credentials were
|
|
# supported and then went stale — telling the operator their work was at risk
|
|
# when it was not, which is the exact failure the credential mount fixes.
|
|
if [ -f "${SYLPH_GIT_CREDENTIALS:-$HOME/.sylph-git-credentials}" ]; then
|
|
echo " It commits to auto/* branches and publishes them with push-work,"
|
|
echo " which refuses any other branch and never force-pushes. Review with:"
|
|
echo " git -C '$PROJECT/Syplheed-Reborn' fetch origin && git log --oneline origin/auto/..."
|
|
else
|
|
echo " It commits to auto/* branches and CANNOT PUSH — no git credentials"
|
|
echo " are mounted, so its work dies with the container. Review it with:"
|
|
echo " git -C '$PROJECT/Syplheed-Reborn' log --oneline auto/..."
|
|
fi
|
|
;;
|
|
|
|
logs) shift; exec docker logs "$@" "$NAME" ;;
|
|
|
|
remote)
|
|
# Fish the Remote Control link out of the session's own output. Claude Code
|
|
# prints it once, when the session registers with your account — which takes
|
|
# a minute or two after launch, so this waits rather than answering "not
|
|
# found" to a question that is really "not yet".
|
|
printf 'waiting for the session to register' >&2
|
|
for i in $(seq 1 90); do
|
|
url=$(docker logs "$NAME" 2>&1 \
|
|
| sed 's/\x1b\[[0-9;]*[a-zA-Z]//g; s/\r//g' \
|
|
| grep -oE 'https://claude\.ai/code/[A-Za-z0-9_-]+' | tail -1)
|
|
if [ -n "${url:-}" ]; then
|
|
printf '\n' >&2
|
|
echo "$url"
|
|
exit 0
|
|
fi
|
|
docker ps -q -f "name=$NAME" | grep -q . || {
|
|
printf '\n' >&2
|
|
echo "container is not running — start it with: ./sylph-agent loose" >&2
|
|
exit 1
|
|
}
|
|
printf '.' >&2
|
|
sleep 4
|
|
done
|
|
printf '\n' >&2
|
|
echo "no Remote Control URL after 6 minutes." >&2
|
|
echo " Launched with SYLPH_REMOTE=0? Or check: ./sylph-agent logs | tail" >&2
|
|
exit 1
|
|
;;
|
|
|
|
shell|agent|run)
|
|
mode=$1; shift
|
|
declare -a ARGS; docker_args ARGS
|
|
echo "==> $mode | cpus=$CPUS mem=${MEM_GB}g shm=${SHM_GB}g (host: ${HOST_CPUS} cpus, $((HOST_MEM_KB/1048576))g)"
|
|
echo "==> project: $PROJECT -> /work"
|
|
docker rm -f "$NAME" >/dev/null 2>&1 || true
|
|
# Allocate a TTY only when stdin actually is one: `docker run -it` fails
|
|
# outright ("cannot attach stdin to a TTY-enabled container") under a
|
|
# pipeline or a CI runner, which is exactly where `run` gets used.
|
|
TTY=(-i); [ -t 0 ] && TTY=(-it)
|
|
case "$mode" in
|
|
shell) exec docker run --rm "${TTY[@]}" "${ARGS[@]}" "$IMAGE" bash ;;
|
|
agent)
|
|
# The flag the user asked for. Refused under root, which is why the
|
|
# image runs as an unprivileged `agent` user.
|
|
ARGS+=(-e SYLPH_AUTONOMOUS=1)
|
|
exec docker run --rm "${TTY[@]}" "${ARGS[@]}" "$IMAGE" "$@"
|
|
;;
|
|
run) exec docker run --rm "${TTY[@]}" "${ARGS[@]}" "$IMAGE" "$@" ;;
|
|
esac
|
|
;;
|
|
|
|
stop) exec docker rm -f "$NAME" ;;
|
|
doctor)
|
|
declare -a ARGS; docker_args ARGS
|
|
exec docker run --rm "${ARGS[@]}" "$IMAGE" sylph-doctor
|
|
;;
|
|
""|-h|--help) usage 0 ;;
|
|
*) echo "unknown command: $1" >&2; usage 2 ;;
|
|
esac
|