Compare commits
21
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
32149603b2 | ||
|
|
f6e5cf3f05 | ||
|
|
43c7ad1478 | ||
|
|
70c0a8bf7a | ||
|
|
92d64cf252 | ||
|
|
ab2c75d0b2 | ||
|
|
00937745f7 | ||
|
|
01e72e4785 | ||
|
|
2b35aa8c16 | ||
|
|
65a3d4eb29 | ||
|
|
2b2d9da606 | ||
|
|
3741e0fef5 | ||
|
|
0e6566d903 | ||
|
|
84a67fcd0d | ||
|
|
f3cc1c4c17 | ||
|
|
7265f55f27 | ||
|
|
88f2e73474 | ||
|
|
fa4940dd7d | ||
|
|
9027fa9ad4 | ||
|
|
be37723c38 | ||
|
|
5f990dd28b |
@@ -39,13 +39,48 @@ jobs:
|
||||
MAJOR_MINOR=$(cat VERSION | tr -d '[:space:]')
|
||||
echo "Major.Minor: ${MAJOR_MINOR}"
|
||||
|
||||
# Find the latest tag matching v{MAJOR_MINOR}.N (exclude -mac, -win suffixes)
|
||||
# `|| true` so an empty grep result doesn't fail the step under pipefail.
|
||||
LATEST_TAG=$(git tag -l "v${MAJOR_MINOR}.*" --sort=-v:refname | grep -E "^v${MAJOR_MINOR}\.[0-9]+$" | head -1 || true)
|
||||
# The patch number is **one past the highest patch already used**, and
|
||||
# never a distance.
|
||||
#
|
||||
# It used to be `git rev-list --count <highest tag>..HEAD`, which is
|
||||
# not a counter at all: it measures how far HEAD has drifted from
|
||||
# whichever tag sorts highest, and that resets to zero every time a
|
||||
# tag is cut. The published history is the proof — each of these is
|
||||
# exactly what the old formula returned at the time:
|
||||
#
|
||||
# v0.4.0 -> 3 commits -> v0.4.3 looked fine
|
||||
# v0.4.3 -> 4 commits -> v0.4.4 fine by luck, 4 > 3
|
||||
# v0.4.4 -> 2 commits -> v0.4.2 went backwards
|
||||
# v0.4.4 -> 6 commits -> v0.4.6 jumped, skipping .5
|
||||
# v0.4.6 -> 3 commits -> v0.4.3 already taken; the upload failed
|
||||
#
|
||||
# Reusing a version is worse than failing to publish one: the macOS
|
||||
# and Windows steps replace assets in place, so a duplicate silently
|
||||
# rewrote a release that had been public for three days. Monotonic
|
||||
# numbering is what stops that at the source.
|
||||
#
|
||||
# Suffixed tags count too. `create-tag` is skipped when any platform
|
||||
# job fails, so a run can publish v0.4.7-mac and never create the
|
||||
# plain v0.4.7 — reading only unsuffixed tags would then hand the
|
||||
# same number out twice.
|
||||
HIGHEST=$(git tag -l "v${MAJOR_MINOR}.*" \
|
||||
| grep -E "^v${MAJOR_MINOR}\.[0-9]+(-mac|-win)?$" \
|
||||
| sed -E "s/^v${MAJOR_MINOR}\.([0-9]+).*/\1/" \
|
||||
| sort -n | tail -1 || true)
|
||||
|
||||
if [ -n "$LATEST_TAG" ]; then
|
||||
echo "Latest matching tag: ${LATEST_TAG}"
|
||||
PATCH=$(git rev-list --count "${LATEST_TAG}..HEAD")
|
||||
# A re-run of a commit that already released must not mint a new
|
||||
# version just because its own tag now exists.
|
||||
EXISTING=$(git tag --points-at HEAD \
|
||||
| grep -E "^v${MAJOR_MINOR}\.[0-9]+$" \
|
||||
| sed -E "s/^v${MAJOR_MINOR}\.([0-9]+)$/\1/" \
|
||||
| sort -n | tail -1 || true)
|
||||
|
||||
if [ -n "$EXISTING" ]; then
|
||||
echo "HEAD is already tagged v${MAJOR_MINOR}.${EXISTING} — reusing it"
|
||||
PATCH="${EXISTING}"
|
||||
elif [ -n "$HIGHEST" ]; then
|
||||
echo "Highest patch already used on this line: ${HIGHEST}"
|
||||
PATCH=$((HIGHEST + 1))
|
||||
else
|
||||
# A minor line nobody has tagged yet is a *new* line, and a new line
|
||||
# starts at .0 — that is what "we are moving to 0.4.x" means. The
|
||||
@@ -165,21 +200,70 @@ jobs:
|
||||
env:
|
||||
TOKEN: ${{ secrets.REGISTRY_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
TAG="v${{ needs.compute-version.outputs.version }}"
|
||||
# Create release
|
||||
curl -s -X POST \
|
||||
|
||||
# Idempotent get-or-create, matching build-macos. This step used to
|
||||
# POST /releases unconditionally: against a tag that already existed
|
||||
# Gitea answered 409, the grep below found no id, and the run died
|
||||
# with a bare "exitcode '1'" and not one line of output explaining
|
||||
# it — `curl -s` with no `-f` swallows the HTTP error, so nothing
|
||||
# ever said "409" or "duplicate tag". Hence -fsS throughout, and
|
||||
# pipefail so a failure cannot be stepped over.
|
||||
HTTP_CODE=$(curl -sS -o release.json -w '%{http_code}' \
|
||||
-H "Authorization: token ${TOKEN}" \
|
||||
"${GITEA_URL}/api/v1/repos/${REPO}/releases/tags/${TAG}")
|
||||
case "${HTTP_CODE}" in
|
||||
200)
|
||||
echo "Release ${TAG} already exists, reusing"
|
||||
;;
|
||||
404)
|
||||
echo "Creating release ${TAG}"
|
||||
curl -fsS -X POST \
|
||||
-H "Authorization: token ${TOKEN}" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d "{\"tag_name\": \"${TAG}\", \"name\": \"Triple-C ${TAG} (Linux)\", \"body\": \"Automated build from commit ${{ gitea.sha }}\"}" \
|
||||
"${GITEA_URL}/api/v1/repos/${REPO}/releases" > release.json
|
||||
RELEASE_ID=$(cat release.json | grep -o '"id":[0-9]*' | head -1 | grep -o '[0-9]*')
|
||||
;;
|
||||
*)
|
||||
echo "Unexpected ${HTTP_CODE} looking up release ${TAG}:" >&2
|
||||
cat release.json >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
|
||||
RELEASE_ID=$(python3 -c "import json,sys; print(json.load(open('release.json')).get('id',''))")
|
||||
if [ -z "${RELEASE_ID}" ]; then
|
||||
echo "No release id for ${TAG}; refusing to upload into nothing:" >&2
|
||||
cat release.json >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "Release ID: ${RELEASE_ID}"
|
||||
# Upload each artifact
|
||||
|
||||
# Replace-not-conflict, so a retry after a partial upload succeeds.
|
||||
# Versions are monotonic now (see compute-version), so this can only
|
||||
# ever be replacing an asset from a failed run of this same commit —
|
||||
# never one belonging to an already-published version.
|
||||
for file in artifacts/*; do
|
||||
[ -f "$file" ] || continue
|
||||
filename=$(basename "$file")
|
||||
|
||||
EXISTING_ID=$(curl -sS \
|
||||
-H "Authorization: token ${TOKEN}" \
|
||||
"${GITEA_URL}/api/v1/repos/${REPO}/releases/${RELEASE_ID}/assets" \
|
||||
| python3 -c "import json,sys; t=sys.argv[1]; print(next((a['id'] for a in json.load(sys.stdin) if a.get('name')==t), ''))" "${filename}" || true)
|
||||
if [ -n "${EXISTING_ID}" ]; then
|
||||
echo "Deleting existing asset ${filename} (id ${EXISTING_ID})"
|
||||
curl -fsS -X DELETE \
|
||||
-H "Authorization: token ${TOKEN}" \
|
||||
"${GITEA_URL}/api/v1/repos/${REPO}/releases/${RELEASE_ID}/assets/${EXISTING_ID}"
|
||||
fi
|
||||
|
||||
echo "Uploading ${filename}..."
|
||||
curl -s -X POST \
|
||||
curl -fsS --http1.1 \
|
||||
--retry 5 --retry-all-errors --retry-delay 5 \
|
||||
--max-time 600 \
|
||||
-X POST \
|
||||
-H "Authorization: token ${TOKEN}" \
|
||||
-H "Content-Type: application/octet-stream" \
|
||||
--data-binary "@${file}" \
|
||||
|
||||
@@ -203,7 +203,8 @@ docker exec stdout → tokio task → emit("terminal-output-{sessionId}") → li
|
||||
### Container (`container/`)
|
||||
|
||||
- **`Dockerfile`** — Ubuntu 24.04 base with Claude Code, Node.js 22, Python 3.12, Rust, Docker CLI, git, gh, AWS CLI v2, ripgrep, pnpm, uv, ruff pre-installed, plus the shared
|
||||
libraries a browser links against (see below)
|
||||
libraries a browser links against (see below) and the VPN tooling the `vpn_support_enabled`
|
||||
toggle grants capability for (`iproute2`, `wireguard-tools`, `iptables`)
|
||||
- **Browser runtime libraries are baked in; browser *binaries* are not.** A layer runs
|
||||
`npx --yes playwright@latest install-deps chromium` as root, so Playwright names its own
|
||||
dependencies and the list cannot rot against Ubuntu 24.04's `t64` renames or a new Chromium
|
||||
@@ -273,6 +274,89 @@ migration and Reset. Four things here are not obvious:
|
||||
actively **removes** `triple-c-*.crt` when the setting is cleared — `/usr/local/share` rides the
|
||||
project's snapshot image, so turning the feature off has to undo, not merely stop.
|
||||
|
||||
### VPN support (`vpn_support_enabled`, `docker/container.rs`)
|
||||
|
||||
An opt-in per-project switch granting the container what a VPN client needs to build a tunnel.
|
||||
`vpn_host_config()` is the single definition of what that means, and it is unit-tested because a
|
||||
container is created once by a very long function where a dropped capability is invisible.
|
||||
|
||||
- **All three pieces or none.** `CAP_NET_ADMIN` (Docker's default set has `net_raw` but *not*
|
||||
`net_admin`, so a client can ping but never connect), the `/dev/net/tun` device (absent
|
||||
entirely from a default container — nothing to open even with the capability), and
|
||||
`net.ipv4.conf.all.src_valid_mark=1` (WireGuard's `wg-quick` sets it and cannot from inside a
|
||||
container, since `/proc/sys` is read-only, so handshake packets die to reverse-path filtering).
|
||||
Any two without the third still presents as a connection that hangs to a timeout, which is why
|
||||
the tests assert the whole set.
|
||||
- **The device is passed through from the host, never `mknod`-ed inside.** The kernel's `tun`
|
||||
module has to back it.
|
||||
- **A missing device fails at `start`, not `create` — verified against Docker 29.7.** `docker
|
||||
create --device /dev/does-not-exist` succeeds and prints an id; runc resolves the device (and
|
||||
validates sysctls) only when it builds the container. So the guard belongs on the start path:
|
||||
`explain_container_failure()` covers both and is called from `start_container`, where it has a
|
||||
container id and no project — which is why it keys off the error naming `/dev/net/tun` rather
|
||||
than off `vpn_support_enabled`. Nothing else in Triple-C requests a device, so that is
|
||||
unambiguous. A version of this check wired to `create` alone is dead code that looks correct.
|
||||
- **`NET_ADMIN` here is not user-namespaced.** Docker does not enable userns remapping by default,
|
||||
so only the *network* namespace confines it: no reach onto host interfaces, but promiscuous
|
||||
mode, arbitrary addresses/routes/NAT on the shared `docker0` segment (sibling containers, the
|
||||
LiteLLM gateway among them, are ARP-spoofable), netlink-triggered host module auto-load, and
|
||||
enough authority to flush in-container netfilter rules that sandbox mode may rely on. Keep the
|
||||
code comments honest about this — an earlier draft claimed it "confers no authority" outside the
|
||||
container, which is too strong.
|
||||
- **`triple-c.vpn-support` is written unconditionally, including `false`.** The usual
|
||||
`docker commit` reason: a `true` stamped once would ride the snapshot image into every future
|
||||
container and make the switch impossible to turn off.
|
||||
- Off is byte-identical to a container created before the feature existed, and a missing label
|
||||
reads as `false`, so no existing project is churned.
|
||||
- **The toggle grants capability and stops there — it routes nothing.** `vpn_host_config()` returns
|
||||
a cap, a device and a sysctl; no client is installed, no route is touched, no tunnel is started
|
||||
or restored. Users read the name as "turn the VPN on" and report the default network not routing
|
||||
through it as a bug. It isn't, and the docs say so explicitly; keep it that way.
|
||||
- **The tooling is baked, not installed at runtime.** `iproute2` and `wireguard-tools` are in
|
||||
`container/Dockerfile` because a runtime install lands in the writable layer and is lost on
|
||||
base-image migration — leaving a project holding the capability with nothing able to exercise it,
|
||||
and no error that points at why. `iptables` is deliberately absent; see the Dockerfile comment.
|
||||
- **Anything built on this fails open.** The network namespace is rebuilt on every start and no
|
||||
service manager runs inside, so a tunnel never survives stop/start or recreation — while leftover
|
||||
`/run` state makes it look as though it did. Note the two different mechanisms: `/run` is in the
|
||||
writable layer, so on a stop/start it is simply the same container's files, and on a recreation
|
||||
`docker commit` has carried it into the snapshot. Traffic silently reverts to the real address.
|
||||
Any future autostart or killswitch work starts here.
|
||||
- **`/run` riding the snapshot means a VPN client's key material can end up in an image.** Verified:
|
||||
a fresh container off the whp snapshot already contained the `wg.priv` a previous tunnel left in
|
||||
`/run`. Anything writing key material there inherits the problem — the same `docker commit`
|
||||
hazard as `triple-c.git-token-hash` and the custom-env fingerprint, in a directory that looks
|
||||
ephemeral and is not. A VPN client that does this should delete its key on teardown.
|
||||
- **`iptables` is baked, and picking `nftables` instead would have been wrong.** `Recommends:
|
||||
nftables | iptables` is stripped by `--no-install-recommends`, and `wg-quick` needs a backend for
|
||||
any `AllowedIPs = 0.0.0.0/0`. `nftables` is the tempting choice — preferred by `wg-quick`, half
|
||||
the size — but `wg-quick` picks nft *unconditionally* when present, and its nft ruleset needs
|
||||
`nft_fib_ipv4`, which LinuxKit (Docker Desktop for Mac) does not build while it *does* build
|
||||
`xt_CONNMARK`. Shipping nftables would therefore have forfeited Mac. See the Dockerfile comment;
|
||||
the kernel-config evidence is quoted there.
|
||||
- **Two `wg-quick` failures remain, and only one is ours to fix.** Full tunnels still need
|
||||
`xt_CONNMARK`, which WSL2 before 6.6 lacks — nothing installable changes that. And every
|
||||
provider's stock config carries a `DNS =` line that fails in `set_dns()` before any routing, so it
|
||||
breaks split tunnels too; `openresolv` has no candidate on noble and `resolvconf` drags in
|
||||
systemd-resolved, so that one is documented rather than fixed. Driving `wg` and `ip route`
|
||||
directly avoids both, which is what the skill does.
|
||||
- **The `pia-vpn` skill is installed *and removed* from `VPN_SUPPORT_ENABLED`.** `container/skills/`
|
||||
is baked to `/opt/triple-c-skills` and `install_feature_skill()` in `entrypoint.sh` copies it into
|
||||
`~/.claude/skills/` on every start — refreshed each time, so a fix reaches any project whose base
|
||||
image has the source, and `rm -rf`'d first, so files dropped from a later version do not linger.
|
||||
The removal branch matters as much as the install: `~/.claude` is a persisted volume, so a skill
|
||||
left behind after the toggle goes off would keep instructing an agent to use a capability the
|
||||
container no longer has. Which is also why the variable is sent as `0` rather than omitted (see
|
||||
`vpn_env_var`, tested), and why it is in `RESERVED_ENV_EXACT` — a custom env var of that name
|
||||
could otherwise claim the skill without the capability behind it.
|
||||
- **Both halves of that live in the base image, so neither reaches an existing project.** A
|
||||
recreation builds from the project's *own snapshot*, which has no `/opt/triple-c-skills` and no
|
||||
updated `entrypoint.sh`; only a migration or a Reset delivers them. The install path says so out
|
||||
loud rather than returning silently, and `/opt/triple-c-skills` is in `FEATURE_PROBES` so the
|
||||
migration pre-flight lists it as missing. Worth knowing before adding anything else behind an
|
||||
existing toggle: the label fingerprints *the setting*, not the set of things the setting drives,
|
||||
so a project already at `true` gets no recreation at all on upgrade.
|
||||
|
||||
### Container Lifecycle
|
||||
|
||||
Containers use a **stop/start** model (not create/destroy). Installed packages persist across stops. The `.claude` config dir uses a named Docker volume (`triple-c-claude-config-{projectId}`), nested inside the home volume (`triple-c-home-{projectId}`), so OAuth tokens and Claude Code config survive container stop/start *and* container recreation.
|
||||
|
||||
+98
-1
@@ -471,6 +471,92 @@ When enabled, the host Docker socket is mounted into the container so Claude Cod
|
||||
|
||||
> Toggling this requires stopping and restarting the container to take effect.
|
||||
|
||||
### VPN Support
|
||||
|
||||
When enabled, the container is given the three things a VPN client needs to build a tunnel:
|
||||
the `NET_ADMIN` capability, the `/dev/net/tun` device, and the `net.ipv4.conf.all.src_valid_mark`
|
||||
sysctl that WireGuard requires. This is **off by default**.
|
||||
|
||||
The `ip`, `wg` and `iptables` commands ship in the container image so there is something able to use
|
||||
them. If your project's container was created from an older base image it will not have them, and
|
||||
`wg` will simply not be found — **migrating the project onto the current base image** is what picks
|
||||
them up. `sudo apt install iproute2 wireguard-tools iptables` works in the meantime, but lives in
|
||||
the writable layer, so it is undone by a **Reset** and by a migration.
|
||||
|
||||
**This setting makes a tunnel possible; it does not make one.** Nothing is connected, no traffic is
|
||||
redirected, and no tunnel is configured or started on your behalf. Enabling it and expecting the
|
||||
container's traffic to start leaving through a VPN is the most common misreading of what it does —
|
||||
configuring a tunnel and routing traffic into it remains yours to do.
|
||||
|
||||
To make that second half easier, enabling this also installs a **`pia-vpn` skill** into the
|
||||
container's `~/.claude/skills/`, so Claude Code can bring up a Private Internet Access tunnel over
|
||||
WireGuard for you — ask it to connect the VPN and it will. The skill carries the parts that are
|
||||
easy to get wrong (see the DNS note below), and it is removed again when you turn the setting off.
|
||||
It needs your PIA credentials in `~/pia-creds`, two lines, username then password. If you use a
|
||||
different provider, ignore it and set up your own client; nothing else depends on it.
|
||||
|
||||
Like the VPN tooling above, the skill ships in the container image, so a project whose container
|
||||
predates it will not get one by toggling the setting — **migrate the project** and it appears; the
|
||||
migration pre-flight lists it among what you would gain.
|
||||
|
||||
With the setting **off**, a client such as PIA or OpenVPN installs and its daemon starts normally,
|
||||
but the connection attempt **hangs until it times out** — a default container has no tun device to open
|
||||
and no permission to add an interface or a route, and most clients report that as a generic timeout
|
||||
rather than a permissions error.
|
||||
|
||||
Things worth knowing:
|
||||
|
||||
- Tailscale is the exception: in its `--tun=userspace-networking` mode it needs neither the
|
||||
capability nor the device, so leave this off if that is all you want.
|
||||
|
||||
- `NET_ADMIN` applies to the container's **own** network namespace — it cannot touch the host's
|
||||
interfaces. It is not nothing, though: within that namespace anything in the container can set
|
||||
promiscuous mode and add arbitrary addresses, routes and firewall rules on the Docker bridge it
|
||||
shares with your other containers, and it can flush firewall rules that sandbox mode relies on.
|
||||
Grant it per project, to projects that need it.
|
||||
- The **Docker host's** kernel must have the `tun` module available. With Docker Desktop that is
|
||||
the Linux VM, not your own machine. If it is missing, the container is created but fails to
|
||||
**start**, with an error naming `/dev/net/tun` and pointing back at this setting.
|
||||
- A VPN client's kill switch applies to everything in the container, Claude Code included. If the
|
||||
tunnel drops, expect API calls to fail until it reconnects or the kill switch is turned off.
|
||||
- **No tunnel survives a restart.** The network namespace is built fresh every time the container
|
||||
starts, and there is no service manager inside to reconnect anything. Leftover state under `/run`
|
||||
makes it *look* like the tunnel is still configured — that directory is in the container's
|
||||
writable layer, so it is simply still there after a stop/start, and `docker commit` carries it
|
||||
into the snapshot that a recreation is built from. Either way the interface and its routes are
|
||||
gone and traffic goes out your real address again, with no error and nothing visibly different.
|
||||
Re-establish it after every start, and check rather than assume.
|
||||
- **A full tunnel breaks DNS unless the client is told to leave private ranges alone.** Your
|
||||
resolver is whatever `/etc/resolv.conf` says, and if that address is outside the container's own
|
||||
subnet then a default route of `0.0.0.0/0` — or a `0.0.0.0/1` plus `128.0.0.0/1` pair — captures
|
||||
it and sends every lookup into a tunnel that cannot carry it. Under Docker Desktop it is
|
||||
`192.168.65.7`, which is exactly that case; on a user-defined Docker network it is `127.0.0.11`,
|
||||
which is loopback and unaffected. Check yours rather than assuming. The symptom when it bites is
|
||||
total: Claude Code reports it cannot connect, because it cannot resolve `api.anthropic.com`.
|
||||
Route `10.0.0.0/8`, `172.16.0.0/12`, `192.168.0.0/16` and `169.254.0.0/16` via the original
|
||||
gateway — and give the tunnel a resolver it can actually reach, normally the VPN provider's own,
|
||||
or you have a tunnel that leaks every DNS query outside itself. Also pin the VPN endpoint's own
|
||||
address via the original gateway, or the tunnel's encrypted packets try to route through the
|
||||
tunnel. Note that a health check which fetches an IP literal such as `1.1.1.1` passes cleanly
|
||||
while DNS is broken — resolve a name instead.
|
||||
- **Delete a client's key material when you tear a tunnel down.** Anything written under `/run` is
|
||||
in the container's writable layer, and recreating or migrating the project runs `docker commit`
|
||||
over it — so a WireGuard private key left there gets baked into the project's snapshot image and
|
||||
copied forward from then on. This is not hypothetical; it has already happened here.
|
||||
- **Strip the `DNS =` line from a provider's `.conf` before `wg-quick up`.** Every commercial
|
||||
provider ships one, and `wg-quick` hands it to `resolvconf`, which is not installed — so it fails
|
||||
at `resolvconf: command not found` and deletes the interface again. This happens before any
|
||||
routing, so it takes **split tunnels down too**. Set the resolver another way instead, or drive
|
||||
`wg` and `ip route` directly rather than going through `wg-quick`.
|
||||
- **`wg-quick` full tunnels also need `xt_CONNMARK` from the host kernel.** Native Linux, Docker
|
||||
Desktop for Mac and WSL2 kernels from 6.6 have it; older WSL2 kernels do not, and a container
|
||||
cannot load one. There the answer is again to add the routes yourself with `ip route`, which
|
||||
needs no firewall backend on any platform.
|
||||
|
||||
> This setting can only be changed when the container is stopped. Capabilities and devices are
|
||||
> fixed when a container is created, so toggling it recreates the container on the next start.
|
||||
> Recreation preserves the home and `.claude` volumes — it is not a Reset.
|
||||
|
||||
### Mission Control
|
||||
|
||||
Toggle **Mission Control** to integrate Flight Control — an AI-first development methodology bundled with Triple-C — into the project. When enabled:
|
||||
@@ -1139,13 +1225,23 @@ triple-c-scheduler list # List all tasks
|
||||
triple-c-scheduler enable --id abc123 # Enable a task
|
||||
triple-c-scheduler disable --id abc123 # Disable a task
|
||||
triple-c-scheduler remove --id abc123 # Delete a task
|
||||
triple-c-scheduler run --id abc123 # Trigger a task immediately
|
||||
triple-c-scheduler run --id abc123 # Trigger a task now, streaming its log
|
||||
triple-c-scheduler status # What is running right now, and for how long
|
||||
triple-c-scheduler status --id abc123 -w # Watch one task until its run finishes
|
||||
triple-c-scheduler logs --id abc123 # View logs for a task
|
||||
triple-c-scheduler logs --tail 20 # View last 20 log entries (all tasks)
|
||||
triple-c-scheduler notifications # View completion notifications
|
||||
triple-c-scheduler notifications --clear # Clear notifications
|
||||
```
|
||||
|
||||
`list` carries a status column, and the Automation tab marks a task **Running** with
|
||||
its elapsed time, so a triggered run is visible rather than silent.
|
||||
|
||||
Note that a log which has stopped growing is not evidence of a stall: `claude -p`
|
||||
writes its answer in one go when it finishes, so a healthy run shows nothing but its
|
||||
header for as long as it is thinking. `status` is what distinguishes a slow run from
|
||||
a dead one — it reports the run only while the runner's process is genuinely alive.
|
||||
|
||||
### Cron Schedule Format
|
||||
|
||||
Standard 5-field cron: `minute hour day-of-month month day-of-week`
|
||||
@@ -1215,6 +1311,7 @@ The sandbox container (Ubuntu 24.04) comes pre-installed with:
|
||||
| ruff | Latest | Python linter/formatter |
|
||||
| Rust | Stable | Rust development (via rustup) |
|
||||
| Docker CLI | Latest | Container management (when spawning is enabled) |
|
||||
| iproute2, WireGuard tools, iptables | Latest | Building a tunnel (when VPN Support is enabled) |
|
||||
| git | Latest | Version control |
|
||||
| GitHub CLI (gh) | Latest | GitHub integration |
|
||||
| AWS CLI | v2 | AWS services and Bedrock |
|
||||
|
||||
@@ -133,6 +133,14 @@ forces that).
|
||||
4. **Stop**: Container halted (its filesystem layer and both named volumes persist)
|
||||
5. **Restart**: Existing container restarted; if any `triple-c.*` label no longer matches the project's settings, the container is committed to a snapshot image, removed, and recreated from that snapshot — so installed packages survive
|
||||
6. **Migrate**: The project is moved onto a newer base image without losing its volumes — see below
|
||||
|
||||
Each recreation moves the `triple-c-snapshot-{projectId}:latest` tag, leaving the image it pointed
|
||||
at before untagged but still on disk — multiple gigabytes per recreation. `sweep_orphaned_snapshots`
|
||||
clears those after a recreation and after a migration is accepted. It only ever removes images that
|
||||
are **both** untagged *and* labelled `triple-c.managed=true`, so a live snapshot tag and a
|
||||
migration's `pre-migration-*` rollback pin are structurally out of reach, and removal is unforced so
|
||||
Docker itself refuses while any container — including a stopped project's — is still built from the
|
||||
image.
|
||||
7. **Reset**: Container, snapshot image **and both named volumes** all removed, then recreated from the clean base image. `remove_project_volumes` deletes `triple-c-home-{projectId}` and `triple-c-claude-config-{projectId}`, so `~/.claude`, `~/.claude.json`, the OAuth login, installed skills, session transcripts and the scheduler's tasks are all lost.
|
||||
|
||||
### Base-Image Migration
|
||||
|
||||
@@ -194,11 +194,24 @@ impl BrowserTarget {
|
||||
}
|
||||
}
|
||||
|
||||
/// The `channel` a launch check must pass. `None` means the bundled build.
|
||||
fn channel(self) -> Option<&'static str> {
|
||||
/// Every `channel` a launch check must pass, comma-separated, where
|
||||
/// `default` means "no channel — the bundled build".
|
||||
///
|
||||
/// Chromium is checked twice because the two consumers of this install do
|
||||
/// not launch the same binary. A script calling `chromium.launch()` with
|
||||
/// no channel gets `chromium-headless-shell`; the viewer reads
|
||||
/// `~/.playwright/cli.config.json`, which pins channel
|
||||
/// `chrome-for-testing`, and that resolves to the *full* `chromium-<rev>`
|
||||
/// build — a separate download under the same `install chromium`.
|
||||
///
|
||||
/// Checking only the first is how a container reaches "verified" and then
|
||||
/// fails in the pane with `Browser "chrome-for-testing" is not installed`.
|
||||
/// Observed on a real project, where a stale `chromium-1217` satisfied the
|
||||
/// headless-shell launch while the viewer wanted `chromium-1237`.
|
||||
fn channels(self) -> &'static str {
|
||||
match self {
|
||||
Self::Chromium => None,
|
||||
Self::Chrome => Some("chrome"),
|
||||
Self::Chromium => "default,chrome-for-testing",
|
||||
Self::Chrome => "chrome",
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -235,7 +248,7 @@ pub async fn install_packages(
|
||||
&format!("Installing @playwright/cli into {}/node_modules…", INSTALL_DIR),
|
||||
);
|
||||
|
||||
let mut step = npm_install(app, project_id, container_id, VIEWER_PACKAGE).await?;
|
||||
let mut step = npm_install(app, project_id, container_id, &[VIEWER_PACKAGE]).await?;
|
||||
if step.exit_code != 0 {
|
||||
return Err(format!(
|
||||
"npm couldn't install the viewer package in this container (exit {}).\n\nnpm said:\n{}",
|
||||
@@ -246,9 +259,15 @@ pub async fn install_packages(
|
||||
|
||||
// Second, `playwright` at the version the viewer package pins — see
|
||||
// `VIEWER_PACKAGE`. Installing it as `@latest` is what splits the tree.
|
||||
//
|
||||
// The viewer package is named *again* here. It is already installed, so
|
||||
// this adds no work, but omitting it is what made npm prune it back out —
|
||||
// see the note on `npm_install`. The pin can only be read after the first
|
||||
// install has written the manifest, which is why this stays two commands
|
||||
// rather than one.
|
||||
let spec = pinned_playwright_spec(container_id).await;
|
||||
emit_progress(app, project_id, &format!("Installing {}…", spec));
|
||||
let second = npm_install(app, project_id, container_id, &spec).await?;
|
||||
let second = npm_install(app, project_id, container_id, &[VIEWER_PACKAGE, &spec]).await?;
|
||||
if second.exit_code != 0 {
|
||||
return Err(format!(
|
||||
"npm couldn't install {} in this container (exit {}).\n\nnpm said:\n{}",
|
||||
@@ -290,7 +309,7 @@ pub async fn install_packages(
|
||||
})
|
||||
}
|
||||
|
||||
/// One `npm install` of one spec, into [`INSTALL_DIR`], as `claude`.
|
||||
/// One `npm install` of one or more specs, into [`INSTALL_DIR`], as `claude`.
|
||||
///
|
||||
/// `env VAR=… cmd` rather than an exec env: it keeps the one exec path in
|
||||
/// `docker/exec.rs` untouched, and `env` is a real binary so no shell is
|
||||
@@ -298,13 +317,24 @@ pub async fn install_packages(
|
||||
/// has no postinstall (verified — `playwright@1.62.1` declares no `scripts` at
|
||||
/// all), but if a future release brings the browser download back, this step
|
||||
/// must stay small and the download must stay the step the user asked for.
|
||||
///
|
||||
/// **Every package that must survive has to appear in `specs`.** `--no-save`
|
||||
/// in a directory with no `package.json` — which [`INSTALL_DIR`] is — leaves
|
||||
/// npm with the command line as its only statement of what the tree should
|
||||
/// contain, and npm ≥7 reconciles the tree against that on every run by
|
||||
/// removing whatever it now considers extraneous. Installing `@playwright/cli`
|
||||
/// and then installing `playwright` in a second command therefore *deletes the
|
||||
/// first one*: verified in a container, `removed 3 packages`, leaving an empty
|
||||
/// `node_modules/@playwright/` behind `playwright` and `playwright-core`. That
|
||||
/// empty directory is why a fresh setup could report success and still leave
|
||||
/// the pane saying `@playwright/cli` was not installed.
|
||||
async fn npm_install(
|
||||
app: &AppHandle,
|
||||
project_id: &str,
|
||||
container_id: &str,
|
||||
spec: &str,
|
||||
specs: &[&str],
|
||||
) -> Result<StepResult, String> {
|
||||
let cmd = vec![
|
||||
let mut cmd = vec![
|
||||
"env".to_string(),
|
||||
"PLAYWRIGHT_SKIP_BROWSER_DOWNLOAD=1".to_string(),
|
||||
"npm".to_string(),
|
||||
@@ -313,8 +343,8 @@ async fn npm_install(
|
||||
"--no-save".to_string(),
|
||||
"--no-fund".to_string(),
|
||||
"--no-audit".to_string(),
|
||||
spec.to_string(),
|
||||
];
|
||||
cmd.extend(specs.iter().map(|s| s.to_string()));
|
||||
run_step(
|
||||
app,
|
||||
project_id,
|
||||
@@ -704,7 +734,7 @@ async fn verify_launch(
|
||||
],
|
||||
vec![
|
||||
format!("TRIPLE_C_PW_DIR={}", dir),
|
||||
format!("TRIPLE_C_PW_CHANNEL={}", target.channel().unwrap_or("")),
|
||||
format!("TRIPLE_C_PW_CHANNELS={}", target.channels()),
|
||||
format!("TRIPLE_C_PW_URL={}", REACHABILITY_URL),
|
||||
],
|
||||
);
|
||||
@@ -795,27 +825,43 @@ fn parse_launch_output(output: &str) -> LaunchVerdict {
|
||||
/// The launch check. One `argv` element, no newlines, same contract as the
|
||||
/// detection probe.
|
||||
///
|
||||
/// Playwright leaves the Chromium sandbox disabled by default, which is what
|
||||
/// makes this work in a container at all. The timeout exists so a browser that
|
||||
/// hangs on a missing library still returns a verdict rather than sitting there
|
||||
/// until the exec is torn down. The navigation is best-effort and never decides
|
||||
/// `ok` — it exists to tell a TLS-intercepted network apart from a broken
|
||||
/// install.
|
||||
/// `chromiumSandbox` is set explicitly rather than left to Playwright's
|
||||
/// default, so this check states the same thing the seeded
|
||||
/// `cli.config.json` does instead of agreeing with it by coincidence. The
|
||||
/// containers forbid unprivileged user namespaces, so a sandboxed Chromium
|
||||
/// aborts on launch; nothing here should be able to drift back into testing a
|
||||
/// configuration the viewer will not use.
|
||||
///
|
||||
/// Each channel in `TRIPLE_C_PW_CHANNELS` is launched in turn — see
|
||||
/// [`BrowserTarget::channels`] for why Chromium needs two — and a failure
|
||||
/// names the channel that failed, because "is not installed" is meaningless
|
||||
/// without it. Only the last launch loads a page: the navigation is
|
||||
/// best-effort, never decides `ok`, and exists to tell a TLS-intercepted
|
||||
/// network apart from a broken install, so doing it once is enough.
|
||||
///
|
||||
/// The timeout exists so a browser that hangs on a missing library still
|
||||
/// returns a verdict rather than sitting there until the exec is torn down.
|
||||
const LAUNCH_PROBE: &str = concat!(
|
||||
r#"const d=process.env.TRIPLE_C_PW_DIR,ch=process.env.TRIPLE_C_PW_CHANNEL||undefined,u=process.env.TRIPLE_C_PW_URL;"#,
|
||||
r#"const d=process.env.TRIPLE_C_PW_DIR,chs=process.env.TRIPLE_C_PW_CHANNELS||"default",u=process.env.TRIPLE_C_PW_URL;"#,
|
||||
r#"let done=false;const say=(ok,detail,nav)=>{if(done)return;done=true;"#,
|
||||
r#"process.stdout.write("\n__TRIPLE_C_BROWSER_LAUNCH__"+JSON.stringify({ok,detail,nav:nav||null})+"\n");};"#,
|
||||
r#"const one=(e)=>String((e&&e.message)||e).split("\n").slice(0,8).join(" | ");"#,
|
||||
r#"const t=setTimeout(()=>{say(false,"the browser did not finish starting within 90s");process.exit(0);},90000);"#,
|
||||
r#"(async()=>{let b=null;try{const {chromium}=require(d);b=await chromium.launch(ch?{channel:ch}:{});"#,
|
||||
r#"let v="";try{v=b.version();}catch(e){}"#,
|
||||
r#"let nav={ok:true,cert:false,detail:""};"#,
|
||||
r#"(async()=>{let b=null,cur="";try{const {chromium}=require(d);"#,
|
||||
r#"const list=chs.split(",").map(s=>s.trim()).filter(Boolean);"#,
|
||||
r#"let v="",nav={ok:true,cert:false,detail:""};"#,
|
||||
r#"for(let i=0;i<list.length;i++){cur=list[i];const c=cur==="default"?undefined:cur;"#,
|
||||
r#"b=await chromium.launch(Object.assign({chromiumSandbox:false},c?{channel:c}:{}));"#,
|
||||
r#"try{v=b.version();}catch(e){}"#,
|
||||
r#"if(i===list.length-1){"#,
|
||||
r#"try{const p=await b.newPage();await p.goto(u,{timeout:20000});}"#,
|
||||
// A certificate failure is classified here, next to the message, because
|
||||
// Chromium's wording is the only place the distinction exists.
|
||||
r#"catch(e){const m=one(e);nav={ok:false,cert:/ERR_CERT|CERT_AUTHORITY|ERR_SSL|SSL_ERROR|self.signed/i.test(m),detail:m};}"#,
|
||||
r#"await b.close();clearTimeout(t);say(true,v,nav);}"#,
|
||||
r#"catch(e){clearTimeout(t);try{if(b)await b.close();}catch(e2){}say(false,one(e));}"#,
|
||||
r#"catch(e){const m=one(e);nav={ok:false,cert:/ERR_CERT|CERT_AUTHORITY|ERR_SSL|SSL_ERROR|self.signed/i.test(m),detail:m};}}"#,
|
||||
r#"await b.close();b=null;}"#,
|
||||
r#"clearTimeout(t);say(true,v,nav);}"#,
|
||||
r#"catch(e){clearTimeout(t);try{if(b)await b.close();}catch(e2){}"#,
|
||||
r#"say(false,(cur&&cur!=="default"?"channel "+cur+": ":"")+one(e));}"#,
|
||||
r#"process.exit(0);})();"#,
|
||||
);
|
||||
|
||||
@@ -984,8 +1030,10 @@ mod tests {
|
||||
// `@playwright/mcp` asks for the chrome channel specifically, so the UI
|
||||
// must be able to say so.
|
||||
assert!(BrowserTarget::Chrome.needed_for().contains("@playwright/mcp"));
|
||||
assert_eq!(BrowserTarget::Chrome.channel(), Some("chrome"));
|
||||
assert_eq!(BrowserTarget::Chromium.channel(), None);
|
||||
assert_eq!(BrowserTarget::Chrome.channels(), "chrome");
|
||||
// Both of Chromium's consumers, or the check passes for a browser the
|
||||
// viewer cannot open — see `channels`.
|
||||
assert_eq!(BrowserTarget::Chromium.channels(), "default,chrome-for-testing");
|
||||
// And a size, before the click, for both.
|
||||
for t in [BrowserTarget::Chromium, BrowserTarget::Chrome] {
|
||||
assert!(t.download_note().to_lowercase().contains("mb"), "{:?}", t);
|
||||
|
||||
@@ -164,6 +164,11 @@ pub struct ScheduledTask {
|
||||
/// Only known for enabled one-shot tasks (their `at` time). Recurring cron
|
||||
/// expressions are not evaluated here.
|
||||
pub next_run: Option<String>,
|
||||
/// Whether a run is in flight right now, from the runner's state file in
|
||||
/// `~/.claude/scheduler/running/<id>.json` with its pid verified live.
|
||||
pub running: bool,
|
||||
/// When the in-flight run started, ISO 8601 (UTC). `None` unless `running`.
|
||||
pub running_since: Option<String>,
|
||||
}
|
||||
|
||||
/// A completion notice written by `triple-c-task-runner` after a task ran.
|
||||
@@ -614,13 +619,25 @@ const SCHEDULER_LIST_SCRIPT: &str = r#"exec 2>/dev/null
|
||||
set -u
|
||||
TASKS="$HOME/.claude/scheduler/tasks"
|
||||
LOGS="$HOME/.claude/scheduler/logs"
|
||||
RUNNING="$HOME/.claude/scheduler/running"
|
||||
[ -d "$TASKS" ] || { echo '[]'; exit 0; }
|
||||
for f in "$TASKS"/*.json; do
|
||||
[ -f "$f" ] || continue
|
||||
id=$(jq -r '.id // ""' "$f") || continue
|
||||
[ -n "$id" ] || id=$(basename "$f" .json)
|
||||
last=$(find "$LOGS/$id" -name '*.log' -type f -printf '%T@\n' | sort -rn | head -1)
|
||||
jq -c --arg fallback_id "$id" --arg lr "${last%%.*}" '{
|
||||
# Live-run state. The pid is checked, not trusted: a container stopped
|
||||
# mid-run cannot fire the runner's cleanup trap, and a task stuck on
|
||||
# "running" forever is a worse lie than showing nothing.
|
||||
started=""
|
||||
state="$RUNNING/$id.json"
|
||||
if [ -f "$state" ]; then
|
||||
pid=$(jq -r '.pid // empty' "$state")
|
||||
if [ -n "$pid" ] && kill -0 "$pid" 2>/dev/null; then
|
||||
started=$(jq -r '.started_epoch // empty' "$state")
|
||||
fi
|
||||
fi
|
||||
jq -c --arg fallback_id "$id" --arg lr "${last%%.*}" --arg started "$started" '{
|
||||
id: (if (.id // "") == "" then $fallback_id else .id end),
|
||||
name: (.name // ""),
|
||||
prompt: (.prompt // ""),
|
||||
@@ -630,7 +647,8 @@ for f in "$TASKS"/*.json; do
|
||||
enabled: (.enabled == true),
|
||||
working_dir: (.working_dir // "/workspace"),
|
||||
created_at: (.created_at // null),
|
||||
last_run_epoch: (if $lr == "" then null else ($lr | tonumber) end)
|
||||
last_run_epoch: (if $lr == "" then null else ($lr | tonumber) end),
|
||||
running_since_epoch: (if $started == "" then null else ($started | tonumber) end)
|
||||
}' "$f"
|
||||
done | jq -s 'sort_by(.name, .id)'
|
||||
"#;
|
||||
@@ -673,6 +691,7 @@ struct RawScheduledTask {
|
||||
working_dir: String,
|
||||
created_at: Option<String>,
|
||||
last_run_epoch: Option<i64>,
|
||||
running_since_epoch: Option<i64>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize)]
|
||||
@@ -723,6 +742,8 @@ pub async fn list_scheduled_tasks(
|
||||
created_at: t.created_at,
|
||||
last_run: t.last_run_epoch.map(epoch_to_iso),
|
||||
next_run,
|
||||
running: t.running_since_epoch.is_some(),
|
||||
running_since: t.running_since_epoch.map(epoch_to_iso),
|
||||
}
|
||||
})
|
||||
.collect())
|
||||
|
||||
@@ -73,6 +73,25 @@ use crate::AppState;
|
||||
/// Report how far behind the current base image a project's container is, and
|
||||
/// what migrating it would actually carry across.
|
||||
///
|
||||
/// Choose the recorded lineage from the two places it can be written, most
|
||||
/// authoritative first: the live container's label, then the snapshot image's.
|
||||
///
|
||||
/// **An empty label is absence, not an answer.** `create_container` always
|
||||
/// writes `triple-c.base-image-id`, even when the value is unknown — that is
|
||||
/// deliberate, because Docker merges an image's labels into a container's and
|
||||
/// an inherited value would otherwise ride a snapshot forever. The consequence
|
||||
/// is that `Some("")` is the *common* reading from a container whose lineage
|
||||
/// was never established, so treating it as an answer silently skips the
|
||||
/// snapshot, which may well have recorded a real one.
|
||||
fn pick_recorded_lineage(
|
||||
from_container: Option<String>,
|
||||
from_snapshot: Option<String>,
|
||||
) -> Option<String> {
|
||||
from_container
|
||||
.filter(|v| !v.is_empty())
|
||||
.or_else(|| from_snapshot.filter(|v| !v.is_empty()))
|
||||
}
|
||||
|
||||
/// Read-only. Runs two filesystem probes (~3 s each) and is therefore meant to
|
||||
/// be called on demand, not polled.
|
||||
#[tauri::command]
|
||||
@@ -98,20 +117,24 @@ pub async fn get_container_staleness(
|
||||
// Lineage, most authoritative source first: the live container's label,
|
||||
// then the snapshot image's. Both are written by `create_container` and
|
||||
// propagated onto the snapshot by `docker commit`.
|
||||
// Each source is filtered for emptiness *before* it is allowed to satisfy
|
||||
// the lookup. `create_container` always writes this label, even when the
|
||||
// value is unknown — deliberately, so an inherited image label cannot ride
|
||||
// a snapshot forever — which means the container's copy is very often
|
||||
// `Some("")`. Filtering only the final result let that empty string count
|
||||
// as an answer and skip the snapshot entirely, so a snapshot that *did*
|
||||
// record a lineage was never consulted and the project reported "unknown"
|
||||
// with the information sitting one lookup away.
|
||||
let container_id = docker::find_existing_container(&project).await.unwrap_or(None);
|
||||
let recorded = match &container_id {
|
||||
let from_container = match &container_id {
|
||||
Some(id) => container_label(id, mig::LABEL_BASE_IMAGE_ID).await,
|
||||
None => None,
|
||||
}
|
||||
.or_else(|| None);
|
||||
let recorded = match recorded {
|
||||
Some(v) => Some(v),
|
||||
None => mig::image_labels(&snapshot_image)
|
||||
};
|
||||
let from_snapshot = mig::image_labels(&snapshot_image)
|
||||
.await
|
||||
.get(mig::LABEL_BASE_IMAGE_ID)
|
||||
.cloned(),
|
||||
}
|
||||
.filter(|v| !v.is_empty());
|
||||
.cloned();
|
||||
let recorded = pick_recorded_lineage(from_container, from_snapshot);
|
||||
|
||||
out.base_image_id = recorded.clone();
|
||||
out.known = recorded.is_some();
|
||||
@@ -833,6 +856,16 @@ pub async fn confirm_migration(
|
||||
migration_store::clear_staging(&project_id)?;
|
||||
migration_store::clear(&project_id)?;
|
||||
log::info!("Migration confirmed for project {}", project_id);
|
||||
|
||||
// Dropping the pin above is what turns the pre-migration image into an
|
||||
// orphan: it was the only tag holding a multi-gigabyte pre-migration
|
||||
// snapshot. Accepting the update is therefore the moment to sweep, and
|
||||
// waiting for the project's next recreation would leave it lying around
|
||||
// indefinitely.
|
||||
tauri::async_runtime::spawn(async {
|
||||
crate::docker::sweep_orphaned_snapshots().await;
|
||||
});
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -1661,6 +1694,32 @@ fn summarize(
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn an_empty_lineage_label_is_absence_and_falls_through_to_the_snapshot() {
|
||||
let some = |s: &str| Some(s.to_string());
|
||||
|
||||
// The regression: the container always carries the label, so an
|
||||
// unknown lineage reads as `Some("")`. Letting that satisfy the lookup
|
||||
// skipped a snapshot that had recorded the real thing.
|
||||
assert_eq!(
|
||||
pick_recorded_lineage(some(""), some("sha256:base")),
|
||||
some("sha256:base")
|
||||
);
|
||||
|
||||
// Ordinary precedence still holds: the container wins when it has one.
|
||||
assert_eq!(
|
||||
pick_recorded_lineage(some("sha256:container"), some("sha256:snapshot")),
|
||||
some("sha256:container")
|
||||
);
|
||||
assert_eq!(pick_recorded_lineage(None, some("sha256:snap")), some("sha256:snap"));
|
||||
|
||||
// Genuinely unknown stays unknown — "probe instead", never a lineage
|
||||
// invented to make the comparison succeed.
|
||||
assert_eq!(pick_recorded_lineage(None, None), None);
|
||||
assert_eq!(pick_recorded_lineage(some(""), some("")), None);
|
||||
assert_eq!(pick_recorded_lineage(some(""), None), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn byte_sizes_read_the_way_a_disk_warning_should() {
|
||||
assert_eq!(human_bytes(512), "512 B");
|
||||
|
||||
@@ -450,6 +450,18 @@ pub async fn start_project_container(
|
||||
).await?;
|
||||
emit_progress(&app_handle, &project_id, "Starting container...");
|
||||
docker::start_container(&new_id).await?;
|
||||
|
||||
// The commit above moved `:latest` and orphaned the image it
|
||||
// used to point at; the container holding that image open was
|
||||
// removed a few lines up, so now is when Docker will actually
|
||||
// let it go. Detached because this is housekeeping and the
|
||||
// project is already running — and it sweeps every orphan, not
|
||||
// just this one, so recreations that happened before the sweep
|
||||
// existed are cleaned up too.
|
||||
tauri::async_runtime::spawn(async {
|
||||
docker::sweep_orphaned_snapshots().await;
|
||||
});
|
||||
|
||||
new_id
|
||||
} else {
|
||||
emit_progress(&app_handle, &project_id, "Starting container...");
|
||||
|
||||
@@ -18,11 +18,12 @@ This container supports scheduled tasks via `triple-c-scheduler`. You can set up
|
||||
### Commands
|
||||
- `triple-c-scheduler add --name "NAME" --schedule "CRON" --prompt "TASK"` — Add a recurring task
|
||||
- `triple-c-scheduler add --name "NAME" --at "YYYY-MM-DD HH:MM" --prompt "TASK"` — Add a one-time task
|
||||
- `triple-c-scheduler list` — List all scheduled tasks
|
||||
- `triple-c-scheduler list` — List all scheduled tasks, with a running/idle status column
|
||||
- `triple-c-scheduler remove --id ID` — Remove a task
|
||||
- `triple-c-scheduler enable --id ID` / `triple-c-scheduler disable --id ID` — Toggle tasks
|
||||
- `triple-c-scheduler status [--id ID] [--watch]` — Show what is running right now, and for how long
|
||||
- `triple-c-scheduler logs [--id ID] [--tail N]` — View execution logs
|
||||
- `triple-c-scheduler run --id ID` — Manually trigger a task immediately
|
||||
- `triple-c-scheduler run --id ID` — Manually trigger a task immediately (streams its log)
|
||||
- `triple-c-scheduler notifications [--clear]` — View or clear completion notifications
|
||||
|
||||
### Cron format
|
||||
@@ -36,7 +37,7 @@ Use `--at "YYYY-MM-DD HH:MM"` instead of `--schedule`. The task automatically re
|
||||
Use `--working-dir /workspace/project` to set where the task runs (default: /workspace).
|
||||
|
||||
### Checking results
|
||||
After tasks run, check notifications with `triple-c-scheduler notifications` and detailed output with `triple-c-scheduler logs`.
|
||||
While a task is running, `triple-c-scheduler status` reports it with elapsed time — a log that has stopped growing is normal, because `claude -p` writes its answer only at the end, so use `status` rather than log silence to tell a slow run from a dead one. After tasks run, check notifications with `triple-c-scheduler notifications` and detailed output with `triple-c-scheduler logs`.
|
||||
|
||||
### Timezone
|
||||
Scheduled times use the container's configured timezone (check with `date`). If no timezone is configured, UTC is used."#;
|
||||
@@ -211,6 +212,12 @@ pub const SECRET_ENV_KEYS: &[&str] = &[
|
||||
];
|
||||
|
||||
/// Env var name prefixes Triple-C manages itself; users cannot set these by hand.
|
||||
/// The label every container Triple-C creates carries — and, because
|
||||
/// `docker commit` copies a container's labels onto the image, every snapshot it
|
||||
/// commits. [`sweep_orphaned_snapshots`] treats it as the mark of provenance,
|
||||
/// which is what keeps the sweep away from the user's own images.
|
||||
const LABEL_MANAGED: &str = "triple-c.managed";
|
||||
|
||||
const RESERVED_ENV_PREFIXES: &[&str] = &["ANTHROPIC_", "AWS_", "GIT_", "HOST_", "TRIPLE_C_"];
|
||||
|
||||
/// Exact env var names Triple-C manages itself. Not covered by
|
||||
@@ -226,6 +233,7 @@ const RESERVED_ENV_EXACT: &[&str] = &[
|
||||
"MCP_SERVERS_JSON",
|
||||
"CLAUDE_CODE_SETTINGS_JSON",
|
||||
"MISSION_CONTROL_ENABLED",
|
||||
"VPN_SUPPORT_ENABLED",
|
||||
"TRIPLE_C_PERMISSION_MODE",
|
||||
CLAUDE_OAUTH_TOKEN_ENV,
|
||||
// The model-alias vars are already covered by the `ANTHROPIC_` prefix
|
||||
@@ -791,6 +799,128 @@ async fn resolve_base_image_id(image_name: &str, base_image_name: &str) -> Strin
|
||||
.unwrap_or_default()
|
||||
}
|
||||
|
||||
/// The `/dev/net/tun` character device, as it is named on both sides.
|
||||
const TUN_DEVICE: &str = "/dev/net/tun";
|
||||
|
||||
/// The `HostConfig` fields "VPN support" contributes: `CapAdd`, `Devices`,
|
||||
/// `Sysctls` — in that order.
|
||||
type VpnHostConfigParts = (
|
||||
Option<Vec<String>>,
|
||||
Option<Vec<bollard::models::DeviceMapping>>,
|
||||
Option<HashMap<String, String>>,
|
||||
);
|
||||
|
||||
/// The three host-config pieces a VPN client needs, or all-`None` when the
|
||||
/// project has not opted in.
|
||||
///
|
||||
/// Returned as a triple rather than set inline so the exact shape is unit
|
||||
/// testable — a container is created once, by a very long async function, and a
|
||||
/// silently-dropped capability looks identical to a VPN server that is simply
|
||||
/// unreachable.
|
||||
///
|
||||
/// All three are required together and each fails differently on its own:
|
||||
/// * **`CAP_NET_ADMIN`** — without it the client cannot create an interface or
|
||||
/// write a route. Docker's default bounding set grants `net_raw` but not
|
||||
/// `net_admin`, which is why a client can ping but never connect.
|
||||
/// * **`/dev/net/tun`** — the device is absent from a default container, so
|
||||
/// there is nothing to open even with the capability. It is passed through
|
||||
/// from the host rather than `mknod`-ed inside, so the kernel's `tun` module
|
||||
/// backs it.
|
||||
/// * **`net.ipv4.conf.all.src_valid_mark`** — WireGuard's own `wg-quick` sets
|
||||
/// this, and cannot from inside a container (`/proc/sys` is read-only), so
|
||||
/// its handshake packets are dropped by reverse-path filtering. Harmless for
|
||||
/// OpenVPN-based clients, so it is set unconditionally with the rest.
|
||||
///
|
||||
/// What it costs, stated accurately: Docker does not enable user-namespace
|
||||
/// remapping by default, so this is a real `CAP_NET_ADMIN` in the *initial*
|
||||
/// user namespace and only the **network** namespace confines it. It cannot
|
||||
/// touch the host's interfaces, but within its own namespace it can set
|
||||
/// promiscuous mode and add arbitrary addresses, routes and NAT rules on the
|
||||
/// shared `docker0` L2 segment — which puts sibling containers (the LiteLLM
|
||||
/// gateway among them) within reach of ARP spoofing, and lets netlink trigger
|
||||
/// host-kernel module auto-loading. It is also enough to flush netfilter rules
|
||||
/// inside the container, so pair it with `sandbox_mode_enabled` advisedly.
|
||||
/// Hence opt-in, per project, rather than on for everyone.
|
||||
/// The env var `entrypoint.sh` installs and removes the `pia-vpn` skill from.
|
||||
///
|
||||
/// **Emitted either way, never omitted.** `~/.claude` is a persisted volume, so
|
||||
/// turning the toggle off has to actively tell entrypoint to remove a skill an
|
||||
/// earlier run left there, and an absent variable cannot say that. It is also
|
||||
/// what stops a `=1` baked into a snapshot by `docker commit` from outliving
|
||||
/// the setting — the explicit `=0` overwrites it.
|
||||
///
|
||||
/// Extracted for the same reason as [`vpn_host_config`]: the emitting code sits
|
||||
/// in a very long function where a dropped or inverted value is invisible, and
|
||||
/// `MISSION_CONTROL_ENABLED` twenty lines above shows the failure this avoids —
|
||||
/// it is pushed only when true, so a snapshot's baked `=1` survives the toggle
|
||||
/// going off.
|
||||
fn vpn_env_var(enabled: bool) -> String {
|
||||
format!("VPN_SUPPORT_ENABLED={}", u8::from(enabled))
|
||||
}
|
||||
|
||||
fn vpn_host_config(enabled: bool) -> VpnHostConfigParts {
|
||||
if !enabled {
|
||||
return (None, None, None);
|
||||
}
|
||||
|
||||
let devices = vec![bollard::models::DeviceMapping {
|
||||
path_on_host: Some(TUN_DEVICE.to_string()),
|
||||
path_in_container: Some(TUN_DEVICE.to_string()),
|
||||
cgroup_permissions: Some("rwm".to_string()),
|
||||
}];
|
||||
|
||||
let sysctls = HashMap::from([(
|
||||
"net.ipv4.conf.all.src_valid_mark".to_string(),
|
||||
"1".to_string(),
|
||||
)]);
|
||||
|
||||
(
|
||||
Some(vec!["NET_ADMIN".to_string()]),
|
||||
Some(devices),
|
||||
Some(sysctls),
|
||||
)
|
||||
}
|
||||
|
||||
/// Turn the daemon's device-passthrough failure into an explanation.
|
||||
///
|
||||
/// **This fires on `start`, not `create`.** Verified against Docker 29.7:
|
||||
/// `docker create --device /dev/does-not-exist` succeeds and prints an id; the
|
||||
/// device is only resolved when runc builds the container, so the failure lands
|
||||
/// on the *next* call. Sysctls validate at the same point. Anything that
|
||||
/// inspects only the create path will never see it — which is why both paths
|
||||
/// route through here and the tests exercise the start-side string.
|
||||
///
|
||||
/// Unmapped, this reads as `Failed to start container: Docker responded with
|
||||
/// status code 500: error gathering device information while adding custom
|
||||
/// device "/dev/net/tun": no such file or directory` — a path the user will go
|
||||
/// looking for on the wrong machine, since with Docker Desktop the relevant
|
||||
/// host is the Linux VM rather than their own, and with nothing pointing back
|
||||
/// at the switch that caused it.
|
||||
///
|
||||
/// Deliberately not gated on `vpn_support_enabled`: nothing else in Triple-C
|
||||
/// ever asks for a device, so an error naming `/dev/net/tun` can only have come
|
||||
/// from a container created with the switch on. That keeps the check usable
|
||||
/// from [`start_container`], which has a container id and no project.
|
||||
fn explain_container_failure(action: &str, err: &str) -> String {
|
||||
let device_missing = err.contains(TUN_DEVICE)
|
||||
&& (err.contains("no such file or directory")
|
||||
|| err.contains("No such file or directory")
|
||||
|| err.contains("error gathering device information"));
|
||||
|
||||
if device_missing {
|
||||
return format!(
|
||||
"Failed to {} container: the Docker host has no {} device, which \
|
||||
\"VPN support\" requires. The host kernel needs the `tun` module \
|
||||
loaded (on Docker Desktop that is the Linux VM, not your own \
|
||||
machine). Turn VPN support off in Config → Runtime to start this \
|
||||
project without it. Original error: {}",
|
||||
action, TUN_DEVICE, err
|
||||
);
|
||||
}
|
||||
|
||||
format!("Failed to {} container: {}", action, err)
|
||||
}
|
||||
|
||||
pub async fn create_container(
|
||||
project: &Project,
|
||||
docker_socket_path: &str,
|
||||
@@ -1163,6 +1293,8 @@ pub async fn create_container(
|
||||
env_vars.push("MISSION_CONTROL_ENABLED=1".to_string());
|
||||
}
|
||||
|
||||
env_vars.push(vpn_env_var(project.vpn_support_enabled));
|
||||
|
||||
// Permission mode — read by triple-c-task-runner for scheduled (headless)
|
||||
// Claude Code runs. Interactive terminals get the flags directly instead.
|
||||
env_vars.push(format!(
|
||||
@@ -1355,7 +1487,7 @@ pub async fn create_container(
|
||||
}
|
||||
|
||||
let mut labels = HashMap::new();
|
||||
labels.insert("triple-c.managed".to_string(), "true".to_string());
|
||||
labels.insert(LABEL_MANAGED.to_string(), "true".to_string());
|
||||
labels.insert("triple-c.project-id".to_string(), project.id.clone());
|
||||
labels.insert("triple-c.project-name".to_string(), project.name.clone());
|
||||
labels.insert("triple-c.backend".to_string(), format!("{:?}", project.backend));
|
||||
@@ -1368,6 +1500,13 @@ pub async fn create_container(
|
||||
labels.insert("triple-c.image".to_string(), image_name.to_string());
|
||||
labels.insert("triple-c.timezone".to_string(), timezone.unwrap_or("").to_string());
|
||||
labels.insert("triple-c.mission-control".to_string(), project.mission_control_enabled.to_string());
|
||||
// Capabilities, devices and sysctls are fixed at creation, so this is
|
||||
// container state and gets the label-and-compare treatment. Written
|
||||
// unconditionally (`false`, not omitted) because `docker commit` copies
|
||||
// container labels onto the snapshot image: a `true` stamped once would
|
||||
// otherwise ride that snapshot into every future container and make the
|
||||
// switch impossible to turn back off.
|
||||
labels.insert("triple-c.vpn-support".to_string(), project.vpn_support_enabled.to_string());
|
||||
labels.insert("triple-c.permission-mode".to_string(),
|
||||
project.effective_permission_mode().as_env_value().to_string());
|
||||
labels.insert("triple-c.custom-env-fingerprint".to_string(), custom_env_fingerprint.clone());
|
||||
@@ -1436,10 +1575,15 @@ pub async fn create_container(
|
||||
labels.insert((*key).to_string(), (*value).to_string());
|
||||
}
|
||||
|
||||
let (cap_add, devices, sysctls) = vpn_host_config(project.vpn_support_enabled);
|
||||
|
||||
let host_config = HostConfig {
|
||||
mounts: Some(mounts),
|
||||
port_bindings: if port_bindings.is_empty() { None } else { Some(port_bindings) },
|
||||
init: Some(true),
|
||||
cap_add,
|
||||
devices,
|
||||
sysctls,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
@@ -1469,7 +1613,7 @@ pub async fn create_container(
|
||||
let response = docker
|
||||
.create_container(Some(options), config)
|
||||
.await
|
||||
.map_err(|e| format!("Failed to create container: {}", e))?;
|
||||
.map_err(|e| explain_container_failure("create", &e.to_string()))?;
|
||||
|
||||
Ok(response.id)
|
||||
}
|
||||
@@ -1479,7 +1623,7 @@ pub async fn start_container(container_id: &str) -> Result<(), String> {
|
||||
docker
|
||||
.start_container(container_id, None::<StartContainerOptions<String>>)
|
||||
.await
|
||||
.map_err(|e| format!("Failed to start container: {}", e))
|
||||
.map_err(|e| explain_container_failure("start", &e.to_string()))
|
||||
}
|
||||
|
||||
pub async fn stop_container(container_id: &str) -> Result<(), String> {
|
||||
@@ -1703,6 +1847,128 @@ fn env_holds_a_secret(env: &[String]) -> bool {
|
||||
})
|
||||
}
|
||||
|
||||
/// Outcome of [`sweep_orphaned_snapshots`].
|
||||
#[derive(Debug, Default, Clone, serde::Serialize)]
|
||||
pub struct SnapshotSweepReport {
|
||||
/// Image ids that were removed.
|
||||
pub removed: Vec<String>,
|
||||
/// Bytes the removed images accounted for, as Docker reported them. A
|
||||
/// shared-layer estimate, not a disk-usage measurement.
|
||||
pub reclaimed_bytes: i64,
|
||||
/// Orphans Docker refused to delete because a container is still built
|
||||
/// from them. Normal, not a failure — the next sweep gets them.
|
||||
pub in_use: usize,
|
||||
/// Orphans that could not be removed for any other reason, with the error.
|
||||
pub failed: Vec<(String, String)>,
|
||||
/// Set when the engine could not be reached or listed at all.
|
||||
pub unavailable: Option<String>,
|
||||
}
|
||||
|
||||
/// The filter every sweep runs under. Extracted so a test can hold the two
|
||||
/// conditions in place: **dangling** and **labelled as ours**. Losing either
|
||||
/// one turns a snapshot sweep into a prune of the user's whole image store.
|
||||
fn orphan_sweep_filters() -> HashMap<String, Vec<String>> {
|
||||
HashMap::from([
|
||||
("dangling".to_string(), vec!["true".to_string()]),
|
||||
(
|
||||
"label".to_string(),
|
||||
vec![format!("{}=true", LABEL_MANAGED)],
|
||||
),
|
||||
])
|
||||
}
|
||||
|
||||
/// Remove the untagged snapshot commits left behind by recreation.
|
||||
///
|
||||
/// Every recreation commits the container to `triple-c-snapshot-{id}:latest`
|
||||
/// and moves that tag; the image the tag pointed at before keeps its layers and
|
||||
/// loses its name. Nothing else deletes those, so a project that has been
|
||||
/// recreated a dozen times leaves a dozen multi-gigabyte orphans behind.
|
||||
///
|
||||
/// Two conditions, and the safety of this whole function rests on them:
|
||||
///
|
||||
/// * **Dangling** — untagged. Every image the app relies on carries a tag:
|
||||
/// `triple-c-snapshot-{id}:latest` is what a project is rebuilt from, and a
|
||||
/// migration's `pre-migration-*` pin is the only copy of a rollback target.
|
||||
/// Neither can ever match this filter, so neither can be swept.
|
||||
/// * **`triple-c.managed=true`** — only images Triple-C itself committed.
|
||||
/// `docker commit` copies the container's labels onto the image, which is what
|
||||
/// makes the label a reliable mark of provenance. The user's own dangling
|
||||
/// images are none of our business.
|
||||
///
|
||||
/// Removal is not forced, so Docker refuses (409) while any container is still
|
||||
/// built from the image — including the stopped containers of projects that are
|
||||
/// not running. That refusal is the third safety net and it is the daemon's,
|
||||
/// not ours; those orphans are simply counted and left for a later sweep.
|
||||
///
|
||||
/// Never fails the caller: this is housekeeping, and a full disk is a better
|
||||
/// outcome than a project that will not start.
|
||||
pub async fn sweep_orphaned_snapshots() -> SnapshotSweepReport {
|
||||
use bollard::image::ListImagesOptions;
|
||||
|
||||
let mut report = SnapshotSweepReport::default();
|
||||
|
||||
let docker = match get_docker() {
|
||||
Ok(d) => d,
|
||||
Err(e) => {
|
||||
report.unavailable = Some(e);
|
||||
return report;
|
||||
}
|
||||
};
|
||||
|
||||
let images = match docker
|
||||
.list_images(Some(ListImagesOptions {
|
||||
all: false,
|
||||
filters: orphan_sweep_filters(),
|
||||
..Default::default()
|
||||
}))
|
||||
.await
|
||||
{
|
||||
Ok(images) => images,
|
||||
Err(e) => {
|
||||
report.unavailable = Some(format!("Could not list orphaned snapshots: {}", e));
|
||||
return report;
|
||||
}
|
||||
};
|
||||
|
||||
for summary in images {
|
||||
match docker
|
||||
.remove_image(
|
||||
&summary.id,
|
||||
Some(RemoveImageOptions {
|
||||
force: false,
|
||||
noprune: false,
|
||||
}),
|
||||
None,
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(_) => {
|
||||
report.reclaimed_bytes += summary.size;
|
||||
report.removed.push(summary.id);
|
||||
}
|
||||
Err(bollard::errors::Error::DockerResponseServerError {
|
||||
status_code: 409, ..
|
||||
}) => {
|
||||
report.in_use += 1;
|
||||
}
|
||||
Err(e) => {
|
||||
report.failed.push((summary.id, e.to_string()));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if !report.removed.is_empty() || report.in_use > 0 {
|
||||
log::info!(
|
||||
"Snapshot sweep: removed {} orphan(s) ({:.2} GB), {} still in use by a container",
|
||||
report.removed.len(),
|
||||
report.reclaimed_bytes as f64 / 1_073_741_824.0,
|
||||
report.in_use
|
||||
);
|
||||
}
|
||||
|
||||
report
|
||||
}
|
||||
|
||||
/// Outcome of [`scrub_secrets_from_snapshots`], so callers can tell the user
|
||||
/// what actually happened rather than guessing.
|
||||
#[derive(Debug, Default, Clone, serde::Serialize)]
|
||||
@@ -2238,6 +2504,19 @@ pub async fn container_needs_recreation(
|
||||
return Ok(true);
|
||||
}
|
||||
|
||||
// ── VPN support (NET_ADMIN + /dev/net/tun + sysctl) ───────────────────
|
||||
// A container's capabilities, devices and sysctls are set at creation and
|
||||
// cannot be changed on a running or stopped container, so recreation is the
|
||||
// only way a toggle here takes effect. A missing label means the container
|
||||
// predates the feature, which is the same thing as having it off — so
|
||||
// existing projects are not churned until someone actually turns it on.
|
||||
let expected_vpn = project.vpn_support_enabled.to_string();
|
||||
let container_vpn = get_label("triple-c.vpn-support").unwrap_or_else(|| "false".to_string());
|
||||
if container_vpn != expected_vpn {
|
||||
log::info!("VPN support mismatch (container={:?}, expected={:?})", container_vpn, expected_vpn);
|
||||
return Ok(true);
|
||||
}
|
||||
|
||||
// ── Permission mode ────────────────────────────────────────────────────
|
||||
// The mode is injected as the TRIPLE_C_PERMISSION_MODE env var, and
|
||||
// container env can only change by recreating the container. A missing
|
||||
@@ -2366,7 +2645,7 @@ pub async fn list_sibling_containers() -> Result<Vec<ContainerSummary>, String>
|
||||
.into_iter()
|
||||
.filter(|c| {
|
||||
if let Some(labels) = &c.labels {
|
||||
!labels.contains_key("triple-c.managed")
|
||||
!labels.contains_key(LABEL_MANAGED)
|
||||
} else {
|
||||
true
|
||||
}
|
||||
@@ -2487,6 +2766,142 @@ mod tests {
|
||||
assert_eq!(fp, "");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn vpn_support_off_touches_nothing_in_the_host_config() {
|
||||
// The default must stay byte-identical to a container created before the
|
||||
// feature existed, or every project recreates on the next start.
|
||||
let (cap_add, devices, sysctls) = vpn_host_config(false);
|
||||
assert_eq!(cap_add, None);
|
||||
assert_eq!(devices, None);
|
||||
assert_eq!(sysctls, None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn vpn_support_on_grants_all_three_pieces() {
|
||||
// Each is useless without the others — a client with the capability but
|
||||
// no device, or the device but no capability, still times out — so this
|
||||
// asserts the whole set rather than any one of them.
|
||||
let (cap_add, devices, sysctls) = vpn_host_config(true);
|
||||
|
||||
assert_eq!(cap_add, Some(vec!["NET_ADMIN".to_string()]));
|
||||
|
||||
let devices = devices.expect("the tun device must be passed through");
|
||||
assert_eq!(devices.len(), 1);
|
||||
assert_eq!(devices[0].path_on_host.as_deref(), Some(TUN_DEVICE));
|
||||
assert_eq!(devices[0].path_in_container.as_deref(), Some(TUN_DEVICE));
|
||||
assert_eq!(devices[0].cgroup_permissions.as_deref(), Some("rwm"));
|
||||
|
||||
assert_eq!(
|
||||
sysctls
|
||||
.expect("wireguard needs src_valid_mark")
|
||||
.get("net.ipv4.conf.all.src_valid_mark")
|
||||
.map(String::as_str),
|
||||
Some("1")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn vpn_support_never_grants_more_than_net_admin() {
|
||||
// NET_ADMIN is already a step out of the sandbox. Anything else added
|
||||
// here (SYS_ADMIN, or a blanket privileged flag) would be a much larger
|
||||
// one, so pin the set.
|
||||
let (cap_add, _, _) = vpn_host_config(true);
|
||||
assert_eq!(cap_add.unwrap(), vec!["NET_ADMIN"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_vpn_skill_flag_is_emitted_either_way_never_omitted() {
|
||||
// The whole removal path depends on this. If `false` ever became "emit
|
||||
// nothing", a project that had the toggle on would keep the skill
|
||||
// forever: the container recreates from a snapshot whose baked
|
||||
// VPN_SUPPORT_ENABLED=1 would then go unchallenged, and entrypoint
|
||||
// would reinstall a skill for a capability the container no longer has.
|
||||
assert_eq!(vpn_env_var(true), "VPN_SUPPORT_ENABLED=1");
|
||||
assert_eq!(vpn_env_var(false), "VPN_SUPPORT_ENABLED=0");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_vpn_skill_flag_is_reserved_from_custom_env() {
|
||||
// entrypoint.sh installs and removes the pia-vpn skill from this
|
||||
// variable. A custom env var of the same name would let a project claim
|
||||
// the skill without the capability behind it — or keep it after the
|
||||
// toggle is off — so it has to be unsettable like the others.
|
||||
assert!(is_reserved_env_key("VPN_SUPPORT_ENABLED"));
|
||||
assert!(is_reserved_env_key("vpn_support_enabled"));
|
||||
assert_eq!(
|
||||
compute_env_fingerprint(&[EnvVar {
|
||||
key: "VPN_SUPPORT_ENABLED".to_string(),
|
||||
value: "1".to_string(),
|
||||
}]),
|
||||
""
|
||||
);
|
||||
}
|
||||
|
||||
/// What bollard actually hands us when a tun-less host rejects the device.
|
||||
///
|
||||
/// Captured verbatim from Docker 29.7: `docker create` with a missing
|
||||
/// device **succeeds**, and this arrives from the subsequent `start`.
|
||||
/// `DockerResponseServerError`'s Display is
|
||||
/// `"Docker responded with status code {code}: {message}"` with the
|
||||
/// daemon's message unaltered.
|
||||
const REAL_TUN_ERROR: &str = "Docker responded with status code 500: error \
|
||||
gathering device information while adding custom device \
|
||||
\"/dev/net/tun\": no such file or directory";
|
||||
|
||||
#[test]
|
||||
fn a_missing_tun_device_is_explained_on_the_path_that_actually_fails() {
|
||||
// The start path is the one that matters: the daemon defers device
|
||||
// resolution to runc, so create returns an id on a host with no tun
|
||||
// module and only start fails. A version of this that checked create
|
||||
// alone would be dead code.
|
||||
let msg = explain_container_failure("start", REAL_TUN_ERROR);
|
||||
assert!(msg.starts_with("Failed to start container:"), "{}", msg);
|
||||
assert!(msg.contains("VPN support"), "should name the switch: {}", msg);
|
||||
assert!(msg.contains("tun` module"), "should name the cause: {}", msg);
|
||||
assert!(msg.contains("Config → Runtime"), "should say where to fix it: {}", msg);
|
||||
assert!(msg.contains(REAL_TUN_ERROR), "should keep the original: {}", msg);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_same_explanation_covers_create_if_the_daemon_ever_checks_earlier() {
|
||||
// Belt and braces — older and future daemons may validate at create.
|
||||
let msg = explain_container_failure("create", REAL_TUN_ERROR);
|
||||
assert!(msg.starts_with("Failed to create container:"), "{}", msg);
|
||||
assert!(msg.contains("VPN support"), "{}", msg);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unrelated_failures_are_left_alone() {
|
||||
for (action, err) in [
|
||||
("create", "Conflict. The container name \"/triple-c-x\" is already in use"),
|
||||
("start", "Docker responded with status code 404: No such container"),
|
||||
("start", "error gathering device information while adding custom device \"/dev/dri/card0\""),
|
||||
] {
|
||||
assert_eq!(
|
||||
explain_container_failure(action, err),
|
||||
format!("Failed to {} container: {}", action, err),
|
||||
"{} should pass through untouched",
|
||||
err
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_orphan_sweep_only_ever_looks_at_our_own_untagged_images() {
|
||||
// Both conditions are load-bearing. Without `dangling` the sweep would
|
||||
// match `triple-c-snapshot-{id}:latest` — what every project is rebuilt
|
||||
// from — and a migration's `pre-migration-*` pin, which is the only copy
|
||||
// of a rollback target. Without the label it would match every dangling
|
||||
// image on the user's machine.
|
||||
let filters = orphan_sweep_filters();
|
||||
assert_eq!(filters.get("dangling"), Some(&vec!["true".to_string()]));
|
||||
assert_eq!(
|
||||
filters.get("label"),
|
||||
Some(&vec!["triple-c.managed=true".to_string()])
|
||||
);
|
||||
assert_eq!(filters.len(), 2, "an extra filter widens or narrows the sweep");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_custom_env_fingerprint_never_carries_the_value() {
|
||||
// It goes into `triple-c.custom-env-fingerprint`, which `docker inspect`
|
||||
|
||||
@@ -135,6 +135,8 @@ pub const FEATURE_PROBES: &[(&str, &str)] = &[
|
||||
("/usr/local/bin/triple-c-task-runner", "Scheduled task runner"),
|
||||
("/usr/local/bin/triple-c-sso-refresh", "AWS SSO auto-refresh"),
|
||||
("/opt/mission-control", "Mission Control (Flight Control)"),
|
||||
("/usr/bin/wg", "VPN tooling (WireGuard, for the VPN Support toggle)"),
|
||||
("/opt/triple-c-skills", "Bundled skills (PIA VPN, for the VPN Support toggle)"),
|
||||
];
|
||||
|
||||
/// Headroom demanded on Docker's storage backend on top of the measured
|
||||
|
||||
@@ -145,6 +145,22 @@ pub struct Project {
|
||||
/// container-recreation label.
|
||||
#[serde(default)]
|
||||
pub browser_view_enabled: bool,
|
||||
/// Grant the container what a VPN client needs to build a tunnel:
|
||||
/// `CAP_NET_ADMIN`, the `/dev/net/tun` device, and the WireGuard
|
||||
/// `src_valid_mark` sysctl. Without all three a client (PIA, WireGuard,
|
||||
/// OpenVPN) installs and runs but its connection attempt hangs until it
|
||||
/// times out, because it cannot create the tunnel interface or touch the
|
||||
/// routing table.
|
||||
///
|
||||
/// Off by default and deliberately opt-in: `NET_ADMIN` lets anything in the
|
||||
/// container reconfigure its own network stack, which reaches further than
|
||||
/// it sounds — see `vpn_host_config` for what it does and does not confer.
|
||||
/// Unlike `auth_bridge_enabled` this *is*
|
||||
/// container state, so it carries a `triple-c.vpn-support` label and is
|
||||
/// compared in `container_needs_recreation` — capabilities and devices are
|
||||
/// fixed at creation and can only change by recreating the container.
|
||||
#[serde(default)]
|
||||
pub vpn_support_enabled: bool,
|
||||
/// Use the shared, long-lived Claude Code OAuth token (from
|
||||
/// `claude setup-token`, held in the OS keychain) for this project instead
|
||||
/// of requiring its own `claude login`. Only consulted when `backend` is
|
||||
@@ -366,6 +382,7 @@ impl Project {
|
||||
mission_control_enabled: false,
|
||||
auth_bridge_enabled: false,
|
||||
browser_view_enabled: false,
|
||||
vpn_support_enabled: false,
|
||||
use_shared_auth_token: default_use_shared_auth_token(),
|
||||
full_permissions: false,
|
||||
permission_mode: None,
|
||||
|
||||
@@ -0,0 +1,112 @@
|
||||
import { describe, it, expect, vi, beforeEach, afterEach } from "vitest";
|
||||
import { render, screen, fireEvent, act } from "@testing-library/react";
|
||||
import AutomationTab from "./AutomationTab";
|
||||
import type { Project, ScheduledTask } from "../../../lib/types";
|
||||
|
||||
const listScheduledTasks = vi.fn(async () => tasks);
|
||||
const getSchedulerNotifications = vi.fn(async () => []);
|
||||
const runScheduledTaskNow = vi.fn(async () => "started");
|
||||
const pushToast = vi.fn();
|
||||
|
||||
vi.mock("../../../lib/tauri-commands", () => ({
|
||||
listScheduledTasks: () => listScheduledTasks(),
|
||||
getSchedulerNotifications: () => getSchedulerNotifications(),
|
||||
runScheduledTaskNow: (p: string, t: string) => runScheduledTaskNow(p, t),
|
||||
clearSchedulerNotifications: vi.fn(async () => {}),
|
||||
getScheduledTaskLog: vi.fn(async () => ""),
|
||||
removeScheduledTask: vi.fn(async () => {}),
|
||||
setScheduledTaskEnabled: vi.fn(async () => {}),
|
||||
}));
|
||||
|
||||
vi.mock("../../../store/appState", () => ({
|
||||
useAppState: (selector: (s: unknown) => unknown) => selector({ pushToast }),
|
||||
}));
|
||||
|
||||
const project = { id: "p1", name: "api", status: "running" } as unknown as Project;
|
||||
|
||||
const baseTask: ScheduledTask = {
|
||||
id: "a1b2c3d4",
|
||||
name: "nightly",
|
||||
prompt: "Run the suite",
|
||||
schedule: "0 3 * * *",
|
||||
task_type: "recurring",
|
||||
at: null,
|
||||
enabled: true,
|
||||
working_dir: "/workspace",
|
||||
created_at: null,
|
||||
last_run: null,
|
||||
next_run: null,
|
||||
running: false,
|
||||
running_since: null,
|
||||
};
|
||||
|
||||
let tasks: ScheduledTask[] = [];
|
||||
|
||||
async function renderTab() {
|
||||
render(<AutomationTab project={project} />);
|
||||
await act(async () => {
|
||||
await Promise.resolve();
|
||||
});
|
||||
}
|
||||
|
||||
beforeEach(() => {
|
||||
vi.useFakeTimers({ shouldAdvanceTime: true });
|
||||
tasks = [baseTask];
|
||||
listScheduledTasks.mockClear();
|
||||
runScheduledTaskNow.mockClear();
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
vi.useRealTimers();
|
||||
});
|
||||
|
||||
describe("AutomationTab run state", () => {
|
||||
it("offers Run now for an idle task and says nothing about running", async () => {
|
||||
await renderTab();
|
||||
expect(screen.getByRole("button", { name: "Run now" })).toBeEnabled();
|
||||
expect(screen.queryByText(/Running/)).toBeNull();
|
||||
});
|
||||
|
||||
it("shows a running task as running, with elapsed time, and blocks a second trigger", async () => {
|
||||
const startedSecondsAgo = new Date(Date.now() - 90_000).toISOString();
|
||||
tasks = [{ ...baseTask, running: true, running_since: startedSecondsAgo }];
|
||||
await renderTab();
|
||||
|
||||
// The whole point: a detached run is visible rather than silent.
|
||||
expect(screen.getByText(/Running for 1m/)).toBeTruthy();
|
||||
expect(screen.getByRole("button", { name: "Running…" })).toBeDisabled();
|
||||
});
|
||||
|
||||
it("keeps polling after a trigger, so a run that has not registered yet still appears", async () => {
|
||||
await renderTab();
|
||||
const callsAfterLoad = listScheduledTasks.mock.calls.length;
|
||||
|
||||
// The runner needs a moment to write its state file; until then the task
|
||||
// still reads as idle, which is exactly the window that used to look dead.
|
||||
await act(async () => {
|
||||
fireEvent.click(screen.getByRole("button", { name: "Run now" }));
|
||||
await Promise.resolve();
|
||||
});
|
||||
expect(runScheduledTaskNow).toHaveBeenCalledWith("p1", "a1b2c3d4");
|
||||
|
||||
tasks = [{ ...baseTask, running: true, running_since: new Date().toISOString() }];
|
||||
await act(async () => {
|
||||
vi.advanceTimersByTime(2000);
|
||||
await Promise.resolve();
|
||||
});
|
||||
|
||||
expect(listScheduledTasks.mock.calls.length).toBeGreaterThan(callsAfterLoad);
|
||||
expect(screen.getByRole("button", { name: "Running…" })).toBeDisabled();
|
||||
});
|
||||
|
||||
it("stops polling once nothing is running", async () => {
|
||||
await renderTab();
|
||||
// No trigger, nothing running: the interval must not be armed at all.
|
||||
const before = listScheduledTasks.mock.calls.length;
|
||||
await act(async () => {
|
||||
vi.advanceTimersByTime(30_000);
|
||||
await Promise.resolve();
|
||||
});
|
||||
expect(listScheduledTasks.mock.calls.length).toBe(before);
|
||||
});
|
||||
});
|
||||
@@ -15,7 +15,7 @@ import Toggle from "../../ui/Toggle";
|
||||
import Modal from "../../ui/Modal";
|
||||
import StatusIndicator from "../../ui/StatusIndicator";
|
||||
import TaskEditorModal from "./TaskEditorModal";
|
||||
import { formatAge } from "./format";
|
||||
import { formatAge, formatRunningFor } from "./format";
|
||||
|
||||
interface Props {
|
||||
project: Project;
|
||||
@@ -59,6 +59,22 @@ export default function AutomationTab({ project }: Props) {
|
||||
|
||||
useEffect(load, [load]);
|
||||
|
||||
// A task in flight is the one state this view cannot sit still for: runs are
|
||||
// detached, so without polling "Run now" looks like it did nothing until the
|
||||
// user reaches for Refresh. Polling stops as soon as nothing is running.
|
||||
//
|
||||
// `justTriggered` covers the gap between firing a run and the runner writing
|
||||
// its state file — a second or two in which the task still reads as idle, and
|
||||
// where giving up on polling would reproduce the exact silence this fixes.
|
||||
const anyTaskRunning = tasks.some((t) => t.running);
|
||||
const [justTriggered, setJustTriggered] = useState(0);
|
||||
useEffect(() => {
|
||||
if (!running) return;
|
||||
if (!anyTaskRunning && Date.now() - justTriggered > 20_000) return;
|
||||
const timer = setInterval(load, anyTaskRunning ? 5000 : 1500);
|
||||
return () => clearInterval(timer);
|
||||
}, [running, anyTaskRunning, justTriggered, load]);
|
||||
|
||||
const withTask = async (taskId: string, label: string, fn: () => Promise<unknown>) => {
|
||||
setBusyTaskId(taskId);
|
||||
try {
|
||||
@@ -185,6 +201,12 @@ export default function AutomationTab({ project }: Props) {
|
||||
<span className="text-[10px] uppercase tracking-wide px-1.5 py-0.5 rounded-[var(--radius-control)] bg-[var(--bg-tertiary)] text-[var(--text-secondary)]">
|
||||
{task.task_type}
|
||||
</span>
|
||||
{task.running && (
|
||||
<StatusIndicator
|
||||
tone="busy"
|
||||
label={`Running ${formatRunningFor(task.running_since) ?? ""}`.trim()}
|
||||
/>
|
||||
)}
|
||||
</div>
|
||||
<div className="text-xs text-[var(--text-secondary)] font-mono truncate">
|
||||
{task.at ?? task.schedule}
|
||||
@@ -202,14 +224,15 @@ export default function AutomationTab({ project }: Props) {
|
||||
}
|
||||
/>
|
||||
<Button
|
||||
disabled={busyTaskId === task.id}
|
||||
disabled={busyTaskId === task.id || task.running}
|
||||
onClick={() =>
|
||||
withTask(task.id, "Run now", () =>
|
||||
runScheduledTaskNow(project.id, task.id),
|
||||
)
|
||||
withTask(task.id, "Run now", async () => {
|
||||
await runScheduledTaskNow(project.id, task.id);
|
||||
setJustTriggered(Date.now());
|
||||
})
|
||||
}
|
||||
>
|
||||
Run now
|
||||
{task.running ? "Running…" : "Run now"}
|
||||
</Button>
|
||||
<Button disabled={busyTaskId === task.id} onClick={() => setEditing(task)}>
|
||||
Edit
|
||||
|
||||
@@ -127,6 +127,46 @@ describe("ContainerMigrationBanner", () => {
|
||||
expect(container).toBeEmptyDOMElement();
|
||||
});
|
||||
|
||||
it("speaks up when an unlabelled container could not be probed at all", () => {
|
||||
// The probe is the only signal a container with no lineage label has. If
|
||||
// it fails and the banner stays silent, that is indistinguishable from
|
||||
// "up to date" — the exact reading that let an out-of-date project go
|
||||
// unnoticed indefinitely.
|
||||
renderBanner(
|
||||
migration({
|
||||
staleness: {
|
||||
...FRESH,
|
||||
known: false,
|
||||
stale: false,
|
||||
probe_error: "output exceeded the inspection limit",
|
||||
},
|
||||
probeSettled: false,
|
||||
}),
|
||||
);
|
||||
expect(
|
||||
screen.getByText(/Container base could not be checked/i),
|
||||
).toBeInTheDocument();
|
||||
expect(
|
||||
screen.getByText(/output exceeded the inspection limit/i),
|
||||
).toBeInTheDocument();
|
||||
// And it must not pose as a finding about the container itself.
|
||||
expect(
|
||||
screen.queryByText(/Container is missing things/i),
|
||||
).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("stays quiet when a labelled container's probe fails but its lineage is current", () => {
|
||||
// `known` means the version comparison already answered the question, so
|
||||
// a failed probe is not grounds to raise anything.
|
||||
const { container } = renderBanner(
|
||||
migration({
|
||||
staleness: { ...FRESH, probe_error: "could not exec in the container" },
|
||||
probeSettled: false,
|
||||
}),
|
||||
);
|
||||
expect(container).toBeEmptyDOMElement();
|
||||
});
|
||||
|
||||
it("disables the action and explains why while the container is running", () => {
|
||||
renderBanner(migration({ staleness: STALE }), false);
|
||||
expect(
|
||||
|
||||
@@ -131,7 +131,15 @@ export default function ContainerMigrationBanner({
|
||||
const probeFoundGaps =
|
||||
!staleness.known &&
|
||||
(staleness.missing_features.length > 0 || staleness.missing_paths.length > 0);
|
||||
if (!staleness.stale && !probeFoundGaps) return null;
|
||||
|
||||
// The probe is the *only* signal a container with no lineage label has, so
|
||||
// when it fails there is nothing left to be quiet about. Staying silent here
|
||||
// is indistinguishable from "everything is fine" — and it is the likeliest
|
||||
// outcome for the oldest, largest projects, whose manifests are the ones apt
|
||||
// to exceed the inspection limit. Say that the check did not run instead.
|
||||
const probeUnavailable = !staleness.known && !!staleness.probe_error;
|
||||
|
||||
if (!staleness.stale && !probeFoundGaps && !probeUnavailable) return null;
|
||||
|
||||
const snapshot = formatSnapshotDate(staleness.snapshot_created_at);
|
||||
const features = joinFeatures(staleness.missing_features);
|
||||
@@ -139,15 +147,24 @@ export default function ContainerMigrationBanner({
|
||||
return (
|
||||
<section
|
||||
className={`${SHELL} border-[var(--warning)]/40 bg-[var(--warning-muted)]`}
|
||||
aria-label="Container base is out of date"
|
||||
aria-label={
|
||||
probeUnavailable
|
||||
? "Container base could not be checked"
|
||||
: "Container base is out of date"
|
||||
}
|
||||
>
|
||||
<div className="flex items-start justify-between gap-3">
|
||||
<div className="min-w-0 space-y-1">
|
||||
<StatusIndicator
|
||||
tone="error"
|
||||
// A check that could not run is not a finding: it gets the
|
||||
// "unresolved" tone rather than the one that says something is
|
||||
// wrong with the container.
|
||||
tone={probeUnavailable ? "unknown" : "error"}
|
||||
label={
|
||||
staleness.known
|
||||
? "Container base is out of date"
|
||||
: probeUnavailable
|
||||
? "Container base could not be checked"
|
||||
: "Container is missing things the current base ships"
|
||||
}
|
||||
className="text-[13px] font-semibold"
|
||||
@@ -158,6 +175,8 @@ export default function ContainerMigrationBanner({
|
||||
? snapshot
|
||||
? `Running on a saved image from ${snapshot}.`
|
||||
: "Running on a saved image older than the current base."
|
||||
: probeUnavailable
|
||||
? "This container predates base-image tracking, so probing it is the only way to tell whether it is behind — and that did not complete."
|
||||
: "This container predates base-image tracking, so it was probed directly."}
|
||||
</p>
|
||||
|
||||
|
||||
@@ -120,6 +120,15 @@ export default function OverviewTab({
|
||||
{project.mission_control_enabled ? "ON" : "OFF"}
|
||||
</span>
|
||||
</span>
|
||||
{/* Only when granted. It is off for nearly every project and an
|
||||
always-present "VPN OFF" would be noise, but where it *is* on the
|
||||
container holds NET_ADMIN, which is worth seeing at a glance. */}
|
||||
{project.vpn_support_enabled && (
|
||||
<span className="text-[var(--text-secondary)]">
|
||||
VPN support{" "}
|
||||
<span className="text-[var(--text-primary)] font-medium">ON</span>
|
||||
</span>
|
||||
)}
|
||||
<button
|
||||
type="button"
|
||||
onClick={() => onOpenTab("config")}
|
||||
|
||||
@@ -60,6 +60,8 @@ const existingTask: ScheduledTask = {
|
||||
created_at: null,
|
||||
last_run: null,
|
||||
next_run: null,
|
||||
running: false,
|
||||
running_since: null,
|
||||
};
|
||||
|
||||
async function renderEditor(task: ScheduledTask | null = null, project = baseProject) {
|
||||
|
||||
@@ -0,0 +1,94 @@
|
||||
import { describe, it, expect, vi, beforeEach } from "vitest";
|
||||
import { render, screen, fireEvent } from "@testing-library/react";
|
||||
import RuntimeSection from "./RuntimeSection";
|
||||
import type { Project } from "../../../../lib/types";
|
||||
|
||||
const baseProject: Project = {
|
||||
id: "p1",
|
||||
name: "api-server",
|
||||
paths: [{ host_path: "/src/api", mount_name: "api" }],
|
||||
container_id: null,
|
||||
status: "stopped",
|
||||
backend: "anthropic",
|
||||
bedrock_config: null,
|
||||
ollama_config: null,
|
||||
llamacpp_config: null,
|
||||
openai_compatible_config: null,
|
||||
allow_docker_access: false,
|
||||
sandbox_mode_enabled: true,
|
||||
mission_control_enabled: false,
|
||||
auth_bridge_enabled: false,
|
||||
browser_view_enabled: false,
|
||||
vpn_support_enabled: false,
|
||||
use_shared_auth_token: true,
|
||||
full_permissions: false,
|
||||
permission_mode: null,
|
||||
ssh_key_path: null,
|
||||
ca_cert_path: null,
|
||||
git_token: null,
|
||||
git_user_name: null,
|
||||
git_user_email: null,
|
||||
custom_env_vars: [],
|
||||
port_mappings: [],
|
||||
claude_instructions: null,
|
||||
claude_code_settings: null,
|
||||
renamed_session_names: {},
|
||||
created_at: "2026-01-01T00:00:00Z",
|
||||
updated_at: "2026-01-01T00:00:00Z",
|
||||
};
|
||||
|
||||
const VPN = "VPN support";
|
||||
|
||||
const save = vi.fn().mockResolvedValue(true);
|
||||
|
||||
function renderSection(over: Partial<Project> = {}, disabled = false) {
|
||||
return render(
|
||||
<RuntimeSection
|
||||
project={{ ...baseProject, ...over }}
|
||||
save={save}
|
||||
disabled={disabled}
|
||||
disabledReason="Container must be stopped to change this setting."
|
||||
/>,
|
||||
);
|
||||
}
|
||||
|
||||
describe("RuntimeSection — VPN support toggle", () => {
|
||||
beforeEach(() => vi.clearAllMocks());
|
||||
|
||||
it("saves only the VPN flag when switched on", () => {
|
||||
renderSection();
|
||||
fireEvent.click(screen.getByRole("switch", { name: VPN }));
|
||||
expect(save).toHaveBeenCalledWith({ vpn_support_enabled: true });
|
||||
});
|
||||
|
||||
it("saves the flag off again, rather than dropping the key", () => {
|
||||
// Off has to be written explicitly: the container carries a
|
||||
// `triple-c.vpn-support` label either way, and an absent value would leave
|
||||
// the capability granted.
|
||||
renderSection({ vpn_support_enabled: true });
|
||||
fireEvent.click(screen.getByRole("switch", { name: VPN }));
|
||||
expect(save).toHaveBeenCalledWith({ vpn_support_enabled: false });
|
||||
});
|
||||
|
||||
it("reflects the project's current state", () => {
|
||||
renderSection({ vpn_support_enabled: true });
|
||||
expect(screen.getByRole("switch", { name: VPN })).toBeChecked();
|
||||
});
|
||||
|
||||
it("cannot be changed while the container is running", () => {
|
||||
// Capabilities and devices are fixed at creation, so this setting is gated
|
||||
// on the container being stopped along with the rest of the tab.
|
||||
renderSection({}, true);
|
||||
const toggle = screen.getByRole("switch", { name: VPN });
|
||||
expect(toggle).toBeDisabled();
|
||||
fireEvent.click(toggle);
|
||||
expect(save).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("warns that the change recreates the container", () => {
|
||||
renderSection();
|
||||
expect(
|
||||
screen.getByText(/recreates the container on its next start/i),
|
||||
).toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
@@ -57,6 +57,19 @@ export default function RuntimeSection({
|
||||
}
|
||||
/>
|
||||
|
||||
<SwitchRow
|
||||
label="VPN support"
|
||||
hint="Grants NET_ADMIN and the /dev/net/tun device so a VPN client (PIA, WireGuard, OpenVPN) can build a tunnel inside the container. Without it a client installs and runs but its connection hangs until it times out. Anything in the container can then reconfigure the container's own network stack; the host's is untouched. Changing this recreates the container on its next start — the home and .claude volumes are preserved."
|
||||
control={
|
||||
<Toggle
|
||||
label="VPN support"
|
||||
checked={project.vpn_support_enabled}
|
||||
disabled={disabled}
|
||||
onChange={(v) => save({ vpn_support_enabled: v })}
|
||||
/>
|
||||
}
|
||||
/>
|
||||
|
||||
<SwitchRow
|
||||
label="Mission Control"
|
||||
hint="A web dashboard for monitoring and managing Claude sessions remotely."
|
||||
|
||||
@@ -26,6 +26,21 @@ export function formatElapsed(ms: number): string {
|
||||
return `${days}d ago`;
|
||||
}
|
||||
|
||||
/** "for 42s" / "for 4m" / "for 1h 12m" — elapsed phrasing for a run in flight.
|
||||
* Seconds are kept below a minute because the first thing anyone wants from a
|
||||
* freshly triggered run is evidence that it started at all. */
|
||||
export function formatRunningFor(iso: string | null | undefined): string | null {
|
||||
if (!iso) return null;
|
||||
const started = Date.parse(iso);
|
||||
if (Number.isNaN(started)) return null;
|
||||
const seconds = Math.max(0, Math.floor((Date.now() - started) / 1000));
|
||||
if (seconds < 60) return `for ${seconds}s`;
|
||||
const minutes = Math.floor(seconds / 60);
|
||||
if (minutes < 60) return `for ${minutes}m`;
|
||||
const hours = Math.floor(minutes / 60);
|
||||
return `for ${hours}h ${minutes % 60}m`;
|
||||
}
|
||||
|
||||
/** Uptime phrasing for a known start timestamp. */
|
||||
export function formatUptime(startedAtMs: number | undefined): string | null {
|
||||
if (startedAtMs === undefined) return null;
|
||||
|
||||
@@ -33,6 +33,10 @@ export interface Project {
|
||||
auth_bridge_enabled: boolean;
|
||||
/** Opt in to the browser-view pane. Host-side only, like `auth_bridge_enabled`. */
|
||||
browser_view_enabled: boolean;
|
||||
/** Grant NET_ADMIN, /dev/net/tun and the WireGuard `src_valid_mark` sysctl so
|
||||
* a VPN client inside the container can build a tunnel. Unlike the two flags
|
||||
* above this is container state — changing it recreates the container. */
|
||||
vpn_support_enabled: boolean;
|
||||
/** Use the shared long-lived Claude Code token (from `claude setup-token`,
|
||||
* held in the OS keychain) instead of this project's own `claude login`.
|
||||
* Defaults to true; only applies when `backend` is "anthropic" and a token
|
||||
@@ -388,6 +392,10 @@ export interface ScheduledTask {
|
||||
last_run: string | null;
|
||||
/** Known only for enabled one-shot tasks; cron is not evaluated. */
|
||||
next_run: string | null;
|
||||
/** A run is in flight right now (the runner's pid was verified live). */
|
||||
running: boolean;
|
||||
/** When that run started. Null unless `running`. */
|
||||
running_since: string | null;
|
||||
}
|
||||
|
||||
/** Mirrors Rust `ScheduleKind` — which of the scheduler's two `add` flags to
|
||||
|
||||
@@ -34,6 +34,9 @@ RUN for i in 1 2 3 4 5; do \
|
||||
cron \
|
||||
bubblewrap \
|
||||
socat \
|
||||
iproute2 \
|
||||
wireguard-tools \
|
||||
iptables \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# `libnss3-tools` above provides `certutil`. Chrome/Chromium read neither
|
||||
@@ -42,6 +45,76 @@ RUN for i in 1 2 3 4 5; do \
|
||||
# corporate CA, no matter what the system trust store says. entrypoint.sh
|
||||
# degrades to a warning if it is ever missing.
|
||||
|
||||
# `iproute2`, `wireguard-tools` and `iptables` above are what the VPN support
|
||||
# toggle (`vpn_support_enabled`) grants capability *for*. That toggle hands a
|
||||
# project CAP_NET_ADMIN and /dev/net/tun; without `ip` there is then no way to
|
||||
# add a route, and without `wg` no way to build the tunnel those two exist to
|
||||
# serve — a capability with nothing able to use it.
|
||||
#
|
||||
# They are baked rather than left to a runtime `apt-get install` for the same
|
||||
# reason as the Playwright libraries below: the writable layer is re-paid after
|
||||
# every Reset and lost on base-image migration. A hand-installed `wg` therefore
|
||||
# works right up until an upgrade, then disappears and takes the tunnel with it
|
||||
# — silently, since a VPN that fails to come up looks exactly like one that was
|
||||
# never started.
|
||||
#
|
||||
# Measured against the *current base image*, since a bare ubuntu:24.04 also
|
||||
# pulls libelf1t64 and netbase, which this base already has, and so over-reports
|
||||
# by ~258 kB: **+12 packages, 7,203 kB on amd64**. The same set on arm64 is
|
||||
# ~14.4 MB — the package list is identical on both arches, the binaries are
|
||||
# simply larger (measured as 14.7 MB on arm64 ubuntu:24.04, less that 258 kB).
|
||||
#
|
||||
# ## Why `iptables`, and not `nftables`
|
||||
#
|
||||
# `wireguard-tools` declares `Recommends: nftables | iptables`, which the
|
||||
# `--no-install-recommends` above strips. That is not cosmetic: `wg-quick`'s
|
||||
# `add_default()` runs whenever a config has `AllowedIPs = 0.0.0.0/0` — i.e.
|
||||
# every stock full-tunnel config every provider hands out — and it shells out to
|
||||
# a firewall backend with no `type -p` guard. Measured with neither installed:
|
||||
#
|
||||
# [#] iptables-restore -n
|
||||
# /usr/bin/wg-quick: line 32: iptables-restore: command not found
|
||||
# wg-quick EXIT=127
|
||||
#
|
||||
# `nftables` looks like the better pick — wg-quick prefers it, it is first in
|
||||
# that Recommends, it is half the size — and it is the wrong one. wg-quick picks
|
||||
# nft *unconditionally* when present (`if type -p nft`, line 241), so installing
|
||||
# it makes the iptables path unreachable; and its nft ruleset needs three
|
||||
# expression families where the iptables path needs one. Isolating them on a
|
||||
# WSL2 host, the two connmark rules install fine and this is what fails:
|
||||
#
|
||||
# nft add rule ... fib saddr type != local drop
|
||||
# Error: Could not process rule: No such file or directory
|
||||
# ^^^^^^^^^^^^^^ needs nft_fib_ipv4
|
||||
#
|
||||
# That matters because of how the two hosts we ship to are configured. From
|
||||
# LinuxKit's kernel config — Docker Desktop for Mac, identical on both arches:
|
||||
#
|
||||
# CONFIG_NETFILTER_XT_CONNMARK=y <- the iptables path works
|
||||
# # CONFIG_NFT_FIB_IPV4 is not set <- the nft path does not
|
||||
#
|
||||
# So shipping `nftables` would forfeit the platform it was meant to fix. With
|
||||
# `iptables`, full tunnels work on native Linux, on Docker Desktop for Mac, and
|
||||
# on WSL2 kernels from 6.6 (which added xt_CONNMARK as a module). Only WSL2
|
||||
# older than that is left out, and nothing installable here changes it — the way
|
||||
# out there is to add the routes with `ip route` instead of using `wg-quick`,
|
||||
# which is what the pia-vpn skill does on every platform.
|
||||
#
|
||||
# ## What this still does not fix
|
||||
#
|
||||
# `wireguard-tools` only *Suggests* `openresolv | resolvconf`, so neither is
|
||||
# installed, and every provider's stock config carries a `DNS =` line. That
|
||||
# fails in `set_dns()`, *before* the firewall step, so it takes split tunnels
|
||||
# down too:
|
||||
#
|
||||
# [#] resolvconf -a wg0 -m 0 -x
|
||||
# /usr/bin/wg-quick: line 32: resolvconf: command not found
|
||||
#
|
||||
# Deliberately not fixed here: `openresolv` has no installation candidate on
|
||||
# noble, and `resolvconf` resolves only by pulling in systemd-resolved — a
|
||||
# resolver daemon and systemd units, into a container with no systemd. Strip the
|
||||
# `DNS =` line and set the resolver another way. Documented in HOW-TO-USE.md.
|
||||
|
||||
# Remove default ubuntu user to free UID 1000 for host-user remapping
|
||||
RUN if id ubuntu >/dev/null 2>&1; then userdel -r ubuntu 2>/dev/null || userdel ubuntu; fi \
|
||||
&& if getent group ubuntu >/dev/null 2>&1; then groupdel ubuntu 2>/dev/null || true; fi
|
||||
@@ -314,10 +387,28 @@ RUN chmod +x /usr/local/bin/triple-c-sso-refresh
|
||||
|
||||
COPY mission-control /opt/mission-control
|
||||
|
||||
# Skills that ship with a Triple-C feature rather than with Mission Control.
|
||||
# entrypoint.sh installs them into ~/.claude/skills/ when the feature that owns
|
||||
# them is enabled, and removes them when it is not — a skill telling an agent to
|
||||
# build a tunnel in a container that no longer has CAP_NET_ADMIN is worse than
|
||||
# no skill at all. Staged in /opt because ~/.claude is a volume mount: an image
|
||||
# copy underneath it would be masked from the project's first start onward.
|
||||
COPY skills /opt/triple-c-skills
|
||||
# `find`, not a `*/*.sh` glob: the glob fails the build the day a skill ships
|
||||
# without a script, which is a legitimate thing for a skill to do.
|
||||
RUN find /opt/triple-c-skills -name '*.sh' -exec chmod +x {} +
|
||||
|
||||
COPY entrypoint.sh /usr/local/bin/entrypoint.sh
|
||||
RUN chmod +x /usr/local/bin/entrypoint.sh
|
||||
COPY triple-c-scheduler /usr/local/bin/triple-c-scheduler
|
||||
RUN chmod +x /usr/local/bin/triple-c-scheduler
|
||||
# Lives in /usr/local/bin rather than under /home/claude on purpose: the home
|
||||
# directory is the mount point of the project's home volume, so an image copy
|
||||
# of it is masked after the project's first start and can never be updated
|
||||
# again. /usr/local/bin rides the snapshot and is replaced on migration, which
|
||||
# is what lets a fix to this script reach an existing project at all.
|
||||
COPY triple-c-playwright-heal /usr/local/bin/triple-c-playwright-heal
|
||||
RUN chmod +x /usr/local/bin/triple-c-playwright-heal
|
||||
COPY triple-c-task-runner /usr/local/bin/triple-c-task-runner
|
||||
RUN chmod +x /usr/local/bin/triple-c-task-runner
|
||||
|
||||
|
||||
+94
-1
@@ -338,6 +338,64 @@ if [ "$MISSION_CONTROL_ENABLED" = "1" ]; then
|
||||
unset MISSION_CONTROL_ENABLED
|
||||
fi
|
||||
|
||||
# ── Feature skills ──────────────────────────────────────────────────────────
|
||||
# Skills owned by a Triple-C feature rather than by Mission Control. Installed
|
||||
# when the feature is on, removed when it is off: ~/.claude is a persisted
|
||||
# volume, so a skill left behind after its feature is disabled would keep
|
||||
# telling an agent to use a capability the container no longer has.
|
||||
#
|
||||
# Copied on every start rather than only when absent, so a fix to a skill
|
||||
# reaches projects that already have the old copy. Local edits under these
|
||||
# directories do not survive — treat /opt/triple-c-skills as the source.
|
||||
#
|
||||
# The source lives in the *base image*, so a project whose container predates it
|
||||
# recreates from its own snapshot and has no /opt/triple-c-skills to copy from.
|
||||
# That case says so rather than returning silently: the toggle is on, the
|
||||
# capability is there, and the skill simply never appears — which is impossible
|
||||
# to work out from the outside.
|
||||
install_feature_skill() {
|
||||
local _name="$1"
|
||||
local _enabled="$2"
|
||||
local _src="/opt/triple-c-skills/$1"
|
||||
local _dest="/home/claude/.claude/skills/$1"
|
||||
|
||||
# Reject anything that is not a plain directory name. The disabled branch
|
||||
# `rm -rf`s $_dest under a *persisted volume*, so a blank name would take the
|
||||
# whole skills directory (Mission Control's included) and `../x` would escape
|
||||
# it entirely. Only the literal `pia-vpn` is passed today; this is so that
|
||||
# stays true.
|
||||
case "$_name" in
|
||||
''|*/*|.*) echo "entrypoint: install_feature_skill: bad skill name '$_name'"; return 1 ;;
|
||||
esac
|
||||
|
||||
if [ "$_enabled" = "1" ]; then
|
||||
if [ ! -d "$_src" ]; then
|
||||
echo "entrypoint: $_name skill unavailable — this container's base image predates it; migrate the project to get it"
|
||||
return 0
|
||||
fi
|
||||
# Checked, not assumed: with no `set -e` in this script every step here
|
||||
# can fail (full volume, read-only mount, a file where the directory
|
||||
# should be) and the success line would still print.
|
||||
mkdir -p /home/claude/.claude/skills || {
|
||||
echo "entrypoint: $_name skill install FAILED (cannot create ~/.claude/skills)"; return 1; }
|
||||
# Not just $_dest: when Mission Control is off nothing else creates the
|
||||
# parent, so root would own it and `claude` could not add a skill there.
|
||||
chown claude:claude /home/claude/.claude/skills
|
||||
rm -rf "$_dest"
|
||||
cp -r "$_src" "$_dest" || {
|
||||
echo "entrypoint: $_name skill install FAILED (copy from $_src)"; return 1; }
|
||||
chown -R claude:claude "$_dest"
|
||||
echo "entrypoint: $_name skill installed to ~/.claude/skills/"
|
||||
elif [ -e "$_dest" ] || [ -L "$_dest" ]; then
|
||||
# -e/-L rather than -d: a leftover *file* at that path must go too.
|
||||
rm -rf "$_dest"
|
||||
echo "entrypoint: $_name skill removed (feature disabled)"
|
||||
fi
|
||||
}
|
||||
|
||||
install_feature_skill pia-vpn "${VPN_SUPPORT_ENABLED:-0}"
|
||||
unset VPN_SUPPORT_ENABLED
|
||||
|
||||
# ── Claude Code settings ────────────────────────────────────────────────────
|
||||
# Merge Claude Code settings into ~/.claude/settings.json (preserves existing
|
||||
# keys). Creates the file if it doesn't exist. These control TUI mode, effort
|
||||
@@ -425,6 +483,32 @@ if [ -x /usr/local/bin/triple-c-open ]; then
|
||||
export BROWSER=/usr/local/bin/triple-c-open
|
||||
fi
|
||||
|
||||
# ── Playwright browser config ───────────────────────────────────────────────
|
||||
# Seed ~/.playwright/cli.config.json on every start.
|
||||
#
|
||||
# Without it `playwright-cli` resolves to channel `chrome` — system Google
|
||||
# Chrome — with the Chromium sandbox ON, and these containers do not permit
|
||||
# unprivileged user namespaces, so the browser aborts with "Failed to move to
|
||||
# new namespace ... Operation not permitted". On a base image that no longer
|
||||
# ships Google Chrome the same default fails the other way, with "Chromium
|
||||
# distribution 'chrome' is not found". One cause, two error messages, and
|
||||
# neither of them looks like a configuration problem.
|
||||
#
|
||||
# Seeded here rather than baked into the image because ~/.playwright is inside
|
||||
# the home volume: an image copy would reach new projects only, and every
|
||||
# existing project would stay broken forever. Written on every start from a
|
||||
# source outside the volume, the way CLAUDE_INSTRUCTIONS and the Mission
|
||||
# Control skills already are.
|
||||
#
|
||||
# --seed-config-only is the cheap path: no npm install, no browser download, no
|
||||
# apt, no verify launch, nothing over the network. It writes one small file if
|
||||
# it is absent and returns. Measured at ~2 ms. The heavier repairs stay
|
||||
# on-demand — run `triple-c-playwright-heal` with no arguments for those.
|
||||
if [ -x /usr/local/bin/triple-c-playwright-heal ]; then
|
||||
/usr/local/bin/triple-c-playwright-heal --seed-config-only --quiet || \
|
||||
echo "entrypoint: warning — playwright config seeding failed (browser view may not launch)"
|
||||
fi
|
||||
|
||||
# ── Scheduler setup ─────────────────────────────────────────────────────────
|
||||
SCHEDULER_DIR="/home/claude/.claude/scheduler"
|
||||
mkdir -p "$SCHEDULER_DIR/tasks" "$SCHEDULER_DIR/logs" "$SCHEDULER_DIR/notifications"
|
||||
@@ -434,17 +518,26 @@ chown -R claude:claude "$SCHEDULER_DIR"
|
||||
cron
|
||||
|
||||
# Save environment variables for cron jobs (cron runs with a minimal env)
|
||||
#
|
||||
# HOME is deliberately NOT captured here. This entrypoint runs as root, so the
|
||||
# snapshot would record HOME=/root — and the task runner sources this file with
|
||||
# `set -a`, which would overwrite the HOME cron gives the job. Claude Code then
|
||||
# looks for its OAuth credential at /root/.claude/.credentials.json instead of
|
||||
# /home/claude/.claude/.credentials.json and every scheduled task dies with
|
||||
# "Not logged in · Please run /login". Cron still needs a HOME, so it is written
|
||||
# explicitly below with the value the `claude` user actually has.
|
||||
ENV_FILE="$SCHEDULER_DIR/.env"
|
||||
: > "$ENV_FILE"
|
||||
env | while IFS='=' read -r key value; do
|
||||
case "$key" in
|
||||
ANTHROPIC_*|AWS_*|CLAUDE_CODE_*|TRIPLE_C_PERMISSION_MODE|PATH|HOME|LANG|TZ|COLORTERM|BROWSER|NODE_EXTRA_CA_CERTS|REQUESTS_CA_BUNDLE|SSL_CERT_FILE)
|
||||
ANTHROPIC_*|AWS_*|CLAUDE_CODE_*|TRIPLE_C_PERMISSION_MODE|PATH|LANG|TZ|COLORTERM|BROWSER|NODE_EXTRA_CA_CERTS|REQUESTS_CA_BUNDLE|SSL_CERT_FILE)
|
||||
# Escape single quotes in value and write as KEY='VALUE'
|
||||
escaped_value=$(printf '%s' "$value" | sed "s/'/'\\\\''/g")
|
||||
printf "%s='%s'\n" "$key" "$escaped_value" >> "$ENV_FILE"
|
||||
;;
|
||||
esac
|
||||
done
|
||||
printf "HOME='/home/claude'\n" >> "$ENV_FILE"
|
||||
chown claude:claude "$ENV_FILE"
|
||||
chmod 600 "$ENV_FILE"
|
||||
|
||||
|
||||
@@ -0,0 +1,225 @@
|
||||
---
|
||||
name: pia-vpn
|
||||
description: Connect this container's traffic through a PIA VPN tunnel over WireGuard, or diagnose one that is not working. Use when asked to enable, route through, check, or tear down a VPN, when traffic needs to leave from a different location, or when DNS or connectivity broke after a VPN was brought up.
|
||||
---
|
||||
|
||||
# PIA VPN
|
||||
|
||||
Bring this container's traffic out through Private Internet Access over
|
||||
WireGuard, using the API PIA documents for headless use.
|
||||
|
||||
Run `sudo ~/.claude/skills/pia-vpn/pia-wg.sh` with `up`, `up --full`, `down` or
|
||||
`status`. Read the rest of this page before the first `up --full` — three of the
|
||||
behaviours below are actively misleading if you meet them without warning, and
|
||||
each one presents as "the VPN is fine" or "Claude is broken" rather than as
|
||||
what it is.
|
||||
|
||||
## Before anything else: what the toggle does not do
|
||||
|
||||
Triple-C's **VPN support** setting grants three things — `CAP_NET_ADMIN`, the
|
||||
`/dev/net/tun` device, and the `net.ipv4.conf.all.src_valid_mark` sysctl — and
|
||||
stops there. It starts no client, builds no tunnel and changes no route.
|
||||
|
||||
So "the VPN is enabled but traffic isn't going through it" is normally not a
|
||||
fault. It means the capability is present and nothing has used it yet. Check
|
||||
with `status` before assuming something is broken.
|
||||
|
||||
If the toggle is off, the script says so and names the setting. It cannot be
|
||||
turned on from inside the container; the user changes it in Config → Runtime,
|
||||
and it recreates the container on the next start (home and `.claude` volumes
|
||||
are preserved — it is not a Reset).
|
||||
|
||||
## Two modes
|
||||
|
||||
| | routes | use when |
|
||||
|---|---|---|
|
||||
| `up` | only `1.1.1.1/32` | verifying the tunnel works without disturbing anything |
|
||||
| `up --full` | all public traffic | you actually want traffic leaving via PIA |
|
||||
|
||||
Prefer `up` first. It proves the handshake, credentials and region are good
|
||||
while your own connectivity is untouched, so a failure is cheap.
|
||||
|
||||
**`up --full` routes Claude Code's own API traffic through PIA.** If the tunnel
|
||||
drops, that traffic stops until it recovers or you run `down`. Say so before
|
||||
running it — the user may be mid-session, and they will experience the failure
|
||||
as Claude going away, not as a VPN problem.
|
||||
|
||||
## Trap 1: a full tunnel takes DNS with it
|
||||
|
||||
The container resolves through an address on the Docker network — under Docker
|
||||
Desktop, `192.168.65.7` — which sits **outside** the container's own subnet. A
|
||||
default route of `0.0.0.0/0`, or the `0.0.0.0/1` + `128.0.0.0/1` pair, captures
|
||||
it and posts every lookup into a tunnel that cannot carry private traffic.
|
||||
|
||||
Nothing resolves after that. The visible symptom is Claude Code reporting it
|
||||
cannot connect, because `api.anthropic.com` no longer resolves:
|
||||
|
||||
```
|
||||
$ curl https://api.anthropic.com/v1/messages
|
||||
* Could not resolve host: api.anthropic.com (rc=6)
|
||||
```
|
||||
|
||||
`pia-wg.sh` already handles this: it routes `10.0.0.0/8`, `172.16.0.0/12`,
|
||||
`192.168.0.0/16` and `169.254.0.0/16` back via the original gateway, then pins
|
||||
PIA's own resolvers through the tunnel with `/32` routes that outrank the
|
||||
`10/8` exclusion. If you ever route traffic by hand, you owe both halves — the
|
||||
exclusions *and* a resolver reachable from wherever you pointed the default.
|
||||
|
||||
The failure has a quiet twin. Do only the first half — exclude the private
|
||||
ranges, leave the resolver alone — and everything *works*, while every DNS
|
||||
query travels outside the tunnel to your ISP. A VPN that leaks the full list of
|
||||
what you looked up is worse than one that is visibly broken, so `up --full`
|
||||
refuses to proceed if PIA does not hand back resolvers rather than carrying on
|
||||
without them.
|
||||
|
||||
The mechanism above is Docker Desktop's. On a user-defined Docker network the
|
||||
resolver is `127.0.0.11`, which is loopback and never captured by a default
|
||||
route — the trap still exists there (that resolver forwards upstream from
|
||||
inside the container's namespace) but arrives by a different path. Check
|
||||
`/etc/resolv.conf` rather than assuming which case you are in.
|
||||
|
||||
## Trap 2: an IP-literal health check cannot see a dead resolver
|
||||
|
||||
`curl https://1.1.1.1/cdn-cgi/trace` needs no DNS, so it returns a cheerful
|
||||
PIA exit address while name resolution is entirely broken. A tunnel verified
|
||||
that way looks perfect and works for nothing.
|
||||
|
||||
`status` resolves a real name for this reason. Trust its `DNS:` line, and if
|
||||
you check by hand, resolve a name rather than fetching an address.
|
||||
|
||||
## Trap 3: in test mode, the obvious probe is the one thing tunnelled
|
||||
|
||||
`up` routes `1.1.1.1` and nothing else. So checking your address by fetching
|
||||
`https://1.1.1.1/cdn-cgi/trace` reports a **PIA** address — not because your
|
||||
traffic is going through PIA, but because that single probe is. Everything else
|
||||
still leaves directly.
|
||||
|
||||
This reads exactly like a working full tunnel, and it is the likeliest reason
|
||||
someone concludes the VPN is on when it is not. `status` prints both exits in
|
||||
test mode for this reason:
|
||||
|
||||
```
|
||||
mode: test route only (1.1.1.1 through the tunnel, nothing else)
|
||||
through the tunnel: 64.113.5.73
|
||||
everything else: 172.116.197.166 <- your real address
|
||||
```
|
||||
|
||||
Two different addresses there is correct and expected in test mode. If you want
|
||||
the second line to change, you want `up --full`.
|
||||
|
||||
## Trap 4: no tunnel survives a restart, and it fails open
|
||||
|
||||
The network namespace is rebuilt every time the container starts, and nothing
|
||||
inside reconnects anything. After a stop/start, Reset or any config change that
|
||||
recreates the container, the interface and its routes are gone.
|
||||
|
||||
State under `/run/pia-wg` rides the snapshot and persists, so leftover files
|
||||
make it look as though the tunnel is still configured. It is not. Traffic goes
|
||||
out the real address with no error and nothing visibly different.
|
||||
|
||||
Never infer from `/run/pia-wg` that a tunnel is up. Run `status` — if the
|
||||
handshake line is missing, there is no tunnel. Re-run `up` after every start.
|
||||
|
||||
## Credentials
|
||||
|
||||
Two lines in `~/pia-creds` — username, then password:
|
||||
|
||||
```
|
||||
p1234567
|
||||
your-password
|
||||
```
|
||||
|
||||
Treat the contents as secret: never print the file, never echo the values, and
|
||||
never include them in a commit, a log or a message. The script reads it directly
|
||||
and does not echo it, and passes PIA's session token to `curl` on stdin rather
|
||||
than in the argv, where `ps` would expose it to everything in the container.
|
||||
|
||||
`PIA_CREDS` points somewhere else — but `sudo` resets the environment, so it
|
||||
only takes effect **after** the word `sudo`:
|
||||
|
||||
```bash
|
||||
sudo PIA_CREDS=/path/to/creds ~/.claude/skills/pia-vpn/pia-wg.sh up # works
|
||||
PIA_CREDS=/path/to/creds sudo ~/.claude/skills/pia-vpn/pia-wg.sh up # ignored
|
||||
```
|
||||
|
||||
The second form fails silently back to the default path. Same for `PIA_REGION`.
|
||||
|
||||
## Regions
|
||||
|
||||
Defaults to `us_chicago`. Override with `PIA_REGION`:
|
||||
|
||||
```bash
|
||||
sudo PIA_REGION=uk_london ~/.claude/skills/pia-vpn/pia-wg.sh up --full
|
||||
```
|
||||
|
||||
List the ids:
|
||||
|
||||
```bash
|
||||
curl -s https://serverlist.piaservers.net/vpninfo/servers/v6 \
|
||||
| head -1 | jq -r '.regions[].id'
|
||||
```
|
||||
|
||||
## Verifying
|
||||
|
||||
`status` prints the handshake, DNS, and which address traffic actually leaves
|
||||
from — labelled by mode, so the answer cannot be misread:
|
||||
|
||||
```
|
||||
latest handshake: 2 seconds ago
|
||||
transfer: 92 B received, 180 B sent
|
||||
DNS: ok (via 10.0.0.243 10.0.0.242)
|
||||
mode: full tunnel
|
||||
all traffic exits: 64.113.5.244
|
||||
```
|
||||
|
||||
All of it matters. A handshake with `DNS: BROKEN` is trap 1. `mode: test route
|
||||
only` with two different addresses is trap 3, and is correct — it means the
|
||||
tunnel works and you have not asked for it to carry anything yet. Report the
|
||||
mode line when telling someone the VPN is on; "the public IP is a PIA one" is
|
||||
true in test mode too, and means much less than it sounds like.
|
||||
|
||||
## Tearing down
|
||||
|
||||
`down` restores `resolv.conf` from its backup (only if that backup still looks
|
||||
like a resolver file — restoring a truncated one would leave the container with
|
||||
no DNS at all), removes exactly the routes that were added, in reverse order,
|
||||
and deletes the interface. It is safe to run when nothing is up. Confirm
|
||||
afterwards that the public address is back to the container's own.
|
||||
|
||||
`up` calls it too, but only *after* every network fetch has succeeded, so a
|
||||
failed `up` leaves an existing tunnel alone rather than tearing it down to
|
||||
report a bad password. From that point on a rollback is armed: if any step of
|
||||
the setup fails, the tunnel is torn down rather than left half-configured.
|
||||
|
||||
The private key is deleted earlier still — the moment `wg set` has read it,
|
||||
while the tunnel is being built. That is not housekeeping: `/run` is in the
|
||||
container's writable layer, and recreating or migrating the project runs
|
||||
`docker commit` over it *without* tearing the tunnel down first. A key that
|
||||
lived for the tunnel's lifetime would be baked into the snapshot image and
|
||||
copied forward from then on. The kernel keeps its own copy, so nothing is lost.
|
||||
|
||||
## What this deliberately does not do
|
||||
|
||||
- **No killswitch.** `iptables` *is* in the image, so one is buildable — this
|
||||
is a deliberate omission, not a missing dependency. Blocking non-tunnel egress
|
||||
cuts Claude Code's own API traffic the moment the tunnel drops, which ends the
|
||||
session that would otherwise fix it. If the user needs guaranteed egress
|
||||
rather than convenient egress, say so plainly and let them decide, rather than
|
||||
improvising one.
|
||||
- **No autostart.** There is no service manager in the container and Triple-C
|
||||
has no start hook, so nothing re-establishes the tunnel on its own. `cron` is
|
||||
in the image and `triple-c-scheduler` runs on it, so a scheduled reconnect is
|
||||
possible if the user wants one — it is just not set up, and a tunnel that
|
||||
reconnects unattended deserves an explicit decision.
|
||||
- **Not PIA's desktop client.** `pia-daemon` and `piactl` are installable but
|
||||
cannot work headless: the daemon never accepts a client connection without
|
||||
the GUI, and `piactl --help` states that connecting requires it. If you find
|
||||
one installed, it is not a working alternative to this script.
|
||||
- **Not `wg-quick`.** Its `Table=auto` full-tunnel mode routes by firewall mark
|
||||
and needs `xt_CONNMARK` from the host kernel, which Docker Desktop for
|
||||
Windows (WSL2) does not have and a container cannot load. This script adds
|
||||
the routes with `ip route` directly, which works on every host.
|
||||
- **IPv4 only.** The `0.0.0.0/1` + `128.0.0.0/1` pair covers v4. A container
|
||||
with a global IPv6 address and a v6 default route would leak all v6 traffic
|
||||
outside the tunnel; Triple-C's containers do not have one by default, but
|
||||
check `ip -6 route show default` before relying on this where it matters.
|
||||
@@ -0,0 +1,318 @@
|
||||
#!/usr/bin/env bash
|
||||
# PIA over WireGuard, headless.
|
||||
#
|
||||
# PIA's desktop client (pia-daemon + piactl) cannot work here: its daemon never
|
||||
# accepts a client connection without the GUI running, and `piactl --help` says
|
||||
# as much. This talks to PIA's public API directly instead, which is the path
|
||||
# PIA themselves document for headless use.
|
||||
#
|
||||
# sudo pia-wg.sh up tunnel up, only 1.1.1.1 routed through it (safe test)
|
||||
# sudo pia-wg.sh up --full tunnel up, all *public* traffic exits via PIA
|
||||
# sudo pia-wg.sh down tear down, restoring DNS and routes
|
||||
# sudo pia-wg.sh status handshake, DNS and current public IP
|
||||
#
|
||||
# Requires the project's "VPN support" setting (Config -> Runtime) to be on.
|
||||
#
|
||||
# Settings are read from the environment, but note that sudo resets it: they
|
||||
# have to be passed *through* sudo, after the word `sudo`, not before it.
|
||||
#
|
||||
# sudo PIA_REGION=uk_london pia-wg.sh up --full # works
|
||||
# PIA_REGION=uk_london sudo pia-wg.sh up --full # silently ignored
|
||||
#
|
||||
# PIA_CREDS credentials file, two lines: username, then password
|
||||
# (default /home/claude/pia-creds; never echoed by this script)
|
||||
# PIA_REGION region id (default us_chicago). List them with:
|
||||
# curl -s https://serverlist.piaservers.net/vpninfo/servers/v6 \
|
||||
# | head -1 | jq -r '.regions[].id'
|
||||
set -euo pipefail
|
||||
|
||||
# Not ~/pia-creds: under sudo, HOME is /root.
|
||||
# Read by up()'s EXIT trap, which runs after the function's locals are gone.
|
||||
SETUP_OK=0
|
||||
|
||||
CREDS=${PIA_CREDS:-/home/claude/pia-creds}
|
||||
REGION=${PIA_REGION:-us_chicago}
|
||||
IFACE=pia0
|
||||
STATE=/run/pia-wg
|
||||
|
||||
# Kept off the tunnel in --full mode. The container's DNS resolver, the Docker
|
||||
# host network (host.docker.internal, any host-side Ollama), sibling containers
|
||||
# and the LAN all live in here. PIA cannot route any of it, so without these
|
||||
# exclusions the container reaches the public internet and nothing else --
|
||||
# including, fatally, its own resolver.
|
||||
PRIVATE_NETS="10.0.0.0/8 172.16.0.0/12 192.168.0.0/16 169.254.0.0/16"
|
||||
|
||||
# Args are joined with spaces so a long message can be written as several
|
||||
# source lines without the indentation ending up in the output.
|
||||
die() { echo "pia-wg: $*" >&2; exit 1; }
|
||||
|
||||
# `x=$(cmd)` is a plain assignment, so `set -e` kills the script on a non-zero
|
||||
# cmd *before* any `[ -z "$x" ] || die` line can run. Every capture below
|
||||
# therefore goes through `run`; without it a wrong password exits 22 with no
|
||||
# output at all, which is the most likely way this is used wrongly and was the
|
||||
# least explained.
|
||||
#
|
||||
# It takes a description rather than reporting the command it ran: one of these
|
||||
# invocations carries the account password in `-u`, and an error message is
|
||||
# exactly the wrong place for that to surface.
|
||||
run() { local what=$1; shift; "$@" || die "$what (exit $?)"; }
|
||||
|
||||
preflight() {
|
||||
[ "$(id -u)" = 0 ] || die "run with sudo"
|
||||
# CAP_NET_ADMIN is bit 12. Checking it by name gives a usable error; without
|
||||
# it the first `ip` call fails with a bare "Operation not permitted" that
|
||||
# points nowhere near the setting that actually needs changing.
|
||||
#
|
||||
# Deliberately NOT checking /dev/net/tun: kernel WireGuard is a netlink
|
||||
# interface and does not use it (verified -- `ip link add type wireguard`
|
||||
# succeeds with NET_ADMIN and no tun device). It is OpenVPN and userspace
|
||||
# wireguard-go that need it. The real kernel dependency here is the
|
||||
# `wireguard` module, which `ip link add` below reports on directly.
|
||||
local caps
|
||||
caps=$(awk '/^CapEff:/{print $2}' /proc/self/status)
|
||||
if [ $(( 0x$caps & 0x1000 )) -eq 0 ]; then
|
||||
die "this container has no CAP_NET_ADMIN." \
|
||||
"Turn on \"VPN support\" in Config -> Runtime and start the project" \
|
||||
"again. That recreates the container; the home and .claude volumes" \
|
||||
"are preserved, so nothing in them is lost."
|
||||
fi
|
||||
command -v wg >/dev/null || \
|
||||
die "wireguard-tools is not installed." \
|
||||
"If this project's container was built from an older base image," \
|
||||
"migrate it onto the current one -- that is what ships \`wg\`."
|
||||
[ -r "$CREDS" ] || \
|
||||
die "no credentials at $CREDS." \
|
||||
"Two lines are expected: username, then password." \
|
||||
"Set PIA_CREDS (after the word \`sudo\`) to read them elsewhere."
|
||||
}
|
||||
|
||||
# Routes that must work. A silent failure here is the worst state this script
|
||||
# can reach: the two half-routes need no gateway and would succeed, so the
|
||||
# tunnel captures everything while the exclusions that keep DNS and the Docker
|
||||
# host reachable are quietly missing -- and `status` still says "full tunnel".
|
||||
add_route() {
|
||||
ip route add "$@" || die "could not add route '$*'"
|
||||
printf '%s\n' "$*" >> "$STATE/routes"
|
||||
}
|
||||
|
||||
up() {
|
||||
case "${1:-}" in
|
||||
""|--full) ;;
|
||||
*) die "unknown option '$1' (expected --full or nothing)." \
|
||||
"Refusing rather than silently giving you a test route." ;;
|
||||
esac
|
||||
preflight
|
||||
|
||||
mkdir -p "$STATE"; cd "$STATE"
|
||||
|
||||
# `curl -o` creates the file before it knows the request failed, so a plain
|
||||
# `[ -f ]` cache check can pin a truncated cert forever -- and /run rides the
|
||||
# snapshot, so "forever" outlives the container. Fetch to a temp name and
|
||||
# rename only on success.
|
||||
if [ ! -s ca.rsa.4096.crt ]; then
|
||||
run "could not download PIA's CA certificate" \
|
||||
curl -sf -m 20 -o ca.crt.part \
|
||||
https://raw.githubusercontent.com/pia-foss/manual-connections/master/ca.rsa.4096.crt
|
||||
[ -s ca.crt.part ] || die "PIA's CA certificate downloaded empty"
|
||||
mv ca.crt.part ca.rsa.4096.crt
|
||||
fi
|
||||
|
||||
local u p tok srv sip scn priv pub resp ep gw dns
|
||||
u=$(sed -n 1p "$CREDS"); p=$(sed -n 2p "$CREDS")
|
||||
[ -n "$u" ] && [ -n "$p" ] || die "$CREDS needs two lines: username, then password"
|
||||
|
||||
tok=$(run "PIA rejected the credentials in $CREDS, or could not be reached" \
|
||||
curl -sf -m 25 -u "$u:$p" \
|
||||
https://www.privateinternetaccess.com/gtoken/generateToken | jq -r .token)
|
||||
[ -n "$tok" ] && [ "$tok" != null ] || die "PIA returned no token - check the credentials in $CREDS"
|
||||
|
||||
run "could not fetch PIA's server list" \
|
||||
curl -sf -m 30 https://serverlist.piaservers.net/vpninfo/servers/v6 \
|
||||
| head -1 > servers.json
|
||||
srv=$(jq -r --arg r "$REGION" '.regions[] | select(.id==$r) | .servers.wg[0]' servers.json)
|
||||
sip=$(echo "$srv" | jq -r .ip); scn=$(echo "$srv" | jq -r .cn)
|
||||
[ -n "$sip" ] && [ "$sip" != null ] || die "no WireGuard server for region '$REGION'"
|
||||
|
||||
# Only now tear down any previous tunnel. Doing it up front (as an earlier
|
||||
# version did) meant a failed token fetch or an unreachable server list took
|
||||
# a *working* tunnel down with it and silently reverted the container to its
|
||||
# real address, while the error talked about credentials. Everything above
|
||||
# this line can fail; nothing above it has touched the network stack.
|
||||
#
|
||||
# It also still does the job it was added for: clearing a stale resolv.conf
|
||||
# backup so a second `up` cannot save PIA's own resolvers over the real ones.
|
||||
down >/dev/null 2>&1 || true
|
||||
|
||||
# From here on the network stack is being modified, so any failure has to put
|
||||
# it back rather than exit half-configured. `down` is idempotent and restores
|
||||
# routes and resolv.conf exactly.
|
||||
#
|
||||
# EXIT rather than ERR, and a flag rather than the trap's own exit status: an
|
||||
# ERR trap is not inherited by shell functions without `set -E`, so a failure
|
||||
# inside add_route would not fire it, and `die` exits explicitly, which is not
|
||||
# an error and would not fire it either. EXIT catches both.
|
||||
SETUP_OK=0
|
||||
trap '[ "$SETUP_OK" = 1 ] || { echo "pia-wg: setup failed - rolling back" >&2; down >/dev/null 2>&1; }' EXIT
|
||||
|
||||
# umask, not a later chmod: the file is created under the inherited 0022
|
||||
# otherwise, so the key is world-readable for the moment in between.
|
||||
( umask 077; priv=$(wg genkey); printf '%s' "$priv" > wg.priv )
|
||||
priv=$(cat wg.priv); pub=$(printf '%s' "$priv" | wg pubkey)
|
||||
|
||||
# The token goes in on stdin as a curl config rather than in the argv, where
|
||||
# `ps` and /proc/*/cmdline expose it to every process in the container --
|
||||
# verified. It is a ~24h bearer credential for the whole PIA account.
|
||||
# PIA pins its certificate to the server's common name, which is why this
|
||||
# connects by CN and lets --connect-to point that name at the real address.
|
||||
resp=$(printf -- '--data-urlencode "pt=%s"\n--data-urlencode "pubkey=%s"\n' "$tok" "$pub" \
|
||||
| run "could not register the key with $scn" \
|
||||
curl -sf -m 25 -G -K - --connect-to "$scn::$sip:" \
|
||||
--cacert ca.rsa.4096.crt "https://$scn:1337/addKey")
|
||||
[ "$(echo "$resp" | jq -r .status)" = OK ] || die "key registration failed: $resp"
|
||||
|
||||
: > "$STATE/routes"
|
||||
ip link add "$IFACE" type wireguard 2>/dev/null || \
|
||||
die "could not create a WireGuard interface." \
|
||||
"The Docker host's kernel has no 'wireguard' module."
|
||||
wg set "$IFACE" private-key wg.priv \
|
||||
peer "$(echo "$resp" | jq -r .server_key)" \
|
||||
endpoint "$(echo "$resp" | jq -r .server_ip):$(echo "$resp" | jq -r .server_port)" \
|
||||
allowed-ips 0.0.0.0/0 persistent-keepalive 25
|
||||
# The kernel holds the key from here, so the file has no reason to outlive
|
||||
# this line -- and every reason not to: /run is in the writable layer, and a
|
||||
# recreate or migrate runs `docker commit` over it without tearing the tunnel
|
||||
# down first, baking the key into the project's snapshot image. `down` also
|
||||
# removes it, for the case where `up` never got this far.
|
||||
rm -f wg.priv
|
||||
ip addr add "$(echo "$resp" | jq -r .peer_ip)/32" dev "$IFACE"
|
||||
ip link set "$IFACE" up
|
||||
|
||||
if [ "${1:-}" = "--full" ]; then
|
||||
ep=$(echo "$resp" | jq -r .server_ip)
|
||||
gw=$(ip route show default | awk '{print $3; exit}')
|
||||
# `default dev eth0` with no `via` yields the literal "eth0" here, which
|
||||
# would make every exclusion below a malformed no-op.
|
||||
[[ $gw =~ ^[0-9]+\.[0-9]+\.[0-9]+\.[0-9]+$ ]] || \
|
||||
die "no usable default gateway to pin the tunnel against (got '${gw:-none}')"
|
||||
|
||||
# PIA's resolvers are required in --full. Without them the 10/8 exclusion
|
||||
# below is already in place, so every lookup would go to the container's
|
||||
# own resolver *outside* the tunnel -- a full tunnel leaking all its DNS,
|
||||
# reported by `status` as perfectly healthy.
|
||||
dns=$(echo "$resp" | jq -r '.dns_servers[]? // empty' | head -2)
|
||||
[ -n "$dns" ] || die "PIA returned no DNS servers; refusing a full tunnel that would leak every lookup"
|
||||
|
||||
# Pin the endpoint to the pre-existing gateway first, so the tunnel's own
|
||||
# packets do not try to route through the tunnel. Then beat the default
|
||||
# route with two half-routes rather than replacing it -- nothing to restore
|
||||
# on teardown, and the container keeps working if this script dies midway.
|
||||
add_route "$ep/32" via "$gw"
|
||||
add_route 0.0.0.0/1 dev "$IFACE"
|
||||
add_route 128.0.0.0/1 dev "$IFACE"
|
||||
|
||||
# Keep container, host and LAN traffic off the tunnel. Longer prefixes than
|
||||
# the two halves above, so these win.
|
||||
for n in $PRIVATE_NETS; do add_route "$n" via "$gw"; done
|
||||
|
||||
# PIA's resolvers live inside 10/8, so pin them back through the tunnel with
|
||||
# /32s -- longer still, so they beat the exclusion just added.
|
||||
cp /etc/resolv.conf "$STATE/resolv.conf.bak"
|
||||
for d in $dns; do add_route "$d/32" dev "$IFACE"; done
|
||||
# resolv.conf is a bind mount: write through it, never replace it.
|
||||
for d in $dns; do echo "nameserver $d"; done > /etc/resolv.conf
|
||||
echo "full tunnel: public traffic exits via PIA; private ranges stay local"
|
||||
else
|
||||
add_route 1.1.1.1/32 dev "$IFACE"
|
||||
echo "test route only: 1.1.1.1 goes via PIA, everything else unchanged"
|
||||
fi
|
||||
|
||||
# A tunnel with no handshake still routes -- into a black hole. Without this
|
||||
# `up --full` would exit 0 having pointed all traffic *and* resolv.conf at a
|
||||
# peer that never answered, and `status` would print "mode: full tunnel".
|
||||
local waited=0
|
||||
until [ "$(wg show "$IFACE" latest-handshakes | awk '{print $2; exit}')" != 0 ]; do
|
||||
waited=$((waited + 1))
|
||||
[ "$waited" -lt 20 ] || die "no handshake from $REGION after 10s - rolled back"
|
||||
sleep 0.5
|
||||
done
|
||||
|
||||
SETUP_OK=1
|
||||
trap - EXIT
|
||||
status
|
||||
}
|
||||
|
||||
down() {
|
||||
[ "$(id -u)" = 0 ] || die "run with sudo"
|
||||
# Only restore something that actually looks like a resolver file. Restoring
|
||||
# an empty or truncated backup leaves the container with no DNS at all, which
|
||||
# is worse than leaving the current one alone.
|
||||
if [ -f "$STATE/resolv.conf.bak" ]; then
|
||||
if grep -q '^nameserver' "$STATE/resolv.conf.bak" 2>/dev/null; then
|
||||
cat "$STATE/resolv.conf.bak" > /etc/resolv.conf
|
||||
else
|
||||
echo "pia-wg: warning - saved resolv.conf looks empty; leaving the current one alone" >&2
|
||||
fi
|
||||
rm -f "$STATE/resolv.conf.bak"
|
||||
fi
|
||||
if [ -f "$STATE/routes" ]; then
|
||||
# Reverse order: the specific overrides go before the ranges they sit in.
|
||||
tac "$STATE/routes" | while read -r r; do
|
||||
[ -n "$r" ] && ip route del $r 2>/dev/null || true
|
||||
done
|
||||
rm -f "$STATE/routes"
|
||||
fi
|
||||
ip link del "$IFACE" 2>/dev/null || true
|
||||
# /run is in the writable layer and `docker commit` bakes it into the
|
||||
# project's snapshot image, so a key left here rides that image into every
|
||||
# future container. Verified: a snapshot already carried one.
|
||||
rm -f "$STATE/wg.priv"
|
||||
echo "tunnel down"
|
||||
}
|
||||
|
||||
# Both are Cloudflare and both answer /cdn-cgi/trace over their bare address, so
|
||||
# neither needs DNS. Only 1.1.1.1 is ever routed into the tunnel, which is what
|
||||
# lets status tell the two exits apart.
|
||||
TRACE_TUNNELLED=https://1.1.1.1/cdn-cgi/trace
|
||||
TRACE_DIRECT=https://1.0.0.1/cdn-cgi/trace
|
||||
|
||||
exit_ip() { curl -s -m 20 "$1" | sed -n 's/^ip=//p'; }
|
||||
|
||||
status() {
|
||||
# `wg show` needs root; `ip route`/`ip link` do not. Without this guard an
|
||||
# unprivileged run prints "no tunnel up" and then "mode: full tunnel" in the
|
||||
# same breath, and an agent reading the first line re-runs `up`.
|
||||
[ "$(id -u)" = 0 ] || die "run with sudo"
|
||||
wg show "$IFACE" 2>/dev/null | grep -E "latest handshake|transfer" || echo "no tunnel up"
|
||||
|
||||
# Resolve a name, not an IP literal. A curl to 1.1.1.1 succeeds while DNS is
|
||||
# completely broken, which is exactly how a dead resolver goes unnoticed.
|
||||
printf 'DNS: '
|
||||
if timeout 10 getent hosts api.anthropic.com >/dev/null 2>&1; then
|
||||
echo "ok (via $(sed -n 's/^nameserver //p' /etc/resolv.conf | tr '\n' ' '))"
|
||||
else
|
||||
echo "BROKEN - cannot resolve api.anthropic.com"
|
||||
fi
|
||||
|
||||
# Report the exit per mode. In test mode the probe address is itself the one
|
||||
# thing inside the tunnel, so a single "public IP" line would print a PIA
|
||||
# address while every other packet leaves directly -- the exact reading that
|
||||
# makes a test tunnel look like a full one.
|
||||
if ip route show 0.0.0.0/1 2>/dev/null | grep -q "$IFACE"; then
|
||||
echo "mode: full tunnel"
|
||||
echo " all traffic exits: $(exit_ip "$TRACE_TUNNELLED")"
|
||||
elif ip link show "$IFACE" >/dev/null 2>&1; then
|
||||
echo "mode: test route only (1.1.1.1 through the tunnel, nothing else)"
|
||||
echo " through the tunnel: $(exit_ip "$TRACE_TUNNELLED")"
|
||||
echo " everything else: $(exit_ip "$TRACE_DIRECT") <- your real address"
|
||||
else
|
||||
echo "mode: no tunnel"
|
||||
echo " all traffic exits: $(exit_ip "$TRACE_DIRECT")"
|
||||
fi
|
||||
}
|
||||
|
||||
case "${1:-}" in
|
||||
up) shift; up "${1:-}" ;;
|
||||
down) down ;;
|
||||
status) status ;;
|
||||
*) sed -n '2,26p' "$0" | sed 's/^# \{0,1\}//'; exit 1 ;;
|
||||
esac
|
||||
Executable
+272
@@ -0,0 +1,272 @@
|
||||
#!/bin/bash
|
||||
# triple-c-playwright-heal — make Playwright usable in a Triple-C container.
|
||||
#
|
||||
# Idempotent: safe to run on every start and safe to re-run after a partial
|
||||
# failure. Each step checks for its own result first, so a healthy container is
|
||||
# a fast no-op that still prints why it is healthy. The last step is the only
|
||||
# one that means anything: it launches a browser for real.
|
||||
#
|
||||
# The things that go wrong, in the order they bite:
|
||||
#
|
||||
# 1. @playwright/cli missing — including the case where it was installed and
|
||||
# then silently removed again. `npm install --no-save <pkg>` in /workspace,
|
||||
# which has no package.json, prunes packages npm considers extraneous, so
|
||||
# installing @playwright/cli and then installing playwright wipes the
|
||||
# first one and leaves an empty node_modules/@playwright/. That directory
|
||||
# reads as "installed" to a naive check, which is why this script tests
|
||||
# the package *entry point*.
|
||||
#
|
||||
# 2. Bundled chromium missing or the wrong revision. Browsers live in the
|
||||
# home volume and outlive any single @playwright/cli install, so a stale
|
||||
# chromium-<old> is routinely present while the installed playwright-core
|
||||
# wants a newer one. Must be installed AS claude: run as root it lands in
|
||||
# /root/.cache/ms-playwright where the agent cannot see it.
|
||||
#
|
||||
# 3. No cli.config.json — the one that breaks an otherwise clean install.
|
||||
# With no config, playwright-cli resolves to channel `chrome` (system
|
||||
# Google Chrome) with the sandbox ON. These containers forbid unprivileged
|
||||
# user namespaces, so Chrome aborts with "Failed to move to new namespace
|
||||
# ... Operation not permitted". On newer base images Chrome is not present
|
||||
# at all and it fails with "Chromium distribution 'chrome' is not found".
|
||||
# Same root cause both ways: the default channel is wrong here.
|
||||
#
|
||||
# 4. The storage-state file the config points at is missing. Playwright
|
||||
# treats an unreadable storageState as a hard error on every launch, not
|
||||
# as "no saved state", so the file has to exist from the very first run.
|
||||
#
|
||||
# 5. xvfb or socat missing (older base images only). Headless Playwright
|
||||
# needs neither; the `playwright-cli show` dashboard needs xvfb, and the
|
||||
# browser-view pane needs socat — without it the pane reports
|
||||
# "127.0.0.1 sent an invalid response" while the container side is fine.
|
||||
#
|
||||
# Usage: triple-c-playwright-heal [--seed-config-only] [--force-config] [--quiet]
|
||||
# --seed-config-only only ensure the config and its storage-state file
|
||||
# exist. No npm install, no browser download, no apt, no
|
||||
# verify launch. Cheap and offline — this is the mode
|
||||
# entrypoint.sh runs on every container start.
|
||||
# --force-config overwrite an existing config instead of keeping it
|
||||
# --quiet print only problems and repairs, not healthy no-ops
|
||||
|
||||
set -u
|
||||
|
||||
TARGET_USER=claude
|
||||
TARGET_HOME=/home/claude
|
||||
PW_DIR=/workspace
|
||||
CONFIG_DIR="$TARGET_HOME/.playwright"
|
||||
CONFIG_FILE="$CONFIG_DIR/cli.config.json"
|
||||
STATE_FILE="$CONFIG_DIR/storage-state.json"
|
||||
CLI_ENTRY="$PW_DIR/node_modules/@playwright/cli/playwright-cli.js"
|
||||
|
||||
FORCE_CONFIG=0
|
||||
QUIET=0
|
||||
SEED_ONLY=0
|
||||
for arg in "$@"; do
|
||||
case "$arg" in
|
||||
--force-config) FORCE_CONFIG=1 ;;
|
||||
--quiet) QUIET=1 ;;
|
||||
--seed-config-only) SEED_ONLY=1 ;;
|
||||
*) echo "playwright-heal: unknown option: $arg" >&2; exit 2 ;;
|
||||
esac
|
||||
done
|
||||
|
||||
changed=0
|
||||
failed=0
|
||||
|
||||
say() { [ "$QUIET" = 1 ] || echo "playwright-heal: $*"; }
|
||||
warn() { echo "playwright-heal: $*" >&2; }
|
||||
did() { changed=1; echo "playwright-heal: $*"; }
|
||||
|
||||
# Run as claude whether we were invoked as root (docker exec / entrypoint) or
|
||||
# as claude (terminal session). Nothing user-visible may end up root-owned.
|
||||
as_claude() {
|
||||
if [ "$(id -u)" = 0 ]; then
|
||||
su "$TARGET_USER" -s /bin/sh -c "$1"
|
||||
else
|
||||
sh -c "$1"
|
||||
fi
|
||||
}
|
||||
|
||||
# ── 1. @playwright/cli ───────────────────────────────────────────────────────
|
||||
if [ "$SEED_ONLY" = 1 ]; then
|
||||
:
|
||||
elif [ -f "$CLI_ENTRY" ]; then
|
||||
say "@playwright/cli present"
|
||||
else
|
||||
# An empty leftover @playwright/ can make npm consider the tree settled.
|
||||
if [ -d "$PW_DIR/node_modules/@playwright" ]; then
|
||||
say "clearing partial @playwright install"
|
||||
rm -rf "$PW_DIR/node_modules/@playwright"
|
||||
fi
|
||||
say "installing @playwright/cli..."
|
||||
if as_claude "cd $PW_DIR && npm install --no-save --no-audit --no-fund @playwright/cli" >/tmp/pw-heal-npm.log 2>&1; then
|
||||
did "installed @playwright/cli"
|
||||
else
|
||||
warn "npm install failed; see /tmp/pw-heal-npm.log"
|
||||
failed=1
|
||||
fi
|
||||
fi
|
||||
|
||||
# ── 2. bundled chromium ──────────────────────────────────────────────────────
|
||||
# Ask Playwright where *this* version's chromium belongs rather than globbing
|
||||
# chromium-*, which would call a stale revision "present" and then fail at
|
||||
# launch with 'Browser "chrome-for-testing" is not installed'. --dry-run prints
|
||||
# the install location for the installed version and downloads nothing.
|
||||
if [ "$SEED_ONLY" = 1 ]; then
|
||||
:
|
||||
else
|
||||
chromium_dir=""
|
||||
if [ -f "$PW_DIR/node_modules/playwright-core/cli.js" ]; then
|
||||
chromium_dir=$(as_claude "cd $PW_DIR && node node_modules/playwright-core/cli.js install --dry-run chromium 2>/dev/null" \
|
||||
| awk '/Install location:/ { print $3; exit }')
|
||||
fi
|
||||
|
||||
if [ -n "$chromium_dir" ] && [ -d "$chromium_dir" ]; then
|
||||
say "chromium present ($(basename "$chromium_dir"))"
|
||||
elif [ -f "$PW_DIR/node_modules/playwright-core/cli.js" ]; then
|
||||
say "downloading chromium (~300 MB)..."
|
||||
if as_claude "cd $PW_DIR && node node_modules/playwright-core/cli.js install chromium" >/tmp/pw-heal-browser.log 2>&1; then
|
||||
did "installed chromium"
|
||||
else
|
||||
warn "chromium install failed; see /tmp/pw-heal-browser.log"
|
||||
failed=1
|
||||
fi
|
||||
else
|
||||
warn "playwright-core missing, cannot install chromium"
|
||||
failed=1
|
||||
fi
|
||||
fi
|
||||
|
||||
# ── 3. cli.config.json ───────────────────────────────────────────────────────
|
||||
# The *global* config, not a project-level .playwright/, because the project
|
||||
# one resolves relative to the current working directory and silently stops
|
||||
# applying the moment you cd elsewhere.
|
||||
#
|
||||
# `chrome-for-testing` is the only recognised chromium alias — "chromium" is
|
||||
# not one and falls back to system Chrome. chromiumSandbox:false is what
|
||||
# actually appends --no-sandbox.
|
||||
write_config() {
|
||||
mkdir -p "$CONFIG_DIR" || return 1
|
||||
cat > "$CONFIG_FILE" <<EOF
|
||||
{
|
||||
"browser": {
|
||||
"browserName": "chromium",
|
||||
"launchOptions": {
|
||||
"channel": "chrome-for-testing",
|
||||
"chromiumSandbox": false,
|
||||
"args": ["--no-sandbox", "--disable-dev-shm-usage"]
|
||||
},
|
||||
"contextOptions": {
|
||||
"storageState": "$STATE_FILE"
|
||||
}
|
||||
}
|
||||
}
|
||||
EOF
|
||||
chown -R "$TARGET_USER:$TARGET_USER" "$CONFIG_DIR" 2>/dev/null || true
|
||||
}
|
||||
|
||||
# storageState is a *load* path, and Playwright reads it at context creation.
|
||||
# A path that does not exist is not treated as "no saved state" — it is a hard
|
||||
# error, "Error reading storage state from …", on every single launch. So the
|
||||
# file has to exist before the config that names it can be used at all, and it
|
||||
# has to be recreated if anything deletes it. An empty state is valid and
|
||||
# behaves exactly like no state.
|
||||
#
|
||||
# The path is read back out of the config rather than assumed, so a
|
||||
# hand-edited config pointing somewhere else still gets its file created
|
||||
# instead of being silently broken by ours.
|
||||
ensure_state_file() {
|
||||
[ -f "$CONFIG_FILE" ] || return 0
|
||||
state_path=$(grep -o '"storageState"[[:space:]]*:[[:space:]]*"[^"]*"' "$CONFIG_FILE" 2>/dev/null \
|
||||
| sed 's/.*"\([^"]*\)"[[:space:]]*$/\1/')
|
||||
[ -n "$state_path" ] || return 0
|
||||
[ -f "$state_path" ] && return 0
|
||||
mkdir -p "$(dirname "$state_path")" 2>/dev/null
|
||||
printf '{\n "cookies": [],\n "origins": []\n}\n' > "$state_path" || return 1
|
||||
chown "$TARGET_USER:$TARGET_USER" "$state_path" 2>/dev/null || true
|
||||
did "created empty $state_path (storageState needs it to exist)"
|
||||
}
|
||||
|
||||
if [ ! -f "$CONFIG_FILE" ]; then
|
||||
if write_config; then did "wrote $CONFIG_FILE"; else warn "could not write $CONFIG_FILE"; failed=1; fi
|
||||
elif [ "$FORCE_CONFIG" = 1 ]; then
|
||||
if write_config; then did "overwrote $CONFIG_FILE (--force-config)"; else warn "could not write $CONFIG_FILE"; failed=1; fi
|
||||
elif grep -q '"chromiumSandbox"[[:space:]]*:[[:space:]]*false' "$CONFIG_FILE" 2>/dev/null; then
|
||||
say "config present and disables the sandbox"
|
||||
else
|
||||
# Present but hand-edited into a state that will not launch. Do not clobber
|
||||
# deliberate config silently; say what is wrong and how to replace it.
|
||||
warn "config at $CONFIG_FILE does not set chromiumSandbox:false — the browser will likely fail to launch. Re-run with --force-config to replace it."
|
||||
fi
|
||||
|
||||
# Unconditional: the config may name a storageState this run did not write —
|
||||
# one seeded by an older version of this script, or edited by hand — and a
|
||||
# missing file there breaks every launch.
|
||||
ensure_state_file || { warn "could not create the storage-state file"; failed=1; }
|
||||
|
||||
if [ "$SEED_ONLY" = 1 ]; then
|
||||
[ "$failed" = 1 ] && exit 1
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# ── 4. xvfb (headed dashboard only) ──────────────────────────────────────────
|
||||
# Current base images get this from `playwright install-deps` (its `tools`
|
||||
# group); older ones predate that layer. Headless never needs it, so a missing
|
||||
# xvfb is a note, not a failure.
|
||||
if command -v Xvfb >/dev/null 2>&1; then
|
||||
say "xvfb present"
|
||||
elif [ "$(id -u)" = 0 ]; then
|
||||
say "installing xvfb (needed only for the headed dashboard)..."
|
||||
if (apt-get update -qq && DEBIAN_FRONTEND=noninteractive apt-get install -y -qq xvfb) >/tmp/pw-heal-xvfb.log 2>&1; then
|
||||
did "installed xvfb"
|
||||
else
|
||||
warn "xvfb install failed (headless still works); see /tmp/pw-heal-xvfb.log"
|
||||
fi
|
||||
else
|
||||
say "xvfb missing and not running as root — skipping (headless still works)"
|
||||
fi
|
||||
|
||||
# ── 4b. socat (the browser-view pane's tunnel) ───────────────────────────────
|
||||
# Not Playwright's, but the same class of failure and it presents as a
|
||||
# Playwright problem: the pane's host-side proxy reaches the dashboard by
|
||||
# running `socat` *inside* the container over a Docker exec. On a container old
|
||||
# enough to predate socat in the base image, that exec produces something that
|
||||
# is not an HTTP response, and the webview reports "127.0.0.1 sent an invalid
|
||||
# response" — with the container side working perfectly. A project keeps the
|
||||
# base image it was first built from until it is migrated, so this is the
|
||||
# normal case on an older project, not an exotic one.
|
||||
if command -v socat >/dev/null 2>&1; then
|
||||
say "socat present"
|
||||
elif [ "$(id -u)" = 0 ]; then
|
||||
say "installing socat (needed by the browser-view pane)..."
|
||||
if (apt-get update -qq && DEBIAN_FRONTEND=noninteractive apt-get install -y -qq socat) >/tmp/pw-heal-socat.log 2>&1; then
|
||||
did "installed socat"
|
||||
else
|
||||
warn "socat install failed; the browser-view pane will report an invalid response. See /tmp/pw-heal-socat.log"
|
||||
failed=1
|
||||
fi
|
||||
else
|
||||
warn "socat missing and not running as root — the browser-view pane will report an invalid response"
|
||||
fi
|
||||
|
||||
# ── 5. verify by actually launching ──────────────────────────────────────────
|
||||
# Every step above can report success while the browser still refuses to
|
||||
# start — that is precisely how this broke. A dedicated session name keeps
|
||||
# this clear of whatever the agent already has open.
|
||||
if [ -f "$CLI_ENTRY" ]; then
|
||||
verify_out=$(as_claude "cd /tmp && timeout 90 node $CLI_ENTRY -s=heal-verify open 'data:text/html,<h1>ok</h1>' 2>&1")
|
||||
if printf '%s' "$verify_out" | grep -q 'opened with pid'; then
|
||||
say "verified: browser launches"
|
||||
as_claude "cd /tmp && timeout 30 node $CLI_ENTRY -s=heal-verify close" >/dev/null 2>&1
|
||||
else
|
||||
warn "browser still fails to launch:"
|
||||
printf '%s\n' "$verify_out" | grep -m4 -E 'namespace|Check failed|is not installed|is not found|missing dependencies|Error' >&2
|
||||
failed=1
|
||||
fi
|
||||
else
|
||||
warn "@playwright/cli not installed — nothing to verify"
|
||||
failed=1
|
||||
fi
|
||||
|
||||
[ "$failed" = 1 ] && exit 1
|
||||
[ "$changed" = 1 ] && say "done — repairs applied" || say "done — nothing to repair"
|
||||
exit 0
|
||||
@@ -8,17 +8,59 @@ SCHEDULER_DIR="${HOME}/.claude/scheduler"
|
||||
TASKS_DIR="${SCHEDULER_DIR}/tasks"
|
||||
LOGS_DIR="${SCHEDULER_DIR}/logs"
|
||||
NOTIFICATIONS_DIR="${SCHEDULER_DIR}/notifications"
|
||||
RUNNING_DIR="${SCHEDULER_DIR}/running"
|
||||
|
||||
# ── Helpers ──────────────────────────────────────────────────────────────────
|
||||
|
||||
ensure_dirs() {
|
||||
mkdir -p "$TASKS_DIR" "$LOGS_DIR" "$NOTIFICATIONS_DIR"
|
||||
mkdir -p "$TASKS_DIR" "$LOGS_DIR" "$NOTIFICATIONS_DIR" "$RUNNING_DIR"
|
||||
}
|
||||
|
||||
generate_id() {
|
||||
head -c 4 /dev/urandom | od -An -tx1 | tr -d ' \n'
|
||||
}
|
||||
|
||||
# Live run state for a task: prints "pid<TAB>started_epoch<TAB>log" and returns
|
||||
# 0 when the task is genuinely running, returns 1 otherwise.
|
||||
#
|
||||
# triple-c-task-runner writes the file and removes it from an EXIT trap, but a
|
||||
# trap cannot fire for SIGKILL or a container stop mid-run. So the pid is
|
||||
# checked rather than believed, and a state file whose process is gone is
|
||||
# cleared here — otherwise one hard stop leaves a task reading as "running"
|
||||
# forever, which is worse than no indicator at all.
|
||||
run_state() {
|
||||
local id="$1"
|
||||
local state_file="${RUNNING_DIR}/${id}.json"
|
||||
[ -f "$state_file" ] || return 1
|
||||
|
||||
local pid
|
||||
pid=$(jq -r '.pid // empty' "$state_file" 2>/dev/null)
|
||||
if [ -z "$pid" ] || ! kill -0 "$pid" 2>/dev/null; then
|
||||
rm -f "$state_file"
|
||||
return 1
|
||||
fi
|
||||
|
||||
printf '%s\t%s\t%s\n' \
|
||||
"$pid" \
|
||||
"$(jq -r '.started_epoch // 0' "$state_file")" \
|
||||
"$(jq -r '.log // ""' "$state_file")"
|
||||
}
|
||||
|
||||
# Compact elapsed time since an epoch, e.g. "8s", "4m12s", "1h07m".
|
||||
elapsed_since() {
|
||||
local start="$1" now delta
|
||||
now=$(date +%s)
|
||||
delta=$(( now - start ))
|
||||
[ "$delta" -lt 0 ] && delta=0
|
||||
if [ "$delta" -ge 3600 ]; then
|
||||
printf '%dh%02dm' $(( delta / 3600 )) $(( (delta % 3600) / 60 ))
|
||||
elif [ "$delta" -ge 60 ]; then
|
||||
printf '%dm%02ds' $(( delta / 60 )) $(( delta % 60 ))
|
||||
else
|
||||
printf '%ds' "$delta"
|
||||
fi
|
||||
}
|
||||
|
||||
# Reject a malformed cron expression at the point of entry.
|
||||
#
|
||||
# Without this an invalid schedule is written to a task file, and the next
|
||||
@@ -85,8 +127,9 @@ Commands:
|
||||
enable Enable a disabled task
|
||||
disable Disable a task
|
||||
list List all tasks
|
||||
status Show which tasks are running right now
|
||||
logs Show execution logs
|
||||
run Manually trigger a task now
|
||||
run Manually trigger a task now (streams its log)
|
||||
notifications Show or clear completion notifications
|
||||
|
||||
Add options:
|
||||
@@ -99,6 +142,10 @@ Add options:
|
||||
Remove/Enable/Disable/Run options:
|
||||
--id ID Task ID (required)
|
||||
|
||||
Status options:
|
||||
--id ID Show one task, including its last result when idle
|
||||
--watch, -w Refresh every 5s until the run finishes
|
||||
|
||||
Logs options:
|
||||
--id ID Show logs for a specific task (optional)
|
||||
--tail N Show last N lines (default: 50)
|
||||
@@ -313,8 +360,8 @@ cmd_disable() {
|
||||
|
||||
cmd_list() {
|
||||
local found=false
|
||||
printf "%-10s %-20s %-10s %-9s %-20s %s\n" "ID" "NAME" "TYPE" "ENABLED" "SCHEDULE" "PROMPT"
|
||||
printf "%-10s %-20s %-10s %-9s %-20s %s\n" "──────────" "────────────────────" "──────────" "─────────" "────────────────────" "──────────────────────────────"
|
||||
printf "%-10s %-20s %-10s %-9s %-20s %-12s %s\n" "ID" "NAME" "TYPE" "ENABLED" "SCHEDULE" "STATUS" "PROMPT"
|
||||
printf "%-10s %-20s %-10s %-9s %-20s %-12s %s\n" "──────────" "────────────────────" "──────────" "─────────" "────────────────────" "────────────" "──────────────────────────────"
|
||||
|
||||
for task_file in "$TASKS_DIR"/*.json; do
|
||||
[ -f "$task_file" ] || continue
|
||||
@@ -333,12 +380,21 @@ cmd_list() {
|
||||
display_schedule="at $at"
|
||||
fi
|
||||
|
||||
local status state started
|
||||
if state=$(run_state "$id"); then
|
||||
started=$(printf '%s' "$state" | cut -f2)
|
||||
status="running $(elapsed_since "$started")"
|
||||
else
|
||||
status="idle"
|
||||
fi
|
||||
|
||||
# Truncate long fields for display
|
||||
[ ${#name} -gt 20 ] && name="${name:0:17}..."
|
||||
[ ${#display_schedule} -gt 20 ] && display_schedule="${display_schedule:0:17}..."
|
||||
[ ${#prompt} -gt 30 ] && prompt="${prompt:0:27}..."
|
||||
|
||||
printf "%-10s %-20s %-10s %-9s %-20s %s\n" "$id" "$name" "$type" "$enabled" "$display_schedule" "$prompt"
|
||||
printf "%-10s %-20s %-10s %-9s %-20s %-12s %s\n" \
|
||||
"$id" "$name" "$type" "$enabled" "$display_schedule" "$status" "$prompt"
|
||||
done
|
||||
|
||||
if [ "$found" = "false" ]; then
|
||||
@@ -346,6 +402,78 @@ cmd_list() {
|
||||
fi
|
||||
}
|
||||
|
||||
# Is anything running, and how far along is it?
|
||||
#
|
||||
# This is the command for the question "did my `run` do anything, or has it
|
||||
# stalled?" — `logs` alone cannot answer it, because a log that stops growing
|
||||
# looks identical whether Claude is thinking or the run is dead.
|
||||
cmd_status() {
|
||||
local id="" watch=false
|
||||
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case "$1" in
|
||||
--id) id="$2"; shift 2 ;;
|
||||
--watch|-w) watch=true; shift ;;
|
||||
*) echo "Unknown option: $1" >&2; return 1 ;;
|
||||
esac
|
||||
done
|
||||
|
||||
while true; do
|
||||
local any=false
|
||||
for task_file in "$TASKS_DIR"/*.json; do
|
||||
[ -f "$task_file" ] || continue
|
||||
local tid
|
||||
tid=$(jq -r '.id' "$task_file")
|
||||
[ -z "$id" ] || [ "$tid" = "$id" ] || continue
|
||||
|
||||
local name state
|
||||
name=$(jq -r '.name' "$task_file")
|
||||
if state=$(run_state "$tid"); then
|
||||
any=true
|
||||
local pid started log
|
||||
pid=$(printf '%s' "$state" | cut -f1)
|
||||
started=$(printf '%s' "$state" | cut -f2)
|
||||
log=$(printf '%s' "$state" | cut -f3)
|
||||
echo "● RUNNING $name ($tid)"
|
||||
echo " elapsed: $(elapsed_since "$started") pid: $pid"
|
||||
echo " log: $log"
|
||||
# `claude -p` writes its answer in one go at the end, so a log
|
||||
# with only its header is the normal state of a healthy run —
|
||||
# print the tail only when there is something to show, rather
|
||||
# than an empty "last output:" that reads like a stall.
|
||||
# `|| true` throughout: under `set -e` a grep matching nothing
|
||||
# would otherwise abort the whole command.
|
||||
local tail_out=""
|
||||
if [ -f "$log" ]; then
|
||||
tail_out=$({ grep -v '^===' "$log" || true; } \
|
||||
| { grep -v '^$' || true; } | tail -n 3)
|
||||
fi
|
||||
if [ -n "$tail_out" ]; then
|
||||
echo " last output:"
|
||||
printf '%s\n' "$tail_out" | sed 's/^/ /'
|
||||
fi
|
||||
elif [ -n "$id" ]; then
|
||||
echo "○ idle $name ($tid)"
|
||||
local latest
|
||||
latest=$(ls -t "$LOGS_DIR/$tid"/*.log 2>/dev/null | head -1) || true
|
||||
if [ -n "$latest" ]; then
|
||||
echo " last run: $(basename "$latest" .log) $(grep -o 'Exit code: [0-9]*' "$latest" | tail -1)"
|
||||
fi
|
||||
fi
|
||||
done
|
||||
|
||||
if [ "$any" = "false" ] && [ -z "$id" ]; then
|
||||
echo "Nothing running."
|
||||
fi
|
||||
|
||||
[ "$watch" = "true" ] || break
|
||||
# Stop watching once the thing being watched has finished.
|
||||
[ "$any" = "true" ] || break
|
||||
sleep 5
|
||||
echo ""
|
||||
done
|
||||
}
|
||||
|
||||
cmd_logs() {
|
||||
local id="" tail_n=50
|
||||
|
||||
@@ -413,8 +541,53 @@ cmd_run() {
|
||||
|
||||
local name
|
||||
name=$(jq -r '.name' "$task_file")
|
||||
|
||||
if run_state "$id" >/dev/null; then
|
||||
echo "Task '$name' ($id) is already running — see: triple-c-scheduler status --id $id"
|
||||
return 0
|
||||
fi
|
||||
|
||||
echo "Manually triggering task '$name' ($id)..."
|
||||
/usr/local/bin/triple-c-task-runner "$id"
|
||||
|
||||
# Run in the background and stream its log. A task can easily think for
|
||||
# minutes, and the previous behaviour — block with no output until it is
|
||||
# over — is indistinguishable from a hang.
|
||||
/usr/local/bin/triple-c-task-runner "$id" &
|
||||
local runner_pid=$!
|
||||
|
||||
local state="" waited=0
|
||||
while [ "$waited" -lt 20 ]; do
|
||||
if state=$(run_state "$id"); then
|
||||
break
|
||||
fi
|
||||
kill -0 "$runner_pid" 2>/dev/null || break
|
||||
sleep 0.5
|
||||
waited=$(( waited + 1 ))
|
||||
done
|
||||
|
||||
local log=""
|
||||
[ -n "$state" ] && log=$(printf '%s' "$state" | cut -f3)
|
||||
|
||||
if [ -n "$log" ]; then
|
||||
echo " log: $log"
|
||||
echo " elsewhere: triple-c-scheduler status --id $id --watch"
|
||||
echo ""
|
||||
# --pid stops the follow when the runner exits, so this returns on its own.
|
||||
tail -n +1 -f --pid="$runner_pid" "$log" 2>/dev/null
|
||||
fi
|
||||
|
||||
local rc=0
|
||||
wait "$runner_pid" || rc=$?
|
||||
|
||||
# A run short enough that its state file was never observed still deserves
|
||||
# its output shown rather than swallowed.
|
||||
if [ -z "$log" ]; then
|
||||
local latest
|
||||
latest=$(ls -t "$LOGS_DIR/$id"/*.log 2>/dev/null | head -1) || true
|
||||
[ -n "$latest" ] && tail -n 20 "$latest"
|
||||
fi
|
||||
|
||||
return $rc
|
||||
}
|
||||
|
||||
cmd_notifications() {
|
||||
@@ -464,6 +637,7 @@ case "$command" in
|
||||
enable) cmd_enable "$@" ;;
|
||||
disable) cmd_disable "$@" ;;
|
||||
list) cmd_list ;;
|
||||
status) cmd_status "$@" ;;
|
||||
logs) cmd_logs "$@" ;;
|
||||
run) cmd_run "$@" ;;
|
||||
notifications) cmd_notifications "$@" ;;
|
||||
|
||||
@@ -9,6 +9,7 @@ SCHEDULER_DIR="${HOME}/.claude/scheduler"
|
||||
TASKS_DIR="${SCHEDULER_DIR}/tasks"
|
||||
LOGS_DIR="${SCHEDULER_DIR}/logs"
|
||||
NOTIFICATIONS_DIR="${SCHEDULER_DIR}/notifications"
|
||||
RUNNING_DIR="${SCHEDULER_DIR}/running"
|
||||
ENV_FILE="${SCHEDULER_DIR}/.env"
|
||||
|
||||
TASK_ID="${1:-}"
|
||||
@@ -34,11 +35,19 @@ if ! flock -n 200; then
|
||||
fi
|
||||
|
||||
# ── Source saved environment ─────────────────────────────────────────────────
|
||||
# The env file is a snapshot taken by the entrypoint, which runs as root. A
|
||||
# snapshot written before the entrypoint stopped capturing HOME still carries
|
||||
# HOME=/root, and `set -a` would apply it to `claude` below — which then finds no
|
||||
# credential under /root/.claude and exits with "Not logged in". The env file
|
||||
# lives on the home volume, so those stale copies outlive an image update until
|
||||
# the container is restarted; keep our own HOME regardless of what it says.
|
||||
if [ -f "$ENV_FILE" ]; then
|
||||
REAL_HOME="${HOME:-/home/claude}"
|
||||
set -a
|
||||
# shellcheck disable=SC1090
|
||||
source "$ENV_FILE"
|
||||
set +a
|
||||
HOME="$REAL_HOME"
|
||||
fi
|
||||
|
||||
# ── Read task definition ────────────────────────────────────────────────────
|
||||
@@ -69,6 +78,27 @@ mkdir -p "$TASK_LOG_DIR"
|
||||
TIMESTAMP=$(date +"%Y%m%d-%H%M%S")
|
||||
LOG_FILE="${TASK_LOG_DIR}/${TIMESTAMP}.log"
|
||||
|
||||
# ── Publish run state ───────────────────────────────────────────────────────
|
||||
# A scheduled run is detached — cron has no terminal, and the app fires it as a
|
||||
# detached exec — so without this there is no way to tell a task that is still
|
||||
# thinking from one that died, and a long run reads as a stall. `list`, `status`
|
||||
# and the app's Automation tab all read this file.
|
||||
#
|
||||
# flock above is what actually prevents overlapping runs; this is purely an
|
||||
# observability record, which is why readers verify the pid rather than trust
|
||||
# the file. The EXIT trap covers the crash paths (OOM, container stop, SIGTERM)
|
||||
# that would otherwise leave a task looking like it had been running for days.
|
||||
mkdir -p "$RUNNING_DIR"
|
||||
RUN_STATE="${RUNNING_DIR}/${TASK_ID}.json"
|
||||
trap 'rm -f "$RUN_STATE"' EXIT
|
||||
jq -n \
|
||||
--arg pid "$$" \
|
||||
--arg started "$(date +%s)" \
|
||||
--arg log "$LOG_FILE" \
|
||||
--arg name "$TASK_NAME" \
|
||||
'{pid: ($pid | tonumber), started_epoch: ($started | tonumber), log: $log, name: $name}' \
|
||||
> "$RUN_STATE"
|
||||
|
||||
# ── Execute Claude agent ────────────────────────────────────────────────────
|
||||
{
|
||||
echo "=== Task: $TASK_NAME ($TASK_ID) ==="
|
||||
|
||||
Reference in New Issue
Block a user