agentbbs/internal/pods/pods.go
Anthony Ettinger 94ce79374c feat(pods): SSH agent forwarding → git push from the pod with your key
Code in your pod and push to git.profullstack.com using the SAME SSH key you
signed in with — nothing is copied into the pod. When a member attaches with
agent forwarding (ssh -A), agentbbs listens on a fresh unix socket in a
per-user agent dir bind-mounted at /run/agentbbs-agent and proxies it back over
the session; the pod shell gets SSH_AUTH_SOCK pointed at it. The pod image's
ssh_config sends git@git.profullstack.com to Forgejo's SSH server (:2222), so
`git clone git@git.profullstack.com:you/repo.git` just works.

- pods.go: agentDir + startAgent (per-session socket, cleaned up on exit);
  Attach injects SSH_AUTH_SOCK when ssh.AgentRequested; ensure() bind-mounts the
  agent dir and self-heals idle pods missing it. No main.go change needed —
  charmbracelet/ssh sets AgentRequested from the session request loop.
- pods/Containerfile: /etc/ssh/ssh_config.d entry (port 2222, user git,
  accept-new) so the conventional git@ URL reaches Forgejo.
- setup.sh: keep using an already-built pod image if a later rebuild
  transient-fails, so a flaky deploy never downgrades pods to the base image.

Build/vet/test/gofmt clean. Image rebuilt on the host; `ssh -G
git.profullstack.com` resolves to port 2222 / user git. End-to-end push needs a
live `ssh -A` session (validate after deploy).

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-23 09:16:29 +00:00

377 lines
13 KiB
Go

// Package pods gives paid members a personal Linux container over SSH
// (`ssh pod@host`) — "run shit in a docker-like" without root on the host.
//
// Engine preference: rootless Podman (daemonless; container root maps to an
// unprivileged host uid via user namespaces), falling back to Docker.
//
// Capability profile differs by engine. Under rootless podman the host is
// protected by the userns mapping itself, so pods keep podman's default
// capability set — the pod's root behaves like real root (apt, chown, su,
// binding :80, ping). Under docker a breakout is host-root, so docker pods run
// non-root (uid 1000) with cap-drop ALL + no-new-privileges. Either way the SSH
// user never touches the host OS. cpu/mem/pids limits apply to both.
package pods
import (
"crypto/rand"
"encoding/hex"
"fmt"
"io"
"net"
"os"
"os/exec"
"path/filepath"
"regexp"
"strings"
"sync"
"github.com/charmbracelet/ssh"
"github.com/creack/pty"
)
// Manager provisions and attaches per-user pods.
type Manager struct {
engine string // "podman" or "docker"
image string
usersDir string // <data>/users on the host; when set, <user>/public_html is bind-mounted into the pod
mu sync.Mutex
attached map[string]int // container name -> live session count
}
// Detect picks the best available engine. Returns an error if neither
// podman nor docker is present. usersDir is the host <data>/users tree: when
// non-empty, each pod bind-mounts <usersDir>/<user>/public_html at
// /home/dev/public_html so a member's ~/public_html is exactly the directory
// Caddy serves at <name>.<host>. Pass "" to disable the bind.
func Detect(usersDir string) (*Manager, error) {
image := os.Getenv("AGENTBBS_POD_IMAGE")
if image == "" {
image = "debian:stable-slim"
}
for _, eng := range []string{"podman", "docker"} {
if _, err := exec.LookPath(eng); err == nil {
return &Manager{engine: eng, image: image, usersDir: usersDir, attached: map[string]int{}}, nil
}
}
return nil, fmt.Errorf("pods: neither podman nor docker found")
}
// publicHTMLMount returns the host path to the member's public_html and the
// "-v src:dst" volume spec that maps it into the pod, or "" for both when the
// bind is disabled. The host directory is created if absent so the mount source
// exists before the container starts.
func (m *Manager) publicHTMLMount(user string) (host, spec string) {
if m.usersDir == "" {
return "", ""
}
host = filepath.Join(m.usersDir, user, "public_html")
if err := os.MkdirAll(host, 0o755); err != nil {
// Non-fatal: fall back to the named-volume-only pod (no homepage bind).
return "", ""
}
return host, host + ":/home/dev/public_html"
}
// hasMount reports whether the named container already has a mount at the given
// destination path. Used to detect pods created before a mount was introduced so
// ensure can recreate them. A failed inspect reports false (treat as missing).
func (m *Manager) hasMount(name, dest string) bool {
out, err := exec.Command(m.engine, "container", "inspect",
"-f", "{{range .Mounts}}{{println .Destination}}{{end}}", name).Output()
if err != nil {
return false
}
for _, line := range strings.Split(string(out), "\n") {
if strings.TrimSpace(line) == dest {
return true
}
}
return false
}
// hasImage reports whether the named container is running the given image.
// Used to roll out a new pod image: a mismatch triggers an idle recreate so
// members pick up added tooling without losing their home volume. A blank or
// unresolvable image name is treated as a match (never heal on uncertainty).
func (m *Manager) hasImage(name, image string) bool {
if image == "" {
return true
}
out, err := exec.Command(m.engine, "container", "inspect", "-f", "{{.ImageName}}", name).Output()
if err != nil {
return true
}
got := strings.TrimSpace(string(out))
// Normalize: inspect may report "localhost/agentbbs-pod:latest" while m.image
// is the same; also tolerate the docker.io/library/ prefix podman adds.
norm := func(s string) string {
s = strings.TrimPrefix(s, "docker.io/library/")
s = strings.TrimPrefix(s, "docker.io/")
s = strings.TrimPrefix(s, "localhost/")
return s
}
return norm(got) == norm(image)
}
// agentDir is the host directory bind-mounted into a pod at /run/agentbbs-agent,
// where Attach drops a forwarded SSH-agent socket. Derived as <data>/agent/<user>
// (a sibling of the users dir). Empty — disabling agent forwarding — when the
// users dir isn't configured or the directory can't be created.
func (m *Manager) agentDir(user string) string {
if m.usersDir == "" {
return ""
}
d := filepath.Join(filepath.Dir(m.usersDir), "agent", unsafeName.ReplaceAllString(strings.ToLower(user), "-"))
if err := os.MkdirAll(d, 0o700); err != nil {
return ""
}
return d
}
// startAgent forwards the connecting client's SSH agent into the member's pod:
// it listens on a fresh unix socket in the bind-mounted agent dir and proxies
// connections back over the SSH session. Returns the in-pod SSH_AUTH_SOCK path
// and a cleanup func, or "" when forwarding can't be set up (no agent dir / no
// socket). With this, `git push git@git.profullstack.com` inside the pod uses
// the member's own key — nothing is copied into the pod.
func (m *Manager) startAgent(s ssh.Session, user string) (sock string, cleanup func()) {
dir := m.agentDir(user)
if dir == "" {
return "", func() {}
}
var b [8]byte
_, _ = rand.Read(b[:])
fname := "agent-" + hex.EncodeToString(b[:]) + ".sock"
hostSock := filepath.Join(dir, fname)
_ = os.Remove(hostSock)
l, err := net.Listen("unix", hostSock)
if err != nil {
return "", func() {}
}
go ssh.ForwardAgentConnections(l, s)
return "/run/agentbbs-agent/" + fname, func() { _ = l.Close(); _ = os.Remove(hostSock) }
}
// Engine reports the active container engine.
func (m *Manager) Engine() string { return m.engine }
var unsafeName = regexp.MustCompile(`[^a-zA-Z0-9_.-]`)
func (m *Manager) containerName(user string) string {
return "agentbbs-pod-" + unsafeName.ReplaceAllString(strings.ToLower(user), "-")
}
// ensure creates (or starts) the user's container and returns its name.
func (m *Manager) ensure(user string) (string, error) {
name := m.containerName(user)
// Bind the host's public_html into the pod so a member's edits at
// ~/public_html are exactly what Caddy serves at <name>.<host>.
_, pubSpec := m.publicHTMLMount(user)
// Bind a per-user agent dir into the pod; Attach drops a forwarded SSH-agent
// socket here so `git push` uses the member's own key (see startAgent).
agentDir := m.agentDir(user)
if m.engine == "docker" {
// Under docker the pod runs as uid 1000 (never container root), so the
// named home volume — and the bind-mounted public_html — must be owned
// by 1000. A trusted one-shot init container enforces that on every
// ensure — mounts can predate the container or survive recreation.
// Rootless podman doesn't need this: container root maps to the
// unprivileged host user that already owns these trees.
initArgs := []string{"run", "--rm", "-v", name + "-home:/home/dev"}
chown := "chown 1000:1000 /home/dev"
if pubSpec != "" {
initArgs = append(initArgs, "-v", pubSpec)
chown += "; chown 1000:1000 /home/dev/public_html"
}
initArgs = append(initArgs, m.image, "sh", "-c", chown)
init := exec.Command(m.engine, initArgs...)
if out, err := init.CombinedOutput(); err != nil {
return "", fmt.Errorf("pods: volume init failed: %v: %s", err, strings.TrimSpace(string(out)))
}
}
// Already exists?
if err := exec.Command(m.engine, "container", "inspect", name).Run(); err == nil {
// Self-heal pods created before the homepage bind existed: recreate so
// ~/public_html maps to the served dir. The named home volume persists
// across rm, so the member's files are kept. Only heal when the pod is
// idle (no live session) — never pull a running pod out from under an
// active session; a still-unbound pod heals on its next idle attach.
m.mu.Lock()
idle := m.attached[name] == 0
m.mu.Unlock()
// Recreate an idle pod when it's missing the public_html bind OR is
// running an out-of-date image (e.g. a new pod image with added tooling).
// The home volume persists across rm, so member data is kept; a busy pod
// heals on its next idle attach instead.
needsHeal := idle && ((pubSpec != "" && !m.hasMount(name, "/home/dev/public_html")) ||
(agentDir != "" && !m.hasMount(name, "/run/agentbbs-agent")) ||
!m.hasImage(name, m.image))
if needsHeal {
_ = exec.Command(m.engine, "rm", "-f", name).Run() // fall through to recreate
} else {
_ = exec.Command(m.engine, "start", name).Run() // no-op if running
return name, nil
}
}
args := []string{
"run", "-d",
"--name", name,
"--hostname", "pod-" + unsafeName.ReplaceAllString(user, "-"),
"--memory", env("AGENTBBS_POD_MEM", "512m"),
"--cpus", env("AGENTBBS_POD_CPUS", "1"),
"--pids-limit", "256",
"--restart", "unless-stopped",
"-v", name + "-home:/home/dev",
"-w", "/home/dev",
"-e", "HOME=/home/dev",
}
if pubSpec != "" {
args = append(args, "-v", pubSpec)
}
if agentDir != "" {
args = append(args, "-v", agentDir+":/run/agentbbs-agent")
}
if m.engine == "docker" {
// Rootful docker: a breakout is host-root, so refuse to hand out
// container root — run as uid 1000 with no caps and no privilege
// escalation.
args = append(args,
"--user", "1000:1000",
"--cap-drop", "ALL",
"--security-opt", "no-new-privileges",
)
} else {
// Rootless podman: container root is already an unprivileged host uid
// via user namespaces, so keep podman's default capability set and let
// the pod's root act like real root (apt, chown, su, binding :80,
// ping). Power users can opt into extra caps (e.g. SYS_PTRACE for
// strace, NET_ADMIN for iptables) via AGENTBBS_POD_EXTRA_CAPS.
for _, c := range strings.Fields(strings.ReplaceAll(env("AGENTBBS_POD_EXTRA_CAPS", ""), ",", " ")) {
args = append(args, "--cap-add", c)
}
}
args = append(args, m.image, "sleep", "infinity")
out, err := exec.Command(m.engine, args...).CombinedOutput()
if err != nil {
return "", fmt.Errorf("pods: create failed: %v: %s", err, strings.TrimSpace(string(out)))
}
return name, nil
}
// Attach provisions the pod and wires the SSH session to a shell inside it.
// Blocks until the shell exits or the session closes.
func (m *Manager) Attach(s ssh.Session, user string) error {
ptyReq, winCh, hasPty := s.Pty()
if !hasPty {
return fmt.Errorf("pods: a PTY is required (ssh -t)")
}
name, err := m.ensure(user)
if err != nil {
return err
}
// Forward the client's SSH agent (ssh -A) into the pod so git push uses the
// member's own key. No-op unless the client requested forwarding.
execEnv := []string{"-e", "TERM=" + ptyReq.Term}
if ssh.AgentRequested(s) {
if sock, cleanup := m.startAgent(s, user); sock != "" {
defer cleanup()
execEnv = append(execEnv, "-e", "SSH_AUTH_SOCK="+sock)
}
}
shell := env("AGENTBBS_POD_SHELL", "/bin/bash")
cmd := exec.Command(m.engine, append(append([]string{"exec", "-it"}, execEnv...), name, shell, "-l")...)
f, err := pty.Start(cmd)
if err != nil {
// busybox-ish images may lack bash
cmd = exec.Command(m.engine, append(append([]string{"exec", "-it"}, execEnv...), name, "/bin/sh", "-l")...)
f, err = pty.Start(cmd)
if err != nil {
return fmt.Errorf("pods: attach failed: %w", err)
}
}
defer f.Close()
m.ref(name, +1)
defer m.deref(name)
_ = pty.Setsize(f, &pty.Winsize{Rows: uint16(ptyReq.Window.Height), Cols: uint16(ptyReq.Window.Width)})
go func() {
for w := range winCh {
_ = pty.Setsize(f, &pty.Winsize{Rows: uint16(w.Height), Cols: uint16(w.Width)})
}
}()
go func() { _, _ = io.Copy(f, s) }() // ssh -> pod
_, _ = io.Copy(s, f) // pod -> ssh
_ = cmd.Wait()
return nil
}
// Exec provisions the user's pod and runs argv inside it wired to the SSH
// session (PTY required). Used for tor@/tor-irc@ so arbitrary or interactive
// commands run sandboxed in the member's container, never on the host. Blocks
// until the command exits or the session closes.
func (m *Manager) Exec(s ssh.Session, user string, argv []string) error {
if len(argv) == 0 {
return fmt.Errorf("pods: no command given")
}
ptyReq, winCh, hasPty := s.Pty()
if !hasPty {
return fmt.Errorf("pods: a PTY is required (ssh -t)")
}
name, err := m.ensure(user)
if err != nil {
return err
}
args := append([]string{"exec", "-it", "-e", "TERM=" + ptyReq.Term, name}, argv...)
cmd := exec.Command(m.engine, args...)
f, err := pty.Start(cmd)
if err != nil {
return fmt.Errorf("pods: exec failed: %w", err)
}
defer f.Close()
m.ref(name, +1)
defer m.deref(name)
_ = pty.Setsize(f, &pty.Winsize{Rows: uint16(ptyReq.Window.Height), Cols: uint16(ptyReq.Window.Width)})
go func() {
for w := range winCh {
_ = pty.Setsize(f, &pty.Winsize{Rows: uint16(w.Height), Cols: uint16(w.Width)})
}
}()
go func() { _, _ = io.Copy(f, s) }() // ssh -> pod
_, _ = io.Copy(s, f) // pod -> ssh
_ = cmd.Wait()
return nil
}
func (m *Manager) ref(name string, d int) {
m.mu.Lock()
defer m.mu.Unlock()
m.attached[name] += d
}
// deref stops the container shortly after the last session detaches, unless
// AGENTBBS_POD_KEEP=1 keeps pods running between visits.
func (m *Manager) deref(name string) {
m.mu.Lock()
m.attached[name]--
last := m.attached[name] <= 0
m.mu.Unlock()
if last && os.Getenv("AGENTBBS_POD_KEEP") != "1" {
go func() { _ = exec.Command(m.engine, "stop", "-t", "2", name).Run() }()
}
}
func env(k, def string) string {
if v := os.Getenv(k); v != "" {
return v
}
return def
}