# cc-ci.nix — everything on this host that exists FOR cc-ci, and nothing else. # # Split out of the orchestrator host config on 2026-08-20. The host it runs on is a general # agent/orchestration box that also serves several unrelated projects; this module is the cc-ci # part of it, so that the two can evolve (and be reviewed) independently. It is exported from this # repo's flake as `nixosModules.cc-ci` and imported by whichever host runs cc-ci. # # All of it assumes the cc-ci workspaces exist on the host: # /srv/cc-ci the loops workspace (+ .cc-ci-logs, upgrader.env) # /srv/cc-ci-orch this repo (the orchestrator's own working dir) # and that a `loops` user, tmux, python3 and the standalone claude/opencode CLIs are present — # those are host concerns, provided by the host config, not by this module. { config, pkgs, lib, ... }: { # cc-ci-loops supervisor — workspace staged 2026-05-31, so ENABLED for reboot-resilience. systemd.services.cc-ci-loops = { description = "cc-ci Builder/Adversary loops + watchdog (launch.sh start)"; wantedBy = [ "multi-user.target" ]; # enabled after workspace staged (Hetzner cutover) after = [ "network-online.target" "tailscaled.service" "claude-install.service" ]; wants = [ "network-online.target" ]; serviceConfig = { # KillMode=process: this unit only LAUNCHES the tmux server, it does not own it. With the # default (control-group) systemd kills every leftover process in the cgroup when the unit # stops — and since one tmux server hosts every agent session on this host, a rebuild that # merely touched this unit wiped all of them (operator 2026-08-01). Only the (already # exited) main process is killed now; `systemctl stop` therefore does NOT tear down agents. KillMode = "process"; Type = "oneshot"; RemainAfterExit = true; User = "loops"; Group = "users"; WorkingDirectory = "/srv/cc-ci/cc-ci"; # Append one line to REBOOTS.md per genuine reboot (boot_id-gated; not on manual restart). ExecStartPre = "${pkgs.bash}/bin/bash /srv/cc-ci/cc-ci-plan/reboot-log.sh"; }; # CLAUDE_BIN points at the standalone CLI installed by claude-install.service; the loops # backend defaults to claude (persisted in .loop-backend). Without this, launch.py's preflight # `which(claude)` fails because the systemd `path` below has no /home/loops/.local/bin. environment = { RESUME_PHASE = "1"; HOME = "/home/loops"; CLAUDE_BIN = "/home/loops/.local/bin/claude"; }; path = [ pkgs.bash pkgs.tmux pkgs.git pkgs.python3 pkgs.openssh pkgs.nettools ]; script = '' # Put the standalone claude/opencode binaries on PATH. On a cold boot this is the env the # tmux server (and thus every agent session) inherits, so bare `claude` resolves everywhere. export PATH="/home/loops/.local/bin:$PATH" [ -x /srv/cc-ci/cc-ci-plan/launch.sh ] && /srv/cc-ci/cc-ci-plan/launch.sh start || \ echo "workspace not staged yet — skipping loop start" ''; }; # cc-ci-orchestrator supervisor — the operator's steering session. Same shape as # lichen-orchestrator / project-orchestrator above: this unit only LAUNCHES the orchestrator's # tmux session via the agent-orchestrator harness (cc-ci-plan/agents.py); it does not own the # session or the tmux server. The orchestrator agent is declared in cc-ci-plan/agents.toml on # the OPencode backend (backend = "opencode", model = "opencode/glm-5.2"), so on boot it # attaches to the shared opencode web server (opencode-web.service below) and is reachable for # Remote Control at https://oc.commoninternet.net under the /srv/cc-ci-orch project. The harness # watchdog (started by `agents.py up`) keeps it alive: heal-only (no stall reboots — a persistent # supervisor must not be killed just for idling). Added 2026-08-03 to give the cc-ci orchestrator # the same reboot-resilience the other two orchestrators already have. systemd.services.cc-ci-orchestrator = { description = "cc-ci orchestrator (operator steering session) — agents.py up orchestrator, opencode backend"; wantedBy = [ "multi-user.target" ]; after = [ "network-online.target" "tailscaled.service" "opencode-web.service" ]; wants = [ "network-online.target" ]; serviceConfig = { # KillMode=process: see the note on cc-ci-loops — a rebuild that merely touches this unit # must not tear down the (shared) tmux server and every agent session with it. KillMode = "process"; Type = "oneshot"; RemainAfterExit = true; User = "loops"; Group = "users"; WorkingDirectory = "/srv/cc-ci-orch"; }; environment = { HOME = "/home/loops"; }; path = [ pkgs.bash pkgs.tmux pkgs.git pkgs.python3 pkgs.openssh pkgs.nettools ]; script = '' export PATH="/home/loops/.local/bin:$PATH" proj="/srv/cc-ci-orch" echo "$(cat /proc/sys/kernel/random/boot_id) boot $(date -u +%FT%TZ) — cc-ci-orchestrator up" \ >> "$proj/cc-ci-plan/.ao-boot.log" 2>/dev/null || true cd "$proj" && python3 cc-ci-plan/agents.py up orchestrator || echo "cc-ci orchestrator agents.py up failed" ''; }; # Weekly recipe upgrade — runs /upgrade-all over every enrolled recipe (opens recipe PRs # verified by !testme, never merges). Replaces the boot-fragile busybox-crond-in-tmux from # phase 5 §4 with a reboot-safe systemd timer. The service is timer-triggered only (NOT # wantedBy multi-user.target) so it never runs on boot/activation — only on the schedule. systemd.services.cc-ci-upgrade-all = { description = "cc-ci weekly /upgrade-all run (recipe upgrade survey + PRs, never merges)"; after = [ "network-online.target" "tailscaled.service" "claude-install.service" ]; wants = [ "network-online.target" ]; serviceConfig = { Type = "oneshot"; # launch-upgrader.py spawns the cc-ci-upgrader tmux session and returns User = "loops"; Group = "users"; WorkingDirectory = "/srv/cc-ci"; # Optional per-run overrides for backend/model (LOOP_BACKEND, LOOP_MODEL, OPENCODE_SHARE, # UPGRADER_ARGS, …). The leading "-" makes it optional: absent file → claude/sonnet defaults. # Current config (as of 2026-08-16): the upgrader + report run on tinfoil/deepseek-v4-pro # (LOOP_MODEL + REPORT_MODEL in the env file); the hourly SUPERVISOR stays on glm-5.2 # (SUPERVISOR_MODEL defaults to opencode-go/glm-5.2 in launch-supervisor.py, NOT overridden # here). Subagents bind deepseek via the cc-ci repo's opencode config. LOOP_TIER=zen is kept # so the tier check passes; the watchdog's usage-limit probe sends the deepseek model name to # the zen endpoint, which returns 200 (not 429) → resume immediately (correct: tinfoil has no # rolling usage limit to wait out). No rebuild needed to switch — the env file is read at each # timer fire. Holds no secrets (the tinfoil API key lives in the opencode config / auth.json). EnvironmentFile = "-/srv/cc-ci/upgrader.env"; }; environment = { HOME = "/home/loops"; CLAUDE_BIN = "/home/loops/.local/bin/claude"; }; path = [ pkgs.bash pkgs.tmux pkgs.git pkgs.python3 pkgs.openssh pkgs.nettools ]; script = '' export PATH="/home/loops/.local/bin:$PATH" python3 /srv/cc-ci/cc-ci-plan/launch-upgrader.py start >> /srv/cc-ci/.cc-ci-logs/upgrader-cron.log 2>&1 ''; }; systemd.timers.cc-ci-upgrade-all = { description = "Weekly trigger for cc-ci-upgrade-all (Thursdays 22:00 America/New_York — Boston 10pm)"; wantedBy = [ "timers.target" ]; timerConfig = { # 10pm Thursday Boston time — DST-aware (EDT→02:00 UTC, EST→03:00 UTC) via the tz in OnCalendar. OnCalendar = "Thu *-*-* 22:00:00 America/New_York"; Persistent = true; # if the box was down at the scheduled time, run once on next boot }; }; # Hourly SUPERVISOR — a glm-5.2 orchestrator wake-up that keeps the weekly run on track. The # log-idle/429 watchdog only handles opencode-go usage-limit stalls; it does NOT cover a host # disk-full crash (which killed the 2026-07-03 run) or any other environmental wedge. This is a # CHEAP deterministic gate: if the weekly run is complete or actively progressing it does NOTHING # (zero model tokens). Only when a run has stalled/died before completing does it launch a # short-lived glm-5.2 agent that diagnoses the blockage and drives the run to a clean DONE. systemd.services.cc-ci-upgrade-supervisor = { description = "cc-ci hourly weekly-run supervisor (glm-5.2 — drives a stalled /upgrade-all to completion)"; after = [ "network-online.target" "tailscaled.service" ]; wants = [ "network-online.target" ]; serviceConfig = { Type = "oneshot"; # launch-supervisor.py check: gate now, spawn the agent into tmux, return User = "loops"; Group = "users"; WorkingDirectory = "/srv/cc-ci"; # Shares the weekly run's optional override file (e.g. SUPERVISOR_MODEL=…); "-" = optional. EnvironmentFile = "-/srv/cc-ci/upgrader.env"; }; environment = { HOME = "/home/loops"; }; path = [ pkgs.bash pkgs.tmux pkgs.git pkgs.python3 pkgs.openssh pkgs.nettools ]; script = '' export PATH="/home/loops/.local/bin:$PATH" python3 /srv/cc-ci/cc-ci-plan/launch-supervisor.py check >> /srv/cc-ci/.cc-ci-logs/supervisor-cron.log 2>&1 ''; }; systemd.timers.cc-ci-upgrade-supervisor = { description = "Hourly trigger for cc-ci-upgrade-supervisor (weekly-run health check + drive)"; wantedBy = [ "timers.target" ]; timerConfig = { OnCalendar = "*-*-* *:07:00"; # every hour at :07 (offset from the weekly :00 fire) Persistent = false; # a missed hourly check is moot — the next hour re-checks }; }; }