The orchestrator's flake now builds the machine it shares with the cc-ci CI
server: `nixosConfigurations.cc-ci` composes cc-ci's nixosModules.cc-ci-server
(new flake input, nixpkgs + sops-nix follow ours), this repo's orchestrator
module (nix/modules/cc-ci.nix, exported as cc-ci-orchestrator, `cc-ci` kept
as an alias for notplants-nix) and the new nix/modules/orchestrator-host.nix
— the host contract those units always assumed (loops user, claude/opencode
CLIs, opencode web server + tailnet-only UI on 8443 since traefik owns
80/443, nix-ld, tool set, `ssh cc-ci` → loopback).
nix/hosts/cc-ci/{hardware,networking}.nix are PROVISIONAL copies of the old
server's layout so the flake evaluates; they get replaced by the
nixos-infect output of 195.201.88.249.
README.md is the deploy guide (Hetzner Debian → nixos-infect → this flake →
staging → data restore → cutover). archive/ holds the retired Incus/Hetzner
orchestrator host configs, the old terraform and the migration plans;
references updated. cc-ci-plan/plan-cc-ci-combined-host.md is the working
plan for the move.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01FqkQq3CDmFWcQ7u1LzoyRz
158 lines
9.7 KiB
Nix
158 lines
9.7 KiB
Nix
# cc-ci.nix — the cc-ci ORCHESTRATOR: the Builder/Adversary loops supervisor, the operator's
|
|
# steering session, and the weekly-upgrade + hourly-supervisor timers. Nothing else.
|
|
#
|
|
# Exported from this repo's flake as `nixosModules.cc-ci-orchestrator` (and, for the host that
|
|
# used to import it under the old name, `nixosModules.cc-ci`). Split out of the shared agent
|
|
# host config on 2026-08-20; since 2026-09 it runs on the same Hetzner host as the CI server
|
|
# itself (`#cc-ci` in flake.nix), next to recipe-maintainers/cc-ci's `nixosModules.cc-ci-server`.
|
|
#
|
|
# All of it assumes the cc-ci workspaces exist on the host:
|
|
# /srv/cc-ci the loops workspace (+ .cc-ci-logs, upgrader.env) — a symlink to
|
|
# /srv/cc-ci-orch this repo (the orchestrator's own working dir), with cc-ci/ checked out
|
|
# and that a `loops` user, tmux, python3 and the standalone claude/opencode CLIs are present —
|
|
# those are host concerns, provided by nix/modules/orchestrator-host.nix, not by this module.
|
|
{ config, pkgs, lib, ... }:
|
|
{
|
|
# cc-ci-loops supervisor — workspace staged 2026-05-31, so ENABLED for reboot-resilience.
|
|
systemd.services.cc-ci-loops = {
|
|
description = "cc-ci Builder/Adversary loops + watchdog (launch.sh start)";
|
|
wantedBy = [ "multi-user.target" ]; # enabled after workspace staged (Hetzner cutover)
|
|
after = [ "network-online.target" "tailscaled.service" "claude-install.service" ];
|
|
wants = [ "network-online.target" ];
|
|
serviceConfig = {
|
|
# KillMode=process: this unit only LAUNCHES the tmux server, it does not own it. With the
|
|
# default (control-group) systemd kills every leftover process in the cgroup when the unit
|
|
# stops — and since one tmux server hosts every agent session on this host, a rebuild that
|
|
# merely touched this unit wiped all of them (operator 2026-08-01). Only the (already
|
|
# exited) main process is killed now; `systemctl stop` therefore does NOT tear down agents.
|
|
KillMode = "process";
|
|
Type = "oneshot"; RemainAfterExit = true;
|
|
User = "loops"; Group = "users";
|
|
WorkingDirectory = "/srv/cc-ci/cc-ci";
|
|
# Append one line to REBOOTS.md per genuine reboot (boot_id-gated; not on manual restart).
|
|
ExecStartPre = "${pkgs.bash}/bin/bash /srv/cc-ci/cc-ci-plan/reboot-log.sh";
|
|
};
|
|
# CLAUDE_BIN points at the standalone CLI installed by claude-install.service; the loops
|
|
# backend defaults to claude (persisted in .loop-backend). Without this, launch.py's preflight
|
|
# `which(claude)` fails because the systemd `path` below has no /home/loops/.local/bin.
|
|
environment = { RESUME_PHASE = "1"; HOME = "/home/loops"; CLAUDE_BIN = "/home/loops/.local/bin/claude"; };
|
|
path = [ pkgs.bash pkgs.tmux pkgs.git pkgs.python3 pkgs.openssh pkgs.nettools ];
|
|
script = ''
|
|
# Put the standalone claude/opencode binaries on PATH. On a cold boot this is the env the
|
|
# tmux server (and thus every agent session) inherits, so bare `claude` resolves everywhere.
|
|
export PATH="/home/loops/.local/bin:$PATH"
|
|
[ -x /srv/cc-ci/cc-ci-plan/launch.sh ] && /srv/cc-ci/cc-ci-plan/launch.sh start || \
|
|
echo "workspace not staged yet — skipping loop start"
|
|
'';
|
|
};
|
|
|
|
# cc-ci-orchestrator supervisor — the operator's steering session. Same shape as
|
|
# lichen-orchestrator / project-orchestrator above: this unit only LAUNCHES the orchestrator's
|
|
# tmux session via the agent-orchestrator harness (cc-ci-plan/agents.py); it does not own the
|
|
# session or the tmux server. The orchestrator agent is declared in cc-ci-plan/agents.toml
|
|
# (backend/model chosen there — Claude Code under Remote Control since 2026-09-07; before that
|
|
# opencode/glm-5.2 attached to the shared opencode web server, opencode-web.service in
|
|
# orchestrator-host.nix, which the upgrader still uses). The harness watchdog (started by
|
|
# `agents.py up`) keeps it alive: heal-only (no stall reboots — a persistent supervisor must not
|
|
# be killed just for idling). Added 2026-08-03 for reboot-resilience.
|
|
systemd.services.cc-ci-orchestrator = {
|
|
description = "cc-ci orchestrator (operator steering session) — agents.py up orchestrator";
|
|
wantedBy = [ "multi-user.target" ];
|
|
after = [ "network-online.target" "tailscaled.service" "opencode-web.service" ];
|
|
wants = [ "network-online.target" ];
|
|
serviceConfig = {
|
|
# KillMode=process: see the note on cc-ci-loops — a rebuild that merely touches this unit
|
|
# must not tear down the (shared) tmux server and every agent session with it.
|
|
KillMode = "process";
|
|
Type = "oneshot"; RemainAfterExit = true;
|
|
User = "loops"; Group = "users";
|
|
WorkingDirectory = "/srv/cc-ci-orch";
|
|
};
|
|
environment = { HOME = "/home/loops"; };
|
|
path = [ pkgs.bash pkgs.tmux pkgs.git pkgs.python3 pkgs.openssh pkgs.nettools ];
|
|
script = ''
|
|
export PATH="/home/loops/.local/bin:$PATH"
|
|
proj="/srv/cc-ci-orch"
|
|
echo "$(cat /proc/sys/kernel/random/boot_id) boot $(date -u +%FT%TZ) — cc-ci-orchestrator up" \
|
|
>> "$proj/cc-ci-plan/.ao-boot.log" 2>/dev/null || true
|
|
cd "$proj" && python3 cc-ci-plan/agents.py up orchestrator || echo "cc-ci orchestrator agents.py up failed"
|
|
'';
|
|
};
|
|
|
|
# Weekly recipe upgrade — runs /upgrade-all over every enrolled recipe (opens recipe PRs
|
|
# verified by !testme, never merges). Replaces the boot-fragile busybox-crond-in-tmux from
|
|
# phase 5 §4 with a reboot-safe systemd timer. The service is timer-triggered only (NOT
|
|
# wantedBy multi-user.target) so it never runs on boot/activation — only on the schedule.
|
|
systemd.services.cc-ci-upgrade-all = {
|
|
description = "cc-ci weekly /upgrade-all run (recipe upgrade survey + PRs, never merges)";
|
|
after = [ "network-online.target" "tailscaled.service" "claude-install.service" ];
|
|
wants = [ "network-online.target" ];
|
|
serviceConfig = {
|
|
Type = "oneshot"; # launch-upgrader.py spawns the cc-ci-upgrader tmux session and returns
|
|
User = "loops"; Group = "users";
|
|
WorkingDirectory = "/srv/cc-ci";
|
|
# Optional per-run overrides for backend/model (LOOP_BACKEND, LOOP_MODEL, OPENCODE_SHARE,
|
|
# UPGRADER_ARGS, …). The leading "-" makes it optional: absent file → claude/sonnet defaults.
|
|
# Current config (as of 2026-08-16): the upgrader + report run on tinfoil/deepseek-v4-pro
|
|
# (LOOP_MODEL + REPORT_MODEL in the env file); the hourly SUPERVISOR stays on glm-5.2
|
|
# (SUPERVISOR_MODEL defaults to opencode-go/glm-5.2 in launch-supervisor.py, NOT overridden
|
|
# here). Subagents bind deepseek via the cc-ci repo's opencode config. LOOP_TIER=zen is kept
|
|
# so the tier check passes; the watchdog's usage-limit probe sends the deepseek model name to
|
|
# the zen endpoint, which returns 200 (not 429) → resume immediately (correct: tinfoil has no
|
|
# rolling usage limit to wait out). No rebuild needed to switch — the env file is read at each
|
|
# timer fire. Holds no secrets (the tinfoil API key lives in the opencode config / auth.json).
|
|
EnvironmentFile = "-/srv/cc-ci/upgrader.env";
|
|
};
|
|
environment = { HOME = "/home/loops"; CLAUDE_BIN = "/home/loops/.local/bin/claude"; };
|
|
path = [ pkgs.bash pkgs.tmux pkgs.git pkgs.python3 pkgs.openssh pkgs.nettools ];
|
|
script = ''
|
|
export PATH="/home/loops/.local/bin:$PATH"
|
|
python3 /srv/cc-ci/cc-ci-plan/launch-upgrader.py start >> /srv/cc-ci/.cc-ci-logs/upgrader-cron.log 2>&1
|
|
'';
|
|
};
|
|
|
|
systemd.timers.cc-ci-upgrade-all = {
|
|
description = "Weekly trigger for cc-ci-upgrade-all (Thursdays 22:00 America/New_York — Boston 10pm)";
|
|
wantedBy = [ "timers.target" ];
|
|
timerConfig = {
|
|
# 10pm Thursday Boston time — DST-aware (EDT→02:00 UTC, EST→03:00 UTC) via the tz in OnCalendar.
|
|
OnCalendar = "Thu *-*-* 22:00:00 America/New_York";
|
|
Persistent = true; # if the box was down at the scheduled time, run once on next boot
|
|
};
|
|
};
|
|
|
|
# Hourly SUPERVISOR — a glm-5.2 orchestrator wake-up that keeps the weekly run on track. The
|
|
# log-idle/429 watchdog only handles opencode-go usage-limit stalls; it does NOT cover a host
|
|
# disk-full crash (which killed the 2026-07-03 run) or any other environmental wedge. This is a
|
|
# CHEAP deterministic gate: if the weekly run is complete or actively progressing it does NOTHING
|
|
# (zero model tokens). Only when a run has stalled/died before completing does it launch a
|
|
# short-lived glm-5.2 agent that diagnoses the blockage and drives the run to a clean DONE.
|
|
systemd.services.cc-ci-upgrade-supervisor = {
|
|
description = "cc-ci hourly weekly-run supervisor (glm-5.2 — drives a stalled /upgrade-all to completion)";
|
|
after = [ "network-online.target" "tailscaled.service" ];
|
|
wants = [ "network-online.target" ];
|
|
serviceConfig = {
|
|
Type = "oneshot"; # launch-supervisor.py check: gate now, spawn the agent into tmux, return
|
|
User = "loops"; Group = "users";
|
|
WorkingDirectory = "/srv/cc-ci";
|
|
# Shares the weekly run's optional override file (e.g. SUPERVISOR_MODEL=…); "-" = optional.
|
|
EnvironmentFile = "-/srv/cc-ci/upgrader.env";
|
|
};
|
|
environment = { HOME = "/home/loops"; };
|
|
path = [ pkgs.bash pkgs.tmux pkgs.git pkgs.python3 pkgs.openssh pkgs.nettools ];
|
|
script = ''
|
|
export PATH="/home/loops/.local/bin:$PATH"
|
|
python3 /srv/cc-ci/cc-ci-plan/launch-supervisor.py check >> /srv/cc-ci/.cc-ci-logs/supervisor-cron.log 2>&1
|
|
'';
|
|
};
|
|
|
|
systemd.timers.cc-ci-upgrade-supervisor = {
|
|
description = "Hourly trigger for cc-ci-upgrade-supervisor (weekly-run health check + drive)";
|
|
wantedBy = [ "timers.target" ];
|
|
timerConfig = {
|
|
OnCalendar = "*-*-* *:07:00"; # every hour at :07 (offset from the weekly :00 fire)
|
|
Persistent = false; # a missed hourly check is moot — the next hour re-checks
|
|
};
|
|
};
|
|
}
|