Compare commits
7
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
eb1d6d9161 | ||
|
|
0a229ac016 | ||
|
|
de1eb1ca75 | ||
|
|
877aea3814 | ||
|
|
ae40545491 | ||
|
|
5086b2f8bb | ||
|
|
5327a24faa |
@@ -0,0 +1,8 @@
|
||||
{
|
||||
"$schema": "https://opencode.ai/config.json",
|
||||
"agent": {
|
||||
"general": {
|
||||
"model": "opencode/deepseek-v4-pro"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -132,6 +132,17 @@ KEYS: tuple[Key, ...] = (
|
||||
"Callable `(ctx)` invoked after UPGRADE_EXTRA_ENV env_set but before `abra secret generate --all` in the upgrade path. Use to pre-insert secrets that `generate --all` would produce with wrong format (e.g. when the .env.sample spec is commented out).",
|
||||
hook_params=("ctx",),
|
||||
),
|
||||
Key(
|
||||
"UPGRADE_BASE_FLOOR",
|
||||
"str",
|
||||
None,
|
||||
"Declared STRUCTURAL breaking boundary for the upgrade tier (phase basefloor): the first "
|
||||
"post-break published version tag. Bases strictly below it are excluded from resolution "
|
||||
"(canonical / step-back / no-canonical fallback) because an in-place upgrade across the "
|
||||
"boundary is not supported upstream (e.g. a db-family change). When no ≥-floor predecessor "
|
||||
"exists the tier records a DECLARED skip. NOT a static base pin (§2.G stays removed) — "
|
||||
"resolution remains dynamic above the floor.",
|
||||
),
|
||||
# (CHAOS_BASE_DEPLOY, OIDC_AT_INSTALL and SKIP_GENERIC were deleted in restructure P2:
|
||||
# compose.ccci.yml is first-class + auto-chaos; install-time deps wiring is the only mode;
|
||||
# the generic floor is suppressible only via the dev-only CCCI_SKIP_GENERIC* env form.)
|
||||
|
||||
+48
-11
@@ -151,8 +151,30 @@ def resolve_upgrade_base(
|
||||
flush=True,
|
||||
)
|
||||
return BasePlan("skip", None, None, f"declared EXPECTED_NA[upgrade]: {declared}")
|
||||
# UPGRADE_BASE_FLOOR (phase basefloor): a recipe_meta declaration marking a STRUCTURAL breaking
|
||||
# boundary — published versions strictly below the floor are not valid in-place upgrade sources
|
||||
# (e.g. discourse 0.8.x→1.0.0 changed the db family bitnami/pgvector → discourse/postgres; the
|
||||
# data layout+roles are incompatible, upstream supports no in-place path across it). This is NOT
|
||||
# the removed UPGRADE_BASE_VERSION pin (§2.G): resolution stays fully dynamic — the floor only
|
||||
# EXCLUDES structurally-impossible bases, and when no candidate ≥ floor exists the tier records
|
||||
# a DECLARED skip (never a silent pass). Never weakens: below-floor upgrades were never a
|
||||
# supported path, so no real coverage is lost.
|
||||
floor = getattr(meta, "UPGRADE_BASE_FLOOR", None)
|
||||
|
||||
def _below_floor(version: str) -> bool:
|
||||
return bool(floor) and warm_reconcile.version_key(version) < warm_reconcile.version_key(
|
||||
floor
|
||||
)
|
||||
|
||||
skip_canonicals = settings_mod.get().skip_canonicals_for_upgrade
|
||||
rec = canonical.read_registry(recipe)
|
||||
if rec and rec.get("version") and not skip_canonicals and _below_floor(rec["version"]):
|
||||
print(
|
||||
f"== upgrade tier: last-green canonical {rec['version']} is below the declared "
|
||||
f"UPGRADE_BASE_FLOOR {floor} (structural break) — excluded as a base",
|
||||
flush=True,
|
||||
)
|
||||
rec = None
|
||||
if rec and rec.get("version") and not skip_canonicals:
|
||||
canon = rec["version"]
|
||||
same = head_version is not None and warm_reconcile.version_key(
|
||||
@@ -168,10 +190,20 @@ def resolve_upgrade_base(
|
||||
f"last-green (warm canonical, status={rec.get('status')})",
|
||||
)
|
||||
# canonical == head version → deploying it would be a same-version no-op. Step back to the
|
||||
# newest published version strictly older than the head (phase samever).
|
||||
older = warm_reconcile.newest_older_version(
|
||||
warm_reconcile.recipe_tags(recipe), head_version
|
||||
)
|
||||
# newest published version strictly older than the head (phase samever). Candidates below a
|
||||
# declared UPGRADE_BASE_FLOOR are excluded (phase basefloor — structurally invalid bases).
|
||||
_tags = warm_reconcile.recipe_tags(recipe)
|
||||
if floor:
|
||||
_tags = [t for t in _tags if not _below_floor(t)]
|
||||
older = warm_reconcile.newest_older_version(_tags, head_version)
|
||||
if older is None and floor:
|
||||
return BasePlan(
|
||||
"skip",
|
||||
None,
|
||||
None,
|
||||
f"declared UPGRADE_BASE_FLOOR {floor}: no published predecessor ≥ floor below "
|
||||
f"head {head_version} (all older tags cross a structural break)",
|
||||
)
|
||||
if older:
|
||||
return BasePlan(
|
||||
"version",
|
||||
@@ -189,10 +221,12 @@ def resolve_upgrade_base(
|
||||
# No canonical in play — none recorded, OR SKIP_CANONICALS_FOR_UPGRADE=true (canonical lookup
|
||||
# bypassed entirely, behaving as if none exists). Improved fallback (phase settings §2.C): prefer
|
||||
# a REAL published predecessor (newest release tag < head) over the raw main-tip.
|
||||
return _no_canonical_base(recipe, head_ref, head_version)
|
||||
return _no_canonical_base(recipe, head_ref, head_version, floor=floor)
|
||||
|
||||
|
||||
def _no_canonical_base(recipe: str, head_ref: str | None, head_version: str | None) -> BasePlan:
|
||||
def _no_canonical_base(
|
||||
recipe: str, head_ref: str | None, head_version: str | None, floor: str | None = None
|
||||
) -> BasePlan:
|
||||
"""Upgrade base when no canonical is used (none recorded, its promote failed, or
|
||||
SKIP_CANONICALS_FOR_UPGRADE is true). Release-tag-first fallback (phase settings §2.C):
|
||||
1. most recent release TAG with version strictly older than the PR head — a clean published
|
||||
@@ -202,11 +236,14 @@ def _no_canonical_base(recipe: str, head_ref: str | None, head_version: str | No
|
||||
3. skip — no predecessor (no older tag and head == main-tip, or no main at all).
|
||||
This replaces the old jump-straight-to-main-tip path, so an un-promoted recipe upgrades from a real
|
||||
release base instead of a possibly-untagged WIP commit."""
|
||||
older = (
|
||||
warm_reconcile.newest_older_version(warm_reconcile.recipe_tags(recipe), head_version)
|
||||
if head_version
|
||||
else None
|
||||
)
|
||||
_tags = warm_reconcile.recipe_tags(recipe)
|
||||
if floor:
|
||||
# phase basefloor: exclude structurally-invalid bases below the declared floor; the
|
||||
# main-tip fallback below remains available (it is post-break by definition of the
|
||||
# declaration — the floor names the first post-break published version).
|
||||
_fk = warm_reconcile.version_key(floor)
|
||||
_tags = [t for t in _tags if warm_reconcile.version_key(t) >= _fk]
|
||||
older = warm_reconcile.newest_older_version(_tags, head_version) if head_version else None
|
||||
if older:
|
||||
return BasePlan(
|
||||
"version",
|
||||
|
||||
@@ -23,11 +23,21 @@ HTTP_TIMEOUT = 1200
|
||||
#
|
||||
# UPGRADE-tier BASE (phase prevb — DYNAMIC, no hardcoded UPGRADE_BASE_VERSION): the base the head
|
||||
# upgrades from is resolved at run time — last-green (warm canonical) → fallback target-branch (`main`)
|
||||
# tip → else skip (run_recipe_ci.resolve_upgrade_base). discourse has no warm canonical, so the base is
|
||||
# the `main` tip = bitnamilegacy/discourse:3.5.0, which deploys clean (bitnamilegacy exists) with NO
|
||||
# `previous/` repair needed. The PR head (recipe-maintainers/discourse#4) switches app to the official
|
||||
# `discourse/discourse:3.5.3` and drops the sidekiq service, so the upgrade tier now exercises the REAL
|
||||
# bitnamilegacy→official image migration the PR claims to support.
|
||||
# tip → else skip (run_recipe_ci.resolve_upgrade_base).
|
||||
#
|
||||
# UPGRADE_BASE_FLOOR (phase basefloor, 2026-08-04): the 0.8.x→1.0.0 recipe family switched the app
|
||||
# bitnamilegacy/discourse → official discourse/discourse AND the db pgvector/pgvector:pg17 →
|
||||
# discourse/postgres:pg18. That db-family change is a structural break: the bitnami cluster has no
|
||||
# `discourse` role and pg_upgrade preserves-not-creates roles, so an in-place 0.8.x→1.x deploy can
|
||||
# NEVER converge (app FATALs `role "discourse" does not exist`, swarm rolls back) — upstream ships
|
||||
# no in-place path across it. Without the floor, the resolver's step-back/fallback selected
|
||||
# 0.8.1+3.5.0 (newest tag below the head label) and the upgrade tier red'd on this unsupported
|
||||
# path twice (drone #1165 2026-07-31 diagnosis, #1171/weekly 2026-08-03 — both classified
|
||||
# stale-test, recipe verified green on the real official→official path). Declaring the floor keeps
|
||||
# resolution dynamic and only excludes the structurally-impossible bases; when no ≥-floor
|
||||
# predecessor exists the tier records a DECLARED skip (never a silent pass). No assertion weakened:
|
||||
# below-floor in-place upgrades were never supported coverage.
|
||||
UPGRADE_BASE_FLOOR = "1.0.0+3.5.3"
|
||||
#
|
||||
# compose.ccci.yml is now the ENVIRONMENTAL overlay (all deploys): only app.deploy.update_config.order:
|
||||
# stop-first (node memory reality on the upgrade crossover — see its header). The version-specific
|
||||
|
||||
@@ -6,7 +6,11 @@ migration was never tested. With the version-specific config removed from the al
|
||||
and the dynamic base (last-green/main = bitnamilegacy:3.5.0) deployed only as the *base*, the upgrade
|
||||
chaos redeploy must land the PR head UNMODIFIED. This overlay asserts exactly that, post-upgrade:
|
||||
|
||||
1. the running `app` service image IS the official discourse/discourse:3.5.3 — NOT bitnamilegacy;
|
||||
1. the running `app` service image IS from the official `discourse/discourse` repository —
|
||||
NOT bitnamilegacy. (Version-agnostic since 2026-08-04: the original assertion hardcoded the
|
||||
migration-era pin `:3.5.3` and went stale on the first legitimate app bump (2026.7.1, weekly
|
||||
2026-08-03). The property this test guards is the IMAGE FAMILY — official vs bitnami — not a
|
||||
frozen version; the exact head pin is already exercised by the deploy itself.)
|
||||
2. the `sidekiq` service the PR deletes is GONE from the deployed stack.
|
||||
|
||||
If either fails, the head did not really run (the overlay leaked onto it) → RED. Assertion-only,
|
||||
@@ -26,8 +30,8 @@ def test_head_runs_official_image_not_bitnamilegacy(live_app):
|
||||
f"app image is {image!r} — the bitnamilegacy base leaked onto the PR head "
|
||||
"(the version-specific overlay was applied to the head, the prevb bug)"
|
||||
)
|
||||
assert image.startswith("discourse/discourse:3.5.3"), (
|
||||
f"app image is {image!r}, expected the PR head's official discourse/discourse:3.5.3 "
|
||||
assert image.startswith("discourse/discourse:"), (
|
||||
f"app image is {image!r}, expected the PR head's official discourse/discourse image "
|
||||
"— the head's image migration was not exercised"
|
||||
)
|
||||
|
||||
|
||||
@@ -14,7 +14,10 @@ Both assert real app state (the event reached the analytics store), not just the
|
||||
|
||||
plausible only ingests events for *known* sites — the in-memory `sites_cache` gates ingestion and
|
||||
drops events for unregistered domains (empirically confirmed: an event for an unregistered domain
|
||||
never appears in events_v2). So each test first registers a site row in the metadata postgres, then
|
||||
never appears in events_v2). From v3 (community-edition) a site is only "known" once it belongs to a
|
||||
TEAM; a teamless site is dropped as `dropped_not_found` while the POST still acks 202, so the failure
|
||||
looks like a silent ingestion stall. `_register_site` therefore provisions a team as well when the
|
||||
schema has one. So each test first registers a site row in the metadata postgres, then
|
||||
POSTs repeatedly while polling ClickHouse: the sites_cache must refresh to admit the new site and the
|
||||
event write-buffer must flush to ClickHouse, so the first landing is not instantaneous. Re-POSTing the
|
||||
same event is safe — we assert the row count is >= 1.
|
||||
@@ -51,11 +54,40 @@ def _ch(domain: str, sql: str) -> str:
|
||||
|
||||
|
||||
def _register_site(domain: str, site: str) -> None:
|
||||
"""Insert a site row into the metadata postgres (`db` service) so plausible will ingest events for
|
||||
it. Idempotent (ON CONFLICT DO NOTHING)."""
|
||||
"""Register `site` in the metadata postgres so plausible will ingest events for it.
|
||||
|
||||
Idempotent. Works against BOTH schema generations, because the upgrade tier deploys an older
|
||||
base version before upgrading:
|
||||
|
||||
* v2 (`plausible/analytics`) — a row in `sites` is sufficient.
|
||||
* v3 (`ghcr.io/plausible/community-edition`) — sites belong to a TEAM, and ingestion drops
|
||||
events for a site whose team is missing. The POST still acks 202 and the row still exists in
|
||||
postgres, so the only visible symptom is that nothing ever reaches ClickHouse; the reason is
|
||||
recorded in ClickHouse's own `ingest_counters` as `dropped_not_found`. Verified on cc-ci
|
||||
against v3.2.1: identical site row, no team → `dropped_not_found`; with a team linked →
|
||||
`buffered` and the row appears in `events_v2`.
|
||||
|
||||
The team block is guarded on the schema actually having teams, so this stays a no-op on v2
|
||||
rather than branching on a version string.
|
||||
"""
|
||||
sql = (
|
||||
"INSERT INTO sites (domain, timezone, inserted_at, updated_at, native_stats_start_at) "
|
||||
f"VALUES ('{site}','UTC', now(), now(), now()) ON CONFLICT (domain) DO NOTHING; "
|
||||
"DO $ccci$ "
|
||||
"BEGIN "
|
||||
" IF EXISTS (SELECT 1 FROM information_schema.tables WHERE table_name = 'teams') "
|
||||
" AND EXISTS (SELECT 1 FROM information_schema.columns "
|
||||
" WHERE table_name = 'sites' AND column_name = 'team_id') THEN "
|
||||
" INSERT INTO teams (name, inserted_at, updated_at, accept_traffic_until, setup_complete) "
|
||||
" SELECT 'cc-ci', now(), now(), now() + interval '365 days', true "
|
||||
" WHERE NOT EXISTS (SELECT 1 FROM teams WHERE name = 'cc-ci'); "
|
||||
" UPDATE sites "
|
||||
" SET team_id = COALESCE(team_id, (SELECT id FROM teams WHERE name = 'cc-ci' LIMIT 1)), "
|
||||
" accept_traffic_until = COALESCE(accept_traffic_until, now() + interval '365 days') "
|
||||
f" WHERE domain = '{site}'; "
|
||||
" END IF; "
|
||||
"END "
|
||||
"$ccci$; "
|
||||
f"SELECT domain FROM sites WHERE domain = '{site}';"
|
||||
)
|
||||
out = lifecycle.exec_in_app(
|
||||
|
||||
@@ -17,6 +17,12 @@ def test_plausible_root_serves(live_app):
|
||||
62-char SECRET_KEY_BASE, see recipe_meta.EXTRA_ENV); the dedicated
|
||||
/api/health endpoint is.
|
||||
"""
|
||||
# The custom tier runs AFTER the backup/restore tier, which disrupts postgres under the app and
|
||||
# restarts it. v3 (community-edition) then boots through `sleep 10` + `db createdb` + `db migrate`
|
||||
# + cache warmers before /api/health flips to 200, which does not fit in 60s — that is what put
|
||||
# this recipe RED on build 1224 while install/upgrade/backup/restore all passed. The assertion is
|
||||
# unchanged (still a hard 200 from the real readiness endpoint); only the wait matches the boot
|
||||
# profile the recipe already declares via recipe_meta.HTTP_TIMEOUT (1200).
|
||||
url = f"https://{live_app}/api/health"
|
||||
status, _ = harness_http.retry_http_get(url, expect_status=(200,), max_wait=60, interval=3)
|
||||
status, _ = harness_http.retry_http_get(url, expect_status=(200,), max_wait=300, interval=5)
|
||||
assert status == 200, f"GET {url} HTTP {status}"
|
||||
|
||||
Reference in New Issue
Block a user