CVE detection engine — correctness fixes, spec, tests, audit #2

Merged
autonomic-bot merged 10 commits from review/1-cve-engine into review/session-base 2026-08-11 19:03:53 +00:00
Showing only changes of commit 98a624a13a - Show all commits
+96 -5
View File
@@ -80,6 +80,57 @@ def _github_token() -> str | None:
return None
def _gh_paginate(url: str, hdrs: dict, max_pages: int = 20):
"""Yield every row from a GitHub list endpoint, following Link rel=\"next\" cursors."""
seen_keys = set()
for _ in range(max_pages):
req = urllib.request.Request(url, headers={"User-Agent": UA, **hdrs})
with urllib.request.urlopen(req, timeout=TIMEOUT) as r:
rows = json.load(r)
link = r.headers.get("Link", "") or ""
fresh = 0
for a in rows:
k = a.get("ghsa_id") or json.dumps(a, sort_keys=True)[:120]
if k not in seen_keys:
seen_keys.add(k); fresh += 1
yield a
nxt = None
for part in link.split(","):
if 'rel="next"' in part:
nxt = part.split(";")[0].strip().strip("<>")
if not nxt or fresh == 0:
return
url = nxt
def _tag_date(owner: str, repo: str, version: str | None) -> str | None:
"""Publish date of a release tag, for DATE-BASED classification (see classify_by_date).
Version strings cannot be ordered across a scheme change (semver → calver), but tag dates
always can. Tries the common tag spellings; returns an ISO timestamp or None."""
if not version:
return None
hdrs = {"Accept": "application/vnd.github+json"}
tok = _github_token()
if tok:
hdrs["Authorization"] = f"Bearer {tok}"
for tag in (f"v{version}", version):
try:
ref = json.loads(_fetch(f"https://api.github.com/repos/{owner}/{repo}/git/ref/tags/{tag}", hdrs))
obj = ref.get("object", {})
sha, typ = obj.get("sha"), obj.get("type")
if typ == "tag":
t = json.loads(_fetch(f"https://api.github.com/repos/{owner}/{repo}/git/tags/{sha}", hdrs))
if t.get("tagger", {}).get("date"):
return t["tagger"]["date"]
sha = t.get("object", {}).get("sha")
c = json.loads(_fetch(f"https://api.github.com/repos/{owner}/{repo}/commits/{sha}", hdrs))
return c["commit"]["committer"]["date"]
except Exception: # noqa: BLE001 — try the next spelling
continue
return None
def _fetch(url: str, headers: dict | None = None) -> str:
h = {"User-Agent": UA, "Accept-Encoding": "gzip"}
h.update(headers or {})
@@ -142,7 +193,11 @@ def github_advisories(urls: list[str]) -> list[dict]:
hdrs["Authorization"] = f"Bearer {tok}"
entry = {"source": f"github-advisories:{owner}/{repo}", "status": "ok", "advisories": []}
try:
for a in json.loads(_fetch(api, hdrs)):
# PAGINATE. This endpoint caps at 100 per response and IGNORES ?page= — it returns the
# same rows again, which silently truncates busy projects (discourse has 286; a hand
# count on 2026-08-10 found 123 CVEs in one upgrade window that a single page missed).
# Follow the Link rel="next" cursor to exhaustion instead.
for a in _gh_paginate(api, hdrs):
# An advisory carries ONE ENTRY PER PATCHED RELEASE LINE. n8n patches three
# (1.123.32, 2.17.4, 2.18.1); reading only vulnerabilities[0] silently dropped the
# line our deployment is actually on, so CVE-2026-42231/42232 classified as
@@ -161,6 +216,7 @@ def github_advisories(urls: list[str]) -> list[dict]:
filter(None, (v.get("patched_versions") for v in vulns))
) or None,
"url": a.get("html_url"),
"published_at": a.get("published_at"),
}
)
except urllib.error.HTTPError as e:
@@ -243,7 +299,8 @@ def scan(recipe: str, v_from: str | None, v_to: str | None, registry_dir: str) -
def record(cve: str, src: str, **extra):
e = report["cves"].setdefault(cve, {"sources": [], "severity": None, "ghsa": None,
"vulnerable_range": None, "patched": None, "context": None})
"vulnerable_range": None, "patched": None,
"context": None, "published_at": None})
if src not in e["sources"]:
e["sources"].append(src)
for k, v in extra.items():
@@ -257,7 +314,7 @@ def scan(recipe: str, v_from: str | None, v_to: str | None, registry_dir: str) -
if a.get("cve"):
record(a["cve"], entry["source"], severity=a.get("severity"), ghsa=a.get("ghsa"),
vulnerable_range=a.get("vulnerable_range"), patched=a.get("patched"),
context=a.get("summary"))
context=a.get("summary"), published_at=a.get("published_at"))
for entry in vendor_pages(urls):
report["sources"].append({"source": entry["source"], "status": entry["status"],
@@ -317,11 +374,41 @@ def scan(recipe: str, v_from: str | None, v_to: str | None, registry_dir: str) -
e["classification"] = "unclassified" if not (kf and kt and kp) else "outside-window"
if e["classification"] == "unclassified":
unknown.append(cve)
report["fixed_by_this_upgrade"] = sorted(fixed)
report["unclassified"] = sorted(unknown)
# NEVER report 0 for something we could not determine. When classification was refused, the
# count is UNKNOWN (null) — a 0 would be read as "no CVEs", which is an assertion this scan
# cannot make. Consumers must distinguish "counted 0" from "could not count".
# DATE-BASED FALLBACK (phase datewin). Version strings cannot be ordered across a scheme change,
# but RELEASE DATES always can. Resolve both versions to their tag dates on the primary repo and
# count advisories PUBLISHED inside that window — the method a hand count used on 2026-08-10 to
# establish discourse 3.5.3 (2025-12-30) → 2026.7.1 (2026-07-31) = 123 CVEs, where version
# comparison had produced first a false 133 and then a refusal. Only used when the version path
# refuses; a successful version classification is always preferred (it is exact, not temporal).
date_window = None
if scheme_change and primary and primary.startswith("github-advisories:"):
owner_repo = primary.split(":", 1)[1]
owner, _, repo = owner_repo.partition("/")
d_from, d_to = _tag_date(owner, repo, v_from), _tag_date(owner, repo, v_to)
if d_from and d_to and d_from < d_to:
date_window = (d_from, d_to)
fixed, unknown = [], []
for cve, e in report["cves"].items():
pub = e.get("published_at")
if primary in e["sources"] and pub and d_from < pub <= d_to:
e["classification"] = "fixed-by-this-upgrade (by advisory publish date)"
fixed.append(cve)
else:
e["classification"] = ("outside-window (by date)" if primary in e["sources"]
else "unclassified: different image than the given window")
if primary not in e["sources"]:
unknown.append(cve)
scheme_change = False # resolved by date; a real count is available
report["classified_by"] = "advisory publish date (version scheme changed)"
report["date_window"] = {"from": d_from, "to": d_to}
if not date_window and not scheme_change:
report["classified_by"] = "patched version ranges"
report["fixed_by_this_upgrade"] = sorted(fixed)
report["unclassified"] = sorted(unknown)
report["count_known"] = not scheme_change
report["cve_count_fixed"] = len(fixed) if not scheme_change else None
report["cve_count_total_seen"] = len(report["cves"])
@@ -361,6 +448,10 @@ def markdown(rep: dict) -> str:
return "\n".join(L)
if rep["fixed_by_this_upgrade"]:
L.append(f"\n**CVEs fixed by this upgrade: {rep['cve_count_fixed']}**\n")
if rep.get("date_window"):
L.append(f"_Counted by advisory PUBLISH DATE ({rep['date_window']['from'][:10]}"
f"{rep['date_window']['to'][:10]}) because the version scheme changed across this "
f"jump; version strings cannot be ordered across it._\n")
L.append("| CVE | severity | fixed in | advisory | source |")
L.append("|---|---|---|---|---|")
for cve in rep["fixed_by_this_upgrade"]: