104 lines
4.3 KiB
Python
104 lines
4.3 KiB
Python
#!/usr/bin/env python3
|
|
"""Read the review comments on a Tangled pull (top-level discussion comments).
|
|
|
|
WHY: the appview has no read API for pull comments, and eyeballing the pull page
|
|
HTML per comment is slow. This fetches the pull (or a specific round) and prints
|
|
every top-level comment as author / time / body — so acting on operator review
|
|
notes is one command, not a WebFetch guess.
|
|
|
|
Same session-cookie auth as tangled_pr.py (cookie from the encrypted store,
|
|
tangled.cookie). Read-only: it never posts.
|
|
|
|
USAGE:
|
|
tangled_comments.py --owner notplants-bot.bsky.social --repo lichen.page.review --pull 75
|
|
tangled_comments.py ... --pull 75 --round 2 # a specific round's page
|
|
tangled_comments.py ... --pull 75 --json # machine-readable
|
|
"""
|
|
import argparse, html, json, os, re, sys, urllib.request
|
|
|
|
BASE = "https://tangled.org"
|
|
|
|
|
|
def load_cookie():
|
|
"""Cookie from the encrypted store (tangled.cookie) — see engine/README.md (Secrets)."""
|
|
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
|
import secrets as _store
|
|
c = _store.get("tangled.cookie")
|
|
if not c:
|
|
sys.exit("no tangled.cookie in the secret store — add it with: sops /secrets/store.yaml")
|
|
return c
|
|
|
|
|
|
def fetch(url, cookie):
|
|
req = urllib.request.Request(url, headers={"Cookie": cookie, "User-Agent": "tangled-pr-bot"})
|
|
return urllib.request.urlopen(req, timeout=60).read().decode()
|
|
|
|
|
|
def _text(fragment):
|
|
"""Strip tags from a body fragment down to readable plaintext."""
|
|
t = re.sub(r"(?is)<br\s*/?>", "\n", fragment)
|
|
t = re.sub(r"(?is)</p\s*>", "\n\n", t)
|
|
t = re.sub(r"(?is)<li[^>]*>", "\n- ", t)
|
|
t = re.sub(r"(?s)<[^>]+>", "", t)
|
|
return html.unescape(t).strip()
|
|
|
|
|
|
def parse_comments(doc):
|
|
"""Return [{cid, author, when, iso, uri, body}] in document order.
|
|
|
|
Anchored on the per-comment header block the appview emits:
|
|
id="comment-header-{cid}" ... <a href="/{handle}">{handle}</a> ...
|
|
<time datetime="{iso}">{when}</time> ... class="...comment-body"><div class="prose...">{body}</div>
|
|
"""
|
|
out = []
|
|
heads = list(re.finditer(r'id="comment-header-([0-9a-z]+)"', doc))
|
|
for i, hm in enumerate(heads):
|
|
cid = hm.group(1)
|
|
# bound by the next comment header, not a fixed window: a comment's own
|
|
# reaction/button markup can push its body several KB past the header.
|
|
end = heads[i + 1].start() if i + 1 < len(heads) else len(doc)
|
|
seg = doc[hm.end(): end]
|
|
am = re.search(r'href="/([^"/]+)"[^>]*>\s*([^<]+?)\s*</a>', seg)
|
|
author = html.unescape(am.group(2)) if am else "?"
|
|
tm = re.search(r'<time datetime="([^"]+)"[^>]*>\s*([^<]+?)\s*</time>', seg)
|
|
iso = html.unescape(tm.group(1)) if tm else ""
|
|
when = html.unescape(tm.group(2)) if tm else ""
|
|
bm = re.search(r'comment-body">\s*<div class="prose[^"]*">(.*?)</div>\s*<div class="reactions', seg, re.S)
|
|
if not bm:
|
|
bm = re.search(r'comment-body">\s*<div class="prose[^"]*">(.*?)</div>', seg, re.S)
|
|
body = _text(bm.group(1)) if bm else ""
|
|
out.append({"cid": cid, "author": author, "when": when, "iso": iso,
|
|
"uri": f"at://.../sh.tangled.feed.comment/{cid}", "body": body})
|
|
return out
|
|
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser(description="read a Tangled pull's review comments")
|
|
ap.add_argument("--owner", required=True)
|
|
ap.add_argument("--repo", required=True)
|
|
ap.add_argument("--pull", required=True, type=int)
|
|
ap.add_argument("--round", type=int, default=None, help="a specific round page (default: latest)")
|
|
ap.add_argument("--json", action="store_true")
|
|
a = ap.parse_args()
|
|
|
|
cookie = load_cookie()
|
|
url = f"{BASE}/{a.owner}/{a.repo}/pulls/{a.pull}"
|
|
if a.round is not None:
|
|
url += f"/round/{a.round}"
|
|
doc = fetch(url, cookie)
|
|
comments = parse_comments(doc)
|
|
|
|
if a.json:
|
|
print(json.dumps({"url": url, "count": len(comments), "comments": comments}, indent=2))
|
|
return
|
|
print(f"# pull #{a.pull} ({url}) — {len(comments)} comment(s)\n")
|
|
for i, c in enumerate(comments, 1):
|
|
print(f"[{i}] {c['author']} · {c['when']} ({c['iso']}) #{c['cid']}")
|
|
for line in (c["body"] or "(empty)").splitlines():
|
|
print(f" {line}")
|
|
print()
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|