From a5bb8a61a186d5c3c74bfbac99e87519c3b22baf Mon Sep 17 00:00:00 2001 From: maziggy Date: Mon, 29 Jun 2026 13:15:07 +0200 Subject: [PATCH] ci(repo-stats): chart container pulls from ghcr.io alongside clones/stars MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit GHCR exposes total + 30-day daily-pull counts only in the package page HTML (no REST or GraphQL endpoint). jgehrcke/github-repo-stats has no notion of container metrics, so post-process the report after it runs. New: .github/scripts/ghcr_inject.py - Scrapes Total downloads (exact integer from title="N", not the K-rounded display) and the 30-day sparkline (rect data-merge-count, data-date). - Merges per-day rows into maziggy/bambuddy/ghcr-pulls.csv on gh-pages. Fresh window overwrites overlapping dates, so GitHub's late revisions to the last 30 days self-correct; days older than 30 stay frozen at whatever was captured while still in-window. - Patches latest-report/report.html: adds a TOC entry, a Container pulls (ghcr.io) section at the top, and a Vega-Lite line+point chart whose theme/config is cloned from the existing Total clones chart so it inherits the report's look-and-feel. - Bracketed by HTML-comment markers so re-runs replace rather than stack (jgehrcke regenerates report.html every tick; we re-inject). - Hard-fails if either scrape pattern stops matching — silent fallbacks would let the chart freeze without notice. Workflow: after run-ghrs, checkout source + gh-pages, run the injector, commit only if the diff is non-empty. Uses the existing contents: write permission; no new secrets. --- .github/scripts/ghcr_inject.py | 326 +++++++++++++++++++++++++++++++ .github/workflows/repo-stats.yml | 37 ++++ 2 files changed, 363 insertions(+) create mode 100644 .github/scripts/ghcr_inject.py diff --git a/.github/scripts/ghcr_inject.py b/.github/scripts/ghcr_inject.py new file mode 100644 index 000000000..ad7413a41 --- /dev/null +++ b/.github/scripts/ghcr_inject.py @@ -0,0 +1,326 @@ +#!/usr/bin/env python3 +"""Inject GHCR container-download stats into the jgehrcke/github-repo-stats report. + +GHCR exposes a container's total + 30-day daily-pull series only in the +package page HTML. There is no REST or GraphQL API for it. This script +scrapes that page once per workflow run, merges the rolling 30-day window +into a sidecar CSV on gh-pages, and re-injects a Vega-Lite chart at the +top of ``latest-report/report.html``. + +Why the merge: each run only sees the last 30 days, but the CSV grows +forever — days that fall off GitHub's 30-day window stay in the CSV +because they were captured while in-window. Overlapping dates are +overwritten on each run, so GitHub's late-arriving revisions to recent +days self-correct. + +Hard-fails if either scrape pattern stops matching. Silent fallbacks +would let the chart freeze at last-known-good and nobody would notice. +""" + +from __future__ import annotations + +import argparse +import csv +import json +import re +import sys +import urllib.request +from datetime import datetime, timezone +from pathlib import Path + +GHCR_URL = "https://github.com/{owner}/{pkg}/pkgs/container/{pkg}" +USER_AGENT = "Mozilla/5.0 (X11; Linux x86_64) bambuddy-stats" + +TOTAL_RE = re.compile( + r'Total downloads\s*

([^<]+)

', + re.DOTALL, +) +RECT_MERGE_FIRST_RE = re.compile(r'data-merge-count="(\d+)"[^>]*data-date="(\d{4}-\d{2}-\d{2})"') +RECT_DATE_FIRST_RE = re.compile(r'data-date="(\d{4}-\d{2}-\d{2})"[^>]*data-merge-count="(\d+)"') + + +def fetch_ghcr(owner: str, pkg: str) -> str: + req = urllib.request.Request( + GHCR_URL.format(owner=owner, pkg=pkg), + headers={"User-Agent": USER_AGENT, "Accept": "text/html"}, + ) + with urllib.request.urlopen(req, timeout=30) as resp: + return resp.read().decode("utf-8") + + +def parse_total(html: str) -> tuple[int, str]: + m = TOTAL_RE.search(html) + if not m: + raise RuntimeError( + "GHCR scrape: 'Total downloads' marker not found. GitHub markup likely changed — update TOTAL_RE." + ) + return int(m.group(1)), m.group(2).strip() + + +def parse_daily(html: str) -> dict[str, int]: + daily: dict[str, int] = {} + for m in RECT_MERGE_FIRST_RE.finditer(html): + daily[m.group(2)] = int(m.group(1)) + for m in RECT_DATE_FIRST_RE.finditer(html): + daily.setdefault(m.group(1), int(m.group(2))) + if not daily: + raise RuntimeError( + "GHCR scrape: 30-day sparkline rects not found. GitHub markup likely changed — update RECT_*_RE." + ) + return daily + + +def merge_csv(csv_path: Path, fresh: dict[str, int]) -> dict[str, int]: + merged: dict[str, int] = {} + if csv_path.exists(): + with csv_path.open() as fp: + for row in csv.DictReader(fp): + merged[row["date"]] = int(row["daily_count"]) + merged.update(fresh) + return dict(sorted(merged.items())) + + +def write_csv(csv_path: Path, series: dict[str, int]) -> None: + csv_path.parent.mkdir(parents=True, exist_ok=True) + with csv_path.open("w", newline="") as fp: + w = csv.writer(fp) + w.writerow(["date", "daily_count"]) + for date, count in series.items(): + w.writerow([date, count]) + + +# Cloned verbatim from jgehrcke's "Total clones" chart so the new chart +# inherits the report's theme (fonts, palette, axis colors). +VEGA_CONFIG = { + "arc": {"fill": "#1b1e23"}, + "area": {"fill": "#1b1e23"}, + "axisBottom": { + "domainColor": "#a9b4c4", + "gridColor": "#a9b4c4", + "labelColor": "#1b1e23", + "labelFont": "relative-mono-11-pitch-pro, Menlo, monospace", + "tickColor": "#a9b4c4", + "titleColor": "#1b1e23", + "titleFont": "relative-mono-11-pitch-pro, Menlo, monospace", + }, + "axisLeft": { + "domainColor": "#a9b4c4", + "gridColor": "#a9b4c4", + "labelColor": "#1b1e23", + "labelFont": "relative-mono-11-pitch-pro, Menlo, monospace", + "tickColor": "#a9b4c4", + "titleColor": "#1b1e23", + "titleFont": "relative-mono-11-pitch-pro, Menlo, monospace", + }, + "axisX": {"grid": False}, + "axisY": {"grid": False, "labelBound": True}, + "background": "#FFFFFF", + "group": {"fill": "#FFFFFF"}, + "header": { + "fontWeight": 400, + "labelFont": "relative-mono-11-pitch-pro, Menlo, monospace", + "titleFont": "relative-mono-11-pitch-pro, Menlo, monospace", + }, + "legend": { + "labelFont": "relative-mono-11-pitch-pro, Menlo, monospace", + "symbolSize": 200, + "symbolType": "circle", + "titleFont": "relative-mono-11-pitch-pro, Menlo, monospace", + }, + "line": {"color": "#1b1e23", "stroke": "#1b1e23"}, + "path": {"stroke": "#1b1e23"}, + "point": { + "color": "#1b1e23", + "cursor": "pointer", + "filled": True, + "size": 20, + }, + "range": { + "category": ["#85a2f7", "#ea9755", "#7eb36a", "#f07071", "#bc85d9", "#e587b6", "#a9b4c4", "#d4c05e", "#64b9c4"], + }, + "style": { + "bar": {"fill": "#1b1e23"}, + "text": { + "font": "relative-mono-11-pitch-pro, Menlo, monospace", + "fontWeight": 400, + }, + }, + "symbol": {"shape": "circle"}, + "title": { + "anchor": "start", + "font": "relative-mono-11-pitch-pro, Menlo, monospace", + "fontWeight": 400, + }, + "trail": {"color": "#1b1e23", "stroke": "#1b1e23"}, + "view": {"stroke": None}, +} + + +def build_vega_spec(series: dict[str, int]) -> dict: + rows = [{"time": f"{date}T00:00:00+00:00", "daily_count": count} for date, count in series.items()] + counts = [r["daily_count"] for r in rows] or [1] + y_max = max(counts) + dates = sorted(series.keys()) + x_domain = [dates[0], dates[-1]] if dates else None + return { + "$schema": "https://vega.github.io/schema/vega-lite/v4.17.0.json", + "config": VEGA_CONFIG, + "data": {"name": "data-ghcr-pulls"}, + "datasets": {"data-ghcr-pulls": rows}, + "encoding": { + "tooltip": [ + {"field": "daily_count", "format": ".0f", "title": "pulls", "type": "quantitative"}, + {"field": "time", "format": "%B %e, %Y", "title": "date", "type": "temporal"}, + ], + "x": { + "axis": {"labelAngle": 25}, + "field": "time", + "scale": {"domain": x_domain} if x_domain else {}, + "timeUnit": "yearmonthdate", + "title": "date", + "type": "temporal", + }, + "y": { + "axis": {"values": [1, 10, 50, 100, 500, 1000, 5000, 10000, 50000]}, + "field": "daily_count", + "scale": { + "domain": [0, y_max * 1.1 if y_max > 0 else 1], + "type": "symlog", + "zero": True, + }, + "title": "container pulls per day", + "type": "quantitative", + }, + }, + "height": 200, + "mark": {"point": True, "type": "line"}, + "padding": 10, + "width": "container", + } + + +TOC_START = "" +TOC_END = "" +SECTION_START = "" +SECTION_END = "" +SCRIPT_START = "" +SCRIPT_END = "" + + +def _strip_existing(html: str, start: str, end: str) -> str: + pattern = re.compile(re.escape(start) + r".*?" + re.escape(end) + r"\n?", re.DOTALL) + return pattern.sub("", html) + + +def patch_report( + report_path: Path, + spec: dict, + cumulative: int, + cumulative_display: str, + fetched_at: str, + owner: str, + pkg: str, +) -> None: + html = report_path.read_text(encoding="utf-8") + + # Idempotency: if a prior run left markers (shouldn't happen because + # jgehrcke regenerates the file, but guard against partial re-runs), + # strip them before re-injecting. + for s, e in ( + (TOC_START, TOC_END), + (SECTION_START, SECTION_END), + (SCRIPT_START, SCRIPT_END), + ): + html = _strip_existing(html, s, e) + + toc_block = f'{TOC_START}\n
  • Container pulls (ghcr.io)
  • \n{TOC_END}\n' + section_block = ( + f"{SECTION_START}\n" + f'

    Container pulls (ghcr.io)

    \n' + f"

    Daily pulls of ghcr.io/{owner}/{pkg}. " + f"Cumulative: {cumulative:,} " + f"({cumulative_display}). Source refreshed {fetched_at}.

    \n" + f'

    Pulls per day

    \n' + f'
    \n\n
    \n' + f'
    \n\n
    \n' + f"{SECTION_END}\n" + ) + script_block = ( + f"{SCRIPT_START}\n" + f'\n" + f"{SCRIPT_END}\n" + ) + + toc_anchor = "

    Table of contents:

    \n