From 5d6bd5a551a41f43ee2f4652d1c76e7f96b1d947 Mon Sep 17 00:00:00 2001 From: priya-sundaram-dev Date: Tue, 8 Sep 2026 23:42:14 +0000 Subject: [PATCH 1/3] fix(ci): open a PR to persist Hacktoberfest tracker instead of pushing to protected master --- .github/workflows/hacktoberfest_prep.yml | 32 ++++++++++++++++++------ 1 file changed, 25 insertions(+), 7 deletions(-) diff --git a/.github/workflows/hacktoberfest_prep.yml b/.github/workflows/hacktoberfest_prep.yml index fed359a0d621..09cb8408712c 100644 --- a/.github/workflows/hacktoberfest_prep.yml +++ b/.github/workflows/hacktoberfest_prep.yml @@ -20,6 +20,7 @@ on: permissions: contents: write + pull-requests: write jobs: hacktoberfest-prep: @@ -57,13 +58,30 @@ jobs: if git diff --quiet -- docs/hacktober_2026_prep.md; then echo "No changes to docs/hacktober_2026_prep.md." fi - - name: Commit any changes + # `master` is a protected branch: direct pushes are rejected with + # `GH006: Protected branch update failed ... Changes must be made through + # a pull request`. So instead of committing straight to master, persist + # the refreshed tracker by opening — or updating in place — a single + # rolling pull request. Re-runs reuse the same branch, so at most one + # open PR exists at any time. + - name: Open or update the tracker pull request if: github.event_name != 'push' && github.event_name != 'pull_request' - run: | - git config --global user.name "$GITHUB_ACTOR" - git config --global user.email "$GITHUB_ACTOR@users.noreply.github.com" - git add docs/hacktober_2026_prep.md - git commit -m "chore: refresh Hacktoberfest 2026 prep tracker" || echo "No changes to commit" - git push || echo "Nothing to push" + uses: peter-evans/create-pull-request@v7 + with: + add-paths: docs/hacktober_2026_prep.md + branch: chore/hacktoberfest-2026-prep-refresh + delete-branch: true + commit-message: "chore: refresh Hacktoberfest 2026 prep tracker" + title: "chore: refresh Hacktoberfest 2026 prep tracker" + body: | + Automated daily refresh of the Hacktoberfest 2026 open-PR cleanup + tracker (`docs/hacktober_2026_prep.md`): ticks off any tracked pull + request that has since been merged/closed and rewrites the + **Automated statistics** section. + + This PR is updated in place by the `hacktoberfest_prep` workflow, so + it always reflects the latest scheduled run. Merge it whenever you + want to capture the current snapshot; a fresh one opens on the next + run if there are new changes. - name: Propagate the script's exit code run: exit ${{ steps.update.outputs.exit_code }} From 049c7df896f3c11164542e7085e8f674cd9a3a47 Mon Sep 17 00:00:00 2001 From: Priya Sundaram Date: Wed, 9 Sep 2026 00:07:58 +0000 Subject: [PATCH 2/3] ci: use bundled gh CLI instead of peter-evans/create-pull-request Per @cclauss / zizmor 'superfluous actions' audit, persist the rolling tracker PR with the gh CLI rather than a third-party action. --- .github/workflows/hacktoberfest_prep.yml | 50 +++++++++++++++--------- 1 file changed, 31 insertions(+), 19 deletions(-) diff --git a/.github/workflows/hacktoberfest_prep.yml b/.github/workflows/hacktoberfest_prep.yml index 09cb8408712c..b767350388ac 100644 --- a/.github/workflows/hacktoberfest_prep.yml +++ b/.github/workflows/hacktoberfest_prep.yml @@ -61,27 +61,39 @@ jobs: # `master` is a protected branch: direct pushes are rejected with # `GH006: Protected branch update failed ... Changes must be made through # a pull request`. So instead of committing straight to master, persist - # the refreshed tracker by opening — or updating in place — a single - # rolling pull request. Re-runs reuse the same branch, so at most one - # open PR exists at any time. + # the refreshed tracker on a single rolling branch and open — or, since + # re-pushing the branch updates the existing PR in place, leave open — one + # pull request. Uses the bundled `gh` CLI rather than a third-party + # action (see zizmor's "superfluous actions" audit). - name: Open or update the tracker pull request if: github.event_name != 'push' && github.event_name != 'pull_request' - uses: peter-evans/create-pull-request@v7 - with: - add-paths: docs/hacktober_2026_prep.md - branch: chore/hacktoberfest-2026-prep-refresh - delete-branch: true - commit-message: "chore: refresh Hacktoberfest 2026 prep tracker" - title: "chore: refresh Hacktoberfest 2026 prep tracker" - body: | - Automated daily refresh of the Hacktoberfest 2026 open-PR cleanup - tracker (`docs/hacktober_2026_prep.md`): ticks off any tracked pull - request that has since been merged/closed and rewrites the - **Automated statistics** section. + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + BRANCH: chore/hacktoberfest-2026-prep-refresh + run: | + set -euo pipefail + if git diff --quiet -- docs/hacktober_2026_prep.md; then + echo "No changes to docs/hacktober_2026_prep.md; nothing to persist." + exit 0 + fi + git config --global user.name "$GITHUB_ACTOR" + git config --global user.email "$GITHUB_ACTOR@users.noreply.github.com" + git switch -c "$BRANCH" + git add docs/hacktober_2026_prep.md + git commit -m "chore: refresh Hacktoberfest 2026 prep tracker" + # Force-push so the rolling branch always carries just the latest + # snapshot on top of master; this also updates any open PR in place. + git push --force origin "$BRANCH" + if [ -z "$(gh pr list --head "$BRANCH" --state open --json number --jq '.[].number')" ]; then + gh pr create \ + --base master \ + --head "$BRANCH" \ + --title "chore: refresh Hacktoberfest 2026 prep tracker" \ + --body "Automated daily refresh of the Hacktoberfest 2026 open-PR cleanup tracker (\`docs/hacktober_2026_prep.md\`): ticks off any tracked pull request that has since been merged/closed and rewrites the **Automated statistics** section. - This PR is updated in place by the `hacktoberfest_prep` workflow, so - it always reflects the latest scheduled run. Merge it whenever you - want to capture the current snapshot; a fresh one opens on the next - run if there are new changes. + This PR is updated in place by the \`hacktoberfest_prep\` workflow, so it always reflects the latest scheduled run. Merge it whenever you want to capture the current snapshot." + else + echo "Open tracker PR already exists; force-push updated it in place." + fi - name: Propagate the script's exit code run: exit ${{ steps.update.outputs.exit_code }} From 3f66155bb432d7552c78c75dd9718e1dbab4e7fa Mon Sep 17 00:00:00 2001 From: Priya Sundaram Date: Wed, 9 Sep 2026 00:42:49 +0000 Subject: [PATCH 3/3] fix(ci): make tracker refresh degrade gracefully when rate limited MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The dry run was failing because a run can exhaust the GITHUB_TOKEN's 1000/hour-per-repo budget (shared across concurrent runs) — chiefly the awaiting-reviews directory scan. A single exhausted request then raised and killed the whole job. - _request now honours Retry-After (secondary limits) and, once retries are exhausted, raises BestEffortError instead of a bare RuntimeError. - Row resolution, the directory scan, and the search counts catch BestEffortError and degrade (keep the row / mark the stat unavailable) instead of failing. Only the post-Oct-1 retirement exits non-zero. - Trim the directory scan to 120 PRs and CONCURRENCY to 5 to stay well under the shared budget in the first place. --- scripts/hacktoberfest_prep_update.py | 128 ++++++++++++++++++++------- 1 file changed, 97 insertions(+), 31 deletions(-) diff --git a/scripts/hacktoberfest_prep_update.py b/scripts/hacktoberfest_prep_update.py index daf32c2f9bd0..5dcd54dd8654 100644 --- a/scripts/hacktoberfest_prep_update.py +++ b/scripts/hacktoberfest_prep_update.py @@ -43,10 +43,14 @@ AWAITING_LABEL = "awaiting reviews" HACKTOBERFEST_START = dt.date(2026, 10, 1) -# How many API requests to keep in flight at once. GitHub's authenticated -# primary limit is 5000/hour, but bursts of concurrent requests can trip the -# secondary limits, so keep this modest. -CONCURRENCY = 8 +# How many API requests to keep in flight at once. A user token's primary +# limit is 5000/hour, but the ``GITHUB_TOKEN`` the runner hands us is capped at +# 1000/hour *per repository* and shared across every concurrent workflow run, so +# several dry runs in the same hour can exhaust it between them. Bursts of +# concurrent requests can also trip the secondary limits. Keep this modest and +# let the callers degrade gracefully when a request can't be satisfied (see +# ``BestEffortError``) rather than failing the whole job. +CONCURRENCY = 5 # A tracked row looks like: ``12. [ ] #15144 awaiting reviews`` ROW_RE = re.compile( @@ -55,6 +59,16 @@ STATS_HEADER = "## Automated statistics" +class BestEffortError(RuntimeError): + """A row/statistic could not be fetched (e.g. rate limited). + + Raised by :func:`_request` once every retry is exhausted. The refresh is + best-effort: callers catch this so an unreachable API degrades the tracker + (keep the old value / omit a stat) instead of failing the whole job. The + only intentional non-zero exit is the post-Oct-1 retirement. + """ + + def _log(message: str) -> None: """Emit a progress line to stderr, flushed so Actions shows it live.""" print(message, file=sys.stderr, flush=True) @@ -86,19 +100,32 @@ async def _request( resp = await client.get(url, params=params) if resp.is_success: return resp.json(), dict(resp.headers) - remaining = resp.headers.get("X-RateLimit-Remaining") - if resp.status_code in (403, 429) and remaining == "0": - reset = int(resp.headers.get("X-RateLimit-Reset", "0")) - wait = max(1, reset - int(time.time())) + 1 - _log(f"Rate limited on {url}; sleeping {min(wait, 90)}s") - await asyncio.sleep(min(wait, 90)) - continue + # Both primary ("remaining == 0") and secondary/abuse rate limits + # come back as 403/429. Primary limits advertise a reset epoch; + # secondary limits instead send a ``Retry-After`` (seconds) and may + # still report a non-zero remaining, so honour either signal. + if resp.status_code in (403, 429): + remaining = resp.headers.get("X-RateLimit-Remaining") + retry_after = resp.headers.get("Retry-After") + if retry_after is not None: + wait = int(retry_after) + 1 + elif remaining == "0": + reset = int(resp.headers.get("X-RateLimit-Reset", "0")) + wait = max(1, reset - int(time.time())) + 1 + else: + wait = 0 + if wait and attempt < 3: + _log(f"Rate limited on {url}; sleeping {min(wait, 90)}s") + await asyncio.sleep(min(wait, 90)) + continue if resp.status_code >= 500 and attempt < 3: await asyncio.sleep(2 * (attempt + 1)) continue resp.raise_for_status() - msg = f"giving up on {url}" - raise RuntimeError(msg) + # Exhausted every retry (typically the shared per-repo budget ran dry). + # Signal the callers to degrade rather than crash the whole run. + msg = f"giving up on {url} after repeated rate limiting" + raise BestEffortError(msg) async def _search_count( @@ -135,7 +162,7 @@ async def top_awaiting_directories( client: httpx2.AsyncClient, sem: asyncio.Semaphore, limit: int = 3, - max_prs: int = 400, + max_prs: int = 120, ) -> list[tuple[str, int]]: """Count open ``awaiting reviews`` PRs by the top-level directory they touch.""" query = f'repo:{REPO} is:pr is:open label:"{AWAITING_LABEL}"' @@ -176,12 +203,22 @@ async def dirs_for(number: int) -> set[str]: _log(f" ...scanned {done}/{total} PR(s)") return dirs - results = await asyncio.gather(*(dirs_for(n) for n in numbers)) + results = await asyncio.gather( + *(dirs_for(n) for n in numbers), return_exceptions=True + ) counts: dict[str, int] = {} + skipped = 0 for dirs in results: + if isinstance(dirs, BestEffortError): + skipped += 1 + continue + if isinstance(dirs, BaseException): + raise dirs for directory in dirs: counts[directory] = counts.get(directory, 0) + 1 + if skipped: + _log(f" ...{skipped} PR(s) skipped (API unavailable); ranking partial.") ranked = sorted(counts.items(), key=lambda kv: (-kv[1], kv[0])) return ranked[:limit] @@ -199,12 +236,23 @@ async def refresh_checkboxes( ] if pending: _log(f"Checking {len(pending)} open tracker row(s) for resolution...") - states = dict( - zip( - pending, - await asyncio.gather(*(pr_state(client, sem, n) for n in pending)), - ) + # ``return_exceptions`` keeps one rate-limited row from cancelling the rest: + # a row we couldn't resolve is simply left unchanged (treated as ``None``). + resolved = await asyncio.gather( + *(pr_state(client, sem, n) for n in pending), return_exceptions=True ) + states: dict[int, str | None] = {} + unresolved = 0 + for number, result in zip(pending, resolved): + if isinstance(result, BestEffortError): + unresolved += 1 + states[number] = None + elif isinstance(result, BaseException): + raise result + else: + states[number] = result + if unresolved: + _log(f" ...{unresolved} row(s) left unchanged (API unavailable).") updated = 0 out: list[str] = [] @@ -225,13 +273,23 @@ async def refresh_checkboxes( async def build_stats_block(client: httpx2.AsyncClient, sem: asyncio.Semaphore) -> str: _log("Collecting open issue/PR counts...") awaiting_query = f'repo:{REPO} is:pr is:open label:"{AWAITING_LABEL}"' + + async def _count_or_none(query: str) -> int | None: + try: + return await _search_count(client, sem, query) + except BestEffortError: + return None + open_issues, open_prs, awaiting = await asyncio.gather( - _search_count(client, sem, f"repo:{REPO} is:issue is:open"), - _search_count(client, sem, f"repo:{REPO} is:pr is:open"), - _search_count(client, sem, awaiting_query), + _count_or_none(f"repo:{REPO} is:issue is:open"), + _count_or_none(f"repo:{REPO} is:pr is:open"), + _count_or_none(awaiting_query), ) today = dt.datetime.now(dt.UTC).date().isoformat() + def _fmt(value: int | None) -> str: + return str(value) if value is not None else "unavailable (rate limited)" + lines = [ STATS_HEADER, "", @@ -240,9 +298,9 @@ async def build_stats_block(client: httpx2.AsyncClient, sem: asyncio.Semaphore) f"`scripts/hacktoberfest_prep_update.py` on {today} (UTC)._" ), "", - f"- **Open issues:** {open_issues}", - f"- **Open pull requests:** {open_prs}", - f"- **Open PRs labelled `{AWAITING_LABEL}`:** {awaiting}", + f"- **Open issues:** {_fmt(open_issues)}", + f"- **Open pull requests:** {_fmt(open_prs)}", + f"- **Open PRs labelled `{AWAITING_LABEL}`:** {_fmt(awaiting)}", "", ( "**Top three directories to work on** (most open pull requests " @@ -250,12 +308,20 @@ async def build_stats_block(client: httpx2.AsyncClient, sem: asyncio.Semaphore) ), "", ] - if top_dirs := await top_awaiting_directories(client, sem): - for rank, (directory, count) in enumerate(top_dirs, start=1): - plural = "PR" if count == 1 else "PRs" - lines.append(f"{rank}. `{directory}/` — {count} awaiting-reviews {plural}") + try: + top_dirs = await top_awaiting_directories(client, sem) + except BestEffortError: + top_dirs = [] + lines.append("_Directory ranking unavailable this run (rate limited)._") else: - lines.append("_No open `awaiting reviews` pull requests found._") + if top_dirs: + for rank, (directory, count) in enumerate(top_dirs, start=1): + plural = "PR" if count == 1 else "PRs" + lines.append( + f"{rank}. `{directory}/` — {count} awaiting-reviews {plural}" + ) + else: + lines.append("_No open `awaiting reviews` pull requests found._") lines.append("") return "\n".join(lines)