Files
Python/scripts/hacktoberfest_prep_update.py
priya-sundaram-dev 15c5af0255 fix(ci): correct the Hacktoberfest tracker's open-issue count and add a countdown (#15235)
* fix(ci): count open issues reliably and add a Hacktoberfest countdown

The tracker's "Open issues" line was showing the pull-request total (e.g.
621) instead of the issue total (107). GitHub's `/search/issues`
`is:issue` / `is:pr` qualifiers are unreliable on this large, high-churn
repo -- some runs return the PR pool for *both* queries, so the two lines
printed the same number.

Count without the flaky qualifiers instead:
- open PRs = the `Link: rel="last"` page number of `/repos/{repo}/pulls`
  (deterministic, page-numbered pagination);
- open issues = the repo endpoint's `open_issues_count` (issues + PRs)
  minus the open-PR total -- self-checking and stable.

Also add the requested Hacktoberfest countdown to the stats block:
- days until 2026-10-01;
- issues to close per day to clear the backlog;
- PRs to merge or close per day to clear the backlog.

Per-day figures round up (finishing a day early beats a day late) and
degrade to a clear message once Hacktoberfest starts, so the block never
divides by zero on the final day.

* chore: re-trigger keeper after marking PR checklist
2026-09-09 10:42:03 +02:00

450 lines
17 KiB
Python

#!/usr/bin/env python3
"""Refresh the Hacktoberfest 2026 open-PR cleanup tracker.
This script is run once a day by the ``hacktoberfest_prep`` GitHub Actions
workflow (see ``.github/workflows/hacktoberfest_prep.yml``). It:
1. Reads ``docs/hacktober_2026_prep.md`` and, for every tracked pull request
that is still an unchecked ``[ ]`` box, checks whether the PR has since
been merged or closed. Resolved rows are ticked (``[ ]`` -> ``[x]``) and
annotated with ``merged`` / ``closed``.
2. Rewrites a machine-generated ``## Automated statistics`` section at the end
of the file with the current number of open issues and open pull requests
and the top three algorithm directories that have the most open pull
requests labelled ``awaiting reviews``.
3. Exits non-zero once Hacktoberfest 2026 has begun (on or after
2026-10-01, UTC), so the prep window closing is loud rather than silent.
The slow part of the job is looking at the files touched by every open
``awaiting reviews`` PR (potentially a few hundred of them). Those requests
have no dependencies on one another, so they are fired concurrently through a
single ``httpx2.AsyncClient`` with a bounded concurrency limit — this turns a
long series of round-trips into a handful of batches and cuts the runtime from
~16 minutes to well under a minute. Progress is reported to stderr as it goes
so a human watching the Actions log can see it is alive.
It only needs ``httpx2`` (the repo-standard HTTP client) and the
``GITHUB_TOKEN`` provided by the Actions runner.
"""
import asyncio
import datetime as dt
import os
import re
import sys
import time
import httpx2
REPO = os.environ.get("GITHUB_REPOSITORY", "TheAlgorithms/Python")
TOKEN = os.environ.get("GITHUB_TOKEN", "")
API = "https://api.github.com"
TRACKER = "docs/hacktober_2026_prep.md"
AWAITING_LABEL = "awaiting reviews"
HACKTOBERFEST_START = dt.date(2026, 10, 1)
# How many API requests to keep in flight at once. A user token's primary
# limit is 5000/hour, but the ``GITHUB_TOKEN`` the runner hands us is capped at
# 1000/hour *per repository* and shared across every concurrent workflow run, so
# several dry runs in the same hour can exhaust it between them. Bursts of
# concurrent requests can also trip the secondary limits. Keep this modest and
# let the callers degrade gracefully when a request can't be satisfied (see
# ``BestEffortError``) rather than failing the whole job.
CONCURRENCY = 5
# A tracked row looks like: ``12. [ ] #15144 awaiting reviews``
ROW_RE = re.compile(
r"^(?P<idx>\d+)\.\s+\[(?P<mark>[ x])\]\s+#(?P<pr>\d+)\b(?P<rest>.*)$"
)
STATS_HEADER = "## Automated statistics"
class BestEffortError(RuntimeError):
"""A row/statistic could not be fetched (e.g. rate limited).
Raised by :func:`_request` once every retry is exhausted. The refresh is
best-effort: callers catch this so an unreachable API degrades the tracker
(keep the old value / omit a stat) instead of failing the whole job. The
only intentional non-zero exit is the post-Oct-1 retirement.
"""
def _log(message: str) -> None:
"""Emit a progress line to stderr, flushed so Actions shows it live."""
print(message, file=sys.stderr, flush=True)
def _headers() -> dict[str, str]:
headers = {
"Accept": "application/vnd.github+json",
"X-GitHub-Api-Version": "2022-11-28",
"User-Agent": "hacktoberfest-prep-bot",
}
if TOKEN:
headers["Authorization"] = f"Bearer {TOKEN}"
return headers
async def _request(
client: httpx2.AsyncClient,
sem: asyncio.Semaphore,
url: str,
params: dict | None = None,
) -> tuple[dict | list, dict]:
"""GET ``url`` and return ``(json_body, headers)``, retrying on 403/rate limit.
The semaphore bounds how many of these run concurrently.
"""
async with sem:
for attempt in range(4):
resp = await client.get(url, params=params)
if resp.is_success:
return resp.json(), dict(resp.headers)
# Both primary ("remaining == 0") and secondary/abuse rate limits
# come back as 403/429. Primary limits advertise a reset epoch;
# secondary limits instead send a ``Retry-After`` (seconds) and may
# still report a non-zero remaining, so honour either signal.
if resp.status_code in (403, 429):
remaining = resp.headers.get("X-RateLimit-Remaining")
retry_after = resp.headers.get("Retry-After")
if retry_after is not None:
wait = int(retry_after) + 1
elif remaining == "0":
reset = int(resp.headers.get("X-RateLimit-Reset", "0"))
wait = max(1, reset - int(time.time())) + 1
else:
wait = 0
if wait and attempt < 3:
_log(f"Rate limited on {url}; sleeping {min(wait, 90)}s")
await asyncio.sleep(min(wait, 90))
continue
if resp.status_code >= 500 and attempt < 3:
await asyncio.sleep(2 * (attempt + 1))
continue
resp.raise_for_status()
# Exhausted every retry (typically the shared per-repo budget ran dry).
# Signal the callers to degrade rather than crash the whole run.
msg = f"giving up on {url} after repeated rate limiting"
raise BestEffortError(msg)
async def _search_count(
client: httpx2.AsyncClient, sem: asyncio.Semaphore, query: str
) -> int:
body, _ = await _request(
client, sem, f"{API}/search/issues", {"q": query, "per_page": 1}
)
return int(body.get("total_count", 0)) # type: ignore[union-attr]
# GitHub's ``/search/issues`` ``is:issue`` / ``is:pr`` qualifiers have proven
# unreliable — some runs return the pull-request pool for *both* queries, so the
# "Open issues" line reported the PR count. The counts below avoid search
# entirely: the repository endpoint's ``open_issues_count`` is issues + PRs, and
# the open-PR total comes from the ``Link: rel="last"`` page number of the pulls
# listing. Open issues is then their difference — deterministic and self-checking.
_LAST_PAGE_RE = re.compile(r'[?&]page=(\d+)>;\s*rel="last"')
async def _last_page_count(
client: httpx2.AsyncClient,
sem: asyncio.Semaphore,
url: str,
params: dict | None = None,
) -> int:
"""Total items in a paginated listing, read from its ``Link`` header.
Requests one item per page so the ``rel="last"`` page number *is* the count.
Falls back to the length of the single returned page when there is no
``Link`` header (0 or 1 items).
"""
query = {"per_page": 1, **(params or {})}
body, headers = await _request(client, sem, url, query)
link = headers.get("link") or headers.get("Link") or ""
if match := _LAST_PAGE_RE.search(link):
return int(match.group(1))
return len(body) if isinstance(body, list) else 0
async def open_issue_and_pr_counts(
client: httpx2.AsyncClient, sem: asyncio.Semaphore
) -> tuple[int, int]:
"""Return ``(open_issues, open_prs)`` without the flaky search qualifiers.
``open_issues_count`` from the repo endpoint counts issues *and* pull
requests; subtracting the pull-request total leaves genuine issues.
"""
repo, open_prs = await asyncio.gather(
_request(client, sem, f"{API}/repos/{REPO}"),
_last_page_count(client, sem, f"{API}/repos/{REPO}/pulls", {"state": "open"}),
)
repo_body, _ = repo
total = int(repo_body.get("open_issues_count", 0)) # type: ignore[union-attr]
open_issues = max(total - open_prs, 0)
return open_issues, open_prs
async def pr_state(
client: httpx2.AsyncClient, sem: asyncio.Semaphore, number: int
) -> str | None:
"""Return ``"merged"`` / ``"closed"`` for a resolved row, else ``None``.
Uses the unified ``/issues/{number}`` endpoint, which resolves for both
pull requests *and* issues. The tracker's "Open issues" section lists
issue numbers, and ``/pulls/{issue}`` 404s on those, so querying
``/issues`` keeps a single issue row from crashing the whole run. A row is
"merged" only when it is a PR whose ``pull_request.merged_at`` is set; any
other closed row is "closed".
"""
body, _ = await _request(client, sem, f"{API}/repos/{REPO}/issues/{number}")
if body.get("state") == "open": # type: ignore[union-attr]
return None
pr = body.get("pull_request") # type: ignore[union-attr]
if pr and pr.get("merged_at"):
return "merged"
return "closed"
async def top_awaiting_directories(
client: httpx2.AsyncClient,
sem: asyncio.Semaphore,
limit: int = 3,
max_prs: int = 120,
) -> list[tuple[str, int]]:
"""Count open ``awaiting reviews`` PRs by the top-level directory they touch."""
query = f'repo:{REPO} is:pr is:open label:"{AWAITING_LABEL}"'
numbers: list[int] = []
page = 1
while len(numbers) < max_prs:
body, _ = await _request(
client,
sem,
f"{API}/search/issues",
{"q": query, "per_page": 100, "page": page},
)
items = body.get("items", []) # type: ignore[union-attr]
if not items:
break
numbers.extend(item["number"] for item in items)
if len(items) < 100:
break
page += 1
numbers = numbers[:max_prs]
total = len(numbers)
_log(f"Scanning changed files for {total} '{AWAITING_LABEL}' PR(s)...")
done = 0
async def dirs_for(number: int) -> set[str]:
nonlocal done
files, _ = await _request(
client, sem, f"{API}/repos/{REPO}/pulls/{number}/files", {"per_page": 100}
)
dirs = set()
for changed in files: # type: ignore[union-attr]
parts = changed["filename"].split("/")
if len(parts) > 1 and not parts[0].startswith("."):
dirs.add(parts[0])
done += 1
if done % 25 == 0 or done == total:
_log(f" ...scanned {done}/{total} PR(s)")
return dirs
results = await asyncio.gather(
*(dirs_for(n) for n in numbers), return_exceptions=True
)
counts: dict[str, int] = {}
skipped = 0
for dirs in results:
if isinstance(dirs, BestEffortError):
skipped += 1
continue
if isinstance(dirs, BaseException):
raise dirs
for directory in dirs:
counts[directory] = counts.get(directory, 0) + 1
if skipped:
_log(f" ...{skipped} PR(s) skipped (API unavailable); ranking partial.")
ranked = sorted(counts.items(), key=lambda kv: (-kv[1], kv[0]))
return ranked[:limit]
async def refresh_checkboxes(
client: httpx2.AsyncClient, sem: asyncio.Semaphore, lines: list[str]
) -> tuple[list[str], int]:
"""Tick rows whose PR is now merged/closed. Returns (new_lines, n_updated)."""
# Collect the PR numbers of every still-unchecked tracked row, then look
# their states up concurrently.
pending = [
int(match.group("pr"))
for line in lines
if (match := ROW_RE.match(line)) and match.group("mark") != "x"
]
if pending:
_log(f"Checking {len(pending)} open tracker row(s) for resolution...")
# ``return_exceptions`` keeps one rate-limited row from cancelling the rest:
# a row we couldn't resolve is simply left unchanged (treated as ``None``).
resolved = await asyncio.gather(
*(pr_state(client, sem, n) for n in pending), return_exceptions=True
)
states: dict[int, str | None] = {}
unresolved = 0
for number, result in zip(pending, resolved):
if isinstance(result, BestEffortError):
unresolved += 1
states[number] = None
elif isinstance(result, BaseException):
raise result
else:
states[number] = result
if unresolved:
_log(f" ...{unresolved} row(s) left unchanged (API unavailable).")
updated = 0
out: list[str] = []
for line in lines:
match = ROW_RE.match(line)
if not match or match.group("mark") == "x":
out.append(line)
continue
state = states.get(int(match.group("pr")))
if state is None:
out.append(line)
continue
out.append(f"{match.group('idx')}. [x] #{match.group('pr')} {state}")
updated += 1
return out, updated
async def build_stats_block(client: httpx2.AsyncClient, sem: asyncio.Semaphore) -> str:
_log("Collecting open issue/PR counts...")
awaiting_query = f'repo:{REPO} is:pr is:open label:"{AWAITING_LABEL}"'
async def _count_or_none(query: str) -> int | None:
try:
return await _search_count(client, sem, query)
except BestEffortError:
return None
async def _counts_or_none() -> tuple[int | None, int | None]:
try:
return await open_issue_and_pr_counts(client, sem)
except BestEffortError:
return None, None
(open_issues, open_prs), awaiting = await asyncio.gather(
_counts_or_none(),
_count_or_none(awaiting_query),
)
today = dt.datetime.now(dt.UTC).date()
today_iso = today.isoformat()
def _fmt(value: int | None) -> str:
return str(value) if value is not None else "unavailable (rate limited)"
# Hacktoberfest countdown: how much daily throughput clears the backlog by
# 2026-10-01. ``days_left`` is inclusive of today so the target date itself
# is not counted as a working day (avoids a divide-by-zero on the last day).
days_left = max((HACKTOBERFEST_START - today).days, 0)
def _per_day(count: int | None) -> str:
if count is None:
return "unavailable (rate limited)"
if days_left <= 0:
return "Hacktoberfest has started"
# Round up: finishing a day early beats finishing a day late.
return f"{-(-count // days_left)} per day (over {days_left} days)"
lines = [
STATS_HEADER,
"",
(
f"_Generated automatically by "
f"`scripts/hacktoberfest_prep_update.py` on {today_iso} (UTC)._"
),
"",
f"- **Open issues:** {_fmt(open_issues)}",
f"- **Open pull requests:** {_fmt(open_prs)}",
f"- **Open PRs labelled `{AWAITING_LABEL}`:** {_fmt(awaiting)}",
(
f"- **Days until Hacktoberfest ({HACKTOBERFEST_START.isoformat()}):** "
f"{days_left}"
),
f"- **Issues to close per day to clear the backlog:** {_per_day(open_issues)}",
(
"- **Pull requests to merge or close per day to clear the backlog:** "
f"{_per_day(open_prs)}"
),
"",
(
"**Top three directories to work on** (most open pull requests "
f"labelled `{AWAITING_LABEL}`):"
),
"",
]
try:
top_dirs = await top_awaiting_directories(client, sem)
except BestEffortError:
top_dirs = []
lines.append("_Directory ranking unavailable this run (rate limited)._")
else:
if top_dirs:
for rank, (directory, count) in enumerate(top_dirs, start=1):
plural = "PR" if count == 1 else "PRs"
lines.append(
f"{rank}. `{directory}/` — {count} awaiting-reviews {plural}"
)
else:
lines.append("_No open `awaiting reviews` pull requests found._")
lines.append("")
return "\n".join(lines)
def splice_stats(text: str, stats_block: str) -> str:
idx = text.find(STATS_HEADER)
head = text[:idx].rstrip("\n") if idx != -1 else text.rstrip("\n")
return f"{head}\n\n{stats_block}\n"
async def compute(lines: list[str]) -> tuple[list[str], int, str]:
"""Do all the network work: resolve tracker rows and build the stats block."""
async with httpx2.AsyncClient(headers=_headers(), timeout=30) as client:
sem = asyncio.Semaphore(CONCURRENCY)
lines, n_updated = await refresh_checkboxes(client, sem, lines)
stats_block = await build_stats_block(client, sem)
return lines, n_updated, stats_block
def main() -> int:
started = time.monotonic()
with open(TRACKER, encoding="utf-8") as handle:
text = handle.read()
body_before_stats = text.split(STATS_HEADER, 1)[0]
lines = body_before_stats.splitlines()
lines, n_updated, stats_block = asyncio.run(compute(lines))
body = "\n".join(lines)
new_text = splice_stats(body, stats_block)
with open(TRACKER, "w", encoding="utf-8") as handle:
handle.write(new_text)
elapsed = time.monotonic() - started
print(f"Checked off {n_updated} newly-resolved pull request(s) in {elapsed:.1f}s.")
today = dt.datetime.now(dt.UTC).date()
if today >= HACKTOBERFEST_START:
_log(
f"Hacktoberfest 2026 has begun ({today} >= {HACKTOBERFEST_START}); "
"the prep window is over — failing on purpose so this job is retired."
)
return 1
return 0
if __name__ == "__main__":
raise SystemExit(main())