From fcac01955c981a0d047315f7d2e381872d952a01 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 16 Aug 2026 01:19:47 +0800 Subject: [PATCH] feat: add pepy.tech spot-check helper for download audits Maintainer registered a free pepy API key (PEPY_TECH_API_KEY in the gitignored repo-root .env). fetch_pepy_downloads.py sums the most recent 30 days from the v2 per-day data, throttled to the free tier's 5 requests/minute, and prints TSV without touching the single-source cache file. The audit-the-list skill's downloads bullet is rewritten in the same commit because the ClickPy-default change made its old --dry-run-first instruction fail; it now documents all three sources (ClickPy, BigQuery, pepy/pypistats) and the never-mix-mirror-counting rule. Co-Authored-By: Claude --- .claude/skills/audit-the-list/SKILL.md | 2 +- website/fetch_pepy_downloads.py | 68 ++++++++++++++++++++++++++ 2 files changed, 69 insertions(+), 1 deletion(-) create mode 100644 website/fetch_pepy_downloads.py diff --git a/.claude/skills/audit-the-list/SKILL.md b/.claude/skills/audit-the-list/SKILL.md index a66851bb..c87cd05f 100644 --- a/.claude/skills/audit-the-list/SKILL.md +++ b/.claude/skills/audit-the-list/SKILL.md @@ -16,7 +16,7 @@ Resolve the scope from the arguments. Named sections mean exactly those, whether Fetch live evidence for every entry in scope before judging anything (CLAUDE.md verification rule): -- **Downloads/month**: `cd website && UV_PYTHON=3.13 uv run python fetch_pypi_downloads.py --names-file --dry-run` first, then without `--dry-run`. The script's docstring carries the BigQuery cost constraints — read it before batching. `https://pypistats.org/api/packages/{name}/recent` for a handful of spot-checks, paced 8s or slower. +- **Downloads/month**: `cd website && UV_PYTHON=3.13 uv run python fetch_pypi_downloads.py --names-file ` — default source is ClickPy (free, keyless, one batched query covers the full README). `--bigquery` switches to the canonical cross-check on the maintainer's own GCP account; there, run `--dry-run` first and read the docstring's cost constraints before batching. Spot-checks: `uv run python fetch_pepy_downloads.py ...` (needs `PEPY_TECH_API_KEY` in repo-root `.env`, throttled to 5 requests/minute) or `https://pypistats.org/api/packages/{name}/recent` paced 8s or slower. pypistats excludes mirror/CI traffic; ClickPy, BigQuery, and pepy include it — never mix sources within one comparison. - **Repo state**: archived flag, last push, created date, stars, description — `gh api repos/{owner}/{repo}`, GitLab API for GitLab-hosted projects. - **PyPI metadata** (`https://pypi.org/pypi/{name}/json`) wherever a name might not be the canonical package — ownership collisions and wrong display names surface here. diff --git a/website/fetch_pepy_downloads.py b/website/fetch_pepy_downloads.py new file mode 100644 index 00000000..cd763f6c --- /dev/null +++ b/website/fetch_pepy_downloads.py @@ -0,0 +1,68 @@ +#!/usr/bin/env python3 +"""Spot-check last-30-day PyPI download counts via the pepy.tech API. + +Cross-check for a handful of packages during audits — not for full-README +sweeps: the free API key is throttled to 5 requests/minute (10 burst), so +this script sleeps 12s between requests and 530 names would take ~2 hours +(use fetch_pypi_downloads.py for bulk). Reads PEPY_TECH_API_KEY from the +environment, falling back to the repo-root .env. The v2 endpoint returns +~90 days of per-day per-version counts; this script sums the most recent +30 days present in the response across all versions. pepy counts include +mirror/CI traffic (CI filtering is a paid pepy feature), matching the +ClickPy/BigQuery figures; pypistats.org excludes mirrors, so never mix +the two in one comparison. Results print to stdout as TSV and are not +cached — data/pypi_downloads.tsv stays single-source. + +Usage: python fetch_pepy_downloads.py NAME [NAME ...] +""" + +import os +import sys +import time +from pathlib import Path + +import httpx +from fetch_pypi_downloads import normalize + +ENV_FILE = Path(__file__).parent.parent / ".env" +PEPY_URL = "https://api.pepy.tech/api/v2/projects/{name}" +SECONDS_BETWEEN_REQUESTS = 12 + + +def load_api_key() -> str: + key = os.environ.get("PEPY_TECH_API_KEY", "") + if not key and ENV_FILE.exists(): + for line in ENV_FILE.read_text().splitlines(): + name, sep, value = line.partition("=") + if sep and name.strip() == "PEPY_TECH_API_KEY": + key = value.strip() + if not key: + print("Error: PEPY_TECH_API_KEY not set (environment or repo-root .env).", file=sys.stderr) + sys.exit(1) + return key + + +def last_30_day_total(downloads_per_day: dict[str, dict[str, int]]) -> int: + recent_days = sorted(downloads_per_day)[-30:] + return sum(sum(downloads_per_day[day].values()) for day in recent_days) + + +def main() -> None: + if len(sys.argv) < 2: + print("Usage: python fetch_pepy_downloads.py NAME [NAME ...]", file=sys.stderr) + sys.exit(1) + names = [normalize(name) for name in sys.argv[1:]] + with httpx.Client(headers={"X-API-Key": load_api_key()}, timeout=30) as client: + for i, name in enumerate(names): + if i: + time.sleep(SECONDS_BETWEEN_REQUESTS) + resp = client.get(PEPY_URL.format(name=name)) + if resp.status_code == 404: + print(f"{name}\tNOT_FOUND") + continue + resp.raise_for_status() + print(f"{name}\t{last_30_day_total(resp.json()['downloads'])}") + + +if __name__ == "__main__": + main()