diff --git a/.claude/skills/audit-the-list/SKILL.md b/.claude/skills/audit-the-list/SKILL.md index 57285d1d..961b3c7f 100644 --- a/.claude/skills/audit-the-list/SKILL.md +++ b/.claude/skills/audit-the-list/SKILL.md @@ -16,7 +16,7 @@ Resolve the scope from the arguments. Named sections mean exactly those, whether Fetch live evidence for every entry in scope before judging anything (CLAUDE.md verification rule): -- **Downloads/month**: `cd website && UV_PYTHON=3.13 uv run python fetch_pypi_downloads_via_clickpy.py` — free keyless ClickPy sweep of the full README, sole writer of `data/pypi_downloads.tsv` (rewritten from scratch each run). Cross-checks print to stdout, take explicit names, and never touch the cache: `fetch_pypi_downloads_via_bigquery.py ...` (canonical source, maintainer's own GCP account, `--dry-run` first — the docstring carries the cost constraints; full-README sweeps exceed the free tier, keep name lists small), `fetch_pypi_downloads_via_pepy.py ...` (needs `PEPY_TECH_API_KEY` in repo-root `.env`, throttled to 5 requests/minute), or `https://pypistats.org/api/packages/{name}/recent` paced 8s or slower. pypistats excludes mirror/CI traffic; ClickPy, BigQuery, and pepy include it — never mix sources within one comparison. +- **Downloads/month**: `cd website && UV_PYTHON=3.13 uv run python fetch_pypi_downloads_via_clickpy.py` — free keyless ClickPy sweep of the full README, sole writer of `data/pypi_downloads.tsv` (rewritten from scratch each run; header row, every row stamped with its `fetched_at` date). A cache whose `fetched_at` is within the last 7 days is current enough for verdicts — skip the sweep; older than that, re-run it (costs ~1s). Cross-checks print to stdout, take explicit names, and never touch the cache: `fetch_pypi_downloads_via_bigquery.py ...` (canonical source, maintainer's own GCP account, `--dry-run` first — the docstring carries the cost constraints; full-README sweeps exceed the free tier, keep name lists small), `fetch_pypi_downloads_via_pepy.py ...` (needs `PEPY_TECH_API_KEY` in repo-root `.env`, throttled to 5 requests/minute), or `https://pypistats.org/api/packages/{name}/recent` paced 8s or slower. pypistats excludes mirror/CI traffic; ClickPy, BigQuery, and pepy include it — never mix sources within one comparison. - **Repo state**: archived flag, last push, created date, stars, description — `gh api repos/{owner}/{repo}`, GitLab API for GitLab-hosted projects. - **PyPI metadata** (`https://pypi.org/pypi/{name}/json`) wherever a name might not be the canonical package — ownership collisions and wrong display names surface here. diff --git a/website/fetch_pypi_downloads_via_clickpy.py b/website/fetch_pypi_downloads_via_clickpy.py index e88815ac..e384b365 100644 --- a/website/fetch_pypi_downloads_via_clickpy.py +++ b/website/fetch_pypi_downloads_via_clickpy.py @@ -12,7 +12,10 @@ excludes mirrors by default. This script is the sole writer of data/pypi_downloads.tsv and rewrites it from scratch each run, so entries removed from README.md drop out -naturally; names with no PyPI rows are written as NOT_FOUND. Cross-checks +naturally; names with no PyPI rows are written as NOT_FOUND. The file +starts with a header row (name, downloads, fetched_at) and every row +carries the sweep date: a cache fetched within the last 7 days is current +enough for audit verdicts, so only re-run when older. Cross-checks against other sources (fetch_pypi_downloads_via_bigquery.py, fetch_pypi_downloads_via_pepy.py) print to stdout and never touch the cache. @@ -21,6 +24,7 @@ Usage: python fetch_pypi_downloads_via_clickpy.py """ import re +from datetime import date from pathlib import Path import httpx @@ -67,7 +71,9 @@ def main() -> None: names = collect_names(README_PATH.read_text()) print(f"Querying {len(names)} package names...") counts = fetch_clickpy(names) - OUT_FILE.write_text("\n".join(f"{name}\t{counts.get(name, 'NOT_FOUND')}" for name in names) + "\n") + fetched_at = date.today().isoformat() + rows = "\n".join(f"{name}\t{counts.get(name, 'NOT_FOUND')}\t{fetched_at}" for name in names) + OUT_FILE.write_text(f"name\tdownloads\tfetched_at\n{rows}\n") print(f"Done. {len(counts)}/{len(names)} names found on PyPI. Cached to {OUT_FILE}")