From b440e64cd0a5cd75ebff0029b250c43b4137d70f Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 16 Aug 2026 17:51:11 +0800 Subject: [PATCH] fix: resolve PyPI download counts through curated package overrides A pypi.org identity sweep of all 438 cached rows (project_urls/home_page vs entry GitHub URL) found download counts were looked up by README display name, so entries whose name differs from the canonical package silently measured squatters or dead predecessors: pytorch measured a squatter (169,737/mo vs torch's 94M), jinja measured Jinja1 (3,168 vs jinja2's 736M), django-rest-framework a dead alias package (real: djangorestframework), django-rules an abandoned fork (real: rules), strawberry an unrelated bookmarking service (real: strawberry-graphql), devpi a deprecated metapackage (mapped to devpi-server). New curated website/data/pypi_name_overrides.json maps normalized README name to the real package, or null for projects not pip-installable whose name is squatted or a relic (cpython, pyenv, renpy, python-patterns, winpython); also maps mem0 to mem0ai, fasthtml to python-fasthtml, and playwright-python to playwright. All three fetch scripts resolve names through it; the clickpy TSV cache gains a package column recording what each row actually measured. .gitignore switches website/data/ to website/data/* with a negation so the curated overrides file is tracked while caches stay ignored. Co-Authored-By: Claude --- .gitignore | 3 +- website/data/pypi_name_overrides.json | 16 +++++++ website/fetch_pypi_downloads_via_bigquery.py | 16 +++++-- website/fetch_pypi_downloads_via_clickpy.py | 44 ++++++++++++++------ website/fetch_pypi_downloads_via_pepy.py | 13 ++++-- 5 files changed, 72 insertions(+), 20 deletions(-) create mode 100644 website/data/pypi_name_overrides.json diff --git a/.gitignore b/.gitignore index 1f2c6f1e..ba307609 100644 --- a/.gitignore +++ b/.gitignore @@ -11,7 +11,8 @@ __pycache__/ # website website/output/ -website/data/ +website/data/* +!website/data/pypi_name_overrides.json # agents .playwright-cli/ diff --git a/website/data/pypi_name_overrides.json b/website/data/pypi_name_overrides.json new file mode 100644 index 00000000..47a400a0 --- /dev/null +++ b/website/data/pypi_name_overrides.json @@ -0,0 +1,16 @@ +{ + "cpython": null, + "devpi": "devpi-server", + "django-rest-framework": "djangorestframework", + "django-rules": "rules", + "fasthtml": "python-fasthtml", + "jinja": "jinja2", + "mem0": "mem0ai", + "playwright-python": "playwright", + "pyenv": null, + "python-patterns": null, + "pytorch": "torch", + "renpy": null, + "strawberry": "strawberry-graphql", + "winpython": null +} diff --git a/website/fetch_pypi_downloads_via_bigquery.py b/website/fetch_pypi_downloads_via_bigquery.py index 21f69479..e8b5bbfc 100644 --- a/website/fetch_pypi_downloads_via_bigquery.py +++ b/website/fetch_pypi_downloads_via_bigquery.py @@ -3,7 +3,8 @@ Queries the canonical source ClickPy mirrors — `bigquery-public-data.pypi.file_downloads` via the `bq` CLI — and prints -namecount TSV to stdout. Maintainer-local (needs a personal GCP +namecount TSV to stdout. Names resolve through +data/pypi_name_overrides.json, matching the cache sweep. Maintainer-local (needs a personal GCP account) and print-only: data/pypi_downloads.tsv is written solely by fetch_pypi_downloads_via_clickpy.py. @@ -27,7 +28,7 @@ import subprocess import sys from json import loads -from fetch_pypi_downloads_via_clickpy import normalize +from fetch_pypi_downloads_via_clickpy import resolve MAX_BYTES_BILLED = 400_000_000_000 @@ -59,7 +60,16 @@ def fetch_bigquery(names: list[str], dry_run: bool) -> dict[str, int]: def main() -> None: dry_run = "--dry-run" in sys.argv - names = sorted({normalize(arg) for arg in sys.argv[1:] if arg != "--dry-run"}) + names = set() + for arg in sys.argv[1:]: + if arg == "--dry-run": + continue + pkg = resolve(arg) + if pkg is None: + print(f"{arg}: not pip-installable per pypi_name_overrides.json, skipping", file=sys.stderr) + else: + names.add(pkg) + names = sorted(names) if not names: print("Usage: python fetch_pypi_downloads_via_bigquery.py [--dry-run] NAME [NAME ...]", file=sys.stderr) sys.exit(1) diff --git a/website/fetch_pypi_downloads_via_clickpy.py b/website/fetch_pypi_downloads_via_clickpy.py index 1e7c1a26..6934dc01 100644 --- a/website/fetch_pypi_downloads_via_clickpy.py +++ b/website/fetch_pypi_downloads_via_clickpy.py @@ -13,20 +13,24 @@ excludes mirrors by default. This script is the sole writer of data/pypi_downloads.tsv and rewrites it from scratch each run, so entries removed from README.md drop out naturally; names with no PyPI rows are written as NOT_FOUND. Counts are -looked up by README display name: when the display name differs from the -canonical PyPI package, the row silently measures the wrong package (a -squatter or a dead predecessor) — verify identity at -pypi.org/pypi/{name}/json before citing a count. The file -starts with a header row (name, downloads, fetched_at) and every row -carries the sweep date: a cache fetched within the last 7 days is current -enough for audit verdicts, so only re-run when older. Cross-checks -against other sources (fetch_pypi_downloads_via_bigquery.py, +looked up by README display name; when the display name differs from the +canonical PyPI package (a squatter or a dead predecessor would be +measured otherwise), add the mapping to the curated +data/pypi_name_overrides.json — normalized README name to the real +package, or null for projects that are not pip-installable so their +row is never queried. The file starts with a header row (name, package, +downloads, fetched_at) — package is the PyPI package the row actually +measured ("-" for null overrides) — and every row carries the sweep +date: a cache fetched within the last 7 days is current enough for +audit verdicts, so only re-run when older. Cross-checks against other +sources (fetch_pypi_downloads_via_bigquery.py, fetch_pypi_downloads_via_pepy.py) print to stdout and never touch the cache. Usage: python fetch_pypi_downloads_via_clickpy.py """ +import json import re from datetime import date from pathlib import Path @@ -36,6 +40,7 @@ from readme_parser import parse_readme DATA_DIR = Path(__file__).parent / "data" OUT_FILE = DATA_DIR / "pypi_downloads.tsv" +OVERRIDES_FILE = DATA_DIR / "pypi_name_overrides.json" README_PATH = Path(__file__).parent.parent / "README.md" CLICKPY_URL = "https://sql-clickhouse.clickhouse.com/?user=demo" @@ -47,6 +52,16 @@ def normalize(name: str) -> str: return re.sub(r"[-_.]+", "-", name.lower()) +def load_overrides() -> dict[str, str | None]: + return json.loads(OVERRIDES_FILE.read_text()) + + +def resolve(name: str) -> str | None: + """Map a README display name to the PyPI package to measure. None = not pip-installable.""" + normalized = normalize(name) + return load_overrides().get(normalized, normalized) + + def collect_names(readme_text: str) -> list[str]: names = set() for group in parse_readme(readme_text): @@ -73,12 +88,15 @@ def fetch_clickpy(names: list[str]) -> dict[str, int]: def main() -> None: names = collect_names(README_PATH.read_text()) - print(f"Querying {len(names)} package names...") - counts = fetch_clickpy(names) + overrides = load_overrides() + packages = {name: overrides.get(name, name) for name in names} + query_names = sorted({pkg for pkg in packages.values() if pkg}) + print(f"Querying {len(query_names)} package names...") + counts = fetch_clickpy(query_names) fetched_at = date.today().isoformat() - rows = "\n".join(f"{name}\t{counts.get(name, 'NOT_FOUND')}\t{fetched_at}" for name in names) - OUT_FILE.write_text(f"name\tdownloads\tfetched_at\n{rows}\n") - print(f"Done. {len(counts)}/{len(names)} names found on PyPI. Cached to {OUT_FILE}") + rows = "\n".join(f"{name}\t{pkg or '-'}\t{counts.get(pkg, 'NOT_FOUND') if pkg else 'NOT_FOUND'}\t{fetched_at}" for name, pkg in packages.items()) + OUT_FILE.write_text(f"name\tpackage\tdownloads\tfetched_at\n{rows}\n") + print(f"Done. {len(counts)}/{len(query_names)} names found on PyPI. Cached to {OUT_FILE}") if __name__ == "__main__": diff --git a/website/fetch_pypi_downloads_via_pepy.py b/website/fetch_pypi_downloads_via_pepy.py index 998f260f..55eb7593 100644 --- a/website/fetch_pypi_downloads_via_pepy.py +++ b/website/fetch_pypi_downloads_via_pepy.py @@ -7,7 +7,8 @@ this script sleeps 12s between requests and 530 names would take ~2 hours (use fetch_pypi_downloads_via_clickpy.py for bulk). Reads PEPY_TECH_API_KEY from the environment, falling back to the repo-root .env. The v2 endpoint returns ~90 days of per-day per-version counts; this script sums the most recent -30 days present in the response across all versions. pepy counts include +30 days present in the response across all versions. Names resolve +through data/pypi_name_overrides.json, matching the cache sweep. pepy counts include mirror/CI traffic (CI filtering is a paid pepy feature), matching the ClickPy/BigQuery figures; pypistats.org excludes mirrors, so never mix the two in one comparison. Results print to stdout as TSV and are not @@ -22,7 +23,7 @@ import time from pathlib import Path import httpx -from fetch_pypi_downloads_via_clickpy import normalize +from fetch_pypi_downloads_via_clickpy import resolve ENV_FILE = Path(__file__).parent.parent / ".env" PEPY_URL = "https://api.pepy.tech/api/v2/projects/{name}" @@ -51,7 +52,13 @@ def main() -> None: if len(sys.argv) < 2: print("Usage: python fetch_pypi_downloads_via_pepy.py NAME [NAME ...]", file=sys.stderr) sys.exit(1) - names = [normalize(name) for name in sys.argv[1:]] + names = [] + for arg in sys.argv[1:]: + pkg = resolve(arg) + if pkg is None: + print(f"{arg}: not pip-installable per pypi_name_overrides.json, skipping", file=sys.stderr) + else: + names.append(pkg) with httpx.Client(headers={"X-API-Key": load_api_key()}, timeout=30) as client: for i, name in enumerate(names): if i: