fix: resolve PyPI download counts through curated package overrides

A pypi.org identity sweep of all 438 cached rows (project_urls/home_page
vs entry GitHub URL) found download counts were looked up by README
display name, so entries whose name differs from the canonical package
silently measured squatters or dead predecessors: pytorch measured a
squatter (169,737/mo vs torch's 94M), jinja measured Jinja1 (3,168 vs
jinja2's 736M), django-rest-framework a dead alias package (real:
djangorestframework), django-rules an abandoned fork (real: rules),
strawberry an unrelated bookmarking service (real: strawberry-graphql),
devpi a deprecated metapackage (mapped to devpi-server).

New curated website/data/pypi_name_overrides.json maps normalized
README name to the real package, or null for projects not
pip-installable whose name is squatted or a relic (cpython, pyenv,
renpy, python-patterns, winpython); also maps mem0 to mem0ai, fasthtml
to python-fasthtml, and playwright-python to playwright.

All three fetch scripts resolve names through it; the clickpy TSV
cache gains a package column recording what each row actually
measured. .gitignore switches website/data/ to website/data/* with a
negation so the curated overrides file is tracked while caches stay
ignored.

Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
Vinta Chen
2026-08-16 17:51:11 +08:00
co-authored by Claude
parent e9329d4d1e
commit b440e64cd0
5 changed files with 72 additions and 20 deletions
+2 -1
View File
@@ -11,7 +11,8 @@ __pycache__/
# website
website/output/
website/data/
website/data/*
!website/data/pypi_name_overrides.json
# agents
.playwright-cli/
+16
View File
@@ -0,0 +1,16 @@
{
"cpython": null,
"devpi": "devpi-server",
"django-rest-framework": "djangorestframework",
"django-rules": "rules",
"fasthtml": "python-fasthtml",
"jinja": "jinja2",
"mem0": "mem0ai",
"playwright-python": "playwright",
"pyenv": null,
"python-patterns": null,
"pytorch": "torch",
"renpy": null,
"strawberry": "strawberry-graphql",
"winpython": null
}
+13 -3
View File
@@ -3,7 +3,8 @@
Queries the canonical source ClickPy mirrors —
`bigquery-public-data.pypi.file_downloads` via the `bq` CLI — and prints
name<TAB>count TSV to stdout. Maintainer-local (needs a personal GCP
name<TAB>count TSV to stdout. Names resolve through
data/pypi_name_overrides.json, matching the cache sweep. Maintainer-local (needs a personal GCP
account) and print-only: data/pypi_downloads.tsv is written solely by
fetch_pypi_downloads_via_clickpy.py.
@@ -27,7 +28,7 @@ import subprocess
import sys
from json import loads
from fetch_pypi_downloads_via_clickpy import normalize
from fetch_pypi_downloads_via_clickpy import resolve
MAX_BYTES_BILLED = 400_000_000_000
@@ -59,7 +60,16 @@ def fetch_bigquery(names: list[str], dry_run: bool) -> dict[str, int]:
def main() -> None:
dry_run = "--dry-run" in sys.argv
names = sorted({normalize(arg) for arg in sys.argv[1:] if arg != "--dry-run"})
names = set()
for arg in sys.argv[1:]:
if arg == "--dry-run":
continue
pkg = resolve(arg)
if pkg is None:
print(f"{arg}: not pip-installable per pypi_name_overrides.json, skipping", file=sys.stderr)
else:
names.add(pkg)
names = sorted(names)
if not names:
print("Usage: python fetch_pypi_downloads_via_bigquery.py [--dry-run] NAME [NAME ...]", file=sys.stderr)
sys.exit(1)
+31 -13
View File
@@ -13,20 +13,24 @@ excludes mirrors by default.
This script is the sole writer of data/pypi_downloads.tsv and rewrites it
from scratch each run, so entries removed from README.md drop out
naturally; names with no PyPI rows are written as NOT_FOUND. Counts are
looked up by README display name: when the display name differs from the
canonical PyPI package, the row silently measures the wrong package (a
squatter or a dead predecessor) — verify identity at
pypi.org/pypi/{name}/json before citing a count. The file
starts with a header row (name, downloads, fetched_at) and every row
carries the sweep date: a cache fetched within the last 7 days is current
enough for audit verdicts, so only re-run when older. Cross-checks
against other sources (fetch_pypi_downloads_via_bigquery.py,
looked up by README display name; when the display name differs from the
canonical PyPI package (a squatter or a dead predecessor would be
measured otherwise), add the mapping to the curated
data/pypi_name_overrides.json — normalized README name to the real
package, or null for projects that are not pip-installable so their
row is never queried. The file starts with a header row (name, package,
downloads, fetched_at) — package is the PyPI package the row actually
measured ("-" for null overrides) — and every row carries the sweep
date: a cache fetched within the last 7 days is current enough for
audit verdicts, so only re-run when older. Cross-checks against other
sources (fetch_pypi_downloads_via_bigquery.py,
fetch_pypi_downloads_via_pepy.py) print to stdout and never touch the
cache.
Usage: python fetch_pypi_downloads_via_clickpy.py
"""
import json
import re
from datetime import date
from pathlib import Path
@@ -36,6 +40,7 @@ from readme_parser import parse_readme
DATA_DIR = Path(__file__).parent / "data"
OUT_FILE = DATA_DIR / "pypi_downloads.tsv"
OVERRIDES_FILE = DATA_DIR / "pypi_name_overrides.json"
README_PATH = Path(__file__).parent.parent / "README.md"
CLICKPY_URL = "https://sql-clickhouse.clickhouse.com/?user=demo"
@@ -47,6 +52,16 @@ def normalize(name: str) -> str:
return re.sub(r"[-_.]+", "-", name.lower())
def load_overrides() -> dict[str, str | None]:
return json.loads(OVERRIDES_FILE.read_text())
def resolve(name: str) -> str | None:
"""Map a README display name to the PyPI package to measure. None = not pip-installable."""
normalized = normalize(name)
return load_overrides().get(normalized, normalized)
def collect_names(readme_text: str) -> list[str]:
names = set()
for group in parse_readme(readme_text):
@@ -73,12 +88,15 @@ def fetch_clickpy(names: list[str]) -> dict[str, int]:
def main() -> None:
names = collect_names(README_PATH.read_text())
print(f"Querying {len(names)} package names...")
counts = fetch_clickpy(names)
overrides = load_overrides()
packages = {name: overrides.get(name, name) for name in names}
query_names = sorted({pkg for pkg in packages.values() if pkg})
print(f"Querying {len(query_names)} package names...")
counts = fetch_clickpy(query_names)
fetched_at = date.today().isoformat()
rows = "\n".join(f"{name}\t{counts.get(name, 'NOT_FOUND')}\t{fetched_at}" for name in names)
OUT_FILE.write_text(f"name\tdownloads\tfetched_at\n{rows}\n")
print(f"Done. {len(counts)}/{len(names)} names found on PyPI. Cached to {OUT_FILE}")
rows = "\n".join(f"{name}\t{pkg or '-'}\t{counts.get(pkg, 'NOT_FOUND') if pkg else 'NOT_FOUND'}\t{fetched_at}" for name, pkg in packages.items())
OUT_FILE.write_text(f"name\tpackage\tdownloads\tfetched_at\n{rows}\n")
print(f"Done. {len(counts)}/{len(query_names)} names found on PyPI. Cached to {OUT_FILE}")
if __name__ == "__main__":
+10 -3
View File
@@ -7,7 +7,8 @@ this script sleeps 12s between requests and 530 names would take ~2 hours
(use fetch_pypi_downloads_via_clickpy.py for bulk). Reads PEPY_TECH_API_KEY from the
environment, falling back to the repo-root .env. The v2 endpoint returns
~90 days of per-day per-version counts; this script sums the most recent
30 days present in the response across all versions. pepy counts include
30 days present in the response across all versions. Names resolve
through data/pypi_name_overrides.json, matching the cache sweep. pepy counts include
mirror/CI traffic (CI filtering is a paid pepy feature), matching the
ClickPy/BigQuery figures; pypistats.org excludes mirrors, so never mix
the two in one comparison. Results print to stdout as TSV and are not
@@ -22,7 +23,7 @@ import time
from pathlib import Path
import httpx
from fetch_pypi_downloads_via_clickpy import normalize
from fetch_pypi_downloads_via_clickpy import resolve
ENV_FILE = Path(__file__).parent.parent / ".env"
PEPY_URL = "https://api.pepy.tech/api/v2/projects/{name}"
@@ -51,7 +52,13 @@ def main() -> None:
if len(sys.argv) < 2:
print("Usage: python fetch_pypi_downloads_via_pepy.py NAME [NAME ...]", file=sys.stderr)
sys.exit(1)
names = [normalize(name) for name in sys.argv[1:]]
names = []
for arg in sys.argv[1:]:
pkg = resolve(arg)
if pkg is None:
print(f"{arg}: not pip-installable per pypi_name_overrides.json, skipping", file=sys.stderr)
else:
names.append(pkg)
with httpx.Client(headers={"X-API-Key": load_api_key()}, timeout=30) as client:
for i, name in enumerate(names):
if i: