mirror of
https://github.com/vinta/awesome-python.git
synced 2026-10-02 08:23:10 +08:00
fix: resolve PyPI download counts through curated package overrides
A pypi.org identity sweep of all 438 cached rows (project_urls/home_page vs entry GitHub URL) found download counts were looked up by README display name, so entries whose name differs from the canonical package silently measured squatters or dead predecessors: pytorch measured a squatter (169,737/mo vs torch's 94M), jinja measured Jinja1 (3,168 vs jinja2's 736M), django-rest-framework a dead alias package (real: djangorestframework), django-rules an abandoned fork (real: rules), strawberry an unrelated bookmarking service (real: strawberry-graphql), devpi a deprecated metapackage (mapped to devpi-server). New curated website/data/pypi_name_overrides.json maps normalized README name to the real package, or null for projects not pip-installable whose name is squatted or a relic (cpython, pyenv, renpy, python-patterns, winpython); also maps mem0 to mem0ai, fasthtml to python-fasthtml, and playwright-python to playwright. All three fetch scripts resolve names through it; the clickpy TSV cache gains a package column recording what each row actually measured. .gitignore switches website/data/ to website/data/* with a negation so the curated overrides file is tracked while caches stay ignored. Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
+2
-1
@@ -11,7 +11,8 @@ __pycache__/
|
|||||||
|
|
||||||
# website
|
# website
|
||||||
website/output/
|
website/output/
|
||||||
website/data/
|
website/data/*
|
||||||
|
!website/data/pypi_name_overrides.json
|
||||||
|
|
||||||
# agents
|
# agents
|
||||||
.playwright-cli/
|
.playwright-cli/
|
||||||
|
|||||||
@@ -0,0 +1,16 @@
|
|||||||
|
{
|
||||||
|
"cpython": null,
|
||||||
|
"devpi": "devpi-server",
|
||||||
|
"django-rest-framework": "djangorestframework",
|
||||||
|
"django-rules": "rules",
|
||||||
|
"fasthtml": "python-fasthtml",
|
||||||
|
"jinja": "jinja2",
|
||||||
|
"mem0": "mem0ai",
|
||||||
|
"playwright-python": "playwright",
|
||||||
|
"pyenv": null,
|
||||||
|
"python-patterns": null,
|
||||||
|
"pytorch": "torch",
|
||||||
|
"renpy": null,
|
||||||
|
"strawberry": "strawberry-graphql",
|
||||||
|
"winpython": null
|
||||||
|
}
|
||||||
@@ -3,7 +3,8 @@
|
|||||||
|
|
||||||
Queries the canonical source ClickPy mirrors —
|
Queries the canonical source ClickPy mirrors —
|
||||||
`bigquery-public-data.pypi.file_downloads` via the `bq` CLI — and prints
|
`bigquery-public-data.pypi.file_downloads` via the `bq` CLI — and prints
|
||||||
name<TAB>count TSV to stdout. Maintainer-local (needs a personal GCP
|
name<TAB>count TSV to stdout. Names resolve through
|
||||||
|
data/pypi_name_overrides.json, matching the cache sweep. Maintainer-local (needs a personal GCP
|
||||||
account) and print-only: data/pypi_downloads.tsv is written solely by
|
account) and print-only: data/pypi_downloads.tsv is written solely by
|
||||||
fetch_pypi_downloads_via_clickpy.py.
|
fetch_pypi_downloads_via_clickpy.py.
|
||||||
|
|
||||||
@@ -27,7 +28,7 @@ import subprocess
|
|||||||
import sys
|
import sys
|
||||||
from json import loads
|
from json import loads
|
||||||
|
|
||||||
from fetch_pypi_downloads_via_clickpy import normalize
|
from fetch_pypi_downloads_via_clickpy import resolve
|
||||||
|
|
||||||
MAX_BYTES_BILLED = 400_000_000_000
|
MAX_BYTES_BILLED = 400_000_000_000
|
||||||
|
|
||||||
@@ -59,7 +60,16 @@ def fetch_bigquery(names: list[str], dry_run: bool) -> dict[str, int]:
|
|||||||
|
|
||||||
def main() -> None:
|
def main() -> None:
|
||||||
dry_run = "--dry-run" in sys.argv
|
dry_run = "--dry-run" in sys.argv
|
||||||
names = sorted({normalize(arg) for arg in sys.argv[1:] if arg != "--dry-run"})
|
names = set()
|
||||||
|
for arg in sys.argv[1:]:
|
||||||
|
if arg == "--dry-run":
|
||||||
|
continue
|
||||||
|
pkg = resolve(arg)
|
||||||
|
if pkg is None:
|
||||||
|
print(f"{arg}: not pip-installable per pypi_name_overrides.json, skipping", file=sys.stderr)
|
||||||
|
else:
|
||||||
|
names.add(pkg)
|
||||||
|
names = sorted(names)
|
||||||
if not names:
|
if not names:
|
||||||
print("Usage: python fetch_pypi_downloads_via_bigquery.py [--dry-run] NAME [NAME ...]", file=sys.stderr)
|
print("Usage: python fetch_pypi_downloads_via_bigquery.py [--dry-run] NAME [NAME ...]", file=sys.stderr)
|
||||||
sys.exit(1)
|
sys.exit(1)
|
||||||
|
|||||||
@@ -13,20 +13,24 @@ excludes mirrors by default.
|
|||||||
This script is the sole writer of data/pypi_downloads.tsv and rewrites it
|
This script is the sole writer of data/pypi_downloads.tsv and rewrites it
|
||||||
from scratch each run, so entries removed from README.md drop out
|
from scratch each run, so entries removed from README.md drop out
|
||||||
naturally; names with no PyPI rows are written as NOT_FOUND. Counts are
|
naturally; names with no PyPI rows are written as NOT_FOUND. Counts are
|
||||||
looked up by README display name: when the display name differs from the
|
looked up by README display name; when the display name differs from the
|
||||||
canonical PyPI package, the row silently measures the wrong package (a
|
canonical PyPI package (a squatter or a dead predecessor would be
|
||||||
squatter or a dead predecessor) — verify identity at
|
measured otherwise), add the mapping to the curated
|
||||||
pypi.org/pypi/{name}/json before citing a count. The file
|
data/pypi_name_overrides.json — normalized README name to the real
|
||||||
starts with a header row (name, downloads, fetched_at) and every row
|
package, or null for projects that are not pip-installable so their
|
||||||
carries the sweep date: a cache fetched within the last 7 days is current
|
row is never queried. The file starts with a header row (name, package,
|
||||||
enough for audit verdicts, so only re-run when older. Cross-checks
|
downloads, fetched_at) — package is the PyPI package the row actually
|
||||||
against other sources (fetch_pypi_downloads_via_bigquery.py,
|
measured ("-" for null overrides) — and every row carries the sweep
|
||||||
|
date: a cache fetched within the last 7 days is current enough for
|
||||||
|
audit verdicts, so only re-run when older. Cross-checks against other
|
||||||
|
sources (fetch_pypi_downloads_via_bigquery.py,
|
||||||
fetch_pypi_downloads_via_pepy.py) print to stdout and never touch the
|
fetch_pypi_downloads_via_pepy.py) print to stdout and never touch the
|
||||||
cache.
|
cache.
|
||||||
|
|
||||||
Usage: python fetch_pypi_downloads_via_clickpy.py
|
Usage: python fetch_pypi_downloads_via_clickpy.py
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
import json
|
||||||
import re
|
import re
|
||||||
from datetime import date
|
from datetime import date
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
@@ -36,6 +40,7 @@ from readme_parser import parse_readme
|
|||||||
|
|
||||||
DATA_DIR = Path(__file__).parent / "data"
|
DATA_DIR = Path(__file__).parent / "data"
|
||||||
OUT_FILE = DATA_DIR / "pypi_downloads.tsv"
|
OUT_FILE = DATA_DIR / "pypi_downloads.tsv"
|
||||||
|
OVERRIDES_FILE = DATA_DIR / "pypi_name_overrides.json"
|
||||||
README_PATH = Path(__file__).parent.parent / "README.md"
|
README_PATH = Path(__file__).parent.parent / "README.md"
|
||||||
CLICKPY_URL = "https://sql-clickhouse.clickhouse.com/?user=demo"
|
CLICKPY_URL = "https://sql-clickhouse.clickhouse.com/?user=demo"
|
||||||
|
|
||||||
@@ -47,6 +52,16 @@ def normalize(name: str) -> str:
|
|||||||
return re.sub(r"[-_.]+", "-", name.lower())
|
return re.sub(r"[-_.]+", "-", name.lower())
|
||||||
|
|
||||||
|
|
||||||
|
def load_overrides() -> dict[str, str | None]:
|
||||||
|
return json.loads(OVERRIDES_FILE.read_text())
|
||||||
|
|
||||||
|
|
||||||
|
def resolve(name: str) -> str | None:
|
||||||
|
"""Map a README display name to the PyPI package to measure. None = not pip-installable."""
|
||||||
|
normalized = normalize(name)
|
||||||
|
return load_overrides().get(normalized, normalized)
|
||||||
|
|
||||||
|
|
||||||
def collect_names(readme_text: str) -> list[str]:
|
def collect_names(readme_text: str) -> list[str]:
|
||||||
names = set()
|
names = set()
|
||||||
for group in parse_readme(readme_text):
|
for group in parse_readme(readme_text):
|
||||||
@@ -73,12 +88,15 @@ def fetch_clickpy(names: list[str]) -> dict[str, int]:
|
|||||||
|
|
||||||
def main() -> None:
|
def main() -> None:
|
||||||
names = collect_names(README_PATH.read_text())
|
names = collect_names(README_PATH.read_text())
|
||||||
print(f"Querying {len(names)} package names...")
|
overrides = load_overrides()
|
||||||
counts = fetch_clickpy(names)
|
packages = {name: overrides.get(name, name) for name in names}
|
||||||
|
query_names = sorted({pkg for pkg in packages.values() if pkg})
|
||||||
|
print(f"Querying {len(query_names)} package names...")
|
||||||
|
counts = fetch_clickpy(query_names)
|
||||||
fetched_at = date.today().isoformat()
|
fetched_at = date.today().isoformat()
|
||||||
rows = "\n".join(f"{name}\t{counts.get(name, 'NOT_FOUND')}\t{fetched_at}" for name in names)
|
rows = "\n".join(f"{name}\t{pkg or '-'}\t{counts.get(pkg, 'NOT_FOUND') if pkg else 'NOT_FOUND'}\t{fetched_at}" for name, pkg in packages.items())
|
||||||
OUT_FILE.write_text(f"name\tdownloads\tfetched_at\n{rows}\n")
|
OUT_FILE.write_text(f"name\tpackage\tdownloads\tfetched_at\n{rows}\n")
|
||||||
print(f"Done. {len(counts)}/{len(names)} names found on PyPI. Cached to {OUT_FILE}")
|
print(f"Done. {len(counts)}/{len(query_names)} names found on PyPI. Cached to {OUT_FILE}")
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|||||||
@@ -7,7 +7,8 @@ this script sleeps 12s between requests and 530 names would take ~2 hours
|
|||||||
(use fetch_pypi_downloads_via_clickpy.py for bulk). Reads PEPY_TECH_API_KEY from the
|
(use fetch_pypi_downloads_via_clickpy.py for bulk). Reads PEPY_TECH_API_KEY from the
|
||||||
environment, falling back to the repo-root .env. The v2 endpoint returns
|
environment, falling back to the repo-root .env. The v2 endpoint returns
|
||||||
~90 days of per-day per-version counts; this script sums the most recent
|
~90 days of per-day per-version counts; this script sums the most recent
|
||||||
30 days present in the response across all versions. pepy counts include
|
30 days present in the response across all versions. Names resolve
|
||||||
|
through data/pypi_name_overrides.json, matching the cache sweep. pepy counts include
|
||||||
mirror/CI traffic (CI filtering is a paid pepy feature), matching the
|
mirror/CI traffic (CI filtering is a paid pepy feature), matching the
|
||||||
ClickPy/BigQuery figures; pypistats.org excludes mirrors, so never mix
|
ClickPy/BigQuery figures; pypistats.org excludes mirrors, so never mix
|
||||||
the two in one comparison. Results print to stdout as TSV and are not
|
the two in one comparison. Results print to stdout as TSV and are not
|
||||||
@@ -22,7 +23,7 @@ import time
|
|||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
import httpx
|
import httpx
|
||||||
from fetch_pypi_downloads_via_clickpy import normalize
|
from fetch_pypi_downloads_via_clickpy import resolve
|
||||||
|
|
||||||
ENV_FILE = Path(__file__).parent.parent / ".env"
|
ENV_FILE = Path(__file__).parent.parent / ".env"
|
||||||
PEPY_URL = "https://api.pepy.tech/api/v2/projects/{name}"
|
PEPY_URL = "https://api.pepy.tech/api/v2/projects/{name}"
|
||||||
@@ -51,7 +52,13 @@ def main() -> None:
|
|||||||
if len(sys.argv) < 2:
|
if len(sys.argv) < 2:
|
||||||
print("Usage: python fetch_pypi_downloads_via_pepy.py NAME [NAME ...]", file=sys.stderr)
|
print("Usage: python fetch_pypi_downloads_via_pepy.py NAME [NAME ...]", file=sys.stderr)
|
||||||
sys.exit(1)
|
sys.exit(1)
|
||||||
names = [normalize(name) for name in sys.argv[1:]]
|
names = []
|
||||||
|
for arg in sys.argv[1:]:
|
||||||
|
pkg = resolve(arg)
|
||||||
|
if pkg is None:
|
||||||
|
print(f"{arg}: not pip-installable per pypi_name_overrides.json, skipping", file=sys.stderr)
|
||||||
|
else:
|
||||||
|
names.append(pkg)
|
||||||
with httpx.Client(headers={"X-API-Key": load_api_key()}, timeout=30) as client:
|
with httpx.Client(headers={"X-API-Key": load_api_key()}, timeout=30) as client:
|
||||||
for i, name in enumerate(names):
|
for i, name in enumerate(names):
|
||||||
if i:
|
if i:
|
||||||
|
|||||||
Reference in New Issue
Block a user