mirror of
https://github.com/vinta/awesome-python.git
synced 2026-10-02 08:23:10 +08:00
A pypi.org identity sweep of all 438 cached rows (project_urls/home_page vs entry GitHub URL) found download counts were looked up by README display name, so entries whose name differs from the canonical package silently measured squatters or dead predecessors: pytorch measured a squatter (169,737/mo vs torch's 94M), jinja measured Jinja1 (3,168 vs jinja2's 736M), django-rest-framework a dead alias package (real: djangorestframework), django-rules an abandoned fork (real: rules), strawberry an unrelated bookmarking service (real: strawberry-graphql), devpi a deprecated metapackage (mapped to devpi-server). New curated website/data/pypi_name_overrides.json maps normalized README name to the real package, or null for projects not pip-installable whose name is squatted or a relic (cpython, pyenv, renpy, python-patterns, winpython); also maps mem0 to mem0ai, fasthtml to python-fasthtml, and playwright-python to playwright. All three fetch scripts resolve names through it; the clickpy TSV cache gains a package column recording what each row actually measured. .gitignore switches website/data/ to website/data/* with a negation so the curated overrides file is tracked while caches stay ignored. Co-Authored-By: Claude <noreply@anthropic.com>
76 lines
2.8 KiB
Python
76 lines
2.8 KiB
Python
#!/usr/bin/env python3
|
|
"""Spot-check last-30-day PyPI download counts via the pepy.tech API.
|
|
|
|
Cross-check for a handful of packages during audits — not for full-README
|
|
sweeps: the free API key is throttled to 5 requests/minute (10 burst), so
|
|
this script sleeps 12s between requests and 530 names would take ~2 hours
|
|
(use fetch_pypi_downloads_via_clickpy.py for bulk). Reads PEPY_TECH_API_KEY from the
|
|
environment, falling back to the repo-root .env. The v2 endpoint returns
|
|
~90 days of per-day per-version counts; this script sums the most recent
|
|
30 days present in the response across all versions. Names resolve
|
|
through data/pypi_name_overrides.json, matching the cache sweep. pepy counts include
|
|
mirror/CI traffic (CI filtering is a paid pepy feature), matching the
|
|
ClickPy/BigQuery figures; pypistats.org excludes mirrors, so never mix
|
|
the two in one comparison. Results print to stdout as TSV and are not
|
|
cached — data/pypi_downloads.tsv stays single-source.
|
|
|
|
Usage: python fetch_pypi_downloads_via_pepy.py NAME [NAME ...]
|
|
"""
|
|
|
|
import os
|
|
import sys
|
|
import time
|
|
from pathlib import Path
|
|
|
|
import httpx
|
|
from fetch_pypi_downloads_via_clickpy import resolve
|
|
|
|
ENV_FILE = Path(__file__).parent.parent / ".env"
|
|
PEPY_URL = "https://api.pepy.tech/api/v2/projects/{name}"
|
|
SECONDS_BETWEEN_REQUESTS = 12
|
|
|
|
|
|
def load_api_key() -> str:
|
|
key = os.environ.get("PEPY_TECH_API_KEY", "")
|
|
if not key and ENV_FILE.exists():
|
|
for line in ENV_FILE.read_text().splitlines():
|
|
name, sep, value = line.partition("=")
|
|
if sep and name.strip() == "PEPY_TECH_API_KEY":
|
|
key = value.strip()
|
|
if not key:
|
|
print("Error: PEPY_TECH_API_KEY not set (environment or repo-root .env).", file=sys.stderr)
|
|
sys.exit(1)
|
|
return key
|
|
|
|
|
|
def last_30_day_total(downloads_per_day: dict[str, dict[str, int]]) -> int:
|
|
recent_days = sorted(downloads_per_day)[-30:]
|
|
return sum(sum(downloads_per_day[day].values()) for day in recent_days)
|
|
|
|
|
|
def main() -> None:
|
|
if len(sys.argv) < 2:
|
|
print("Usage: python fetch_pypi_downloads_via_pepy.py NAME [NAME ...]", file=sys.stderr)
|
|
sys.exit(1)
|
|
names = []
|
|
for arg in sys.argv[1:]:
|
|
pkg = resolve(arg)
|
|
if pkg is None:
|
|
print(f"{arg}: not pip-installable per pypi_name_overrides.json, skipping", file=sys.stderr)
|
|
else:
|
|
names.append(pkg)
|
|
with httpx.Client(headers={"X-API-Key": load_api_key()}, timeout=30) as client:
|
|
for i, name in enumerate(names):
|
|
if i:
|
|
time.sleep(SECONDS_BETWEEN_REQUESTS)
|
|
resp = client.get(PEPY_URL.format(name=name))
|
|
if resp.status_code == 404:
|
|
print(f"{name}\tNOT_FOUND")
|
|
continue
|
|
resp.raise_for_status()
|
|
print(f"{name}\t{last_30_day_total(resp.json()['downloads'])}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|