feat: add Downloads/Month column to website table

Sourced from website/data/pypi_downloads.tsv the same way
github_stars.json feeds the stars column. The new sortable column
sits between GitHub Stars and Last Commit on the homepage and
category pages, formatted with thousands separators like stars, with
an em dash when no PyPI data exists. Rows are matched by normalized
README display name; Built-in entries never show counts since
same-named PyPI packages are stdlib backports (e.g. the asyncio
package).

Below 960px the column hides and the count moves into the expand
row, mirroring the existing Last Commit treatment. main.js gains the
downloads sort branch and URL param.

The deploy workflow fetches the TSV via the new
make fetch_pypi_downloads target with a daily actions/cache
fallback, mirroring the stars fetch, but non-fatal: the column
degrades to dashes when the fetch fails, unlike stars which the
build requires.

Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
Vinta Chen
2026-08-16 17:52:04 +08:00
co-authored by Claude
parent b440e64cd0
commit cc804b1de2
8 changed files with 146 additions and 8 deletions
+23
View File
@@ -11,6 +11,7 @@ from datetime import UTC, datetime
from pathlib import Path
from typing import TypedDict
from fetch_pypi_downloads_via_clickpy import normalize
from jinja2 import Environment, FileSystemLoader
from readme_parser import AlsoSee, ParsedGroup, ParsedSection, parse_readme, parse_sponsors, slugify
@@ -52,6 +53,7 @@ class TemplateEntry(TypedDict):
groups: list[str]
subcategories: list[TemplateSubcategory]
stars: int | None
downloads: int | None
owner: str | None
last_commit_at: str | None
source_type: str | None
@@ -96,6 +98,22 @@ def load_stars(path: Path) -> dict[str, dict]:
return {}
def load_downloads(path: Path) -> dict[str, int]:
"""Load last-30-day download counts from the TSV cache, keyed by normalized README name.
Columns: name, package, downloads, fetched_at. Skips the header and
NOT_FOUND rows. Returns empty dict if the file doesn't exist.
"""
if not path.exists():
return {}
downloads: dict[str, int] = {}
for line in path.read_text(encoding="utf-8").splitlines()[1:]:
parts = line.split("\t")
if len(parts) >= 3 and parts[2].isdigit():
downloads[parts[0]] = int(parts[2])
return downloads
def sort_entries(entries: Sequence[TemplateEntry]) -> list[TemplateEntry]:
"""Sort entries by stars descending, then name ascending.
@@ -481,6 +499,7 @@ def extract_entries(
groups=[],
subcategories=[],
stars=None,
downloads=None,
owner=None,
last_commit_at=None,
source_type=detect_source_type(entry["url"]),
@@ -535,6 +554,7 @@ def build(repo_root: Path) -> None:
build_date = datetime.now(UTC)
stars_data = load_stars(website / "data" / "github_stars.json")
downloads_data = load_downloads(website / "data" / "pypi_downloads.tsv")
repo_self = stars_data.get("vinta/awesome-python", {})
repo_stars = None
@@ -551,6 +571,9 @@ def build(repo_root: Path) -> None:
entry["stars"] = sd["stars"]
entry["owner"] = sd["owner"]
entry["last_commit_at"] = sd.get("last_commit_at", "")
# Built-in entries would hit same-named PyPI backports (e.g. asyncio), not the stdlib.
if entry.get("source_type") != "Built-in":
entry["downloads"] = downloads_data.get(normalize(entry["name"]))
entries = sort_entries(entries)
category_urls = {cat["name"]: category_path(cat) for cat in categories}