From cc804b1de2b85e7dcc46f04e4c334f6d37d4cbca Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 16 Aug 2026 17:52:04 +0800 Subject: [PATCH] feat: add Downloads/Month column to website table Sourced from website/data/pypi_downloads.tsv the same way github_stars.json feeds the stars column. The new sortable column sits between GitHub Stars and Last Commit on the homepage and category pages, formatted with thousands separators like stars, with an em dash when no PyPI data exists. Rows are matched by normalized README display name; Built-in entries never show counts since same-named PyPI packages are stdlib backports (e.g. the asyncio package). Below 960px the column hides and the count moves into the expand row, mirroring the existing Last Commit treatment. main.js gains the downloads sort branch and URL param. The deploy workflow fetches the TSV via the new make fetch_pypi_downloads target with a daily actions/cache fallback, mirroring the stars fetch, but non-fatal: the column degrades to dashes when the fetch fails, unlike stars which the build requires. Co-Authored-By: Claude --- .github/workflows/deploy-website.yml | 19 +++++++++++ Makefile | 3 ++ website/build.py | 23 +++++++++++++ website/static/main.js | 13 +++++++- website/static/style.css | 17 ++++++++-- website/templates/category.html | 15 +++++++-- website/templates/index.html | 15 +++++++-- website/tests/test_build.py | 49 ++++++++++++++++++++++++++++ 8 files changed, 146 insertions(+), 8 deletions(-) diff --git a/.github/workflows/deploy-website.yml b/.github/workflows/deploy-website.yml index dd748bb5..0c23761c 100644 --- a/.github/workflows/deploy-website.yml +++ b/.github/workflows/deploy-website.yml @@ -71,6 +71,25 @@ jobs: fi python -m json.tool website/data/github_stars.json > /dev/null + - name: Restore PyPI download data cache + uses: actions/cache/restore@v4 + with: + path: website/data/pypi_downloads.tsv + key: pypi-downloads-${{ steps.date.outputs.today }} + restore-keys: pypi-downloads- + + - name: Fetch PyPI downloads + id: fetch-downloads + continue-on-error: true + run: make fetch_pypi_downloads + + - name: Save PyPI download data cache + if: steps.fetch-downloads.outcome == 'success' + uses: actions/cache/save@v4 + with: + path: website/data/pypi_downloads.tsv + key: pypi-downloads-${{ steps.date.outputs.today }} + - name: Build website run: make build diff --git a/Makefile b/Makefile index 8debe6a9..5a07ad91 100644 --- a/Makefile +++ b/Makefile @@ -7,6 +7,9 @@ install: fetch_github_stars: uv run python website/fetch_github_stars.py +fetch_pypi_downloads: + uv run python website/fetch_pypi_downloads_via_clickpy.py + test: uv run pytest website/tests/ -v diff --git a/website/build.py b/website/build.py index 347c4fb0..cba524fa 100644 --- a/website/build.py +++ b/website/build.py @@ -11,6 +11,7 @@ from datetime import UTC, datetime from pathlib import Path from typing import TypedDict +from fetch_pypi_downloads_via_clickpy import normalize from jinja2 import Environment, FileSystemLoader from readme_parser import AlsoSee, ParsedGroup, ParsedSection, parse_readme, parse_sponsors, slugify @@ -52,6 +53,7 @@ class TemplateEntry(TypedDict): groups: list[str] subcategories: list[TemplateSubcategory] stars: int | None + downloads: int | None owner: str | None last_commit_at: str | None source_type: str | None @@ -96,6 +98,22 @@ def load_stars(path: Path) -> dict[str, dict]: return {} +def load_downloads(path: Path) -> dict[str, int]: + """Load last-30-day download counts from the TSV cache, keyed by normalized README name. + + Columns: name, package, downloads, fetched_at. Skips the header and + NOT_FOUND rows. Returns empty dict if the file doesn't exist. + """ + if not path.exists(): + return {} + downloads: dict[str, int] = {} + for line in path.read_text(encoding="utf-8").splitlines()[1:]: + parts = line.split("\t") + if len(parts) >= 3 and parts[2].isdigit(): + downloads[parts[0]] = int(parts[2]) + return downloads + + def sort_entries(entries: Sequence[TemplateEntry]) -> list[TemplateEntry]: """Sort entries by stars descending, then name ascending. @@ -481,6 +499,7 @@ def extract_entries( groups=[], subcategories=[], stars=None, + downloads=None, owner=None, last_commit_at=None, source_type=detect_source_type(entry["url"]), @@ -535,6 +554,7 @@ def build(repo_root: Path) -> None: build_date = datetime.now(UTC) stars_data = load_stars(website / "data" / "github_stars.json") + downloads_data = load_downloads(website / "data" / "pypi_downloads.tsv") repo_self = stars_data.get("vinta/awesome-python", {}) repo_stars = None @@ -551,6 +571,9 @@ def build(repo_root: Path) -> None: entry["stars"] = sd["stars"] entry["owner"] = sd["owner"] entry["last_commit_at"] = sd.get("last_commit_at", "") + # Built-in entries would hit same-named PyPI backports (e.g. asyncio), not the stdlib. + if entry.get("source_type") != "Built-in": + entry["downloads"] = downloads_data.get(normalize(entry["name"])) entries = sort_entries(entries) category_urls = {cat["name"]: category_path(cat) for cat in categories} diff --git a/website/static/main.js b/website/static/main.js index d5b337b3..073bb6c4 100644 --- a/website/static/main.js +++ b/website/static/main.js @@ -238,6 +238,14 @@ function getSortValue(row, col) { const num = parseInt(text, 10); return isNaN(num) ? -1 : num; } + if (col === "downloads") { + const text = row + .querySelector(".col-downloads") + .textContent.trim() + .replace(/,/g, ""); + const num = parseInt(text, 10); + return isNaN(num) ? -1 : num; + } if (col === "commit-time") { const attr = row.querySelector(".col-commit").getAttribute("data-commit"); return attr ? new Date(attr).getTime() : 0; @@ -467,7 +475,10 @@ if (backToTop) { const order = params.get("order"); if (q && searchInput) searchInput.value = q; if ( - (sort === "name" || sort === "stars" || sort === "commit-time") && + (sort === "name" || + sort === "stars" || + sort === "downloads" || + sort === "commit-time") && (order === "desc" || order === "asc") ) { activeSort = { col: sort, order: order }; diff --git a/website/static/style.css b/website/static/style.css index 93056570..1cd3a9f9 100644 --- a/website/static/style.css +++ b/website/static/style.css @@ -893,6 +893,14 @@ th[data-sort].sort-asc::after { letter-spacing: 0.02em; } +.col-downloads { + width: 9.5rem; + text-align: right; + white-space: nowrap; + font-variant-numeric: tabular-nums; + color: var(--ink-soft); +} + .col-commit { width: 9rem; white-space: nowrap; @@ -1041,7 +1049,8 @@ th[data-sort].sort-asc::after { color: var(--line-strong); } -.expand-commit { +.expand-commit, +.expand-downloads { display: none; } @@ -1606,11 +1615,13 @@ th[data-sort].sort-asc::after { justify-self: start; } - .col-commit { + .col-commit, + .col-downloads { display: none; } - .expand-commit { + .expand-commit, + .expand-downloads { display: inline; } diff --git a/website/templates/category.html b/website/templates/category.html index 6e4b03e7..51542767 100644 --- a/website/templates/category.html +++ b/website/templates/category.html @@ -120,6 +120,9 @@ + + + @@ -156,6 +159,10 @@ >{{ entry.source_type }}{% else %}—{% endif %} + + {% if entry.downloads is not none %}{{ + "{:,}".format(entry.downloads) }}{% else %}—{% endif %} + - +
{{ entry.description | safe }}
{% endif %} - +
{% if entry.also_see %}
@@ -249,6 +256,10 @@ >{% endif %} {% if entry.downloads is not none %}/{{ + "{:,}".format(entry.downloads) }} downloads/month{% endif %}
diff --git a/website/templates/index.html b/website/templates/index.html index e7f1c9ea..dc20049f 100644 --- a/website/templates/index.html +++ b/website/templates/index.html @@ -171,6 +171,9 @@ + + + @@ -207,6 +210,10 @@ >{{ entry.source_type }}{% else %}—{% endif %} + + {% if entry.downloads is not none %}{{ + "{:,}".format(entry.downloads) }}{% else %}—{% endif %} + - +
{{ entry.description | safe }}
{% endif %} - +
{% if entry.description %}
{{ entry.description | safe }}
@@ -298,6 +305,10 @@ >{% endif %} {% if entry.downloads is not none %}/{{ + "{:,}".format(entry.downloads) }} downloads/month{% endif %}
diff --git a/website/tests/test_build.py b/website/tests/test_build.py index 0af6c0de..3b417b24 100644 --- a/website/tests/test_build.py +++ b/website/tests/test_build.py @@ -17,6 +17,7 @@ from build import ( detect_source_type, extract_entries, extract_github_repo, + load_downloads, load_stars, sort_entries, subcategory_path, @@ -429,6 +430,40 @@ class TestBuild: # Expand content present assert "expand-content" in html + def test_build_with_downloads_renders_column(self, tmp_path): + readme = textwrap.dedent("""\ + # T + + ## Projects + + ## Stuff + + - [My-Lib](https://github.com/org/mylib) - On PyPI. + - [no-pypi](https://example.com/none) - Not on PyPI. + - [asyncio](https://docs.python.org/3/library/asyncio.html) - Built-in. + + # Contributing + + Done. + """) + (tmp_path / "README.md").write_text(readme, encoding="utf-8") + self._copy_real_templates(tmp_path) + + data_dir = tmp_path / "website" / "data" + data_dir.mkdir(parents=True) + # Keyed by normalized README display name, like fetch_pypi_downloads_via_clickpy.py writes it + (data_dir / "pypi_downloads.tsv").write_text( + "name\tpackage\tdownloads\tfetched_at\nasyncio\tasyncio\t26305454\t2026-08-16\nmy-lib\tmy-lib\t1234567\t2026-08-16\n", + encoding="utf-8", + ) + + build(tmp_path) + + html = (tmp_path / "website" / "output" / "index.html").read_text(encoding="utf-8") + assert "1,234,567" in html + # Built-in entries never show PyPI counts: the asyncio row is the backport package + assert "26,305,454" not in html + def test_build_fails_when_group_and_category_slug_collide(self, tmp_path): readme = textwrap.dedent("""\ # T @@ -974,6 +1009,7 @@ def _template_entry(name: str, stars: int | None, source_type: str | None = None groups=[], subcategories=[], stars=stars, + downloads=None, owner=None, last_commit_at=None, source_type=source_type, @@ -1199,3 +1235,16 @@ class TestAnnotateEntriesWithStars: markdown = "- [foo](https://github.com/owner/foo) - A foo." stars = {"owner/foo": {"stars": 5, "owner": "owner"}} assert annotate_entries_with_stars(markdown, stars) == ("- [foo](https://github.com/owner/foo) - A foo. (5 GitHub stars)") + + +class TestLoadDownloads: + def test_parses_tsv_and_skips_not_found(self, tmp_path): + tsv = tmp_path / "pypi_downloads.tsv" + tsv.write_text( + "name\tpackage\tdownloads\tfetched_at\naiohttp\taiohttp\t649105404\t2026-08-16\npytorch\ttorch\t50000000\t2026-08-16\ndead-pkg\t-\tNOT_FOUND\t2026-08-16\n", + encoding="utf-8", + ) + assert load_downloads(tsv) == {"aiohttp": 649105404, "pytorch": 50000000} + + def test_missing_file_returns_empty(self, tmp_path): + assert load_downloads(tmp_path / "nope.tsv") == {}