feat: annotate llms.txt entries with PyPI download counts

Entries now end with "(PyPI downloads/month: N, GitHub stars: M)" where
known, replacing the stars-only note, since download counts are the
list's stated primary evidence signal. annotate_entries_with_stars is
renamed to annotate_entries_with_stats and looks downloads up by the
first link's display name, skipping category-index bullets (which link
into the site itself) and Built-in entries (which would otherwise hit
same-named PyPI backports like logging or asyncio).

The intro now mirrors the README subtitle verbatim with the
project/category totals on their own line below, and the "opinionated
catalog" wording is gone since the shortlist ADR is literally titled
"shortlist, not a catalog".

Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
Vinta Chen
2026-08-16 19:43:52 +08:00
co-authored by Claude
parent b7e12cfcad
commit 30c9f18bc2
3 changed files with 68 additions and 39 deletions
+36 -23
View File
@@ -16,7 +16,7 @@ from jinja2 import Environment, FileSystemLoader
from readme_parser import AlsoSee, ParsedGroup, ParsedSection, parse_readme, parse_sponsors, slugify
GITHUB_REPO_URL_RE = re.compile(r"^https?://github\.com/([^/]+/[^/]+?)(?:\.git)?/?$")
MARKDOWN_LINK_RE = re.compile(r"\[[^\]]+\]\(([^)\s]+)\)")
MARKDOWN_LINK_RE = re.compile(r"\[([^\]]+)\]\(([^)\s]+)\)")
BULLET_LINE_RE = re.compile(r"^\s*-\s")
SITE_URL = "https://awesome-python.com/"
SITEMAP_URL = f"{SITE_URL}sitemap.xml"
@@ -377,7 +377,7 @@ def link_llms_category_index_to_canonical_pages(markdown: str, categories: Seque
out: list[str] = []
def replace_link(match: re.Match[str]) -> str:
target = match.group(1)
target = match.group(2)
url = category_urls.get(target)
if url is None:
return match.group(0)
@@ -393,21 +393,24 @@ def build_llms_txt(
template_text: str,
*,
readme_text: str,
subtitle: str,
stars_data: dict[str, dict],
downloads_data: dict[str, int],
categories: Sequence[ParsedSection],
total_entries: int,
) -> str:
"""Render the llms.txt entry point with the curated category catalog."""
categories_md = annotate_entries_with_stars(
"""Render the llms.txt entry point with the curated category shortlist."""
categories_md = annotate_entries_with_stats(
link_llms_category_index_to_canonical_pages(
extract_categories_body(readme_text).rstrip(),
categories,
),
stars_data,
format_stars=lambda n: f"GitHub stars: {n}",
downloads_data,
)
text_env = Environment(autoescape=False, trim_blocks=True, lstrip_blocks=True)
rendered = text_env.from_string(template_text).render(
subtitle=subtitle,
site_url=SITE_URL,
github_repo_url="https://github.com/vinta/awesome-python",
contributing_url="https://github.com/vinta/awesome-python/blob/master/CONTRIBUTING.md",
@@ -420,38 +423,46 @@ def build_llms_txt(
return rendered.rstrip() + "\n"
def annotate_entries_with_stars(
def annotate_entries_with_stats(
markdown: str,
stars_data: dict[str, dict],
*,
format_stars=None,
downloads_data: dict[str, int],
) -> str:
"""Append the star count to bullet entry lines whose first GitHub link has known star data.
"""Append download and star counts to bullet entry lines.
`format_stars` controls the parenthesized text. Defaults to "{N} GitHub stars".
Pass `str` for a bare number.
Downloads are looked up by the first link's display name (the same key the
TSV cache uses); stars by the first GitHub link with known star data.
"""
if format_stars is None:
format_stars = lambda n: f"{n} GitHub stars" # noqa: E731 lambda-assignment
lines = markdown.splitlines(keepends=True)
out: list[str] = []
for line in lines:
if not BULLET_LINE_RE.match(line):
out.append(line)
continue
annotated = line
for match in MARKDOWN_LINK_RE.finditer(line):
repo_key = extract_github_repo(match.group(1))
links = MARKDOWN_LINK_RE.findall(line)
parts: list[str] = []
if links:
name, url = links[0]
# Category-index bullets link into the site itself, and Built-in entries
# would hit same-named PyPI backports (e.g. logging) — no counts for either.
if not url.startswith((SITE_URL, "#")) and detect_source_type(url) != "Built-in":
downloads = downloads_data.get(normalize(name))
if downloads is not None:
parts.append(f"PyPI downloads/month: {downloads}")
for _, url in links:
repo_key = extract_github_repo(url)
if not repo_key:
continue
entry = stars_data.get(repo_key)
if not entry or "stars" not in entry:
continue
stripped = line.rstrip("\n")
ending = line[len(stripped) :]
annotated = f"{stripped} ({format_stars(entry['stars'])}){ending}"
break
out.append(annotated)
if entry and "stars" in entry:
parts.append(f"GitHub stars: {entry['stars']}")
break
if not parts:
out.append(line)
continue
stripped = line.rstrip("\n")
ending = line[len(stripped) :]
out.append(f"{stripped} ({', '.join(parts)}){ending}")
return "".join(out)
@@ -751,7 +762,9 @@ def build(repo_root: Path) -> None:
llms_txt = build_llms_txt(
llms_template,
readme_text=readme_text,
subtitle=subtitle,
stars_data=stars_data,
downloads_data=downloads_data,
categories=categories,
total_entries=total_entries,
)
+4 -2
View File
@@ -1,8 +1,10 @@
# Awesome Python
Awesome Python is an opinionated catalog of {{ total_entries }} Python frameworks, libraries, tools, and resources across {{ total_categories }} {% if total_categories == 1 %}category{% else %}categories{% endif %}.
{{ subtitle }}
Scan the category index, then jump to the matching section for direct project links and short descriptions. GitHub entries with known star data end with a `GitHub stars: N` note in parentheses; treat it as popularity context, not a quality guarantee. Use the homepage for project context outside the catalog.
{{ total_entries }} projects across {{ total_categories }} {% if total_categories == 1 %}category{% else %}categories{% endif %}.
Scan the category index, then jump to the matching section for direct project links and short descriptions. Entries end with `PyPI downloads/month: N` (last 30 days, mirrors and CI included) and `GitHub stars: N` notes in parentheses where known; treat both as popularity context, not a quality guarantee. Use the homepage for project context outside the shortlist.
## Primary Links
+28 -14
View File
@@ -12,7 +12,7 @@ from pathlib import Path
import pytest
from build import (
TemplateEntry,
annotate_entries_with_stars,
annotate_entries_with_stats,
build,
detect_source_type,
extract_entries,
@@ -334,6 +334,10 @@ class TestBuild:
"owner/w2": {"stars": 42, "owner": "owner", "fetched_at": "2026-01-01T00:00:00+00:00"},
}
(data_dir / "github_stars.json").write_text(json.dumps(stars), encoding="utf-8")
(data_dir / "pypi_downloads.tsv").write_text(
"name\tpackage\tdownloads\tfetched_at\nw1\tw1\t777\t2026-08-16\n",
encoding="utf-8",
)
build(tmp_path)
@@ -343,7 +347,8 @@ class TestBuild:
assert '<link rel="alternate" type="text/plain" href="/llms.txt" title="LLMs text entry point" />' in index_html
assert llms_txt.startswith("# Awesome Python\n")
assert llms_txt.startswith("# Awesome Python\n\nIntro.\n")
assert "2 projects across 1 category." in llms_txt
assert "Scan the category index" in llms_txt
assert "Homepage: https://awesome-python.com/" in llms_txt
assert "Markdown homepage" not in llms_txt
@@ -357,7 +362,7 @@ class TestBuild:
assert "- [Widgets](https://awesome-python.com/categories/widgets/)" in llms_txt
assert "- [Widgets](#widgets)" not in llms_txt
assert "### Widgets" in llms_txt
assert "- [w1](https://example.com) - A widget." in llms_txt
assert "- [w1](https://example.com) - A widget. (PyPI downloads/month: 777)" in llms_txt
assert "- [w2](https://github.com/owner/w2) - A starred widget. (GitHub stars: 42)" in llms_txt
assert llms_txt != readme
assert "# Contributing" not in llms_txt
@@ -1196,15 +1201,24 @@ class TestExtractEntries:
# ---------------------------------------------------------------------------
# annotate_entries_with_stars
# annotate_entries_with_stats
# ---------------------------------------------------------------------------
class TestAnnotateEntriesWithStars:
class TestAnnotateEntriesWithStats:
def test_appends_star_count_to_bullet(self):
markdown = "- [foo](https://github.com/owner/foo) - A foo.\n"
stars = {"owner/foo": {"stars": 123, "owner": "owner"}}
assert annotate_entries_with_stars(markdown, stars) == ("- [foo](https://github.com/owner/foo) - A foo. (123 GitHub stars)\n")
assert annotate_entries_with_stats(markdown, stars, {}) == ("- [foo](https://github.com/owner/foo) - A foo. (GitHub stars: 123)\n")
def test_appends_downloads_by_display_name(self):
markdown = "- [Foo.py](https://example.com) - A foo.\n"
assert annotate_entries_with_stats(markdown, {}, {"foo-py": 777}) == ("- [Foo.py](https://example.com) - A foo. (PyPI downloads/month: 777)\n")
def test_appends_downloads_and_stars_together(self):
markdown = "- [foo](https://github.com/owner/foo) - A foo.\n"
stars = {"owner/foo": {"stars": 123, "owner": "owner"}}
assert annotate_entries_with_stats(markdown, stars, {"foo": 777}) == ("- [foo](https://github.com/owner/foo) - A foo. (PyPI downloads/month: 777, GitHub stars: 123)\n")
def test_uses_first_github_link(self):
markdown = "- [foo](https://github.com/owner/foo) - A foo. Also [bar](https://github.com/owner/bar).\n"
@@ -1212,31 +1226,31 @@ class TestAnnotateEntriesWithStars:
"owner/foo": {"stars": 10, "owner": "owner"},
"owner/bar": {"stars": 99, "owner": "owner"},
}
assert annotate_entries_with_stars(markdown, stars) == ("- [foo](https://github.com/owner/foo) - A foo. Also [bar](https://github.com/owner/bar). (10 GitHub stars)\n")
assert annotate_entries_with_stats(markdown, stars, {}) == ("- [foo](https://github.com/owner/foo) - A foo. Also [bar](https://github.com/owner/bar). (GitHub stars: 10)\n")
def test_skips_entries_without_star_data(self):
def test_skips_entries_without_data(self):
markdown = "- [foo](https://github.com/owner/foo) - A foo.\n"
assert annotate_entries_with_stars(markdown, {}) == markdown
assert annotate_entries_with_stats(markdown, {}, {}) == markdown
def test_skips_non_github_links(self):
def test_skips_non_github_links_for_stars(self):
markdown = "- [foo](https://example.com) - A foo.\n"
stars = {"owner/foo": {"stars": 1, "owner": "owner"}}
assert annotate_entries_with_stars(markdown, stars) == markdown
assert annotate_entries_with_stats(markdown, stars, {}) == markdown
def test_skips_non_bullet_lines(self):
markdown = "See [foo](https://github.com/owner/foo) for details.\n"
stars = {"owner/foo": {"stars": 1, "owner": "owner"}}
assert annotate_entries_with_stars(markdown, stars) == markdown
assert annotate_entries_with_stats(markdown, stars, {"foo": 5}) == markdown
def test_handles_indented_bullets(self):
markdown = " - [foo](https://github.com/owner/foo)\n"
stars = {"owner/foo": {"stars": 7, "owner": "owner"}}
assert annotate_entries_with_stars(markdown, stars) == (" - [foo](https://github.com/owner/foo) (7 GitHub stars)\n")
assert annotate_entries_with_stats(markdown, stars, {}) == (" - [foo](https://github.com/owner/foo) (GitHub stars: 7)\n")
def test_preserves_lines_without_trailing_newline(self):
markdown = "- [foo](https://github.com/owner/foo) - A foo."
stars = {"owner/foo": {"stars": 5, "owner": "owner"}}
assert annotate_entries_with_stars(markdown, stars) == ("- [foo](https://github.com/owner/foo) - A foo. (5 GitHub stars)")
assert annotate_entries_with_stats(markdown, stars, {}) == ("- [foo](https://github.com/owner/foo) - A foo. (GitHub stars: 5)")
class TestLoadDownloads: