" in category_html
assert 'Widget libraries. Also see awesome-widgets.' in category_html
assert 'href="https://example.com/w1"' in category_html
assert "A widget." in category_html
assert 'href="https://github.com/owner/w2"' in category_html
assert '
' in category_html
assert "42" in category_html
assert "2026-01-01T00:00:00+00:00" in category_html
def test_build_links_description_anchors_to_category_pages(self, tmp_path):
readme = textwrap.dedent("""\
# Awesome Python
Intro.
## Projects
**Tools**
## Audio & Video
_Media tools._
- [a1](https://example.com/a1) - A media tool.
## Widgets
_Widget libraries. Also see [Audio & Video](#audio--video) and [Gadgets](#gadgets)._
- [w1](https://example.com/w1) - A widget.
# Contributing
Help!
""")
(tmp_path / "README.md").write_text(readme, encoding="utf-8")
self._copy_real_templates(tmp_path)
build(tmp_path)
category_html = (tmp_path / "website" / "output" / "categories" / "widgets" / "index.html").read_text(encoding="utf-8")
assert 'Also see Audio & Video' in category_html
assert 'Gadgets' in category_html
def test_build_creates_llms_text_alternate_without_sponsors(self, tmp_path):
readme = textwrap.dedent("""\
# Awesome Python
Intro.
## **Sponsors**
- **[Sponsor](https://sponsor.example.com)**: Sponsored tool.
> Become a sponsor: [Sponsor us](SPONSORSHIP.md).
## Categories
**Tools**
- [Widgets](#widgets)
## Projects
**Tools**
### Widgets
- [w1](https://example.com) - A widget.
- [w2](https://github.com/owner/w2) - A starred widget.
## Contributing
Help!
""")
(tmp_path / "README.md").write_text(readme, encoding="utf-8")
self._copy_real_templates(tmp_path)
data_dir = tmp_path / "website" / "data"
data_dir.mkdir(parents=True)
stars = {
"owner/w2": {"stars": 42, "owner": "owner", "fetched_at": "2026-01-01T00:00:00+00:00"},
}
(data_dir / "github_stars.json").write_text(json.dumps(stars), encoding="utf-8")
(data_dir / "pypi_downloads.tsv").write_text(
"name\tpackage\tdownloads\tfetched_at\nw1\tw1\t777\t2026-08-16\n",
encoding="utf-8",
)
build(tmp_path)
site = tmp_path / "website" / "output"
index_html = (site / "index.html").read_text(encoding="utf-8")
llms_txt = (site / "llms.txt").read_text(encoding="utf-8")
assert '' in index_html
assert llms_txt.startswith("# Awesome Python\n\nIntro.\n")
assert "2 projects across 1 category, updated on " in llms_txt
assert "Scan the category index" in llms_txt
assert "Homepage: https://awesome-python.com/" in llms_txt
assert "Markdown homepage" not in llms_txt
assert "https://awesome-python.com/index.md" not in llms_txt
assert "GitHub repository: https://github.com/vinta/awesome-python" in llms_txt
assert "Contributing guide: https://github.com/vinta/awesome-python/blob/master/CONTRIBUTING.md" in llms_txt
assert "Sponsorship: https://awesome-python.com/sponsorship/" in llms_txt
assert "Sitemap: https://awesome-python.com/sitemap.xml" in llms_txt
assert "## Categories" in llms_txt
assert "**Tools**" in llms_txt
assert "- [Widgets](https://awesome-python.com/categories/widgets/)" in llms_txt
assert "- [Widgets](#widgets)" not in llms_txt
assert "### Widgets" in llms_txt
assert "- [w1](https://example.com) - A widget. (PyPI downloads/month: 777)" in llms_txt
assert "- [w2](https://github.com/owner/w2) - A starred widget. (GitHub stars: 42)" in llms_txt
assert llms_txt != readme
assert "# Contributing" not in llms_txt
def test_build_cleans_stale_output(self, tmp_path):
readme = textwrap.dedent("""\
# T
## Projects
## Only
- [x](https://x.com) - X.
# Contributing
Done.
""")
self._make_repo(tmp_path, readme)
stale = tmp_path / "website" / "output" / "categories" / "stale"
stale.mkdir(parents=True)
(stale / "index.html").write_text("old", encoding="utf-8")
build(tmp_path)
assert not (tmp_path / "website" / "output" / "categories" / "stale").exists()
def test_build_with_stars_sorts_by_stars(self, tmp_path):
readme = textwrap.dedent("""\
# T
## Projects
## Stuff
- [low-stars](https://github.com/org/low) - Low.
- [high-stars](https://github.com/org/high) - High.
- [no-stars](https://example.com/none) - None.
# Contributing
Done.
""")
(tmp_path / "README.md").write_text(readme, encoding="utf-8")
# Copy real templates
real_tpl = Path(__file__).parent / ".." / "templates"
tpl_dir = tmp_path / "website" / "templates"
shutil.copytree(real_tpl, tpl_dir)
# Create mock star data
data_dir = tmp_path / "website" / "data"
data_dir.mkdir(parents=True)
stars = {
"org/high": {"stars": 5000, "owner": "org", "fetched_at": "2026-01-01T00:00:00+00:00"},
"org/low": {"stars": 100, "owner": "org", "fetched_at": "2026-01-01T00:00:00+00:00"},
}
(data_dir / "github_stars.json").write_text(json.dumps(stars), encoding="utf-8")
build(tmp_path)
html = (tmp_path / "website" / "output" / "index.html").read_text(encoding="utf-8")
# Star-sorted: high-stars (5000) before low-stars (100) before no-stars (None)
assert html.index("high-stars") < html.index("low-stars")
assert html.index("low-stars") < html.index("no-stars")
# Formatted star counts
assert "5,000" in html
assert "100" in html
# Expand content present
assert "expand-content" in html
def test_build_with_downloads_renders_column(self, tmp_path):
readme = textwrap.dedent("""\
# T
## Projects
## Stuff
- [My-Lib](https://github.com/org/mylib) - On PyPI.
- [no-pypi](https://example.com/none) - Not on PyPI.
- [asyncio](https://docs.python.org/3/library/asyncio.html) - Stdlib.
- [my-lib.thing](https://github.com/org/mylib) - (part of My-Lib) A bundled feature.
# Contributing
Done.
""")
(tmp_path / "README.md").write_text(readme, encoding="utf-8")
self._copy_real_templates(tmp_path)
data_dir = tmp_path / "website" / "data"
data_dir.mkdir(parents=True)
# Keyed by normalized README display name, like fetch_pypi_downloads_via_clickpy.py writes it
(data_dir / "pypi_downloads.tsv").write_text(
"name\tpackage\tdownloads\tfetched_at\nasyncio\tasyncio\t26305454\t2026-08-16\nmy-lib\tmy-lib\t1234567\t2026-08-16\n",
encoding="utf-8",
)
build(tmp_path)
html = (tmp_path / "website" / "output" / "index.html").read_text(encoding="utf-8")
assert "1,234,567" in html
# Stdlib entries never show PyPI counts: the asyncio row is the backport package
assert "26,305,454" not in html
# Default sort: entries with download counts come first
assert html.index("My-Lib") < html.index("no-pypi")
# Each no-download entry gets the badge matching why it has no count
assert html.count('Stdlib') == 1
assert html.count('Bundled') == 1
assert html.count('Not on PyPI') == 1
def test_build_fails_when_group_and_category_slug_collide(self, tmp_path):
readme = textwrap.dedent("""\
# T
## Projects
**Widgets**
## Widgets
- [w1](https://example.com) - W.
# Contributing
Done.
""")
self._make_repo(tmp_path, readme)
with pytest.raises(ValueError, match="slug collision"):
build(tmp_path)
def test_index_contains_aligned_homepage_metadata(self, tmp_path):
readme = (Path(__file__).parents[2] / "README.md").read_text(encoding="utf-8")
(tmp_path / "README.md").write_text(readme, encoding="utf-8")
self._copy_real_templates(tmp_path)
build(tmp_path)
parsed_groups = parse_readme(readme)
categories = [cat for group in parsed_groups for cat in group["categories"]]
entries = extract_entries(categories, parsed_groups)
html = (tmp_path / "website" / "output" / "index.html").read_text(encoding="utf-8")
parser = HeadMetadataParser()
parser.feed(html)
expected_title = "Awesome Python"
expected_description = f"An opinionated guide to the best Python frameworks, libraries, and tools. Explore {len(entries)} curated projects across {len(categories)} categories, from AI and agents to data science and web development."
expected_url = "https://awesome-python.com/"
expected_image = "https://awesome-python.com/static/og-image.png"
assert parser.title_count == 1
assert parser.title.strip() == expected_title
assert parser.meta_by_name["description"] == expected_description
assert parser.links_by_rel["canonical"] == expected_url
assert parser.meta_by_property["og:type"] == "website"
assert parser.meta_by_property["og:title"] == expected_title
assert parser.meta_by_property["og:description"] == expected_description
assert parser.meta_by_property["og:image"] == expected_image
assert parser.meta_by_property["og:url"] == expected_url
assert parser.meta_by_name["twitter:card"] == "summary_large_image"
assert parser.meta_by_name["twitter:title"] == expected_title
assert parser.meta_by_name["twitter:description"] == expected_description
assert parser.meta_by_name["twitter:image"] == expected_image
assert "\n Sponsorship' in html
assert 'id="hero-category-heading">Browse by category' in html
assert 'class="hero-category-link" href="/categories/ai-and-agents/"' in html
def test_index_contains_homepage_json_ld(self, tmp_path):
readme = (Path(__file__).parents[2] / "README.md").read_text(encoding="utf-8")
(tmp_path / "README.md").write_text(readme, encoding="utf-8")
self._copy_real_templates(tmp_path)
build(tmp_path)
parsed_groups = parse_readme(readme)
categories = [cat for group in parsed_groups for cat in group["categories"]]
entries = extract_entries(categories, parsed_groups)
html = (tmp_path / "website" / "output" / "index.html").read_text(encoding="utf-8")
marker = '", start)
block = html[start:end]
assert "" not in block
data = json.loads(block)
assert data["@context"] == "https://schema.org"
graph = {node["@type"]: node for node in data["@graph"]}
assert set(graph) == {"WebSite", "CollectionPage"}
assert graph["WebSite"]["url"] == "https://awesome-python.com/"
assert graph["WebSite"]["name"] == "Awesome Python"
assert graph["WebSite"]["@id"] == "https://awesome-python.com/#website"
collection = graph["CollectionPage"]
assert collection["@id"] == "https://awesome-python.com/"
assert collection["url"] == "https://awesome-python.com/"
assert collection["isPartOf"] == {"@type": "WebSite", "@id": graph["WebSite"]["@id"]}
expected_description = f"An opinionated guide to the best Python frameworks, libraries, and tools. Explore {len(entries)} curated projects across {len(categories)} categories, from AI and agents to data science and web development."
assert collection["description"] == expected_description
item_list = collection["mainEntity"]
assert item_list["@type"] == "ItemList"
assert item_list["numberOfItems"] == len(entries)
assert len(item_list["itemListElement"]) == len(entries)
positions = [item["position"] for item in item_list["itemListElement"]]
assert positions == list(range(1, len(entries) + 1))
assert all(item["@type"] == "ListItem" for item in item_list["itemListElement"])
assert all(item["url"].startswith(("http://", "https://")) for item in item_list["itemListElement"])
rendered_names = {item["name"] for item in item_list["itemListElement"]}
rendered_urls = {item["url"] for item in item_list["itemListElement"]}
assert rendered_names == {e["name"] for e in entries}
assert rendered_urls == {e["url"] for e in entries}
def test_category_page_contains_json_ld(self, tmp_path):
readme = textwrap.dedent("""\
# Awesome Python
Intro.
## Projects
**Tools**
## Widgets
_Widget libraries._
- [w1](https://example.com/w1) - A widget.
- [w2](https://github.com/owner/w2) - A starred widget.
# Contributing
Help!
""")
(tmp_path / "README.md").write_text(readme, encoding="utf-8")
self._copy_real_templates(tmp_path)
build(tmp_path)
category_html = (tmp_path / "website" / "output" / "categories" / "widgets" / "index.html").read_text(encoding="utf-8")
marker = '", start)
block = category_html[start:end]
assert "" not in block
data = json.loads(block)
assert data["@context"] == "https://schema.org"
graph = {node["@type"]: node for node in data["@graph"]}
assert set(graph) == {"WebSite", "CollectionPage", "BreadcrumbList"}
assert graph["WebSite"]["@id"] == "https://awesome-python.com/#website"
collection = graph["CollectionPage"]
assert collection["name"] == "Python Widgets Libraries"
assert collection["@id"] == "https://awesome-python.com/categories/widgets/"
assert collection["url"] == "https://awesome-python.com/categories/widgets/"
assert collection["description"] == "Widget libraries. Explore 2 curated Python projects in Widgets."
assert collection["isPartOf"] == {"@type": "WebSite", "@id": "https://awesome-python.com/#website"}
item_list = collection["mainEntity"]
assert item_list["@type"] == "ItemList"
assert item_list["numberOfItems"] == 2
names = {item["name"] for item in item_list["itemListElement"]}
urls = {item["url"] for item in item_list["itemListElement"]}
assert names == {"w1", "w2"}
assert urls == {"https://example.com/w1", "https://github.com/owner/w2"}
positions = sorted(item["position"] for item in item_list["itemListElement"])
assert positions == [1, 2]
breadcrumbs = graph["BreadcrumbList"]["itemListElement"]
assert breadcrumbs == [
{"@type": "ListItem", "position": 1, "name": "Awesome Python", "item": "https://awesome-python.com/"},
{"@type": "ListItem", "position": 2, "name": "Widgets", "item": "https://awesome-python.com/categories/widgets/"},
]
def test_group_page_falls_back_to_default_description_in_json_ld(self, tmp_path):
readme = textwrap.dedent("""\
# T
## Projects
**AI & ML**
## Deep Learning
- [dl1](https://example.com/dl1) - DL.
# Contributing
Done.
""")
self._copy_real_templates(tmp_path)
(tmp_path / "README.md").write_text(readme, encoding="utf-8")
build(tmp_path)
group_html = (tmp_path / "website" / "output" / "categories" / "ai-ml" / "index.html").read_text(encoding="utf-8")
marker = '", start)
data = json.loads(group_html[start:end])
graph = {node["@type"]: node for node in data["@graph"]}
collection = graph["CollectionPage"]
assert collection["name"] == "Python AI & ML Libraries"
assert collection["@id"] == "https://awesome-python.com/categories/ai-ml/"
assert collection["url"] == "https://awesome-python.com/categories/ai-ml/"
assert collection["description"] == "Explore 1 curated Python projects in AI & ML. Part of the Awesome Python catalog."
def test_build_creates_subcategory_pages(self, tmp_path):
readme = textwrap.dedent("""\
# T
## Projects
**Web**
## Web Frameworks
- Synchronous
- [django](https://example.com/django) - Sync framework.
- Asynchronous
- [fastapi](https://example.com/fastapi) - Async framework.
# Contributing
Done.
""")
self._copy_real_templates(tmp_path)
(tmp_path / "README.md").write_text(readme, encoding="utf-8")
build(tmp_path)
site = tmp_path / "website" / "output"
sync = (site / "categories" / "web-frameworks" / "synchronous" / "index.html").read_text(encoding="utf-8")
async_ = (site / "categories" / "web-frameworks" / "asynchronous" / "index.html").read_text(encoding="utf-8")
assert "django" in sync
assert "fastapi" not in sync
assert "fastapi" in async_
assert "django" not in async_
def test_subcategory_page_shows_breadcrumb(self, tmp_path):
readme = textwrap.dedent("""\
# T
## Projects
**Web**
## Web Frameworks
- Synchronous
- [django](https://example.com/django) - Sync.
# Contributing
Done.
""")
self._copy_real_templates(tmp_path)
(tmp_path / "README.md").write_text(readme, encoding="utf-8")
build(tmp_path)
site = tmp_path / "website" / "output"
sync = (site / "categories" / "web-frameworks" / "synchronous" / "index.html").read_text(encoding="utf-8")
assert 'href="/categories/web-frameworks/"' in sync
assert "Web Frameworks" in sync
assert "
Synchronous
" in sync
assert "category-breadcrumb" in sync
parser = HeadMetadataParser()
parser.feed(sync)
assert parser.title.strip() == "Synchronous for Web Frameworks - Awesome Python"
assert parser.meta_by_name["description"] == "Explore 1 curated Python projects in Synchronous for Web Frameworks. Part of the Awesome Python catalog."
marker = '", start)
graph = {node["@type"]: node for node in json.loads(sync[start:end])["@graph"]}
assert graph["CollectionPage"]["name"] == "Synchronous for Web Frameworks"
assert graph["BreadcrumbList"]["itemListElement"] == [
{"@type": "ListItem", "position": 1, "name": "Awesome Python", "item": "https://awesome-python.com/"},
{
"@type": "ListItem",
"position": 2,
"name": "Web Frameworks",
"item": "https://awesome-python.com/categories/web-frameworks/",
},
{
"@type": "ListItem",
"position": 3,
"name": "Synchronous",
"item": "https://awesome-python.com/categories/web-frameworks/synchronous/",
},
]
parent = (site / "categories" / "web-frameworks" / "index.html").read_text(encoding="utf-8")
assert "category-breadcrumb" not in parent
def test_sponsorship_page_contains_json_ld(self, tmp_path):
readme = textwrap.dedent("""\
# T
## Projects
**Tools**
## Widgets
- [w1](https://example.com/w1) - A widget.
# Contributing
Done.
""")
self._copy_real_templates(tmp_path)
(tmp_path / "README.md").write_text(readme, encoding="utf-8")
build(tmp_path)
site = tmp_path / "website" / "output"
html = (site / "sponsorship" / "index.html").read_text(encoding="utf-8")
parser = HeadMetadataParser()
parser.feed(html)
assert parser.title.strip() == "Sponsor Awesome Python"
assert parser.meta_by_name["description"] == (
"Sponsorship for awesome-python: tiers, audience, and how to get your product in front of professional Python developers evaluating tools for production use."
)
assert parser.links_by_rel["canonical"] == "https://awesome-python.com/sponsorship/"
assert 'Sponsorship' in html
marker = '", start)
graph = {node["@type"]: node for node in json.loads(html[start:end])["@graph"]}
assert set(graph) == {"WebSite", "WebPage", "BreadcrumbList"}
assert graph["WebPage"]["@id"] == "https://awesome-python.com/sponsorship/"
assert graph["WebPage"]["url"] == "https://awesome-python.com/sponsorship/"
assert graph["BreadcrumbList"]["itemListElement"] == [
{"@type": "ListItem", "position": 1, "name": "Awesome Python", "item": "https://awesome-python.com/"},
{"@type": "ListItem", "position": 2, "name": "Sponsorship", "item": "https://awesome-python.com/sponsorship/"},
]
def test_build_creates_group_pages(self, tmp_path):
readme = textwrap.dedent("""\
# T
## Projects
**AI & ML**
## Deep Learning
- [dl1](https://example.com/dl1) - DL.
## Machine Learning
- [ml1](https://example.com/ml1) - ML.
**Web Development**
## Web Frameworks
- [wf1](https://example.com/wf1) - WF.
# Contributing
Done.
""")
self._copy_real_templates(tmp_path)
(tmp_path / "README.md").write_text(readme, encoding="utf-8")
build(tmp_path)
site = tmp_path / "website" / "output"
ai_ml = (site / "categories" / "ai-ml" / "index.html").read_text(encoding="utf-8")
web_dev = (site / "categories" / "web-development" / "index.html").read_text(encoding="utf-8")
assert "dl1" in ai_ml
assert "ml1" in ai_ml
assert "wf1" not in ai_ml
assert "wf1" in web_dev
assert "dl1" not in web_dev
def test_tags_link_to_pages_and_subcategory_anchors(self, tmp_path):
readme = textwrap.dedent("""\
# T
## Projects
**AI & ML**
## Deep Learning
- Vision
- [v1](https://example.com/v1) - Vision lib.
# Contributing
Done.
""")
self._copy_real_templates(tmp_path)
(tmp_path / "README.md").write_text(readme, encoding="utf-8")
build(tmp_path)
site = tmp_path / "website" / "output"
index_html = (site / "index.html").read_text(encoding="utf-8")
category_html = (site / "categories" / "deep-learning" / "index.html").read_text(encoding="utf-8")
for html in (index_html, category_html):
assert 'href="/categories/deep-learning/#vision"' in html
assert 'href="/categories/ai-ml/"' in html
assert "data-url=" not in html
assert 'href="/categories/deep-learning/"' in index_html
assert 'id="vision"' in category_html
_REDIRECT_README = textwrap.dedent("""\
# Awesome Python
Intro.
## Projects
**Tools**
### Widgets
- [w1](https://example.com/w1) - A widget.
## Contributing
Help!
""")
def _write_redirects(self, tmp_path, redirects):
data_dir = tmp_path / "website" / "data"
data_dir.mkdir(parents=True)
(data_dir / "redirects.json").write_text(json.dumps(redirects), encoding="utf-8")
def test_build_writes_redirect_stub_outside_sitemap(self, tmp_path):
self._copy_real_templates(tmp_path)
(tmp_path / "README.md").write_text(self._REDIRECT_README, encoding="utf-8")
self._write_redirects(tmp_path, {"/categories/old-widgets/": "/categories/widgets/"})
build(tmp_path)
site = tmp_path / "website" / "output"
stub = (site / "categories" / "old-widgets" / "index.html").read_text(encoding="utf-8")
assert '' in stub
assert '' in stub
assert '' in stub
assert "old-widgets" not in (site / "sitemap.xml").read_text(encoding="utf-8")
def test_build_renders_category_intro_and_uses_lead_as_meta_description(self, tmp_path):
self._copy_real_templates(tmp_path)
(tmp_path / "README.md").write_text(self._REDIRECT_README, encoding="utf-8")
intros_dir = tmp_path / "website" / "data" / "category_intros"
intros_dir.mkdir(parents=True)
(intros_dir / "widgets.md").write_text("Use `w1` for most apps.\n\nSee [the docs](https://example.com/docs).\n\nHow to choose:\n\n- Small apps: w1\n", encoding="utf-8")
build(tmp_path)
category_html = (tmp_path / "website" / "output" / "categories" / "widgets" / "index.html").read_text(encoding="utf-8")
parser = HeadMetadataParser()
parser.feed(category_html)
assert parser.meta_by_name["description"] == "Use w1 for most apps."
assert '
Use w1 for most apps.
' in category_html
assert "
Small apps: w1
" in category_html
assert 'the docs' in category_html
def test_build_renders_category_guide_below_table(self, tmp_path):
self._copy_real_templates(tmp_path)
(tmp_path / "README.md").write_text(self._REDIRECT_README, encoding="utf-8")
intros_dir = tmp_path / "website" / "data" / "category_intros"
intros_dir.mkdir(parents=True)
(intros_dir / "widgets.md").write_text("Use w1.\n\nHow to choose:\n\n- Small apps: w1\n\nSet up w1 once per process.\n", encoding="utf-8")
build(tmp_path)
category_html = (tmp_path / "website" / "output" / "categories" / "widgets" / "index.html").read_text(encoding="utf-8")
intro_html = category_html.split('
', 1)[1].split("
", 1)[0]
assert "
Small apps: w1
" in intro_html
assert "Set up w1" not in intro_html
guide_html = category_html.split('', 1)[1]
assert "
Widgets guide
" in guide_html
assert "
Set up w1 once per process.
" in guide_html
assert category_html.index('id="guide"') > category_html.index("
")
assert 'Widgets guide' in category_html
def test_section_page_groups_rows_by_use_case_in_readme_order(self, tmp_path):
readme = textwrap.dedent("""\
# T
## Projects
**Tools**
### Widgets
- Small
- [w2](https://example.com/w2) - Second.
- [w1](https://example.com/w1) - First.
- Large
- [w3](https://example.com/w3) - Third.
- [sqlite3](https://docs.python.org/3/library/sqlite3.html) - Stdlib.
# Contributing
Done.
""")
self._copy_real_templates(tmp_path)
(tmp_path / "README.md").write_text(readme, encoding="utf-8")
build(tmp_path)
site = tmp_path / "website" / "output" / "categories"
html = (site / "widgets" / "index.html").read_text(encoding="utf-8")
assert 'data-default-sort="editorial"' in html
positions = [html.index(marker) for marker in ('