Merge branch 'fix/description-category-links' into feature/seo-redesign

This commit is contained in:
Vinta Chen
2026-09-27 09:10:59 +08:00
2 changed files with 52 additions and 0 deletions
+17
View File
@@ -20,6 +20,7 @@ from readme_parser import AlsoSee, ParsedGroup, ParsedSection, parse_readme, par
GITHUB_REPO_URL_RE = re.compile(r"^https?://github\.com/([^/]+/[^/]+?)(?:\.git)?/?$") GITHUB_REPO_URL_RE = re.compile(r"^https?://github\.com/([^/]+/[^/]+?)(?:\.git)?/?$")
MARKDOWN_LINK_RE = re.compile(r"\[([^\]]+)\]\(([^)\s]+)\)") MARKDOWN_LINK_RE = re.compile(r"\[([^\]]+)\]\(([^)\s]+)\)")
BULLET_LINE_RE = re.compile(r"^\s*-\s") BULLET_LINE_RE = re.compile(r"^\s*-\s")
ANCHOR_LINK_ATTRS_RE = re.compile(r'href="(#[^"]*)" target="_blank" rel="noopener"')
SITE_URL = "https://awesome-python.com/" SITE_URL = "https://awesome-python.com/"
SITEMAP_URL = f"{SITE_URL}sitemap.xml" SITEMAP_URL = f"{SITE_URL}sitemap.xml"
SITEMAP_NS = "http://www.sitemaps.org/schemas/sitemap/0.9" SITEMAP_NS = "http://www.sitemaps.org/schemas/sitemap/0.9"
@@ -466,6 +467,21 @@ def link_llms_category_index_to_canonical_pages(markdown: str, categories: Seque
return "".join(out) return "".join(out)
def link_description_anchors_to_category_pages(categories: Sequence[ParsedSection]) -> None:
"""Point README anchor links in section descriptions at category pages, which lack those anchors."""
category_paths = {}
for category in categories:
category_paths[f"#{category['slug']}"] = category_path(category)
category_paths[github_markdown_anchor(category["name"])] = category_path(category)
def replace_attrs(match: re.Match[str]) -> str:
path = category_paths.get(match.group(1))
return f'href="{path}"' if path else match.group(0)
for category in categories:
category["description_html"] = ANCHOR_LINK_ATTRS_RE.sub(replace_attrs, category["description_html"])
def build_llms_txt( def build_llms_txt(
template_text: str, template_text: str,
*, *,
@@ -645,6 +661,7 @@ def build(repo_root: Path) -> None:
duplicates = {s for s, n in Counter(all_top_level_slugs).items() if n > 1} duplicates = {s for s, n in Counter(all_top_level_slugs).items() if n > 1}
if duplicates: if duplicates:
raise ValueError(f"slug collision in /categories/ namespace: {sorted(duplicates)}. Rename a category or group so their slugs differ.") raise ValueError(f"slug collision in /categories/ namespace: {sorted(duplicates)}. Rename a category or group so their slugs differ.")
link_description_anchors_to_category_pages(categories)
total_entries = sum(c["entry_count"] for c in categories) total_entries = sum(c["entry_count"] for c in categories)
entries = extract_entries(categories, parsed_groups) entries = extract_entries(categories, parsed_groups)
build_date = datetime.now(UTC) build_date = datetime.now(UTC)
+35
View File
@@ -295,6 +295,41 @@ class TestBuild:
assert "42" in category_html assert "42" in category_html
assert "2026-01-01T00:00:00+00:00" in category_html assert "2026-01-01T00:00:00+00:00" in category_html
def test_build_links_description_anchors_to_category_pages(self, tmp_path):
readme = textwrap.dedent("""\
# Awesome Python
Intro.
## Projects
**Tools**
## Audio & Video
_Media tools._
- [a1](https://example.com/a1) - A media tool.
## Widgets
_Widget libraries. Also see [Audio & Video](#audio--video) and [Gadgets](#gadgets)._
- [w1](https://example.com/w1) - A widget.
# Contributing
Help!
""")
(tmp_path / "README.md").write_text(readme, encoding="utf-8")
self._copy_real_templates(tmp_path)
build(tmp_path)
category_html = (tmp_path / "website" / "output" / "categories" / "widgets" / "index.html").read_text(encoding="utf-8")
assert 'Also see <a href="/categories/audio-video/">Audio &amp; Video</a>' in category_html
assert '<a href="#gadgets" target="_blank" rel="noopener">Gadgets</a>' in category_html
def test_build_creates_llms_text_alternate_without_sponsors(self, tmp_path): def test_build_creates_llms_text_alternate_without_sponsors(self, tmp_path):
readme = textwrap.dedent("""\ readme = textwrap.dedent("""\
# Awesome Python # Awesome Python