From c10264977ef13b581c196fa417bb14e152074719 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sat, 26 Sep 2026 18:45:53 +0800 Subject: [PATCH 001/104] feat: serve redirect stubs for renamed and dissolved category slugs Renamed or dissolved category slugs (e.g. /categories/web-servers/rpc/, /categories/code-analysis/code-linters/) returned 404, Search Console lists 7 of them, and each audit re-home was dropping the old URL's ranking. Co-Authored-By: Claude --- .gitignore | 1 + website/build.py | 12 ++++++++ website/data/redirects.json | 10 +++++++ website/templates/redirect.html | 11 ++++++++ website/tests/test_build.py | 50 +++++++++++++++++++++++++++++++++ 5 files changed, 84 insertions(+) create mode 100644 website/data/redirects.json create mode 100644 website/templates/redirect.html diff --git a/.gitignore b/.gitignore index ba307609..9f878b42 100644 --- a/.gitignore +++ b/.gitignore @@ -13,6 +13,7 @@ __pycache__/ website/output/ website/data/* !website/data/pypi_name_overrides.json +!website/data/redirects.json # agents .playwright-cli/ diff --git a/website/build.py b/website/build.py index 67346cec..89dbc4a3 100644 --- a/website/build.py +++ b/website/build.py @@ -772,6 +772,18 @@ def build(repo_root: Path) -> None: parent_category=cat_by_slug[cat_slug], ) + redirects_file = website / "data" / "redirects.json" + redirects = json.loads(redirects_file.read_text(encoding="utf-8")) if redirects_file.exists() else {} + for old_path, new_path in redirects.items(): + tpl_redirect = env.get_template("redirect.html") + stub = site_dir / old_path.strip("/") / "index.html" + if stub.exists(): + raise ValueError(f"redirect source {old_path} is a live page; remove it from redirects.json") + if not (site_dir / new_path.strip("/") / "index.html").exists(): + raise ValueError(f"redirect target {new_path} for {old_path} does not exist") + stub.parent.mkdir(parents=True, exist_ok=True) + stub.write_text(tpl_redirect.render(target_url=SITE_URL + new_path.lstrip("/")), encoding="utf-8") + static_src = website / "static" static_dst = site_dir / "static" if static_src.exists(): diff --git a/website/data/redirects.json b/website/data/redirects.json new file mode 100644 index 00000000..af60550e --- /dev/null +++ b/website/data/redirects.json @@ -0,0 +1,10 @@ +{ + "/categories/": "/", + "/categories/ai-and-agents/pre-trained-models-and-inference/": "/categories/ai-and-agents/pre-trained-models/", + "/categories/cli-tools/productivity-tools/": "/categories/cli-tools/", + "/categories/code-analysis/code-linters/": "/categories/code-analysis/linters-and-formatters/", + "/categories/data-analysis/financial-data/": "/categories/data-ingestion-etl/financial-data/", + "/categories/distributed-computing/batch-processing/": "/categories/distributed-computing/", + "/categories/gui-development/terminal/": "/categories/cli-development/tui-frameworks/", + "/categories/web-servers/rpc/": "/categories/web-apis/rpc/" +} diff --git a/website/templates/redirect.html b/website/templates/redirect.html new file mode 100644 index 00000000..292c443c --- /dev/null +++ b/website/templates/redirect.html @@ -0,0 +1,11 @@ + + + +Redirecting… + + + + +

Redirecting…

+Click here if you are not redirected. + diff --git a/website/tests/test_build.py b/website/tests/test_build.py index 2f0af663..a1d11503 100644 --- a/website/tests/test_build.py +++ b/website/tests/test_build.py @@ -955,6 +955,56 @@ class TestBuild: assert 'data-url="/categories/ai-ml/"' in index_html assert 'data-url="/categories/deep-learning/vision/"' in index_html + _REDIRECT_README = textwrap.dedent("""\ + # Awesome Python + + Intro. + + ## Projects + + **Tools** + + ### Widgets + + - [w1](https://example.com/w1) - A widget. + + ## Contributing + + Help! + """) + + def _write_redirects(self, tmp_path, redirects): + data_dir = tmp_path / "website" / "data" + data_dir.mkdir(parents=True) + (data_dir / "redirects.json").write_text(json.dumps(redirects), encoding="utf-8") + + def test_build_writes_redirect_stub_outside_sitemap(self, tmp_path): + self._copy_real_templates(tmp_path) + (tmp_path / "README.md").write_text(self._REDIRECT_README, encoding="utf-8") + self._write_redirects(tmp_path, {"/categories/old-widgets/": "/categories/widgets/"}) + build(tmp_path) + + site = tmp_path / "website" / "output" + stub = (site / "categories" / "old-widgets" / "index.html").read_text(encoding="utf-8") + assert '' in stub + assert '' in stub + assert '' in stub + assert "old-widgets" not in (site / "sitemap.xml").read_text(encoding="utf-8") + + def test_build_rejects_redirect_to_missing_page(self, tmp_path): + self._copy_real_templates(tmp_path) + (tmp_path / "README.md").write_text(self._REDIRECT_README, encoding="utf-8") + self._write_redirects(tmp_path, {"/categories/old-widgets/": "/categories/gone/"}) + with pytest.raises(ValueError, match="does not exist"): + build(tmp_path) + + def test_build_rejects_redirect_over_live_page(self, tmp_path): + self._copy_real_templates(tmp_path) + (tmp_path / "README.md").write_text(self._REDIRECT_README, encoding="utf-8") + self._write_redirects(tmp_path, {"/categories/widgets/": "/"}) + with pytest.raises(ValueError, match="is a live page"): + build(tmp_path) + # --------------------------------------------------------------------------- # extract_github_repo From a74ca52860b7092e96e2af5c84dd0e8e75349cf9 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sat, 26 Sep 2026 19:42:22 +0800 Subject: [PATCH 002/104] feat: render per-category intro text above the entry list Category pages carried no text of their own beyond the README one-line description, and most meta descriptions fell back to a generic "Explore N curated Python projects" line, which correlated with weak search rankings for category queries. This adds optional per-category intro markdown files rendered under the H1, with the first paragraph used as the meta description and links opening in a new tab, starting with the ORM category. Co-Authored-By: Claude --- .gitignore | 1 + website/build.py | 26 ++++++++++++++++++++++-- website/data/category_intros/orm.md | 20 +++++++++++++++++++ website/static/style.css | 31 +++++++++++++++++++++++++++-- website/templates/category.html | 3 +++ website/tests/test_build.py | 16 +++++++++++++++ 6 files changed, 93 insertions(+), 4 deletions(-) create mode 100644 website/data/category_intros/orm.md diff --git a/.gitignore b/.gitignore index 9f878b42..de692147 100644 --- a/.gitignore +++ b/.gitignore @@ -14,6 +14,7 @@ website/output/ website/data/* !website/data/pypi_name_overrides.json !website/data/redirects.json +!website/data/category_intros/ # agents .playwright-cli/ diff --git a/website/build.py b/website/build.py index 89dbc4a3..46985670 100644 --- a/website/build.py +++ b/website/build.py @@ -13,7 +13,9 @@ from typing import TypedDict from fetch_pypi_downloads_via_clickpy import OVERRIDES_FILE, normalize from jinja2 import Environment, FileSystemLoader -from readme_parser import AlsoSee, ParsedGroup, ParsedSection, parse_readme, parse_sponsors, slugify +from markdown_it import MarkdownIt +from markdown_it.tree import SyntaxTreeNode +from readme_parser import AlsoSee, ParsedGroup, ParsedSection, parse_readme, parse_sponsors, render_inline_text, slugify GITHUB_REPO_URL_RE = re.compile(r"^https?://github\.com/([^/]+/[^/]+?)(?:\.git)?/?$") MARKDOWN_LINK_RE = re.compile(r"\[([^\]]+)\]\(([^)\s]+)\)") @@ -235,6 +237,24 @@ def category_meta_description(name: str, entry_count: int, description: str, par return f"{count_sentence} Part of the Awesome Python catalog." +def load_category_intro(path: Path) -> tuple[str, str]: + """Render a category intro file to HTML, plus its first paragraph as plain text for the meta description. + + Returns empty strings if the category has no intro file. + """ + if not path.exists(): + return "", "" + md = MarkdownIt("commonmark") + tokens = md.parse(path.read_text(encoding="utf-8")) + for token in tokens: + for child in token.children or []: + if child.type == "link_open": + child.attrSet("target", "_blank") + child.attrSet("rel", "noopener") + lead = next(node for node in SyntaxTreeNode(tokens).children if node.type == "paragraph") + return md.renderer.render(tokens, md.options, {}), render_inline_text(lead.children[0].children) + + def build_breadcrumb_json_ld(items: Sequence[tuple[str, str]]) -> dict: return { "@type": "BreadcrumbList", @@ -676,7 +696,8 @@ def build(repo_root: Path) -> None: page_dir.mkdir(parents=True, exist_ok=True) parent_name = parent_category["name"] if parent_category else None category_title = category_meta_title(category["name"], parent_name) - category_description = category_meta_description(category["name"], len(entries), category["description"], parent_name) + intro_html, intro_lead = load_category_intro(website / "data" / "category_intros" / f"{current_path.removeprefix('/categories/').strip('/')}.md") + category_description = intro_lead or category_meta_description(category["name"], len(entries), category["description"], parent_name) breadcrumbs = [("Awesome Python", SITE_URL)] if parent_category: breadcrumbs.append((parent_category["name"], category_public_url(parent_category))) @@ -691,6 +712,7 @@ def build(repo_root: Path) -> None: category_title=category_title, category_url=category_url, category_description=category_description, + intro_html=intro_html, entries=entries, total_categories=len(categories), category_urls=category_urls, diff --git a/website/data/category_intros/orm.md b/website/data/category_intros/orm.md new file mode 100644 index 00000000..8b669ef1 --- /dev/null +++ b/website/data/category_intros/orm.md @@ -0,0 +1,20 @@ +Use SQLAlchemy for most projects, and the Django ORM inside Django. With FastAPI, use SQLModel, so your tables and API schemas share the same fields. + +How to choose: + +- Any framework, or full control over the SQL: SQLAlchemy +- A Django project: the Django ORM, which needs Django's settings even outside a web app +- A FastAPI app: SQLModel +- A simple ORM with few concepts to learn: peewee +- MongoDB: Beanie for async code, MongoEngine for sync code +- DynamoDB: PynamoDB + +With SQLAlchemy, write [typed models](https://docs.sqlalchemy.org/en/latest/orm/declarative_styles.html) with `DeclarativeBase`, `Mapped[]`, and `mapped_column()`. Create one `sessionmaker` at startup and open one session per request. Load relationships with `selectinload()`, and run migrations with Alembic, which comes from the SQLAlchemy project itself. + +Sync code is the safe default, even in an async framework. FastAPI's own docs put it plainly: ["If you just don't know, use normal `def`."](https://fastapi.tiangolo.com/async/) If you do go async, give each task its own `AsyncSession`. + +In Django, let `makemigrations` and `migrate` own the schema, and use `select_related()` or `prefetch_related()` whenever you touch related rows. Reach for raw SQL last: ["Explore the ORM before using raw SQL!"](https://docs.djangoproject.com/en/stable/topics/db/sql/) + +With SQLModel, follow its FastAPI tutorial: a base model for the shared fields, a `table=True` model for the database, and [separate models for create, read, and update](https://sqlmodel.tiangolo.com/tutorial/fastapi/multiple-models/). When you outgrow it, plug SQLAlchemy in directly. + +For MongoDB and DynamoDB, model around your queries, not your tables. MongoDB's rule is that ["data that's accessed together should be stored together."](https://www.mongodb.com/docs/manual/core/data-modeling-introduction/) AWS goes further: [don't design a DynamoDB schema](https://docs.aws.amazon.com/amazondynamodb/latest/developerguide/bp-general-nosql-design.html) until you know the questions it needs to answer. diff --git a/website/static/style.css b/website/static/style.css index 2313c010..bdad3835 100644 --- a/website/static/style.css +++ b/website/static/style.css @@ -496,17 +496,44 @@ kbd { text-wrap: pretty; } -.category-subtitle a { +.category-subtitle a, +.category-intro a { color: var(--hero-text); text-decoration: underline; text-decoration-color: oklch(100% 0 0 / 0.32); text-underline-offset: 0.2em; } -.category-subtitle a:hover { +.category-subtitle a:hover, +.category-intro a:hover { text-decoration-color: oklch(100% 0 0 / 0.7); } +.category-intro { + margin-top: 1.75rem; + display: grid; + gap: 0.9rem; + color: var(--hero-muted); + font-size: clamp(1.05rem, 1.8vw, 1.2rem); + text-wrap: pretty; +} + +.category-intro > p:first-child { + color: var(--hero-text); + font-size: clamp(1.2rem, 2.2vw, 1.45rem); + line-height: 1.45; +} + +.category-intro ul { + display: grid; + gap: 0.35rem; + padding-left: 1.25rem; +} + +.category-intro code { + font-size: 0.9em; +} + .sponsor-band { padding-block: clamp(2.5rem, 5.5vw, 4rem); background: diff --git a/website/templates/category.html b/website/templates/category.html index 0a868bc2..7325aa2b 100644 --- a/website/templates/category.html +++ b/website/templates/category.html @@ -37,6 +37,9 @@ {% if category.description_html %}

{{ category.description_html | safe }}

{% endif %} + {% if intro_html %} +
{{ intro_html | safe }}
+ {% endif %} {% if group_categories %} diff --git a/website/tests/test_build.py b/website/tests/test_build.py index a1d11503..7998525f 100644 --- a/website/tests/test_build.py +++ b/website/tests/test_build.py @@ -991,6 +991,22 @@ class TestBuild: assert '' in stub assert "old-widgets" not in (site / "sitemap.xml").read_text(encoding="utf-8") + def test_build_renders_category_intro_and_uses_lead_as_meta_description(self, tmp_path): + self._copy_real_templates(tmp_path) + (tmp_path / "README.md").write_text(self._REDIRECT_README, encoding="utf-8") + intros_dir = tmp_path / "website" / "data" / "category_intros" + intros_dir.mkdir(parents=True) + (intros_dir / "widgets.md").write_text("Use `w1` for most apps.\n\nSee [the docs](https://example.com/docs).\n\nHow to choose:\n\n- Small apps: w1\n", encoding="utf-8") + build(tmp_path) + + category_html = (tmp_path / "website" / "output" / "categories" / "widgets" / "index.html").read_text(encoding="utf-8") + parser = HeadMetadataParser() + parser.feed(category_html) + assert parser.meta_by_name["description"] == "Use w1 for most apps." + assert '

Use w1 for most apps.

' in category_html + assert "
  • Small apps: w1
  • " in category_html + assert 'the docs' in category_html + def test_build_rejects_redirect_to_missing_page(self, tmp_path): self._copy_real_templates(tmp_path) (tmp_path / "README.md").write_text(self._REDIRECT_README, encoding="utf-8") From 5b9bc45c2b1878367b534c83e3d7c0c9cf4e86a3 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sat, 26 Sep 2026 19:42:33 +0800 Subject: [PATCH 003/104] docs: require a redirect entry when an audit renames or dissolves a category Audits that rename or dissolve a category never added a redirect, which is how the 404s appearing in Search Console traced back to renamed category pages. Co-Authored-By: Claude --- .claude/skills/audit-the-list/SKILL.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.claude/skills/audit-the-list/SKILL.md b/.claude/skills/audit-the-list/SKILL.md index bb008997..ccb433e2 100644 --- a/.claude/skills/audit-the-list/SKILL.md +++ b/.claude/skills/audit-the-list/SKILL.md @@ -33,7 +33,7 @@ Run the `preview-verdicts` skill: it generates the interactive review page and d ## 5. Execute -One commit per section: body lists each removal with its reason and downloads figure; restructures, tier moves, and reorders ride the same commit. Format-only outcomes (no removals) are a single style commit. `make test` before every commit, `make build` after the last one. Generic commit helpers tend to split a section audit into structural and per-subcategory commits — if that happens, squash back to one commit per section. Done when the tree is clean, tests passed before each commit, and the build count reconciles with the adjudicated changes. +One commit per section: body lists each removal with its reason and downloads figure; restructures, tier moves, and reorders ride the same commit. A restructure that renames or dissolves a category or subcategory adds `"old path": "new path"` to `website/data/redirects.json` in that commit, so the old URL keeps its search ranking; the build fails when a redirect target no longer exists, so a later rename also updates the entries pointing at the renamed path. Format-only outcomes (no removals) are a single style commit. `make test` before every commit, `make build` after the last one. Generic commit helpers tend to split a section audit into structural and per-subcategory commits — if that happens, squash back to one commit per section. Done when the tree is clean, tests passed before each commit, and the build count reconciles with the adjudicated changes. ## 6. Record From 8ea48cbecdfbe5080afb84b742af0fc79b3a7f75 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sat, 26 Sep 2026 19:43:59 +0800 Subject: [PATCH 004/104] build: add Claude Code launch config for website preview Adds a "website" launch config that runs `make preview` (build, rebuild on file changes, serve website/output), so the desktop app's preview pane can start the site in one step. Co-Authored-By: Claude --- .claude/launch.json | 11 +++++++++++ 1 file changed, 11 insertions(+) create mode 100644 .claude/launch.json diff --git a/.claude/launch.json b/.claude/launch.json new file mode 100644 index 00000000..de8c1ce1 --- /dev/null +++ b/.claude/launch.json @@ -0,0 +1,11 @@ +{ + "version": "0.0.1", + "configurations": [ + { + "name": "website", + "runtimeExecutable": "make", + "runtimeArgs": ["preview"], + "port": 8000 + } + ] +} From fd300c44537580e294d86d05619ecafb5edaa67b Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sat, 26 Sep 2026 20:09:42 +0800 Subject: [PATCH 005/104] docs: add Testing category intro Recommends pytest, Hypothesis, Playwright, and tox/nox based on each project's own docs. Co-Authored-By: Claude --- website/data/category_intros/testing.md | 28 +++++++++++++++++++++++++ 1 file changed, 28 insertions(+) create mode 100644 website/data/category_intros/testing.md diff --git a/website/data/category_intros/testing.md b/website/data/category_intros/testing.md new file mode 100644 index 00000000..e7034229 --- /dev/null +++ b/website/data/category_intros/testing.md @@ -0,0 +1,28 @@ +Use pytest as your Python testing framework. Add Hypothesis to find the edge cases you missed, and Playwright to test in a real browser. + +How to choose: + +- Unit and integration tests: pytest +- Property-based tests: Hypothesis +- Running the suite across Python versions or dependency sets: tox, or nox if you'd rather configure it in Python +- End-to-end browser tests: Playwright +- An existing Selenium suite, or tests spread across many machines with Selenium Grid: Selenium, or SeleniumBase for a pytest-ready framework on top of it +- Acceptance tests in a readable keyword syntax: Robot Framework +- Load testing: Locust +- An API with an OpenAPI or GraphQL schema: Schemathesis +- Faking HTTP: responses for Requests, RESPX for HTTPX, VCR.py to record and replay real traffic +- Freezing the clock: freezegun +- Test data: factory_boy for ORM models, Polyfactory for dataclasses and Pydantic models, Faker or Mimesis for single fake values +- Coverage: coverage.py + +With pytest, write tests as plain functions with `assert`, and share setup through fixtures, which pytest says [offer dramatic improvements](https://docs.pytest.org/en/stable/explanation/fixtures.html) over xUnit-style setup and teardown. For a new project, its good practices recommend [the `importlib` import mode](https://docs.pytest.org/en/stable/explanation/goodpractices.html) and a `src` layout. You don't have to rewrite an old unittest suite first: pytest [runs it as is](https://docs.pytest.org/en/stable/how-to/unittest.html), so you can move it over one file at a time. + +A Hypothesis test is a pytest test with `@given` on top. You describe the inputs with strategies, and Hypothesis picks the values, [including edge cases you might not have thought about](https://hypothesis.readthedocs.io/en/latest/). It still works with fixtures and `parametrize`. + +For the browser, Playwright calls its pytest plugin, pytest-playwright, [the recommended way to write end-to-end tests](https://playwright.dev/python/docs/intro). Find elements by what the user sees, [starting with `get_by_role()`](https://playwright.dev/python/docs/locators), and assert with `expect()`, which keeps retrying until the condition is met or it times out. + +Run tox or nox locally and in CI. pytest's own docs point to tox because it [tests the installed package, not your checkout](https://docs.pytest.org/en/stable/explanation/goodpractices.html), which catches packaging mistakes. + +For coverage, run `coverage run --branch -m pytest`. You don't need a pytest plugin for it: coverage.py calls one [unnecessary for most purposes](https://coverage.readthedocs.io/en/latest/). + +When you mock with `unittest.mock`, [patch where an object is looked up](https://docs.python.org/3/library/unittest.mock.html#where-to-patch), not where it's defined. Pass `autospec=True` too, so a call with the wrong signature raises a `TypeError`. From 7d19f84e63a43646a1e22bda5ae272bd218043e6 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sat, 26 Sep 2026 20:10:14 +0800 Subject: [PATCH 006/104] docs: add Data Validation category intro Co-Authored-By: Claude --- website/data/category_intros/data-validation.md | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) create mode 100644 website/data/category_intros/data-validation.md diff --git a/website/data/category_intros/data-validation.md b/website/data/category_intros/data-validation.md new file mode 100644 index 00000000..3729ae2a --- /dev/null +++ b/website/data/category_intros/data-validation.md @@ -0,0 +1,17 @@ +For a Python validation library, use Pydantic for data coming into your app, jsonschema when a JSON Schema is the contract, and Pandera for dataframes. + +How to choose: + +- API payloads, config files, anything you can describe with type hints: Pydantic +- A JSON Schema shared with other languages or teams: jsonschema +- pandas, Polars, or PySpark dataframes: Pandera + +With Pydantic, describe your data as a `BaseModel` with type hints. Parse JSON with `model_validate_json()`, not `model_validate(json.loads(...))`, as [its performance tips](https://pydantic.dev/docs/validation/latest/concepts/performance/) recommend. For a type that isn't a model, like `list[Item]`, create one `TypeAdapter` and reuse it. + +Pydantic is forgiving by default: it turns `"123"` into `123` and ignores fields your model doesn't declare. When that's too loose, turn on [strict mode](https://pydantic.dev/docs/validation/latest/concepts/strict_mode/) or set `extra='forbid'`. + +With jsonschema, `validate()` checks the schema itself on every call. To validate many documents against one schema, [create a validator once](https://python-jsonschema.readthedocs.io/en/stable/validate/) and reuse it, like `Draft202012Validator(schema)`. The `format` keyword checks nothing until you pass a `format_checker`, and [`default` doesn't fill in missing fields](https://python-jsonschema.readthedocs.io/en/stable/faq/). + +With Pandera, write a `DataFrameModel`, which [works like a Pydantic model](https://pandera.readthedocs.io/en/stable/dataframe_models.html) for your columns. Decorate pipeline functions with `@pa.check_types` to validate what goes in and what comes out. Pass `lazy=True` to [get every failure in one report](https://pandera.readthedocs.io/en/stable/lazy_validation.html), not just the first. + +Validate once, where untrusted data enters your program: a request body, a config file, a CSV upload. After that, [Pydantic guarantees](https://pydantic.dev/docs/validation/latest/concepts/models/) the fields match their types, so don't check them again deeper in. The same goes for schemas: if other teams need a JSON Schema, [generate it from your Pydantic models](https://pydantic.dev/docs/validation/latest/concepts/json_schema/) instead of writing it twice. From ec54348e4ca6b84e42041697dbb5e42a5f5d4927 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sat, 26 Sep 2026 20:10:52 +0800 Subject: [PATCH 007/104] docs: add CLI Development category intro Covers recommended usage from each project's own docs for Typer, argparse, Click, Rich, and Textual. Co-Authored-By: Claude --- .../data/category_intros/cli-development.md | 26 +++++++++++++++++++ 1 file changed, 26 insertions(+) create mode 100644 website/data/category_intros/cli-development.md diff --git a/website/data/category_intros/cli-development.md b/website/data/category_intros/cli-development.md new file mode 100644 index 00000000..018debf9 --- /dev/null +++ b/website/data/category_intros/cli-development.md @@ -0,0 +1,26 @@ +For a Python CLI framework, use Typer, or argparse when a script can't take dependencies. For a full-screen terminal app, use Textual. + +How to choose: + +- A new CLI: Typer +- A script that runs on the standard library alone: argparse +- Decorators instead of type hints, or subcommands loaded lazily at runtime: Click +- A quick CLI over an existing function or class, for debugging or exploring code: Fire +- An interactive prompt with completion, history, and key bindings: prompt_toolkit +- Colored output, tables, and pretty logs: Rich +- A progress bar on a loop: tqdm, or alive-progress for animated bars +- A full-screen terminal UI: Textual, or urwid inside an event loop you already run, such as Twisted or Trio +- ASCII animations and effects: asciimatics +- ANSI colors on Windows consoles: colorama + +With Typer, declare each parameter with `Annotated`. Its tutorial keeps repeating one line: ["Prefer to use the `Annotated` version if possible."](https://typer.tiangolo.com/tutorial/arguments/optional/) Create one `typer.Typer()` app and register your commands on it. Test them with `CliRunner`, which calls the app with a list of arguments and gives you back the exit code and output. + +Typer builds on Click, so reach for Click itself when you'd rather write decorators, or when your app has so many subcommands that they should [load lazily at runtime](https://click.palletsprojects.com/en/stable/). Group commands with `@click.group()`, and test them with `click.testing.CliRunner`. + +Use argparse when the tool has to run without installing anything. Python's own docs call it [the recommended choice](https://docs.python.org/3/library/optparse.html#choosing-an-argument-parser) among the standard library's parsers, and `add_subparsers()` handles subcommands. + +With Rich, create one `Console` and share it across the app: ["Most applications will require a single Console instance"](https://rich.readthedocs.io/en/stable/console.html). Add a second one with `stderr=True` for errors, and pass `RichHandler` to `logging.basicConfig()` so your logs get the same formatting. + +With Textual, keep styles in a separate `.tcss` file and run `textual run --dev` to [live edit them](https://textual.textualize.io/guide/CSS/). For tests, `run_test()` starts the app headless and gives you a `Pilot` that presses keys and clicks widgets. + +Whatever you pick, ship the CLI as a package, not a script. Click's docs put it plainly: ["It's recommended to write command line utilities as installable packages with entry points instead of telling users to run `python hello.py`."](https://click.palletsprojects.com/en/stable/entry-points/) Declare the command under `[project.scripts]` in `pyproject.toml`, and users can install it with pipx or `uv tool install`. From b71151a8080e977771bcddad8a8ce169eef660d7 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sat, 26 Sep 2026 20:40:29 +0800 Subject: [PATCH 008/104] docs: add Computer Vision category intro Co-Authored-By: Claude --- .../data/category_intros/computer-vision.md | 24 +++++++++++++++++++ 1 file changed, 24 insertions(+) create mode 100644 website/data/category_intros/computer-vision.md diff --git a/website/data/category_intros/computer-vision.md b/website/data/category_intros/computer-vision.md new file mode 100644 index 00000000..048fe971 --- /dev/null +++ b/website/data/category_intros/computer-vision.md @@ -0,0 +1,24 @@ +For a Python computer vision library, start with OpenCV, and add Ultralytics YOLO to detect objects. For OCR, use pytesseract for scans and EasyOCR for photos. + +How to choose: + +- Reading, transforming, and writing images and video: OpenCV +- Detecting, segmenting, or tracking objects with a pretrained model: Ultralytics YOLO, which is AGPL-3.0 unless you buy an Enterprise License +- Image processing and augmentation on GPU batches, inside a PyTorch model or training loop: Kornia +- Browsing a dataset, finding label mistakes, and seeing where a model fails: FiftyOne +- Scanned pages and documents: pytesseract, on top of a Tesseract engine you install yourself +- Text in photos, without a separate OCR engine to install: EasyOCR + +OpenCV ships as four wheels on PyPI, and you install exactly one, since they all use the same `cv2` namespace. On a server or in Docker, pick `opencv-python-headless`, which leaves out the GUI libraries. The opencv-python README says [you should always use it](https://github.com/opencv/opencv-python#installation-and-usage) unless you call `cv2.imshow`. The wheels are CPU-only; for CUDA, you build OpenCV from source. + +With Ultralytics YOLO, load a pretrained checkpoint with `YOLO()`, fine-tune it with `model.train()`, and ship it with `model.export(format="onnx")`. Check the license before you build a product on it. Ultralytics says an Enterprise License is ["required if you want to use Ultralytics YOLO without open-sourcing your entire project"](https://www.ultralytics.com/license), even for models you trained yourself. + +Kornia works on PyTorch tensors, so its operators ["run wherever the tensor lives"](https://kornia.readthedocs.io/en/latest/) and take a whole batch in one call. Gradients flow through them too, so they can sit inside your model or your loss. Its augmentations expect [float tensors in [0, 1]](https://kornia.readthedocs.io/en/latest/augmentation.html): one still in [0, 255] doesn't raise an error, it just clips. + +With FiftyOne, load your images and your model's predictions into a `fo.Dataset` and browse them with `fo.launch_app()`. Then run `compute_mistakenness()` to rank likely label mistakes, which [create an artificial ceiling](https://docs.voxel51.com/brain/index.html) on how good your model can get. Before you trust a test score, run `compute_leaky_splits()` too: duplicates across train and test make it easy to [overestimate your model](https://docs.voxel51.com/brain/index.html#leaky-splits). + +pytesseract wraps the `tesseract` command, so install Tesseract separately and make sure it's on your `PATH`. Pass `lang=` for anything but English. Tesseract works best at 300 DPI or more, so scale small images up. It also [expects a page of text](https://tesseract-ocr.github.io/tessdoc/ImproveQuality.html) by default, so for a single line, pass `config="--psm 7"`. + +With EasyOCR, create one `easyocr.Reader` and reuse it: loading the model takes a while, and it [needs to be run only once](https://github.com/JaidedAI/EasyOCR#usage). + +Watch the channel order when you pass images between libraries. OpenCV [loads color images as BGR](https://docs.opencv.org/4.x/d4/da8/group__imgcodecs.html). Ultralytics YOLO and EasyOCR take OpenCV images as they are, but pytesseract assumes RGB, so convert first with `cv2.cvtColor(img, cv2.COLOR_BGR2RGB)`. From 3870b0a7238c20dc31ed439e93c6a65213edcaae Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sat, 26 Sep 2026 20:40:53 +0800 Subject: [PATCH 009/104] docs: add Audio & Video Processing category intro Co-Authored-By: Claude --- .../category_intros/audio-video-processing.md | 25 +++++++++++++++++++ 1 file changed, 25 insertions(+) create mode 100644 website/data/category_intros/audio-video-processing.md diff --git a/website/data/category_intros/audio-video-processing.md b/website/data/category_intros/audio-video-processing.md new file mode 100644 index 00000000..55c34c36 --- /dev/null +++ b/website/data/category_intros/audio-video-processing.md @@ -0,0 +1,25 @@ +For a Python audio processing library, use librosa to analyze audio and music. For video, use MoviePy to edit clips in code, or VidGear for live streams. + +How to choose: + +- Audio and music analysis, such as beat tracking or spectrograms: librosa +- Quick cuts, fades, and format conversion: pydub +- Editing video in code, such as cuts, titles, and compositing: MoviePy +- Webcams, live streams, and screen capture in real time: VidGear +- Reading tags, with one API for every format and an MIT license: tinytag +- Writing tags: Mutagen, which is GPL-licensed +- Organizing a whole music collection and fixing its tags from MusicBrainz: beets + +With librosa, `librosa.load()` resamples everything to 22050 Hz by default. Pass `sr=None` to [keep the file's native sampling rate](https://librosa.org/doc/latest/api/generated/librosa.load.html), or `offset` and `duration` to load only part of a file. + +With pydub, load a file with `AudioSegment.from_file()`, slice it in milliseconds, and save it with `export()`. WAV works in pure Python, but every other format [needs FFmpeg](https://github.com/jiaaro/pydub#dependencies). The standard library dropped the `audioop` module pydub imports, so [install `audioop-lts`](https://github.com/jiaaro/pydub/issues/863) next to it. + +A MoviePy script loads clips, changes them with `subclipped()` and the `with_*` methods, and calls `write_videofile()`. Open file clips in a `with` block, since each file clip [locks the file](https://zulko.github.io/moviepy/user_guide/loading.html) until it's closed. To only convert a video file, MoviePy's docs tell you to call FFmpeg directly: it's ["faster and more memory-efficient"](https://zulko.github.io/moviepy/getting_started/quick_presentation.html). + +MoviePy can't stream video, such as reading from a webcam, so use VidGear for that. It [needs OpenCV](https://abhitronix.github.io/vidgear/latest/installation/pip_install/) installed first. Its gears keep OpenCV's read loop: `CamGear(source=0).start()`, call `read()` until it returns `None`, then `stop()`. + +tinytag only reads tags: ["Support for changing/writing metadata will not be added."](https://github.com/tinytag/tinytag) To write them, use Mutagen. `mutagen.File()` guesses the format for you, and Mutagen [saves ID3 tags as v2.4](https://mutagen.readthedocs.io/en/latest/user/id3.html) unless you open and save them with `v2_version=3`. + +beets runs from the command line: set `directory` and `library` in its config file, then run `beet import`. Its getting started guide recommends the autotagger and [importing a few albums at a time](https://beets.readthedocs.io/en/stable/guides/main.html). It also warns that importing can modify and move your files, so back them up first. + +For long files, `librosa.stream()` reads audio [block by block](https://librosa.org/doc/latest/api/generated/librosa.stream.html) instead of loading it all into memory; set `center=False` in the analyses you run on those blocks. pydub [loads the whole file into RAM](https://github.com/jiaaro/pydub/issues/51#issuecomment-35839094), so pass `start_second` and `duration` to `from_file()` when you only need a slice. From 1dfa95834d16d702a621ed6c0930363fc8232bf5 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sat, 26 Sep 2026 20:41:18 +0800 Subject: [PATCH 010/104] docs: add GUI Development category intro Co-Authored-By: Claude --- .../data/category_intros/gui-development.md | 28 +++++++++++++++++++ 1 file changed, 28 insertions(+) create mode 100644 website/data/category_intros/gui-development.md diff --git a/website/data/category_intros/gui-development.md b/website/data/category_intros/gui-development.md new file mode 100644 index 00000000..8e393325 --- /dev/null +++ b/website/data/category_intros/gui-development.md @@ -0,0 +1,28 @@ +For a Python GUI library, use PySide6 for desktop apps, or tkinter to stay on the standard library. For a UI that runs in the browser, use NiceGUI. + +How to choose: + +- A full desktop app: PySide6, or PyQt6 for a codebase already on it +- A small tool on the standard library alone: tkinter +- A modern look on top of tkinter: CustomTkinter, or Tkinter Designer to turn a Figma design into tkinter code +- A GNOME or GTK app: PyGObject +- Native widgets on Windows, macOS, and Linux: wxPython +- Native widgets on desktop and mobile: Toga +- A touch-first app for desktop and mobile: Kivy +- Fast, interactive tools with lots of plots: Dear PyGui +- A UI in the browser, or in its own desktop window: NiceGUI +- An HTML, CSS, and JavaScript frontend in a native window: pywebview +- Web, desktop, and mobile apps from one codebase: Flet +- A GUI for an existing argparse script: Gooey + +With PySide6, build desktop apps with Qt Widgets, which [respect the system style](https://doc.qt.io/qtforpython-6/faq/whatisqt.html), and save Qt Quick for fluid, touch-style UIs. Lay out windows in Qt Widgets Designer (`pyside6-designer`), then turn each `.ui` file into a Python class with `pyside6-uic`, which the docs call [the standard way](https://doc.qt.io/qtforpython-6/tutorials/basictutorial/uifiles.html) to use it. To ship the app, the docs recommend `pyside6-deploy` over PyInstaller and similar tools, since it's ["easier to use and also to get the most optimized executable."](https://doc.qt.io/qtforpython-6/deployment/index.html) + +PySide6 and PyQt6 both bind Qt, so the license usually decides. PySide6 is available under the LGPLv3/GPLv3 and a commercial license. PyQt6 is GPLv3 or commercial, and in Riverbank's words: ["Unlike Qt, PyQt is not available under the LGPL."](https://www.riverbankcomputing.com/software/pyqt/) If your app can't be GPL, PyQt6 means buying a license. + +With tkinter, import `tkinter.ttk` too and use its themed widgets. Be careful with tutorials: the docs warn that [most documentation you will find online still uses the old API](https://docs.python.org/3/library/tkinter.html) and can be woefully outdated. CustomTkinter adds modern, customizable widgets on top of tkinter, with [a consistent look across all desktop platforms](https://customtkinter.tomschimansky.com/). + +NiceGUI serves your UI to the browser by default. Pass `native=True` to `ui.run()` and it [opens in a desktop window](https://nicegui.io/documentation/section_configuration_deployment#native_mode) instead, through pywebview. Use pywebview directly for your own HTML frontend, and pass a Python object as `js_api` to [call its methods from JavaScript](https://pywebview.flowrl.com/guide/interdomain) as `pywebview.api`. + +Toga, Kivy, and Flet each have their own packager. Toga apps ship with Briefcase, BeeWare's tool for [turning a Python project into a standalone native app](https://briefcase.beeware.org/en/stable/). Kivy [recommends Buildozer](https://kivy.org/doc/stable/guide/packaging-android.html) for Android, and Flet has [`flet build`](https://flet.dev/docs/publish/). + +In any toolkit, keep slow work out of event handlers, or the UI stops responding. The tkinter docs say [long-running computations](https://docs.python.org/3/library/tkinter.html#threading-model) belong in smaller pieces on a timer or in another thread, not in a handler. Run the slow work with Qt's threads and signals, or NiceGUI's `run.io_bound()` and `run.cpu_bound()`. In wxPython and Kivy, hand the result back to the UI thread with `wx.CallAfter()` or `@mainthread`. From 17bb6270d632c3fca02fcac32f033065c2532276 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sat, 26 Sep 2026 22:09:10 +0800 Subject: [PATCH 011/104] docs: add AI and Agents category intro The AI and Agents category page had no intro, so readers got a 35-project list with no guidance on which library to pick for building an agent, serving a model, or fine-tuning. Co-Authored-By: Claude --- website/data/category_intros/ai-and-agents.md | 58 +++++++++++++++++++ 1 file changed, 58 insertions(+) create mode 100644 website/data/category_intros/ai-and-agents.md diff --git a/website/data/category_intros/ai-and-agents.md b/website/data/category_intros/ai-and-agents.md new file mode 100644 index 00000000..6b1b3ee8 --- /dev/null +++ b/website/data/category_intros/ai-and-agents.md @@ -0,0 +1,58 @@ +LangChain is the place to start among Python libraries for AI agents, and LangGraph gives you control of every step. vLLM serves your own models. + +How to choose: + +- A first agent, or a prebuilt tool-calling loop: LangChain +- Long-running, stateful agents that mix fixed steps with LLM-driven ones: LangGraph +- Typed agents whose outputs are validated: Pydantic AI +- A team of role-playing agents: CrewAI +- An agent built on one vendor's platform: OpenAI Agents SDK or Claude Agent SDK +- Structured data from an LLM, without an agent framework: Instructor +- Prompts tuned against a metric instead of by hand: DSPy +- RAG over your own documents: LlamaIndex +- Memory that survives across sessions: Mem0 +- Agent context you can browse and edit like files: OpenViking +- A knowledge graph with provenance for regulated domains: Semantica +- Running pre-trained models: Transformers +- Serving a model on GPUs: vLLM, or SGLang when requests share long prompts +- Running a model on Apple silicon: MLX LM +- One API for many LLM providers: LiteLLM +- Image and video generation: Diffusers +- Fine-tuning: PEFT for adapters, Unsloth for fast low-memory training, Axolotl for YAML-configured runs across GPUs +- Speech to text: Whisper, or FunASR for streaming and edge deployment +- Text to speech: Kitten TTS on CPU, gTTS for a quick online voice +- Speech research: VibeVoice +- A ready-made personal assistant: Hermes Agent, or AstrBot for chat apps like Telegram, Slack, and QQ +- Skills for your coding agent: Django AI Skills for Django, Sentry Skills for code review, Trail of Bits Skills for security work + +New to agents? LangGraph's own docs [recommend LangChain's prebuilt agents](https://docs.langchain.com/oss/python/langgraph/overview), which run on LangGraph: give an agent a model, tools, and a prompt, and the loop is handled for you. Drop down to LangGraph for [needs that combine deterministic and agentic workflows](https://docs.langchain.com/oss/python/langchain/overview). You don't need LangChain to use LangGraph. + +Pydantic AI is the pick when you want your type checker to cover the agent too. Give the agent an output type, and [every run comes back as a validated Pydantic model](https://pydantic.dev/docs/ai/overview/); when validation fails, the model is asked to try again. Tools and instructions get their dependencies through typed injection, so you can swap in a test double in unit tests. + +CrewAI splits the work into Crews, teams of role-playing agents, and Flows, event-driven workflows that hold state. For production apps, its docs recommend [starting with a Flow](https://docs.crewai.com/en/concepts/production-architecture) and calling Crews from it. + +OpenAI Agents SDK keeps [the primitives few](https://openai.github.io/openai-agents-python/): agents, handoffs, and guardrails, with tracing built in. It also runs [non-OpenAI models](https://openai.github.io/openai-agents-python/models/). Claude Agent SDK runs [Claude Code as a library](https://code.claude.com/docs/en/agent-sdk/overview): the same built-in tools, permissions, sessions, and hooks, inside your own process. + +Instructor gets validated data out of an LLM into a Pydantic model, with retries when validation fails. Its own docs draw the line: [Instructor for extraction, Pydantic AI for agents](https://python.useinstructor.com/). + +DSPy has you [write signatures, not prompts](https://dspy.ai/). Give it examples and a metric, and its optimizers tune the prompts for you. + +LlamaIndex is a [data framework](https://github.com/run-llama/llama_index) for LLM apps: it loads, indexes, and queries your documents. Install `llama-index` to start, or `llama-index-core` plus only the integrations you need. + +Mem0 adds [memory that persists across sessions](https://docs.mem0.ai/). Self-host the open-source version, or use the managed platform. OpenViking is [AGPL-licensed](https://github.com/volcengine/OpenViking/blob/main/LICENSE), where Mem0 is Apache 2.0. + +Transformers runs pre-trained models from the Hugging Face Hub. Start with [`pipeline()`](https://huggingface.co/docs/transformers/pipeline_tutorial): pick a task and a model, and it handles preprocessing and output. Diffusers works the same way for [diffusion models](https://huggingface.co/docs/diffusers/index). + +vLLM serves a model behind an [OpenAI-compatible API](https://docs.vllm.ai/en/latest/getting_started/quickstart.html). SGLang does too, and its [RadixAttention caches shared prefixes](https://docs.sglang.io/), which helps when requests share a long prompt. On Apple silicon, [MLX LM](https://github.com/ml-explore/mlx-lm) runs and fine-tunes models locally. + +LiteLLM puts many LLM providers behind one OpenAI-style API. Use [the Python SDK in your code, or run the proxy as a gateway](https://docs.litellm.ai/docs/) when a platform team needs keys, budgets, and spend tracking across projects. + +PEFT [trains a small set of extra parameters](https://huggingface.co/docs/peft/index) instead of the whole model, and works with Transformers and Diffusers. Unsloth's docs [recommend starting with QLoRA](https://unsloth.ai/docs/get-started/fine-tuning-llms-guide). Axolotl drives [the whole pipeline from one YAML file](https://docs.axolotl.ai/): preprocessing, training, evaluation, quantization, and inference. + +Whisper is a [general-purpose speech recognition model](https://github.com/openai/whisper) that also translates speech and identifies languages. Microsoft marks VibeVoice for [research and development only](https://github.com/microsoft/VibeVoice). + +For text to speech, [Kitten TTS runs on CPU](https://github.com/KittenML/KittenTTS) without a GPU. gTTS calls [Google Translate's undocumented speech endpoint](https://github.com/pndurette/gTTS), so it needs the internet and can break without notice. + +The skill repos aren't pip packages: they install into your coding agent, not your app. Django AI Skills and Sentry Skills follow the [Agent Skills](https://agentskills.io/) open format. + +Write your app against the OpenAI API format, and you can switch between a hosted model and your own: vLLM, SGLang, and the LiteLLM proxy all speak it. From 4d2c9906e391365b827fbd90efe9e29d72311aed Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sat, 26 Sep 2026 22:09:16 +0800 Subject: [PATCH 012/104] docs: rewrite GUI Development category intro MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The old intro's lead used the same "For a Python X library, use …" template as other category pages, and it carried version-bound usage tips (Qt Widgets vs Qt Quick, pyside6-uic, per-toolkit threading helpers) instead of each project's recommended setup. Co-Authored-By: Claude --- .../data/category_intros/gui-development.md | 41 +++++++++++-------- 1 file changed, 23 insertions(+), 18 deletions(-) diff --git a/website/data/category_intros/gui-development.md b/website/data/category_intros/gui-development.md index 8e393325..ce8fc7dd 100644 --- a/website/data/category_intros/gui-development.md +++ b/website/data/category_intros/gui-development.md @@ -1,28 +1,33 @@ -For a Python GUI library, use PySide6 for desktop apps, or tkinter to stay on the standard library. For a UI that runs in the browser, use NiceGUI. +Among Python GUI libraries, PySide6 is the default for desktop apps and tkinter for small tools. For a UI in the browser, pick NiceGUI. How to choose: -- A full desktop app: PySide6, or PyQt6 for a codebase already on it -- A small tool on the standard library alone: tkinter -- A modern look on top of tkinter: CustomTkinter, or Tkinter Designer to turn a Figma design into tkinter code -- A GNOME or GTK app: PyGObject -- Native widgets on Windows, macOS, and Linux: wxPython +- Full desktop app: PySide6, or PyQt6 if your app can be GPL or you buy a license +- Small tool without third-party packages: tkinter +- Modern look for a tkinter app: [CustomTkinter](https://customtkinter.tomschimansky.com/) +- tkinter layout drawn in Figma: [Tkinter Designer](https://github.com/ParthJadhav/Tkinter-Designer) +- Native widgets on Windows, macOS, and Linux: [wxPython](https://wxpython.org/pages/overview/) +- GNOME app on Linux: [PyGObject](https://pygobject.gnome.org/) +- GPU-rendered tools for your scripts: [Dear PyGui](https://dearpygui.readthedocs.io/en/latest/about/what-why.html) +- Multi-touch apps on Android and iOS: Kivy - Native widgets on desktop and mobile: Toga -- A touch-first app for desktop and mobile: Kivy -- Fast, interactive tools with lots of plots: Dear PyGui -- A UI in the browser, or in its own desktop window: NiceGUI -- An HTML, CSS, and JavaScript frontend in a native window: pywebview -- Web, desktop, and mobile apps from one codebase: Flet -- A GUI for an existing argparse script: Gooey +- One codebase for web, desktop, and mobile: Flet +- Dashboards and web UIs: NiceGUI +- HTML/JavaScript frontend in a desktop window: [pywebview](https://pywebview.flowrl.com/guide/) +- GUI for an existing argparse script: [Gooey](https://github.com/chriskiehl/Gooey) -With PySide6, build desktop apps with Qt Widgets, which [respect the system style](https://doc.qt.io/qtforpython-6/faq/whatisqt.html), and save Qt Quick for fluid, touch-style UIs. Lay out windows in Qt Widgets Designer (`pyside6-designer`), then turn each `.ui` file into a Python class with `pyside6-uic`, which the docs call [the standard way](https://doc.qt.io/qtforpython-6/tutorials/basictutorial/uifiles.html) to use it. To ship the app, the docs recommend `pyside6-deploy` over PyInstaller and similar tools, since it's ["easier to use and also to get the most optimized executable."](https://doc.qt.io/qtforpython-6/deployment/index.html) +PySide6 is Qt for Python, the [official Python bindings for Qt](https://doc.qt.io/qtforpython-6/), under the LGPL, the GPL, or a commercial license. Qt's docs [recommend a virtual environment](https://doc.qt.io/qtforpython-6/gettingstarted.html) over installing it into your system Python. Ship it with [pyside6-deploy](https://doc.qt.io/qtforpython-6/deployment/index.html). -PySide6 and PyQt6 both bind Qt, so the license usually decides. PySide6 is available under the LGPLv3/GPLv3 and a commercial license. PyQt6 is GPLv3 or commercial, and in Riverbank's words: ["Unlike Qt, PyQt is not available under the LGPL."](https://www.riverbankcomputing.com/software/pyqt/) If your app can't be GPL, PyQt6 means buying a license. +PyQt6 wraps the same Qt, but Riverbank [licenses it under the GPL or a commercial license, not the LGPL](https://www.riverbankcomputing.com/software/pyqt/). So a closed-source app needs a commercial PyQt6 license, while PySide6 can stay on the LGPL. -With tkinter, import `tkinter.ttk` too and use its themed widgets. Be careful with tutorials: the docs warn that [most documentation you will find online still uses the old API](https://docs.python.org/3/library/tkinter.html) and can be woefully outdated. CustomTkinter adds modern, customizable widgets on top of tkinter, with [a consistent look across all desktop platforms](https://customtkinter.tomschimansky.com/). +tkinter is the standard Python interface to Tcl/Tk. Python's docs recommend the [themed tkinter.ttk widgets](https://docs.python.org/3/library/tkinter.ttk.html), which follow the platform's native theme, over the classic ones most online docs still use. -NiceGUI serves your UI to the browser by default. Pass `native=True` to `ui.run()` and it [opens in a desktop window](https://nicegui.io/documentation/section_configuration_deployment#native_mode) instead, through pywebview. Use pywebview directly for your own HTML frontend, and pass a Python object as `js_api` to [call its methods from JavaScript](https://pywebview.flowrl.com/guide/interdomain) as `pywebview.api`. +NiceGUI runs a web server and shows your UI in the browser, which suits dashboards, micro web apps, and robotics projects. Pass `native=True` to `ui.run()` to [open it in a desktop window](https://nicegui.io/documentation/section_configuration_deployment) instead, or bundle it into an executable with nicegui-pack. -Toga, Kivy, and Flet each have their own packager. Toga apps ship with Briefcase, BeeWare's tool for [turning a Python project into a standalone native app](https://briefcase.beeware.org/en/stable/). Kivy [recommends Buildozer](https://kivy.org/doc/stable/guide/packaging-android.html) for Android, and Flet has [`flet build`](https://flet.dev/docs/publish/). +Kivy runs the same code on Android, iOS, Linux, macOS, and Windows. Declare the widget tree in the [KV language](https://kivy.org/doc/stable/guide/lang.html) to keep the UI apart from your logic, and build Android packages with [Buildozer](https://kivy.org/doc/stable/guide/packaging-android.html). -In any toolkit, keep slow work out of event handlers, or the UI stops responding. The tkinter docs say [long-running computations](https://docs.python.org/3/library/tkinter.html#threading-model) belong in smaller pieces on a timer or in another thread, not in a handler. Run the slow work with Qt's threads and signals, or NiceGUI's `run.io_bound()` and `run.cpu_bound()`. In wxPython and Kivy, hand the result back to the UI thread with `wx.CallAfter()` or `@mainthread`. +Toga [uses native system widgets, not themes](https://toga.beeware.org/en/stable/about/philosophy/), so a Toga app is a native app on each platform. Start with the [BeeWare tutorial](https://tutorial.beeware.org/), which packages your app with Briefcase. + +Flet builds web, desktop, and mobile apps from one Python codebase, [without HTML, CSS, or JavaScript](https://flet.dev/docs/). Package it for each platform with [flet build](https://flet.dev/docs/publish/). + +Whatever you pick, keep slow work out of event handlers, or the window freezes. The [tkinter docs](https://docs.python.org/3/library/tkinter.html) say to break it into smaller pieces with timers or run it in another thread, and Qt's docs suggest threads for the same reason. From 2cc1bb440d2679de1590d7ce5d2aff385a3f0219 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sat, 26 Sep 2026 22:09:22 +0800 Subject: [PATCH 013/104] docs: rewrite ORM category intro The old intro told every FastAPI app to use SQLModel, which SQLModel's own docs don't claim beyond simple cases, and leaned on API names from SQLAlchemy's 2.0 rename (Mapped, mapped_column). Co-Authored-By: Claude --- website/data/category_intros/orm.md | 31 ++++++++++++++++++----------- 1 file changed, 19 insertions(+), 12 deletions(-) diff --git a/website/data/category_intros/orm.md b/website/data/category_intros/orm.md index 8b669ef1..24415ed8 100644 --- a/website/data/category_intros/orm.md +++ b/website/data/category_intros/orm.md @@ -1,20 +1,27 @@ -Use SQLAlchemy for most projects, and the Django ORM inside Django. With FastAPI, use SQLModel, so your tables and API schemas share the same fields. +SQLAlchemy is the Python ORM for most projects, and Django projects use the Django ORM. On MongoDB, pick Beanie for async code and MongoEngine for sync. How to choose: -- Any framework, or full control over the SQL: SQLAlchemy -- A Django project: the Django ORM, which needs Django's settings even outside a web app -- A FastAPI app: SQLModel -- A simple ORM with few concepts to learn: peewee -- MongoDB: Beanie for async code, MongoEngine for sync code -- DynamoDB: PynamoDB +- A Django project: Django ORM +- Any other app on a relational database: SQLAlchemy +- A FastAPI app with simple tables: SQLModel +- A small app that wants one module and no dependencies: peewee +- MongoDB from async code: Beanie +- MongoDB from sync code: MongoEngine +- Amazon DynamoDB: PynamoDB -With SQLAlchemy, write [typed models](https://docs.sqlalchemy.org/en/latest/orm/declarative_styles.html) with `DeclarativeBase`, `Mapped[]`, and `mapped_column()`. Create one `sessionmaker` at startup and open one session per request. Load relationships with `selectinload()`, and run migrations with Alembic, which comes from the SQLAlchemy project itself. +SQLAlchemy follows the [data mapper pattern](https://www.sqlalchemy.org/philosophy.html), so your classes and your schema can change separately. Open the Session [in a `with` block](https://docs.sqlalchemy.org/en/latest/orm/quickstart.html), and [keep its lifecycle outside](https://docs.sqlalchemy.org/en/latest/orm/session_basics.html) the functions that read or write data: one Session per thread, one AsyncSession per task. To avoid N+1 queries, load related objects up front with [eager loading](https://docs.sqlalchemy.org/en/latest/orm/queryguide/relationships.html). -Sync code is the safe default, even in an async framework. FastAPI's own docs put it plainly: ["If you just don't know, use normal `def`."](https://fastapi.tiangolo.com/async/) If you do go async, give each task its own `AsyncSession`. +The Django ORM follows the [Active Record pattern](https://docs.djangoproject.com/en/stable/misc/design-philosophies/): each model holds both the fields and the behavior of its data. Before you optimize, [find out what queries you run](https://docs.djangoproject.com/en/stable/topics/db/optimization/) and what they cost. -In Django, let `makemigrations` and `migrate` own the schema, and use `select_related()` or `prefetch_related()` whenever you touch related rows. Reach for raw SQL last: ["Explore the ORM before using raw SQL!"](https://docs.djangoproject.com/en/stable/topics/db/sql/) +SQLModel is [designed for FastAPI apps](https://sqlmodel.tiangolo.com/): one class is both a Pydantic model and a SQLAlchemy model. Get the session from a [FastAPI dependency](https://sqlmodel.tiangolo.com/tutorial/fastapi/session-with-dependency/), so each request gets its own. When you need more complex features, [plug SQLAlchemy in directly](https://sqlmodel.tiangolo.com/features/). -With SQLModel, follow its FastAPI tutorial: a base model for the shared fields, a `table=True` model for the database, and [separate models for create, read, and update](https://sqlmodel.tiangolo.com/tutorial/fastapi/multiple-models/). When you outgrow it, plug SQLAlchemy in directly. +For a small app, peewee is a [single module with no required dependencies](https://docs.peewee-orm.com/en/latest/), with few concepts to learn. In a web app, [open a connection per request](https://docs.peewee-orm.com/en/latest/peewee/database.html) and close it when the request ends. Once traffic grows, [switch to a pooled database](https://docs.peewee-orm.com/en/latest/peewee/framework_integration.html). -For MongoDB and DynamoDB, model around your queries, not your tables. MongoDB's rule is that ["data that's accessed together should be stored together."](https://www.mongodb.com/docs/manual/core/data-modeling-introduction/) AWS goes further: [don't design a DynamoDB schema](https://docs.aws.amazon.com/amazondynamodb/latest/developerguide/bp-general-nosql-design.html) until you know the questions it needs to answer. +On the NoSQL side, Beanie is an [async MongoDB ODM built on Pydantic](https://beanie-odm.dev/), with one Document class per collection. Pass your document models to [`init_beanie()`](https://beanie-odm.dev/tutorial/initialization/), which creates the collections and the indexes you declared. + +MongoEngine is the sync pick: it's [built on PyMongo only](https://mongoengine-odm.readthedocs.io/faq.html) and doesn't support async drivers. Its document schemas are [enforced in your app, not by MongoDB](https://mongoengine-odm.readthedocs.io/tutorial.html), and they catch wrong types and missing fields. + +PynamoDB is a [Pythonic interface to DynamoDB](https://pynamodb.readthedocs.io/en/stable/). Build on its [Model API](https://pynamodb.readthedocs.io/en/stable/tutorial.html): one model class per table, each with a hash key. For tests, point it at a [local DynamoDB-compatible server](https://pynamodb.readthedocs.io/en/stable/local.html). + +Track schema changes in migrations: Django [has them built in](https://docs.djangoproject.com/en/stable/topics/migrations/), SQLAlchemy has [Alembic](https://alembic.sqlalchemy.org/en/latest/), and Beanie [ships its own](https://beanie-odm.dev/tutorial/migrations/). Review every migration Alembic's autogenerate writes, since it's [not meant to be perfect](https://alembic.sqlalchemy.org/en/latest/autogenerate.html). From 00915b03001decae3acccf012d2d543337049a8d Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sat, 26 Sep 2026 22:14:01 +0800 Subject: [PATCH 014/104] docs: trim ai-and-agents how-to-choose list to 12 items The list had 22 items, which in the upcoming layout sits above the entry table and would push it 4-5 phone screens down; it now has one item per README subcategory, in README order, reusing the intro's own wording. Co-Authored-By: Claude --- website/data/category_intros/ai-and-agents.md | 24 ++++++------------- 1 file changed, 7 insertions(+), 17 deletions(-) diff --git a/website/data/category_intros/ai-and-agents.md b/website/data/category_intros/ai-and-agents.md index 6b1b3ee8..ebb14eb0 100644 --- a/website/data/category_intros/ai-and-agents.md +++ b/website/data/category_intros/ai-and-agents.md @@ -2,28 +2,18 @@ LangChain is the place to start among Python libraries for AI agents, and LangGr How to choose: -- A first agent, or a prebuilt tool-calling loop: LangChain -- Long-running, stateful agents that mix fixed steps with LLM-driven ones: LangGraph -- Typed agents whose outputs are validated: Pydantic AI -- A team of role-playing agents: CrewAI +- Skills for your coding agent: Django AI Skills, Sentry Skills, or Trail of Bits Skills +- A first agent: LangChain, or LangGraph to control every step - An agent built on one vendor's platform: OpenAI Agents SDK or Claude Agent SDK -- Structured data from an LLM, without an agent framework: Instructor +- A ready-made personal assistant: Hermes Agent, or AstrBot for chat apps - Prompts tuned against a metric instead of by hand: DSPy -- RAG over your own documents: LlamaIndex -- Memory that survives across sessions: Mem0 -- Agent context you can browse and edit like files: OpenViking -- A knowledge graph with provenance for regulated domains: Semantica +- Structured output, RAG, or agent memory: Instructor, LlamaIndex, or Mem0 - Running pre-trained models: Transformers -- Serving a model on GPUs: vLLM, or SGLang when requests share long prompts -- Running a model on Apple silicon: MLX LM +- Serving a model: vLLM, or MLX LM on Apple silicon - One API for many LLM providers: LiteLLM - Image and video generation: Diffusers -- Fine-tuning: PEFT for adapters, Unsloth for fast low-memory training, Axolotl for YAML-configured runs across GPUs -- Speech to text: Whisper, or FunASR for streaming and edge deployment -- Text to speech: Kitten TTS on CPU, gTTS for a quick online voice -- Speech research: VibeVoice -- A ready-made personal assistant: Hermes Agent, or AstrBot for chat apps like Telegram, Slack, and QQ -- Skills for your coding agent: Django AI Skills for Django, Sentry Skills for code review, Trail of Bits Skills for security work +- Fine-tuning: PEFT, Unsloth, or Axolotl +- Speech: Whisper for speech to text, Kitten TTS for text to speech New to agents? LangGraph's own docs [recommend LangChain's prebuilt agents](https://docs.langchain.com/oss/python/langgraph/overview), which run on LangGraph: give an agent a model, tools, and a prompt, and the loop is handled for you. Drop down to LangGraph for [needs that combine deterministic and agentic workflows](https://docs.langchain.com/oss/python/langchain/overview). You don't need LangChain to use LangGraph. From eaddf9441eaf8c549b1cb8e1a2f452b1fd79dff2 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sat, 26 Sep 2026 22:14:23 +0800 Subject: [PATCH 015/104] docs: drop links from gui-development how-to-choose list 7 of its 13 items carried links while the other intros' lists carry none, and the entry table already links every project. Co-Authored-By: Claude --- website/data/category_intros/gui-development.md | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/website/data/category_intros/gui-development.md b/website/data/category_intros/gui-development.md index ce8fc7dd..425ac8a8 100644 --- a/website/data/category_intros/gui-development.md +++ b/website/data/category_intros/gui-development.md @@ -4,17 +4,17 @@ How to choose: - Full desktop app: PySide6, or PyQt6 if your app can be GPL or you buy a license - Small tool without third-party packages: tkinter -- Modern look for a tkinter app: [CustomTkinter](https://customtkinter.tomschimansky.com/) -- tkinter layout drawn in Figma: [Tkinter Designer](https://github.com/ParthJadhav/Tkinter-Designer) -- Native widgets on Windows, macOS, and Linux: [wxPython](https://wxpython.org/pages/overview/) -- GNOME app on Linux: [PyGObject](https://pygobject.gnome.org/) -- GPU-rendered tools for your scripts: [Dear PyGui](https://dearpygui.readthedocs.io/en/latest/about/what-why.html) +- Modern look for a tkinter app: CustomTkinter +- tkinter layout drawn in Figma: Tkinter Designer +- Native widgets on Windows, macOS, and Linux: wxPython +- GNOME app on Linux: PyGObject +- GPU-rendered tools for your scripts: Dear PyGui - Multi-touch apps on Android and iOS: Kivy - Native widgets on desktop and mobile: Toga - One codebase for web, desktop, and mobile: Flet - Dashboards and web UIs: NiceGUI -- HTML/JavaScript frontend in a desktop window: [pywebview](https://pywebview.flowrl.com/guide/) -- GUI for an existing argparse script: [Gooey](https://github.com/chriskiehl/Gooey) +- HTML/JavaScript frontend in a desktop window: pywebview +- GUI for an existing argparse script: Gooey PySide6 is Qt for Python, the [official Python bindings for Qt](https://doc.qt.io/qtforpython-6/), under the LGPL, the GPL, or a commercial license. Qt's docs [recommend a virtual environment](https://doc.qt.io/qtforpython-6/gettingstarted.html) over installing it into your system Python. Ship it with [pyside6-deploy](https://doc.qt.io/qtforpython-6/deployment/index.html). From eb51124608a8e38de53de82af486d48d78983a51 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sat, 26 Sep 2026 22:20:37 +0800 Subject: [PATCH 016/104] refactor: extract category table row markup into entry_rows macro Co-Authored-By: Claude --- website/templates/category.html | 163 ++++++++++++++++---------------- 1 file changed, 83 insertions(+), 80 deletions(-) diff --git a/website/templates/category.html b/website/templates/category.html index 7325aa2b..6edb3061 100644 --- a/website/templates/category.html +++ b/website/templates/category.html @@ -62,91 +62,15 @@ {% endblock %} {% block content %} - -
    -
    -
    -

    Search every project in one place

    -
    -

    - Press / to search. Tap a tag to filter. Click any row for - details. -

    -
    - -
    -

    Search and filter

    -
    - - - - - -
    -
    - Filtering for - -
    -
    - -

    Results

    -
    - - - - - - - - - - - - - - {% for entry in entries %} +{% macro entry_rows(entry, index) %} - + {% endif %} - + +{% endmacro %} + +
    +
    +
    +

    Search every project in one place

    +
    +

    + Press / to search. Tap a tag to filter. Click any row for + details. +

    +
    + +
    +

    Search and filter

    +
    + + + + + +
    +
    + Filtering for + +
    +
    + +

    Results

    +
    +
    Row number - - - - - - - - Tags - -
    @@ -273,6 +197,85 @@
    + + + + + + + + + + + + + {% for entry in entries %} + {{ entry_rows(entry, loop.index) }} {% endfor %}
    Row number + + + + + + + + Tags + +
    From b3fe53bf51a90237fb346f910236ad2fa8be1a88 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sat, 26 Sep 2026 23:16:53 +0800 Subject: [PATCH 017/104] fix: stop tracking .claude/launch.json Personal Claude Code preview config, now holding a local-worktree server with an absolute path, was tracked so local edits showed as repo changes; excluded via .git/info/exclude instead. Co-Authored-By: Claude --- .claude/launch.json | 11 ----------- 1 file changed, 11 deletions(-) delete mode 100644 .claude/launch.json diff --git a/.claude/launch.json b/.claude/launch.json deleted file mode 100644 index de8c1ce1..00000000 --- a/.claude/launch.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "version": "0.0.1", - "configurations": [ - { - "name": "website", - "runtimeExecutable": "make", - "runtimeArgs": ["preview"], - "port": 8000 - } - ] -} From 9cc721f533c1ff239ccd10c38f785b05c32d02bc Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sat, 26 Sep 2026 23:28:51 +0800 Subject: [PATCH 018/104] fix: match category meta titles to search query word order Category page titles read "ORM Python Libraries", but people search "python orm", so the title word order didn't match the query. Co-Authored-By: Claude --- website/build.py | 2 +- website/tests/test_build.py | 6 +++--- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/website/build.py b/website/build.py index 46985670..fefc2f77 100644 --- a/website/build.py +++ b/website/build.py @@ -222,7 +222,7 @@ def category_meta_title(name: str, parent_name: str | None = None) -> str: if len(title) <= 60: return title return f"{name} - Awesome Python" - title = f"{name} Python Libraries - Awesome Python" + title = f"Python {name} Libraries - Awesome Python" if len(title) <= 60: return title return f"{name} - Awesome Python" diff --git a/website/tests/test_build.py b/website/tests/test_build.py index 7998525f..d32486b3 100644 --- a/website/tests/test_build.py +++ b/website/tests/test_build.py @@ -280,7 +280,7 @@ class TestBuild: assert 'href="/categories/widgets/"' in index_html assert 'data-value="Widgets"' in index_html - assert parser.title.strip() == "Widgets Python Libraries - Awesome Python" + assert parser.title.strip() == "Python Widgets Libraries - Awesome Python" assert parser.meta_by_name["description"] == "Widget libraries. Also see awesome-widgets. Explore 2 curated Python projects in Widgets." assert parser.links_by_rel["canonical"] == "https://awesome-python.com/categories/widgets/" assert parser.meta_by_property["og:url"] == "https://awesome-python.com/categories/widgets/" @@ -622,7 +622,7 @@ class TestBuild: assert set(graph) == {"WebSite", "CollectionPage", "BreadcrumbList"} assert graph["WebSite"]["@id"] == "https://awesome-python.com/#website" collection = graph["CollectionPage"] - assert collection["name"] == "Widgets Python Libraries" + assert collection["name"] == "Python Widgets Libraries" assert collection["@id"] == "https://awesome-python.com/categories/widgets/" assert collection["url"] == "https://awesome-python.com/categories/widgets/" assert collection["description"] == "Widget libraries. Explore 2 curated Python projects in Widgets." @@ -672,7 +672,7 @@ class TestBuild: graph = {node["@type"]: node for node in data["@graph"]} collection = graph["CollectionPage"] - assert collection["name"] == "AI & ML Python Libraries" + assert collection["name"] == "Python AI & ML Libraries" assert collection["@id"] == "https://awesome-python.com/categories/ai-ml/" assert collection["url"] == "https://awesome-python.com/categories/ai-ml/" assert collection["description"] == "Explore 1 curated Python projects in AI & ML. Part of the Awesome Python catalog." From 528f9edbb7d5523b381e5c8a5b1f73985c828e7b Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sat, 26 Sep 2026 23:32:20 +0800 Subject: [PATCH 019/104] fix: expose entry description rows to screen readers Every entry description row carried aria-hidden="true", hiding descriptions that sighted readers can see from screen readers. Co-Authored-By: Claude --- website/templates/category.html | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/website/templates/category.html b/website/templates/category.html index 6edb3061..9fca1b26 100644 --- a/website/templates/category.html +++ b/website/templates/category.html @@ -149,7 +149,7 @@ → {% if entry.description %} - +
    {{ entry.description | safe }}
    From 2b9469c0ae1c8f9fd3fb3f47d5fb4e7f2db3c0f1 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sat, 26 Sep 2026 23:33:57 +0800 Subject: [PATCH 020/104] feat: group category page rows by use case, restructure intro Category pages sorted rows by downloads, which buried editorial leads (the Django ORM showed as row 7 of 7, tkinter as 14 of 14) against CONTRIBUTING.md's "position is the marker"; the full intro in the hero pushed the table 2.5 screens down on a phone; and every page repeated the same "Search every project in one place" H2. Co-Authored-By: Claude --- website/build.py | 61 ++++++++++++++-- website/static/main.js | 35 ++++++++- website/static/style.css | 118 +++++++++++++++++++++++++++++-- website/templates/category.html | 54 +++++++++++++- website/tests/test_build.py | 121 +++++++++++++++++++++++++++++++- 5 files changed, 370 insertions(+), 19 deletions(-) diff --git a/website/build.py b/website/build.py index fefc2f77..9164ca1a 100644 --- a/website/build.py +++ b/website/build.py @@ -66,6 +66,13 @@ class TemplateEntry(TypedDict): also_see: list[AlsoSee] +class EntryGroup(TypedDict): + name: str # empty for a page with a single unnamed group + slug: str + url: str # links the group heading to its own page, empty when it has none + entries: list[TemplateEntry] + + class SyntheticCategory(TypedDict): name: str slug: str @@ -237,13 +244,16 @@ def category_meta_description(name: str, entry_count: int, description: str, par return f"{count_sentence} Part of the Awesome Python catalog." -def load_category_intro(path: Path) -> tuple[str, str]: - """Render a category intro file to HTML, plus its first paragraph as plain text for the meta description. +def load_category_intro(path: Path) -> tuple[str, str, str]: + """Render a category intro file to HTML, split at the end of its "How to choose:" list. - Returns empty strings if the category has no intro file. + Returns the part shown above the table, the guide shown below it, and the + first paragraph as plain text for the meta description. A file without the + list keeps everything above the table. Returns empty strings if the category + has no intro file. """ if not path.exists(): - return "", "" + return "", "", "" md = MarkdownIt("commonmark") tokens = md.parse(path.read_text(encoding="utf-8")) for token in tokens: @@ -252,7 +262,38 @@ def load_category_intro(path: Path) -> tuple[str, str]: child.attrSet("target", "_blank") child.attrSet("rel", "noopener") lead = next(node for node in SyntaxTreeNode(tokens).children if node.type == "paragraph") - return md.renderer.render(tokens, md.options, {}), render_inline_text(lead.children[0].children) + split_at = len(tokens) + for i, token in enumerate(tokens): + if token.type == "inline" and token.level == 1 and token.content == "How to choose:" and i + 2 < len(tokens) and tokens[i + 2].type == "bullet_list_open": + split_at = next(j for j in range(i + 3, len(tokens)) if tokens[j].type == "bullet_list_close" and tokens[j].level == 0) + 1 + break + render = md.renderer.render + return render(tokens[:split_at], md.options, {}), render(tokens[split_at:], md.options, {}), render_inline_text(lead.children[0].children) + + +def group_section_entries(section: ParsedSection, entries_by_key: dict[tuple[str, str], TemplateEntry]) -> list[EntryGroup]: + """Group a section's entries by use case (subcategory), both in README order.""" + groups: dict[str, EntryGroup] = {} + for parsed in section["entries"]: + name = parsed["subcategory"] + group = groups.setdefault(name, EntryGroup(name=name, slug=slugify(name) if name else "", url="", entries=[])) + group["entries"].append(entries_by_key[(parsed["url"], parsed["name"])]) + return list(groups.values()) + + +def group_entries_by_section(sections: Sequence[ParsedSection], entries_by_key: dict[tuple[str, str], TemplateEntry]) -> list[EntryGroup]: + """Group a thematic group's entries by section, both in README order, listing each entry once.""" + placed: set[tuple[str, str]] = set() + groups: list[EntryGroup] = [] + for section in sections: + entries: list[TemplateEntry] = [] + for parsed in section["entries"]: + key = (parsed["url"], parsed["name"]) + if key not in placed: + placed.add(key) + entries.append(entries_by_key[key]) + groups.append(EntryGroup(name=section["name"], slug=section["slug"], url=category_path(section), entries=entries)) + return groups def build_breadcrumb_json_ld(items: Sequence[tuple[str, str]]) -> dict: @@ -692,11 +733,12 @@ def build(repo_root: Path) -> None: page_dir: Path, parent_category: ParsedSection | None = None, group_categories: Sequence[ParsedSection] | None = None, + entry_groups: Sequence[EntryGroup] = (), ) -> None: page_dir.mkdir(parents=True, exist_ok=True) parent_name = parent_category["name"] if parent_category else None category_title = category_meta_title(category["name"], parent_name) - intro_html, intro_lead = load_category_intro(website / "data" / "category_intros" / f"{current_path.removeprefix('/categories/').strip('/')}.md") + intro_html, guide_html, intro_lead = load_category_intro(website / "data" / "category_intros" / f"{current_path.removeprefix('/categories/').strip('/')}.md") category_description = intro_lead or category_meta_description(category["name"], len(entries), category["description"], parent_name) breadcrumbs = [("Awesome Python", SITE_URL)] if parent_category: @@ -713,7 +755,9 @@ def build(repo_root: Path) -> None: category_url=category_url, category_description=category_description, intro_html=intro_html, + guide_html=guide_html, entries=entries, + entry_groups=entry_groups, total_categories=len(categories), category_urls=category_urls, current_path=current_path, @@ -726,6 +770,8 @@ def build(repo_root: Path) -> None: encoding="utf-8", ) + entries_by_key = {(e["url"], e["name"]): e for e in entries} + section_groups = {category["name"]: group_section_entries(category, entries_by_key) for category in categories} for category in categories: render_category( category, @@ -733,6 +779,7 @@ def build(repo_root: Path) -> None: entries=[e for e in entries if category["name"] in e["categories"]], current_path=category_path(category), page_dir=categories_dir / category["slug"], + entry_groups=section_groups[category["name"]], ) for group in parsed_groups: @@ -743,6 +790,7 @@ def build(repo_root: Path) -> None: current_path=group_path(group["slug"]), page_dir=categories_dir / group["slug"], group_categories=group["categories"], + entry_groups=group_entries_by_section(group["categories"], entries_by_key), ) if builtin_entries: @@ -792,6 +840,7 @@ def build(repo_root: Path) -> None: current_path=subcategory_path(cat_slug, sub_slug), page_dir=categories_dir / cat_slug / sub_slug, parent_category=cat_by_slug[cat_slug], + entry_groups=[EntryGroup(name="", slug="", url="", entries=group["entries"]) for group in section_groups[cat_by_slug[cat_slug]["name"]] if group["name"] == sub_name], ) redirects_file = website / "data" / "redirects.json" diff --git a/website/static/main.js b/website/static/main.js index 11d43861..307e3f92 100644 --- a/website/static/main.js +++ b/website/static/main.js @@ -4,8 +4,14 @@ function getScrollBehavior() { return reducedMotion.matches ? "auto" : "smooth"; } +const table = document.querySelector(".table"); +// Category pages list rows in editorial (README) order until a column is sorted +const defaultSort = + table && table.dataset.defaultSort === "editorial" + ? { col: "editorial", order: "asc" } + : { col: "downloads", order: "desc" }; let activeFilter = null; -let activeSort = { col: "downloads", order: "desc" }; +let activeSort = defaultSort; const searchInput = document.querySelector(".search"); const filterBar = document.querySelector(".filter-bar"); const filterValue = document.querySelector(".filter-value"); @@ -14,6 +20,7 @@ const noResults = document.querySelector(".no-results"); const rows = document.querySelectorAll(".table tbody tr.row"); const tags = document.querySelectorAll(".tag"); const tbody = document.querySelector(".table tbody"); +const groupRows = document.querySelectorAll(".table tbody tr.group-row"); function initRevealSections() { const sections = document.querySelectorAll("[data-reveal]"); @@ -111,6 +118,12 @@ document time.textContent = relativeTime(time.getAttribute("datetime")); }); +let currentGroupRow = null; +Array.prototype.forEach.call(tbody ? tbody.rows : [], function (tr) { + if (tr.classList.contains("group-row")) currentGroupRow = tr; + else if (tr.classList.contains("row")) tr._groupRow = currentGroupRow; +}); + rows.forEach(function (row, i) { row._origIndex = i; let next = row.nextElementSibling; @@ -176,6 +189,15 @@ function applyFilters() { } }); + groupRows.forEach(function (groupRow) { + groupRow.hidden = true; + }); + if (activeSort.col === "editorial") { + rows.forEach(function (row) { + if (!row.hidden && row._groupRow) row._groupRow.hidden = false; + }); + } + if (noResults) noResults.hidden = visibleCount > 0; tags.forEach(function (tag) { @@ -211,7 +233,7 @@ function buildQueryString() { const params = new URLSearchParams(); const query = searchInput ? searchInput.value.trim() : ""; if (query) params.set("q", query); - if (activeSort.col !== "downloads" || activeSort.order !== "desc") { + if (activeSort.col !== defaultSort.col || activeSort.order !== defaultSort.order) { params.set("sort", activeSort.col); params.set("order", activeSort.order); } @@ -227,6 +249,8 @@ function updateURL() { } function getSortValue(row, col) { + // +1 keeps the first row above the "no value" cutoff in sortRows + if (col === "editorial") return row._origIndex + 1; if (col === "name") { return row.querySelector(".col-name a").textContent.trim().toLowerCase(); } @@ -282,7 +306,12 @@ function sortRows() { }); const frag = document.createDocumentFragment(); + let lastGroupRow = null; arr.forEach(function (row) { + if (col === "editorial" && row._groupRow && row._groupRow !== lastGroupRow) { + frag.appendChild(row._groupRow); + lastGroupRow = row._groupRow; + } frag.appendChild(row); if (row._descRow) frag.appendChild(row._descRow); if (row._expandRow) frag.appendChild(row._expandRow); @@ -392,7 +421,7 @@ sortHeaders.forEach(function (th) { if (activeSort.col === col) { if (activeSort.order === defaultOrder) activeSort = { col: col, order: altOrder }; - else activeSort = { col: "downloads", order: "desc" }; + else activeSort = defaultSort; } else { activeSort = { col: col, order: defaultOrder }; } diff --git a/website/static/style.css b/website/static/style.css index bdad3835..f7bddd67 100644 --- a/website/static/style.css +++ b/website/static/style.css @@ -438,6 +438,7 @@ kbd { .search:focus-visible, .filter-clear:focus-visible, .tag:focus-visible, +.jump-link:focus-visible, .back-to-top:focus-visible, .no-results-clear:focus-visible, .table a:focus-visible, @@ -459,9 +460,9 @@ kbd { z-index: 1; width: min(100%, calc(var(--shell-max) + (var(--shell-pad) * 2))); margin: 0 auto; - padding: 1.25rem var(--shell-pad) clamp(3.75rem, 8vw, 6.75rem); + padding: 1.25rem var(--shell-pad) clamp(2.25rem, 5vw, 3.5rem); display: grid; - gap: clamp(3rem, 8vw, 5.5rem); + gap: clamp(2rem, 5vw, 3rem); } .category-hero h1 { @@ -497,7 +498,9 @@ kbd { } .category-subtitle a, -.category-intro a { +.category-intro a, +.guide-body a, +.jump-link { color: var(--hero-text); text-decoration: underline; text-decoration-color: oklch(100% 0 0 / 0.32); @@ -505,12 +508,14 @@ kbd { } .category-subtitle a:hover, -.category-intro a:hover { +.category-intro a:hover, +.guide-body a:hover, +.jump-link:hover { text-decoration-color: oklch(100% 0 0 / 0.7); } .category-intro { - margin-top: 1.75rem; + margin-top: 1.25rem; display: grid; gap: 0.9rem; color: var(--hero-muted); @@ -530,10 +535,65 @@ kbd { padding-left: 1.25rem; } -.category-intro code { +.category-intro code, +.guide-body code { font-size: 0.9em; } +.jump-links { + display: flex; + flex-wrap: wrap; + gap: 0.45rem 1.6rem; + margin-top: 1.5rem; +} + +.jump-link { + font-size: var(--text-base); + font-weight: 600; +} + +/* The arrow marks an in-page jump; inline-block keeps it out of the underline */ +.jump-link::after { + content: "\2193"; + display: inline-block; + margin-left: 0.3rem; + color: var(--hero-kicker); +} + +.guide-band { + padding-block: clamp(3rem, 7vw, 5.5rem); + background: linear-gradient(140deg, var(--hero-bg-start) 0%, var(--hero-bg-mid) 58%, var(--hero-bg-end) 100%); + color: var(--hero-text); +} + +.guide-section { + display: grid; + gap: 1.5rem; +} + +.guide-section h2 { + font-family: var(--font-display); + font-size: clamp(2.2rem, 4vw, 3.3rem); + font-weight: 600; + line-height: 0.94; + letter-spacing: -0.03em; +} + +.guide-body { + display: grid; + gap: 1rem; + color: var(--hero-muted); + font-size: var(--text-lg); + line-height: 1.7; + text-wrap: pretty; +} + +.guide-body ul { + display: grid; + gap: 0.35rem; + padding-left: 1.25rem; +} + .sponsor-band { padding-block: clamp(2.5rem, 5.5vw, 4rem); background: @@ -651,6 +711,11 @@ kbd { padding-bottom: 1.75rem; } +.order-note { + color: var(--ink-soft); + font-size: var(--text-base); +} + .results-note { color: var(--ink-soft); font-size: var(--text-sm); @@ -819,6 +884,44 @@ kbd { cursor: pointer; } +.group-row th { + padding-top: 2.25rem; + padding-bottom: 0.85rem; + padding-left: max(var(--shell-pad), calc(50vw - (var(--shell-max) / 2) + var(--shell-pad))); + text-align: left; + border-bottom: 1px solid var(--line-strong); + background: var(--bg-paper); +} + +.group-row h2 { + display: flex; + align-items: baseline; + gap: 0.75rem; + font-family: var(--font-display); + font-size: clamp(1.7rem, 2.8vw, 2.2rem); + font-weight: 600; + line-height: 1; + scroll-margin-top: 4.5rem; +} + +.group-row h2 a { + color: var(--ink); +} + +.group-row h2 a:hover { + color: var(--accent-deep); + text-decoration: underline; + text-decoration-color: var(--accent-underline); + text-underline-offset: 0.2em; +} + +.group-count { + font-family: var(--font-body); + font-size: var(--text-sm); + font-weight: 600; + color: var(--ink-muted); +} + .row:not(.open):hover td { background: var(--row-hover); } @@ -1735,7 +1838,8 @@ th[data-sort].sort-asc::after { } .table thead th:first-child, - .table tbody td:first-child { + .table tbody td:first-child, + .group-row th { padding-left: 0.8rem; } diff --git a/website/templates/category.html b/website/templates/category.html index 9fca1b26..a9618aef 100644 --- a/website/templates/category.html +++ b/website/templates/category.html @@ -40,6 +40,19 @@ {% if intro_html %}
    {{ intro_html | safe }}
    {% endif %} + {% set named_groups = entry_groups | selectattr("name") | list %} + {% if (named_groups and not group_categories) or guide_html %} + + {% endif %}
    {% if group_categories %} @@ -201,9 +214,17 @@
    + {% if entry_groups %} + {% set named_groups = entry_groups | selectattr("name") | list %} +

    + {% if group_categories %}Listed by section, in editorial order.{% elif named_groups %}Listed in editorial order, grouped by use case.{% else %}Listed in editorial order.{% endif %} + Click a column to re-sort the whole list. +

    + {% else %}
    -

    Search every project in one place

    +

    {{ entries | length }} {{ category.name }} projects

    + {% endif %}

    Press / to search. Tap a tag to filter. Click any row for details. @@ -249,7 +270,7 @@ role="region" aria-label="Libraries table" > - +
    @@ -274,9 +295,29 @@ + {% if entry_groups %} + {% set row = namespace(index=0) %} + {% for group in entry_groups %} + {% if group.name %} + + + + {% endif %} + {% for entry in group.entries %} + {% set row.index = row.index + 1 %} + {{ entry_rows(entry, row.index) }} + {% endfor %} + {% endfor %} + {% else %} {% for entry in entries %} {{ entry_rows(entry, loop.index) }} {% endfor %} + {% endif %}
    Row number
    +

    + {% if group.url %}{{ group.name }}{% else %}{{ group.name }}{% endif %} + {{ group.entries | length }} project{{ "s" if group.entries | length != 1 }} +

    +

    @@ -290,6 +331,15 @@
    +{% if guide_html %} +
    +
    +

    {{ category.name }} guide

    +
    {{ guide_html | safe }}
    +
    +
    +{% endif %} +
    diff --git a/website/tests/test_build.py b/website/tests/test_build.py index d32486b3..cfebc616 100644 --- a/website/tests/test_build.py +++ b/website/tests/test_build.py @@ -17,6 +17,7 @@ from build import ( detect_source_type, extract_entries, extract_github_repo, + load_category_intro, load_downloads, load_pypi_badges, load_stars, @@ -291,7 +292,7 @@ class TestBuild: assert 'href="https://example.com/w1"' in category_html assert "A widget." in category_html assert 'href="https://github.com/owner/w2"' in category_html - assert '' in category_html + assert '
    ' in category_html assert "42" in category_html assert "2026-01-01T00:00:00+00:00" in category_html @@ -1007,6 +1008,99 @@ class TestBuild: assert "
  • Small apps: w1
  • " in category_html assert 'the docs' in category_html + def test_build_renders_category_guide_below_table(self, tmp_path): + self._copy_real_templates(tmp_path) + (tmp_path / "README.md").write_text(self._REDIRECT_README, encoding="utf-8") + intros_dir = tmp_path / "website" / "data" / "category_intros" + intros_dir.mkdir(parents=True) + (intros_dir / "widgets.md").write_text("Use w1.\n\nHow to choose:\n\n- Small apps: w1\n\nSet up w1 once per process.\n", encoding="utf-8") + build(tmp_path) + + category_html = (tmp_path / "website" / "output" / "categories" / "widgets" / "index.html").read_text(encoding="utf-8") + intro_html = category_html.split('
    ', 1)[1].split("
    ", 1)[0] + assert "
  • Small apps: w1
  • " in intro_html + assert "Set up w1" not in intro_html + guide_html = category_html.split('
    ', 1)[1] + assert "

    Widgets guide

    " in guide_html + assert "

    Set up w1 once per process.

    " in guide_html + assert category_html.index('id="guide"') > category_html.index("
    ") + assert 'Widgets guide' in category_html + + def test_section_page_groups_rows_by_use_case_in_readme_order(self, tmp_path): + readme = textwrap.dedent("""\ + # T + + ## Projects + + **Tools** + + ### Widgets + + - Small + - [w2](https://example.com/w2) - Second. + - [w1](https://example.com/w1) - First. + - Large + - [w3](https://example.com/w3) - Third. + - [sqlite3](https://docs.python.org/3/library/sqlite3.html) - Stdlib. + + # Contributing + + Done. + """) + self._copy_real_templates(tmp_path) + (tmp_path / "README.md").write_text(readme, encoding="utf-8") + build(tmp_path) + + site = tmp_path / "website" / "output" / "categories" + html = (site / "widgets" / "index.html").read_text(encoding="utf-8") + assert 'data-default-sort="editorial"' in html + positions = [html.index(marker) for marker in ('

    ', ">w2w1', ">w3Small' in html + assert '' in html + + subcategory_html = (site / "widgets" / "small" / "index.html").read_text(encoding="utf-8") + assert 'data-default-sort="editorial"' in subcategory_html + assert "group-row" not in subcategory_html + assert subcategory_html.index(">w2w1Machine Learning') + dl_heading = html.index('Deep Learning') + assert ml_heading < html.index(">ml1dl1ml1Use w1 for most apps.

    \n

    How to choose:

    \n
      \n
    • Small apps: w1
    • \n
    • Big apps: w2
    • \n
    \n" + assert guide_html == "

    Configure w1 once.

    \n

    Pin w2.

    \n" + assert lead == "Use w1 for most apps." + + def test_keeps_everything_above_table_without_how_to_choose_list(self, tmp_path): + path = tmp_path / "widgets.md" + path.write_text("Use w1.\n\n- Small apps: w1\n\nConfigure w1 once.\n", encoding="utf-8") + intro_html, guide_html, _ = load_category_intro(path) + assert "Configure w1 once." in intro_html + assert guide_html == "" + + def test_returns_empty_strings_without_intro_file(self, tmp_path): + assert load_category_intro(tmp_path / "missing.md") == ("", "", "") From d5febbea1f99eedbe9a97e5dc662e700d49930eb Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sat, 26 Sep 2026 23:34:01 +0800 Subject: [PATCH 021/104] docs: document category page group rows and guide band Co-Authored-By: Claude --- DESIGN.md | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/DESIGN.md b/DESIGN.md index 1e678f2b..56ea32bc 100644 --- a/DESIGN.md +++ b/DESIGN.md @@ -140,7 +140,7 @@ Hard-won sizing rules (do not relax): Depth comes from **tonal layers**, not heavy shadows. - The page is a quiet warm canvas (`--bg-page`). The content shell is slightly brighter paper (`--bg-paper`). The sponsor band, CTA backgrounds, and inline decorative blocks step up to `--bg-paper-strong`. -- The hero is the one place that uses real atmosphere: subtle grid, slow sheen, warm radial gradients on a dark earthy ground (`--hero-bg-start` → `--hero-bg-mid` → `--hero-bg-end`). The sheen and any other motion respect `prefers-reduced-motion`. +- The hero and the category guide band are the only places with real atmosphere: warm gradients on a dark earthy ground (`--hero-bg-start` → `--hero-bg-mid` → `--hero-bg-end`). Only the hero adds the subtle grid and slow sheen. The sheen and any other motion respect `prefers-reduced-motion`. - The footer is a single tonal block in `--footer-bg`, no internal gradients. - Two depth treatments are allowed and only these two. The search input combines a 1px inset highlight (`--search-inset`) with a soft warm drop shadow (`--search-shadow`, intensified by `--search-focus-shadow` on focus). The primary CTA button (`.hero-action-primary`) carries a warm drop shadow for press affordance. Both shadows are soft, warm-tinted, and tied to interactive elements. No new drop shadows on cards, panels, rows, or static decoration. - No glassmorphism as default decoration. @@ -160,8 +160,11 @@ The shape language is overwhelmingly **pill on small, zero radius on large**. The component vocabulary is small and table-led. Source of truth: `website/static/style.css`. - **Table-driven index** (the hero of the page). Sticky header, sortable columns, click-to-expand rows that indent under the Name column. Modeled on placestoread.xyz. Not a card grid. +- **Group rows**. Section, subcategory, and group pages list rows in README order. Section pages add one H2 row per use case; group pages add one per section, linking to that section's page. A note replaces the sort arrow until someone sorts a column, which flattens the list and hides the group rows. - **Filter tags** (`.tag`). `--accent-soft` background with `--accent-deep` text. Pill shape. Hover swaps to `--highlight` background with `--tag-hover-border` border and ink text. Active state uses the warm `--tag-active-start` → `--tag-active-end` gradient with hero-ink text. Tag variants (`tag-group`, `tag-source`) inherit the base `.tag` style today and differ only at narrow widths (`tag-group` hides under 960px). Add a new variant only when a real visual difference is needed. - **Hero**. Magazine-cover headline, dark earthy ground, kicker and proof microcopy, primary CTA button using `--hero-btn-start` / `--hero-btn-end`. Subtle grid plus slow sheen. Respects `prefers-reduced-motion`. +- **Jump links**. One per use case in a category hero, plus one for the guide. Underlined like the intro's links, with a trailing ↓. Never tag-styled: tags filter the table or open another page, jump links only scroll. +- **Guide band**. Everything in a category intro after its "How to choose" list. It sits below the table on the hero's dark gradient, with its H2 above the text. - **Sponsor band**. Sits in the README header on `--bg-paper-strong`. Editorial layout, not a logo wall. Sponsor links share the global accent treatment. - **CTA**. Warm `--cta-bg`, full-bleed within shell. The button itself uses accent tokens. - **Footer**. Dark warm charcoal, part of the same system. Footer links share the global hover and focus treatment. From 2930319d48384a58f3ebaa921e833caae765b455 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 00:16:51 +0800 Subject: [PATCH 022/104] docs: add CMS category intro Explains when to pick Wagtail (developer-defined page types) versus django CMS (editors composing pages live), based on each project's own documentation. Co-Authored-By: Claude --- website/data/category_intros/cms.md | 12 ++++++++++++ 1 file changed, 12 insertions(+) create mode 100644 website/data/category_intros/cms.md diff --git a/website/data/category_intros/cms.md b/website/data/category_intros/cms.md new file mode 100644 index 00000000..fa32524f --- /dev/null +++ b/website/data/category_intros/cms.md @@ -0,0 +1,12 @@ +Both Python CMS picks run on Django: Wagtail for page types your developers define, like articles, and django CMS for editors building pages on the live site. + +How to choose: + +- Content types your developers define, like articles and events: Wagtail +- Editors composing pages from reusable components on the live site: django CMS + +Wagtail is [not an instant website in a box](https://docs.wagtail.org/en/stable/getting_started/the_zen_of_wagtail.html#wagtail-is-not-an-instant-website-in-a-box): expect to write code. Start a project with [`wagtail start`](https://docs.wagtail.org/en/stable/getting_started/quick_install.html). Each page type is [a Django model that inherits from `Page`](https://docs.wagtail.org/en/stable/topics/pages.html), so give each kind of content its own type with its own fields. An event page with a date and a location [can show up in a calendar](https://docs.wagtail.org/en/stable/getting_started/the_zen_of_wagtail.html#a-cms-should-get-information-out-of-an-editor-s-head-and-into-a-database-as-efficiently-and-directly-as-possible), and a styled heading on a generic page can't. Use [StreamField](https://docs.wagtail.org/en/stable/topics/streamfield.html) for pages without a fixed structure, like blog posts, and [snippets](https://docs.wagtail.org/en/stable/topics/snippets/index.html) for content that doesn't need its own page, like headers and footers. For a headless site, use its [built-in API](https://docs.wagtail.org/en/stable/advanced_topics/api/index.html). + +In django CMS, [editors compose pages from plugins](https://docs.django-cms.org/en/stable/explanation/philosophy.html#three-disciplines-three-surfaces) in a toolbar on the live site, and developers write standard Django. Each template [declares its placeholders](https://docs.django-cms.org/en/stable/tutorials/02-templates-placeholders.html) with `{% placeholder %}`, and `CMS_TEMPLATES` lists the templates editors can pick from. For a region that is the same on every page, like a footer, use [`{% static_alias %}`](https://docs.django-cms.org/en/stable/tutorials/02-templates-placeholders.html#a-reusable-region-with-static-alias), so the content is stored once. For content from your own app, ask [where it lives](https://docs.django-cms.org/en/stable/explanation/composition.html). If it fits in a placeholder on a page, write a plugin. If it has its own list view, detail view, and URL, mount it as an app with an apphook. The core [publishes as you edit](https://docs.django-cms.org/en/stable/explanation/publishing.html), which is rarely enough for a production site with editors, so add a versioning package. + +Both are built on Django. Wagtail [deploys like a Django site](https://docs.wagtail.org/en/stable/deployment/index.html). You can add either one to an existing Django project: Wagtail [integrates into one](https://docs.wagtail.org/en/stable/getting_started/integrating_into_django.html), and django CMS [doesn't make you rebuild around it](https://docs.django-cms.org/en/stable/explanation/philosophy.html#implications-for-projects). Neither fits a very small or static site. For a two-page brochure, django CMS calls itself [overkill](https://docs.django-cms.org/en/stable/explanation/philosophy.html#when-django-cms-may-not-be-the-right-fit), and a static site generator is lighter. From 5216fb3c1a0a4c721091a4db219d30ed6b275072 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 00:17:34 +0800 Subject: [PATCH 023/104] docs: add Image Processing category intro Co-Authored-By: Claude --- .../data/category_intros/image-processing.md | 30 +++++++++++++++++++ 1 file changed, 30 insertions(+) create mode 100644 website/data/category_intros/image-processing.md diff --git a/website/data/category_intros/image-processing.md b/website/data/category_intros/image-processing.md new file mode 100644 index 00000000..0a7bbcdb --- /dev/null +++ b/website/data/category_intros/image-processing.md @@ -0,0 +1,30 @@ +Pillow handles everyday edits, so it's the default Python image processing library. Use scikit-image for scientific analysis, and pyvips when memory is tight. + +How to choose: + +- Resizing, cropping, and converting images: Pillow +- Scientific image analysis on NumPy arrays: scikit-image +- Large images, or many of them, with little memory: pyvips +- ImageMagick's features from Python: Wand +- Removing photo backgrounds: rembg +- Resizing and cropping images on demand over HTTP: thumbor +- QR codes: qrcode +- Barcodes: python-barcode + +Pillow is [ideal for batch processing](https://pillow.readthedocs.io/en/stable/handbook/overview.html), like making thumbnails and converting between file formats. Open each file in a `with Image.open(path) as im:` block. Opening is [fast and independent of the file size](https://pillow.readthedocs.io/en/stable/handbook/tutorial.html), since Pillow reads the pixels only when it has to. To fit an image to a size, use `ImageOps.contain()`, `cover()`, `fit()`, or `pad()`. `thumbnail()` works too, but it [changes the image in place](https://pillow.readthedocs.io/en/stable/handbook/tutorial.html#relative-resizing). + +scikit-image [aims to be the reference library for scientific image analysis](https://scikit-image.org/docs/stable/about/values.html) in Python, and puts science ahead of photo editing. Images are plain NumPy arrays, so [standard NumPy operations](https://scikit-image.org/docs/stable/user_guide/numpy_images.html) work on them. To change an image's dtype, use `img_as_float()` or `img_as_ubyte()`, [never `astype`](https://scikit-image.org/docs/stable/user_guide/data_types.html), which doesn't rescale the values to the new dtype's range. + +pyvips builds a pipeline of operations and runs it only when you write the result. It streams the image a section at a time, so it [doesn't keep whole images in memory](https://github.com/libvips/pyvips). To shrink an image, use `pyvips.Image.thumbnail()` instead of resizing: it [loads and resizes in one step](https://www.libvips.org/API/current/developer-checklist.html), which is faster and uses less memory. When you read an image from top to bottom, open it with [`access="sequential"`](https://libvips.github.io/pyvips/intro.html). + +Wand is a [ctypes-based ImageMagick binding](https://docs.wand-py.org/en/latest/), so install ImageMagick's MagickWand library first. Its objects are resources like open files: [use them in a `with` block](https://docs.wand-py.org/en/latest/guide/resource.html) so they get closed. Wand's docs say to [never use Wand directly in an HTTP service](https://docs.wand-py.org/en/latest/guide/security.html) or on any public server. Hand the images to a background worker through a queue, and limit ImageMagick's resources and formats in its `policy.xml`. + +rembg runs as a [CLI, a Python library, an HTTP server, or a Docker container](https://github.com/danielgatis/rembg). In code, create a session once with `new_session()` and pass it to each `remove()` call, since `remove` otherwise [starts a new session every call](https://github.com/danielgatis/rembg/blob/main/USAGE.md). The model weights [carry their own licenses](https://github.com/danielgatis/rembg), separate from rembg's MIT license, so check the one you use before you ship it in a commercial product. + +thumbor is an HTTP server: you [set the size and crop in the image URL](https://github.com/thumbor/thumbor), and it [detects faces and important features](https://thumbor.readthedocs.io/en/latest/) to crop around them. Set a `SECURITY_KEY` so [every URL is signed](https://thumbor.readthedocs.io/en/latest/security.html) and nobody can tamper with it, and build those URLs in Python with [libthumbor](https://thumbor.readthedocs.io/en/latest/libraries.html). In production, [turn off `ALLOW_UNSAFE_URL`](https://thumbor.readthedocs.io/en/latest/configuration.html) and run [more than one instance](https://thumbor.readthedocs.io/en/latest/hosting.html) behind a load balancer. + +qrcode makes a QR code in one call: `qrcode.make("Some data")`. For more control, use the `QRCode` class, and pass `version=None` with `make(fit=True)` to pick the size for you. For SVG output, the README [recommends the path factory](https://github.com/lincolnloop/python-qrcode), `SvgPathImage`. If you style the code or embed an image, set error correction to high, since styled codes aren't guaranteed to work with all readers. + +python-barcode writes SVG with [no external dependencies](https://python-barcode.readthedocs.io/en/latest/). For PNG and other images, install the `python-barcode[images]` extra, which adds Pillow. Its docs [recommend SVG](https://python-barcode.readthedocs.io/en/latest/getting-started.html) unless your target can't use it, since vectors scale better. It calculates the checksum for you. + +Treat every uploaded image as untrusted. Pillow [guards against decompression bombs](https://pillow.readthedocs.io/en/stable/reference/Image.html) with a pixel limit, and its `Image.open(formats=...)` argument restricts the formats it tries. With pyvips, check the image dimensions before you process it, and [block the loaders](https://www.libvips.org/API/current/developer-checklist.html) libvips hasn't tested for security. With Wand, check each file's [magic bytes](https://docs.wand-py.org/en/latest/guide/security.html), never its extension or MIME type. From a9bb827b4abcd610cf8d2f129ae91ba7ea47d7fa Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 00:17:54 +0800 Subject: [PATCH 024/104] docs: add Code Analysis category intro Ruff for linting and formatting, a type checker alongside it, and pre-commit to run them, with per-tool guidance sourced from each project's own docs. Co-Authored-By: Claude --- website/data/category_intros/code-analysis.md | 43 +++++++++++++++++++ 1 file changed, 43 insertions(+) create mode 100644 website/data/category_intros/code-analysis.md diff --git a/website/data/category_intros/code-analysis.md b/website/data/category_intros/code-analysis.md new file mode 100644 index 00000000..4da863c7 --- /dev/null +++ b/website/data/category_intros/code-analysis.md @@ -0,0 +1,43 @@ +Run Ruff for Python code analysis: it lints and formats. Add a type checker, since Ruff doesn't check types, and pre-commit to run Ruff on every commit. + +How to choose: + +- Rules on which modules may import which: Import Linter +- Dead code: Vulture, or repowise to index the repo for your agent +- Deeply nested functions that are hard to read: complexipy +- Several analysis tools behind one command: Prospector +- Checks before every commit: pre-commit +- Linting, formatting, and import sorting: Ruff +- Formatting without Ruff: Black, with isort on its black profile +- Deeper inference and checks of your own: Pylint +- Flake8 plugins Ruff doesn't have: Flake8 +- Security issues: Bandit +- Refactoring: Rope +- Type checking: mypy, or Pyright, ty, or Pyrefly to also check unannotated code +- Type hints from the types your code sees at runtime: MonkeyType + +Ruff lints and formats in one tool, and its FAQ lists what it [can replace](https://docs.astral.sh/ruff/faq/#which-tools-does-ruff-replace): Flake8 and dozens of its plugins, Black, and isort. To sort imports and format, [run the linter, then the formatter](https://docs.astral.sh/ruff/formatter/#sorting-imports): `ruff check --select I --fix`, then `ruff format`. When you turn on a new rule in an existing codebase, [`--add-noqa`](https://docs.astral.sh/ruff/tutorial/#adding-rules) marks the current violations, so the rule only applies to new code. + +Pick one formatter and stay with it. Ruff's formatter is designed as a drop-in replacement for Black, but it's [not meant to be used interchangeably with Black](https://docs.astral.sh/ruff/formatter/) over time. If you pick Black, set isort's black profile [in a config file at the root of your repo](https://isort.readthedocs.io/en/latest/configuration/black_compatibility.html), so it applies however isort runs. To adopt Black, [reformat everything in one commit](https://black.readthedocs.io/en/stable/guides/introducing_black_to_your_project.html) and list that commit in `.git-blame-ignore-revs`, so `git blame` skips it. + +Ruff can be a [drop-in replacement for Flake8](https://docs.astral.sh/ruff/faq/#how-does-ruffs-linter-compare-to-flake8) when your code is formatted with Black and uses few or no Flake8 plugins. Keep Flake8 when you depend on a plugin Ruff doesn't cover. In pre-commit, install the plugin [through `additional_dependencies`](https://flake8.pycqa.org/en/latest/user/using-hooks.html). + +Pylint [doesn't trust your type hints](https://pylint.readthedocs.io/en/stable/). It infers the actual values instead, which makes it slower but finds more issues in code that isn't fully typed. You can also write plugins for checks of your own. On a legacy project, start with `--errors-only`, then turn more messages on over time. Because of its speed, Pylint's docs suggest running it [in CI or a pre-push hook](https://pylint.readthedocs.io/en/stable/user_guide/installation/pre-commit-integration.html), not on every commit. + +Bandit [finds common security issues](https://github.com/PyCQA/bandit) by walking each file's syntax tree. On an existing project, [save a baseline](https://bandit.readthedocs.io/en/latest/start.html) to ignore the findings you've judged to be non-issues. When you skip a line, write `# nosec B602` instead of a bare `# nosec`, so [a new issue on that line still shows up](https://bandit.readthedocs.io/en/latest/config.html). + +Ruff is [a linter, not a type checker](https://docs.astral.sh/ruff/faq/#how-does-ruff-compare-to-mypy-or-pyright-or-pyre), so add one. mypy is [designed for gradual typing](https://mypy.readthedocs.io/en/stable/), and by default it skips functions without annotations. On an existing codebase, its docs suggest [starting with part of the code](https://mypy.readthedocs.io/en/stable/existing_code.html), running it in CI early, turning on `check_untyped_defs` as soon as you can, and aiming for `mypy --strict`. + +Pyright, ty, and Pyrefly each come with a language server for your editor, and they check unannotated code too. Pyright [checks all code regardless of annotations](https://github.com/microsoft/pyright/blob/main/docs/mypy-comparison.md). Commit its config, [run it in CI](https://github.com/microsoft/pyright/blob/main/docs/getting-started.md), and turn on strict mode file by file with `# pyright: strict`. + +ty is [designed for adoption](https://docs.astral.sh/ty/), with support for partially typed code. Add it as a [dev dependency](https://docs.astral.sh/ty/installation/), so everyone runs the same version. Pick Pyrefly when your code uses [Pydantic, Django, or attrs](https://pyrefly.org/en/docs/compare/) and you don't want the overhead of plugins. To switch to it, `pyrefly init` [migrates your existing type checker config](https://pyrefly.org/en/docs/installation/), and `pyrefly suppress` marks the current errors as ignored, so you start from a clean check. + +Import Linter checks [contracts on your imports](https://import-linter.readthedocs.io/en/stable/), such as layers: higher layers may import lower ones, not the other way around. Vulture finds unused code. For its false positives, it recommends [a whitelist over `noqa` comments](https://github.com/jendrikseipp/vulture). After you delete dead code, run it again, since it may find more. complexipy measures how hard code is for people to understand, and its docs say to [run it alongside Ruff](https://complexipy.com/comparison-with-ruff/): Ruff catches wide functions, complexipy catches deep ones. On a large existing codebase, [take a snapshot](https://complexipy.com/usage-guide/) first, so only new complexity fails. + +Prospector wraps several analysis tools and aims to be [useful out of the box](https://github.com/prospector-dev/prospector). repowise indexes your repo for coding agents and developers, with dead code and git history in the same index. It's licensed under the [AGPL, or a commercial license](https://github.com/repowise-dev/repowise). + +Rope is a refactoring library, and its wiki suggests [starting with its language server plugin](https://github.com/python-rope/rope/wiki/How-to-use-Rope-in-my-IDE-or-Text-editor%3F) before the native editor integrations. MonkeyType records the types your code sees at runtime and writes annotations from them, which its docs call [an informative first draft](https://monkeytype.readthedocs.io/en/latest/) for you to check and fix. mypy's docs suggest [collecting types from test runs](https://mypy.readthedocs.io/en/stable/existing_code.html) this way. + +pre-commit runs your hooks [before every commit](https://pre-commit.com/). Run `pre-commit install` after every clone, `pre-commit run --all-files` when you add a hook, and the same command in CI. With Ruff's hooks, [put the lint hook with `--fix` before the formatter](https://docs.astral.sh/ruff/integrations/#pre-commit), since its fixes can leave code that needs reformatting. + +The checkers here do static code analysis: they check your code without running it. MonkeyType is the exception, since it records types while your code runs. Run Ruff [in your editor](https://docs.astral.sh/ruff/editors/) and on every commit, and Pylint and your type checker in CI. From 88cfcf6f4da79fc0e797b0918893c6fa28d7391d Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 00:27:02 +0800 Subject: [PATCH 025/104] docs: rewrite Computer Vision category intro Covers OpenCV as the default, Ultralytics YOLO for detection/segmentation/pose models and its AGPL-3.0/Enterprise License terms, Kornia for GPU-batch vision ops, FiftyOne for dataset curation, and pytesseract vs EasyOCR for OCR. Co-Authored-By: Claude --- .../data/category_intros/computer-vision.md | 28 +++++++++---------- 1 file changed, 14 insertions(+), 14 deletions(-) diff --git a/website/data/category_intros/computer-vision.md b/website/data/category_intros/computer-vision.md index 048fe971..26e1e35b 100644 --- a/website/data/category_intros/computer-vision.md +++ b/website/data/category_intros/computer-vision.md @@ -1,24 +1,24 @@ -For a Python computer vision library, start with OpenCV, and add Ultralytics YOLO to detect objects. For OCR, use pytesseract for scans and EasyOCR for photos. +Start with OpenCV when you need a Python computer vision library for images and video. To train and run detection models, use Ultralytics YOLO. How to choose: -- Reading, transforming, and writing images and video: OpenCV -- Detecting, segmenting, or tracking objects with a pretrained model: Ultralytics YOLO, which is AGPL-3.0 unless you buy an Enterprise License -- Image processing and augmentation on GPU batches, inside a PyTorch model or training loop: Kornia -- Browsing a dataset, finding label mistakes, and seeing where a model fails: FiftyOne -- Scanned pages and documents: pytesseract, on top of a Tesseract engine you install yourself -- Text in photos, without a separate OCR engine to install: EasyOCR +- Image and video processing: OpenCV +- Detection, segmentation, and pose models: Ultralytics YOLO +- Vision ops inside a PyTorch model: Kornia +- Dataset curation and model evaluation: FiftyOne +- OCR on clean, printed documents: pytesseract +- OCR on text in photos: EasyOCR -OpenCV ships as four wheels on PyPI, and you install exactly one, since they all use the same `cv2` namespace. On a server or in Docker, pick `opencv-python-headless`, which leaves out the GUI libraries. The opencv-python README says [you should always use it](https://github.com/opencv/opencv-python#installation-and-usage) unless you call `cv2.imshow`. The wheels are CPU-only; for CUDA, you build OpenCV from source. +OpenCV comes as [four pip packages that share the `cv2` namespace](https://github.com/opencv/opencv-python), so install only one: opencv-python for the main modules, or opencv-contrib-python to add the extra modules. If you never call `cv2.imshow` or you build your GUI with another toolkit, install the headless variant of either one, which also makes Docker images smaller. -With Ultralytics YOLO, load a pretrained checkpoint with `YOLO()`, fine-tune it with `model.train()`, and ship it with `model.export(format="onnx")`. Check the license before you build a product on it. Ultralytics says an Enterprise License is ["required if you want to use Ultralytics YOLO without open-sourcing your entire project"](https://www.ultralytics.com/license), even for models you trained yourself. +Ultralytics YOLO covers the [whole life of a model](https://docs.ultralytics.com/modes/): train, validate, predict, export, and track, from Python or the `yolo` command. Its docs recommend [starting training from a pretrained model](https://docs.ultralytics.com/modes/train/). To deploy, [export it](https://docs.ultralytics.com/modes/export/) to ONNX, TensorRT, CoreML, or another format. The code and the models you train with it are [AGPL-3.0](https://www.ultralytics.com/license), so unless you open-source your whole project, you need an Enterprise License. -Kornia works on PyTorch tensors, so its operators ["run wherever the tensor lives"](https://kornia.readthedocs.io/en/latest/) and take a whole batch in one call. Gradients flow through them too, so they can sit inside your model or your loss. Its augmentations expect [float tensors in [0, 1]](https://kornia.readthedocs.io/en/latest/augmentation.html): one still in [0, 255] doesn't raise an error, it just clips. +Kornia is a [differentiable computer vision library like OpenCV, with strong GPU support](https://kornia.readthedocs.io/en/latest/get-started/introduction.html). Every operator works on PyTorch tensors and supports autograd, so vision ops can run on the GPU and sit inside your training loop. -With FiftyOne, load your images and your model's predictions into a `fo.Dataset` and browse them with `fo.launch_app()`. Then run `compute_mistakenness()` to rank likely label mistakes, which [create an artificial ceiling](https://docs.voxel51.com/brain/index.html) on how good your model can get. Before you trust a test score, run `compute_leaky_splits()` too: duplicates across train and test make it easy to [overestimate your model](https://docs.voxel51.com/brain/index.html#leaky-splits). +FiftyOne works on the data side of a model. [Load your dataset and your model's predictions into it](https://docs.voxel51.com/user_guide/basics.html), then see where the model succeeds and fails, and find mistakes in your labels. It [integrates with Ultralytics](https://docs.voxel51.com/integrations/ultralytics.html), so you can run and fine-tune YOLO models on FiftyOne datasets. -pytesseract wraps the `tesseract` command, so install Tesseract separately and make sure it's on your `PATH`. Pass `lang=` for anything but English. Tesseract works best at 300 DPI or more, so scale small images up. It also [expects a page of text](https://tesseract-ocr.github.io/tessdoc/ImproveQuality.html) by default, so for a single line, pass `config="--psm 7"`. +pytesseract [wraps the Tesseract OCR engine](https://github.com/madmaze/pytesseract), which you install on its own, then put on your PATH or point `tesseract_cmd` at. Tesseract suits clean, printed text and needs no GPU. To get better results, [improve the image first](https://tesseract-ocr.github.io/tessdoc/ImproveQuality.html): Tesseract works best at 300 DPI or more, and retraining rarely helps unless you use an unusual font or a new language. -With EasyOCR, create one `easyocr.Reader` and reuse it: loading the model takes a while, and it [needs to be run only once](https://github.com/JaidedAI/EasyOCR#usage). +EasyOCR is a [general OCR that reads text in photos as well as in documents](https://www.jaided.ai/easyocr), in dozens of languages. It handles text in photos, where Tesseract struggles, but it's slow without a GPU. Create a `Reader` for your languages [once](https://github.com/JaidedAI/EasyOCR) and reuse it for every image, since that call loads the model into memory. -Watch the channel order when you pass images between libraries. OpenCV [loads color images as BGR](https://docs.opencv.org/4.x/d4/da8/group__imgcodecs.html). Ultralytics YOLO and EasyOCR take OpenCV images as they are, but pytesseract assumes RGB, so convert first with `cv2.cvtColor(img, cv2.COLOR_BGR2RGB)`. +Ultralytics YOLO, Kornia, and EasyOCR run on PyTorch. When you need a specific CUDA build, install PyTorch before the library, as [Ultralytics](https://docs.ultralytics.com/quickstart/) and [Kornia](https://kornia.readthedocs.io/en/latest/get-started/installation.html) recommend. EasyOCR's README says the same for Windows. From 3c2afa13bbdd400bcee0291f93c2273f8a446786 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 00:27:20 +0800 Subject: [PATCH 026/104] docs: rewrite Audio & Video Processing category intro Covers librosa for audio analysis, MoviePy for scripted video editing, VidGear for real-time video, Mutagen vs tinytag for tag read/write and licensing, and beets as a CLI tag/organization tool. Co-Authored-By: Claude --- .../category_intros/audio-video-processing.md | 32 ++++++++++--------- 1 file changed, 17 insertions(+), 15 deletions(-) diff --git a/website/data/category_intros/audio-video-processing.md b/website/data/category_intros/audio-video-processing.md index 55c34c36..857e36a1 100644 --- a/website/data/category_intros/audio-video-processing.md +++ b/website/data/category_intros/audio-video-processing.md @@ -1,25 +1,27 @@ -For a Python audio processing library, use librosa to analyze audio and music. For video, use MoviePy to edit clips in code, or VidGear for live streams. +Each job has its own pick: librosa is the Python audio processing library for music analysis, MoviePy edits video from a script, and Mutagen tags audio files. How to choose: -- Audio and music analysis, such as beat tracking or spectrograms: librosa -- Quick cuts, fades, and format conversion: pydub -- Editing video in code, such as cuts, titles, and compositing: MoviePy -- Webcams, live streams, and screen capture in real time: VidGear -- Reading tags, with one API for every format and an MIT license: tinytag -- Writing tags: Mutagen, which is GPL-licensed -- Organizing a whole music collection and fixing its tags from MusicBrainz: beets +- Cutting, joining, and fading audio files: pydub +- Music and audio analysis: librosa +- Editing video or making GIFs from a script: MoviePy +- Real-time video from cameras and network streams: VidGear +- Reading and writing tags across formats: Mutagen +- Reading tags only: tinytag +- Tagging and organizing your music collection: beets -With librosa, `librosa.load()` resamples everything to 22050 Hz by default. Pass `sr=None` to [keep the file's native sampling rate](https://librosa.org/doc/latest/api/generated/librosa.load.html), or `offset` and `duration` to load only part of a file. +pydub gives you a simple, high-level interface to cut, join, and fade audio. It opens and saves WAV files in pure Python, but [other formats like MP3 need FFmpeg](https://github.com/jiaaro/pydub#dependencies), so install FFmpeg with it. An AudioSegment is [immutable](https://github.com/jiaaro/pydub#quickstart): every operation returns a new one, so you can chain them, and every length and position is in milliseconds. -With pydub, load a file with `AudioSegment.from_file()`, slice it in milliseconds, and save it with `export()`. WAV works in pure Python, but every other format [needs FFmpeg](https://github.com/jiaaro/pydub#dependencies). The standard library dropped the `audioop` module pydub imports, so [install `audioop-lts`](https://github.com/jiaaro/pydub/issues/863) next to it. +librosa gives you [the foundational algorithms and tools for music information retrieval](https://librosa.org/doc/latest/index.html). By default, `librosa.load` resamples the signal and mixes stereo down to mono, and those defaults [suit most analysis tasks](https://librosa.org/doc/latest/auto_tutorials/01-intro/01-load.html); pass `sr=None` to keep the file's own sampling rate. When you need more control than `load` gives you, such as writing files, [its docs recommend using its audio I/O backend directly](https://librosa.org/doc/latest/ioformats.html). -A MoviePy script loads clips, changes them with `subclipped()` and the `with_*` methods, and calls `write_videofile()`. Open file clips in a `with` block, since each file clip [locks the file](https://zulko.github.io/moviepy/user_guide/loading.html) until it's closed. To only convert a video file, MoviePy's docs tell you to call FFmpeg directly: it's ["faster and more memory-efficient"](https://zulko.github.io/moviepy/getting_started/quick_presentation.html). +MoviePy is for [automating video editing](https://zulko.github.io/moviepy/getting_started/quick_presentation.html): processing many videos, composing them in complicated ways, or making videos and GIFs on a web server. A script loads clips, modifies them, puts them together, and writes the result. Modifying a clip [returns a new clip and leaves the original alone](https://zulko.github.io/moviepy/user_guide/modifying.html), and the computation happens at the final render. Open file clips in a `with` block, or [call `close()`](https://zulko.github.io/moviepy/user_guide/loading.html) when you're done, since each one holds a subprocess and a lock on the file. MoviePy can't stream video. For frame-by-frame analysis, its docs send you to a computer vision library. -MoviePy can't stream video, such as reading from a webcam, so use VidGear for that. It [needs OpenCV](https://abhitronix.github.io/vidgear/latest/installation/pip_install/) installed first. Its gears keep OpenCV's read loop: `CamGear(source=0).start()`, call `read()` until it returns `None`, then `stop()`. +VidGear is a framework for [real-time media applications](https://abhitronix.github.io/vidgear/latest/) built on OpenCV and FFmpeg. All its APIs [keep OpenCV's coding syntax](https://abhitronix.github.io/vidgear/latest/switch_from_cv/). Each task has [its own gear](https://abhitronix.github.io/vidgear/latest/gears/): CamGear reads cameras, network streams, and streaming sites in multiple threads. WriteGear writes frames to a video file or network stream, and StreamGear transcodes video into adaptive streaming formats. [Install OpenCV first](https://abhitronix.github.io/vidgear/latest/installation/pip_install/), since the core functions need it. -tinytag only reads tags: ["Support for changing/writing metadata will not be added."](https://github.com/tinytag/tinytag) To write them, use Mutagen. `mutagen.File()` guesses the format for you, and Mutagen [saves ID3 tags as v2.4](https://mutagen.readthedocs.io/en/latest/user/id3.html) unless you open and save them with `v2_version=3`. +Mutagen reads and writes tags with [roughly the same API across all tag formats](https://mutagen.readthedocs.io/en/latest/). `mutagen.File` [guesses the file type](https://mutagen.readthedocs.io/en/latest/user/gettingstarted.html). ID3 tags in MP3 files are highly structured; for common keys, use [the simpler EasyID3 interface](https://mutagen.readthedocs.io/en/latest/user/id3.html). Mutagen is licensed under GPL-2.0-or-later; if you only read tags, MIT-licensed tinytag avoids that. -beets runs from the command line: set `directory` and `library` in its config file, then run `beet import`. Its getting started guide recommends the autotagger and [importing a few albums at a time](https://beets.readthedocs.io/en/stable/guides/main.html). It also warns that importing can modify and move your files, so back them up first. +tinytag only reads metadata, and [writing support will not be added](https://github.com/tinytag/tinytag): its README points you to Mutagen for that. It's pure Python with no dependencies and gives you the same API for every format. `TinyTag.get()` returns an object with attributes like `artist` and `duration`. -For long files, `librosa.stream()` reads audio [block by block](https://librosa.org/doc/latest/api/generated/librosa.stream.html) instead of loading it all into memory; set `center=False` in the analyses you run on those blocks. pydub [loads the whole file into RAM](https://github.com/jiaaro/pydub/issues/51#issuecomment-35839094), so pass `start_second` and `duration` to `from_file()` when you only need a slice. +beets is a command-line music library manager, not a library you import: it [catalogs your collection and improves its metadata](https://beets.io/) as it goes. Install it [as a standalone tool](https://beets.readthedocs.io/en/stable/guides/installation.html), isolated from your system Python and other packages. `beet import` can modify and move your files, so [back up first and import a few albums at a time](https://beets.readthedocs.io/en/stable/guides/main.html). [Plugins](https://beets.readthedocs.io/en/stable/plugins/index.html) add commands, fetch extra data during import, and add metadata sources. + +FFmpeg sits under most of these projects: pydub needs it for any format other than WAV, MoviePy runs on it, and VidGear's WriteGear and StreamGear wrap it. When you only want to convert a video file or turn images into a movie, [call FFmpeg directly](https://zulko.github.io/moviepy/getting_started/quick_presentation.html). MoviePy's own docs say it's faster and uses less memory than going through MoviePy. From ee2874a463dfbfff27782ed2b12a8320e3fdd037 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 00:27:25 +0800 Subject: [PATCH 027/104] docs: rewrite Testing category intro Covers pytest as the default, a how-to-choose item per README subcategory, and a guide on Hypothesis, Playwright, tox/Nox, mocks, and coverage. Co-Authored-By: Claude --- website/data/category_intros/testing.md | 51 ++++++++++++++++--------- 1 file changed, 32 insertions(+), 19 deletions(-) diff --git a/website/data/category_intros/testing.md b/website/data/category_intros/testing.md index e7034229..2afb74b8 100644 --- a/website/data/category_intros/testing.md +++ b/website/data/category_intros/testing.md @@ -1,28 +1,41 @@ -Use pytest as your Python testing framework. Add Hypothesis to find the edge cases you missed, and Playwright to test in a real browser. +Plain assert statements and fixtures make pytest the Python testing framework for new code. Add Hypothesis for property-based tests, Playwright for browsers. How to choose: -- Unit and integration tests: pytest -- Property-based tests: Hypothesis -- Running the suite across Python versions or dependency sets: tox, or nox if you'd rather configure it in Python -- End-to-end browser tests: Playwright -- An existing Selenium suite, or tests spread across many machines with Selenium Grid: Selenium, or SeleniumBase for a pytest-ready framework on top of it -- Acceptance tests in a readable keyword syntax: Robot Framework -- Load testing: Locust -- An API with an OpenAPI or GraphQL schema: Schemathesis -- Faking HTTP: responses for Requests, RESPX for HTTPX, VCR.py to record and replay real traffic -- Freezing the clock: freezegun -- Test data: factory_boy for ORM models, Polyfactory for dataclasses and Pydantic models, Faker or Mimesis for single fake values -- Coverage: coverage.py +- Writing tests: pytest, plus Hypothesis for edge cases; Robot Framework for non-programmers +- Tests across Python versions: tox, or Nox to configure them in Python +- Browser tests: Playwright; Selenium or SeleniumBase for WebDriver suites +- Load tests written in Python: Locust +- Tests generated from an OpenAPI or GraphQL schema: Schemathesis +- Mocking: unittest.mock; responses, RESPX, or VCR.py for HTTP; FreezeGun for time +- Test objects: factory_boy for ORM models, Polyfactory for type hints +- Code coverage: Coverage.py +- Fake data: Faker or Mimesis -With pytest, write tests as plain functions with `assert`, and share setup through fixtures, which pytest says [offer dramatic improvements](https://docs.pytest.org/en/stable/explanation/fixtures.html) over xUnit-style setup and teardown. For a new project, its good practices recommend [the `importlib` import mode](https://docs.pytest.org/en/stable/explanation/goodpractices.html) and a `src` layout. You don't have to rewrite an old unittest suite first: pytest [runs it as is](https://docs.pytest.org/en/stable/how-to/unittest.html), so you can move it over one file at a time. +pytest lets you write tests with [plain `assert` statements](https://docs.pytest.org/en/stable/) and shows you what failed. It also runs your unittest suites as they are, so you can [move an old suite over bit by bit](https://docs.pytest.org/en/stable/how-to/unittest.html). Share setup through [fixtures](https://docs.pytest.org/en/stable/how-to/fixtures.html): use `yield` fixtures for teardown, and put the ones several test modules need in `conftest.py`. For a new project, pytest's docs recommend a src layout and the [importlib import mode](https://docs.pytest.org/en/stable/explanation/goodpractices.html). -A Hypothesis test is a pytest test with `@given` on top. You describe the inputs with strategies, and Hypothesis picks the values, [including edge cases you might not have thought about](https://hypothesis.readthedocs.io/en/latest/). It still works with fixtures and `parametrize`. +Hypothesis adds property-based tests to pytest or unittest. You describe the inputs with a strategy passed to [`@given`](https://hypothesis.readthedocs.io/en/latest/quickstart.html), and Hypothesis picks which ones to try, including edge cases you didn't think of. It's [an addition to unit tests, not always a replacement](https://hypothesis.readthedocs.io/en/latest/tutorial/introduction.html): start with round trips like encode/decode, and with tests you already parametrize. Use the [most general strategy](https://hypothesis.readthedocs.io/en/latest/explanation/domain.html) your test should pass for. -For the browser, Playwright calls its pytest plugin, pytest-playwright, [the recommended way to write end-to-end tests](https://playwright.dev/python/docs/intro). Find elements by what the user sees, [starting with `get_by_role()`](https://playwright.dev/python/docs/locators), and assert with `expect()`, which keeps retrying until the condition is met or it times out. +Robot Framework is a [keyword-driven framework for acceptance testing](https://robotframework.org/robotframework/latest/RobotFrameworkUserGuide.html). Tests are tables of keywords, and you build higher-level keywords out of existing ones. That suits teams where people who don't write Python read or write the tests. -Run tox or nox locally and in CI. pytest's own docs point to tox because it [tests the installed package, not your checkout](https://docs.pytest.org/en/stable/explanation/goodpractices.html), which catches packaging mistakes. +tox and Nox both run your tests in separate virtual environments, one per Python version or task. Configure tox [in TOML](https://tox.wiki/en/latest/tutorial/getting-started.html), in `tox.toml` or `pyproject.toml`. It tests the installed package, not your checkout, so it [catches packaging mistakes](https://docs.pytest.org/en/stable/explanation/goodpractices.html). Nox is configured in Python, in a `noxfile.py`, and tox's own docs point you to it [if tox configuration is too limiting](https://tox.wiki/en/latest/explanation.html). -For coverage, run `coverage run --branch -m pytest`. You don't need a pytest plugin for it: coverage.py calls one [unnecessary for most purposes](https://coverage.readthedocs.io/en/latest/). +Playwright was [created for end-to-end testing](https://playwright.dev/python/docs/intro) and runs Chromium, Firefox, and WebKit. Write your tests with its pytest plugin, which gives each test its own browser context. Playwright [waits for elements to be ready](https://playwright.dev/python/docs/actionability) before each action, so you don't add waits yourself. Find elements [by role, text, or test id](https://playwright.dev/python/docs/locators) rather than CSS or XPath, which break when the page changes. -When you mock with `unittest.mock`, [patch where an object is looked up](https://docs.python.org/3/library/unittest.mock.html#where-to-patch), not where it's defined. Pass `autospec=True` too, so a call with the wrong signature raises a `TypeError`. +Selenium drives real browsers through WebDriver, on your machine or on remote ones through Selenium Grid. Keep it for the WebDriver suites you already have. It [doesn't structure your test suite for you](https://www.selenium.dev/documentation/test_practices/), so run it under a test runner like pytest, and use [explicit waits](https://www.selenium.dev/documentation/webdriver/waits/) for the exact condition you need. SeleniumBase builds on Selenium's WebDriver APIs and [runs under pytest](https://github.com/seleniumbase/SeleniumBase), and its methods wait for elements that need time to load. + +For load tests, Selenium's docs [advise against using it](https://www.selenium.dev/documentation/test_practices/discouraged/performance_testing/); use Locust. You [write the tests in regular Python code](https://docs.locust.io/en/stable/what-is-locust.html): a `User` class with `@task` methods. When you need more load, [run one worker per CPU core](https://docs.locust.io/en/stable/running-distributed.html). + +Schemathesis generates property-based tests from your OpenAPI or GraphQL schema, using Hypothesis under the hood. Its docs [recommend the CLI for most users](https://schemathesis.readthedocs.io/en/stable/faq/), since the pytest integration has fewer features. + +unittest.mock ships with Python. [Patch where an object is looked up](https://docs.python.org/3/library/unittest.mock.html), not where it's defined, and add `autospec=True` so your tests fail when the real API changes. For code you own, pytest's docs suggest you [pass dependencies in](https://docs.pytest.org/en/stable/how-to/monkeypatch.html) rather than patch them. + +For HTTP, pick the mock that matches your client: responses for requests, and RESPX for HTTPX. Both raise an error on requests you didn't mock. VCR.py records real responses to a cassette file and replays them, with many clients including both. [Filter out credentials](https://vcrpy.readthedocs.io/en/latest/advanced.html) before you commit cassettes. For time, FreezeGun [freezes `datetime` and `time`](https://github.com/spulec/freezegun) at the moment you choose. + +factory_boy [replaces static fixtures with factories](https://factoryboy.readthedocs.io/en/stable/) that set only the fields a test cares about, and works with Django, SQLAlchemy, and MongoDB models. Polyfactory [builds objects from type hints](https://polyfactory.litestar.dev/latest/): dataclasses, TypedDicts, Pydantic models, and more. + +For the data itself, Faker [generates localized fake data](https://faker.readthedocs.io/en/master/) and comes with a pytest fixture; factory_boy uses it too. Mimesis is [fully typed and generates data from schemas](https://mimesis.name/latest/about.html), in many languages. + +Coverage.py measures which lines your tests run. Run pytest under it with [`coverage run -m pytest`](https://coverage.readthedocs.io/en/latest/), which its docs say is enough for most purposes, and include your tests in the measurement. It measures lines by default; add [`--branch`](https://coverage.readthedocs.io/en/latest/branch.html) to see which branches never ran. + +Write each test so it [runs in any order](https://www.selenium.dev/documentation/test_practices/discouraged/test_dependency/), without relying on other tests. A [flaky test](https://docs.pytest.org/en/stable/explanation/flaky.html) usually means state the test doesn't control, and random data is one such state: seed Faker, Mimesis, factory_boy, and Polyfactory so a [failing build reproduces](https://factoryboy.readthedocs.io/en/stable/). From c52f21f99d18b0e1e324f6391ec3ce542e364bb9 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 00:27:29 +0800 Subject: [PATCH 028/104] docs: rewrite CLI Development category intro Covers argparse for basic apps, Click or Typer beyond that, Rich for output, and TUI frameworks. Co-Authored-By: Claude --- .../data/category_intros/cli-development.md | 46 ++++++++++++------- 1 file changed, 29 insertions(+), 17 deletions(-) diff --git a/website/data/category_intros/cli-development.md b/website/data/category_intros/cli-development.md index 018debf9..9c2009bc 100644 --- a/website/data/category_intros/cli-development.md +++ b/website/data/category_intros/cli-development.md @@ -1,26 +1,38 @@ -For a Python CLI framework, use Typer, or argparse when a script can't take dependencies. For a full-screen terminal app, use Textual. +Python's own argparse covers basic command-line apps. Beyond that, pick Click or Typer as your Python CLI library, and style the output with Rich. How to choose: -- A new CLI: Typer -- A script that runs on the standard library alone: argparse -- Decorators instead of type hints, or subcommands loaded lazily at runtime: Click -- A quick CLI over an existing function or class, for debugging or exploring code: Fire -- An interactive prompt with completion, history, and key bindings: prompt_toolkit -- Colored output, tables, and pretty logs: Rich -- A progress bar on a loop: tqdm, or alive-progress for animated bars -- A full-screen terminal UI: Textual, or urwid inside an event loop you already run, such as Twisted or Trio -- ASCII animations and effects: asciimatics -- ANSI colors on Windows consoles: colorama +- Basic command-line app with no dependencies: argparse +- Nested commands, or subcommands loaded lazily: Click +- Options declared once as type-hinted function parameters: Typer +- Interactive prompts and REPLs: prompt_toolkit +- CLI from existing code without writing a parser: Python Fire +- Progress bar for a loop: tqdm, or alive-progress for animated bars +- Colored text, tables, and logs: Rich, or colorama for ANSI colors on Windows +- Full-screen app in the terminal or a browser: Textual +- Terminal widgets on an event loop you already run: Urwid +- Full-screen forms and ASCII animations: asciimatics -With Typer, declare each parameter with `Annotated`. Its tutorial keeps repeating one line: ["Prefer to use the `Annotated` version if possible."](https://typer.tiangolo.com/tutorial/arguments/optional/) Create one `typer.Typer()` app and register your commands on it. Test them with `CliRunner`, which calls the app with a list of arguments and gives you back the exit code and output. +argparse is in the standard library. Python's docs call it [the recommended choice](https://docs.python.org/3/library/optparse.html#choosing-an-argument-parser) when you have no more specific needs, since it gives you the most out of the box for the least code. It [writes the help and usage messages](https://docs.python.org/3/library/argparse.html) and reports invalid arguments for you. For commands written as decorated functions, the same docs point to Click, and for a CLI that works with static type checking, to Typer. -Typer builds on Click, so reach for Click itself when you'd rather write decorators, or when your app has so many subcommands that they should [load lazily at runtime](https://click.palletsprojects.com/en/stable/). Group commands with `@click.group()`, and test them with `click.testing.CliRunner`. +Click builds a CLI from [commands declared with decorators](https://click.palletsprojects.com/en/stable/quickstart/). You can nest them to any depth, and Click can [load subcommands lazily](https://click.palletsprojects.com/en/stable/) at runtime. It also [reads option values from environment variables](https://click.palletsprojects.com/en/stable/why/) and comes with helpers for ANSI colors, terminal size, and launching editors. -Use argparse when the tool has to run without installing anything. Python's own docs call it [the recommended choice](https://docs.python.org/3/library/optparse.html#choosing-an-argument-parser) among the standard library's parsers, and `add_subparsers()` handles subcommands. +Typer builds a CLI from your function signatures: you [declare each argument and option once](https://typer.tiangolo.com/), as a type-hinted function parameter. Write those hints with `Annotated`, which its tutorial [recommends wherever possible](https://typer.tiangolo.com/tutorial/arguments/optional/). Its docs say you get [the best results by pairing it with Rich](https://typer.tiangolo.com/tutorial/printing/): Typer structures the commands and options, and Rich displays the output. -With Rich, create one `Console` and share it across the app: ["Most applications will require a single Console instance"](https://rich.readthedocs.io/en/stable/console.html). Add a second one with `stderr=True` for errors, and pass `RichHandler` to `logging.basicConfig()` so your logs get the same formatting. +prompt_toolkit is for interactive input. It can [replace GNU readline](https://python-prompt-toolkit.readthedocs.io/en/stable/) or build full-screen apps. For a REPL, use a [`PromptSession`](https://python-prompt-toolkit.readthedocs.io/en/stable/pages/asking_for_input.html), which keeps the history for the whole session. -With Textual, keep styles in a separate `.tcss` file and run `textual run --dev` to [live edit them](https://textual.textualize.io/guide/CSS/). For tests, `run_test()` starts the app headless and gives you a `Pilot` that presses keys and clicks widgets. +Python Fire turns any Python object into a CLI: [call `fire.Fire()`](https://github.com/google/python-fire/blob/master/docs/guide.md) at the end of your program. Its README pitches it for [developing and debugging your code](https://github.com/google/python-fire), exploring existing code, and turning other people's code into a CLI. It takes each argument's type from the value you pass, not from the function signature. -Whatever you pick, ship the CLI as a package, not a script. Click's docs put it plainly: ["It's recommended to write command line utilities as installable packages with entry points instead of telling users to run `python hello.py`."](https://click.palletsprojects.com/en/stable/entry-points/) Declare the command under `[project.scripts]` in `pyproject.toml`, and users can install it with pipx or `uv tool install`. +tqdm adds a progress bar to any loop: [wrap the iterable in `tqdm()`](https://github.com/tqdm/tqdm). Import it from `tqdm.auto`, which picks the console bar or the Jupyter widget for you. alive-progress wraps the loop in an [`alive_bar` context manager](https://github.com/rsalmei/alive-progress) instead, with a spinner that speeds up and slows down with your throughput. + +Rich writes colored text, tables, Markdown, and syntax-highlighted code to the terminal. Start with its [drop-in `print`](https://rich.readthedocs.io/en/stable/introduction.html), then create [one `Console` at module level](https://rich.readthedocs.io/en/stable/console.html) for the rest of your app. For colored logs, send the logging module's output through [`RichHandler`](https://rich.readthedocs.io/en/stable/logging.html). + +colorama makes ANSI escape codes work on Windows and does nothing on other platforms. If that's all you need, call [`just_fix_windows_console()`](https://github.com/tartley/colorama). + +Textual apps run in the terminal, and the `textual serve` command from textual-dev [serves them in a browser](https://textual.textualize.io/guide/devtools/). Build one by [subclassing `App`](https://textual.textualize.io/guide/app/), and style its widgets with [CSS](https://textual.textualize.io/guide/CSS/). + +Urwid is a [console widget construction set](https://urwid.org/manual/overview.html) rather than a finished UI library. It runs on [your choice of event loop](https://github.com/urwid/urwid): asyncio, or another one you already use. It's [licensed under the LGPL](https://github.com/urwid/urwid/blob/master/COPYING), while Textual and asciimatics use permissive licenses. + +asciimatics does [full-screen text UIs, from interactive forms to ASCII animations](https://github.com/peterbrittain/asciimatics): create a Screen, build a Scene from Effect objects, and let the Screen play it. + +Ship your CLI as an installable package with a `[project.scripts]` entry point, not a file users run with `python`. [Click recommends it](https://click.palletsprojects.com/en/stable/entry-points/), and shell completion in Click and [Typer](https://typer.tiangolo.com/tutorial/package/) works through that entry point. Test your commands in-process: Click and Typer both provide a [`CliRunner`](https://click.palletsprojects.com/en/stable/testing/), and Textual's [`run_test()`](https://textual.textualize.io/guide/testing/) drives your app as if you were using the keyboard and mouse. From e5088a6f5caf38ec1c9f8010e7fe71711c71b850 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 00:27:34 +0800 Subject: [PATCH 029/104] docs: rewrite Data Validation category intro Covers Pydantic for API input and config, Pandera for dataframes, and jsonschema for JSON Schema validation. Co-Authored-By: Claude --- .../data/category_intros/data-validation.md | 18 ++++++++---------- 1 file changed, 8 insertions(+), 10 deletions(-) diff --git a/website/data/category_intros/data-validation.md b/website/data/category_intros/data-validation.md index 3729ae2a..f5bd96ea 100644 --- a/website/data/category_intros/data-validation.md +++ b/website/data/category_intros/data-validation.md @@ -1,17 +1,15 @@ -For a Python validation library, use Pydantic for data coming into your app, jsonschema when a JSON Schema is the contract, and Pandera for dataframes. +Validate API input and config with Pydantic, the Python data validation library built on type hints. Use Pandera for dataframes, jsonschema for JSON Schema. How to choose: -- API payloads, config files, anything you can describe with type hints: Pydantic -- A JSON Schema shared with other languages or teams: jsonschema -- pandas, Polars, or PySpark dataframes: Pandera +- API input, forms, and config: Pydantic +- Data checked against a JSON Schema document: jsonschema +- pandas, polars, or PySpark dataframes: Pandera -With Pydantic, describe your data as a `BaseModel` with type hints. Parse JSON with `model_validate_json()`, not `model_validate(json.loads(...))`, as [its performance tips](https://pydantic.dev/docs/validation/latest/concepts/performance/) recommend. For a type that isn't a model, like `list[Item]`, create one `TypeAdapter` and reuse it. +Pydantic builds the schema from your [type hints](https://pydantic.dev/docs/validation/latest/get-started/why/) and guarantees the types of the [output, not the input](https://pydantic.dev/docs/validation/latest/concepts/models/): by default, a numeric string passed to an int field comes out as an int. Where a wrong type should raise an error instead, turn on [strict mode](https://pydantic.dev/docs/validation/latest/concepts/strict_mode/) per field or per model. [Validate incoming JSON directly](https://pydantic.dev/docs/validation/latest/concepts/performance/) instead of parsing it into a dict first. To load config from environment variables, use [Pydantic Settings](https://pydantic.dev/docs/validation/latest/concepts/pydantic_settings/). -Pydantic is forgiving by default: it turns `"123"` into `123` and ignores fields your model doesn't declare. When that's too loose, turn on [strict mode](https://pydantic.dev/docs/validation/latest/concepts/strict_mode/) or set `extra='forbid'`. +jsonschema is an [implementation of the JSON Schema specification](https://python-jsonschema.readthedocs.io/en/stable/), so one schema can [work across different systems and platforms](https://json-schema.org/overview/what-is-jsonschema). When you validate many instances against one schema, [create a validator for your schema's draft once and call its `validate` method](https://python-jsonschema.readthedocs.io/en/stable/validate/). Use [`iter_errors()`](https://python-jsonschema.readthedocs.io/en/stable/errors/) to report every error, not only the first. To enforce `format` keywords such as dates or emails, [hook a format checker](https://python-jsonschema.readthedocs.io/en/stable/validate/#validating-formats) into the validator. -With jsonschema, `validate()` checks the schema itself on every call. To validate many documents against one schema, [create a validator once](https://python-jsonschema.readthedocs.io/en/stable/validate/) and reuse it, like `Draft202012Validator(schema)`. The `format` keyword checks nothing until you pass a `format_checker`, and [`default` doesn't fill in missing fields](https://python-jsonschema.readthedocs.io/en/stable/faq/). +Pandera validates [dataframe-like objects](https://pandera.readthedocs.io/en/stable/): define a schema once and use it on pandas, polars, PySpark, and other dataframe libraries. Write the schema as a DataFrameModel class, [much like a Pydantic model](https://pandera.readthedocs.io/en/stable/dataframe_models.html), and add the `check_types()` decorator to validate at run time. To check an existing pipeline, put [`check_input()` and `check_output()`](https://pandera.readthedocs.io/en/stable/decorators.html) on its functions. To see every failure in one run instead of only the first, validate with [`lazy=True`](https://pandera.readthedocs.io/en/stable/lazy_validation.html). -With Pandera, write a `DataFrameModel`, which [works like a Pydantic model](https://pandera.readthedocs.io/en/stable/dataframe_models.html) for your columns. Decorate pipeline functions with `@pa.check_types` to validate what goes in and what comes out. Pass `lazy=True` to [get every failure in one report](https://pandera.readthedocs.io/en/stable/lazy_validation.html), not just the first. - -Validate once, where untrusted data enters your program: a request body, a config file, a CSV upload. After that, [Pydantic guarantees](https://pydantic.dev/docs/validation/latest/concepts/models/) the fields match their types, so don't check them again deeper in. The same goes for schemas: if other teams need a JSON Schema, [generate it from your Pydantic models](https://pydantic.dev/docs/validation/latest/concepts/json_schema/) instead of writing it twice. +Pick by the shape of your data. Running a dataframe through a Pydantic model row by row [might not scale](https://pandera.readthedocs.io/en/stable/pydantic_integration.html) to larger datasets, so use Pandera there; a DataFrameModel can still be a field in a Pydantic model. Pydantic can [generate a JSON Schema](https://pydantic.dev/docs/validation/latest/concepts/json_schema/) from any model for tools that read the format. jsonschema works the other way: it validates data against a JSON Schema document you already have. From 62a4e7c4e5a50c4583bfa853dbab00a15dd67b10 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 00:20:56 +0800 Subject: [PATCH 030/104] fix: make tags link to category pages instead of filtering in place Clicking a category or group tag filtered the homepage table in place while rewriting the address bar to the category URL, so one URL rendered two different pages and readers never reached category pages with their intros and guides; category pages also showed a redundant self-referential filter bar. Co-Authored-By: Claude --- website/build.py | 8 --- website/static/main.js | 91 +++------------------------------ website/static/style.css | 49 +----------------- website/templates/category.html | 16 +----- website/templates/index.html | 16 +----- website/tests/test_build.py | 85 ++++-------------------------- 6 files changed, 18 insertions(+), 247 deletions(-) diff --git a/website/build.py b/website/build.py index 9164ca1a..70bbdc14 100644 --- a/website/build.py +++ b/website/build.py @@ -678,12 +678,7 @@ def build(repo_root: Path) -> None: filter_urls: dict[str, str] = dict(category_urls) for group in parsed_groups: filter_urls[group["name"]] = group_path(group["slug"]) - for entry in entries: - for sub in entry.get("subcategories", []): - filter_urls[sub["value"]] = sub["url"] builtin_entries = [e for e in entries if e.get("source_type") == BUILTIN_FILTER] - if builtin_entries: - filter_urls[BUILTIN_FILTER] = BUILTIN_PATH env = Environment( loader=FileSystemLoader(website / "templates"), @@ -696,7 +691,6 @@ def build(repo_root: Path) -> None: shutil.rmtree(site_dir) site_dir.mkdir(parents=True) - filter_urls_json = json.dumps(filter_urls, sort_keys=True, ensure_ascii=False).replace(" None: sponsors=sponsors, category_urls=category_urls, filter_urls=filter_urls, - filter_urls_json=filter_urls_json, homepage_json_ld=homepage_json_ld, ), encoding="utf-8", @@ -762,7 +755,6 @@ def build(repo_root: Path) -> None: category_urls=category_urls, current_path=current_path, filter_urls=filter_urls, - filter_urls_json=filter_urls_json, parent_category=parent_category, group_categories=group_categories, category_json_ld=category_json_ld, diff --git a/website/static/main.js b/website/static/main.js index 307e3f92..919aefd4 100644 --- a/website/static/main.js +++ b/website/static/main.js @@ -10,15 +10,10 @@ const defaultSort = table && table.dataset.defaultSort === "editorial" ? { col: "editorial", order: "asc" } : { col: "downloads", order: "desc" }; -let activeFilter = null; let activeSort = defaultSort; const searchInput = document.querySelector(".search"); -const filterBar = document.querySelector(".filter-bar"); -const filterValue = document.querySelector(".filter-value"); -const filterClear = document.querySelector(".filter-clear"); const noResults = document.querySelector(".no-results"); const rows = document.querySelectorAll(".table tbody tr.row"); -const tags = document.querySelectorAll(".tag"); const tbody = document.querySelector(".table tbody"); const groupRows = document.querySelectorAll(".table tbody tr.group-row"); @@ -145,7 +140,7 @@ function collapseAll() { function applyFilters() { const query = searchInput ? searchInput.value.toLowerCase().trim() : ""; - const descRowsVisible = !isIndexDocument || activeFilter !== null; + const descRowsVisible = !isIndexDocument; let visibleCount = 0; collapseAll(); @@ -153,12 +148,7 @@ function applyFilters() { rows.forEach(function (row) { let show = true; - if (activeFilter) { - const rowTags = row.dataset.tags; - show = rowTags ? rowTags.split("||").includes(activeFilter) : false; - } - - if (show && query) { + if (query) { if (!row._searchText) { let text = row.textContent.toLowerCase(); if (row._descRow) { @@ -200,35 +190,12 @@ function applyFilters() { if (noResults) noResults.hidden = visibleCount > 0; - tags.forEach(function (tag) { - tag.classList.toggle("active", activeFilter === tag.dataset.value); - }); - - if (filterBar) { - if (activeFilter) { - filterBar.classList.add("visible"); - if (filterValue) filterValue.textContent = activeFilter; - } else { - filterBar.classList.remove("visible"); - } - } - updateURL(); } -const filterUrlsScript = document.getElementById("filter-urls"); -const filterToUrl = filterUrlsScript - ? JSON.parse(filterUrlsScript.textContent) - : {}; - const isIndexDocument = location.pathname === "/" || location.pathname === "/index.html"; -const urlToFilter = {}; -Object.keys(filterToUrl).forEach(function (k) { - urlToFilter[filterToUrl[k]] = k; -}); - function buildQueryString() { const params = new URLSearchParams(); const query = searchInput ? searchInput.value.trim() : ""; @@ -243,9 +210,7 @@ function buildQueryString() { function updateURL() { if (!isIndexDocument) return; - const path = - activeFilter && filterToUrl[activeFilter] ? filterToUrl[activeFilter] : "/"; - history.replaceState(null, "", path + buildQueryString()); + history.replaceState(null, "", "/" + buildQueryString()); } function getSortValue(row, col) { @@ -340,8 +305,8 @@ function updateSortIndicators() { // Expand/collapse: event delegation on tbody if (tbody) { tbody.addEventListener("click", function (e) { - // Don't toggle if clicking a link or tag button - if (e.target.closest("a") || e.target.closest(".tag")) return; + // Don't toggle if clicking a link + if (e.target.closest("a")) return; let row = e.target.closest("tr.row"); if (!row) { @@ -370,36 +335,6 @@ if (tbody) { }); } -tags.forEach(function (tag) { - tag.addEventListener("click", function (e) { - e.preventDefault(); - const value = tag.dataset.value; - const url = tag.dataset.url; - if (isIndexDocument) { - activeFilter = activeFilter === value ? null : value; - if (activeFilter && url) { - history.pushState(null, "", url + buildQueryString()); - } else { - history.pushState(null, "", "/" + buildQueryString()); - } - applyFilters(); - } else if (url) { - window.location.href = url + "#library-index"; - } - }); -}); - -if (filterClear) { - filterClear.addEventListener("click", function () { - if (!isIndexDocument) { - window.location.href = "/#library-index"; - return; - } - activeFilter = null; - applyFilters(); - }); -} - const noResultsClear = document.querySelector(".no-results-clear"); if (noResultsClear) { noResultsClear.addEventListener("click", function () { @@ -408,7 +343,6 @@ if (noResultsClear) { return; } if (searchInput) searchInput.value = ""; - activeFilter = null; applyFilters(); }); } @@ -451,7 +385,6 @@ if (searchInput) { } if (e.key === "Escape" && document.activeElement === searchInput) { searchInput.value = ""; - activeFilter = null; applyFilters(); searchInput.blur(); } @@ -512,20 +445,8 @@ if (backToTop) { ) { activeSort = { col: sort, order: order }; } - const matched = urlToFilter[location.pathname]; - if (matched) activeFilter = matched; - if (q || activeFilter || sort) { + if (q || sort) { sortRows(); } - if (activeFilter) { - applyFilters(); - } updateSortIndicators(); })(); - -window.addEventListener("popstate", function () { - if (!isIndexDocument) return; - const matched = urlToFilter[location.pathname]; - activeFilter = matched || null; - applyFilters(); -}); diff --git a/website/static/style.css b/website/static/style.css index f7bddd67..4e067c06 100644 --- a/website/static/style.css +++ b/website/static/style.css @@ -265,8 +265,7 @@ kbd { .hero-topbar-link:active, .hero-action:active, -.tag:active, -.filter-clear:active { +.tag:active { transform: translateY(1px); } @@ -436,7 +435,6 @@ kbd { .hero-topbar-link:focus-visible, .hero-category-link:focus-visible, .search:focus-visible, -.filter-clear:focus-visible, .tag:focus-visible, .jump-link:focus-visible, .back-to-top:focus-visible, @@ -771,51 +769,6 @@ kbd { 0 1.6rem 3rem -2rem var(--search-focus-shadow); } -.filter-bar { - display: flex; - align-items: center; - gap: 0.75rem; - min-height: 2.3rem; - font-size: var(--text-sm); - color: var(--ink-soft); - opacity: 0; - transform: translateY(-0.4rem); - pointer-events: none; - transition: - opacity 180ms ease, - transform 180ms cubic-bezier(0.22, 1, 0.36, 1); -} - -.filter-bar.visible { - opacity: 1; - transform: translateY(0); - pointer-events: auto; -} - -.filter-bar strong { - color: var(--ink); -} - -.filter-clear { - border: 1px solid var(--line); - border-radius: 999px; - background: var(--bg-paper); - color: var(--ink-soft); - padding: 0.42rem 0.82rem; - cursor: pointer; - transition: - border-color 180ms ease, - color 180ms ease, - background-color 180ms ease, - transform 180ms ease; -} - -.filter-clear:hover { - color: var(--ink); - background: var(--accent-soft); - border-color: oklch(68% 0.08 58 / 0.5); -} - .table-wrap { width: 100%; border-top: 1px solid var(--line); diff --git a/website/templates/category.html b/website/templates/category.html index a9618aef..7528da00 100644 --- a/website/templates/category.html +++ b/website/templates/category.html @@ -78,7 +78,6 @@ {% macro entry_rows(entry, index) %} {% for subcat in entry.subcategories %} - + {{ subcat.name }} {% endfor %} @@ -132,8 +131,6 @@ {{ cat }} {% endfor %} @@ -142,8 +139,6 @@ {{ entry.groups[0] }} @@ -152,8 +147,6 @@ Stdlib @@ -211,7 +204,6 @@ {% endmacro %} -
    {% if entry_groups %} @@ -255,12 +247,6 @@ aria-label="Search projects" />
    -
    - Filtering for - -

    Results

    diff --git a/website/templates/index.html b/website/templates/index.html index b5540624..e4d95d10 100644 --- a/website/templates/index.html +++ b/website/templates/index.html @@ -96,7 +96,6 @@
    {% endif %} -
    @@ -132,12 +131,6 @@ aria-label="Search projects" />
    -
    - Filtering for - -

    Results

    @@ -175,7 +168,6 @@ {% for entry in entries %} {% for subcat in entry.subcategories %} - + {{ subcat.name }} {% endfor %} {% for cat in entry.categories %} {{ cat }} {% endfor %} {{ entry.groups[0] }} @@ -245,8 +233,6 @@ Stdlib diff --git a/website/tests/test_build.py b/website/tests/test_build.py index cfebc616..f2448e90 100644 --- a/website/tests/test_build.py +++ b/website/tests/test_build.py @@ -280,7 +280,6 @@ class TestBuild: parser.feed(category_html) assert 'href="/categories/widgets/"' in index_html - assert 'data-value="Widgets"' in index_html assert parser.title.strip() == "Python Widgets Libraries - Awesome Python" assert parser.meta_by_name["description"] == "Widget libraries. Also see awesome-widgets. Explore 2 curated Python projects in Widgets." assert parser.links_by_rel["canonical"] == "https://awesome-python.com/categories/widgets/" @@ -816,75 +815,6 @@ class TestBuild: {"@type": "ListItem", "position": 2, "name": "Sponsorship", "item": "https://awesome-python.com/sponsorship/"}, ] - def test_index_embeds_filter_urls_json(self, tmp_path): - readme = textwrap.dedent("""\ - # T - - ## Projects - - **AI & ML** - - ## Deep Learning - - - [dl1](https://example.com/dl1) - DL. - - ## Machine Learning - - - Classical - - - [ml1](https://example.com/ml1) - ML. - - # Contributing - - Done. - """) - self._copy_real_templates(tmp_path) - (tmp_path / "README.md").write_text(readme, encoding="utf-8") - build(tmp_path) - - site = tmp_path / "website" / "output" - index_html = (site / "index.html").read_text(encoding="utf-8") - - marker = '", start) - data = json.loads(index_html[start:end]) - - assert data["Deep Learning"] == "/categories/deep-learning/" - assert data["Machine Learning"] == "/categories/machine-learning/" - assert data["AI & ML"] == "/categories/ai-ml/" - assert data["Machine Learning > Classical"] == "/categories/machine-learning/classical/" - - def test_filter_urls_json_escapes_closing_script_tag(self, tmp_path): - readme = textwrap.dedent("""\ - # T - - ## Projects - - ## Sneaky - - - [a](https://example.com) - A. - - # Contributing - - Done. - """) - self._copy_real_templates(tmp_path) - (tmp_path / "README.md").write_text(readme, encoding="utf-8") - build(tmp_path) - - site = tmp_path / "website" / "output" - index_html = (site / "index.html").read_text(encoding="utf-8") - - marker = '", start) - block = index_html[start:end] - assert "" not in block - data = json.loads(block) - assert any("Sneaky" in key for key in data) - def test_build_creates_group_pages(self, tmp_path): readme = textwrap.dedent("""\ # T @@ -925,7 +855,7 @@ class TestBuild: assert "wf1" in web_dev assert "dl1" not in web_dev - def test_tag_buttons_have_data_url(self, tmp_path): + def test_tags_link_to_pages_and_subcategory_anchors(self, tmp_path): readme = textwrap.dedent("""\ # T @@ -950,11 +880,14 @@ class TestBuild: site = tmp_path / "website" / "output" index_html = (site / "index.html").read_text(encoding="utf-8") - assert 'data-value="Deep Learning"' in index_html - assert 'data-url="/categories/deep-learning/"' in index_html - assert 'data-value="AI & ML"' in index_html or 'data-value="AI & ML"' in index_html - assert 'data-url="/categories/ai-ml/"' in index_html - assert 'data-url="/categories/deep-learning/vision/"' in index_html + category_html = (site / "categories" / "deep-learning" / "index.html").read_text(encoding="utf-8") + + for html in (index_html, category_html): + assert 'href="/categories/deep-learning/#vision"' in html + assert 'href="/categories/ai-ml/"' in html + assert "data-url=" not in html + assert 'href="/categories/deep-learning/"' in index_html + assert 'id="vision"' in category_html _REDIRECT_README = textwrap.dedent("""\ # Awesome Python From 52915911d48dcdce82ffff88789399df5a35e9c8 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 00:21:12 +0800 Subject: [PATCH 031/104] feat: link section use-case headings to their subcategory pages With tags now pointing at section anchors instead of filtering in place, subcategory pages (74 clicks and 19.7K impressions in the last 3 months) had no other internal links pointing to them. Co-Authored-By: Claude --- website/build.py | 3 ++- website/tests/test_build.py | 1 + 2 files changed, 3 insertions(+), 1 deletion(-) diff --git a/website/build.py b/website/build.py index 70bbdc14..d447c384 100644 --- a/website/build.py +++ b/website/build.py @@ -276,7 +276,8 @@ def group_section_entries(section: ParsedSection, entries_by_key: dict[tuple[str groups: dict[str, EntryGroup] = {} for parsed in section["entries"]: name = parsed["subcategory"] - group = groups.setdefault(name, EntryGroup(name=name, slug=slugify(name) if name else "", url="", entries=[])) + slug = slugify(name) if name else "" + group = groups.setdefault(name, EntryGroup(name=name, slug=slug, url=subcategory_path(section["slug"], slug) if name else "", entries=[])) group["entries"].append(entries_by_key[(parsed["url"], parsed["name"])]) return list(groups.values()) diff --git a/website/tests/test_build.py b/website/tests/test_build.py index f2448e90..a322f1c2 100644 --- a/website/tests/test_build.py +++ b/website/tests/test_build.py @@ -990,6 +990,7 @@ class TestBuild: positions = [html.index(marker) for marker in ('

    ', ">w2w1', ">w3Small' in html + assert 'Small' in html assert '' in html subcategory_html = (site / "widgets" / "small" / "index.html").read_text(encoding="utf-8") From ef00fe3adb9acfdc590fcade3ba60a9f5d690f30 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 02:04:41 +0800 Subject: [PATCH 032/104] fix: restore hidden group headings when navigating to a jump link After sorting by a column, or while a search hid a group, category page group headings stayed hidden, so the header's jump links (and any use-case heading link) pointed at a hidden element and did nothing. Co-Authored-By: Claude --- website/static/main.js | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/website/static/main.js b/website/static/main.js index 919aefd4..30871c9f 100644 --- a/website/static/main.js +++ b/website/static/main.js @@ -364,6 +364,24 @@ sortHeaders.forEach(function (th) { }); }); +// Group headings are hidden while sorted flat or filtered out by search, so a link to one must restore them first +document.addEventListener("click", function (e) { + const link = e.target.closest('a[href*="#"]'); + if (!link || link.pathname !== location.pathname) return; + const heading = document.getElementById(decodeURIComponent(link.hash.slice(1))); + const groupRow = heading ? heading.closest(".group-row") : null; + if (!groupRow) return; + if (activeSort.col !== "editorial") { + activeSort = defaultSort; + sortRows(); + updateSortIndicators(); + } + if (groupRow.hidden && searchInput) { + searchInput.value = ""; + applyFilters(); + } +}); + if (searchInput) { let searchTimer; searchInput.addEventListener("input", function () { From 77b1b70df5d88e1098a10cc23f3b8e9a33ab0efa Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 02:04:57 +0800 Subject: [PATCH 033/104] fix: hide use-case and page tags that repeat the group heading On section and subcategory pages, every row repeated its use case and current page as tags (and as the label under the name on phones), right under a heading that already said the same thing. They are now hidden in editorial order and shown again when a column sort flattens the groups. Co-Authored-By: Claude --- website/static/main.js | 1 + website/static/style.css | 5 +++++ website/templates/category.html | 9 ++++++--- website/tests/test_build.py | 2 ++ 4 files changed, 14 insertions(+), 3 deletions(-) diff --git a/website/static/main.js b/website/static/main.js index 30871c9f..93eac711 100644 --- a/website/static/main.js +++ b/website/static/main.js @@ -288,6 +288,7 @@ function sortRows() { const sortHeaders = document.querySelectorAll("th[data-sort]"); function updateSortIndicators() { + if (table) table.classList.toggle("sorted", activeSort.col !== "editorial"); sortHeaders.forEach(function (th) { th.classList.remove("sort-asc", "sort-desc"); if (th.dataset.sort === activeSort.col) { diff --git a/website/static/style.css b/website/static/style.css index 4e067c06..12ff6cdf 100644 --- a/website/static/style.css +++ b/website/static/style.css @@ -868,6 +868,11 @@ kbd { text-underline-offset: 0.2em; } +/* main.js adds .sorted once a column sort flattens the groups */ +.table:not(.sorted) .repeats-heading { + display: none; +} + .group-count { font-family: var(--font-body); font-size: var(--text-sm); diff --git a/website/templates/category.html b/website/templates/category.html index 7528da00..e8d05860 100644 --- a/website/templates/category.html +++ b/website/templates/category.html @@ -76,6 +76,9 @@ {% endblock %} {% block content %} {% macro entry_rows(entry, index) %} + {# On section and subcategory pages the group heading already names the row's use case #} + {% set grouped_page = entry_groups and not group_categories %} + {% set section_path = "/categories/" ~ parent_category.slug ~ "/" if parent_category else current_path %} {{ entry.name }} - {% if entry.subcategories %}{{ entry.subcategories[0].name }}{% else %}{{ category.name }}{% endif %} @@ -123,13 +126,13 @@ {% for subcat in entry.subcategories %} - + {{ subcat.name }} {% endfor %} {% for cat in entry.categories %} {{ cat }} diff --git a/website/tests/test_build.py b/website/tests/test_build.py index a322f1c2..78495fa8 100644 --- a/website/tests/test_build.py +++ b/website/tests/test_build.py @@ -991,6 +991,7 @@ class TestBuild: assert positions == sorted(positions) assert 'Small' in html assert 'Small' in html + assert '' in html assert '' in html subcategory_html = (site / "widgets" / "small" / "index.html").read_text(encoding="utf-8") @@ -1033,6 +1034,7 @@ class TestBuild: dl_heading = html.index('Deep Learning') assert ml_heading < html.index(">ml1dl1ml1 Date: Sun, 27 Sep 2026 02:35:12 +0800 Subject: [PATCH 034/104] docs: add file format processing category intro The category page had no intro, giving readers no guidance for picking among the PDF, Office, Markdown, and config libraries listed. Co-Authored-By: Claude --- .../category_intros/file-format-processing.md | 51 +++++++++++++++++++ 1 file changed, 51 insertions(+) create mode 100644 website/data/category_intros/file-format-processing.md diff --git a/website/data/category_intros/file-format-processing.md b/website/data/category_intros/file-format-processing.md new file mode 100644 index 00000000..069f0bec --- /dev/null +++ b/website/data/category_intros/file-format-processing.md @@ -0,0 +1,51 @@ +PDF, Word, and Excel files open with pypdf, python-docx, and openpyxl, a Python file format library for each. MarkItDown turns all three into Markdown. + +How to choose: + +- Splitting, merging, and reading PDFs: pypdf +- Word documents: python-docx +- Excel: openpyxl to read or edit, XlsxWriter for new reports +- Documents to Markdown for an LLM: MarkItDown +- ELF binaries and DWARF debug info: pyelftools +- One table exported to CSV, JSON, Excel, and more: Tablib +- Scanned PDFs, tables, and complex layouts: Docling +- PowerPoint decks: python-pptx +- New PDFs drawn from Python code: ReportLab +- PDF text with its position and font: pdfminer.six +- PDFs from HTML and CSS: WeasyPrint +- Markdown to HTML: markdown-it-py for CommonMark, Python-Markdown for its extensions, Mistune for speed +- Config files: tomllib for TOML, PyYAML for YAML + +pypdf works on PDFs that already exist: it [splits, merges, crops, and transforms pages](https://pypdf.readthedocs.io/en/stable/), adds passwords, and pulls out text and metadata. It's [pure Python with no C dependency](https://pypdf.readthedocs.io/en/stable/meta/comparisons.html), and it doesn't create PDFs. pypdf [isn't OCR software](https://pypdf.readthedocs.io/en/stable/user/extract-text.html), so run scanned pages through OCR instead. + +python-docx [only edits existing documents](https://python-docx.readthedocs.io/en/latest/user/documents.html): `Document()` opens a built-in template with no content. Start from your own .docx instead, so its styles, headers, and footers carry over. + +openpyxl reads and writes Excel files. For big workbooks, open them with `read_only=True` or create them with `write_only=True`, which [keep memory near constant](https://openpyxl.readthedocs.io/en/stable/optimized.html). A cell with a formula loads the formula; pass [`data_only=True`](https://openpyxl.readthedocs.io/en/stable/tutorial.html) to get the value Excel last calculated. + +XlsxWriter [only writes new files](https://xlsxwriter.readthedocs.io/introduction.html) and can't read or modify existing ones, but it supports more Excel features than the alternatives. Open the workbook [in a `with` block](https://xlsxwriter.readthedocs.io/workbook.html) so it gets closed and saved. For large files, turn on [`constant_memory`](https://xlsxwriter.readthedocs.io/working_with_memory.html) and write the rows in order. From pandas, pass [`engine='xlsxwriter'`](https://xlsxwriter.readthedocs.io/working_with_pandas.html) to `pd.ExcelWriter`. + +MarkItDown converts files to Markdown [for LLMs and text analysis](https://github.com/microsoft/markitdown), not for high-fidelity conversions that people read. Install `markitdown[all]`, or only the extras for your formats, like `markitdown[pdf, docx, pptx]`. + +Docling [understands PDF layout](https://docling-project.github.io/docling/): reading order, tables, and formulas. It also runs OCR on scanned pages. Its models [run locally and send no data out](https://docling-project.github.io/docling/usage/advanced_options/#using-remote-services) unless you turn remote services on. Each file becomes a DoclingDocument, which you export to Markdown or [split into chunks](https://docling-project.github.io/docling/concepts/chunking/) for an embedding model. + +python-pptx builds decks from data, like a database query or analytics output, and [doesn't need PowerPoint installed](https://python-pptx.readthedocs.io/en/latest/). It [only edits existing presentations](https://python-pptx.readthedocs.io/en/latest/user/presentations.html), so start from your own deck: its theme, slide master, and slide layouts set how the slides look. Add each slide from one of those layouts, picked by [its index in your deck](https://python-pptx.readthedocs.io/en/latest/user/slides.html). + +ReportLab draws new PDFs from Python code. Learn it on [`pdfgen`, its lowest-level interface](https://docs.reportlab.com/developerfaqs/), then build multi-page documents with [Platypus](https://docs.reportlab.com/reportlab/userguide/ch5_platypus/). Platypus lets you keep paragraph styles and page layouts in one shared file, so restyling takes a few lines. + +pdfminer.six [focuses on text](https://github.com/pdfminer/pdfminer.six): it gets each piece of text with its exact location, font, and color. Start with `extract_text()` from its [high-level API](https://pdfminersix.readthedocs.io/en/latest/tutorial/highlevel.html). A PDF [stores only characters and their positions](https://pdfminersix.readthedocs.io/en/latest/topic/converting_pdf_to_text.html), so pdfminer.six guesses words, lines, and paragraphs from the layout. Tune those guesses with `LAParams`. + +WeasyPrint turns HTML and CSS into PDFs, like [reports, invoices, and tickets](https://doc.courtbouillon.org/weasyprint/stable/). It runs [no JavaScript](https://doc.courtbouillon.org/weasyprint/stable/going_further.html). Set page size and margins [with the CSS `@page` rule](https://doc.courtbouillon.org/weasyprint/stable/common_use_cases.html). + +markdown-it-py [follows the CommonMark spec](https://markdown-it-py.readthedocs.io/en/latest/) and takes plugins for more syntax. For content your users submit, use the [`js-default` preset](https://markdown-it-py.readthedocs.io/en/latest/security.html), since the default settings aren't safe for it. + +Python-Markdown [isn't a CommonMark implementation](https://python-markdown.github.io/): it follows the original Markdown syntax and has an extension API. It [doesn't sanitize its HTML output](https://python-markdown.github.io/sanitization/), so sanitize it yourself when the input is untrusted. + +Mistune is [fast and has no dependencies](https://mistune.lepture.com/en/latest/). For untrusted input, build the parser with [`mistune.create_markdown()`](https://mistune.lepture.com/en/latest/guide.html), which escapes HTML tags, since `mistune.html()` doesn't. + +tomllib [only reads TOML](https://docs.python.org/3/library/tomllib.html), from a file opened in binary mode. + +Tablib holds one dataset and exports it to many formats; Excel, YAML, and pandas [are optional extras](https://tablib.readthedocs.io/en/stable/formats.html), like `tablib[xlsx]`. + +pyelftools is [pure Python with no dependencies](https://github.com/eliben/pyelftools); start from its [`ELFFile` class](https://github.com/eliben/pyelftools/blob/main/doc/user-guide.md) and stay on the high-level API. + +Treat every file you didn't create as untrusted. With PyYAML, call [`yaml.safe_load()`](https://pyyaml.org/wiki/PyYAMLDocumentation#loading-yaml), never `yaml.load()`, which can run any Python function. Install [defusedxml](https://openpyxl.readthedocs.io/en/stable/#security) next to openpyxl to guard against XML attacks like billion laughs. [Catch pypdf's exceptions](https://pypdf.readthedocs.io/en/stable/user/security.html) yourself, so a broken PDF can't crash your service. For MarkItDown, call [`convert_local()` or `convert_stream()`](https://github.com/microsoft/markitdown#security-considerations) instead of `convert()`, which also fetches remote URIs. Cap Docling's input with [`max_num_pages` and `max_file_size`](https://docling-project.github.io/docling/usage/advanced_options/#impose-limits-on-the-document-size). Run WeasyPrint on untrusted HTML [as a user with limited access](https://doc.courtbouillon.org/weasyprint/stable/first_steps.html#security), with a URL fetcher that blocks local files. With tomllib, [limit the size](https://docs.python.org/3/library/tomllib.html) of the data you parse. From fd40096421bb96edf39b9a9992fb90c012b6f63b Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 02:35:36 +0800 Subject: [PATCH 035/104] docs: add text processing category intro The page had no intro, so readers got no pick among the encoding detection, fuzzy matching, parser, slug, and ID libraries. Co-Authored-By: Claude --- .../data/category_intros/text-processing.md | 42 +++++++++++++++++++ 1 file changed, 42 insertions(+) create mode 100644 website/data/category_intros/text-processing.md diff --git a/website/data/category_intros/text-processing.md b/website/data/category_intros/text-processing.md new file mode 100644 index 00000000..70743995 --- /dev/null +++ b/website/data/category_intros/text-processing.md @@ -0,0 +1,42 @@ +Need a Python text processing library? charset-normalizer reads text in unknown encodings, RapidFuzz does fuzzy matching, and pyparsing builds parsers. + +How to choose: + +- Bytes in an unknown encoding: charset-normalizer +- Fuzzy matching against many strings: RapidFuzz +- Parsers for your own grammar: pyparsing, or parsy for small languages +- Text garbled by a wrong decode (mojibake): ftfy +- Diffs, or close matches without a dependency: difflib +- ASCII-art banners: pyfiglet +- Translations, plus localized dates and numbers: Babel +- Syntax highlighting: Pygments +- Splitting and formatting SQL: sqlparse +- Phone numbers: phonenumbers +- URL slugs: python-slugify, or Unidecode for ASCII transliteration alone +- Short public IDs: shortuuid, or Sqids to encode integer IDs + +Decode with the encoding you know before you guess one. ftfy's docs say to [assume UTF-8](https://ftfy.readthedocs.io/en/latest/avoid.html#assume-utf-8) until proven otherwise, and charset-normalizer's FAQ calls detection [the last resort](https://charset-normalizer.readthedocs.io/en/latest/community/faq.html#should-i-bother-using-detection). For bytes you have no clue about, charset-normalizer is the detector [requests installs](https://github.com/psf/requests/blob/main/pyproject.toml). Call `from_bytes()` or `from_path()`, then `.best()` to [get the most probable result](https://charset-normalizer.readthedocs.io/en/latest/user/advanced_search.html). Its `detect()` is [backward compatible with chardet's](https://charset-normalizer.readthedocs.io/en/latest/user/getstarted.html), so switching is one import. If you stay on chardet, read large files and streams with [`UniversalDetector`](https://chardet.readthedocs.io/en/latest/usage.html). + +ftfy fixes text that was already decoded wrong, and it [doesn't take bytes](https://ftfy.readthedocs.io/en/latest/detect.html), so it's no replacement for a detector. Call `ftfy.fix_text()`, [the function you'll use most](https://ftfy.readthedocs.io/en/latest/explain.html#ftfy.fix_text). Every fix is on by default, so [check which ones fit your text](https://ftfy.readthedocs.io/en/latest/config.html). + +RapidFuzz is a fast string matching library with a C++ core. To compare one string with a list, use the process module's `process.extractOne()` or `process.cdist()`, which is [faster than calling the scorers yourself](https://github.com/rapidfuzz/RapidFuzz). It doesn't lowercase or strip punctuation for you: pass `processor=utils.default_process` when case and punctuation shouldn't count. + +difflib ships with Python, so it works where you can't add a dependency. `get_close_matches()` returns the [good enough matches](https://docs.python.org/3/library/difflib.html#difflib.get_close_matches) for a word, and its diffs come out in unified, context, or HTML format. + +pyparsing lets you [build the grammar in Python code](https://github.com/pyparsing/pyparsing) instead of regular expressions. Since a grammar can accept invalid input, its docs say to use it on [input you assume is well-formatted](https://pyparsing-docs.readthedocs.io/en/latest/HowToUsePyparsing.html#usage-notes). Its [best practices](https://github.com/pyparsing/pyparsing/blob/master/pyparsing/ai/best_practices.md) say to write the grammar out in BNF first. When the grammar is recursive, call `enable_packrat()` right after the import, and give results names to the fields you read back. + +parsy combines small parsers into bigger ones. Its docs say it [excels at small languages](https://parsy.readthedocs.io/en/latest/overview.html) and is easy to read, but to look elsewhere when you need speed or good error messages. For more complex parsers, the [`@generate` decorator](https://parsy.readthedocs.io/en/latest/ref/generating.html) is both more readable and more powerful. + +Babel does two jobs: gettext message catalogs, and [CLDR locale data](https://babel.pocoo.org/en/latest/intro.html) for localized dates, numbers, and names. Run the catalog steps with the `pybabel` command: [extract, init, update, and compile](https://babel.pocoo.org/en/latest/cmdline.html). Keep times in UTC, and [convert to the user's time zone](https://babel.pocoo.org/en/latest/dates.html#time-zone-support) only for input and display. + +Pygments turns code into HTML, LaTeX, ANSI, and more, as a command-line tool or a library. For HTML, it writes CSS classes instead of inline styles, so [generate the stylesheet](https://pygments.org/docs/quickstart/#example) with `HtmlFormatter().get_style_defs()`. Pick the lexer by name or file name, and [guess it](https://pygments.org/docs/quickstart/#guessing-lexers) only when you don't know the language. Pygments [doesn't guarantee how long it runs](https://pygments.org/docs/security/), so on user input, run it with a short timeout and cap how many run at once. + +sqlparse is a [non-validating SQL parser](https://sqlparse.readthedocs.io/en/latest/): it splits scripts into statements, formats them, and walks their tokens, without assuming a SQL dialect. Use `split()`, `format()`, and `parse()`. On SQL from untrusted sources, [keep its grouping limits](https://sqlparse.readthedocs.io/en/latest/api.html#security-and-performance-considerations) as they are. + +phonenumbers is a Python port of Google's libphonenumber. Pass `parse()` the region the number was dialed from, unless it's in E.164 format. Then [check it's possible and valid](https://github.com/daviddrysdale/python-phonenumbers) with `is_possible_number()` and `is_valid_number()`. + +python-slugify makes URL slugs, and Unidecode turns Unicode text into ASCII. Their licenses differ. [Unidecode is GPL](https://github.com/avian2/unidecode). python-slugify's own code is MIT, and by default it runs on text-unidecode, which offers the Artistic license or GPL. But python-slugify [switches to Unidecode](https://github.com/un33k/python-slugify) whenever it's installed. + +shortuuid turns UUIDs into [short IDs for users to see](https://github.com/skorokithakis/shortuuid), and leaves out look-alike characters like l, 1, I, O, and 0. Sqids turns database keys and other integers into short IDs. Anyone can [decode them back into numbers](https://sqids.org/faq#not-recommended), so keep them away from sensitive data and user IDs. To check an ID is the canonical one, [re-encode the decoded numbers](https://sqids.org/faq#valid-ids) and compare. + +Store what these libraries generate, or pin their versions, since slugs and IDs can change between releases. python-slugify says to [pin the package and its backend](https://github.com/un33k/python-slugify), and Unidecode says to [store each slug once or lock the version](https://github.com/avian2/unidecode). Sqids says to [pass your own blocklist](https://sqids.org/faq#future-blocklist), even one identical to the default. From 9e8e38eb2eddce7b96417b355f6a572a624cde39 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 02:36:04 +0800 Subject: [PATCH 036/104] docs: add web scraping category intro The web scraping category page had no intro, leaving readers without guidance for picking among crawling frameworks, LLM-ready crawlers, AI browser agents, and content extractors. Co-Authored-By: Claude --- website/data/category_intros/web-scraping.md | 28 ++++++++++++++++++++ 1 file changed, 28 insertions(+) create mode 100644 website/data/category_intros/web-scraping.md diff --git a/website/data/category_intros/web-scraping.md b/website/data/category_intros/web-scraping.md new file mode 100644 index 00000000..0409e072 --- /dev/null +++ b/website/data/category_intros/web-scraping.md @@ -0,0 +1,28 @@ +To crawl a whole site, use Scrapy as your Python web scraping library; to feed pages to an LLM, Crawl4AI; to pull an article's main text, Trafilatura. + +How to choose: + +- Crawling a whole site, or many sites: Scrapy +- Web pages as clean Markdown for an LLM: Crawl4AI +- An article's main text and metadata: Trafilatura +- An AI agent that does a task in the browser for you: Browser Use +- Browser automation in your own code, with plain-language steps: Stagehand +- A browser agent that picks each action from a table of page elements: jev-ultrafast +- RSS, Atom, and JSON feeds: feedparser +- A whole HTML page as Markdown text: html2text + +Scrapy is a framework [for crawling websites and extracting structured data](https://docs.scrapy.org/en/latest/intro/overview.html) from them, and it sends requests asynchronously. Create a project with `scrapy startproject` and [write your spiders in it](https://docs.scrapy.org/en/latest/intro/tutorial.html). When a page loads its data with JavaScript, [find the request that returns the data](https://docs.scrapy.org/en/latest/topics/dynamic-content.html) and send it yourself. Use a headless browser only when that fails. To run your spiders in production, deploy them to [Scrapyd or Zyte Scrapy Cloud](https://docs.scrapy.org/en/latest/topics/deploy.html). + +Crawl4AI [turns websites into clean, LLM-ready Markdown](https://docs.crawl4ai.com/). It runs Chromium headless by default, so run `crawl4ai-setup` once to install the browser. Open an `AsyncWebCrawler` in an `async with` block and [call `arun()` for each URL](https://docs.crawl4ai.com/core/quickstart/). When pages share one layout, extract the data with CSS or XPath schemas: they [do exactly what you specify](https://docs.crawl4ai.com/extraction/no-llm-strategies/), while an LLM's output can vary. Keep [LLM extraction](https://docs.crawl4ai.com/extraction/llm-strategies/), which is slower and costlier, for content an AI has to interpret. + +Trafilatura [extracts a page's main text and metadata](https://trafilatura.readthedocs.io/en/latest/) and skips boilerplate like headers and footers. Call `extract()`, and pass `favor_precision=True` or `favor_recall=True` to [tune what it keeps](https://trafilatura.readthedocs.io/en/latest/usage-python.html). It works on raw HTML, so [render JavaScript pages first](https://trafilatura.readthedocs.io/en/latest/troubleshooting.html) with a browser and pass the result to `extract()`. It pairs with Scrapy: [Scrapy crawls, Trafilatura extracts](https://trafilatura.readthedocs.io/en/latest/faq.html#how-does-trafilatura-compare-to-beautifulsoup-or-scrapy). + +Browser Use is an AI browser agent: you [give it a task and an LLM](https://docs.browser-use.com/open-source/quickstart), and it runs the task in a browser. Its docs say to [be specific about the actions](https://docs.browser-use.com/open-source/customize/agent/prompting-guide) you want. Pass a Pydantic model as `output_model_schema` to get structured results. Don't take the agent's word for it: `is_successful()` is [only its own assessment](https://docs.browser-use.com/open-source/customize/agent/output-format), so check important results yourself, like whether a form really got submitted. Pass logins as `sensitive_data`, so the model [sees only placeholders](https://docs.browser-use.com/open-source/examples/templates/sensitive-data) in the text it reads. Set `use_vision=False` too, or the real values can leak through screenshots. + +Stagehand leaves the steps to your code: you [mix plain-language actions with regular page calls](https://docs.stagehand.dev/first-steps/introduction) in one script, and decide how much AI each step uses. Give each `act()` call [one focused action](https://docs.stagehand.dev/best-practices/prompting-best-practices), and pass credentials as variables, so their values never reach the model. To get data out, pass `extract()` a Pydantic model, and Stagehand [validates the result against it](https://docs.stagehand.dev/basics/extract). + +feedparser parses RSS, Atom, and JSON feeds with [one function, `parse()`](https://feedparser.readthedocs.io/en/latest/introduction/), which takes a URL, a file, or a string. When you poll a feed, send back the [ETag and Last-Modified values](https://feedparser.readthedocs.io/en/latest/http-etag/) from the last response. Otherwise you download unchanged feeds again, and the publisher may ban you. Content feedparser marks as `text/plain` [hasn't been sanitized](https://feedparser.readthedocs.io/en/latest/html-sanitization/), so escape it before you render it. + +html2text [converts a page of HTML into Markdown](https://github.com/Alir3z4/html2text), with options such as [`ignore_links`](https://github.com/Alir3z4/html2text/blob/master/docs/usage.md). It's GPL, while Trafilatura is [Apache](https://trafilatura.readthedocs.io/en/latest/). + +Whatever you pick, tell sites who you are and go easy on them. Scrapy's docs say to [set `USER_AGENT` to a value that identifies you](https://docs.scrapy.org/en/latest/topics/practices.html#avoiding-getting-banned) and space out your requests. The feedparser docs say to [set the User-Agent to your app's name and URL](https://feedparser.readthedocs.io/en/latest/http-useragent/), and Trafilatura's to [throttle per domain and follow robots.txt](https://trafilatura.readthedocs.io/en/latest/downloads.html). From 432a25e384bc573fff9174bbf914ff2b802d6e1b Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 02:43:41 +0800 Subject: [PATCH 037/104] docs: add Game Development category intro The page had no guidance on which Python game library to pick. Recommends pygame-ce over upstream pygame, which fails to install on Python 3.14 (no cp314 wheel, open issues #4627/#4810). Co-Authored-By: Claude --- .../data/category_intros/game-development.md | 21 +++++++++++++++++++ 1 file changed, 21 insertions(+) create mode 100644 website/data/category_intros/game-development.md diff --git a/website/data/category_intros/game-development.md b/website/data/category_intros/game-development.md new file mode 100644 index 00000000..60c4a8c2 --- /dev/null +++ b/website/data/category_intros/game-development.md @@ -0,0 +1,21 @@ +A 2D game needs a loop: write your own with pygame-ce, the Python game development library, or let Arcade run it. Panda3D does 3D, and Ren'Py visual novels. + +How to choose: + +- 2D game where you write the game loop: pygame-ce +- 2D game with a ready-made loop and physics engines: Arcade +- 3D game: Panda3D +- Visual novel or life simulation game: Ren'Py +- Windowing, OpenGL graphics, and sound with no other dependencies: pyglet + +pygame-ce is the community edition of pygame, a fork by its former core developers that aims for more frequent releases. Your code still says `import pygame`, so if pygame is already installed, [uninstall it first](https://github.com/pygame-community/pygame-ce/wiki/Installing-pygame%E2%80%90ce), then run `pip install pygame-ce` in a virtual environment. Its [quick start](https://pyga.me/docs/) gives you full control of the game loop: handle events, draw the frame, `flip()` the display, and cap the frame rate with `clock.tick()`. Convert each image once after you load it, with [`convert()`, or `convert_alpha()` if it has transparency](https://pyga.me/docs/ref/surface.html#pygame.Surface.convert), so it blits fast. It's under the LGPL, and its README says [closed-source and commercial games are fine](https://github.com/pygame-community/pygame-ce). + +Arcade is [an easy-to-learn library for 2D games](https://github.com/pythonarcade/arcade), built on pyglet and OpenGL, and meant for beginning programmers too. Start with the [Platformer Tutorial](https://api.arcade.academy/en/stable/tutorials/platform_tutorial/step_01.html): you subclass `arcade.Window`, draw in `on_draw()`, and `arcade.run()` runs the loop until the window closes. For movement and collisions, it comes with [physics engines](https://api.arcade.academy/en/stable/api_docs/api/physics_engines.html) for top-down and platformer games. Its code is MIT, and its [built-in assets need no attribution](https://api.arcade.academy/en/stable/), so you can ship them in a commercial game. + +Panda3D is [a 3D engine written in C++ with Python bindings](https://docs.panda3d.org/latest/python/introduction/index), and its manual says it's a tool for skilled programmers, not beginners. Install it with [`pip install panda3d`](https://github.com/panda3d/panda3d), subclass [`ShowBase`](https://docs.panda3d.org/latest/python/introduction/tutorial/starting-panda3d), and call `run()`, which holds the main loop. Its [`build_apps` tool](https://docs.panda3d.org/latest/python/distribution/index) builds self-contained executables for Windows, Linux, and macOS without needing each system. It's BSD, free for commercial games. + +Ren'Py is a [visual novel engine](https://www.renpy.org/) with its own script language, for stories that run on computers and mobile devices. It isn't a pip package: [download Ren'Py and run its launcher](https://www.renpy.org/doc/html/quickstart.html), create a project there, and write your story in `script.rpy`. Python works inside the scripts, and [third-party pure-Python packages](https://www.renpy.org/doc/html/python.html#first-and-third-party-python-modules-and-packages) go in `game/python-packages`. Ship with [Build Distributions](https://www.renpy.org/doc/html/build.html) in the launcher, which also builds a package for itch.io and Steam. Most of Ren'Py is MIT, but some parts are LGPL, so [distribute your game in a way that satisfies the LGPL](https://www.renpy.org/doc/html/license.html). + +pyglet is a [windowing and multimedia library with no external dependencies](https://pyglet.readthedocs.io/en/latest/), written in pure Python: windows, input, OpenGL graphics, images, video, and sound. Start with [Writing a pyglet application](https://pyglet.readthedocs.io/en/latest/programming_guide/quickstart.html), which attaches handlers with `@window.event` and calls `pyglet.app.run()`. Draw through a `Batch`, since the docs say [you always want batched rendering](https://pyglet.readthedocs.io/en/latest/programming_guide/shapes.html) for performance. It's under the BSD license. + +Move things by the time since the last frame, so your game runs at the same speed at any frame rate. pygame-ce's quick start gets it in seconds by [dividing `clock.tick()` by 1000](https://pyga.me/docs/), pyglet passes it as `dt` to [scheduled functions](https://pyglet.readthedocs.io/en/latest/programming_guide/time.html#sprite-movement-techniques), and Arcade passes it as `delta_time` to [`on_update()`](https://api.arcade.academy/en/stable/api_docs/api/window.html#arcade.Window.on_update). From e2308c087d349a39b20a93201adb4ed8f97b97ac Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 02:43:45 +0800 Subject: [PATCH 038/104] docs: add Date and Time category intro The page had no guidance on which Python date library to pick. Co-Authored-By: Claude --- website/data/category_intros/date-and-time.md | 21 +++++++++++++++++++ 1 file changed, 21 insertions(+) create mode 100644 website/data/category_intros/date-and-time.md diff --git a/website/data/category_intros/date-and-time.md b/website/data/category_intros/date-and-time.md new file mode 100644 index 00000000..eeda6507 --- /dev/null +++ b/website/data/category_intros/date-and-time.md @@ -0,0 +1,21 @@ +Most code should handle time zones with zoneinfo and pick python-dateutil, the Python date library for parsing date strings and adding months. + +How to choose: + +- Time zones by IANA name, like Europe/Paris: zoneinfo +- Parsing date strings, adding months, and recurring dates: python-dateutil +- Dates as people write them, like "3 days ago", in many languages: dateparser +- An easier datetime API whose objects are still datetimes: pendulum +- Exact and local times as separate types, with DST-safe math: whenever + +zoneinfo brings the IANA time zone database to Python, and the datetime docs say [its usage is recommended](https://docs.python.org/3/library/datetime.html#tzinfo-objects). Attach a ZoneInfo to a datetime [through the constructor, `replace()`, or `astimezone()`](https://docs.python.org/3/library/zoneinfo.html#using-zoneinfo). Some systems, Windows among them, have no IANA database, so if your code runs across platforms, [declare a dependency on tzdata](https://docs.python.org/3/library/zoneinfo.html#data-sources). + +python-dateutil adds [extensions to the standard datetime module](https://dateutil.readthedocs.io/en/stable/); install it as `python-dateutil` and import it as `dateutil`. Its parser [reads most known formats](https://dateutil.readthedocs.io/en/stable/parser.html) and returns a datetime even for an ambiguous date. For input like 01/05/09, set `dayfirst` or `yearfirst` to match your data. For calendar math, `relativedelta` takes [plural arguments that add and singular ones that replace](https://dateutil.readthedocs.io/en/stable/relativedelta.html): `months=+1` moves a month ahead, and `day=1` jumps to the first. `rrule` builds [recurring dates from iCalendar rules](https://dateutil.readthedocs.io/en/stable/rrule.html). + +dateparser reads relative dates like "two weeks ago" and absolute ones in more than 200 language locales. Its docs say it [stands out](https://dateparser.readthedocs.io/en/latest/#common-use-cases) for scraped pages, logs, and other data from mixed sources, and for letting users type dates in their own words. Call `dateparser.parse()`, and [pass `languages` when you know them](https://dateparser.readthedocs.io/en/latest/#how-to-use), so it skips language detection. When you parse many dates from one source, [use `DateDataParser`](https://dateparser.readthedocs.io/en/latest/usage.html), which remembers the languages it has found. + +pendulum's classes are [drop-in replacements for the native ones](https://pendulum.eustace.io/docs/#introduction), since they inherit from datetime. Every instance is time zone aware and in UTC by default. Its docs call aware datetimes [the preferred and recommended way](https://pendulum.eustace.io/docs/#instantiation) to use it. For tests, install `pendulum[test]` and [travel in time](https://pendulum.eustace.io/docs/#testing). + +whenever puts exact time and local time in [separate types](https://whenever.readthedocs.io/en/latest/guide/choosing-a-type.html): an instant when only the moment matters, a zoned datetime when the local time matters too. Mixing up naive and aware [becomes a type error](https://whenever.readthedocs.io/en/latest/), and DST is handled in all arithmetic. A standard datetime [does no time zone adjustment](https://docs.python.org/3/library/datetime.html#datetime-objects) when you add a timedelta to it. In production, [turn whenever's DST warnings into errors](https://whenever.readthedocs.io/en/latest/faq.html#why-warnings-instead-of-errors) with Python's standard warnings filter. + +Decide whether you extend datetime or replace it. zoneinfo, python-dateutil, and dateparser all use standard datetime objects, so they work together. pendulum's objects are datetimes too, but code that checks the exact type, like sqlite3 and some database drivers, [needs an adapter registered](https://pendulum.eustace.io/docs/#limitations). whenever [doesn't subclass datetime at all](https://whenever.readthedocs.io/en/latest/faq.html#why-no-drop-in-replacement-for-datetime), so [convert to and from standard datetimes](https://whenever.readthedocs.io/en/latest/guide/stdlib-convert.html) where other code needs one. From 4d09bcb7aba35f9f519a75d12eb0779a2728818b Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 02:43:48 +0800 Subject: [PATCH 039/104] docs: add Implementations category intro The page had no guidance on when to run CPython vs MicroPython, Pyodide, Cython, or PyPy. Co-Authored-By: Claude --- .../data/category_intros/implementations.md | 21 +++++++++++++++++++ 1 file changed, 21 insertions(+) create mode 100644 website/data/category_intros/implementations.md diff --git a/website/data/category_intros/implementations.md b/website/data/category_intros/implementations.md new file mode 100644 index 00000000..1ea70ecd --- /dev/null +++ b/website/data/category_intros/implementations.md @@ -0,0 +1,21 @@ +Unless you need MicroPython on a microcontroller or Pyodide in the browser, CPython is the Python implementation to run. Cython compiles your slow code to C. + +How to choose: + +- Most projects: CPython +- Microcontrollers and other constrained devices: MicroPython +- Python in the browser or Node.js: Pyodide +- Slow code to compile, or a C library to wrap: Cython +- Long-running programs in pure Python: PyPy + +CPython is the [original and most-maintained implementation](https://docs.python.org/3/reference/introduction.html#alternate-implementations) of Python, written in C, and new language features generally show up there first. Give each app its own [virtual environment](https://docs.python.org/3/installing/index.html) from venv, and install packages into it with pip. + +MicroPython is a lean implementation of Python 3, [optimized to run on microcontrollers](https://micropython.org/) and in constrained environments. It includes a small subset of the standard library, plus modules like `machine` for the hardware. Flash your board with firmware from [the download page](https://micropython.org/download/), then manage it from your computer with [mpremote](https://docs.micropython.org/en/latest/reference/mpremote.html). `mpremote mip install` [installs packages](https://docs.micropython.org/en/latest/reference/packages.html#installing-packages-with-mpremote) from micropython-lib by default, not PyPI. When RAM runs short, [freeze the modules that rarely change](https://docs.micropython.org/en/latest/reference/packages.html#freezing-packages) into the firmware. + +Pyodide is [a port of CPython to WebAssembly](https://pyodide.org/en/stable/) that runs in the browser and Node.js. It installs any pure Python wheel from PyPI, and many packages with C extensions, like NumPy and pandas, have been ported for it. Install packages with [`micropip.install()`](https://pyodide.org/en/stable/usage/loading-packages.html#how-to-chose-between-micropip-install-and-pyodide-loadpackage), which its docs recommend for almost everything. In a web page, run Pyodide [in a web worker](https://pyodide.org/en/stable/usage/webworker.html), so long computations don't freeze your UI. + +Cython isn't a separate interpreter: it [translates Python code to C](https://cython.readthedocs.io/en/latest/src/quickstart/overview.html) that runs inside CPython, to speed up slow code or wrap C libraries. Write your code in [pure Python syntax](https://cython.readthedocs.io/en/latest/src/tutorial/pure.html) to keep the file runnable by the plain interpreter, and use `.pyx` files for what that syntax can't express. The annotation report from `cython -a` shows [where types help](https://cython.readthedocs.io/en/latest/src/quickstart/cythonize.html#determining-where-to-add-types). Build your package with [a build backend](https://cython.readthedocs.io/en/latest/src/userguide/source_files_and_compilation.html#compiling-with-a-build-backend), and ship [prebuilt wheels](https://cython.readthedocs.io/en/latest/src/userguide/source_files_and_compilation.html#compiling-with-pyximport) to your users. + +PyPy is a replacement for CPython, and speed is the reason to use it. It works best on [long-running programs that spend much of their time in Python code](https://pypy.org/features.html#speed), not on short scripts. Code built on C extension modules is a poor fit, since they [often run much slower on PyPy than on CPython](https://doc.pypy.org/faq.html#do-c-extension-modules-work-with-pypy). Packages installed for CPython aren't available to PyPy, so [install them for PyPy](https://doc.pypy.org/faq.html#module-xyz-does-not-work-with-pypy-importerror) in its own virtual environment with `pypy -m pip`. + +Before you switch implementations or compile anything for speed, [measure first](https://pypy.org/performance.html#profiling-vmprof) to confirm the slowdown is real. Then profile to find the slow parts, and only optimize those. From 80958eb516721cdefda3a07d56e50f1f8ecf8b65 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 03:02:51 +0800 Subject: [PATCH 040/104] docs: add deep learning category intro The Deep Learning page had no intro, leaving readers with no pick among PyTorch, Keras, JAX, Lightning, and the reinforcement learning libraries. Co-Authored-By: Claude --- website/data/category_intros/deep-learning.md | 24 +++++++++++++++++++ 1 file changed, 24 insertions(+) create mode 100644 website/data/category_intros/deep-learning.md diff --git a/website/data/category_intros/deep-learning.md b/website/data/category_intros/deep-learning.md new file mode 100644 index 00000000..9201d434 --- /dev/null +++ b/website/data/category_intros/deep-learning.md @@ -0,0 +1,24 @@ +Three Python deep learning frameworks, three strengths: PyTorch for research and new architectures, Keras for a high-level API, JAX for compiled code on TPUs. + +How to choose: + +- Research and new model architectures: PyTorch +- High-level model building on JAX or PyTorch: Keras +- NumPy-style array code compiled for GPUs and TPUs: JAX +- PyTorch training without the loop boilerplate: PyTorch Lightning +- Reinforcement learning environments: Gymnasium +- Reinforcement learning algorithms: Stable-Baselines3 + +PyTorch models are `nn.Module` subclasses, and autograd [builds the computational graph as your code runs](https://docs.pytorch.org/docs/stable/user_guide/pytorch_main_components.html). Install it with [the command from its selector](https://pytorch.org/get-started/locally/), which matches your OS and GPU. To keep a model, [save its `state_dict`](https://docs.pytorch.org/tutorials/beginner/saving_loading_models.html#save-load-state-dict-recommended) instead of pickling the whole module. Wrap your model in [`torch.compile`](https://docs.pytorch.org/tutorials/intermediate/torch_compile_tutorial.html) to speed it up with minimal code changes. To train on more than one GPU, use [DistributedDataParallel](https://docs.pytorch.org/docs/stable/notes/cuda.html#use-nn-parallel-distributeddataparallel-instead-of-multiprocessing-or-nn-dataparallel). + +Keras is a [multi-framework API](https://keras.io/getting_started/about/): a Keras model can run as a PyTorch Module or as a JAX function. Install a backend next to it and [set `KERAS_BACKEND`](https://keras.io/getting_started/#configuring-your-backend) before you import Keras. Build simple models as a Sequential stack of layers and anything more complex with the functional API. Save the whole model to [a `.keras` file](https://keras.io/getting_started/faq/#what-are-my-options-for-saving-models) with `model.save()` instead of pickling it. The file [reloads with any backend](https://keras.io/keras_3/). + +JAX does [accelerator-oriented array computation](https://docs.jax.dev/en/latest/) with a NumPy-style API and composable transformations: `jax.grad` for derivatives, `jax.jit` for compilation, and `jax.vmap` for batching. The transformations only work on [functionally pure functions](https://docs.jax.dev/en/latest/notebooks/Common_Gotchas_in_JAX.html#pure-functions), so pass all data in as arguments and return every result. JAX itself stays narrow: to train neural networks, use the [JAX AI Stack](https://docs.jaxstack.ai/en/latest/getting_started.html), with Flax NNX for models and Optax for optimizers. On NVIDIA GPUs, the JAX team strongly recommends [installing CUDA and cuDNN from pip wheels](https://docs.jax.dev/en/latest/installation.html#pip-installation-nvidia-gpu-cuda-installed-via-pip-easier). + +PyTorch Lightning [organizes PyTorch code to remove boilerplate](https://lightning.ai/docs/pytorch/stable/home/introduction): you write the model logic in a LightningModule, and the Trainer handles devices, precision, and distributed training. Install it as [the `lightning` package](https://lightning.ai/docs/pytorch/stable/home/installation). Its [style guide](https://lightning.ai/docs/pytorch/stable/reference/starter/style_guide) recommends keeping each LightningModule self-contained, the model separate from the system that trains it, and data loading in a LightningDataModule. To keep your own training loop, use [Lightning Fabric](https://lightning.ai/docs/fabric/stable), which scales a plain PyTorch script after you change a few lines. + +Gymnasium is [an API standard for reinforcement learning](https://gymnasium.farama.org/), with a collection of reference environments. It's the maintained fork of OpenAI's Gym, and many older tutorials still use Gym's old API, so follow its [migration guide](https://gymnasium.farama.org/introduction/migration_guide/) when you port one. [Register your own environment](https://gymnasium.farama.org/introduction/create_custom_env/#registering-and-making-the-environment) so `gymnasium.make()` creates it like a built-in one, and run [`check_env`](https://gymnasium.farama.org/introduction/create_custom_env/#check-environment-validity) on it to catch common issues. + +Stable-Baselines3 is a set of [reliable implementations of reinforcement learning algorithms in PyTorch](https://stable-baselines3.readthedocs.io/en/master/), and it trains on any environment that [follows the Gymnasium interface](https://stable-baselines3.readthedocs.io/en/master/guide/custom_env.html). It [assumes you know some reinforcement learning](https://github.com/DLR-RM/stable-baselines3). Its tips page recommends [starting from the RL Zoo's tuned hyperparameters](https://stable-baselines3.readthedocs.io/en/master/guide/rl_tips.html#general-advice-when-using-reinforcement-learning) and normalizing the agent's input. Evaluate the agent on [a separate test environment](https://stable-baselines3.readthedocs.io/en/master/guide/rl_tips.html#how-to-evaluate-an-rl-algorithm), since training adds exploration noise. [Pick an algorithm](https://stable-baselines3.readthedocs.io/en/master/guide/rl_tips.html#which-algorithm-should-i-use) by your action space first: DQN handles only discrete actions, and SAC only continuous ones. + +PyTorch's security policy says [running untrusted models is equivalent to running untrusted code](https://github.com/pytorch/pytorch/blob/main/SECURITY.md), so run untrusted ones in a sandbox. Load checkpoints with [`weights_only=True`](https://docs.pytorch.org/docs/stable/notes/serialization.html#weights-only-security) in `torch.load`, and leave [`safe_mode`](https://keras.io/api/models/model_saving_apis/model_saving_and_loading/) on when Keras loads a model. From 579ea6bf85b1c8d0179f558c5db85fa3bef993e0 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 03:03:12 +0800 Subject: [PATCH 041/104] docs: add machine learning category intro The page had no intro, so readers got no pick among scikit-learn, the boosting libraries, pgmpy, Feature-engine, and TimesFM. Co-Authored-By: Claude --- .../data/category_intros/machine-learning.md | 29 +++++++++++++++++++ 1 file changed, 29 insertions(+) create mode 100644 website/data/category_intros/machine-learning.md diff --git a/website/data/category_intros/machine-learning.md b/website/data/category_intros/machine-learning.md new file mode 100644 index 00000000..586bbbc5 --- /dev/null +++ b/website/data/category_intros/machine-learning.md @@ -0,0 +1,29 @@ +Tabular data goes to scikit-learn, and boosted trees to LightGBM or CatBoost. Both fit in that Python machine learning library's Pipeline. + +How to choose: + +- Classification, regression, and clustering on tabular data: scikit-learn +- Boosted trees that train fast on large datasets: LightGBM +- Boosted trees on data with many categorical columns: CatBoost +- Bayesian networks and causal models: pgmpy +- Feature engineering on pandas dataframes: Feature-engine +- Boosted trees on a Spark, Dask, or Ray cluster: XGBoost +- Forecasting time series without training a model first: TimesFM + +scikit-learn covers [supervised and unsupervised learning](https://scikit-learn.org/stable/getting_started.html), plus preprocessing, model selection, and evaluation. Deep learning is [out of its scope](https://scikit-learn.org/stable/faq.html#why-is-there-no-support-for-deep-or-reinforcement-learning-will-there-be-such-support-in-the-future). To pick a model, follow its [Choosing the right estimator](https://scikit-learn.org/stable/machine_learning_map.html) chart. Split your data into train and test sets [before any preprocessing](https://scikit-learn.org/stable/common_pitfalls.html#how-to-avoid-data-leakage). Then put the preprocessing and the model in one Pipeline, so cross-validation and search never fit on the data they score. For gradient boosting without another dependency, its HistGradientBoostingClassifier and HistGradientBoostingRegressor [handle missing values and categorical data](https://scikit-learn.org/stable/modules/ensemble.html#gradient-boosted-trees) with no preprocessing. + +LightGBM aims at [faster training and lower memory use](https://lightgbm.readthedocs.io/en/latest/). It grows trees leaf-wise, so `num_leaves` is [the main parameter to tune](https://lightgbm.readthedocs.io/en/latest/Parameters-Tuning.html#tune-parameters-for-the-leaf-wise-best-first-tree): keep it below 2^(max_depth). To prevent overfitting, raise `min_data_in_leaf`. Instead of one-hot encoding, mark categorical columns with `categorical_feature`: it [often performs better](https://lightgbm.readthedocs.io/en/latest/Advanced-Topics.html#categorical-feature-support). With a validation set, use [early stopping](https://lightgbm.readthedocs.io/en/latest/Python-Intro.html#early-stopping) to find the number of boosting rounds. + +CatBoost takes [non-numeric features without preprocessing](https://catboost.ai/) and gives good results with its default parameters. Its docs say [not to one-hot encode](https://catboost.ai/docs/en/features/categorical-features) during preprocessing: list the categorical columns in `cat_features` instead. Before tuning anything else, [rule out underfitting and overfitting](https://catboost.ai/docs/en/concepts/parameter-tuning): set a large number of iterations, and turn on the overfitting detector and the use-best-model option. The learning rate is set from your data by default. + +pgmpy does [causal and probabilistic reasoning with graphical models](https://pgmpy.org/), from learning a graph from data to running inference on the fitted model. scikit-learn [leaves graphical models out](https://scikit-learn.org/stable/faq.html#will-you-add-graphical-models-or-sequence-prediction-to-scikit-learn), so use pgmpy for Bayesian networks. Every discovery algorithm [follows one pattern](https://pgmpy.org/guides/causal_discovery.html#api): instantiate, fit, and read the result. Switching algorithms means changing only the class. Pass what you already know about the domain as [required or forbidden edges](https://pgmpy.org/guides/causal_discovery.html#expert-knowledge) with `ExpertKnowledge`. For queries, Variable Elimination is [the default choice](https://pgmpy.org/guides/probabilistic_inference.html#exact-inference) while the model is small enough for exact inference. + +Feature-engine is for work where [pandas and scikit-learn are your main tools](https://feature-engine.trainindata.com/en/latest/#sitting-at-the-interface-of-pandas-and-scikit-learn). Each transformer takes the columns it changes in its `variables` argument, so it [applies steps to selected groups of variables](https://scikit-learn.org/stable/related_projects.html). Fit the transformers on the training set and transform both sets, as the [quick start](https://feature-engine.trainindata.com/en/latest/quickstart/index.html) does. Put them in a scikit-learn Pipeline, and your [whole feature engineering pipeline](https://feature-engine.trainindata.com/en/latest/quickstart/index.html#feature-engine-within-scikit-learn-s-pipeline) saves as one object. + +XGBoost is built to be [efficient, flexible, and portable](https://xgboost.readthedocs.io/en/stable/). The same code runs on distributed environments, and its docs cover training on Dask, Spark, and Ray. Use its scikit-learn interface, like `XGBClassifier`, so it [works with scikit-learn's tools](https://xgboost.readthedocs.io/en/stable/python/sklearn_estimator.html) such as cross-validation. For categorical columns, pass a dataframe with the `category` dtype and set [`enable_categorical`](https://xgboost.readthedocs.io/en/stable/tutorials/categorical.html#training-with-scikit-learn-interface). Its docs suggest you tune with cross-validation, then [retrain with the best parameters and early stopping](https://xgboost.readthedocs.io/en/stable/python/sklearn_estimator.html#early-stopping). To keep a model, [save it with `save_model`](https://xgboost.readthedocs.io/en/stable/tutorials/saving_model.html), since a pickle is a memory snapshot meant only for checkpoints. + +TimesFM is a forecasting model that Google Research pretrained on a large time-series corpus. It [does well zero-shot](https://research.google/blog/a-decoder-only-foundation-model-for-time-series-forecasting/) on benchmarks from many domains, so you can forecast without training a model first. Install it with the extra for your backend, and load a checkpoint from the Hugging Face Hub. The code is Apache licensed, but [the pretrained weights carry their own license](https://github.com/google-research/timesfm), so check the model card before commercial or production use. + +Every pick but TimesFM works with scikit-learn. XGBoost, LightGBM, and CatBoost ship scikit-learn estimators, Feature-engine's transformers go in a Pipeline, and pgmpy is [scikit-learn compatible where possible](https://pgmpy.org/). Independent benchmarks find no single winner among the three boosting libraries, so compare them on your own data in the same cross-validation. + +Treat a model file you didn't make like code. scikit-learn's docs say to [never load a pickle from an untrusted source](https://scikit-learn.org/stable/model_persistence.html#security-maintainability-limitations), and point to skops.io or ONNX instead. XGBoost's [security notes](https://xgboost.readthedocs.io/en/stable/security.html#use-of-python-pickle) say the same about pickles. From 74513330aa7b23ae19e456dc6a97ff6cac99e0b1 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 03:03:31 +0800 Subject: [PATCH 042/104] docs: add natural language processing category intro The page had no intro, so readers got no pick among spaCy, NLTK, Stanza, and the Chinese text libraries. Co-Authored-By: Claude --- .../natural-language-processing.md | 24 +++++++++++++++++++ 1 file changed, 24 insertions(+) create mode 100644 website/data/category_intros/natural-language-processing.md diff --git a/website/data/category_intros/natural-language-processing.md b/website/data/category_intros/natural-language-processing.md new file mode 100644 index 00000000..de0180fc --- /dev/null +++ b/website/data/category_intros/natural-language-processing.md @@ -0,0 +1,24 @@ +Whether you're building an app or learning the field, there's a Python NLP library for it: spaCy for apps, NLTK for learning. Stanza covers many languages. + +How to choose: + +- Production NLP pipelines: spaCy +- Learning NLP, or lexical resources like WordNet: NLTK +- Many languages, or CoreNLP from Python: Stanza +- Chinese word segmentation: jieba +- Chinese characters to pinyin: pypinyin +- Spaces between CJK text and letters or digits: pangu.py + +spaCy is [designed for production use](https://spacy.io/usage/facts-figures#comparison-usage): you build and train NLP pipelines, then package them to deploy. Its trained pipelines are Python packages. In an automated build, [install them with pip from a direct link](https://spacy.io/usage/models#download-pip) instead of spaCy's download command, and put that link in your requirements.txt. In a larger code base, [import the pipeline as a module](https://spacy.io/usage/models#models-loading), so a missing one raises an ImportError right away. Run your texts through `nlp.pipe` [in batches](https://spacy.io/usage/processing-pipelines#processing), and disable the components you don't need. To train your own pipeline, run [`spacy train` with one config file](https://spacy.io/usage/training#quickstart). + +NLTK comes with [corpora and lexical resources such as WordNet](https://www.nltk.org/), plus libraries to tokenize, tag, parse, and classify text. Its creators wrote a book that teaches NLP with it, and you can read it online. The book also says NLTK [isn't highly optimized for runtime performance](https://www.nltk.org/book/ch00.html#natural-language-toolkit-nltk), so use it to learn and experiment. The data ships separately: [get it with NLTK's data downloader](https://www.nltk.org/data.html), and set `NLTK_DATA` when you install it somewhere other than the standard locations. + +Stanza is [designed to work across many languages, using the Universal Dependencies formalism](https://stanfordnlp.github.io/stanza/#about). It's also the official Python interface to Stanford's Java CoreNLP, though its own pipeline doesn't need CoreNLP. You can [list the processors to load](https://stanfordnlp.github.io/stanza/getting_started.html#specifying-processors) with `processors=`. Pass all your documents to the pipeline [at once](https://stanfordnlp.github.io/stanza/getting_started.html#processing-multiple-documents), since a for loop over one sentence at a time is very slow. For a lot of text, [run it on a GPU](https://stanfordnlp.github.io/stanza/getting_started.html#controlling-devices). To keep the pipeline from downloading anything at runtime, [download the models ahead of time](https://stanfordnlp.github.io/stanza/getting_started.html#downloading-models-for-offline-usage). + +jieba [segments Chinese text into words](https://github.com/fxsjy/jieba): accurate mode suits text analysis, and search engine mode cuts long words into short ones for better recall. Add your own words with `jieba.load_userdict()` to get higher accuracy, and for Traditional Chinese, switch to its bigger dictionary with `jieba.set_dictionary()`. The other picks work with it: spaCy can [use jieba as its Chinese segmenter](https://spacy.io/usage/models#chinese), and Stanza [supports it as a tokenizer](https://stanfordnlp.github.io/stanza/pipeline.html). + +pypinyin [matches pinyin by whole words](https://github.com/mozillazg/python-pinyin), so it handles characters with more than one reading. It also writes zhuyin (Bopomofo) and Wade-Giles. When a wrong word split gives a wrong reading, [segment the text with jieba first](https://pypinyin.readthedocs.io/zh-cn/latest/faq.html) and pass in the list of words. For readings that are still wrong, [add your own](https://pypinyin.readthedocs.io/zh-cn/latest/usage.html#custom-dict) with `load_phrases_dict()` or `load_single_dict()`. + +pangu.py [inserts spaces between CJK characters and letters, digits, and symbols](https://github.com/vinta/pangu.py). Call `pangu.space_text()` on a string or `pangu.space_file()` on a file. From the command line, `pangu-py -c` prints the corrected text and exits with 1 when the spacing needed fixing. + +Check the license of the models and data, not only the library. spaCy is [MIT](https://github.com/explosion/spaCy/blob/master/LICENSE), but its [Spanish pipelines](https://spacy.io/models/es) are GPL and its [Italian ones](https://spacy.io/models/it) are for non-commercial use only. NLTK is [Apache](https://github.com/nltk/nltk/blob/develop/LICENSE.txt), and its corpora come [under various licenses](https://github.com/nltk/nltk/wiki/FAQ), each listed in its own README. From 3c05cd63d61d2a114a8619f7fac111478634467a Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 03:26:34 +0800 Subject: [PATCH 043/104] docs: add web frameworks category intro The page had no intro, so readers got no pick among Django, Flask, and the other full-stack and async frameworks. Co-Authored-By: Claude --- .../data/category_intros/web-frameworks.md | 35 +++++++++++++++++++ 1 file changed, 35 insertions(+) create mode 100644 website/data/category_intros/web-frameworks.md diff --git a/website/data/category_intros/web-frameworks.md b/website/data/category_intros/web-frameworks.md new file mode 100644 index 00000000..653d607d --- /dev/null +++ b/website/data/category_intros/web-frameworks.md @@ -0,0 +1,35 @@ +When you want an ORM, an admin, and auth built in, use Django. Flask, a smaller Python web framework, gives you a core you extend. + +How to choose: + +- ORM, admin, and auth built in: Django +- A small core you extend with your own picks: Flask +- One file with no dependencies: Bottle +- An app that starts small and may grow large: Pyramid +- HTML from the server with HTMX: FastHTML +- A minimal async toolkit to build on: Starlette +- Long polling, WebSockets, and other long-lived connections: Tornado +- Async with sessions, caching, and ORM integration built in: Litestar +- Frontend and backend in pure Python: Reflex + +Django comes with [a full stack for convenience](https://docs.djangoproject.com/en/stable/misc/design-philosophies/), and its pieces stay independent where possible. You describe your database layout as models in Python, and Django [builds an admin interface from them](https://docs.djangoproject.com/en/stable/intro/overview/). [User authentication](https://docs.djangoproject.com/en/stable/topics/auth/) with accounts, groups, and permissions is built in too. A project holds your settings and [one or more apps](https://docs.djangoproject.com/en/stable/intro/tutorial01/#creating-the-polls-app), and an app can move between projects. Before you deploy, run [`manage.py check --deploy`](https://docs.djangoproject.com/en/stable/howto/deployment/checklist/#run-manage-py-check-deploy) against your production settings. Async views work under WSGI, but for slow streaming and long polling, [deploy Django under ASGI](https://docs.djangoproject.com/en/stable/topics/async/#async-views). + +Flask keeps [the core simple but extensible](https://flask.palletsprojects.com/en/stable/design/#what-does-micro-mean): it doesn't pick your database or form library, and extensions add them. Create the app in an [application factory](https://flask.palletsprojects.com/en/stable/patterns/appfactories/) and bind each extension with `init_app`, so tests can build instances with their own settings. Flask runs each async view on a separate thread, not an event loop, which [costs performance compared with ASGI frameworks](https://flask.palletsprojects.com/en/stable/design/#async-await-and-asgi-support). For a mainly async codebase, its docs point you to an async-first framework. + +Bottle is [a single file module](https://bottlepy.org/docs/stable/) with no dependencies outside the standard library. Its FAQ pitches it for [prototyping, weekend projects, and small applications](https://bottlepy.org/docs/stable/faq.html#is-bottle-suitable-for-complex-applications), and suggests a full-stack framework like Django when you have tight deadlines. In production, have [Gunicorn or another WSGI server load your app](https://bottlepy.org/docs/stable/deployment.html) instead of calling `run()`. + +Pyramid is built so you [don't have to rewrite a small app in another framework](https://docs.pylonsproject.org/projects/pyramid/en/latest/narr/introduction.html#what-makes-pyramid-unique) when it gets too big. Its core maps URLs to code, handles security, and serves static assets, and it makes no assertions about which database or template system you use. Start from [its cookiecutter](https://docs.pylonsproject.org/projects/pyramid/en/latest/narr/project.html), which asks for your template language, persistence, and URL mapping. Deploy with the `production.ini` it generates, which turns off the interactive debugger. + +FastHTML is [designed to create hypermedia applications](https://www.fastht.ml/about/tech): it returns HTML from the server, the approach HTMX uses, and you often won't write any JavaScript at all. It's built on Starlette and Uvicorn. Its docs say [not to assume other frameworks' best practices apply](https://www.fastht.ml/docs/ref/best_practice.html): let the function name define each route, and use only GET and POST. + +Starlette is [a lightweight ASGI framework/toolkit](https://starlette.dev/#framework-or-toolkit): use it as a complete framework, or take any of its components on their own. Install an ASGI server such as Uvicorn next to it, and pick [any async database library](https://starlette.dev/database/) you like. Keep configuration [in environment variables or a `.env` file](https://starlette.dev/config/) that you don't commit. + +Tornado [isn't based on WSGI](https://www.tornadoweb.org/en/stable/#threads-and-wsgi) and typically runs one thread per process, with its own web framework and HTTP server used together. Its non-blocking I/O makes it a fit for [long polling, WebSockets, and other long-lived connections](https://www.tornadoweb.org/en/stable/). Hand blocking code to `run_in_executor`, and [run one process per CPU](https://www.tornadoweb.org/en/stable/guide/running.html#processes-and-ports). + +Litestar is [not a microframework](https://docs.litestar.dev/latest/#philosophy): it comes with ORM integration, client- and server-side sessions, and caching, though it will never have its own ORM. Class-based controllers sit at its core. Set dependencies, guards, and middleware on [any layer](https://docs.litestar.dev/latest/onboarding/flask.html), from the app down to one handler, and the setting closest to the handler wins. + +Reflex builds the [frontend, backend, and database in pure Python](https://reflex.dev/docs/getting-started/introduction/). It [compiles your UI to a React frontend](https://reflex.dev/docs/advanced-onboarding/how-reflex-works/), runs your state handlers on the server, and syncs the two over WebSockets. Change state only through event handlers on your `State` class. In production, the Reflex team runs Redis as the state manager. Keep auth data and other sensitive state in [backend-only vars](https://reflex.dev/docs/vars/base-vars/#backend-only-vars). + +Don't deploy on the development server: [Flask](https://flask.palletsprojects.com/en/stable/deploying/) and [Django](https://docs.djangoproject.com/en/stable/howto/deployment/checklist/#switch-away-from-manage-py-runserver) both tell you to switch to a production server, and to turn debug mode off. + +For a site with forms and logins, turn on CSRF protection. Django's [CSRF middleware is on by default](https://docs.djangoproject.com/en/stable/howto/csrf/). Tornado, Pyramid, and Litestar have it as a setting you turn on. Flask [leaves it to a form library](https://flask.palletsprojects.com/en/stable/web-security/#cross-site-request-forgery-csrf), and Starlette to third-party middleware. From d45070592679adf7a970c051200a093fb2bc9065 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 03:26:57 +0800 Subject: [PATCH 044/104] docs: add web APIs category intro The Web APIs category page had no intro, leaving readers without a pick among Django REST framework, Django Ninja, FastAPI, and the GraphQL and RPC libraries. Co-Authored-By: Claude --- website/data/category_intros/web-apis.md | 33 ++++++++++++++++++++++++ 1 file changed, 33 insertions(+) create mode 100644 website/data/category_intros/web-apis.md diff --git a/website/data/category_intros/web-apis.md b/website/data/category_intros/web-apis.md new file mode 100644 index 00000000..87bc81c7 --- /dev/null +++ b/website/data/category_intros/web-apis.md @@ -0,0 +1,33 @@ +For a Django project, pick Django REST framework or Django Ninja. Outside Django, build on FastAPI, a Python API framework based on type hints. + +How to choose: + +- Django, with CRUD endpoints over your models: Django REST framework +- Django, with endpoints declared by type hints: Django Ninja +- A new service outside Django: FastAPI +- GraphQL over Django models: Strawberry GraphQL Django +- Django, with msgspec, attrs, or dataclass schemas: django-modern-rest +- A Flask app: APIFlask +- An OpenAPI spec written before the code: Connexion +- GraphQL on FastAPI, Flask, or another framework: Strawberry +- RPC between services in any language: gRPC + +Django REST framework adds serializers and a [browsable API](https://www.django-rest-framework.org/) to Django. For CRUD over your models, [group the views in a `ModelViewSet` and register it with a router](https://www.django-rest-framework.org/tutorial/quickstart/). For an OpenAPI schema, its docs [recommend drf-spectacular](https://www.django-rest-framework.org/topics/documenting-your-api/#drf-spectacular). By default, the API [allows unrestricted access](https://www.django-rest-framework.org/api-guide/permissions/#setting-the-permission-policy), so set a policy in `DEFAULT_PERMISSION_CLASSES`. + +Django Ninja is [heavily inspired by FastAPI](https://django-ninja.dev/motivation/) and works with Django's ORM, URLs, views, and auth. You write each endpoint as a function with type hints, and it [generates the OpenAPI docs](https://django-ninja.dev/) from them. Run async views on [an ASGI server](https://django-ninja.dev/guides/async-support/). If you use Django's cookie-based auth, [keep Django's CSRF protection on](https://django-ninja.dev/reference/csrf/#use-djangos-built-in-csrf-protection). + +FastAPI builds [request validation and OpenAPI docs from standard type hints](https://fastapi.tiangolo.com/), on top of Pydantic and Starlette. Install `fastapi[standard]`, which brings Uvicorn and the `fastapi` command: run `fastapi dev` while you work and [`fastapi run` in production](https://fastapi.tiangolo.com/fastapi-cli/). + +Strawberry GraphQL Django [builds types from your Django models](https://strawberry.rocks/docs/django), while Strawberry alone [only provides a GraphQL view](https://strawberry.rocks/docs/integrations/django) for Django. Add its [query optimizer](https://strawberry.rocks/docs/django/guide/optimizer), which calls `select_related()` and `prefetch_related()` for you to avoid N+1 queries. + +django-modern-rest [drops into an existing Django app](https://django-modern-rest.readthedocs.io/en/latest/) and validates both requests and responses against your schemas: msgspec, Pydantic, attrs, dataclasses, and more. Its docs [recommend always installing msgspec](https://django-modern-rest.readthedocs.io/en/latest/pages/getting-started.html#installation) to parse JSON, even when your schemas are Pydantic models. + +APIFlask is a [thin wrapper on Flask](https://apiflask.com/migrations/flask/): swap `Flask` for `APIFlask` and `Blueprint` for `APIBlueprint`, and your [Flask extensions keep working](https://apiflask.com/comparison/#apiflask-vs-fastapi). Declare each endpoint's input and output with `@app.input()` and `@app.output()`, as [marshmallow schemas or Pydantic models](https://apiflask.com/), and APIFlask generates the OpenAPI docs from them. + +Connexion is spec-first: you write the OpenAPI spec, and Connexion [routes and validates requests against it](https://connexion.readthedocs.io/en/latest/#why-connexion), so server and client can be built in parallel. FastAPI goes the other way, and its maintainer says it's [not meant for writing the schema first](https://github.com/fastapi/fastapi/discussions/6169). Start a new project on [`AsyncApp`](https://connexion.readthedocs.io/en/latest/quickstart.html#creating-your-application), or on `FlaskApp` to keep the Flask ecosystem. + +Strawberry builds a GraphQL schema from [dataclasses and type hints](https://strawberry.rocks/docs), and FastAPI's docs [recommend it for GraphQL](https://fastapi.tiangolo.com/how-to/graphql/#graphql-with-strawberry). Before you deploy, [turn off GraphiQL and introspection](https://strawberry.rocks/docs/operations/deployment), and add the [security extensions](https://strawberry.rocks/docs/operations/deployment#security-extensions) that limit query depth, aliases, and tokens. + +gRPC has you [define a service once in a `.proto` file](https://grpc.io/docs/languages/python/basics/) and generate its clients and servers in any language gRPC supports. Install `grpcio` and `grpcio-tools`, and [compile the `.proto` file into Python code with `grpc_tools.protoc`](https://grpc.io/docs/languages/python/quickstart/). [Use TLS](https://grpc.io/docs/guides/auth/) to authenticate the server and encrypt the traffic. + +Keep separate models for what an endpoint takes in and what it sends back. FastAPI [filters the response through the output model](https://fastapi.tiangolo.com/tutorial/response-model/#add-an-output-model), so fields the output model leaves out, like a password, never reach the client. APIFlask [recommends separate input and output schemas](https://apiflask.com/schema/#marshmallow) too. From 10a7f12861ce64ef0567c85ff77e08e45da15f7a Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 03:27:35 +0800 Subject: [PATCH 045/104] docs: add web servers category intro The Web Servers page had no intro, so readers got no pick among Gunicorn, Uvicorn, Waitress, Granian, and Hypercorn. Co-Authored-By: Claude --- website/data/category_intros/web-servers.md | 21 +++++++++++++++++++++ 1 file changed, 21 insertions(+) create mode 100644 website/data/category_intros/web-servers.md diff --git a/website/data/category_intros/web-servers.md b/website/data/category_intros/web-servers.md new file mode 100644 index 00000000..68fa6400 --- /dev/null +++ b/website/data/category_intros/web-servers.md @@ -0,0 +1,21 @@ +In production, Gunicorn serves WSGI apps like Django, and Uvicorn serves ASGI apps like FastAPI. Waitress, a pure-Python web server, runs WSGI apps on Windows. + +How to choose: + +- A WSGI app (Flask, Django) on Linux or macOS: Gunicorn +- An ASGI app (FastAPI, Starlette): Uvicorn +- A WSGI app on Windows, or a server in pure Python: Waitress +- One server for both ASGI and WSGI apps, or throughput above all: Granian +- An app on Trio, or HTTP/3 from the Python server: Hypercorn + +Gunicorn is a [pre-fork server](https://gunicorn.org/design/#server-model): one process manages a pool of worker processes. It's [built for Unix](https://gunicorn.org/). With its default sync workers, your proxy must [buffer slow clients](https://gunicorn.org/deploy/#nginx-configuration), or Gunicorn is open to denial-of-service attacks. Start with (2 × CPU cores) + 1 workers and [adjust under load](https://gunicorn.org/design/#how-many-workers). For long blocking calls, streaming, or WebSockets, switch to [async workers](https://gunicorn.org/design/#when-to-use-async-workers) like gevent. + +Uvicorn is an [ASGI web server](https://uvicorn.dev/). In production, run it under a [process manager](https://uvicorn.dev/deployment/#using-a-process-manager). Its [built-in one](https://uvicorn.dev/deployment/#built-in) starts workers with `--workers` and restarts any that die, and its docs also cover [running Uvicorn workers under Gunicorn](https://uvicorn.dev/deployment/#gunicorn). In a container, run a single Uvicorn process and [let your orchestrator scale](https://uvicorn.dev/deployment/docker/) the number of containers. + +Granian is a [Rust HTTP server](https://github.com/emmett-framework/granian) that serves ASGI, WSGI, and RSGI apps from one package. Its README [suggests it](https://github.com/emmett-framework/granian#rationale) when you care about throughput above all, and not when you want pure Python or your app relies on Trio or gevent. Pass [`--interface asgi` or `--interface wsgi`](https://github.com/emmett-framework/granian#options), since the default is RSGI. Start with [one worker per CPU core](https://github.com/emmett-framework/granian#workers-and-threads), or one per container on Docker or Kubernetes, rather than numbers suggested for other servers. + +Hypercorn is an ASGI server [inspired by Gunicorn](https://hypercorn.readthedocs.io/en/latest/). It speaks HTTP/1 and HTTP/2, with WebSockets over both, and runs on [Trio](https://hypercorn.readthedocs.io/en/latest/discussion/workers.html#trio) as well as asyncio. For HTTP/3, install its [`h3` extra](https://github.com/pgjones/hypercorn). Its docs recommend setting [`server_names`](https://hypercorn.readthedocs.io/en/latest/how_to_guides/server_names.html#dns-rebinding-attacks) to the hosts you serve, to block DNS rebinding attacks. + +Waitress is a [pure-Python WSGI server](https://docs.pylonsproject.org/projects/waitress/en/latest/index.html) with no dependencies outside the standard library, and it runs on both Unix and Windows. Waitress [doesn't support TLS](https://docs.pylonsproject.org/projects/waitress/en/latest/reverse-proxy.html) itself, so put a reverse proxy in front of it for HTTPS. + +Put a proxy like Nginx in front: Gunicorn's docs [strongly recommend it](https://gunicorn.org/deploy/), and Uvicorn's recommend it [for resilience](https://uvicorn.dev/deployment/#running-behind-nginx). Behind a proxy, tell your server which proxies to trust for `X-Forwarded-*` headers, since any client can set them. [Uvicorn](https://uvicorn.dev/deployment/#proxies-and-forwarded-headers) and Gunicorn take `--forwarded-allow-ips`, Granian [`trusted_hosts`](https://github.com/emmett-framework/granian#proxies-and-forwarded-headers), Hypercorn [`ProxyFixMiddleware`](https://hypercorn.readthedocs.io/en/latest/how_to_guides/proxy_fix.html), and Waitress [`trusted_proxy`](https://docs.pylonsproject.org/projects/waitress/en/latest/reverse-proxy.html#passing-the-proxy-headers-to-setup-the-wsgi-environment). In Uvicorn, Gunicorn, Granian, and Waitress, trust every address with `*` only when [no client can reach the server directly](https://gunicorn.org/deploy/#nginx-configuration). From cabd05ec7e193f0b2b9c4dd035ea0ae6f88f730e Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 06:26:18 +0800 Subject: [PATCH 046/104] docs: add WebSocket category intro The page had no intro, so readers got no pick among Channels, Flask-SocketIO, websockets, and Autobahn|Python. Co-Authored-By: Claude --- website/data/category_intros/websocket.md | 21 +++++++++++++++++++++ 1 file changed, 21 insertions(+) create mode 100644 website/data/category_intros/websocket.md diff --git a/website/data/category_intros/websocket.md b/website/data/category_intros/websocket.md new file mode 100644 index 00000000..163e4eb7 --- /dev/null +++ b/website/data/category_intros/websocket.md @@ -0,0 +1,21 @@ +With Django, Channels is the pick; with Flask and Socket.IO clients, Flask-SocketIO. Standalone apps run on websockets, a Python WebSocket library. + +How to choose: + +- A Django project: Channels +- A Flask app with Socket.IO clients: Flask-SocketIO +- Plain WebSocket servers and clients: websockets +- A few real-time features next to a Django project: websockets as a separate server +- RPC and pub/sub over WAMP, or a Twisted app: Autobahn|Python + +Channels [extends Django beyond HTTP](https://channels.readthedocs.io/en/latest/) to handle WebSockets, and it [integrates with Django's auth and sessions](https://channels.readthedocs.io/en/latest/introduction.html). Write [sync consumers by default](https://channels.readthedocs.io/en/latest/topics/consumers.html#basic-layout). Switch to async ones only when async handling helps and every library you call is async-native. Serve everything with Daphne, or [keep HTTP on your WSGI server](https://channels.readthedocs.io/en/latest/deploying.html#http-and-websocket) and send only WebSockets to Daphne. + +Flask-SocketIO speaks Socket.IO, which is [not a WebSocket implementation](https://socket.io/docs/v4/#what-socketio-is-not): a plain WebSocket client can't connect to it. Pick it when your clients use [a Socket.IO client library](https://flask-socketio.readthedocs.io/en/latest/intro.html#requirements). In return, you get events, [rooms, and broadcasting](https://flask-socketio.readthedocs.io/en/latest/getting_started.html#rooms). Start the server with `socketio.run()`, since `flask run` [lacks WebSocket support](https://flask-socketio.readthedocs.io/en/latest/getting_started.html#initialization). + +websockets runs on asyncio by default, which is [ideal for servers with many connections](https://websockets.readthedocs.io/en/stable/). For clients, its threading implementation is a good alternative. It [isn't an HTTP server](https://websockets.readthedocs.io/en/stable/faq/server.html#how-do-i-run-http-and-websocket-servers-on-the-same-port), so [run it as its own program](https://websockets.readthedocs.io/en/stable/deploy/index.html#how-do-i-start-a-process) that calls `serve()`, not under a WSGI or ASGI server. For a few real-time features in a Django project, like notifications, its docs find [a separate websockets server next to Django](https://websockets.readthedocs.io/en/stable/howto/django.html) well suited, where Channels means switching to a new deployment architecture. + +Autobahn|Python implements [both WebSocket and WAMP, on Twisted or asyncio](https://autobahn.readthedocs.io/en/latest/). WAMP adds [RPC and pub/sub over WebSocket](https://github.com/crossbario/autobahn-python), and every WAMP client [needs a WAMP router to talk to](https://autobahn.readthedocs.io/en/latest/wamp/programming.html#wamp-programming-1). Write components with functions and decorators, [the recommended approach](https://autobahn.readthedocs.io/en/latest/wamp/programming.html#creating-components-1). + +Always [secure WebSocket connections with TLS](https://websockets.readthedocs.io/en/stable/topics/security.html#encryption) in production. Any site can open a WebSocket to yours, with your users' cookies attached. If you serve private data, [restrict the allowed origins](https://channels.readthedocs.io/en/latest/topics/security.html#websockets): Channels has `AllowedHostsOriginValidator`, websockets has the [`origins` argument](https://websockets.readthedocs.io/en/stable/reference/asyncio/server.html#websockets.asyncio.server.serve), and Flask-SocketIO [allows only the same origin by default](https://flask-socketio.readthedocs.io/en/latest/deployment.html#cross-origin-controls). + +Broadcasting across processes needs a message broker such as Redis. Channels' [production channel layer](https://channels.readthedocs.io/en/latest/topics/channel_layers.html#redis-channel-layer) runs on it, Flask-SocketIO [takes it as a message queue](https://flask-socketio.readthedocs.io/en/latest/deployment.html#using-multiple-workers), with sticky sessions at the load balancer, and websockets' docs [suggest it for pub/sub](https://websockets.readthedocs.io/en/stable/faq/server.html#how-do-i-send-a-message-to-a-channel-a-topic-or-some-users). From d7c709371fdbe4e16f022ba18afc586ade336cfd Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 06:26:22 +0800 Subject: [PATCH 047/104] docs: add template engines category intro The page had no intro, so readers got no pick between Jinja and Mako. Co-Authored-By: Claude --- website/data/category_intros/template-engines.md | 15 +++++++++++++++ 1 file changed, 15 insertions(+) create mode 100644 website/data/category_intros/template-engines.md diff --git a/website/data/category_intros/template-engines.md b/website/data/category_intros/template-engines.md new file mode 100644 index 00000000..b59f345d --- /dev/null +++ b/website/data/category_intros/template-engines.md @@ -0,0 +1,15 @@ +If your templates should hold no Python code, Jinja fits: a Python template engine whose sandbox also renders untrusted templates. Mako embeds plain Python. + +How to choose: + +- Templates without embedded Python: Jinja +- Templates your users write: Jinja, in its sandbox +- Plain Python inside your templates: Mako + +Jinja [doesn't allow arbitrary Python code in templates](https://jinja.palletsprojects.com/en/stable/faq/): you get blocks, filters, and function calls, and the rest of your logic stays in Python. Outside a framework, [create one `Environment`](https://jinja.palletsprojects.com/en/stable/api/) when your app starts and load templates through a `PackageLoader`. Put the layout your pages share in a base template, and let each page [extend it and override its blocks](https://jinja.palletsprojects.com/en/stable/templates/#template-inheritance). Flask [sets up Jinja for you](https://flask.palletsprojects.com/en/stable/templating/#jinja-setup), Django has a [built-in Jinja2 backend](https://docs.djangoproject.com/en/stable/topics/templates/#django.template.backends.jinja2.Jinja2), and FastAPI [supports it with `Jinja2Templates`](https://fastapi.tiangolo.com/advanced/templates/). + +When your users write templates, like a report layout, render them in Jinja's [sandbox](https://jinja.palletsprojects.com/en/stable/sandbox/#sandbox), which can block attribute access, method calls, and other operations. Its docs say the sandbox alone [is not a solution for perfect security](https://jinja.palletsprojects.com/en/stable/sandbox/#security-considerations): catch errors when a template renders, limit CPU and memory, and pass the template only the data it needs. + +Mako is [an embedded Python language](https://www.makotemplates.org/): inside `<% %>` tags you [write regular Python](https://docs.makotemplates.org/en/latest/syntax.html#python-blocks). A real application [loads its templates from a `TemplateLookup`](https://docs.makotemplates.org/en/latest/usage.html#using-templatelookup) with a `module_directory`, which caches each compiled template on disk as a Python module. + +Jinja leaves HTML escaping [off by default](https://jinja.palletsprojects.com/en/stable/faq/#why-is-html-escaping-not-the-default), since it also renders plain text, emails, and config files. When you create the `Environment` yourself, turn it on with [`select_autoescape()`](https://jinja.palletsprojects.com/en/stable/api/#jinja2.select_autoescape); Flask turns it on for HTML templates, and Django's Jinja2 backend turns it on for all. Mako escapes HTML only through [the `h` filter](https://docs.makotemplates.org/en/latest/filtering.html#expression-filtering): add it to `default_filters` on your `TemplateLookup` to escape every expression. From 8b04838903f45132885c30dc439ca4c8a9157350 Mon Sep 17 00:00:00 2001 From: Vinta Chen Date: Sun, 27 Sep 2026 06:26:25 +0800 Subject: [PATCH 048/104] docs: add web asset management category intro The page had no intro, so readers got no pick between django-storages and Django Compressor. Co-Authored-By: Claude --- website/data/category_intros/web-asset-management.md | 12 ++++++++++++ 1 file changed, 12 insertions(+) create mode 100644 website/data/category_intros/web-asset-management.md diff --git a/website/data/category_intros/web-asset-management.md b/website/data/category_intros/web-asset-management.md new file mode 100644 index 00000000..03733638 --- /dev/null +++ b/website/data/category_intros/web-asset-management.md @@ -0,0 +1,12 @@ +Django serves static files only in development, but django-storages puts them and user uploads on S3 or other clouds. Django Compressor bundles CSS and JS. + +How to choose: + +- Static files and user uploads on S3, Google Cloud Storage, or Azure: django-storages +- CSS and JavaScript combined and minified from your templates: Django Compressor + +django-storages is [a collection of storage backends for Django](https://django-storages.readthedocs.io/en/latest/): Amazon S3, Google Cloud Storage, Azure Storage, and more. Install the extra for your backend, like [`django-storages[s3]`](https://django-storages.readthedocs.io/en/latest/backends/amazon-S3.html#installation). Configure it through Django's [`STORAGES` setting](https://docs.djangoproject.com/en/stable/ref/settings/#storages), with the backend's settings under `OPTIONS`. The `default` key stores user uploads, and a `staticfiles` key has `collectstatic` [put your static files on S3](https://django-storages.readthedocs.io/en/latest/backends/amazon-S3.html#configuration-settings) too. Django's security docs say to [serve user uploads from a separate domain](https://docs.djangoproject.com/en/stable/topics/security/#user-uploaded-content), not a subdomain of your site. + +Django Compressor [processes, combines, and minifies](https://github.com/django-compressor/django-compressor) the CSS and JavaScript in your Django templates into cacheable static files. It also supports compilers like Sass and LESS. Its maintainer sees it as [an alternative for people who don't want to keep up](https://github.com/django-compressor/django-compressor/discussions/1095) with JavaScript build tools and want one easy setup for development and production. Add `compressor` to `INSTALLED_APPS` and [its finder](https://django-compressor.readthedocs.io/en/stable/quickstart.html) to `STATICFILES_FINDERS`. Then wrap your `` and `