This is an automated email from the ASF dual-hosted git repository.

Lee-W pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/airflow.git


The following commit(s) were added to refs/heads/main by this push:
     new 07bed4fde80 Link registry modules to the guide section that documents 
them (#71477)
07bed4fde80 is described below

commit 07bed4fde809386b40b37941008253230af1c65b
Author: Wei Lee <[email protected]>
AuthorDate: Wed Oct 7 14:28:48 2026 +0900

    Link registry modules to the guide section that documents them (#71477)
---
 dev/registry/extract_parameters.py                 |  46 ++-
 dev/registry/extract_versions.py                   |  85 +++++
 dev/registry/registry_contract_models.py           |   3 +
 dev/registry/registry_tools/docs_guides.py         | 186 +++++++++++
 dev/registry/tests/test_docs_guides.py             | 369 +++++++++++++++++++++
 dev/registry/tests/test_extract_parameters.py      | 107 +++++-
 dev/registry/tests/test_extract_versions.py        | 195 ++++++++++-
 .../tests/test_registry_contract_models.py         |   9 +
 providers/common/ai/docs/operators/llm_branch.rst  |   4 +-
 .../common/ai/docs/operators/llm_file_analysis.rst |   4 +-
 .../ai/docs/operators/llm_schema_compare.rst       |   4 +-
 providers/common/ai/docs/operators/llm_sql.rst     |   4 +-
 registry/AGENTS.md                                 |  33 ++
 registry/src/css/main.css                          |  10 +
 registry/src/provider-version.njk                  |   6 +
 15 files changed, 1051 insertions(+), 14 deletions(-)

diff --git a/dev/registry/extract_parameters.py 
b/dev/registry/extract_parameters.py
index 6e09084bc04..df9f344ab00 100644
--- a/dev/registry/extract_parameters.py
+++ b/dev/registry/extract_parameters.py
@@ -55,7 +55,9 @@ from pathlib import Path
 
 import yaml
 from extract_metadata import fetch_provider_inventory, read_inventory
+from extract_versions import detect_layout, git_tag_exists, read_guide_docs as 
read_guide_docs_at_tag
 from registry_contract_models import validate_modules_catalog, 
validate_provider_parameters
+from registry_tools.docs_guides import attach_guide_urls, 
collect_guide_anchors, is_guide_page
 from registry_tools.types import (
     BASE_CLASS_IMPORTS,
     CLASS_LEVEL_CATEGORY_OVERRIDES,
@@ -98,6 +100,7 @@ class Module:
     provider_name: str
     supports_durable_execution: bool
     supports_deferrable: bool
+    guide_url: str | None = None
 
 
 def get_category(integration_name: str) -> str:
@@ -738,6 +741,41 @@ def _resolve_decorated_operator_class(decorator_fn: 
object) -> type | None:
     return candidate if inspect.isclass(candidate) else None
 
 
+def read_guide_docs(docs_dir: Path) -> dict[str, str]:
+    """Read a provider's authored reST docs from the working tree, keyed by 
path relative to ``docs_dir``."""
+    if not docs_dir.is_dir():
+        return {}
+    docs = {}
+    for path in sorted(docs_dir.rglob("*.rst")):
+        relative = path.relative_to(docs_dir).as_posix()
+        if not is_guide_page(relative):
+            continue
+        docs[relative] = path.read_text(encoding="utf-8")
+    return docs
+
+
+def read_released_guide_docs(
+    provider_id: str, version: str, provider_rel_path: Path
+) -> dict[str, str] | None:
+    """Read a provider's guide docs at its release tag, or None when there is 
no tag to read.
+
+    The guide links point at ``/stable``, which serves the released docs, so 
the
+    anchors must come from the same content; the working tree may be ahead of 
it.
+    A tag from the old flat layout yields an empty dict, meaning no guide links
+    rather than a fallback to the working tree.
+    """
+    if not version:
+        return None
+    tag = f"providers-{provider_id}/{version}"
+    if not git_tag_exists(tag):
+        return None
+    dir_path = provider_rel_path.as_posix()
+    layout = detect_layout(tag, dir_path)
+    if layout is None:
+        return None
+    return read_guide_docs_at_tag(tag, layout, dir_path)
+
+
 def discover_classes_from_provider(
     provider_yaml_path: Path,
     base_classes: dict[str, type],
@@ -748,7 +786,8 @@ def discover_classes_from_provider(
     """Discover classes from a single provider by importing its modules at 
runtime.
 
     Reads the provider.yaml to find which modules/classes to inspect, imports 
them,
-    and returns metadata for each discovered class with every `Module` 
dataclass field.
+    and returns metadata for each discovered class with every required `Module`
+    dataclass field, plus ``guide_url`` when a how-to guide documents the 
class.
     """
     with open(provider_yaml_path) as f:
         provider_yaml = yaml.safe_load(f)
@@ -985,6 +1024,11 @@ def discover_classes_from_provider(
             }
         )
 
+    guide_docs = read_released_guide_docs(provider_id, version, 
provider_rel_path)
+    if guide_docs is None:
+        guide_docs = read_guide_docs(provider_yaml_path.parent / "docs")
+    attach_guide_urls(discovered, collect_guide_anchors(guide_docs), 
base_docs_url)
+
     return discovered
 
 
diff --git a/dev/registry/extract_versions.py b/dev/registry/extract_versions.py
index d88530b1a10..d7f86251aa9 100644
--- a/dev/registry/extract_versions.py
+++ b/dev/registry/extract_versions.py
@@ -57,6 +57,7 @@ except ImportError:
     sys.exit(1)
 
 from extract_metadata import fetch_provider_inventory, read_connection_urls, 
resolve_connection_docs_url
+from registry_tools.docs_guides import attach_guide_urls, 
collect_guide_anchors, is_guide_page
 from registry_tools.types import (
     CLASS_LEVEL_CATEGORY_OVERRIDES,
     CLASS_LEVEL_SECTIONS,
@@ -127,6 +128,62 @@ def git_show(tag: str, path: str) -> str | None:
         return None
 
 
+def git_ls_tree(tag: str, prefix: str) -> list[str]:
+    """List the file paths under a prefix at a specific git tag."""
+    try:
+        result = subprocess.run(
+            ["git", "-c", "core.quotePath=false", "ls-tree", "-r", 
"--name-only", tag, "--", prefix],
+            capture_output=True,
+            cwd=AIRFLOW_ROOT,
+            check=True,
+        )
+    except subprocess.CalledProcessError:
+        return []
+    return [line for line in result.stdout.decode("utf-8").splitlines() if 
line]
+
+
+def git_cat_file_batch(tag: str, paths: list[str]) -> dict[str, str]:
+    """Read multiple files at a specific git tag in one `git cat-file --batch` 
call.
+
+    Returns a mapping of path -> content for paths that exist at the tag; a 
path
+    git reports as missing is simply absent from the result, matching 
git_show's
+    "return None for a missing path" semantics.
+
+    Decode failures are left unguarded on purpose: .rst files are Sphinx
+    convention UTF-8, an explicit "utf-8" decode is more predictable than
+    following the process locale, and a UnicodeDecodeError should surface 
loudly
+    rather than being swallowed. A failing ``git cat-file`` call also raises
+    (``check=True``); only git_show turns CalledProcessError into ``None``.
+    """
+    if not paths:
+        return {}
+
+    stdin = ("\n".join(f"{tag}:{p}" for p in paths) + "\n").encode("utf-8")
+    result = subprocess.run(
+        ["git", "cat-file", "--batch"],
+        input=stdin,
+        capture_output=True,
+        cwd=AIRFLOW_ROOT,
+        check=True,
+    )
+
+    output = result.stdout
+    pos = 0
+    contents: dict[str, str] = {}
+    for path in paths:
+        newline_idx = output.index(b"\n", pos)
+        header = output[pos:newline_idx].decode("utf-8")
+        pos = newline_idx + 1
+        if header.endswith(" missing"):
+            continue
+        _sha1, _obj_type, size_str = header.split(" ")
+        size = int(size_str)
+        content_bytes = output[pos : pos + size]
+        pos += size + 1  # skip the protocol's trailing LF, which isn't 
counted in size
+        contents[path] = content_bytes.decode("utf-8")
+    return contents
+
+
 def git_tag_exists(tag: str) -> bool:
     """Check if a git tag exists locally."""
     result = subprocess.run(
@@ -179,6 +236,32 @@ def get_source_file_path(layout: str, dir_path: str, 
module_path: str) -> str:
     return f"providers/src/{rel_file}"
 
 
+def read_guide_docs(tag: str, layout: str, dir_path: str) -> dict[str, str]:
+    """Read a provider's authored reST docs at a tag, keyed by path relative 
to its docs dir.
+
+    Only the per-provider layout keeps docs beside the provider; under the old 
flat
+    layout they lived in a top-level ``docs/`` tree, so those tags get no guide
+    links rather than links guessed from a path that moved.
+    """
+    if layout != "new":
+        return {}
+
+    docs_prefix = f"providers/{dir_path}/docs/"
+    survivors: list[tuple[str, str]] = []
+    for path in git_ls_tree(tag, docs_prefix):
+        if not path.endswith(".rst"):
+            continue
+        relative = path[len(docs_prefix) :]
+        if not is_guide_page(relative):
+            continue
+        survivors.append((relative, path))
+
+    batch_result = git_cat_file_batch(tag, [full_path for _relative, full_path 
in survivors])
+    return {
+        relative: batch_result[full_path] for relative, full_path in survivors 
if batch_result.get(full_path)
+    }
+
+
 def parse_pyproject_toml_content(content: str, layout: str) -> dict[str, Any]:
     """Parse pyproject.toml content for dependencies, requires-python, and 
extras."""
     result: dict[str, Any] = {"requires_python": "", "dependencies": [], 
"optional_extras": {}}
@@ -380,6 +463,8 @@ def extract_modules_from_yaml(
                 }
             )
 
+    attach_guide_urls(modules, collect_guide_anchors(read_guide_docs(tag, 
layout, dir_path)), base_docs_url)
+
     return modules
 
 
diff --git a/dev/registry/registry_contract_models.py 
b/dev/registry/registry_contract_models.py
index 31f4001dc49..02eb221b1ce 100644
--- a/dev/registry/registry_contract_models.py
+++ b/dev/registry/registry_contract_models.py
@@ -146,6 +146,9 @@ class ModuleContract(BaseModel):
     provider_name: str | None = None
     supports_durable_execution: bool = False
     supports_deferrable: bool = False
+    # Only set for classes and task decorators (e.g. ``@task.agent``) that a 
how-to
+    # guide documents in a section of their own.
+    guide_url: str | None = None
 
 
 class ModulesCatalogContract(BaseModel):
diff --git a/dev/registry/registry_tools/docs_guides.py 
b/dev/registry/registry_tools/docs_guides.py
new file mode 100644
index 00000000000..d19f9775704
--- /dev/null
+++ b/dev/registry/registry_tools/docs_guides.py
@@ -0,0 +1,186 @@
+# Licensed to the Apache Software Foundation (ASF) under one
+# or more contributor license agreements.  See the NOTICE file
+# distributed with this work for additional information
+# regarding copyright ownership.  The ASF licenses this file
+# to you under the Apache License, Version 2.0 (the
+# "License"); you may not use this file except in compliance
+# with the License.  You may obtain a copy of the License at
+#
+#   http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing,
+# software distributed under the License is distributed on an
+# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+# KIND, either express or implied.  See the License for the
+# specific language governing permissions and limitations
+# under the License.
+"""Map a provider's classes to the how-to guide sections that document them.
+
+A module's ``docs_url`` points at generated API reference, which tells a reader
+what the arguments are but not how the thing is meant to be used. The prose
+guides carry that, and they already mark it: a how-to guide documents one class
+(or a class and its task-flow decorator) per section, titled with the name(s)
+either at the start (``HookToolset``, ``SQLToolset``, ``AgentOperator`` &
+``@task.agent``) or after a colon at the very end, as in a section titled
+"Airflow hooks as tools: ``HookToolset``".
+
+So the mapping is read back out of the guides rather than curated anywhere: a
+hand-maintained name-to-guide table would rot silently every time a guide is
+split, renamed, or a class is dropped, and a rotten link is worse than none.
+Callers supply the reST they can see (a git tag, or the working tree) and get
+back only the anchors those sources actually contain.
+"""
+
+from __future__ import annotations
+
+import re
+from collections.abc import Mapping
+from pathlib import PurePosixPath
+from typing import Any
+
+# reST underlines an (optionally overlined) section title with a run of one
+# punctuation character, at least as long as the title itself.
+_ADORNMENT_CHARS = "!\"#$%&'()*+,-./:;<=>?@[\\]^_`{|}~"
+
+_SKIPPED_PAGE_NAMES = frozenset({"changelog.rst", "commits.rst"})
+
+
+def is_guide_page(relative_path: str) -> bool:
+    """Whether a path relative to a provider's docs directory is a how-to 
guide page.
+
+    Callers hand every ``.rst`` they can see to this before it ever reaches
+    ``collect_guide_anchors``. Two kinds of real, built pages must not go
+    further:
+
+    - Anything under a ``_``-prefixed path segment, at any depth
+      (``_api/hook/index.rst``, ``operators/_partials/foo.rst``, top-level
+      ``_partials/foo.rst``): Sphinx/autoapi output and partials are directive
+      markup, not the hand-written, reST-underlined titles this module's
+      inline-literal title convention parses.
+    - ``changelog.rst`` and ``commits.rst``: real release-note pages, not
+      how-to guides, that can carry inline-literal-formatted headings by
+      coincidence.
+    """
+    path = PurePosixPath(relative_path)
+    if any(part.startswith("_") for part in path.parts):
+        return False
+    return path.name not in _SKIPPED_PAGE_NAMES
+
+
+# A single inline-literal name: a class (``HookToolset``) or a task-flow
+# decorator (``@task.llm_file_analysis``) -- narrow enough that it still can't
+# match arbitrary prose wrapped in backticks.
+_INLINE_LITERAL_NAME = 
r"``(@?[A-Za-z_][A-Za-z0-9_]*(?:\.[A-Za-z_][A-Za-z0-9_]*)*)``"
+_INLINE_LITERAL_NAME_RE = re.compile(_INLINE_LITERAL_NAME)
+
+# Only titles that consist solely of, or end a colon-led clause with, a run of
+# inline-literal names are treated as documenting them, so prose headings
+# ("Bounded query results") never produce a link. A run is one or more names
+# joined by "&", ",", "/" or "and" -- how guides write a section that covers 
both
+# an operator and its decorator (``AgentOperator`` & ``@task.agent``).
+_NAME_SEPARATOR = r"(?:\s*[&,/]\s*|\s+and\s+)"
+_NAME_RUN = 
rf"{_INLINE_LITERAL_NAME}(?:{_NAME_SEPARATOR}{_INLINE_LITERAL_NAME})*"
+
+# Shape one (what older release tags' docs use):
+# the title is nothing but the name run. Anchoring to "$" keeps a title that
+# merely opens with a literal and continues in prose ("``SandboxToolset``
+# parameters") from claiming to document that class.
+_LEADING_LITERAL_NAME_RUN = re.compile(rf"^{_NAME_RUN}\s*$")
+# Shape two (what current docs use): a prose lead-in, a colon, then the name 
run runs to
+# the very end of the title. Anchoring to "$" is what keeps a colon earlier in
+# the title, with prose after it, from being mistaken for this shape.
+_TRAILING_LITERAL_NAME_RUN = re.compile(rf":\s+{_NAME_RUN}\s*$")
+
+
+def slugify_section_anchor(title: str) -> str:
+    """Return the HTML id Sphinx gives a section with this title.
+
+    Mirrors docutils' ``make_id``: lower-case, every run of non-alphanumeric
+    characters becomes a single hyphen, and leading/trailing hyphens are
+    dropped -- e.g. the section titled ``HookToolset`` is served at
+    ``#hooktoolset``.
+    """
+    return re.sub(r"[^a-z0-9]+", "-", title.lower()).strip("-")
+
+
+def _extract_names_from_title(title: str) -> list[str]:
+    """Return the names a section title documents, or [] if it names prose.
+
+    A guide marks a section as being *about* one or more names by titling it
+    with them as inline literals, either as the whole title (``HookToolset``, 
or
+    ``AgentOperator`` & ``@task.agent`` where one section covers the operator
+    and its decorator) or after a colon at the very end ("Airflow hooks as
+    tools: ``HookToolset``"). Requiring that markup is what keeps a single-word
+    prose heading ("Guidelines") -- or a literal appearing elsewhere in a prose
+    title -- from claiming to document a class of the same name, and it is a
+    convention the guides already follow rather than one imposed on them.
+    """
+    match = _LEADING_LITERAL_NAME_RUN.match(title) or 
_TRAILING_LITERAL_NAME_RUN.search(title)
+    return _INLINE_LITERAL_NAME_RE.findall(match.group(0)) if match else []
+
+
+def _is_adornment(line: str) -> bool:
+    """Whether a line is a reST title overline/underline rather than a 
title."""
+    return bool(line) and len(set(line)) == 1 and line[0] in _ADORNMENT_CHARS
+
+
+def _extract_section_titles(text: str) -> list[str]:
+    """Return every section title in a reST document, in document order."""
+    titles = []
+    lines = text.splitlines()
+    for index, line in enumerate(lines[:-1]):
+        title = line.strip()
+        # Guides title these sections with an inline literal (``HookToolset``),
+        # so a title can legitimately start with an adornment character; only a
+        # line that is *entirely* one repeated character is an adornment.
+        if not title or _is_adornment(title):
+            continue
+        underline = lines[index + 1].strip()
+        if len(underline) >= len(title) and _is_adornment(underline):
+            titles.append(title)
+    return titles
+
+
+def collect_guide_anchors(docs: Mapping[str, str]) -> dict[str, str]:
+    """Map name -> ``<page>.html#<anchor>`` for every documented class or 
decorator.
+
+    ``docs`` maps a page path relative to the provider's docs directory (e.g.
+    ``toolsets.rst``) to its reST source. A title can name more than one name
+    (``AgentOperator`` & ``@task.agent``), in which case every one gets the
+    same anchor. When two pages document the same name: a page's own title
+    (its first section) beats a subsection found on any other page, since that
+    page is the one dedicated to the class; among two page titles -- or two
+    subsections neither page titles -- the first page in sorted order wins, so
+    a rebuild of the same sources always produces the same link. "Page title"
+    is simply the first title _extract_section_titles finds, not a checked
+    top-level adornment, so a heading-shaped block earlier on the page (say,
+    inside a directive) would take that role.
+    """
+    found: dict[str, tuple[bool, str]] = {}  # name -> (from a page title?, 
anchor)
+    for page in sorted(docs):
+        page_url = re.sub(r"\.rst$", ".html", page)
+        for index, title in enumerate(_extract_section_titles(docs[page])):
+            is_page_title = index == 0
+            for name in _extract_names_from_title(title):
+                current = found.get(name)
+                # A page title replaces a subsection found earlier; nothing 
else
+                # replaces what was found first, so rebuilds stay 
deterministic.
+                if current is not None and (current[0] or not is_page_title):
+                    continue
+                found[name] = (is_page_title, 
f"{page_url}#{slugify_section_anchor(title)}")
+    return {name: anchor for name, (_, anchor) in found.items()}
+
+
+def attach_guide_urls(modules: list[dict[str, Any]], anchors: Mapping[str, 
str], base_docs_url: str) -> int:
+    """Set ``guide_url`` on every module a guide section documents.
+
+    Mutates ``modules`` in place; returns how many got a link.
+    """
+    attached = 0
+    for module in modules:
+        anchor = anchors.get(module["name"])
+        if not anchor:
+            continue
+        module["guide_url"] = f"{base_docs_url.rstrip('/')}/{anchor}"
+        attached += 1
+    return attached
diff --git a/dev/registry/tests/test_docs_guides.py 
b/dev/registry/tests/test_docs_guides.py
new file mode 100644
index 00000000000..53fcfe4367d
--- /dev/null
+++ b/dev/registry/tests/test_docs_guides.py
@@ -0,0 +1,369 @@
+# Licensed to the Apache Software Foundation (ASF) under one
+# or more contributor license agreements.  See the NOTICE file
+# distributed with this work for additional information
+# regarding copyright ownership.  The ASF licenses this file
+# to you under the Apache License, Version 2.0 (the
+# "License"); you may not use this file except in compliance
+# with the License.  You may obtain a copy of the License at
+#
+#   http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing,
+# software distributed under the License is distributed on an
+# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+# KIND, either express or implied.  See the License for the
+# specific language governing permissions and limitations
+# under the License.
+from __future__ import annotations
+
+from unittest.mock import patch
+
+import pytest
+from extract_parameters import read_guide_docs as read_guide_docs_from_worktree
+from extract_versions import read_guide_docs as read_guide_docs_from_tag
+from registry_tools.docs_guides import (
+    attach_guide_urls,
+    collect_guide_anchors,
+    is_guide_page,
+    slugify_section_anchor,
+)
+
+TOOLSETS_GUIDE = """
+.. _howto/toolsets:
+
+Toolsets: Airflow hooks as AI agent tools
+==========================================
+
+Intro prose.
+
+Airflow hooks as tools: ``HookToolset``
+----------------------------------------
+
+How to use it.
+
+Guidelines
+^^^^^^^^^^
+
+More prose.
+
+.. _bounded-query-results:
+
+Bounded query results
+^^^^^^^^^^^^^^^^^^^^^
+
+``SQLToolset`` bounds that.
+
+Files with DataFusion: ``DataFusionToolset``
+-----------------------------------------------
+
+Another one.
+"""
+
+
[email protected](
+    ("title", "expected"),
+    [
+        # Verified against the published guide: the section titled 
``HookToolset``
+        # is served at .../toolsets.html#hooktoolset.
+        ("HookToolset", "hooktoolset"),
+        ("AgentSkillsToolset", "agentskillstoolset"),
+        ("Bounded query results", "bounded-query-results"),
+        ("Agent_Skills", "agent-skills"),
+        ("Direct PydanticAI MCP toolsets", "direct-pydanticai-mcp-toolsets"),
+        ("``AgentOperator`` & ``@task.agent``", "agentoperator-task-agent"),
+    ],
+)
+def test_slugify_section_anchor_matches_sphinx_ids(title, expected):
+    assert slugify_section_anchor(title) == expected
+
+
+def test_collect_guide_anchors_finds_class_named_sections_at_any_depth():
+    anchors = collect_guide_anchors({"toolsets.rst": TOOLSETS_GUIDE})
+
+    assert anchors == {
+        "HookToolset": "toolsets.html#airflow-hooks-as-tools-hooktoolset",
+        "DataFusionToolset": 
"toolsets.html#files-with-datafusion-datafusiontoolset",
+    }
+
+
+def test_collect_guide_anchors_ignores_prose_headings():
+    anchors = collect_guide_anchors({"toolsets.rst": TOOLSETS_GUIDE})
+
+    # "Guidelines" is shaped like a class name but isn't marked up as one.
+    assert "Guidelines" not in anchors
+    assert "Bounded query results" not in anchors
+
+
+def test_collect_guide_anchors_ignores_classes_only_mentioned_in_prose():
+    # SQLToolset appears in the guide's body but has no section of its own, so
+    # there is no anchor to link to.
+    assert "SQLToolset" not in collect_guide_anchors({"toolsets.rst": 
TOOLSETS_GUIDE})
+
+
+def test_collect_guide_anchors_keeps_nested_page_paths():
+    guide = "Agents with tools: 
``AgentOperator``\n-------------------------------------\n\nProse.\n"
+
+    assert collect_guide_anchors({"operators/agent.rst": guide}) == {
+        "AgentOperator": "operators/agent.html#agents-with-tools-agentoperator"
+    }
+
+
+def 
test_collect_guide_anchors_ignores_a_title_that_opens_with_a_name_then_continues_in_prose():
+    guide = "``SandboxToolset`` parameters\n-----------------------------\n\nA 
parameter table.\n"
+
+    assert collect_guide_anchors({"sandbox/configuration.rst": guide}) == {}
+
+
+def test_collect_guide_anchors_names_only_the_trailing_run_of_a_colon_title():
+    guide = "``A``: ``B``\n-------------\n\nProse.\n"
+
+    assert collect_guide_anchors({"page.rst": guide}) == {"B": "page.html#a-b"}
+
+
+def test_collect_guide_anchors_prefers_first_sorted_page_among_page_titles():
+    # Both pages title themselves after SQLToolset, so neither title beats the
+    # other on that basis alone; the tie is broken by sorted page order.
+    guide = "SQL databases: 
``SQLToolset``\n------------------------------\n\nProse.\n"
+
+    anchors = collect_guide_anchors({"toolsets.rst": guide, 
"operators/sql.rst": guide})
+
+    assert anchors["SQLToolset"] == 
"operators/sql.html#sql-databases-sqltoolset"
+
+
+def test_collect_guide_anchors_handles_a_title_covering_more_than_the_class():
+    # Verified against the published guide: this heading is served at
+    # .../operators/agent.html#agentoperator-task-agent, so the anchor comes 
from
+    # the whole title while both the operator and its decorator get linked to 
it.
+    # This is the older leading shape; older release tags' docs (read by
+    # extract_versions.py) still use it, so it must keep working.
+    guide = "``AgentOperator`` & 
``@task.agent``\n===================================\n\nProse.\n"
+
+    assert collect_guide_anchors({"operators/agent.rst": guide}) == {
+        "AgentOperator": "operators/agent.html#agentoperator-task-agent",
+        "@task.agent": "operators/agent.html#agentoperator-task-agent",
+    }
+
+
+def test_collect_guide_anchors_links_a_decorator_name_with_underscores():
+    # ``@task.llm_file_analysis`` is the real decorator name for the common.ai
+    # provider's LLMFileAnalysisOperator; the leading-literal charset must 
admit
+    # "@" and "." for it to ever get a link. Also a leading-shape fixture kept 
for
+    # the same reason as the test above: older release tags' docs still use it.
+    guide = (
+        "``LLMFileAnalysisOperator`` & ``@task.llm_file_analysis``\n"
+        "==========================================================\n\n"
+        "Prose.\n"
+    )
+
+    assert collect_guide_anchors({"operators/llm_file_analysis.rst": guide}) 
== {
+        "LLMFileAnalysisOperator": (
+            
"operators/llm_file_analysis.html#llmfileanalysisoperator-task-llm-file-analysis"
+        ),
+        "@task.llm_file_analysis": (
+            
"operators/llm_file_analysis.html#llmfileanalysisoperator-task-llm-file-analysis"
+        ),
+    }
+
+
+def test_collect_guide_anchors_ignores_a_prose_title_mentioning_a_literal():
+    # The title neither opens with the literal nor ends a colon-led clause with
+    # it, so a prose heading that happens to mention one in passing -- anywhere
+    # in the title -- must not produce a link.
+    guide = "Using ``foo`` in a 
pipeline\n============================\n\nProse.\n"
+
+    assert collect_guide_anchors({"toolsets.rst": guide}) == {}
+
+
+def test_collect_guide_anchors_requires_a_long_enough_underline():
+    # An underline shorter than the title isn't a section in reST, so it must 
not
+    # produce a link to an anchor Sphinx never emitted.
+    guide = "Airflow hooks as tools: ``HookToolset``\n---\n\nProse.\n"
+
+    assert collect_guide_anchors({"toolsets.rst": guide}) == {}
+
+
+def test_collect_guide_anchors_reads_a_name_trailing_a_colon():
+    guide = "Airflow hooks as tools: 
``HookToolset``\n========================================\n\nProse.\n"
+
+    assert collect_guide_anchors({"toolsets/hook.rst": guide}) == {
+        "HookToolset": "toolsets/hook.html#airflow-hooks-as-tools-hooktoolset"
+    }
+
+
+def test_collect_guide_anchors_reads_two_names_joined_by_and():
+    guide = (
+        "Agents with tools: ``AgentOperator`` and ``@task.agent``\n"
+        "=========================================================\n\n"
+        "Prose.\n"
+    )
+
+    assert collect_guide_anchors({"operators/agent.rst": guide}) == {
+        "AgentOperator": 
"operators/agent.html#agents-with-tools-agentoperator-and-task-agent",
+        "@task.agent": 
"operators/agent.html#agents-with-tools-agentoperator-and-task-agent",
+    }
+
+
[email protected](
+    "title",
+    [
+        # Leading shape, one separator per case.
+        "``A`` & ``B``",
+        "``A``, ``B``",
+        "``A``/``B``",
+        "``A`` and ``B``",
+        # Trailing shape, one separator per case.
+        "Both: ``A`` & ``B``",
+        "Both: ``A``, ``B``",
+        "Both: ``A``/``B``",
+        "Both: ``A`` and ``B``",
+    ],
+)
+def 
test_collect_guide_anchors_accepts_every_separator_in_both_title_shapes(title):
+    underline = "=" * (len(title) + 1)
+
+    anchors = collect_guide_anchors({"page.rst": 
f"{title}\n{underline}\n\nProse.\n"})
+
+    expected_anchor = f"page.html#{slugify_section_anchor(title)}"
+    assert anchors == {"A": expected_anchor, "B": expected_anchor}
+
+
+def 
test_collect_guide_anchors_prefers_a_page_title_over_an_earlier_pages_subsection():
+    # A subsection on an unrelated page happens to be titled after the class,
+    # but a later page is dedicated to it.
+    docs = {
+        "agent_security.rst": (
+            "Securing agent tools\n"
+            "=====================\n\n"
+            "``HookToolset`` guidelines\n"
+            "^^^^^^^^^^^^^^^^^^^^^^^^^^^\n\n"
+            "Prose.\n"
+        ),
+        "toolsets/hook.rst": (
+            "Airflow hooks as tools: 
``HookToolset``\n========================================\n\nProse.\n"
+        ),
+    }
+
+    assert collect_guide_anchors(docs) == {
+        "HookToolset": "toolsets/hook.html#airflow-hooks-as-tools-hooktoolset"
+    }
+
+
+def 
test_collect_guide_anchors_prefers_the_page_title_over_its_own_subsection():
+    guide = (
+        "Batch processing: ``LLMBatchOperator``\n"
+        "=======================================\n\n"
+        "How it works.\n\n"
+        "``LLMBatchOperator`` or the vendor batch operators?\n"
+        "----------------------------------------------------\n\n"
+        "Prose.\n"
+    )
+
+    assert collect_guide_anchors({"operators/llm_batch.rst": guide}) == {
+        "LLMBatchOperator": 
"operators/llm_batch.html#batch-processing-llmbatchoperator"
+    }
+
+
+def 
test_collect_guide_anchors_keeps_the_first_subsection_when_no_page_title_names_it():
+    docs = {
+        "a.rst": "Prose page\n===========\n\n``X``\n-----\n",
+        "b.rst": "Other page\n===========\n\n``X``\n-----\n",
+    }
+
+    assert collect_guide_anchors(docs) == {"X": "a.html#x"}
+
+
[email protected](
+    "title",
+    [
+        # Colon present, but prose follows it before the literal -- must not be
+        # scanned for a literal anywhere after the colon.
+        "Role in a Dag: use ``MCPToolset``, not the hook directly",
+        # Title ends with the literal, but there's no colon to lead it.
+        "Using HITL review with ``AgentOperator``",
+        # Colon immediately precedes the literal, but prose follows it -- the
+        # literal doesn't reach the end of the title.
+        "Guidelines: ``HookToolset`` and its allow-list",
+    ],
+)
+def 
test_collect_guide_anchors_ignores_a_literal_that_does_not_end_the_title(title):
+    underline = "=" * (len(title) + 1)
+
+    assert collect_guide_anchors({"toolsets.rst": 
f"{title}\n{underline}\n\nProse.\n"}) == {}
+
+
+def test_attach_guide_urls_only_links_documented_classes():
+    modules = [
+        {"name": "HookToolset", "docs_url": 
"https://example.test/_api/hook/index.html"},
+        {"name": "UndocumentedToolset", "docs_url": 
"https://example.test/_api/other/index.html"},
+    ]
+
+    attached = attach_guide_urls(
+        modules,
+        {"HookToolset": "toolsets.html#hooktoolset"},
+        
"https://airflow.apache.org/docs/apache-airflow-providers-common-ai/0.7.0";,
+    )
+
+    assert attached == 1
+    assert modules[0]["guide_url"] == (
+        
"https://airflow.apache.org/docs/apache-airflow-providers-common-ai/0.7.0/toolsets.html#hooktoolset";
+    )
+    assert "guide_url" not in modules[1]
+
+
+def test_attach_guide_urls_does_not_double_up_the_base_separator():
+    modules = [{"name": "HookToolset"}]
+
+    attach_guide_urls(modules, {"HookToolset": "toolsets.html#hooktoolset"}, 
"https://example.test/docs/";)
+
+    assert modules[0]["guide_url"] == 
"https://example.test/docs/toolsets.html#hooktoolset";
+
+
[email protected](
+    ("relative_path", "expected"),
+    [
+        # A `_`-prefixed path segment marks autoapi/partial content, at any 
depth.
+        ("_api/index.rst", False),
+        ("_api/hook/index.rst", False),
+        ("operators/_partials/foo.rst", False),
+        ("_partials/foo.rst", False),
+        # Real, built release-note pages, not how-to guides.
+        ("changelog.rst", False),
+        ("commits.rst", False),
+        ("toolsets.rst", True),
+        ("operators/agent.rst", True),
+    ],
+)
+def test_is_guide_page(relative_path, expected):
+    assert is_guide_page(relative_path) == expected
+
+
+def test_readers_agree_on_which_pages_are_guides(tmp_path):
+    """Both `read_guide_docs` implementations delegate to `is_guide_page`, so a
+    working-tree read and a git-tag read of the same paths must end up with the
+    same set of pages -- regardless of which source produced them."""
+    relative_paths = ["_api/index.rst", "changelog.rst", "commits.rst", 
"toolsets.rst"]
+    for relative in relative_paths:
+        target = tmp_path / relative
+        target.parent.mkdir(parents=True, exist_ok=True)
+        target.write_text("Prose.\n")
+
+    from_worktree = read_guide_docs_from_worktree(tmp_path)
+
+    docs_prefix = "providers/test/docs/"
+    with (
+        patch(
+            "extract_versions.git_ls_tree",
+            autospec=True,
+            return_value=[docs_prefix + relative for relative in 
relative_paths],
+        ),
+        patch(
+            "extract_versions.git_cat_file_batch",
+            autospec=True,
+            side_effect=lambda tag, paths: {p: "Prose.\n" for p in paths},
+        ),
+    ):
+        from_tag = read_guide_docs_from_tag("providers-test/1.0.0", "new", 
"test")
+
+    # Both readers must apply the same filter; each reader's own test can pass
+    # while the two drift apart.
+    assert set(from_worktree) == set(from_tag) == {"toolsets.rst"}
diff --git a/dev/registry/tests/test_extract_parameters.py 
b/dev/registry/tests/test_extract_parameters.py
index eac805cc49a..b751ae64f08 100644
--- a/dev/registry/tests/test_extract_parameters.py
+++ b/dev/registry/tests/test_extract_parameters.py
@@ -22,7 +22,7 @@ import abc
 import builtins
 import json
 import types
-from dataclasses import fields
+from dataclasses import MISSING, fields
 from unittest.mock import patch
 
 import pytest
@@ -38,6 +38,7 @@ from extract_parameters import (
     get_category,
     is_durable_capable,
     load_resumable_job_mixin,
+    read_guide_docs,
     supports_deferrable,
 )
 
@@ -911,6 +912,20 @@ FAKE_PROVIDER_YAML = {
 }
 
 
+# ---------------------------------------------------------------------------
+# read_guide_docs
+# ---------------------------------------------------------------------------
+def test_read_guide_docs_skips_generated_and_release_note_pages(tmp_path):
+    (tmp_path / "_api" / "x").mkdir(parents=True)
+    (tmp_path / "_api" / "x" / "index.rst").write_text("Generated.\n")
+    (tmp_path / "changelog.rst").write_text("Release notes.\n")
+    (tmp_path / 
"toolsets.rst").write_text("``HookToolset``\n---------------\n\nProse.\n")
+
+    result = read_guide_docs(tmp_path)
+
+    assert set(result) == {"toolsets.rst"}
+
+
 # ---------------------------------------------------------------------------
 # TestDiscoverClassesFromProvider
 # ---------------------------------------------------------------------------
@@ -999,6 +1014,88 @@ class TestDiscoverClassesFromProvider:
         assert operators[0]["import_path"] == 
"airflow.providers.amazon.aws.operators.s3.FakeOperator"
         assert operators[0]["provider_id"] == "amazon"
 
+    def test_guide_section_becomes_a_guide_url(self, provider_yaml_path, 
base_classes):
+        """A class documented by a section of its own gets a link to that 
section;
+        one that is only in the API reference keeps just its ``docs_url``."""
+        docs_dir = provider_yaml_path.parent / "docs" / "operators"
+        docs_dir.mkdir(parents=True)
+        (docs_dir / 
"s3.rst").write_text("``FakeOperator``\n----------------\n\nProse.\n")
+
+        with (
+            patch("extract_parameters.PROVIDERS_DIR", 
provider_yaml_path.parent.parent),
+            patch("extract_parameters.importlib.import_module", 
side_effect=self._mock_import),
+        ):
+            result = discover_classes_from_provider(provider_yaml_path, 
base_classes)
+
+        by_name = {r["name"]: r for r in result}
+        assert by_name["FakeOperator"]["guide_url"] == (
+            
"https://airflow.apache.org/docs/apache-airflow-providers-amazon/stable";
+            "/operators/s3.html#fakeoperator"
+        )
+        assert "guide_url" not in by_name["FakeSensor"]
+
+    @pytest.mark.parametrize(
+        ("tag_exists", "expected_anchor"),
+        [
+            pytest.param(True, "released-title-fakeoperator", 
id="tag-exists-reads-released-docs"),
+            pytest.param(False, "fakeoperator", 
id="no-tag-reads-working-tree"),
+        ],
+    )
+    def test_guide_docs_come_from_release_tag_when_it_exists(
+        self, provider_yaml_path, base_classes, tag_exists, expected_anchor
+    ):
+        docs_dir = provider_yaml_path.parent / "docs" / "operators"
+        docs_dir.mkdir(parents=True)
+        (docs_dir / 
"s3.rst").write_text("``FakeOperator``\n----------------\n\nUnreleased 
title.\n")
+        released = {
+            "operators/s3.rst": "Released title: 
``FakeOperator``\n--------------------------------\n"
+        }
+
+        with (
+            patch("extract_parameters.PROVIDERS_DIR", 
provider_yaml_path.parent.parent),
+            patch("extract_parameters.git_tag_exists", 
return_value=tag_exists) as tag_check,
+            patch("extract_parameters.detect_layout", return_value="new"),
+            patch("extract_parameters.read_guide_docs_at_tag", 
return_value=released) as read_at_tag,
+            patch("extract_parameters.importlib.import_module", 
side_effect=self._mock_import),
+        ):
+            result = discover_classes_from_provider(provider_yaml_path, 
base_classes, version="1.2.3")
+
+        tag_check.assert_called_once_with("providers-amazon/1.2.3")
+        if tag_exists:
+            read_at_tag.assert_called_once_with("providers-amazon/1.2.3", 
"new", "amazon")
+        else:
+            read_at_tag.assert_not_called()
+        guide_url = {r["name"]: r for r in result}["FakeOperator"]["guide_url"]
+        assert guide_url.endswith(f"/operators/s3.html#{expected_anchor}")
+
+    @pytest.mark.parametrize(
+        ("version", "layout", "expect_tag_lookup"),
+        [
+            pytest.param("", "new", False, id="no-version-skips-tag-lookup"),
+            pytest.param("1.2.3", None, True, 
id="undetectable-layout-falls-back"),
+        ],
+    )
+    def test_guide_docs_fall_back_to_working_tree(
+        self, provider_yaml_path, base_classes, version, layout, 
expect_tag_lookup
+    ):
+        docs_dir = provider_yaml_path.parent / "docs" / "operators"
+        docs_dir.mkdir(parents=True)
+        (docs_dir / 
"s3.rst").write_text("``FakeOperator``\n----------------\n\nProse.\n")
+
+        with (
+            patch("extract_parameters.PROVIDERS_DIR", 
provider_yaml_path.parent.parent),
+            patch("extract_parameters.git_tag_exists", return_value=True) as 
tag_check,
+            patch("extract_parameters.detect_layout", return_value=layout),
+            patch("extract_parameters.read_guide_docs_at_tag") as read_at_tag,
+            patch("extract_parameters.importlib.import_module", 
side_effect=self._mock_import),
+        ):
+            result = discover_classes_from_provider(provider_yaml_path, 
base_classes, version=version)
+
+        assert tag_check.called is expect_tag_lookup
+        read_at_tag.assert_not_called()
+        guide_url = {r["name"]: r for r in result}["FakeOperator"]["guide_url"]
+        assert guide_url.endswith("/operators/s3.html#fakeoperator")
+
     def test_discovers_sensor(self, provider_yaml_path, base_classes):
         with (
             patch("extract_parameters.PROVIDERS_DIR", 
provider_yaml_path.parent.parent),
@@ -1093,14 +1190,18 @@ class TestDiscoverClassesFromProvider:
         assert operators[0]["short_description"] == "Copy objects in S3."
 
     def test_all_module_fields_present(self, provider_yaml_path, base_classes):
-        """Every discovered entry has every `Module` dataclass field (derived, 
not hardcoded)."""
+        """Every discovered entry has every required `Module` dataclass field 
(derived, not hardcoded).
+
+        Fields with a default (e.g. ``guide_url``) are attached separately and 
only
+        when applicable, so they are allowed to be absent here.
+        """
         with (
             patch("extract_parameters.PROVIDERS_DIR", 
provider_yaml_path.parent.parent),
             patch("extract_parameters.importlib.import_module", 
side_effect=self._mock_import),
         ):
             result = discover_classes_from_provider(provider_yaml_path, 
base_classes)
 
-        required_fields = {f.name for f in fields(Module)}
+        required_fields = {f.name for f in fields(Module) if f.default is 
MISSING}
         for entry in result:
             missing = required_fields - entry.keys()
             assert not missing, f"Missing fields {missing} in {entry['name']}"
diff --git a/dev/registry/tests/test_extract_versions.py 
b/dev/registry/tests/test_extract_versions.py
index 6a3128ce01a..7ecc7d94dff 100644
--- a/dev/registry/tests/test_extract_versions.py
+++ b/dev/registry/tests/test_extract_versions.py
@@ -18,8 +18,9 @@
 
 from __future__ import annotations
 
+import subprocess
 import textwrap
-from unittest.mock import patch
+from unittest.mock import MagicMock, call, patch
 
 import pytest
 from extract_versions import (
@@ -28,6 +29,9 @@ from extract_versions import (
     SCRIPT_DIR,
     extract_modules_from_yaml,
     extract_version_data,
+    git_cat_file_batch,
+    git_ls_tree,
+    read_guide_docs,
 )
 from registry_tools.types import CLASS_LEVEL_SECTIONS, 
DICT_SHAPED_CLASS_LEVEL_SECTIONS
 
@@ -84,7 +88,8 @@ EXPECTED_CLASS_LEVEL_DESC_SUFFIXES = {
 }
 
 
-def _extract_class_level_modules(provider_yaml: dict) -> list[dict]:
+@patch("extract_versions.read_guide_docs", autospec=True, return_value={})
+def _extract_class_level_modules(provider_yaml: dict, _mock_read_guide_docs) 
-> list[dict]:
     return extract_modules_from_yaml(
         provider_yaml,
         tag="providers-test/1.0.0",
@@ -266,3 +271,189 @@ class TestExtractVersionDataUriSchemes:
             {"scheme": "tfs", "filesystem": 
"airflow.providers.test.fs.testfs"},
         ]
         mock_git_show.assert_any_call("providers-test/1.0.0", fs_source_path)
+
+
+class TestExtractModulesGuideUrls:
+    """A class the provider's guides document in a section of its own must get 
a
+    ``guide_url`` for every version, not just the latest. A superseded 
version's
+    page is rendered only from the per-version file this module writes, so a 
link
+    resolved on the latest path alone disappears the moment a new version 
lands.
+    """
+
+    PROVIDER_YAML = {
+        "toolsets": [
+            {
+                "integration-name": "Test",
+                "python-modules": ["airflow.providers.test.toolsets.hook"],
+            }
+        ]
+    }
+    SOURCE = 'class HookToolset:\n    """A toolset."""\n'
+    GUIDE = "``HookToolset``\n---------------\n\nProse.\n"
+
+    def _extract(self, layout="new", 
docs_paths=("providers/test/docs/toolsets.rst",)):
+        def fake_git_show(_tag, path):
+            return self.SOURCE if path.endswith(".py") else None
+
+        def fake_git_cat_file_batch(_tag, paths):
+            return {p: self.GUIDE for p in paths}
+
+        with (
+            patch("extract_versions.git_ls_tree", autospec=True, 
return_value=list(docs_paths)),
+            patch("extract_versions.git_show", autospec=True, 
side_effect=fake_git_show),
+            patch("extract_versions.git_cat_file_batch", autospec=True, 
side_effect=fake_git_cat_file_batch),
+        ):
+            return extract_modules_from_yaml(
+                self.PROVIDER_YAML, "providers-test/1.0.0", layout, "test", 
"test", "1.0.0"
+            )
+
+    def test_documented_class_gets_a_versioned_guide_url(self):
+        modules = self._extract()
+
+        assert [m["name"] for m in modules] == ["HookToolset"]
+        assert modules[0]["guide_url"] == (
+            
"https://airflow.apache.org/docs/apache-airflow-providers-test/1.0.0/toolsets.html#hooktoolset";
+        )
+
+    def test_undocumented_class_gets_no_guide_url(self):
+        modules = self._extract(docs_paths=())
+
+        assert [m["name"] for m in modules] == ["HookToolset"]
+        assert "guide_url" not in modules[0]
+
+    def test_old_layout_gets_no_guide_url(self):
+        # Pre-per-provider tags kept docs in a top-level tree, so there is no
+        # provider-relative page path to build a link from.
+        modules = self._extract(layout="old")
+
+        assert "guide_url" not in modules[0]
+
+
+class TestReadGuideDocs:
+    def 
test_skips_generated_and_release_note_pages_before_calling_git_show(self):
+        docs_prefix = "providers/test/docs/"
+        paths = [
+            docs_prefix + "_api/x/index.rst",
+            docs_prefix + "changelog.rst",
+            docs_prefix + "diagram.png",
+            docs_prefix + "conf.py",
+            docs_prefix + "toolsets.rst",
+        ]
+
+        with (
+            patch("extract_versions.git_ls_tree", autospec=True, 
return_value=paths),
+            patch(
+                "extract_versions.git_cat_file_batch",
+                autospec=True,
+                side_effect=lambda tag, paths: {p: "Prose.\n" for p in paths},
+            ) as mock_git_cat_file_batch,
+        ):
+            result = read_guide_docs("providers-test/1.0.0", "new", "test")
+
+        assert set(result) == {"toolsets.rst"}
+        # Filtering must happen before the batch call, not just before the 
dict write.
+        assert mock_git_cat_file_batch.call_args_list == [
+            call("providers-test/1.0.0", [docs_prefix + "toolsets.rst"])
+        ]
+
+    def test_skips_a_page_whose_content_is_an_empty_string(self):
+        docs_prefix = "providers/test/docs/"
+        paths = [docs_prefix + "empty.rst"]
+
+        with (
+            patch("extract_versions.git_ls_tree", autospec=True, 
return_value=paths),
+            patch(
+                "extract_versions.git_cat_file_batch",
+                autospec=True,
+                return_value={docs_prefix + "empty.rst": ""},
+            ),
+        ):
+            result = read_guide_docs("providers-test/1.0.0", "new", "test")
+
+        assert result == {}
+
+
+class TestGitLsTree:
+    def test_passes_quote_path_false_to_git(self):
+        mock_result = MagicMock(spec=subprocess.CompletedProcess)
+        mock_result.stdout = b"providers/test/docs/toolsets.rst\n"
+        with patch("extract_versions.subprocess.run", autospec=True, 
return_value=mock_result) as mock_run:
+            git_ls_tree("providers-test/1.0.0", "providers/test/docs/")
+
+        assert mock_run.call_args.args[0] == [
+            "git",
+            "-c",
+            "core.quotePath=false",
+            "ls-tree",
+            "-r",
+            "--name-only",
+            "providers-test/1.0.0",
+            "--",
+            "providers/test/docs/",
+        ]
+
+    def test_decodes_stdout_as_utf8(self):
+        mock_result = MagicMock(spec=subprocess.CompletedProcess)
+        mock_result.stdout = "docs/café.rst\ndocs/b.rst\n".encode()
+        with patch("extract_versions.subprocess.run", autospec=True, 
return_value=mock_result):
+            result = git_ls_tree("providers-test/1.0.0", 
"providers/test/docs/")
+
+        assert result == ["docs/café.rst", "docs/b.rst"]
+
+
+def _batch_hit(sha1: str, obj_type: str, content: bytes) -> bytes:
+    return f"{sha1} {obj_type} {len(content)}\n".encode() + content + b"\n"
+
+
+def _batch_missing(spec: str) -> bytes:
+    return f"{spec} missing\n".encode()
+
+
+class TestGitCatFileBatch:
+    def test_empty_paths_returns_empty_dict_without_subprocess(self):
+        with patch("extract_versions.subprocess.run", autospec=True) as 
mock_run:
+            result = git_cat_file_batch("providers-test/1.0.0", [])
+
+        assert result == {}
+        mock_run.assert_not_called()
+
+    def test_two_hits_with_different_sizes_and_multibyte_content(self):
+        tag = "providers-test/1.0.0"
+        paths = ["providers/test/docs/a.rst", "providers/test/docs/b.rst"]
+        first_content = b"short\n"
+        second_content = "café prôse with more text\n".encode()
+        payload = _batch_hit("aaa1", "blob", first_content) + 
_batch_hit("bbb2", "blob", second_content)
+
+        mock_result = MagicMock(spec=subprocess.CompletedProcess)
+        mock_result.stdout = payload
+        with patch("extract_versions.subprocess.run", autospec=True, 
return_value=mock_result):
+            result = git_cat_file_batch(tag, paths)
+
+        assert result == {
+            paths[0]: "short\n",
+            paths[1]: "café prôse with more text\n",
+        }
+
+    def test_hit_followed_by_missing_path(self):
+        tag = "providers-test/1.0.0"
+        paths = ["providers/test/docs/a.rst", 
"providers/test/docs/missing.rst"]
+        payload = _batch_hit("aaa1", "blob", b"content\n") + 
_batch_missing(f"{tag}:{paths[1]}")
+
+        mock_result = MagicMock(spec=subprocess.CompletedProcess)
+        mock_result.stdout = payload
+        with patch("extract_versions.subprocess.run", autospec=True, 
return_value=mock_result):
+            result = git_cat_file_batch(tag, paths)
+
+        assert result == {paths[0]: "content\n"}
+
+    def test_all_missing_returns_empty_dict(self):
+        tag = "providers-test/1.0.0"
+        paths = ["providers/test/docs/a.rst", "providers/test/docs/b.rst"]
+        payload = _batch_missing(f"{tag}:{paths[0]}") + 
_batch_missing(f"{tag}:{paths[1]}")
+
+        mock_result = MagicMock(spec=subprocess.CompletedProcess)
+        mock_result.stdout = payload
+        with patch("extract_versions.subprocess.run", autospec=True, 
return_value=mock_result):
+            result = git_cat_file_batch(tag, paths)
+
+        assert result == {}
diff --git a/dev/registry/tests/test_registry_contract_models.py 
b/dev/registry/tests/test_registry_contract_models.py
index 4e477682f66..0e29b7aa2c4 100644
--- a/dev/registry/tests/test_registry_contract_models.py
+++ b/dev/registry/tests/test_registry_contract_models.py
@@ -114,6 +114,15 @@ def 
test_module_contract_preserves_supports_deferrable_true():
     assert validated["modules"][0]["supports_deferrable"] is True
 
 
+def test_module_contract_omits_guide_url_for_undocumented_classes():
+    assert ModuleContract.model_validate(_module_payload()).guide_url is None
+
+
+def test_module_contract_preserves_guide_url_value():
+    guide_url = "https://example.invalid/docs/toolsets.html#exampletoolset";
+    assert 
ModuleContract.model_validate(_module_payload(guide_url=guide_url)).guide_url 
== guide_url
+
+
 def test_connection_type_contract_defaults_external_services_to_empty_list():
     """Legacy connection-types entries (provider.yaml without 
`external-services`)
     must still validate, with the field defaulting to an empty list."""
diff --git a/providers/common/ai/docs/operators/llm_branch.rst 
b/providers/common/ai/docs/operators/llm_branch.rst
index 35b15f0f2b9..0e113aed050 100644
--- a/providers/common/ai/docs/operators/llm_branch.rst
+++ b/providers/common/ai/docs/operators/llm_branch.rst
@@ -17,8 +17,8 @@
 
 .. _howto/operator:llm_branch:
 
-Branch on an answer: ``LLMBranchOperator``
-==========================================
+Branch on an answer: ``LLMBranchOperator`` and ``@task.llm_branch``
+===================================================================
 
 Use 
:class:`~airflow.providers.common.ai.operators.llm_branch.LLMBranchOperator`
 for LLM-driven branching, where the LLM decides which downstream task(s) to
diff --git a/providers/common/ai/docs/operators/llm_file_analysis.rst 
b/providers/common/ai/docs/operators/llm_file_analysis.rst
index 4fb0931410d..eed8c866281 100644
--- a/providers/common/ai/docs/operators/llm_file_analysis.rst
+++ b/providers/common/ai/docs/operators/llm_file_analysis.rst
@@ -17,8 +17,8 @@
 
 .. _howto/operator:llm_file_analysis:
 
-Analyze files and images: ``LLMFileAnalysisOperator``
-=====================================================
+Analyze files and images: ``LLMFileAnalysisOperator`` and 
``@task.llm_file_analysis``
+=====================================================================================
 
 .. note::
 
diff --git a/providers/common/ai/docs/operators/llm_schema_compare.rst 
b/providers/common/ai/docs/operators/llm_schema_compare.rst
index cafde3900de..47b06cf2225 100644
--- a/providers/common/ai/docs/operators/llm_schema_compare.rst
+++ b/providers/common/ai/docs/operators/llm_schema_compare.rst
@@ -17,8 +17,8 @@
 
 .. _howto/operator:llm_schema_compare:
 
-Detect schema drift: ``LLMSchemaCompareOperator``
-=================================================
+Detect schema drift: ``LLMSchemaCompareOperator`` and 
``@task.llm_schema_compare``
+==================================================================================
 
 .. note::
 
diff --git a/providers/common/ai/docs/operators/llm_sql.rst 
b/providers/common/ai/docs/operators/llm_sql.rst
index 5d06ec97e39..42e51de7955 100644
--- a/providers/common/ai/docs/operators/llm_sql.rst
+++ b/providers/common/ai/docs/operators/llm_sql.rst
@@ -17,8 +17,8 @@
 
 .. _howto/operator:llm_sql_query:
 
-Natural language to SQL: ``LLMSQLQueryOperator``
-================================================
+Natural language to SQL: ``LLMSQLQueryOperator`` and ``@task.llm_sql``
+======================================================================
 
 .. note::
 
diff --git a/registry/AGENTS.md b/registry/AGENTS.md
index 99416cb8931..b82377a2569 100644
--- a/registry/AGENTS.md
+++ b/registry/AGENTS.md
@@ -462,6 +462,39 @@ They run inside Breeze where all providers are installed. 
`extract_metadata.py`
 the CI workflow can run the fast scripts (metadata, ~30s per provider) without 
spinning
 up Breeze, while parameter/connection extraction is a separate step.
 
+### How a module gets a "Guide" link
+
+A module card links to the how-to guide section that documents it, alongside 
the
+generated API reference. `provider.yaml`'s `how-to-guide` fields are 
CI-enforced
+by `check_doc_files`, but they name a whole page, never a section, and cover 
only
+operators, sensors and transfers — not the toolset, hook and decorator pages 
this
+needs. So `registry_tools/docs_guides.py` reads the provider's own `docs/*.rst`
+and matches a class to a section when the section's title names it as an inline
+literal, either at the start — ``` ``MCPHook`` ``` or after a colon at the very
+end — ``` Airflow hooks as tools: ``HookToolset`` ``` or ``` Agents with tools:
+``AgentOperator`` and ``@task.agent`` ```. A literal elsewhere in a prose title
+does not count. The anchor is derived from the whole title the way docutils
+derives its HTML id. When a name is titled in more than one place, a page's own
+title wins over a subsection on any other page, so the link lands on the page
+dedicated to the class rather than on a passing section about it.
+
+That convention is what the guides already do, and it is deliberately the only
+signal this resolves a section from: a hand-maintained class-to-guide table
+would keep pointing at sections that have since been renamed or split, and a
+link that lands on the wrong section is worse than no link. A class documented
+only in prose gets no Guide link.
+
+Growing the set of modules that get a Guide link means titling that provider's
+sections in one of those two shapes, not touching this extractor. `common/ai`
+titles its dedicated operator, hook and toolset pages this way; a couple of 
other
+providers use the same shapes for a config option name or a single decorator
+rather than a class. Having the right title doesn't guarantee a link — that 
still
+depends on a same-named module existing in the catalog.
+
+Both extraction paths resolve it — `extract_parameters.py` from the working 
tree
+for the latest release, `extract_versions.py` from the git tag for superseded 
ones
+— because a superseded version's page is rendered only from its own metadata 
file.
+
 ### Relationship to `run_provider_yaml_files_check.py`
 
 `scripts/in_container/run_provider_yaml_files_check.py` (run by the
diff --git a/registry/src/css/main.css b/registry/src/css/main.css
index d5077f8c888..42b6d1a618e 100644
--- a/registry/src/css/main.css
+++ b/registry/src/css/main.css
@@ -3552,12 +3552,14 @@ main {
 /* Module Actions (View Docs, Source) */
 .provider-detail-page .module-actions {
   display: flex;
+  flex-wrap: wrap;
   align-items: center;
   gap: var(--space-3);
   margin-top: var(--space-3);
 }
 
 .provider-detail-page .module-actions .docs-link,
+.provider-detail-page .module-actions .guide-link,
 .provider-detail-page .module-actions .source-link {
   display: inline-flex;
   align-items: center;
@@ -3575,6 +3577,14 @@ main {
   color: var(--color-cyan-300);
 }
 
+.provider-detail-page .module-actions .guide-link {
+  color: var(--accent-secondary);
+}
+
+.provider-detail-page .module-actions .guide-link:hover {
+  color: var(--color-cyan-300);
+}
+
 .provider-detail-page .module-actions .source-link {
   color: var(--text-muted);
 }
diff --git a/registry/src/provider-version.njk 
b/registry/src/provider-version.njk
index e28be209f1a..7d0c16b9c49 100644
--- a/registry/src/provider-version.njk
+++ b/registry/src/provider-version.njk
@@ -475,6 +475,12 @@ eleventyComputed:
                 <svg fill="none" stroke="currentColor" viewBox="0 0 24 24" 
aria-hidden="true"><path stroke-linecap="round" stroke-linejoin="round" 
stroke-width="2" d="M10 6H6a2 2 0 00-2 2v10a2 2 0 002 2h10a2 2 0 002-2v-4M14 
4h6m0 0v6m0-6L10 14" /></svg>
               </a>
               {% endif %}
+              {% if module.guide_url %}
+              <a href="{{ module.guide_url }}" target="_blank" rel="noopener" 
class="guide-link" title="How-to guide section for {{ module.name }}">
+                Guide
+                <svg fill="none" stroke="currentColor" viewBox="0 0 24 24" 
aria-hidden="true"><path stroke-linecap="round" stroke-linejoin="round" 
stroke-width="2" d="M10 6H6a2 2 0 00-2 2v10a2 2 0 002 2h10a2 2 0 002-2v-4M14 
4h6m0 0v6m0-6L10 14" /></svg>
+              </a>
+              {% endif %}
               {% if module.source_url %}
               <a href="{{ module.source_url }}" target="_blank" rel="noopener" 
class="source-link">
                 Source

Reply via email to