This is an automated email from the ASF dual-hosted git repository.
Lee-W pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/airflow.git
The following commit(s) were added to refs/heads/main by this push:
new 07bed4fde80 Link registry modules to the guide section that documents
them (#71477)
07bed4fde80 is described below
commit 07bed4fde809386b40b37941008253230af1c65b
Author: Wei Lee <[email protected]>
AuthorDate: Wed Oct 7 14:28:48 2026 +0900
Link registry modules to the guide section that documents them (#71477)
---
dev/registry/extract_parameters.py | 46 ++-
dev/registry/extract_versions.py | 85 +++++
dev/registry/registry_contract_models.py | 3 +
dev/registry/registry_tools/docs_guides.py | 186 +++++++++++
dev/registry/tests/test_docs_guides.py | 369 +++++++++++++++++++++
dev/registry/tests/test_extract_parameters.py | 107 +++++-
dev/registry/tests/test_extract_versions.py | 195 ++++++++++-
.../tests/test_registry_contract_models.py | 9 +
providers/common/ai/docs/operators/llm_branch.rst | 4 +-
.../common/ai/docs/operators/llm_file_analysis.rst | 4 +-
.../ai/docs/operators/llm_schema_compare.rst | 4 +-
providers/common/ai/docs/operators/llm_sql.rst | 4 +-
registry/AGENTS.md | 33 ++
registry/src/css/main.css | 10 +
registry/src/provider-version.njk | 6 +
15 files changed, 1051 insertions(+), 14 deletions(-)
diff --git a/dev/registry/extract_parameters.py
b/dev/registry/extract_parameters.py
index 6e09084bc04..df9f344ab00 100644
--- a/dev/registry/extract_parameters.py
+++ b/dev/registry/extract_parameters.py
@@ -55,7 +55,9 @@ from pathlib import Path
import yaml
from extract_metadata import fetch_provider_inventory, read_inventory
+from extract_versions import detect_layout, git_tag_exists, read_guide_docs as
read_guide_docs_at_tag
from registry_contract_models import validate_modules_catalog,
validate_provider_parameters
+from registry_tools.docs_guides import attach_guide_urls,
collect_guide_anchors, is_guide_page
from registry_tools.types import (
BASE_CLASS_IMPORTS,
CLASS_LEVEL_CATEGORY_OVERRIDES,
@@ -98,6 +100,7 @@ class Module:
provider_name: str
supports_durable_execution: bool
supports_deferrable: bool
+ guide_url: str | None = None
def get_category(integration_name: str) -> str:
@@ -738,6 +741,41 @@ def _resolve_decorated_operator_class(decorator_fn:
object) -> type | None:
return candidate if inspect.isclass(candidate) else None
+def read_guide_docs(docs_dir: Path) -> dict[str, str]:
+ """Read a provider's authored reST docs from the working tree, keyed by
path relative to ``docs_dir``."""
+ if not docs_dir.is_dir():
+ return {}
+ docs = {}
+ for path in sorted(docs_dir.rglob("*.rst")):
+ relative = path.relative_to(docs_dir).as_posix()
+ if not is_guide_page(relative):
+ continue
+ docs[relative] = path.read_text(encoding="utf-8")
+ return docs
+
+
+def read_released_guide_docs(
+ provider_id: str, version: str, provider_rel_path: Path
+) -> dict[str, str] | None:
+ """Read a provider's guide docs at its release tag, or None when there is
no tag to read.
+
+ The guide links point at ``/stable``, which serves the released docs, so
the
+ anchors must come from the same content; the working tree may be ahead of
it.
+ A tag from the old flat layout yields an empty dict, meaning no guide links
+ rather than a fallback to the working tree.
+ """
+ if not version:
+ return None
+ tag = f"providers-{provider_id}/{version}"
+ if not git_tag_exists(tag):
+ return None
+ dir_path = provider_rel_path.as_posix()
+ layout = detect_layout(tag, dir_path)
+ if layout is None:
+ return None
+ return read_guide_docs_at_tag(tag, layout, dir_path)
+
+
def discover_classes_from_provider(
provider_yaml_path: Path,
base_classes: dict[str, type],
@@ -748,7 +786,8 @@ def discover_classes_from_provider(
"""Discover classes from a single provider by importing its modules at
runtime.
Reads the provider.yaml to find which modules/classes to inspect, imports
them,
- and returns metadata for each discovered class with every `Module`
dataclass field.
+ and returns metadata for each discovered class with every required `Module`
+ dataclass field, plus ``guide_url`` when a how-to guide documents the
class.
"""
with open(provider_yaml_path) as f:
provider_yaml = yaml.safe_load(f)
@@ -985,6 +1024,11 @@ def discover_classes_from_provider(
}
)
+ guide_docs = read_released_guide_docs(provider_id, version,
provider_rel_path)
+ if guide_docs is None:
+ guide_docs = read_guide_docs(provider_yaml_path.parent / "docs")
+ attach_guide_urls(discovered, collect_guide_anchors(guide_docs),
base_docs_url)
+
return discovered
diff --git a/dev/registry/extract_versions.py b/dev/registry/extract_versions.py
index d88530b1a10..d7f86251aa9 100644
--- a/dev/registry/extract_versions.py
+++ b/dev/registry/extract_versions.py
@@ -57,6 +57,7 @@ except ImportError:
sys.exit(1)
from extract_metadata import fetch_provider_inventory, read_connection_urls,
resolve_connection_docs_url
+from registry_tools.docs_guides import attach_guide_urls,
collect_guide_anchors, is_guide_page
from registry_tools.types import (
CLASS_LEVEL_CATEGORY_OVERRIDES,
CLASS_LEVEL_SECTIONS,
@@ -127,6 +128,62 @@ def git_show(tag: str, path: str) -> str | None:
return None
+def git_ls_tree(tag: str, prefix: str) -> list[str]:
+ """List the file paths under a prefix at a specific git tag."""
+ try:
+ result = subprocess.run(
+ ["git", "-c", "core.quotePath=false", "ls-tree", "-r",
"--name-only", tag, "--", prefix],
+ capture_output=True,
+ cwd=AIRFLOW_ROOT,
+ check=True,
+ )
+ except subprocess.CalledProcessError:
+ return []
+ return [line for line in result.stdout.decode("utf-8").splitlines() if
line]
+
+
+def git_cat_file_batch(tag: str, paths: list[str]) -> dict[str, str]:
+ """Read multiple files at a specific git tag in one `git cat-file --batch`
call.
+
+ Returns a mapping of path -> content for paths that exist at the tag; a
path
+ git reports as missing is simply absent from the result, matching
git_show's
+ "return None for a missing path" semantics.
+
+ Decode failures are left unguarded on purpose: .rst files are Sphinx
+ convention UTF-8, an explicit "utf-8" decode is more predictable than
+ following the process locale, and a UnicodeDecodeError should surface
loudly
+ rather than being swallowed. A failing ``git cat-file`` call also raises
+ (``check=True``); only git_show turns CalledProcessError into ``None``.
+ """
+ if not paths:
+ return {}
+
+ stdin = ("\n".join(f"{tag}:{p}" for p in paths) + "\n").encode("utf-8")
+ result = subprocess.run(
+ ["git", "cat-file", "--batch"],
+ input=stdin,
+ capture_output=True,
+ cwd=AIRFLOW_ROOT,
+ check=True,
+ )
+
+ output = result.stdout
+ pos = 0
+ contents: dict[str, str] = {}
+ for path in paths:
+ newline_idx = output.index(b"\n", pos)
+ header = output[pos:newline_idx].decode("utf-8")
+ pos = newline_idx + 1
+ if header.endswith(" missing"):
+ continue
+ _sha1, _obj_type, size_str = header.split(" ")
+ size = int(size_str)
+ content_bytes = output[pos : pos + size]
+ pos += size + 1 # skip the protocol's trailing LF, which isn't
counted in size
+ contents[path] = content_bytes.decode("utf-8")
+ return contents
+
+
def git_tag_exists(tag: str) -> bool:
"""Check if a git tag exists locally."""
result = subprocess.run(
@@ -179,6 +236,32 @@ def get_source_file_path(layout: str, dir_path: str,
module_path: str) -> str:
return f"providers/src/{rel_file}"
+def read_guide_docs(tag: str, layout: str, dir_path: str) -> dict[str, str]:
+ """Read a provider's authored reST docs at a tag, keyed by path relative
to its docs dir.
+
+ Only the per-provider layout keeps docs beside the provider; under the old
flat
+ layout they lived in a top-level ``docs/`` tree, so those tags get no guide
+ links rather than links guessed from a path that moved.
+ """
+ if layout != "new":
+ return {}
+
+ docs_prefix = f"providers/{dir_path}/docs/"
+ survivors: list[tuple[str, str]] = []
+ for path in git_ls_tree(tag, docs_prefix):
+ if not path.endswith(".rst"):
+ continue
+ relative = path[len(docs_prefix) :]
+ if not is_guide_page(relative):
+ continue
+ survivors.append((relative, path))
+
+ batch_result = git_cat_file_batch(tag, [full_path for _relative, full_path
in survivors])
+ return {
+ relative: batch_result[full_path] for relative, full_path in survivors
if batch_result.get(full_path)
+ }
+
+
def parse_pyproject_toml_content(content: str, layout: str) -> dict[str, Any]:
"""Parse pyproject.toml content for dependencies, requires-python, and
extras."""
result: dict[str, Any] = {"requires_python": "", "dependencies": [],
"optional_extras": {}}
@@ -380,6 +463,8 @@ def extract_modules_from_yaml(
}
)
+ attach_guide_urls(modules, collect_guide_anchors(read_guide_docs(tag,
layout, dir_path)), base_docs_url)
+
return modules
diff --git a/dev/registry/registry_contract_models.py
b/dev/registry/registry_contract_models.py
index 31f4001dc49..02eb221b1ce 100644
--- a/dev/registry/registry_contract_models.py
+++ b/dev/registry/registry_contract_models.py
@@ -146,6 +146,9 @@ class ModuleContract(BaseModel):
provider_name: str | None = None
supports_durable_execution: bool = False
supports_deferrable: bool = False
+ # Only set for classes and task decorators (e.g. ``@task.agent``) that a
how-to
+ # guide documents in a section of their own.
+ guide_url: str | None = None
class ModulesCatalogContract(BaseModel):
diff --git a/dev/registry/registry_tools/docs_guides.py
b/dev/registry/registry_tools/docs_guides.py
new file mode 100644
index 00000000000..d19f9775704
--- /dev/null
+++ b/dev/registry/registry_tools/docs_guides.py
@@ -0,0 +1,186 @@
+# Licensed to the Apache Software Foundation (ASF) under one
+# or more contributor license agreements. See the NOTICE file
+# distributed with this work for additional information
+# regarding copyright ownership. The ASF licenses this file
+# to you under the Apache License, Version 2.0 (the
+# "License"); you may not use this file except in compliance
+# with the License. You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing,
+# software distributed under the License is distributed on an
+# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+# KIND, either express or implied. See the License for the
+# specific language governing permissions and limitations
+# under the License.
+"""Map a provider's classes to the how-to guide sections that document them.
+
+A module's ``docs_url`` points at generated API reference, which tells a reader
+what the arguments are but not how the thing is meant to be used. The prose
+guides carry that, and they already mark it: a how-to guide documents one class
+(or a class and its task-flow decorator) per section, titled with the name(s)
+either at the start (``HookToolset``, ``SQLToolset``, ``AgentOperator`` &
+``@task.agent``) or after a colon at the very end, as in a section titled
+"Airflow hooks as tools: ``HookToolset``".
+
+So the mapping is read back out of the guides rather than curated anywhere: a
+hand-maintained name-to-guide table would rot silently every time a guide is
+split, renamed, or a class is dropped, and a rotten link is worse than none.
+Callers supply the reST they can see (a git tag, or the working tree) and get
+back only the anchors those sources actually contain.
+"""
+
+from __future__ import annotations
+
+import re
+from collections.abc import Mapping
+from pathlib import PurePosixPath
+from typing import Any
+
+# reST underlines an (optionally overlined) section title with a run of one
+# punctuation character, at least as long as the title itself.
+_ADORNMENT_CHARS = "!\"#$%&'()*+,-./:;<=>?@[\\]^_`{|}~"
+
+_SKIPPED_PAGE_NAMES = frozenset({"changelog.rst", "commits.rst"})
+
+
+def is_guide_page(relative_path: str) -> bool:
+ """Whether a path relative to a provider's docs directory is a how-to
guide page.
+
+ Callers hand every ``.rst`` they can see to this before it ever reaches
+ ``collect_guide_anchors``. Two kinds of real, built pages must not go
+ further:
+
+ - Anything under a ``_``-prefixed path segment, at any depth
+ (``_api/hook/index.rst``, ``operators/_partials/foo.rst``, top-level
+ ``_partials/foo.rst``): Sphinx/autoapi output and partials are directive
+ markup, not the hand-written, reST-underlined titles this module's
+ inline-literal title convention parses.
+ - ``changelog.rst`` and ``commits.rst``: real release-note pages, not
+ how-to guides, that can carry inline-literal-formatted headings by
+ coincidence.
+ """
+ path = PurePosixPath(relative_path)
+ if any(part.startswith("_") for part in path.parts):
+ return False
+ return path.name not in _SKIPPED_PAGE_NAMES
+
+
+# A single inline-literal name: a class (``HookToolset``) or a task-flow
+# decorator (``@task.llm_file_analysis``) -- narrow enough that it still can't
+# match arbitrary prose wrapped in backticks.
+_INLINE_LITERAL_NAME =
r"``(@?[A-Za-z_][A-Za-z0-9_]*(?:\.[A-Za-z_][A-Za-z0-9_]*)*)``"
+_INLINE_LITERAL_NAME_RE = re.compile(_INLINE_LITERAL_NAME)
+
+# Only titles that consist solely of, or end a colon-led clause with, a run of
+# inline-literal names are treated as documenting them, so prose headings
+# ("Bounded query results") never produce a link. A run is one or more names
+# joined by "&", ",", "/" or "and" -- how guides write a section that covers
both
+# an operator and its decorator (``AgentOperator`` & ``@task.agent``).
+_NAME_SEPARATOR = r"(?:\s*[&,/]\s*|\s+and\s+)"
+_NAME_RUN =
rf"{_INLINE_LITERAL_NAME}(?:{_NAME_SEPARATOR}{_INLINE_LITERAL_NAME})*"
+
+# Shape one (what older release tags' docs use):
+# the title is nothing but the name run. Anchoring to "$" keeps a title that
+# merely opens with a literal and continues in prose ("``SandboxToolset``
+# parameters") from claiming to document that class.
+_LEADING_LITERAL_NAME_RUN = re.compile(rf"^{_NAME_RUN}\s*$")
+# Shape two (what current docs use): a prose lead-in, a colon, then the name
run runs to
+# the very end of the title. Anchoring to "$" is what keeps a colon earlier in
+# the title, with prose after it, from being mistaken for this shape.
+_TRAILING_LITERAL_NAME_RUN = re.compile(rf":\s+{_NAME_RUN}\s*$")
+
+
+def slugify_section_anchor(title: str) -> str:
+ """Return the HTML id Sphinx gives a section with this title.
+
+ Mirrors docutils' ``make_id``: lower-case, every run of non-alphanumeric
+ characters becomes a single hyphen, and leading/trailing hyphens are
+ dropped -- e.g. the section titled ``HookToolset`` is served at
+ ``#hooktoolset``.
+ """
+ return re.sub(r"[^a-z0-9]+", "-", title.lower()).strip("-")
+
+
+def _extract_names_from_title(title: str) -> list[str]:
+ """Return the names a section title documents, or [] if it names prose.
+
+ A guide marks a section as being *about* one or more names by titling it
+ with them as inline literals, either as the whole title (``HookToolset``,
or
+ ``AgentOperator`` & ``@task.agent`` where one section covers the operator
+ and its decorator) or after a colon at the very end ("Airflow hooks as
+ tools: ``HookToolset``"). Requiring that markup is what keeps a single-word
+ prose heading ("Guidelines") -- or a literal appearing elsewhere in a prose
+ title -- from claiming to document a class of the same name, and it is a
+ convention the guides already follow rather than one imposed on them.
+ """
+ match = _LEADING_LITERAL_NAME_RUN.match(title) or
_TRAILING_LITERAL_NAME_RUN.search(title)
+ return _INLINE_LITERAL_NAME_RE.findall(match.group(0)) if match else []
+
+
+def _is_adornment(line: str) -> bool:
+ """Whether a line is a reST title overline/underline rather than a
title."""
+ return bool(line) and len(set(line)) == 1 and line[0] in _ADORNMENT_CHARS
+
+
+def _extract_section_titles(text: str) -> list[str]:
+ """Return every section title in a reST document, in document order."""
+ titles = []
+ lines = text.splitlines()
+ for index, line in enumerate(lines[:-1]):
+ title = line.strip()
+ # Guides title these sections with an inline literal (``HookToolset``),
+ # so a title can legitimately start with an adornment character; only a
+ # line that is *entirely* one repeated character is an adornment.
+ if not title or _is_adornment(title):
+ continue
+ underline = lines[index + 1].strip()
+ if len(underline) >= len(title) and _is_adornment(underline):
+ titles.append(title)
+ return titles
+
+
+def collect_guide_anchors(docs: Mapping[str, str]) -> dict[str, str]:
+ """Map name -> ``<page>.html#<anchor>`` for every documented class or
decorator.
+
+ ``docs`` maps a page path relative to the provider's docs directory (e.g.
+ ``toolsets.rst``) to its reST source. A title can name more than one name
+ (``AgentOperator`` & ``@task.agent``), in which case every one gets the
+ same anchor. When two pages document the same name: a page's own title
+ (its first section) beats a subsection found on any other page, since that
+ page is the one dedicated to the class; among two page titles -- or two
+ subsections neither page titles -- the first page in sorted order wins, so
+ a rebuild of the same sources always produces the same link. "Page title"
+ is simply the first title _extract_section_titles finds, not a checked
+ top-level adornment, so a heading-shaped block earlier on the page (say,
+ inside a directive) would take that role.
+ """
+ found: dict[str, tuple[bool, str]] = {} # name -> (from a page title?,
anchor)
+ for page in sorted(docs):
+ page_url = re.sub(r"\.rst$", ".html", page)
+ for index, title in enumerate(_extract_section_titles(docs[page])):
+ is_page_title = index == 0
+ for name in _extract_names_from_title(title):
+ current = found.get(name)
+ # A page title replaces a subsection found earlier; nothing
else
+ # replaces what was found first, so rebuilds stay
deterministic.
+ if current is not None and (current[0] or not is_page_title):
+ continue
+ found[name] = (is_page_title,
f"{page_url}#{slugify_section_anchor(title)}")
+ return {name: anchor for name, (_, anchor) in found.items()}
+
+
+def attach_guide_urls(modules: list[dict[str, Any]], anchors: Mapping[str,
str], base_docs_url: str) -> int:
+ """Set ``guide_url`` on every module a guide section documents.
+
+ Mutates ``modules`` in place; returns how many got a link.
+ """
+ attached = 0
+ for module in modules:
+ anchor = anchors.get(module["name"])
+ if not anchor:
+ continue
+ module["guide_url"] = f"{base_docs_url.rstrip('/')}/{anchor}"
+ attached += 1
+ return attached
diff --git a/dev/registry/tests/test_docs_guides.py
b/dev/registry/tests/test_docs_guides.py
new file mode 100644
index 00000000000..53fcfe4367d
--- /dev/null
+++ b/dev/registry/tests/test_docs_guides.py
@@ -0,0 +1,369 @@
+# Licensed to the Apache Software Foundation (ASF) under one
+# or more contributor license agreements. See the NOTICE file
+# distributed with this work for additional information
+# regarding copyright ownership. The ASF licenses this file
+# to you under the Apache License, Version 2.0 (the
+# "License"); you may not use this file except in compliance
+# with the License. You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing,
+# software distributed under the License is distributed on an
+# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+# KIND, either express or implied. See the License for the
+# specific language governing permissions and limitations
+# under the License.
+from __future__ import annotations
+
+from unittest.mock import patch
+
+import pytest
+from extract_parameters import read_guide_docs as read_guide_docs_from_worktree
+from extract_versions import read_guide_docs as read_guide_docs_from_tag
+from registry_tools.docs_guides import (
+ attach_guide_urls,
+ collect_guide_anchors,
+ is_guide_page,
+ slugify_section_anchor,
+)
+
+TOOLSETS_GUIDE = """
+.. _howto/toolsets:
+
+Toolsets: Airflow hooks as AI agent tools
+==========================================
+
+Intro prose.
+
+Airflow hooks as tools: ``HookToolset``
+----------------------------------------
+
+How to use it.
+
+Guidelines
+^^^^^^^^^^
+
+More prose.
+
+.. _bounded-query-results:
+
+Bounded query results
+^^^^^^^^^^^^^^^^^^^^^
+
+``SQLToolset`` bounds that.
+
+Files with DataFusion: ``DataFusionToolset``
+-----------------------------------------------
+
+Another one.
+"""
+
+
[email protected](
+ ("title", "expected"),
+ [
+ # Verified against the published guide: the section titled
``HookToolset``
+ # is served at .../toolsets.html#hooktoolset.
+ ("HookToolset", "hooktoolset"),
+ ("AgentSkillsToolset", "agentskillstoolset"),
+ ("Bounded query results", "bounded-query-results"),
+ ("Agent_Skills", "agent-skills"),
+ ("Direct PydanticAI MCP toolsets", "direct-pydanticai-mcp-toolsets"),
+ ("``AgentOperator`` & ``@task.agent``", "agentoperator-task-agent"),
+ ],
+)
+def test_slugify_section_anchor_matches_sphinx_ids(title, expected):
+ assert slugify_section_anchor(title) == expected
+
+
+def test_collect_guide_anchors_finds_class_named_sections_at_any_depth():
+ anchors = collect_guide_anchors({"toolsets.rst": TOOLSETS_GUIDE})
+
+ assert anchors == {
+ "HookToolset": "toolsets.html#airflow-hooks-as-tools-hooktoolset",
+ "DataFusionToolset":
"toolsets.html#files-with-datafusion-datafusiontoolset",
+ }
+
+
+def test_collect_guide_anchors_ignores_prose_headings():
+ anchors = collect_guide_anchors({"toolsets.rst": TOOLSETS_GUIDE})
+
+ # "Guidelines" is shaped like a class name but isn't marked up as one.
+ assert "Guidelines" not in anchors
+ assert "Bounded query results" not in anchors
+
+
+def test_collect_guide_anchors_ignores_classes_only_mentioned_in_prose():
+ # SQLToolset appears in the guide's body but has no section of its own, so
+ # there is no anchor to link to.
+ assert "SQLToolset" not in collect_guide_anchors({"toolsets.rst":
TOOLSETS_GUIDE})
+
+
+def test_collect_guide_anchors_keeps_nested_page_paths():
+ guide = "Agents with tools:
``AgentOperator``\n-------------------------------------\n\nProse.\n"
+
+ assert collect_guide_anchors({"operators/agent.rst": guide}) == {
+ "AgentOperator": "operators/agent.html#agents-with-tools-agentoperator"
+ }
+
+
+def
test_collect_guide_anchors_ignores_a_title_that_opens_with_a_name_then_continues_in_prose():
+ guide = "``SandboxToolset`` parameters\n-----------------------------\n\nA
parameter table.\n"
+
+ assert collect_guide_anchors({"sandbox/configuration.rst": guide}) == {}
+
+
+def test_collect_guide_anchors_names_only_the_trailing_run_of_a_colon_title():
+ guide = "``A``: ``B``\n-------------\n\nProse.\n"
+
+ assert collect_guide_anchors({"page.rst": guide}) == {"B": "page.html#a-b"}
+
+
+def test_collect_guide_anchors_prefers_first_sorted_page_among_page_titles():
+ # Both pages title themselves after SQLToolset, so neither title beats the
+ # other on that basis alone; the tie is broken by sorted page order.
+ guide = "SQL databases:
``SQLToolset``\n------------------------------\n\nProse.\n"
+
+ anchors = collect_guide_anchors({"toolsets.rst": guide,
"operators/sql.rst": guide})
+
+ assert anchors["SQLToolset"] ==
"operators/sql.html#sql-databases-sqltoolset"
+
+
+def test_collect_guide_anchors_handles_a_title_covering_more_than_the_class():
+ # Verified against the published guide: this heading is served at
+ # .../operators/agent.html#agentoperator-task-agent, so the anchor comes
from
+ # the whole title while both the operator and its decorator get linked to
it.
+ # This is the older leading shape; older release tags' docs (read by
+ # extract_versions.py) still use it, so it must keep working.
+ guide = "``AgentOperator`` &
``@task.agent``\n===================================\n\nProse.\n"
+
+ assert collect_guide_anchors({"operators/agent.rst": guide}) == {
+ "AgentOperator": "operators/agent.html#agentoperator-task-agent",
+ "@task.agent": "operators/agent.html#agentoperator-task-agent",
+ }
+
+
+def test_collect_guide_anchors_links_a_decorator_name_with_underscores():
+ # ``@task.llm_file_analysis`` is the real decorator name for the common.ai
+ # provider's LLMFileAnalysisOperator; the leading-literal charset must
admit
+ # "@" and "." for it to ever get a link. Also a leading-shape fixture kept
for
+ # the same reason as the test above: older release tags' docs still use it.
+ guide = (
+ "``LLMFileAnalysisOperator`` & ``@task.llm_file_analysis``\n"
+ "==========================================================\n\n"
+ "Prose.\n"
+ )
+
+ assert collect_guide_anchors({"operators/llm_file_analysis.rst": guide})
== {
+ "LLMFileAnalysisOperator": (
+
"operators/llm_file_analysis.html#llmfileanalysisoperator-task-llm-file-analysis"
+ ),
+ "@task.llm_file_analysis": (
+
"operators/llm_file_analysis.html#llmfileanalysisoperator-task-llm-file-analysis"
+ ),
+ }
+
+
+def test_collect_guide_anchors_ignores_a_prose_title_mentioning_a_literal():
+ # The title neither opens with the literal nor ends a colon-led clause with
+ # it, so a prose heading that happens to mention one in passing -- anywhere
+ # in the title -- must not produce a link.
+ guide = "Using ``foo`` in a
pipeline\n============================\n\nProse.\n"
+
+ assert collect_guide_anchors({"toolsets.rst": guide}) == {}
+
+
+def test_collect_guide_anchors_requires_a_long_enough_underline():
+ # An underline shorter than the title isn't a section in reST, so it must
not
+ # produce a link to an anchor Sphinx never emitted.
+ guide = "Airflow hooks as tools: ``HookToolset``\n---\n\nProse.\n"
+
+ assert collect_guide_anchors({"toolsets.rst": guide}) == {}
+
+
+def test_collect_guide_anchors_reads_a_name_trailing_a_colon():
+ guide = "Airflow hooks as tools:
``HookToolset``\n========================================\n\nProse.\n"
+
+ assert collect_guide_anchors({"toolsets/hook.rst": guide}) == {
+ "HookToolset": "toolsets/hook.html#airflow-hooks-as-tools-hooktoolset"
+ }
+
+
+def test_collect_guide_anchors_reads_two_names_joined_by_and():
+ guide = (
+ "Agents with tools: ``AgentOperator`` and ``@task.agent``\n"
+ "=========================================================\n\n"
+ "Prose.\n"
+ )
+
+ assert collect_guide_anchors({"operators/agent.rst": guide}) == {
+ "AgentOperator":
"operators/agent.html#agents-with-tools-agentoperator-and-task-agent",
+ "@task.agent":
"operators/agent.html#agents-with-tools-agentoperator-and-task-agent",
+ }
+
+
[email protected](
+ "title",
+ [
+ # Leading shape, one separator per case.
+ "``A`` & ``B``",
+ "``A``, ``B``",
+ "``A``/``B``",
+ "``A`` and ``B``",
+ # Trailing shape, one separator per case.
+ "Both: ``A`` & ``B``",
+ "Both: ``A``, ``B``",
+ "Both: ``A``/``B``",
+ "Both: ``A`` and ``B``",
+ ],
+)
+def
test_collect_guide_anchors_accepts_every_separator_in_both_title_shapes(title):
+ underline = "=" * (len(title) + 1)
+
+ anchors = collect_guide_anchors({"page.rst":
f"{title}\n{underline}\n\nProse.\n"})
+
+ expected_anchor = f"page.html#{slugify_section_anchor(title)}"
+ assert anchors == {"A": expected_anchor, "B": expected_anchor}
+
+
+def
test_collect_guide_anchors_prefers_a_page_title_over_an_earlier_pages_subsection():
+ # A subsection on an unrelated page happens to be titled after the class,
+ # but a later page is dedicated to it.
+ docs = {
+ "agent_security.rst": (
+ "Securing agent tools\n"
+ "=====================\n\n"
+ "``HookToolset`` guidelines\n"
+ "^^^^^^^^^^^^^^^^^^^^^^^^^^^\n\n"
+ "Prose.\n"
+ ),
+ "toolsets/hook.rst": (
+ "Airflow hooks as tools:
``HookToolset``\n========================================\n\nProse.\n"
+ ),
+ }
+
+ assert collect_guide_anchors(docs) == {
+ "HookToolset": "toolsets/hook.html#airflow-hooks-as-tools-hooktoolset"
+ }
+
+
+def
test_collect_guide_anchors_prefers_the_page_title_over_its_own_subsection():
+ guide = (
+ "Batch processing: ``LLMBatchOperator``\n"
+ "=======================================\n\n"
+ "How it works.\n\n"
+ "``LLMBatchOperator`` or the vendor batch operators?\n"
+ "----------------------------------------------------\n\n"
+ "Prose.\n"
+ )
+
+ assert collect_guide_anchors({"operators/llm_batch.rst": guide}) == {
+ "LLMBatchOperator":
"operators/llm_batch.html#batch-processing-llmbatchoperator"
+ }
+
+
+def
test_collect_guide_anchors_keeps_the_first_subsection_when_no_page_title_names_it():
+ docs = {
+ "a.rst": "Prose page\n===========\n\n``X``\n-----\n",
+ "b.rst": "Other page\n===========\n\n``X``\n-----\n",
+ }
+
+ assert collect_guide_anchors(docs) == {"X": "a.html#x"}
+
+
[email protected](
+ "title",
+ [
+ # Colon present, but prose follows it before the literal -- must not be
+ # scanned for a literal anywhere after the colon.
+ "Role in a Dag: use ``MCPToolset``, not the hook directly",
+ # Title ends with the literal, but there's no colon to lead it.
+ "Using HITL review with ``AgentOperator``",
+ # Colon immediately precedes the literal, but prose follows it -- the
+ # literal doesn't reach the end of the title.
+ "Guidelines: ``HookToolset`` and its allow-list",
+ ],
+)
+def
test_collect_guide_anchors_ignores_a_literal_that_does_not_end_the_title(title):
+ underline = "=" * (len(title) + 1)
+
+ assert collect_guide_anchors({"toolsets.rst":
f"{title}\n{underline}\n\nProse.\n"}) == {}
+
+
+def test_attach_guide_urls_only_links_documented_classes():
+ modules = [
+ {"name": "HookToolset", "docs_url":
"https://example.test/_api/hook/index.html"},
+ {"name": "UndocumentedToolset", "docs_url":
"https://example.test/_api/other/index.html"},
+ ]
+
+ attached = attach_guide_urls(
+ modules,
+ {"HookToolset": "toolsets.html#hooktoolset"},
+
"https://airflow.apache.org/docs/apache-airflow-providers-common-ai/0.7.0",
+ )
+
+ assert attached == 1
+ assert modules[0]["guide_url"] == (
+
"https://airflow.apache.org/docs/apache-airflow-providers-common-ai/0.7.0/toolsets.html#hooktoolset"
+ )
+ assert "guide_url" not in modules[1]
+
+
+def test_attach_guide_urls_does_not_double_up_the_base_separator():
+ modules = [{"name": "HookToolset"}]
+
+ attach_guide_urls(modules, {"HookToolset": "toolsets.html#hooktoolset"},
"https://example.test/docs/")
+
+ assert modules[0]["guide_url"] ==
"https://example.test/docs/toolsets.html#hooktoolset"
+
+
[email protected](
+ ("relative_path", "expected"),
+ [
+ # A `_`-prefixed path segment marks autoapi/partial content, at any
depth.
+ ("_api/index.rst", False),
+ ("_api/hook/index.rst", False),
+ ("operators/_partials/foo.rst", False),
+ ("_partials/foo.rst", False),
+ # Real, built release-note pages, not how-to guides.
+ ("changelog.rst", False),
+ ("commits.rst", False),
+ ("toolsets.rst", True),
+ ("operators/agent.rst", True),
+ ],
+)
+def test_is_guide_page(relative_path, expected):
+ assert is_guide_page(relative_path) == expected
+
+
+def test_readers_agree_on_which_pages_are_guides(tmp_path):
+ """Both `read_guide_docs` implementations delegate to `is_guide_page`, so a
+ working-tree read and a git-tag read of the same paths must end up with the
+ same set of pages -- regardless of which source produced them."""
+ relative_paths = ["_api/index.rst", "changelog.rst", "commits.rst",
"toolsets.rst"]
+ for relative in relative_paths:
+ target = tmp_path / relative
+ target.parent.mkdir(parents=True, exist_ok=True)
+ target.write_text("Prose.\n")
+
+ from_worktree = read_guide_docs_from_worktree(tmp_path)
+
+ docs_prefix = "providers/test/docs/"
+ with (
+ patch(
+ "extract_versions.git_ls_tree",
+ autospec=True,
+ return_value=[docs_prefix + relative for relative in
relative_paths],
+ ),
+ patch(
+ "extract_versions.git_cat_file_batch",
+ autospec=True,
+ side_effect=lambda tag, paths: {p: "Prose.\n" for p in paths},
+ ),
+ ):
+ from_tag = read_guide_docs_from_tag("providers-test/1.0.0", "new",
"test")
+
+ # Both readers must apply the same filter; each reader's own test can pass
+ # while the two drift apart.
+ assert set(from_worktree) == set(from_tag) == {"toolsets.rst"}
diff --git a/dev/registry/tests/test_extract_parameters.py
b/dev/registry/tests/test_extract_parameters.py
index eac805cc49a..b751ae64f08 100644
--- a/dev/registry/tests/test_extract_parameters.py
+++ b/dev/registry/tests/test_extract_parameters.py
@@ -22,7 +22,7 @@ import abc
import builtins
import json
import types
-from dataclasses import fields
+from dataclasses import MISSING, fields
from unittest.mock import patch
import pytest
@@ -38,6 +38,7 @@ from extract_parameters import (
get_category,
is_durable_capable,
load_resumable_job_mixin,
+ read_guide_docs,
supports_deferrable,
)
@@ -911,6 +912,20 @@ FAKE_PROVIDER_YAML = {
}
+# ---------------------------------------------------------------------------
+# read_guide_docs
+# ---------------------------------------------------------------------------
+def test_read_guide_docs_skips_generated_and_release_note_pages(tmp_path):
+ (tmp_path / "_api" / "x").mkdir(parents=True)
+ (tmp_path / "_api" / "x" / "index.rst").write_text("Generated.\n")
+ (tmp_path / "changelog.rst").write_text("Release notes.\n")
+ (tmp_path /
"toolsets.rst").write_text("``HookToolset``\n---------------\n\nProse.\n")
+
+ result = read_guide_docs(tmp_path)
+
+ assert set(result) == {"toolsets.rst"}
+
+
# ---------------------------------------------------------------------------
# TestDiscoverClassesFromProvider
# ---------------------------------------------------------------------------
@@ -999,6 +1014,88 @@ class TestDiscoverClassesFromProvider:
assert operators[0]["import_path"] ==
"airflow.providers.amazon.aws.operators.s3.FakeOperator"
assert operators[0]["provider_id"] == "amazon"
+ def test_guide_section_becomes_a_guide_url(self, provider_yaml_path,
base_classes):
+ """A class documented by a section of its own gets a link to that
section;
+ one that is only in the API reference keeps just its ``docs_url``."""
+ docs_dir = provider_yaml_path.parent / "docs" / "operators"
+ docs_dir.mkdir(parents=True)
+ (docs_dir /
"s3.rst").write_text("``FakeOperator``\n----------------\n\nProse.\n")
+
+ with (
+ patch("extract_parameters.PROVIDERS_DIR",
provider_yaml_path.parent.parent),
+ patch("extract_parameters.importlib.import_module",
side_effect=self._mock_import),
+ ):
+ result = discover_classes_from_provider(provider_yaml_path,
base_classes)
+
+ by_name = {r["name"]: r for r in result}
+ assert by_name["FakeOperator"]["guide_url"] == (
+
"https://airflow.apache.org/docs/apache-airflow-providers-amazon/stable"
+ "/operators/s3.html#fakeoperator"
+ )
+ assert "guide_url" not in by_name["FakeSensor"]
+
+ @pytest.mark.parametrize(
+ ("tag_exists", "expected_anchor"),
+ [
+ pytest.param(True, "released-title-fakeoperator",
id="tag-exists-reads-released-docs"),
+ pytest.param(False, "fakeoperator",
id="no-tag-reads-working-tree"),
+ ],
+ )
+ def test_guide_docs_come_from_release_tag_when_it_exists(
+ self, provider_yaml_path, base_classes, tag_exists, expected_anchor
+ ):
+ docs_dir = provider_yaml_path.parent / "docs" / "operators"
+ docs_dir.mkdir(parents=True)
+ (docs_dir /
"s3.rst").write_text("``FakeOperator``\n----------------\n\nUnreleased
title.\n")
+ released = {
+ "operators/s3.rst": "Released title:
``FakeOperator``\n--------------------------------\n"
+ }
+
+ with (
+ patch("extract_parameters.PROVIDERS_DIR",
provider_yaml_path.parent.parent),
+ patch("extract_parameters.git_tag_exists",
return_value=tag_exists) as tag_check,
+ patch("extract_parameters.detect_layout", return_value="new"),
+ patch("extract_parameters.read_guide_docs_at_tag",
return_value=released) as read_at_tag,
+ patch("extract_parameters.importlib.import_module",
side_effect=self._mock_import),
+ ):
+ result = discover_classes_from_provider(provider_yaml_path,
base_classes, version="1.2.3")
+
+ tag_check.assert_called_once_with("providers-amazon/1.2.3")
+ if tag_exists:
+ read_at_tag.assert_called_once_with("providers-amazon/1.2.3",
"new", "amazon")
+ else:
+ read_at_tag.assert_not_called()
+ guide_url = {r["name"]: r for r in result}["FakeOperator"]["guide_url"]
+ assert guide_url.endswith(f"/operators/s3.html#{expected_anchor}")
+
+ @pytest.mark.parametrize(
+ ("version", "layout", "expect_tag_lookup"),
+ [
+ pytest.param("", "new", False, id="no-version-skips-tag-lookup"),
+ pytest.param("1.2.3", None, True,
id="undetectable-layout-falls-back"),
+ ],
+ )
+ def test_guide_docs_fall_back_to_working_tree(
+ self, provider_yaml_path, base_classes, version, layout,
expect_tag_lookup
+ ):
+ docs_dir = provider_yaml_path.parent / "docs" / "operators"
+ docs_dir.mkdir(parents=True)
+ (docs_dir /
"s3.rst").write_text("``FakeOperator``\n----------------\n\nProse.\n")
+
+ with (
+ patch("extract_parameters.PROVIDERS_DIR",
provider_yaml_path.parent.parent),
+ patch("extract_parameters.git_tag_exists", return_value=True) as
tag_check,
+ patch("extract_parameters.detect_layout", return_value=layout),
+ patch("extract_parameters.read_guide_docs_at_tag") as read_at_tag,
+ patch("extract_parameters.importlib.import_module",
side_effect=self._mock_import),
+ ):
+ result = discover_classes_from_provider(provider_yaml_path,
base_classes, version=version)
+
+ assert tag_check.called is expect_tag_lookup
+ read_at_tag.assert_not_called()
+ guide_url = {r["name"]: r for r in result}["FakeOperator"]["guide_url"]
+ assert guide_url.endswith("/operators/s3.html#fakeoperator")
+
def test_discovers_sensor(self, provider_yaml_path, base_classes):
with (
patch("extract_parameters.PROVIDERS_DIR",
provider_yaml_path.parent.parent),
@@ -1093,14 +1190,18 @@ class TestDiscoverClassesFromProvider:
assert operators[0]["short_description"] == "Copy objects in S3."
def test_all_module_fields_present(self, provider_yaml_path, base_classes):
- """Every discovered entry has every `Module` dataclass field (derived,
not hardcoded)."""
+ """Every discovered entry has every required `Module` dataclass field
(derived, not hardcoded).
+
+ Fields with a default (e.g. ``guide_url``) are attached separately and
only
+ when applicable, so they are allowed to be absent here.
+ """
with (
patch("extract_parameters.PROVIDERS_DIR",
provider_yaml_path.parent.parent),
patch("extract_parameters.importlib.import_module",
side_effect=self._mock_import),
):
result = discover_classes_from_provider(provider_yaml_path,
base_classes)
- required_fields = {f.name for f in fields(Module)}
+ required_fields = {f.name for f in fields(Module) if f.default is
MISSING}
for entry in result:
missing = required_fields - entry.keys()
assert not missing, f"Missing fields {missing} in {entry['name']}"
diff --git a/dev/registry/tests/test_extract_versions.py
b/dev/registry/tests/test_extract_versions.py
index 6a3128ce01a..7ecc7d94dff 100644
--- a/dev/registry/tests/test_extract_versions.py
+++ b/dev/registry/tests/test_extract_versions.py
@@ -18,8 +18,9 @@
from __future__ import annotations
+import subprocess
import textwrap
-from unittest.mock import patch
+from unittest.mock import MagicMock, call, patch
import pytest
from extract_versions import (
@@ -28,6 +29,9 @@ from extract_versions import (
SCRIPT_DIR,
extract_modules_from_yaml,
extract_version_data,
+ git_cat_file_batch,
+ git_ls_tree,
+ read_guide_docs,
)
from registry_tools.types import CLASS_LEVEL_SECTIONS,
DICT_SHAPED_CLASS_LEVEL_SECTIONS
@@ -84,7 +88,8 @@ EXPECTED_CLASS_LEVEL_DESC_SUFFIXES = {
}
-def _extract_class_level_modules(provider_yaml: dict) -> list[dict]:
+@patch("extract_versions.read_guide_docs", autospec=True, return_value={})
+def _extract_class_level_modules(provider_yaml: dict, _mock_read_guide_docs)
-> list[dict]:
return extract_modules_from_yaml(
provider_yaml,
tag="providers-test/1.0.0",
@@ -266,3 +271,189 @@ class TestExtractVersionDataUriSchemes:
{"scheme": "tfs", "filesystem":
"airflow.providers.test.fs.testfs"},
]
mock_git_show.assert_any_call("providers-test/1.0.0", fs_source_path)
+
+
+class TestExtractModulesGuideUrls:
+ """A class the provider's guides document in a section of its own must get
a
+ ``guide_url`` for every version, not just the latest. A superseded
version's
+ page is rendered only from the per-version file this module writes, so a
link
+ resolved on the latest path alone disappears the moment a new version
lands.
+ """
+
+ PROVIDER_YAML = {
+ "toolsets": [
+ {
+ "integration-name": "Test",
+ "python-modules": ["airflow.providers.test.toolsets.hook"],
+ }
+ ]
+ }
+ SOURCE = 'class HookToolset:\n """A toolset."""\n'
+ GUIDE = "``HookToolset``\n---------------\n\nProse.\n"
+
+ def _extract(self, layout="new",
docs_paths=("providers/test/docs/toolsets.rst",)):
+ def fake_git_show(_tag, path):
+ return self.SOURCE if path.endswith(".py") else None
+
+ def fake_git_cat_file_batch(_tag, paths):
+ return {p: self.GUIDE for p in paths}
+
+ with (
+ patch("extract_versions.git_ls_tree", autospec=True,
return_value=list(docs_paths)),
+ patch("extract_versions.git_show", autospec=True,
side_effect=fake_git_show),
+ patch("extract_versions.git_cat_file_batch", autospec=True,
side_effect=fake_git_cat_file_batch),
+ ):
+ return extract_modules_from_yaml(
+ self.PROVIDER_YAML, "providers-test/1.0.0", layout, "test",
"test", "1.0.0"
+ )
+
+ def test_documented_class_gets_a_versioned_guide_url(self):
+ modules = self._extract()
+
+ assert [m["name"] for m in modules] == ["HookToolset"]
+ assert modules[0]["guide_url"] == (
+
"https://airflow.apache.org/docs/apache-airflow-providers-test/1.0.0/toolsets.html#hooktoolset"
+ )
+
+ def test_undocumented_class_gets_no_guide_url(self):
+ modules = self._extract(docs_paths=())
+
+ assert [m["name"] for m in modules] == ["HookToolset"]
+ assert "guide_url" not in modules[0]
+
+ def test_old_layout_gets_no_guide_url(self):
+ # Pre-per-provider tags kept docs in a top-level tree, so there is no
+ # provider-relative page path to build a link from.
+ modules = self._extract(layout="old")
+
+ assert "guide_url" not in modules[0]
+
+
+class TestReadGuideDocs:
+ def
test_skips_generated_and_release_note_pages_before_calling_git_show(self):
+ docs_prefix = "providers/test/docs/"
+ paths = [
+ docs_prefix + "_api/x/index.rst",
+ docs_prefix + "changelog.rst",
+ docs_prefix + "diagram.png",
+ docs_prefix + "conf.py",
+ docs_prefix + "toolsets.rst",
+ ]
+
+ with (
+ patch("extract_versions.git_ls_tree", autospec=True,
return_value=paths),
+ patch(
+ "extract_versions.git_cat_file_batch",
+ autospec=True,
+ side_effect=lambda tag, paths: {p: "Prose.\n" for p in paths},
+ ) as mock_git_cat_file_batch,
+ ):
+ result = read_guide_docs("providers-test/1.0.0", "new", "test")
+
+ assert set(result) == {"toolsets.rst"}
+ # Filtering must happen before the batch call, not just before the
dict write.
+ assert mock_git_cat_file_batch.call_args_list == [
+ call("providers-test/1.0.0", [docs_prefix + "toolsets.rst"])
+ ]
+
+ def test_skips_a_page_whose_content_is_an_empty_string(self):
+ docs_prefix = "providers/test/docs/"
+ paths = [docs_prefix + "empty.rst"]
+
+ with (
+ patch("extract_versions.git_ls_tree", autospec=True,
return_value=paths),
+ patch(
+ "extract_versions.git_cat_file_batch",
+ autospec=True,
+ return_value={docs_prefix + "empty.rst": ""},
+ ),
+ ):
+ result = read_guide_docs("providers-test/1.0.0", "new", "test")
+
+ assert result == {}
+
+
+class TestGitLsTree:
+ def test_passes_quote_path_false_to_git(self):
+ mock_result = MagicMock(spec=subprocess.CompletedProcess)
+ mock_result.stdout = b"providers/test/docs/toolsets.rst\n"
+ with patch("extract_versions.subprocess.run", autospec=True,
return_value=mock_result) as mock_run:
+ git_ls_tree("providers-test/1.0.0", "providers/test/docs/")
+
+ assert mock_run.call_args.args[0] == [
+ "git",
+ "-c",
+ "core.quotePath=false",
+ "ls-tree",
+ "-r",
+ "--name-only",
+ "providers-test/1.0.0",
+ "--",
+ "providers/test/docs/",
+ ]
+
+ def test_decodes_stdout_as_utf8(self):
+ mock_result = MagicMock(spec=subprocess.CompletedProcess)
+ mock_result.stdout = "docs/café.rst\ndocs/b.rst\n".encode()
+ with patch("extract_versions.subprocess.run", autospec=True,
return_value=mock_result):
+ result = git_ls_tree("providers-test/1.0.0",
"providers/test/docs/")
+
+ assert result == ["docs/café.rst", "docs/b.rst"]
+
+
+def _batch_hit(sha1: str, obj_type: str, content: bytes) -> bytes:
+ return f"{sha1} {obj_type} {len(content)}\n".encode() + content + b"\n"
+
+
+def _batch_missing(spec: str) -> bytes:
+ return f"{spec} missing\n".encode()
+
+
+class TestGitCatFileBatch:
+ def test_empty_paths_returns_empty_dict_without_subprocess(self):
+ with patch("extract_versions.subprocess.run", autospec=True) as
mock_run:
+ result = git_cat_file_batch("providers-test/1.0.0", [])
+
+ assert result == {}
+ mock_run.assert_not_called()
+
+ def test_two_hits_with_different_sizes_and_multibyte_content(self):
+ tag = "providers-test/1.0.0"
+ paths = ["providers/test/docs/a.rst", "providers/test/docs/b.rst"]
+ first_content = b"short\n"
+ second_content = "café prôse with more text\n".encode()
+ payload = _batch_hit("aaa1", "blob", first_content) +
_batch_hit("bbb2", "blob", second_content)
+
+ mock_result = MagicMock(spec=subprocess.CompletedProcess)
+ mock_result.stdout = payload
+ with patch("extract_versions.subprocess.run", autospec=True,
return_value=mock_result):
+ result = git_cat_file_batch(tag, paths)
+
+ assert result == {
+ paths[0]: "short\n",
+ paths[1]: "café prôse with more text\n",
+ }
+
+ def test_hit_followed_by_missing_path(self):
+ tag = "providers-test/1.0.0"
+ paths = ["providers/test/docs/a.rst",
"providers/test/docs/missing.rst"]
+ payload = _batch_hit("aaa1", "blob", b"content\n") +
_batch_missing(f"{tag}:{paths[1]}")
+
+ mock_result = MagicMock(spec=subprocess.CompletedProcess)
+ mock_result.stdout = payload
+ with patch("extract_versions.subprocess.run", autospec=True,
return_value=mock_result):
+ result = git_cat_file_batch(tag, paths)
+
+ assert result == {paths[0]: "content\n"}
+
+ def test_all_missing_returns_empty_dict(self):
+ tag = "providers-test/1.0.0"
+ paths = ["providers/test/docs/a.rst", "providers/test/docs/b.rst"]
+ payload = _batch_missing(f"{tag}:{paths[0]}") +
_batch_missing(f"{tag}:{paths[1]}")
+
+ mock_result = MagicMock(spec=subprocess.CompletedProcess)
+ mock_result.stdout = payload
+ with patch("extract_versions.subprocess.run", autospec=True,
return_value=mock_result):
+ result = git_cat_file_batch(tag, paths)
+
+ assert result == {}
diff --git a/dev/registry/tests/test_registry_contract_models.py
b/dev/registry/tests/test_registry_contract_models.py
index 4e477682f66..0e29b7aa2c4 100644
--- a/dev/registry/tests/test_registry_contract_models.py
+++ b/dev/registry/tests/test_registry_contract_models.py
@@ -114,6 +114,15 @@ def
test_module_contract_preserves_supports_deferrable_true():
assert validated["modules"][0]["supports_deferrable"] is True
+def test_module_contract_omits_guide_url_for_undocumented_classes():
+ assert ModuleContract.model_validate(_module_payload()).guide_url is None
+
+
+def test_module_contract_preserves_guide_url_value():
+ guide_url = "https://example.invalid/docs/toolsets.html#exampletoolset"
+ assert
ModuleContract.model_validate(_module_payload(guide_url=guide_url)).guide_url
== guide_url
+
+
def test_connection_type_contract_defaults_external_services_to_empty_list():
"""Legacy connection-types entries (provider.yaml without
`external-services`)
must still validate, with the field defaulting to an empty list."""
diff --git a/providers/common/ai/docs/operators/llm_branch.rst
b/providers/common/ai/docs/operators/llm_branch.rst
index 35b15f0f2b9..0e113aed050 100644
--- a/providers/common/ai/docs/operators/llm_branch.rst
+++ b/providers/common/ai/docs/operators/llm_branch.rst
@@ -17,8 +17,8 @@
.. _howto/operator:llm_branch:
-Branch on an answer: ``LLMBranchOperator``
-==========================================
+Branch on an answer: ``LLMBranchOperator`` and ``@task.llm_branch``
+===================================================================
Use
:class:`~airflow.providers.common.ai.operators.llm_branch.LLMBranchOperator`
for LLM-driven branching, where the LLM decides which downstream task(s) to
diff --git a/providers/common/ai/docs/operators/llm_file_analysis.rst
b/providers/common/ai/docs/operators/llm_file_analysis.rst
index 4fb0931410d..eed8c866281 100644
--- a/providers/common/ai/docs/operators/llm_file_analysis.rst
+++ b/providers/common/ai/docs/operators/llm_file_analysis.rst
@@ -17,8 +17,8 @@
.. _howto/operator:llm_file_analysis:
-Analyze files and images: ``LLMFileAnalysisOperator``
-=====================================================
+Analyze files and images: ``LLMFileAnalysisOperator`` and
``@task.llm_file_analysis``
+=====================================================================================
.. note::
diff --git a/providers/common/ai/docs/operators/llm_schema_compare.rst
b/providers/common/ai/docs/operators/llm_schema_compare.rst
index cafde3900de..47b06cf2225 100644
--- a/providers/common/ai/docs/operators/llm_schema_compare.rst
+++ b/providers/common/ai/docs/operators/llm_schema_compare.rst
@@ -17,8 +17,8 @@
.. _howto/operator:llm_schema_compare:
-Detect schema drift: ``LLMSchemaCompareOperator``
-=================================================
+Detect schema drift: ``LLMSchemaCompareOperator`` and
``@task.llm_schema_compare``
+==================================================================================
.. note::
diff --git a/providers/common/ai/docs/operators/llm_sql.rst
b/providers/common/ai/docs/operators/llm_sql.rst
index 5d06ec97e39..42e51de7955 100644
--- a/providers/common/ai/docs/operators/llm_sql.rst
+++ b/providers/common/ai/docs/operators/llm_sql.rst
@@ -17,8 +17,8 @@
.. _howto/operator:llm_sql_query:
-Natural language to SQL: ``LLMSQLQueryOperator``
-================================================
+Natural language to SQL: ``LLMSQLQueryOperator`` and ``@task.llm_sql``
+======================================================================
.. note::
diff --git a/registry/AGENTS.md b/registry/AGENTS.md
index 99416cb8931..b82377a2569 100644
--- a/registry/AGENTS.md
+++ b/registry/AGENTS.md
@@ -462,6 +462,39 @@ They run inside Breeze where all providers are installed.
`extract_metadata.py`
the CI workflow can run the fast scripts (metadata, ~30s per provider) without
spinning
up Breeze, while parameter/connection extraction is a separate step.
+### How a module gets a "Guide" link
+
+A module card links to the how-to guide section that documents it, alongside
the
+generated API reference. `provider.yaml`'s `how-to-guide` fields are
CI-enforced
+by `check_doc_files`, but they name a whole page, never a section, and cover
only
+operators, sensors and transfers — not the toolset, hook and decorator pages
this
+needs. So `registry_tools/docs_guides.py` reads the provider's own `docs/*.rst`
+and matches a class to a section when the section's title names it as an inline
+literal, either at the start — ``` ``MCPHook`` ``` or after a colon at the very
+end — ``` Airflow hooks as tools: ``HookToolset`` ``` or ``` Agents with tools:
+``AgentOperator`` and ``@task.agent`` ```. A literal elsewhere in a prose title
+does not count. The anchor is derived from the whole title the way docutils
+derives its HTML id. When a name is titled in more than one place, a page's own
+title wins over a subsection on any other page, so the link lands on the page
+dedicated to the class rather than on a passing section about it.
+
+That convention is what the guides already do, and it is deliberately the only
+signal this resolves a section from: a hand-maintained class-to-guide table
+would keep pointing at sections that have since been renamed or split, and a
+link that lands on the wrong section is worse than no link. A class documented
+only in prose gets no Guide link.
+
+Growing the set of modules that get a Guide link means titling that provider's
+sections in one of those two shapes, not touching this extractor. `common/ai`
+titles its dedicated operator, hook and toolset pages this way; a couple of
other
+providers use the same shapes for a config option name or a single decorator
+rather than a class. Having the right title doesn't guarantee a link — that
still
+depends on a same-named module existing in the catalog.
+
+Both extraction paths resolve it — `extract_parameters.py` from the working
tree
+for the latest release, `extract_versions.py` from the git tag for superseded
ones
+— because a superseded version's page is rendered only from its own metadata
file.
+
### Relationship to `run_provider_yaml_files_check.py`
`scripts/in_container/run_provider_yaml_files_check.py` (run by the
diff --git a/registry/src/css/main.css b/registry/src/css/main.css
index d5077f8c888..42b6d1a618e 100644
--- a/registry/src/css/main.css
+++ b/registry/src/css/main.css
@@ -3552,12 +3552,14 @@ main {
/* Module Actions (View Docs, Source) */
.provider-detail-page .module-actions {
display: flex;
+ flex-wrap: wrap;
align-items: center;
gap: var(--space-3);
margin-top: var(--space-3);
}
.provider-detail-page .module-actions .docs-link,
+.provider-detail-page .module-actions .guide-link,
.provider-detail-page .module-actions .source-link {
display: inline-flex;
align-items: center;
@@ -3575,6 +3577,14 @@ main {
color: var(--color-cyan-300);
}
+.provider-detail-page .module-actions .guide-link {
+ color: var(--accent-secondary);
+}
+
+.provider-detail-page .module-actions .guide-link:hover {
+ color: var(--color-cyan-300);
+}
+
.provider-detail-page .module-actions .source-link {
color: var(--text-muted);
}
diff --git a/registry/src/provider-version.njk
b/registry/src/provider-version.njk
index e28be209f1a..7d0c16b9c49 100644
--- a/registry/src/provider-version.njk
+++ b/registry/src/provider-version.njk
@@ -475,6 +475,12 @@ eleventyComputed:
<svg fill="none" stroke="currentColor" viewBox="0 0 24 24"
aria-hidden="true"><path stroke-linecap="round" stroke-linejoin="round"
stroke-width="2" d="M10 6H6a2 2 0 00-2 2v10a2 2 0 002 2h10a2 2 0 002-2v-4M14
4h6m0 0v6m0-6L10 14" /></svg>
</a>
{% endif %}
+ {% if module.guide_url %}
+ <a href="{{ module.guide_url }}" target="_blank" rel="noopener"
class="guide-link" title="How-to guide section for {{ module.name }}">
+ Guide
+ <svg fill="none" stroke="currentColor" viewBox="0 0 24 24"
aria-hidden="true"><path stroke-linecap="round" stroke-linejoin="round"
stroke-width="2" d="M10 6H6a2 2 0 00-2 2v10a2 2 0 002 2h10a2 2 0 002-2v-4M14
4h6m0 0v6m0-6L10 14" /></svg>
+ </a>
+ {% endif %}
{% if module.source_url %}
<a href="{{ module.source_url }}" target="_blank" rel="noopener"
class="source-link">
Source