Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
19 commits
Select commit Hold shift + click to select a range
258180f
Link registry modules to the guide section that documents them
Lee-W Aug 12, 2026
b4a11d8
Give a section's decorator name a guide link too
Lee-W Sep 18, 2026
65fa06b
Resolve guide links from guide pages only
Lee-W Sep 18, 2026
849976a
Say where Guide-link coverage comes from
Lee-W Sep 18, 2026
57823f5
Say guide_url is present only when a guide documents the class
Lee-W Sep 19, 2026
e08f244
Pin the guide_url contract tests to the model, not the payload
Lee-W Sep 19, 2026
d298157
Say what provider.yaml's how-to-guide does not reach
Lee-W Sep 19, 2026
ae9cca9
Let a module's action links wrap
Lee-W Sep 19, 2026
f6ecd39
Read a release tag's guide pages in one git call
Lee-W Sep 19, 2026
e969cbb
Decode ls-tree output the same way as cat-file --batch
Lee-W Sep 28, 2026
0893fd8
Link guide pages titled "Prose: ``Class``" to their class again
Lee-W Sep 28, 2026
e6a4182
Give common.ai's remaining task decorators a guide link
Lee-W Sep 28, 2026
763245f
Stop linking parameter-table pages as a class's guide
Lee-W Oct 6, 2026
3fa88f8
Keep extract_versions module tests off real git and pin the .rst filter
Lee-W Oct 6, 2026
fe22544
Tidy comments the guide-link work left behind
Lee-W Oct 6, 2026
cac4877
Pin that a colon title names only its trailing literal run
Lee-W Oct 6, 2026
8a9c8e4
Build registry guide links from the docs at the provider's release tag
Lee-W Oct 6, 2026
3e2758f
Pin the release-tag guide docs call and cover both fallbacks to the w…
Lee-W Oct 6, 2026
72e40d0
Use the title shape the anchored guide matcher accepts in the release…
Lee-W Oct 6, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
46 changes: 45 additions & 1 deletion dev/registry/extract_parameters.py
Original file line number Diff line number Diff line change
Expand Up @@ -55,7 +55,9 @@

import yaml
from extract_metadata import fetch_provider_inventory, read_inventory
from extract_versions import detect_layout, git_tag_exists, read_guide_docs as read_guide_docs_at_tag
from registry_contract_models import validate_modules_catalog, validate_provider_parameters
from registry_tools.docs_guides import attach_guide_urls, collect_guide_anchors, is_guide_page
from registry_tools.types import (
BASE_CLASS_IMPORTS,
CLASS_LEVEL_CATEGORY_OVERRIDES,
Expand Down Expand Up @@ -98,6 +100,7 @@ class Module:
provider_name: str
supports_durable_execution: bool
supports_deferrable: bool
guide_url: str | None = None
Comment thread
Lee-W marked this conversation as resolved.


def get_category(integration_name: str) -> str:
Expand Down Expand Up @@ -738,6 +741,41 @@ def _resolve_decorated_operator_class(decorator_fn: object) -> type | None:
return candidate if inspect.isclass(candidate) else None


def read_guide_docs(docs_dir: Path) -> dict[str, str]:
"""Read a provider's authored reST docs from the working tree, keyed by path relative to ``docs_dir``."""
if not docs_dir.is_dir():
return {}
docs = {}
for path in sorted(docs_dir.rglob("*.rst")):
relative = path.relative_to(docs_dir).as_posix()
if not is_guide_page(relative):
continue
docs[relative] = path.read_text(encoding="utf-8")
return docs


def read_released_guide_docs(
provider_id: str, version: str, provider_rel_path: Path
) -> dict[str, str] | None:
"""Read a provider's guide docs at its release tag, or None when there is no tag to read.

The guide links point at ``/stable``, which serves the released docs, so the
anchors must come from the same content; the working tree may be ahead of it.
A tag from the old flat layout yields an empty dict, meaning no guide links
rather than a fallback to the working tree.
"""
if not version:
return None
tag = f"providers-{provider_id}/{version}"
if not git_tag_exists(tag):
return None
dir_path = provider_rel_path.as_posix()
layout = detect_layout(tag, dir_path)
if layout is None:
return None
return read_guide_docs_at_tag(tag, layout, dir_path)


def discover_classes_from_provider(
provider_yaml_path: Path,
base_classes: dict[str, type],
Expand All @@ -748,7 +786,8 @@ def discover_classes_from_provider(
"""Discover classes from a single provider by importing its modules at runtime.

Reads the provider.yaml to find which modules/classes to inspect, imports them,
and returns metadata for each discovered class with every `Module` dataclass field.
and returns metadata for each discovered class with every required `Module`
dataclass field, plus ``guide_url`` when a how-to guide documents the class.
"""
with open(provider_yaml_path) as f:
provider_yaml = yaml.safe_load(f)
Expand Down Expand Up @@ -985,6 +1024,11 @@ def make_entry(
}
)

guide_docs = read_released_guide_docs(provider_id, version, provider_rel_path)
if guide_docs is None:
guide_docs = read_guide_docs(provider_yaml_path.parent / "docs")
attach_guide_urls(discovered, collect_guide_anchors(guide_docs), base_docs_url)

return discovered


Expand Down
85 changes: 85 additions & 0 deletions dev/registry/extract_versions.py
Original file line number Diff line number Diff line change
Expand Up @@ -60,6 +60,7 @@
sys.exit(1)

from extract_metadata import fetch_provider_inventory, read_connection_urls, resolve_connection_docs_url
from registry_tools.docs_guides import attach_guide_urls, collect_guide_anchors, is_guide_page
from registry_tools.types import (
CLASS_LEVEL_CATEGORY_OVERRIDES,
CLASS_LEVEL_SECTIONS,
Expand Down Expand Up @@ -130,6 +131,62 @@ def git_show(tag: str, path: str) -> str | None:
return None


def git_ls_tree(tag: str, prefix: str) -> list[str]:
"""List the file paths under a prefix at a specific git tag."""
try:
result = subprocess.run(
["git", "-c", "core.quotePath=false", "ls-tree", "-r", "--name-only", tag, "--", prefix],
capture_output=True,
cwd=AIRFLOW_ROOT,
check=True,
)
except subprocess.CalledProcessError:
return []
return [line for line in result.stdout.decode("utf-8").splitlines() if line]


def git_cat_file_batch(tag: str, paths: list[str]) -> dict[str, str]:
"""Read multiple files at a specific git tag in one `git cat-file --batch` call.

Returns a mapping of path -> content for paths that exist at the tag; a path
git reports as missing is simply absent from the result, matching git_show's
"return None for a missing path" semantics.

Decode failures are left unguarded on purpose: .rst files are Sphinx
convention UTF-8, an explicit "utf-8" decode is more predictable than
following the process locale, and a UnicodeDecodeError should surface loudly
rather than being swallowed. A failing ``git cat-file`` call also raises
(``check=True``); only git_show turns CalledProcessError into ``None``.
"""
if not paths:
return {}

stdin = ("\n".join(f"{tag}:{p}" for p in paths) + "\n").encode("utf-8")
result = subprocess.run(
["git", "cat-file", "--batch"],
input=stdin,
capture_output=True,
cwd=AIRFLOW_ROOT,
check=True,
)

output = result.stdout
pos = 0
contents: dict[str, str] = {}
for path in paths:
newline_idx = output.index(b"\n", pos)
header = output[pos:newline_idx].decode("utf-8")
pos = newline_idx + 1
if header.endswith(" missing"):
continue
_sha1, _obj_type, size_str = header.split(" ")
size = int(size_str)
content_bytes = output[pos : pos + size]
pos += size + 1 # skip the protocol's trailing LF, which isn't counted in size
contents[path] = content_bytes.decode("utf-8")
return contents


def git_tag_exists(tag: str) -> bool:
"""Check if a git tag exists locally."""
result = subprocess.run(
Expand Down Expand Up @@ -182,6 +239,32 @@ def get_source_file_path(layout: str, dir_path: str, module_path: str) -> str:
return f"providers/src/{rel_file}"


def read_guide_docs(tag: str, layout: str, dir_path: str) -> dict[str, str]:
"""Read a provider's authored reST docs at a tag, keyed by path relative to its docs dir.

Only the per-provider layout keeps docs beside the provider; under the old flat
layout they lived in a top-level ``docs/`` tree, so those tags get no guide
links rather than links guessed from a path that moved.
"""
if layout != "new":
return {}

docs_prefix = f"providers/{dir_path}/docs/"
survivors: list[tuple[str, str]] = []
for path in git_ls_tree(tag, docs_prefix):
if not path.endswith(".rst"):
Comment thread
Lee-W marked this conversation as resolved.
continue
relative = path[len(docs_prefix) :]
if not is_guide_page(relative):
continue
survivors.append((relative, path))

batch_result = git_cat_file_batch(tag, [full_path for _relative, full_path in survivors])
return {
relative: batch_result[full_path] for relative, full_path in survivors if batch_result.get(full_path)
}


def parse_pyproject_toml_content(content: str, layout: str) -> dict[str, Any]:
"""Parse pyproject.toml content for dependencies, requires-python, and extras."""
result: dict[str, Any] = {"requires_python": "", "dependencies": [], "optional_extras": {}}
Expand Down Expand Up @@ -383,6 +466,8 @@ def process_module(module_path: str, module_type: str, integration: str, categor
}
)

attach_guide_urls(modules, collect_guide_anchors(read_guide_docs(tag, layout, dir_path)), base_docs_url)

return modules


Expand Down
3 changes: 3 additions & 0 deletions dev/registry/registry_contract_models.py
Original file line number Diff line number Diff line change
Expand Up @@ -146,6 +146,9 @@ class ModuleContract(BaseModel):
provider_name: str | None = None
supports_durable_execution: bool = False
supports_deferrable: bool = False
# Only set for classes and task decorators (e.g. ``@task.agent``) that a how-to
# guide documents in a section of their own.
guide_url: str | None = None


class ModulesCatalogContract(BaseModel):
Expand Down
186 changes: 186 additions & 0 deletions dev/registry/registry_tools/docs_guides.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,186 @@
# Licensed to the Apache Software Foundation (ASF) under one
# or more contributor license agreements. See the NOTICE file
# distributed with this work for additional information
# regarding copyright ownership. The ASF licenses this file
# to you under the Apache License, Version 2.0 (the
# "License"); you may not use this file except in compliance
# with the License. You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing,
# software distributed under the License is distributed on an
# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
# KIND, either express or implied. See the License for the
# specific language governing permissions and limitations
# under the License.
"""Map a provider's classes to the how-to guide sections that document them.

A module's ``docs_url`` points at generated API reference, which tells a reader
what the arguments are but not how the thing is meant to be used. The prose
guides carry that, and they already mark it: a how-to guide documents one class
(or a class and its task-flow decorator) per section, titled with the name(s)
either at the start (``HookToolset``, ``SQLToolset``, ``AgentOperator`` &
``@task.agent``) or after a colon at the very end, as in a section titled
"Airflow hooks as tools: ``HookToolset``".

So the mapping is read back out of the guides rather than curated anywhere: a
hand-maintained name-to-guide table would rot silently every time a guide is
split, renamed, or a class is dropped, and a rotten link is worse than none.
Callers supply the reST they can see (a git tag, or the working tree) and get
back only the anchors those sources actually contain.
"""

from __future__ import annotations

import re
from collections.abc import Mapping
from pathlib import PurePosixPath
from typing import Any

# reST underlines an (optionally overlined) section title with a run of one
# punctuation character, at least as long as the title itself.
_ADORNMENT_CHARS = "!\"#$%&'()*+,-./:;<=>?@[\\]^_`{|}~"

_SKIPPED_PAGE_NAMES = frozenset({"changelog.rst", "commits.rst"})


def is_guide_page(relative_path: str) -> bool:
"""Whether a path relative to a provider's docs directory is a how-to guide page.

Callers hand every ``.rst`` they can see to this before it ever reaches
``collect_guide_anchors``. Two kinds of real, built pages must not go
further:

- Anything under a ``_``-prefixed path segment, at any depth
(``_api/hook/index.rst``, ``operators/_partials/foo.rst``, top-level
``_partials/foo.rst``): Sphinx/autoapi output and partials are directive
markup, not the hand-written, reST-underlined titles this module's
inline-literal title convention parses.
- ``changelog.rst`` and ``commits.rst``: real release-note pages, not
how-to guides, that can carry inline-literal-formatted headings by
coincidence.
"""
path = PurePosixPath(relative_path)
if any(part.startswith("_") for part in path.parts):
return False
return path.name not in _SKIPPED_PAGE_NAMES


# A single inline-literal name: a class (``HookToolset``) or a task-flow
# decorator (``@task.llm_file_analysis``) -- narrow enough that it still can't
# match arbitrary prose wrapped in backticks.
_INLINE_LITERAL_NAME = r"``(@?[A-Za-z_][A-Za-z0-9_]*(?:\.[A-Za-z_][A-Za-z0-9_]*)*)``"
_INLINE_LITERAL_NAME_RE = re.compile(_INLINE_LITERAL_NAME)

# Only titles that consist solely of, or end a colon-led clause with, a run of
# inline-literal names are treated as documenting them, so prose headings
# ("Bounded query results") never produce a link. A run is one or more names
# joined by "&", ",", "/" or "and" -- how guides write a section that covers both
# an operator and its decorator (``AgentOperator`` & ``@task.agent``).
_NAME_SEPARATOR = r"(?:\s*[&,/]\s*|\s+and\s+)"
_NAME_RUN = rf"{_INLINE_LITERAL_NAME}(?:{_NAME_SEPARATOR}{_INLINE_LITERAL_NAME})*"

# Shape one (what older release tags' docs use):
# the title is nothing but the name run. Anchoring to "$" keeps a title that
# merely opens with a literal and continues in prose ("``SandboxToolset``
# parameters") from claiming to document that class.
_LEADING_LITERAL_NAME_RUN = re.compile(rf"^{_NAME_RUN}\s*$")
# Shape two (what current docs use): a prose lead-in, a colon, then the name run runs to
# the very end of the title. Anchoring to "$" is what keeps a colon earlier in
# the title, with prose after it, from being mistaken for this shape.
_TRAILING_LITERAL_NAME_RUN = re.compile(rf":\s+{_NAME_RUN}\s*$")


def slugify_section_anchor(title: str) -> str:
"""Return the HTML id Sphinx gives a section with this title.

Mirrors docutils' ``make_id``: lower-case, every run of non-alphanumeric
characters becomes a single hyphen, and leading/trailing hyphens are
dropped -- e.g. the section titled ``HookToolset`` is served at
``#hooktoolset``.
"""
return re.sub(r"[^a-z0-9]+", "-", title.lower()).strip("-")


def _extract_names_from_title(title: str) -> list[str]:
"""Return the names a section title documents, or [] if it names prose.

A guide marks a section as being *about* one or more names by titling it
with them as inline literals, either as the whole title (``HookToolset``, or
``AgentOperator`` & ``@task.agent`` where one section covers the operator
and its decorator) or after a colon at the very end ("Airflow hooks as
tools: ``HookToolset``"). Requiring that markup is what keeps a single-word
prose heading ("Guidelines") -- or a literal appearing elsewhere in a prose
title -- from claiming to document a class of the same name, and it is a
convention the guides already follow rather than one imposed on them.
"""
match = _LEADING_LITERAL_NAME_RUN.match(title) or _TRAILING_LITERAL_NAME_RUN.search(title)
return _INLINE_LITERAL_NAME_RE.findall(match.group(0)) if match else []


def _is_adornment(line: str) -> bool:
"""Whether a line is a reST title overline/underline rather than a title."""
return bool(line) and len(set(line)) == 1 and line[0] in _ADORNMENT_CHARS


def _extract_section_titles(text: str) -> list[str]:
"""Return every section title in a reST document, in document order."""
titles = []
lines = text.splitlines()
for index, line in enumerate(lines[:-1]):
title = line.strip()
# Guides title these sections with an inline literal (``HookToolset``),
# so a title can legitimately start with an adornment character; only a
# line that is *entirely* one repeated character is an adornment.
if not title or _is_adornment(title):
continue
underline = lines[index + 1].strip()
if len(underline) >= len(title) and _is_adornment(underline):
titles.append(title)
return titles


def collect_guide_anchors(docs: Mapping[str, str]) -> dict[str, str]:
"""Map name -> ``<page>.html#<anchor>`` for every documented class or decorator.

``docs`` maps a page path relative to the provider's docs directory (e.g.
``toolsets.rst``) to its reST source. A title can name more than one name
(``AgentOperator`` & ``@task.agent``), in which case every one gets the
same anchor. When two pages document the same name: a page's own title
(its first section) beats a subsection found on any other page, since that
page is the one dedicated to the class; among two page titles -- or two
subsections neither page titles -- the first page in sorted order wins, so
a rebuild of the same sources always produces the same link. "Page title"
is simply the first title _extract_section_titles finds, not a checked
top-level adornment, so a heading-shaped block earlier on the page (say,
inside a directive) would take that role.
"""
found: dict[str, tuple[bool, str]] = {} # name -> (from a page title?, anchor)
for page in sorted(docs):
page_url = re.sub(r"\.rst$", ".html", page)
for index, title in enumerate(_extract_section_titles(docs[page])):
is_page_title = index == 0
for name in _extract_names_from_title(title):
current = found.get(name)
# A page title replaces a subsection found earlier; nothing else
# replaces what was found first, so rebuilds stay deterministic.
if current is not None and (current[0] or not is_page_title):
continue
found[name] = (is_page_title, f"{page_url}#{slugify_section_anchor(title)}")
return {name: anchor for name, (_, anchor) in found.items()}


def attach_guide_urls(modules: list[dict[str, Any]], anchors: Mapping[str, str], base_docs_url: str) -> int:
"""Set ``guide_url`` on every module a guide section documents.

Mutates ``modules`` in place; returns how many got a link.
"""
attached = 0
for module in modules:
anchor = anchors.get(module["name"])
if not anchor:
continue
module["guide_url"] = f"{base_docs_url.rstrip('/')}/{anchor}"
attached += 1
return attached
Loading