diff --git a/dev/registry/extract_parameters.py b/dev/registry/extract_parameters.py index 7f776ae164599..ad3813a4bd21d 100644 --- a/dev/registry/extract_parameters.py +++ b/dev/registry/extract_parameters.py @@ -55,7 +55,9 @@ import yaml from extract_metadata import fetch_provider_inventory, read_inventory +from extract_versions import detect_layout, git_tag_exists, read_guide_docs as read_guide_docs_at_tag from registry_contract_models import validate_modules_catalog, validate_provider_parameters +from registry_tools.docs_guides import attach_guide_urls, collect_guide_anchors, is_guide_page from registry_tools.types import ( BASE_CLASS_IMPORTS, CLASS_LEVEL_CATEGORY_OVERRIDES, @@ -98,6 +100,7 @@ class Module: provider_name: str supports_durable_execution: bool supports_deferrable: bool + guide_url: str | None = None def get_category(integration_name: str) -> str: @@ -738,6 +741,41 @@ def _resolve_decorated_operator_class(decorator_fn: object) -> type | None: return candidate if inspect.isclass(candidate) else None +def read_guide_docs(docs_dir: Path) -> dict[str, str]: + """Read a provider's authored reST docs from the working tree, keyed by path relative to ``docs_dir``.""" + if not docs_dir.is_dir(): + return {} + docs = {} + for path in sorted(docs_dir.rglob("*.rst")): + relative = path.relative_to(docs_dir).as_posix() + if not is_guide_page(relative): + continue + docs[relative] = path.read_text(encoding="utf-8") + return docs + + +def read_released_guide_docs( + provider_id: str, version: str, provider_rel_path: Path +) -> dict[str, str] | None: + """Read a provider's guide docs at its release tag, or None when there is no tag to read. + + The guide links point at ``/stable``, which serves the released docs, so the + anchors must come from the same content; the working tree may be ahead of it. + A tag from the old flat layout yields an empty dict, meaning no guide links + rather than a fallback to the working tree. + """ + if not version: + return None + tag = f"providers-{provider_id}/{version}" + if not git_tag_exists(tag): + return None + dir_path = provider_rel_path.as_posix() + layout = detect_layout(tag, dir_path) + if layout is None: + return None + return read_guide_docs_at_tag(tag, layout, dir_path) + + def discover_classes_from_provider( provider_yaml_path: Path, base_classes: dict[str, type], @@ -748,7 +786,8 @@ def discover_classes_from_provider( """Discover classes from a single provider by importing its modules at runtime. Reads the provider.yaml to find which modules/classes to inspect, imports them, - and returns metadata for each discovered class with every `Module` dataclass field. + and returns metadata for each discovered class with every required `Module` + dataclass field, plus ``guide_url`` when a how-to guide documents the class. """ with open(provider_yaml_path) as f: provider_yaml = yaml.safe_load(f) @@ -985,6 +1024,11 @@ def make_entry( } ) + guide_docs = read_released_guide_docs(provider_id, version, provider_rel_path) + if guide_docs is None: + guide_docs = read_guide_docs(provider_yaml_path.parent / "docs") + attach_guide_urls(discovered, collect_guide_anchors(guide_docs), base_docs_url) + return discovered diff --git a/dev/registry/extract_versions.py b/dev/registry/extract_versions.py index d831725f1c31e..8c73192d8ff18 100644 --- a/dev/registry/extract_versions.py +++ b/dev/registry/extract_versions.py @@ -60,6 +60,7 @@ sys.exit(1) from extract_metadata import fetch_provider_inventory, read_connection_urls, resolve_connection_docs_url +from registry_tools.docs_guides import attach_guide_urls, collect_guide_anchors, is_guide_page from registry_tools.types import ( CLASS_LEVEL_CATEGORY_OVERRIDES, CLASS_LEVEL_SECTIONS, @@ -130,6 +131,62 @@ def git_show(tag: str, path: str) -> str | None: return None +def git_ls_tree(tag: str, prefix: str) -> list[str]: + """List the file paths under a prefix at a specific git tag.""" + try: + result = subprocess.run( + ["git", "-c", "core.quotePath=false", "ls-tree", "-r", "--name-only", tag, "--", prefix], + capture_output=True, + cwd=AIRFLOW_ROOT, + check=True, + ) + except subprocess.CalledProcessError: + return [] + return [line for line in result.stdout.decode("utf-8").splitlines() if line] + + +def git_cat_file_batch(tag: str, paths: list[str]) -> dict[str, str]: + """Read multiple files at a specific git tag in one `git cat-file --batch` call. + + Returns a mapping of path -> content for paths that exist at the tag; a path + git reports as missing is simply absent from the result, matching git_show's + "return None for a missing path" semantics. + + Decode failures are left unguarded on purpose: .rst files are Sphinx + convention UTF-8, an explicit "utf-8" decode is more predictable than + following the process locale, and a UnicodeDecodeError should surface loudly + rather than being swallowed. A failing ``git cat-file`` call also raises + (``check=True``); only git_show turns CalledProcessError into ``None``. + """ + if not paths: + return {} + + stdin = ("\n".join(f"{tag}:{p}" for p in paths) + "\n").encode("utf-8") + result = subprocess.run( + ["git", "cat-file", "--batch"], + input=stdin, + capture_output=True, + cwd=AIRFLOW_ROOT, + check=True, + ) + + output = result.stdout + pos = 0 + contents: dict[str, str] = {} + for path in paths: + newline_idx = output.index(b"\n", pos) + header = output[pos:newline_idx].decode("utf-8") + pos = newline_idx + 1 + if header.endswith(" missing"): + continue + _sha1, _obj_type, size_str = header.split(" ") + size = int(size_str) + content_bytes = output[pos : pos + size] + pos += size + 1 # skip the protocol's trailing LF, which isn't counted in size + contents[path] = content_bytes.decode("utf-8") + return contents + + def git_tag_exists(tag: str) -> bool: """Check if a git tag exists locally.""" result = subprocess.run( @@ -182,6 +239,32 @@ def get_source_file_path(layout: str, dir_path: str, module_path: str) -> str: return f"providers/src/{rel_file}" +def read_guide_docs(tag: str, layout: str, dir_path: str) -> dict[str, str]: + """Read a provider's authored reST docs at a tag, keyed by path relative to its docs dir. + + Only the per-provider layout keeps docs beside the provider; under the old flat + layout they lived in a top-level ``docs/`` tree, so those tags get no guide + links rather than links guessed from a path that moved. + """ + if layout != "new": + return {} + + docs_prefix = f"providers/{dir_path}/docs/" + survivors: list[tuple[str, str]] = [] + for path in git_ls_tree(tag, docs_prefix): + if not path.endswith(".rst"): + continue + relative = path[len(docs_prefix) :] + if not is_guide_page(relative): + continue + survivors.append((relative, path)) + + batch_result = git_cat_file_batch(tag, [full_path for _relative, full_path in survivors]) + return { + relative: batch_result[full_path] for relative, full_path in survivors if batch_result.get(full_path) + } + + def parse_pyproject_toml_content(content: str, layout: str) -> dict[str, Any]: """Parse pyproject.toml content for dependencies, requires-python, and extras.""" result: dict[str, Any] = {"requires_python": "", "dependencies": [], "optional_extras": {}} @@ -383,6 +466,8 @@ def process_module(module_path: str, module_type: str, integration: str, categor } ) + attach_guide_urls(modules, collect_guide_anchors(read_guide_docs(tag, layout, dir_path)), base_docs_url) + return modules diff --git a/dev/registry/registry_contract_models.py b/dev/registry/registry_contract_models.py index 31f4001dc4927..02eb221b1ce10 100644 --- a/dev/registry/registry_contract_models.py +++ b/dev/registry/registry_contract_models.py @@ -146,6 +146,9 @@ class ModuleContract(BaseModel): provider_name: str | None = None supports_durable_execution: bool = False supports_deferrable: bool = False + # Only set for classes and task decorators (e.g. ``@task.agent``) that a how-to + # guide documents in a section of their own. + guide_url: str | None = None class ModulesCatalogContract(BaseModel): diff --git a/dev/registry/registry_tools/docs_guides.py b/dev/registry/registry_tools/docs_guides.py new file mode 100644 index 0000000000000..d19f97757040f --- /dev/null +++ b/dev/registry/registry_tools/docs_guides.py @@ -0,0 +1,186 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. +"""Map a provider's classes to the how-to guide sections that document them. + +A module's ``docs_url`` points at generated API reference, which tells a reader +what the arguments are but not how the thing is meant to be used. The prose +guides carry that, and they already mark it: a how-to guide documents one class +(or a class and its task-flow decorator) per section, titled with the name(s) +either at the start (``HookToolset``, ``SQLToolset``, ``AgentOperator`` & +``@task.agent``) or after a colon at the very end, as in a section titled +"Airflow hooks as tools: ``HookToolset``". + +So the mapping is read back out of the guides rather than curated anywhere: a +hand-maintained name-to-guide table would rot silently every time a guide is +split, renamed, or a class is dropped, and a rotten link is worse than none. +Callers supply the reST they can see (a git tag, or the working tree) and get +back only the anchors those sources actually contain. +""" + +from __future__ import annotations + +import re +from collections.abc import Mapping +from pathlib import PurePosixPath +from typing import Any + +# reST underlines an (optionally overlined) section title with a run of one +# punctuation character, at least as long as the title itself. +_ADORNMENT_CHARS = "!\"#$%&'()*+,-./:;<=>?@[\\]^_`{|}~" + +_SKIPPED_PAGE_NAMES = frozenset({"changelog.rst", "commits.rst"}) + + +def is_guide_page(relative_path: str) -> bool: + """Whether a path relative to a provider's docs directory is a how-to guide page. + + Callers hand every ``.rst`` they can see to this before it ever reaches + ``collect_guide_anchors``. Two kinds of real, built pages must not go + further: + + - Anything under a ``_``-prefixed path segment, at any depth + (``_api/hook/index.rst``, ``operators/_partials/foo.rst``, top-level + ``_partials/foo.rst``): Sphinx/autoapi output and partials are directive + markup, not the hand-written, reST-underlined titles this module's + inline-literal title convention parses. + - ``changelog.rst`` and ``commits.rst``: real release-note pages, not + how-to guides, that can carry inline-literal-formatted headings by + coincidence. + """ + path = PurePosixPath(relative_path) + if any(part.startswith("_") for part in path.parts): + return False + return path.name not in _SKIPPED_PAGE_NAMES + + +# A single inline-literal name: a class (``HookToolset``) or a task-flow +# decorator (``@task.llm_file_analysis``) -- narrow enough that it still can't +# match arbitrary prose wrapped in backticks. +_INLINE_LITERAL_NAME = r"``(@?[A-Za-z_][A-Za-z0-9_]*(?:\.[A-Za-z_][A-Za-z0-9_]*)*)``" +_INLINE_LITERAL_NAME_RE = re.compile(_INLINE_LITERAL_NAME) + +# Only titles that consist solely of, or end a colon-led clause with, a run of +# inline-literal names are treated as documenting them, so prose headings +# ("Bounded query results") never produce a link. A run is one or more names +# joined by "&", ",", "/" or "and" -- how guides write a section that covers both +# an operator and its decorator (``AgentOperator`` & ``@task.agent``). +_NAME_SEPARATOR = r"(?:\s*[&,/]\s*|\s+and\s+)" +_NAME_RUN = rf"{_INLINE_LITERAL_NAME}(?:{_NAME_SEPARATOR}{_INLINE_LITERAL_NAME})*" + +# Shape one (what older release tags' docs use): +# the title is nothing but the name run. Anchoring to "$" keeps a title that +# merely opens with a literal and continues in prose ("``SandboxToolset`` +# parameters") from claiming to document that class. +_LEADING_LITERAL_NAME_RUN = re.compile(rf"^{_NAME_RUN}\s*$") +# Shape two (what current docs use): a prose lead-in, a colon, then the name run runs to +# the very end of the title. Anchoring to "$" is what keeps a colon earlier in +# the title, with prose after it, from being mistaken for this shape. +_TRAILING_LITERAL_NAME_RUN = re.compile(rf":\s+{_NAME_RUN}\s*$") + + +def slugify_section_anchor(title: str) -> str: + """Return the HTML id Sphinx gives a section with this title. + + Mirrors docutils' ``make_id``: lower-case, every run of non-alphanumeric + characters becomes a single hyphen, and leading/trailing hyphens are + dropped -- e.g. the section titled ``HookToolset`` is served at + ``#hooktoolset``. + """ + return re.sub(r"[^a-z0-9]+", "-", title.lower()).strip("-") + + +def _extract_names_from_title(title: str) -> list[str]: + """Return the names a section title documents, or [] if it names prose. + + A guide marks a section as being *about* one or more names by titling it + with them as inline literals, either as the whole title (``HookToolset``, or + ``AgentOperator`` & ``@task.agent`` where one section covers the operator + and its decorator) or after a colon at the very end ("Airflow hooks as + tools: ``HookToolset``"). Requiring that markup is what keeps a single-word + prose heading ("Guidelines") -- or a literal appearing elsewhere in a prose + title -- from claiming to document a class of the same name, and it is a + convention the guides already follow rather than one imposed on them. + """ + match = _LEADING_LITERAL_NAME_RUN.match(title) or _TRAILING_LITERAL_NAME_RUN.search(title) + return _INLINE_LITERAL_NAME_RE.findall(match.group(0)) if match else [] + + +def _is_adornment(line: str) -> bool: + """Whether a line is a reST title overline/underline rather than a title.""" + return bool(line) and len(set(line)) == 1 and line[0] in _ADORNMENT_CHARS + + +def _extract_section_titles(text: str) -> list[str]: + """Return every section title in a reST document, in document order.""" + titles = [] + lines = text.splitlines() + for index, line in enumerate(lines[:-1]): + title = line.strip() + # Guides title these sections with an inline literal (``HookToolset``), + # so a title can legitimately start with an adornment character; only a + # line that is *entirely* one repeated character is an adornment. + if not title or _is_adornment(title): + continue + underline = lines[index + 1].strip() + if len(underline) >= len(title) and _is_adornment(underline): + titles.append(title) + return titles + + +def collect_guide_anchors(docs: Mapping[str, str]) -> dict[str, str]: + """Map name -> ``.html#`` for every documented class or decorator. + + ``docs`` maps a page path relative to the provider's docs directory (e.g. + ``toolsets.rst``) to its reST source. A title can name more than one name + (``AgentOperator`` & ``@task.agent``), in which case every one gets the + same anchor. When two pages document the same name: a page's own title + (its first section) beats a subsection found on any other page, since that + page is the one dedicated to the class; among two page titles -- or two + subsections neither page titles -- the first page in sorted order wins, so + a rebuild of the same sources always produces the same link. "Page title" + is simply the first title _extract_section_titles finds, not a checked + top-level adornment, so a heading-shaped block earlier on the page (say, + inside a directive) would take that role. + """ + found: dict[str, tuple[bool, str]] = {} # name -> (from a page title?, anchor) + for page in sorted(docs): + page_url = re.sub(r"\.rst$", ".html", page) + for index, title in enumerate(_extract_section_titles(docs[page])): + is_page_title = index == 0 + for name in _extract_names_from_title(title): + current = found.get(name) + # A page title replaces a subsection found earlier; nothing else + # replaces what was found first, so rebuilds stay deterministic. + if current is not None and (current[0] or not is_page_title): + continue + found[name] = (is_page_title, f"{page_url}#{slugify_section_anchor(title)}") + return {name: anchor for name, (_, anchor) in found.items()} + + +def attach_guide_urls(modules: list[dict[str, Any]], anchors: Mapping[str, str], base_docs_url: str) -> int: + """Set ``guide_url`` on every module a guide section documents. + + Mutates ``modules`` in place; returns how many got a link. + """ + attached = 0 + for module in modules: + anchor = anchors.get(module["name"]) + if not anchor: + continue + module["guide_url"] = f"{base_docs_url.rstrip('/')}/{anchor}" + attached += 1 + return attached diff --git a/dev/registry/tests/test_docs_guides.py b/dev/registry/tests/test_docs_guides.py new file mode 100644 index 0000000000000..53fcfe4367d89 --- /dev/null +++ b/dev/registry/tests/test_docs_guides.py @@ -0,0 +1,369 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. +from __future__ import annotations + +from unittest.mock import patch + +import pytest +from extract_parameters import read_guide_docs as read_guide_docs_from_worktree +from extract_versions import read_guide_docs as read_guide_docs_from_tag +from registry_tools.docs_guides import ( + attach_guide_urls, + collect_guide_anchors, + is_guide_page, + slugify_section_anchor, +) + +TOOLSETS_GUIDE = """ +.. _howto/toolsets: + +Toolsets: Airflow hooks as AI agent tools +========================================== + +Intro prose. + +Airflow hooks as tools: ``HookToolset`` +---------------------------------------- + +How to use it. + +Guidelines +^^^^^^^^^^ + +More prose. + +.. _bounded-query-results: + +Bounded query results +^^^^^^^^^^^^^^^^^^^^^ + +``SQLToolset`` bounds that. + +Files with DataFusion: ``DataFusionToolset`` +----------------------------------------------- + +Another one. +""" + + +@pytest.mark.parametrize( + ("title", "expected"), + [ + # Verified against the published guide: the section titled ``HookToolset`` + # is served at .../toolsets.html#hooktoolset. + ("HookToolset", "hooktoolset"), + ("AgentSkillsToolset", "agentskillstoolset"), + ("Bounded query results", "bounded-query-results"), + ("Agent_Skills", "agent-skills"), + ("Direct PydanticAI MCP toolsets", "direct-pydanticai-mcp-toolsets"), + ("``AgentOperator`` & ``@task.agent``", "agentoperator-task-agent"), + ], +) +def test_slugify_section_anchor_matches_sphinx_ids(title, expected): + assert slugify_section_anchor(title) == expected + + +def test_collect_guide_anchors_finds_class_named_sections_at_any_depth(): + anchors = collect_guide_anchors({"toolsets.rst": TOOLSETS_GUIDE}) + + assert anchors == { + "HookToolset": "toolsets.html#airflow-hooks-as-tools-hooktoolset", + "DataFusionToolset": "toolsets.html#files-with-datafusion-datafusiontoolset", + } + + +def test_collect_guide_anchors_ignores_prose_headings(): + anchors = collect_guide_anchors({"toolsets.rst": TOOLSETS_GUIDE}) + + # "Guidelines" is shaped like a class name but isn't marked up as one. + assert "Guidelines" not in anchors + assert "Bounded query results" not in anchors + + +def test_collect_guide_anchors_ignores_classes_only_mentioned_in_prose(): + # SQLToolset appears in the guide's body but has no section of its own, so + # there is no anchor to link to. + assert "SQLToolset" not in collect_guide_anchors({"toolsets.rst": TOOLSETS_GUIDE}) + + +def test_collect_guide_anchors_keeps_nested_page_paths(): + guide = "Agents with tools: ``AgentOperator``\n-------------------------------------\n\nProse.\n" + + assert collect_guide_anchors({"operators/agent.rst": guide}) == { + "AgentOperator": "operators/agent.html#agents-with-tools-agentoperator" + } + + +def test_collect_guide_anchors_ignores_a_title_that_opens_with_a_name_then_continues_in_prose(): + guide = "``SandboxToolset`` parameters\n-----------------------------\n\nA parameter table.\n" + + assert collect_guide_anchors({"sandbox/configuration.rst": guide}) == {} + + +def test_collect_guide_anchors_names_only_the_trailing_run_of_a_colon_title(): + guide = "``A``: ``B``\n-------------\n\nProse.\n" + + assert collect_guide_anchors({"page.rst": guide}) == {"B": "page.html#a-b"} + + +def test_collect_guide_anchors_prefers_first_sorted_page_among_page_titles(): + # Both pages title themselves after SQLToolset, so neither title beats the + # other on that basis alone; the tie is broken by sorted page order. + guide = "SQL databases: ``SQLToolset``\n------------------------------\n\nProse.\n" + + anchors = collect_guide_anchors({"toolsets.rst": guide, "operators/sql.rst": guide}) + + assert anchors["SQLToolset"] == "operators/sql.html#sql-databases-sqltoolset" + + +def test_collect_guide_anchors_handles_a_title_covering_more_than_the_class(): + # Verified against the published guide: this heading is served at + # .../operators/agent.html#agentoperator-task-agent, so the anchor comes from + # the whole title while both the operator and its decorator get linked to it. + # This is the older leading shape; older release tags' docs (read by + # extract_versions.py) still use it, so it must keep working. + guide = "``AgentOperator`` & ``@task.agent``\n===================================\n\nProse.\n" + + assert collect_guide_anchors({"operators/agent.rst": guide}) == { + "AgentOperator": "operators/agent.html#agentoperator-task-agent", + "@task.agent": "operators/agent.html#agentoperator-task-agent", + } + + +def test_collect_guide_anchors_links_a_decorator_name_with_underscores(): + # ``@task.llm_file_analysis`` is the real decorator name for the common.ai + # provider's LLMFileAnalysisOperator; the leading-literal charset must admit + # "@" and "." for it to ever get a link. Also a leading-shape fixture kept for + # the same reason as the test above: older release tags' docs still use it. + guide = ( + "``LLMFileAnalysisOperator`` & ``@task.llm_file_analysis``\n" + "==========================================================\n\n" + "Prose.\n" + ) + + assert collect_guide_anchors({"operators/llm_file_analysis.rst": guide}) == { + "LLMFileAnalysisOperator": ( + "operators/llm_file_analysis.html#llmfileanalysisoperator-task-llm-file-analysis" + ), + "@task.llm_file_analysis": ( + "operators/llm_file_analysis.html#llmfileanalysisoperator-task-llm-file-analysis" + ), + } + + +def test_collect_guide_anchors_ignores_a_prose_title_mentioning_a_literal(): + # The title neither opens with the literal nor ends a colon-led clause with + # it, so a prose heading that happens to mention one in passing -- anywhere + # in the title -- must not produce a link. + guide = "Using ``foo`` in a pipeline\n============================\n\nProse.\n" + + assert collect_guide_anchors({"toolsets.rst": guide}) == {} + + +def test_collect_guide_anchors_requires_a_long_enough_underline(): + # An underline shorter than the title isn't a section in reST, so it must not + # produce a link to an anchor Sphinx never emitted. + guide = "Airflow hooks as tools: ``HookToolset``\n---\n\nProse.\n" + + assert collect_guide_anchors({"toolsets.rst": guide}) == {} + + +def test_collect_guide_anchors_reads_a_name_trailing_a_colon(): + guide = "Airflow hooks as tools: ``HookToolset``\n========================================\n\nProse.\n" + + assert collect_guide_anchors({"toolsets/hook.rst": guide}) == { + "HookToolset": "toolsets/hook.html#airflow-hooks-as-tools-hooktoolset" + } + + +def test_collect_guide_anchors_reads_two_names_joined_by_and(): + guide = ( + "Agents with tools: ``AgentOperator`` and ``@task.agent``\n" + "=========================================================\n\n" + "Prose.\n" + ) + + assert collect_guide_anchors({"operators/agent.rst": guide}) == { + "AgentOperator": "operators/agent.html#agents-with-tools-agentoperator-and-task-agent", + "@task.agent": "operators/agent.html#agents-with-tools-agentoperator-and-task-agent", + } + + +@pytest.mark.parametrize( + "title", + [ + # Leading shape, one separator per case. + "``A`` & ``B``", + "``A``, ``B``", + "``A``/``B``", + "``A`` and ``B``", + # Trailing shape, one separator per case. + "Both: ``A`` & ``B``", + "Both: ``A``, ``B``", + "Both: ``A``/``B``", + "Both: ``A`` and ``B``", + ], +) +def test_collect_guide_anchors_accepts_every_separator_in_both_title_shapes(title): + underline = "=" * (len(title) + 1) + + anchors = collect_guide_anchors({"page.rst": f"{title}\n{underline}\n\nProse.\n"}) + + expected_anchor = f"page.html#{slugify_section_anchor(title)}" + assert anchors == {"A": expected_anchor, "B": expected_anchor} + + +def test_collect_guide_anchors_prefers_a_page_title_over_an_earlier_pages_subsection(): + # A subsection on an unrelated page happens to be titled after the class, + # but a later page is dedicated to it. + docs = { + "agent_security.rst": ( + "Securing agent tools\n" + "=====================\n\n" + "``HookToolset`` guidelines\n" + "^^^^^^^^^^^^^^^^^^^^^^^^^^^\n\n" + "Prose.\n" + ), + "toolsets/hook.rst": ( + "Airflow hooks as tools: ``HookToolset``\n========================================\n\nProse.\n" + ), + } + + assert collect_guide_anchors(docs) == { + "HookToolset": "toolsets/hook.html#airflow-hooks-as-tools-hooktoolset" + } + + +def test_collect_guide_anchors_prefers_the_page_title_over_its_own_subsection(): + guide = ( + "Batch processing: ``LLMBatchOperator``\n" + "=======================================\n\n" + "How it works.\n\n" + "``LLMBatchOperator`` or the vendor batch operators?\n" + "----------------------------------------------------\n\n" + "Prose.\n" + ) + + assert collect_guide_anchors({"operators/llm_batch.rst": guide}) == { + "LLMBatchOperator": "operators/llm_batch.html#batch-processing-llmbatchoperator" + } + + +def test_collect_guide_anchors_keeps_the_first_subsection_when_no_page_title_names_it(): + docs = { + "a.rst": "Prose page\n===========\n\n``X``\n-----\n", + "b.rst": "Other page\n===========\n\n``X``\n-----\n", + } + + assert collect_guide_anchors(docs) == {"X": "a.html#x"} + + +@pytest.mark.parametrize( + "title", + [ + # Colon present, but prose follows it before the literal -- must not be + # scanned for a literal anywhere after the colon. + "Role in a Dag: use ``MCPToolset``, not the hook directly", + # Title ends with the literal, but there's no colon to lead it. + "Using HITL review with ``AgentOperator``", + # Colon immediately precedes the literal, but prose follows it -- the + # literal doesn't reach the end of the title. + "Guidelines: ``HookToolset`` and its allow-list", + ], +) +def test_collect_guide_anchors_ignores_a_literal_that_does_not_end_the_title(title): + underline = "=" * (len(title) + 1) + + assert collect_guide_anchors({"toolsets.rst": f"{title}\n{underline}\n\nProse.\n"}) == {} + + +def test_attach_guide_urls_only_links_documented_classes(): + modules = [ + {"name": "HookToolset", "docs_url": "https://example.test/_api/hook/index.html"}, + {"name": "UndocumentedToolset", "docs_url": "https://example.test/_api/other/index.html"}, + ] + + attached = attach_guide_urls( + modules, + {"HookToolset": "toolsets.html#hooktoolset"}, + "https://airflow.apache.org/docs/apache-airflow-providers-common-ai/0.7.0", + ) + + assert attached == 1 + assert modules[0]["guide_url"] == ( + "https://airflow.apache.org/docs/apache-airflow-providers-common-ai/0.7.0/toolsets.html#hooktoolset" + ) + assert "guide_url" not in modules[1] + + +def test_attach_guide_urls_does_not_double_up_the_base_separator(): + modules = [{"name": "HookToolset"}] + + attach_guide_urls(modules, {"HookToolset": "toolsets.html#hooktoolset"}, "https://example.test/docs/") + + assert modules[0]["guide_url"] == "https://example.test/docs/toolsets.html#hooktoolset" + + +@pytest.mark.parametrize( + ("relative_path", "expected"), + [ + # A `_`-prefixed path segment marks autoapi/partial content, at any depth. + ("_api/index.rst", False), + ("_api/hook/index.rst", False), + ("operators/_partials/foo.rst", False), + ("_partials/foo.rst", False), + # Real, built release-note pages, not how-to guides. + ("changelog.rst", False), + ("commits.rst", False), + ("toolsets.rst", True), + ("operators/agent.rst", True), + ], +) +def test_is_guide_page(relative_path, expected): + assert is_guide_page(relative_path) == expected + + +def test_readers_agree_on_which_pages_are_guides(tmp_path): + """Both `read_guide_docs` implementations delegate to `is_guide_page`, so a + working-tree read and a git-tag read of the same paths must end up with the + same set of pages -- regardless of which source produced them.""" + relative_paths = ["_api/index.rst", "changelog.rst", "commits.rst", "toolsets.rst"] + for relative in relative_paths: + target = tmp_path / relative + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text("Prose.\n") + + from_worktree = read_guide_docs_from_worktree(tmp_path) + + docs_prefix = "providers/test/docs/" + with ( + patch( + "extract_versions.git_ls_tree", + autospec=True, + return_value=[docs_prefix + relative for relative in relative_paths], + ), + patch( + "extract_versions.git_cat_file_batch", + autospec=True, + side_effect=lambda tag, paths: {p: "Prose.\n" for p in paths}, + ), + ): + from_tag = read_guide_docs_from_tag("providers-test/1.0.0", "new", "test") + + # Both readers must apply the same filter; each reader's own test can pass + # while the two drift apart. + assert set(from_worktree) == set(from_tag) == {"toolsets.rst"} diff --git a/dev/registry/tests/test_extract_parameters.py b/dev/registry/tests/test_extract_parameters.py index eac805cc49a37..b751ae64f0893 100644 --- a/dev/registry/tests/test_extract_parameters.py +++ b/dev/registry/tests/test_extract_parameters.py @@ -22,7 +22,7 @@ import builtins import json import types -from dataclasses import fields +from dataclasses import MISSING, fields from unittest.mock import patch import pytest @@ -38,6 +38,7 @@ get_category, is_durable_capable, load_resumable_job_mixin, + read_guide_docs, supports_deferrable, ) @@ -911,6 +912,20 @@ def fake_decorator_task(python_callable=None, **kwargs): } +# --------------------------------------------------------------------------- +# read_guide_docs +# --------------------------------------------------------------------------- +def test_read_guide_docs_skips_generated_and_release_note_pages(tmp_path): + (tmp_path / "_api" / "x").mkdir(parents=True) + (tmp_path / "_api" / "x" / "index.rst").write_text("Generated.\n") + (tmp_path / "changelog.rst").write_text("Release notes.\n") + (tmp_path / "toolsets.rst").write_text("``HookToolset``\n---------------\n\nProse.\n") + + result = read_guide_docs(tmp_path) + + assert set(result) == {"toolsets.rst"} + + # --------------------------------------------------------------------------- # TestDiscoverClassesFromProvider # --------------------------------------------------------------------------- @@ -999,6 +1014,88 @@ def test_discovers_operator(self, provider_yaml_path, base_classes): assert operators[0]["import_path"] == "airflow.providers.amazon.aws.operators.s3.FakeOperator" assert operators[0]["provider_id"] == "amazon" + def test_guide_section_becomes_a_guide_url(self, provider_yaml_path, base_classes): + """A class documented by a section of its own gets a link to that section; + one that is only in the API reference keeps just its ``docs_url``.""" + docs_dir = provider_yaml_path.parent / "docs" / "operators" + docs_dir.mkdir(parents=True) + (docs_dir / "s3.rst").write_text("``FakeOperator``\n----------------\n\nProse.\n") + + with ( + patch("extract_parameters.PROVIDERS_DIR", provider_yaml_path.parent.parent), + patch("extract_parameters.importlib.import_module", side_effect=self._mock_import), + ): + result = discover_classes_from_provider(provider_yaml_path, base_classes) + + by_name = {r["name"]: r for r in result} + assert by_name["FakeOperator"]["guide_url"] == ( + "https://airflow.apache.org/docs/apache-airflow-providers-amazon/stable" + "/operators/s3.html#fakeoperator" + ) + assert "guide_url" not in by_name["FakeSensor"] + + @pytest.mark.parametrize( + ("tag_exists", "expected_anchor"), + [ + pytest.param(True, "released-title-fakeoperator", id="tag-exists-reads-released-docs"), + pytest.param(False, "fakeoperator", id="no-tag-reads-working-tree"), + ], + ) + def test_guide_docs_come_from_release_tag_when_it_exists( + self, provider_yaml_path, base_classes, tag_exists, expected_anchor + ): + docs_dir = provider_yaml_path.parent / "docs" / "operators" + docs_dir.mkdir(parents=True) + (docs_dir / "s3.rst").write_text("``FakeOperator``\n----------------\n\nUnreleased title.\n") + released = { + "operators/s3.rst": "Released title: ``FakeOperator``\n--------------------------------\n" + } + + with ( + patch("extract_parameters.PROVIDERS_DIR", provider_yaml_path.parent.parent), + patch("extract_parameters.git_tag_exists", return_value=tag_exists) as tag_check, + patch("extract_parameters.detect_layout", return_value="new"), + patch("extract_parameters.read_guide_docs_at_tag", return_value=released) as read_at_tag, + patch("extract_parameters.importlib.import_module", side_effect=self._mock_import), + ): + result = discover_classes_from_provider(provider_yaml_path, base_classes, version="1.2.3") + + tag_check.assert_called_once_with("providers-amazon/1.2.3") + if tag_exists: + read_at_tag.assert_called_once_with("providers-amazon/1.2.3", "new", "amazon") + else: + read_at_tag.assert_not_called() + guide_url = {r["name"]: r for r in result}["FakeOperator"]["guide_url"] + assert guide_url.endswith(f"/operators/s3.html#{expected_anchor}") + + @pytest.mark.parametrize( + ("version", "layout", "expect_tag_lookup"), + [ + pytest.param("", "new", False, id="no-version-skips-tag-lookup"), + pytest.param("1.2.3", None, True, id="undetectable-layout-falls-back"), + ], + ) + def test_guide_docs_fall_back_to_working_tree( + self, provider_yaml_path, base_classes, version, layout, expect_tag_lookup + ): + docs_dir = provider_yaml_path.parent / "docs" / "operators" + docs_dir.mkdir(parents=True) + (docs_dir / "s3.rst").write_text("``FakeOperator``\n----------------\n\nProse.\n") + + with ( + patch("extract_parameters.PROVIDERS_DIR", provider_yaml_path.parent.parent), + patch("extract_parameters.git_tag_exists", return_value=True) as tag_check, + patch("extract_parameters.detect_layout", return_value=layout), + patch("extract_parameters.read_guide_docs_at_tag") as read_at_tag, + patch("extract_parameters.importlib.import_module", side_effect=self._mock_import), + ): + result = discover_classes_from_provider(provider_yaml_path, base_classes, version=version) + + assert tag_check.called is expect_tag_lookup + read_at_tag.assert_not_called() + guide_url = {r["name"]: r for r in result}["FakeOperator"]["guide_url"] + assert guide_url.endswith("/operators/s3.html#fakeoperator") + def test_discovers_sensor(self, provider_yaml_path, base_classes): with ( patch("extract_parameters.PROVIDERS_DIR", provider_yaml_path.parent.parent), @@ -1093,14 +1190,18 @@ def test_extracts_short_description(self, provider_yaml_path, base_classes): assert operators[0]["short_description"] == "Copy objects in S3." def test_all_module_fields_present(self, provider_yaml_path, base_classes): - """Every discovered entry has every `Module` dataclass field (derived, not hardcoded).""" + """Every discovered entry has every required `Module` dataclass field (derived, not hardcoded). + + Fields with a default (e.g. ``guide_url``) are attached separately and only + when applicable, so they are allowed to be absent here. + """ with ( patch("extract_parameters.PROVIDERS_DIR", provider_yaml_path.parent.parent), patch("extract_parameters.importlib.import_module", side_effect=self._mock_import), ): result = discover_classes_from_provider(provider_yaml_path, base_classes) - required_fields = {f.name for f in fields(Module)} + required_fields = {f.name for f in fields(Module) if f.default is MISSING} for entry in result: missing = required_fields - entry.keys() assert not missing, f"Missing fields {missing} in {entry['name']}" diff --git a/dev/registry/tests/test_extract_versions.py b/dev/registry/tests/test_extract_versions.py index 6a3128ce01a32..7ecc7d94dff6e 100644 --- a/dev/registry/tests/test_extract_versions.py +++ b/dev/registry/tests/test_extract_versions.py @@ -18,8 +18,9 @@ from __future__ import annotations +import subprocess import textwrap -from unittest.mock import patch +from unittest.mock import MagicMock, call, patch import pytest from extract_versions import ( @@ -28,6 +29,9 @@ SCRIPT_DIR, extract_modules_from_yaml, extract_version_data, + git_cat_file_batch, + git_ls_tree, + read_guide_docs, ) from registry_tools.types import CLASS_LEVEL_SECTIONS, DICT_SHAPED_CLASS_LEVEL_SECTIONS @@ -84,7 +88,8 @@ def test_no_other_candidates(self): } -def _extract_class_level_modules(provider_yaml: dict) -> list[dict]: +@patch("extract_versions.read_guide_docs", autospec=True, return_value={}) +def _extract_class_level_modules(provider_yaml: dict, _mock_read_guide_docs) -> list[dict]: return extract_modules_from_yaml( provider_yaml, tag="providers-test/1.0.0", @@ -266,3 +271,189 @@ def test_filesystem_schemes_read_from_release_tag( {"scheme": "tfs", "filesystem": "airflow.providers.test.fs.testfs"}, ] mock_git_show.assert_any_call("providers-test/1.0.0", fs_source_path) + + +class TestExtractModulesGuideUrls: + """A class the provider's guides document in a section of its own must get a + ``guide_url`` for every version, not just the latest. A superseded version's + page is rendered only from the per-version file this module writes, so a link + resolved on the latest path alone disappears the moment a new version lands. + """ + + PROVIDER_YAML = { + "toolsets": [ + { + "integration-name": "Test", + "python-modules": ["airflow.providers.test.toolsets.hook"], + } + ] + } + SOURCE = 'class HookToolset:\n """A toolset."""\n' + GUIDE = "``HookToolset``\n---------------\n\nProse.\n" + + def _extract(self, layout="new", docs_paths=("providers/test/docs/toolsets.rst",)): + def fake_git_show(_tag, path): + return self.SOURCE if path.endswith(".py") else None + + def fake_git_cat_file_batch(_tag, paths): + return {p: self.GUIDE for p in paths} + + with ( + patch("extract_versions.git_ls_tree", autospec=True, return_value=list(docs_paths)), + patch("extract_versions.git_show", autospec=True, side_effect=fake_git_show), + patch("extract_versions.git_cat_file_batch", autospec=True, side_effect=fake_git_cat_file_batch), + ): + return extract_modules_from_yaml( + self.PROVIDER_YAML, "providers-test/1.0.0", layout, "test", "test", "1.0.0" + ) + + def test_documented_class_gets_a_versioned_guide_url(self): + modules = self._extract() + + assert [m["name"] for m in modules] == ["HookToolset"] + assert modules[0]["guide_url"] == ( + "https://airflow.apache.org/docs/apache-airflow-providers-test/1.0.0/toolsets.html#hooktoolset" + ) + + def test_undocumented_class_gets_no_guide_url(self): + modules = self._extract(docs_paths=()) + + assert [m["name"] for m in modules] == ["HookToolset"] + assert "guide_url" not in modules[0] + + def test_old_layout_gets_no_guide_url(self): + # Pre-per-provider tags kept docs in a top-level tree, so there is no + # provider-relative page path to build a link from. + modules = self._extract(layout="old") + + assert "guide_url" not in modules[0] + + +class TestReadGuideDocs: + def test_skips_generated_and_release_note_pages_before_calling_git_show(self): + docs_prefix = "providers/test/docs/" + paths = [ + docs_prefix + "_api/x/index.rst", + docs_prefix + "changelog.rst", + docs_prefix + "diagram.png", + docs_prefix + "conf.py", + docs_prefix + "toolsets.rst", + ] + + with ( + patch("extract_versions.git_ls_tree", autospec=True, return_value=paths), + patch( + "extract_versions.git_cat_file_batch", + autospec=True, + side_effect=lambda tag, paths: {p: "Prose.\n" for p in paths}, + ) as mock_git_cat_file_batch, + ): + result = read_guide_docs("providers-test/1.0.0", "new", "test") + + assert set(result) == {"toolsets.rst"} + # Filtering must happen before the batch call, not just before the dict write. + assert mock_git_cat_file_batch.call_args_list == [ + call("providers-test/1.0.0", [docs_prefix + "toolsets.rst"]) + ] + + def test_skips_a_page_whose_content_is_an_empty_string(self): + docs_prefix = "providers/test/docs/" + paths = [docs_prefix + "empty.rst"] + + with ( + patch("extract_versions.git_ls_tree", autospec=True, return_value=paths), + patch( + "extract_versions.git_cat_file_batch", + autospec=True, + return_value={docs_prefix + "empty.rst": ""}, + ), + ): + result = read_guide_docs("providers-test/1.0.0", "new", "test") + + assert result == {} + + +class TestGitLsTree: + def test_passes_quote_path_false_to_git(self): + mock_result = MagicMock(spec=subprocess.CompletedProcess) + mock_result.stdout = b"providers/test/docs/toolsets.rst\n" + with patch("extract_versions.subprocess.run", autospec=True, return_value=mock_result) as mock_run: + git_ls_tree("providers-test/1.0.0", "providers/test/docs/") + + assert mock_run.call_args.args[0] == [ + "git", + "-c", + "core.quotePath=false", + "ls-tree", + "-r", + "--name-only", + "providers-test/1.0.0", + "--", + "providers/test/docs/", + ] + + def test_decodes_stdout_as_utf8(self): + mock_result = MagicMock(spec=subprocess.CompletedProcess) + mock_result.stdout = "docs/café.rst\ndocs/b.rst\n".encode() + with patch("extract_versions.subprocess.run", autospec=True, return_value=mock_result): + result = git_ls_tree("providers-test/1.0.0", "providers/test/docs/") + + assert result == ["docs/café.rst", "docs/b.rst"] + + +def _batch_hit(sha1: str, obj_type: str, content: bytes) -> bytes: + return f"{sha1} {obj_type} {len(content)}\n".encode() + content + b"\n" + + +def _batch_missing(spec: str) -> bytes: + return f"{spec} missing\n".encode() + + +class TestGitCatFileBatch: + def test_empty_paths_returns_empty_dict_without_subprocess(self): + with patch("extract_versions.subprocess.run", autospec=True) as mock_run: + result = git_cat_file_batch("providers-test/1.0.0", []) + + assert result == {} + mock_run.assert_not_called() + + def test_two_hits_with_different_sizes_and_multibyte_content(self): + tag = "providers-test/1.0.0" + paths = ["providers/test/docs/a.rst", "providers/test/docs/b.rst"] + first_content = b"short\n" + second_content = "café prôse with more text\n".encode() + payload = _batch_hit("aaa1", "blob", first_content) + _batch_hit("bbb2", "blob", second_content) + + mock_result = MagicMock(spec=subprocess.CompletedProcess) + mock_result.stdout = payload + with patch("extract_versions.subprocess.run", autospec=True, return_value=mock_result): + result = git_cat_file_batch(tag, paths) + + assert result == { + paths[0]: "short\n", + paths[1]: "café prôse with more text\n", + } + + def test_hit_followed_by_missing_path(self): + tag = "providers-test/1.0.0" + paths = ["providers/test/docs/a.rst", "providers/test/docs/missing.rst"] + payload = _batch_hit("aaa1", "blob", b"content\n") + _batch_missing(f"{tag}:{paths[1]}") + + mock_result = MagicMock(spec=subprocess.CompletedProcess) + mock_result.stdout = payload + with patch("extract_versions.subprocess.run", autospec=True, return_value=mock_result): + result = git_cat_file_batch(tag, paths) + + assert result == {paths[0]: "content\n"} + + def test_all_missing_returns_empty_dict(self): + tag = "providers-test/1.0.0" + paths = ["providers/test/docs/a.rst", "providers/test/docs/b.rst"] + payload = _batch_missing(f"{tag}:{paths[0]}") + _batch_missing(f"{tag}:{paths[1]}") + + mock_result = MagicMock(spec=subprocess.CompletedProcess) + mock_result.stdout = payload + with patch("extract_versions.subprocess.run", autospec=True, return_value=mock_result): + result = git_cat_file_batch(tag, paths) + + assert result == {} diff --git a/dev/registry/tests/test_registry_contract_models.py b/dev/registry/tests/test_registry_contract_models.py index 3c2b44daf7433..ea926d131881f 100644 --- a/dev/registry/tests/test_registry_contract_models.py +++ b/dev/registry/tests/test_registry_contract_models.py @@ -114,6 +114,15 @@ def test_module_contract_preserves_supports_deferrable_true(): assert validated["modules"][0]["supports_deferrable"] is True +def test_module_contract_omits_guide_url_for_undocumented_classes(): + assert ModuleContract.model_validate(_module_payload()).guide_url is None + + +def test_module_contract_preserves_guide_url_value(): + guide_url = "https://example.invalid/docs/toolsets.html#exampletoolset" + assert ModuleContract.model_validate(_module_payload(guide_url=guide_url)).guide_url == guide_url + + def test_connection_type_contract_defaults_external_services_to_empty_list(): """Legacy connection-types entries (provider.yaml without `external-services`) must still validate, with the field defaulting to an empty list.""" diff --git a/providers/common/ai/docs/operators/llm_branch.rst b/providers/common/ai/docs/operators/llm_branch.rst index 35b15f0f2b970..0e113aed05028 100644 --- a/providers/common/ai/docs/operators/llm_branch.rst +++ b/providers/common/ai/docs/operators/llm_branch.rst @@ -17,8 +17,8 @@ .. _howto/operator:llm_branch: -Branch on an answer: ``LLMBranchOperator`` -========================================== +Branch on an answer: ``LLMBranchOperator`` and ``@task.llm_branch`` +=================================================================== Use :class:`~airflow.providers.common.ai.operators.llm_branch.LLMBranchOperator` for LLM-driven branching, where the LLM decides which downstream task(s) to diff --git a/providers/common/ai/docs/operators/llm_file_analysis.rst b/providers/common/ai/docs/operators/llm_file_analysis.rst index 4fb0931410d77..eed8c86628189 100644 --- a/providers/common/ai/docs/operators/llm_file_analysis.rst +++ b/providers/common/ai/docs/operators/llm_file_analysis.rst @@ -17,8 +17,8 @@ .. _howto/operator:llm_file_analysis: -Analyze files and images: ``LLMFileAnalysisOperator`` -===================================================== +Analyze files and images: ``LLMFileAnalysisOperator`` and ``@task.llm_file_analysis`` +===================================================================================== .. note:: diff --git a/providers/common/ai/docs/operators/llm_schema_compare.rst b/providers/common/ai/docs/operators/llm_schema_compare.rst index cafde3900de5e..47b06cf22255f 100644 --- a/providers/common/ai/docs/operators/llm_schema_compare.rst +++ b/providers/common/ai/docs/operators/llm_schema_compare.rst @@ -17,8 +17,8 @@ .. _howto/operator:llm_schema_compare: -Detect schema drift: ``LLMSchemaCompareOperator`` -================================================= +Detect schema drift: ``LLMSchemaCompareOperator`` and ``@task.llm_schema_compare`` +================================================================================== .. note:: diff --git a/providers/common/ai/docs/operators/llm_sql.rst b/providers/common/ai/docs/operators/llm_sql.rst index 5d06ec97e3978..42e51de795582 100644 --- a/providers/common/ai/docs/operators/llm_sql.rst +++ b/providers/common/ai/docs/operators/llm_sql.rst @@ -17,8 +17,8 @@ .. _howto/operator:llm_sql_query: -Natural language to SQL: ``LLMSQLQueryOperator`` -================================================ +Natural language to SQL: ``LLMSQLQueryOperator`` and ``@task.llm_sql`` +====================================================================== .. note:: diff --git a/registry/AGENTS.md b/registry/AGENTS.md index 99416cb89311c..b82377a2569cc 100644 --- a/registry/AGENTS.md +++ b/registry/AGENTS.md @@ -462,6 +462,39 @@ They run inside Breeze where all providers are installed. `extract_metadata.py` the CI workflow can run the fast scripts (metadata, ~30s per provider) without spinning up Breeze, while parameter/connection extraction is a separate step. +### How a module gets a "Guide" link + +A module card links to the how-to guide section that documents it, alongside the +generated API reference. `provider.yaml`'s `how-to-guide` fields are CI-enforced +by `check_doc_files`, but they name a whole page, never a section, and cover only +operators, sensors and transfers — not the toolset, hook and decorator pages this +needs. So `registry_tools/docs_guides.py` reads the provider's own `docs/*.rst` +and matches a class to a section when the section's title names it as an inline +literal, either at the start — ``` ``MCPHook`` ``` or after a colon at the very +end — ``` Airflow hooks as tools: ``HookToolset`` ``` or ``` Agents with tools: +``AgentOperator`` and ``@task.agent`` ```. A literal elsewhere in a prose title +does not count. The anchor is derived from the whole title the way docutils +derives its HTML id. When a name is titled in more than one place, a page's own +title wins over a subsection on any other page, so the link lands on the page +dedicated to the class rather than on a passing section about it. + +That convention is what the guides already do, and it is deliberately the only +signal this resolves a section from: a hand-maintained class-to-guide table +would keep pointing at sections that have since been renamed or split, and a +link that lands on the wrong section is worse than no link. A class documented +only in prose gets no Guide link. + +Growing the set of modules that get a Guide link means titling that provider's +sections in one of those two shapes, not touching this extractor. `common/ai` +titles its dedicated operator, hook and toolset pages this way; a couple of other +providers use the same shapes for a config option name or a single decorator +rather than a class. Having the right title doesn't guarantee a link — that still +depends on a same-named module existing in the catalog. + +Both extraction paths resolve it — `extract_parameters.py` from the working tree +for the latest release, `extract_versions.py` from the git tag for superseded ones +— because a superseded version's page is rendered only from its own metadata file. + ### Relationship to `run_provider_yaml_files_check.py` `scripts/in_container/run_provider_yaml_files_check.py` (run by the diff --git a/registry/src/css/main.css b/registry/src/css/main.css index d5077f8c8887e..42b6d1a618e91 100644 --- a/registry/src/css/main.css +++ b/registry/src/css/main.css @@ -3552,12 +3552,14 @@ main { /* Module Actions (View Docs, Source) */ .provider-detail-page .module-actions { display: flex; + flex-wrap: wrap; align-items: center; gap: var(--space-3); margin-top: var(--space-3); } .provider-detail-page .module-actions .docs-link, +.provider-detail-page .module-actions .guide-link, .provider-detail-page .module-actions .source-link { display: inline-flex; align-items: center; @@ -3575,6 +3577,14 @@ main { color: var(--color-cyan-300); } +.provider-detail-page .module-actions .guide-link { + color: var(--accent-secondary); +} + +.provider-detail-page .module-actions .guide-link:hover { + color: var(--color-cyan-300); +} + .provider-detail-page .module-actions .source-link { color: var(--text-muted); } diff --git a/registry/src/provider-version.njk b/registry/src/provider-version.njk index e28be209f1a82..7d0c16b9c4912 100644 --- a/registry/src/provider-version.njk +++ b/registry/src/provider-version.njk @@ -475,6 +475,12 @@ eleventyComputed: {% endif %} + {% if module.guide_url %} + + Guide + + + {% endif %} {% if module.source_url %} Source