Skip to content
108 changes: 98 additions & 10 deletions packages/gcp-sphinx-docfx-yaml/docfx_yaml/extension.py
Original file line number Diff line number Diff line change
Expand Up @@ -29,8 +29,8 @@
import shutil
from collections import defaultdict
from collections.abc import Mapping, MutableSet, Sequence
from functools import partial
from itertools import zip_longest
from functools import lru_cache, partial
from itertools import chain, zip_longest
from pathlib import Path
from typing import Any, Iterable

Expand All @@ -46,13 +46,21 @@
import subprocess

import sphinx.application
import yaml
from docuploader import shell
from sphinx.builders.html import StandaloneHTMLBuilder
from sphinx.errors import ExtensionError
from sphinx.ext.napoleon import Config, GoogleDocstring, _process_docstring
from sphinx.util import ensuredir
from sphinx.util.console import bold, darkgreen
from sphinx.util.nodes import make_refnode
from yaml import safe_dump as dump

try:
from yaml import CSafeDumper as SafeDumper
except ImportError:
from yaml import SafeDumper

dump = partial(yaml.dump, Dumper=SafeDumper)

from docfx_yaml import markdown_utils

Expand Down Expand Up @@ -184,6 +192,32 @@ def _grab_repo_metadata() -> Mapping[str, str] | None:
return None


class DocFXHTMLBuilder(StandaloneHTMLBuilder):
"""HTML builder subclass that skips rendering unused HTML pages during DocFX builds."""

def write(self, *args: Any, **kwargs: Any) -> None:
pass

def finish(self) -> None:
pass


def _configure_docfx(app: sphinx.application.Sphinx, config: Any) -> None:
"""Disables Sphinx extensions from shared conf.py files that DocFX does not need.

Package conf.py files are shared between HTML docs and DocFX builds and
enable `sphinx.ext.intersphinx` and `sphinx.ext.viewcode` by default.
Neither is used in DocFX YAML output, so we clear `intersphinx_mapping`
(avoiding remote inventory downloads) and disconnect `viewcode` listeners
(avoiding source file tokenization during `doctree-read`).
"""
config.intersphinx_mapping = {}
for listeners in getattr(getattr(app, "events", None), "listeners", {}).values():
for listener in list(listeners):
if getattr(listener.handler, "__module__", "") == "sphinx.ext.viewcode":
app.disconnect(listener.id)


def build_init(app: sphinx.application.Sphinx) -> None:
"""Initializes the build.

Expand All @@ -197,9 +231,6 @@ def build_init(app: sphinx.application.Sphinx) -> None:
else:
print("Successfully retrieved repository metadata.")
app.env.library_shortname = repo_metadata["name"]
print("Running sphinx-build with Markdown first...")
markdown_utils.run_sphinx_markdown(app)
print("Completed running sphinx-build with Markdown files.")

"""
Set up environment data
Expand Down Expand Up @@ -1025,6 +1056,35 @@ def _extract_type_name(annotation: Any) -> str:
return type_name


@lru_cache(maxsize=512)
def _get_class_lines(full_path: str) -> dict[str, int]:
"""Parses a file once and maps class qualnames to their starting line numbers."""
lines: dict[str, int] = {}

def _visit(node: ast.AST, prefix: str = "") -> None:
for child in ast.iter_child_nodes(node):
if isinstance(child, ast.ClassDef):
qual = f"{prefix}{child.name}"
lines.setdefault(
qual,
child.decorator_list[0].lineno
if child.decorator_list
else child.lineno,
)
_visit(child, f"{qual}.")
elif isinstance(child, (ast.FunctionDef, ast.AsyncFunctionDef)):
_visit(child, f"{prefix}{child.name}.<locals>.")
else:
_visit(child, prefix)

try:
with open(full_path, "rb") as f:
_visit(ast.parse(f.read()))
except Exception:
pass
return lines


def _create_datam(
app: sphinx.application.Sphinx,
cls: str | None,
Expand Down Expand Up @@ -1180,7 +1240,12 @@ def _update_friendly_package_name(path):

# Make relative
path = path.replace(os.sep, "", 1)
start_line = inspect.getsourcelines(obj)[1]
unwrapped = inspect.unwrap(obj)
start_line = (
_get_class_lines(full_path).get(getattr(unwrapped, "__qualname__", ""), 0)
if inspect.isclass(unwrapped)
else 0
) or inspect.getsourcelines(obj)[1]

path = _update_friendly_package_name(path)

Expand Down Expand Up @@ -1475,6 +1540,7 @@ def _reformat_pattern(code: str, pattern: str) -> str:
return code


@lru_cache(maxsize=4096)
def format_code(code: str) -> str:
"""Reformats code using black.format_str().

Expand Down Expand Up @@ -1937,7 +2003,11 @@ def find_uid_to_convert(
None if current word does not contain any reference `uid`, or the `uid`
that should be converted.
"""
for uid in known_uids:
# All Python UIDs are dotted paths (e.g. `pkg.module.Symbol`), so skip
# plain words and sentence-ending periods before scanning `known_uids`.
if "." not in current_word.strip("."):
return None
for uid in chain(known_uids, hard_coded_references or ()):
Comment thread
daniel-sanche marked this conversation as resolved.
# Do not convert references to itself or containing partial
# references. This could result in `storage.types.ReadSession` being
# prematurely converted to
Expand Down Expand Up @@ -1994,6 +2064,19 @@ def convert_cross_references(
Returns:
content that has been modified with proper cross references if found.
"""
# Every UID in `google-*` packages (and in `hard_coded_references`) starts
# with "google.", so if "google." isn't in `content`, no cross-reference can
# match. However, a few packages in the repo don't use the `google.*`
# namespace (e.g. `pandas_gbq.Context`), so we only take this shortcut when
# the package's `known_uids` actually start with "google.".
if (
known_uids
and known_uids[0].startswith("google.")
and known_uids[-1].startswith("google.")
and "google." not in content
):
return content

example_text = "Examples:"
words = content.split(" ")

Expand All @@ -2014,7 +2097,6 @@ def convert_cross_references(
"google.iam.v1.iam_policy_pb2.TestIamPermissionsResponse": iam_policy_link
+ "#L120-L131",
}
known_uids.extend(hard_coded_references.keys())

Comment thread
daniel-sanche marked this conversation as resolved.
# Used to keep track of current position to avoid converting if needed.
example_index = len(content)
Expand Down Expand Up @@ -2267,6 +2349,7 @@ def convert_module_to_package_if_needed(obj):
ensuredir(normalized_outdir)

# Add markdown pages to the configured output directory.
markdown_utils.run_sphinx_markdown(app)
markdown_utils.move_markdown_pages(app, normalized_outdir)

pkg_toc_yaml = []
Expand All @@ -2277,6 +2360,8 @@ def convert_module_to_package_if_needed(obj):
# Used to disambiguate entry names
yaml_map = {}

known_uids = sorted(app.env.docfx_uid_names.keys(), reverse=True)

# Order matters here, we need modules before lower level classes,
# so that we can make sure to inject the TOC properly
for data_set in (
Expand Down Expand Up @@ -2433,7 +2518,6 @@ def convert_module_to_package_if_needed(obj):
# google.cloud.aiplatform.AutoMLForecastingTrainingJob

current_object_name = obj["fullName"]
known_uids = sorted(app.env.docfx_uid_names.keys(), reverse=True)
# Currently we only need to look in summary, syntax and
# attributes for cross references.
search_cross_references(obj, current_object_name, known_uids)
Expand Down Expand Up @@ -2643,6 +2727,8 @@ def missing_reference(
Returns:
Any: The new node.
"""
if getattr(app.builder, "name", None) == "markdown":
return None
reftarget = ""
refdoc = ""
reftype = ""
Expand Down Expand Up @@ -2686,6 +2772,8 @@ def setup(app: sphinx.application.Sphinx) -> None:
app.add_directive("remarks", RemarksDirective)
app.add_directive("todo", TodoDirective)

app.add_builder(DocFXHTMLBuilder, override=True)
app.connect("config-inited", _configure_docfx)
app.connect("builder-inited", build_init)
app.connect("autodoc-process-docstring", process_docstring)
app.connect("autodoc-process-signature", process_signature)
Expand Down
43 changes: 27 additions & 16 deletions packages/gcp-sphinx-docfx-yaml/docfx_yaml/markdown_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -510,26 +510,37 @@ def remove_unused_pages(


def run_sphinx_markdown(app: sphinx.application) -> None:
"""Runs sphinx-build with Markdown builder in the plugin.
"""Runs Markdown builder in-process reusing the already-read Sphinx environment.

Args:
app (sphinx.application): The sphinx application.
"""
cwd = os.getcwd()
relative_srcdir = app.srcdir.removeprefix(f"{cwd}/")
relative_outdir = app.outdir.removeprefix(f"{cwd}/").removesuffix("/html")
# Skip running sphinx-build for Markdown for some unit tests.
# Skip running Markdown builder for some unit tests.
# Not required other than to output DocFX YAML.
if "docs" in cwd:
markdown_outdir = Path(app.builder.outdir).parent / "markdown"
if (
"docs" in os.getcwd()
or markdown_outdir.exists()
or not getattr(app.env, "found_docs", None)
):
return

return shell.run(
[
"sphinx-build",
"-M",
"markdown",
relative_srcdir,
relative_outdir,
],
hide_output=False,
)
from sphinx.util.osutil import ensuredir
from sphinx_markdown_builder.markdown_builder import MarkdownBuilder

ensuredir(str(markdown_outdir))
docnames = sorted(app.env.found_docs)
orig_builder = app.builder
md_builder = MarkdownBuilder(app)
md_builder.outdir = str(markdown_outdir)
md_builder.set_environment(app.env)
md_builder.init()
md_builder.prepare_writing(docnames)
app.builder = md_builder
try:
for docname in docnames:
doctree = app.env.get_and_resolve_doctree(docname, md_builder)
md_builder.write_doc_serialized(docname, doctree)
md_builder.write_doc(docname, doctree)
finally:
app.builder = orig_builder
26 changes: 17 additions & 9 deletions packages/gcp-sphinx-docfx-yaml/docfx_yaml/utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,12 +17,16 @@
from inspect import signature

from docutils import nodes
from docutils.io import StringOutput
from docutils.frontend import OptionParser
from docutils.utils import new_document
from sphinx import addnodes
from sphinx.application import Sphinx

from .writer import MarkdownTranslator
from .writer import MarkdownWriter as Writer

_DEFAULT_SETTINGS = OptionParser(components=(Writer,)).get_default_values()


def slugify(value: str) -> str:
"""Converts to lowercase, removes non-word characters.
Expand Down Expand Up @@ -70,14 +74,18 @@ def transform_node(app: Sphinx, node: nodes.Node) -> str:
Returns:
str: The transformed node as a string.
"""
destination = StringOutput(encoding="utf-8")
doc = new_document(b"<partial node>")
if node.parent is not None:
node = node.deepcopy()
doc = new_document(b"<partial node>", _DEFAULT_SETTINGS)
doc.append(node)

# Resolve refs
# Resolve refs only when the node actually contains pending cross-references
doc["docname"] = "inmemory"
app.env.resolve_references(doctree=doc, fromdocname="inmemory", builder=app.builder)

writer = Writer(app.builder)
writer.write(doc, destination)
return destination.destination.decode("utf-8")
if any(True for _ in node.traverse(addnodes.pending_xref)):
app.env.resolve_references(
doctree=doc, fromdocname="inmemory", builder=app.builder
)

visitor = MarkdownTranslator(doc, app.builder)
doc.walkabout(visitor)
return visitor.body
22 changes: 22 additions & 0 deletions packages/gcp-sphinx-docfx-yaml/tests/test_helpers.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
import tempfile
import unittest
import unittest.mock

from parameterized import parameterized
from yaml import Loader, load
Expand Down Expand Up @@ -411,6 +412,27 @@ def test_is_not_valid_python_code(self, invalid_syntax):
result = extension.is_valid_python_code(invalid_syntax)
self.assertFalse(result)

def test_configure_docfx_and_builder(self):
app = unittest.mock.MagicMock()
app.config.intersphinx_mapping = {"python": ("https://example.com", None)}
viewcode_listener = unittest.mock.MagicMock(id=1)
viewcode_listener.handler.__module__ = "sphinx.ext.viewcode"
other_listener = unittest.mock.MagicMock(id=2)
other_listener.handler.__module__ = "docfx_yaml.extension"
app.events.listeners = {"doctree-read": [viewcode_listener, other_listener]}

extension._configure_docfx(app, app.config)

self.assertEqual(app.config.intersphinx_mapping, {})
app.disconnect.assert_called_once_with(1)
self.assertIsNone(extension.DocFXHTMLBuilder.write(None))
self.assertIsNone(extension.DocFXHTMLBuilder.finish(None))

def test_missing_reference_skips_markdown_builder(self):
app = unittest.mock.MagicMock()
app.builder.name = "markdown"
self.assertIsNone(extension.missing_reference(app, None, None, None))


if __name__ == "__main__":
unittest.main()
Loading
Loading