From 35d68eda91f70dc5d8c1e58fe5f967de14b95b94 Mon Sep 17 00:00:00 2001 From: Alexander Lanin Date: Wed, 2 Sep 2026 10:56:36 +0200 Subject: [PATCH 1/7] POC: add documentation delta tool --- .github/workflows/on-pr.yml | 92 ++++ tools/BUILD | 8 + tools/docs_delta.py | 914 +++++++++++++++++++++++++++++++++ tools/tests/BUILD | 9 + tools/tests/docs_delta_test.py | 443 ++++++++++++++++ 5 files changed, 1466 insertions(+) create mode 100644 tools/docs_delta.py create mode 100644 tools/tests/docs_delta_test.py diff --git a/.github/workflows/on-pr.yml b/.github/workflows/on-pr.yml index 1a6c653bf..642b1a42c 100644 --- a/.github/workflows/on-pr.yml +++ b/.github/workflows/on-pr.yml @@ -43,3 +43,95 @@ jobs: with: bazel-target: "//:docs" tests-report-artifact: tests-report + + docs-delta: + needs: [docs-build] + if: >- + github.event_name == 'pull_request' && + github.event.pull_request.head.repo.full_name == github.repository + runs-on: ubuntu-24.04 + permissions: + actions: read + contents: read + steps: + - name: Check out pull request + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + + - name: Check out published documentation history + continue-on-error: true + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + repository: ${{ github.repository }} + ref: gh-pages + path: .docs-baseline + fetch-depth: 0 + persist-credentials: false + + - name: Download documentation artifact + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: github-pages + path: docs-artifact + + - name: Set up uv + uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 + + - name: Generate documentation delta report + run: | + set -euo pipefail + uv run --no-project python tools/docs_delta.py + + - name: Upload documentation delta artifact + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0 # v7.0.1 + with: + name: docs-delta + path: docs-artifact/docs-delta.md + if-no-files-found: error + + docs-comment: + needs: [docs-delta] + if: >- + github.event_name == 'pull_request' && + github.event.pull_request.head.repo.full_name == github.repository + runs-on: ubuntu-24.04 + permissions: + actions: read + pull-requests: write + steps: + - name: Download documentation delta artifact + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: docs-delta + path: docs-delta + + - name: Prepare documentation preview comment + env: + PR_NUMBER: ${{ github.event.pull_request.number }} + REPOSITORY_OWNER: ${{ github.repository_owner }} + REPOSITORY_NAME: ${{ github.event.repository.name }} + run: | + set -euo pipefail + { + echo "Documentation preview for this pull request is available at:" + echo "**pr-${PR_NUMBER}**: https://${REPOSITORY_OWNER}.github.io/${REPOSITORY_NAME}/pr-${PR_NUMBER}/" + echo + cat docs-delta/docs-delta.md + } > "$RUNNER_TEMP/docs-comment.md" + + - name: Find existing documentation preview comment + id: find-comment + uses: peter-evans/find-comment@b30e6a3c0ed37e7c023ccd3f1db5c6c0b0c23aad # v4.0.0 + with: + issue-number: ${{ github.event.pull_request.number }} + comment-author: github-actions[bot] + body-includes: Documentation preview for this pull request + + - name: Create or update documentation preview comment + uses: peter-evans/create-or-update-comment@e8674b075228eee787fea43ef493e45ece1004c9 # v5.0.0 + with: + issue-number: ${{ github.event.pull_request.number }} + comment-id: ${{ steps.find-comment.outputs.comment-id }} + body-path: ${{ runner.temp }}/docs-comment.md + edit-mode: replace diff --git a/tools/BUILD b/tools/BUILD index 93b6741da..82aa238c3 100644 --- a/tools/BUILD +++ b/tools/BUILD @@ -32,3 +32,11 @@ py_binary( "//src/helper_lib", ], ) + +py_binary( + name = "docs_delta", + srcs = ["docs_delta.py"], + main = "docs_delta.py", + visibility = ["//visibility:public"], + deps = [], +) diff --git a/tools/docs_delta.py b/tools/docs_delta.py new file mode 100644 index 000000000..c14552d92 --- /dev/null +++ b/tools/docs_delta.py @@ -0,0 +1,914 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License, Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +"""Compare rendered documentation trees and write a Markdown delta report. + +The tool works on artifacts already produced by a docs build. In GitHub Actions +it can additionally resolve the matching published baseline from a local, +full-history checkout of ``gh-pages``. It never fetches from GitHub or invokes +Sphinx, which keeps the comparison useful both in CI and as a local command. +""" + +from __future__ import annotations + +import argparse +import io +import json +import os +import re +import shutil +import subprocess +import sys +import tarfile +import tempfile +from collections.abc import Iterator, Mapping, Sequence +from contextlib import contextmanager +from dataclasses import dataclass +from pathlib import Path +from urllib.parse import quote + +JsonObject = dict[str, object] +NeedMap = Mapping[str, JsonObject] + +DETAIL_LIMIT = 15 +MAX_RENDERED_VALUE_LENGTH = 600 + +# These values are generated from source/test links or Sphinx's internal +# bookkeeping. They can change when a build is moved to another checkout or +# when the same source is rebuilt, without representing a documentation delta. +# Keep this list deliberately explicit: fields not listed here are part of the +# comparison contract and must be reviewed when Sphinx-Needs adds new output. +VOLATILE_NEED_FIELDS = frozenset( + { + "lineno", + "lineno_content", + "source", + "target_id", + "is_modified", + "source_code_link", + "testlink", + } +) + +_MISSING = object() +_META_TAG = re.compile(r"]*>", re.IGNORECASE) +_META_ATTRIBUTE = re.compile( + r"(?:name|property|itemprop)\s*=\s*([\"'])(.*?)\1", re.IGNORECASE +) +_HTML_COMMENT = re.compile(r"", re.IGNORECASE | re.DOTALL) +_VOLATILE_ATTRIBUTE = re.compile( + r"\s+data-(?:build|generated|timestamp|last-modified)(?:-[\w-]+)?\s*=\s*(?:\"[^\"]*\"|'[^']*'|[^\s>]+)", + re.IGNORECASE, +) +_VOLATILE_META_NAMES = frozenset( + { + "build-date", + "build_date", + "created", + "date", + "generated", + "generated-at", + "generated_at", + "generator", + "last-modified", + "last_modified", + "timestamp", + } +) +_VOLATILE_COMMENT_WORDS = re.compile( + r"\b(?:build|built|generated|generator|last\s+updated|timestamp)\b", + re.IGNORECASE, +) + + +class DocsDeltaError(ValueError): + """A user-actionable input or report-generation error.""" + + +@dataclass(frozen=True) +class GithubPullRequest: + """Pull request provenance obtained from the Actions environment.""" + + base_ref: str + base_sha: str + number: str | None + + +@dataclass(frozen=True) +class NeedChange: + """One Need and, for modified entries, its two versions.""" + + need_id: str + baseline: JsonObject | None + current: JsonObject | None + + @property + def changed_fields(self) -> tuple[str, ...]: + """Return changed non-volatile fields in deterministic order.""" + + if self.baseline is None or self.current is None: + return () + names = set(self.baseline) | set(self.current) + return tuple( + sorted( + name + for name in names + if name not in VOLATILE_NEED_FIELDS + and not _json_values_equal( + self.baseline.get(name, _MISSING), + self.current.get(name, _MISSING), + ) + ) + ) + + +@dataclass(frozen=True) +class NeedComparison: + """Categorized comparison result for a pair of Need inventories.""" + + added: tuple[NeedChange, ...] + removed: tuple[NeedChange, ...] + modified: tuple[NeedChange, ...] + unchanged_count: int + + +@dataclass(frozen=True) +class PageChange: + """One rendered HTML path that was added, removed, or modified.""" + + path: str + + +@dataclass(frozen=True) +class PageComparison: + """Categorized comparison result for rendered HTML pages.""" + + added: tuple[PageChange, ...] + removed: tuple[PageChange, ...] + modified: tuple[PageChange, ...] + unchanged_count: int + + +def _json_values_equal(left: object, right: object) -> bool: + """Compare JSON values without conflating booleans and numbers.""" + + if left is _MISSING or right is _MISSING: + return left is right + return json.dumps(left, sort_keys=True, ensure_ascii=False) == json.dumps( + right, sort_keys=True, ensure_ascii=False + ) + + +def _flatten_needs(path: Path) -> dict[str, JsonObject]: + """Load all Need entries from every version in a Sphinx-Needs export.""" + + try: + payload = json.loads(path.read_text(encoding="utf-8")) + except (OSError, UnicodeDecodeError, json.JSONDecodeError) as exc: + raise DocsDeltaError(f"cannot read needs JSON {path}: {exc}") from exc + + if not isinstance(payload, dict) or not isinstance(payload.get("versions"), dict): + raise DocsDeltaError(f"needs JSON has no versions map: {path}") + + result: dict[str, JsonObject] = {} + for version in payload["versions"].values(): + if not isinstance(version, dict) or not isinstance(version.get("needs"), dict): + continue + for need_id, need in version["needs"].items(): + if not isinstance(need_id, str) or not isinstance(need, dict): + raise DocsDeltaError(f"invalid Need entry in {path}") + result[need_id] = need + return result + + +def load_needs(directory: Path) -> dict[str, JsonObject]: + """Load the root ``needs.json`` file from a documentation directory.""" + + needs_path = directory / "needs.json" + if not needs_path.is_file(): + raise DocsDeltaError(f"needs JSON is missing: {needs_path}") + return _flatten_needs(needs_path) + + +def compare_needs(baseline: NeedMap, current: NeedMap) -> NeedComparison: + """Compare Needs by ID while ignoring only the volatile field allowlist.""" + + added: list[NeedChange] = [] + removed: list[NeedChange] = [] + modified: list[NeedChange] = [] + unchanged_count = 0 + + for need_id in sorted(set(baseline) | set(current)): + old = baseline.get(need_id) + new = current.get(need_id) + change = NeedChange(need_id, old, new) + if old is None: + added.append(change) + elif new is None: + removed.append(change) + elif change.changed_fields: + modified.append(change) + else: + unchanged_count += 1 + + return NeedComparison( + added=tuple(added), + removed=tuple(removed), + modified=tuple(modified), + unchanged_count=unchanged_count, + ) + + +def _html_metadata_name(tag: str) -> str | None: + for match in _META_ATTRIBUTE.finditer(tag): + name = match.group(2).lower() + if name in _VOLATILE_META_NAMES: + return name + return None + + +def normalize_html(content: str) -> str: + """Normalize generated HTML without hiding ordinary page content changes.""" + + normalized = content.replace("\r\n", "\n").replace("\r", "\n") + + def remove_meta(match: re.Match[str]) -> str: + return "" if _html_metadata_name(match.group(0)) else match.group(0) + + normalized = _META_TAG.sub(remove_meta, normalized) + + def remove_comment(match: re.Match[str]) -> str: + return "" if _VOLATILE_COMMENT_WORDS.search(match.group(0)) else match.group(0) + + normalized = _HTML_COMMENT.sub(remove_comment, normalized) + normalized = _VOLATILE_ATTRIBUTE.sub("", normalized) + normalized = "\n".join(line.rstrip() for line in normalized.splitlines()) + return normalized.strip() + + +def _html_files(directory: Path) -> dict[str, str]: + """Return normalized HTML keyed by POSIX-relative path.""" + + if not directory.is_dir(): + raise DocsDeltaError(f"documentation directory is missing: {directory}") + result: dict[str, str] = {} + for path in sorted(directory.rglob("*.html")): + if path.is_file(): + relative = path.relative_to(directory).as_posix() + try: + result[relative] = normalize_html(path.read_text(encoding="utf-8")) + except (OSError, UnicodeDecodeError) as exc: + raise DocsDeltaError( + f"cannot read rendered page {path}: {exc}" + ) from exc + return result + + +def compare_html(baseline_dir: Path, current_dir: Path) -> PageComparison: + """Compare normalized recursive HTML output from two build directories.""" + + baseline = _html_files(baseline_dir) + current = _html_files(current_dir) + added: list[PageChange] = [] + removed: list[PageChange] = [] + modified: list[PageChange] = [] + unchanged_count = 0 + + for path in sorted(set(baseline) | set(current)): + old = baseline.get(path) + new = current.get(path) + change = PageChange(path) + if old is None: + added.append(change) + elif new is None: + removed.append(change) + elif old != new: + modified.append(change) + else: + unchanged_count += 1 + + return PageComparison( + added=tuple(added), + removed=tuple(removed), + modified=tuple(modified), + unchanged_count=unchanged_count, + ) + + +def _url_with_path(base_url: str, relative_path: str, anchor: str | None = None) -> str: + url = ( + base_url.rstrip("/") + + "/" + + "/".join( + quote(part, safe="-._~") for part in relative_path.strip("/").split("/") + ) + ) + if anchor: + url += "#" + quote(anchor, safe="-._~") + return url + + +def _need_url(base_url: str, need: Mapping[str, object]) -> str | None: + docname = need.get("docname") + if not isinstance(docname, str) or not docname.strip(): + return None + rendered_name = docname.strip("/") + if not rendered_name.endswith(".html"): + rendered_name += ".html" + return _url_with_path(base_url, rendered_name) + + +def _need_link(base_url: str, need_id: str, need: Mapping[str, object]) -> str | None: + page_url = _need_url(base_url, need) + if page_url is None: + return None + return page_url + "#" + quote(need_id, safe="-._~") + + +def _markdown_link(label: str, url: str | None) -> str: + if url is None: + return f"`{label}`" + safe_label = label.replace("[", "\\[").replace("]", "\\]") + return f"[{safe_label}]({url})" + + +def _display_name(need: Mapping[str, object]) -> str: + for field in ("title", "name"): + value = need.get(field) + if isinstance(value, str) and value: + return value + return "" + + +def _format_value(value: object) -> str: + if value is _MISSING: + return "(missing)" + rendered = json.dumps( + value, sort_keys=True, ensure_ascii=False, separators=(",", ":") + ) + if len(rendered) > MAX_RENDERED_VALUE_LENGTH: + rendered = rendered[:MAX_RENDERED_VALUE_LENGTH] + "…" + return rendered.replace("`", "\\`").replace("\n", " ") + + +def _format_need_entry( + change: NeedChange, *, base_url: str, pr_url: str, include_diff: bool = False +) -> list[str]: + need = change.current or change.baseline + assert need is not None + links = [] + old_link = ( + _need_link(base_url, change.need_id, change.baseline) + if change.baseline + else None + ) + new_link = ( + _need_link(pr_url, change.need_id, change.current) if change.current else None + ) + if old_link: + links.append(_markdown_link("old", old_link)) + if new_link: + links.append(_markdown_link("new", new_link)) + link_text = " / ".join(links) + title = _display_name(need) + suffix = f" — {title}" if title else "" + line = f"- `{change.need_id}`{suffix}" + if link_text: + line += f" ({link_text})" + result = [line] + if include_diff: + assert change.baseline is not None and change.current is not None + for field in change.changed_fields: + result.append( + f" - `{field}`: {_format_value(change.baseline.get(field, _MISSING))} " + f"→ {_format_value(change.current.get(field, _MISSING))}" + ) + return result + + +def _format_page_entry( + change: PageChange, *, base_url: str, pr_url: str, kind: str +) -> str: + old = _url_with_path(base_url, change.path) if kind != "added" else None + new = _url_with_path(pr_url, change.path) if kind != "removed" else None + links = [] + if old: + links.append(_markdown_link("old", old)) + if new: + links.append(_markdown_link("new", new)) + return f"- `{change.path}` ({' / '.join(links)})" + + +def _section( + lines: list[str], + title: str, + entries: Sequence[object], + formatter, + *, + base_url: str, + pr_url: str, + kind: str = "", +) -> None: + if not entries: + return + lines.extend([f"### {title} ({len(entries)})", ""]) + if len(entries) > DETAIL_LIMIT: + lines.extend([f"{len(entries)} entries changed; details omitted.", ""]) + return + for entry in entries: + if isinstance(entry, NeedChange): + lines.extend( + formatter( + entry, + base_url=base_url, + pr_url=pr_url, + include_diff=kind == "modified", + ) + ) + else: + lines.append(formatter(entry, base_url=base_url, pr_url=pr_url, kind=kind)) + lines.append("") + + +def render_report( + needs: NeedComparison, + pages: PageComparison, + *, + base_url: str, + pr_url: str, +) -> str: + """Render a deterministic Markdown report for comparison results.""" + + lines = [ + "# Documentation delta", + "", + f"Baseline: {_markdown_link(base_url, base_url)} ", + f"PR preview: {_markdown_link(pr_url, pr_url)}", + "", + "## Summary", + "", + f"- Needs: {len(needs.added)} added, {len(needs.removed)} removed, " + f"{len(needs.modified)} modified, {needs.unchanged_count} unchanged", + f"- Rendered pages: {len(pages.added)} added, {len(pages.removed)} removed, " + f"{len(pages.modified)} modified, {pages.unchanged_count} unchanged", + "", + "## Needs", + "", + ] + _section( + lines, + "Added", + needs.added, + _format_need_entry, + base_url=base_url, + pr_url=pr_url, + kind="added", + ) + _section( + lines, + "Removed", + needs.removed, + _format_need_entry, + base_url=base_url, + pr_url=pr_url, + kind="removed", + ) + _section( + lines, + "Modified", + needs.modified, + _format_need_entry, + base_url=base_url, + pr_url=pr_url, + kind="modified", + ) + lines.extend(["## Rendered HTML pages", ""]) + _section( + lines, + "Added", + pages.added, + _format_page_entry, + base_url=base_url, + pr_url=pr_url, + kind="added", + ) + _section( + lines, + "Removed", + pages.removed, + _format_page_entry, + base_url=base_url, + pr_url=pr_url, + kind="removed", + ) + _section( + lines, + "Modified", + pages.modified, + _format_page_entry, + base_url=base_url, + pr_url=pr_url, + kind="modified", + ) + return "\n".join(lines).rstrip() + "\n" + + +def unavailable_report(reason: str) -> str: + """Render the successful, explicit report used when baseline is unavailable.""" + + return ( + "# Documentation delta\n\n" + "## Delta unavailable\n\n" + f"The documentation baseline is unavailable: {reason}\n" + ) + + +def _write_report(path: Path, content: str) -> None: + """Atomically write the report so a failed run never leaves partial Markdown.""" + + path.parent.mkdir(parents=True, exist_ok=True) + temporary_path: Path | None = None + file_descriptor: int | None = None + try: + file_descriptor, temporary_name = tempfile.mkstemp( + dir=path.parent, prefix=f".{path.name}.", suffix=".tmp" + ) + temporary_path = Path(temporary_name) + with os.fdopen(file_descriptor, "w", encoding="utf-8") as output: + file_descriptor = None + output.write(content) + output.flush() + os.replace(temporary_path, path) + temporary_path = None + finally: + if file_descriptor is not None: + os.close(file_descriptor) + if temporary_path is not None: + temporary_path.unlink(missing_ok=True) + + +def _workspace_path(environ: Mapping[str, str], relative_path: str) -> Path: + workspace = environ.get("GITHUB_WORKSPACE") + if workspace: + return Path(workspace) / relative_path + return Path(relative_path) + + +def _github_pull_request(environ: Mapping[str, str]) -> GithubPullRequest | None: + """Read pull request provenance from standard GitHub Actions variables.""" + + event_name = environ.get("GITHUB_EVENT_NAME") + event_path = environ.get("GITHUB_EVENT_PATH") + if event_name != "pull_request": + return None + if not event_path: + raise DocsDeltaError( + "GITHUB_EVENT_PATH is required to resolve a pull request baseline" + ) + + try: + payload = json.loads(Path(event_path).read_text(encoding="utf-8")) + except (OSError, UnicodeDecodeError, json.JSONDecodeError) as exc: + raise DocsDeltaError( + f"cannot read GitHub event payload {event_path}: {exc}" + ) from exc + if not isinstance(payload, dict): + raise DocsDeltaError(f"GitHub event payload is not an object: {event_path}") + + pull_request = payload.get("pull_request") + if not isinstance(pull_request, dict): + return None + base = pull_request.get("base") + if not isinstance(base, dict): + return None + + base_ref = environ.get("GITHUB_BASE_REF") or base.get("ref") + base_sha = environ.get("GITHUB_BASE_SHA") or base.get("sha") + if not isinstance(base_ref, str) or not base_ref: + return None + if not isinstance(base_sha, str) or not base_sha: + return None + + number = pull_request.get("number") + if not isinstance(number, int | str) or isinstance(number, bool): + number = None + else: + number = str(number) + return GithubPullRequest(base_ref=base_ref, base_sha=base_sha, number=number) + + +def _find_published_baseline_commit( + gh_pages_dir: Path, pull_request: GithubPullRequest +) -> tuple[str | None, str]: + """Find the newest gh-pages commit whose generated Needs contain the base SHA.""" + + if not gh_pages_dir.is_dir(): + return None, f"gh-pages checkout is missing: {gh_pages_dir}" + + try: + git_check = subprocess.run( + ["git", "-C", str(gh_pages_dir), "rev-parse", "--git-dir"], + check=False, + capture_output=True, + text=True, + ) + except OSError as exc: + return None, f"cannot inspect gh-pages checkout {gh_pages_dir}: {exc}" + if git_check.returncode != 0: + return None, f"gh-pages checkout is not a Git repository: {gh_pages_dir}" + + needs_path = f"{pull_request.base_ref}/needs.json" + try: + result = subprocess.run( + [ + "git", + "-C", + str(gh_pages_dir), + "log", + "--all", + "--full-history", + "--format=%H", + "--", + needs_path, + ], + check=False, + capture_output=True, + text=True, + ) + except OSError as exc: + return None, f"cannot search gh-pages history: {exc}" + if result.returncode != 0: + detail = result.stderr.strip() or "git log failed" + return None, f"cannot search gh-pages history: {detail}" + + candidate_commits = [ + line.strip() for line in result.stdout.splitlines() if line.strip() + ] + for commit in candidate_commits: + try: + needs = subprocess.run( + [ + "git", + "-C", + str(gh_pages_dir), + "show", + f"{commit}:{needs_path}", + ], + check=False, + capture_output=True, + ) + except OSError as exc: + return None, f"cannot inspect published needs JSON: {exc}" + if needs.returncode == 0 and pull_request.base_sha.encode() in needs.stdout: + return commit, "" + + return ( + None, + f"no published {pull_request.base_ref} baseline contains " + f"source commit {pull_request.base_sha}", + ) + + +def _extract_git_tree( + gh_pages_dir: Path, commit: str, tree_path: str, destination: Path +) -> None: + """Extract one published documentation tree without changing the checkout.""" + + try: + result = subprocess.run( + [ + "git", + "-C", + str(gh_pages_dir), + "archive", + "--format=tar", + f"{commit}:{tree_path}", + ], + check=False, + capture_output=True, + ) + except OSError as exc: + raise DocsDeltaError(f"cannot extract gh-pages baseline: {exc}") from exc + if result.returncode != 0: + detail = result.stderr.decode(errors="replace").strip() or "git archive failed" + raise DocsDeltaError(f"cannot extract gh-pages baseline: {detail}") + + destination.mkdir(parents=True, exist_ok=True) + try: + with tarfile.open(fileobj=io.BytesIO(result.stdout), mode="r:") as archive: + for member in archive.getmembers(): + relative = Path(member.name) + if relative.is_absolute() or ".." in relative.parts: + raise DocsDeltaError( + f"gh-pages archive contains unsafe path: {member.name}" + ) + target = destination / relative + if member.isdir(): + target.mkdir(parents=True, exist_ok=True) + elif member.isfile(): + target.parent.mkdir(parents=True, exist_ok=True) + source = archive.extractfile(member) + if source is None: + raise DocsDeltaError( + f"cannot read gh-pages archive member: {member.name}" + ) + with source, target.open("wb") as output: + shutil.copyfileobj(source, output) + else: + raise DocsDeltaError( + f"unsupported gh-pages archive member: {member.name}" + ) + except (OSError, tarfile.TarError) as exc: + raise DocsDeltaError(f"cannot extract gh-pages baseline: {exc}") from exc + + +@contextmanager +def _resolved_baseline( + args: argparse.Namespace, + github_pull_request: GithubPullRequest | None, + environ: Mapping[str, str], +) -> Iterator[tuple[Path | None, str]]: + """Resolve either a directory baseline or a published gh-pages baseline.""" + + if args.baseline_mode is None: + mode = ( + "directory" + if args.baseline_dir is not None or github_pull_request is None + else "gh-pages" + ) + else: + mode = args.baseline_mode + + if mode == "directory": + if args.baseline_dir is None: + raise DocsDeltaError( + "--baseline-dir is required outside GitHub pull request mode" + ) + yield args.baseline_dir, "" + return + + if args.baseline_dir is not None: + raise DocsDeltaError( + "--baseline-dir cannot be used with --baseline-mode gh-pages" + ) + if github_pull_request is None: + raise DocsDeltaError( + "gh-pages baseline mode requires a GitHub pull request event" + ) + + gh_pages_dir = args.gh_pages_dir or _workspace_path(environ, ".docs-baseline") + commit, reason = _find_published_baseline_commit(gh_pages_dir, github_pull_request) + if commit is None: + yield None, reason + return + + with tempfile.TemporaryDirectory(prefix="docs-delta-baseline-") as temporary_dir: + baseline_dir = Path(temporary_dir) + try: + _extract_git_tree( + gh_pages_dir, + commit, + github_pull_request.base_ref, + baseline_dir, + ) + except DocsDeltaError as exc: + yield None, str(exc) + return + yield baseline_dir, "" + + +def _github_url_defaults( + environ: Mapping[str, str], + github_pull_request: GithubPullRequest | None = None, +) -> tuple[str | None, str | None]: + """Derive conventional GitHub Pages URLs from standard Actions variables.""" + + repository = environ.get("GITHUB_REPOSITORY", "") + if "/" not in repository: + return None, None + owner, name = repository.split("/", 1) + if not owner or not name: + return None, None + pages_root = environ.get("GITHUB_PAGES_URL") or f"https://{owner}.github.io/{name}" + base_ref = ( + github_pull_request.base_ref + if github_pull_request + else (environ.get("GITHUB_BASE_REF") or "main") + ) + if github_pull_request and github_pull_request.number: + pr_ref = f"pr-{github_pull_request.number}" + else: + pr_ref = ( + environ.get("GITHUB_HEAD_REF") or environ.get("GITHUB_REF_NAME") or base_ref + ) + + def pages_ref_url(ref: str) -> str: + # A branch name is one URL path segment. In particular, a slash in a + # feature branch must not become a second documentation path segment. + return pages_root.rstrip("/") + "/" + quote(ref, safe="-._~") + + return ( + pages_ref_url(base_ref), + pages_ref_url(pr_ref), + ) + + +def _resolve_urls( + args: argparse.Namespace, github_pull_request: GithubPullRequest | None +) -> tuple[str, str]: + defaults = _github_url_defaults(os.environ, github_pull_request) + base_url = args.base_url or defaults[0] + pr_url = args.pr_url or defaults[1] + if not base_url or not pr_url: + raise DocsDeltaError( + "documentation URLs are required; provide --base-url and --pr-url " + "or run in GitHub Actions with GITHUB_REPOSITORY set" + ) + return base_url, pr_url + + +def argument_parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--baseline-dir", + type=Path, + help="directory baseline; defaults to automatic gh-pages mode in PR Actions", + ) + parser.add_argument( + "--baseline-mode", + choices=("directory", "gh-pages"), + help="baseline source; inferred when omitted", + ) + parser.add_argument( + "--gh-pages-dir", + type=Path, + help="local full-history gh-pages checkout (defaults to .docs-baseline)", + ) + parser.add_argument( + "--current-dir", + type=Path, + help="current documentation directory (defaults to docs-artifact)", + ) + parser.add_argument("--base-url") + parser.add_argument("--pr-url") + parser.add_argument( + "--output", + type=Path, + help="report path (defaults to docs-artifact/docs-delta.md)", + ) + return parser + + +def main(argv: Sequence[str] | None = None) -> int: + """Run the CLI and return a non-zero status only for current-input errors.""" + + args = argument_parser().parse_args(argv) + try: + github_pull_request = _github_pull_request(os.environ) + current_dir = args.current_dir or _workspace_path(os.environ, "docs-artifact") + output = args.output or current_dir / "docs-delta.md" + + with _resolved_baseline(args, github_pull_request, os.environ) as ( + baseline_dir, + baseline_reason, + ): + if baseline_dir is None or not baseline_dir.is_dir(): + reason = baseline_reason or f"directory is missing: {baseline_dir}" + _write_report(output, unavailable_report(reason)) + return 0 + try: + baseline_needs = load_needs(baseline_dir) + except DocsDeltaError as exc: + _write_report(output, unavailable_report(str(exc))) + return 0 + + if not current_dir.is_dir(): + raise DocsDeltaError( + f"documentation directory is missing: {current_dir}" + ) + current_needs = load_needs(current_dir) + base_url, pr_url = _resolve_urls(args, github_pull_request) + report = render_report( + compare_needs(baseline_needs, current_needs), + compare_html(baseline_dir, current_dir), + base_url=base_url, + pr_url=pr_url, + ) + _write_report(output, report) + except (DocsDeltaError, OSError) as exc: + print(f"docs_delta: error: {exc}", file=sys.stderr) + return 2 + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tools/tests/BUILD b/tools/tests/BUILD index e84b502a7..73772cdb9 100644 --- a/tools/tests/BUILD +++ b/tools/tests/BUILD @@ -23,3 +23,12 @@ score_pytest( ] + all_requirements, pytest_config = "//:pyproject.toml", ) + +score_pytest( + name = "docs_delta_test", + srcs = ["docs_delta_test.py"], + deps = [ + "//tools:docs_delta", + ] + all_requirements, + pytest_config = "//:pyproject.toml", +) diff --git a/tools/tests/docs_delta_test.py b/tools/tests/docs_delta_test.py new file mode 100644 index 000000000..2fb658b65 --- /dev/null +++ b/tools/tests/docs_delta_test.py @@ -0,0 +1,443 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License, Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +"""Unit and fixture-based integration tests for the documentation delta tool.""" + +from __future__ import annotations + +import json +import subprocess +from pathlib import Path + +import pytest + +from tools import docs_delta + + +def _need(docname: str, title: str, **extra: object) -> dict[str, object]: + return {"id": title.lower(), "docname": docname, "title": title, **extra} + + +def _write_needs(directory: Path, needs: dict[str, dict[str, object]]) -> None: + (directory / "needs.json").write_text( + json.dumps({"versions": {"1": {"needs": needs}}}), encoding="utf-8" + ) + + +def _git(directory: Path, *arguments: str) -> None: + subprocess.run( + ["git", "-C", str(directory), *arguments], + check=True, + capture_output=True, + text=True, + ) + + +def test_compare_needs_categorizes_changes_and_ignores_volatile_fields() -> None: + baseline = { + "same": _need( + "same", + "Same", + id="same", + lineno=10, + source="old.rst", + source_code_link="old-commit", + testlink="old-result", + ), + "changed": _need( + "changed", + "Old", + id="changed", + lineno=10, + source="old.rst", + source_code_link="old-commit", + testlink="old-result", + ), + "removed": _need("removed", "Removed", id="removed"), + } + current = { + "same": _need( + "same", + "Same", + id="same", + lineno=999, + source="new.rst", + source_code_link="new-commit", + testlink="new-result", + ), + "changed": _need( + "changed", + "New", + id="changed", + lineno=20, + source="new.rst", + source_code_link="new-commit", + testlink="new-result", + ), + "added": _need("added", "Added", id="added"), + } + + result = docs_delta.compare_needs(baseline, current) + + assert [entry.need_id for entry in result.added] == ["added"] + assert [entry.need_id for entry in result.removed] == ["removed"] + assert [entry.need_id for entry in result.modified] == ["changed"] + assert result.modified[0].changed_fields == ("title",) + assert result.unchanged_count == 1 + + +def test_modified_need_reports_all_changed_fields_and_one_sided_links() -> None: + result = docs_delta.render_report( + docs_delta.NeedComparison( + added=(docs_delta.NeedChange("added", None, _need("new/page", "Added")),), + removed=( + docs_delta.NeedChange("removed", _need("old/page", "Removed"), None), + ), + modified=( + docs_delta.NeedChange( + "changed", + _need("old/page", "Old", tags=["old"], extra="before"), + _need("new/page", "New", tags=["new"], extra="after"), + ), + ), + unchanged_count=0, + ), + docs_delta.PageComparison((), (), (), 0), + base_url="https://docs.example/main", + pr_url="https://docs.example/pr/1", + ) + + assert "[old](https://docs.example/main/old/page.html#changed)" in result + assert "[new](https://docs.example/pr/1/new/page.html#changed)" in result + assert '`extra`: "before" → "after"' in result + assert '`tags`: ["old"] → ["new"]' in result + assert "https://docs.example/pr/1/new/page.html#added" in result + assert "https://docs.example/main/old/page.html#removed" in result + + +def test_large_changed_values_are_truncated() -> None: + large_old = "a" * (docs_delta.MAX_RENDERED_VALUE_LENGTH + 20) + large_new = "b" * (docs_delta.MAX_RENDERED_VALUE_LENGTH + 20) + change = docs_delta.NeedChange( + "changed", + _need("page", "Old", details=large_old), + _need("page", "New", details=large_new), + ) + + report = docs_delta.render_report( + docs_delta.NeedComparison((), (), (change,), 0), + docs_delta.PageComparison((), (), (), 0), + base_url="https://docs.example/main", + pr_url="https://docs.example/pr/1", + ) + + assert "…" in report + assert large_old not in report + assert large_new not in report + + +def test_normalize_html_ignores_generated_metadata_but_keeps_content_changes() -> None: + old = ( + '\r\n' + '\r\n' + "\r\n" + "

Documentation

\r\n" + ) + new = ( + '\n' + '\n' + "\n" + "

Documentation

\n" + ) + + assert docs_delta.normalize_html(old) == docs_delta.normalize_html(new) + assert docs_delta.normalize_html(new).replace( + "Documentation", "Changed" + ) != docs_delta.normalize_html(old) + + +def test_compare_html_categorizes_added_removed_modified_and_unchanged( + tmp_path: Path, +) -> None: + baseline = tmp_path / "baseline" + current = tmp_path / "current" + (baseline / "nested").mkdir(parents=True) + (current / "nested").mkdir(parents=True) + (baseline / "same.html").write_text("

same

\n", encoding="utf-8") + (current / "same.html").write_text("

same

\n", encoding="utf-8") + (baseline / "changed.html").write_text("

old

\n", encoding="utf-8") + (current / "changed.html").write_text("

new

\n", encoding="utf-8") + (baseline / "removed.html").write_text("

removed

\n", encoding="utf-8") + (current / "added.html").write_text("

added

\n", encoding="utf-8") + (baseline / "nested/page.html").write_text("

old

\n", encoding="utf-8") + (current / "nested/page.html").write_text("

old

\n", encoding="utf-8") + + result = docs_delta.compare_html(baseline, current) + + assert [entry.path for entry in result.added] == ["added.html"] + assert [entry.path for entry in result.removed] == ["removed.html"] + assert [entry.path for entry in result.modified] == ["changed.html"] + assert result.unchanged_count == 2 + + +def test_detail_threshold_is_independent_for_needs_and_pages() -> None: + needs = docs_delta.NeedComparison( + added=tuple( + docs_delta.NeedChange(str(index), None, _need("page", str(index))) + for index in range(16) + ), + removed=(), + modified=(), + unchanged_count=0, + ) + pages = docs_delta.PageComparison( + added=tuple(docs_delta.PageChange(f"page-{index}.html") for index in range(2)), + removed=(), + modified=(), + unchanged_count=0, + ) + + report = docs_delta.render_report( + needs, + pages, + base_url="https://docs.example/main", + pr_url="https://docs.example/pr/1", + ) + + assert "16 entries changed; details omitted." in report + assert "`page-0.html`" in report + assert "`0`" not in report + + +def test_missing_baseline_writes_unavailable_report(tmp_path: Path) -> None: + output = tmp_path / "nested" / "delta.md" + + assert ( + docs_delta.main( + [ + "--baseline-dir", + str(tmp_path / "missing"), + "--current-dir", + str(tmp_path / "current"), + "--base-url", + "https://docs.example/main", + "--pr-url", + "https://docs.example/pr/1", + "--output", + str(output), + ] + ) + == 0 + ) + assert "## Delta unavailable" in output.read_text(encoding="utf-8") + + +def test_cli_writes_complete_fixture_report(tmp_path: Path) -> None: + baseline = tmp_path / "baseline" + current = tmp_path / "current" + baseline.mkdir() + current.mkdir() + _write_needs( + baseline, + { + "same": _need("same", "Same", id="same"), + "changed": _need("guide", "Old title", id="changed"), + "removed": _need("removed", "Removed", id="removed"), + }, + ) + _write_needs( + current, + { + "same": _need("same", "Same", id="same"), + "changed": _need("guide", "New title", id="changed"), + "added": _need("new", "Added", id="added"), + }, + ) + (baseline / "guide.html").write_text("

old

\n", encoding="utf-8") + (current / "guide.html").write_text("

new

\n", encoding="utf-8") + (current / "new.html").write_text("

new page

\n", encoding="utf-8") + output = tmp_path / "delta.md" + + assert ( + docs_delta.main( + [ + "--baseline-dir", + str(baseline), + "--current-dir", + str(current), + "--base-url", + "https://docs.example/main", + "--pr-url", + "https://docs.example/pr/1", + "--output", + str(output), + ] + ) + == 0 + ) + report = output.read_text(encoding="utf-8") + assert "Needs: 1 added, 1 removed, 1 modified, 1 unchanged" in report + assert "Rendered pages: 1 added, 0 removed, 1 modified, 0 unchanged" in report + assert '`title`: "Old title" → "New title"' in report + assert ( + "`guide.html` ([old](https://docs.example/main/guide.html) / [new](https://docs.example/pr/1/guide.html))" + in report + ) + + +def test_cli_uses_github_environment_urls_when_options_are_omitted( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + baseline = tmp_path / "baseline" + current = tmp_path / "current" + baseline.mkdir() + current.mkdir() + _write_needs(baseline, {}) + _write_needs(current, {}) + monkeypatch.setenv("GITHUB_REPOSITORY", "eclipse-score/docs-as-code") + monkeypatch.setenv("GITHUB_BASE_REF", "main") + monkeypatch.setenv("GITHUB_HEAD_REF", "feature/docs-delta") + output = tmp_path / "delta.md" + + assert ( + docs_delta.main( + [ + "--baseline-dir", + str(baseline), + "--current-dir", + str(current), + "--output", + str(output), + ] + ) + == 0 + ) + report = output.read_text(encoding="utf-8") + assert "https://eclipse-score.github.io/docs-as-code/main" in report + assert "https://eclipse-score.github.io/docs-as-code/feature%2Fdocs-delta" in report + + +def test_cli_automatically_selects_matching_gh_pages_baseline( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """PR mode must select the historical publication matching the base SHA.""" + + gh_pages = tmp_path / ".docs-baseline" + gh_pages.mkdir() + subprocess.run( + ["git", "init", "--initial-branch=gh-pages", str(gh_pages)], + check=True, + capture_output=True, + text=True, + ) + _git(gh_pages, "config", "user.name", "Documentation Delta Test") + _git(gh_pages, "config", "user.email", "docs-delta@example.invalid") + + base_sha = "0123456789abcdef0123456789abcdef01234567" + latest_sha = "fedcba9876543210fedcba9876543210fedcba98" + main_docs = gh_pages / "main" + main_docs.mkdir() + _write_needs( + main_docs, + { + "need": _need( + "guide", "Base publication", id="need", source_code_link=base_sha + ) + }, + ) + (main_docs / "guide.html").write_text("

Base publication

\n", encoding="utf-8") + _git(gh_pages, "add", ".") + _git(gh_pages, "commit", "-m", "Publish base documentation") + + _write_needs( + main_docs, + { + "need": _need( + "guide", + "Newer base publication", + id="need", + source_code_link=base_sha, + ) + }, + ) + (main_docs / "guide.html").write_text( + "

Newer base publication

\n", encoding="utf-8" + ) + _git(gh_pages, "add", ".") + _git(gh_pages, "commit", "-m", "Republish base documentation") + newer_base_publication_commit = subprocess.run( + ["git", "-C", str(gh_pages), "rev-parse", "HEAD"], + check=True, + capture_output=True, + text=True, + ).stdout.strip() + + _write_needs( + main_docs, + { + "need": _need( + "guide", "Latest publication", id="need", source_code_link=latest_sha + ) + }, + ) + (main_docs / "guide.html").write_text( + "

Latest publication

\n", encoding="utf-8" + ) + _git(gh_pages, "add", ".") + _git(gh_pages, "commit", "-m", "Publish latest documentation") + + selected_commit, selection_reason = docs_delta._find_published_baseline_commit( + gh_pages, + docs_delta.GithubPullRequest("main", base_sha, "123"), + ) + assert selected_commit == newer_base_publication_commit, selection_reason + + current = tmp_path / "docs-artifact" + current.mkdir() + _write_needs( + current, + { + "need": _need( + "guide", + "Newer base publication", + id="need", + source_code_link="pr-sha", + ) + }, + ) + (current / "guide.html").write_text( + "

Newer base publication

\n", encoding="utf-8" + ) + + event = tmp_path / "event.json" + event.write_text( + json.dumps( + { + "pull_request": { + "number": 123, + "base": {"ref": "main", "sha": base_sha}, + } + } + ), + encoding="utf-8", + ) + monkeypatch.setenv("GITHUB_EVENT_NAME", "pull_request") + monkeypatch.setenv("GITHUB_EVENT_PATH", str(event)) + monkeypatch.setenv("GITHUB_WORKSPACE", str(tmp_path)) + monkeypatch.setenv("GITHUB_REPOSITORY", "eclipse-score/docs-as-code") + monkeypatch.setenv("GITHUB_BASE_REF", "main") + + assert docs_delta.main([]) == 0 + report = (current / "docs-delta.md").read_text(encoding="utf-8") + assert "Needs: 0 added, 0 removed, 0 modified, 1 unchanged" in report + assert "Rendered pages: 0 added, 0 removed, 0 modified, 1 unchanged" in report + assert "https://eclipse-score.github.io/docs-as-code/pr-123" in report From 5217f603523c5075d34a9d1a2116c78c307880be Mon Sep 17 00:00:00 2001 From: Alexander Lanin Date: Wed, 2 Sep 2026 11:12:18 +0200 Subject: [PATCH 2/7] Fix upload artifact action pin --- .github/workflows/on-pr.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/on-pr.yml b/.github/workflows/on-pr.yml index 642b1a42c..5e4dd1936 100644 --- a/.github/workflows/on-pr.yml +++ b/.github/workflows/on-pr.yml @@ -84,7 +84,7 @@ jobs: uv run --no-project python tools/docs_delta.py - name: Upload documentation delta artifact - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0 # v7.0.1 + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: docs-delta path: docs-artifact/docs-delta.md From 40381b1bd898e52d9e3f0c08f321e02dd41a74c6 Mon Sep 17 00:00:00 2001 From: Alexander Lanin Date: Tue, 29 Sep 2026 23:29:54 +0200 Subject: [PATCH 3/7] Fix docs delta checks and stale preview comments --- .github/workflows/on-pr.yml | 30 ++++++--- tools/docs_delta.py | 111 ++++++++++++++++++++++++--------- tools/tests/docs_delta_test.py | 13 ---- 3 files changed, 106 insertions(+), 48 deletions(-) diff --git a/.github/workflows/on-pr.yml b/.github/workflows/on-pr.yml index 5e4dd1936..9794b2729 100644 --- a/.github/workflows/on-pr.yml +++ b/.github/workflows/on-pr.yml @@ -91,8 +91,9 @@ jobs: if-no-files-found: error docs-comment: - needs: [docs-delta] + needs: [docs-build, docs-delta] if: >- + always() && github.event_name == 'pull_request' && github.event.pull_request.head.repo.full_name == github.repository runs-on: ubuntu-24.04 @@ -101,6 +102,7 @@ jobs: pull-requests: write steps: - name: Download documentation delta artifact + if: needs.docs-delta.result == 'success' uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 with: name: docs-delta @@ -111,14 +113,28 @@ jobs: PR_NUMBER: ${{ github.event.pull_request.number }} REPOSITORY_OWNER: ${{ github.repository_owner }} REPOSITORY_NAME: ${{ github.event.repository.name }} + DOCS_BUILD_RESULT: ${{ needs.docs-build.result }} + DOCS_DELTA_RESULT: ${{ needs.docs-delta.result }} run: | set -euo pipefail - { - echo "Documentation preview for this pull request is available at:" - echo "**pr-${PR_NUMBER}**: https://${REPOSITORY_OWNER}.github.io/${REPOSITORY_NAME}/pr-${PR_NUMBER}/" - echo - cat docs-delta/docs-delta.md - } > "$RUNNER_TEMP/docs-comment.md" + if [ "$DOCS_BUILD_RESULT" = success ] && [ "$DOCS_DELTA_RESULT" = success ]; then + { + echo "Documentation preview for this pull request is available at:" + echo "**pr-${PR_NUMBER}**: https://${REPOSITORY_OWNER}.github.io/${REPOSITORY_NAME}/pr-${PR_NUMBER}/" + echo + cat docs-delta/docs-delta.md + } > "$RUNNER_TEMP/docs-comment.md" + else + { + echo "Documentation preview for this pull request is unavailable." + echo + if [ "$DOCS_BUILD_RESULT" != success ]; then + echo "The documentation build did not complete successfully (result: ${DOCS_BUILD_RESULT})." + else + echo "The documentation delta job did not complete successfully (result: ${DOCS_DELTA_RESULT})." + fi + } > "$RUNNER_TEMP/docs-comment.md" + fi - name: Find existing documentation preview comment id: find-comment diff --git a/tools/docs_delta.py b/tools/docs_delta.py index c14552d92..6125865f4 100644 --- a/tools/docs_delta.py +++ b/tools/docs_delta.py @@ -34,11 +34,13 @@ from contextlib import contextmanager from dataclasses import dataclass from pathlib import Path +from typing import Protocol, cast from urllib.parse import quote JsonObject = dict[str, object] NeedMap = Mapping[str, JsonObject] + DETAIL_LIMIT = 15 MAX_RENDERED_VALUE_LENGTH = 600 @@ -158,6 +160,32 @@ class PageComparison: unchanged_count: int +class NeedFormatter(Protocol): + """Format a Need change for a Markdown report section.""" + + def __call__( + self, + change: NeedChange, + *, + base_url: str, + pr_url: str, + include_diff: bool = False, + ) -> list[str]: ... + + +class PageFormatter(Protocol): + """Format a rendered page change for a Markdown report section.""" + + def __call__( + self, + change: PageChange, + *, + base_url: str, + pr_url: str, + kind: str, + ) -> str: ... + + def _json_values_equal(left: object, right: object) -> bool: """Compare JSON values without conflating booleans and numbers.""" @@ -179,14 +207,20 @@ def _flatten_needs(path: Path) -> dict[str, JsonObject]: if not isinstance(payload, dict) or not isinstance(payload.get("versions"), dict): raise DocsDeltaError(f"needs JSON has no versions map: {path}") + versions = cast(dict[str, object], payload["versions"]) result: dict[str, JsonObject] = {} - for version in payload["versions"].values(): - if not isinstance(version, dict) or not isinstance(version.get("needs"), dict): + for version_value in versions.values(): + if not isinstance(version_value, dict): + continue + version = cast(dict[str, object], version_value) + needs_value = version.get("needs") + if not isinstance(needs_value, dict): continue - for need_id, need in version["needs"].items(): + needs = cast(dict[object, object], needs_value) + for need_id, need in needs.items(): if not isinstance(need_id, str) or not isinstance(need, dict): raise DocsDeltaError(f"invalid Need entry in {path}") - result[need_id] = need + result[need_id] = cast(JsonObject, need) return result @@ -408,11 +442,11 @@ def _format_page_entry( return f"- `{change.path}` ({' / '.join(links)})" -def _section( +def _need_section( lines: list[str], title: str, - entries: Sequence[object], - formatter, + entries: Sequence[NeedChange], + formatter: NeedFormatter, *, base_url: str, pr_url: str, @@ -425,17 +459,35 @@ def _section( lines.extend([f"{len(entries)} entries changed; details omitted.", ""]) return for entry in entries: - if isinstance(entry, NeedChange): - lines.extend( - formatter( - entry, - base_url=base_url, - pr_url=pr_url, - include_diff=kind == "modified", - ) + lines.extend( + formatter( + entry, + base_url=base_url, + pr_url=pr_url, + include_diff=kind == "modified", ) - else: - lines.append(formatter(entry, base_url=base_url, pr_url=pr_url, kind=kind)) + ) + lines.append("") + + +def _page_section( + lines: list[str], + title: str, + entries: Sequence[PageChange], + formatter: PageFormatter, + *, + base_url: str, + pr_url: str, + kind: str, +) -> None: + if not entries: + return + lines.extend([f"### {title} ({len(entries)})", ""]) + if len(entries) > DETAIL_LIMIT: + lines.extend([f"{len(entries)} entries changed; details omitted.", ""]) + return + for entry in entries: + lines.append(formatter(entry, base_url=base_url, pr_url=pr_url, kind=kind)) lines.append("") @@ -464,7 +516,7 @@ def render_report( "## Needs", "", ] - _section( + _need_section( lines, "Added", needs.added, @@ -473,7 +525,7 @@ def render_report( pr_url=pr_url, kind="added", ) - _section( + _need_section( lines, "Removed", needs.removed, @@ -482,7 +534,7 @@ def render_report( pr_url=pr_url, kind="removed", ) - _section( + _need_section( lines, "Modified", needs.modified, @@ -492,7 +544,7 @@ def render_report( kind="modified", ) lines.extend(["## Rendered HTML pages", ""]) - _section( + _page_section( lines, "Added", pages.added, @@ -501,7 +553,7 @@ def render_report( pr_url=pr_url, kind="added", ) - _section( + _page_section( lines, "Removed", pages.removed, @@ -510,7 +562,7 @@ def render_report( pr_url=pr_url, kind="removed", ) - _section( + _page_section( lines, "Modified", pages.modified, @@ -584,12 +636,15 @@ def _github_pull_request(environ: Mapping[str, str]) -> GithubPullRequest | None if not isinstance(payload, dict): raise DocsDeltaError(f"GitHub event payload is not an object: {event_path}") - pull_request = payload.get("pull_request") - if not isinstance(pull_request, dict): + event_payload = cast(dict[str, object], payload) + pull_request_value = event_payload.get("pull_request") + if not isinstance(pull_request_value, dict): return None - base = pull_request.get("base") - if not isinstance(base, dict): + pull_request = cast(dict[str, object], pull_request_value) + base_value = pull_request.get("base") + if not isinstance(base_value, dict): return None + base = cast(dict[str, object], base_value) base_ref = environ.get("GITHUB_BASE_REF") or base.get("ref") base_sha = environ.get("GITHUB_BASE_SHA") or base.get("sha") @@ -598,7 +653,7 @@ def _github_pull_request(environ: Mapping[str, str]) -> GithubPullRequest | None if not isinstance(base_sha, str) or not base_sha: return None - number = pull_request.get("number") + number: object = pull_request.get("number") if not isinstance(number, int | str) or isinstance(number, bool): number = None else: diff --git a/tools/tests/docs_delta_test.py b/tools/tests/docs_delta_test.py index 2fb658b65..764186251 100644 --- a/tools/tests/docs_delta_test.py +++ b/tools/tests/docs_delta_test.py @@ -374,13 +374,6 @@ def test_cli_automatically_selects_matching_gh_pages_baseline( ) _git(gh_pages, "add", ".") _git(gh_pages, "commit", "-m", "Republish base documentation") - newer_base_publication_commit = subprocess.run( - ["git", "-C", str(gh_pages), "rev-parse", "HEAD"], - check=True, - capture_output=True, - text=True, - ).stdout.strip() - _write_needs( main_docs, { @@ -395,12 +388,6 @@ def test_cli_automatically_selects_matching_gh_pages_baseline( _git(gh_pages, "add", ".") _git(gh_pages, "commit", "-m", "Publish latest documentation") - selected_commit, selection_reason = docs_delta._find_published_baseline_commit( - gh_pages, - docs_delta.GithubPullRequest("main", base_sha, "123"), - ) - assert selected_commit == newer_base_publication_commit, selection_reason - current = tmp_path / "docs-artifact" current.mkdir() _write_needs( From a53c133fb741048970c4e9746cd8f58f82329c09 Mon Sep 17 00:00:00 2001 From: Alexander Lanin Date: Tue, 29 Sep 2026 23:51:31 +0200 Subject: [PATCH 4/7] Ignore generated revision noise in docs delta --- tools/docs_delta.py | 9 +++++++++ tools/tests/docs_delta_test.py | 5 +++++ 2 files changed, 14 insertions(+) diff --git a/tools/docs_delta.py b/tools/docs_delta.py index 6125865f4..ece943f4e 100644 --- a/tools/docs_delta.py +++ b/tools/docs_delta.py @@ -71,6 +71,13 @@ r"\s+data-(?:build|generated|timestamp|last-modified)(?:-[\w-]+)?\s*=\s*(?:\"[^\"]*\"|'[^']*'|[^\s>]+)", re.IGNORECASE, ) +# PR documentation builds run on GitHub's synthetic merge commit, while the +# published baseline was built from the base commit. Keep source paths and line +# numbers comparable, but ignore the revision segment in generated blob links. +_GITHUB_BLOB_COMMIT = re.compile(r"(?<=/blob/)[0-9a-f]{40}(?=/)", re.IGNORECASE) +# Sphinx-Needs derives these wrapper IDs from rendered content, including the +# source-link revision above. Need IDs and visible content remain compared. +_NEED_CONTAINER_ID = re.compile(r"\bSNCB-[0-9a-f]{8}\b", re.IGNORECASE) _VOLATILE_META_NAMES = frozenset( { "build-date", @@ -285,6 +292,8 @@ def remove_comment(match: re.Match[str]) -> str: normalized = _HTML_COMMENT.sub(remove_comment, normalized) normalized = _VOLATILE_ATTRIBUTE.sub("", normalized) + normalized = _GITHUB_BLOB_COMMIT.sub("", normalized) + normalized = _NEED_CONTAINER_ID.sub("SNCB-", normalized) normalized = "\n".join(line.rstrip() for line in normalized.splitlines()) return normalized.strip() diff --git a/tools/tests/docs_delta_test.py b/tools/tests/docs_delta_test.py index 764186251..ab73d35e6 100644 --- a/tools/tests/docs_delta_test.py +++ b/tools/tests/docs_delta_test.py @@ -150,16 +150,21 @@ def test_normalize_html_ignores_generated_metadata_but_keeps_content_changes() - '\r\n' '\r\n' "\r\n" + '\r\n' "

Documentation

\r\n" ) new = ( '\n' '\n' "\n" + '\n' "

Documentation

\n" ) assert docs_delta.normalize_html(old) == docs_delta.normalize_html(new) + assert docs_delta.normalize_html( + new.replace("src/example.py", "src/changed.py") + ) != docs_delta.normalize_html(old) assert docs_delta.normalize_html(new).replace( "Documentation", "Changed" ) != docs_delta.normalize_html(old) From 86dd80037e0bc8740e7e0149ca3895591cd67113 Mon Sep 17 00:00:00 2001 From: Alexander Lanin Date: Wed, 30 Sep 2026 00:01:32 +0200 Subject: [PATCH 5/7] Make docs delta PR comment readable --- .github/workflows/on-pr.yml | 10 +-- tools/docs_delta.py | 154 +++++++++++++++++++-------------- tools/tests/docs_delta_test.py | 3 +- 3 files changed, 94 insertions(+), 73 deletions(-) diff --git a/.github/workflows/on-pr.yml b/.github/workflows/on-pr.yml index 9794b2729..73414479f 100644 --- a/.github/workflows/on-pr.yml +++ b/.github/workflows/on-pr.yml @@ -110,20 +110,12 @@ jobs: - name: Prepare documentation preview comment env: - PR_NUMBER: ${{ github.event.pull_request.number }} - REPOSITORY_OWNER: ${{ github.repository_owner }} - REPOSITORY_NAME: ${{ github.event.repository.name }} DOCS_BUILD_RESULT: ${{ needs.docs-build.result }} DOCS_DELTA_RESULT: ${{ needs.docs-delta.result }} run: | set -euo pipefail if [ "$DOCS_BUILD_RESULT" = success ] && [ "$DOCS_DELTA_RESULT" = success ]; then - { - echo "Documentation preview for this pull request is available at:" - echo "**pr-${PR_NUMBER}**: https://${REPOSITORY_OWNER}.github.io/${REPOSITORY_NAME}/pr-${PR_NUMBER}/" - echo - cat docs-delta/docs-delta.md - } > "$RUNNER_TEMP/docs-comment.md" + cat docs-delta/docs-delta.md > "$RUNNER_TEMP/docs-comment.md" else { echo "Documentation preview for this pull request is unavailable." diff --git a/tools/docs_delta.py b/tools/docs_delta.py index ece943f4e..7dcc4ce69 100644 --- a/tools/docs_delta.py +++ b/tools/docs_delta.py @@ -492,9 +492,6 @@ def _page_section( if not entries: return lines.extend([f"### {title} ({len(entries)})", ""]) - if len(entries) > DETAIL_LIMIT: - lines.extend([f"{len(entries)} entries changed; details omitted.", ""]) - return for entry in entries: lines.append(formatter(entry, base_url=base_url, pr_url=pr_url, kind=kind)) lines.append("") @@ -512,8 +509,9 @@ def render_report( lines = [ "# Documentation delta", "", - f"Baseline: {_markdown_link(base_url, base_url)} ", - f"PR preview: {_markdown_link(pr_url, pr_url)}", + "Documentation preview for this pull request: " + f"{_markdown_link('open preview', pr_url)} · " + f"Baseline: {_markdown_link('open baseline', base_url)}", "", "## Summary", "", @@ -521,65 +519,95 @@ def render_report( f"{len(needs.modified)} modified, {needs.unchanged_count} unchanged", f"- Rendered pages: {len(pages.added)} added, {len(pages.removed)} removed, " f"{len(pages.modified)} modified, {pages.unchanged_count} unchanged", - "", - "## Needs", - "", ] - _need_section( - lines, - "Added", - needs.added, - _format_need_entry, - base_url=base_url, - pr_url=pr_url, - kind="added", - ) - _need_section( - lines, - "Removed", - needs.removed, - _format_need_entry, - base_url=base_url, - pr_url=pr_url, - kind="removed", - ) - _need_section( - lines, - "Modified", - needs.modified, - _format_need_entry, - base_url=base_url, - pr_url=pr_url, - kind="modified", - ) - lines.extend(["## Rendered HTML pages", ""]) - _page_section( - lines, - "Added", - pages.added, - _format_page_entry, - base_url=base_url, - pr_url=pr_url, - kind="added", - ) - _page_section( - lines, - "Removed", - pages.removed, - _format_page_entry, - base_url=base_url, - pr_url=pr_url, - kind="removed", - ) - _page_section( - lines, - "Modified", - pages.modified, - _format_page_entry, - base_url=base_url, - pr_url=pr_url, - kind="modified", - ) + + need_change_count = len(needs.added) + len(needs.removed) + len(needs.modified) + if need_change_count: + need_lines: list[str] = [] + _need_section( + need_lines, + "Added", + needs.added, + _format_need_entry, + base_url=base_url, + pr_url=pr_url, + kind="added", + ) + _need_section( + need_lines, + "Removed", + needs.removed, + _format_need_entry, + base_url=base_url, + pr_url=pr_url, + kind="removed", + ) + _need_section( + need_lines, + "Modified", + needs.modified, + _format_need_entry, + base_url=base_url, + pr_url=pr_url, + kind="modified", + ) + if need_change_count > DETAIL_LIMIT: + lines.extend( + [ + "", + "
", + f"Need changes ({need_change_count})", + "", + *need_lines, + "
", + ] + ) + else: + lines.extend(["", "## Need changes", "", *need_lines]) + + page_change_count = len(pages.added) + len(pages.removed) + len(pages.modified) + if page_change_count: + page_lines: list[str] = [] + _page_section( + page_lines, + "Added", + pages.added, + _format_page_entry, + base_url=base_url, + pr_url=pr_url, + kind="added", + ) + _page_section( + page_lines, + "Removed", + pages.removed, + _format_page_entry, + base_url=base_url, + pr_url=pr_url, + kind="removed", + ) + _page_section( + page_lines, + "Modified", + pages.modified, + _format_page_entry, + base_url=base_url, + pr_url=pr_url, + kind="modified", + ) + if page_change_count > DETAIL_LIMIT: + lines.extend( + [ + "", + "
", + f"Page changes ({page_change_count})", + "", + *page_lines, + "
", + ] + ) + else: + lines.extend(["", "## Page changes", "", *page_lines]) return "\n".join(lines).rstrip() + "\n" diff --git a/tools/tests/docs_delta_test.py b/tools/tests/docs_delta_test.py index ab73d35e6..7f9a98a1f 100644 --- a/tools/tests/docs_delta_test.py +++ b/tools/tests/docs_delta_test.py @@ -205,7 +205,7 @@ def test_detail_threshold_is_independent_for_needs_and_pages() -> None: unchanged_count=0, ) pages = docs_delta.PageComparison( - added=tuple(docs_delta.PageChange(f"page-{index}.html") for index in range(2)), + added=tuple(docs_delta.PageChange(f"page-{index}.html") for index in range(16)), removed=(), modified=(), unchanged_count=0, @@ -219,6 +219,7 @@ def test_detail_threshold_is_independent_for_needs_and_pages() -> None: ) assert "16 entries changed; details omitted." in report + assert "Page changes (16)" in report assert "`page-0.html`" in report assert "`0`" not in report From 8340a67661ee77f7ed7a00992632b750fb6bedc2 Mon Sep 17 00:00:00 2001 From: Alexander Lanin Date: Wed, 30 Sep 2026 00:49:39 +0200 Subject: [PATCH 6/7] Split docs delta into a uv package --- .github/workflows/on-pr.yml | 2 +- tools/BUILD | 16 +- tools/docs_delta.py | 1006 --------------------- tools/docs_delta/README.md | 39 + tools/docs_delta/__init__.py | 50 + tools/docs_delta/__main__.py | 17 + tools/docs_delta/cli.py | 109 +++ tools/docs_delta/comparison.py | 205 +++++ tools/docs_delta/github.py | 369 ++++++++ tools/docs_delta/pyproject.toml | 26 + tools/docs_delta/rendered_html.py | 154 ++++ tools/docs_delta/report.py | 378 ++++++++ tools/docs_delta/uv.lock | 8 + tools/docs_delta_main.py | 18 + tools/tests/BUILD | 6 +- tools/tests/docs_delta_cli_test.py | 301 ++++++ tools/tests/docs_delta_comparison_test.py | 123 +++ tools/tests/docs_delta_github_test.py | 300 ++++++ tools/tests/docs_delta_html_test.py | 83 ++ tools/tests/docs_delta_report_test.py | 155 ++++ tools/tests/docs_delta_test.py | 436 --------- tools/tests/docs_delta_test_support.py | 41 + 22 files changed, 2392 insertions(+), 1450 deletions(-) delete mode 100644 tools/docs_delta.py create mode 100644 tools/docs_delta/README.md create mode 100644 tools/docs_delta/__init__.py create mode 100644 tools/docs_delta/__main__.py create mode 100644 tools/docs_delta/cli.py create mode 100644 tools/docs_delta/comparison.py create mode 100644 tools/docs_delta/github.py create mode 100644 tools/docs_delta/pyproject.toml create mode 100644 tools/docs_delta/rendered_html.py create mode 100644 tools/docs_delta/report.py create mode 100644 tools/docs_delta/uv.lock create mode 100644 tools/docs_delta_main.py create mode 100644 tools/tests/docs_delta_cli_test.py create mode 100644 tools/tests/docs_delta_comparison_test.py create mode 100644 tools/tests/docs_delta_github_test.py create mode 100644 tools/tests/docs_delta_html_test.py create mode 100644 tools/tests/docs_delta_report_test.py delete mode 100644 tools/tests/docs_delta_test.py create mode 100644 tools/tests/docs_delta_test_support.py diff --git a/.github/workflows/on-pr.yml b/.github/workflows/on-pr.yml index 73414479f..67e78cd01 100644 --- a/.github/workflows/on-pr.yml +++ b/.github/workflows/on-pr.yml @@ -81,7 +81,7 @@ jobs: - name: Generate documentation delta report run: | set -euo pipefail - uv run --no-project python tools/docs_delta.py + uv run --locked --project tools/docs_delta docs-delta - name: Upload documentation delta artifact uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 diff --git a/tools/BUILD b/tools/BUILD index 82aa238c3..b9d093c5b 100644 --- a/tools/BUILD +++ b/tools/BUILD @@ -5,7 +5,7 @@ # information regarding copyright ownership. # # This program and the accompanying materials are made available under the -# terms of the Apache License, Version 2.0 which is available at +# terms of the Apache License Version 2.0 which is available at # https://www.apache.org/licenses/LICENSE-2.0 # # SPDX-License-Identifier: Apache-2.0 @@ -34,9 +34,17 @@ py_binary( ) py_binary( - name = "docs_delta", - srcs = ["docs_delta.py"], - main = "docs_delta.py", + name = "docs_delta_cli", + srcs = ["docs_delta_main.py"] + glob(["docs_delta/*.py"]), + main = "docs_delta_main.py", visibility = ["//visibility:public"], deps = [], ) + +# Retain the original Bazel label for callers while the package itself now +# occupies the ``tools/docs_delta`` source path. +alias( + name = "docs_delta", + actual = ":docs_delta_cli", + visibility = ["//visibility:public"], +) diff --git a/tools/docs_delta.py b/tools/docs_delta.py deleted file mode 100644 index 7dcc4ce69..000000000 --- a/tools/docs_delta.py +++ /dev/null @@ -1,1006 +0,0 @@ -# ******************************************************************************* -# Copyright (c) 2026 Contributors to the Eclipse Foundation -# -# See the NOTICE file(s) distributed with this work for additional -# information regarding copyright ownership. -# -# This program and the accompanying materials are made available under the -# terms of the Apache License, Version 2.0 which is available at -# https://www.apache.org/licenses/LICENSE-2.0 -# -# SPDX-License-Identifier: Apache-2.0 -# ******************************************************************************* -"""Compare rendered documentation trees and write a Markdown delta report. - -The tool works on artifacts already produced by a docs build. In GitHub Actions -it can additionally resolve the matching published baseline from a local, -full-history checkout of ``gh-pages``. It never fetches from GitHub or invokes -Sphinx, which keeps the comparison useful both in CI and as a local command. -""" - -from __future__ import annotations - -import argparse -import io -import json -import os -import re -import shutil -import subprocess -import sys -import tarfile -import tempfile -from collections.abc import Iterator, Mapping, Sequence -from contextlib import contextmanager -from dataclasses import dataclass -from pathlib import Path -from typing import Protocol, cast -from urllib.parse import quote - -JsonObject = dict[str, object] -NeedMap = Mapping[str, JsonObject] - - -DETAIL_LIMIT = 15 -MAX_RENDERED_VALUE_LENGTH = 600 - -# These values are generated from source/test links or Sphinx's internal -# bookkeeping. They can change when a build is moved to another checkout or -# when the same source is rebuilt, without representing a documentation delta. -# Keep this list deliberately explicit: fields not listed here are part of the -# comparison contract and must be reviewed when Sphinx-Needs adds new output. -VOLATILE_NEED_FIELDS = frozenset( - { - "lineno", - "lineno_content", - "source", - "target_id", - "is_modified", - "source_code_link", - "testlink", - } -) - -_MISSING = object() -_META_TAG = re.compile(r"]*>", re.IGNORECASE) -_META_ATTRIBUTE = re.compile( - r"(?:name|property|itemprop)\s*=\s*([\"'])(.*?)\1", re.IGNORECASE -) -_HTML_COMMENT = re.compile(r"", re.IGNORECASE | re.DOTALL) -_VOLATILE_ATTRIBUTE = re.compile( - r"\s+data-(?:build|generated|timestamp|last-modified)(?:-[\w-]+)?\s*=\s*(?:\"[^\"]*\"|'[^']*'|[^\s>]+)", - re.IGNORECASE, -) -# PR documentation builds run on GitHub's synthetic merge commit, while the -# published baseline was built from the base commit. Keep source paths and line -# numbers comparable, but ignore the revision segment in generated blob links. -_GITHUB_BLOB_COMMIT = re.compile(r"(?<=/blob/)[0-9a-f]{40}(?=/)", re.IGNORECASE) -# Sphinx-Needs derives these wrapper IDs from rendered content, including the -# source-link revision above. Need IDs and visible content remain compared. -_NEED_CONTAINER_ID = re.compile(r"\bSNCB-[0-9a-f]{8}\b", re.IGNORECASE) -_VOLATILE_META_NAMES = frozenset( - { - "build-date", - "build_date", - "created", - "date", - "generated", - "generated-at", - "generated_at", - "generator", - "last-modified", - "last_modified", - "timestamp", - } -) -_VOLATILE_COMMENT_WORDS = re.compile( - r"\b(?:build|built|generated|generator|last\s+updated|timestamp)\b", - re.IGNORECASE, -) - - -class DocsDeltaError(ValueError): - """A user-actionable input or report-generation error.""" - - -@dataclass(frozen=True) -class GithubPullRequest: - """Pull request provenance obtained from the Actions environment.""" - - base_ref: str - base_sha: str - number: str | None - - -@dataclass(frozen=True) -class NeedChange: - """One Need and, for modified entries, its two versions.""" - - need_id: str - baseline: JsonObject | None - current: JsonObject | None - - @property - def changed_fields(self) -> tuple[str, ...]: - """Return changed non-volatile fields in deterministic order.""" - - if self.baseline is None or self.current is None: - return () - names = set(self.baseline) | set(self.current) - return tuple( - sorted( - name - for name in names - if name not in VOLATILE_NEED_FIELDS - and not _json_values_equal( - self.baseline.get(name, _MISSING), - self.current.get(name, _MISSING), - ) - ) - ) - - -@dataclass(frozen=True) -class NeedComparison: - """Categorized comparison result for a pair of Need inventories.""" - - added: tuple[NeedChange, ...] - removed: tuple[NeedChange, ...] - modified: tuple[NeedChange, ...] - unchanged_count: int - - -@dataclass(frozen=True) -class PageChange: - """One rendered HTML path that was added, removed, or modified.""" - - path: str - - -@dataclass(frozen=True) -class PageComparison: - """Categorized comparison result for rendered HTML pages.""" - - added: tuple[PageChange, ...] - removed: tuple[PageChange, ...] - modified: tuple[PageChange, ...] - unchanged_count: int - - -class NeedFormatter(Protocol): - """Format a Need change for a Markdown report section.""" - - def __call__( - self, - change: NeedChange, - *, - base_url: str, - pr_url: str, - include_diff: bool = False, - ) -> list[str]: ... - - -class PageFormatter(Protocol): - """Format a rendered page change for a Markdown report section.""" - - def __call__( - self, - change: PageChange, - *, - base_url: str, - pr_url: str, - kind: str, - ) -> str: ... - - -def _json_values_equal(left: object, right: object) -> bool: - """Compare JSON values without conflating booleans and numbers.""" - - if left is _MISSING or right is _MISSING: - return left is right - return json.dumps(left, sort_keys=True, ensure_ascii=False) == json.dumps( - right, sort_keys=True, ensure_ascii=False - ) - - -def _flatten_needs(path: Path) -> dict[str, JsonObject]: - """Load all Need entries from every version in a Sphinx-Needs export.""" - - try: - payload = json.loads(path.read_text(encoding="utf-8")) - except (OSError, UnicodeDecodeError, json.JSONDecodeError) as exc: - raise DocsDeltaError(f"cannot read needs JSON {path}: {exc}") from exc - - if not isinstance(payload, dict) or not isinstance(payload.get("versions"), dict): - raise DocsDeltaError(f"needs JSON has no versions map: {path}") - - versions = cast(dict[str, object], payload["versions"]) - result: dict[str, JsonObject] = {} - for version_value in versions.values(): - if not isinstance(version_value, dict): - continue - version = cast(dict[str, object], version_value) - needs_value = version.get("needs") - if not isinstance(needs_value, dict): - continue - needs = cast(dict[object, object], needs_value) - for need_id, need in needs.items(): - if not isinstance(need_id, str) or not isinstance(need, dict): - raise DocsDeltaError(f"invalid Need entry in {path}") - result[need_id] = cast(JsonObject, need) - return result - - -def load_needs(directory: Path) -> dict[str, JsonObject]: - """Load the root ``needs.json`` file from a documentation directory.""" - - needs_path = directory / "needs.json" - if not needs_path.is_file(): - raise DocsDeltaError(f"needs JSON is missing: {needs_path}") - return _flatten_needs(needs_path) - - -def compare_needs(baseline: NeedMap, current: NeedMap) -> NeedComparison: - """Compare Needs by ID while ignoring only the volatile field allowlist.""" - - added: list[NeedChange] = [] - removed: list[NeedChange] = [] - modified: list[NeedChange] = [] - unchanged_count = 0 - - for need_id in sorted(set(baseline) | set(current)): - old = baseline.get(need_id) - new = current.get(need_id) - change = NeedChange(need_id, old, new) - if old is None: - added.append(change) - elif new is None: - removed.append(change) - elif change.changed_fields: - modified.append(change) - else: - unchanged_count += 1 - - return NeedComparison( - added=tuple(added), - removed=tuple(removed), - modified=tuple(modified), - unchanged_count=unchanged_count, - ) - - -def _html_metadata_name(tag: str) -> str | None: - for match in _META_ATTRIBUTE.finditer(tag): - name = match.group(2).lower() - if name in _VOLATILE_META_NAMES: - return name - return None - - -def normalize_html(content: str) -> str: - """Normalize generated HTML without hiding ordinary page content changes.""" - - normalized = content.replace("\r\n", "\n").replace("\r", "\n") - - def remove_meta(match: re.Match[str]) -> str: - return "" if _html_metadata_name(match.group(0)) else match.group(0) - - normalized = _META_TAG.sub(remove_meta, normalized) - - def remove_comment(match: re.Match[str]) -> str: - return "" if _VOLATILE_COMMENT_WORDS.search(match.group(0)) else match.group(0) - - normalized = _HTML_COMMENT.sub(remove_comment, normalized) - normalized = _VOLATILE_ATTRIBUTE.sub("", normalized) - normalized = _GITHUB_BLOB_COMMIT.sub("", normalized) - normalized = _NEED_CONTAINER_ID.sub("SNCB-", normalized) - normalized = "\n".join(line.rstrip() for line in normalized.splitlines()) - return normalized.strip() - - -def _html_files(directory: Path) -> dict[str, str]: - """Return normalized HTML keyed by POSIX-relative path.""" - - if not directory.is_dir(): - raise DocsDeltaError(f"documentation directory is missing: {directory}") - result: dict[str, str] = {} - for path in sorted(directory.rglob("*.html")): - if path.is_file(): - relative = path.relative_to(directory).as_posix() - try: - result[relative] = normalize_html(path.read_text(encoding="utf-8")) - except (OSError, UnicodeDecodeError) as exc: - raise DocsDeltaError( - f"cannot read rendered page {path}: {exc}" - ) from exc - return result - - -def compare_html(baseline_dir: Path, current_dir: Path) -> PageComparison: - """Compare normalized recursive HTML output from two build directories.""" - - baseline = _html_files(baseline_dir) - current = _html_files(current_dir) - added: list[PageChange] = [] - removed: list[PageChange] = [] - modified: list[PageChange] = [] - unchanged_count = 0 - - for path in sorted(set(baseline) | set(current)): - old = baseline.get(path) - new = current.get(path) - change = PageChange(path) - if old is None: - added.append(change) - elif new is None: - removed.append(change) - elif old != new: - modified.append(change) - else: - unchanged_count += 1 - - return PageComparison( - added=tuple(added), - removed=tuple(removed), - modified=tuple(modified), - unchanged_count=unchanged_count, - ) - - -def _url_with_path(base_url: str, relative_path: str, anchor: str | None = None) -> str: - url = ( - base_url.rstrip("/") - + "/" - + "/".join( - quote(part, safe="-._~") for part in relative_path.strip("/").split("/") - ) - ) - if anchor: - url += "#" + quote(anchor, safe="-._~") - return url - - -def _need_url(base_url: str, need: Mapping[str, object]) -> str | None: - docname = need.get("docname") - if not isinstance(docname, str) or not docname.strip(): - return None - rendered_name = docname.strip("/") - if not rendered_name.endswith(".html"): - rendered_name += ".html" - return _url_with_path(base_url, rendered_name) - - -def _need_link(base_url: str, need_id: str, need: Mapping[str, object]) -> str | None: - page_url = _need_url(base_url, need) - if page_url is None: - return None - return page_url + "#" + quote(need_id, safe="-._~") - - -def _markdown_link(label: str, url: str | None) -> str: - if url is None: - return f"`{label}`" - safe_label = label.replace("[", "\\[").replace("]", "\\]") - return f"[{safe_label}]({url})" - - -def _display_name(need: Mapping[str, object]) -> str: - for field in ("title", "name"): - value = need.get(field) - if isinstance(value, str) and value: - return value - return "" - - -def _format_value(value: object) -> str: - if value is _MISSING: - return "(missing)" - rendered = json.dumps( - value, sort_keys=True, ensure_ascii=False, separators=(",", ":") - ) - if len(rendered) > MAX_RENDERED_VALUE_LENGTH: - rendered = rendered[:MAX_RENDERED_VALUE_LENGTH] + "…" - return rendered.replace("`", "\\`").replace("\n", " ") - - -def _format_need_entry( - change: NeedChange, *, base_url: str, pr_url: str, include_diff: bool = False -) -> list[str]: - need = change.current or change.baseline - assert need is not None - links = [] - old_link = ( - _need_link(base_url, change.need_id, change.baseline) - if change.baseline - else None - ) - new_link = ( - _need_link(pr_url, change.need_id, change.current) if change.current else None - ) - if old_link: - links.append(_markdown_link("old", old_link)) - if new_link: - links.append(_markdown_link("new", new_link)) - link_text = " / ".join(links) - title = _display_name(need) - suffix = f" — {title}" if title else "" - line = f"- `{change.need_id}`{suffix}" - if link_text: - line += f" ({link_text})" - result = [line] - if include_diff: - assert change.baseline is not None and change.current is not None - for field in change.changed_fields: - result.append( - f" - `{field}`: {_format_value(change.baseline.get(field, _MISSING))} " - f"→ {_format_value(change.current.get(field, _MISSING))}" - ) - return result - - -def _format_page_entry( - change: PageChange, *, base_url: str, pr_url: str, kind: str -) -> str: - old = _url_with_path(base_url, change.path) if kind != "added" else None - new = _url_with_path(pr_url, change.path) if kind != "removed" else None - links = [] - if old: - links.append(_markdown_link("old", old)) - if new: - links.append(_markdown_link("new", new)) - return f"- `{change.path}` ({' / '.join(links)})" - - -def _need_section( - lines: list[str], - title: str, - entries: Sequence[NeedChange], - formatter: NeedFormatter, - *, - base_url: str, - pr_url: str, - kind: str = "", -) -> None: - if not entries: - return - lines.extend([f"### {title} ({len(entries)})", ""]) - if len(entries) > DETAIL_LIMIT: - lines.extend([f"{len(entries)} entries changed; details omitted.", ""]) - return - for entry in entries: - lines.extend( - formatter( - entry, - base_url=base_url, - pr_url=pr_url, - include_diff=kind == "modified", - ) - ) - lines.append("") - - -def _page_section( - lines: list[str], - title: str, - entries: Sequence[PageChange], - formatter: PageFormatter, - *, - base_url: str, - pr_url: str, - kind: str, -) -> None: - if not entries: - return - lines.extend([f"### {title} ({len(entries)})", ""]) - for entry in entries: - lines.append(formatter(entry, base_url=base_url, pr_url=pr_url, kind=kind)) - lines.append("") - - -def render_report( - needs: NeedComparison, - pages: PageComparison, - *, - base_url: str, - pr_url: str, -) -> str: - """Render a deterministic Markdown report for comparison results.""" - - lines = [ - "# Documentation delta", - "", - "Documentation preview for this pull request: " - f"{_markdown_link('open preview', pr_url)} · " - f"Baseline: {_markdown_link('open baseline', base_url)}", - "", - "## Summary", - "", - f"- Needs: {len(needs.added)} added, {len(needs.removed)} removed, " - f"{len(needs.modified)} modified, {needs.unchanged_count} unchanged", - f"- Rendered pages: {len(pages.added)} added, {len(pages.removed)} removed, " - f"{len(pages.modified)} modified, {pages.unchanged_count} unchanged", - ] - - need_change_count = len(needs.added) + len(needs.removed) + len(needs.modified) - if need_change_count: - need_lines: list[str] = [] - _need_section( - need_lines, - "Added", - needs.added, - _format_need_entry, - base_url=base_url, - pr_url=pr_url, - kind="added", - ) - _need_section( - need_lines, - "Removed", - needs.removed, - _format_need_entry, - base_url=base_url, - pr_url=pr_url, - kind="removed", - ) - _need_section( - need_lines, - "Modified", - needs.modified, - _format_need_entry, - base_url=base_url, - pr_url=pr_url, - kind="modified", - ) - if need_change_count > DETAIL_LIMIT: - lines.extend( - [ - "", - "
", - f"Need changes ({need_change_count})", - "", - *need_lines, - "
", - ] - ) - else: - lines.extend(["", "## Need changes", "", *need_lines]) - - page_change_count = len(pages.added) + len(pages.removed) + len(pages.modified) - if page_change_count: - page_lines: list[str] = [] - _page_section( - page_lines, - "Added", - pages.added, - _format_page_entry, - base_url=base_url, - pr_url=pr_url, - kind="added", - ) - _page_section( - page_lines, - "Removed", - pages.removed, - _format_page_entry, - base_url=base_url, - pr_url=pr_url, - kind="removed", - ) - _page_section( - page_lines, - "Modified", - pages.modified, - _format_page_entry, - base_url=base_url, - pr_url=pr_url, - kind="modified", - ) - if page_change_count > DETAIL_LIMIT: - lines.extend( - [ - "", - "
", - f"Page changes ({page_change_count})", - "", - *page_lines, - "
", - ] - ) - else: - lines.extend(["", "## Page changes", "", *page_lines]) - return "\n".join(lines).rstrip() + "\n" - - -def unavailable_report(reason: str) -> str: - """Render the successful, explicit report used when baseline is unavailable.""" - - return ( - "# Documentation delta\n\n" - "## Delta unavailable\n\n" - f"The documentation baseline is unavailable: {reason}\n" - ) - - -def _write_report(path: Path, content: str) -> None: - """Atomically write the report so a failed run never leaves partial Markdown.""" - - path.parent.mkdir(parents=True, exist_ok=True) - temporary_path: Path | None = None - file_descriptor: int | None = None - try: - file_descriptor, temporary_name = tempfile.mkstemp( - dir=path.parent, prefix=f".{path.name}.", suffix=".tmp" - ) - temporary_path = Path(temporary_name) - with os.fdopen(file_descriptor, "w", encoding="utf-8") as output: - file_descriptor = None - output.write(content) - output.flush() - os.replace(temporary_path, path) - temporary_path = None - finally: - if file_descriptor is not None: - os.close(file_descriptor) - if temporary_path is not None: - temporary_path.unlink(missing_ok=True) - - -def _workspace_path(environ: Mapping[str, str], relative_path: str) -> Path: - workspace = environ.get("GITHUB_WORKSPACE") - if workspace: - return Path(workspace) / relative_path - return Path(relative_path) - - -def _github_pull_request(environ: Mapping[str, str]) -> GithubPullRequest | None: - """Read pull request provenance from standard GitHub Actions variables.""" - - event_name = environ.get("GITHUB_EVENT_NAME") - event_path = environ.get("GITHUB_EVENT_PATH") - if event_name != "pull_request": - return None - if not event_path: - raise DocsDeltaError( - "GITHUB_EVENT_PATH is required to resolve a pull request baseline" - ) - - try: - payload = json.loads(Path(event_path).read_text(encoding="utf-8")) - except (OSError, UnicodeDecodeError, json.JSONDecodeError) as exc: - raise DocsDeltaError( - f"cannot read GitHub event payload {event_path}: {exc}" - ) from exc - if not isinstance(payload, dict): - raise DocsDeltaError(f"GitHub event payload is not an object: {event_path}") - - event_payload = cast(dict[str, object], payload) - pull_request_value = event_payload.get("pull_request") - if not isinstance(pull_request_value, dict): - return None - pull_request = cast(dict[str, object], pull_request_value) - base_value = pull_request.get("base") - if not isinstance(base_value, dict): - return None - base = cast(dict[str, object], base_value) - - base_ref = environ.get("GITHUB_BASE_REF") or base.get("ref") - base_sha = environ.get("GITHUB_BASE_SHA") or base.get("sha") - if not isinstance(base_ref, str) or not base_ref: - return None - if not isinstance(base_sha, str) or not base_sha: - return None - - number: object = pull_request.get("number") - if not isinstance(number, int | str) or isinstance(number, bool): - number = None - else: - number = str(number) - return GithubPullRequest(base_ref=base_ref, base_sha=base_sha, number=number) - - -def _find_published_baseline_commit( - gh_pages_dir: Path, pull_request: GithubPullRequest -) -> tuple[str | None, str]: - """Find the newest gh-pages commit whose generated Needs contain the base SHA.""" - - if not gh_pages_dir.is_dir(): - return None, f"gh-pages checkout is missing: {gh_pages_dir}" - - try: - git_check = subprocess.run( - ["git", "-C", str(gh_pages_dir), "rev-parse", "--git-dir"], - check=False, - capture_output=True, - text=True, - ) - except OSError as exc: - return None, f"cannot inspect gh-pages checkout {gh_pages_dir}: {exc}" - if git_check.returncode != 0: - return None, f"gh-pages checkout is not a Git repository: {gh_pages_dir}" - - needs_path = f"{pull_request.base_ref}/needs.json" - try: - result = subprocess.run( - [ - "git", - "-C", - str(gh_pages_dir), - "log", - "--all", - "--full-history", - "--format=%H", - "--", - needs_path, - ], - check=False, - capture_output=True, - text=True, - ) - except OSError as exc: - return None, f"cannot search gh-pages history: {exc}" - if result.returncode != 0: - detail = result.stderr.strip() or "git log failed" - return None, f"cannot search gh-pages history: {detail}" - - candidate_commits = [ - line.strip() for line in result.stdout.splitlines() if line.strip() - ] - for commit in candidate_commits: - try: - needs = subprocess.run( - [ - "git", - "-C", - str(gh_pages_dir), - "show", - f"{commit}:{needs_path}", - ], - check=False, - capture_output=True, - ) - except OSError as exc: - return None, f"cannot inspect published needs JSON: {exc}" - if needs.returncode == 0 and pull_request.base_sha.encode() in needs.stdout: - return commit, "" - - return ( - None, - f"no published {pull_request.base_ref} baseline contains " - f"source commit {pull_request.base_sha}", - ) - - -def _extract_git_tree( - gh_pages_dir: Path, commit: str, tree_path: str, destination: Path -) -> None: - """Extract one published documentation tree without changing the checkout.""" - - try: - result = subprocess.run( - [ - "git", - "-C", - str(gh_pages_dir), - "archive", - "--format=tar", - f"{commit}:{tree_path}", - ], - check=False, - capture_output=True, - ) - except OSError as exc: - raise DocsDeltaError(f"cannot extract gh-pages baseline: {exc}") from exc - if result.returncode != 0: - detail = result.stderr.decode(errors="replace").strip() or "git archive failed" - raise DocsDeltaError(f"cannot extract gh-pages baseline: {detail}") - - destination.mkdir(parents=True, exist_ok=True) - try: - with tarfile.open(fileobj=io.BytesIO(result.stdout), mode="r:") as archive: - for member in archive.getmembers(): - relative = Path(member.name) - if relative.is_absolute() or ".." in relative.parts: - raise DocsDeltaError( - f"gh-pages archive contains unsafe path: {member.name}" - ) - target = destination / relative - if member.isdir(): - target.mkdir(parents=True, exist_ok=True) - elif member.isfile(): - target.parent.mkdir(parents=True, exist_ok=True) - source = archive.extractfile(member) - if source is None: - raise DocsDeltaError( - f"cannot read gh-pages archive member: {member.name}" - ) - with source, target.open("wb") as output: - shutil.copyfileobj(source, output) - else: - raise DocsDeltaError( - f"unsupported gh-pages archive member: {member.name}" - ) - except (OSError, tarfile.TarError) as exc: - raise DocsDeltaError(f"cannot extract gh-pages baseline: {exc}") from exc - - -@contextmanager -def _resolved_baseline( - args: argparse.Namespace, - github_pull_request: GithubPullRequest | None, - environ: Mapping[str, str], -) -> Iterator[tuple[Path | None, str]]: - """Resolve either a directory baseline or a published gh-pages baseline.""" - - if args.baseline_mode is None: - mode = ( - "directory" - if args.baseline_dir is not None or github_pull_request is None - else "gh-pages" - ) - else: - mode = args.baseline_mode - - if mode == "directory": - if args.baseline_dir is None: - raise DocsDeltaError( - "--baseline-dir is required outside GitHub pull request mode" - ) - yield args.baseline_dir, "" - return - - if args.baseline_dir is not None: - raise DocsDeltaError( - "--baseline-dir cannot be used with --baseline-mode gh-pages" - ) - if github_pull_request is None: - raise DocsDeltaError( - "gh-pages baseline mode requires a GitHub pull request event" - ) - - gh_pages_dir = args.gh_pages_dir or _workspace_path(environ, ".docs-baseline") - commit, reason = _find_published_baseline_commit(gh_pages_dir, github_pull_request) - if commit is None: - yield None, reason - return - - with tempfile.TemporaryDirectory(prefix="docs-delta-baseline-") as temporary_dir: - baseline_dir = Path(temporary_dir) - try: - _extract_git_tree( - gh_pages_dir, - commit, - github_pull_request.base_ref, - baseline_dir, - ) - except DocsDeltaError as exc: - yield None, str(exc) - return - yield baseline_dir, "" - - -def _github_url_defaults( - environ: Mapping[str, str], - github_pull_request: GithubPullRequest | None = None, -) -> tuple[str | None, str | None]: - """Derive conventional GitHub Pages URLs from standard Actions variables.""" - - repository = environ.get("GITHUB_REPOSITORY", "") - if "/" not in repository: - return None, None - owner, name = repository.split("/", 1) - if not owner or not name: - return None, None - pages_root = environ.get("GITHUB_PAGES_URL") or f"https://{owner}.github.io/{name}" - base_ref = ( - github_pull_request.base_ref - if github_pull_request - else (environ.get("GITHUB_BASE_REF") or "main") - ) - if github_pull_request and github_pull_request.number: - pr_ref = f"pr-{github_pull_request.number}" - else: - pr_ref = ( - environ.get("GITHUB_HEAD_REF") or environ.get("GITHUB_REF_NAME") or base_ref - ) - - def pages_ref_url(ref: str) -> str: - # A branch name is one URL path segment. In particular, a slash in a - # feature branch must not become a second documentation path segment. - return pages_root.rstrip("/") + "/" + quote(ref, safe="-._~") - - return ( - pages_ref_url(base_ref), - pages_ref_url(pr_ref), - ) - - -def _resolve_urls( - args: argparse.Namespace, github_pull_request: GithubPullRequest | None -) -> tuple[str, str]: - defaults = _github_url_defaults(os.environ, github_pull_request) - base_url = args.base_url or defaults[0] - pr_url = args.pr_url or defaults[1] - if not base_url or not pr_url: - raise DocsDeltaError( - "documentation URLs are required; provide --base-url and --pr-url " - "or run in GitHub Actions with GITHUB_REPOSITORY set" - ) - return base_url, pr_url - - -def argument_parser() -> argparse.ArgumentParser: - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument( - "--baseline-dir", - type=Path, - help="directory baseline; defaults to automatic gh-pages mode in PR Actions", - ) - parser.add_argument( - "--baseline-mode", - choices=("directory", "gh-pages"), - help="baseline source; inferred when omitted", - ) - parser.add_argument( - "--gh-pages-dir", - type=Path, - help="local full-history gh-pages checkout (defaults to .docs-baseline)", - ) - parser.add_argument( - "--current-dir", - type=Path, - help="current documentation directory (defaults to docs-artifact)", - ) - parser.add_argument("--base-url") - parser.add_argument("--pr-url") - parser.add_argument( - "--output", - type=Path, - help="report path (defaults to docs-artifact/docs-delta.md)", - ) - return parser - - -def main(argv: Sequence[str] | None = None) -> int: - """Run the CLI and return a non-zero status only for current-input errors.""" - - args = argument_parser().parse_args(argv) - try: - github_pull_request = _github_pull_request(os.environ) - current_dir = args.current_dir or _workspace_path(os.environ, "docs-artifact") - output = args.output or current_dir / "docs-delta.md" - - with _resolved_baseline(args, github_pull_request, os.environ) as ( - baseline_dir, - baseline_reason, - ): - if baseline_dir is None or not baseline_dir.is_dir(): - reason = baseline_reason or f"directory is missing: {baseline_dir}" - _write_report(output, unavailable_report(reason)) - return 0 - try: - baseline_needs = load_needs(baseline_dir) - except DocsDeltaError as exc: - _write_report(output, unavailable_report(str(exc))) - return 0 - - if not current_dir.is_dir(): - raise DocsDeltaError( - f"documentation directory is missing: {current_dir}" - ) - current_needs = load_needs(current_dir) - base_url, pr_url = _resolve_urls(args, github_pull_request) - report = render_report( - compare_needs(baseline_needs, current_needs), - compare_html(baseline_dir, current_dir), - base_url=base_url, - pr_url=pr_url, - ) - _write_report(output, report) - except (DocsDeltaError, OSError) as exc: - print(f"docs_delta: error: {exc}", file=sys.stderr) - return 2 - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/tools/docs_delta/README.md b/tools/docs_delta/README.md new file mode 100644 index 000000000..8520648e1 --- /dev/null +++ b/tools/docs_delta/README.md @@ -0,0 +1,39 @@ + + +# Documentation Delta + +Compare the Sphinx-Needs inventory and rendered HTML from two documentation +builds, then write a Markdown report. The command uses only the Python standard +library. + +Run the CLI from the repository root with: + +```sh +uv run --locked --project tools/docs_delta docs-delta --help +``` + +For a local comparison, provide the baseline and current build directories and +the URLs that reviewers should open: + +```sh +uv run --locked --project tools/docs_delta docs-delta \ + --baseline-dir /path/to/baseline \ + --current-dir /path/to/current \ + --base-url https://example.org/docs/main \ + --pr-url https://example.org/docs/pr-123 +``` + +In GitHub pull-request Actions, the CLI can resolve the baseline from a local, +full-history `gh-pages` checkout and derive the preview URLs from Actions +metadata. `docs-delta --help` lists the options for overriding those defaults. diff --git a/tools/docs_delta/__init__.py b/tools/docs_delta/__init__.py new file mode 100644 index 000000000..aec8fd0d2 --- /dev/null +++ b/tools/docs_delta/__init__.py @@ -0,0 +1,50 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +"""Compare documentation builds and format a Markdown delta.""" + +from .cli import main +from .comparison import ( + DocsDeltaError, + NeedChange, + NeedComparison, + NeedMap, + PageChange, + PageComparison, + compare_needs, + load_needs, +) +from .rendered_html import compare_html, normalize_html +from .report import ( + MAX_RENDERED_VALUE_LENGTH, + need_link, + render_report, + unavailable_report, +) + +__all__ = [ + "DocsDeltaError", + "MAX_RENDERED_VALUE_LENGTH", + "NeedChange", + "NeedComparison", + "NeedMap", + "PageChange", + "PageComparison", + "compare_html", + "compare_needs", + "load_needs", + "main", + "need_link", + "normalize_html", + "render_report", + "unavailable_report", +] diff --git a/tools/docs_delta/__main__.py b/tools/docs_delta/__main__.py new file mode 100644 index 000000000..d24966006 --- /dev/null +++ b/tools/docs_delta/__main__.py @@ -0,0 +1,17 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +"""Run the docs-delta command as ``python -m docs_delta``.""" + +from .cli import main + +raise SystemExit(main()) diff --git a/tools/docs_delta/cli.py b/tools/docs_delta/cli.py new file mode 100644 index 000000000..8f41643ab --- /dev/null +++ b/tools/docs_delta/cli.py @@ -0,0 +1,109 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +"""Command-line interface for creating documentation delta reports.""" + +from __future__ import annotations + +import argparse +import os +import sys +from collections.abc import Sequence +from pathlib import Path + +from .comparison import DocsDeltaError, compare_needs, load_needs +from .github import github_pull_request, resolve_urls, resolved_baseline, workspace_path +from .rendered_html import compare_html +from .report import render_report, unavailable_report, write_report + + +def argument_parser() -> argparse.ArgumentParser: + """Describe the local and GitHub Actions forms of the command line.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--baseline-dir", + type=Path, + help="directory baseline; defaults to automatic gh-pages mode in PR Actions", + ) + parser.add_argument( + "--baseline-mode", + choices=("directory", "gh-pages"), + help="baseline source; inferred when omitted", + ) + parser.add_argument( + "--gh-pages-dir", + type=Path, + help="local full-history gh-pages checkout (defaults to .docs-baseline)", + ) + parser.add_argument( + "--current-dir", + type=Path, + help="current documentation directory (defaults to docs-artifact)", + ) + parser.add_argument("--base-url") + parser.add_argument("--pr-url") + parser.add_argument( + "--output", + type=Path, + help="report path (defaults to docs-artifact/docs-delta.md)", + ) + return parser + + +def main(argv: Sequence[str] | None = None) -> int: + """Run the CLI and write either a delta or an explicit unavailable report. + + A baseline may legitimately be absent while publishing is in progress; in + that case the report is still successful and explains why it has no delta. + Missing or malformed current build data is an error because it would make + even the PR side of the comparison unreliable. + """ + + args = argument_parser().parse_args(argv) + try: + pull_request = github_pull_request(os.environ) + current_dir = args.current_dir or workspace_path(os.environ, "docs-artifact") + output = args.output or current_dir / "docs-delta.md" + + with resolved_baseline(args, pull_request, os.environ) as ( + baseline_dir, + baseline_reason, + ): + if baseline_dir is None or not baseline_dir.is_dir(): + # A stale baseline would produce a plausible but misleading + # diff. Surface the missing comparison in the comment instead. + reason = baseline_reason or f"directory is missing: {baseline_dir}" + write_report(output, unavailable_report(reason)) + return 0 + try: + baseline_needs = load_needs(baseline_dir) + except DocsDeltaError as exc: + write_report(output, unavailable_report(str(exc))) + return 0 + + if not current_dir.is_dir(): + raise DocsDeltaError( + f"documentation directory is missing: {current_dir}" + ) + current_needs = load_needs(current_dir) + base_url, pr_url = resolve_urls(args, pull_request, os.environ) + report = render_report( + compare_needs(baseline_needs, current_needs), + compare_html(baseline_dir, current_dir), + base_url=base_url, + pr_url=pr_url, + ) + write_report(output, report) + except (DocsDeltaError, OSError) as exc: + print(f"docs_delta: error: {exc}", file=sys.stderr) + return 2 + return 0 diff --git a/tools/docs_delta/comparison.py b/tools/docs_delta/comparison.py new file mode 100644 index 000000000..ba23a42d3 --- /dev/null +++ b/tools/docs_delta/comparison.py @@ -0,0 +1,205 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +"""Need inventory types, loading, and comparison rules.""" + +from __future__ import annotations + +import json +from collections.abc import Mapping +from dataclasses import dataclass +from pathlib import Path +from typing import cast + +JsonObject = dict[str, object] +NeedMap = Mapping[str, JsonObject] +MISSING = object() +# Source locations, generated IDs, and test-result links vary with the build +# revision without changing the requirement represented by a Need. +VOLATILE_NEED_FIELDS = frozenset( + { + "lineno", + "lineno_content", + "source", + "target_id", + "is_modified", + "source_code_link", + "testlink", + } +) + + +class DocsDeltaError(ValueError): + """A user-actionable input or report-generation error.""" + + +@dataclass(frozen=True) +class NeedChange: + """A Need identified by ID, with whichever before/after versions exist. + + An absent baseline means the Need was added; an absent current value means + it was removed. When both versions exist, ``changed_fields`` reports only + differences that are meaningful under the comparison's volatile-field + policy. + """ + + need_id: str + baseline: JsonObject | None + current: JsonObject | None + + @property + def changed_fields(self) -> tuple[str, ...]: + """Return changed non-volatile fields in deterministic order.""" + + if self.baseline is None or self.current is None: + return () + names = set(self.baseline) | set(self.current) + return tuple( + sorted( + name + for name in names + if name not in VOLATILE_NEED_FIELDS + and not _json_values_equal( + self.baseline.get(name, MISSING), + self.current.get(name, MISSING), + ) + ) + ) + + +@dataclass(frozen=True) +class NeedComparison: + """Added, removed, modified, and unchanged counts for Need inventories. + + Changed entries are stored in sorted-ID order by :func:`compare_needs`, + which makes both the Markdown output and its review order reproducible. + """ + + added: tuple[NeedChange, ...] + removed: tuple[NeedChange, ...] + modified: tuple[NeedChange, ...] + unchanged_count: int + + +@dataclass(frozen=True) +class PageChange: + """One rendered HTML path that was added, removed, or modified.""" + + path: str + + +@dataclass(frozen=True) +class PageComparison: + """Added, removed, modified, and unchanged counts for rendered pages. + + A page's relative path determines its identity; a rename therefore appears + as one removal and one addition rather than a guessed rename relationship. + """ + + added: tuple[PageChange, ...] + removed: tuple[PageChange, ...] + modified: tuple[PageChange, ...] + unchanged_count: int + + +def _json_values_equal(left: object, right: object) -> bool: + """Compare JSON-shaped values consistently and distinguish missing keys. + + Python considers ``True == 1``. Serializing the values keeps those Need + field changes visible, while the private sentinel separates a missing + field from an explicit JSON ``null``. + """ + + if left is MISSING or right is MISSING: + return left is right + return json.dumps(left, sort_keys=True, ensure_ascii=False) == json.dumps( + right, sort_keys=True, ensure_ascii=False + ) + + +def _flatten_needs(path: Path) -> dict[str, JsonObject]: + """Flatten versioned Sphinx-Needs output into one ID-to-record mapping. + + Sphinx-Needs exports a ``versions`` map, with each version holding its own + ``needs`` map. Version names are not part of a Need's identity for this + report, so entries from all versions are combined by ID; if an ID occurs + more than once, the last entry in JSON iteration order supplies its record. + Unexpectedly shaped version containers are skipped; malformed individual + Need entries fail with a useful error rather than silently disappearing. + """ + + try: + payload = json.loads(path.read_text(encoding="utf-8")) + except (OSError, UnicodeDecodeError, json.JSONDecodeError) as exc: + raise DocsDeltaError(f"cannot read needs JSON {path}: {exc}") from exc + + if not isinstance(payload, dict) or not isinstance(payload.get("versions"), dict): + raise DocsDeltaError(f"needs JSON has no versions map: {path}") + + versions = cast(dict[str, object], payload["versions"]) + result: dict[str, JsonObject] = {} + for version_value in versions.values(): + if not isinstance(version_value, dict): + continue + version = cast(dict[str, object], version_value) + needs_value = version.get("needs") + if not isinstance(needs_value, dict): + continue + needs = cast(dict[object, object], needs_value) + for need_id, need in needs.items(): + if not isinstance(need_id, str) or not isinstance(need, dict): + raise DocsDeltaError(f"invalid Need entry in {path}") + result[need_id] = cast(JsonObject, need) + return result + + +def load_needs(directory: Path) -> dict[str, JsonObject]: + """Load the root ``needs.json`` inventory from a rendered docs directory.""" + + needs_path = directory / "needs.json" + if not needs_path.is_file(): + raise DocsDeltaError(f"needs JSON is missing: {needs_path}") + return _flatten_needs(needs_path) + + +def compare_needs(baseline: NeedMap, current: NeedMap) -> NeedComparison: + """Compare Needs by ID, ignoring only explicitly volatile fields. + + A field added or removed from a record counts as a modification. All + result groups are sorted by Need ID so report order does not depend on JSON + object insertion order. + """ + + added: list[NeedChange] = [] + removed: list[NeedChange] = [] + modified: list[NeedChange] = [] + unchanged_count = 0 + + for need_id in sorted(set(baseline) | set(current)): + old = baseline.get(need_id) + new = current.get(need_id) + change = NeedChange(need_id, old, new) + if old is None: + added.append(change) + elif new is None: + removed.append(change) + elif change.changed_fields: + modified.append(change) + else: + unchanged_count += 1 + + return NeedComparison( + added=tuple(added), + removed=tuple(removed), + modified=tuple(modified), + unchanged_count=unchanged_count, + ) diff --git a/tools/docs_delta/github.py b/tools/docs_delta/github.py new file mode 100644 index 000000000..66a3f8ddd --- /dev/null +++ b/tools/docs_delta/github.py @@ -0,0 +1,369 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +"""GitHub Actions metadata and published gh-pages baseline resolution.""" + +from __future__ import annotations + +import argparse +import io +import json +import shutil +import subprocess +import tarfile +import tempfile +from collections.abc import Iterator, Mapping +from contextlib import contextmanager +from dataclasses import dataclass +from pathlib import Path +from typing import cast +from urllib.parse import quote + +from .comparison import DocsDeltaError + + +@dataclass(frozen=True) +class GithubPullRequest: + """The source revision that a PR documentation baseline must represent. + + ``base_ref`` selects the published documentation directory and ``base_sha`` + identifies the commit that directory must contain. These are separate + because gh-pages may publish the branch after the PR event was created. + """ + + base_ref: str + base_sha: str + number: str | None + + +def workspace_path(environ: Mapping[str, str], relative_path: str) -> Path: + """Resolve a workflow-relative path, with a cwd-based local fallback.""" + workspace = environ.get("GITHUB_WORKSPACE") + if workspace: + return Path(workspace) / relative_path + return Path(relative_path) + + +def github_pull_request(environ: Mapping[str, str]) -> GithubPullRequest | None: + """Read PR base ref/SHA from Actions metadata, if this is a PR event. + + The event payload is authoritative for PR identity. The dedicated base + variables take precedence when present, since they describe the checkout + context in which the workflow is running. + """ + + event_name = environ.get("GITHUB_EVENT_NAME") + event_path = environ.get("GITHUB_EVENT_PATH") + if event_name != "pull_request": + return None + if not event_path: + raise DocsDeltaError( + "GITHUB_EVENT_PATH is required to resolve a pull request baseline" + ) + + try: + payload = json.loads(Path(event_path).read_text(encoding="utf-8")) + except (OSError, UnicodeDecodeError, json.JSONDecodeError) as exc: + raise DocsDeltaError( + f"cannot read GitHub event payload {event_path}: {exc}" + ) from exc + if not isinstance(payload, dict): + raise DocsDeltaError(f"GitHub event payload is not an object: {event_path}") + + event_payload = cast(dict[str, object], payload) + pull_request_value = event_payload.get("pull_request") + if not isinstance(pull_request_value, dict): + return None + pull_request = cast(dict[str, object], pull_request_value) + base_value = pull_request.get("base") + if not isinstance(base_value, dict): + return None + base = cast(dict[str, object], base_value) + + base_ref = environ.get("GITHUB_BASE_REF") or base.get("ref") + base_sha = environ.get("GITHUB_BASE_SHA") or base.get("sha") + if not isinstance(base_ref, str) or not base_ref: + return None + if not isinstance(base_sha, str) or not base_sha: + return None + + number: object = pull_request.get("number") + if not isinstance(number, int | str) or isinstance(number, bool): + number = None + else: + number = str(number) + return GithubPullRequest(base_ref=base_ref, base_sha=base_sha, number=number) + + +def _find_published_baseline_commit( + gh_pages_dir: Path, pull_request: GithubPullRequest +) -> tuple[str | None, str]: + """Find the newest published base-branch tree containing the PR base SHA. + + The docs publishing workflow updates a branch directory on ``gh-pages`` + only after a successful docs build. Looking up ``needs.json`` history for + the base branch and checking its embedded source SHA avoids selecting a + previous successful build when the current base commit has not been + published. ``git log`` returns newest commits first, so the first match is + the latest publication that represents this exact source revision. + """ + + if not gh_pages_dir.is_dir(): + return None, f"gh-pages checkout is missing: {gh_pages_dir}" + + try: + git_check = subprocess.run( + ["git", "-C", str(gh_pages_dir), "rev-parse", "--git-dir"], + check=False, + capture_output=True, + text=True, + ) + except OSError as exc: + return None, f"cannot inspect gh-pages checkout {gh_pages_dir}: {exc}" + if git_check.returncode != 0: + return None, f"gh-pages checkout is not a Git repository: {gh_pages_dir}" + + # The Needs export carries the source revision that produced a published + # docs tree. Searching this file's history also avoids treating an older + # successful publication as the baseline for a newer, unpublished commit. + needs_path = f"{pull_request.base_ref}/needs.json" + try: + result = subprocess.run( + [ + "git", + "-C", + str(gh_pages_dir), + "log", + "--all", + "--full-history", + "--format=%H", + "--", + needs_path, + ], + check=False, + capture_output=True, + text=True, + ) + except OSError as exc: + return None, f"cannot search gh-pages history: {exc}" + if result.returncode != 0: + detail = result.stderr.strip() or "git log failed" + return None, f"cannot search gh-pages history: {detail}" + + candidate_commits = [ + line.strip() for line in result.stdout.splitlines() if line.strip() + ] + # ``git log`` is newest-first; stop at the most recent publication that + # proves it was built from the PR's exact base SHA. + for commit in candidate_commits: + try: + needs = subprocess.run( + [ + "git", + "-C", + str(gh_pages_dir), + "show", + f"{commit}:{needs_path}", + ], + check=False, + capture_output=True, + ) + except OSError as exc: + return None, f"cannot inspect published needs JSON: {exc}" + if needs.returncode == 0 and pull_request.base_sha.encode() in needs.stdout: + return commit, "" + + return ( + None, + f"no published {pull_request.base_ref} baseline contains " + f"source commit {pull_request.base_sha}", + ) + + +def extract_git_tree( + gh_pages_dir: Path, commit: str, tree_path: str, destination: Path +) -> None: + """Copy a docs subtree from a commit into a temporary directory. + + ``git archive`` reads the selected immutable tree without checking out or + disturbing the shared gh-pages worktree. Members are copied explicitly + instead of using ``extractall`` so unexpected paths or special files in + the archive cannot escape the temporary baseline directory. + """ + + try: + result = subprocess.run( + [ + "git", + "-C", + str(gh_pages_dir), + "archive", + "--format=tar", + f"{commit}:{tree_path}", + ], + check=False, + capture_output=True, + ) + except OSError as exc: + raise DocsDeltaError(f"cannot extract gh-pages baseline: {exc}") from exc + if result.returncode != 0: + detail = result.stderr.decode(errors="replace").strip() or "git archive failed" + raise DocsDeltaError(f"cannot extract gh-pages baseline: {detail}") + + destination.mkdir(parents=True, exist_ok=True) + try: + with tarfile.open(fileobj=io.BytesIO(result.stdout), mode="r:") as archive: + for member in archive.getmembers(): + relative = Path(member.name) + if relative.is_absolute() or ".." in relative.parts: + raise DocsDeltaError( + f"gh-pages archive contains unsafe path: {member.name}" + ) + target = destination / relative + if member.isdir(): + target.mkdir(parents=True, exist_ok=True) + elif member.isfile(): + target.parent.mkdir(parents=True, exist_ok=True) + source = archive.extractfile(member) + if source is None: + raise DocsDeltaError( + f"cannot read gh-pages archive member: {member.name}" + ) + with source, target.open("wb") as output: + shutil.copyfileobj(source, output) + else: + raise DocsDeltaError( + f"unsupported gh-pages archive member: {member.name}" + ) + except (OSError, tarfile.TarError) as exc: + raise DocsDeltaError(f"cannot extract gh-pages baseline: {exc}") from exc + + +@contextmanager +def resolved_baseline( + args: argparse.Namespace, + github_pull_request: GithubPullRequest | None, + environ: Mapping[str, str], +) -> Iterator[tuple[Path | None, str]]: + """Yield a baseline directory, or ``None`` plus a reportable reason. + + Directory mode is useful for local comparisons and explicit artifacts. + PR Actions defaults to gh-pages mode, where the extracted directory only + lives for the duration of this context manager. A baseline lookup failure + is data for the report; malformed CLI choices remain hard errors. + """ + + if args.baseline_mode is None: + # Local callers usually provide a directory. PR Actions has no such + # artifact, so use the published branch history by default. + mode = ( + "directory" + if args.baseline_dir is not None or github_pull_request is None + else "gh-pages" + ) + else: + mode = args.baseline_mode + + if mode == "directory": + if args.baseline_dir is None: + raise DocsDeltaError( + "--baseline-dir is required outside GitHub pull request mode" + ) + yield args.baseline_dir, "" + return + + if args.baseline_dir is not None: + raise DocsDeltaError( + "--baseline-dir cannot be used with --baseline-mode gh-pages" + ) + if github_pull_request is None: + raise DocsDeltaError( + "gh-pages baseline mode requires a GitHub pull request event" + ) + + gh_pages_dir = args.gh_pages_dir or workspace_path(environ, ".docs-baseline") + commit, reason = _find_published_baseline_commit(gh_pages_dir, github_pull_request) + if commit is None: + yield None, reason + return + + with tempfile.TemporaryDirectory(prefix="docs-delta-baseline-") as temporary_dir: + baseline_dir = Path(temporary_dir) + try: + extract_git_tree( + gh_pages_dir, + commit, + github_pull_request.base_ref, + baseline_dir, + ) + except DocsDeltaError as exc: + yield None, str(exc) + return + yield baseline_dir, "" + + +def _github_url_defaults( + environ: Mapping[str, str], + github_pull_request: GithubPullRequest | None = None, +) -> tuple[str | None, str | None]: + """Derive base and PR preview URLs from the repository and Actions context. + + PR previews conventionally live under ``pr-N``. Branch refs are URL-quoted + as one segment so a feature branch containing ``/`` cannot be mistaken for + a nested path under the Pages site. + """ + + repository = environ.get("GITHUB_REPOSITORY", "") + if "/" not in repository: + return None, None + owner, name = repository.split("/", 1) + if not owner or not name: + return None, None + pages_root = environ.get("GITHUB_PAGES_URL") or f"https://{owner}.github.io/{name}" + base_ref = ( + github_pull_request.base_ref + if github_pull_request + else (environ.get("GITHUB_BASE_REF") or "main") + ) + if github_pull_request and github_pull_request.number: + pr_ref = f"pr-{github_pull_request.number}" + else: + pr_ref = ( + environ.get("GITHUB_HEAD_REF") or environ.get("GITHUB_REF_NAME") or base_ref + ) + + def pages_ref_url(ref: str) -> str: + # A branch name is one URL path segment. In particular, a slash in a + # feature branch must not become a second documentation path segment. + return pages_root.rstrip("/") + "/" + quote(ref, safe="-._~") + + return ( + pages_ref_url(base_ref), + pages_ref_url(pr_ref), + ) + + +def resolve_urls( + args: argparse.Namespace, + github_pull_request: GithubPullRequest | None, + environ: Mapping[str, str], +) -> tuple[str, str]: + """Use explicit URLs where supplied, otherwise require usable CI defaults.""" + defaults = _github_url_defaults(environ, github_pull_request) + base_url = args.base_url or defaults[0] + pr_url = args.pr_url or defaults[1] + if not base_url or not pr_url: + raise DocsDeltaError( + "documentation URLs are required; provide --base-url and --pr-url " + "or run in GitHub Actions with GITHUB_REPOSITORY set" + ) + return base_url, pr_url diff --git a/tools/docs_delta/pyproject.toml b/tools/docs_delta/pyproject.toml new file mode 100644 index 000000000..cfdd71780 --- /dev/null +++ b/tools/docs_delta/pyproject.toml @@ -0,0 +1,26 @@ +[project] +name = "score-docs-delta" +version = "0.1.0" +description = "Compare rendered S-CORE documentation builds" +requires-python = ">=3.12" +dependencies = [] + +[project.scripts] +docs-delta = "docs_delta.cli:main" + +[build-system] +requires = ["hatchling"] +build-backend = "hatchling.build" + +[tool.hatch.build.targets.wheel] +only-include = [ + "__init__.py", + "__main__.py", + "cli.py", + "comparison.py", + "github.py", + "rendered_html.py", + "report.py", +] +sources = { "" = "docs_delta" } +dev-mode-dirs = [".."] diff --git a/tools/docs_delta/rendered_html.py b/tools/docs_delta/rendered_html.py new file mode 100644 index 000000000..7dec98f98 --- /dev/null +++ b/tools/docs_delta/rendered_html.py @@ -0,0 +1,154 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +"""Normalization and comparison of rendered HTML pages.""" + +from __future__ import annotations + +import re +from pathlib import Path + +from .comparison import DocsDeltaError, PageChange, PageComparison + +_META_TAG = re.compile(r"]*>", re.IGNORECASE) +_META_ATTRIBUTE = re.compile( + r"(?:name|property|itemprop)\s*=\s*([\"'])(.*?)\1", re.IGNORECASE +) +_HTML_COMMENT = re.compile(r"", re.IGNORECASE | re.DOTALL) +_VOLATILE_ATTRIBUTE = re.compile( + r"\s+data-(?:build|generated|timestamp|last-modified)(?:-[\w-]+)?\s*=\s*(?:\"[^\"]*\"|'[^']*'|[^\s>]+)", + re.IGNORECASE, +) +# PR documentation builds run on GitHub's synthetic merge commit, while the +# published baseline was built from the base commit. Keep source paths and line +# numbers comparable, but ignore the revision segment in generated blob links. +_GITHUB_BLOB_COMMIT = re.compile(r"(?<=/blob/)[0-9a-f]{40}(?=/)", re.IGNORECASE) +# Sphinx-Needs derives these wrapper IDs from rendered content, including the +# source-link revision above. Need IDs and visible content remain compared. +_NEED_CONTAINER_ID = re.compile(r"\bSNCB-[0-9a-f]{8}\b", re.IGNORECASE) +_VOLATILE_META_NAMES = frozenset( + { + "build-date", + "build_date", + "created", + "date", + "generated", + "generated-at", + "generated_at", + "generator", + "last-modified", + "last_modified", + "timestamp", + } +) +_VOLATILE_COMMENT_WORDS = re.compile( + r"\b(?:build|built|generated|generator|last\s+updated|timestamp)\b", + re.IGNORECASE, +) + + +def _html_metadata_name(tag: str) -> str | None: + for match in _META_ATTRIBUTE.finditer(tag): + name = match.group(2).lower() + if name in _VOLATILE_META_NAMES: + return name + return None + + +def normalize_html(content: str) -> str: + """Remove known build-to-build noise while preserving page changes. + + The rules are intentionally narrow. Build dates, generated attributes, + GitHub source revision hashes, and Sphinx-Needs wrapper IDs vary between + otherwise identical builds. Visible text, links, element structure, and + other attributes remain significant so a real rendered change is reported. + Keep each rule tied to a known source of nondeterminism; broad HTML cleanup + could hide a user-facing regression. + """ + + # Normalize line endings first so the later line cleanup behaves the same + # for artifacts produced on different operating systems. + normalized = content.replace("\r\n", "\n").replace("\r", "\n") + + def remove_meta(match: re.Match[str]) -> str: + return "" if _html_metadata_name(match.group(0)) else match.group(0) + + # Only discard metadata tags whose name is on the explicit volatile list; + # unrelated meta tags can affect consumers and remain part of the diff. + normalized = _META_TAG.sub(remove_meta, normalized) + + def remove_comment(match: re.Match[str]) -> str: + return "" if _VOLATILE_COMMENT_WORDS.search(match.group(0)) else match.group(0) + + # Generated comments are noisy, but ordinary HTML comments can document + # page structure and therefore stay comparison-visible. + normalized = _HTML_COMMENT.sub(remove_comment, normalized) + normalized = _VOLATILE_ATTRIBUTE.sub("", normalized) + normalized = _GITHUB_BLOB_COMMIT.sub("", normalized) + normalized = _NEED_CONTAINER_ID.sub("SNCB-", normalized) + # Trailing whitespace and blank padding do not change rendered pages. + normalized = "\n".join(line.rstrip() for line in normalized.splitlines()) + return normalized.strip() + + +def _html_files(directory: Path) -> dict[str, str]: + """Read and normalize every HTML page below ``directory``. + + POSIX separators make paths stable across operating systems and suitable + for URLs in the report. Sorting traversal also avoids filesystem-order + leaking into the report's comparison order. + """ + + if not directory.is_dir(): + raise DocsDeltaError(f"documentation directory is missing: {directory}") + result: dict[str, str] = {} + for path in sorted(directory.rglob("*.html")): + if path.is_file(): + relative = path.relative_to(directory).as_posix() + try: + result[relative] = normalize_html(path.read_text(encoding="utf-8")) + except (OSError, UnicodeDecodeError) as exc: + raise DocsDeltaError( + f"cannot read rendered page {path}: {exc}" + ) from exc + return result + + +def compare_html(baseline_dir: Path, current_dir: Path) -> PageComparison: + """Compare pages by relative path and normalized rendered content.""" + + baseline = _html_files(baseline_dir) + current = _html_files(current_dir) + added: list[PageChange] = [] + removed: list[PageChange] = [] + modified: list[PageChange] = [] + unchanged_count = 0 + + for path in sorted(set(baseline) | set(current)): + old = baseline.get(path) + new = current.get(path) + change = PageChange(path) + if old is None: + added.append(change) + elif new is None: + removed.append(change) + elif old != new: + modified.append(change) + else: + unchanged_count += 1 + + return PageComparison( + added=tuple(added), + removed=tuple(removed), + modified=tuple(modified), + unchanged_count=unchanged_count, + ) diff --git a/tools/docs_delta/report.py b/tools/docs_delta/report.py new file mode 100644 index 000000000..e86f1e0db --- /dev/null +++ b/tools/docs_delta/report.py @@ -0,0 +1,378 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +"""Markdown formatting for documentation delta reports.""" + +from __future__ import annotations + +import json +import os +import tempfile +from collections.abc import Mapping, Sequence +from pathlib import Path +from typing import Protocol +from urllib.parse import quote + +from .comparison import MISSING, NeedChange, NeedComparison +from .rendered_html import PageChange, PageComparison + +DETAIL_LIMIT = 15 +MAX_RENDERED_VALUE_LENGTH = 600 + + +class NeedFormatter(Protocol): + """Format a Need change for a Markdown report section.""" + + def __call__( + self, + change: NeedChange, + *, + base_url: str, + pr_url: str, + include_diff: bool = False, + ) -> list[str]: ... + + +class PageFormatter(Protocol): + """Format a rendered page change for a Markdown report section.""" + + def __call__( + self, + change: PageChange, + *, + base_url: str, + pr_url: str, + kind: str, + ) -> str: ... + + +def _url_with_path(base_url: str, relative_path: str, anchor: str | None = None) -> str: + """Append an encoded documentation path and optional fragment to a URL. + + Encode path segments independently: slashes in a page path remain + separators, while characters inside each segment cannot alter the URL + structure. + """ + url = ( + base_url.rstrip("/") + + "/" + + "/".join( + quote(part, safe="-._~") for part in relative_path.strip("/").split("/") + ) + ) + if anchor: + url += "#" + quote(anchor, safe="-._~") + return url + + +def _need_url(base_url: str, need: Mapping[str, object]) -> str | None: + """Build the page URL for a Need, returning ``None`` if it has no docname.""" + docname = need.get("docname") + if not isinstance(docname, str) or not docname.strip(): + return None + rendered_name = docname.strip("/") + if not rendered_name.endswith(".html"): + rendered_name += ".html" + return _url_with_path(base_url, rendered_name) + + +def need_link(base_url: str, need_id: str, need: Mapping[str, object]) -> str | None: + """Link directly to a Need anchor when its exported page is known.""" + page_url = _need_url(base_url, need) + if page_url is None: + return None + return page_url + "#" + quote(need_id, safe="-._~") + + +def _markdown_link(label: str, url: str | None) -> str: + if url is None: + return f"`{label}`" + safe_label = label.replace("[", "\\[").replace("]", "\\]") + return f"[{safe_label}]({url})" + + +def _display_name(need: Mapping[str, object]) -> str: + for field in ("title", "name"): + value = need.get(field) + if isinstance(value, str) and value: + return value + return "" + + +def _format_value(value: object) -> str: + """Render one JSON field value compactly and safely for Markdown inline code.""" + if value is MISSING: + return "(missing)" + rendered = json.dumps( + value, sort_keys=True, ensure_ascii=False, separators=(",", ":") + ) + if len(rendered) > MAX_RENDERED_VALUE_LENGTH: + rendered = rendered[:MAX_RENDERED_VALUE_LENGTH] + "…" + return rendered.replace("`", "\\`").replace("\n", " ") + + +def _format_need_entry( + change: NeedChange, *, base_url: str, pr_url: str, include_diff: bool = False +) -> list[str]: + """Render one Need row, with old/new links and optional field details.""" + need = change.current or change.baseline + assert need is not None + links = [] + old_link = ( + need_link(base_url, change.need_id, change.baseline) + if change.baseline + else None + ) + new_link = ( + need_link(pr_url, change.need_id, change.current) if change.current else None + ) + if old_link: + links.append(_markdown_link("old", old_link)) + if new_link: + links.append(_markdown_link("new", new_link)) + link_text = " / ".join(links) + title = _display_name(need) + suffix = f" — {title}" if title else "" + line = f"- `{change.need_id}`{suffix}" + if link_text: + line += f" ({link_text})" + result = [line] + if include_diff: + assert change.baseline is not None and change.current is not None + for field in change.changed_fields: + result.append( + f" - `{field}`: {_format_value(change.baseline.get(field, MISSING))} " + f"→ {_format_value(change.current.get(field, MISSING))}" + ) + return result + + +def _format_page_entry( + change: PageChange, *, base_url: str, pr_url: str, kind: str +) -> str: + """Render a page row with links appropriate to its change category.""" + old = _url_with_path(base_url, change.path) if kind != "added" else None + new = _url_with_path(pr_url, change.path) if kind != "removed" else None + links = [] + if old: + links.append(_markdown_link("old", old)) + if new: + links.append(_markdown_link("new", new)) + return f"- `{change.path}` ({' / '.join(links)})" + + +def _need_section( + lines: list[str], + title: str, + entries: Sequence[NeedChange], + formatter: NeedFormatter, + *, + base_url: str, + pr_url: str, + kind: str = "", +) -> None: + """Append a Need section, suppressing field-by-field detail for large groups. + + The entries themselves are retained so reviewers can still see which Needs + changed. Modified-field values are the part that can make a large comment + unwieldy, so those are omitted once the group exceeds ``DETAIL_LIMIT``. + """ + if not entries: + return + lines.extend([f"### {title} ({len(entries)})", ""]) + include_field_diff = kind == "modified" and len(entries) <= DETAIL_LIMIT + if len(entries) > DETAIL_LIMIT: + lines.extend([f"{len(entries)} entries changed; field details omitted.", ""]) + for entry in entries: + lines.extend( + formatter( + entry, + base_url=base_url, + pr_url=pr_url, + include_diff=include_field_diff, + ) + ) + lines.append("") + + +def _page_section( + lines: list[str], + title: str, + entries: Sequence[PageChange], + formatter: PageFormatter, + *, + base_url: str, + pr_url: str, + kind: str, +) -> None: + """Append a page section; every entry remains directly reviewable.""" + if not entries: + return + lines.extend([f"### {title} ({len(entries)})", ""]) + for entry in entries: + lines.append(formatter(entry, base_url=base_url, pr_url=pr_url, kind=kind)) + lines.append("") + + +def render_report( + needs: NeedComparison, + pages: PageComparison, + *, + base_url: str, + pr_url: str, +) -> str: + """Render a deterministic, link-rich Markdown summary of both comparisons. + + Small change sets are expanded for immediate review. Larger sets are + wrapped in GitHub's collapsible-details markup so the comment stays + scannable while preserving every changed path or Need in the report. + """ + + lines = [ + "# Documentation delta", + "", + "Documentation preview for this pull request: " + f"{_markdown_link('open preview', pr_url)} · " + f"Baseline: {_markdown_link('open baseline', base_url)}", + "", + "## Summary", + "", + f"- Needs: {len(needs.added)} added, {len(needs.removed)} removed, " + f"{len(needs.modified)} modified, {needs.unchanged_count} unchanged", + f"- Rendered pages: {len(pages.added)} added, {len(pages.removed)} removed, " + f"{len(pages.modified)} modified, {pages.unchanged_count} unchanged", + ] + + need_change_count = len(needs.added) + len(needs.removed) + len(needs.modified) + if need_change_count: + need_lines: list[str] = [] + _need_section( + need_lines, + "Added", + needs.added, + _format_need_entry, + base_url=base_url, + pr_url=pr_url, + kind="added", + ) + _need_section( + need_lines, + "Removed", + needs.removed, + _format_need_entry, + base_url=base_url, + pr_url=pr_url, + kind="removed", + ) + _need_section( + need_lines, + "Modified", + needs.modified, + _format_need_entry, + base_url=base_url, + pr_url=pr_url, + kind="modified", + ) + if need_change_count > DETAIL_LIMIT: + lines.extend( + [ + "", + "
", + f"Need changes ({need_change_count})", + "", + *need_lines, + "
", + ] + ) + else: + lines.extend(["", "## Need changes", "", *need_lines]) + + page_change_count = len(pages.added) + len(pages.removed) + len(pages.modified) + if page_change_count: + page_lines: list[str] = [] + _page_section( + page_lines, + "Added", + pages.added, + _format_page_entry, + base_url=base_url, + pr_url=pr_url, + kind="added", + ) + _page_section( + page_lines, + "Removed", + pages.removed, + _format_page_entry, + base_url=base_url, + pr_url=pr_url, + kind="removed", + ) + _page_section( + page_lines, + "Modified", + pages.modified, + _format_page_entry, + base_url=base_url, + pr_url=pr_url, + kind="modified", + ) + if page_change_count > DETAIL_LIMIT: + # Keep the complete page list available, but collapsed by default + # so a broad Sphinx rebuild does not bury the useful summary. + lines.extend( + [ + "", + "
", + f"Page changes ({page_change_count})", + "", + *page_lines, + "
", + ] + ) + else: + lines.extend(["", "## Page changes", "", *page_lines]) + return "\n".join(lines).rstrip() + "\n" + + +def unavailable_report(reason: str) -> str: + """Render the successful, explicit report used when baseline is unavailable.""" + + return ( + "# Documentation delta\n\n" + "## Delta unavailable\n\n" + f"The documentation baseline is unavailable: {reason}\n" + ) + + +def write_report(path: Path, content: str) -> None: + """Atomically write the report so a failed run never leaves partial Markdown.""" + + path.parent.mkdir(parents=True, exist_ok=True) + temporary_path: Path | None = None + file_descriptor: int | None = None + try: + file_descriptor, temporary_name = tempfile.mkstemp( + dir=path.parent, prefix=f".{path.name}.", suffix=".tmp" + ) + temporary_path = Path(temporary_name) + with os.fdopen(file_descriptor, "w", encoding="utf-8") as output: + file_descriptor = None + output.write(content) + output.flush() + os.replace(temporary_path, path) + temporary_path = None + finally: + if file_descriptor is not None: + os.close(file_descriptor) + if temporary_path is not None: + temporary_path.unlink(missing_ok=True) diff --git a/tools/docs_delta/uv.lock b/tools/docs_delta/uv.lock new file mode 100644 index 000000000..57b2a9873 --- /dev/null +++ b/tools/docs_delta/uv.lock @@ -0,0 +1,8 @@ +version = 1 +revision = 3 +requires-python = ">=3.12" + +[[package]] +name = "score-docs-delta" +version = "0.1.0" +source = { editable = "." } diff --git a/tools/docs_delta_main.py b/tools/docs_delta_main.py new file mode 100644 index 000000000..86745af81 --- /dev/null +++ b/tools/docs_delta_main.py @@ -0,0 +1,18 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +"""Bazel launcher for the docs-delta CLI package.""" + +from tools.docs_delta.cli import main + +if __name__ == "__main__": # pragma: no cover - exercised via the Bazel binary + raise SystemExit(main()) diff --git a/tools/tests/BUILD b/tools/tests/BUILD index 73772cdb9..38daa623d 100644 --- a/tools/tests/BUILD +++ b/tools/tests/BUILD @@ -5,7 +5,7 @@ # information regarding copyright ownership. # # This program and the accompanying materials are made available under the -# terms of the Apache License, Version 2.0 which is available at +# terms of the Apache License Version 2.0 which is available at # https://www.apache.org/licenses/LICENSE-2.0 # # SPDX-License-Identifier: Apache-2.0 @@ -26,9 +26,9 @@ score_pytest( score_pytest( name = "docs_delta_test", - srcs = ["docs_delta_test.py"], + srcs = glob(["docs_delta_*_test.py", "docs_delta_test_support.py"]), deps = [ - "//tools:docs_delta", + "//tools:docs_delta_cli", ] + all_requirements, pytest_config = "//:pyproject.toml", ) diff --git a/tools/tests/docs_delta_cli_test.py b/tools/tests/docs_delta_cli_test.py new file mode 100644 index 000000000..e7c271250 --- /dev/null +++ b/tools/tests/docs_delta_cli_test.py @@ -0,0 +1,301 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +"""Tests for local and GitHub Actions command-line behavior.""" + +from __future__ import annotations + +import json +import subprocess +from pathlib import Path + +import pytest + +from tools import docs_delta +from tools.tests.docs_delta_test_support import git, need, write_needs + + +def test_missing_baseline_writes_unavailable_report(tmp_path: Path) -> None: + output = tmp_path / "nested" / "delta.md" + + assert ( + docs_delta.main( + [ + "--baseline-dir", + str(tmp_path / "missing"), + "--current-dir", + str(tmp_path / "current"), + "--base-url", + "https://docs.example/main", + "--pr-url", + "https://docs.example/pr/1", + "--output", + str(output), + ] + ) + == 0 + ) + assert "## Delta unavailable" in output.read_text(encoding="utf-8") + + +@pytest.mark.parametrize( + ("mode", "extra_args", "message"), + [ + ("directory", [], "--baseline-dir is required"), + ("gh-pages", [], "requires a GitHub pull request event"), + ( + "gh-pages", + ["--baseline-dir", "somewhere"], + "cannot be used with --baseline-mode gh-pages", + ), + ], +) +def test_baseline_mode_rejects_incompatible_options( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + capsys: pytest.CaptureFixture[str], + mode: str, + extra_args: list[str], + message: str, +) -> None: + monkeypatch.delenv("GITHUB_EVENT_NAME", raising=False) + monkeypatch.delenv("GITHUB_EVENT_PATH", raising=False) + + result = docs_delta.main( + [ + "--baseline-mode", + mode, + *extra_args, + "--current-dir", + str(tmp_path / "current"), + "--output", + str(tmp_path / "delta.md"), + ] + ) + + assert result == 2 + assert message in capsys.readouterr().err + + +def test_cli_reports_missing_current_directory_as_error( + tmp_path: Path, capsys: pytest.CaptureFixture[str] +) -> None: + baseline = tmp_path / "baseline" + baseline.mkdir() + write_needs(baseline, {}) + + result = docs_delta.main( + [ + "--baseline-dir", + str(baseline), + "--current-dir", + str(tmp_path / "missing-current"), + "--base-url", + "https://docs.example/main", + "--pr-url", + "https://docs.example/pr/1", + "--output", + str(tmp_path / "delta.md"), + ] + ) + + assert result == 2 + assert "documentation directory is missing" in capsys.readouterr().err + + +def test_cli_writes_complete_fixture_report(tmp_path: Path) -> None: + baseline = tmp_path / "baseline" + current = tmp_path / "current" + baseline.mkdir() + current.mkdir() + write_needs( + baseline, + { + "same": need("same", "Same", id="same"), + "changed": need("guide", "Old title", id="changed"), + "removed": need("removed", "Removed", id="removed"), + }, + ) + write_needs( + current, + { + "same": need("same", "Same", id="same"), + "changed": need("guide", "New title", id="changed"), + "added": need("new", "Added", id="added"), + }, + ) + (baseline / "guide.html").write_text("

old

\n", encoding="utf-8") + (current / "guide.html").write_text("

new

\n", encoding="utf-8") + (current / "new.html").write_text("

new page

\n", encoding="utf-8") + output = tmp_path / "delta.md" + + assert ( + docs_delta.main( + [ + "--baseline-dir", + str(baseline), + "--current-dir", + str(current), + "--base-url", + "https://docs.example/main", + "--pr-url", + "https://docs.example/pr/1", + "--output", + str(output), + ] + ) + == 0 + ) + report = output.read_text(encoding="utf-8") + assert "Needs: 1 added, 1 removed, 1 modified, 1 unchanged" in report + assert "Rendered pages: 1 added, 0 removed, 1 modified, 0 unchanged" in report + assert '`title`: "Old title" → "New title"' in report + assert ( + "`guide.html` ([old](https://docs.example/main/guide.html) / [new](https://docs.example/pr/1/guide.html))" + in report + ) + + +def test_cli_uses_github_environment_urls_when_options_are_omitted( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + baseline = tmp_path / "baseline" + current = tmp_path / "current" + baseline.mkdir() + current.mkdir() + write_needs(baseline, {}) + write_needs(current, {}) + monkeypatch.setenv("GITHUB_REPOSITORY", "eclipse-score/docs-as-code") + monkeypatch.setenv("GITHUB_BASE_REF", "main") + monkeypatch.setenv("GITHUB_HEAD_REF", "feature/docs-delta") + output = tmp_path / "delta.md" + + assert ( + docs_delta.main( + [ + "--baseline-dir", + str(baseline), + "--current-dir", + str(current), + "--output", + str(output), + ] + ) + == 0 + ) + report = output.read_text(encoding="utf-8") + assert "https://eclipse-score.github.io/docs-as-code/main" in report + assert "https://eclipse-score.github.io/docs-as-code/feature%2Fdocs-delta" in report + + +def test_cli_automatically_selects_matching_gh_pages_baseline( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """PR mode must select the historical publication matching the base SHA.""" + + gh_pages = tmp_path / ".docs-baseline" + gh_pages.mkdir() + subprocess.run( + ["git", "init", "--initial-branch=gh-pages", str(gh_pages)], + check=True, + capture_output=True, + text=True, + ) + git(gh_pages, "config", "user.name", "Documentation Delta Test") + git(gh_pages, "config", "user.email", "docs-delta@example.invalid") + + base_sha = "0123456789abcdef0123456789abcdef01234567" + latest_sha = "fedcba9876543210fedcba9876543210fedcba98" + main_docs = gh_pages / "main" + main_docs.mkdir() + write_needs( + main_docs, + { + "need": need( + "guide", "Base publication", id="need", source_code_link=base_sha + ) + }, + ) + (main_docs / "guide.html").write_text("

Base publication

\n", encoding="utf-8") + git(gh_pages, "add", ".") + git(gh_pages, "commit", "-m", "Publish base documentation") + + write_needs( + main_docs, + { + "need": need( + "guide", + "Newer base publication", + id="need", + source_code_link=base_sha, + ) + }, + ) + (main_docs / "guide.html").write_text( + "

Newer base publication

\n", encoding="utf-8" + ) + git(gh_pages, "add", ".") + git(gh_pages, "commit", "-m", "Republish base documentation") + write_needs( + main_docs, + { + "need": need( + "guide", "Latest publication", id="need", source_code_link=latest_sha + ) + }, + ) + (main_docs / "guide.html").write_text( + "

Latest publication

\n", encoding="utf-8" + ) + git(gh_pages, "add", ".") + git(gh_pages, "commit", "-m", "Publish latest documentation") + + current = tmp_path / "docs-artifact" + current.mkdir() + write_needs( + current, + { + "need": need( + "guide", + "Newer base publication", + id="need", + source_code_link="pr-sha", + ) + }, + ) + (current / "guide.html").write_text( + "

Newer base publication

\n", encoding="utf-8" + ) + + event = tmp_path / "event.json" + event.write_text( + json.dumps( + { + "pull_request": { + "number": 123, + "base": {"ref": "main", "sha": base_sha}, + } + } + ), + encoding="utf-8", + ) + monkeypatch.setenv("GITHUB_EVENT_NAME", "pull_request") + monkeypatch.setenv("GITHUB_EVENT_PATH", str(event)) + monkeypatch.setenv("GITHUB_WORKSPACE", str(tmp_path)) + monkeypatch.setenv("GITHUB_REPOSITORY", "eclipse-score/docs-as-code") + monkeypatch.setenv("GITHUB_BASE_REF", "main") + + assert docs_delta.main([]) == 0 + report = (current / "docs-delta.md").read_text(encoding="utf-8") + assert "Needs: 0 added, 0 removed, 0 modified, 1 unchanged" in report + assert "Rendered pages: 0 added, 0 removed, 0 modified, 1 unchanged" in report + assert "https://eclipse-score.github.io/docs-as-code/pr-123" in report diff --git a/tools/tests/docs_delta_comparison_test.py b/tools/tests/docs_delta_comparison_test.py new file mode 100644 index 000000000..728c90904 --- /dev/null +++ b/tools/tests/docs_delta_comparison_test.py @@ -0,0 +1,123 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +"""Tests for Need inventory loading and comparison.""" + +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +from tools import docs_delta +from tools.tests.docs_delta_test_support import need + + +def test_compare_needs_categorizes_changes_and_ignores_volatile_fields() -> None: + baseline = { + "same": need( + "same", + "Same", + id="same", + lineno=10, + source="old.rst", + source_code_link="old-commit", + testlink="old-result", + ), + "changed": need( + "changed", + "Old", + id="changed", + lineno=10, + source="old.rst", + source_code_link="old-commit", + testlink="old-result", + ), + "removed": need("removed", "Removed", id="removed"), + } + current = { + "same": need( + "same", + "Same", + id="same", + lineno=999, + source="new.rst", + source_code_link="new-commit", + testlink="new-result", + ), + "changed": need( + "changed", + "New", + id="changed", + lineno=20, + source="new.rst", + source_code_link="new-commit", + testlink="new-result", + ), + "added": need("added", "Added", id="added"), + } + + result = docs_delta.compare_needs(baseline, current) + + assert [entry.need_id for entry in result.added] == ["added"] + assert [entry.need_id for entry in result.removed] == ["removed"] + assert [entry.need_id for entry in result.modified] == ["changed"] + assert result.modified[0].changed_fields == ("title",) + assert result.unchanged_count == 1 + + +def test_need_comparison_distinguishes_missing_null_and_boolean_number() -> None: + baseline: dict[str, dict[str, object]] = { + "need": {"id": "need", "optional": None, "enabled": True} + } + current: dict[str, dict[str, object]] = {"need": {"id": "need", "enabled": 1}} + + result = docs_delta.compare_needs(baseline, current) + + assert result.modified[0].changed_fields == ("enabled", "optional") + + +@pytest.mark.parametrize( + "payload", + [ + {}, + {"versions": []}, + {"versions": {"1": {"needs": {"need": "not a record"}}}}, + ], +) +def test_load_needs_rejects_malformed_exports(tmp_path: Path, payload: object) -> None: + (tmp_path / "needs.json").write_text(json.dumps(payload), encoding="utf-8") + + with pytest.raises(docs_delta.DocsDeltaError, match="needs JSON|Need entry"): + docs_delta.load_needs(tmp_path) + + +def test_load_needs_flattens_versions_and_uses_last_duplicate_id( + tmp_path: Path, +) -> None: + (tmp_path / "needs.json").write_text( + json.dumps( + { + "versions": { + "1": {"needs": {"same": {"id": "same", "title": "Old"}}}, + "ignored": "malformed version container", + "2": { + "needs": {"same": {"id": "same", "title": "New"}}, + }, + } + } + ), + encoding="utf-8", + ) + + assert docs_delta.load_needs(tmp_path) == {"same": {"id": "same", "title": "New"}} diff --git a/tools/tests/docs_delta_github_test.py b/tools/tests/docs_delta_github_test.py new file mode 100644 index 000000000..af31a20a4 --- /dev/null +++ b/tools/tests/docs_delta_github_test.py @@ -0,0 +1,300 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +"""Tests for Actions metadata and gh-pages baseline resolution.""" + +from __future__ import annotations + +import io +import json +import subprocess +import tarfile +from pathlib import Path +from types import SimpleNamespace + +import pytest + +from tools import docs_delta +from tools.docs_delta import github +from tools.tests.docs_delta_test_support import git, need, write_needs + + +def test_pull_request_metadata_is_only_read_for_pr_events( + tmp_path: Path, +) -> None: + assert github.github_pull_request({"GITHUB_EVENT_NAME": "push"}) is None + + event = tmp_path / "event.json" + event.write_text("not JSON", encoding="utf-8") + with pytest.raises(docs_delta.DocsDeltaError, match="cannot read GitHub event"): + github.github_pull_request( + { + "GITHUB_EVENT_NAME": "pull_request", + "GITHUB_EVENT_PATH": str(event), + } + ) + + +def test_workspace_path_falls_back_to_relative_paths_locally() -> None: + assert github.workspace_path({}, "docs-artifact") == Path("docs-artifact") + assert github.workspace_path( + {"GITHUB_WORKSPACE": "/workspace"}, "docs-artifact" + ) == Path("/workspace/docs-artifact") + + +@pytest.mark.parametrize( + "payload", + [ + {}, + {"pull_request": "not an object"}, + {"pull_request": {"base": "not an object"}}, + {"pull_request": {"base": {"ref": "", "sha": ""}}}, + ], +) +def test_incomplete_pull_request_payload_does_not_create_a_baseline( + tmp_path: Path, payload: object +) -> None: + event = tmp_path / "event.json" + event.write_text(json.dumps(payload), encoding="utf-8") + + assert ( + github.github_pull_request( + { + "GITHUB_EVENT_NAME": "pull_request", + "GITHUB_EVENT_PATH": str(event), + } + ) + is None + ) + + +def test_non_object_event_payload_is_rejected(tmp_path: Path) -> None: + event = tmp_path / "event.json" + event.write_text("[]", encoding="utf-8") + + with pytest.raises(docs_delta.DocsDeltaError, match="payload is not an object"): + github.github_pull_request( + { + "GITHUB_EVENT_NAME": "pull_request", + "GITHUB_EVENT_PATH": str(event), + } + ) + + +def test_pull_request_payload_rejects_missing_event_path() -> None: + with pytest.raises( + docs_delta.DocsDeltaError, match="GITHUB_EVENT_PATH is required" + ): + github.github_pull_request({"GITHUB_EVENT_NAME": "pull_request"}) + + +def test_pull_request_metadata_uses_actions_overrides_and_ignores_boolean_number( + tmp_path: Path, +) -> None: + event = tmp_path / "event.json" + event.write_text( + json.dumps( + { + "pull_request": { + "number": True, + "base": {"ref": "event-main", "sha": "event-sha"}, + } + } + ), + encoding="utf-8", + ) + + pull_request = github.github_pull_request( + { + "GITHUB_EVENT_NAME": "pull_request", + "GITHUB_EVENT_PATH": str(event), + "GITHUB_BASE_REF": "actions-main", + "GITHUB_BASE_SHA": "actions-sha", + } + ) + + assert pull_request == github.GithubPullRequest( + base_ref="actions-main", base_sha="actions-sha", number=None + ) + + +def test_resolve_urls_requires_complete_defaults() -> None: + args = docs_delta.cli.argument_parser().parse_args([]) + + with pytest.raises( + docs_delta.DocsDeltaError, match="documentation URLs are required" + ): + github.resolve_urls(args, None, {}) + + +def test_unsafe_archive_path_is_rejected( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + archive_data = io.BytesIO() + with tarfile.open(fileobj=archive_data, mode="w") as archive: + member = tarfile.TarInfo("../escaped.html") + member.size = len(b"unsafe") + archive.addfile(member, io.BytesIO(b"unsafe")) + + monkeypatch.setattr( + github.subprocess, + "run", + lambda *args, **kwargs: SimpleNamespace( + returncode=0, stdout=archive_data.getvalue(), stderr=b"" + ), + ) + destination = tmp_path / "baseline" + + with pytest.raises(docs_delta.DocsDeltaError, match="unsafe path"): + github.extract_git_tree(tmp_path, "commit", "main", destination) + + assert not (tmp_path / "escaped.html").exists() + + +def test_unsupported_archive_member_is_rejected( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + archive_data = io.BytesIO() + with tarfile.open(fileobj=archive_data, mode="w") as archive: + member = tarfile.TarInfo("external-link") + member.type = tarfile.SYMTYPE + member.linkname = "outside" + archive.addfile(member) + + monkeypatch.setattr( + github.subprocess, + "run", + lambda *args, **kwargs: SimpleNamespace( + returncode=0, stdout=archive_data.getvalue(), stderr=b"" + ), + ) + + with pytest.raises( + docs_delta.DocsDeltaError, match="unsupported gh-pages archive member" + ): + github.extract_git_tree(tmp_path, "commit", "main", tmp_path / "baseline") + + +def test_archive_command_failure_is_reported( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + monkeypatch.setattr( + github.subprocess, + "run", + lambda *args, **kwargs: SimpleNamespace( + returncode=1, stdout=b"", stderr=b"missing tree" + ), + ) + + with pytest.raises(docs_delta.DocsDeltaError, match="missing tree"): + github.extract_git_tree(tmp_path, "commit", "main", tmp_path / "baseline") + + +def test_missing_gh_pages_checkout_writes_unavailable_report( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + current = tmp_path / "docs-artifact" + current.mkdir() + event = tmp_path / "event.json" + event.write_text( + json.dumps( + { + "pull_request": { + "number": 125, + "base": {"ref": "main", "sha": "base-sha"}, + } + } + ), + encoding="utf-8", + ) + monkeypatch.setenv("GITHUB_EVENT_NAME", "pull_request") + monkeypatch.setenv("GITHUB_EVENT_PATH", str(event)) + monkeypatch.setenv("GITHUB_WORKSPACE", str(tmp_path)) + + assert docs_delta.main([]) == 0 + report_text = (current / "docs-delta.md").read_text(encoding="utf-8") + assert "gh-pages checkout is missing" in report_text + + +def test_non_git_gh_pages_checkout_writes_unavailable_report( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + gh_pages = tmp_path / ".docs-baseline" + gh_pages.mkdir() + current = tmp_path / "docs-artifact" + current.mkdir() + event = tmp_path / "event.json" + event.write_text( + json.dumps( + { + "pull_request": { + "number": 126, + "base": {"ref": "main", "sha": "base-sha"}, + } + } + ), + encoding="utf-8", + ) + monkeypatch.setenv("GITHUB_EVENT_NAME", "pull_request") + monkeypatch.setenv("GITHUB_EVENT_PATH", str(event)) + monkeypatch.setenv("GITHUB_WORKSPACE", str(tmp_path)) + + assert docs_delta.main([]) == 0 + report_text = (current / "docs-delta.md").read_text(encoding="utf-8") + assert "gh-pages checkout is not a Git repository" in report_text + + +def test_unpublished_base_commit_writes_unavailable_report( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + gh_pages = tmp_path / ".docs-baseline" + gh_pages.mkdir() + subprocess.run( + ["git", "init", "--initial-branch=gh-pages", str(gh_pages)], + check=True, + capture_output=True, + text=True, + ) + git(gh_pages, "config", "user.name", "Documentation Delta Test") + git(gh_pages, "config", "user.email", "docs-delta@example.invalid") + main_docs = gh_pages / "main" + main_docs.mkdir() + write_needs(main_docs, {"need": need("guide", "Published", id="need")}) + git(gh_pages, "add", ".") + git(gh_pages, "commit", "-m", "Publish older documentation") + + current = tmp_path / "docs-artifact" + current.mkdir() + write_needs(current, {}) + event = tmp_path / "event.json" + event.write_text( + json.dumps( + { + "pull_request": { + "number": 124, + "base": { + "ref": "main", + "sha": "abcdef0123456789abcdef0123456789abcdef01", + }, + } + } + ), + encoding="utf-8", + ) + monkeypatch.setenv("GITHUB_EVENT_NAME", "pull_request") + monkeypatch.setenv("GITHUB_EVENT_PATH", str(event)) + monkeypatch.setenv("GITHUB_WORKSPACE", str(tmp_path)) + + assert docs_delta.main([]) == 0 + report_text = (current / "docs-delta.md").read_text(encoding="utf-8") + assert "## Delta unavailable" in report_text + assert "no published main baseline contains source commit" in report_text diff --git a/tools/tests/docs_delta_html_test.py b/tools/tests/docs_delta_html_test.py new file mode 100644 index 000000000..b8cf0f19d --- /dev/null +++ b/tools/tests/docs_delta_html_test.py @@ -0,0 +1,83 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +"""Tests for generated HTML normalization and page comparison.""" + +from __future__ import annotations + +from pathlib import Path + +import pytest + +from tools import docs_delta + + +def test_normalize_html_ignores_generated_metadata_but_keeps_content_changes() -> None: + old = ( + '\r\n' + '\r\n' + "\r\n" + '\r\n' + "

Documentation

\r\n" + ) + new = ( + '\n' + '\n' + "\n" + '\n' + "

Documentation

\n" + ) + + assert docs_delta.normalize_html(old) == docs_delta.normalize_html(new) + assert docs_delta.normalize_html( + new.replace("src/example.py", "src/changed.py") + ) != docs_delta.normalize_html(old) + assert docs_delta.normalize_html(new).replace( + "Documentation", "Changed" + ) != docs_delta.normalize_html(old) + + +def test_normalize_html_preserves_user_metadata_and_ordinary_comments() -> None: + content = '' + + assert docs_delta.normalize_html(content) == content + + +def test_compare_html_categorizes_added_removed_modified_and_unchanged( + tmp_path: Path, +) -> None: + baseline = tmp_path / "baseline" + current = tmp_path / "current" + (baseline / "nested").mkdir(parents=True) + (current / "nested").mkdir(parents=True) + (baseline / "same.html").write_text("

same

\n", encoding="utf-8") + (current / "same.html").write_text("

same

\n", encoding="utf-8") + (baseline / "changed.html").write_text("

old

\n", encoding="utf-8") + (current / "changed.html").write_text("

new

\n", encoding="utf-8") + (baseline / "removed.html").write_text("

removed

\n", encoding="utf-8") + (current / "added.html").write_text("

added

\n", encoding="utf-8") + (baseline / "nested/page.html").write_text("

old

\n", encoding="utf-8") + (current / "nested/page.html").write_text("

old

\n", encoding="utf-8") + + result = docs_delta.compare_html(baseline, current) + + assert [entry.path for entry in result.added] == ["added.html"] + assert [entry.path for entry in result.removed] == ["removed.html"] + assert [entry.path for entry in result.modified] == ["changed.html"] + assert result.unchanged_count == 2 + + +def test_compare_html_rejects_a_missing_directory(tmp_path: Path) -> None: + with pytest.raises( + docs_delta.DocsDeltaError, match="documentation directory is missing" + ): + docs_delta.compare_html(tmp_path / "missing", tmp_path / "current") diff --git a/tools/tests/docs_delta_report_test.py b/tools/tests/docs_delta_report_test.py new file mode 100644 index 000000000..bc7fe5752 --- /dev/null +++ b/tools/tests/docs_delta_report_test.py @@ -0,0 +1,155 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +"""Tests for Markdown report content, links, and file output.""" + +from __future__ import annotations + +from pathlib import Path + +import pytest + +from tools import docs_delta +from tools.docs_delta import report as report_module +from tools.tests.docs_delta_test_support import need + + +def test_modified_need_reports_all_changed_fields_and_one_sided_links() -> None: + result = docs_delta.render_report( + docs_delta.NeedComparison( + added=(docs_delta.NeedChange("added", None, need("new/page", "Added")),), + removed=( + docs_delta.NeedChange("removed", need("old/page", "Removed"), None), + ), + modified=( + docs_delta.NeedChange( + "changed", + need("old/page", "Old", tags=["old"], extra="before"), + need("new/page", "New", tags=["new"], extra="after"), + ), + ), + unchanged_count=0, + ), + docs_delta.PageComparison((), (), (), 0), + base_url="https://docs.example/main", + pr_url="https://docs.example/pr/1", + ) + + assert "[old](https://docs.example/main/old/page.html#changed)" in result + assert "[new](https://docs.example/pr/1/new/page.html#changed)" in result + assert '`extra`: "before" → "after"' in result + assert '`tags`: ["old"] → ["new"]' in result + assert "https://docs.example/pr/1/new/page.html#added" in result + assert "https://docs.example/main/old/page.html#removed" in result + + +def test_large_changed_values_are_truncated() -> None: + large_old = "a" * (docs_delta.MAX_RENDERED_VALUE_LENGTH + 20) + large_new = "b" * (docs_delta.MAX_RENDERED_VALUE_LENGTH + 20) + change = docs_delta.NeedChange( + "changed", + need("page", "Old", details=large_old), + need("page", "New", details=large_new), + ) + + report = docs_delta.render_report( + docs_delta.NeedComparison((), (), (change,), 0), + docs_delta.PageComparison((), (), (), 0), + base_url="https://docs.example/main", + pr_url="https://docs.example/pr/1", + ) + + assert "…" in report + assert large_old not in report + assert large_new not in report + + +def test_detail_threshold_is_independent_for_needs_and_pages() -> None: + needs = docs_delta.NeedComparison( + added=tuple( + docs_delta.NeedChange(str(index), None, need("page", str(index))) + for index in range(16) + ), + removed=(), + modified=(), + unchanged_count=0, + ) + pages = docs_delta.PageComparison( + added=tuple(docs_delta.PageChange(f"page-{index}.html") for index in range(16)), + removed=(), + modified=(), + unchanged_count=0, + ) + + report = docs_delta.render_report( + needs, + pages, + base_url="https://docs.example/main", + pr_url="https://docs.example/pr/1", + ) + + assert "16 entries changed; field details omitted." in report + assert "Page changes (16)" in report + assert "`page-0.html`" in report + assert "- `0`" in report + + +def test_large_modified_need_section_keeps_ids_but_hides_field_values() -> None: + changes = tuple( + docs_delta.NeedChange( + f"REQ-{index:02}", + need("page", "Old title", details=f"old secret {index}"), + need("page", "New title", details=f"new secret {index}"), + ) + for index in range(16) + ) + + result = docs_delta.render_report( + docs_delta.NeedComparison((), (), changes, 0), + docs_delta.PageComparison((), (), (), 0), + base_url="https://docs.example/main", + pr_url="https://docs.example/pr/1", + ) + + assert "Need changes (16)" in result + assert "16 entries changed; field details omitted." in result + assert "`REQ-00`" in result and "`REQ-15`" in result + assert "old secret" not in result + assert "new secret" not in result + + +def test_need_links_encode_path_segments_and_anchors() -> None: + url = docs_delta.need_link( + "https://docs.example/main/", + "REQ #1", + {"docname": "nested/a guide"}, + ) + + assert url == "https://docs.example/main/nested/a%20guide.html#REQ%20%231" + + +def test_report_write_failure_preserves_previous_output_and_cleans_up( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + output = tmp_path / "delta.md" + output.write_text("previous report", encoding="utf-8") + + def fail_replace(source: Path, destination: Path) -> None: + raise OSError("simulated replace failure") + + monkeypatch.setattr(report_module.os, "replace", fail_replace) + + with pytest.raises(OSError, match="simulated replace failure"): + report_module.write_report(output, "new report") + + assert output.read_text(encoding="utf-8") == "previous report" + assert list(tmp_path.glob(".delta.md.*.tmp")) == [] diff --git a/tools/tests/docs_delta_test.py b/tools/tests/docs_delta_test.py deleted file mode 100644 index 7f9a98a1f..000000000 --- a/tools/tests/docs_delta_test.py +++ /dev/null @@ -1,436 +0,0 @@ -# ******************************************************************************* -# Copyright (c) 2026 Contributors to the Eclipse Foundation -# -# See the NOTICE file(s) distributed with this work for additional -# information regarding copyright ownership. -# -# This program and the accompanying materials are made available under the -# terms of the Apache License, Version 2.0 which is available at -# https://www.apache.org/licenses/LICENSE-2.0 -# -# SPDX-License-Identifier: Apache-2.0 -# ******************************************************************************* -"""Unit and fixture-based integration tests for the documentation delta tool.""" - -from __future__ import annotations - -import json -import subprocess -from pathlib import Path - -import pytest - -from tools import docs_delta - - -def _need(docname: str, title: str, **extra: object) -> dict[str, object]: - return {"id": title.lower(), "docname": docname, "title": title, **extra} - - -def _write_needs(directory: Path, needs: dict[str, dict[str, object]]) -> None: - (directory / "needs.json").write_text( - json.dumps({"versions": {"1": {"needs": needs}}}), encoding="utf-8" - ) - - -def _git(directory: Path, *arguments: str) -> None: - subprocess.run( - ["git", "-C", str(directory), *arguments], - check=True, - capture_output=True, - text=True, - ) - - -def test_compare_needs_categorizes_changes_and_ignores_volatile_fields() -> None: - baseline = { - "same": _need( - "same", - "Same", - id="same", - lineno=10, - source="old.rst", - source_code_link="old-commit", - testlink="old-result", - ), - "changed": _need( - "changed", - "Old", - id="changed", - lineno=10, - source="old.rst", - source_code_link="old-commit", - testlink="old-result", - ), - "removed": _need("removed", "Removed", id="removed"), - } - current = { - "same": _need( - "same", - "Same", - id="same", - lineno=999, - source="new.rst", - source_code_link="new-commit", - testlink="new-result", - ), - "changed": _need( - "changed", - "New", - id="changed", - lineno=20, - source="new.rst", - source_code_link="new-commit", - testlink="new-result", - ), - "added": _need("added", "Added", id="added"), - } - - result = docs_delta.compare_needs(baseline, current) - - assert [entry.need_id for entry in result.added] == ["added"] - assert [entry.need_id for entry in result.removed] == ["removed"] - assert [entry.need_id for entry in result.modified] == ["changed"] - assert result.modified[0].changed_fields == ("title",) - assert result.unchanged_count == 1 - - -def test_modified_need_reports_all_changed_fields_and_one_sided_links() -> None: - result = docs_delta.render_report( - docs_delta.NeedComparison( - added=(docs_delta.NeedChange("added", None, _need("new/page", "Added")),), - removed=( - docs_delta.NeedChange("removed", _need("old/page", "Removed"), None), - ), - modified=( - docs_delta.NeedChange( - "changed", - _need("old/page", "Old", tags=["old"], extra="before"), - _need("new/page", "New", tags=["new"], extra="after"), - ), - ), - unchanged_count=0, - ), - docs_delta.PageComparison((), (), (), 0), - base_url="https://docs.example/main", - pr_url="https://docs.example/pr/1", - ) - - assert "[old](https://docs.example/main/old/page.html#changed)" in result - assert "[new](https://docs.example/pr/1/new/page.html#changed)" in result - assert '`extra`: "before" → "after"' in result - assert '`tags`: ["old"] → ["new"]' in result - assert "https://docs.example/pr/1/new/page.html#added" in result - assert "https://docs.example/main/old/page.html#removed" in result - - -def test_large_changed_values_are_truncated() -> None: - large_old = "a" * (docs_delta.MAX_RENDERED_VALUE_LENGTH + 20) - large_new = "b" * (docs_delta.MAX_RENDERED_VALUE_LENGTH + 20) - change = docs_delta.NeedChange( - "changed", - _need("page", "Old", details=large_old), - _need("page", "New", details=large_new), - ) - - report = docs_delta.render_report( - docs_delta.NeedComparison((), (), (change,), 0), - docs_delta.PageComparison((), (), (), 0), - base_url="https://docs.example/main", - pr_url="https://docs.example/pr/1", - ) - - assert "…" in report - assert large_old not in report - assert large_new not in report - - -def test_normalize_html_ignores_generated_metadata_but_keeps_content_changes() -> None: - old = ( - '\r\n' - '\r\n' - "\r\n" - '\r\n' - "

Documentation

\r\n" - ) - new = ( - '\n' - '\n' - "\n" - '\n' - "

Documentation

\n" - ) - - assert docs_delta.normalize_html(old) == docs_delta.normalize_html(new) - assert docs_delta.normalize_html( - new.replace("src/example.py", "src/changed.py") - ) != docs_delta.normalize_html(old) - assert docs_delta.normalize_html(new).replace( - "Documentation", "Changed" - ) != docs_delta.normalize_html(old) - - -def test_compare_html_categorizes_added_removed_modified_and_unchanged( - tmp_path: Path, -) -> None: - baseline = tmp_path / "baseline" - current = tmp_path / "current" - (baseline / "nested").mkdir(parents=True) - (current / "nested").mkdir(parents=True) - (baseline / "same.html").write_text("

same

\n", encoding="utf-8") - (current / "same.html").write_text("

same

\n", encoding="utf-8") - (baseline / "changed.html").write_text("

old

\n", encoding="utf-8") - (current / "changed.html").write_text("

new

\n", encoding="utf-8") - (baseline / "removed.html").write_text("

removed

\n", encoding="utf-8") - (current / "added.html").write_text("

added

\n", encoding="utf-8") - (baseline / "nested/page.html").write_text("

old

\n", encoding="utf-8") - (current / "nested/page.html").write_text("

old

\n", encoding="utf-8") - - result = docs_delta.compare_html(baseline, current) - - assert [entry.path for entry in result.added] == ["added.html"] - assert [entry.path for entry in result.removed] == ["removed.html"] - assert [entry.path for entry in result.modified] == ["changed.html"] - assert result.unchanged_count == 2 - - -def test_detail_threshold_is_independent_for_needs_and_pages() -> None: - needs = docs_delta.NeedComparison( - added=tuple( - docs_delta.NeedChange(str(index), None, _need("page", str(index))) - for index in range(16) - ), - removed=(), - modified=(), - unchanged_count=0, - ) - pages = docs_delta.PageComparison( - added=tuple(docs_delta.PageChange(f"page-{index}.html") for index in range(16)), - removed=(), - modified=(), - unchanged_count=0, - ) - - report = docs_delta.render_report( - needs, - pages, - base_url="https://docs.example/main", - pr_url="https://docs.example/pr/1", - ) - - assert "16 entries changed; details omitted." in report - assert "Page changes (16)" in report - assert "`page-0.html`" in report - assert "`0`" not in report - - -def test_missing_baseline_writes_unavailable_report(tmp_path: Path) -> None: - output = tmp_path / "nested" / "delta.md" - - assert ( - docs_delta.main( - [ - "--baseline-dir", - str(tmp_path / "missing"), - "--current-dir", - str(tmp_path / "current"), - "--base-url", - "https://docs.example/main", - "--pr-url", - "https://docs.example/pr/1", - "--output", - str(output), - ] - ) - == 0 - ) - assert "## Delta unavailable" in output.read_text(encoding="utf-8") - - -def test_cli_writes_complete_fixture_report(tmp_path: Path) -> None: - baseline = tmp_path / "baseline" - current = tmp_path / "current" - baseline.mkdir() - current.mkdir() - _write_needs( - baseline, - { - "same": _need("same", "Same", id="same"), - "changed": _need("guide", "Old title", id="changed"), - "removed": _need("removed", "Removed", id="removed"), - }, - ) - _write_needs( - current, - { - "same": _need("same", "Same", id="same"), - "changed": _need("guide", "New title", id="changed"), - "added": _need("new", "Added", id="added"), - }, - ) - (baseline / "guide.html").write_text("

old

\n", encoding="utf-8") - (current / "guide.html").write_text("

new

\n", encoding="utf-8") - (current / "new.html").write_text("

new page

\n", encoding="utf-8") - output = tmp_path / "delta.md" - - assert ( - docs_delta.main( - [ - "--baseline-dir", - str(baseline), - "--current-dir", - str(current), - "--base-url", - "https://docs.example/main", - "--pr-url", - "https://docs.example/pr/1", - "--output", - str(output), - ] - ) - == 0 - ) - report = output.read_text(encoding="utf-8") - assert "Needs: 1 added, 1 removed, 1 modified, 1 unchanged" in report - assert "Rendered pages: 1 added, 0 removed, 1 modified, 0 unchanged" in report - assert '`title`: "Old title" → "New title"' in report - assert ( - "`guide.html` ([old](https://docs.example/main/guide.html) / [new](https://docs.example/pr/1/guide.html))" - in report - ) - - -def test_cli_uses_github_environment_urls_when_options_are_omitted( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - baseline = tmp_path / "baseline" - current = tmp_path / "current" - baseline.mkdir() - current.mkdir() - _write_needs(baseline, {}) - _write_needs(current, {}) - monkeypatch.setenv("GITHUB_REPOSITORY", "eclipse-score/docs-as-code") - monkeypatch.setenv("GITHUB_BASE_REF", "main") - monkeypatch.setenv("GITHUB_HEAD_REF", "feature/docs-delta") - output = tmp_path / "delta.md" - - assert ( - docs_delta.main( - [ - "--baseline-dir", - str(baseline), - "--current-dir", - str(current), - "--output", - str(output), - ] - ) - == 0 - ) - report = output.read_text(encoding="utf-8") - assert "https://eclipse-score.github.io/docs-as-code/main" in report - assert "https://eclipse-score.github.io/docs-as-code/feature%2Fdocs-delta" in report - - -def test_cli_automatically_selects_matching_gh_pages_baseline( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """PR mode must select the historical publication matching the base SHA.""" - - gh_pages = tmp_path / ".docs-baseline" - gh_pages.mkdir() - subprocess.run( - ["git", "init", "--initial-branch=gh-pages", str(gh_pages)], - check=True, - capture_output=True, - text=True, - ) - _git(gh_pages, "config", "user.name", "Documentation Delta Test") - _git(gh_pages, "config", "user.email", "docs-delta@example.invalid") - - base_sha = "0123456789abcdef0123456789abcdef01234567" - latest_sha = "fedcba9876543210fedcba9876543210fedcba98" - main_docs = gh_pages / "main" - main_docs.mkdir() - _write_needs( - main_docs, - { - "need": _need( - "guide", "Base publication", id="need", source_code_link=base_sha - ) - }, - ) - (main_docs / "guide.html").write_text("

Base publication

\n", encoding="utf-8") - _git(gh_pages, "add", ".") - _git(gh_pages, "commit", "-m", "Publish base documentation") - - _write_needs( - main_docs, - { - "need": _need( - "guide", - "Newer base publication", - id="need", - source_code_link=base_sha, - ) - }, - ) - (main_docs / "guide.html").write_text( - "

Newer base publication

\n", encoding="utf-8" - ) - _git(gh_pages, "add", ".") - _git(gh_pages, "commit", "-m", "Republish base documentation") - _write_needs( - main_docs, - { - "need": _need( - "guide", "Latest publication", id="need", source_code_link=latest_sha - ) - }, - ) - (main_docs / "guide.html").write_text( - "

Latest publication

\n", encoding="utf-8" - ) - _git(gh_pages, "add", ".") - _git(gh_pages, "commit", "-m", "Publish latest documentation") - - current = tmp_path / "docs-artifact" - current.mkdir() - _write_needs( - current, - { - "need": _need( - "guide", - "Newer base publication", - id="need", - source_code_link="pr-sha", - ) - }, - ) - (current / "guide.html").write_text( - "

Newer base publication

\n", encoding="utf-8" - ) - - event = tmp_path / "event.json" - event.write_text( - json.dumps( - { - "pull_request": { - "number": 123, - "base": {"ref": "main", "sha": base_sha}, - } - } - ), - encoding="utf-8", - ) - monkeypatch.setenv("GITHUB_EVENT_NAME", "pull_request") - monkeypatch.setenv("GITHUB_EVENT_PATH", str(event)) - monkeypatch.setenv("GITHUB_WORKSPACE", str(tmp_path)) - monkeypatch.setenv("GITHUB_REPOSITORY", "eclipse-score/docs-as-code") - monkeypatch.setenv("GITHUB_BASE_REF", "main") - - assert docs_delta.main([]) == 0 - report = (current / "docs-delta.md").read_text(encoding="utf-8") - assert "Needs: 0 added, 0 removed, 0 modified, 1 unchanged" in report - assert "Rendered pages: 0 added, 0 removed, 0 modified, 1 unchanged" in report - assert "https://eclipse-score.github.io/docs-as-code/pr-123" in report diff --git a/tools/tests/docs_delta_test_support.py b/tools/tests/docs_delta_test_support.py new file mode 100644 index 000000000..7fe3859e8 --- /dev/null +++ b/tools/tests/docs_delta_test_support.py @@ -0,0 +1,41 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +"""Shared fixture helpers for documentation delta tests.""" + +from __future__ import annotations + +import json +import subprocess +from pathlib import Path + + +def need(docname: str, title: str, **extra: object) -> dict[str, object]: + """Make a small representative Sphinx-Needs record for unit tests.""" + return {"id": title.lower(), "docname": docname, "title": title, **extra} + + +def write_needs(directory: Path, needs: dict[str, dict[str, object]]) -> None: + """Write the versioned JSON envelope expected from Sphinx-Needs.""" + (directory / "needs.json").write_text( + json.dumps({"versions": {"1": {"needs": needs}}}), encoding="utf-8" + ) + + +def git(directory: Path, *arguments: str) -> None: + """Run a Git command against a temporary repository in an integration test.""" + subprocess.run( + ["git", "-C", str(directory), *arguments], + check=True, + capture_output=True, + text=True, + ) From 9a062c3333c5c49922580583c30d1a8a59a5e2fc Mon Sep 17 00:00:00 2001 From: Alexander Lanin Date: Wed, 30 Sep 2026 09:40:04 +0200 Subject: [PATCH 7/7] Document pull request need summaries --- docs/internals/requirements/tool_verification.rst | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/docs/internals/requirements/tool_verification.rst b/docs/internals/requirements/tool_verification.rst index cc588cd5e..90e63a67d 100644 --- a/docs/internals/requirements/tool_verification.rst +++ b/docs/internals/requirements/tool_verification.rst @@ -82,7 +82,9 @@ Report record :post_template: tool_qualification_report Evaluates the S-CORE Docs-as-Code tool for building and checking - documentation and traceability data from RST/Markdown sources. + documentation and traceability data from RST/Markdown sources. It also + summarizes changed Needs between the published documentation and a pull + request's proposed documentation. Details -------