diff --git a/.github/workflows/on-pr.yml b/.github/workflows/on-pr.yml
index 1a6c653bf..67e78cd01 100644
--- a/.github/workflows/on-pr.yml
+++ b/.github/workflows/on-pr.yml
@@ -43,3 +43,103 @@ jobs:
with:
bazel-target: "//:docs"
tests-report-artifact: tests-report
+
+ docs-delta:
+ needs: [docs-build]
+ if: >-
+ github.event_name == 'pull_request' &&
+ github.event.pull_request.head.repo.full_name == github.repository
+ runs-on: ubuntu-24.04
+ permissions:
+ actions: read
+ contents: read
+ steps:
+ - name: Check out pull request
+ uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
+ with:
+ persist-credentials: false
+
+ - name: Check out published documentation history
+ continue-on-error: true
+ uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
+ with:
+ repository: ${{ github.repository }}
+ ref: gh-pages
+ path: .docs-baseline
+ fetch-depth: 0
+ persist-credentials: false
+
+ - name: Download documentation artifact
+ uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
+ with:
+ name: github-pages
+ path: docs-artifact
+
+ - name: Set up uv
+ uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
+
+ - name: Generate documentation delta report
+ run: |
+ set -euo pipefail
+ uv run --locked --project tools/docs_delta docs-delta
+
+ - name: Upload documentation delta artifact
+ uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
+ with:
+ name: docs-delta
+ path: docs-artifact/docs-delta.md
+ if-no-files-found: error
+
+ docs-comment:
+ needs: [docs-build, docs-delta]
+ if: >-
+ always() &&
+ github.event_name == 'pull_request' &&
+ github.event.pull_request.head.repo.full_name == github.repository
+ runs-on: ubuntu-24.04
+ permissions:
+ actions: read
+ pull-requests: write
+ steps:
+ - name: Download documentation delta artifact
+ if: needs.docs-delta.result == 'success'
+ uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
+ with:
+ name: docs-delta
+ path: docs-delta
+
+ - name: Prepare documentation preview comment
+ env:
+ DOCS_BUILD_RESULT: ${{ needs.docs-build.result }}
+ DOCS_DELTA_RESULT: ${{ needs.docs-delta.result }}
+ run: |
+ set -euo pipefail
+ if [ "$DOCS_BUILD_RESULT" = success ] && [ "$DOCS_DELTA_RESULT" = success ]; then
+ cat docs-delta/docs-delta.md > "$RUNNER_TEMP/docs-comment.md"
+ else
+ {
+ echo "Documentation preview for this pull request is unavailable."
+ echo
+ if [ "$DOCS_BUILD_RESULT" != success ]; then
+ echo "The documentation build did not complete successfully (result: ${DOCS_BUILD_RESULT})."
+ else
+ echo "The documentation delta job did not complete successfully (result: ${DOCS_DELTA_RESULT})."
+ fi
+ } > "$RUNNER_TEMP/docs-comment.md"
+ fi
+
+ - name: Find existing documentation preview comment
+ id: find-comment
+ uses: peter-evans/find-comment@b30e6a3c0ed37e7c023ccd3f1db5c6c0b0c23aad # v4.0.0
+ with:
+ issue-number: ${{ github.event.pull_request.number }}
+ comment-author: github-actions[bot]
+ body-includes: Documentation preview for this pull request
+
+ - name: Create or update documentation preview comment
+ uses: peter-evans/create-or-update-comment@e8674b075228eee787fea43ef493e45ece1004c9 # v5.0.0
+ with:
+ issue-number: ${{ github.event.pull_request.number }}
+ comment-id: ${{ steps.find-comment.outputs.comment-id }}
+ body-path: ${{ runner.temp }}/docs-comment.md
+ edit-mode: replace
diff --git a/docs/internals/requirements/tool_verification.rst b/docs/internals/requirements/tool_verification.rst
index cc588cd5e..90e63a67d 100644
--- a/docs/internals/requirements/tool_verification.rst
+++ b/docs/internals/requirements/tool_verification.rst
@@ -82,7 +82,9 @@ Report record
:post_template: tool_qualification_report
Evaluates the S-CORE Docs-as-Code tool for building and checking
- documentation and traceability data from RST/Markdown sources.
+ documentation and traceability data from RST/Markdown sources. It also
+ summarizes changed Needs between the published documentation and a pull
+ request's proposed documentation.
Details
-------
diff --git a/tools/BUILD b/tools/BUILD
index 93b6741da..b9d093c5b 100644
--- a/tools/BUILD
+++ b/tools/BUILD
@@ -5,7 +5,7 @@
# information regarding copyright ownership.
#
# This program and the accompanying materials are made available under the
-# terms of the Apache License, Version 2.0 which is available at
+# terms of the Apache License Version 2.0 which is available at
# https://www.apache.org/licenses/LICENSE-2.0
#
# SPDX-License-Identifier: Apache-2.0
@@ -32,3 +32,19 @@ py_binary(
"//src/helper_lib",
],
)
+
+py_binary(
+ name = "docs_delta_cli",
+ srcs = ["docs_delta_main.py"] + glob(["docs_delta/*.py"]),
+ main = "docs_delta_main.py",
+ visibility = ["//visibility:public"],
+ deps = [],
+)
+
+# Retain the original Bazel label for callers while the package itself now
+# occupies the ``tools/docs_delta`` source path.
+alias(
+ name = "docs_delta",
+ actual = ":docs_delta_cli",
+ visibility = ["//visibility:public"],
+)
diff --git a/tools/docs_delta/README.md b/tools/docs_delta/README.md
new file mode 100644
index 000000000..8520648e1
--- /dev/null
+++ b/tools/docs_delta/README.md
@@ -0,0 +1,39 @@
+
+
+# Documentation Delta
+
+Compare the Sphinx-Needs inventory and rendered HTML from two documentation
+builds, then write a Markdown report. The command uses only the Python standard
+library.
+
+Run the CLI from the repository root with:
+
+```sh
+uv run --locked --project tools/docs_delta docs-delta --help
+```
+
+For a local comparison, provide the baseline and current build directories and
+the URLs that reviewers should open:
+
+```sh
+uv run --locked --project tools/docs_delta docs-delta \
+ --baseline-dir /path/to/baseline \
+ --current-dir /path/to/current \
+ --base-url https://example.org/docs/main \
+ --pr-url https://example.org/docs/pr-123
+```
+
+In GitHub pull-request Actions, the CLI can resolve the baseline from a local,
+full-history `gh-pages` checkout and derive the preview URLs from Actions
+metadata. `docs-delta --help` lists the options for overriding those defaults.
diff --git a/tools/docs_delta/__init__.py b/tools/docs_delta/__init__.py
new file mode 100644
index 000000000..aec8fd0d2
--- /dev/null
+++ b/tools/docs_delta/__init__.py
@@ -0,0 +1,50 @@
+# *******************************************************************************
+# Copyright (c) 2026 Contributors to the Eclipse Foundation
+#
+# See the NOTICE file(s) distributed with this work for additional
+# information regarding copyright ownership.
+#
+# This program and the accompanying materials are made available under the
+# terms of the Apache License Version 2.0 which is available at
+# https://www.apache.org/licenses/LICENSE-2.0
+#
+# SPDX-License-Identifier: Apache-2.0
+# *******************************************************************************
+"""Compare documentation builds and format a Markdown delta."""
+
+from .cli import main
+from .comparison import (
+ DocsDeltaError,
+ NeedChange,
+ NeedComparison,
+ NeedMap,
+ PageChange,
+ PageComparison,
+ compare_needs,
+ load_needs,
+)
+from .rendered_html import compare_html, normalize_html
+from .report import (
+ MAX_RENDERED_VALUE_LENGTH,
+ need_link,
+ render_report,
+ unavailable_report,
+)
+
+__all__ = [
+ "DocsDeltaError",
+ "MAX_RENDERED_VALUE_LENGTH",
+ "NeedChange",
+ "NeedComparison",
+ "NeedMap",
+ "PageChange",
+ "PageComparison",
+ "compare_html",
+ "compare_needs",
+ "load_needs",
+ "main",
+ "need_link",
+ "normalize_html",
+ "render_report",
+ "unavailable_report",
+]
diff --git a/tools/docs_delta/__main__.py b/tools/docs_delta/__main__.py
new file mode 100644
index 000000000..d24966006
--- /dev/null
+++ b/tools/docs_delta/__main__.py
@@ -0,0 +1,17 @@
+# *******************************************************************************
+# Copyright (c) 2026 Contributors to the Eclipse Foundation
+#
+# See the NOTICE file(s) distributed with this work for additional
+# information regarding copyright ownership.
+#
+# This program and the accompanying materials are made available under the
+# terms of the Apache License Version 2.0 which is available at
+# https://www.apache.org/licenses/LICENSE-2.0
+#
+# SPDX-License-Identifier: Apache-2.0
+# *******************************************************************************
+"""Run the docs-delta command as ``python -m docs_delta``."""
+
+from .cli import main
+
+raise SystemExit(main())
diff --git a/tools/docs_delta/cli.py b/tools/docs_delta/cli.py
new file mode 100644
index 000000000..8f41643ab
--- /dev/null
+++ b/tools/docs_delta/cli.py
@@ -0,0 +1,109 @@
+# *******************************************************************************
+# Copyright (c) 2026 Contributors to the Eclipse Foundation
+#
+# See the NOTICE file(s) distributed with this work for additional
+# information regarding copyright ownership.
+#
+# This program and the accompanying materials are made available under the
+# terms of the Apache License Version 2.0 which is available at
+# https://www.apache.org/licenses/LICENSE-2.0
+#
+# SPDX-License-Identifier: Apache-2.0
+# *******************************************************************************
+"""Command-line interface for creating documentation delta reports."""
+
+from __future__ import annotations
+
+import argparse
+import os
+import sys
+from collections.abc import Sequence
+from pathlib import Path
+
+from .comparison import DocsDeltaError, compare_needs, load_needs
+from .github import github_pull_request, resolve_urls, resolved_baseline, workspace_path
+from .rendered_html import compare_html
+from .report import render_report, unavailable_report, write_report
+
+
+def argument_parser() -> argparse.ArgumentParser:
+ """Describe the local and GitHub Actions forms of the command line."""
+ parser = argparse.ArgumentParser(description=__doc__)
+ parser.add_argument(
+ "--baseline-dir",
+ type=Path,
+ help="directory baseline; defaults to automatic gh-pages mode in PR Actions",
+ )
+ parser.add_argument(
+ "--baseline-mode",
+ choices=("directory", "gh-pages"),
+ help="baseline source; inferred when omitted",
+ )
+ parser.add_argument(
+ "--gh-pages-dir",
+ type=Path,
+ help="local full-history gh-pages checkout (defaults to .docs-baseline)",
+ )
+ parser.add_argument(
+ "--current-dir",
+ type=Path,
+ help="current documentation directory (defaults to docs-artifact)",
+ )
+ parser.add_argument("--base-url")
+ parser.add_argument("--pr-url")
+ parser.add_argument(
+ "--output",
+ type=Path,
+ help="report path (defaults to docs-artifact/docs-delta.md)",
+ )
+ return parser
+
+
+def main(argv: Sequence[str] | None = None) -> int:
+ """Run the CLI and write either a delta or an explicit unavailable report.
+
+ A baseline may legitimately be absent while publishing is in progress; in
+ that case the report is still successful and explains why it has no delta.
+ Missing or malformed current build data is an error because it would make
+ even the PR side of the comparison unreliable.
+ """
+
+ args = argument_parser().parse_args(argv)
+ try:
+ pull_request = github_pull_request(os.environ)
+ current_dir = args.current_dir or workspace_path(os.environ, "docs-artifact")
+ output = args.output or current_dir / "docs-delta.md"
+
+ with resolved_baseline(args, pull_request, os.environ) as (
+ baseline_dir,
+ baseline_reason,
+ ):
+ if baseline_dir is None or not baseline_dir.is_dir():
+ # A stale baseline would produce a plausible but misleading
+ # diff. Surface the missing comparison in the comment instead.
+ reason = baseline_reason or f"directory is missing: {baseline_dir}"
+ write_report(output, unavailable_report(reason))
+ return 0
+ try:
+ baseline_needs = load_needs(baseline_dir)
+ except DocsDeltaError as exc:
+ write_report(output, unavailable_report(str(exc)))
+ return 0
+
+ if not current_dir.is_dir():
+ raise DocsDeltaError(
+ f"documentation directory is missing: {current_dir}"
+ )
+ current_needs = load_needs(current_dir)
+ base_url, pr_url = resolve_urls(args, pull_request, os.environ)
+ report = render_report(
+ compare_needs(baseline_needs, current_needs),
+ compare_html(baseline_dir, current_dir),
+ base_url=base_url,
+ pr_url=pr_url,
+ )
+ write_report(output, report)
+ except (DocsDeltaError, OSError) as exc:
+ print(f"docs_delta: error: {exc}", file=sys.stderr)
+ return 2
+ return 0
diff --git a/tools/docs_delta/comparison.py b/tools/docs_delta/comparison.py
new file mode 100644
index 000000000..ba23a42d3
--- /dev/null
+++ b/tools/docs_delta/comparison.py
@@ -0,0 +1,205 @@
+# *******************************************************************************
+# Copyright (c) 2026 Contributors to the Eclipse Foundation
+#
+# See the NOTICE file(s) distributed with this work for additional
+# information regarding copyright ownership.
+#
+# This program and the accompanying materials are made available under the
+# terms of the Apache License Version 2.0 which is available at
+# https://www.apache.org/licenses/LICENSE-2.0
+#
+# SPDX-License-Identifier: Apache-2.0
+# *******************************************************************************
+"""Need inventory types, loading, and comparison rules."""
+
+from __future__ import annotations
+
+import json
+from collections.abc import Mapping
+from dataclasses import dataclass
+from pathlib import Path
+from typing import cast
+
+JsonObject = dict[str, object]
+NeedMap = Mapping[str, JsonObject]
+MISSING = object()
+# Source locations, generated IDs, and test-result links vary with the build
+# revision without changing the requirement represented by a Need.
+VOLATILE_NEED_FIELDS = frozenset(
+ {
+ "lineno",
+ "lineno_content",
+ "source",
+ "target_id",
+ "is_modified",
+ "source_code_link",
+ "testlink",
+ }
+)
+
+
+class DocsDeltaError(ValueError):
+ """A user-actionable input or report-generation error."""
+
+
+@dataclass(frozen=True)
+class NeedChange:
+ """A Need identified by ID, with whichever before/after versions exist.
+
+ An absent baseline means the Need was added; an absent current value means
+ it was removed. When both versions exist, ``changed_fields`` reports only
+ differences that are meaningful under the comparison's volatile-field
+ policy.
+ """
+
+ need_id: str
+ baseline: JsonObject | None
+ current: JsonObject | None
+
+ @property
+ def changed_fields(self) -> tuple[str, ...]:
+ """Return changed non-volatile fields in deterministic order."""
+
+ if self.baseline is None or self.current is None:
+ return ()
+ names = set(self.baseline) | set(self.current)
+ return tuple(
+ sorted(
+ name
+ for name in names
+ if name not in VOLATILE_NEED_FIELDS
+ and not _json_values_equal(
+ self.baseline.get(name, MISSING),
+ self.current.get(name, MISSING),
+ )
+ )
+ )
+
+
+@dataclass(frozen=True)
+class NeedComparison:
+ """Added, removed, modified, and unchanged counts for Need inventories.
+
+ Changed entries are stored in sorted-ID order by :func:`compare_needs`,
+ which makes both the Markdown output and its review order reproducible.
+ """
+
+ added: tuple[NeedChange, ...]
+ removed: tuple[NeedChange, ...]
+ modified: tuple[NeedChange, ...]
+ unchanged_count: int
+
+
+@dataclass(frozen=True)
+class PageChange:
+ """One rendered HTML path that was added, removed, or modified."""
+
+ path: str
+
+
+@dataclass(frozen=True)
+class PageComparison:
+ """Added, removed, modified, and unchanged counts for rendered pages.
+
+ A page's relative path determines its identity; a rename therefore appears
+ as one removal and one addition rather than a guessed rename relationship.
+ """
+
+ added: tuple[PageChange, ...]
+ removed: tuple[PageChange, ...]
+ modified: tuple[PageChange, ...]
+ unchanged_count: int
+
+
+def _json_values_equal(left: object, right: object) -> bool:
+ """Compare JSON-shaped values consistently and distinguish missing keys.
+
+ Python considers ``True == 1``. Serializing the values keeps those Need
+ field changes visible, while the private sentinel separates a missing
+ field from an explicit JSON ``null``.
+ """
+
+ if left is MISSING or right is MISSING:
+ return left is right
+ return json.dumps(left, sort_keys=True, ensure_ascii=False) == json.dumps(
+ right, sort_keys=True, ensure_ascii=False
+ )
+
+
+def _flatten_needs(path: Path) -> dict[str, JsonObject]:
+ """Flatten versioned Sphinx-Needs output into one ID-to-record mapping.
+
+ Sphinx-Needs exports a ``versions`` map, with each version holding its own
+ ``needs`` map. Version names are not part of a Need's identity for this
+ report, so entries from all versions are combined by ID; if an ID occurs
+ more than once, the last entry in JSON iteration order supplies its record.
+ Unexpectedly shaped version containers are skipped; malformed individual
+ Need entries fail with a useful error rather than silently disappearing.
+ """
+
+ try:
+ payload = json.loads(path.read_text(encoding="utf-8"))
+ except (OSError, UnicodeDecodeError, json.JSONDecodeError) as exc:
+ raise DocsDeltaError(f"cannot read needs JSON {path}: {exc}") from exc
+
+ if not isinstance(payload, dict) or not isinstance(payload.get("versions"), dict):
+ raise DocsDeltaError(f"needs JSON has no versions map: {path}")
+
+ versions = cast(dict[str, object], payload["versions"])
+ result: dict[str, JsonObject] = {}
+ for version_value in versions.values():
+ if not isinstance(version_value, dict):
+ continue
+ version = cast(dict[str, object], version_value)
+ needs_value = version.get("needs")
+ if not isinstance(needs_value, dict):
+ continue
+ needs = cast(dict[object, object], needs_value)
+ for need_id, need in needs.items():
+ if not isinstance(need_id, str) or not isinstance(need, dict):
+ raise DocsDeltaError(f"invalid Need entry in {path}")
+ result[need_id] = cast(JsonObject, need)
+ return result
+
+
+def load_needs(directory: Path) -> dict[str, JsonObject]:
+ """Load the root ``needs.json`` inventory from a rendered docs directory."""
+
+ needs_path = directory / "needs.json"
+ if not needs_path.is_file():
+ raise DocsDeltaError(f"needs JSON is missing: {needs_path}")
+ return _flatten_needs(needs_path)
+
+
+def compare_needs(baseline: NeedMap, current: NeedMap) -> NeedComparison:
+ """Compare Needs by ID, ignoring only explicitly volatile fields.
+
+ A field added or removed from a record counts as a modification. All
+ result groups are sorted by Need ID so report order does not depend on JSON
+ object insertion order.
+ """
+
+ added: list[NeedChange] = []
+ removed: list[NeedChange] = []
+ modified: list[NeedChange] = []
+ unchanged_count = 0
+
+ for need_id in sorted(set(baseline) | set(current)):
+ old = baseline.get(need_id)
+ new = current.get(need_id)
+ change = NeedChange(need_id, old, new)
+ if old is None:
+ added.append(change)
+ elif new is None:
+ removed.append(change)
+ elif change.changed_fields:
+ modified.append(change)
+ else:
+ unchanged_count += 1
+
+ return NeedComparison(
+ added=tuple(added),
+ removed=tuple(removed),
+ modified=tuple(modified),
+ unchanged_count=unchanged_count,
+ )
diff --git a/tools/docs_delta/github.py b/tools/docs_delta/github.py
new file mode 100644
index 000000000..66a3f8ddd
--- /dev/null
+++ b/tools/docs_delta/github.py
@@ -0,0 +1,369 @@
+# *******************************************************************************
+# Copyright (c) 2026 Contributors to the Eclipse Foundation
+#
+# See the NOTICE file(s) distributed with this work for additional
+# information regarding copyright ownership.
+#
+# This program and the accompanying materials are made available under the
+# terms of the Apache License Version 2.0 which is available at
+# https://www.apache.org/licenses/LICENSE-2.0
+#
+# SPDX-License-Identifier: Apache-2.0
+# *******************************************************************************
+"""GitHub Actions metadata and published gh-pages baseline resolution."""
+
+from __future__ import annotations
+
+import argparse
+import io
+import json
+import shutil
+import subprocess
+import tarfile
+import tempfile
+from collections.abc import Iterator, Mapping
+from contextlib import contextmanager
+from dataclasses import dataclass
+from pathlib import Path
+from typing import cast
+from urllib.parse import quote
+
+from .comparison import DocsDeltaError
+
+
+@dataclass(frozen=True)
+class GithubPullRequest:
+ """The source revision that a PR documentation baseline must represent.
+
+ ``base_ref`` selects the published documentation directory and ``base_sha``
+ identifies the commit that directory must contain. These are separate
+ because gh-pages may publish the branch after the PR event was created.
+ """
+
+ base_ref: str
+ base_sha: str
+ number: str | None
+
+
+def workspace_path(environ: Mapping[str, str], relative_path: str) -> Path:
+ """Resolve a workflow-relative path, with a cwd-based local fallback."""
+ workspace = environ.get("GITHUB_WORKSPACE")
+ if workspace:
+ return Path(workspace) / relative_path
+ return Path(relative_path)
+
+
+def github_pull_request(environ: Mapping[str, str]) -> GithubPullRequest | None:
+ """Read PR base ref/SHA from Actions metadata, if this is a PR event.
+
+ The event payload is authoritative for PR identity. The dedicated base
+ variables take precedence when present, since they describe the checkout
+ context in which the workflow is running.
+ """
+
+ event_name = environ.get("GITHUB_EVENT_NAME")
+ event_path = environ.get("GITHUB_EVENT_PATH")
+ if event_name != "pull_request":
+ return None
+ if not event_path:
+ raise DocsDeltaError(
+ "GITHUB_EVENT_PATH is required to resolve a pull request baseline"
+ )
+
+ try:
+ payload = json.loads(Path(event_path).read_text(encoding="utf-8"))
+ except (OSError, UnicodeDecodeError, json.JSONDecodeError) as exc:
+ raise DocsDeltaError(
+ f"cannot read GitHub event payload {event_path}: {exc}"
+ ) from exc
+ if not isinstance(payload, dict):
+ raise DocsDeltaError(f"GitHub event payload is not an object: {event_path}")
+
+ event_payload = cast(dict[str, object], payload)
+ pull_request_value = event_payload.get("pull_request")
+ if not isinstance(pull_request_value, dict):
+ return None
+ pull_request = cast(dict[str, object], pull_request_value)
+ base_value = pull_request.get("base")
+ if not isinstance(base_value, dict):
+ return None
+ base = cast(dict[str, object], base_value)
+
+ base_ref = environ.get("GITHUB_BASE_REF") or base.get("ref")
+ base_sha = environ.get("GITHUB_BASE_SHA") or base.get("sha")
+ if not isinstance(base_ref, str) or not base_ref:
+ return None
+ if not isinstance(base_sha, str) or not base_sha:
+ return None
+
+ number: object = pull_request.get("number")
+ if not isinstance(number, int | str) or isinstance(number, bool):
+ number = None
+ else:
+ number = str(number)
+ return GithubPullRequest(base_ref=base_ref, base_sha=base_sha, number=number)
+
+
+def _find_published_baseline_commit(
+ gh_pages_dir: Path, pull_request: GithubPullRequest
+) -> tuple[str | None, str]:
+ """Find the newest published base-branch tree containing the PR base SHA.
+
+ The docs publishing workflow updates a branch directory on ``gh-pages``
+ only after a successful docs build. Looking up ``needs.json`` history for
+ the base branch and checking its embedded source SHA avoids selecting a
+ previous successful build when the current base commit has not been
+ published. ``git log`` returns newest commits first, so the first match is
+ the latest publication that represents this exact source revision.
+ """
+
+ if not gh_pages_dir.is_dir():
+ return None, f"gh-pages checkout is missing: {gh_pages_dir}"
+
+ try:
+ git_check = subprocess.run(
+ ["git", "-C", str(gh_pages_dir), "rev-parse", "--git-dir"],
+ check=False,
+ capture_output=True,
+ text=True,
+ )
+ except OSError as exc:
+ return None, f"cannot inspect gh-pages checkout {gh_pages_dir}: {exc}"
+ if git_check.returncode != 0:
+ return None, f"gh-pages checkout is not a Git repository: {gh_pages_dir}"
+
+ # The Needs export carries the source revision that produced a published
+ # docs tree. Searching this file's history also avoids treating an older
+ # successful publication as the baseline for a newer, unpublished commit.
+ needs_path = f"{pull_request.base_ref}/needs.json"
+ try:
+ result = subprocess.run(
+ [
+ "git",
+ "-C",
+ str(gh_pages_dir),
+ "log",
+ "--all",
+ "--full-history",
+ "--format=%H",
+ "--",
+ needs_path,
+ ],
+ check=False,
+ capture_output=True,
+ text=True,
+ )
+ except OSError as exc:
+ return None, f"cannot search gh-pages history: {exc}"
+ if result.returncode != 0:
+ detail = result.stderr.strip() or "git log failed"
+ return None, f"cannot search gh-pages history: {detail}"
+
+ candidate_commits = [
+ line.strip() for line in result.stdout.splitlines() if line.strip()
+ ]
+ # ``git log`` is newest-first; stop at the most recent publication that
+ # proves it was built from the PR's exact base SHA.
+ for commit in candidate_commits:
+ try:
+ needs = subprocess.run(
+ [
+ "git",
+ "-C",
+ str(gh_pages_dir),
+ "show",
+ f"{commit}:{needs_path}",
+ ],
+ check=False,
+ capture_output=True,
+ )
+ except OSError as exc:
+ return None, f"cannot inspect published needs JSON: {exc}"
+ if needs.returncode == 0 and pull_request.base_sha.encode() in needs.stdout:
+ return commit, ""
+
+ return (
+ None,
+ f"no published {pull_request.base_ref} baseline contains "
+ f"source commit {pull_request.base_sha}",
+ )
+
+
+def extract_git_tree(
+ gh_pages_dir: Path, commit: str, tree_path: str, destination: Path
+) -> None:
+ """Copy a docs subtree from a commit into a temporary directory.
+
+ ``git archive`` reads the selected immutable tree without checking out or
+ disturbing the shared gh-pages worktree. Members are copied explicitly
+ instead of using ``extractall`` so unexpected paths or special files in
+ the archive cannot escape the temporary baseline directory.
+ """
+
+ try:
+ result = subprocess.run(
+ [
+ "git",
+ "-C",
+ str(gh_pages_dir),
+ "archive",
+ "--format=tar",
+ f"{commit}:{tree_path}",
+ ],
+ check=False,
+ capture_output=True,
+ )
+ except OSError as exc:
+ raise DocsDeltaError(f"cannot extract gh-pages baseline: {exc}") from exc
+ if result.returncode != 0:
+ detail = result.stderr.decode(errors="replace").strip() or "git archive failed"
+ raise DocsDeltaError(f"cannot extract gh-pages baseline: {detail}")
+
+ destination.mkdir(parents=True, exist_ok=True)
+ try:
+ with tarfile.open(fileobj=io.BytesIO(result.stdout), mode="r:") as archive:
+ for member in archive.getmembers():
+ relative = Path(member.name)
+ if relative.is_absolute() or ".." in relative.parts:
+ raise DocsDeltaError(
+ f"gh-pages archive contains unsafe path: {member.name}"
+ )
+ target = destination / relative
+ if member.isdir():
+ target.mkdir(parents=True, exist_ok=True)
+ elif member.isfile():
+ target.parent.mkdir(parents=True, exist_ok=True)
+ source = archive.extractfile(member)
+ if source is None:
+ raise DocsDeltaError(
+ f"cannot read gh-pages archive member: {member.name}"
+ )
+ with source, target.open("wb") as output:
+ shutil.copyfileobj(source, output)
+ else:
+ raise DocsDeltaError(
+ f"unsupported gh-pages archive member: {member.name}"
+ )
+ except (OSError, tarfile.TarError) as exc:
+ raise DocsDeltaError(f"cannot extract gh-pages baseline: {exc}") from exc
+
+
+@contextmanager
+def resolved_baseline(
+ args: argparse.Namespace,
+ github_pull_request: GithubPullRequest | None,
+ environ: Mapping[str, str],
+) -> Iterator[tuple[Path | None, str]]:
+ """Yield a baseline directory, or ``None`` plus a reportable reason.
+
+ Directory mode is useful for local comparisons and explicit artifacts.
+ PR Actions defaults to gh-pages mode, where the extracted directory only
+ lives for the duration of this context manager. A baseline lookup failure
+ is data for the report; malformed CLI choices remain hard errors.
+ """
+
+ if args.baseline_mode is None:
+ # Local callers usually provide a directory. PR Actions has no such
+ # artifact, so use the published branch history by default.
+ mode = (
+ "directory"
+ if args.baseline_dir is not None or github_pull_request is None
+ else "gh-pages"
+ )
+ else:
+ mode = args.baseline_mode
+
+ if mode == "directory":
+ if args.baseline_dir is None:
+ raise DocsDeltaError(
+ "--baseline-dir is required outside GitHub pull request mode"
+ )
+ yield args.baseline_dir, ""
+ return
+
+ if args.baseline_dir is not None:
+ raise DocsDeltaError(
+ "--baseline-dir cannot be used with --baseline-mode gh-pages"
+ )
+ if github_pull_request is None:
+ raise DocsDeltaError(
+ "gh-pages baseline mode requires a GitHub pull request event"
+ )
+
+ gh_pages_dir = args.gh_pages_dir or workspace_path(environ, ".docs-baseline")
+ commit, reason = _find_published_baseline_commit(gh_pages_dir, github_pull_request)
+ if commit is None:
+ yield None, reason
+ return
+
+ with tempfile.TemporaryDirectory(prefix="docs-delta-baseline-") as temporary_dir:
+ baseline_dir = Path(temporary_dir)
+ try:
+ extract_git_tree(
+ gh_pages_dir,
+ commit,
+ github_pull_request.base_ref,
+ baseline_dir,
+ )
+ except DocsDeltaError as exc:
+ yield None, str(exc)
+ return
+ yield baseline_dir, ""
+
+
+def _github_url_defaults(
+ environ: Mapping[str, str],
+ github_pull_request: GithubPullRequest | None = None,
+) -> tuple[str | None, str | None]:
+ """Derive base and PR preview URLs from the repository and Actions context.
+
+ PR previews conventionally live under ``pr-N``. Branch refs are URL-quoted
+ as one segment so a feature branch containing ``/`` cannot be mistaken for
+ a nested path under the Pages site.
+ """
+
+ repository = environ.get("GITHUB_REPOSITORY", "")
+ if "/" not in repository:
+ return None, None
+ owner, name = repository.split("/", 1)
+ if not owner or not name:
+ return None, None
+ pages_root = environ.get("GITHUB_PAGES_URL") or f"https://{owner}.github.io/{name}"
+ base_ref = (
+ github_pull_request.base_ref
+ if github_pull_request
+ else (environ.get("GITHUB_BASE_REF") or "main")
+ )
+ if github_pull_request and github_pull_request.number:
+ pr_ref = f"pr-{github_pull_request.number}"
+ else:
+ pr_ref = (
+ environ.get("GITHUB_HEAD_REF") or environ.get("GITHUB_REF_NAME") or base_ref
+ )
+
+ def pages_ref_url(ref: str) -> str:
+ # A branch name is one URL path segment. In particular, a slash in a
+ # feature branch must not become a second documentation path segment.
+ return pages_root.rstrip("/") + "/" + quote(ref, safe="-._~")
+
+ return (
+ pages_ref_url(base_ref),
+ pages_ref_url(pr_ref),
+ )
+
+
+def resolve_urls(
+ args: argparse.Namespace,
+ github_pull_request: GithubPullRequest | None,
+ environ: Mapping[str, str],
+) -> tuple[str, str]:
+ """Use explicit URLs where supplied, otherwise require usable CI defaults."""
+ defaults = _github_url_defaults(environ, github_pull_request)
+ base_url = args.base_url or defaults[0]
+ pr_url = args.pr_url or defaults[1]
+ if not base_url or not pr_url:
+ raise DocsDeltaError(
+ "documentation URLs are required; provide --base-url and --pr-url "
+ "or run in GitHub Actions with GITHUB_REPOSITORY set"
+ )
+ return base_url, pr_url
diff --git a/tools/docs_delta/pyproject.toml b/tools/docs_delta/pyproject.toml
new file mode 100644
index 000000000..cfdd71780
--- /dev/null
+++ b/tools/docs_delta/pyproject.toml
@@ -0,0 +1,26 @@
+[project]
+name = "score-docs-delta"
+version = "0.1.0"
+description = "Compare rendered S-CORE documentation builds"
+requires-python = ">=3.12"
+dependencies = []
+
+[project.scripts]
+docs-delta = "docs_delta.cli:main"
+
+[build-system]
+requires = ["hatchling"]
+build-backend = "hatchling.build"
+
+[tool.hatch.build.targets.wheel]
+only-include = [
+ "__init__.py",
+ "__main__.py",
+ "cli.py",
+ "comparison.py",
+ "github.py",
+ "rendered_html.py",
+ "report.py",
+]
+sources = { "" = "docs_delta" }
+dev-mode-dirs = [".."]
diff --git a/tools/docs_delta/rendered_html.py b/tools/docs_delta/rendered_html.py
new file mode 100644
index 000000000..7dec98f98
--- /dev/null
+++ b/tools/docs_delta/rendered_html.py
@@ -0,0 +1,154 @@
+# *******************************************************************************
+# Copyright (c) 2026 Contributors to the Eclipse Foundation
+#
+# See the NOTICE file(s) distributed with this work for additional
+# information regarding copyright ownership.
+#
+# This program and the accompanying materials are made available under the
+# terms of the Apache License Version 2.0 which is available at
+# https://www.apache.org/licenses/LICENSE-2.0
+#
+# SPDX-License-Identifier: Apache-2.0
+# *******************************************************************************
+"""Normalization and comparison of rendered HTML pages."""
+
+from __future__ import annotations
+
+import re
+from pathlib import Path
+
+from .comparison import DocsDeltaError, PageChange, PageComparison
+
+_META_TAG = re.compile(r"]*>", re.IGNORECASE)
+_META_ATTRIBUTE = re.compile(
+ r"(?:name|property|itemprop)\s*=\s*([\"'])(.*?)\1", re.IGNORECASE
+)
+_HTML_COMMENT = re.compile(r"", re.IGNORECASE | re.DOTALL)
+_VOLATILE_ATTRIBUTE = re.compile(
+ r"\s+data-(?:build|generated|timestamp|last-modified)(?:-[\w-]+)?\s*=\s*(?:\"[^\"]*\"|'[^']*'|[^\s>]+)",
+ re.IGNORECASE,
+)
+# PR documentation builds run on GitHub's synthetic merge commit, while the
+# published baseline was built from the base commit. Keep source paths and line
+# numbers comparable, but ignore the revision segment in generated blob links.
+_GITHUB_BLOB_COMMIT = re.compile(r"(?<=/blob/)[0-9a-f]{40}(?=/)", re.IGNORECASE)
+# Sphinx-Needs derives these wrapper IDs from rendered content, including the
+# source-link revision above. Need IDs and visible content remain compared.
+_NEED_CONTAINER_ID = re.compile(r"\bSNCB-[0-9a-f]{8}\b", re.IGNORECASE)
+_VOLATILE_META_NAMES = frozenset(
+ {
+ "build-date",
+ "build_date",
+ "created",
+ "date",
+ "generated",
+ "generated-at",
+ "generated_at",
+ "generator",
+ "last-modified",
+ "last_modified",
+ "timestamp",
+ }
+)
+_VOLATILE_COMMENT_WORDS = re.compile(
+ r"\b(?:build|built|generated|generator|last\s+updated|timestamp)\b",
+ re.IGNORECASE,
+)
+
+
+def _html_metadata_name(tag: str) -> str | None:
+ for match in _META_ATTRIBUTE.finditer(tag):
+ name = match.group(2).lower()
+ if name in _VOLATILE_META_NAMES:
+ return name
+ return None
+
+
+def normalize_html(content: str) -> str:
+ """Remove known build-to-build noise while preserving page changes.
+
+ The rules are intentionally narrow. Build dates, generated attributes,
+ GitHub source revision hashes, and Sphinx-Needs wrapper IDs vary between
+ otherwise identical builds. Visible text, links, element structure, and
+ other attributes remain significant so a real rendered change is reported.
+ Keep each rule tied to a known source of nondeterminism; broad HTML cleanup
+ could hide a user-facing regression.
+ """
+
+ # Normalize line endings first so the later line cleanup behaves the same
+ # for artifacts produced on different operating systems.
+ normalized = content.replace("\r\n", "\n").replace("\r", "\n")
+
+ def remove_meta(match: re.Match[str]) -> str:
+ return "" if _html_metadata_name(match.group(0)) else match.group(0)
+
+ # Only discard metadata tags whose name is on the explicit volatile list;
+ # unrelated meta tags can affect consumers and remain part of the diff.
+ normalized = _META_TAG.sub(remove_meta, normalized)
+
+ def remove_comment(match: re.Match[str]) -> str:
+ return "" if _VOLATILE_COMMENT_WORDS.search(match.group(0)) else match.group(0)
+
+ # Generated comments are noisy, but ordinary HTML comments can document
+ # page structure and therefore stay comparison-visible.
+ normalized = _HTML_COMMENT.sub(remove_comment, normalized)
+ normalized = _VOLATILE_ATTRIBUTE.sub("", normalized)
+ normalized = _GITHUB_BLOB_COMMIT.sub("", normalized)
+ normalized = _NEED_CONTAINER_ID.sub("SNCB-", normalized)
+ # Trailing whitespace and blank padding do not change rendered pages.
+ normalized = "\n".join(line.rstrip() for line in normalized.splitlines())
+ return normalized.strip()
+
+
+def _html_files(directory: Path) -> dict[str, str]:
+ """Read and normalize every HTML page below ``directory``.
+
+ POSIX separators make paths stable across operating systems and suitable
+ for URLs in the report. Sorting traversal also avoids filesystem-order
+ leaking into the report's comparison order.
+ """
+
+ if not directory.is_dir():
+ raise DocsDeltaError(f"documentation directory is missing: {directory}")
+ result: dict[str, str] = {}
+ for path in sorted(directory.rglob("*.html")):
+ if path.is_file():
+ relative = path.relative_to(directory).as_posix()
+ try:
+ result[relative] = normalize_html(path.read_text(encoding="utf-8"))
+ except (OSError, UnicodeDecodeError) as exc:
+ raise DocsDeltaError(
+ f"cannot read rendered page {path}: {exc}"
+ ) from exc
+ return result
+
+
+def compare_html(baseline_dir: Path, current_dir: Path) -> PageComparison:
+ """Compare pages by relative path and normalized rendered content."""
+
+ baseline = _html_files(baseline_dir)
+ current = _html_files(current_dir)
+ added: list[PageChange] = []
+ removed: list[PageChange] = []
+ modified: list[PageChange] = []
+ unchanged_count = 0
+
+ for path in sorted(set(baseline) | set(current)):
+ old = baseline.get(path)
+ new = current.get(path)
+ change = PageChange(path)
+ if old is None:
+ added.append(change)
+ elif new is None:
+ removed.append(change)
+ elif old != new:
+ modified.append(change)
+ else:
+ unchanged_count += 1
+
+ return PageComparison(
+ added=tuple(added),
+ removed=tuple(removed),
+ modified=tuple(modified),
+ unchanged_count=unchanged_count,
+ )
diff --git a/tools/docs_delta/report.py b/tools/docs_delta/report.py
new file mode 100644
index 000000000..e86f1e0db
--- /dev/null
+++ b/tools/docs_delta/report.py
@@ -0,0 +1,378 @@
+# *******************************************************************************
+# Copyright (c) 2026 Contributors to the Eclipse Foundation
+#
+# See the NOTICE file(s) distributed with this work for additional
+# information regarding copyright ownership.
+#
+# This program and the accompanying materials are made available under the
+# terms of the Apache License Version 2.0 which is available at
+# https://www.apache.org/licenses/LICENSE-2.0
+#
+# SPDX-License-Identifier: Apache-2.0
+# *******************************************************************************
+"""Markdown formatting for documentation delta reports."""
+
+from __future__ import annotations
+
+import json
+import os
+import tempfile
+from collections.abc import Mapping, Sequence
+from pathlib import Path
+from typing import Protocol
+from urllib.parse import quote
+
+from .comparison import MISSING, NeedChange, NeedComparison
+from .rendered_html import PageChange, PageComparison
+
+DETAIL_LIMIT = 15
+MAX_RENDERED_VALUE_LENGTH = 600
+
+
+class NeedFormatter(Protocol):
+ """Format a Need change for a Markdown report section."""
+
+ def __call__(
+ self,
+ change: NeedChange,
+ *,
+ base_url: str,
+ pr_url: str,
+ include_diff: bool = False,
+ ) -> list[str]: ...
+
+
+class PageFormatter(Protocol):
+ """Format a rendered page change for a Markdown report section."""
+
+ def __call__(
+ self,
+ change: PageChange,
+ *,
+ base_url: str,
+ pr_url: str,
+ kind: str,
+ ) -> str: ...
+
+
+def _url_with_path(base_url: str, relative_path: str, anchor: str | None = None) -> str:
+ """Append an encoded documentation path and optional fragment to a URL.
+
+ Encode path segments independently: slashes in a page path remain
+ separators, while characters inside each segment cannot alter the URL
+ structure.
+ """
+ url = (
+ base_url.rstrip("/")
+ + "/"
+ + "/".join(
+ quote(part, safe="-._~") for part in relative_path.strip("/").split("/")
+ )
+ )
+ if anchor:
+ url += "#" + quote(anchor, safe="-._~")
+ return url
+
+
+def _need_url(base_url: str, need: Mapping[str, object]) -> str | None:
+ """Build the page URL for a Need, returning ``None`` if it has no docname."""
+ docname = need.get("docname")
+ if not isinstance(docname, str) or not docname.strip():
+ return None
+ rendered_name = docname.strip("/")
+ if not rendered_name.endswith(".html"):
+ rendered_name += ".html"
+ return _url_with_path(base_url, rendered_name)
+
+
+def need_link(base_url: str, need_id: str, need: Mapping[str, object]) -> str | None:
+ """Link directly to a Need anchor when its exported page is known."""
+ page_url = _need_url(base_url, need)
+ if page_url is None:
+ return None
+ return page_url + "#" + quote(need_id, safe="-._~")
+
+
+def _markdown_link(label: str, url: str | None) -> str:
+ if url is None:
+ return f"`{label}`"
+ safe_label = label.replace("[", "\\[").replace("]", "\\]")
+ return f"[{safe_label}]({url})"
+
+
+def _display_name(need: Mapping[str, object]) -> str:
+ for field in ("title", "name"):
+ value = need.get(field)
+ if isinstance(value, str) and value:
+ return value
+ return ""
+
+
+def _format_value(value: object) -> str:
+ """Render one JSON field value compactly and safely for Markdown inline code."""
+ if value is MISSING:
+ return "(missing)"
+ rendered = json.dumps(
+ value, sort_keys=True, ensure_ascii=False, separators=(",", ":")
+ )
+ if len(rendered) > MAX_RENDERED_VALUE_LENGTH:
+ rendered = rendered[:MAX_RENDERED_VALUE_LENGTH] + "…"
+ return rendered.replace("`", "\\`").replace("\n", " ")
+
+
+def _format_need_entry(
+ change: NeedChange, *, base_url: str, pr_url: str, include_diff: bool = False
+) -> list[str]:
+ """Render one Need row, with old/new links and optional field details."""
+ need = change.current or change.baseline
+ assert need is not None
+ links = []
+ old_link = (
+ need_link(base_url, change.need_id, change.baseline)
+ if change.baseline
+ else None
+ )
+ new_link = (
+ need_link(pr_url, change.need_id, change.current) if change.current else None
+ )
+ if old_link:
+ links.append(_markdown_link("old", old_link))
+ if new_link:
+ links.append(_markdown_link("new", new_link))
+ link_text = " / ".join(links)
+ title = _display_name(need)
+ suffix = f" — {title}" if title else ""
+ line = f"- `{change.need_id}`{suffix}"
+ if link_text:
+ line += f" ({link_text})"
+ result = [line]
+ if include_diff:
+ assert change.baseline is not None and change.current is not None
+ for field in change.changed_fields:
+ result.append(
+ f" - `{field}`: {_format_value(change.baseline.get(field, MISSING))} "
+ f"→ {_format_value(change.current.get(field, MISSING))}"
+ )
+ return result
+
+
+def _format_page_entry(
+ change: PageChange, *, base_url: str, pr_url: str, kind: str
+) -> str:
+ """Render a page row with links appropriate to its change category."""
+ old = _url_with_path(base_url, change.path) if kind != "added" else None
+ new = _url_with_path(pr_url, change.path) if kind != "removed" else None
+ links = []
+ if old:
+ links.append(_markdown_link("old", old))
+ if new:
+ links.append(_markdown_link("new", new))
+ return f"- `{change.path}` ({' / '.join(links)})"
+
+
+def _need_section(
+ lines: list[str],
+ title: str,
+ entries: Sequence[NeedChange],
+ formatter: NeedFormatter,
+ *,
+ base_url: str,
+ pr_url: str,
+ kind: str = "",
+) -> None:
+ """Append a Need section, suppressing field-by-field detail for large groups.
+
+ The entries themselves are retained so reviewers can still see which Needs
+ changed. Modified-field values are the part that can make a large comment
+ unwieldy, so those are omitted once the group exceeds ``DETAIL_LIMIT``.
+ """
+ if not entries:
+ return
+ lines.extend([f"### {title} ({len(entries)})", ""])
+ include_field_diff = kind == "modified" and len(entries) <= DETAIL_LIMIT
+ if len(entries) > DETAIL_LIMIT:
+ lines.extend([f"{len(entries)} entries changed; field details omitted.", ""])
+ for entry in entries:
+ lines.extend(
+ formatter(
+ entry,
+ base_url=base_url,
+ pr_url=pr_url,
+ include_diff=include_field_diff,
+ )
+ )
+ lines.append("")
+
+
+def _page_section(
+ lines: list[str],
+ title: str,
+ entries: Sequence[PageChange],
+ formatter: PageFormatter,
+ *,
+ base_url: str,
+ pr_url: str,
+ kind: str,
+) -> None:
+ """Append a page section; every entry remains directly reviewable."""
+ if not entries:
+ return
+ lines.extend([f"### {title} ({len(entries)})", ""])
+ for entry in entries:
+ lines.append(formatter(entry, base_url=base_url, pr_url=pr_url, kind=kind))
+ lines.append("")
+
+
+def render_report(
+ needs: NeedComparison,
+ pages: PageComparison,
+ *,
+ base_url: str,
+ pr_url: str,
+) -> str:
+ """Render a deterministic, link-rich Markdown summary of both comparisons.
+
+ Small change sets are expanded for immediate review. Larger sets are
+ wrapped in GitHub's collapsible-details markup so the comment stays
+ scannable while preserving every changed path or Need in the report.
+ """
+
+ lines = [
+ "# Documentation delta",
+ "",
+ "Documentation preview for this pull request: "
+ f"{_markdown_link('open preview', pr_url)} · "
+ f"Baseline: {_markdown_link('open baseline', base_url)}",
+ "",
+ "## Summary",
+ "",
+ f"- Needs: {len(needs.added)} added, {len(needs.removed)} removed, "
+ f"{len(needs.modified)} modified, {needs.unchanged_count} unchanged",
+ f"- Rendered pages: {len(pages.added)} added, {len(pages.removed)} removed, "
+ f"{len(pages.modified)} modified, {pages.unchanged_count} unchanged",
+ ]
+
+ need_change_count = len(needs.added) + len(needs.removed) + len(needs.modified)
+ if need_change_count:
+ need_lines: list[str] = []
+ _need_section(
+ need_lines,
+ "Added",
+ needs.added,
+ _format_need_entry,
+ base_url=base_url,
+ pr_url=pr_url,
+ kind="added",
+ )
+ _need_section(
+ need_lines,
+ "Removed",
+ needs.removed,
+ _format_need_entry,
+ base_url=base_url,
+ pr_url=pr_url,
+ kind="removed",
+ )
+ _need_section(
+ need_lines,
+ "Modified",
+ needs.modified,
+ _format_need_entry,
+ base_url=base_url,
+ pr_url=pr_url,
+ kind="modified",
+ )
+ if need_change_count > DETAIL_LIMIT:
+ lines.extend(
+ [
+ "",
+ "",
+ f"Need changes ({need_change_count})
",
+ "",
+ *need_lines,
+ " ",
+ ]
+ )
+ else:
+ lines.extend(["", "## Need changes", "", *need_lines])
+
+ page_change_count = len(pages.added) + len(pages.removed) + len(pages.modified)
+ if page_change_count:
+ page_lines: list[str] = []
+ _page_section(
+ page_lines,
+ "Added",
+ pages.added,
+ _format_page_entry,
+ base_url=base_url,
+ pr_url=pr_url,
+ kind="added",
+ )
+ _page_section(
+ page_lines,
+ "Removed",
+ pages.removed,
+ _format_page_entry,
+ base_url=base_url,
+ pr_url=pr_url,
+ kind="removed",
+ )
+ _page_section(
+ page_lines,
+ "Modified",
+ pages.modified,
+ _format_page_entry,
+ base_url=base_url,
+ pr_url=pr_url,
+ kind="modified",
+ )
+ if page_change_count > DETAIL_LIMIT:
+ # Keep the complete page list available, but collapsed by default
+ # so a broad Sphinx rebuild does not bury the useful summary.
+ lines.extend(
+ [
+ "",
+ "",
+ f"Page changes ({page_change_count})
",
+ "",
+ *page_lines,
+ " ",
+ ]
+ )
+ else:
+ lines.extend(["", "## Page changes", "", *page_lines])
+ return "\n".join(lines).rstrip() + "\n"
+
+
+def unavailable_report(reason: str) -> str:
+ """Render the successful, explicit report used when baseline is unavailable."""
+
+ return (
+ "# Documentation delta\n\n"
+ "## Delta unavailable\n\n"
+ f"The documentation baseline is unavailable: {reason}\n"
+ )
+
+
+def write_report(path: Path, content: str) -> None:
+ """Atomically write the report so a failed run never leaves partial Markdown."""
+
+ path.parent.mkdir(parents=True, exist_ok=True)
+ temporary_path: Path | None = None
+ file_descriptor: int | None = None
+ try:
+ file_descriptor, temporary_name = tempfile.mkstemp(
+ dir=path.parent, prefix=f".{path.name}.", suffix=".tmp"
+ )
+ temporary_path = Path(temporary_name)
+ with os.fdopen(file_descriptor, "w", encoding="utf-8") as output:
+ file_descriptor = None
+ output.write(content)
+ output.flush()
+ os.replace(temporary_path, path)
+ temporary_path = None
+ finally:
+ if file_descriptor is not None:
+ os.close(file_descriptor)
+ if temporary_path is not None:
+ temporary_path.unlink(missing_ok=True)
diff --git a/tools/docs_delta/uv.lock b/tools/docs_delta/uv.lock
new file mode 100644
index 000000000..57b2a9873
--- /dev/null
+++ b/tools/docs_delta/uv.lock
@@ -0,0 +1,8 @@
+version = 1
+revision = 3
+requires-python = ">=3.12"
+
+[[package]]
+name = "score-docs-delta"
+version = "0.1.0"
+source = { editable = "." }
diff --git a/tools/docs_delta_main.py b/tools/docs_delta_main.py
new file mode 100644
index 000000000..86745af81
--- /dev/null
+++ b/tools/docs_delta_main.py
@@ -0,0 +1,18 @@
+# *******************************************************************************
+# Copyright (c) 2026 Contributors to the Eclipse Foundation
+#
+# See the NOTICE file(s) distributed with this work for additional
+# information regarding copyright ownership.
+#
+# This program and the accompanying materials are made available under the
+# terms of the Apache License Version 2.0 which is available at
+# https://www.apache.org/licenses/LICENSE-2.0
+#
+# SPDX-License-Identifier: Apache-2.0
+# *******************************************************************************
+"""Bazel launcher for the docs-delta CLI package."""
+
+from tools.docs_delta.cli import main
+
+if __name__ == "__main__": # pragma: no cover - exercised via the Bazel binary
+ raise SystemExit(main())
diff --git a/tools/tests/BUILD b/tools/tests/BUILD
index e84b502a7..38daa623d 100644
--- a/tools/tests/BUILD
+++ b/tools/tests/BUILD
@@ -5,7 +5,7 @@
# information regarding copyright ownership.
#
# This program and the accompanying materials are made available under the
-# terms of the Apache License, Version 2.0 which is available at
+# terms of the Apache License Version 2.0 which is available at
# https://www.apache.org/licenses/LICENSE-2.0
#
# SPDX-License-Identifier: Apache-2.0
@@ -23,3 +23,12 @@ score_pytest(
] + all_requirements,
pytest_config = "//:pyproject.toml",
)
+
+score_pytest(
+ name = "docs_delta_test",
+ srcs = glob(["docs_delta_*_test.py", "docs_delta_test_support.py"]),
+ deps = [
+ "//tools:docs_delta_cli",
+ ] + all_requirements,
+ pytest_config = "//:pyproject.toml",
+)
diff --git a/tools/tests/docs_delta_cli_test.py b/tools/tests/docs_delta_cli_test.py
new file mode 100644
index 000000000..e7c271250
--- /dev/null
+++ b/tools/tests/docs_delta_cli_test.py
@@ -0,0 +1,301 @@
+# *******************************************************************************
+# Copyright (c) 2026 Contributors to the Eclipse Foundation
+#
+# See the NOTICE file(s) distributed with this work for additional
+# information regarding copyright ownership.
+#
+# This program and the accompanying materials are made available under the
+# terms of the Apache License Version 2.0 which is available at
+# https://www.apache.org/licenses/LICENSE-2.0
+#
+# SPDX-License-Identifier: Apache-2.0
+# *******************************************************************************
+"""Tests for local and GitHub Actions command-line behavior."""
+
+from __future__ import annotations
+
+import json
+import subprocess
+from pathlib import Path
+
+import pytest
+
+from tools import docs_delta
+from tools.tests.docs_delta_test_support import git, need, write_needs
+
+
+def test_missing_baseline_writes_unavailable_report(tmp_path: Path) -> None:
+ output = tmp_path / "nested" / "delta.md"
+
+ assert (
+ docs_delta.main(
+ [
+ "--baseline-dir",
+ str(tmp_path / "missing"),
+ "--current-dir",
+ str(tmp_path / "current"),
+ "--base-url",
+ "https://docs.example/main",
+ "--pr-url",
+ "https://docs.example/pr/1",
+ "--output",
+ str(output),
+ ]
+ )
+ == 0
+ )
+ assert "## Delta unavailable" in output.read_text(encoding="utf-8")
+
+
+@pytest.mark.parametrize(
+ ("mode", "extra_args", "message"),
+ [
+ ("directory", [], "--baseline-dir is required"),
+ ("gh-pages", [], "requires a GitHub pull request event"),
+ (
+ "gh-pages",
+ ["--baseline-dir", "somewhere"],
+ "cannot be used with --baseline-mode gh-pages",
+ ),
+ ],
+)
+def test_baseline_mode_rejects_incompatible_options(
+ tmp_path: Path,
+ monkeypatch: pytest.MonkeyPatch,
+ capsys: pytest.CaptureFixture[str],
+ mode: str,
+ extra_args: list[str],
+ message: str,
+) -> None:
+ monkeypatch.delenv("GITHUB_EVENT_NAME", raising=False)
+ monkeypatch.delenv("GITHUB_EVENT_PATH", raising=False)
+
+ result = docs_delta.main(
+ [
+ "--baseline-mode",
+ mode,
+ *extra_args,
+ "--current-dir",
+ str(tmp_path / "current"),
+ "--output",
+ str(tmp_path / "delta.md"),
+ ]
+ )
+
+ assert result == 2
+ assert message in capsys.readouterr().err
+
+
+def test_cli_reports_missing_current_directory_as_error(
+ tmp_path: Path, capsys: pytest.CaptureFixture[str]
+) -> None:
+ baseline = tmp_path / "baseline"
+ baseline.mkdir()
+ write_needs(baseline, {})
+
+ result = docs_delta.main(
+ [
+ "--baseline-dir",
+ str(baseline),
+ "--current-dir",
+ str(tmp_path / "missing-current"),
+ "--base-url",
+ "https://docs.example/main",
+ "--pr-url",
+ "https://docs.example/pr/1",
+ "--output",
+ str(tmp_path / "delta.md"),
+ ]
+ )
+
+ assert result == 2
+ assert "documentation directory is missing" in capsys.readouterr().err
+
+
+def test_cli_writes_complete_fixture_report(tmp_path: Path) -> None:
+ baseline = tmp_path / "baseline"
+ current = tmp_path / "current"
+ baseline.mkdir()
+ current.mkdir()
+ write_needs(
+ baseline,
+ {
+ "same": need("same", "Same", id="same"),
+ "changed": need("guide", "Old title", id="changed"),
+ "removed": need("removed", "Removed", id="removed"),
+ },
+ )
+ write_needs(
+ current,
+ {
+ "same": need("same", "Same", id="same"),
+ "changed": need("guide", "New title", id="changed"),
+ "added": need("new", "Added", id="added"),
+ },
+ )
+ (baseline / "guide.html").write_text("old
\n", encoding="utf-8")
+ (current / "guide.html").write_text("new
\n", encoding="utf-8")
+ (current / "new.html").write_text("new page
\n", encoding="utf-8")
+ output = tmp_path / "delta.md"
+
+ assert (
+ docs_delta.main(
+ [
+ "--baseline-dir",
+ str(baseline),
+ "--current-dir",
+ str(current),
+ "--base-url",
+ "https://docs.example/main",
+ "--pr-url",
+ "https://docs.example/pr/1",
+ "--output",
+ str(output),
+ ]
+ )
+ == 0
+ )
+ report = output.read_text(encoding="utf-8")
+ assert "Needs: 1 added, 1 removed, 1 modified, 1 unchanged" in report
+ assert "Rendered pages: 1 added, 0 removed, 1 modified, 0 unchanged" in report
+ assert '`title`: "Old title" → "New title"' in report
+ assert (
+ "`guide.html` ([old](https://docs.example/main/guide.html) / [new](https://docs.example/pr/1/guide.html))"
+ in report
+ )
+
+
+def test_cli_uses_github_environment_urls_when_options_are_omitted(
+ tmp_path: Path, monkeypatch: pytest.MonkeyPatch
+) -> None:
+ baseline = tmp_path / "baseline"
+ current = tmp_path / "current"
+ baseline.mkdir()
+ current.mkdir()
+ write_needs(baseline, {})
+ write_needs(current, {})
+ monkeypatch.setenv("GITHUB_REPOSITORY", "eclipse-score/docs-as-code")
+ monkeypatch.setenv("GITHUB_BASE_REF", "main")
+ monkeypatch.setenv("GITHUB_HEAD_REF", "feature/docs-delta")
+ output = tmp_path / "delta.md"
+
+ assert (
+ docs_delta.main(
+ [
+ "--baseline-dir",
+ str(baseline),
+ "--current-dir",
+ str(current),
+ "--output",
+ str(output),
+ ]
+ )
+ == 0
+ )
+ report = output.read_text(encoding="utf-8")
+ assert "https://eclipse-score.github.io/docs-as-code/main" in report
+ assert "https://eclipse-score.github.io/docs-as-code/feature%2Fdocs-delta" in report
+
+
+def test_cli_automatically_selects_matching_gh_pages_baseline(
+ tmp_path: Path, monkeypatch: pytest.MonkeyPatch
+) -> None:
+ """PR mode must select the historical publication matching the base SHA."""
+
+ gh_pages = tmp_path / ".docs-baseline"
+ gh_pages.mkdir()
+ subprocess.run(
+ ["git", "init", "--initial-branch=gh-pages", str(gh_pages)],
+ check=True,
+ capture_output=True,
+ text=True,
+ )
+ git(gh_pages, "config", "user.name", "Documentation Delta Test")
+ git(gh_pages, "config", "user.email", "docs-delta@example.invalid")
+
+ base_sha = "0123456789abcdef0123456789abcdef01234567"
+ latest_sha = "fedcba9876543210fedcba9876543210fedcba98"
+ main_docs = gh_pages / "main"
+ main_docs.mkdir()
+ write_needs(
+ main_docs,
+ {
+ "need": need(
+ "guide", "Base publication", id="need", source_code_link=base_sha
+ )
+ },
+ )
+ (main_docs / "guide.html").write_text("Base publication
\n", encoding="utf-8")
+ git(gh_pages, "add", ".")
+ git(gh_pages, "commit", "-m", "Publish base documentation")
+
+ write_needs(
+ main_docs,
+ {
+ "need": need(
+ "guide",
+ "Newer base publication",
+ id="need",
+ source_code_link=base_sha,
+ )
+ },
+ )
+ (main_docs / "guide.html").write_text(
+ "Newer base publication
\n", encoding="utf-8"
+ )
+ git(gh_pages, "add", ".")
+ git(gh_pages, "commit", "-m", "Republish base documentation")
+ write_needs(
+ main_docs,
+ {
+ "need": need(
+ "guide", "Latest publication", id="need", source_code_link=latest_sha
+ )
+ },
+ )
+ (main_docs / "guide.html").write_text(
+ "Latest publication
\n", encoding="utf-8"
+ )
+ git(gh_pages, "add", ".")
+ git(gh_pages, "commit", "-m", "Publish latest documentation")
+
+ current = tmp_path / "docs-artifact"
+ current.mkdir()
+ write_needs(
+ current,
+ {
+ "need": need(
+ "guide",
+ "Newer base publication",
+ id="need",
+ source_code_link="pr-sha",
+ )
+ },
+ )
+ (current / "guide.html").write_text(
+ "Newer base publication
\n", encoding="utf-8"
+ )
+
+ event = tmp_path / "event.json"
+ event.write_text(
+ json.dumps(
+ {
+ "pull_request": {
+ "number": 123,
+ "base": {"ref": "main", "sha": base_sha},
+ }
+ }
+ ),
+ encoding="utf-8",
+ )
+ monkeypatch.setenv("GITHUB_EVENT_NAME", "pull_request")
+ monkeypatch.setenv("GITHUB_EVENT_PATH", str(event))
+ monkeypatch.setenv("GITHUB_WORKSPACE", str(tmp_path))
+ monkeypatch.setenv("GITHUB_REPOSITORY", "eclipse-score/docs-as-code")
+ monkeypatch.setenv("GITHUB_BASE_REF", "main")
+
+ assert docs_delta.main([]) == 0
+ report = (current / "docs-delta.md").read_text(encoding="utf-8")
+ assert "Needs: 0 added, 0 removed, 0 modified, 1 unchanged" in report
+ assert "Rendered pages: 0 added, 0 removed, 0 modified, 1 unchanged" in report
+ assert "https://eclipse-score.github.io/docs-as-code/pr-123" in report
diff --git a/tools/tests/docs_delta_comparison_test.py b/tools/tests/docs_delta_comparison_test.py
new file mode 100644
index 000000000..728c90904
--- /dev/null
+++ b/tools/tests/docs_delta_comparison_test.py
@@ -0,0 +1,123 @@
+# *******************************************************************************
+# Copyright (c) 2026 Contributors to the Eclipse Foundation
+#
+# See the NOTICE file(s) distributed with this work for additional
+# information regarding copyright ownership.
+#
+# This program and the accompanying materials are made available under the
+# terms of the Apache License Version 2.0 which is available at
+# https://www.apache.org/licenses/LICENSE-2.0
+#
+# SPDX-License-Identifier: Apache-2.0
+# *******************************************************************************
+"""Tests for Need inventory loading and comparison."""
+
+from __future__ import annotations
+
+import json
+from pathlib import Path
+
+import pytest
+
+from tools import docs_delta
+from tools.tests.docs_delta_test_support import need
+
+
+def test_compare_needs_categorizes_changes_and_ignores_volatile_fields() -> None:
+ baseline = {
+ "same": need(
+ "same",
+ "Same",
+ id="same",
+ lineno=10,
+ source="old.rst",
+ source_code_link="old-commit",
+ testlink="old-result",
+ ),
+ "changed": need(
+ "changed",
+ "Old",
+ id="changed",
+ lineno=10,
+ source="old.rst",
+ source_code_link="old-commit",
+ testlink="old-result",
+ ),
+ "removed": need("removed", "Removed", id="removed"),
+ }
+ current = {
+ "same": need(
+ "same",
+ "Same",
+ id="same",
+ lineno=999,
+ source="new.rst",
+ source_code_link="new-commit",
+ testlink="new-result",
+ ),
+ "changed": need(
+ "changed",
+ "New",
+ id="changed",
+ lineno=20,
+ source="new.rst",
+ source_code_link="new-commit",
+ testlink="new-result",
+ ),
+ "added": need("added", "Added", id="added"),
+ }
+
+ result = docs_delta.compare_needs(baseline, current)
+
+ assert [entry.need_id for entry in result.added] == ["added"]
+ assert [entry.need_id for entry in result.removed] == ["removed"]
+ assert [entry.need_id for entry in result.modified] == ["changed"]
+ assert result.modified[0].changed_fields == ("title",)
+ assert result.unchanged_count == 1
+
+
+def test_need_comparison_distinguishes_missing_null_and_boolean_number() -> None:
+ baseline: dict[str, dict[str, object]] = {
+ "need": {"id": "need", "optional": None, "enabled": True}
+ }
+ current: dict[str, dict[str, object]] = {"need": {"id": "need", "enabled": 1}}
+
+ result = docs_delta.compare_needs(baseline, current)
+
+ assert result.modified[0].changed_fields == ("enabled", "optional")
+
+
+@pytest.mark.parametrize(
+ "payload",
+ [
+ {},
+ {"versions": []},
+ {"versions": {"1": {"needs": {"need": "not a record"}}}},
+ ],
+)
+def test_load_needs_rejects_malformed_exports(tmp_path: Path, payload: object) -> None:
+ (tmp_path / "needs.json").write_text(json.dumps(payload), encoding="utf-8")
+
+ with pytest.raises(docs_delta.DocsDeltaError, match="needs JSON|Need entry"):
+ docs_delta.load_needs(tmp_path)
+
+
+def test_load_needs_flattens_versions_and_uses_last_duplicate_id(
+ tmp_path: Path,
+) -> None:
+ (tmp_path / "needs.json").write_text(
+ json.dumps(
+ {
+ "versions": {
+ "1": {"needs": {"same": {"id": "same", "title": "Old"}}},
+ "ignored": "malformed version container",
+ "2": {
+ "needs": {"same": {"id": "same", "title": "New"}},
+ },
+ }
+ }
+ ),
+ encoding="utf-8",
+ )
+
+ assert docs_delta.load_needs(tmp_path) == {"same": {"id": "same", "title": "New"}}
diff --git a/tools/tests/docs_delta_github_test.py b/tools/tests/docs_delta_github_test.py
new file mode 100644
index 000000000..af31a20a4
--- /dev/null
+++ b/tools/tests/docs_delta_github_test.py
@@ -0,0 +1,300 @@
+# *******************************************************************************
+# Copyright (c) 2026 Contributors to the Eclipse Foundation
+#
+# See the NOTICE file(s) distributed with this work for additional
+# information regarding copyright ownership.
+#
+# This program and the accompanying materials are made available under the
+# terms of the Apache License Version 2.0 which is available at
+# https://www.apache.org/licenses/LICENSE-2.0
+#
+# SPDX-License-Identifier: Apache-2.0
+# *******************************************************************************
+"""Tests for Actions metadata and gh-pages baseline resolution."""
+
+from __future__ import annotations
+
+import io
+import json
+import subprocess
+import tarfile
+from pathlib import Path
+from types import SimpleNamespace
+
+import pytest
+
+from tools import docs_delta
+from tools.docs_delta import github
+from tools.tests.docs_delta_test_support import git, need, write_needs
+
+
+def test_pull_request_metadata_is_only_read_for_pr_events(
+ tmp_path: Path,
+) -> None:
+ assert github.github_pull_request({"GITHUB_EVENT_NAME": "push"}) is None
+
+ event = tmp_path / "event.json"
+ event.write_text("not JSON", encoding="utf-8")
+ with pytest.raises(docs_delta.DocsDeltaError, match="cannot read GitHub event"):
+ github.github_pull_request(
+ {
+ "GITHUB_EVENT_NAME": "pull_request",
+ "GITHUB_EVENT_PATH": str(event),
+ }
+ )
+
+
+def test_workspace_path_falls_back_to_relative_paths_locally() -> None:
+ assert github.workspace_path({}, "docs-artifact") == Path("docs-artifact")
+ assert github.workspace_path(
+ {"GITHUB_WORKSPACE": "/workspace"}, "docs-artifact"
+ ) == Path("/workspace/docs-artifact")
+
+
+@pytest.mark.parametrize(
+ "payload",
+ [
+ {},
+ {"pull_request": "not an object"},
+ {"pull_request": {"base": "not an object"}},
+ {"pull_request": {"base": {"ref": "", "sha": ""}}},
+ ],
+)
+def test_incomplete_pull_request_payload_does_not_create_a_baseline(
+ tmp_path: Path, payload: object
+) -> None:
+ event = tmp_path / "event.json"
+ event.write_text(json.dumps(payload), encoding="utf-8")
+
+ assert (
+ github.github_pull_request(
+ {
+ "GITHUB_EVENT_NAME": "pull_request",
+ "GITHUB_EVENT_PATH": str(event),
+ }
+ )
+ is None
+ )
+
+
+def test_non_object_event_payload_is_rejected(tmp_path: Path) -> None:
+ event = tmp_path / "event.json"
+ event.write_text("[]", encoding="utf-8")
+
+ with pytest.raises(docs_delta.DocsDeltaError, match="payload is not an object"):
+ github.github_pull_request(
+ {
+ "GITHUB_EVENT_NAME": "pull_request",
+ "GITHUB_EVENT_PATH": str(event),
+ }
+ )
+
+
+def test_pull_request_payload_rejects_missing_event_path() -> None:
+ with pytest.raises(
+ docs_delta.DocsDeltaError, match="GITHUB_EVENT_PATH is required"
+ ):
+ github.github_pull_request({"GITHUB_EVENT_NAME": "pull_request"})
+
+
+def test_pull_request_metadata_uses_actions_overrides_and_ignores_boolean_number(
+ tmp_path: Path,
+) -> None:
+ event = tmp_path / "event.json"
+ event.write_text(
+ json.dumps(
+ {
+ "pull_request": {
+ "number": True,
+ "base": {"ref": "event-main", "sha": "event-sha"},
+ }
+ }
+ ),
+ encoding="utf-8",
+ )
+
+ pull_request = github.github_pull_request(
+ {
+ "GITHUB_EVENT_NAME": "pull_request",
+ "GITHUB_EVENT_PATH": str(event),
+ "GITHUB_BASE_REF": "actions-main",
+ "GITHUB_BASE_SHA": "actions-sha",
+ }
+ )
+
+ assert pull_request == github.GithubPullRequest(
+ base_ref="actions-main", base_sha="actions-sha", number=None
+ )
+
+
+def test_resolve_urls_requires_complete_defaults() -> None:
+ args = docs_delta.cli.argument_parser().parse_args([])
+
+ with pytest.raises(
+ docs_delta.DocsDeltaError, match="documentation URLs are required"
+ ):
+ github.resolve_urls(args, None, {})
+
+
+def test_unsafe_archive_path_is_rejected(
+ tmp_path: Path, monkeypatch: pytest.MonkeyPatch
+) -> None:
+ archive_data = io.BytesIO()
+ with tarfile.open(fileobj=archive_data, mode="w") as archive:
+ member = tarfile.TarInfo("../escaped.html")
+ member.size = len(b"unsafe")
+ archive.addfile(member, io.BytesIO(b"unsafe"))
+
+ monkeypatch.setattr(
+ github.subprocess,
+ "run",
+ lambda *args, **kwargs: SimpleNamespace(
+ returncode=0, stdout=archive_data.getvalue(), stderr=b""
+ ),
+ )
+ destination = tmp_path / "baseline"
+
+ with pytest.raises(docs_delta.DocsDeltaError, match="unsafe path"):
+ github.extract_git_tree(tmp_path, "commit", "main", destination)
+
+ assert not (tmp_path / "escaped.html").exists()
+
+
+def test_unsupported_archive_member_is_rejected(
+ tmp_path: Path, monkeypatch: pytest.MonkeyPatch
+) -> None:
+ archive_data = io.BytesIO()
+ with tarfile.open(fileobj=archive_data, mode="w") as archive:
+ member = tarfile.TarInfo("external-link")
+ member.type = tarfile.SYMTYPE
+ member.linkname = "outside"
+ archive.addfile(member)
+
+ monkeypatch.setattr(
+ github.subprocess,
+ "run",
+ lambda *args, **kwargs: SimpleNamespace(
+ returncode=0, stdout=archive_data.getvalue(), stderr=b""
+ ),
+ )
+
+ with pytest.raises(
+ docs_delta.DocsDeltaError, match="unsupported gh-pages archive member"
+ ):
+ github.extract_git_tree(tmp_path, "commit", "main", tmp_path / "baseline")
+
+
+def test_archive_command_failure_is_reported(
+ tmp_path: Path, monkeypatch: pytest.MonkeyPatch
+) -> None:
+ monkeypatch.setattr(
+ github.subprocess,
+ "run",
+ lambda *args, **kwargs: SimpleNamespace(
+ returncode=1, stdout=b"", stderr=b"missing tree"
+ ),
+ )
+
+ with pytest.raises(docs_delta.DocsDeltaError, match="missing tree"):
+ github.extract_git_tree(tmp_path, "commit", "main", tmp_path / "baseline")
+
+
+def test_missing_gh_pages_checkout_writes_unavailable_report(
+ tmp_path: Path, monkeypatch: pytest.MonkeyPatch
+) -> None:
+ current = tmp_path / "docs-artifact"
+ current.mkdir()
+ event = tmp_path / "event.json"
+ event.write_text(
+ json.dumps(
+ {
+ "pull_request": {
+ "number": 125,
+ "base": {"ref": "main", "sha": "base-sha"},
+ }
+ }
+ ),
+ encoding="utf-8",
+ )
+ monkeypatch.setenv("GITHUB_EVENT_NAME", "pull_request")
+ monkeypatch.setenv("GITHUB_EVENT_PATH", str(event))
+ monkeypatch.setenv("GITHUB_WORKSPACE", str(tmp_path))
+
+ assert docs_delta.main([]) == 0
+ report_text = (current / "docs-delta.md").read_text(encoding="utf-8")
+ assert "gh-pages checkout is missing" in report_text
+
+
+def test_non_git_gh_pages_checkout_writes_unavailable_report(
+ tmp_path: Path, monkeypatch: pytest.MonkeyPatch
+) -> None:
+ gh_pages = tmp_path / ".docs-baseline"
+ gh_pages.mkdir()
+ current = tmp_path / "docs-artifact"
+ current.mkdir()
+ event = tmp_path / "event.json"
+ event.write_text(
+ json.dumps(
+ {
+ "pull_request": {
+ "number": 126,
+ "base": {"ref": "main", "sha": "base-sha"},
+ }
+ }
+ ),
+ encoding="utf-8",
+ )
+ monkeypatch.setenv("GITHUB_EVENT_NAME", "pull_request")
+ monkeypatch.setenv("GITHUB_EVENT_PATH", str(event))
+ monkeypatch.setenv("GITHUB_WORKSPACE", str(tmp_path))
+
+ assert docs_delta.main([]) == 0
+ report_text = (current / "docs-delta.md").read_text(encoding="utf-8")
+ assert "gh-pages checkout is not a Git repository" in report_text
+
+
+def test_unpublished_base_commit_writes_unavailable_report(
+ tmp_path: Path, monkeypatch: pytest.MonkeyPatch
+) -> None:
+ gh_pages = tmp_path / ".docs-baseline"
+ gh_pages.mkdir()
+ subprocess.run(
+ ["git", "init", "--initial-branch=gh-pages", str(gh_pages)],
+ check=True,
+ capture_output=True,
+ text=True,
+ )
+ git(gh_pages, "config", "user.name", "Documentation Delta Test")
+ git(gh_pages, "config", "user.email", "docs-delta@example.invalid")
+ main_docs = gh_pages / "main"
+ main_docs.mkdir()
+ write_needs(main_docs, {"need": need("guide", "Published", id="need")})
+ git(gh_pages, "add", ".")
+ git(gh_pages, "commit", "-m", "Publish older documentation")
+
+ current = tmp_path / "docs-artifact"
+ current.mkdir()
+ write_needs(current, {})
+ event = tmp_path / "event.json"
+ event.write_text(
+ json.dumps(
+ {
+ "pull_request": {
+ "number": 124,
+ "base": {
+ "ref": "main",
+ "sha": "abcdef0123456789abcdef0123456789abcdef01",
+ },
+ }
+ }
+ ),
+ encoding="utf-8",
+ )
+ monkeypatch.setenv("GITHUB_EVENT_NAME", "pull_request")
+ monkeypatch.setenv("GITHUB_EVENT_PATH", str(event))
+ monkeypatch.setenv("GITHUB_WORKSPACE", str(tmp_path))
+
+ assert docs_delta.main([]) == 0
+ report_text = (current / "docs-delta.md").read_text(encoding="utf-8")
+ assert "## Delta unavailable" in report_text
+ assert "no published main baseline contains source commit" in report_text
diff --git a/tools/tests/docs_delta_html_test.py b/tools/tests/docs_delta_html_test.py
new file mode 100644
index 000000000..b8cf0f19d
--- /dev/null
+++ b/tools/tests/docs_delta_html_test.py
@@ -0,0 +1,83 @@
+# *******************************************************************************
+# Copyright (c) 2026 Contributors to the Eclipse Foundation
+#
+# See the NOTICE file(s) distributed with this work for additional
+# information regarding copyright ownership.
+#
+# This program and the accompanying materials are made available under the
+# terms of the Apache License Version 2.0 which is available at
+# https://www.apache.org/licenses/LICENSE-2.0
+#
+# SPDX-License-Identifier: Apache-2.0
+# *******************************************************************************
+"""Tests for generated HTML normalization and page comparison."""
+
+from __future__ import annotations
+
+from pathlib import Path
+
+import pytest
+
+from tools import docs_delta
+
+
+def test_normalize_html_ignores_generated_metadata_but_keeps_content_changes() -> None:
+ old = (
+ '\r\n'
+ '\r\n'
+ "\r\n"
+ '\r\n'
+ "Documentation
\r\n"
+ )
+ new = (
+ '\n'
+ '\n'
+ "\n"
+ '\n'
+ "Documentation
\n"
+ )
+
+ assert docs_delta.normalize_html(old) == docs_delta.normalize_html(new)
+ assert docs_delta.normalize_html(
+ new.replace("src/example.py", "src/changed.py")
+ ) != docs_delta.normalize_html(old)
+ assert docs_delta.normalize_html(new).replace(
+ "Documentation", "Changed"
+ ) != docs_delta.normalize_html(old)
+
+
+def test_normalize_html_preserves_user_metadata_and_ordinary_comments() -> None:
+ content = ''
+
+ assert docs_delta.normalize_html(content) == content
+
+
+def test_compare_html_categorizes_added_removed_modified_and_unchanged(
+ tmp_path: Path,
+) -> None:
+ baseline = tmp_path / "baseline"
+ current = tmp_path / "current"
+ (baseline / "nested").mkdir(parents=True)
+ (current / "nested").mkdir(parents=True)
+ (baseline / "same.html").write_text("same
\n", encoding="utf-8")
+ (current / "same.html").write_text("same
\n", encoding="utf-8")
+ (baseline / "changed.html").write_text("old
\n", encoding="utf-8")
+ (current / "changed.html").write_text("new
\n", encoding="utf-8")
+ (baseline / "removed.html").write_text("removed
\n", encoding="utf-8")
+ (current / "added.html").write_text("added
\n", encoding="utf-8")
+ (baseline / "nested/page.html").write_text("old
\n", encoding="utf-8")
+ (current / "nested/page.html").write_text("old
\n", encoding="utf-8")
+
+ result = docs_delta.compare_html(baseline, current)
+
+ assert [entry.path for entry in result.added] == ["added.html"]
+ assert [entry.path for entry in result.removed] == ["removed.html"]
+ assert [entry.path for entry in result.modified] == ["changed.html"]
+ assert result.unchanged_count == 2
+
+
+def test_compare_html_rejects_a_missing_directory(tmp_path: Path) -> None:
+ with pytest.raises(
+ docs_delta.DocsDeltaError, match="documentation directory is missing"
+ ):
+ docs_delta.compare_html(tmp_path / "missing", tmp_path / "current")
diff --git a/tools/tests/docs_delta_report_test.py b/tools/tests/docs_delta_report_test.py
new file mode 100644
index 000000000..bc7fe5752
--- /dev/null
+++ b/tools/tests/docs_delta_report_test.py
@@ -0,0 +1,155 @@
+# *******************************************************************************
+# Copyright (c) 2026 Contributors to the Eclipse Foundation
+#
+# See the NOTICE file(s) distributed with this work for additional
+# information regarding copyright ownership.
+#
+# This program and the accompanying materials are made available under the
+# terms of the Apache License Version 2.0 which is available at
+# https://www.apache.org/licenses/LICENSE-2.0
+#
+# SPDX-License-Identifier: Apache-2.0
+# *******************************************************************************
+"""Tests for Markdown report content, links, and file output."""
+
+from __future__ import annotations
+
+from pathlib import Path
+
+import pytest
+
+from tools import docs_delta
+from tools.docs_delta import report as report_module
+from tools.tests.docs_delta_test_support import need
+
+
+def test_modified_need_reports_all_changed_fields_and_one_sided_links() -> None:
+ result = docs_delta.render_report(
+ docs_delta.NeedComparison(
+ added=(docs_delta.NeedChange("added", None, need("new/page", "Added")),),
+ removed=(
+ docs_delta.NeedChange("removed", need("old/page", "Removed"), None),
+ ),
+ modified=(
+ docs_delta.NeedChange(
+ "changed",
+ need("old/page", "Old", tags=["old"], extra="before"),
+ need("new/page", "New", tags=["new"], extra="after"),
+ ),
+ ),
+ unchanged_count=0,
+ ),
+ docs_delta.PageComparison((), (), (), 0),
+ base_url="https://docs.example/main",
+ pr_url="https://docs.example/pr/1",
+ )
+
+ assert "[old](https://docs.example/main/old/page.html#changed)" in result
+ assert "[new](https://docs.example/pr/1/new/page.html#changed)" in result
+ assert '`extra`: "before" → "after"' in result
+ assert '`tags`: ["old"] → ["new"]' in result
+ assert "https://docs.example/pr/1/new/page.html#added" in result
+ assert "https://docs.example/main/old/page.html#removed" in result
+
+
+def test_large_changed_values_are_truncated() -> None:
+ large_old = "a" * (docs_delta.MAX_RENDERED_VALUE_LENGTH + 20)
+ large_new = "b" * (docs_delta.MAX_RENDERED_VALUE_LENGTH + 20)
+ change = docs_delta.NeedChange(
+ "changed",
+ need("page", "Old", details=large_old),
+ need("page", "New", details=large_new),
+ )
+
+ report = docs_delta.render_report(
+ docs_delta.NeedComparison((), (), (change,), 0),
+ docs_delta.PageComparison((), (), (), 0),
+ base_url="https://docs.example/main",
+ pr_url="https://docs.example/pr/1",
+ )
+
+ assert "…" in report
+ assert large_old not in report
+ assert large_new not in report
+
+
+def test_detail_threshold_is_independent_for_needs_and_pages() -> None:
+ needs = docs_delta.NeedComparison(
+ added=tuple(
+ docs_delta.NeedChange(str(index), None, need("page", str(index)))
+ for index in range(16)
+ ),
+ removed=(),
+ modified=(),
+ unchanged_count=0,
+ )
+ pages = docs_delta.PageComparison(
+ added=tuple(docs_delta.PageChange(f"page-{index}.html") for index in range(16)),
+ removed=(),
+ modified=(),
+ unchanged_count=0,
+ )
+
+ report = docs_delta.render_report(
+ needs,
+ pages,
+ base_url="https://docs.example/main",
+ pr_url="https://docs.example/pr/1",
+ )
+
+ assert "16 entries changed; field details omitted." in report
+ assert "Page changes (16)" in report
+ assert "`page-0.html`" in report
+ assert "- `0`" in report
+
+
+def test_large_modified_need_section_keeps_ids_but_hides_field_values() -> None:
+ changes = tuple(
+ docs_delta.NeedChange(
+ f"REQ-{index:02}",
+ need("page", "Old title", details=f"old secret {index}"),
+ need("page", "New title", details=f"new secret {index}"),
+ )
+ for index in range(16)
+ )
+
+ result = docs_delta.render_report(
+ docs_delta.NeedComparison((), (), changes, 0),
+ docs_delta.PageComparison((), (), (), 0),
+ base_url="https://docs.example/main",
+ pr_url="https://docs.example/pr/1",
+ )
+
+ assert "Need changes (16)" in result
+ assert "16 entries changed; field details omitted." in result
+ assert "`REQ-00`" in result and "`REQ-15`" in result
+ assert "old secret" not in result
+ assert "new secret" not in result
+
+
+def test_need_links_encode_path_segments_and_anchors() -> None:
+ url = docs_delta.need_link(
+ "https://docs.example/main/",
+ "REQ #1",
+ {"docname": "nested/a guide"},
+ )
+
+ assert url == "https://docs.example/main/nested/a%20guide.html#REQ%20%231"
+
+
+def test_report_write_failure_preserves_previous_output_and_cleans_up(
+ tmp_path: Path, monkeypatch: pytest.MonkeyPatch
+) -> None:
+ output = tmp_path / "delta.md"
+ output.write_text("previous report", encoding="utf-8")
+
+ def fail_replace(source: Path, destination: Path) -> None:
+ raise OSError("simulated replace failure")
+
+ monkeypatch.setattr(report_module.os, "replace", fail_replace)
+
+ with pytest.raises(OSError, match="simulated replace failure"):
+ report_module.write_report(output, "new report")
+
+ assert output.read_text(encoding="utf-8") == "previous report"
+ assert list(tmp_path.glob(".delta.md.*.tmp")) == []
diff --git a/tools/tests/docs_delta_test_support.py b/tools/tests/docs_delta_test_support.py
new file mode 100644
index 000000000..7fe3859e8
--- /dev/null
+++ b/tools/tests/docs_delta_test_support.py
@@ -0,0 +1,41 @@
+# *******************************************************************************
+# Copyright (c) 2026 Contributors to the Eclipse Foundation
+#
+# See the NOTICE file(s) distributed with this work for additional
+# information regarding copyright ownership.
+#
+# This program and the accompanying materials are made available under the
+# terms of the Apache License Version 2.0 which is available at
+# https://www.apache.org/licenses/LICENSE-2.0
+#
+# SPDX-License-Identifier: Apache-2.0
+# *******************************************************************************
+"""Shared fixture helpers for documentation delta tests."""
+
+from __future__ import annotations
+
+import json
+import subprocess
+from pathlib import Path
+
+
+def need(docname: str, title: str, **extra: object) -> dict[str, object]:
+ """Make a small representative Sphinx-Needs record for unit tests."""
+ return {"id": title.lower(), "docname": docname, "title": title, **extra}
+
+
+def write_needs(directory: Path, needs: dict[str, dict[str, object]]) -> None:
+ """Write the versioned JSON envelope expected from Sphinx-Needs."""
+ (directory / "needs.json").write_text(
+ json.dumps({"versions": {"1": {"needs": needs}}}), encoding="utf-8"
+ )
+
+
+def git(directory: Path, *arguments: str) -> None:
+ """Run a Git command against a temporary repository in an integration test."""
+ subprocess.run(
+ ["git", "-C", str(directory), *arguments],
+ check=True,
+ capture_output=True,
+ text=True,
+ )