From 140203afdbfbc840f9b87adc475ed72471a48d2f Mon Sep 17 00:00:00 2001 From: Vincent Vatelot Date: Wed, 30 Sep 2026 17:16:31 +0200 Subject: [PATCH 1/3] feat(scraper): add YAML-driven RWEB best practices analysis Introduce a best_practices brick that evaluates a first set of RWEB rules alongside EcoIndex metrics, without changing the existing analysis contract. Co-authored-by: Cursor --- .../ecoindex/best_practices/__init__.py | 24 +++ components/ecoindex/best_practices/context.py | 14 ++ .../best_practices/default_rules.yaml | 64 ++++++ components/ecoindex/best_practices/engine.py | 97 +++++++++ components/ecoindex/best_practices/models.py | 86 ++++++++ .../ecoindex/best_practices/registry.py | 27 +++ .../ecoindex/best_practices/rules/__init__.py | 4 + .../ecoindex/best_practices/rules/dom.py | 29 +++ .../ecoindex/best_practices/rules/network.py | 26 +++ components/ecoindex/scraper/scrap.py | 62 ++++++ development/best_practices_scraper.py | 62 ++++++ projects/ecoindex_scraper/pyproject.toml | 4 + .../ecoindex/best_practices/test_engine.py | 203 ++++++++++++++++++ .../best_practices/test_scraper_api.py | 27 +++ 14 files changed, 729 insertions(+) create mode 100644 components/ecoindex/best_practices/__init__.py create mode 100644 components/ecoindex/best_practices/context.py create mode 100644 components/ecoindex/best_practices/default_rules.yaml create mode 100644 components/ecoindex/best_practices/engine.py create mode 100644 components/ecoindex/best_practices/models.py create mode 100644 components/ecoindex/best_practices/registry.py create mode 100644 components/ecoindex/best_practices/rules/__init__.py create mode 100644 components/ecoindex/best_practices/rules/dom.py create mode 100644 components/ecoindex/best_practices/rules/network.py create mode 100644 development/best_practices_scraper.py create mode 100644 test/components/ecoindex/best_practices/test_engine.py create mode 100644 test/components/ecoindex/best_practices/test_scraper_api.py diff --git a/components/ecoindex/best_practices/__init__.py b/components/ecoindex/best_practices/__init__.py new file mode 100644 index 0000000..7f695b1 --- /dev/null +++ b/components/ecoindex/best_practices/__init__.py @@ -0,0 +1,24 @@ +from ecoindex.best_practices.context import BestPracticesContext, DomMetrics +from ecoindex.best_practices.engine import ( + DEFAULT_RULES_PATH, + BestPracticesEngine, + load_rules_config, +) +from ecoindex.best_practices.models import ( + BestPracticesReport, + RuleCategory, + RuleResult, + RuleStatus, +) + +__all__ = [ + "BestPracticesContext", + "BestPracticesEngine", + "BestPracticesReport", + "DEFAULT_RULES_PATH", + "DomMetrics", + "RuleCategory", + "RuleResult", + "RuleStatus", + "load_rules_config", +] diff --git a/components/ecoindex/best_practices/context.py b/components/ecoindex/best_practices/context.py new file mode 100644 index 0000000..452e08e --- /dev/null +++ b/components/ecoindex/best_practices/context.py @@ -0,0 +1,14 @@ +from pydantic import BaseModel, Field + +from ecoindex.models.scraper import Requests + + +class DomMetrics(BaseModel): + inline_js: int = 0 + inline_css: int = 0 + print_stylesheet: int = 0 + + +class BestPracticesContext(BaseModel): + requests: Requests = Field(default_factory=Requests) + dom: DomMetrics = Field(default_factory=DomMetrics) diff --git a/components/ecoindex/best_practices/default_rules.yaml b/components/ecoindex/best_practices/default_rules.yaml new file mode 100644 index 0000000..7031c12 --- /dev/null +++ b/components/ecoindex/best_practices/default_rules.yaml @@ -0,0 +1,64 @@ +version: 1 +rules: + - id: http_requests + enabled: true + checker: http_requests + category: network + rweb_id: RWEB_0047 + url: https://rweb.greenit.fr/en/fiches/RWEB_0047-limit-the-number-of-http-requests + title: "Limit the number of HTTP requests" + description: > + Page load time correlates with the number of files the browser must + download and their individual size. Reducing HTTP requests per page + lowers the number of servers needed and the associated environmental + impacts. + thresholds: + # Soft warn inspired by GreenIT-Analysis intermediate target; fail = RWEB maxValue + warn: 26 + fail: 40 + + - id: domains_number + enabled: true + checker: domains_number + category: network + rweb_id: RWEB_0082 + url: https://rweb.greenit.fr/en/fiches/RWEB_0082-limit-the-number-of-domains-serving-resources + title: "Limit the number of domains serving resources" + description: > + When page components are hosted on several domains, the browser must + open an HTTP connection to each of them. Resources should be grouped + on as few domains as possible (a separate cookie-less domain for + static assets remaining an accepted exception). + thresholds: + warn: 3 + fail: 5 + + - id: externalize_css_js + enabled: true + checker: externalize_css_js + category: network + rweb_id: RWEB_0042 + url: https://rweb.greenit.fr/en/fiches/RWEB_0042-externalize-css-and-javascript + title: "Externalize CSS and Javascript" + description: > + CSS and JavaScript should not be inlined in the HTML of the page + (except small configuration variables). External files can be cached + by the browser and avoid re-downloading the same code on every page. + thresholds: + warn: 1 + fail: 2 + + - id: print_stylesheet + enabled: true + checker: print_stylesheet + category: user-device + rweb_id: RWEB_0031 + url: https://rweb.greenit.fr/en/fiches/RWEB_0031-provide-a-css-print + title: "Provide a CSS print" + description: > + A print stylesheet reduces unnecessary ink and paper by hiding chrome + (header, footer, menu, sidebars) and non-content images when printing. + # Presence check (GreenIT-Analysis style): at least one print stylesheet required + higher_is_worse: false + thresholds: + fail: 1 diff --git a/components/ecoindex/best_practices/engine.py b/components/ecoindex/best_practices/engine.py new file mode 100644 index 0000000..3111ae7 --- /dev/null +++ b/components/ecoindex/best_practices/engine.py @@ -0,0 +1,97 @@ +from pathlib import Path + +from yaml import safe_load + +from ecoindex.best_practices.context import BestPracticesContext +from ecoindex.best_practices.models import ( + BestPracticesReport, + RuleDefinition, + RuleResult, + RuleStatus, + RulesConfig, + Thresholds, +) +from ecoindex.best_practices.registry import get_checker + +# Ensure checkers are registered when the engine is used. +from ecoindex.best_practices import rules as _rules # noqa: F401 + +DEFAULT_RULES_PATH = Path(__file__).parent / "default_rules.yaml" + + +def load_rules_config(path: Path | None = None) -> RulesConfig: + config_path = path or DEFAULT_RULES_PATH + with open(config_path) as fp: + raw = safe_load(fp) or {} + return RulesConfig.model_validate(raw) + + +def evaluate_status( + value: float, + thresholds: Thresholds, + *, + higher_is_worse: bool = True, +) -> RuleStatus: + """Evaluate a measured value against RWEB-style maximum thresholds. + + When ``higher_is_worse`` is True (count metrics), ``fail`` / ``warn`` are + maximum acceptable values: status is fail when ``value > fail``, warn when + ``value > warn``. This matches RWEB ``maxValue`` ("less than or equal to"). + """ + warn = thresholds.warn + fail = thresholds.fail + + if higher_is_worse: + if fail is not None and value > fail: + return RuleStatus.fail + if warn is not None and value > warn: + return RuleStatus.warn + return RuleStatus.ok + + if fail is not None and value < fail: + return RuleStatus.fail + if warn is not None and value < warn: + return RuleStatus.warn + return RuleStatus.ok + + +def run_rule(context: BestPracticesContext, rule: RuleDefinition) -> RuleResult: + checker = get_checker(rule.checker) + outcome = checker(context, rule) + status = evaluate_status( + outcome.value, + rule.thresholds, + higher_is_worse=rule.higher_is_worse, + ) + message = outcome.message or f"{rule.id}={outcome.value}" + return RuleResult( + id=rule.id, + category=rule.category, + title=rule.title, + description=rule.description, + rweb_id=rule.rweb_id, + url=rule.url, + status=status, + value=outcome.value, + thresholds=rule.thresholds, + message=message, + details=outcome.details, + ) + + +class BestPracticesEngine: + def __init__(self, config: RulesConfig | Path | None = None): + if isinstance(config, RulesConfig): + self.config = config + elif isinstance(config, Path): + self.config = load_rules_config(config) + else: + self.config = load_rules_config(None) + + def run(self, context: BestPracticesContext) -> BestPracticesReport: + results: list[RuleResult] = [] + for rule in self.config.rules: + if not rule.enabled: + continue + results.append(run_rule(context, rule)) + return BestPracticesReport(results=results) diff --git a/components/ecoindex/best_practices/models.py b/components/ecoindex/best_practices/models.py new file mode 100644 index 0000000..b9ead77 --- /dev/null +++ b/components/ecoindex/best_practices/models.py @@ -0,0 +1,86 @@ +from enum import Enum + +from pydantic import BaseModel, Field, HttpUrl + + +class RuleStatus(str, Enum): + ok = "ok" + warn = "warn" + fail = "fail" + + +class RuleCategory(str, Enum): + """RWEB `tiers` values used by the référentiel.""" + + network = "network" + user_device = "user-device" + + +class Thresholds(BaseModel): + """Maximum acceptable values (RWEB `maxValue` semantics: value must be <= fail).""" + + warn: float | None = None + fail: float | None = None + + +class RuleDefinition(BaseModel): + id: str + enabled: bool = True + checker: str + category: RuleCategory + title: str + description: str = "" + rweb_id: str | None = None + url: HttpUrl | str | None = None + thresholds: Thresholds = Field(default_factory=Thresholds) + higher_is_worse: bool = True + params: dict = Field(default_factory=dict) + + +class RulesConfig(BaseModel): + version: int = 1 + rules: list[RuleDefinition] = Field(default_factory=list) + + +class RuleResult(BaseModel): + id: str + category: RuleCategory + title: str + description: str = "" + rweb_id: str | None = None + url: HttpUrl | str | None = None + status: RuleStatus + value: float + thresholds: Thresholds + message: str + details: list[str] = Field(default_factory=list) + + +class BestPracticesReport(BaseModel): + results: list[RuleResult] = Field(default_factory=list) + + @property + def ok_count(self) -> int: + return sum(1 for r in self.results if r.status == RuleStatus.ok) + + @property + def warn_count(self) -> int: + return sum(1 for r in self.results if r.status == RuleStatus.warn) + + @property + def fail_count(self) -> int: + return sum(1 for r in self.results if r.status == RuleStatus.fail) + + def results_by_category(self) -> dict[RuleCategory, list[RuleResult]]: + grouped: dict[RuleCategory, list[RuleResult]] = {} + for result in self.results: + grouped.setdefault(result.category, []).append(result) + return grouped + + +class CheckOutcome(BaseModel): + """Raw measurement returned by a checker before threshold evaluation.""" + + value: float + details: list[str] = Field(default_factory=list) + message: str | None = None diff --git a/components/ecoindex/best_practices/registry.py b/components/ecoindex/best_practices/registry.py new file mode 100644 index 0000000..275f700 --- /dev/null +++ b/components/ecoindex/best_practices/registry.py @@ -0,0 +1,27 @@ +from collections.abc import Callable + +from ecoindex.best_practices.context import BestPracticesContext +from ecoindex.best_practices.models import CheckOutcome, RuleDefinition + +Checker = Callable[[BestPracticesContext, RuleDefinition], CheckOutcome] + +_REGISTRY: dict[str, Checker] = {} + + +def register(name: str) -> Callable[[Checker], Checker]: + def decorator(func: Checker) -> Checker: + _REGISTRY[name] = func + return func + + return decorator + + +def get_checker(name: str) -> Checker: + try: + return _REGISTRY[name] + except KeyError as exc: + raise KeyError(f"Unknown best-practice checker: {name!r}") from exc + + +def list_checkers() -> list[str]: + return sorted(_REGISTRY.keys()) diff --git a/components/ecoindex/best_practices/rules/__init__.py b/components/ecoindex/best_practices/rules/__init__.py new file mode 100644 index 0000000..e474574 --- /dev/null +++ b/components/ecoindex/best_practices/rules/__init__.py @@ -0,0 +1,4 @@ +from ecoindex.best_practices.rules import dom as dom +from ecoindex.best_practices.rules import network as network + +__all__ = ["dom", "network"] diff --git a/components/ecoindex/best_practices/rules/dom.py b/components/ecoindex/best_practices/rules/dom.py new file mode 100644 index 0000000..588041c --- /dev/null +++ b/components/ecoindex/best_practices/rules/dom.py @@ -0,0 +1,29 @@ +from ecoindex.best_practices.context import BestPracticesContext +from ecoindex.best_practices.models import CheckOutcome, RuleDefinition +from ecoindex.best_practices.registry import register + + +@register("externalize_css_js") +def check_externalize_css_js( + context: BestPracticesContext, rule: RuleDefinition +) -> CheckOutcome: + value = float(context.dom.inline_js + context.dom.inline_css) + return CheckOutcome( + value=value, + message=( + f"{int(value)} inline CSS/JavaScript block(s) " + f"({context.dom.inline_css} CSS, {context.dom.inline_js} JS)" + ), + ) + + +@register("print_stylesheet") +def check_print_stylesheet( + context: BestPracticesContext, rule: RuleDefinition +) -> CheckOutcome: + value = float(context.dom.print_stylesheet) + if value >= 1: + message = f"{int(value)} print stylesheet(s) found" + else: + message = "No print stylesheet found" + return CheckOutcome(value=value, message=message) diff --git a/components/ecoindex/best_practices/rules/network.py b/components/ecoindex/best_practices/rules/network.py new file mode 100644 index 0000000..8d32dbf --- /dev/null +++ b/components/ecoindex/best_practices/rules/network.py @@ -0,0 +1,26 @@ +from ecoindex.best_practices.context import BestPracticesContext +from ecoindex.best_practices.models import CheckOutcome, RuleDefinition +from ecoindex.best_practices.registry import register + + +@register("http_requests") +def check_http_requests( + context: BestPracticesContext, rule: RuleDefinition +) -> CheckOutcome: + value = float(context.requests.total_count) + return CheckOutcome( + value=value, + message=f"{int(value)} HTTP request(s)", + ) + + +@register("domains_number") +def check_domains_number( + context: BestPracticesContext, rule: RuleDefinition +) -> CheckOutcome: + domains = sorted(context.requests.domain_aggregation.keys()) + return CheckOutcome( + value=float(len(domains)), + details=domains, + message=f"{len(domains)} distinct domain(s)", + ) diff --git a/components/ecoindex/scraper/scrap.py b/components/ecoindex/scraper/scrap.py index 4d5ba51..786f97d 100644 --- a/components/ecoindex/scraper/scrap.py +++ b/components/ecoindex/scraper/scrap.py @@ -1,6 +1,7 @@ import json import os from datetime import datetime +from pathlib import Path from time import sleep from uuid import uuid4 from typing import Any, cast @@ -10,6 +11,12 @@ from camoufox.async_api import AsyncCamoufox +from ecoindex.best_practices import ( + BestPracticesContext, + BestPracticesEngine, + BestPracticesReport, + DomMetrics, +) from ecoindex.compute import compute_ecoindex from ecoindex.exceptions.scraper import EcoindexScraperStatusException from ecoindex.models.compute import PageMetrics, Result, ScreenShot, WindowSize @@ -23,6 +30,24 @@ from ecoindex.utils.screenshots import convert_screenshot_to_webp, set_screenshot_rights from typing_extensions import deprecated +DOM_METRICS_SCRIPT = """ +() => { + const scripts = Array.from(document.scripts); + const inlineJs = scripts.filter((s) => !s.src).length; + const styles = Array.from(document.querySelectorAll("style")).filter( + (el) => !(el instanceof SVGStyleElement) + ); + const printStyles = + document.querySelectorAll("link[rel=stylesheet][media~=print]").length + + document.querySelectorAll("style[media~=print]").length; + return { + inline_js: inlineJs, + inline_css: styles.length, + print_stylesheet: printStyles, + }; +} +""" + class EcoindexScraper: def __init__( @@ -40,6 +65,7 @@ def __init__( cookies: list[dict[str, object]] = [], custom_headers: dict[str, str] = {}, logger=None, + best_practices: bool | Path = False, ): self.url = url self.window_size = window_size @@ -59,6 +85,12 @@ def __init__( self.cookies = cookies self.custom_headers = custom_headers self.logger = logger + self.best_practices_enabled = best_practices is not False + self.best_practices_config_path: Path | None = ( + best_practices if isinstance(best_practices, Path) else None + ) + self.dom_metrics = DomMetrics() + self._best_practices_report: BestPracticesReport | None = None @staticmethod def get_user_agent() -> UserAgent: @@ -93,6 +125,20 @@ async def get_requests_by_category(self) -> MimetypeAggregation: async def get_requests_by_domain(self) -> dict[str, DomainMetrics]: return self.all_requests.domain_aggregation + async def get_best_practices(self) -> BestPracticesReport: + if not self.best_practices_enabled: + raise RuntimeError( + "Best practices analysis is disabled. " + "Initialize EcoindexScraper with best_practices=True " + "or a Path to a YAML config." + ) + if self._best_practices_report is None: + raise RuntimeError( + "Best practices report is not available yet. " + "Call get_page_analysis() or scrap_page() first." + ) + return self._best_practices_report + async def scrap_page(self) -> PageMetrics: async with AsyncCamoufox( headless=self.headless, @@ -123,11 +169,15 @@ async def scrap_page(self) -> PageMetrics: ) sleep(self.wait_after_scroll) total_nodes = await self.get_nodes_count() + if self.best_practices_enabled: + self.dom_metrics = await self.get_dom_metrics() await self.page.close() await self.context.close() await browser.close() await self.get_requests_from_har_file() + if self.best_practices_enabled: + self._best_practices_report = self._run_best_practices() return PageMetrics( size=self.all_requests.total_size / 1000, @@ -135,6 +185,18 @@ async def scrap_page(self) -> PageMetrics: requests=self.all_requests.total_count, ) + def _run_best_practices(self) -> BestPracticesReport: + engine = BestPracticesEngine(self.best_practices_config_path) + context = BestPracticesContext( + requests=self.all_requests, + dom=self.dom_metrics, + ) + return engine.run(context) + + async def get_dom_metrics(self) -> DomMetrics: + raw = await self.page.evaluate(DOM_METRICS_SCRIPT) + return DomMetrics.model_validate(raw) + async def generate_screenshot(self) -> None: if self.screenshot and self.screenshot.folder and self.screenshot.id: await self.page.screenshot(path=self.screenshot.get_png()) diff --git a/development/best_practices_scraper.py b/development/best_practices_scraper.py new file mode 100644 index 0000000..cdf0ec5 --- /dev/null +++ b/development/best_practices_scraper.py @@ -0,0 +1,62 @@ +"""Manual smoke test for EcoIndex + RWEB best practices analysis. + +Run from the repo root: + + uv run development/best_practices_scraper.py +""" + +from __future__ import annotations + +import asyncio +import sys +from pathlib import Path + +# Prefer local polylith bricks over stale copies in site-packages. +_ROOT = Path(__file__).resolve().parents[1] +for _path in (_ROOT / "components", _ROOT / "bases", _ROOT / "development"): + sys.path.insert(0, str(_path)) + +from ecoindex.scraper import EcoindexScraper + +URL = "https://www.ecoindex.fr" +# Optional custom rules file; leave as None to use the default RWEB rules +RULES_PATH: Path | None = None + + +async def main() -> None: + best_practices: bool | Path = RULES_PATH if RULES_PATH else True + scraper = EcoindexScraper(url=URL, best_practices=best_practices) + + print(f"Analyzing {URL} ...") + result = await scraper.get_page_analysis() + report = await scraper.get_best_practices() + + print() + print("=== EcoIndex ===") + print(f"grade={result.grade} score={result.score}") + print( + f"nodes={result.nodes} size_kb={result.size:.1f} requests={result.requests}" + ) + print() + print( + "=== Best practices " + f"(ok={report.ok_count} warn={report.warn_count} fail={report.fail_count}) ===" + ) + + for rule in report.results: + print() + print(f"[{rule.status.value.upper()}] {rule.rweb_id} — {rule.title}") + print(f" category : {rule.category.value}") + print(f" value : {rule.value}") + print(f" thresholds: warn={rule.thresholds.warn} fail={rule.thresholds.fail}") + print(f" message : {rule.message}") + if rule.details: + print(f" details : {', '.join(rule.details[:5])}") + print(f" url : {rule.url}") + + if report.fail_count: + sys.exit(1) + + +if __name__ == "__main__": + asyncio.run(main()) diff --git a/projects/ecoindex_scraper/pyproject.toml b/projects/ecoindex_scraper/pyproject.toml index 982ee26..69cd917 100644 --- a/projects/ecoindex_scraper/pyproject.toml +++ b/projects/ecoindex_scraper/pyproject.toml @@ -31,7 +31,11 @@ build-backend = "hatchling.build" [tool.hatch.build.targets.wheel] packages = ["ecoindex"] +[tool.hatch.build.targets.wheel.force-include] +"../../components/ecoindex/best_practices/default_rules.yaml" = "ecoindex/best_practices/default_rules.yaml" + [tool.polylith.bricks] +"../../components/ecoindex/best_practices" = "ecoindex/best_practices" "../../components/ecoindex/compute" = "ecoindex/compute" "../../components/ecoindex/data" = "ecoindex/data" "../../components/ecoindex/exceptions" = "ecoindex/exceptions" diff --git a/test/components/ecoindex/best_practices/test_engine.py b/test/components/ecoindex/best_practices/test_engine.py new file mode 100644 index 0000000..5a950c7 --- /dev/null +++ b/test/components/ecoindex/best_practices/test_engine.py @@ -0,0 +1,203 @@ +from pathlib import Path + +import pytest +from ecoindex.best_practices import ( + BestPracticesContext, + BestPracticesEngine, + DomMetrics, + RuleStatus, + load_rules_config, +) +from ecoindex.best_practices.engine import evaluate_status +from ecoindex.best_practices.models import ( + RuleCategory, + RuleDefinition, + RulesConfig, + Thresholds, +) +from ecoindex.best_practices.registry import get_checker, list_checkers +from ecoindex.best_practices.rules import dom, network # noqa: F401 +from ecoindex.models.scraper import DomainMetrics, RequestItem, Requests + + +def _request( + url: str, + *, + status: int = 200, + category: str = "other", + mime_type: str = "application/octet-stream", + size: float = 100, +) -> RequestItem: + return RequestItem( + url=url, + domain=url.split("/")[2], + mime_type=mime_type, + status=status, + size=size, + category=category, + ) + + +def test_default_rules_config_loads() -> None: + config = load_rules_config() + assert config.version == 1 + assert len(config.rules) == 4 + assert {rule.checker for rule in config.rules} <= set(list_checkers()) + assert {rule.category for rule in config.rules} == { + RuleCategory.network, + RuleCategory.user_device, + } + assert all(rule.rweb_id and rule.rweb_id.startswith("RWEB_") for rule in config.rules) + assert all(rule.url for rule in config.rules) + assert all(rule.description for rule in config.rules) + + by_id = {rule.id: rule for rule in config.rules} + assert by_id["http_requests"].thresholds.fail == 40 + assert by_id["http_requests"].thresholds.warn == 26 + assert by_id["domains_number"].thresholds.fail == 5 + assert by_id["externalize_css_js"].thresholds.fail == 2 + assert by_id["print_stylesheet"].higher_is_worse is False + + +def test_evaluate_status_higher_is_worse_uses_rweb_max_semantics() -> None: + thresholds = Thresholds(warn=10, fail=20) + assert evaluate_status(5, thresholds) == RuleStatus.ok + assert evaluate_status(10, thresholds) == RuleStatus.ok + assert evaluate_status(11, thresholds) == RuleStatus.warn + assert evaluate_status(20, thresholds) == RuleStatus.warn + assert evaluate_status(21, thresholds) == RuleStatus.fail + + +def test_evaluate_status_higher_is_better() -> None: + thresholds = Thresholds(fail=1) + assert evaluate_status(0, thresholds, higher_is_worse=False) == RuleStatus.fail + assert evaluate_status(1, thresholds, higher_is_worse=False) == RuleStatus.ok + + +def test_network_checkers() -> None: + requests = Requests( + total_count=3, + items=[ + _request("https://a.example/page", status=200, category="html"), + _request("https://b.example/img.png", status=404, category="image"), + _request("https://c.example/go", status=301, category="other"), + ], + domain_aggregation={ + "a.example": DomainMetrics(total_count=1, total_size=100), + "b.example": DomainMetrics(total_count=1, total_size=100), + "c.example": DomainMetrics(total_count=1, total_size=100), + }, + ) + context = BestPracticesContext(requests=requests) + rule = RuleDefinition( + id="x", + checker="http_requests", + category=RuleCategory.network, + title="t", + thresholds=Thresholds(), + ) + + assert get_checker("http_requests")(context, rule).value == 3 + assert get_checker("domains_number")(context, rule).value == 3 + + +def test_dom_checkers() -> None: + context = BestPracticesContext( + dom=DomMetrics(inline_js=2, inline_css=1, print_stylesheet=0) + ) + rule = RuleDefinition( + id="x", + checker="externalize_css_js", + category=RuleCategory.network, + title="t", + thresholds=Thresholds(), + ) + assert get_checker("externalize_css_js")(context, rule).value == 3 + assert get_checker("print_stylesheet")(context, rule).value == 0 + + +def test_engine_skips_disabled_rules_and_applies_thresholds() -> None: + config = RulesConfig( + rules=[ + RuleDefinition( + id="http_requests", + enabled=True, + checker="http_requests", + category=RuleCategory.network, + title="HTTP requests", + rweb_id="RWEB_0047", + url="https://rweb.greenit.fr/en/fiches/RWEB_0047-limit-the-number-of-http-requests", + thresholds=Thresholds(warn=2, fail=5), + ), + RuleDefinition( + id="disabled_rule", + enabled=False, + checker="domains_number", + category=RuleCategory.network, + title="Disabled", + thresholds=Thresholds(warn=1, fail=2), + ), + RuleDefinition( + id="print_stylesheet", + enabled=True, + checker="print_stylesheet", + category=RuleCategory.user_device, + title="Print CSS", + rweb_id="RWEB_0031", + higher_is_worse=False, + thresholds=Thresholds(fail=1), + ), + ] + ) + context = BestPracticesContext( + requests=Requests(total_count=3), + dom=DomMetrics(print_stylesheet=0), + ) + report = BestPracticesEngine(config).run(context) + + assert len(report.results) == 2 + by_id = {r.id: r for r in report.results} + assert by_id["http_requests"].status == RuleStatus.warn + assert by_id["http_requests"].rweb_id == "RWEB_0047" + assert by_id["http_requests"].category == RuleCategory.network + assert by_id["print_stylesheet"].status == RuleStatus.fail + assert by_id["print_stylesheet"].category == RuleCategory.user_device + assert report.warn_count == 1 + assert report.fail_count == 1 + assert report.ok_count == 0 + by_category = report.results_by_category() + assert len(by_category[RuleCategory.network]) == 1 + assert len(by_category[RuleCategory.user_device]) == 1 + + +def test_engine_loads_custom_yaml(tmp_path: Path) -> None: + yaml_path = tmp_path / "rules.yaml" + yaml_path.write_text( + """ +version: 1 +rules: + - id: http_requests + enabled: true + checker: http_requests + category: network + rweb_id: RWEB_0047 + url: https://rweb.greenit.fr/en/fiches/RWEB_0047-limit-the-number-of-http-requests + title: Custom HTTP + thresholds: + warn: 1 + fail: 2 +""" + ) + report = BestPracticesEngine(yaml_path).run( + BestPracticesContext(requests=Requests(total_count=3)) + ) + assert len(report.results) == 1 + assert report.results[0].title == "Custom HTTP" + assert report.results[0].category == RuleCategory.network + assert report.results[0].rweb_id == "RWEB_0047" + assert report.results[0].status == RuleStatus.fail + + +def test_unknown_checker_raises() -> None: + with pytest.raises(KeyError, match="Unknown best-practice checker"): + get_checker("does_not_exist") diff --git a/test/components/ecoindex/best_practices/test_scraper_api.py b/test/components/ecoindex/best_practices/test_scraper_api.py new file mode 100644 index 0000000..5b40c67 --- /dev/null +++ b/test/components/ecoindex/best_practices/test_scraper_api.py @@ -0,0 +1,27 @@ +import pytest +from ecoindex.scraper import EcoindexScraper + + +@pytest.mark.asyncio +async def test_get_best_practices_requires_enabled() -> None: + scraper = EcoindexScraper(url="https://www.example.com") + with pytest.raises(RuntimeError, match="disabled"): + await scraper.get_best_practices() + + +@pytest.mark.asyncio +async def test_get_best_practices_requires_analysis_first() -> None: + scraper = EcoindexScraper(url="https://www.example.com", best_practices=True) + with pytest.raises(RuntimeError, match="not available yet"): + await scraper.get_best_practices() + + +def test_scraper_best_practices_init_with_path(tmp_path) -> None: + config = tmp_path / "rules.yaml" + config.write_text("version: 1\nrules: []\n") + scraper = EcoindexScraper( + url="https://www.example.com", + best_practices=config, + ) + assert scraper.best_practices_enabled is True + assert scraper.best_practices_config_path == config From 2fe6554c0a6525105bff1fa86ac547f85a5f47da Mon Sep 17 00:00:00 2001 From: Vincent Vatelot Date: Thu, 1 Oct 2026 16:36:50 +0200 Subject: [PATCH 2/3] feat(scraper): include best_practices brick in api and cli packages Polylith check requires projects that ship EcoindexScraper to also declare the new best_practices dependency brick. Co-authored-by: Cursor --- projects/ecoindex_api/pyproject.toml | 4 ++++ projects/ecoindex_cli/pyproject.toml | 4 ++++ 2 files changed, 8 insertions(+) diff --git a/projects/ecoindex_api/pyproject.toml b/projects/ecoindex_api/pyproject.toml index 65e9196..ec87f75 100644 --- a/projects/ecoindex_api/pyproject.toml +++ b/projects/ecoindex_api/pyproject.toml @@ -61,9 +61,13 @@ build-backend = "hatchling.build" [tool.hatch.build.targets.wheel] packages = ["ecoindex"] +[tool.hatch.build.targets.wheel.force-include] +"../../components/ecoindex/best_practices/default_rules.yaml" = "ecoindex/best_practices/default_rules.yaml" + [tool.polylith.bricks] "../../bases/ecoindex/backend" = "ecoindex/backend" "../../bases/ecoindex/worker" = "ecoindex/worker" +"../../components/ecoindex/best_practices" = "ecoindex/best_practices" "../../components/ecoindex/compute" = "ecoindex/compute" "../../components/ecoindex/config" = "ecoindex/config" "../../components/ecoindex/data" = "ecoindex/data" diff --git a/projects/ecoindex_cli/pyproject.toml b/projects/ecoindex_cli/pyproject.toml index ec2f54f..32a038b 100644 --- a/projects/ecoindex_cli/pyproject.toml +++ b/projects/ecoindex_cli/pyproject.toml @@ -40,8 +40,12 @@ build-backend = "hatchling.build" [tool.hatch.build.targets.wheel] packages = ["ecoindex"] +[tool.hatch.build.targets.wheel.force-include] +"../../components/ecoindex/best_practices/default_rules.yaml" = "ecoindex/best_practices/default_rules.yaml" + [tool.polylith.bricks] "../../bases/ecoindex/cli" = "ecoindex/cli" +"../../components/ecoindex/best_practices" = "ecoindex/best_practices" "../../components/ecoindex/compute" = "ecoindex/compute" "../../components/ecoindex/config" = "ecoindex/config" "../../components/ecoindex/data" = "ecoindex/data" From 46bccdaf8fb1683497ec9f38c7183b0c1ceb0643 Mon Sep 17 00:00:00 2001 From: Vincent Vatelot Date: Thu, 1 Oct 2026 16:54:03 +0200 Subject: [PATCH 3/3] fix(scraper): silence ruff E402 in best practices smoke script Path bootstrap must run before importing EcoindexScraper so local polylith bricks win over site-packages. Co-authored-by: Cursor --- development/best_practices_scraper.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/development/best_practices_scraper.py b/development/best_practices_scraper.py index cdf0ec5..b5e0101 100644 --- a/development/best_practices_scraper.py +++ b/development/best_practices_scraper.py @@ -16,7 +16,7 @@ for _path in (_ROOT / "components", _ROOT / "bases", _ROOT / "development"): sys.path.insert(0, str(_path)) -from ecoindex.scraper import EcoindexScraper +from ecoindex.scraper import EcoindexScraper # noqa: E402 URL = "https://www.ecoindex.fr" # Optional custom rules file; leave as None to use the default RWEB rules