Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -53,8 +53,8 @@ dependencies = [
all = ["crawlee[adaptive-crawler,pydantic-ai,beautifulsoup,cli,curl-impersonate,httpx,parsel,playwright,otel,sql_sqlite,sql_postgres,sql_mysql,stagehand,redis]"]
adaptive-crawler = [
"crawlee[beautifulsoup,parsel]",
"jaro-winkler>=2.0.3",
"playwright>=1.27.0",
"rapidfuzz>=3.0.0",
"scikit-learn>=1.6.0",
"apify_fingerprint_datapoints>=0.0.3",
"browserforge>=1.2.4"
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -9,8 +9,8 @@
from typing import TYPE_CHECKING, Annotated, Literal
from urllib.parse import urlparse

from jaro import jaro_winkler_metric
from pydantic import BaseModel, ConfigDict, Field, PlainSerializer, PlainValidator
from rapidfuzz.distance import JaroWinkler
from sklearn.linear_model import LogisticRegression
from typing_extensions import override

Expand Down Expand Up @@ -259,10 +259,10 @@ def calculate_url_similarity(url_1: UrlComponents, url_2: UrlComponents) -> floa
"""Calculate url similarity based on host name and path components similarity.

Return 0 if different host names.
Compare path components using jaro-wrinkler method and assign 1 or 0 value based on similarity_cutoff for each
Compare path components using Jaro-Winkler method and assign 1 or 0 value based on similarity_cutoff for each
path component. Return their weighted average.
"""
# Anything with jaro_winkler_metric less than this value is considered completely different,
# Anything with Jaro-Winkler similarity less than this value is considered completely different,
# otherwise considered the same.
similarity_cutoff = 0.8

Expand All @@ -273,6 +273,6 @@ def calculate_url_similarity(url_1: UrlComponents, url_2: UrlComponents) -> floa

# Each additional path component from longer path is compared to empty string.
return mean(
1 if jaro_winkler_metric(path_1, path_2) > similarity_cutoff else 0
1 if JaroWinkler.similarity(path_1, path_2) > similarity_cutoff else 0
for path_1, path_2 in zip_longest(url_1[1:], url_2[1:], fillvalue='')
)
Original file line number Diff line number Diff line change
Expand Up @@ -1079,7 +1079,7 @@ async def handler(context: AdaptivePlaywrightCrawlingContext) -> None:
'optional_module_name',
[
pytest.param('playwright', id='playwright'),
pytest.param('jaro', id='jaro'),
pytest.param('rapidfuzz', id='rapidfuzz'),
pytest.param('sklearn', id='sklearn'),
],
)
Expand Down
10 changes: 6 additions & 4 deletions tests/unit/crawlers/_adaptive_playwright/test_predictor.py
Original file line number Diff line number Diff line change
Expand Up @@ -200,14 +200,16 @@ async def test_persistent_prediction_recovery(*, persistence_enabled: bool, same
@pytest.mark.parametrize(
('url_1', 'url_2', 'expected_rounded_similarity'),
[
(
pytest.param(
'https://docs.python.org/3/library/itertools.html#itertools.zip_longest',
'https://docs.python.org/3.7/library/itertools.html#itertools.zip_longest',
0.67,
id='different path segment',
),
('https://differente.com/same', 'https://differenta.com/same', 0),
('https://same.com/almost_the_same', 'https://same.com/almost_the_sama', 1),
('https://same.com/same/extra', 'https://same.com/same', 0.5),
pytest.param('https://differente.com/same', 'https://differenta.com/same', 0, id='different domain'),
pytest.param('https://same.com/almost_the_same', 'https://same.com/almost_the_sama', 1, id='similar path'),
pytest.param('https://same.com/same/extra', 'https://same.com/same', 0.5, id='extra path segment'),
pytest.param('https://same.com/item/12345', 'https://same.com/item/12399', 1, id='shared numeric prefix'),
],
)
def test_url_similarity(url_1: str, url_2: str, expected_rounded_similarity: float) -> None:
Expand Down
Loading
Loading