Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
27 changes: 14 additions & 13 deletions src/apify/scrapy/requests.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,7 @@
from scrapy import Spider
from scrapy.http.headers import Headers
from scrapy.utils.misc import load_object
from scrapy.utils.request import request_from_dict
from scrapy.utils.request import RequestFingerprinter, request_from_dict

from crawlee._request import UserData
from crawlee._types import HttpHeaders
Expand Down Expand Up @@ -61,23 +61,28 @@ def to_apify_request(scrapy_request: ScrapyRequest, spider: Spider) -> ApifyRequ
logger.warning('Failed to convert to Apify request: Scrapy request must be a ScrapyRequest instance.')
return None

# Configuration to behave as similarly as possible to Scrapy's default RFPDupeFilter.
#
# The body is stored twice on purpose: as `payload` (used for the extended unique key) and inside the serialized
# Scrapy request below (used to reconstruct it). Both come from `scrapy_request.body`.
# The body is stored as `payload` for the Apify-platform view of the request; the authoritative copy used to
# reconstruct the Scrapy request travels inside the serialized blob below. Both come from `scrapy_request.body`.
request_kwargs: dict[str, Any] = {
'url': scrapy_request.url,
'method': scrapy_request.method,
'payload': scrapy_request.body,
'use_extended_unique_key': True,
'keep_url_fragment': False,
}

try:
if scrapy_request.dont_filter:
request_kwargs['always_enqueue'] = True
elif scrapy_request.meta.get('apify_request_unique_key'):
request_kwargs['unique_key'] = scrapy_request.meta['apify_request_unique_key']
else:
# Deduplicate exactly like Scrapy's own RFPDupeFilter, whose fingerprint canonicalizes the URL
# case-sensitively (`w3lib.url.canonicalize_url`), keeps `utm_*` params, and ignores headers. Left to
# Crawlee, the unique key would instead come from `normalize_url`, which lowercases the whole URL and
# strips `utm_*` — silently collapsing distinct pages Scrapy would crawl — while its extended key hashes
# headers Scrapy ignores. A custom `REQUEST_FINGERPRINTER_CLASS` is honored when the spider has a crawler.
crawler = getattr(spider, 'crawler', None)
fingerprinter = getattr(crawler, 'request_fingerprinter', None) or RequestFingerprinter()
request_kwargs['unique_key'] = fingerprinter.fingerprint(scrapy_request).hex()

# Serialize the Scrapy request now, before `Request.from_url()` runs below. `from_url()` mutates the
# `user_data` dict it receives in place (it injects a live `CrawleeRequestData` under `__crawlee`), and that
Expand Down Expand Up @@ -105,12 +110,8 @@ def to_apify_request(scrapy_request: ScrapyRequest, spider: Spider) -> ApifyRequ

# Store an Apify-platform view of the headers. The authoritative copy with exact bytes travels in
# the serialized scrapy_request below, so non-UTF-8 headers (which make `to_unicode_dict()` raise) are
# tolerated rather than dropping the whole request.
#
# Trade-off: with `use_extended_unique_key=True` the unique key includes the headers, so when non-UTF-8
# headers are omitted here two requests differing only in those headers share a unique key and one is
# deduplicated away. This is rare (header values are normally ASCII/UTF-8) and still strictly better than
# the old behavior, which dropped such requests entirely.
# omitted here rather than dropping the whole request. Dedup is unaffected: the unique key comes from
# Scrapy's fingerprint, which ignores headers.
if isinstance(scrapy_request.headers, Headers):
try:
headers = cast('dict[str, str]', dict(scrapy_request.headers.to_unicode_dict()))
Expand Down
41 changes: 41 additions & 0 deletions tests/unit/scrapy/requests/test_to_apify_request.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@
import pytest
from scrapy import Request, Spider
from scrapy.http.headers import Headers
from scrapy.utils.request import fingerprint

from crawlee._types import HttpHeaders

Expand Down Expand Up @@ -176,6 +177,46 @@ def test_dont_filter_request_is_always_enqueued(spider: Spider) -> None:
assert first.unique_key != second.unique_key


def test_unique_key_matches_scrapy_fingerprint(spider: Spider) -> None:
"""The computed unique key equals Scrapy's own request fingerprint, matching RFPDupeFilter dedup semantics."""
scrapy_request = Request(url='https://example.com/Products/Item-A?b=2&a=1')

apify_request = to_apify_request(scrapy_request, spider)

assert apify_request is not None
assert apify_request.unique_key == fingerprint(scrapy_request).hex()


def test_case_variant_urls_get_distinct_unique_keys(spider: Spider) -> None:
"""Case-variant paths must not collapse; Scrapy distinguishes them, so the unique keys must differ."""
upper = to_apify_request(Request(url='https://example.com/Products/Item-A'), spider)
lower = to_apify_request(Request(url='https://example.com/products/item-a'), spider)

assert upper is not None
assert lower is not None
assert upper.unique_key != lower.unique_key


def test_utm_params_kept_in_unique_key(spider: Spider) -> None:
"""`utm_*` tracking params must not be stripped; Scrapy keeps them, so the two URLs get distinct keys."""
with_utm = to_apify_request(Request(url='https://example.com/x?utm_source=x&id=1'), spider)
without_utm = to_apify_request(Request(url='https://example.com/x?id=1'), spider)

assert with_utm is not None
assert without_utm is not None
assert with_utm.unique_key != without_utm.unique_key


def test_header_only_difference_shares_unique_key(spider: Spider) -> None:
"""Scrapy ignores headers when deduplicating, so requests differing only in headers share a unique key."""
en = to_apify_request(Request(url='https://example.com', headers={'Accept-Language': 'en'}), spider)
de = to_apify_request(Request(url='https://example.com', headers={'Accept-Language': 'de'}), spider)

assert en is not None
assert de is not None
assert en.unique_key == de.unique_key


def test_apify_request_id_in_meta_is_ignored(spider: Spider) -> None:
"""An `apify_request_id` in `meta` is ignored and does not break conversion; the unique key still applies."""
scrapy_request = Request(
Expand Down
Loading