From 2f42739ed6d1871e42021602811379acc46a42bc Mon Sep 17 00:00:00 2001 From: Max Bohomolov Date: Wed, 30 Sep 2026 15:37:41 +0000 Subject: [PATCH 1/8] parse HTML responses starting with an XML declaration as HTML in `ParselCrawler` --- .../fill_and_submit_web_form_automated.py | 39 + docs/examples/fill_and_submit_web_form.mdx | 21 +- .../scrapy_migration/crawlee_post.py | 36 +- docs/guides/scrapy_migration.mdx | 4 +- src/crawlee/_utils/html.py | 422 ++++++++++- .../_beautifulsoup_crawling_context.py | 108 ++- .../_parsel/_parsel_crawling_context.py | 66 +- tests/unit/_utils/test_html.py | 716 +++++++++++++++++- .../test_beautifulsoup_crawler.py | 24 +- .../crawlers/_parsel/test_parsel_crawler.py | 22 + 10 files changed, 1418 insertions(+), 40 deletions(-) create mode 100644 docs/examples/code_examples/fill_and_submit_web_form_automated.py diff --git a/docs/examples/code_examples/fill_and_submit_web_form_automated.py b/docs/examples/code_examples/fill_and_submit_web_form_automated.py new file mode 100644 index 0000000000..844b0d9fe3 --- /dev/null +++ b/docs/examples/code_examples/fill_and_submit_web_form_automated.py @@ -0,0 +1,39 @@ +import asyncio + +from crawlee.crawlers import ParselCrawler, ParselCrawlingContext + + +async def main() -> None: + crawler = ParselCrawler() + + # Fill in the form on the page and enqueue its submission. + @crawler.router.default_handler + async def request_handler(context: ParselCrawlingContext) -> None: + context.log.info(f'Filling in the form on {context.request.url} ...') + requests = await context.extract_form_requests( + fields={ + 'custname': 'John Doe', + 'custtel': '1234567890', + 'custemail': 'johndoe@example.com', + 'size': 'large', + 'topping': ['bacon', 'cheese', 'mushroom'], + 'delivery': '13:00', + 'comments': 'Please ring the doorbell upon arrival.', + }, + label='form-result', + ) + await context.add_requests(requests) + + # Process the response to the form submission. + @crawler.router.handler('form-result') + async def form_result_handler(context: ParselCrawlingContext) -> None: + context.log.info(f'Processing {context.request.url} ...') + response = (await context.http_response.read()).decode('utf-8') + context.log.info(f'Response: {response}') # To see the response in the logs. + + # Run the crawler with the page containing the form. + await crawler.run(['https://httpbin.org/forms/post']) + + +if __name__ == '__main__': + asyncio.run(main()) diff --git a/docs/examples/fill_and_submit_web_form.mdx b/docs/examples/fill_and_submit_web_form.mdx index bda46c1d97..e81af4623f 100644 --- a/docs/examples/fill_and_submit_web_form.mdx +++ b/docs/examples/fill_and_submit_web_form.mdx @@ -10,8 +10,9 @@ import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; import RequestExample from '!!raw-loader!roa-loader!./code_examples/fill_and_submit_web_form_request.py'; import CrawlerExample from '!!raw-loader!roa-loader!./code_examples/fill_and_submit_web_form_crawler.py'; +import AutomatedExample from '!!raw-loader!roa-loader!./code_examples/fill_and_submit_web_form_automated.py'; -This example demonstrates how to fill and submit a web form using the `HttpCrawler` crawler. The same approach applies to any crawler that inherits from it, such as the `BeautifulSoupCrawler` or `ParselCrawler`. +This example demonstrates how to fill and submit a web form using the `HttpCrawler` crawler. The same approach applies to any crawler that inherits from it, such as the `BeautifulSoupCrawler` or `ParselCrawler`. These two crawlers can also [fill in the form automatically](#fill-in-the-form-automatically). We are going to use the [httpbin.org](https://httpbin.org) website to demonstrate how it works. @@ -118,3 +119,21 @@ Finally, run your crawler. Your logs should show something like this: ``` This log output confirms that the crawler successfully submitted the form and processed the response. Congratulations! You have successfully filled and submitted a web form using the `HttpCrawler`. + +## Fill in the form automatically + +The `ParselCrawler` and `BeautifulSoupCrawler` can build the form request for you. Their crawling contexts provide the `extract_form_requests` helper, which reads the form from the page, fills in your values and returns a list with the request that submits it the way a browser does. The action URL, the method and the encoding come from the form itself, so you only need the field names from [Investigate the form fields](#investigate-the-form-fields). + +The crawler below opens the page with the form. The default handler fills in the form with the `fields` argument and enqueues the submission with a label. A separate handler for that label processes the response. + + + {AutomatedExample} + + +Note that: + +- `fields` replaces the values of the listed fields and adds the ones the form doesn't have. A list submits the field once per value, as with the `topping` checkboxes. +- Fields you don't list keep the values from the page, so hidden inputs such as CSRF tokens are submitted as they are. A CSRF token is tied to the session cookie, so pass `session_id=context.session.id` to send the form in the same session. The request also carries the `Referer` and `Origin` headers a browser sends, which some CSRF checks require. +- On a page with several forms, the helper submits the one sharing the most field names with `fields`, or the first one if none shares any. It skips forms that can't be submitted, for example because their action is JavaScript, but never falls back to a form sharing fewer names, so the list can be empty. To pick a form yourself, pass a CSS selector such as `selector='#order'`. To submit each form, pass `all_forms=True`. Then `fields` only replaces the fields each form has. +- The first enabled submit button of the form is clicked by default, and a form without one is submitted anyway. Use the `click` argument to pick another button by its attributes, even a disabled one, or to submit without one. +- The page decides where its form is sent. To enqueue only requests to the same host, call `context.add_requests(requests, strategy='same-hostname')`. diff --git a/docs/guides/code_examples/scrapy_migration/crawlee_post.py b/docs/guides/code_examples/scrapy_migration/crawlee_post.py index 2f332426ef..b97a77552e 100644 --- a/docs/guides/code_examples/scrapy_migration/crawlee_post.py +++ b/docs/guides/code_examples/scrapy_migration/crawlee_post.py @@ -1,7 +1,5 @@ import asyncio -from urllib.parse import urlencode -from crawlee import Request from crawlee.crawlers import ParselCrawler, ParselCrawlingContext @@ -14,34 +12,16 @@ async def login_page(context: ParselCrawlingContext) -> None: if not context.session: raise RuntimeError('Session not found') - token = context.selector.css('input[name="csrf_token"]::attr(value)').get() - - # The CSRF token is required for the POST to succeed. If it's missing, - # the login will fail. - if not token: - raise RuntimeError('CSRF token not found') - - form = {'csrf_token': token, 'username': 'user', 'password': 'pass'} - # highlight-start - # Crawlee's `payload` is the raw request body, so encode the fields yourself - # and set the `Content-Type`. Scrapy's `FormRequest` does both for you. - await context.add_requests( - [ - Request.from_url( - 'https://quotes.toscrape.com/login', - method='POST', - payload=urlencode(form), - headers={'content-type': 'application/x-www-form-urlencoded'}, - label='after-login', - # Bind the POST to the same session so its CSRF cookie matches. - session_id=context.session.id, - # The POST shares the GET's URL. Include the method and payload - # in the unique key, or the queue drops it as a duplicate. - use_extended_unique_key=True, - ) - ] + # Like Scrapy's `FormRequest.from_response`, the helper keeps the hidden + # `csrf_token` field, encodes the data and sets the `Content-Type` header. + requests = await context.extract_form_requests( + fields={'username': 'user', 'password': 'pass'}, + label='after-login', + # Bind the POST to the same session so its CSRF cookie matches. + session_id=context.session.id, ) + await context.add_requests(requests) # highlight-end @crawler.router.handler('after-login') diff --git a/docs/guides/scrapy_migration.mdx b/docs/guides/scrapy_migration.mdx index 5acbfa3824..a5e4c29a9e 100644 --- a/docs/guides/scrapy_migration.mdx +++ b/docs/guides/scrapy_migration.mdx @@ -80,7 +80,7 @@ Both frameworks give you a request scheduler, filtering of duplicate requests, r | `response.follow()` / `yield Request(...)` | `enqueue_links` / `add_requests` | | `dont_filter=True` | `Request.from_url(always_enqueue=True)` | | `allowed_domains` | `enqueue_links(strategy=...)` | -| `scrapy.FormRequest` | `Request.from_url(method='POST', payload=...)` | +| `scrapy.FormRequest` | `Request.from_url(method='POST', payload=...)` / `context.extract_form_requests(...)` | | Item pipelines | `Dataset` | | Downloader / spider middlewares | `router.use()`, navigation hooks, HTTP clients | | `settings.py` | `Configuration` + crawler arguments | @@ -263,7 +263,7 @@ Scrapy retries failed requests with `RetryMiddleware` and reports terminal failu ## Forms and login -Scrapy submits forms with `FormRequest`, which encodes `formdata` as `form-urlencoded` and sets the header for you. Crawlee's `payload` takes the raw request body, so encode the fields yourself with `urllib.parse.urlencode` and set the `Content-Type` through `headers=`. For a full login flow with session reuse, see the [Logging in with a crawler guide](./logging-in-with-a-crawler). +Scrapy submits forms with `FormRequest.from_response`, which reads the form from the page, keeps its hidden fields and encodes the data for you. Crawlee's `extract_form_requests` helper does the same in the `ParselCrawler` and `BeautifulSoupCrawler`. Pass your values in `fields` and request options such as `label` or `session_id` as keyword arguments. Like `from_response`, it submits a single form. Scrapy takes the first form by default, while the helper prefers the one sharing the most field names with `fields`. It returns a list, which is empty when no form matches, so enqueue it with `add_requests`. For a plain `FormRequest` that doesn't come from a form on the page, use `Request.from_url`. Its `payload` is the raw request body, so encode the fields with `urllib.parse.urlencode` and set the `Content-Type` through `headers=`. For a full login flow with session reuse, see the [Logging in with a crawler guide](./logging-in-with-a-crawler). diff --git a/src/crawlee/_utils/html.py b/src/crawlee/_utils/html.py index 5b357f312a..9eef8d8910 100644 --- a/src/crawlee/_utils/html.py +++ b/src/crawlee/_utils/html.py @@ -4,8 +4,24 @@ import codecs import re +from typing import TYPE_CHECKING, NamedTuple, TypedDict +from urllib.parse import urlencode, urlsplit +from yarl import URL + +from crawlee._request import Request +from crawlee._types import HttpHeaders +from crawlee._utils.crypto import compute_short_hash from crawlee._utils.http import parse_content_type_charset +from crawlee._utils.urls import convert_to_absolute_url, is_url_absolute, validate_http_url + +if TYPE_CHECKING: + from collections.abc import Mapping, Sequence + + from lxml.html import HtmlElement + from typing_extensions import NotRequired, Unpack + + from crawlee._types import JsonSerializable # Matches the `encoding` of an XML declaration, which XHTML pages may use instead of a `` tag. _XML_ENCODING_PATTERN = re.compile(rb'^\s*<\?xml[^>]*\sencoding\s*=\s*["\']([a-z0-9_:.+-]+)', re.IGNORECASE) @@ -90,6 +106,55 @@ _ENCODING_BY_LABEL = {label: codec for codec, labels in _WHATWG_ENCODING_LABELS.items() for label in labels.split()} +_FIELD_TAGS = ('input', 'button', 'select', 'textarea') +_BUTTON_INPUT_TYPES = ('submit', 'image', 'reset', 'button') + +# Browsers cut a longer referrer down to the origin. +_MAX_REFERRER_LENGTH = 4096 + + +class FormRequestOptions(TypedDict): + """Options for the `Request` created from a form. + + Mirrors `RequestOptions` without the URL, method and payload, which come from the form, without `id` and + `unique_key`, which can't be shared by several forms, and without `enqueue_strategy`, which enqueuing sets. + """ + + label: NotRequired[str | None] + """A label routing the request to a specific handler.""" + + headers: NotRequired[HttpHeaders | dict[str, str] | None] + """HTTP headers of the request, replacing those the form sets, like `Content-Type` or `Referer`.""" + + session_id: NotRequired[str | None] + """ID of the `Session` the request is bound to.""" + + keep_url_fragment: NotRequired[bool] + """Whether the URL fragment counts towards the unique key of the request.""" + + use_extended_unique_key: NotRequired[bool] + """Whether the method and payload count towards the unique key. Defaults to `True` for POST forms.""" + + always_enqueue: NotRequired[bool] + """Whether to enqueue the request even if it's already in the queue.""" + + user_data: NotRequired[Mapping[str, JsonSerializable]] + """Custom data stored with the request.""" + + no_retry: NotRequired[bool] + """Whether to skip retrying the request if it fails.""" + + max_retries: NotRequired[int | None] + """The maximum number of retries of the request.""" + + +class _Field(NamedTuple): + """A single entry the form submits.""" + + name: str + value: str + is_file: bool = False + def get_declared_html_encoding(body: bytes, content_type: str | None) -> str | None: """Get the Python codec for the encoding an HTML response body declares. @@ -110,7 +175,7 @@ def get_declared_html_encoding(body: bytes, content_type: str | None) -> str | N return encoding header_charset = parse_content_type_charset(content_type) - return _resolve_encoding(header_charset) or _find_declared_encoding(body) + return resolve_encoding(header_charset) or _find_declared_encoding(body) def decode_html_body(body: bytes, encoding: str) -> str: @@ -125,18 +190,94 @@ def decode_html_body(body: bytes, encoding: str) -> str: return body.decode(encoding, 'replace').removeprefix('\ufeff') -def _resolve_encoding(label: str | None) -> str | None: +def resolve_encoding(label: str | None) -> str | None: """Get the Python codec for a WHATWG encoding label, or `None` if browsers don't know the label.""" if not label: return None return _ENCODING_BY_LABEL.get(label.lower()) +def strip_html_comments(body: bytes) -> bytes: + """Remove the HTML comments from the body, so a commented-out declaration doesn't count.""" + return _HTML_COMMENT_PATTERN.sub(b'', body) + + +def forms_to_requests( + forms: Sequence[HtmlElement], + page_url: str, + page_encoding: str, + *, + fields: Mapping[str, str | Sequence[str] | None] | None = None, + click: bool | Mapping[str, str] = True, + all_forms: bool = False, + **kwargs: Unpack[FormRequestOptions], +) -> list[Request]: + """Create a `Request` submitting one of the forms, or each of them, the way a browser does. + + Args: + forms: The form elements, each within the lxml tree of the whole page. + page_url: The URL of the page, which forms without an action submit to and the `Referer` comes from. + page_encoding: The Python codec the page was decoded with, which forms submit in by default. + fields: Field values to submit, see `extract_form_requests` of the crawling contexts. + click: The submit button to click, see `extract_form_requests`. + all_forms: Whether to submit each form, filling in only the fields it has, see `extract_form_requests`. + **kwargs: Additional options passed to `Request.from_url`. + """ + if not forms: + return [] + + root = forms[0].getroottree().getroot() + base = root.find('.//base[@href]') + try: + base_url = convert_to_absolute_url(page_url, '' if base is None else base.get('href')) + except ValueError: + base_url = page_url + + page_elements = list(root.iter(*_FIELD_TAGS)) + elements_by_form = _elements_by_form(root, page_elements) + disabled = _disabled_elements(root, page_elements) + + if not all_forms: + # The form sharing the most field names with `fields` is the one to fill in, and ties keep document order. + # Forms sharing fewer names never get the values, even if the best one can't be submitted. + wanted = set(fields or ()) + shared = {form: len(wanted & _field_names(elements_by_form.get(form, []))) for form in forms} + most_shared = max(shared.values()) + forms = [form for form in forms if shared[form] == most_shared] + + requests = [] + for form in forms: + elements = elements_by_form.get(form, []) + form_fields = fields + if all_forms and fields: + # Values meant for one form don't spread to the others, like credentials into a search form. + names = _field_names(elements) + form_fields = {name: value for name, value in fields.items() if name in names} + + request = _form_to_request( + form, + elements, + page_url=page_url, + base_url=base_url, + page_encoding=page_encoding, + disabled=disabled, + fields=form_fields, + click=click, + **kwargs, + ) + if request is None: + continue + requests.append(request) + if not all_forms: + break + return requests + + def _find_declared_encoding(body: bytes) -> str | None: """Find the encoding declared by an XML declaration or a `` tag near the start of the body.""" - prescan = _HTML_COMMENT_PATTERN.sub(b'', body[:_PRESCAN_BYTES]) + prescan = strip_html_comments(body[:_PRESCAN_BYTES]) xml_match = _XML_ENCODING_PATTERN.match(prescan) - xml_encoding = _resolve_encoding(xml_match.group(1).decode('ascii')) if xml_match else None + xml_encoding = resolve_encoding(xml_match.group(1).decode('ascii')) if xml_match else None encoding = xml_encoding or _find_meta_encoding(prescan) # A declaration readable as ASCII rules out UTF-16, so browsers read such pages as UTF-8. return 'utf-8' if encoding and encoding.startswith('utf-16') else encoding @@ -161,7 +302,278 @@ def _find_meta_encoding(prescan: bytes) -> str | None: else: continue - encoding = _resolve_encoding(label) + encoding = resolve_encoding(label) if encoding: return encoding return None + + +def _elements_by_form(root: HtmlElement, elements: list[HtmlElement]) -> dict[HtmlElement, list[HtmlElement]]: + """Group the fields and buttons of the page by the form they belong to, in document order.""" + # Walking each form avoids an ancestor walk per field. + enclosing_form: dict[HtmlElement, HtmlElement] = {} + for form in root.iter('form'): + enclosing_form.update(dict.fromkeys(form.iter(*_FIELD_TAGS), form)) + + # The first element with a given ID wins, as in `getElementById`. Only the `form` attribute needs them. + elements_by_id: dict[str, HtmlElement] = {} + if any(element.get('form') is not None for element in elements): + for element in root.xpath('//*[@id!=""]'): + elements_by_id.setdefault(element.get('id'), element) + + elements_by_form: dict[HtmlElement, list[HtmlElement]] = {} + for element in elements: + form_id = element.get('form') + owner = enclosing_form.get(element) if form_id is None else elements_by_id.get(form_id) + if owner is not None and owner.tag == 'form': + elements_by_form.setdefault(owner, []).append(element) + return elements_by_form + + +def _disabled_elements(root: HtmlElement, elements: list[HtmlElement]) -> set[HtmlElement]: + """Find the disabled fields and buttons, including those inside a disabled `
`.""" + disabled = {element for element in elements if 'disabled' in element.attrib} + for fieldset in root.iter('fieldset'): + # A fieldset inside a disabled one is already covered, so each element is visited once. + if 'disabled' in fieldset.attrib and fieldset not in disabled: + disabled.update(fieldset.iter('fieldset', *_FIELD_TAGS)) + return disabled + + +def _field_names(elements: list[HtmlElement]) -> set[str]: + """Get the names of the given fields and buttons.""" + return {name for element in elements if (name := element.get('name'))} + + +def _form_to_request( + form: HtmlElement, + elements: list[HtmlElement], + *, + page_url: str, + base_url: str, + page_encoding: str, + disabled: set[HtmlElement], + fields: Mapping[str, str | Sequence[str] | None] | None, + click: bool | Mapping[str, str], + **kwargs: Unpack[FormRequestOptions], +) -> Request | None: + """Create a `Request` submitting a single form, or `None` if a browser wouldn't send one.""" + button = None + if click is not False: + button = _find_clickable(elements, {} if click is True else click, disabled) + if click is not True and button is None: + # Pages often enable a button with JavaScript, so a disabled one matching `click` is clicked too. + button = _find_clickable(elements, click, set()) + if button is None: + return None + + method = (_submission_attribute(form, button, 'method') or 'get').upper() + enctype = _submission_attribute(form, button, 'enctype').lower() + action = _submission_attribute(form, button, 'action').strip() + + # A dialog form only closes its `` on the client. + if method == 'DIALOG': + return None + + url = _resolve_action(page_url, base_url, action) + if url is None: + return None + + entries = _collect_fields(elements, button, fields, disabled) + encoding = _form_encoding(form, page_encoding) + # CSRF checks may require the `Referer` or `Origin` a browser sends. Headers passed in `headers` win. + form_headers = _referrer_headers(page_url, url, method) + + if method != 'POST': + query = urlencode( + [(entry.name, entry.value) for entry in entries], encoding=encoding, errors='xmlcharrefreplace' + ) + kwargs['headers'] = HttpHeaders(form_headers) | HttpHeaders(kwargs.get('headers') or {}) + return Request.from_url(urlsplit(url)._replace(query=query).geturl(), method='GET', **kwargs) + + payload, form_headers['Content-Type'] = _encode_body(entries, enctype, encoding) + kwargs['headers'] = HttpHeaders(form_headers) | HttpHeaders(kwargs.get('headers') or {}) + kwargs.setdefault('use_extended_unique_key', True) + return Request.from_url(url, method='POST', payload=payload, **kwargs) + + +def _find_clickable( + elements: list[HtmlElement], attributes: Mapping[str, str], disabled: set[HtmlElement] +) -> HtmlElement | None: + """Find the first submit button having all the given attributes, skipping the `disabled` ones.""" + for element in elements: + if not _is_submit_button(element) or element in disabled: + continue + if all(element.get(key) == value for key, value in attributes.items()): + return element + return None + + +def _is_submit_button(element: HtmlElement) -> bool: + """Check whether the element submits the form when clicked.""" + button_type = element.get('type', '').lower() + if element.tag == 'button': + return button_type in ('', 'submit') + if element.tag == 'input': + return button_type in ('submit', 'image') + return False + + +def _submission_attribute(form: HtmlElement, button: HtmlElement | None, name: str) -> str: + """Get a form attribute like `action`, which the clicked button can override with its `form*` counterpart.""" + if button is not None: + button_value = button.get(f'form{name}') + if button_value: + return button_value + return form.get(name) or '' + + +def _resolve_action(page_url: str, base_url: str, action: str) -> str | None: + """Resolve the form action to an absolute URL, or `None` if it isn't a valid HTTP(S) URL.""" + if not action: + return page_url + + try: + url = convert_to_absolute_url(base_url, action) + validate_http_url(url) + except ValueError: + return None + # `validate_http_url` accepts a URL without a host, like `http:x` resolved against an HTTPS page. + return url if is_url_absolute(url) else None + + +def _collect_fields( + elements: list[HtmlElement], + button: HtmlElement | None, + fields: Mapping[str, str | Sequence[str] | None] | None, + disabled: set[HtmlElement], +) -> list[_Field]: + """Collect the entries the form submits: its fields, the clicked button and the `fields` overrides.""" + # Only one radio button of a group is checked, the last one in the markup. + checked_radios = {element.get('name'): element for element in elements if _is_radio(element) and element.checked} + + entries: list[_Field] = [] + for element in elements: + if element is button: + entries.extend(_button_fields(button)) + continue + if not element.get('name') or element in disabled: + continue + if _is_radio(element) and checked_radios.get(element.get('name')) is not element: + continue + entries.extend(_element_fields(element)) + + return _apply_fields(entries, fields or {}) + + +def _is_radio(element: HtmlElement) -> bool: + """Check whether the element is a radio button.""" + return element.tag == 'input' and element.type == 'radio' + + +def _button_fields(button: HtmlElement) -> list[_Field]: + """Get the entries the clicked button adds.""" + name = button.get('name') + + # An image button sends the click coordinates instead of its value. + if button.get('type', '').lower() == 'image': + prefix = f'{name}.' if name else '' + return [_Field(f'{prefix}x', '0'), _Field(f'{prefix}y', '0')] + + if name: + return [_Field(name, button.get('value', ''))] + return [] + + +def _element_fields(element: HtmlElement) -> list[_Field]: + """Get the entries a single enabled, named field submits.""" + name = element.get('name') + + if element.tag == 'select': + values = element.value if element.multiple else [element.value] + return [_Field(name, value) for value in values if value is not None] + + if element.tag == 'textarea': + # Browsers drop the newline right after ` + +
+ + + + + + + + """ + + [request] = await extract_form_requests(html) + + assert _submitted_fields(request) == { + 'text': 't', + 'checked': 'c1', + 'no-value': 'on', + 'radio': 'r2', + 'first': 'o1', + 'last': 'b', + 'first-enabled': 'on', + 'multi': ['m1', 'm3'], + 'area': '\nlong text', + 'file': '', + 'empty': '', + 'hidden': '', + } + + +async def test_form_attribute(extract_form_requests: ExtractFormRequests) -> None: + """Fields are assigned to forms by their `form` attribute, and `fields` fill in only the fields a form has.""" + html = """ + + + +
+
+ """ + + search, login, empty = await extract_form_requests(html, fields={'q': 'y', 'other': 'p'}, all_forms=True) + + assert _submitted_fields(search) == {'q': 'y', 'lang': 'en', 'go': ''} + assert _submitted_fields(login) == {'other': 'p'} + assert _submitted_fields(empty) == {} + + +async def test_disabled_fieldset(extract_form_requests: ExtractFormRequests) -> None: + """A disabled `
` disables all fields inside it, and an enabled one none.""" + html = """ +
+
+
+
+ """ + + [request] = await extract_form_requests(html) + + assert _submitted_fields(request) == {'c': ''} + + +async def test_fields_argument(extract_form_requests: ExtractFormRequests) -> None: + """Values from `fields` replace, drop and add fields.""" + html = '
' + + [request] = await extract_form_requests(html, fields={'replace': 'new', 'drop': None, 'tags': ['a', 'b']}) + + assert _submitted_fields(request) == {'keep': 'k', 'replace': 'new', 'tags': ['a', 'b']} + + +@pytest.mark.parametrize( + ('html', 'expected_query'), + [ + pytest.param('
', 'q=x', id='unknown-method-is-get'), + pytest.param( + '
', 'q=%C3%A9', id='utf-16-submits-utf-8' + ), + pytest.param( + '
', + 'q=x', + id='first-duplicate-id-wins', + ), + pytest.param( + '
', + 'q=x', + id='first-id-not-a-form', + ), + pytest.param('
', 'q.x=0&q.y=0', id='uppercase-tags'), + pytest.param( + '
', + 'q=x', + id='empty-form-attribute', + ), + pytest.param('
', 'q=x', id='unnamed-button'), + pytest.param('
', 'map.x=0&map.y=0', id='image-button'), + pytest.param('
', 'x=0&y=0', id='unnamed-image-button'), + ], +) +async def test_browser_rules(extract_form_requests: ExtractFormRequests, html: str, expected_query: str) -> None: + """Less obvious browser rules are followed.""" + [request] = await extract_form_requests(html) + + assert request.method == 'GET' + assert urlsplit(request.url).query == expected_query + + +@pytest.mark.parametrize( + ('html', 'expected_url'), + [ + pytest.param('
', 'https://example.com/page/index.html', id='no-action'), + pytest.param('
', 'https://example.com/up', id='relative'), + pytest.param('
', 'https://example.com/go', id='whitespace-around-action'), + pytest.param( + '
', + 'https://other.com/dir/go', + id='base-href', + ), + pytest.param( + '
', + 'https://example.com/page/index.html', + id='base-href-no-action', + ), + pytest.param( + '
', + 'https://example.com/page/go', + id='invalid-base-href', + ), + pytest.param( + '
', + 'https://example.com/page/go', + id='base-without-href', + ), + ], +) +async def test_action_resolution(extract_form_requests: ExtractFormRequests, html: str, expected_url: str) -> None: + """The action is resolved against ``, and a missing action submits to the page URL.""" + [request] = await extract_form_requests(html) + + assert request.url == expected_url + + +async def test_action_uses_loaded_url(extract_form_requests: ExtractFormRequests) -> None: + """The action is resolved against the URL the page was loaded from after redirects.""" + page_request = Request.from_url(_PAGE_URL, loaded_url='https://example.com/moved/index.html') + + [request] = await extract_form_requests('
', page_request=page_request) + + assert request.url == 'https://example.com/moved/go' + + +async def test_unsubmittable_forms_skipped(extract_form_requests: ExtractFormRequests) -> None: + """Dialog forms and forms whose action isn't a valid HTTP(S) URL are skipped.""" + html = ( + '
' + '
' + '
' + ) + + assert _paths(await extract_form_requests(html, all_forms=True)) == ['/ok'] + + +async def test_request_options(extract_form_requests: ExtractFormRequests) -> None: + """Request options are applied, and `headers` are merged over the ones the form sets.""" + html = '
' + headers = {'X-Custom': '1', 'referer': 'https://example.com/'} + + [request] = await extract_form_requests(html, headers=headers, label='detail', use_extended_unique_key=False) + [get_request] = await extract_form_requests(html.replace('post', 'get'), headers=headers) + + assert request.label == 'detail' + assert request.unique_key == 'https://example.com/login' + assert request.headers['x-custom'] == '1' + assert request.headers['content-type'] == 'application/x-www-form-urlencoded' + assert request.headers['referer'] == 'https://example.com/' + assert get_request.headers['referer'] == 'https://example.com/' + + +@pytest.mark.parametrize( + ('page_url', 'html', 'expected_referer', 'expected_origin'), + [ + pytest.param( + _REFERRER_PAGE_URL, + '
', + 'https://example.com/page?token=1', + 'https://example.com', + id='same-origin', + ), + pytest.param( + _REFERRER_PAGE_URL, + '', + 'https://example.com/', + 'https://example.com', + id='other-host', + ), + pytest.param( + _REFERRER_PAGE_URL, + '', + 'https://example.com/', + 'https://example.com', + id='other-port', + ), + pytest.param( + 'https://example.com:443/page', + '', + 'https://example.com/page', + 'https://example.com', + id='default-port', + ), + pytest.param( + 'https://example.com', + '', + 'https://example.com/', + 'https://example.com', + id='empty-path', + ), + pytest.param( + _REFERRER_PAGE_URL, + '', + 'https://example.com/', + 'https://example.com', + id='base-href-ignored', + ), + pytest.param( + _REFERRER_PAGE_URL, '', None, 'null', id='https-to-http' + ), + pytest.param( + 'http://example.com/page', + '', + 'http://example.com/', + 'http://example.com', + id='http-to-https', + ), + pytest.param( + _REFERRER_PAGE_URL, + '', + 'https://example.com/page?token=1', + None, + id='get-without-origin', + ), + pytest.param( + f'https://example.com/page?q={"a" * 4096}', + '', + 'https://example.com/', + None, + id='long-referrer-cut', + ), + ], +) +async def test_referrer_headers( + extract_form_requests: ExtractFormRequests, + page_url: str, + html: str, + expected_referer: str | None, + expected_origin: str | None, +) -> None: + """The `Referer` and `Origin` headers come from the page URL under the default referrer policy of browsers.""" + [request] = await extract_form_requests(f'{html}
', page_request=Request.from_url(page_url)) + + assert request.headers.get('referer') == expected_referer + assert request.headers.get('origin') == expected_origin + + +async def test_click_first_button(extract_form_requests: ExtractFormRequests) -> None: + """The first enabled submit button is submitted in document order by default, and none with `click=False`.""" + html = """ +
+ + +
+ + + +
+ """ + + [request] = await extract_form_requests(html) + [unclicked] = await extract_form_requests(html, click=False) + + assert urlsplit(request.url).query == 'go=first&q=x' + assert urlsplit(unclicked.url).query == 'q=x' + + +async def test_click_button_overrides(extract_form_requests: ExtractFormRequests) -> None: + """`click` picks the button by its attributes and the button overrides apply.""" + html = """ +
+ + + +
+ """ + + [request] = await extract_form_requests(html, click={'value': 'delete'}) + + assert request.method == 'POST' + assert request.url == 'https://example.com/delete' + assert _submitted_fields(request) == {'q': 'x', 'action': 'delete'} + + +async def test_click_disabled_button(extract_form_requests: ExtractFormRequests) -> None: + """`click` picks a disabled button only if no enabled one matches.""" + html = ( + '
' + ) + + [enabled] = await extract_form_requests(html, click={'name': 'go'}) + [disabled] = await extract_form_requests(html, click={'value': 'back'}) + + assert _submitted_fields(enabled) == {'go': 'next'} + assert _submitted_fields(disabled) == {'go': 'back'} + + +@pytest.mark.parametrize( + ('click', 'expected_paths'), + [ + pytest.param({'name': 'log-in'}, ['/login'], id='by-name'), + pytest.param({'id': 'find-button'}, ['/search'], id='by-id'), + pytest.param({'name': 'missing'}, [], id='no-match'), + pytest.param({}, ['/search'], id='any-button'), + ], +) +async def test_click_picks_form( + extract_form_requests: ExtractFormRequests, click: dict[str, str], expected_paths: list[str] +) -> None: + """A `click` mapping submits the form with a matching button.""" + html = """ +
+
+
+ """ + + assert _paths(await extract_form_requests(html, click=click)) == expected_paths + + +async def test_selector(extract_form_requests: ExtractFormRequests) -> None: + """`selector` narrows the forms, skipping other elements it matches.""" + html = '
' + + assert _paths(await extract_form_requests(html, selector='#c, #b')) == ['/c'] + + +@pytest.mark.parametrize( + 'selector', + [ + pytest.param('#missing', id='no-match'), + pytest.param('form::text, form::attr(action)', id='text-and-attributes'), + pytest.param('form[', id='syntax-error'), + pytest.param('form:unknown', id='unknown-pseudo-class'), + pytest.param('ns|form', id='unknown-namespace'), + pytest.param('#\\110000', id='escape-past-unicode'), + ], +) +async def test_selector_without_forms(extract_form_requests: ExtractFormRequests, selector: str) -> None: + """A selector that is invalid or matches no form yields no requests.""" + assert await extract_form_requests('
text
', selector=selector) == [] + + +@pytest.mark.parametrize( + ('html', 'fields', 'expected_paths'), + [ + pytest.param( + '
' + '
', + {'email': 'me@example.com', 'password': 'secret'}, + ['/login'], + id='most-shared-names', + ), + pytest.param( + '
', + {'q': 'shoes'}, + ['/header'], + id='tie-keeps-document-order', + ), + pytest.param( + '
', + {'token': 'x'}, + ['/a'], + id='no-shared-name-first-form', + ), + pytest.param( + '
', + {}, + ['/ok'], + id='unsubmittable-forms-passed-over', + ), + pytest.param( + '
' + '
', + {'username': 'me', 'password': 'secret'}, + [], + id='best-form-unsubmittable', + ), + ], +) +async def test_form_choice( + extract_form_requests: ExtractFormRequests, html: str, fields: dict[str, str], expected_paths: list[str] +) -> None: + """The first submittable form among those sharing the most field names with `fields` is submitted.""" + assert _paths(await extract_form_requests(html, fields=fields)) == expected_paths + + +@pytest.mark.parametrize( + 'html', + [ + pytest.param('

No forms here.

', id='no-form'), + pytest.param('', id='commented-out-form'), + ], +) +async def test_page_without_forms(extract_form_requests: ExtractFormRequests, html: str) -> None: + """A page without forms yields no requests.""" + assert await extract_form_requests(html) == [] + + +async def test_xml_declaration_page(extract_form_requests: ExtractFormRequests) -> None: + """An HTML page starting with an XML declaration is read as HTML.""" + html = '
' + + assert _paths(await extract_form_requests(html, content_type='text/html')) == ['/ok'] + + +async def test_form_after_html_end(extract_form_requests: ExtractFormRequests) -> None: + """A form after `` doesn't shift the forms before it.""" + html = '
' + + # Newer libxml2 puts content after `` into a second root, which isn't searched, older versions keep it. + assert _paths(await extract_form_requests(html, all_forms=True)) in (['/ok'], ['/ok', '/after']) + assert _paths(await extract_form_requests(html, selector='#ok')) == ['/ok'] + assert _paths(await extract_form_requests(html, selector='[action="/after"]')) in ([], ['/after']) + + +@pytest.mark.parametrize('parser', [pytest.param('lxml', id='lxml'), pytest.param('html5lib', id='html5lib')]) +async def test_beautifulsoup_form_inside_select(parser: BeautifulSoupParserType) -> None: + """Soup forms are matched to the right ones when the parser moves a form out of a `
' + + # With a declared encoding, a soup built by lxml is reused. + content_type = 'text/html; charset=utf-8' + requests = await _extract_with_beautifulsoup(html, content_type=content_type, parser=parser, selector='#y') + + assert _paths(requests) == ['/y'] + + +async def test_deeply_nested_form(extract_form_requests: ExtractFormRequests) -> None: + """Fields of a form nested deeper than the default libxml2 limit are collected.""" + html = '
' * 300 + '
' + '
' * 300 + + [request] = await extract_form_requests(html) + + assert _submitted_fields(request) == {'q': 'x'} + + +@pytest.mark.parametrize( + ('attributes', 'content_type'), + [ + pytest.param('accept-charset="bogus,windows-1250"', None, id='accept-charset'), + pytest.param('', 'text/html; charset=windows-1250', id='header'), + ], +) +@pytest.mark.parametrize('method', [pytest.param('get', id='get'), pytest.param('post', id='post')]) +async def test_submission_encoding( + extract_form_requests: ExtractFormRequests, attributes: str, content_type: str | None, method: str +) -> None: + """Values are encoded in the form or page charset, with unsupported characters as character references.""" + html = f'
' + + [request] = await extract_form_requests(html, content_type=content_type, fields={'q': 'Příliš ✓'}) + + encoded = urlsplit(request.url).query if method == 'get' else (request.payload or b'').decode() + assert encoded == 'q=P%F8%EDli%9A+%26%2310003%3B' + + +@pytest.mark.parametrize( + ('html', 'value', 'encoding'), + [ + pytest.param(f'

{_CZECH * 5}

', _CZECH, 'cp1250', id='guess'), + # `BeautifulSoup` names this encoding `CP932`, which isn't a WHATWG label. + pytest.param(f'

{_JAPANESE * 5}

', _JAPANESE, 'shift_jis', id='guess-outside-whatwg-labels'), + pytest.param(f'

{_CZECH * 5}

', _CZECH, 'cp1250', id='unknown-label'), + pytest.param(f'

{_CZECH * 5}

', _CZECH, 'cp1250', id='commented-out-meta'), + pytest.param( + f'

{_CZECH * 5}

', + _CZECH, + 'utf-8', + id='commented-out-meta-same-as-guess', + ), + # `BeautifulSoup` looks for a `` in the first 5% of a page, so it takes a large one. + pytest.param( + f'{"

text

" * 20_000}', + _CZECH, + 'utf-8', + id='late-meta', + ), + pytest.param('', _CZECH, 'cp1252', id='ascii'), + ], +) +async def test_beautifulsoup_undeclared_encoding(html: str, value: str, encoding: str) -> None: + """A page declaring no encoding in the prescan submits in the one browsers would pick.""" + page = f'{html}
' + + [request] = await _extract_with_beautifulsoup( + page, content_type='text/html', encoding=encoding, fields={'q': value} + ) + + assert urlsplit(request.url).query == urlencode({'q': value}, encoding=encoding, errors='xmlcharrefreplace') + + +async def test_beautifulsoup_guess_fallback() -> None: + """A page whose guessed encoding Python can't decode submits in UTF-8.""" + detector = Mock(encodings=['EUC-TW'], declared_encoding=None) + target = 'crawlee.crawlers._beautifulsoup._beautifulsoup_crawling_context.EncodingDetector' + + with patch(target, return_value=detector): + [request] = await _extract_with_beautifulsoup( + f'
', content_type='text/html' + ) + + assert urlsplit(request.url).query == urlencode({'q': _CZECH}) + + +async def test_beautifulsoup_undeclared_selector() -> None: + """A selector is matched on the page decoded as for the forms, not as `BeautifulSoup` decoded it.""" + # `BeautifulSoup` reads the page as UTF-7, which browsers don't know, and finds a form with the `first` ID. + html = '+ADw-form id=first+AD4-+ADw-/form+AD4-
' + + assert await _extract_with_beautifulsoup(html, content_type='text/html', selector='#first') == [] + + +async def test_parsel_undeclared_encoding() -> None: + """A page declaring no encoding submits in UTF-8 with Parsel.""" + html = '
' + + [request] = await _extract_with_parsel(html, content_type='text/html', fields={'q': _CZECH}) + + assert urlsplit(request.url).query == urlencode({'q': _CZECH}) + + +async def test_parsel_json_response() -> None: + """A JSON response yields no requests with Parsel.""" + body = '{"html": "
"}' + + assert await _extract_with_parsel(body, content_type='application/json') == [] diff --git a/tests/unit/crawlers/_beautifulsoup/test_beautifulsoup_crawler.py b/tests/unit/crawlers/_beautifulsoup/test_beautifulsoup_crawler.py index f3bac73206..1f332d8368 100644 --- a/tests/unit/crawlers/_beautifulsoup/test_beautifulsoup_crawler.py +++ b/tests/unit/crawlers/_beautifulsoup/test_beautifulsoup_crawler.py @@ -1,9 +1,10 @@ from __future__ import annotations import asyncio +import json import sys from datetime import timedelta -from typing import TYPE_CHECKING +from typing import TYPE_CHECKING, Any from unittest import mock import pytest @@ -314,6 +315,27 @@ async def request_handler(context: BeautifulSoupCrawlingContext) -> None: handler.assert_awaited_once_with([str(server_url.with_path(expected_path))]) +async def test_extract_form_requests(server_url: URL, http_client: HttpClient) -> None: + crawler = BeautifulSoupCrawler(http_client=http_client) + responses: list[dict[str, Any]] = [] + page = '
' + + @crawler.router.default_handler + async def request_handler(context: BeautifulSoupCrawlingContext) -> None: + requests = await context.extract_form_requests(fields={'q': 'x'}, label='result') + await context.add_requests(requests) + + @crawler.router.handler('result') + async def result_handler(context: BeautifulSoupCrawlingContext) -> None: + responses.append(json.loads(await context.http_response.read())) + + await crawler.run([str((server_url / 'echo_content').with_query(content=page))]) + + [response] = responses + assert response['form'] == {'token': 't', 'q': 'x'} + assert response['headers']['referer'].startswith(str(server_url / 'echo_content')) + + async def test_extract_non_href_links(server_url: URL, http_client: HttpClient) -> None: crawler = BeautifulSoupCrawler(http_client=http_client) extracted_links: list[str] = [] diff --git a/tests/unit/crawlers/_parsel/test_parsel_crawler.py b/tests/unit/crawlers/_parsel/test_parsel_crawler.py index 51613346c0..0bc960a4f4 100644 --- a/tests/unit/crawlers/_parsel/test_parsel_crawler.py +++ b/tests/unit/crawlers/_parsel/test_parsel_crawler.py @@ -1,6 +1,7 @@ from __future__ import annotations import codecs +import json import sys from typing import TYPE_CHECKING, Any from unittest import mock @@ -402,6 +403,27 @@ async def request_handler(context: ParselCrawlingContext) -> None: handler.assert_awaited_once_with([str(server_url.with_path(expected_path))]) +async def test_extract_form_requests(server_url: URL, http_client: HttpClient) -> None: + crawler = ParselCrawler(http_client=http_client) + responses: list[dict[str, Any]] = [] + page = '
' + + @crawler.router.default_handler + async def request_handler(context: ParselCrawlingContext) -> None: + requests = await context.extract_form_requests(fields={'q': 'x'}, label='result') + await context.add_requests(requests) + + @crawler.router.handler('result') + async def result_handler(context: ParselCrawlingContext) -> None: + responses.append(json.loads(await context.http_response.read())) + + await crawler.run([str((server_url / 'echo_content').with_query(content=page))]) + + [response] = responses + assert response['form'] == {'token': 't', 'q': 'x'} + assert response['headers']['referer'].startswith(str(server_url / 'echo_content')) + + async def test_extract_non_href_links(server_url: URL, http_client: HttpClient) -> None: crawler = ParselCrawler(http_client=http_client) extracted_links: list[str] = [] From a44838e5a9bf8a5c0087c4f8ee4eaf211e313943 Mon Sep 17 00:00:00 2001 From: Vlada Dusek Date: Tue, 6 Oct 2026 10:40:27 +0200 Subject: [PATCH 2/8] test: add docstrings to the crawler form tests --- tests/unit/crawlers/_beautifulsoup/test_beautifulsoup_crawler.py | 1 + tests/unit/crawlers/_parsel/test_parsel_crawler.py | 1 + 2 files changed, 2 insertions(+) diff --git a/tests/unit/crawlers/_beautifulsoup/test_beautifulsoup_crawler.py b/tests/unit/crawlers/_beautifulsoup/test_beautifulsoup_crawler.py index 1f332d8368..9d15ca3cd4 100644 --- a/tests/unit/crawlers/_beautifulsoup/test_beautifulsoup_crawler.py +++ b/tests/unit/crawlers/_beautifulsoup/test_beautifulsoup_crawler.py @@ -316,6 +316,7 @@ async def request_handler(context: BeautifulSoupCrawlingContext) -> None: async def test_extract_form_requests(server_url: URL, http_client: HttpClient) -> None: + """A form request built in the handler submits the form fields with the `Referer` of the page.""" crawler = BeautifulSoupCrawler(http_client=http_client) responses: list[dict[str, Any]] = [] page = '
' diff --git a/tests/unit/crawlers/_parsel/test_parsel_crawler.py b/tests/unit/crawlers/_parsel/test_parsel_crawler.py index 0bc960a4f4..a37d6b9a71 100644 --- a/tests/unit/crawlers/_parsel/test_parsel_crawler.py +++ b/tests/unit/crawlers/_parsel/test_parsel_crawler.py @@ -404,6 +404,7 @@ async def request_handler(context: ParselCrawlingContext) -> None: async def test_extract_form_requests(server_url: URL, http_client: HttpClient) -> None: + """A form request built in the handler submits the form fields with the `Referer` of the page.""" crawler = ParselCrawler(http_client=http_client) responses: list[dict[str, Any]] = [] page = '
' From 05399278b73877c54fd908f41e0c83e3ca83a6c2 Mon Sep 17 00:00:00 2001 From: Vlada Dusek Date: Tue, 6 Oct 2026 10:40:46 +0200 Subject: [PATCH 3/8] fix: keep fields in the first legend of a disabled fieldset enabled --- src/crawlee/_utils/html.py | 5 ++++- tests/unit/_utils/test_html.py | 6 ++++-- 2 files changed, 8 insertions(+), 3 deletions(-) diff --git a/src/crawlee/_utils/html.py b/src/crawlee/_utils/html.py index 9eef8d8910..4f8d8da561 100644 --- a/src/crawlee/_utils/html.py +++ b/src/crawlee/_utils/html.py @@ -336,7 +336,10 @@ def _disabled_elements(root: HtmlElement, elements: list[HtmlElement]) -> set[Ht for fieldset in root.iter('fieldset'): # A fieldset inside a disabled one is already covered, so each element is visited once. if 'disabled' in fieldset.attrib and fieldset not in disabled: - disabled.update(fieldset.iter('fieldset', *_FIELD_TAGS)) + # The contents of the first `` child stay enabled. + legend = next((child for child in fieldset if child.tag == 'legend'), None) + exempt = set() if legend is None else set(legend.iter('fieldset', *_FIELD_TAGS)) + disabled.update(element for element in fieldset.iter('fieldset', *_FIELD_TAGS) if element not in exempt) return disabled diff --git a/tests/unit/_utils/test_html.py b/tests/unit/_utils/test_html.py index 9741e5f2c8..4b739b83a9 100644 --- a/tests/unit/_utils/test_html.py +++ b/tests/unit/_utils/test_html.py @@ -355,9 +355,11 @@ async def test_form_attribute(extract_form_requests: ExtractFormRequests) -> Non async def test_disabled_fieldset(extract_form_requests: ExtractFormRequests) -> None: - """A disabled `
` disables all fields inside it, and an enabled one none.""" + """A disabled `
` disables all fields inside it except its first ``, and an enabled one none.""" html = """
+
+
@@ -365,7 +367,7 @@ async def test_disabled_fieldset(extract_form_requests: ExtractFormRequests) -> [request] = await extract_form_requests(html) - assert _submitted_fields(request) == {'c': ''} + assert _submitted_fields(request) == {'first-legend': '', 'c': ''} async def test_fields_argument(extract_form_requests: ExtractFormRequests) -> None: From 3750db9a9579706cb470740a735760d1eca6d4ff Mon Sep 17 00:00:00 2001 From: Vlada Dusek Date: Tue, 6 Oct 2026 10:41:31 +0200 Subject: [PATCH 4/8] fix: submit select options the way browsers do --- src/crawlee/_utils/html.py | 31 +++++++++++++++++++++++++++++-- tests/unit/_utils/test_html.py | 19 +++++++++++++++++++ 2 files changed, 48 insertions(+), 2 deletions(-) diff --git a/src/crawlee/_utils/html.py b/src/crawlee/_utils/html.py index 4f8d8da561..af168626ca 100644 --- a/src/crawlee/_utils/html.py +++ b/src/crawlee/_utils/html.py @@ -112,6 +112,9 @@ # Browsers cut a longer referrer down to the origin. _MAX_REFERRER_LENGTH = 4096 +# Browsers collapse only ASCII whitespace in an option text, keeping non-breaking spaces. +_ASCII_WHITESPACE_PATTERN = re.compile(r'[ \t\n\f\r]+') + class FormRequestOptions(TypedDict): """Options for the `Request` created from a form. @@ -493,8 +496,7 @@ def _element_fields(element: HtmlElement) -> list[_Field]: name = element.get('name') if element.tag == 'select': - values = element.value if element.multiple else [element.value] - return [_Field(name, value) for value in values if value is not None] + return [_Field(name, value) for value in _select_values(element)] if element.tag == 'textarea': # Browsers drop the newline right after `' + + [request] = await extract_form_requests(html, fields={'c\nd': 'v'}) + + assert b'name="a%22b"\r\n\r\nx\r\ny\r\n' in (request.payload or b'') + assert b'name="c%0D%0Ad"\r\n\r\nv\r\n' in (request.payload or b'') + + async def test_text_plain_form(extract_form_requests: ExtractFormRequests) -> None: """A `text/plain` form sends one field per line.""" html = '
' @@ -339,7 +349,7 @@ async def test_submitted_fields(extract_form_requests: ExtractFormRequests) -> N 'multi-disabled': 'e', 'text-value': 'spaced text', 'multi': ['m1', 'm3'], - 'area': '\nlong text', + 'area': '\r\nlong text', 'file': '', 'empty': '', 'hidden': '', From 53c19ca9179b49705b0af564fb4f309f3ab36c8b Mon Sep 17 00:00:00 2001 From: Vlada Dusek Date: Tue, 6 Oct 2026 10:42:03 +0200 Subject: [PATCH 6/8] fix: let empty form* attributes on the clicked button override the form --- src/crawlee/_utils/html.py | 7 +++---- tests/unit/_utils/test_html.py | 15 +++++++++++++++ 2 files changed, 18 insertions(+), 4 deletions(-) diff --git a/src/crawlee/_utils/html.py b/src/crawlee/_utils/html.py index c55b6990aa..179320851e 100644 --- a/src/crawlee/_utils/html.py +++ b/src/crawlee/_utils/html.py @@ -431,10 +431,9 @@ def _is_submit_button(element: HtmlElement) -> bool: def _submission_attribute(form: HtmlElement, button: HtmlElement | None, name: str) -> str: """Get a form attribute like `action`, which the clicked button can override with its `form*` counterpart.""" - if button is not None: - button_value = button.get(f'form{name}') - if button_value: - return button_value + # A present but empty override still wins, like `formaction=""` submitting to the page URL. + if button is not None and (button_value := button.get(f'form{name}')) is not None: + return button_value return form.get(name) or '' diff --git a/tests/unit/_utils/test_html.py b/tests/unit/_utils/test_html.py index 7b66252165..b770d7864f 100644 --- a/tests/unit/_utils/test_html.py +++ b/tests/unit/_utils/test_html.py @@ -637,6 +637,21 @@ async def test_click_button_overrides(extract_form_requests: ExtractFormRequests assert _submitted_fields(request) == {'q': 'x', 'action': 'delete'} +async def test_empty_button_overrides(extract_form_requests: ExtractFormRequests) -> None: + """Empty `formaction`, `formmethod` and `formenctype` override the form attributes with their defaults.""" + html = """ +
+ + +
+ """ + + [request] = await extract_form_requests(html) + + assert request.method == 'GET' + assert request.url == f'{_PAGE_URL}?q=x' + + async def test_click_disabled_button(extract_form_requests: ExtractFormRequests) -> None: """`click` picks a disabled button only if no enabled one matches.""" html = ( From 68c4924748cc210ea9d4f98c14c8215727a485e1 Mon Sep 17 00:00:00 2001 From: Vlada Dusek Date: Tue, 6 Oct 2026 10:42:16 +0200 Subject: [PATCH 7/8] fix: resolve a whitespace-only action and a padded base href like browsers --- src/crawlee/_utils/html.py | 6 +++--- tests/unit/_utils/test_html.py | 10 ++++++++++ 2 files changed, 13 insertions(+), 3 deletions(-) diff --git a/src/crawlee/_utils/html.py b/src/crawlee/_utils/html.py index 179320851e..2558a574b8 100644 --- a/src/crawlee/_utils/html.py +++ b/src/crawlee/_utils/html.py @@ -236,7 +236,7 @@ def forms_to_requests( root = forms[0].getroottree().getroot() base = root.find('.//base[@href]') try: - base_url = convert_to_absolute_url(page_url, '' if base is None else base.get('href')) + base_url = convert_to_absolute_url(page_url, '' if base is None else base.get('href').strip()) except ValueError: base_url = page_url @@ -379,7 +379,7 @@ def _form_to_request( method = (_submission_attribute(form, button, 'method') or 'get').upper() enctype = _submission_attribute(form, button, 'enctype').lower() - action = _submission_attribute(form, button, 'action').strip() + action = _submission_attribute(form, button, 'action') # A dialog form only closes its `` on the client. if method == 'DIALOG': @@ -443,7 +443,7 @@ def _resolve_action(page_url: str, base_url: str, action: str) -> str | None: return page_url try: - url = convert_to_absolute_url(base_url, action) + url = convert_to_absolute_url(base_url, action.strip()) validate_http_url(url) except ValueError: return None diff --git a/tests/unit/_utils/test_html.py b/tests/unit/_utils/test_html.py index b770d7864f..bc602f94f3 100644 --- a/tests/unit/_utils/test_html.py +++ b/tests/unit/_utils/test_html.py @@ -450,6 +450,16 @@ async def test_browser_rules(extract_form_requests: ExtractFormRequests, html: s pytest.param('
', 'https://example.com/page/index.html', id='no-action'), pytest.param('
', 'https://example.com/up', id='relative'), pytest.param('
', 'https://example.com/go', id='whitespace-around-action'), + pytest.param( + '
', + 'https://other.com/dir/', + id='whitespace-around-base-href', + ), + pytest.param( + '
', + 'https://other.com/dir/', + id='whitespace-action-base-href', + ), pytest.param( '
', 'https://other.com/dir/go', From b00b969f3d5a96d27dbb3272e4081dbbef95f7a5 Mon Sep 17 00:00:00 2001 From: Vlada Dusek Date: Tue, 6 Oct 2026 10:42:23 +0200 Subject: [PATCH 8/8] fix: submit input values the way browsers do --- src/crawlee/_utils/html.py | 9 ++++++++- tests/unit/_utils/test_html.py | 17 +++++++++++++++++ 2 files changed, 25 insertions(+), 1 deletion(-) diff --git a/src/crawlee/_utils/html.py b/src/crawlee/_utils/html.py index 2558a574b8..206aa087d4 100644 --- a/src/crawlee/_utils/html.py +++ b/src/crawlee/_utils/html.py @@ -519,7 +519,14 @@ def _element_fields(element: HtmlElement) -> list[_Field]: if element.checkable and not element.checked: return [] - return [_Field(name, element.value or '')] + # lxml reports `on` for a checkbox or radio button with an empty `value`, which browsers submit as is. + value = element.get('value', 'on' if element.checkable else '') + # Browsers strip line breaks from the value of a text-like input, and surrounding whitespace from an email or URL. + if element.type not in ('hidden', 'checkbox', 'radio'): + value = value.replace('\r', '').replace('\n', '') + if element.type in ('email', 'url'): + value = value.strip(' \t\n\f\r') + return [_Field(name, value)] def _select_values(select: HtmlElement) -> list[str]: diff --git a/tests/unit/_utils/test_html.py b/tests/unit/_utils/test_html.py index bc602f94f3..1099a79c9b 100644 --- a/tests/unit/_utils/test_html.py +++ b/tests/unit/_utils/test_html.py @@ -306,6 +306,7 @@ async def test_submitted_fields(extract_form_requests: ExtractFormRequests) -> N + @@ -342,6 +343,7 @@ async def test_submitted_fields(extract_form_requests: ExtractFormRequests) -> N 'text': 't', 'checked': 'c1', 'no-value': 'on', + 'empty-value': '', 'radio': 'r2', 'first': 'o1', 'last': 'b', @@ -383,6 +385,21 @@ async def test_option_text_keeps_nbsp(extract_form_requests: ExtractFormRequests assert _submitted_fields(request) == {'s': '\xa0a\xa0 b'} +async def test_input_value_sanitization(extract_form_requests: ExtractFormRequests) -> None: + """Text-like inputs drop line breaks, email and URL inputs also surrounding whitespace, hidden inputs neither.""" + html = """ +
+ + + +
+ """ + + [request] = await extract_form_requests(html) + + assert _submitted_fields(request) == {'text': 'ab', 'email': 'x@y.z', 'hidden': 'c\r\nd'} + + async def test_disabled_fieldset(extract_form_requests: ExtractFormRequests) -> None: """A disabled `
` disables all fields inside it except its first ``, and an enabled one none.""" html = """