diff --git a/docs/examples/code_examples/fill_and_submit_web_form_automated.py b/docs/examples/code_examples/fill_and_submit_web_form_automated.py new file mode 100644 index 0000000000..844b0d9fe3 --- /dev/null +++ b/docs/examples/code_examples/fill_and_submit_web_form_automated.py @@ -0,0 +1,39 @@ +import asyncio + +from crawlee.crawlers import ParselCrawler, ParselCrawlingContext + + +async def main() -> None: + crawler = ParselCrawler() + + # Fill in the form on the page and enqueue its submission. + @crawler.router.default_handler + async def request_handler(context: ParselCrawlingContext) -> None: + context.log.info(f'Filling in the form on {context.request.url} ...') + requests = await context.extract_form_requests( + fields={ + 'custname': 'John Doe', + 'custtel': '1234567890', + 'custemail': 'johndoe@example.com', + 'size': 'large', + 'topping': ['bacon', 'cheese', 'mushroom'], + 'delivery': '13:00', + 'comments': 'Please ring the doorbell upon arrival.', + }, + label='form-result', + ) + await context.add_requests(requests) + + # Process the response to the form submission. + @crawler.router.handler('form-result') + async def form_result_handler(context: ParselCrawlingContext) -> None: + context.log.info(f'Processing {context.request.url} ...') + response = (await context.http_response.read()).decode('utf-8') + context.log.info(f'Response: {response}') # To see the response in the logs. + + # Run the crawler with the page containing the form. + await crawler.run(['https://httpbin.org/forms/post']) + + +if __name__ == '__main__': + asyncio.run(main()) diff --git a/docs/examples/fill_and_submit_web_form.mdx b/docs/examples/fill_and_submit_web_form.mdx index bda46c1d97..e81af4623f 100644 --- a/docs/examples/fill_and_submit_web_form.mdx +++ b/docs/examples/fill_and_submit_web_form.mdx @@ -10,8 +10,9 @@ import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; import RequestExample from '!!raw-loader!roa-loader!./code_examples/fill_and_submit_web_form_request.py'; import CrawlerExample from '!!raw-loader!roa-loader!./code_examples/fill_and_submit_web_form_crawler.py'; +import AutomatedExample from '!!raw-loader!roa-loader!./code_examples/fill_and_submit_web_form_automated.py'; -This example demonstrates how to fill and submit a web form using the `HttpCrawler` crawler. The same approach applies to any crawler that inherits from it, such as the `BeautifulSoupCrawler` or `ParselCrawler`. +This example demonstrates how to fill and submit a web form using the `HttpCrawler` crawler. The same approach applies to any crawler that inherits from it, such as the `BeautifulSoupCrawler` or `ParselCrawler`. These two crawlers can also [fill in the form automatically](#fill-in-the-form-automatically). We are going to use the [httpbin.org](https://httpbin.org) website to demonstrate how it works. @@ -118,3 +119,21 @@ Finally, run your crawler. Your logs should show something like this: ``` This log output confirms that the crawler successfully submitted the form and processed the response. Congratulations! You have successfully filled and submitted a web form using the `HttpCrawler`. + +## Fill in the form automatically + +The `ParselCrawler` and `BeautifulSoupCrawler` can build the form request for you. Their crawling contexts provide the `extract_form_requests` helper, which reads the form from the page, fills in your values and returns a list with the request that submits it the way a browser does. The action URL, the method and the encoding come from the form itself, so you only need the field names from [Investigate the form fields](#investigate-the-form-fields). + +The crawler below opens the page with the form. The default handler fills in the form with the `fields` argument and enqueues the submission with a label. A separate handler for that label processes the response. + + + {AutomatedExample} + + +Note that: + +- `fields` replaces the values of the listed fields and adds the ones the form doesn't have. A list submits the field once per value, as with the `topping` checkboxes. +- Fields you don't list keep the values from the page, so hidden inputs such as CSRF tokens are submitted as they are. A CSRF token is tied to the session cookie, so pass `session_id=context.session.id` to send the form in the same session. The request also carries the `Referer` and `Origin` headers a browser sends, which some CSRF checks require. +- On a page with several forms, the helper submits the one sharing the most field names with `fields`, or the first one if none shares any. It skips forms that can't be submitted, for example because their action is JavaScript, but never falls back to a form sharing fewer names, so the list can be empty. To pick a form yourself, pass a CSS selector such as `selector='#order'`. To submit each form, pass `all_forms=True`. Then `fields` only replaces the fields each form has. +- The first enabled submit button of the form is clicked by default, and a form without one is submitted anyway. Use the `click` argument to pick another button by its attributes, even a disabled one, or to submit without one. +- The page decides where its form is sent. To enqueue only requests to the same host, call `context.add_requests(requests, strategy='same-hostname')`. diff --git a/docs/guides/code_examples/scrapy_migration/crawlee_post.py b/docs/guides/code_examples/scrapy_migration/crawlee_post.py index 2f332426ef..b97a77552e 100644 --- a/docs/guides/code_examples/scrapy_migration/crawlee_post.py +++ b/docs/guides/code_examples/scrapy_migration/crawlee_post.py @@ -1,7 +1,5 @@ import asyncio -from urllib.parse import urlencode -from crawlee import Request from crawlee.crawlers import ParselCrawler, ParselCrawlingContext @@ -14,34 +12,16 @@ async def login_page(context: ParselCrawlingContext) -> None: if not context.session: raise RuntimeError('Session not found') - token = context.selector.css('input[name="csrf_token"]::attr(value)').get() - - # The CSRF token is required for the POST to succeed. If it's missing, - # the login will fail. - if not token: - raise RuntimeError('CSRF token not found') - - form = {'csrf_token': token, 'username': 'user', 'password': 'pass'} - # highlight-start - # Crawlee's `payload` is the raw request body, so encode the fields yourself - # and set the `Content-Type`. Scrapy's `FormRequest` does both for you. - await context.add_requests( - [ - Request.from_url( - 'https://quotes.toscrape.com/login', - method='POST', - payload=urlencode(form), - headers={'content-type': 'application/x-www-form-urlencoded'}, - label='after-login', - # Bind the POST to the same session so its CSRF cookie matches. - session_id=context.session.id, - # The POST shares the GET's URL. Include the method and payload - # in the unique key, or the queue drops it as a duplicate. - use_extended_unique_key=True, - ) - ] + # Like Scrapy's `FormRequest.from_response`, the helper keeps the hidden + # `csrf_token` field, encodes the data and sets the `Content-Type` header. + requests = await context.extract_form_requests( + fields={'username': 'user', 'password': 'pass'}, + label='after-login', + # Bind the POST to the same session so its CSRF cookie matches. + session_id=context.session.id, ) + await context.add_requests(requests) # highlight-end @crawler.router.handler('after-login') diff --git a/docs/guides/scrapy_migration.mdx b/docs/guides/scrapy_migration.mdx index 5acbfa3824..a5e4c29a9e 100644 --- a/docs/guides/scrapy_migration.mdx +++ b/docs/guides/scrapy_migration.mdx @@ -80,7 +80,7 @@ Both frameworks give you a request scheduler, filtering of duplicate requests, r | `response.follow()` / `yield Request(...)` | `enqueue_links` / `add_requests` | | `dont_filter=True` | `Request.from_url(always_enqueue=True)` | | `allowed_domains` | `enqueue_links(strategy=...)` | -| `scrapy.FormRequest` | `Request.from_url(method='POST', payload=...)` | +| `scrapy.FormRequest` | `Request.from_url(method='POST', payload=...)` / `context.extract_form_requests(...)` | | Item pipelines | `Dataset` | | Downloader / spider middlewares | `router.use()`, navigation hooks, HTTP clients | | `settings.py` | `Configuration` + crawler arguments | @@ -263,7 +263,7 @@ Scrapy retries failed requests with `RetryMiddleware` and reports terminal failu ## Forms and login -Scrapy submits forms with `FormRequest`, which encodes `formdata` as `form-urlencoded` and sets the header for you. Crawlee's `payload` takes the raw request body, so encode the fields yourself with `urllib.parse.urlencode` and set the `Content-Type` through `headers=`. For a full login flow with session reuse, see the [Logging in with a crawler guide](./logging-in-with-a-crawler). +Scrapy submits forms with `FormRequest.from_response`, which reads the form from the page, keeps its hidden fields and encodes the data for you. Crawlee's `extract_form_requests` helper does the same in the `ParselCrawler` and `BeautifulSoupCrawler`. Pass your values in `fields` and request options such as `label` or `session_id` as keyword arguments. Like `from_response`, it submits a single form. Scrapy takes the first form by default, while the helper prefers the one sharing the most field names with `fields`. It returns a list, which is empty when no form matches, so enqueue it with `add_requests`. For a plain `FormRequest` that doesn't come from a form on the page, use `Request.from_url`. Its `payload` is the raw request body, so encode the fields with `urllib.parse.urlencode` and set the `Content-Type` through `headers=`. For a full login flow with session reuse, see the [Logging in with a crawler guide](./logging-in-with-a-crawler). diff --git a/src/crawlee/_utils/html.py b/src/crawlee/_utils/html.py index 5b357f312a..206aa087d4 100644 --- a/src/crawlee/_utils/html.py +++ b/src/crawlee/_utils/html.py @@ -4,8 +4,24 @@ import codecs import re +from typing import TYPE_CHECKING, NamedTuple, TypedDict +from urllib.parse import urlencode, urlsplit +from yarl import URL + +from crawlee._request import Request +from crawlee._types import HttpHeaders +from crawlee._utils.crypto import compute_short_hash from crawlee._utils.http import parse_content_type_charset +from crawlee._utils.urls import convert_to_absolute_url, is_url_absolute, validate_http_url + +if TYPE_CHECKING: + from collections.abc import Mapping, Sequence + + from lxml.html import HtmlElement + from typing_extensions import NotRequired, Unpack + + from crawlee._types import JsonSerializable # Matches the `encoding` of an XML declaration, which XHTML pages may use instead of a `` tag. _XML_ENCODING_PATTERN = re.compile(rb'^\s*<\?xml[^>]*\sencoding\s*=\s*["\']([a-z0-9_:.+-]+)', re.IGNORECASE) @@ -90,6 +106,62 @@ _ENCODING_BY_LABEL = {label: codec for codec, labels in _WHATWG_ENCODING_LABELS.items() for label in labels.split()} +_FIELD_TAGS = ('input', 'button', 'select', 'textarea') +_BUTTON_INPUT_TYPES = ('submit', 'image', 'reset', 'button') + +# Browsers cut a longer referrer down to the origin. +_MAX_REFERRER_LENGTH = 4096 + +_LINE_BREAK_PATTERN = re.compile(r'\r\n|\r|\n') + +# Browsers collapse only ASCII whitespace in an option text, keeping non-breaking spaces. +_ASCII_WHITESPACE_PATTERN = re.compile(r'[ \t\n\f\r]+') + +_MULTIPART_NAME_ESCAPES = str.maketrans({'"': '%22', '\r': '%0D', '\n': '%0A'}) + + +class FormRequestOptions(TypedDict): + """Options for the `Request` created from a form. + + Mirrors `RequestOptions` without the URL, method and payload, which come from the form, without `id` and + `unique_key`, which can't be shared by several forms, and without `enqueue_strategy`, which enqueuing sets. + """ + + label: NotRequired[str | None] + """A label routing the request to a specific handler.""" + + headers: NotRequired[HttpHeaders | dict[str, str] | None] + """HTTP headers of the request, replacing those the form sets, like `Content-Type` or `Referer`.""" + + session_id: NotRequired[str | None] + """ID of the `Session` the request is bound to.""" + + keep_url_fragment: NotRequired[bool] + """Whether the URL fragment counts towards the unique key of the request.""" + + use_extended_unique_key: NotRequired[bool] + """Whether the method and payload count towards the unique key. Defaults to `True` for POST forms.""" + + always_enqueue: NotRequired[bool] + """Whether to enqueue the request even if it's already in the queue.""" + + user_data: NotRequired[Mapping[str, JsonSerializable]] + """Custom data stored with the request.""" + + no_retry: NotRequired[bool] + """Whether to skip retrying the request if it fails.""" + + max_retries: NotRequired[int | None] + """The maximum number of retries of the request.""" + + +class _Field(NamedTuple): + """A single entry the form submits.""" + + name: str + value: str + is_file: bool = False + def get_declared_html_encoding(body: bytes, content_type: str | None) -> str | None: """Get the Python codec for the encoding an HTML response body declares. @@ -110,7 +182,7 @@ def get_declared_html_encoding(body: bytes, content_type: str | None) -> str | N return encoding header_charset = parse_content_type_charset(content_type) - return _resolve_encoding(header_charset) or _find_declared_encoding(body) + return resolve_encoding(header_charset) or _find_declared_encoding(body) def decode_html_body(body: bytes, encoding: str) -> str: @@ -125,18 +197,94 @@ def decode_html_body(body: bytes, encoding: str) -> str: return body.decode(encoding, 'replace').removeprefix('\ufeff') -def _resolve_encoding(label: str | None) -> str | None: +def resolve_encoding(label: str | None) -> str | None: """Get the Python codec for a WHATWG encoding label, or `None` if browsers don't know the label.""" if not label: return None return _ENCODING_BY_LABEL.get(label.lower()) +def strip_html_comments(body: bytes) -> bytes: + """Remove the HTML comments from the body, so a commented-out declaration doesn't count.""" + return _HTML_COMMENT_PATTERN.sub(b'', body) + + +def forms_to_requests( + forms: Sequence[HtmlElement], + page_url: str, + page_encoding: str, + *, + fields: Mapping[str, str | Sequence[str] | None] | None = None, + click: bool | Mapping[str, str] = True, + all_forms: bool = False, + **kwargs: Unpack[FormRequestOptions], +) -> list[Request]: + """Create a `Request` submitting one of the forms, or each of them, the way a browser does. + + Args: + forms: The form elements, each within the lxml tree of the whole page. + page_url: The URL of the page, which forms without an action submit to and the `Referer` comes from. + page_encoding: The Python codec the page was decoded with, which forms submit in by default. + fields: Field values to submit, see `extract_form_requests` of the crawling contexts. + click: The submit button to click, see `extract_form_requests`. + all_forms: Whether to submit each form, filling in only the fields it has, see `extract_form_requests`. + **kwargs: Additional options passed to `Request.from_url`. + """ + if not forms: + return [] + + root = forms[0].getroottree().getroot() + base = root.find('.//base[@href]') + try: + base_url = convert_to_absolute_url(page_url, '' if base is None else base.get('href').strip()) + except ValueError: + base_url = page_url + + page_elements = list(root.iter(*_FIELD_TAGS)) + elements_by_form = _elements_by_form(root, page_elements) + disabled = _disabled_elements(root, page_elements) + + if not all_forms: + # The form sharing the most field names with `fields` is the one to fill in, and ties keep document order. + # Forms sharing fewer names never get the values, even if the best one can't be submitted. + wanted = set(fields or ()) + shared = {form: len(wanted & _field_names(elements_by_form.get(form, []))) for form in forms} + most_shared = max(shared.values()) + forms = [form for form in forms if shared[form] == most_shared] + + requests = [] + for form in forms: + elements = elements_by_form.get(form, []) + form_fields = fields + if all_forms and fields: + # Values meant for one form don't spread to the others, like credentials into a search form. + names = _field_names(elements) + form_fields = {name: value for name, value in fields.items() if name in names} + + request = _form_to_request( + form, + elements, + page_url=page_url, + base_url=base_url, + page_encoding=page_encoding, + disabled=disabled, + fields=form_fields, + click=click, + **kwargs, + ) + if request is None: + continue + requests.append(request) + if not all_forms: + break + return requests + + def _find_declared_encoding(body: bytes) -> str | None: """Find the encoding declared by an XML declaration or a `` tag near the start of the body.""" - prescan = _HTML_COMMENT_PATTERN.sub(b'', body[:_PRESCAN_BYTES]) + prescan = strip_html_comments(body[:_PRESCAN_BYTES]) xml_match = _XML_ENCODING_PATTERN.match(prescan) - xml_encoding = _resolve_encoding(xml_match.group(1).decode('ascii')) if xml_match else None + xml_encoding = resolve_encoding(xml_match.group(1).decode('ascii')) if xml_match else None encoding = xml_encoding or _find_meta_encoding(prescan) # A declaration readable as ASCII rules out UTF-16, so browsers read such pages as UTF-8. return 'utf-8' if encoding and encoding.startswith('utf-16') else encoding @@ -161,7 +309,315 @@ def _find_meta_encoding(prescan: bytes) -> str | None: else: continue - encoding = _resolve_encoding(label) + encoding = resolve_encoding(label) if encoding: return encoding return None + + +def _elements_by_form(root: HtmlElement, elements: list[HtmlElement]) -> dict[HtmlElement, list[HtmlElement]]: + """Group the fields and buttons of the page by the form they belong to, in document order.""" + # Walking each form avoids an ancestor walk per field. + enclosing_form: dict[HtmlElement, HtmlElement] = {} + for form in root.iter('form'): + enclosing_form.update(dict.fromkeys(form.iter(*_FIELD_TAGS), form)) + + # The first element with a given ID wins, as in `getElementById`. Only the `form` attribute needs them. + elements_by_id: dict[str, HtmlElement] = {} + if any(element.get('form') is not None for element in elements): + for element in root.xpath('//*[@id!=""]'): + elements_by_id.setdefault(element.get('id'), element) + + elements_by_form: dict[HtmlElement, list[HtmlElement]] = {} + for element in elements: + form_id = element.get('form') + owner = enclosing_form.get(element) if form_id is None else elements_by_id.get(form_id) + if owner is not None and owner.tag == 'form': + elements_by_form.setdefault(owner, []).append(element) + return elements_by_form + + +def _disabled_elements(root: HtmlElement, elements: list[HtmlElement]) -> set[HtmlElement]: + """Find the disabled fields and buttons, including those inside a disabled `
`.""" + disabled = {element for element in elements if 'disabled' in element.attrib} + for fieldset in root.iter('fieldset'): + # A fieldset inside a disabled one is already covered, so each element is visited once. + if 'disabled' in fieldset.attrib and fieldset not in disabled: + # The contents of the first `` child stay enabled. + legend = next((child for child in fieldset if child.tag == 'legend'), None) + exempt = set() if legend is None else set(legend.iter('fieldset', *_FIELD_TAGS)) + disabled.update(element for element in fieldset.iter('fieldset', *_FIELD_TAGS) if element not in exempt) + return disabled + + +def _field_names(elements: list[HtmlElement]) -> set[str]: + """Get the names of the given fields and buttons.""" + return {name for element in elements if (name := element.get('name'))} + + +def _form_to_request( + form: HtmlElement, + elements: list[HtmlElement], + *, + page_url: str, + base_url: str, + page_encoding: str, + disabled: set[HtmlElement], + fields: Mapping[str, str | Sequence[str] | None] | None, + click: bool | Mapping[str, str], + **kwargs: Unpack[FormRequestOptions], +) -> Request | None: + """Create a `Request` submitting a single form, or `None` if a browser wouldn't send one.""" + button = None + if click is not False: + button = _find_clickable(elements, {} if click is True else click, disabled) + if click is not True and button is None: + # Pages often enable a button with JavaScript, so a disabled one matching `click` is clicked too. + button = _find_clickable(elements, click, set()) + if button is None: + return None + + method = (_submission_attribute(form, button, 'method') or 'get').upper() + enctype = _submission_attribute(form, button, 'enctype').lower() + action = _submission_attribute(form, button, 'action') + + # A dialog form only closes its `` on the client. + if method == 'DIALOG': + return None + + url = _resolve_action(page_url, base_url, action) + if url is None: + return None + + entries = _collect_fields(elements, button, fields, disabled) + encoding = _form_encoding(form, page_encoding) + # CSRF checks may require the `Referer` or `Origin` a browser sends. Headers passed in `headers` win. + form_headers = _referrer_headers(page_url, url, method) + + if method != 'POST': + query = urlencode( + [(entry.name, entry.value) for entry in entries], encoding=encoding, errors='xmlcharrefreplace' + ) + kwargs['headers'] = HttpHeaders(form_headers) | HttpHeaders(kwargs.get('headers') or {}) + return Request.from_url(urlsplit(url)._replace(query=query).geturl(), method='GET', **kwargs) + + payload, form_headers['Content-Type'] = _encode_body(entries, enctype, encoding) + kwargs['headers'] = HttpHeaders(form_headers) | HttpHeaders(kwargs.get('headers') or {}) + kwargs.setdefault('use_extended_unique_key', True) + return Request.from_url(url, method='POST', payload=payload, **kwargs) + + +def _find_clickable( + elements: list[HtmlElement], attributes: Mapping[str, str], disabled: set[HtmlElement] +) -> HtmlElement | None: + """Find the first submit button having all the given attributes, skipping the `disabled` ones.""" + for element in elements: + if not _is_submit_button(element) or element in disabled: + continue + if all(element.get(key) == value for key, value in attributes.items()): + return element + return None + + +def _is_submit_button(element: HtmlElement) -> bool: + """Check whether the element submits the form when clicked.""" + button_type = element.get('type', '').lower() + if element.tag == 'button': + return button_type in ('', 'submit') + if element.tag == 'input': + return button_type in ('submit', 'image') + return False + + +def _submission_attribute(form: HtmlElement, button: HtmlElement | None, name: str) -> str: + """Get a form attribute like `action`, which the clicked button can override with its `form*` counterpart.""" + # A present but empty override still wins, like `formaction=""` submitting to the page URL. + if button is not None and (button_value := button.get(f'form{name}')) is not None: + return button_value + return form.get(name) or '' + + +def _resolve_action(page_url: str, base_url: str, action: str) -> str | None: + """Resolve the form action to an absolute URL, or `None` if it isn't a valid HTTP(S) URL.""" + if not action: + return page_url + + try: + url = convert_to_absolute_url(base_url, action.strip()) + validate_http_url(url) + except ValueError: + return None + # `validate_http_url` accepts a URL without a host, like `http:x` resolved against an HTTPS page. + return url if is_url_absolute(url) else None + + +def _collect_fields( + elements: list[HtmlElement], + button: HtmlElement | None, + fields: Mapping[str, str | Sequence[str] | None] | None, + disabled: set[HtmlElement], +) -> list[_Field]: + """Collect the entries the form submits: its fields, the clicked button and the `fields` overrides.""" + # Only one radio button of a group is checked, the last one in the markup. + checked_radios = {element.get('name'): element for element in elements if _is_radio(element) and element.checked} + + entries: list[_Field] = [] + for element in elements: + if element is button: + entries.extend(_button_fields(button)) + continue + if not element.get('name') or element in disabled: + continue + if _is_radio(element) and checked_radios.get(element.get('name')) is not element: + continue + entries.extend(_element_fields(element)) + + # Browsers submit every line break as CRLF. + return [ + _Field(_LINE_BREAK_PATTERN.sub('\r\n', entry.name), _LINE_BREAK_PATTERN.sub('\r\n', entry.value), entry.is_file) + for entry in _apply_fields(entries, fields or {}) + ] + + +def _is_radio(element: HtmlElement) -> bool: + """Check whether the element is a radio button.""" + return element.tag == 'input' and element.type == 'radio' + + +def _button_fields(button: HtmlElement) -> list[_Field]: + """Get the entries the clicked button adds.""" + name = button.get('name') + + # An image button sends the click coordinates instead of its value. + if button.get('type', '').lower() == 'image': + prefix = f'{name}.' if name else '' + return [_Field(f'{prefix}x', '0'), _Field(f'{prefix}y', '0')] + + if name: + return [_Field(name, button.get('value', ''))] + return [] + + +def _element_fields(element: HtmlElement) -> list[_Field]: + """Get the entries a single enabled, named field submits.""" + name = element.get('name') + + if element.tag == 'select': + return [_Field(name, value) for value in _select_values(element)] + + if element.tag == 'textarea': + # Browsers drop the newline right after `' + + [request] = await extract_form_requests(html, fields={'c\nd': 'v'}) + + assert b'name="a%22b"\r\n\r\nx\r\ny\r\n' in (request.payload or b'') + assert b'name="c%0D%0Ad"\r\n\r\nv\r\n' in (request.payload or b'') + + +async def test_text_plain_form(extract_form_requests: ExtractFormRequests) -> None: + """A `text/plain` form sends one field per line.""" + html = '
' + + [request] = await extract_form_requests(html) + + assert request.headers['content-type'] == 'text/plain' + assert request.payload == b'a=1\r\nb=2\r\n' + + +async def test_submitted_fields(extract_form_requests: ExtractFormRequests) -> None: + """Only the fields a browser would submit are collected.""" + html = """ +
+ + + + + + + + + + + + + + + + + + + +
+ + + + + + +
+ """ + + [request] = await extract_form_requests(html) + + assert _submitted_fields(request) == { + 'text': 't', + 'checked': 'c1', + 'no-value': 'on', + 'empty-value': '', + 'radio': 'r2', + 'first': 'o1', + 'last': 'b', + 'first-enabled': 'on', + 'multi-disabled': 'e', + 'text-value': 'spaced text', + 'multi': ['m1', 'm3'], + 'area': '\r\nlong text', + 'file': '', + 'empty': '', + 'hidden': '', + } + + +async def test_form_attribute(extract_form_requests: ExtractFormRequests) -> None: + """Fields are assigned to forms by their `form` attribute, and `fields` fill in only the fields a form has.""" + html = """ + + + +
+
+ """ + + search, login, empty = await extract_form_requests(html, fields={'q': 'y', 'other': 'p'}, all_forms=True) + + assert _submitted_fields(search) == {'q': 'y', 'lang': 'en', 'go': ''} + assert _submitted_fields(login) == {'other': 'p'} + assert _submitted_fields(empty) == {} + + +async def test_option_text_keeps_nbsp(extract_form_requests: ExtractFormRequests) -> None: + """An `