diff --git a/docs/examples/code_examples/fill_and_submit_web_form_automated.py b/docs/examples/code_examples/fill_and_submit_web_form_automated.py
new file mode 100644
index 0000000000..844b0d9fe3
--- /dev/null
+++ b/docs/examples/code_examples/fill_and_submit_web_form_automated.py
@@ -0,0 +1,39 @@
+import asyncio
+
+from crawlee.crawlers import ParselCrawler, ParselCrawlingContext
+
+
+async def main() -> None:
+ crawler = ParselCrawler()
+
+ # Fill in the form on the page and enqueue its submission.
+ @crawler.router.default_handler
+ async def request_handler(context: ParselCrawlingContext) -> None:
+ context.log.info(f'Filling in the form on {context.request.url} ...')
+ requests = await context.extract_form_requests(
+ fields={
+ 'custname': 'John Doe',
+ 'custtel': '1234567890',
+ 'custemail': 'johndoe@example.com',
+ 'size': 'large',
+ 'topping': ['bacon', 'cheese', 'mushroom'],
+ 'delivery': '13:00',
+ 'comments': 'Please ring the doorbell upon arrival.',
+ },
+ label='form-result',
+ )
+ await context.add_requests(requests)
+
+ # Process the response to the form submission.
+ @crawler.router.handler('form-result')
+ async def form_result_handler(context: ParselCrawlingContext) -> None:
+ context.log.info(f'Processing {context.request.url} ...')
+ response = (await context.http_response.read()).decode('utf-8')
+ context.log.info(f'Response: {response}') # To see the response in the logs.
+
+ # Run the crawler with the page containing the form.
+ await crawler.run(['https://httpbin.org/forms/post'])
+
+
+if __name__ == '__main__':
+ asyncio.run(main())
diff --git a/docs/examples/fill_and_submit_web_form.mdx b/docs/examples/fill_and_submit_web_form.mdx
index bda46c1d97..e81af4623f 100644
--- a/docs/examples/fill_and_submit_web_form.mdx
+++ b/docs/examples/fill_and_submit_web_form.mdx
@@ -10,8 +10,9 @@ import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock';
import RequestExample from '!!raw-loader!roa-loader!./code_examples/fill_and_submit_web_form_request.py';
import CrawlerExample from '!!raw-loader!roa-loader!./code_examples/fill_and_submit_web_form_crawler.py';
+import AutomatedExample from '!!raw-loader!roa-loader!./code_examples/fill_and_submit_web_form_automated.py';
-This example demonstrates how to fill and submit a web form using the `HttpCrawler` crawler. The same approach applies to any crawler that inherits from it, such as the `BeautifulSoupCrawler` or `ParselCrawler`.
+This example demonstrates how to fill and submit a web form using the `HttpCrawler` crawler. The same approach applies to any crawler that inherits from it, such as the `BeautifulSoupCrawler` or `ParselCrawler`. These two crawlers can also [fill in the form automatically](#fill-in-the-form-automatically).
We are going to use the [httpbin.org](https://httpbin.org) website to demonstrate how it works.
@@ -118,3 +119,21 @@ Finally, run your crawler. Your logs should show something like this:
```
This log output confirms that the crawler successfully submitted the form and processed the response. Congratulations! You have successfully filled and submitted a web form using the `HttpCrawler`.
+
+## Fill in the form automatically
+
+The `ParselCrawler` and `BeautifulSoupCrawler` can build the form request for you. Their crawling contexts provide the `extract_form_requests` helper, which reads the form from the page, fills in your values and returns a list with the request that submits it the way a browser does. The action URL, the method and the encoding come from the form itself, so you only need the field names from [Investigate the form fields](#investigate-the-form-fields).
+
+The crawler below opens the page with the form. The default handler fills in the form with the `fields` argument and enqueues the submission with a label. A separate handler for that label processes the response.
+
+
+ {AutomatedExample}
+
+
+Note that:
+
+- `fields` replaces the values of the listed fields and adds the ones the form doesn't have. A list submits the field once per value, as with the `topping` checkboxes.
+- Fields you don't list keep the values from the page, so hidden inputs such as CSRF tokens are submitted as they are. A CSRF token is tied to the session cookie, so pass `session_id=context.session.id` to send the form in the same session. The request also carries the `Referer` and `Origin` headers a browser sends, which some CSRF checks require.
+- On a page with several forms, the helper submits the one sharing the most field names with `fields`, or the first one if none shares any. It skips forms that can't be submitted, for example because their action is JavaScript, but never falls back to a form sharing fewer names, so the list can be empty. To pick a form yourself, pass a CSS selector such as `selector='#order'`. To submit each form, pass `all_forms=True`. Then `fields` only replaces the fields each form has.
+- The first enabled submit button of the form is clicked by default, and a form without one is submitted anyway. Use the `click` argument to pick another button by its attributes, even a disabled one, or to submit without one.
+- The page decides where its form is sent. To enqueue only requests to the same host, call `context.add_requests(requests, strategy='same-hostname')`.
diff --git a/docs/guides/code_examples/scrapy_migration/crawlee_post.py b/docs/guides/code_examples/scrapy_migration/crawlee_post.py
index 2f332426ef..b97a77552e 100644
--- a/docs/guides/code_examples/scrapy_migration/crawlee_post.py
+++ b/docs/guides/code_examples/scrapy_migration/crawlee_post.py
@@ -1,7 +1,5 @@
import asyncio
-from urllib.parse import urlencode
-from crawlee import Request
from crawlee.crawlers import ParselCrawler, ParselCrawlingContext
@@ -14,34 +12,16 @@ async def login_page(context: ParselCrawlingContext) -> None:
if not context.session:
raise RuntimeError('Session not found')
- token = context.selector.css('input[name="csrf_token"]::attr(value)').get()
-
- # The CSRF token is required for the POST to succeed. If it's missing,
- # the login will fail.
- if not token:
- raise RuntimeError('CSRF token not found')
-
- form = {'csrf_token': token, 'username': 'user', 'password': 'pass'}
-
# highlight-start
- # Crawlee's `payload` is the raw request body, so encode the fields yourself
- # and set the `Content-Type`. Scrapy's `FormRequest` does both for you.
- await context.add_requests(
- [
- Request.from_url(
- 'https://quotes.toscrape.com/login',
- method='POST',
- payload=urlencode(form),
- headers={'content-type': 'application/x-www-form-urlencoded'},
- label='after-login',
- # Bind the POST to the same session so its CSRF cookie matches.
- session_id=context.session.id,
- # The POST shares the GET's URL. Include the method and payload
- # in the unique key, or the queue drops it as a duplicate.
- use_extended_unique_key=True,
- )
- ]
+ # Like Scrapy's `FormRequest.from_response`, the helper keeps the hidden
+ # `csrf_token` field, encodes the data and sets the `Content-Type` header.
+ requests = await context.extract_form_requests(
+ fields={'username': 'user', 'password': 'pass'},
+ label='after-login',
+ # Bind the POST to the same session so its CSRF cookie matches.
+ session_id=context.session.id,
)
+ await context.add_requests(requests)
# highlight-end
@crawler.router.handler('after-login')
diff --git a/docs/guides/scrapy_migration.mdx b/docs/guides/scrapy_migration.mdx
index 5acbfa3824..a5e4c29a9e 100644
--- a/docs/guides/scrapy_migration.mdx
+++ b/docs/guides/scrapy_migration.mdx
@@ -80,7 +80,7 @@ Both frameworks give you a request scheduler, filtering of duplicate requests, r
| `response.follow()` / `yield Request(...)` | `enqueue_links` / `add_requests` |
| `dont_filter=True` | `Request.from_url(always_enqueue=True)` |
| `allowed_domains` | `enqueue_links(strategy=...)` |
-| `scrapy.FormRequest` | `Request.from_url(method='POST', payload=...)` |
+| `scrapy.FormRequest` | `Request.from_url(method='POST', payload=...)` / `context.extract_form_requests(...)` |
| Item pipelines | `Dataset` |
| Downloader / spider middlewares | `router.use()`, navigation hooks, HTTP clients |
| `settings.py` | `Configuration` + crawler arguments |
@@ -263,7 +263,7 @@ Scrapy retries failed requests with `RetryMiddleware` and reports terminal failu
## Forms and login
-Scrapy submits forms with `FormRequest`, which encodes `formdata` as `form-urlencoded` and sets the header for you. Crawlee's `payload` takes the raw request body, so encode the fields yourself with `urllib.parse.urlencode` and set the `Content-Type` through `headers=`. For a full login flow with session reuse, see the [Logging in with a crawler guide](./logging-in-with-a-crawler).
+Scrapy submits forms with `FormRequest.from_response`, which reads the form from the page, keeps its hidden fields and encodes the data for you. Crawlee's `extract_form_requests` helper does the same in the `ParselCrawler` and `BeautifulSoupCrawler`. Pass your values in `fields` and request options such as `label` or `session_id` as keyword arguments. Like `from_response`, it submits a single form. Scrapy takes the first form by default, while the helper prefers the one sharing the most field names with `fields`. It returns a list, which is empty when no form matches, so enqueue it with `add_requests`. For a plain `FormRequest` that doesn't come from a form on the page, use `Request.from_url`. Its `payload` is the raw request body, so encode the fields with `urllib.parse.urlencode` and set the `Content-Type` through `headers=`. For a full login flow with session reuse, see the [Logging in with a crawler guide](./logging-in-with-a-crawler).
diff --git a/src/crawlee/_utils/html.py b/src/crawlee/_utils/html.py
index 5b357f312a..206aa087d4 100644
--- a/src/crawlee/_utils/html.py
+++ b/src/crawlee/_utils/html.py
@@ -4,8 +4,24 @@
import codecs
import re
+from typing import TYPE_CHECKING, NamedTuple, TypedDict
+from urllib.parse import urlencode, urlsplit
+from yarl import URL
+
+from crawlee._request import Request
+from crawlee._types import HttpHeaders
+from crawlee._utils.crypto import compute_short_hash
from crawlee._utils.http import parse_content_type_charset
+from crawlee._utils.urls import convert_to_absolute_url, is_url_absolute, validate_http_url
+
+if TYPE_CHECKING:
+ from collections.abc import Mapping, Sequence
+
+ from lxml.html import HtmlElement
+ from typing_extensions import NotRequired, Unpack
+
+ from crawlee._types import JsonSerializable
# Matches the `encoding` of an XML declaration, which XHTML pages may use instead of a `` tag.
_XML_ENCODING_PATTERN = re.compile(rb'^\s*<\?xml[^>]*\sencoding\s*=\s*["\']([a-z0-9_:.+-]+)', re.IGNORECASE)
@@ -90,6 +106,62 @@
_ENCODING_BY_LABEL = {label: codec for codec, labels in _WHATWG_ENCODING_LABELS.items() for label in labels.split()}
+_FIELD_TAGS = ('input', 'button', 'select', 'textarea')
+_BUTTON_INPUT_TYPES = ('submit', 'image', 'reset', 'button')
+
+# Browsers cut a longer referrer down to the origin.
+_MAX_REFERRER_LENGTH = 4096
+
+_LINE_BREAK_PATTERN = re.compile(r'\r\n|\r|\n')
+
+# Browsers collapse only ASCII whitespace in an option text, keeping non-breaking spaces.
+_ASCII_WHITESPACE_PATTERN = re.compile(r'[ \t\n\f\r]+')
+
+_MULTIPART_NAME_ESCAPES = str.maketrans({'"': '%22', '\r': '%0D', '\n': '%0A'})
+
+
+class FormRequestOptions(TypedDict):
+ """Options for the `Request` created from a form.
+
+ Mirrors `RequestOptions` without the URL, method and payload, which come from the form, without `id` and
+ `unique_key`, which can't be shared by several forms, and without `enqueue_strategy`, which enqueuing sets.
+ """
+
+ label: NotRequired[str | None]
+ """A label routing the request to a specific handler."""
+
+ headers: NotRequired[HttpHeaders | dict[str, str] | None]
+ """HTTP headers of the request, replacing those the form sets, like `Content-Type` or `Referer`."""
+
+ session_id: NotRequired[str | None]
+ """ID of the `Session` the request is bound to."""
+
+ keep_url_fragment: NotRequired[bool]
+ """Whether the URL fragment counts towards the unique key of the request."""
+
+ use_extended_unique_key: NotRequired[bool]
+ """Whether the method and payload count towards the unique key. Defaults to `True` for POST forms."""
+
+ always_enqueue: NotRequired[bool]
+ """Whether to enqueue the request even if it's already in the queue."""
+
+ user_data: NotRequired[Mapping[str, JsonSerializable]]
+ """Custom data stored with the request."""
+
+ no_retry: NotRequired[bool]
+ """Whether to skip retrying the request if it fails."""
+
+ max_retries: NotRequired[int | None]
+ """The maximum number of retries of the request."""
+
+
+class _Field(NamedTuple):
+ """A single entry the form submits."""
+
+ name: str
+ value: str
+ is_file: bool = False
+
def get_declared_html_encoding(body: bytes, content_type: str | None) -> str | None:
"""Get the Python codec for the encoding an HTML response body declares.
@@ -110,7 +182,7 @@ def get_declared_html_encoding(body: bytes, content_type: str | None) -> str | N
return encoding
header_charset = parse_content_type_charset(content_type)
- return _resolve_encoding(header_charset) or _find_declared_encoding(body)
+ return resolve_encoding(header_charset) or _find_declared_encoding(body)
def decode_html_body(body: bytes, encoding: str) -> str:
@@ -125,18 +197,94 @@ def decode_html_body(body: bytes, encoding: str) -> str:
return body.decode(encoding, 'replace').removeprefix('\ufeff')
-def _resolve_encoding(label: str | None) -> str | None:
+def resolve_encoding(label: str | None) -> str | None:
"""Get the Python codec for a WHATWG encoding label, or `None` if browsers don't know the label."""
if not label:
return None
return _ENCODING_BY_LABEL.get(label.lower())
+def strip_html_comments(body: bytes) -> bytes:
+ """Remove the HTML comments from the body, so a commented-out declaration doesn't count."""
+ return _HTML_COMMENT_PATTERN.sub(b'', body)
+
+
+def forms_to_requests(
+ forms: Sequence[HtmlElement],
+ page_url: str,
+ page_encoding: str,
+ *,
+ fields: Mapping[str, str | Sequence[str] | None] | None = None,
+ click: bool | Mapping[str, str] = True,
+ all_forms: bool = False,
+ **kwargs: Unpack[FormRequestOptions],
+) -> list[Request]:
+ """Create a `Request` submitting one of the forms, or each of them, the way a browser does.
+
+ Args:
+ forms: The form elements, each within the lxml tree of the whole page.
+ page_url: The URL of the page, which forms without an action submit to and the `Referer` comes from.
+ page_encoding: The Python codec the page was decoded with, which forms submit in by default.
+ fields: Field values to submit, see `extract_form_requests` of the crawling contexts.
+ click: The submit button to click, see `extract_form_requests`.
+ all_forms: Whether to submit each form, filling in only the fields it has, see `extract_form_requests`.
+ **kwargs: Additional options passed to `Request.from_url`.
+ """
+ if not forms:
+ return []
+
+ root = forms[0].getroottree().getroot()
+ base = root.find('.//base[@href]')
+ try:
+ base_url = convert_to_absolute_url(page_url, '' if base is None else base.get('href').strip())
+ except ValueError:
+ base_url = page_url
+
+ page_elements = list(root.iter(*_FIELD_TAGS))
+ elements_by_form = _elements_by_form(root, page_elements)
+ disabled = _disabled_elements(root, page_elements)
+
+ if not all_forms:
+ # The form sharing the most field names with `fields` is the one to fill in, and ties keep document order.
+ # Forms sharing fewer names never get the values, even if the best one can't be submitted.
+ wanted = set(fields or ())
+ shared = {form: len(wanted & _field_names(elements_by_form.get(form, []))) for form in forms}
+ most_shared = max(shared.values())
+ forms = [form for form in forms if shared[form] == most_shared]
+
+ requests = []
+ for form in forms:
+ elements = elements_by_form.get(form, [])
+ form_fields = fields
+ if all_forms and fields:
+ # Values meant for one form don't spread to the others, like credentials into a search form.
+ names = _field_names(elements)
+ form_fields = {name: value for name, value in fields.items() if name in names}
+
+ request = _form_to_request(
+ form,
+ elements,
+ page_url=page_url,
+ base_url=base_url,
+ page_encoding=page_encoding,
+ disabled=disabled,
+ fields=form_fields,
+ click=click,
+ **kwargs,
+ )
+ if request is None:
+ continue
+ requests.append(request)
+ if not all_forms:
+ break
+ return requests
+
+
def _find_declared_encoding(body: bytes) -> str | None:
"""Find the encoding declared by an XML declaration or a `` tag near the start of the body."""
- prescan = _HTML_COMMENT_PATTERN.sub(b'', body[:_PRESCAN_BYTES])
+ prescan = strip_html_comments(body[:_PRESCAN_BYTES])
xml_match = _XML_ENCODING_PATTERN.match(prescan)
- xml_encoding = _resolve_encoding(xml_match.group(1).decode('ascii')) if xml_match else None
+ xml_encoding = resolve_encoding(xml_match.group(1).decode('ascii')) if xml_match else None
encoding = xml_encoding or _find_meta_encoding(prescan)
# A declaration readable as ASCII rules out UTF-16, so browsers read such pages as UTF-8.
return 'utf-8' if encoding and encoding.startswith('utf-16') else encoding
@@ -161,7 +309,315 @@ def _find_meta_encoding(prescan: bytes) -> str | None:
else:
continue
- encoding = _resolve_encoding(label)
+ encoding = resolve_encoding(label)
if encoding:
return encoding
return None
+
+
+def _elements_by_form(root: HtmlElement, elements: list[HtmlElement]) -> dict[HtmlElement, list[HtmlElement]]:
+ """Group the fields and buttons of the page by the form they belong to, in document order."""
+ # Walking each form avoids an ancestor walk per field.
+ enclosing_form: dict[HtmlElement, HtmlElement] = {}
+ for form in root.iter('form'):
+ enclosing_form.update(dict.fromkeys(form.iter(*_FIELD_TAGS), form))
+
+ # The first element with a given ID wins, as in `getElementById`. Only the `form` attribute needs them.
+ elements_by_id: dict[str, HtmlElement] = {}
+ if any(element.get('form') is not None for element in elements):
+ for element in root.xpath('//*[@id!=""]'):
+ elements_by_id.setdefault(element.get('id'), element)
+
+ elements_by_form: dict[HtmlElement, list[HtmlElement]] = {}
+ for element in elements:
+ form_id = element.get('form')
+ owner = enclosing_form.get(element) if form_id is None else elements_by_id.get(form_id)
+ if owner is not None and owner.tag == 'form':
+ elements_by_form.setdefault(owner, []).append(element)
+ return elements_by_form
+
+
+def _disabled_elements(root: HtmlElement, elements: list[HtmlElement]) -> set[HtmlElement]:
+ """Find the disabled fields and buttons, including those inside a disabled `