diff --git a/docs/examples/code_examples/fill_and_submit_web_form_automated.py b/docs/examples/code_examples/fill_and_submit_web_form_automated.py new file mode 100644 index 0000000000..844b0d9fe3 --- /dev/null +++ b/docs/examples/code_examples/fill_and_submit_web_form_automated.py @@ -0,0 +1,39 @@ +import asyncio + +from crawlee.crawlers import ParselCrawler, ParselCrawlingContext + + +async def main() -> None: + crawler = ParselCrawler() + + # Fill in the form on the page and enqueue its submission. + @crawler.router.default_handler + async def request_handler(context: ParselCrawlingContext) -> None: + context.log.info(f'Filling in the form on {context.request.url} ...') + requests = await context.extract_form_requests( + fields={ + 'custname': 'John Doe', + 'custtel': '1234567890', + 'custemail': 'johndoe@example.com', + 'size': 'large', + 'topping': ['bacon', 'cheese', 'mushroom'], + 'delivery': '13:00', + 'comments': 'Please ring the doorbell upon arrival.', + }, + label='form-result', + ) + await context.add_requests(requests) + + # Process the response to the form submission. + @crawler.router.handler('form-result') + async def form_result_handler(context: ParselCrawlingContext) -> None: + context.log.info(f'Processing {context.request.url} ...') + response = (await context.http_response.read()).decode('utf-8') + context.log.info(f'Response: {response}') # To see the response in the logs. + + # Run the crawler with the page containing the form. + await crawler.run(['https://httpbin.org/forms/post']) + + +if __name__ == '__main__': + asyncio.run(main()) diff --git a/docs/examples/fill_and_submit_web_form.mdx b/docs/examples/fill_and_submit_web_form.mdx index bda46c1d97..e81af4623f 100644 --- a/docs/examples/fill_and_submit_web_form.mdx +++ b/docs/examples/fill_and_submit_web_form.mdx @@ -10,8 +10,9 @@ import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; import RequestExample from '!!raw-loader!roa-loader!./code_examples/fill_and_submit_web_form_request.py'; import CrawlerExample from '!!raw-loader!roa-loader!./code_examples/fill_and_submit_web_form_crawler.py'; +import AutomatedExample from '!!raw-loader!roa-loader!./code_examples/fill_and_submit_web_form_automated.py'; -This example demonstrates how to fill and submit a web form using the `HttpCrawler` crawler. The same approach applies to any crawler that inherits from it, such as the `BeautifulSoupCrawler` or `ParselCrawler`. +This example demonstrates how to fill and submit a web form using the `HttpCrawler` crawler. The same approach applies to any crawler that inherits from it, such as the `BeautifulSoupCrawler` or `ParselCrawler`. These two crawlers can also [fill in the form automatically](#fill-in-the-form-automatically). We are going to use the [httpbin.org](https://httpbin.org) website to demonstrate how it works. @@ -118,3 +119,21 @@ Finally, run your crawler. Your logs should show something like this: ``` This log output confirms that the crawler successfully submitted the form and processed the response. Congratulations! You have successfully filled and submitted a web form using the `HttpCrawler`. + +## Fill in the form automatically + +The `ParselCrawler` and `BeautifulSoupCrawler` can build the form request for you. Their crawling contexts provide the `extract_form_requests` helper, which reads the form from the page, fills in your values and returns a list with the request that submits it the way a browser does. The action URL, the method and the encoding come from the form itself, so you only need the field names from [Investigate the form fields](#investigate-the-form-fields). + +The crawler below opens the page with the form. The default handler fills in the form with the `fields` argument and enqueues the submission with a label. A separate handler for that label processes the response. + + + {AutomatedExample} + + +Note that: + +- `fields` replaces the values of the listed fields and adds the ones the form doesn't have. A list submits the field once per value, as with the `topping` checkboxes. +- Fields you don't list keep the values from the page, so hidden inputs such as CSRF tokens are submitted as they are. A CSRF token is tied to the session cookie, so pass `session_id=context.session.id` to send the form in the same session. The request also carries the `Referer` and `Origin` headers a browser sends, which some CSRF checks require. +- On a page with several forms, the helper submits the one sharing the most field names with `fields`, or the first one if none shares any. It skips forms that can't be submitted, for example because their action is JavaScript, but never falls back to a form sharing fewer names, so the list can be empty. To pick a form yourself, pass a CSS selector such as `selector='#order'`. To submit each form, pass `all_forms=True`. Then `fields` only replaces the fields each form has. +- The first enabled submit button of the form is clicked by default, and a form without one is submitted anyway. Use the `click` argument to pick another button by its attributes, even a disabled one, or to submit without one. +- The page decides where its form is sent. To enqueue only requests to the same host, call `context.add_requests(requests, strategy='same-hostname')`. diff --git a/docs/guides/code_examples/scrapy_migration/crawlee_post.py b/docs/guides/code_examples/scrapy_migration/crawlee_post.py index 2f332426ef..b97a77552e 100644 --- a/docs/guides/code_examples/scrapy_migration/crawlee_post.py +++ b/docs/guides/code_examples/scrapy_migration/crawlee_post.py @@ -1,7 +1,5 @@ import asyncio -from urllib.parse import urlencode -from crawlee import Request from crawlee.crawlers import ParselCrawler, ParselCrawlingContext @@ -14,34 +12,16 @@ async def login_page(context: ParselCrawlingContext) -> None: if not context.session: raise RuntimeError('Session not found') - token = context.selector.css('input[name="csrf_token"]::attr(value)').get() - - # The CSRF token is required for the POST to succeed. If it's missing, - # the login will fail. - if not token: - raise RuntimeError('CSRF token not found') - - form = {'csrf_token': token, 'username': 'user', 'password': 'pass'} - # highlight-start - # Crawlee's `payload` is the raw request body, so encode the fields yourself - # and set the `Content-Type`. Scrapy's `FormRequest` does both for you. - await context.add_requests( - [ - Request.from_url( - 'https://quotes.toscrape.com/login', - method='POST', - payload=urlencode(form), - headers={'content-type': 'application/x-www-form-urlencoded'}, - label='after-login', - # Bind the POST to the same session so its CSRF cookie matches. - session_id=context.session.id, - # The POST shares the GET's URL. Include the method and payload - # in the unique key, or the queue drops it as a duplicate. - use_extended_unique_key=True, - ) - ] + # Like Scrapy's `FormRequest.from_response`, the helper keeps the hidden + # `csrf_token` field, encodes the data and sets the `Content-Type` header. + requests = await context.extract_form_requests( + fields={'username': 'user', 'password': 'pass'}, + label='after-login', + # Bind the POST to the same session so its CSRF cookie matches. + session_id=context.session.id, ) + await context.add_requests(requests) # highlight-end @crawler.router.handler('after-login') diff --git a/docs/guides/scrapy_migration.mdx b/docs/guides/scrapy_migration.mdx index 5acbfa3824..a5e4c29a9e 100644 --- a/docs/guides/scrapy_migration.mdx +++ b/docs/guides/scrapy_migration.mdx @@ -80,7 +80,7 @@ Both frameworks give you a request scheduler, filtering of duplicate requests, r | `response.follow()` / `yield Request(...)` | `enqueue_links` / `add_requests` | | `dont_filter=True` | `Request.from_url(always_enqueue=True)` | | `allowed_domains` | `enqueue_links(strategy=...)` | -| `scrapy.FormRequest` | `Request.from_url(method='POST', payload=...)` | +| `scrapy.FormRequest` | `Request.from_url(method='POST', payload=...)` / `context.extract_form_requests(...)` | | Item pipelines | `Dataset` | | Downloader / spider middlewares | `router.use()`, navigation hooks, HTTP clients | | `settings.py` | `Configuration` + crawler arguments | @@ -263,7 +263,7 @@ Scrapy retries failed requests with `RetryMiddleware` and reports terminal failu ## Forms and login -Scrapy submits forms with `FormRequest`, which encodes `formdata` as `form-urlencoded` and sets the header for you. Crawlee's `payload` takes the raw request body, so encode the fields yourself with `urllib.parse.urlencode` and set the `Content-Type` through `headers=`. For a full login flow with session reuse, see the [Logging in with a crawler guide](./logging-in-with-a-crawler). +Scrapy submits forms with `FormRequest.from_response`, which reads the form from the page, keeps its hidden fields and encodes the data for you. Crawlee's `extract_form_requests` helper does the same in the `ParselCrawler` and `BeautifulSoupCrawler`. Pass your values in `fields` and request options such as `label` or `session_id` as keyword arguments. Like `from_response`, it submits a single form. Scrapy takes the first form by default, while the helper prefers the one sharing the most field names with `fields`. It returns a list, which is empty when no form matches, so enqueue it with `add_requests`. For a plain `FormRequest` that doesn't come from a form on the page, use `Request.from_url`. Its `payload` is the raw request body, so encode the fields with `urllib.parse.urlencode` and set the `Content-Type` through `headers=`. For a full login flow with session reuse, see the [Logging in with a crawler guide](./logging-in-with-a-crawler). diff --git a/src/crawlee/_utils/html.py b/src/crawlee/_utils/html.py index 5b357f312a..511a55e800 100644 --- a/src/crawlee/_utils/html.py +++ b/src/crawlee/_utils/html.py @@ -4,8 +4,24 @@ import codecs import re +from typing import TYPE_CHECKING, NamedTuple, TypedDict +from urllib.parse import urlencode, urlsplit +from yarl import URL + +from crawlee._request import Request +from crawlee._types import HttpHeaders +from crawlee._utils.crypto import compute_short_hash from crawlee._utils.http import parse_content_type_charset +from crawlee._utils.urls import convert_to_absolute_url, is_url_absolute, validate_http_url + +if TYPE_CHECKING: + from collections.abc import Mapping, Sequence + + from lxml.html import HtmlElement + from typing_extensions import NotRequired, Unpack + + from crawlee._types import JsonSerializable # Matches the `encoding` of an XML declaration, which XHTML pages may use instead of a `` tag. _XML_ENCODING_PATTERN = re.compile(rb'^\s*<\?xml[^>]*\sencoding\s*=\s*["\']([a-z0-9_:.+-]+)', re.IGNORECASE) @@ -90,6 +106,68 @@ _ENCODING_BY_LABEL = {label: codec for codec, labels in _WHATWG_ENCODING_LABELS.items() for label in labels.split()} +_FIELD_TAGS = ('input', 'button', 'select', 'textarea') +_BUTTON_INPUT_TYPES = ('submit', 'image', 'reset', 'button') + +# Browsers cut a longer referrer down to the origin. +_MAX_REFERRER_LENGTH = 4096 + +_LINE_BREAK_PATTERN = re.compile(r'\r\n|\r|\n') + +# Browsers collapse only ASCII whitespace in an option text, keeping non-breaking spaces. +_ASCII_WHITESPACE_PATTERN = re.compile(r'[ \t\n\f\r]+') + +_MULTIPART_NAME_ESCAPES = str.maketrans({'"': '%22', '\r': '%0D', '\n': '%0A'}) + +# The URL parser strips C0 controls and spaces around a URL, but no other whitespace. +_URL_STRIP_CHARS = ''.join(map(chr, range(0x21))) + +_COLOR_PATTERN = re.compile(r'#[0-9a-fA-F]{6}') +_FLOAT_PATTERN = re.compile(r'-?(?:\d+(?:\.\d+)?|\.\d+)(?:[eE][-+]?\d+)?', re.ASCII) + + +class FormRequestOptions(TypedDict): + """Options for the `Request` created from a form. + + Mirrors `RequestOptions` without the URL, method and payload, which come from the form, without `id` and + `unique_key`, which can't be shared by several forms, and without `enqueue_strategy`, which enqueuing sets. + """ + + label: NotRequired[str | None] + """A label routing the request to a specific handler.""" + + headers: NotRequired[HttpHeaders | dict[str, str] | None] + """HTTP headers of the request, replacing those the form sets, like `Content-Type` or `Referer`.""" + + session_id: NotRequired[str | None] + """ID of the `Session` the request is bound to.""" + + keep_url_fragment: NotRequired[bool] + """Whether the URL fragment counts towards the unique key of the request.""" + + use_extended_unique_key: NotRequired[bool] + """Whether the method and payload count towards the unique key. Defaults to `True` for POST forms.""" + + always_enqueue: NotRequired[bool] + """Whether to enqueue the request even if it's already in the queue.""" + + user_data: NotRequired[Mapping[str, JsonSerializable]] + """Custom data stored with the request.""" + + no_retry: NotRequired[bool] + """Whether to skip retrying the request if it fails.""" + + max_retries: NotRequired[int | None] + """The maximum number of retries of the request.""" + + +class _Field(NamedTuple): + """A single entry the form submits.""" + + name: str + value: str + is_file: bool = False + def get_declared_html_encoding(body: bytes, content_type: str | None) -> str | None: """Get the Python codec for the encoding an HTML response body declares. @@ -110,7 +188,7 @@ def get_declared_html_encoding(body: bytes, content_type: str | None) -> str | N return encoding header_charset = parse_content_type_charset(content_type) - return _resolve_encoding(header_charset) or _find_declared_encoding(body) + return resolve_encoding(header_charset) or _find_declared_encoding(body) def decode_html_body(body: bytes, encoding: str) -> str: @@ -125,18 +203,99 @@ def decode_html_body(body: bytes, encoding: str) -> str: return body.decode(encoding, 'replace').removeprefix('\ufeff') -def _resolve_encoding(label: str | None) -> str | None: +def resolve_encoding(label: str | None) -> str | None: """Get the Python codec for a WHATWG encoding label, or `None` if browsers don't know the label.""" if not label: return None return _ENCODING_BY_LABEL.get(label.lower()) +def strip_html_comments(body: bytes) -> bytes: + """Remove the HTML comments from the body, so a commented-out declaration doesn't count.""" + return _HTML_COMMENT_PATTERN.sub(b'', body) + + +def forms_to_requests( + forms: Sequence[HtmlElement], + page_url: str, + page_encoding: str, + *, + fields: Mapping[str, str | Sequence[str] | None] | None = None, + click: bool | Mapping[str, str] = True, + all_forms: bool = False, + **kwargs: Unpack[FormRequestOptions], +) -> list[Request]: + """Create a `Request` submitting one of the forms, or each of them, the way a browser does. + + Args: + forms: The form elements, each within the lxml tree of the whole page. + page_url: The URL of the page, which forms without an action submit to and the `Referer` comes from. + page_encoding: The Python codec the page was decoded with, which forms submit in by default. + fields: Field values to submit, see `extract_form_requests` of the crawling contexts. + click: The submit button to click, see `extract_form_requests`. + all_forms: Whether to submit each form, filling in only the fields it has, see `extract_form_requests`. + **kwargs: Additional options passed to `Request.from_url`. + """ + if not forms: + return [] + + root = forms[0].getroottree().getroot() + base = next(iter(root.xpath('//base[@href][not(ancestor::template)]')), None) + try: + base_url = convert_to_absolute_url(page_url, '' if base is None else base.get('href').strip(_URL_STRIP_CHARS)) + except ValueError: + base_url = page_url + + # Browsers keep the contents of a `