Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
39 changes: 20 additions & 19 deletions docling/backend/html_backend.py
Original file line number Diff line number Diff line change
Expand Up @@ -485,18 +485,21 @@ def __init__(
if isinstance(path_or_stream, BytesIO)
else Path(path_or_stream).read_bytes()
)
markup: str | bytes = raw
if self.input_format == InputFormat.MHTML:
if options.render_page:
raise DocumentLoadError(
"Browser rendering is not supported for MHTML input."
)
raw, resources, root_location = self._parse_mhtml(
markup, resources, root_location = self._parse_mhtml(
raw, configured_base_path
)
self._mhtml_resources = resources
self.base_path = root_location or configured_base_path
self._raw_html_bytes = raw
self.soup = BeautifulSoup(raw, "html.parser")
self._raw_html_bytes = (
markup.encode("utf-8") if isinstance(markup, str) else markup
)
self.soup = BeautifulSoup(markup, "html.parser")
except DocumentLoadError:
raise
except Exception as e:
Expand Down Expand Up @@ -613,23 +616,20 @@ def _decode_mime_payload(part: Message) -> bytes:
return payload if isinstance(payload, bytes) else b""

@classmethod
def _decode_mhtml_html(cls, part: Message) -> bytes:
"""Decode the transfer encoding and re-encode the HTML payload as UTF-8.

Note:
Re-encoding to UTF-8 does not strip or update any ``<meta charset>``
or ``Content-Type`` meta tags inside the HTML. If the document declares
a non-UTF-8 charset internally, BeautifulSoup may attempt to re-decode
the already-UTF-8 bytes using that charset, which can produce corrupted
text. Stripping the meta charset declaration before passing the bytes to
the parser would fix this but is left as a future improvement.
def _decode_mhtml_html(cls, part: Message) -> str | bytes:
"""Decode the transfer encoding and the declared charset of the HTML part.

A charset declared on the MIME part takes precedence over any
``<meta charset>`` in the page, so the HTML is returned as text and the
parser cannot decode it a second time. Without a usable charset the raw
bytes are returned for the parser to detect the encoding itself.
"""
payload = cls._decode_mime_payload(part)
charset = part.get_content_charset()
if not charset:
return payload
try:
return payload.decode(charset, errors="replace").encode("utf-8")
return payload.decode(charset, errors="replace")
except LookupError:
return payload

Expand Down Expand Up @@ -797,7 +797,7 @@ def _collect_mhtml_resources(
@classmethod
def _parse_mhtml(
cls, raw: bytes, configured_base: str | None
) -> tuple[bytes, dict[str, bytes], str]:
) -> tuple[str | bytes, dict[str, bytes], str]:
"""Parse an MHTML archive and extract the HTML root, image resources, and base URL.

Args:
Expand All @@ -806,7 +806,8 @@ def _parse_mhtml(
``HTMLBackendOptions.source_uri``), used to resolve local roots.

Returns:
A tuple of ``(html_bytes, resources, effective_base)`` where
A tuple of ``(html, resources, effective_base)`` where ``html`` is
text if the root part declares a charset and raw bytes otherwise,
``resources`` maps location keys to raw image bytes and
``effective_base`` is the resolved base URL for further reference
resolution.
Expand All @@ -819,15 +820,15 @@ def _parse_mhtml(
raise ValueError("MHTML input has no MIME Content-Type header.")

related_scope, root_part = cls._find_mhtml_root(message)
html_bytes = cls._decode_mhtml_html(root_part)
if not html_bytes.strip():
html = cls._decode_mhtml_html(root_part)
if not html.strip():
raise ValueError("The MHTML HTML root part is empty.")

root_header = root_part.get("Content-Location")
root_location = str(root_header).strip() if root_header else None
effective_base = cls._resolve_mhtml_base(root_location, configured_base)
resources = cls._collect_mhtml_resources(related_scope, effective_base)
return html_bytes, resources, effective_base
return html, resources, effective_base

@override
def convert(self) -> DoclingDocument:
Expand Down
15 changes: 12 additions & 3 deletions tests/test_backend_mhtml.py
Original file line number Diff line number Diff line change
Expand Up @@ -166,9 +166,18 @@ def test_related_without_start_does_not_select_later_html():
assert result.errors


def test_declared_non_utf8_charset_is_decoded():
html = "<html><body><p>Привет, мир</p></body></html>".encode("windows-1251")
encoded = base64.b64encode(html).decode()
@pytest.mark.parametrize(
"head",
[
"",
'<meta charset="windows-1251">',
'<meta http-equiv="Content-Type" content="text/html; charset=windows-1251">',
],
ids=["no-meta", "meta-charset", "meta-http-equiv"],
)
def test_declared_non_utf8_charset_is_decoded(head: str):
html = f"<html><head>{head}</head><body><p>Привет, мир</p></body></html>"
encoded = base64.b64encode(html.encode("windows-1251")).decode()
data = (
"MIME-Version: 1.0\r\n"
"Content-Type: text/html; charset=windows-1251\r\n"
Expand Down
Loading