diff --git a/docling/backend/html_backend.py b/docling/backend/html_backend.py index e123f6c107..57db701776 100644 --- a/docling/backend/html_backend.py +++ b/docling/backend/html_backend.py @@ -17,6 +17,7 @@ from email.parser import BytesParser from functools import cache from io import BytesIO +from itertools import takewhile from pathlib import Path, PureWindowsPath from typing import Any, Final, Iterator, Literal, Optional, Union, cast from urllib.parse import unquote, urljoin, urlparse @@ -39,6 +40,7 @@ GraphLinkLabel, GroupItem, GroupLabel, + ListGroup, PictureClassificationLabel, PictureClassificationMetaField, PictureClassificationPrediction, @@ -2052,7 +2054,7 @@ def parse_table_data( doc.add_table_cell(table_item=docling_table, cell=simple_cell) return data - def _walk(self, element: Tag, doc: DoclingDocument) -> list[RefItem]: # noqa: C901 + def _walk(self, element: Tag, doc: DoclingDocument) -> list[RefItem]: """Parse an XML tag by recursively walking its content. While walking, the method buffers inline text across tags like or , @@ -2062,6 +2064,18 @@ def _walk(self, element: Tag, doc: DoclingDocument) -> list[RefItem]: # noqa: C element: The XML tag to parse. doc: The Docling document to be updated with the parsed content. """ + return self._walk_nodes(element, element.contents, doc) + + def _walk_nodes( # noqa: C901 + self, element: Tag, nodes: list[PageElement], doc: DoclingDocument + ) -> list[RefItem]: + """Parse some children of an XML tag, like `_walk` does for all of them. + + Args: + element: The XML tag whose children are parsed. + nodes: The children of `element` to parse. + doc: The Docling document to be updated with the parsed content. + """ added_refs: list[RefItem] = [] buffer: AnnotatedTextList = AnnotatedTextList() @@ -2128,7 +2142,7 @@ def _flush_buffer() -> None: if inline_ref is not None: added_refs.append(inline_ref) - for node in element.contents: + for node in nodes: if isinstance(node, Tag): name = node.name.lower() if form_field := self._consume_form_field_for_tag(node): @@ -2903,7 +2917,11 @@ def _description_list_children(dl: Tag) -> list[PageElement]: children.append(child) return children - def _handle_list(self, tag: Tag, doc: DoclingDocument) -> RefItem: + def _handle_list( # noqa: C901 + self, tag: Tag, doc: DoclingDocument + ) -> list[RefItem]: + """Parse a list tag and return the items added at the current level.""" + added_refs: list[RefItem] = [] tag_name = tag.name.lower() start: Optional[int] = None name: str = "" @@ -2920,17 +2938,40 @@ def _handle_list(self, tag: Tag, doc: DoclingDocument) -> RefItem: else: name = "list" + # Children of + + +

Paragraph between list items

+ +

After.

\ No newline at end of file diff --git a/tests/test_backend_html.py b/tests/test_backend_html.py index c789dec204..f71aa1ce57 100644 --- a/tests/test_backend_html.py +++ b/tests/test_backend_html.py @@ -13,7 +13,12 @@ import pytest import requests from bs4 import BeautifulSoup -from docling_core.types.doc import DocItemLabel, PictureItem, RichTableCell +from docling_core.types.doc import ( + DocItemLabel, + GroupLabel, + PictureItem, + RichTableCell, +) from docling_core.types.doc.document import ContentLayer from pydantic import AnyUrl, ValidationError @@ -543,6 +548,79 @@ def test_nested_table_in_list_item(): assert md.count("Fault type.") == 1 +def test_list_non_li_children(): + """Regression for #4424: children of