Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
119 changes: 93 additions & 26 deletions docling/backend/html_backend.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,7 @@
from email.parser import BytesParser
from functools import cache
from io import BytesIO
from itertools import takewhile
from pathlib import Path, PureWindowsPath
from typing import Any, Final, Iterator, Literal, Optional, Union, cast
from urllib.parse import unquote, urljoin, urlparse
Expand All @@ -39,6 +40,7 @@
GraphLinkLabel,
GroupItem,
GroupLabel,
ListGroup,
PictureClassificationLabel,
PictureClassificationMetaField,
PictureClassificationPrediction,
Expand Down Expand Up @@ -2052,7 +2054,7 @@ def parse_table_data(
doc.add_table_cell(table_item=docling_table, cell=simple_cell)
return data

def _walk(self, element: Tag, doc: DoclingDocument) -> list[RefItem]: # noqa: C901
def _walk(self, element: Tag, doc: DoclingDocument) -> list[RefItem]:
"""Parse an XML tag by recursively walking its content.

While walking, the method buffers inline text across tags like <b> or <span>,
Expand All @@ -2062,6 +2064,18 @@ def _walk(self, element: Tag, doc: DoclingDocument) -> list[RefItem]: # noqa: C
element: The XML tag to parse.
doc: The Docling document to be updated with the parsed content.
"""
return self._walk_nodes(element, element.contents, doc)

def _walk_nodes( # noqa: C901
self, element: Tag, nodes: list[PageElement], doc: DoclingDocument
) -> list[RefItem]:
"""Parse some children of an XML tag, like `_walk` does for all of them.

Args:
element: The XML tag whose children are parsed.
nodes: The children of `element` to parse.
doc: The Docling document to be updated with the parsed content.
"""
added_refs: list[RefItem] = []
buffer: AnnotatedTextList = AnnotatedTextList()

Expand Down Expand Up @@ -2128,7 +2142,7 @@ def _flush_buffer() -> None:
if inline_ref is not None:
added_refs.append(inline_ref)

for node in element.contents:
for node in nodes:
if isinstance(node, Tag):
name = node.name.lower()
if form_field := self._consume_form_field_for_tag(node):
Expand Down Expand Up @@ -2903,7 +2917,11 @@ def _description_list_children(dl: Tag) -> list[PageElement]:
children.append(child)
return children

def _handle_list(self, tag: Tag, doc: DoclingDocument) -> RefItem:
def _handle_list( # noqa: C901
self, tag: Tag, doc: DoclingDocument
) -> list[RefItem]:
"""Parse a list tag and return the items added at the current level."""
added_refs: list[RefItem] = []
tag_name = tag.name.lower()
start: Optional[int] = None
name: str = ""
Expand All @@ -2920,17 +2938,40 @@ def _handle_list(self, tag: Tag, doc: DoclingDocument) -> RefItem:
else:
name = "list"

# Children of <ul>/<ol> other than <li> are invalid HTML, but common in CMS
# output. Those before the first <li> precede the list. The others interrupt
# it, like the interleaved blocks of the DOCX backend (#3896): the list is
# closed, the content is emitted at the parent level, and the following items
# open a new list group that continues the numbering.
leading: list[PageElement] = []
if not is_description:
leading = list(
takewhile(
lambda node: not (isinstance(node, Tag) and node.name == "li"),
tag.contents,
)
)
added_refs.extend(self._walk_nodes(tag, leading, doc))

# Create the list container
list_group = doc.add_list_group(
name=name,
parent=self.parents[self.level],
content_layer=self.content_layer,
)
self.parents[self.level + 1] = list_group
self.ctx.list_ordered_flag_by_ref[list_group.self_ref] = is_ordered
if is_ordered and start is not None:
self.ctx.list_start_by_ref[list_group.self_ref] = start
self.level += 1
def open_list_group(group_start: Optional[int]) -> ListGroup:
group_name = name
if is_ordered and group_start is not None:
group_name = f"ordered list start {group_start}"
group = doc.add_list_group(
name=group_name,
parent=self.parents[self.level],
content_layer=self.content_layer,
)
added_refs.append(group.get_ref())
self.parents[self.level + 1] = group
self.ctx.list_ordered_flag_by_ref[group.self_ref] = is_ordered
if is_ordered and group_start is not None:
self.ctx.list_start_by_ref[group.self_ref] = group_start
self.level += 1
return group

list_group: Optional[ListGroup] = open_list_group(start)

# Track the number of list items added (not all children)
list_item_counter: int = 0
Expand Down Expand Up @@ -3004,19 +3045,44 @@ def _handle_list(self, tag: Tag, doc: DoclingDocument) -> RefItem:

self.parents[self.level + 1] = None
self.level -= 1
return list_group.get_ref()
return added_refs

# For each top-level <li> in this list (ul/ol)
for li in tag.find_all({"li", "ul", "ol"}, recursive=False):
if not isinstance(li, Tag):
# For each child of this list (ul/ol) after the leading content
prev_item: Optional[RefItem] = None
for li in tag.contents[len(leading) :]:
if isinstance(li, NavigableString) and (
isinstance(li, PreformattedString) or not li.strip()
):
continue

# sub-list items should be indented under main list items, but temporarily
# addressing invalid HTML (docling-core/issues/357)
if li.name in {"ul", "ol"}:
self._handle_block(li, doc)
if not (isinstance(li, Tag) and li.name == "li"):
if isinstance(li, Tag) and li.name in {"ul", "ol"} and prev_item:
# A nested list is a sub-list of the preceding list item
with self._use_list_item_context(prev_item):
self._walk_nodes(tag, [li], doc)
else:
if list_group is not None:
self.parents[self.level + 1] = None
self.level -= 1
parent = self.parents[self.level] or doc.body
n_children = len(parent.children)
added_refs.extend(self._walk_nodes(tag, [li], doc))
if list_group is not None and len(parent.children) == n_children:
# Nothing was added, e.g. a <br>: the list stays open
self.parents[self.level + 1] = list_group
self.level += 1
else:
list_group = None
prev_item = None

else:
if list_group is None:
if is_ordered and start is None:
start = 1
list_group = open_list_group(
start + list_item_counter if start is not None else None
)

# 1) determine the marker using the counter
marker: str = (
f"{start + list_item_counter}."
Expand Down Expand Up @@ -3063,6 +3129,7 @@ def _handle_list(self, tag: Tag, doc: DoclingDocument) -> RefItem:
# Increment counter only when a list item is actually added
if list_item:
list_item_counter += 1
prev_item = list_item

if list_item or inputs_in_li or custom_checkboxes_in_li:
if task_list_inputs:
Expand Down Expand Up @@ -3094,9 +3161,10 @@ def _handle_list(self, tag: Tag, doc: DoclingDocument) -> RefItem:
if not has_list_ancestor:
self._handle_block(sublist, doc)

self.parents[self.level + 1] = None
self.level -= 1
return list_group.get_ref()
if list_group is not None:
self.parents[self.level + 1] = None
self.level -= 1
return added_refs

@staticmethod
def get_html_table_row_col(tag: Tag) -> tuple[int, int]:
Expand Down Expand Up @@ -3171,8 +3239,7 @@ def _handle_block(self, tag: Tag, doc: DoclingDocument) -> list[RefItem]: # noq
added_refs.extend(heading_refs)

elif tag_name in {"ul", "ol", "dl"}:
list_ref = self._handle_list(tag, doc)
added_refs.append(list_ref)
added_refs.extend(self._handle_list(tag, doc))

elif tag_name in {"p", "address", "summary"}:
text_list = self._extract_text_and_hyperlink_recursively(
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -30,4 +30,11 @@ item-0 at level 0: unspecified: group _root_
item-29 at level 4: list_item: Step instruction text
item-30 at level 5: list: group list
item-31 at level 6: list_item: Nested detail 1
item-32 at level 6: list_item: Nested detail 2
item-32 at level 6: list_item: Nested detail 2
item-33 at level 2: section_header: Paragraph between list items
item-34 at level 3: list: group list
item-35 at level 4: list_item: First item
item-36 at level 3: text: A paragraph placed directly inside the list.
item-37 at level 3: list: group list
item-38 at level 4: list_item: Second item
item-39 at level 3: text: After.
111 changes: 110 additions & 1 deletion tests/data/html/groundtruth/html_nested_block_in_list_item.html.json
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,7 @@
"name": "html_nested_block_in_list_item",
"origin": {
"mimetype": "text/html",
"binary_hash": 2267456859147812713,
"binary_hash": 2633762625787780656,
"filename": "html_nested_block_in_list_item.html"
},
"furniture": {
Expand Down Expand Up @@ -40,6 +40,9 @@
},
{
"$ref": "#/texts/17"
},
{
"$ref": "#/texts/21"
}
],
"content_layer": "body",
Expand Down Expand Up @@ -187,6 +190,34 @@
"content_layer": "body",
"name": "list",
"label": "list"
},
{
"self_ref": "#/groups/9",
"parent": {
"$ref": "#/texts/21"
},
"children": [
{
"$ref": "#/texts/22"
}
],
"content_layer": "body",
"name": "list",
"label": "list"
},
{
"self_ref": "#/groups/10",
"parent": {
"$ref": "#/texts/21"
},
"children": [
{
"$ref": "#/texts/24"
}
],
"content_layer": "body",
"name": "list",
"label": "list"
}
],
"texts": [
Expand Down Expand Up @@ -517,6 +548,84 @@
"text": "Nested detail 2",
"enumerated": false,
"marker": ""
},
{
"self_ref": "#/texts/21",
"parent": {
"$ref": "#/groups/0"
},
"children": [
{
"$ref": "#/groups/9"
},
{
"$ref": "#/texts/23"
},
{
"$ref": "#/groups/10"
},
{
"$ref": "#/texts/25"
}
],
"content_layer": "body",
"label": "section_header",
"prov": [],
"orig": "Paragraph between list items",
"text": "Paragraph between list items",
"level": 1
},
{
"self_ref": "#/texts/22",
"parent": {
"$ref": "#/groups/9"
},
"children": [],
"content_layer": "body",
"label": "list_item",
"prov": [],
"orig": "First item",
"text": "First item",
"enumerated": false,
"marker": ""
},
{
"self_ref": "#/texts/23",
"parent": {
"$ref": "#/texts/21"
},
"children": [],
"content_layer": "body",
"label": "text",
"prov": [],
"orig": "A paragraph placed directly inside the list.",
"text": "A paragraph placed directly inside the list."
},
{
"self_ref": "#/texts/24",
"parent": {
"$ref": "#/groups/10"
},
"children": [],
"content_layer": "body",
"label": "list_item",
"prov": [],
"orig": "Second item",
"text": "Second item",
"enumerated": false,
"marker": ""
},
{
"self_ref": "#/texts/25",
"parent": {
"$ref": "#/texts/21"
},
"children": [],
"content_layer": "body",
"label": "text",
"prov": [],
"orig": "After.",
"text": "After."
}
],
"pictures": [
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -19,4 +19,14 @@

- Step instruction text
- Nested detail 1
- Nested detail 2
- Nested detail 2

## Paragraph between list items

- First item

A paragraph placed directly inside the list.

- Second item

After.
11 changes: 11 additions & 0 deletions tests/data/html/sources/html_nested_block_in_list_item.html
Original file line number Diff line number Diff line change
Expand Up @@ -34,5 +34,16 @@ <h2>Nested list in divs</h2>
</div>
</li>
</ul>

<!-- Test case 4: Paragraph directly inside a list. This is not valid HTML, but it
is common in CMS output: the paragraph closes the list, and the items that
follow open a new list. -->
<h2>Paragraph between list items</h2>
<ul>
<li>First item</li>
<p>A paragraph placed directly inside the list.</p>
<li>Second item</li>
</ul>
<p>After.</p>
</body>
</html>
Loading
Loading