Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 3 additions & 3 deletions docling/datamodel/pipeline_options.py
Original file line number Diff line number Diff line change
Expand Up @@ -2027,9 +2027,9 @@ class HeadingHierarchyOptions(BaseModel):
Field(
description=(
"Optional override of the numbering-scheme precedence (highest level first). "
"Known schemes: 'part', 'chapter', 'article', 'roman_u', 'arabic', "
"'alpha_u', 'alpha_l', 'roman_l'. When None, a default legal/regulatory "
"ordering is used."
"Known schemes: 'part', 'division', 'subdivision', 'chapter', 'article', "
"'roman_u', 'arabic', 'alpha_u', 'alpha_l', 'roman_l'. When None, a default "
"legal/regulatory ordering is used."
)
),
] = None
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -51,6 +51,8 @@
# the ``arabic`` rank and is ordered below it by its segment depth (1.1 below 1.).
_DEFAULT_FAMILY_ORDER = [
"part", # PART I / TITLE I / BOOK I
"division", # DIVISION I / Division 1 / Division A
"subdivision", # SUBDIVISION A / Subdivision a
"chapter", # CHAPTER 1
"article", # ARTICLE 1 / SECTION 1 / Clause / § 1
"roman_u", # I. II. III.
Expand All @@ -74,6 +76,14 @@
_KW_ARTICLE = re.compile(
r"^(article|section|clause|schedule|annex|appendix|rule)\b", re.IGNORECASE
)
# "Division"/"Subdivision" are ordinary words too ("Division of Powers"), so -- unlike
# part/chapter/article -- they only count as numbering when an explicit enumerator (an
# Arabic index, a Roman numeral, or a single letter) follows. The enumerator must be
# terminated by whitespace, end-of-string, or separator punctuation, so a French elision
# such as "Division d'appel" (a letter followed by an apostrophe) is not read as a marker.
_KW_ENUM = r"\s+(\d+|[a-z]|[ivxlcdm]+)(?=\s|$|[.,:;)\]\u2013\u2014-])"
_KW_DIVISION = re.compile(r"^division" + _KW_ENUM, re.IGNORECASE)
_KW_SUBDIVISION = re.compile(r"^subdivision" + _KW_ENUM, re.IGNORECASE)
_SECTION_SYMBOL = re.compile(r"^§+\s*\d") # § 1 / §§ 1.2
_SEP = r"(?:[)\]]|[:\-\u2013\u2014](?=\s|$))"
# Dotted decimal outline (1.1, 1.1.1, ...), terminated by punctuation/space/end.
Expand Down Expand Up @@ -117,6 +127,18 @@ def _classify_letter(token: str) -> _Marker | None:
return None


def _keyword_enumerator(m: re.Match[str] | None) -> bool:
"""True when a keyword match carries a real enumerator (digit, letter, or Roman).

Multi-letter tokens are only accepted when they are valid Roman numerals, so
"Division Civil Remedies" (all Roman-numeral letters) is not mistaken for a marker.
"""
if m is None:
return False
tok = m.group(1)
return tok.isdigit() or len(tok) == 1 or _is_roman(tok)


def _parse_marker(text: str) -> _Marker | None:
"""Extract the leading numbering marker from a heading, or None if unnumbered."""
s = (text or "").strip()
Expand All @@ -125,6 +147,10 @@ def _parse_marker(text: str) -> _Marker | None:

if _KW_PART.match(s):
return _Marker(family="part")
if _keyword_enumerator(_KW_DIVISION.match(s)):
return _Marker(family="division")
if _keyword_enumerator(_KW_SUBDIVISION.match(s)):
return _Marker(family="subdivision")
if _KW_CHAPTER.match(s):
return _Marker(family="chapter")
if _KW_ARTICLE.match(s) or _SECTION_SYMBOL.match(s):
Expand Down Expand Up @@ -380,7 +406,7 @@ def _infer_from_style(
# matches an on-page heading "1.1 Definitions" (and vice-versa).
_LEADING_MARKER = re.compile(
r"^\s*(?:"
r"(?:part|title|book|chapter|article|section|clause|schedule|annex|appendix|rule)"
r"(?:part|title|book|division|subdivision|chapter|article|section|clause|schedule|annex|appendix|rule)"
r"\b[\s.:]*[0-9ivxlcdm]*"
r"|§+\s*[0-9.]+"
r"|\(?[0-9]+(?:\.[0-9]+)*[.)\]]?"
Expand Down
10 changes: 7 additions & 3 deletions docs/usage/heading_levels.md
Original file line number Diff line number Diff line change
Expand Up @@ -88,15 +88,19 @@ Two things are worth knowing about this pass:

For everything the outline does not cover, the leading marker of the heading text is the most
reliable signal — on legal and regulatory documents far more reliable than styling, which tends to
be uniform throughout. Docling recognizes keyword markers (`PART`, `TITLE`, `BOOK`, `CHAPTER`,
be uniform throughout. Docling recognizes keyword markers (`PART`, `TITLE`, `BOOK`, `DIVISION`, `SUBDIVISION`, `CHAPTER`,
`ARTICLE`, `SECTION`, `CLAUSE`, `SCHEDULE`, `ANNEX`, `APPENDIX`, `RULE`, `§`), Roman and Arabic
numerals, dotted decimals and parenthesized letters, and ranks them in this default order:

```text
part → chapter → article → roman_u → arabic → alpha_u → alpha_l → roman_l
PART I CHAPTER 1 ARTICLE 1 I. 1. A. (a) (i)
part → division → subdivision → chapter → article → roman_u → arabic → alpha_u → alpha_l → roman_l
PART I DIVISION I SUBDIVISION A CHAPTER 1 ARTICLE 1 I. 1. A. (a) (i)
```

`DIVISION` and `SUBDIVISION` are ordinary words too, so they are read as markers only when a number, letter, or Roman numeral follows.

The default order places `part`/`title` above `division`; in codes where a `Division` sits above a `Title` (some US omnibus acts), reorder with `numbering_schemes` so `division` comes first.

Dotted decimals share the `arabic` rank and sort by their depth, so `1.1` lands one level below
`1.` and `1.1.1` one below that. If your documents follow a different convention, reorder the
scheme names with `numbering_schemes` (highest level first).
Expand Down
38 changes: 38 additions & 0 deletions tests/test_heading_hierarchy.py
Original file line number Diff line number Diff line change
Expand Up @@ -171,10 +171,38 @@ def test_keyword_part_and_article():
assert _parse_marker("§ 1.2 Liability").family == "article"


def test_keyword_division_and_subdivision():
assert _parse_marker("DIVISION I").family == "division"
assert _parse_marker("DIVISION A").family == "division"
assert _parse_marker("Division 1").family == "division"
assert _parse_marker("Division 1 Offences and Penalties").family == "division"
assert _parse_marker("SUBDIVISION A").family == "subdivision"
assert _parse_marker("Subdivision a").family == "subdivision"
# Enumerators may be terminated by separator punctuation, not only whitespace.
assert _parse_marker("Division I, General").family == "division"
assert _parse_marker("Division A; Scope").family == "division"

# Division/subdivision outrank article subsections in the default order.
levels = _levels(["Division 1 General", "Article 3 Powers"])
assert levels == {0: 1, 1: 2}

# Stripping the leading marker stays in sync for bookmark matching.
assert _strip_marker("Division I Powers").strip() == "Powers"


def test_non_marker_text_is_ignored():
assert _parse_marker("Summary") is None
assert _parse_marker("Introduction to the topic") is None
assert _parse_marker("ABSTRACT") is None
assert _parse_marker("Division of Property") is None
assert _parse_marker("Division of weeks of benefits") is None
assert _parse_marker("Division du cong\u00e9") is None
assert _parse_marker("Subdivision of land") is None
assert _parse_marker("Division Civil Remedies") is None
assert _parse_marker("Division d'appel") is None
assert _parse_marker("Division l'emploi") is None
assert _parse_marker("Division s'applique") is None
assert _parse_marker("Division d\u2019appel") is None
assert _parse_marker("2024 Annual Financial Report") is None
assert _parse_marker("2024-2025 Budget") is None
assert _parse_marker("2024\u201325 Outlook") is None
Expand Down Expand Up @@ -228,6 +256,16 @@ def test_custom_numbering_scheme_order():
assert levels == {0: 2, 1: 1}


def test_omitted_scheme_family_ranks_lowest():
# A custom order that omits "division": the heading still parses (family "division")
# but, absent from the order, it ranks below any listed family instead of above it.
levels = _levels(
["Division 1 Offences", "1. Scope"],
numbering_schemes=["arabic", "roman_u"],
)
assert levels == {0: 2, 1: 1}


def test_max_level_clamping_on_document():
doc = DoclingDocument(name="t")
for text in ["1. A", "1.1 B", "1.1.1 C", "1.1.1.1 D"]:
Expand Down
Loading