Skip to content

Commit aee09c3

Browse files
committed
fix: nest DOCX sub-lists that Word stores as a separate numbering definition
Word can express a nested list either as a deeper w:ilvl within the parent's w:numId, or as a new w:numId at w:ilvl 0 that is set apart only by its indentation. Both render identically in Word, but mammoth derives nesting from w:ilvl alone, so the second form was flattened into the parent list and its items were renumbered as siblings. Extend the existing pre_process_docx step to resolve each level's effective indentation from numbering.xml and walk the document tracking the open list levels, so nesting implied by indentation is restored before mammoth reads the file. Within one w:numId the declared w:ilvl stays authoritative, since some numbering definitions give several levels the same indentation. Indentation only ever adds nesting that the declared levels missed and never removes nesting a document states outright, which leaves documents that already convert correctly untouched. Remapped paragraphs are pointed at a generated w:abstractNum carrying their original w:numFmt, so a bulleted sub-list is not silently converted into a numbered one, and depth is capped at the last level mammoth's default style map defines. Fixes #2323
1 parent 4cc9fa1 commit aee09c3

3 files changed

Lines changed: 507 additions & 0 deletions

File tree

‎packages/markitdown/src/markitdown/converter_utils/docx/pre_process.py‎

Lines changed: 318 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -247,6 +247,309 @@ def _pre_process_styles(content: bytes) -> bytes:
247247
return etree.tostring(root.getroottree(), encoding="utf-8", xml_declaration=True)
248248

249249

250+
W_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
251+
252+
# mammoth's default style map only defines list nesting up to five levels (see
253+
# "p:ordered-list(5)" in mammoth.options). A paragraph promoted past the last
254+
# mapped level matches no rule at all and drops out of the list entirely, so
255+
# nesting is capped at the deepest level mammoth can still represent.
256+
MAX_LIST_LEVEL = 4
257+
258+
# Word's default indentation step between consecutive list levels, in twips.
259+
# Used only when a level definition carries no explicit indentation.
260+
DEFAULT_LEVEL_INDENT = 720
261+
262+
263+
def _read_indent(element: Tag | None) -> int | None:
264+
"""
265+
Reads the left indentation (in twips) from the "w:ind" child of an element.
266+
267+
Args:
268+
element (Tag | None): The element whose "w:ind" child should be read.
269+
270+
Returns:
271+
int | None: The left indentation, or None if it is absent or malformed.
272+
"""
273+
if element is None:
274+
return None
275+
ind = element.find("ind", recursive=False)
276+
if ind is None:
277+
return None
278+
# "w:start" is the ISO/strict equivalent of the transitional "w:left".
279+
for attribute in ("w:left", "w:start"):
280+
value = ind.get(attribute)
281+
if value is not None:
282+
try:
283+
return int(value)
284+
except ValueError:
285+
pass
286+
return None
287+
288+
289+
def _read_numbering_definitions(numbering_soup: BeautifulSoup) -> dict:
290+
"""
291+
Resolves every "w:num" in numbering.xml to its per-level indentation and format.
292+
293+
Args:
294+
numbering_soup (BeautifulSoup): The parsed numbering.xml.
295+
296+
Returns:
297+
dict: Maps num_id -> level_index -> {"indent": int | None, "fmt": str | None}.
298+
"""
299+
abstract_nums = {}
300+
for abstract_num in numbering_soup.find_all("abstractNum"):
301+
levels = {}
302+
for lvl in abstract_num.find_all("lvl"):
303+
num_fmt = lvl.find("numFmt")
304+
levels[lvl.get("w:ilvl")] = {
305+
"indent": _read_indent(lvl.find("pPr")),
306+
"fmt": num_fmt.get("w:val") if num_fmt is not None else None,
307+
}
308+
abstract_nums[abstract_num.get("w:abstractNumId")] = levels
309+
310+
nums = {}
311+
for num in numbering_soup.find_all("num"):
312+
abstract_num_id = num.find("abstractNumId")
313+
if abstract_num_id is None:
314+
continue
315+
levels = dict(abstract_nums.get(abstract_num_id.get("w:val"), {}))
316+
# A "w:lvlOverride" replaces the inherited definition for a single level.
317+
for override in num.find_all("lvlOverride"):
318+
lvl = override.find("lvl")
319+
if lvl is None:
320+
continue
321+
num_fmt = lvl.find("numFmt")
322+
level_index = lvl.get("w:ilvl", override.get("w:ilvl"))
323+
levels[level_index] = {
324+
"indent": _read_indent(lvl.find("pPr")),
325+
"fmt": num_fmt.get("w:val") if num_fmt is not None else None,
326+
}
327+
nums[num.get("w:numId")] = levels
328+
return nums
329+
330+
331+
def _iter_list_paragraphs(document_soup: BeautifulSoup):
332+
"""
333+
Yields each paragraph of a document alongside its numbering reference.
334+
335+
Args:
336+
document_soup (BeautifulSoup): The parsed document.xml.
337+
338+
Yields:
339+
tuple: (paragraph, ilvl_tag, num_id_tag). The tags are None for any
340+
paragraph that does not carry direct numbering.
341+
"""
342+
for paragraph in document_soup.find_all("p"):
343+
p_pr = paragraph.find("pPr", recursive=False)
344+
num_pr = p_pr.find("numPr", recursive=False) if p_pr is not None else None
345+
if num_pr is None:
346+
yield paragraph, None, None
347+
continue
348+
ilvl_tag = num_pr.find("ilvl", recursive=False)
349+
num_id_tag = num_pr.find("numId", recursive=False)
350+
# A "w:numId" of 0 explicitly removes numbering from the paragraph.
351+
if num_id_tag is None or num_id_tag.get("w:val") == "0":
352+
yield paragraph, None, None
353+
continue
354+
yield paragraph, ilvl_tag, num_id_tag
355+
356+
357+
def _is_parent_of(candidate: dict, item: dict) -> bool:
358+
"""
359+
Determines whether an open list level is an ancestor of the current item.
360+
361+
Within a single "w:numId" the declared "w:ilvl" is authoritative, because
362+
Word uses it directly and levels of one list are directly comparable.
363+
Across different "w:numId" values the levels are unrelated, so only the
364+
rendered indentation can establish which list is nested inside the other.
365+
366+
Args:
367+
candidate (dict): An open level from the stack.
368+
item (dict): The list paragraph being placed.
369+
370+
Returns:
371+
bool: True if candidate is strictly shallower than item.
372+
"""
373+
if candidate["num_id"] == item["num_id"]:
374+
return candidate["ilvl"] < item["ilvl"]
375+
return candidate["indent"] < item["indent"]
376+
377+
378+
def _resolve_list_depths(document_soup: BeautifulSoup, numbering: dict) -> list:
379+
"""
380+
Computes the true nesting depth of every numbered paragraph in a document.
381+
382+
Word represents a nested list in either of two ways: as a deeper "w:ilvl"
383+
within the parent's "w:numId", or as an entirely new "w:numId" at
384+
"w:ilvl" 0 that is simply indented further. Both render identically, but
385+
mammoth derives nesting from "w:ilvl" alone and so flattens the second
386+
form. Walking the document while tracking the open levels recovers the
387+
nesting that indentation implies.
388+
389+
Args:
390+
document_soup (BeautifulSoup): The parsed document.xml.
391+
numbering (dict): Numbering definitions from _read_numbering_definitions.
392+
393+
Returns:
394+
list: One (paragraph, ilvl_tag, num_id_tag, depth) tuple per paragraph
395+
whose depth differs from its declared level.
396+
"""
397+
remappings = []
398+
stack: list[dict] = []
399+
400+
for paragraph, ilvl_tag, num_id_tag in _iter_list_paragraphs(document_soup):
401+
if num_id_tag is None:
402+
# Body text interrupts the surrounding list, exactly as it does for
403+
# mammoth, so no level stays open across it.
404+
stack.clear()
405+
continue
406+
407+
num_id = num_id_tag.get("w:val")
408+
# A missing "w:ilvl" means the first level.
409+
raw_ilvl = ilvl_tag.get("w:val") if ilvl_tag is not None else "0"
410+
try:
411+
ilvl = int(raw_ilvl)
412+
except (TypeError, ValueError):
413+
ilvl = 0
414+
415+
level = numbering.get(num_id, {}).get(str(ilvl), {})
416+
indent = level.get("indent")
417+
if indent is None:
418+
indent = ilvl * DEFAULT_LEVEL_INDENT
419+
# Indentation applied directly to the paragraph overrides the level's.
420+
paragraph_indent = _read_indent(paragraph.find("pPr", recursive=False))
421+
if paragraph_indent is not None:
422+
indent = paragraph_indent
423+
424+
item = {"num_id": num_id, "ilvl": ilvl, "indent": indent}
425+
while stack and not _is_parent_of(stack[-1], item):
426+
stack.pop()
427+
428+
implied_depth = stack[-1]["depth"] + 1 if stack else 0
429+
# Indentation is only ever used to reveal nesting the declared levels
430+
# missed, never to remove nesting a document states outright. This
431+
# keeps documents that mammoth already handles correctly untouched.
432+
depth = min(max(implied_depth, ilvl), MAX_LIST_LEVEL)
433+
434+
item["depth"] = depth
435+
stack.append(item)
436+
437+
if depth != ilvl:
438+
remappings.append((paragraph, ilvl_tag, num_id_tag, depth))
439+
440+
return remappings
441+
442+
443+
def _apply_list_depths(
444+
document_soup: BeautifulSoup, numbering_soup: BeautifulSoup, remappings: list
445+
) -> None:
446+
"""
447+
Rewrites paragraphs whose nesting depth was mis-declared, in place.
448+
449+
Simply raising "w:ilvl" would make the paragraph resolve against whatever
450+
unrelated level its numbering happens to define at that index, which can
451+
silently flip an ordered list to a bulleted one. Instead each remapped
452+
(num_id, ilvl, depth) combination gets a minimal generated definition that
453+
places the original format at the required depth.
454+
455+
Args:
456+
document_soup (BeautifulSoup): The parsed document.xml.
457+
numbering_soup (BeautifulSoup): The parsed numbering.xml.
458+
remappings (list): Output of _resolve_list_depths.
459+
"""
460+
numbering_root = numbering_soup.find("numbering")
461+
if numbering_root is None:
462+
return
463+
464+
existing_num_ids = {
465+
int(num.get("w:numId"))
466+
for num in numbering_soup.find_all("num")
467+
if (num.get("w:numId") or "").isdigit()
468+
}
469+
existing_abstract_ids = {
470+
int(abstract_num.get("w:abstractNumId"))
471+
for abstract_num in numbering_soup.find_all("abstractNum")
472+
if (abstract_num.get("w:abstractNumId") or "").isdigit()
473+
}
474+
next_num_id = max(existing_num_ids, default=0) + 1
475+
next_abstract_id = max(existing_abstract_ids, default=0) + 1
476+
477+
numbering = _read_numbering_definitions(numbering_soup)
478+
generated: dict = {}
479+
480+
for paragraph, ilvl_tag, num_id_tag, depth in remappings:
481+
num_id = num_id_tag.get("w:val")
482+
ilvl = ilvl_tag.get("w:val") if ilvl_tag is not None else "0"
483+
key = (num_id, ilvl, depth)
484+
485+
if key not in generated:
486+
num_fmt = numbering.get(num_id, {}).get(ilvl, {}).get("fmt")
487+
488+
abstract_num = numbering_soup.new_tag(
489+
"abstractNum", namespace=W_NS, nsprefix="w"
490+
)
491+
abstract_num["w:abstractNumId"] = str(next_abstract_id)
492+
lvl = numbering_soup.new_tag("lvl", namespace=W_NS, nsprefix="w")
493+
lvl["w:ilvl"] = str(depth)
494+
if num_fmt is not None:
495+
num_fmt_tag = numbering_soup.new_tag(
496+
"numFmt", namespace=W_NS, nsprefix="w"
497+
)
498+
num_fmt_tag["w:val"] = num_fmt
499+
lvl.append(num_fmt_tag)
500+
abstract_num.append(lvl)
501+
502+
num = numbering_soup.new_tag("num", namespace=W_NS, nsprefix="w")
503+
num["w:numId"] = str(next_num_id)
504+
abstract_num_id = numbering_soup.new_tag(
505+
"abstractNumId", namespace=W_NS, nsprefix="w"
506+
)
507+
abstract_num_id["w:val"] = str(next_abstract_id)
508+
num.append(abstract_num_id)
509+
510+
# "w:abstractNum" elements must precede "w:num" elements.
511+
first_num = numbering_root.find("num", recursive=False)
512+
if first_num is not None:
513+
first_num.insert_before(abstract_num)
514+
else:
515+
numbering_root.append(abstract_num)
516+
numbering_root.append(num)
517+
518+
generated[key] = str(next_num_id)
519+
next_abstract_id += 1
520+
next_num_id += 1
521+
522+
num_id_tag["w:val"] = generated[key]
523+
if ilvl_tag is None:
524+
num_pr = num_id_tag.parent
525+
ilvl_tag = document_soup.new_tag("ilvl", namespace=W_NS, nsprefix="w")
526+
num_pr.insert(0, ilvl_tag)
527+
ilvl_tag["w:val"] = str(depth)
528+
529+
530+
def _pre_process_lists(document_content: bytes, numbering_content: bytes) -> tuple:
531+
"""
532+
Restores list nesting that is expressed through indentation rather than levels.
533+
534+
Args:
535+
document_content (bytes): The XML content of word/document.xml.
536+
numbering_content (bytes): The XML content of word/numbering.xml.
537+
538+
Returns:
539+
tuple: The processed (document_content, numbering_content) as bytes.
540+
"""
541+
document_soup = BeautifulSoup(document_content.decode(), features="xml")
542+
numbering_soup = BeautifulSoup(numbering_content.decode(), features="xml")
543+
544+
numbering = _read_numbering_definitions(numbering_soup)
545+
remappings = _resolve_list_depths(document_soup, numbering)
546+
if not remappings:
547+
return document_content, numbering_content
548+
549+
_apply_list_depths(document_soup, numbering_soup, remappings)
550+
return str(document_soup).encode(), str(numbering_soup).encode()
551+
552+
250553
def pre_process_docx(input_docx: BinaryIO) -> BinaryIO:
251554
"""
252555
Pre-processes a DOCX file with provided steps.
@@ -274,6 +577,21 @@ def pre_process_docx(input_docx: BinaryIO) -> BinaryIO:
274577
}
275578
with zipfile.ZipFile(input_docx, mode="r") as zip_input:
276579
files = {name: zip_input.read(name) for name in zip_input.namelist()}
580+
581+
# List nesting spans document.xml and numbering.xml, so both are
582+
# rewritten together rather than one file at a time below.
583+
if "word/document.xml" in files and "word/numbering.xml" in files:
584+
try:
585+
(
586+
files["word/document.xml"],
587+
files["word/numbering.xml"],
588+
) = _pre_process_lists(
589+
files["word/document.xml"], files["word/numbering.xml"]
590+
)
591+
except Exception:
592+
# If there is an error in processing the content, keep the original content
593+
pass
594+
277595
with zipfile.ZipFile(output_docx, mode="w") as zip_output:
278596
zip_output.comment = zip_input.comment
279597
for name, content in files.items():

0 commit comments

Comments
 (0)