@@ -247,6 +247,309 @@ def _pre_process_styles(content: bytes) -> bytes:
247247 return etree .tostring (root .getroottree (), encoding = "utf-8" , xml_declaration = True )
248248
249249
250+ W_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
251+
252+ # mammoth's default style map only defines list nesting up to five levels (see
253+ # "p:ordered-list(5)" in mammoth.options). A paragraph promoted past the last
254+ # mapped level matches no rule at all and drops out of the list entirely, so
255+ # nesting is capped at the deepest level mammoth can still represent.
256+ MAX_LIST_LEVEL = 4
257+
258+ # Word's default indentation step between consecutive list levels, in twips.
259+ # Used only when a level definition carries no explicit indentation.
260+ DEFAULT_LEVEL_INDENT = 720
261+
262+
263+ def _read_indent (element : Tag | None ) -> int | None :
264+ """
265+ Reads the left indentation (in twips) from the "w:ind" child of an element.
266+
267+ Args:
268+ element (Tag | None): The element whose "w:ind" child should be read.
269+
270+ Returns:
271+ int | None: The left indentation, or None if it is absent or malformed.
272+ """
273+ if element is None :
274+ return None
275+ ind = element .find ("ind" , recursive = False )
276+ if ind is None :
277+ return None
278+ # "w:start" is the ISO/strict equivalent of the transitional "w:left".
279+ for attribute in ("w:left" , "w:start" ):
280+ value = ind .get (attribute )
281+ if value is not None :
282+ try :
283+ return int (value )
284+ except ValueError :
285+ pass
286+ return None
287+
288+
289+ def _read_numbering_definitions (numbering_soup : BeautifulSoup ) -> dict :
290+ """
291+ Resolves every "w:num" in numbering.xml to its per-level indentation and format.
292+
293+ Args:
294+ numbering_soup (BeautifulSoup): The parsed numbering.xml.
295+
296+ Returns:
297+ dict: Maps num_id -> level_index -> {"indent": int | None, "fmt": str | None}.
298+ """
299+ abstract_nums = {}
300+ for abstract_num in numbering_soup .find_all ("abstractNum" ):
301+ levels = {}
302+ for lvl in abstract_num .find_all ("lvl" ):
303+ num_fmt = lvl .find ("numFmt" )
304+ levels [lvl .get ("w:ilvl" )] = {
305+ "indent" : _read_indent (lvl .find ("pPr" )),
306+ "fmt" : num_fmt .get ("w:val" ) if num_fmt is not None else None ,
307+ }
308+ abstract_nums [abstract_num .get ("w:abstractNumId" )] = levels
309+
310+ nums = {}
311+ for num in numbering_soup .find_all ("num" ):
312+ abstract_num_id = num .find ("abstractNumId" )
313+ if abstract_num_id is None :
314+ continue
315+ levels = dict (abstract_nums .get (abstract_num_id .get ("w:val" ), {}))
316+ # A "w:lvlOverride" replaces the inherited definition for a single level.
317+ for override in num .find_all ("lvlOverride" ):
318+ lvl = override .find ("lvl" )
319+ if lvl is None :
320+ continue
321+ num_fmt = lvl .find ("numFmt" )
322+ level_index = lvl .get ("w:ilvl" , override .get ("w:ilvl" ))
323+ levels [level_index ] = {
324+ "indent" : _read_indent (lvl .find ("pPr" )),
325+ "fmt" : num_fmt .get ("w:val" ) if num_fmt is not None else None ,
326+ }
327+ nums [num .get ("w:numId" )] = levels
328+ return nums
329+
330+
331+ def _iter_list_paragraphs (document_soup : BeautifulSoup ):
332+ """
333+ Yields each paragraph of a document alongside its numbering reference.
334+
335+ Args:
336+ document_soup (BeautifulSoup): The parsed document.xml.
337+
338+ Yields:
339+ tuple: (paragraph, ilvl_tag, num_id_tag). The tags are None for any
340+ paragraph that does not carry direct numbering.
341+ """
342+ for paragraph in document_soup .find_all ("p" ):
343+ p_pr = paragraph .find ("pPr" , recursive = False )
344+ num_pr = p_pr .find ("numPr" , recursive = False ) if p_pr is not None else None
345+ if num_pr is None :
346+ yield paragraph , None , None
347+ continue
348+ ilvl_tag = num_pr .find ("ilvl" , recursive = False )
349+ num_id_tag = num_pr .find ("numId" , recursive = False )
350+ # A "w:numId" of 0 explicitly removes numbering from the paragraph.
351+ if num_id_tag is None or num_id_tag .get ("w:val" ) == "0" :
352+ yield paragraph , None , None
353+ continue
354+ yield paragraph , ilvl_tag , num_id_tag
355+
356+
357+ def _is_parent_of (candidate : dict , item : dict ) -> bool :
358+ """
359+ Determines whether an open list level is an ancestor of the current item.
360+
361+ Within a single "w:numId" the declared "w:ilvl" is authoritative, because
362+ Word uses it directly and levels of one list are directly comparable.
363+ Across different "w:numId" values the levels are unrelated, so only the
364+ rendered indentation can establish which list is nested inside the other.
365+
366+ Args:
367+ candidate (dict): An open level from the stack.
368+ item (dict): The list paragraph being placed.
369+
370+ Returns:
371+ bool: True if candidate is strictly shallower than item.
372+ """
373+ if candidate ["num_id" ] == item ["num_id" ]:
374+ return candidate ["ilvl" ] < item ["ilvl" ]
375+ return candidate ["indent" ] < item ["indent" ]
376+
377+
378+ def _resolve_list_depths (document_soup : BeautifulSoup , numbering : dict ) -> list :
379+ """
380+ Computes the true nesting depth of every numbered paragraph in a document.
381+
382+ Word represents a nested list in either of two ways: as a deeper "w:ilvl"
383+ within the parent's "w:numId", or as an entirely new "w:numId" at
384+ "w:ilvl" 0 that is simply indented further. Both render identically, but
385+ mammoth derives nesting from "w:ilvl" alone and so flattens the second
386+ form. Walking the document while tracking the open levels recovers the
387+ nesting that indentation implies.
388+
389+ Args:
390+ document_soup (BeautifulSoup): The parsed document.xml.
391+ numbering (dict): Numbering definitions from _read_numbering_definitions.
392+
393+ Returns:
394+ list: One (paragraph, ilvl_tag, num_id_tag, depth) tuple per paragraph
395+ whose depth differs from its declared level.
396+ """
397+ remappings = []
398+ stack : list [dict ] = []
399+
400+ for paragraph , ilvl_tag , num_id_tag in _iter_list_paragraphs (document_soup ):
401+ if num_id_tag is None :
402+ # Body text interrupts the surrounding list, exactly as it does for
403+ # mammoth, so no level stays open across it.
404+ stack .clear ()
405+ continue
406+
407+ num_id = num_id_tag .get ("w:val" )
408+ # A missing "w:ilvl" means the first level.
409+ raw_ilvl = ilvl_tag .get ("w:val" ) if ilvl_tag is not None else "0"
410+ try :
411+ ilvl = int (raw_ilvl )
412+ except (TypeError , ValueError ):
413+ ilvl = 0
414+
415+ level = numbering .get (num_id , {}).get (str (ilvl ), {})
416+ indent = level .get ("indent" )
417+ if indent is None :
418+ indent = ilvl * DEFAULT_LEVEL_INDENT
419+ # Indentation applied directly to the paragraph overrides the level's.
420+ paragraph_indent = _read_indent (paragraph .find ("pPr" , recursive = False ))
421+ if paragraph_indent is not None :
422+ indent = paragraph_indent
423+
424+ item = {"num_id" : num_id , "ilvl" : ilvl , "indent" : indent }
425+ while stack and not _is_parent_of (stack [- 1 ], item ):
426+ stack .pop ()
427+
428+ implied_depth = stack [- 1 ]["depth" ] + 1 if stack else 0
429+ # Indentation is only ever used to reveal nesting the declared levels
430+ # missed, never to remove nesting a document states outright. This
431+ # keeps documents that mammoth already handles correctly untouched.
432+ depth = min (max (implied_depth , ilvl ), MAX_LIST_LEVEL )
433+
434+ item ["depth" ] = depth
435+ stack .append (item )
436+
437+ if depth != ilvl :
438+ remappings .append ((paragraph , ilvl_tag , num_id_tag , depth ))
439+
440+ return remappings
441+
442+
443+ def _apply_list_depths (
444+ document_soup : BeautifulSoup , numbering_soup : BeautifulSoup , remappings : list
445+ ) -> None :
446+ """
447+ Rewrites paragraphs whose nesting depth was mis-declared, in place.
448+
449+ Simply raising "w:ilvl" would make the paragraph resolve against whatever
450+ unrelated level its numbering happens to define at that index, which can
451+ silently flip an ordered list to a bulleted one. Instead each remapped
452+ (num_id, ilvl, depth) combination gets a minimal generated definition that
453+ places the original format at the required depth.
454+
455+ Args:
456+ document_soup (BeautifulSoup): The parsed document.xml.
457+ numbering_soup (BeautifulSoup): The parsed numbering.xml.
458+ remappings (list): Output of _resolve_list_depths.
459+ """
460+ numbering_root = numbering_soup .find ("numbering" )
461+ if numbering_root is None :
462+ return
463+
464+ existing_num_ids = {
465+ int (num .get ("w:numId" ))
466+ for num in numbering_soup .find_all ("num" )
467+ if (num .get ("w:numId" ) or "" ).isdigit ()
468+ }
469+ existing_abstract_ids = {
470+ int (abstract_num .get ("w:abstractNumId" ))
471+ for abstract_num in numbering_soup .find_all ("abstractNum" )
472+ if (abstract_num .get ("w:abstractNumId" ) or "" ).isdigit ()
473+ }
474+ next_num_id = max (existing_num_ids , default = 0 ) + 1
475+ next_abstract_id = max (existing_abstract_ids , default = 0 ) + 1
476+
477+ numbering = _read_numbering_definitions (numbering_soup )
478+ generated : dict = {}
479+
480+ for paragraph , ilvl_tag , num_id_tag , depth in remappings :
481+ num_id = num_id_tag .get ("w:val" )
482+ ilvl = ilvl_tag .get ("w:val" ) if ilvl_tag is not None else "0"
483+ key = (num_id , ilvl , depth )
484+
485+ if key not in generated :
486+ num_fmt = numbering .get (num_id , {}).get (ilvl , {}).get ("fmt" )
487+
488+ abstract_num = numbering_soup .new_tag (
489+ "abstractNum" , namespace = W_NS , nsprefix = "w"
490+ )
491+ abstract_num ["w:abstractNumId" ] = str (next_abstract_id )
492+ lvl = numbering_soup .new_tag ("lvl" , namespace = W_NS , nsprefix = "w" )
493+ lvl ["w:ilvl" ] = str (depth )
494+ if num_fmt is not None :
495+ num_fmt_tag = numbering_soup .new_tag (
496+ "numFmt" , namespace = W_NS , nsprefix = "w"
497+ )
498+ num_fmt_tag ["w:val" ] = num_fmt
499+ lvl .append (num_fmt_tag )
500+ abstract_num .append (lvl )
501+
502+ num = numbering_soup .new_tag ("num" , namespace = W_NS , nsprefix = "w" )
503+ num ["w:numId" ] = str (next_num_id )
504+ abstract_num_id = numbering_soup .new_tag (
505+ "abstractNumId" , namespace = W_NS , nsprefix = "w"
506+ )
507+ abstract_num_id ["w:val" ] = str (next_abstract_id )
508+ num .append (abstract_num_id )
509+
510+ # "w:abstractNum" elements must precede "w:num" elements.
511+ first_num = numbering_root .find ("num" , recursive = False )
512+ if first_num is not None :
513+ first_num .insert_before (abstract_num )
514+ else :
515+ numbering_root .append (abstract_num )
516+ numbering_root .append (num )
517+
518+ generated [key ] = str (next_num_id )
519+ next_abstract_id += 1
520+ next_num_id += 1
521+
522+ num_id_tag ["w:val" ] = generated [key ]
523+ if ilvl_tag is None :
524+ num_pr = num_id_tag .parent
525+ ilvl_tag = document_soup .new_tag ("ilvl" , namespace = W_NS , nsprefix = "w" )
526+ num_pr .insert (0 , ilvl_tag )
527+ ilvl_tag ["w:val" ] = str (depth )
528+
529+
530+ def _pre_process_lists (document_content : bytes , numbering_content : bytes ) -> tuple :
531+ """
532+ Restores list nesting that is expressed through indentation rather than levels.
533+
534+ Args:
535+ document_content (bytes): The XML content of word/document.xml.
536+ numbering_content (bytes): The XML content of word/numbering.xml.
537+
538+ Returns:
539+ tuple: The processed (document_content, numbering_content) as bytes.
540+ """
541+ document_soup = BeautifulSoup (document_content .decode (), features = "xml" )
542+ numbering_soup = BeautifulSoup (numbering_content .decode (), features = "xml" )
543+
544+ numbering = _read_numbering_definitions (numbering_soup )
545+ remappings = _resolve_list_depths (document_soup , numbering )
546+ if not remappings :
547+ return document_content , numbering_content
548+
549+ _apply_list_depths (document_soup , numbering_soup , remappings )
550+ return str (document_soup ).encode (), str (numbering_soup ).encode ()
551+
552+
250553def pre_process_docx (input_docx : BinaryIO ) -> BinaryIO :
251554 """
252555 Pre-processes a DOCX file with provided steps.
@@ -274,6 +577,21 @@ def pre_process_docx(input_docx: BinaryIO) -> BinaryIO:
274577 }
275578 with zipfile .ZipFile (input_docx , mode = "r" ) as zip_input :
276579 files = {name : zip_input .read (name ) for name in zip_input .namelist ()}
580+
581+ # List nesting spans document.xml and numbering.xml, so both are
582+ # rewritten together rather than one file at a time below.
583+ if "word/document.xml" in files and "word/numbering.xml" in files :
584+ try :
585+ (
586+ files ["word/document.xml" ],
587+ files ["word/numbering.xml" ],
588+ ) = _pre_process_lists (
589+ files ["word/document.xml" ], files ["word/numbering.xml" ]
590+ )
591+ except Exception :
592+ # If there is an error in processing the content, keep the original content
593+ pass
594+
277595 with zipfile .ZipFile (output_docx , mode = "w" ) as zip_output :
278596 zip_output .comment = zip_input .comment
279597 for name , content in files .items ():
0 commit comments