55import html
66import json
77import re
8+ import string
89import unicodedata
910from copy import deepcopy
1011
@@ -398,17 +399,68 @@ def _evidence_word_character(character):
398399 return character .isalnum () or category [0 ] == 'M' or category == 'Pc'
399400
400401
401- def _evidence_markup (text ):
402+ def _evidence_literals (text ):
403+ """Map Markdown escapes and code content to literal original spans, before markup removal."""
404+ literals = {}
405+ index = 0
406+ while index + 1 < len (text ):
407+ if text [index ] == '\\ ' and text [index + 1 ] in string .punctuation :
408+ unit = (text [index + 1 ], index , index + 2 )
409+ literals [index ] = literals [index + 1 ] = unit
410+ index += 2
411+ else :
412+ index += 1
413+
414+ def protect (start , end ):
415+ for offset in range (start , end ):
416+ literals [offset ] = (text [offset ], offset , offset + 1 )
417+
418+ fence = None
419+ for line in _EVIDENCE_LINE .finditer (text ):
420+ marker = re .match (r' {0,3}(`{3,}|~{3,})(.*)$' , line .group (0 ))
421+ if fence is not None :
422+ protect (line .start (), line .end ())
423+ if marker and marker [1 ][0 ] == fence [0 ] and len (marker [1 ]) >= len (fence ) and not marker [2 ].strip ():
424+ fence = None
425+ elif marker and (marker [1 ][0 ] != '`' or '`' not in marker [2 ]):
426+ fence = marker [1 ]
427+ protect (line .start (), line .end ())
428+ elif line .group (0 ).startswith ((' ' , '\t ' )):
429+ protect (line .start (), line .end ())
430+
431+ runs = [match for match in re .finditer (r'`+' , text ) if match .start () not in literals ]
432+ following , closers = {}, {}
433+ for index in range (len (runs ) - 1 , - 1 , - 1 ):
434+ length = len (runs [index ].group (0 ))
435+ closers [index ] = following .get (length )
436+ following [length ] = index
437+ index = 0
438+ while index < len (runs ):
439+ closer = closers [index ]
440+ if closer is not None :
441+ protect (runs [index ].end (), runs [closer ].start ())
442+ index = closer + 1
443+ else :
444+ index += 1
445+ for match in re .finditer (r'<(code|pre)\b[^>]*>(.*?)(?:</\1\s*>|\Z)' , text , re .DOTALL | re .IGNORECASE ):
446+ if match .start () not in literals :
447+ protect (match .start (2 ), match .end (2 ))
448+ return literals
449+
450+
451+ def _evidence_markup (text , literals ):
402452 """Presentation markup as (start, end, separates): tags, comments and table rule lines."""
403453 spans = [
404454 (match .start (), match .end (), (match .group (1 ) or '' ).lower () not in _EVIDENCE_INLINE_TAGS )
405455 for match in _EVIDENCE_MARKUP .finditer (text )
456+ if match .start () not in literals
406457 ]
407458 for line in _EVIDENCE_LINE .finditer (text ):
408459 content = line .group (0 ).strip ()
409460 if (
410461 content and '-' in content and not content .strip ('|:- \t ' )
411462 and ('|' in content or content .count ('-' ) >= 3 )
463+ and not any (offset in literals for offset in range (line .start (), line .end ()))
412464 ):
413465 spans .append ((line .start (), line .end (), True ))
414466 merged = []
@@ -430,51 +482,54 @@ def _evidence_form(text):
430482 """
431483 units = []
432484
433- def add (character , start , end ):
485+ def add (character , start , end , literal = False ):
434486 character = character .translate (_EVIDENCE_FOLDS )
435487 if character in _EVIDENCE_INVISIBLE :
436488 return
437489 # Marks, including vowel signs with no combining class, stay with their base character.
438- if units and units [- 1 ][0 ] == 'text' and unicodedata .category (character )[0 ] == 'M' :
490+ kind = 'literal' if literal else 'text'
491+ if units and units [- 1 ][0 ] != 'gap' and unicodedata .category (character )[0 ] == 'M' :
439492 units [- 1 ][1 ] += character
440493 units [- 1 ][3 ] = end
441494 else :
442- units .append (['text' , character , start , end ])
495+ units .append ([kind , character , start , end ])
443496
497+ literals = _evidence_literals (text )
444498 position = 0
445- for start , end , separates in [* _evidence_markup (text ), (len (text ), len (text ), False )]:
499+ for start , end , separates in [* _evidence_markup (text , literals ), (len (text ), len (text ), False )]:
446500 if start > position :
447501 index = position
448502 # Entities are decoded only after markup removal, so escaped text never becomes markup.
449- for entity in _EVIDENCE_ENTITY .finditer (text , position , start ):
450- for offset in range (index , entity .start ()):
451- add (text [offset ], offset , offset + 1 )
452- value = html .unescape (entity .group (0 ))
453- if value == entity .group (0 ):
454- for offset in range (entity .start (), entity .end ()):
455- add (text [offset ], offset , offset + 1 )
503+ while index < start :
504+ if index in literals :
505+ character , literal_start , literal_end = literals [index ]
506+ add (character , literal_start , literal_end , literal = True )
507+ index = literal_end
508+ continue
509+ entity = _EVIDENCE_ENTITY .match (text , index , start )
510+ if entity is not None and html .unescape (entity .group (0 )) != entity .group (0 ):
511+ for character in html .unescape (entity .group (0 )):
512+ add (character , entity .start (), entity .end (), literal = True )
513+ index = entity .end ()
456514 else :
457- for character in value :
458- add (character , entity .start (), entity .end ())
459- index = entity .end ()
460- for offset in range (index , start ):
461- add (text [offset ], offset , offset + 1 )
515+ add (text [index ], index , index + 1 )
516+ index += 1
462517 if separates :
463518 units .append (['gap' , ' ' , start , end ])
464519 position = max (position , end )
465520
466521 items = []
467522 for kind , characters , start , end in units :
468- if kind == 'text ' and not characters .isascii ():
523+ if kind != 'gap ' and not characters .isascii ():
469524 kept = any (
470525 unicodedata .decomposition (character ).startswith (_EVIDENCE_KEPT_FORMS )
471526 for character in characters
472527 )
473528 characters = unicodedata .normalize ('NFC' if kept else 'NFKC' , characters ).translate (_EVIDENCE_FOLDS )
474529 for character in characters :
475- if kind == 'gap' or character .isspace () or character == '|' :
530+ if kind == 'gap' or character .isspace () or ( character == '|' and kind != 'literal' ) :
476531 items .append (('gap' , ' ' , start , end ))
477- elif character in _EVIDENCE_EMPHASIS :
532+ elif character in _EVIDENCE_EMPHASIS and kind != 'literal' :
478533 items .append (('mark' , character , start , end ))
479534 elif character not in _EVIDENCE_INVISIBLE :
480535 items .append (('char' , character , start , end ))
0 commit comments