Skip to content

Commit ad25114

Browse files
authored
Merge pull request #1743 from microsoft/paullizer-document-analysis-validation
Fix Markdown escape differences in document evidence validation
2 parents f886f7c + 7b78f4c commit ad25114

6 files changed

Lines changed: 316 additions & 26 deletions

‎application/single_app/config.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -101,7 +101,7 @@
101101
EXECUTOR_TYPE = 'thread'
102102
EXECUTOR_MAX_WORKERS = 30
103103
SESSION_TYPE = 'filesystem'
104-
VERSION = "0.261.315"
104+
VERSION = "0.261.316"
105105
IS_DEVELOPMENT = is_development_env_enabled()
106106

107107
# Opt-out for deployments where App Service Easy Auth is active but the platform

‎application/single_app/functions_document_analysis_results.py‎

Lines changed: 75 additions & 20 deletions
Original file line numberDiff line numberDiff line change
@@ -5,6 +5,7 @@
55
import html
66
import json
77
import re
8+
import string
89
import unicodedata
910
from copy import deepcopy
1011

@@ -398,17 +399,68 @@ def _evidence_word_character(character):
398399
return character.isalnum() or category[0] == 'M' or category == 'Pc'
399400

400401

401-
def _evidence_markup(text):
402+
def _evidence_literals(text):
403+
"""Map Markdown escapes and code content to literal original spans, before markup removal."""
404+
literals = {}
405+
index = 0
406+
while index + 1 < len(text):
407+
if text[index] == '\\' and text[index + 1] in string.punctuation:
408+
unit = (text[index + 1], index, index + 2)
409+
literals[index] = literals[index + 1] = unit
410+
index += 2
411+
else:
412+
index += 1
413+
414+
def protect(start, end):
415+
for offset in range(start, end):
416+
literals[offset] = (text[offset], offset, offset + 1)
417+
418+
fence = None
419+
for line in _EVIDENCE_LINE.finditer(text):
420+
marker = re.match(r' {0,3}(`{3,}|~{3,})(.*)$', line.group(0))
421+
if fence is not None:
422+
protect(line.start(), line.end())
423+
if marker and marker[1][0] == fence[0] and len(marker[1]) >= len(fence) and not marker[2].strip():
424+
fence = None
425+
elif marker and (marker[1][0] != '`' or '`' not in marker[2]):
426+
fence = marker[1]
427+
protect(line.start(), line.end())
428+
elif line.group(0).startswith((' ', '\t')):
429+
protect(line.start(), line.end())
430+
431+
runs = [match for match in re.finditer(r'`+', text) if match.start() not in literals]
432+
following, closers = {}, {}
433+
for index in range(len(runs) - 1, -1, -1):
434+
length = len(runs[index].group(0))
435+
closers[index] = following.get(length)
436+
following[length] = index
437+
index = 0
438+
while index < len(runs):
439+
closer = closers[index]
440+
if closer is not None:
441+
protect(runs[index].end(), runs[closer].start())
442+
index = closer + 1
443+
else:
444+
index += 1
445+
for match in re.finditer(r'<(code|pre)\b[^>]*>(.*?)(?:</\1\s*>|\Z)', text, re.DOTALL | re.IGNORECASE):
446+
if match.start() not in literals:
447+
protect(match.start(2), match.end(2))
448+
return literals
449+
450+
451+
def _evidence_markup(text, literals):
402452
"""Presentation markup as (start, end, separates): tags, comments and table rule lines."""
403453
spans = [
404454
(match.start(), match.end(), (match.group(1) or '').lower() not in _EVIDENCE_INLINE_TAGS)
405455
for match in _EVIDENCE_MARKUP.finditer(text)
456+
if match.start() not in literals
406457
]
407458
for line in _EVIDENCE_LINE.finditer(text):
408459
content = line.group(0).strip()
409460
if (
410461
content and '-' in content and not content.strip('|:- \t')
411462
and ('|' in content or content.count('-') >= 3)
463+
and not any(offset in literals for offset in range(line.start(), line.end()))
412464
):
413465
spans.append((line.start(), line.end(), True))
414466
merged = []
@@ -430,51 +482,54 @@ def _evidence_form(text):
430482
"""
431483
units = []
432484

433-
def add(character, start, end):
485+
def add(character, start, end, literal=False):
434486
character = character.translate(_EVIDENCE_FOLDS)
435487
if character in _EVIDENCE_INVISIBLE:
436488
return
437489
# Marks, including vowel signs with no combining class, stay with their base character.
438-
if units and units[-1][0] == 'text' and unicodedata.category(character)[0] == 'M':
490+
kind = 'literal' if literal else 'text'
491+
if units and units[-1][0] != 'gap' and unicodedata.category(character)[0] == 'M':
439492
units[-1][1] += character
440493
units[-1][3] = end
441494
else:
442-
units.append(['text', character, start, end])
495+
units.append([kind, character, start, end])
443496

497+
literals = _evidence_literals(text)
444498
position = 0
445-
for start, end, separates in [*_evidence_markup(text), (len(text), len(text), False)]:
499+
for start, end, separates in [*_evidence_markup(text, literals), (len(text), len(text), False)]:
446500
if start > position:
447501
index = position
448502
# Entities are decoded only after markup removal, so escaped text never becomes markup.
449-
for entity in _EVIDENCE_ENTITY.finditer(text, position, start):
450-
for offset in range(index, entity.start()):
451-
add(text[offset], offset, offset + 1)
452-
value = html.unescape(entity.group(0))
453-
if value == entity.group(0):
454-
for offset in range(entity.start(), entity.end()):
455-
add(text[offset], offset, offset + 1)
503+
while index < start:
504+
if index in literals:
505+
character, literal_start, literal_end = literals[index]
506+
add(character, literal_start, literal_end, literal=True)
507+
index = literal_end
508+
continue
509+
entity = _EVIDENCE_ENTITY.match(text, index, start)
510+
if entity is not None and html.unescape(entity.group(0)) != entity.group(0):
511+
for character in html.unescape(entity.group(0)):
512+
add(character, entity.start(), entity.end(), literal=True)
513+
index = entity.end()
456514
else:
457-
for character in value:
458-
add(character, entity.start(), entity.end())
459-
index = entity.end()
460-
for offset in range(index, start):
461-
add(text[offset], offset, offset + 1)
515+
add(text[index], index, index + 1)
516+
index += 1
462517
if separates:
463518
units.append(['gap', ' ', start, end])
464519
position = max(position, end)
465520

466521
items = []
467522
for kind, characters, start, end in units:
468-
if kind == 'text' and not characters.isascii():
523+
if kind != 'gap' and not characters.isascii():
469524
kept = any(
470525
unicodedata.decomposition(character).startswith(_EVIDENCE_KEPT_FORMS)
471526
for character in characters
472527
)
473528
characters = unicodedata.normalize('NFC' if kept else 'NFKC', characters).translate(_EVIDENCE_FOLDS)
474529
for character in characters:
475-
if kind == 'gap' or character.isspace() or character == '|':
530+
if kind == 'gap' or character.isspace() or (character == '|' and kind != 'literal'):
476531
items.append(('gap', ' ', start, end))
477-
elif character in _EVIDENCE_EMPHASIS:
532+
elif character in _EVIDENCE_EMPHASIS and kind != 'literal':
478533
items.append(('mark', character, start, end))
479534
elif character not in _EVIDENCE_INVISIBLE:
480535
items.append(('char', character, start, end))
Lines changed: 92 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,92 @@
1+
# Document analysis Markdown escape fix
2+
3+
**Version: 0.261.316**
4+
5+
Fixed in version: **0.261.316**, recorded in
6+
`application/single_app/config.py`.
7+
8+
## Issue and root cause
9+
10+
A fresh PDF analysis read its entire source window but rejected two of six
11+
findings. The extracted Markdown contained `2\. Partnership` and
12+
`6\) TELEPHONE NUMBER`; the model quoted the rendered punctuation as
13+
`2. Partnership` and `6) TELEPHONE NUMBER`. The shared evidence locator
14+
treated the formatting backslashes as written content.
15+
16+
Each affected finding also had valid citations, but one unmatched citation makes
17+
the finding unresolved. The final analysis therefore retained four findings and
18+
was partial. Complete-only Compose correctly refused that input, before Word
19+
rendering started. Ordinary search-backed chat does not use this same narrative
20+
Analyze validation path.
21+
22+
## General formatting comparison
23+
24+
The shared normalized comparison now decodes Markdown backslash escapes of ASCII
25+
punctuation in both the quote and source. It retains the punctuation itself and
26+
maps it back to the original two-character source span. For example, `6\)`
27+
can match `6)`, while saved evidence still contains `6\)` and its original
28+
character offsets.
29+
30+
This is deterministic formatting normalization, not a document-specific
31+
exception, an additional model evaluation, or a retry strategy. Exact matching
32+
still runs first, preserving the [literal evidence fix](DOCUMENT_ANALYSIS_LITERAL_EVIDENCE_FIX.md).
33+
The result contract, tiers, and `evidence-matcher-v2` identifier are unchanged.
34+
35+
Escaped punctuation stays literal after decoding: escaped asterisks and backticks
36+
do not become emphasis, escaped angle brackets do not become HTML tags, escaped
37+
ampersands do not begin entities, and escaped pipes do not become table separators.
38+
Decoded character entities receive the same literal treatment.
39+
40+
## Safeguards and limitations
41+
42+
Backslash pairs are processed left to right. A doubled backslash represents a
43+
literal backslash; a backslash before a letter is not a punctuation escape.
44+
Backslashes inside recognized inline code, top-level backtick or tilde fences,
45+
indented code lines, and HTML `code` or `pre` content remain literal. Unclosed
46+
fences and HTML code regions are protected through the end of the text.
47+
This is a conservative evidence normalizer, not a complete Markdown renderer;
48+
it does not add support for every nested Markdown construct.
49+
50+
Punctuation, numbers, signs, and checkbox states must still agree. Missing,
51+
wrong-location, ambiguous, paraphrased, and unsupported citations remain
52+
unresolved. Complete-only Compose, source coverage, access checks, and findings
53+
validation are unchanged. No citation is silently dropped to make a report pass.
54+
55+
Matching proves that a quotation can be located, not that every claim in a
56+
finding is factually entailed by the quotation. No factual-accuracy evaluation
57+
was added. Old failed attempts are not repaired or migrated; use a fresh request
58+
after deployment. This change adds no settings, routes, dependencies, or
59+
source-reprocessing requirements.
60+
61+
## Files and regression coverage
62+
63+
- `application/single_app/functions_document_analysis_results.py`: literal-span
64+
handling before markup and entity normalization.
65+
- `functional_tests/test_document_analysis_evidence_matching.py`: punctuation
66+
escapes, backslash parity, literal code, source offsets, changed values,
67+
checkbox states, citation locations, and real Analyze finalization without
68+
correction calls.
69+
- `functional_tests/test_orchestration_document_derivation_reliability.py`:
70+
escaped source evidence through real Analyze, Compose, Word rendering,
71+
publication, and download, using the existing offline harness.
72+
- `application/single_app/config.py`: patch version update.
73+
74+
## Validation
75+
76+
The unchanged saved response was replayed offline against the original source,
77+
with its content fingerprint verified. Before the fix, four findings were
78+
accepted, two unresolved, and 23 evidence passages retained. After the fix,
79+
all six findings were accepted, none unresolved, and all 25 evidence passages
80+
retained. Source coverage remained complete. Normal and optimized Python
81+
produced the same result, with no model calls or Azure writes.
82+
83+
The earlier footer incident also remains valid: all 25 findings pass unchanged.
84+
The automated regressions exercise both positive matches and rejected content
85+
changes, and the Word publication cases require exactly the existing two
86+
analysis calls plus one composition call, with no response correction.
87+
88+
The targeted matcher, finalization, derivation, saved-result, workflow-publication,
89+
and attempt-fence suites passed **379 tests and 272 subtests**. Documentation
90+
surface coverage and site quality checks also passed. Validation used isolated
91+
test dependencies; no shared Python packages or application manifests were
92+
modified. No live deployment was performed.

‎docs/explanation/fixes/V2_COMPARISON_ANALYZE_EVIDENCE_MATCHING_FIX.md‎

Lines changed: 5 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -119,6 +119,11 @@ renders as nothing:
119119
marks (`*`, `_` and backticks) at the edge of a word, as in `**Note**:`.
120120
- Character entities are decoded only after tags are handled, so escaped text is never
121121
read as markup.
122+
- As of **0.261.316**, Markdown backslash escapes of ASCII punctuation compare
123+
with their rendered punctuation, while evidence retains the original source
124+
spans. Escaped symbols remain literal rather than becoming markup, and
125+
backslashes in recognized code contexts are preserved. See the
126+
[Markdown escape fix](DOCUMENT_ANALYSIS_MARKDOWN_ESCAPE_FIX.md).
122127
- Typographic quotes, and the prime and double prime, fold to `'` or `"`. An acute
123128
accent typed as an apostrophe, as in `can´t`, also folds to `'`. Hyphen, dash and
124129
minus variants fold to `-`.

0 commit comments

Comments
 (0)