Spaces:
Sleeping
Sleeping
File size: 6,434 Bytes
5ecf925 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 | #!/usr/bin/env python3
"""
Regression test for the "lone w:ins" bug: a CR paragraph containing a
standalone <w:ins> with no adjacent <w:del> (a pure insertion, nothing marked
deleted) used to be silently dropped by cr_parser.py, so a CR with real
content parsed to "0 changes" and the TS was never updated.
Fixtures are real CRs that exhibit this exact shape:
SETTEC(26)000050r1.docx β 1 lone <w:ins> (", 3.8")
SETTEC(26)000048r1.docx β 2 lone <w:ins> occurrences (same text, two clauses)
Run: python3 -m unittest scripts/tests/test_lone_insertion.py -v
"""
import sys
import tempfile
import unittest
from pathlib import Path
SCRIPT_DIR = Path(__file__).resolve().parent.parent
sys.path.insert(0, str(SCRIPT_DIR))
import docx
from docx.oxml.ns import qn
from cr_parser import parse_cr, _extract_inline_replacements
from ts_applicator import apply_manifest
from verify_applied import scan_revision_marks, verify_manifest_applied
FIXTURES = Path(__file__).parent / 'fixtures'
CR_050 = FIXTURES / 'SETTEC(26)000050r1.docx'
CR_048 = FIXTURES / 'SETTEC(26)000048r1.docx'
class LoneInsertionParseTests(unittest.TestCase):
def test_cr050_yields_one_text_insert_change(self):
changes = parse_cr(CR_050)
self.assertEqual(len(changes), 1)
change = changes[0]
self.assertEqual(change['type'], 'text_insert')
self.assertEqual(change['text'], ', 3.8')
self.assertTrue(change['before'])
self.assertTrue(change['after'])
# The anchors must not themselves contain the inserted text β otherwise
# the splice point would be ambiguous.
self.assertNotIn(', 3.8', change['before'] + change['after'])
def test_cr048_yields_two_text_insert_changes(self):
changes = parse_cr(CR_048)
self.assertEqual(len(changes), 2)
self.assertTrue(all(c['type'] == 'text_insert' for c in changes))
def test_scan_revision_marks_reports_one_handled_no_unhandled(self):
scan = scan_revision_marks(CR_050)
self.assertEqual(scan['handled'], 1)
self.assertEqual(scan['unhandled'], {})
def test_old_extractor_saw_zero_changes_here(self):
"""Pins the exact mechanism of the original bug: a lone w:ins is
invisible to _extract_inline_replacements (it only pairs w:del with
an adjacent w:ins), which is why the parser used to report 0 changes
for this CR. If this assertion ever starts failing, _extract_lone_
insertions may have become redundant β but it should not be removed
without re-verifying this class of CR is still handled."""
doc = docx.Document(str(CR_050))
found_target_para = False
for elem in doc.element.body:
if elem.tag != qn('w:p'):
continue
if any(c.tag == qn('w:ins') for c in elem):
found_target_para = True
self.assertEqual(_extract_inline_replacements(elem), [])
self.assertTrue(found_target_para, 'fixture no longer contains the expected w:ins paragraph')
class LoneInsertionApplyTests(unittest.TestCase):
def _make_ts_doc(self, paragraph_text):
doc = docx.Document()
doc.add_paragraph(paragraph_text)
tmp = tempfile.NamedTemporaryFile(suffix='.docx', delete=False)
doc.save(tmp.name)
return Path(tmp.name)
def test_end_to_end_apply_produces_ins_with_no_del(self):
changes = parse_cr(CR_050)
change = changes[0]
ts_text = change['before'] + change['after']
ts_path = self._make_ts_doc(ts_text)
out_path = ts_path.with_name('out.docx')
try:
n_ok, n_skip, log_lines, n_parsed, n_merged = apply_manifest(
ts_path, changes, out_path)
self.assertEqual(n_ok, 1)
self.assertEqual(n_skip, 0)
# python-docx's Paragraph.text only reads direct-child <w:r> runs β
# it does not descend into <w:ins>/<w:del>, so the inserted text
# must be located via the raw element tree, not p.text.
out_doc = docx.Document(str(out_path))
target = next(
p for p in out_doc.paragraphs
if change['text'] in ''.join(
t.text or '' for t in p._element.findall('.//' + qn('w:t')))
)
p_el = target._element
ins_elems = p_el.findall(qn('w:ins'))
del_elems = p_el.findall(qn('w:del'))
self.assertEqual(len(ins_elems), 1)
self.assertEqual(len(del_elems), 0)
inserted_text = ''.join(t.text or '' for t in ins_elems[0].findall('.//' + qn('w:t')))
self.assertEqual(inserted_text, change['text'])
verify_errors = verify_manifest_applied(out_doc, changes)
self.assertEqual(verify_errors, [])
finally:
ts_path.unlink(missing_ok=True)
out_path.unlink(missing_ok=True)
def test_nbsp_variant_between_cr_and_ts_still_applies(self):
"""CR 48r1's second change anchors on '... and Annex A do NOT ...'
(plain space), but the real TS 102267 stores 'Annex\xa0A' with a non-
breaking space β discovered live when re-running this exact CR
against the real TS. The splice point must still be found via
normalized matching, and the TS's real NBSP must be preserved
(not silently overwritten with a plain space)."""
changes = parse_cr(CR_048)
change = next(c for c in changes if 'SCP82' in c['before'])
ts_text = change['before'] + change['after'].replace('Annex A', 'Annex\xa0A')
self.assertIn('\xa0A', ts_text)
ts_path = self._make_ts_doc(ts_text)
out_path = ts_path.with_name('out_nbsp.docx')
try:
n_ok, n_skip, log_lines, n_parsed, n_merged = apply_manifest(
ts_path, [change], out_path)
self.assertEqual(n_ok, 1)
self.assertEqual(n_skip, 0)
out_doc = docx.Document(str(out_path))
full = ''.join(
t.text or '' for t in out_doc.paragraphs[0]._element.findall('.//' + qn('w:t')))
self.assertIn(', 3.8', full)
self.assertIn('Annex\xa0A', full) # original NBSP preserved, not overwritten
finally:
ts_path.unlink(missing_ok=True)
out_path.unlink(missing_ok=True)
if __name__ == '__main__':
unittest.main()
|