File size: 6,434 Bytes
5ecf925
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
#!/usr/bin/env python3
"""
Regression test for the "lone w:ins" bug: a CR paragraph containing a
standalone <w:ins> with no adjacent <w:del> (a pure insertion, nothing marked
deleted) used to be silently dropped by cr_parser.py, so a CR with real
content parsed to "0 changes" and the TS was never updated.

Fixtures are real CRs that exhibit this exact shape:
  SETTEC(26)000050r1.docx β€” 1 lone <w:ins> (", 3.8")
  SETTEC(26)000048r1.docx β€” 2 lone <w:ins> occurrences (same text, two clauses)

Run: python3 -m unittest scripts/tests/test_lone_insertion.py -v
"""

import sys
import tempfile
import unittest
from pathlib import Path

SCRIPT_DIR = Path(__file__).resolve().parent.parent
sys.path.insert(0, str(SCRIPT_DIR))

import docx
from docx.oxml.ns import qn

from cr_parser import parse_cr, _extract_inline_replacements
from ts_applicator import apply_manifest
from verify_applied import scan_revision_marks, verify_manifest_applied

FIXTURES = Path(__file__).parent / 'fixtures'
CR_050 = FIXTURES / 'SETTEC(26)000050r1.docx'
CR_048 = FIXTURES / 'SETTEC(26)000048r1.docx'


class LoneInsertionParseTests(unittest.TestCase):

    def test_cr050_yields_one_text_insert_change(self):
        changes = parse_cr(CR_050)
        self.assertEqual(len(changes), 1)
        change = changes[0]
        self.assertEqual(change['type'], 'text_insert')
        self.assertEqual(change['text'], ', 3.8')
        self.assertTrue(change['before'])
        self.assertTrue(change['after'])
        # The anchors must not themselves contain the inserted text β€” otherwise
        # the splice point would be ambiguous.
        self.assertNotIn(', 3.8', change['before'] + change['after'])

    def test_cr048_yields_two_text_insert_changes(self):
        changes = parse_cr(CR_048)
        self.assertEqual(len(changes), 2)
        self.assertTrue(all(c['type'] == 'text_insert' for c in changes))

    def test_scan_revision_marks_reports_one_handled_no_unhandled(self):
        scan = scan_revision_marks(CR_050)
        self.assertEqual(scan['handled'], 1)
        self.assertEqual(scan['unhandled'], {})

    def test_old_extractor_saw_zero_changes_here(self):
        """Pins the exact mechanism of the original bug: a lone w:ins is
        invisible to _extract_inline_replacements (it only pairs w:del with
        an adjacent w:ins), which is why the parser used to report 0 changes
        for this CR. If this assertion ever starts failing, _extract_lone_
        insertions may have become redundant β€” but it should not be removed
        without re-verifying this class of CR is still handled."""
        doc = docx.Document(str(CR_050))
        found_target_para = False
        for elem in doc.element.body:
            if elem.tag != qn('w:p'):
                continue
            if any(c.tag == qn('w:ins') for c in elem):
                found_target_para = True
                self.assertEqual(_extract_inline_replacements(elem), [])
        self.assertTrue(found_target_para, 'fixture no longer contains the expected w:ins paragraph')


class LoneInsertionApplyTests(unittest.TestCase):

    def _make_ts_doc(self, paragraph_text):
        doc = docx.Document()
        doc.add_paragraph(paragraph_text)
        tmp = tempfile.NamedTemporaryFile(suffix='.docx', delete=False)
        doc.save(tmp.name)
        return Path(tmp.name)

    def test_end_to_end_apply_produces_ins_with_no_del(self):
        changes = parse_cr(CR_050)
        change = changes[0]
        ts_text = change['before'] + change['after']

        ts_path = self._make_ts_doc(ts_text)
        out_path = ts_path.with_name('out.docx')
        try:
            n_ok, n_skip, log_lines, n_parsed, n_merged = apply_manifest(
                ts_path, changes, out_path)
            self.assertEqual(n_ok, 1)
            self.assertEqual(n_skip, 0)

            # python-docx's Paragraph.text only reads direct-child <w:r> runs β€”
            # it does not descend into <w:ins>/<w:del>, so the inserted text
            # must be located via the raw element tree, not p.text.
            out_doc = docx.Document(str(out_path))
            target = next(
                p for p in out_doc.paragraphs
                if change['text'] in ''.join(
                    t.text or '' for t in p._element.findall('.//' + qn('w:t')))
            )
            p_el = target._element
            ins_elems = p_el.findall(qn('w:ins'))
            del_elems = p_el.findall(qn('w:del'))
            self.assertEqual(len(ins_elems), 1)
            self.assertEqual(len(del_elems), 0)
            inserted_text = ''.join(t.text or '' for t in ins_elems[0].findall('.//' + qn('w:t')))
            self.assertEqual(inserted_text, change['text'])

            verify_errors = verify_manifest_applied(out_doc, changes)
            self.assertEqual(verify_errors, [])
        finally:
            ts_path.unlink(missing_ok=True)
            out_path.unlink(missing_ok=True)

    def test_nbsp_variant_between_cr_and_ts_still_applies(self):
        """CR 48r1's second change anchors on '... and Annex A do NOT ...'
        (plain space), but the real TS 102267 stores 'Annex\xa0A' with a non-
        breaking space β€” discovered live when re-running this exact CR
        against the real TS. The splice point must still be found via
        normalized matching, and the TS's real NBSP must be preserved
        (not silently overwritten with a plain space)."""
        changes = parse_cr(CR_048)
        change = next(c for c in changes if 'SCP82' in c['before'])
        ts_text = change['before'] + change['after'].replace('Annex A', 'Annex\xa0A')
        self.assertIn('\xa0A', ts_text)

        ts_path = self._make_ts_doc(ts_text)
        out_path = ts_path.with_name('out_nbsp.docx')
        try:
            n_ok, n_skip, log_lines, n_parsed, n_merged = apply_manifest(
                ts_path, [change], out_path)
            self.assertEqual(n_ok, 1)
            self.assertEqual(n_skip, 0)

            out_doc = docx.Document(str(out_path))
            full = ''.join(
                t.text or '' for t in out_doc.paragraphs[0]._element.findall('.//' + qn('w:t')))
            self.assertIn(', 3.8', full)
            self.assertIn('Annex\xa0A', full)  # original NBSP preserved, not overwritten
        finally:
            ts_path.unlink(missing_ok=True)
            out_path.unlink(missing_ok=True)


if __name__ == '__main__':
    unittest.main()