"""Build the defence deck as .pptx from the same content as the Beamer source. Beamer is the primary deliverable; this exists so the deck can be handed to someone who needs to edit it in PowerPoint. Numbers come from slides/numbers.json (generated), figures from slides/figures/ (generated), so the two decks cannot disagree with each other or with the thesis. """ from __future__ import annotations import json from pathlib import Path from pptx import Presentation from pptx.dml.color import RGBColor from pptx.enum.text import PP_ALIGN from pptx.util import Emu, Inches, Pt ROOT = Path(__file__).resolve().parents[1] SLIDES = ROOT / 'slides' FIGS = SLIDES / 'figures' INK = RGBColor(0x0B, 0x0B, 0x0B) INK2 = RGBColor(0x52, 0x51, 0x4E) MUTED = RGBColor(0x89, 0x87, 0x81) MINORITY = RGBColor(0x2A, 0x78, 0xD6) SURFACE = RGBColor(0xFC, 0xFC, 0xFB) N = json.loads((SLIDES / 'numbers.json').read_text()) W, H = Inches(13.333), Inches(7.5) def background(slide): fill = slide.background.fill fill.solid() fill.fore_color.rgb = SURFACE def textbox(slide, left, top, width, height, text, size=18, color=INK, bold=False, align=PP_ALIGN.LEFT, spacing=1.15): box = slide.shapes.add_textbox(left, top, width, height) frame = box.text_frame frame.word_wrap = True for i, line in enumerate(text.split('\n')): para = frame.paragraphs[0] if i == 0 else frame.add_paragraph() para.alignment = align para.line_spacing = spacing run = para.add_run() run.text = line run.font.size = Pt(size) run.font.color.rgb = color run.font.bold = bold run.font.name = 'Calibri' return box def blank(prs): slide = prs.slides.add_slide(prs.slide_layouts[6]) background(slide) return slide def titled(prs, title): slide = blank(prs) textbox(slide, Inches(0.7), Inches(0.4), Inches(12.0), Inches(0.9), title, size=30, color=INK, bold=True) return slide def picture(slide, name, top=Inches(1.45), height=Inches(4.6)): path = FIGS / f'{name}.png' pic = slide.shapes.add_picture(str(path), Inches(0), top, height=height) pic.left = Emu(int((W - pic.width) / 2)) return pic def takeaway(slide, text, size=16): textbox(slide, Inches(0.7), Inches(6.25), Inches(12.0), Inches(1.0), text, size=size, color=INK2) def bullets(slide, items, top=Inches(1.6), size=20, gap=0.78): for i, item in enumerate(items): textbox(slide, Inches(1.0), top + Inches(gap * i), Inches(11.4), Inches(0.9), f'• {item}', size=size, color=INK) def build(): prs = Presentation() prs.slide_width, prs.slide_height = W, H s = blank(prs) textbox(s, Inches(1.0), Inches(2.5), Inches(11.3), Inches(2.0), 'Majority Accumulation and Minority False Matches\nin Locally Adapted Face Verification', size=34, color=INK, bold=True, align=PP_ALIGN.CENTER) textbox(s, Inches(1.0), Inches(4.6), Inches(11.3), Inches(0.6), 'MSc thesis defence', size=18, color=MUTED, align=PP_ALIGN.CENTER) s = titled(prs, 'Institutions adapt recognisers. They do not train them.') bullets(s, ['Take a model pretrained on a very large public corpus.', 'Fine-tune it on the identities you have already enrolled.', 'Cheap, needs no new data, and reliably improves measured accuracy.']) textbox(s, Inches(0.7), Inches(4.5), Inches(12.0), Inches(1.4), f"But you do not choose who is in that set. In our corpus, " f"{N['pctMale']}% of {N['Nidentities']} enrolled identities are male — " f"not by decision, but because the cohort comes from engineering programmes.", size=19, color=INK) takeaway(s, 'The practitioner faces a fork that does not look like a fork: ' 'use the data as it arrives, or curate it.') s = titled(prs, 'The question') textbox(s, Inches(1.0), Inches(2.1), Inches(11.3), Inches(1.4), 'What does the natural choice cost,\nand who pays for it?', size=32, color=MINORITY, bold=True, align=PP_ALIGN.CENTER) bullets(s, ['RQ1 How does adaptation-set composition affect subgroup performance?', 'RQ2 Do protocol choices change what we observe?', 'RQ3 What in the representation explains it?'], top=Inches(4.3), size=18, gap=0.62) s = titled(prs, 'Seven models. The metric everybody reports.') picture(s, 's_hook_genuine') takeaway(s, 'Flat. Overall accuracy spans 0.025 across every condition.') s = titled(prs, 'The same seven models. The metric almost nobody reports.') picture(s, 's_hook_impostor') takeaway(s, f"The rate at which two different minority users are confused spans " f"{N['FPRFFspan']}×, from {N['FPRFFbest']} to {N['FPRFFworst']}.") s = titled(prs, 'Three things worth noticing') bullets(s, ['The two impostor rates move in opposite directions. ' 'A merely worse model would degrade both.', 'The harm is entirely on the impostor side — minority users are not locked out, ' 'they are let into each other’s accounts.', f"The minority true-accept rate is highest in the worst condition ({N['TPRFworst']})."], size=19, gap=1.05) takeaway(s, 'An auditor checking whether minority users are wrongly rejected ' 'would see nothing wrong at all.') s = titled(prs, 'Why prior work cannot answer this') bullets(s, ['The female share falls from 50% to 22%, and', 'the female count falls from 87 to 39 — at the same time.']) textbox(s, Inches(0.7), Inches(3.5), Inches(12.0), Inches(1.6), 'Those imply opposite remedies: rebalance what you have, or go and collect more ' 'minority data. Varying a ratio at fixed total size cannot separate them.', size=19, color=INK) takeaway(s, 'So we vary the two counts independently.') s = titled(prs, 'The factorial design') pic = s.shapes.add_picture(str(FIGS / 's_design.png'), Inches(0.6), Inches(1.5), height=Inches(4.9)) textbox(s, Inches(7.6), Inches(1.8), Inches(5.2), Inches(4.0), 'Four cells cross minority count with majority count.\n\n' 'Across a row: adding majority identities.\n\n' 'Down a column: adding minority identities.\n\n' 'Fixed throughout: optimiser budget, identity splits, comparison manifests, ' 'and the pretrained checkpoint.', size=17, color=INK) s = titled(prs, 'Protocol discipline') bullets(s, [f"{N['Nidentities']} audited identities ({N['Nmale']} male, {N['Nfemale']} female), " f"{N['Nimages']} images.", 'Four disjoint roles per fold: fit, select, operate (calibration only), test (reporting only).', 'No calibration or test identity ever receives a gradient update.', 'Thresholds fitted on operate, applied unchanged to test.'], size=18, gap=0.72) takeaway(s, 'The central quantities are differences of a few thousandths. ' 'Minor leakage would not support them.') s = titled(prs, 'Both levers are real') picture(s, 's_forest') takeaway(s, f"Per identity added, the minority lever is {N['leverRatio']}× stronger " f"than the majority one.") s = titled(prs, 'And they interact') s.shapes.add_picture(str(FIGS / 's_interaction.png'), Inches(0.6), Inches(1.5), height=Inches(4.9)) textbox(s, Inches(7.6), Inches(2.0), Inches(5.2), Inches(3.6), 'Majority accumulation is 2.2× more harmful when the minority is scarce.\n\n' 'Minority collection is 2.4× more beneficial when the majority is plentiful.\n\n' 'Each remedy is worth most exactly where the other is worth least.', size=17, color=INK) s = titled(prs, 'A result we had to withdraw') picture(s, 's_seeds') takeaway(s, f"At three seeds the accuracy gain read +0.0124 [+0.0051, +0.0187] — an interval " f"excluding zero. At eight seeds it is {N['seedTPR']} (t = {N['seedTPRt']}). " f"The minority harm survives at {N['seedFPR']} (t = {N['seedFPRt']}).", size=15) s = titled(prs, 'What that changed') textbox(s, Inches(0.7), Inches(1.7), Inches(12.0), Inches(1.6), 'The earlier interval was honest but conditional: it described variation over ' 'comparisons, not over retraining. Adding the omitted variance component decided ' 'the contrast.', size=20, color=INK) textbox(s, Inches(1.0), Inches(3.7), Inches(11.3), Inches(1.2), 'There is no reliable accuracy to gain.\nOnly minority security to lose.', size=28, color=MINORITY, bold=True, align=PP_ALIGN.CENTER) takeaway(s, 'Skewed adaptation is also less predictable: 4.5× less stable in minority ' 'false matches across seeds.') s = titled(prs, 'Why: the embedding space') picture(s, 's_geometry') takeaway(s, f"Crowding between identity centroids predicts the error (r = {N['rCrowding']}). " f"Compactness within clusters does not (r = {N['rCompactness']}) — and a false " f"match is an event between two different identities, so it could not.", size=15) s = titled(prs, 'Why: the threshold follows the majority') picture(s, 's_threshold', height=Inches(4.5)) takeaway(s, f"Female–female pairs are only {N['calibFFshare']}% of the calibration pool, so the " f"threshold tracks the male distribution and leaves the minority tail above it.", size=15) s = titled(prs, 'Can threshold policy fix it?') picture(s, 's_policy') takeaway(s, 'Correction moves every skewed model left and down. Balanced adaptation under a ' 'plain global threshold beats all of them on both axes at once.', size=15) s = titled(prs, 'Two things the policy analysis reveals') bullets(s, [f"Unconstrained per-subgroup thresholds create a second harm: {N['mixedHarm']} " f"mixed-gender false matches. A tighten-only constraint gets identical gap closure " f"with a cross-group effect of exactly +0.0000.", f"The required offset is not a model constant: {N['offsetBal']} under balanced " f"adaptation (t = {N['offsetBalT']}, indistinguishable from zero) rising to " f"{N['offsetNat']} under the natural composition."], size=18, gap=1.5) takeaway(s, 'An operator who measures an offset once and reuses it after retraining holds one ' 'of the wrong magnitude, possibly the wrong sign.') s = titled(prs, 'Who pays for equalisation') rows = [('Aggregate true-accept cost', N['policyAggCost']), ('Cost to the protected group', N['policyFemCost']), ('Understatement factor', f"{N['policyRatio']}×")] for i, (label, value) in enumerate(rows): top = Inches(1.9 + 0.75 * i) bold = i == 1 textbox(s, Inches(2.2), top, Inches(6.4), Inches(0.6), label, size=21, color=INK if not bold else MINORITY, bold=bold) textbox(s, Inches(8.6), top, Inches(2.4), Inches(0.6), value, size=21, color=INK if not bold else MINORITY, bold=bold, align=PP_ALIGN.RIGHT) textbox(s, Inches(0.7), Inches(4.6), Inches(12.0), Inches(1.2), 'The protected group carries only 22% of the population weight, so the aggregate ' 'hides most of what the intervention charges them.', size=19, color=INK) takeaway(s, 'That may be the right trade — a false match is unauthorised access, a false ' 'rejection an inconvenience. But it is a trade, and it should be stated as one.') s = titled(prs, 'The comparison that should govern practice') hdr = [('', 'Minority TPR', 'Gap'), ('Balanced adaptation, no intervention', N['balancedTPRF'], N['balancedGap']), ('Skewed adaptation, best intervention', N['skewedTPRF'], N['skewedGap'])] for i, (a, b, c) in enumerate(hdr): top = Inches(1.8 + 0.7 * i) bold = i == 1 col = MINORITY if bold else (MUTED if i == 0 else INK) textbox(s, Inches(1.4), top, Inches(7.0), Inches(0.6), a, size=19, color=col, bold=bold) textbox(s, Inches(8.4), top, Inches(2.0), Inches(0.6), b, size=19, color=col, bold=bold, align=PP_ALIGN.RIGHT) textbox(s, Inches(10.5), top, Inches(1.8), Inches(0.6), c, size=19, color=col, bold=bold, align=PP_ALIGN.RIGHT) textbox(s, Inches(0.7), Inches(4.6), Inches(12.0), Inches(1.2), f"Minority users are {N['pointsBetter']} points better off under balanced adaptation " f"with no fairness intervention.", size=23, color=MINORITY, bold=True, align=PP_ALIGN.CENTER) takeaway(s, 'Threshold policy is a mitigation for a model you are stuck with. It is not a ' 'substitute for the composition decision.') s = titled(prs, 'Recommendations') bullets(s, ['Do not add majority identities just because you have them. ' 'The accuracy benefit is not reliable; the minority cost is.', 'Treat curation as dominating threshold correction.', 'If you use per-subgroup thresholds, constrain them to tighten-only.', 'Report within-group impostor rates — nothing else reveals this harm.'], size=19, gap=1.0) s = titled(prs, 'Limitations') bullets(s, ['One checkpoint, one architecture, one recipe, one institution.', f"{N['Nfemale']} female identities bound the precision of every minority estimate.", 'Session and modality labels come from filenames; manual validation outstanding.', 'Binary recorded gender labels — nothing establishes a causal effect of gender.'], size=19, gap=0.95) s = titled(prs, 'Conclusion') textbox(s, Inches(0.9), Inches(1.7), Inches(11.5), Inches(1.2), 'Using the data as it arrives costs nothing measurable\non the metrics these systems ' 'are judged by.', size=26, color=INK, bold=True, align=PP_ALIGN.CENTER) textbox(s, Inches(0.9), Inches(3.5), Inches(11.5), Inches(1.6), 'Underneath them, minority users are confused with one another 2.5 to 2.7× more ' 'often — through a mechanism we can measure in the embedding space, predict from ' 'calibration data, and trace to a representational change meeting a calibration ' 'convention.', size=19, color=INK, align=PP_ALIGN.CENTER) textbox(s, Inches(0.9), Inches(5.5), Inches(11.5), Inches(0.8), 'The unfairness is purchased for nothing.', size=26, color=MINORITY, bold=True, align=PP_ALIGN.CENTER) s = blank(prs) textbox(s, Inches(0.9), Inches(3.2), Inches(11.5), Inches(0.9), 'Backup slides', size=32, color=MUTED, bold=True, align=PP_ALIGN.CENTER) s = titled(prs, 'Backup: adaptation specialises to capture modality') picture(s, 's_protocol') takeaway(s, f"Roughly two thirds ({N['leakLo']}–{N['leakHi']}% by fold) of genuine test " f"comparisons come from within one capture session. Adaptation helps within a " f"capture mode and hurts across one.", size=15) s = titled(prs, 'Backup: the offset across conditions') picture(s, 's_offset') s = titled(prs, f"Backup: is r = {N['rCrowding']} pseudo-replicated?") bullets(s, ['Partly, and we say so.', f"{N['Nruns']} runs, but only four conditions.", f"Between conditions: r = {N['rBetween']} (4 points).", 'Pooled within condition: r = +0.268, not significant.', 'Leave-one-condition-out MAE: 0.0066 for crowding against 0.0105 for a ' 'two-moment score summary.'], size=18, gap=0.8) takeaway(s, 'Claimed as a composition-level diagnostic — not as a run-level predictor ' 'at fixed composition.') out = SLIDES / 'defense.pptx' prs.save(str(out)) print(f'wrote {out} ({len(prs.slides.__iter__.__self__._sldIdLst)} slides)') if __name__ == '__main__': build()