File size: 79,967 Bytes
11dde75
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2cf2ccd
 
 
 
 
 
03f160f
11dde75
2cf2ccd
 
 
 
 
 
 
 
 
 
67c9044
8dfd1d1
 
67c9044
 
 
 
 
 
 
 
 
11dde75
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
03f160f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
11dde75
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
03f160f
11dde75
 
 
 
03f160f
 
 
 
 
 
 
 
 
 
 
 
 
11dde75
 
 
03f160f
11dde75
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
03f160f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
11dde75
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
03f160f
 
11dde75
 
 
 
 
 
 
03f160f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
11dde75
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2cf2ccd
11dde75
 
03f160f
 
5ea82ce
2cf2ccd
 
5ea82ce
 
2cf2ccd
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5ea82ce
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
11dde75
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2cf2ccd
11dde75
de5014e
 
 
 
 
e5bfacd
 
 
 
 
 
 
 
 
 
03f160f
 
 
 
 
 
 
 
 
 
11dde75
 
 
 
 
 
 
00f9c01
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
11dde75
 
 
 
 
3f67f54
de5014e
3f67f54
11dde75
 
 
 
de5014e
 
11dde75
 
de5014e
11dde75
 
de5014e
11dde75
de5014e
11dde75
de5014e
 
 
 
11dde75
 
 
de5014e
11dde75
 
 
de5014e
00f9c01
11dde75
 
00f9c01
 
 
de5014e
 
00f9c01
 
 
 
 
 
 
 
 
 
de5014e
 
 
 
 
 
 
 
 
 
 
 
 
00f9c01
 
de5014e
00f9c01
 
 
11dde75
 
 
67c9044
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
de5014e
 
 
 
 
 
 
 
 
 
 
 
67c9044
 
 
 
 
 
de5014e
67c9044
 
 
 
 
 
 
 
 
 
 
 
 
 
 
de5014e
 
 
 
 
 
 
 
 
 
 
 
 
11dde75
 
 
de5014e
 
 
11dde75
 
de5014e
 
11dde75
 
 
 
 
 
 
 
de5014e
11dde75
de5014e
 
11dde75
 
 
 
 
de5014e
 
11dde75
 
 
 
 
 
de5014e
11dde75
de5014e
 
11dde75
 
 
 
 
 
de5014e
 
 
11dde75
 
 
 
 
de5014e
 
 
 
 
 
 
 
 
 
11dde75
de5014e
 
11dde75
 
 
 
 
2cf2ccd
 
11dde75
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2cf2ccd
 
11dde75
 
2cf2ccd
 
 
 
 
 
 
 
 
11dde75
 
5ea82ce
2cf2ccd
 
 
5ea82ce
2cf2ccd
 
5ea82ce
11dde75
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
03f160f
11dde75
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
03f160f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
11dde75
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
de5014e
 
 
 
11dde75
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
de5014e
 
 
 
11dde75
 
 
 
 
 
 
 
de5014e
11dde75
 
 
 
de5014e
11dde75
 
de5014e
 
 
11dde75
 
 
 
 
 
de5014e
11dde75
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
de5014e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
11dde75
 
 
de5014e
 
 
 
 
11dde75
de5014e
 
 
 
 
 
 
 
 
 
 
 
 
 
11dde75
de5014e
 
11dde75
 
 
 
 
 
 
 
 
 
 
 
 
 
 
de5014e
 
 
11dde75
 
 
 
 
 
 
 
 
 
 
 
de5014e
 
 
 
 
 
 
 
 
11dde75
de5014e
 
 
 
 
 
 
 
 
 
 
 
 
 
11dde75
 
 
 
 
 
 
de5014e
11dde75
 
 
 
de5014e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
11dde75
de5014e
 
 
 
 
11dde75
 
 
 
de5014e
 
 
 
11dde75
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
de5014e
 
 
11dde75
 
 
 
 
de5014e
 
 
 
 
 
11dde75
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
de5014e
 
11dde75
de5014e
 
11dde75
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
de5014e
11dde75
 
 
 
 
 
 
 
 
de5014e
11dde75
 
 
de5014e
 
 
 
 
 
11dde75
 
 
 
 
 
de5014e
 
11dde75
 
de5014e
 
 
 
11dde75
 
 
 
 
 
de5014e
 
 
11dde75
de5014e
 
 
 
 
 
 
 
11dde75
 
 
de5014e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
11dde75
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
de5014e
 
 
 
 
 
 
 
 
 
11dde75
de5014e
 
 
 
 
 
11dde75
 
 
de5014e
 
11dde75
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
03f160f
11dde75
 
 
03f160f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
11dde75
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
de5014e
 
 
 
 
 
 
 
 
11dde75
 
de5014e
 
 
 
 
 
11dde75
 
 
 
 
 
 
 
 
 
03f160f
11dde75
 
 
 
 
03f160f
 
11dde75
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
03f160f
 
 
 
 
 
11dde75
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
1039
1040
1041
1042
1043
1044
1045
1046
1047
1048
1049
1050
1051
1052
1053
1054
1055
1056
1057
1058
1059
1060
1061
1062
1063
1064
1065
1066
1067
1068
1069
1070
1071
1072
1073
1074
1075
1076
1077
1078
1079
1080
1081
1082
1083
1084
1085
1086
1087
1088
1089
1090
1091
1092
1093
1094
1095
1096
1097
1098
1099
1100
1101
1102
1103
1104
1105
1106
1107
1108
1109
1110
1111
1112
1113
1114
1115
1116
1117
1118
1119
1120
1121
1122
1123
1124
1125
1126
1127
1128
1129
1130
1131
1132
1133
1134
1135
1136
1137
1138
1139
1140
1141
1142
1143
1144
1145
1146
1147
1148
1149
1150
1151
1152
1153
1154
1155
1156
1157
1158
1159
1160
1161
1162
1163
1164
1165
1166
1167
1168
1169
1170
1171
1172
1173
1174
1175
1176
1177
1178
1179
1180
1181
1182
1183
1184
1185
1186
1187
1188
1189
1190
1191
1192
1193
1194
1195
1196
1197
1198
1199
1200
1201
1202
1203
1204
1205
1206
1207
1208
1209
1210
1211
1212
1213
1214
1215
1216
1217
1218
1219
1220
1221
1222
1223
1224
1225
1226
1227
1228
1229
1230
1231
1232
1233
1234
1235
1236
1237
1238
1239
1240
1241
1242
1243
1244
1245
1246
1247
1248
1249
1250
1251
1252
1253
1254
1255
1256
1257
1258
1259
1260
1261
1262
1263
1264
1265
1266
1267
1268
1269
1270
1271
1272
1273
1274
1275
1276
1277
1278
1279
1280
1281
1282
1283
1284
1285
1286
1287
1288
1289
1290
1291
1292
1293
1294
1295
1296
1297
1298
1299
1300
1301
1302
1303
1304
1305
1306
1307
1308
1309
1310
1311
1312
1313
1314
1315
1316
1317
1318
1319
1320
1321
1322
1323
1324
1325
1326
1327
1328
1329
1330
1331
1332
1333
1334
1335
1336
1337
1338
1339
1340
1341
1342
1343
1344
1345
1346
1347
1348
1349
1350
1351
1352
1353
1354
1355
1356
1357
1358
1359
1360
1361
1362
1363
1364
1365
1366
1367
1368
1369
1370
1371
1372
1373
1374
1375
1376
1377
1378
1379
1380
1381
1382
1383
1384
1385
1386
1387
1388
1389
1390
1391
1392
1393
1394
1395
1396
1397
1398
1399
1400
1401
1402
1403
1404
1405
1406
1407
1408
1409
1410
1411
1412
1413
1414
1415
1416
1417
1418
1419
1420
1421
1422
1423
1424
1425
1426
1427
1428
1429
1430
1431
1432
1433
1434
1435
1436
1437
1438
1439
1440
1441
1442
1443
1444
1445
1446
1447
1448
1449
1450
1451
1452
1453
1454
1455
1456
1457
1458
1459
1460
1461
1462
1463
1464
1465
1466
1467
1468
1469
1470
1471
1472
1473
1474
1475
1476
1477
1478
1479
1480
1481
1482
1483
1484
1485
1486
1487
1488
1489
1490
1491
1492
1493
1494
1495
1496
1497
1498
1499
1500
1501
1502
1503
1504
1505
1506
1507
1508
1509
1510
1511
1512
1513
1514
1515
1516
1517
1518
1519
1520
1521
1522
1523
1524
1525
1526
1527
1528
1529
1530
1531
1532
1533
1534
1535
1536
1537
1538
1539
1540
1541
1542
1543
1544
1545
1546
1547
1548
1549
1550
1551
1552
1553
1554
1555
1556
1557
1558
1559
1560
1561
1562
1563
1564
1565
1566
1567
1568
1569
1570
1571
1572
1573
1574
1575
1576
1577
1578
1579
1580
1581
1582
1583
1584
1585
1586
1587
1588
1589
1590
1591
1592
1593
1594
1595
1596
1597
1598
1599
1600
1601
1602
1603
1604
1605
1606
1607
1608
1609
1610
1611
1612
1613
1614
1615
1616
1617
1618
1619
1620
1621
1622
1623
1624
1625
1626
1627
1628
1629
1630
1631
1632
1633
1634
1635
1636
1637
1638
1639
1640
1641
1642
1643
1644
1645
1646
1647
1648
1649
1650
1651
1652
1653
1654
1655
1656
1657
1658
1659
1660
1661
1662
1663
1664
1665
1666
1667
1668
1669
1670
1671
1672
1673
1674
1675
1676
1677
1678
1679
1680
1681
1682
1683
1684
1685
1686
1687
1688
1689
1690
1691
1692
1693
1694
1695
1696
1697
1698
1699
1700
1701
1702
1703
1704
1705
1706
1707
1708
1709
1710
1711
1712
1713
1714
1715
1716
1717
1718
1719
1720
1721
1722
1723
1724
1725
1726
1727
1728
1729
1730
1731
1732
1733
1734
1735
1736
1737
1738
1739
1740
1741
1742
1743
1744
1745
1746
1747
1748
1749
1750
1751
1752
1753
1754
1755
1756
1757
1758
1759
1760
1761
1762
1763
1764
1765
1766
1767
1768
1769
1770
1771
1772
1773
1774
1775
1776
1777
1778
1779
1780
1781
1782
1783
1784
1785
1786
1787
1788
1789
1790
1791
1792
1793
1794
1795
1796
1797
1798
1799
1800
1801
1802
1803
1804
1805
1806
1807
1808
1809
1810
1811
1812
1813
1814
1815
1816
1817
1818
1819
1820
1821
1822
1823
1824
1825
1826
1827
1828
1829
1830
1831
1832
1833
1834
1835
1836
1837
1838
1839
1840
1841
1842
1843
1844
1845
1846
1847
1848
1849
1850
1851
r"""
extraction.py — single source of truth for the AIM Composites extraction pipeline.

This module consolidates the Gemini prompt/schema and all post-extraction
processing that used to live (and drift) inside ``batch_ingest.py`` and the
Streamlit app repo's ``page_files/categorized/page6.py`` /
``Backend/Pdf_DataExtraction.py``.

Pipeline shape::

    pdf_bytes
       -> extract_from_pdf()        # Gemini structured output -> Extraction
       -> verify_against_text()     # ground every value in the PDF text (Task 1)
       -> to_rows()                 # Extraction -> list[PropertyRow] (Task 2/3/5/6)
       -> (caller inserts rows, carrying `status`/`flag_reason`)

Design notes
------------
* The model is asked for a *list* of materials, each with structured,
  unit-aware property values plus a verbatim ``source_quote`` and ``page``.
* Numbers are parsed into ``value_num`` / ``value_min`` / ``value_max`` /
  ``qualifier`` and normalized to a canonical unit (``unit_canonical`` +
  ``value_si``) with :mod:`pint`. Plausibility is range-checked *after*
  conversion, so a GPa-vs-MPa mix no longer false-flags.
* Nothing is silently dropped: every row carries a ``status`` and, when not
  ``ok``, a ``flag_reason``.

Keep ``temperature=0`` and bump :data:`PROMPT_VERSION` on any prompt change.
"""

from __future__ import annotations

import base64
import dataclasses
import json
import logging
import os
import re
import time
import unicodedata
from typing import Any, Optional

import requests

try:  # PyMuPDF — used for text grounding (Task 1) and scanned detection (Task 10)
    import fitz  # type: ignore
except Exception:  # pragma: no cover - import guard
    fitz = None  # type: ignore

try:
    import pint  # unit normalization (Task 3)
except Exception:  # pragma: no cover - import guard
    pint = None  # type: ignore

log = logging.getLogger("extraction")


# ---------------------------------------------------------------------------
# Configuration
# ---------------------------------------------------------------------------

# Stable release only — the previous pin ("gemini-2.5-flash-preview-09-2025",
# a *preview*) was shut down by Google on 2026-02-17 and silently killed every
# extraction call. Previews get retired with ~2 weeks notice; stable models
# don't. Overridable via env so a future swap needs no code change.
GEMINI_MODEL = os.environ.get("GEMINI_MODEL", "gemini-3.5-flash")
# Model for the figure (vision) calls: classify + mining. Empty = same as
# GEMINI_MODEL. Classification is an easy task (one call per PDF picking a
# figure kind), so a cheaper model here costs little quality. The agent sets
# both from Run Control each cycle; read them at call time via text_model() /
# vision_model(), never through a name imported at module load.
GEMINI_VISION_MODEL = os.environ.get("GEMINI_VISION_MODEL", "").strip()
PROMPT_VERSION = "2.1"   # 2.1: processing route + conditions per property


def text_model() -> str:
    """The model the text extraction call uses right now."""
    return (GEMINI_MODEL or "gemini-3.5-flash").strip()


def vision_model() -> str:
    """The model the figure classify / mining calls use right now."""
    return (GEMINI_VISION_MODEL or "").strip() or text_model()

# How much the model may "think" per call. Thinking tokens are billed as
# output (6x the input price on gemini-3.5-flash): on 28 Sep 2026 a 25-paper
# cycle cost ~$7.50, $6.69 of it output tokens, ~30k per PDF where the extraction JSON itself
# is ~4k. Values: "dynamic" (the API default: unbounded), "low" / "minimal" /
# "high" (Gemini 3 thinking levels), "off" (budget 0), or an integer token
# budget (Gemini 2.x style). The agent sets this from its config each cycle
# (Run Control); GEMINI_THINKING is the first-boot default and what
# batch_ingest.py uses. If the model rejects the option (400), the call is
# repeated without it and the option is dropped for the rest of the process.
THINKING = os.environ.get("GEMINI_THINKING", "low").strip().lower() or "dynamic"
_THINKING_UNSUPPORTED = False

GEMINI_URL_TEMPLATE = (
    "https://generativelanguage.googleapis.com/v1beta/models/"
    "{model}:generateContent?key={key}"
)
GEMINI_UPLOAD_URL = (
    "https://generativelanguage.googleapis.com/upload/v1beta/files?key={key}"
)
GEMINI_FILE_STATUS_URL = (
    "https://generativelanguage.googleapis.com/v1beta/{name}?key={key}"
)
REQUEST_TIMEOUT_S = 300

# Transient-failure retry policy, mirrored from pdf_crawler.http_get(). The
# Gemini free tier returns 429s under load; a request that fails once is often
# fine moments later. Retry connection errors and 429/5xx, honoring Retry-After.
MAX_RETRIES = 3
RETRY_STATUS = frozenset({429, 500, 502, 503, 504})
BACKOFF_BASE = 2.0  # seconds; exponential per attempt

# Inline a PDF up to this size; above it, upload via the Gemini File API so the
# base64-inflated request stays under the generateContent size limit (Task 10).
INLINE_PDF_LIMIT_BYTES = 15 * 1024 * 1024
INLINE_PAGE_LIMIT = 80
# Average extractable characters/page below this => treat as scanned/image-only.
SCANNED_TEXT_PER_PAGE = 50


# --- enums (Task 8) --------------------------------------------------------

# Aligned with the app repo's page6.py PROPERTY_CATEGORIES.
SECTION_ENUM = [
    "Mechanical",
    "Thermal",
    "Electrical",
    "Physical",
    "Optical",
    "Rheological",
    "Processing",
    "Descriptive",
    "Composition/Reinforcement",
    "Architecture/Structure",
]

MATERIAL_CLASS_ENUM = ["Polymer", "Fiber", "Composite"]

# Processing route of the specimen a value was measured on (prompt 2.1). A
# property of a composite depends on how the part was made, so every row
# carries the route (process_type, normalized to this list), its name as
# printed and the processing conditions as printed. "Other" keeps a route the
# list does not cover; the printed name is always stored beside it.
PROCESS_TYPE_ENUM = [
    "Injection molding",
    "Compression molding / hot press",
    "Autoclave",
    "Out-of-autoclave (vacuum bag / oven)",
    "Automated fiber placement / tape laying",
    "Thermoforming / stamp forming",
    "Filament winding",
    "Pultrusion",
    "Extrusion / compounding",
    "Additive manufacturing",
    "Liquid molding (RTM / infusion)",
    "Welding / joining",
    "Fiber spinning",
    "Film / solution casting",
    "Electrospinning",
    "Heat treatment / annealing",
    "Other",
]

# process_status values (grounding of the route, independent of the row's
# value status — the route is context for the value, never a value itself):
#   ''                       no processing route reported for this row
#   'grounded'               the quote naming the route is on the cited page
#                            and every number in the conditions is in the PDF
#   'grounded_off_page'      same, but the quote is on another page
#   'conditions_unverified'  quote found, a number in the conditions is not
#   'ungrounded'             the quote is not in the PDF text
#   'unchecked'              no PDF text was available to check against
PROCESS_STATUS_GROUNDED = ("grounded", "grounded_off_page")


# --- schema (Tasks 2/3/8) --------------------------------------------------

_PROPERTY_SCHEMA: dict[str, Any] = {
    "type": "OBJECT",
    "properties": {
        "section": {"type": "STRING", "enum": SECTION_ENUM},
        "property_name": {"type": "STRING"},
        "value_raw": {"type": "STRING"},
        "value_num": {"type": "NUMBER", "nullable": True},
        "value_min": {"type": "NUMBER", "nullable": True},
        "value_max": {"type": "NUMBER", "nullable": True},
        "qualifier": {"type": "STRING"},
        "unit": {"type": "STRING"},
        "test_condition": {"type": "STRING"},
        "comments": {"type": "STRING"},
        "source_quote": {"type": "STRING"},
        "page": {"type": "INTEGER"},
        "process_id": {"type": "STRING"},
    },
    "required": ["section", "property_name", "value_raw", "unit", "source_quote", "page"],
}

_PROCESS_SCHEMA: dict[str, Any] = {
    "type": "OBJECT",
    "properties": {
        "process_id": {"type": "STRING"},
        "process_type": {"type": "STRING", "enum": PROCESS_TYPE_ENUM},
        "process_name": {"type": "STRING"},
        "process_conditions": {"type": "STRING"},
        "source_quote": {"type": "STRING"},
        "page": {"type": "INTEGER"},
    },
    "required": ["process_id", "process_type", "source_quote", "page"],
}

EXTRACTION_SCHEMA: dict[str, Any] = {
    "type": "OBJECT",
    "properties": {
        "processes": {"type": "ARRAY", "items": _PROCESS_SCHEMA},
        "materials": {
            "type": "ARRAY",
            "items": {
                "type": "OBJECT",
                "properties": {
                    "material_name": {"type": "STRING"},
                    "material_abbreviation": {"type": "STRING"},
                    "material_class": {"type": "STRING", "enum": MATERIAL_CLASS_ENUM},
                    "trade_grade": {"type": "STRING"},
                    "manufacturer": {"type": "STRING"},
                    "matrix": {"type": "STRING"},
                    "fiber": {"type": "STRING"},
                    "fiber_volume_fraction": {"type": "STRING"},
                    "properties": {"type": "ARRAY", "items": _PROPERTY_SCHEMA},
                },
                "required": ["material_name", "material_class", "properties"],
            },
        }
    },
    "required": ["materials"],
}

EXTRACTION_PROMPT = (
    "You are an expert materials scientist. From the attached PDF, extract every "
    "distinct material it characterizes as a list under `materials`. A datasheet "
    "or paper may describe MORE THAN ONE material (e.g. a neat resin and its "
    "composite, or several grades) — return each as its own entry with only its "
    "own properties. Do NOT merge two materials into one.\n\n"
    "For each material provide:\n"
    "- material_name (generic material, e.g. 'isotactic polypropylene')\n"
    "- material_abbreviation\n"
    "- material_class: one of Polymer | Fiber | Composite. Choose Composite when a "
    "matrix is reinforced with fibers (laminate, prepreg, CF/PEEK, glass-filled, a "
    "reported fiber volume fraction); Fiber for a bare fiber/yarn/tow datasheet; "
    "Polymer otherwise.\n"
    "- trade_grade (commercial/trade name; '' if absent)\n"
    "- manufacturer (company; '' if absent)\n"
    "- matrix, fiber, fiber_volume_fraction (e.g. '55%') — for composites; '' otherwise\n\n"
    "Extract ALL numeric and descriptive properties across every category. For "
    "each property return:\n"
    "- section: one of " + ", ".join(SECTION_ENUM) + "\n"
    "- property_name\n"
    "- value_raw: the value EXACTLY as printed, including ranges and qualifiers "
    "(e.g. '100-120', '≤ -18.0', '~3.5')\n"
    "- value_num: the single numeric value, or null if it is a range/non-numeric\n"
    "- value_min, value_max: the low/high of a range, else null\n"
    "- qualifier: one of '', '<', '<=', '>', '>=', '~', '±'\n"
    "- unit: the unit exactly as printed ('' if dimensionless)\n"
    "- test_condition (e.g. '23 °C, 50% RH'; '' if none)\n"
    "- comments ('' if none)\n"
    "- source_quote: the VERBATIM sentence or table cell from the PDF that states "
    "this value. Copy it character-for-character; do not paraphrase.\n"
    "- page: the 1-based PDF page number the value appears on.\n"
    "- process_id: the id of the processing route (see below) by which the "
    "specimen this value was measured on was made; '' if the document does "
    "not say.\n\n"
    "PROCESSING ROUTES. Under `processes`, list every distinct manufacturing or "
    "processing route by which the characterized materials or test specimens "
    "were made, as the document states it (materials-and-methods or processing "
    "section of a paper; processing notes of a datasheet). For each route:\n"
    "- process_id: 'P1', 'P2', ...\n"
    "- process_type: one of " + " | ".join(PROCESS_TYPE_ENUM) + "\n"
    "- process_name: the route as the document names it (e.g. 'laser-assisted "
    "automated tape placement with in-situ consolidation')\n"
    "- process_conditions: the processing parameters EXACTLY as printed, joined "
    "by '; ' (e.g. 'consolidation temperature 385 °C; pressure 1 MPa; hold 20 "
    "min; cooling rate 5 °C/min' — or nozzle temperature, bed temperature, "
    "layer height and print speed for printing; melt and mold temperature for "
    "molding). Only values stated in the document; '' if none are given.\n"
    "- source_quote: the VERBATIM sentence that names this route.\n"
    "- page: the 1-based PDF page of that sentence.\n"
    "The same route run at different settings (two cooling rates, two nozzle "
    "temperatures) is two entries. Then give every property the process_id of "
    "the route its specimen came from. Do NOT guess: when the document does "
    "not state how a material was processed (a vendor datasheet without "
    "processing notes, an as-received fiber or resin, a value quoted from "
    "other literature), leave process_id ''.\n\n"
    "Never invent values. If a value is not in the document, do not include it. "
    "Respond ONLY with valid JSON following the schema."
)


# ---------------------------------------------------------------------------
# Data classes
# ---------------------------------------------------------------------------


@dataclasses.dataclass
class Property:
    section: str
    property_name: str
    value_raw: str
    unit: str = ""
    value_num: Optional[float] = None
    value_min: Optional[float] = None
    value_max: Optional[float] = None
    qualifier: str = ""
    test_condition: str = ""
    comments: str = ""
    source_quote: str = ""
    page: Optional[int] = None
    # id of the ProcessRoute (Extraction.processes) the specimen was made by
    process_id: str = ""
    # filled by verify_against_text() / canonicalization:
    unit_canonical: str = ""
    value_si: Optional[float] = None
    status: str = "ok"
    flag_reason: str = ""


@dataclasses.dataclass
class ProcessRoute:
    """One processing route a document reports (prompt 2.1); properties point
    at it through ``Property.process_id``."""

    process_id: str
    process_type: str = ""
    process_name: str = ""
    process_conditions: str = ""
    source_quote: str = ""
    page: Optional[int] = None
    # filled by verify_against_text(): see PROCESS_TYPE_ENUM's status note
    status: str = ""


@dataclasses.dataclass
class Material:
    material_name: str
    material_abbreviation: str = ""
    material_class: str = ""
    trade_grade: str = ""
    manufacturer: str = ""
    matrix: str = ""
    fiber: str = ""
    fiber_volume_fraction: str = ""
    properties: list[Property] = dataclasses.field(default_factory=list)


@dataclasses.dataclass
class Extraction:
    materials: list[Material] = dataclasses.field(default_factory=list)
    model: str = dataclasses.field(default_factory=text_model)
    prompt_version: str = PROMPT_VERSION
    doc_status: str = "ok"  # "ok" | "scanned_no_text" | "empty_extraction"
    # processing routes the document reports (prompt 2.1); see ProcessRoute
    processes: list["ProcessRoute"] = dataclasses.field(default_factory=list)
    # Gemini usage for the text call (0 when no call was made). tokens_out
    # includes the model's thinking tokens, which are billed as output;
    # tokens_thinking is that thinking share on its own (for cost analysis).
    tokens_in: int = 0
    tokens_out: int = 0
    tokens_thinking: int = 0


def usage_detail(resp: Any) -> tuple[int, int, int]:
    """(tokens_in, tokens_out, tokens_thinking) from a generateContent
    response's ``usageMetadata``. tokens_out is billed output (candidates +
    thinking); tokens_thinking is the thinking part alone. (0, 0, 0) when
    absent or unparseable; never raises."""
    try:
        data = resp.json() if hasattr(resp, "json") else (resp or {})
        u = (data or {}).get("usageMetadata") or {}
        tin = int(u.get("promptTokenCount") or 0)
        think = int(u.get("thoughtsTokenCount") or 0)
        tout = int(u.get("candidatesTokenCount") or 0) + think
        return max(0, tin), max(0, tout), max(0, think)
    except Exception:
        return 0, 0, 0


def usage_from_response(resp: Any) -> tuple[int, int]:
    """(tokens_in, tokens_out) from a generateContent response's
    ``usageMetadata``; (0, 0) when absent or unparseable. Output counts
    thinking tokens (``thoughtsTokenCount``) with the candidate tokens, which
    is how they are billed. Never raises: cost accounting must not be able
    to fail an extraction."""
    try:
        data = resp.json() if hasattr(resp, "json") else (resp or {})
        u = (data or {}).get("usageMetadata") or {}
        tin = int(u.get("promptTokenCount") or 0)
        tout = int(u.get("candidatesTokenCount") or 0) + int(u.get("thoughtsTokenCount") or 0)
        return max(0, tin), max(0, tout)
    except Exception:
        return 0, 0


@dataclasses.dataclass
class PropertyRow:
    """One flattened, fully-resolved row ready for the SQLite mirror.

    Carries the legacy columns (`value`, `unit`, `english`, ...) so the CSV
    export and page1.py keep working, plus the new structured/provenance
    columns added in this phase.
    """

    # identity / routing
    material_name: str
    material_abbreviation: str
    material_key: str
    material_class: str
    # legacy columns (kept populated for backward compat)
    section: str
    property_name: str
    value: str
    unit: str
    english: str
    test_condition: str
    comments: str
    # composite descriptors promoted to real columns (Task 2)
    trade_grade: str = ""
    manufacturer: str = ""
    matrix: str = ""
    fiber: str = ""
    fiber_volume_fraction: str = ""
    # structured numeric value (Task 3)
    value_raw: str = ""
    value_num: Optional[float] = None
    value_min: Optional[float] = None
    value_max: Optional[float] = None
    qualifier: str = ""
    unit_canonical: str = ""
    value_si: Optional[float] = None
    # provenance (Tasks 1/6)
    source_pdf: str = ""
    source_sha1: str = ""
    page: Optional[int] = None
    source_quote: str = ""
    # status (Tasks 1/3/4)
    status: str = "ok"
    flag_reason: str = ""
    # bookkeeping
    model: str = dataclasses.field(default_factory=text_model)
    prompt_version: str = PROMPT_VERSION
    # provenance kind (figure-mining phase): 'text' (grounded in the PDF
    # text) or 'figure' (read off a plot/table image — an estimate; see
    # figures.py). figure_id points at the harvested PNG.
    origin: str = "text"
    figure_id: str = ""
    # figure-linking phase (figure_links.py, ported from the InDeS mapper):
    # for a TEXT row, figure_id may instead point at the harvested figure the
    # row's evidence CITES ("see Fig. 3"). The link is grounding evidence,
    # never a value — status is untouched. image_url / image are the columns
    # the shared Space reads to show a row's plot (S3 ref / PNG bytes).
    figure_ref: str = ""                       # citation as parsed, e.g. "Fig. 3(b)"
    figure_link_score: Optional[float] = None  # mapper5 score: 0.90-1.00 citation, <0.9 fallback
    figure_link_signals: str = ""              # "figure_citation, page_confirm" / "page, tokens"
    image_url: str = ""
    image: bytes = dataclasses.field(default=b"", repr=False, compare=False)
    # processing route of the specimen (prompt 2.1): normalized type, the
    # name and conditions as printed, the quote + page that name the route,
    # and how that quote grounded (see PROCESS_TYPE_ENUM's status note).
    # All empty when the document reports no route for this value.
    process_type: str = ""
    process_name: str = ""
    process_conditions: str = ""
    process_quote: str = ""
    process_page: Optional[int] = None
    process_status: str = ""


# ---------------------------------------------------------------------------
# Gemini request with retry/backoff (Task 7)
# ---------------------------------------------------------------------------


_KEY_PARAM_RE = re.compile(r"([?&](?:key|api_key|apikey|token|access_token)=)[^&\s'\")\]]+", re.I)


def redact_secrets(text: Any) -> str:
    """Strip credential query parameters from error text. requests' HTTPError
    and urllib3's connection errors quote the full request URL — for Gemini
    that URL carries `?key=<API key>` — so any error message that is logged
    or stored has to pass through here first."""
    return _KEY_PARAM_RE.sub(r"\1<redacted>", str(text))


class GeminiHTTPError(requests.HTTPError):
    """HTTPError for a Gemini endpoint whose message names the status and
    the API's own error detail (e.g. RESOURCE_EXHAUSTED: "Your project has
    exceeded its monthly spending cap") instead of the keyed URL."""

    def __init__(self, message: str, response: requests.Response, status: str = "",
                 detail: str = "") -> None:
        super().__init__(message, response=response)
        self.status = status        # Gemini "status" field, e.g. RESOURCE_EXHAUSTED
        self.detail = detail        # Gemini "message" field, human-readable

    @property
    def quota(self) -> bool:
        """True when the API refused for budget/quota reasons (spending cap,
        daily quota): a condition that will not clear by retrying now."""
        return (self.status == "RESOURCE_EXHAUSTED"
                or "spending cap" in self.detail.lower()
                or "quota" in self.detail.lower())


def _gemini_error_fields(resp: requests.Response) -> tuple[str, str]:
    """(status, message) from a Gemini error body, or ('', '') if not JSON."""
    try:
        body = resp.json()
    except ValueError:
        return "", ""
    err = (body or {}).get("error") if isinstance(body, dict) else None
    if not isinstance(err, dict):
        return "", ""
    return str(err.get("status") or ""), str(err.get("message") or "")[:300]


def _retry_after(resp: requests.Response) -> Optional[float]:
    val = resp.headers.get("Retry-After")
    if not val:
        return None
    try:
        # Cap: a hostile/buggy server sending Retry-After: 86400 must not
        # stall a cycle for hours (Space anti-hang hardening, commit 3f67f54).
        return max(0.0, min(float(val), 300.0))
    except ValueError:
        return None


def _request_with_retry(
    method: str,
    url: str,
    *,
    what: str = "Gemini",
    timeout: int = REQUEST_TIMEOUT_S,
    _sleep=time.sleep,
    **kw: Any,
) -> Optional[requests.Response]:
    """Issue one HTTP request, retrying 429/5xx and connection errors.

    Shared policy for every Gemini endpoint (generateContent, File API start /
    upload / status poll): bounded retries with exponential backoff, honoring
    a Retry-After header. Returns the 200 Response, or None if retries are
    exhausted / the status is non-retryable (after raise_for_status).
    """
    for attempt in range(MAX_RETRIES + 1):
        try:
            resp = requests.request(method, url, timeout=timeout, **kw)
        except requests.RequestException as exc:
            if attempt < MAX_RETRIES:
                delay = BACKOFF_BASE * (2 ** attempt)
                log.warning("%s request failed: %s; retry %d/%d in %.1fs",
                            what, redact_secrets(exc), attempt + 1, MAX_RETRIES, delay)
                _sleep(delay)
                continue
            # Connection errors quote the URL (with the key) — re-raise clean.
            raise requests.ConnectionError(
                f"{what} request failed: {redact_secrets(exc)}") from None
        if resp.status_code == 200:
            return resp
        status, detail = _gemini_error_fields(resp)
        if resp.status_code == 429 and (status == "RESOURCE_EXHAUSTED"
                                        or "spending cap" in detail.lower()):
            # A spent budget (monthly spending cap, daily quota) is not a
            # momentary rate limit: retrying it in 2/4/8 s only burns time,
            # 14 s per PDF, and on 28 Sep 2026 the retries' error text put
            # the API key into the public event stream. Fail at once.
            log.error("%s -> 429 %s: %s", what, status, detail)
            raise GeminiHTTPError(f"429 {status or 'RESOURCE_EXHAUSTED'} from {what}: {detail}",
                                  resp, status, detail)
        if resp.status_code in RETRY_STATUS and attempt < MAX_RETRIES:
            delay = _retry_after(resp) or BACKOFF_BASE * (2 ** attempt)
            log.warning(
                "%s -> %s; retry %d/%d in %.1fs",
                what, resp.status_code, attempt + 1, MAX_RETRIES, delay,
            )
            _sleep(delay)
            continue
        # Non-retryable (4xx) or retries exhausted: fail loudly, once. This
        # raise deliberately sits OUTSIDE the try above — HTTPError is a
        # RequestException, and raising it inside used to be caught by the
        # connection-error branch and retried, so a 400/401/403 (bad key,
        # invalid schema) burned MAX_RETRIES backoffs before surfacing.
        # The message names the endpoint and the API's detail, never the
        # keyed URL (requests' own raise_for_status() message does).
        log.error("%s -> %s: %s", what, resp.status_code, resp.text[:300])
        raise GeminiHTTPError(f"{resp.status_code} {resp.reason or 'error'} from {what}"
                              + (f": {status} {detail}".rstrip() if (status or detail) else ""),
                              resp, status, detail)
    return None


def thinking_config(setting: Optional[str] = None) -> Optional[dict]:
    """The generationConfig.thinkingConfig for the current THINKING setting,
    or None for "dynamic" (send nothing: the API default) and after the
    model rejected the option once."""
    if _THINKING_UNSUPPORTED:
        return None
    v = (setting if setting is not None else THINKING or "").strip().lower()
    if not v or v == "dynamic":
        return None
    if v == "off":
        return {"thinkingBudget": 0}
    if v in ("minimal", "low", "medium", "high"):
        return {"thinkingLevel": v}
    try:
        return {"thinkingBudget": max(0, int(v))}
    except ValueError:
        log.warning("GEMINI_THINKING=%r not understood; using the API default", v)
        return None


def apply_thinking(payload: dict) -> dict:
    """Add the thinking option to a generateContent payload (in place)."""
    tc = thinking_config()
    gc = payload.setdefault("generationConfig", {})
    if tc is None:
        gc.pop("thinkingConfig", None)
    else:
        gc["thinkingConfig"] = tc
    return payload


def gemini_request(
    url: str,
    payload: dict[str, Any],
    *,
    timeout: int = REQUEST_TIMEOUT_S,
    _sleep=time.sleep,
) -> Optional[requests.Response]:
    """POST `payload` to `url`, retrying 429/5xx and connection errors.

    Mirrors pdf_crawler.http_get()'s policy: bounded retries with exponential
    backoff, honoring a Retry-After header. Returns the 200 Response, or None
    if retries are exhausted / the status is non-retryable.

    The thinking option (THINKING) is applied here, so every Gemini call in
    extraction.py and figures.py gets it. A 400 that names the option
    (a model without that field, an out-of-range value) is retried once
    without it and disables it for the process: cost tuning must never stop
    an extraction.
    """
    global _THINKING_UNSUPPORTED
    apply_thinking(payload)
    try:
        return _request_with_retry("POST", url, json=payload, timeout=timeout, _sleep=_sleep)
    except GeminiHTTPError as exc:
        gc = payload.get("generationConfig") or {}
        if (exc.response is not None and exc.response.status_code == 400
                and "thinkingConfig" in gc and "think" in (exc.detail or "").lower()):
            log.warning("model rejected thinkingConfig=%s (%s); retrying without it and "
                        "leaving thinking at the API default from now on",
                        gc.get("thinkingConfig"), exc.detail[:120])
            _THINKING_UNSUPPORTED = True
            gc.pop("thinkingConfig", None)
            return _request_with_retry("POST", url, json=payload, timeout=timeout, _sleep=_sleep)
        raise


# File API processing wait: exponential poll (1, 2, 4, 8, ... capped at 15 s)
# up to this many seconds. Was a fixed 30 x 1 s, so a large PDF that took
# longer than 30 s to process came back None -> empty_extraction and was
# re-uploaded from scratch on every run.
FILE_ACTIVE_WAIT_S = 180
FILE_POLL_CAP_S = 15.0


def _upload_pdf_file(
    pdf_bytes: bytes, filename: str, api_key: str, *, _sleep=time.sleep
) -> Optional[str]:
    """Upload a PDF via the Gemini File API (resumable protocol); return file URI.

    Used for large/long PDFs (Task 10) where base64 inlining would blow the
    generateContent request-size limit. Every hop (start, upload, status poll)
    goes through the same retry/backoff policy as generateContent — a single
    free-tier 429 used to fail the whole PDF.
    """
    start_url = GEMINI_UPLOAD_URL.format(key=api_key)
    start = _request_with_retry(
        "POST", start_url, what="File API start",
        headers={
            "X-Goog-Upload-Protocol": "resumable",
            "X-Goog-Upload-Command": "start",
            "X-Goog-Upload-Header-Content-Length": str(len(pdf_bytes)),
            "X-Goog-Upload-Header-Content-Type": "application/pdf",
            "Content-Type": "application/json",
        },
        json={"file": {"display_name": filename}},
        _sleep=_sleep,
    )
    if start is None:
        return None
    upload_url = start.headers.get("X-Goog-Upload-URL")
    if not upload_url:
        log.error("File API did not return an upload URL")
        return None

    up = _request_with_retry(
        "POST", upload_url, what="File API upload",
        headers={
            "X-Goog-Upload-Offset": "0",
            "X-Goog-Upload-Command": "upload, finalize",
            "Content-Length": str(len(pdf_bytes)),
        },
        data=pdf_bytes,
        _sleep=_sleep,
    )
    if up is None:
        return None
    info = up.json().get("file", {})
    name = info.get("name")
    uri = info.get("uri")
    state = info.get("state")

    # Wait for the file to become ACTIVE before referencing it.
    waited = 0.0
    delay = 1.0
    while True:
        if state == "ACTIVE":
            return uri
        if state == "FAILED":
            log.error("File API processing failed for %s", filename)
            return None
        if waited >= FILE_ACTIVE_WAIT_S or not name:
            log.error("File API: %s not ACTIVE after %.0fs (state=%s)",
                      filename, waited, state)
            return None
        _sleep(delay)
        waited += delay
        delay = min(delay * 2, FILE_POLL_CAP_S)
        poll = _request_with_retry(
            "GET", GEMINI_FILE_STATUS_URL.format(name=name, key=api_key),
            what="File API status", _sleep=_sleep,
        )
        if poll is None:
            return None
        info = poll.json()
        state = info.get("state")
        uri = info.get("uri", uri)


def _parse_extraction_json(resp: Any) -> Optional[dict[str, Any]]:
    data = resp.json() if hasattr(resp, "json") else (resp or {})
    candidates = data.get("candidates", [])
    if not candidates:
        return None
    parts = candidates[0].get("content", {}).get("parts", [])
    for part in parts:
        text = (part.get("text") or "").strip()
        if text.startswith("{"):
            try:
                return json.loads(text)
            except json.JSONDecodeError:
                log.error("Gemini returned non-JSON text")
                return None
    return None


# ---------------------------------------------------------------------------
# PDF text (grounding + scanned detection)
# ---------------------------------------------------------------------------


def pdf_page_texts(pdf_bytes: bytes) -> list[str]:
    """Return per-page extractable text. Empty list if PyMuPDF is unavailable."""
    if fitz is None:
        return []
    try:
        doc = fitz.open(stream=pdf_bytes, filetype="pdf")
    except Exception as exc:  # pragma: no cover
        log.warning("Could not open PDF for text extraction: %s", exc)
        return []
    try:
        return [p.get_text() for p in doc]
    finally:
        doc.close()


def _is_scanned(page_texts: list[str]) -> bool:
    if not page_texts:
        return False  # can't tell without PyMuPDF; don't falsely flag
    total = sum(len(t.strip()) for t in page_texts)
    return (total / max(1, len(page_texts))) < SCANNED_TEXT_PER_PAGE


# ---------------------------------------------------------------------------
# Extraction entry point
# ---------------------------------------------------------------------------


def extract_from_pdf(pdf_bytes: bytes, filename: str, api_key: str) -> Extraction:
    """Extract structured materials from a PDF (Tasks 2/3/8/10).

    Returns an Extraction whose ``doc_status`` is ``scanned_no_text`` (and
    ``materials`` empty) when the PDF has no extractable text, so the caller
    can record the source without fabricating rows.
    """
    page_texts = pdf_page_texts(pdf_bytes)
    if _is_scanned(page_texts):
        log.info("%s looks scanned/image-only; skipping extraction", filename)
        return Extraction(doc_status="scanned_no_text")

    parts: list[dict[str, Any]] = [{"text": EXTRACTION_PROMPT}]
    use_file_api = (
        len(pdf_bytes) > INLINE_PDF_LIMIT_BYTES or len(page_texts) > INLINE_PAGE_LIMIT
    )
    if use_file_api:
        uri = _upload_pdf_file(pdf_bytes, filename, api_key)
        if not uri:
            return Extraction(doc_status="empty_extraction")
        parts.append({"fileData": {"mimeType": "application/pdf", "fileUri": uri}})
    else:
        parts.append(
            {
                "inlineData": {
                    "mimeType": "application/pdf",
                    "data": base64.b64encode(pdf_bytes).decode("utf-8"),
                }
            }
        )

    payload = {
        "contents": [{"parts": parts}],
        "generationConfig": {
            "temperature": 0,
            "responseMimeType": "application/json",
            "responseSchema": EXTRACTION_SCHEMA,
        },
    }
    model = text_model()
    url = GEMINI_URL_TEMPLATE.format(model=model, key=api_key)
    resp = gemini_request(url, payload)
    if resp is None:
        return Extraction(doc_status="empty_extraction", model=model)
    return extraction_from_response(resp, model)


def extraction_from_response(resp: Any, model: Optional[str] = None) -> Extraction:
    """Extraction (with usage) from a generateContent response object or its
    parsed JSON dict (the Batch API hands back dicts)."""
    model = model or text_model()
    tokens_in, tokens_out, tokens_thinking = usage_detail(resp)
    raw = _parse_extraction_json(resp)
    if not raw:
        # The call was made and billed even though nothing usable came back.
        return Extraction(doc_status="empty_extraction", model=model,
                          tokens_in=tokens_in, tokens_out=tokens_out,
                          tokens_thinking=tokens_thinking)
    out = _extraction_from_json(raw)
    out.model = model
    out.tokens_in, out.tokens_out, out.tokens_thinking = tokens_in, tokens_out, tokens_thinking
    return out


def _extraction_from_json(raw: dict[str, Any]) -> Extraction:
    """Coerce raw model JSON into a typed Extraction (tolerates the old shape)."""
    materials_json = raw.get("materials")
    if materials_json is None:
        # Back-compat: a single-material response with a flat property list.
        flat = raw.get("mechanical_properties") or raw.get("properties") or []
        materials_json = [
            {
                "material_name": raw.get("material_name", ""),
                "material_abbreviation": raw.get("material_abbreviation", ""),
                "material_class": raw.get("material_class", ""),
                "trade_grade": raw.get("trade_grade", ""),
                "manufacturer": raw.get("manufacturer", ""),
                "properties": flat,
            }
        ]

    materials: list[Material] = []
    for mj in materials_json or []:
        props: list[Property] = []
        for pj in mj.get("properties") or []:
            value_raw = (pj.get("value_raw") or pj.get("value") or "").strip()
            prop = Property(
                section=_norm_section(pj.get("section")),
                property_name=(pj.get("property_name") or "").strip() or "Unknown property",
                value_raw=value_raw,
                unit=(pj.get("unit") or "").strip(),
                value_num=_as_float(pj.get("value_num")),
                value_min=_as_float(pj.get("value_min")),
                value_max=_as_float(pj.get("value_max")),
                qualifier=(pj.get("qualifier") or "").strip(),
                test_condition=(pj.get("test_condition") or "").strip(),
                comments=(pj.get("comments") or "").strip(),
                source_quote=(pj.get("source_quote") or "").strip(),
                page=_as_int(pj.get("page")),
                process_id=str(pj.get("process_id") or "").strip(),
            )
            # Fill structured numeric fields from value_raw if the model omitted them.
            _fill_numeric(prop)
            props.append(prop)
        materials.append(
            Material(
                material_name=(mj.get("material_name") or "").strip(),
                material_abbreviation=(mj.get("material_abbreviation") or "").strip(),
                material_class=(mj.get("material_class") or "").strip(),
                trade_grade=(mj.get("trade_grade") or "").strip(),
                manufacturer=(mj.get("manufacturer") or "").strip(),
                matrix=(mj.get("matrix") or "").strip(),
                fiber=(mj.get("fiber") or "").strip(),
                fiber_volume_fraction=(mj.get("fiber_volume_fraction") or "").strip(),
                properties=props,
            )
        )
    return Extraction(materials=materials, processes=_processes_from_json(raw))


def _norm_process_type(ptype: Optional[str], name: str = "") -> str:
    """Normalize a model-returned process type (or, failing that, the printed
    route name) to PROCESS_TYPE_ENUM; 'Other' when nothing matches."""
    for cand in ((ptype or "").strip(), (name or "").strip()):
        if not cand:
            continue
        low = cand.lower()
        for canon in PROCESS_TYPE_ENUM:
            if low == canon.lower():
                return canon
        for keys, canon in _PROCESS_SYNONYMS:
            if any(_process_key_in(k, low) for k in keys):
                return canon
    return "Other"


def _process_key_in(key: str, low: str) -> bool:
    """Substring match for word stems; whole-word match for short keys and
    acronyms, so 'oven' does not fire on 'woven' or 'atl' on 'atlas'."""
    if len(key) <= 4:
        return re.search(r"(?<![a-z])" + re.escape(key) + r"(?![a-z])", low) is not None
    return key in low


# Ordered: the first hit wins, so the specific routes come before the generic
# ones ('injection' before 'molding', 'out-of-autoclave' before 'autoclave').
_PROCESS_SYNONYMS: list[tuple[tuple[str, ...], str]] = [
    (("injection",), "Injection molding"),
    (("out-of-autoclave", "out of autoclave", "ooa", "vacuum bag", "vbo", "oven"),
     "Out-of-autoclave (vacuum bag / oven)"),
    (("autoclave",), "Autoclave"),
    (("fiber placement", "fibre placement", "tape placement", "tape laying",
      "tape layup", "afp", "atp", "atl"), "Automated fiber placement / tape laying"),
    (("thermoform", "stamp", "press forming", "diaphragm forming"),
     "Thermoforming / stamp forming"),
    (("filament wind",), "Filament winding"),
    (("pultru",), "Pultrusion"),
    (("additive", "3d print", "3d-print", "fused filament", "fused deposition",
      "fff", "fdm", "material extrusion", "laser sinter", "sls", "powder bed"),
     "Additive manufacturing"),
    (("rtm", "resin transfer", "infusion", "liquid mold", "liquid mould"),
     "Liquid molding (RTM / infusion)"),
    (("weld", "joining", "bonding"), "Welding / joining"),
    (("electrospin",), "Electrospinning"),
    (("spinning", "spun"), "Fiber spinning"),
    (("solution cast", "solvent cast", "film cast", "casting"), "Film / solution casting"),
    (("anneal", "heat treat", "heat-treat"), "Heat treatment / annealing"),
    (("compression", "hot press", "hot-press", "press consolidat", "press mold",
      "press mould"), "Compression molding / hot press"),
    (("extru", "compound"), "Extrusion / compounding"),
]


def _processes_from_json(raw: dict[str, Any]) -> list[ProcessRoute]:
    """The top-level `processes` list as typed routes (absent in pre-2.1
    responses -> []). Entries without an id, or repeating one, are dropped."""
    out: list[ProcessRoute] = []
    seen: set[str] = set()
    for pj in raw.get("processes") or []:
        if not isinstance(pj, dict):
            continue
        pid = str(pj.get("process_id") or "").strip()
        if not pid or pid in seen:
            continue
        seen.add(pid)
        name = (pj.get("process_name") or "").strip()
        out.append(ProcessRoute(
            process_id=pid,
            process_type=_norm_process_type(pj.get("process_type"), name),
            process_name=name,
            process_conditions=(pj.get("process_conditions") or "").strip(),
            source_quote=(pj.get("source_quote") or "").strip(),
            page=_as_int(pj.get("page")),
        ))
    return out


def _as_float(v: Any) -> Optional[float]:
    if v is None or v == "":
        return None
    try:
        return float(v)
    except (TypeError, ValueError):
        return None


def _as_int(v: Any) -> Optional[int]:
    if v is None or v == "":
        return None
    try:
        return int(v)
    except (TypeError, ValueError):
        return None


def _norm_section(section: Optional[str]) -> str:
    """Normalize free-text section drift to the enum (Task 8)."""
    s = (section or "").strip()
    if not s:
        return "Descriptive"
    low = s.lower()
    for canon in SECTION_ENUM:
        if low == canon.lower():
            return canon
    # substring / synonym normalization, e.g. "Mechanical Properties" -> "Mechanical"
    synonyms = {
        "mechanical": "Mechanical",
        "thermal": "Thermal",
        "electrical": "Electrical",
        "dielectric": "Electrical",
        "physical": "Physical",
        "optical": "Optical",
        "rheolog": "Rheological",
        "viscosity": "Rheological",
        "processing": "Processing",
        "descriptive": "Descriptive",
        "composition": "Composition/Reinforcement",
        "reinforcement": "Composition/Reinforcement",
        "architecture": "Architecture/Structure",
        "structure": "Architecture/Structure",
    }
    for key, canon in synonyms.items():
        if key in low:
            return canon
    return "Descriptive"


# ---------------------------------------------------------------------------
# Value parsing (Task 3)
# ---------------------------------------------------------------------------

_QUALIFIER_MAP = [
    ("≤", "<="), ("≥", ">="), ("≈", "~"),
    ("<=", "<="), (">=", ">="), ("<", "<"), (">", ">"),
    ("~", "~"), ("±", "±"),
]
_NUM_RE = re.compile(r"[-+]?\d{1,3}(?:,\d{3})+(?:\.\d+)?|[-+]?\d*\.?\d+(?:[eE][-+]?\d+)?")
# Plus/minus built from _NUM_RE so both operands accept thousands separators and
# scientific notation ('1,200 ± 100', '1e-3 ± 2e-4') — a plain \d*\.?\d+ operand
# used to anchor after the comma and read '1,200 ± 100' as 200 ± 100.
_PM_RE = re.compile(rf"({_NUM_RE.pattern})\s*(?:±|\+/-|\+-)\s*({_NUM_RE.pattern})")


def _clean_number(tok: str) -> Optional[float]:
    try:
        return float(tok.replace(",", ""))
    except ValueError:
        return None


def parse_value_raw(value_raw: str) -> tuple[Optional[float], Optional[float], Optional[float], str]:
    """Parse a printed value into (value_num, value_min, value_max, qualifier).

    Handles ranges ('100-120', '100 to 120'), qualifiers ('<= -18', '~3.5'),
    plus-minus ('5.0 +/- 0.2'), thousands separators and scientific notation.
    """
    if not value_raw:
        return None, None, None, ""
    s = unicodedata.normalize("NFKC", value_raw).strip()
    # NFKC does NOT fold the typographic minus (U+2212) or dashes to '-'. PDFs
    # typeset negatives and ranges with them, so without this a Tg of '−60 °C'
    # parses as +60 and '20−30' loses its range. Same folding as _normalize_text.
    s = s.replace("−", "-").replace("–", "-").replace("—", "-")

    qualifier = ""
    for needle, canon in _QUALIFIER_MAP:
        if needle in s:
            qualifier = canon
            break

    # plus/minus -> midpoint value with min/max
    pm = _PM_RE.search(s)
    if pm:
        base = _clean_number(pm.group(1))
        delta = _clean_number(pm.group(2))
        if base is not None and delta is not None:
            delta = abs(delta)
            return base, base - delta, base + delta, "±"

    # Range: two numbers separated by a dash/"to" (all dash variants were folded
    # to '-' above). Guard against a leading sign being misread as a separator
    # by working on the sign-stripped remainder.
    body = s
    lead_sign = ""
    if body[:1] in "+-":
        lead_sign, body = body[0], body[1:]
    range_match = re.match(
        r"\s*(\d{1,3}(?:,\d{3})+(?:\.\d+)?|\d*\.?\d+(?:[eE][-+]?\d+)?)"
        r"\s*(?:-|to|\.\.\.|…)\s*"
        r"([-+]?\d{1,3}(?:,\d{3})+(?:\.\d+)?|[-+]?\d*\.?\d+(?:[eE][-+]?\d+)?)",
        body,
    )
    if range_match:
        lo = _clean_number(lead_sign + range_match.group(1))
        hi = _clean_number(range_match.group(2))
        if lo is not None and hi is not None:
            if lo > hi:
                lo, hi = hi, lo
            return None, lo, hi, qualifier

    nums = _NUM_RE.findall(s)
    if not nums:
        return None, None, None, qualifier
    val = _clean_number(nums[0])
    return val, None, None, qualifier


def _fill_numeric(prop: Property) -> None:
    """Backfill value_num/min/max/qualifier from value_raw when model omitted them."""
    has_any = (
        prop.value_num is not None
        or prop.value_min is not None
        or prop.value_max is not None
    )
    if has_any and prop.qualifier:
        return
    num, lo, hi, qual = parse_value_raw(prop.value_raw)
    if prop.value_num is None and prop.value_min is None and prop.value_max is None:
        prop.value_num, prop.value_min, prop.value_max = num, lo, hi
    if not prop.qualifier:
        prop.qualifier = qual


def _representative_value(prop: Property) -> Optional[float]:
    if prop.value_num is not None:
        return prop.value_num
    if prop.value_min is not None and prop.value_max is not None:
        return (prop.value_min + prop.value_max) / 2.0
    return prop.value_min if prop.value_min is not None else prop.value_max


# ---------------------------------------------------------------------------
# Unit normalization + plausibility (Task 3)
# ---------------------------------------------------------------------------

if pint is not None:  # pragma: no branch
    _UREG = pint.UnitRegistry()
    _UREG.define("ksi = 1000 * psi")
    _UREG.define("Msi = 1000000 * psi")
else:  # pragma: no cover
    _UREG = None


@dataclasses.dataclass
class _Family:
    name: str
    keywords: tuple[str, ...]
    canonical: str          # pint unit string (or display for non-pint families)
    display: str            # human-facing unit label
    lo: float               # plausibility low, in `display` units
    hi: float               # plausibility high, in `display` units
    kind: str               # "physical" | "temperature" | "raw" | "passthrough"
    si_factor: float = 1.0  # for "raw" families: value_si = value_in_display * si_factor
    # For "raw" families: accepted printed-unit spellings (normalized via
    # _norm_unit_key) -> factor that converts the printed value into `display`
    # units. '' may be present to accept a bare number. Anything else is a
    # unit_review, never a silent default. Populated below.
    unit_map: dict[str, float] = dataclasses.field(default_factory=dict)


# Keyword matching. Multi-word keywords match as substrings ("tensile modulus"
# in "Tensile modulus (0°)"). SHORT tokens (tg, tm, hdt, cte) must match on
# token boundaries — bare substring matching made 'tm' hit "ASTM" (so every
# property citing an ASTM standard became a melting temperature) and 'tg' hit
# "outgassing". Left boundary = not a letter/digit; right boundary = not a
# LETTER (a trailing digit is allowed: 'Tg2', 'CTE1', 'HDT1.8' are how DSC/DMA
# reports and laminate datasheets label second-heating Tg and the two CTE
# regimes).
_SHORT_TOKENS = frozenset({"tg", "tm", "tc", "hdt", "cte", "clte", "cai", "sbs",
                           "young", "young's", "youngs"})
_KEYWORD_RE_CACHE: dict[str, "re.Pattern[str]"] = {}


def _keyword_matches(kw: str, name: str) -> bool:
    if kw not in _SHORT_TOKENS:
        return kw in name
    pat = _KEYWORD_RE_CACHE.get(kw)
    if pat is None:
        pat = re.compile(r"(?<![a-z0-9])" + re.escape(kw) + r"(?![a-z])")
        _KEYWORD_RE_CACHE[kw] = pat
    return pat.search(name) is not None


# Accepted printed spellings for the raw families. Keys are normalized by
# _norm_unit_key() (lowercased, NFKC, whitespace/µ/degree-sign folded,
# 'x⁻¹'/'x^-1' suffixes rewritten to '/x').
_CTE_UNIT_MAP: dict[str, float] = {
    # -> ppm/°C
    "": 1.0,                       # bare: see _bare_value_factor (ppm vs 1/K by magnitude)
    "ppm/c": 1.0, "ppm/k": 1.0, "ppm/degc": 1.0, "ppm/°c": 1.0, "ppm": 1.0,
    "um/m/c": 1.0, "um/m/k": 1.0, "um/m/degc": 1.0, "um/m/°c": 1.0,
    "um/(m*c)": 1.0, "um/(m*k)": 1.0, "um/(m·c)": 1.0, "um/(m·k)": 1.0,
    "um/m·c": 1.0, "um/m·k": 1.0, "um/m°c": 1.0, "um/mk": 1.0, "um/m*c": 1.0, "um/m*k": 1.0,
    "um·/m/k": 1.0, "um/m/m/k": 1.0,
    "10^-6/c": 1.0, "10^-6/k": 1.0, "10^-6/degc": 1.0, "10^-6/°c": 1.0,
    "10-6/c": 1.0, "10-6/k": 1.0, "10-6/°c": 1.0,
    "x10^-6/c": 1.0, "x10^-6/k": 1.0, "x10^-6/°c": 1.0,
    "x10-6/c": 1.0, "x10-6/k": 1.0, "x10-6/°c": 1.0,
    "e-6/c": 1.0, "e-6/k": 1.0, "e-6/°c": 1.0,
    "1e-6/c": 1.0, "1e-6/k": 1.0, "1e-6/°c": 1.0,
    "uin/in/f": 1.8, "uin/in/°f": 1.8, "uin/in/degf": 1.8, "ppm/f": 1.8,
    "ppm/degf": 1.8, "ppm/°f": 1.8, "10^-6/f": 1.8, "10^-6/°f": 1.8,
    "10-6/f": 1.8, "10-6/°f": 1.8, "x10^-6/f": 1.8, "x10^-6/°f": 1.8,
    "10^-6in/in/f": 1.8, "10^-6in/in/°f": 1.8, "10-6in/in/f": 1.8, "10-6in/in/°f": 1.8,
    "1/c": 1e6, "1/k": 1e6, "1/degc": 1e6, "1/°c": 1e6, "/c": 1e6, "/k": 1e6,
    "/degc": 1e6, "/°c": 1e6, "m/m/c": 1e6, "m/m/k": 1e6, "m/m/°c": 1e6,
    "m/(m*k)": 1e6, "m/(m·k)": 1e6, "mm/mm/c": 1e6, "mm/mm/k": 1e6, "mm/mm/°c": 1e6,
    "1/f": 1.8e6, "1/degf": 1.8e6, "1/°f": 1.8e6, "/f": 1.8e6, "/°f": 1.8e6,
    "in/in/f": 1.8e6, "in/in/°f": 1.8e6,
}
_ELONGATION_UNIT_MAP: dict[str, float] = {
    # -> %
    "%": 1.0, "percent": 1.0, "pct": 1.0,
    # bare / ratio: see _bare_value_factor — a strain FRACTION (0.024) or a
    # percent printed without its unit (2.4), decided by magnitude.
    "": 1.0, "-": 1.0, "mm/mm": 100.0, "m/m": 100.0, "in/in": 100.0,
    "ratio": 100.0, "strain": 100.0, "fraction": 100.0,
}
_SPECIFIC_GRAVITY_UNIT_MAP: dict[str, float] = {
    # dimensionless by definition; tolerate g/cm³ spellings (SG ≈ density in g/cm³)
    "": 1.0, "-": 1.0, "g/cm3": 1.0, "g/cm^3": 1.0, "g/cc": 1.0, "g/ml": 1.0,
    "g/l": 0.001, "kg/m3": 0.001, "kg/m^3": 0.001, "kg/dm3": 1.0, "kg/dm^3": 1.0,
    "kg/l": 1.0,
}


def _bare_value_factor(fam: _Family, rep: float, value_raw: str) -> Optional[float]:
    """Factor for a raw-family value printed with NO unit (or a lone dash).

    The bare case is genuinely ambiguous, so decide by what the number can be:

    * elongation: '%' inside value_raw wins (a datasheet cell "2.4 %" the model
      copied whole) -> percent. Otherwise |v| < 0.5 is a strain fraction
      (0.024 = 2.4 %; no thermoplastic composite fails below 0.5 % strain);
      |v| >= 0.5 is a percent printed without its unit (2.4 -> 2.4 %). The
      x100 rule used to apply to every bare number and turned "2.4 %"/'' into
      240 % with status ok.
    * cte: |v| < 1e-3 is a 1/K fraction (2.3e-5 -> 23 ppm/°C), else ppm/°C.
    * specific gravity: dimensionless, factor 1.
    Returns None to say "don't know" -> unit_review.
    """
    if fam.name == "elongation":
        if "%" in value_raw:
            return 1.0
        return 100.0 if abs(rep) < 0.5 else 1.0
    if fam.name == "cte":
        return 1e6 if 0 < abs(rep) < 1e-3 else 1.0
    if fam.name == "specific_gravity":
        return 1.0
    return None


# Ordered: specific keywords before generic ones (first match wins).
# "passthrough" families exist to *shield* generic keywords: e.g. dielectric,
# impact and tear strength are not pressures, so they must not fall into the
# MPa families below them. They carry no unit check and no plausibility range.
# Compression-after-impact (CAI) is a real MPa strength and must be caught
# BEFORE the impact shield.
PROPERTY_FAMILIES: list[_Family] = [
    _Family("cai_strength", ("compression after impact", "compression-after-impact",
                             "cai"),
            "MPa", "MPa", 0.5, 6000.0, "physical"),
    _Family("dielectric_strength", ("dielectric strength", "breakdown strength",
                                    "breakdown voltage"),
            "", "", 0.0, 0.0, "passthrough"),
    _Family("impact_strength", ("impact strength", "impact resistance", "izod",
                                "charpy", "impact energy"),
            "", "", 0.0, 0.0, "passthrough"),
    _Family("tear_strength", ("tear strength", "tear resistance"),
            "", "", 0.0, 0.0, "passthrough"),
    _Family("peel_strength", ("peel strength", "peel resistance", "bond strength",
                              "adhesion strength", "adhesive strength", "weld strength"),
            "", "", 0.0, 0.0, "passthrough"),
    _Family("tensile_modulus",
            ("tensile modulus", "modulus of elasticity", "young", "young's",
             "youngs", "elastic modulus"),
            "GPa", "GPa", 0.01, 1000.0, "physical"),
    _Family("flexural_modulus", ("flexural modulus", "bending modulus"),
            "GPa", "GPa", 0.01, 800.0, "physical"),
    _Family("shear_modulus", ("shear modulus", "modulus of rigidity"),
            "GPa", "GPa", 0.005, 500.0, "physical"),
    _Family("storage_modulus", ("storage modulus", "compressive modulus", "modulus"),
            "GPa", "GPa", 0.001, 1000.0, "physical"),
    _Family("tensile_strength",
            ("tensile strength", "ultimate tensile", "strength at break",
             "yield strength", "tensile stress"),
            "MPa", "MPa", 0.5, 10000.0, "physical"),
    _Family("flexural_strength", ("flexural strength", "bending strength"),
            "MPa", "MPa", 0.5, 4000.0, "physical"),
    _Family("compressive_strength", ("compressive strength", "compression strength"),
            "MPa", "MPa", 0.5, 6000.0, "physical"),
    # Bare "strength" was removed from this family: it captured dielectric /
    # impact / tear strength (see passthrough guards above) into an MPa check.
    _Family("shear_strength", ("shear strength", "ilss", "interlaminar shear"),
            "MPa", "MPa", 0.5, 4000.0, "physical"),
    _Family("glass_transition", ("glass transition", "tg"),
            "degC", "°C", -150.0, 600.0, "temperature"),
    _Family("melting", ("melting", "melt temperature", "tm"),
            "degC", "°C", 50.0, 500.0, "temperature"),
    _Family("crystallization", ("crystallization",),
            "degC", "°C", 0.0, 500.0, "temperature"),
    _Family("decomposition", ("decomposition", "degradation temperature"),
            "degC", "°C", 100.0, 1200.0, "temperature"),
    _Family("hdt", ("heat deflection", "deflection temperature", "hdt",
                    "heat distortion"),
            "degC", "°C", 0.0, 600.0, "temperature"),
    _Family("cte", ("thermal expansion", "cte", "clte", "expansion coefficient"),
            "ppm/degC", "ppm/°C", -50.0, 500.0, "raw", si_factor=1e-6,
            unit_map=_CTE_UNIT_MAP),
    # Specific gravity is dimensionless by definition; keep it OUT of the pint
    # density family or every SG row is false-flagged missing_unit.
    _Family("specific_gravity", ("specific gravity", "relative density"),
            "", "", 0.1, 12.0, "raw", si_factor=1.0,
            unit_map=_SPECIFIC_GRAVITY_UNIT_MAP),
    _Family("density", ("density",),
            "g/cm**3", "g/cm³", 0.1, 12.0, "physical"),
    _Family("elongation", ("elongation", "strain at break", "strain to failure",
                           "failure strain", "ultimate strain"),
            "%", "%", 0.001, 2000.0, "raw", si_factor=0.01,
            unit_map=_ELONGATION_UNIT_MAP),
    # Generic composite strengths that are real pressures (MPa). Kept LAST so
    # every shield / specific family above wins first; the bare "strength"
    # keyword is safe here because dielectric / impact / tear / peel are
    # already routed to passthrough families.
    _Family("other_strength",
            ("short beam", "short-beam", "sbs", "bearing strength", "open hole",
             "open-hole", "filled hole", "filled-hole", "notched tensile",
             "unnotched", "interlaminar strength", "transverse strength",
             "strength"),
            "MPa", "MPa", 0.5, 10000.0, "physical"),
]


def _match_family(prop: Property) -> Optional[_Family]:
    name = prop.property_name.lower()
    for fam in PROPERTY_FAMILIES:
        for kw in fam.keywords:
            if _keyword_matches(kw, name):
                return fam
    return None


# 'X⁻¹' / 'X^-1' / 'X-1' suffix -> '/X' so 'K⁻¹', '10⁻⁶ K⁻¹', 'µm·m⁻¹·K⁻¹'
# collapse onto the '/k' spellings already in the maps.
_INVERSE_SUFFIX_RE = re.compile(r"([a-z°]+)\^?-1")


def _norm_unit_key(unit: str) -> str:
    """Normalize a printed unit into a lookup key for the raw-family unit maps."""
    u = unit.strip()
    # BEFORE NFKC: it folds 'º' (masculine ordinal, often typed for a degree
    # sign) to 'o' and superscripts to plain digits, so handle those first.
    u = u.replace("º", "°").replace("˚", "°")
    u = u.replace("⁻", "-").replace("⁺", "+")
    u = u.translate(str.maketrans("⁰¹²³⁴⁵⁶⁷⁸⁹", "0123456789"))
    u = unicodedata.normalize("NFKC", u).lower()
    u = u.replace("µ", "u").replace("μ", "u")
    u = u.replace("−", "-").replace("–", "-").replace("—", "-")
    u = u.replace(" ", "").replace("(", "").replace(")", "")
    u = u.replace("**", "^").replace("×", "x").replace("*10", "x10").replace("e-06", "e-6")
    u = u.replace("·", "*")
    # 'x⁻¹' style: 'k^-1' -> '/k'; 'um*m-1*k-1' -> 'um/m/k'
    u = _INVERSE_SUFFIX_RE.sub(r"/\1", u)
    u = u.replace("*/", "/").replace("//", "/")
    if u.startswith("*"):
        u = u[1:]
    return u


def _raw_factor(fam: _Family, unit: str, rep: Optional[float] = None,
                value_raw: str = "") -> Optional[float]:
    """Factor converting a printed raw-family value into `fam.display` units, or None."""
    key = _norm_unit_key(unit)
    if key in ("", "-") and rep is not None:
        return _bare_value_factor(fam, rep, value_raw)
    if key in fam.unit_map:
        return fam.unit_map[key]
    # tolerate '°c' vs 'c' vs 'degc' interchangeably
    for a, b in (("°c", "c"), ("degc", "c"), ("°f", "f"), ("degf", "f"),
                 ("°k", "k"), ("degk", "k")):
        alt = key.replace(a, b)
        if alt in fam.unit_map:
            return fam.unit_map[alt]
    return None


# A letter followed by 2/3 is a power (cm3, mm2, in2, ft3, kJ/m2 ...) — but NOT
# the exponent marker of scientific notation ('1e3 psi', '10E3 psi').
_UNIT_POWER_RE = re.compile(r"(?<=[A-Za-z])(?<![0-9.][eE])([23])(?![0-9])")
_SUPERSCRIPT_RE = re.compile(r"([⁰¹²³⁴⁵⁶⁷⁸⁹⁻⁺]+)")
_SUP_TRANS = str.maketrans("⁰¹²³⁴⁵⁶⁷⁸⁹⁻⁺", "0123456789-+")


def _preprocess_unit(unit: str) -> str:
    # Superscripts FIRST: NFKC folds '³' to a plain '3', so '10³ psi' became
    # '103 psi' (a silent 10x error on the standard US spelling of ksi) and
    # the later '³' replace was dead code.
    u = _SUPERSCRIPT_RE.sub(lambda m: "**" + m.group(1).translate(_SUP_TRANS), unit.strip())
    u = unicodedata.normalize("NFKC", u)
    u = u.replace("·", "*").replace("−", "-")
    u = u.replace("µ", "u").replace("μ", "u")
    u = u.replace("^", "**")
    u = u.replace("g/cc", "g/cm**3")
    # Any letter immediately followed by 2/3 is a power: cm3, mm2, m3, in2, ft3…
    # (was a hardcoded list that missed 'mm2' — the standard European MPa
    # spelling 'N/mm2' failed to parse and landed in unit_review).
    u = _UNIT_POWER_RE.sub(r"**\1", u)
    return u


_TEMP_UNITS = {
    "": "degC", "c": "degC", "degc": "degC", "°c": "degC", "celsius": "degC",
    "k": "kelvin", "kelvin": "kelvin",
    "f": "degF", "degf": "degF", "°f": "degF", "fahrenheit": "degF",
}


def canonicalize(prop: Property) -> tuple[str, Optional[float], Optional[str]]:
    """Return (unit_canonical, value_si, problem).

    `problem` is None on success, or a short reason ("unit_review:...") when the
    unit is missing/dimensionally wrong for the property family.
    """
    fam = _match_family(prop)
    rep = _representative_value(prop)
    if fam is None or fam.kind == "passthrough":
        # No known family (or a shield family): pass the unit through, no SI
        # conversion, no check.
        return (prop.unit, None, None)
    if rep is None:
        return (fam.display, None, None)

    if fam.kind == "raw":
        # Never apply si_factor blindly: CTE '2.3e-5 1/K' is 23 ppm/°C, not
        # 2.3e-11; elongation '0.024' (a strain fraction) is 2.4 %, not 0.024 %.
        factor = _raw_factor(fam, prop.unit, rep, prop.value_raw)
        if factor is None:
            return (fam.display, None, f"unit_review:unexpected_unit:{prop.unit}!~{fam.display}")
        return (fam.display, rep * factor * fam.si_factor, None)

    if _UREG is None:  # pragma: no cover
        return (fam.display, None, None)

    try:
        if fam.kind == "temperature":
            key = unicodedata.normalize("NFKC", prop.unit).strip().lower()
            src = _TEMP_UNITS.get(key)
            if src is None:
                return (fam.display, None, f"unit_review:bad_temp_unit:{prop.unit}")
            q = _UREG.Quantity(rep, src)
            value_si = q.to("kelvin").magnitude
            return (fam.display, value_si, None)

        # physical (pressure, density, ...)
        pre = _preprocess_unit(prop.unit)
        if not pre:
            return (fam.display, None, "unit_review:missing_unit")
        q = rep * _UREG(pre)
        target = _UREG(fam.canonical)
        if q.dimensionality != target.dimensionality:
            return (fam.display, None,
                    f"unit_review:dim_mismatch:{prop.unit}!~{fam.display}")
        value_si = q.to_base_units().magnitude
        return (fam.display, value_si, None)
    except Exception as exc:  # pint parse failure, undefined unit, etc.
        return (fam.display, None, f"unit_review:unparseable:{prop.unit}:{exc}")


def _canonical_value(prop: Property, fam: _Family) -> Optional[float]:
    """Representative value expressed in the family's `display` unit (for range check)."""
    rep = _representative_value(prop)
    if rep is None:
        return None
    if fam.kind == "passthrough":
        return None
    if fam.kind == "raw":
        factor = _raw_factor(fam, prop.unit, rep, prop.value_raw)
        return None if factor is None else rep * factor
    if _UREG is None:  # pragma: no cover
        return rep
    try:
        if fam.kind == "temperature":
            key = unicodedata.normalize("NFKC", prop.unit).strip().lower()
            src = _TEMP_UNITS.get(key)
            if src is None:
                return None
            return _UREG.Quantity(rep, src).to(fam.canonical).magnitude
        pre = _preprocess_unit(prop.unit)
        if not pre:
            return None
        q = rep * _UREG(pre)
        if q.dimensionality != _UREG(fam.canonical).dimensionality:
            return None
        return q.to(fam.canonical).magnitude
    except Exception:
        return None


def plausibility_problem(prop: Property) -> Optional[str]:
    """Range-check the value *after* unit conversion. Returns reason or None."""
    fam = _match_family(prop)
    if fam is None or fam.kind == "passthrough":
        return None
    cval = _canonical_value(prop, fam)
    if cval is None:
        return None  # couldn't convert -> handled as unit_review elsewhere
    if not (fam.lo <= cval <= fam.hi):
        return f"out_of_range[{fam.lo},{fam.hi}{fam.display}]:{cval:.4g}"
    return None


_PLACEHOLDER_VALUES = {"", "n/a", "na", "-", "--", "n.a.", "none", "tbd", "…", "..."}


def _empty_value(prop: Property) -> bool:
    # Fold the typographic dashes a table cell is actually printed with ('–',
    # '—', '−') before the lookup — a verbatim '–' cell used to pass as a
    # value and even ground (its folded '-' matched any hyphen on the page).
    v = prop.value_raw.strip().lower()
    v = v.replace("–", "-").replace("—", "-").replace("−", "-")
    return v in _PLACEHOLDER_VALUES


# ---------------------------------------------------------------------------
# Text grounding (Task 1)
# ---------------------------------------------------------------------------

_SUPERSCRIPT_CHARS_RE = re.compile(r"[⁰¹²³⁴⁵⁶⁷⁸⁹⁺⁻₀₁₂₃₄₅₆₇₈₉]+")


def _normalize_text(s: str) -> str:
    # Superscript/subscript digits are footnote markers or degree signs glued
    # to a value ('776¹', '0⁰/45⁰/90⁰'); NFKC would fold them into the number
    # ('7761'). Replace them with a space BEFORE normalizing.
    s = _SUPERSCRIPT_CHARS_RE.sub(" ", s)
    s = unicodedata.normalize("NFKC", s)
    s = s.replace("–", "-").replace("—", "-").replace("−", "-")
    s = re.sub(r"\s+", " ", s)
    return s.lower().strip()


_NUMERIC_NEEDLE_RE = re.compile(r"^[-+~<>=≤≥≈±.,\d\s/e]+$", re.IGNORECASE)


def _grounded(needle: str, haystacks: list[str]) -> bool:
    """True if `needle` occurs in any (already-normalized) haystack.

    Purely numeric needles are matched on **digit boundaries**: a value of "3"
    must not be "verified" by the "3" inside "ISO 527-3" or "23 °C", "1.2"
    must not match "11.25", and "200" must not match "1,200". Text needles
    (source_quote chunks) keep plain substring matching. Haystacks are
    expected to be `_normalize_text` output.
    """
    n = _normalize_text(needle)
    if not n:
        return False
    if _NUMERIC_NEEDLE_RE.match(n):
        # Strip qualifiers/whitespace so "<= -18.0" grounds on "-18.0" (the
        # sign is part of the number; the qualifier may be typeset elsewhere).
        core = re.sub(r"^[~<>=≤≥≈±\s]+", "", n).strip()
        if not core or not re.search(r"\d", core):
            return False   # '-', '.', 'e', '/' alone can never ground
        # Tighten the needle's own range dash ('70 - 75' -> '70-75'); the
        # haystack side is handled by the tolerant body pattern below.
        core = re.sub(r"(?<=\d) ?- ?(?=\d)", "-", core)
        # Left guard: no digit/dot/thousands-comma, and — for an unsigned
        # needle — no '-' either, so '3' cannot match the tail of 'ISO 527-3'.
        # A signed needle ('-18.0') legitimately starts with the '-'.
        left = r"(?<![\d.\-])(?<!\d,)" if core[0] not in "+-" else r"(?<![\d.])(?<!\d,)"
        # Right guard: no digit, no '.digit' (11.25), no ',digit' (1,200) — a
        # sentence-ending '.' or a list ', ' is a boundary.
        right = r"(?!\d|\.\d|,\d)"
        # Inside the needle, a range/minus dash may be typeset with spaces on
        # the page ('70 – 75', '2818 -0.46'); tolerate optional spaces around
        # a digit-dash-digit instead of rewriting the haystack, so a value
        # after a placeholder dash ('– 2250') still grounds.
        body = re.sub(r"(?<=\d)\\-(?=\d)", r"\\s?-\\s?", re.escape(core))
        pat = re.compile(left + body + right)
        return any(pat.search(h) for h in haystacks)
    return any(n in h for h in haystacks)


def verify_against_text(extraction: Extraction, page_texts: list[str]) -> Extraction:
    """Ground each property's value in the PDF text (Task 1).

    For every property: if `value_raw` (or a chunk of `source_quote`) appears on
    the cited page (falling back to any page), mark `status='ok'`; otherwise
    `status='unverified'`. Also folds in empty-value, unit, and range checks,
    using this precedence:
        empty_value > unverified > unit_review > out_of_range > ok
    """
    if not page_texts:
        norm_pages: list[str] = []
    else:
        norm_pages = [_normalize_text(t) for t in page_texts]

    for material in extraction.materials:
        for prop in material.properties:
            # always compute canonicalization so rows carry unit_canonical/value_si
            unit_canonical, value_si, unit_problem = canonicalize(prop)
            prop.unit_canonical = unit_canonical
            prop.value_si = value_si

            reasons: list[str] = []
            status = "ok"

            if _empty_value(prop):
                status = "empty_value"
                reasons.append("empty_value")
            else:
                # grounding
                if norm_pages:
                    page_idx = (prop.page - 1) if prop.page else None
                    page_in_range = page_idx is not None and 0 <= page_idx < len(norm_pages)
                    cited = [norm_pages[page_idx]] if page_in_range else []
                    if page_idx is not None and not page_in_range:
                        # A cited page that doesn't exist is a wrong page too:
                        # ground anywhere, but say so.
                        found = _grounded(prop.value_raw, norm_pages)
                        if found:
                            reasons.append("grounded_off_page")
                    else:
                        found = _grounded(prop.value_raw, cited or norm_pages)
                    if not found and cited:
                        # Fallback: any page. Still grounded (the number IS in
                        # the PDF) but the cited page was wrong — record it as
                        # a soft reason so reviewers can see the weaker chain.
                        found = _grounded(prop.value_raw, norm_pages)
                        if found:
                            reasons.append("grounded_off_page")
                    if not found and prop.source_quote:
                        chunk = prop.source_quote[:40]
                        found = _grounded(chunk, norm_pages)
                        if found:
                            reasons.append("grounded_via_quote")
                    if not found:
                        status = "unverified"
                        reasons.append("value_not_in_pdf_text")
                # unit problem
                if status == "ok" and unit_problem:
                    status = "unit_review"
                    reasons.append(unit_problem)
                # plausibility (only meaningful once unit is sane)
                if status == "ok":
                    rng = plausibility_problem(prop)
                    if rng:
                        status = "out_of_range"
                        reasons.append(rng)

            prop.status = status
            prop.flag_reason = "; ".join(reasons)
    verify_processes(extraction, norm_pages)
    return extraction


# How much of the route's quote must be found. Like the value fallback above,
# a leading chunk: PDF text layers break long sentences across lines/columns.
_PROCESS_QUOTE_CHUNK = 40


def verify_processes(extraction: Extraction, norm_pages: list[str]) -> None:
    """Ground each processing route in the PDF text (sets ``route.status``).

    Deterministic, like the value check: the quote that names the route must
    be in the text (cited page first, then anywhere), and every number in
    ``process_conditions`` must occur in the text on digit boundaries. The
    route is context for a value, so a failure here never changes a row's own
    ``status`` — it is recorded in ``process_status`` and the route stays
    attached, visibly unverified, instead of being dropped.
    """
    for route in getattr(extraction, "processes", None) or []:
        if not norm_pages:
            route.status = "unchecked"
            continue
        chunk = (route.source_quote or "")[:_PROCESS_QUOTE_CHUNK]
        if not chunk.strip():
            route.status = "ungrounded"
            continue
        page_idx = (route.page - 1) if route.page else None
        on_page = (page_idx is not None and 0 <= page_idx < len(norm_pages)
                   and _grounded(chunk, [norm_pages[page_idx]]))
        if on_page:
            status = "grounded"
        elif _grounded(chunk, norm_pages):
            status = "grounded_off_page"
        else:
            route.status = "ungrounded"
            continue
        for m in _NUM_RE.finditer(route.process_conditions or ""):
            # Unsigned: in 'cooling rate -5', '5-10 °C/min' or 'min-1' the
            # dash is a typographic choice, not part of what must be found.
            tok = m.group(0).lstrip("+-")
            if tok and not _grounded(tok, norm_pages):
                status = "conditions_unverified"
                break
        route.status = status


# ---------------------------------------------------------------------------
# Classification (Task 5)
# ---------------------------------------------------------------------------

COMPOSITE_KEYWORDS = (
    "composite", "laminate", "reinforced", "prepreg",
    "cf/", "gf/", "fiber-reinforced", "fibre-reinforced",
    "ud ", "uni-directional", "unidirectional", "woven",
    "glass-filled", "glass filled", "carbon-filled",
)
FIBER_KEYWORDS = ("fiber", "fibre", "yarn", "tow", "roving", "filament")


def classify_material(material: Material) -> str:
    """Return 'Polymer' | 'Fiber' | 'Composite' (Task 5).

    Primary signal is the model's `material_class` enum; a deterministic
    keyword fallback (using sections + Vf presence) only runs when the field is
    missing or invalid. The old haystack bug (repeating material_name N times,
    never reading section) is gone.
    """
    field = (material.material_class or "").strip().capitalize()
    if field in MATERIAL_CLASS_ENUM:
        return field

    # --- deterministic fallback ---
    # A reported fiber volume fraction is a strong composite signal.
    if material.fiber_volume_fraction.strip():
        return "Composite"

    haystack = " ".join(
        [
            material.material_name or "",
            material.material_abbreviation or "",
            material.trade_grade or "",
            material.matrix or "",
            material.fiber or "",
        ]
        + [p.section or "" for p in material.properties]
        + [p.property_name or "" for p in material.properties]
    ).lower()

    if any(k in haystack for k in COMPOSITE_KEYWORDS):
        return "Composite"
    if any(k in haystack for k in FIBER_KEYWORDS):
        return "Fiber"
    return "Polymer"


def _autoabbr(name: str) -> str:
    if not name:
        return "UNKNOWN"
    parts = [p[0] for p in name.split() if p and p[0].isalpha()]
    return "".join(parts).upper() or name[:6].upper()


def material_key(material: Material) -> str:
    """Stable material identity (Task 6): normalized name, plus the trade grade.

    ``<name>`` when no trade grade is known, else ``<name>|<trade_grade>``. The
    grade is part of the identity: a datasheet describing PEEK 150G and PEEK
    450G yields two materials named "PEEK", and without the grade in the key
    any property both grades share (density, Tg, ...) dedup-collapsed into
    one row and the other was silently dropped. Falls back to the
    abbreviation when the name is empty.
    """
    name = (material.material_name or "").strip().lower()
    name = re.sub(r"\s+", " ", name)
    if not name:
        name = (material.material_abbreviation or "unknown").strip().lower()
    grade = re.sub(r"\s+", " ", (material.trade_grade or "").strip().lower())
    if grade and grade != name and grade not in name:
        return f"{name}|{grade}"
    return name


# ---------------------------------------------------------------------------
# Flatten to DB rows (Tasks 2/5/6)
# ---------------------------------------------------------------------------


def to_rows(extraction: Extraction, source_pdf: str, source_sha1: str) -> list[PropertyRow]:
    """Flatten an Extraction into PropertyRows, iterating materials x properties."""
    rows: list[PropertyRow] = []
    routes = {r.process_id: r for r in (getattr(extraction, "processes", None) or [])}
    for material in extraction.materials:
        mclass = classify_material(material)
        mkey = material_key(material)
        abbr = material.material_abbreviation or _autoabbr(material.material_name)
        for prop in material.properties:
            # an id the document-level list does not define resolves to no route
            route = routes.get(getattr(prop, "process_id", "") or "")
            rows.append(
                PropertyRow(
                    material_name=material.material_name,
                    material_abbreviation=abbr,
                    material_key=mkey,
                    material_class=mclass,
                    section=prop.section,
                    property_name=prop.property_name,
                    value=prop.value_raw,  # legacy column == value_raw (back-compat)
                    unit=prop.unit,
                    english="",  # legacy alt-units column; model no longer emits it
                    test_condition=prop.test_condition,
                    comments=prop.comments,
                    trade_grade=material.trade_grade,
                    manufacturer=material.manufacturer,
                    matrix=material.matrix,
                    fiber=material.fiber,
                    fiber_volume_fraction=material.fiber_volume_fraction,
                    value_raw=prop.value_raw,
                    value_num=prop.value_num,
                    value_min=prop.value_min,
                    value_max=prop.value_max,
                    qualifier=prop.qualifier,
                    unit_canonical=prop.unit_canonical,
                    value_si=prop.value_si,
                    source_pdf=source_pdf,
                    source_sha1=source_sha1,
                    page=prop.page,
                    source_quote=prop.source_quote,
                    status=prop.status,
                    flag_reason=prop.flag_reason,
                    model=extraction.model,
                    prompt_version=extraction.prompt_version,
                    process_type=route.process_type if route else "",
                    process_name=route.process_name if route else "",
                    process_conditions=route.process_conditions if route else "",
                    process_quote=route.source_quote if route else "",
                    process_page=route.page if route else None,
                    process_status=route.status if route else "",
                )
            )
    return rows