๐Ÿ“ฆ EqualifyEverything / equalify-iris

๐Ÿ“„ extraction.ts ยท 6263 lines
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
1039
1040
1041
1042
1043
1044
1045
1046
1047
1048
1049
1050
1051
1052
1053
1054
1055
1056
1057
1058
1059
1060
1061
1062
1063
1064
1065
1066
1067
1068
1069
1070
1071
1072
1073
1074
1075
1076
1077
1078
1079
1080
1081
1082
1083
1084
1085
1086
1087
1088
1089
1090
1091
1092
1093
1094
1095
1096
1097
1098
1099
1100
1101
1102
1103
1104
1105
1106
1107
1108
1109
1110
1111
1112
1113
1114
1115
1116
1117
1118
1119
1120
1121
1122
1123
1124
1125
1126
1127
1128
1129
1130
1131
1132
1133
1134
1135
1136
1137
1138
1139
1140
1141
1142
1143
1144
1145
1146
1147
1148
1149
1150
1151
1152
1153
1154
1155
1156
1157
1158
1159
1160
1161
1162
1163
1164
1165
1166
1167
1168
1169
1170
1171
1172
1173
1174
1175
1176
1177
1178
1179
1180
1181
1182
1183
1184
1185
1186
1187
1188
1189
1190
1191
1192
1193
1194
1195
1196
1197
1198
1199
1200
1201
1202
1203
1204
1205
1206
1207
1208
1209
1210
1211
1212
1213
1214
1215
1216
1217
1218
1219
1220
1221
1222
1223
1224
1225
1226
1227
1228
1229
1230
1231
1232
1233
1234
1235
1236
1237
1238
1239
1240
1241
1242
1243
1244
1245
1246
1247
1248
1249
1250
1251
1252
1253
1254
1255
1256
1257
1258
1259
1260
1261
1262
1263
1264
1265
1266
1267
1268
1269
1270
1271
1272
1273
1274
1275
1276
1277
1278
1279
1280
1281
1282
1283
1284
1285
1286
1287
1288
1289
1290
1291
1292
1293
1294
1295
1296
1297
1298
1299
1300
1301
1302
1303
1304
1305
1306
1307
1308
1309
1310
1311
1312
1313
1314
1315
1316
1317
1318
1319
1320
1321
1322
1323
1324
1325
1326
1327
1328
1329
1330
1331
1332
1333
1334
1335
1336
1337
1338
1339
1340
1341
1342
1343
1344
1345
1346
1347
1348
1349
1350
1351
1352
1353
1354
1355
1356
1357
1358
1359
1360
1361
1362
1363
1364
1365
1366
1367
1368
1369
1370
1371
1372
1373
1374
1375
1376
1377
1378
1379
1380
1381
1382
1383
1384
1385
1386
1387
1388
1389
1390
1391
1392
1393
1394
1395
1396
1397
1398
1399
1400
1401
1402
1403
1404
1405
1406
1407
1408
1409
1410
1411
1412
1413
1414
1415
1416
1417
1418
1419
1420
1421
1422
1423
1424
1425
1426
1427
1428
1429
1430
1431
1432
1433
1434
1435
1436
1437
1438
1439
1440
1441
1442
1443
1444
1445
1446
1447
1448
1449
1450
1451
1452
1453
1454
1455
1456
1457
1458
1459
1460
1461
1462
1463
1464
1465
1466
1467
1468
1469
1470
1471
1472
1473
1474
1475
1476
1477
1478
1479
1480
1481
1482
1483
1484
1485
1486
1487
1488
1489
1490
1491
1492
1493
1494
1495
1496
1497
1498
1499
1500
1501
1502
1503
1504
1505
1506
1507
1508
1509
1510
1511
1512
1513
1514
1515
1516
1517
1518
1519
1520
1521
1522
1523
1524
1525
1526
1527
1528
1529
1530
1531
1532
1533
1534
1535
1536
1537
1538
1539
1540
1541
1542
1543
1544
1545
1546
1547
1548
1549
1550
1551
1552
1553
1554
1555
1556
1557
1558
1559
1560
1561
1562
1563
1564
1565
1566
1567
1568
1569
1570
1571
1572
1573
1574
1575
1576
1577
1578
1579
1580
1581
1582
1583
1584
1585
1586
1587
1588
1589
1590
1591
1592
1593
1594
1595
1596
1597
1598
1599
1600
1601
1602
1603
1604
1605
1606
1607
1608
1609
1610
1611
1612
1613
1614
1615
1616
1617
1618
1619
1620
1621
1622
1623
1624
1625
1626
1627
1628
1629
1630
1631
1632
1633
1634
1635
1636
1637
1638
1639
1640
1641
1642
1643
1644
1645
1646
1647
1648
1649
1650
1651
1652
1653
1654
1655
1656
1657
1658
1659
1660
1661
1662
1663
1664
1665
1666
1667
1668
1669
1670
1671
1672
1673
1674
1675
1676
1677
1678
1679
1680
1681
1682
1683
1684
1685
1686
1687
1688
1689
1690
1691
1692
1693
1694
1695
1696
1697
1698
1699
1700
1701
1702
1703
1704
1705
1706
1707
1708
1709
1710
1711
1712
1713
1714
1715
1716
1717
1718
1719
1720
1721
1722
1723
1724
1725
1726
1727
1728
1729
1730
1731
1732
1733
1734
1735
1736
1737
1738
1739
1740
1741
1742
1743
1744
1745
1746
1747
1748
1749
1750
1751
1752
1753
1754
1755
1756
1757
1758
1759
1760
1761
1762
1763
1764
1765
1766
1767
1768
1769
1770
1771
1772
1773
1774
1775
1776
1777
1778
1779
1780
1781
1782
1783
1784
1785
1786
1787
1788
1789
1790
1791
1792
1793
1794
1795
1796
1797
1798
1799
1800
1801
1802
1803
1804
1805
1806
1807
1808
1809
1810
1811
1812
1813
1814
1815
1816
1817
1818
1819
1820
1821
1822
1823
1824
1825
1826
1827
1828
1829
1830
1831
1832
1833
1834
1835
1836
1837
1838
1839
1840
1841
1842
1843
1844
1845
1846
1847
1848
1849
1850
1851
1852
1853
1854
1855
1856
1857
1858
1859
1860
1861
1862
1863
1864
1865
1866
1867
1868
1869
1870
1871
1872
1873
1874
1875
1876
1877
1878
1879
1880
1881
1882
1883
1884
1885
1886
1887
1888
1889
1890
1891
1892
1893
1894
1895
1896
1897
1898
1899
1900
1901
1902
1903
1904
1905
1906
1907
1908
1909
1910
1911
1912
1913
1914
1915
1916
1917
1918
1919
1920
1921
1922
1923
1924
1925
1926
1927
1928
1929
1930
1931
1932
1933
1934
1935
1936
1937
1938
1939
1940
1941
1942
1943
1944
1945
1946
1947
1948
1949
1950
1951
1952
1953
1954
1955
1956
1957
1958
1959
1960
1961
1962
1963
1964
1965
1966
1967
1968
1969
1970
1971
1972
1973
1974
1975
1976
1977
1978
1979
1980
1981
1982
1983
1984
1985
1986
1987
1988
1989
1990
1991
1992
1993
1994
1995
1996
1997
1998
1999
2000
2001
2002
2003
2004
2005
2006
2007
2008
2009
2010
2011
2012
2013
2014
2015
2016
2017
2018
2019
2020
2021
2022
2023
2024
2025
2026
2027
2028
2029
2030
2031
2032
2033
2034
2035
2036
2037
2038
2039
2040
2041
2042
2043
2044
2045
2046
2047
2048
2049
2050
2051
2052
2053
2054
2055
2056
2057
2058
2059
2060
2061
2062
2063
2064
2065
2066
2067
2068
2069
2070
2071
2072
2073
2074
2075
2076
2077
2078
2079
2080
2081
2082
2083
2084
2085
2086
2087
2088
2089
2090
2091
2092
2093
2094
2095
2096
2097
2098
2099
2100
2101
2102
2103
2104
2105
2106
2107
2108
2109
2110
2111
2112
2113
2114
2115
2116
2117
2118
2119
2120
2121
2122
2123
2124
2125
2126
2127
2128
2129
2130
2131
2132
2133
2134
2135
2136
2137
2138
2139
2140
2141
2142
2143
2144
2145
2146
2147
2148
2149
2150
2151
2152
2153
2154
2155
2156
2157
2158
2159
2160
2161
2162
2163
2164
2165
2166
2167
2168
2169
2170
2171
2172
2173
2174
2175
2176
2177
2178
2179
2180
2181
2182
2183
2184
2185
2186
2187
2188
2189
2190
2191
2192
2193
2194
2195
2196
2197
2198
2199
2200
2201
2202
2203
2204
2205
2206
2207
2208
2209
2210
2211
2212
2213
2214
2215
2216
2217
2218
2219
2220
2221
2222
2223
2224
2225
2226
2227
2228
2229
2230
2231
2232
2233
2234
2235
2236
2237
2238
2239
2240
2241
2242
2243
2244
2245
2246
2247
2248
2249
2250
2251
2252
2253
2254
2255
2256
2257
2258
2259
2260
2261
2262
2263
2264
2265
2266
2267
2268
2269
2270
2271
2272
2273
2274
2275
2276
2277
2278
2279
2280
2281
2282
2283
2284
2285
2286
2287
2288
2289
2290
2291
2292
2293
2294
2295
2296
2297
2298
2299
2300
2301
2302
2303
2304
2305
2306
2307
2308
2309
2310
2311
2312
2313
2314
2315
2316
2317
2318
2319
2320
2321
2322
2323
2324
2325
2326
2327
2328
2329
2330
2331
2332
2333
2334
2335
2336
2337
2338
2339
2340
2341
2342
2343
2344
2345
2346
2347
2348
2349
2350
2351
2352
2353
2354
2355
2356
2357
2358
2359
2360
2361
2362
2363
2364
2365
2366
2367
2368
2369
2370
2371
2372
2373
2374
2375
2376
2377
2378
2379
2380
2381
2382
2383
2384
2385
2386
2387
2388
2389
2390
2391
2392
2393
2394
2395
2396
2397
2398
2399
2400
2401
2402
2403
2404
2405
2406
2407
2408
2409
2410
2411
2412
2413
2414
2415
2416
2417
2418
2419
2420
2421
2422
2423
2424
2425
2426
2427
2428
2429
2430
2431
2432
2433
2434
2435
2436
2437
2438
2439
2440
2441
2442
2443
2444
2445
2446
2447
2448
2449
2450
2451
2452
2453
2454
2455
2456
2457
2458
2459
2460
2461
2462
2463
2464
2465
2466
2467
2468
2469
2470
2471
2472
2473
2474
2475
2476
2477
2478
2479
2480
2481
2482
2483
2484
2485
2486
2487
2488
2489
2490
2491
2492
2493
2494
2495
2496
2497
2498
2499
2500
2501
2502
2503
2504
2505
2506
2507
2508
2509
2510
2511
2512
2513
2514
2515
2516
2517
2518
2519
2520
2521
2522
2523
2524
2525
2526
2527
2528
2529
2530
2531
2532
2533
2534
2535
2536
2537
2538
2539
2540
2541
2542
2543
2544
2545
2546
2547
2548
2549
2550
2551
2552
2553
2554
2555
2556
2557
2558
2559
2560
2561
2562
2563
2564
2565
2566
2567
2568
2569
2570
2571
2572
2573
2574
2575
2576
2577
2578
2579
2580
2581
2582
2583
2584
2585
2586
2587
2588
2589
2590
2591
2592
2593
2594
2595
2596
2597
2598
2599
2600
2601
2602
2603
2604
2605
2606
2607
2608
2609
2610
2611
2612
2613
2614
2615
2616
2617
2618
2619
2620
2621
2622
2623
2624
2625
2626
2627
2628
2629
2630
2631
2632
2633
2634
2635
2636
2637
2638
2639
2640
2641
2642
2643
2644
2645
2646
2647
2648
2649
2650
2651
2652
2653
2654
2655
2656
2657
2658
2659
2660
2661
2662
2663
2664
2665
2666
2667
2668
2669
2670
2671
2672
2673
2674
2675
2676
2677
2678
2679
2680
2681
2682
2683
2684
2685
2686
2687
2688
2689
2690
2691
2692
2693
2694
2695
2696
2697
2698
2699
2700
2701
2702
2703
2704
2705
2706
2707
2708
2709
2710
2711
2712
2713
2714
2715
2716
2717
2718
2719
2720
2721
2722
2723
2724
2725
2726
2727
2728
2729
2730
2731
2732
2733
2734
2735
2736
2737
2738
2739
2740
2741
2742
2743
2744
2745
2746
2747
2748
2749
2750
2751
2752
2753
2754
2755
2756
2757
2758
2759
2760
2761
2762
2763
2764
2765
2766
2767
2768
2769
2770
2771
2772
2773
2774
2775
2776
2777
2778
2779
2780
2781
2782
2783
2784
2785
2786
2787
2788
2789
2790
2791
2792
2793
2794
2795
2796
2797
2798
2799
2800
2801
2802
2803
2804
2805
2806
2807
2808
2809
2810
2811
2812
2813
2814
2815
2816
2817
2818
2819
2820
2821
2822
2823
2824
2825
2826
2827
2828
2829
2830
2831
2832
2833
2834
2835
2836
2837
2838
2839
2840
2841
2842
2843
2844
2845
2846
2847
2848
2849
2850
2851
2852
2853
2854
2855
2856
2857
2858
2859
2860
2861
2862
2863
2864
2865
2866
2867
2868
2869
2870
2871
2872
2873
2874
2875
2876
2877
2878
2879
2880
2881
2882
2883
2884
2885
2886
2887
2888
2889
2890
2891
2892
2893
2894
2895
2896
2897
2898
2899
2900
2901
2902
2903
2904
2905
2906
2907
2908
2909
2910
2911
2912
2913
2914
2915
2916
2917
2918
2919
2920
2921
2922
2923
2924
2925
2926
2927
2928
2929
2930
2931
2932
2933
2934
2935
2936
2937
2938
2939
2940
2941
2942
2943
2944
2945
2946
2947
2948
2949
2950
2951
2952
2953
2954
2955
2956
2957
2958
2959
2960
2961
2962
2963
2964
2965
2966
2967
2968
2969
2970
2971
2972
2973
2974
2975
2976
2977
2978
2979
2980
2981
2982
2983
2984
2985
2986
2987
2988
2989
2990
2991
2992
2993
2994
2995
2996
2997
2998
2999
3000
3001
3002
3003
3004
3005
3006
3007
3008
3009
3010
3011
3012
3013
3014
3015
3016
3017
3018
3019
3020
3021
3022
3023
3024
3025
3026
3027
3028
3029
3030
3031
3032
3033
3034
3035
3036
3037
3038
3039
3040
3041
3042
3043
3044
3045
3046
3047
3048
3049
3050
3051
3052
3053
3054
3055
3056
3057
3058
3059
3060
3061
3062
3063
3064
3065
3066
3067
3068
3069
3070
3071
3072
3073
3074
3075
3076
3077
3078
3079
3080
3081
3082
3083
3084
3085
3086
3087
3088
3089
3090
3091
3092
3093
3094
3095
3096
3097
3098
3099
3100
3101
3102
3103
3104
3105
3106
3107
3108
3109
3110
3111
3112
3113
3114
3115
3116
3117
3118
3119
3120
3121
3122
3123
3124
3125
3126
3127
3128
3129
3130
3131
3132
3133
3134
3135
3136
3137
3138
3139
3140
3141
3142
3143
3144
3145
3146
3147
3148
3149
3150
3151
3152
3153
3154
3155
3156
3157
3158
3159
3160
3161
3162
3163
3164
3165
3166
3167
3168
3169
3170
3171
3172
3173
3174
3175
3176
3177
3178
3179
3180
3181
3182
3183
3184
3185
3186
3187
3188
3189
3190
3191
3192
3193
3194
3195
3196
3197
3198
3199
3200
3201
3202
3203
3204
3205
3206
3207
3208
3209
3210
3211
3212
3213
3214
3215
3216
3217
3218
3219
3220
3221
3222
3223
3224
3225
3226
3227
3228
3229
3230
3231
3232
3233
3234
3235
3236
3237
3238
3239
3240
3241
3242
3243
3244
3245
3246
3247
3248
3249
3250
3251
3252
3253
3254
3255
3256
3257
3258
3259
3260
3261
3262
3263
3264
3265
3266
3267
3268
3269
3270
3271
3272
3273
3274
3275
3276
3277
3278
3279
3280
3281
3282
3283
3284
3285
3286
3287
3288
3289
3290
3291
3292
3293
3294
3295
3296
3297
3298
3299
3300
3301
3302
3303
3304
3305
3306
3307
3308
3309
3310
3311
3312
3313
3314
3315
3316
3317
3318
3319
3320
3321
3322
3323
3324
3325
3326
3327
3328
3329
3330
3331
3332
3333
3334
3335
3336
3337
3338
3339
3340
3341
3342
3343
3344
3345
3346
3347
3348
3349
3350
3351
3352
3353
3354
3355
3356
3357
3358
3359
3360
3361
3362
3363
3364
3365
3366
3367
3368
3369
3370
3371
3372
3373
3374
3375
3376
3377
3378
3379
3380
3381
3382
3383
3384
3385
3386
3387
3388
3389
3390
3391
3392
3393
3394
3395
3396
3397
3398
3399
3400
3401
3402
3403
3404
3405
3406
3407
3408
3409
3410
3411
3412
3413
3414
3415
3416
3417
3418
3419
3420
3421
3422
3423
3424
3425
3426
3427
3428
3429
3430
3431
3432
3433
3434
3435
3436
3437
3438
3439
3440
3441
3442
3443
3444
3445
3446
3447
3448
3449
3450
3451
3452
3453
3454
3455
3456
3457
3458
3459
3460
3461
3462
3463
3464
3465
3466
3467
3468
3469
3470
3471
3472
3473
3474
3475
3476
3477
3478
3479
3480
3481
3482
3483
3484
3485
3486
3487
3488
3489
3490
3491
3492
3493
3494
3495
3496
3497
3498
3499
3500
3501
3502
3503
3504
3505
3506
3507
3508
3509
3510
3511
3512
3513
3514
3515
3516
3517
3518
3519
3520
3521
3522
3523
3524
3525
3526
3527
3528
3529
3530
3531
3532
3533
3534
3535
3536
3537
3538
3539
3540
3541
3542
3543
3544
3545
3546
3547
3548
3549
3550
3551
3552
3553
3554
3555
3556
3557
3558
3559
3560
3561
3562
3563
3564
3565
3566
3567
3568
3569
3570
3571
3572
3573
3574
3575
3576
3577
3578
3579
3580
3581
3582
3583
3584
3585
3586
3587
3588
3589
3590
3591
3592
3593
3594
3595
3596
3597
3598
3599
3600
3601
3602
3603
3604
3605
3606
3607
3608
3609
3610
3611
3612
3613
3614
3615
3616
3617
3618
3619
3620
3621
3622
3623
3624
3625
3626
3627
3628
3629
3630
3631
3632
3633
3634
3635
3636
3637
3638
3639
3640
3641
3642
3643
3644
3645
3646
3647
3648
3649
3650
3651
3652
3653
3654
3655
3656
3657
3658
3659
3660
3661
3662
3663
3664
3665
3666
3667
3668
3669
3670
3671
3672
3673
3674
3675
3676
3677
3678
3679
3680
3681
3682
3683
3684
3685
3686
3687
3688
3689
3690
3691
3692
3693
3694
3695
3696
3697
3698
3699
3700
3701
3702
3703
3704
3705
3706
3707
3708
3709
3710
3711
3712
3713
3714
3715
3716
3717
3718
3719
3720
3721
3722
3723
3724
3725
3726
3727
3728
3729
3730
3731
3732
3733
3734
3735
3736
3737
3738
3739
3740
3741
3742
3743
3744
3745
3746
3747
3748
3749
3750
3751
3752
3753
3754
3755
3756
3757
3758
3759
3760
3761
3762
3763
3764
3765
3766
3767
3768
3769
3770
3771
3772
3773
3774
3775
3776
3777
3778
3779
3780
3781
3782
3783
3784
3785
3786
3787
3788
3789
3790
3791
3792
3793
3794
3795
3796
3797
3798
3799
3800
3801
3802
3803
3804
3805
3806
3807
3808
3809
3810
3811
3812
3813
3814
3815
3816
3817
3818
3819
3820
3821
3822
3823
3824
3825
3826
3827
3828
3829
3830
3831
3832
3833
3834
3835
3836
3837
3838
3839
3840
3841
3842
3843
3844
3845
3846
3847
3848
3849
3850
3851
3852
3853
3854
3855
3856
3857
3858
3859
3860
3861
3862
3863
3864
3865
3866
3867
3868
3869
3870
3871
3872
3873
3874
3875
3876
3877
3878
3879
3880
3881
3882
3883
3884
3885
3886
3887
3888
3889
3890
3891
3892
3893
3894
3895
3896
3897
3898
3899
3900
3901
3902
3903
3904
3905
3906
3907
3908
3909
3910
3911
3912
3913
3914
3915
3916
3917
3918
3919
3920
3921
3922
3923
3924
3925
3926
3927
3928
3929
3930
3931
3932
3933
3934
3935
3936
3937
3938
3939
3940
3941
3942
3943
3944
3945
3946
3947
3948
3949
3950
3951
3952
3953
3954
3955
3956
3957
3958
3959
3960
3961
3962
3963
3964
3965
3966
3967
3968
3969
3970
3971
3972
3973
3974
3975
3976
3977
3978
3979
3980
3981
3982
3983
3984
3985
3986
3987
3988
3989
3990
3991
3992
3993
3994
3995
3996
3997
3998
3999
4000
4001
4002
4003
4004
4005
4006
4007
4008
4009
4010
4011
4012
4013
4014
4015
4016
4017
4018
4019
4020
4021
4022
4023
4024
4025
4026
4027
4028
4029
4030
4031
4032
4033
4034
4035
4036
4037
4038
4039
4040
4041
4042
4043
4044
4045
4046
4047
4048
4049
4050
4051
4052
4053
4054
4055
4056
4057
4058
4059
4060
4061
4062
4063
4064
4065
4066
4067
4068
4069
4070
4071
4072
4073
4074
4075
4076
4077
4078
4079
4080
4081
4082
4083
4084
4085
4086
4087
4088
4089
4090
4091
4092
4093
4094
4095
4096
4097
4098
4099
4100
4101
4102
4103
4104
4105
4106
4107
4108
4109
4110
4111
4112
4113
4114
4115
4116
4117
4118
4119
4120
4121
4122
4123
4124
4125
4126
4127
4128
4129
4130
4131
4132
4133
4134
4135
4136
4137
4138
4139
4140
4141
4142
4143
4144
4145
4146
4147
4148
4149
4150
4151
4152
4153
4154
4155
4156
4157
4158
4159
4160
4161
4162
4163
4164
4165
4166
4167
4168
4169
4170
4171
4172
4173
4174
4175
4176
4177
4178
4179
4180
4181
4182
4183
4184
4185
4186
4187
4188
4189
4190
4191
4192
4193
4194
4195
4196
4197
4198
4199
4200
4201
4202
4203
4204
4205
4206
4207
4208
4209
4210
4211
4212
4213
4214
4215
4216
4217
4218
4219
4220
4221
4222
4223
4224
4225
4226
4227
4228
4229
4230
4231
4232
4233
4234
4235
4236
4237
4238
4239
4240
4241
4242
4243
4244
4245
4246
4247
4248
4249
4250
4251
4252
4253
4254
4255
4256
4257
4258
4259
4260
4261
4262
4263
4264
4265
4266
4267
4268
4269
4270
4271
4272
4273
4274
4275
4276
4277
4278
4279
4280
4281
4282
4283
4284
4285
4286
4287
4288
4289
4290
4291
4292
4293
4294
4295
4296
4297
4298
4299
4300
4301
4302
4303
4304
4305
4306
4307
4308
4309
4310
4311
4312
4313
4314
4315
4316
4317
4318
4319
4320
4321
4322
4323
4324
4325
4326
4327
4328
4329
4330
4331
4332
4333
4334
4335
4336
4337
4338
4339
4340
4341
4342
4343
4344
4345
4346
4347
4348
4349
4350
4351
4352
4353
4354
4355
4356
4357
4358
4359
4360
4361
4362
4363
4364
4365
4366
4367
4368
4369
4370
4371
4372
4373
4374
4375
4376
4377
4378
4379
4380
4381
4382
4383
4384
4385
4386
4387
4388
4389
4390
4391
4392
4393
4394
4395
4396
4397
4398
4399
4400
4401
4402
4403
4404
4405
4406
4407
4408
4409
4410
4411
4412
4413
4414
4415
4416
4417
4418
4419
4420
4421
4422
4423
4424
4425
4426
4427
4428
4429
4430
4431
4432
4433
4434
4435
4436
4437
4438
4439
4440
4441
4442
4443
4444
4445
4446
4447
4448
4449
4450
4451
4452
4453
4454
4455
4456
4457
4458
4459
4460
4461
4462
4463
4464
4465
4466
4467
4468
4469
4470
4471
4472
4473
4474
4475
4476
4477
4478
4479
4480
4481
4482
4483
4484
4485
4486
4487
4488
4489
4490
4491
4492
4493
4494
4495
4496
4497
4498
4499
4500
4501
4502
4503
4504
4505
4506
4507
4508
4509
4510
4511
4512
4513
4514
4515
4516
4517
4518
4519
4520
4521
4522
4523
4524
4525
4526
4527
4528
4529
4530
4531
4532
4533
4534
4535
4536
4537
4538
4539
4540
4541
4542
4543
4544
4545
4546
4547
4548
4549
4550
4551
4552
4553
4554
4555
4556
4557
4558
4559
4560
4561
4562
4563
4564
4565
4566
4567
4568
4569
4570
4571
4572
4573
4574
4575
4576
4577
4578
4579
4580
4581
4582
4583
4584
4585
4586
4587
4588
4589
4590
4591
4592
4593
4594
4595
4596
4597
4598
4599
4600
4601
4602
4603
4604
4605
4606
4607
4608
4609
4610
4611
4612
4613
4614
4615
4616
4617
4618
4619
4620
4621
4622
4623
4624
4625
4626
4627
4628
4629
4630
4631
4632
4633
4634
4635
4636
4637
4638
4639
4640
4641
4642
4643
4644
4645
4646
4647
4648
4649
4650
4651
4652
4653
4654
4655
4656
4657
4658
4659
4660
4661
4662
4663
4664
4665
4666
4667
4668
4669
4670
4671
4672
4673
4674
4675
4676
4677
4678
4679
4680
4681
4682
4683
4684
4685
4686
4687
4688
4689
4690
4691
4692
4693
4694
4695
4696
4697
4698
4699
4700
4701
4702
4703
4704
4705
4706
4707
4708
4709
4710
4711
4712
4713
4714
4715
4716
4717
4718
4719
4720
4721
4722
4723
4724
4725
4726
4727
4728
4729
4730
4731
4732
4733
4734
4735
4736
4737
4738
4739
4740
4741
4742
4743
4744
4745
4746
4747
4748
4749
4750
4751
4752
4753
4754
4755
4756
4757
4758
4759
4760
4761
4762
4763
4764
4765
4766
4767
4768
4769
4770
4771
4772
4773
4774
4775
4776
4777
4778
4779
4780
4781
4782
4783
4784
4785
4786
4787
4788
4789
4790
4791
4792
4793
4794
4795
4796
4797
4798
4799
4800
4801
4802
4803
4804
4805
4806
4807
4808
4809
4810
4811
4812
4813
4814
4815
4816
4817
4818
4819
4820
4821
4822
4823
4824
4825
4826
4827
4828
4829
4830
4831
4832
4833
4834
4835
4836
4837
4838
4839
4840
4841
4842
4843
4844
4845
4846
4847
4848
4849
4850
4851
4852
4853
4854
4855
4856
4857
4858
4859
4860
4861
4862
4863
4864
4865
4866
4867
4868
4869
4870
4871
4872
4873
4874
4875
4876
4877
4878
4879
4880
4881
4882
4883
4884
4885
4886
4887
4888
4889
4890
4891
4892
4893
4894
4895
4896
4897
4898
4899
4900
4901
4902
4903
4904
4905
4906
4907
4908
4909
4910
4911
4912
4913
4914
4915
4916
4917
4918
4919
4920
4921
4922
4923
4924
4925
4926
4927
4928
4929
4930
4931
4932
4933
4934
4935
4936
4937
4938
4939
4940
4941
4942
4943
4944
4945
4946
4947
4948
4949
4950
4951
4952
4953
4954
4955
4956
4957
4958
4959
4960
4961
4962
4963
4964
4965
4966
4967
4968
4969
4970
4971
4972
4973
4974
4975
4976
4977
4978
4979
4980
4981
4982
4983
4984
4985
4986
4987
4988
4989
4990
4991
4992
4993
4994
4995
4996
4997
4998
4999
5000
5001
5002
5003
5004
5005
5006
5007
5008
5009
5010
5011
5012
5013
5014
5015
5016
5017
5018
5019
5020
5021
5022
5023
5024
5025
5026
5027
5028
5029
5030
5031
5032
5033
5034
5035
5036
5037
5038
5039
5040
5041
5042
5043
5044
5045
5046
5047
5048
5049
5050
5051
5052
5053
5054
5055
5056
5057
5058
5059
5060
5061
5062
5063
5064
5065
5066
5067
5068
5069
5070
5071
5072
5073
5074
5075
5076
5077
5078
5079
5080
5081
5082
5083
5084
5085
5086
5087
5088
5089
5090
5091
5092
5093
5094
5095
5096
5097
5098
5099
5100
5101
5102
5103
5104
5105
5106
5107
5108
5109
5110
5111
5112
5113
5114
5115
5116
5117
5118
5119
5120
5121
5122
5123
5124
5125
5126
5127
5128
5129
5130
5131
5132
5133
5134
5135
5136
5137
5138
5139
5140
5141
5142
5143
5144
5145
5146
5147
5148
5149
5150
5151
5152
5153
5154
5155
5156
5157
5158
5159
5160
5161
5162
5163
5164
5165
5166
5167
5168
5169
5170
5171
5172
5173
5174
5175
5176
5177
5178
5179
5180
5181
5182
5183
5184
5185
5186
5187
5188
5189
5190
5191
5192
5193
5194
5195
5196
5197
5198
5199
5200
5201
5202
5203
5204
5205
5206
5207
5208
5209
5210
5211
5212
5213
5214
5215
5216
5217
5218
5219
5220
5221
5222
5223
5224
5225
5226
5227
5228
5229
5230
5231
5232
5233
5234
5235
5236
5237
5238
5239
5240
5241
5242
5243
5244
5245
5246
5247
5248
5249
5250
5251
5252
5253
5254
5255
5256
5257
5258
5259
5260
5261
5262
5263
5264
5265
5266
5267
5268
5269
5270
5271
5272
5273
5274
5275
5276
5277
5278
5279
5280
5281
5282
5283
5284
5285
5286
5287
5288
5289
5290
5291
5292
5293
5294
5295
5296
5297
5298
5299
5300
5301
5302
5303
5304
5305
5306
5307
5308
5309
5310
5311
5312
5313
5314
5315
5316
5317
5318
5319
5320
5321
5322
5323
5324
5325
5326
5327
5328
5329
5330
5331
5332
5333
5334
5335
5336
5337
5338
5339
5340
5341
5342
5343
5344
5345
5346
5347
5348
5349
5350
5351
5352
5353
5354
5355
5356
5357
5358
5359
5360
5361
5362
5363
5364
5365
5366
5367
5368
5369
5370
5371
5372
5373
5374
5375
5376
5377
5378
5379
5380
5381
5382
5383
5384
5385
5386
5387
5388
5389
5390
5391
5392
5393
5394
5395
5396
5397
5398
5399
5400
5401
5402
5403
5404
5405
5406
5407
5408
5409
5410
5411
5412
5413
5414
5415
5416
5417
5418
5419
5420
5421
5422
5423
5424
5425
5426
5427
5428
5429
5430
5431
5432
5433
5434
5435
5436
5437
5438
5439
5440
5441
5442
5443
5444
5445
5446
5447
5448
5449
5450
5451
5452
5453
5454
5455
5456
5457
5458
5459
5460
5461
5462
5463
5464
5465
5466
5467
5468
5469
5470
5471
5472
5473
5474
5475
5476
5477
5478
5479
5480
5481
5482
5483
5484
5485
5486
5487
5488
5489
5490
5491
5492
5493
5494
5495
5496
5497
5498
5499
5500
5501
5502
5503
5504
5505
5506
5507
5508
5509
5510
5511
5512
5513
5514
5515
5516
5517
5518
5519
5520
5521
5522
5523
5524
5525
5526
5527
5528
5529
5530
5531
5532
5533
5534
5535
5536
5537
5538
5539
5540
5541
5542
5543
5544
5545
5546
5547
5548
5549
5550
5551
5552
5553
5554
5555
5556
5557
5558
5559
5560
5561
5562
5563
5564
5565
5566
5567
5568
5569
5570
5571
5572
5573
5574
5575
5576
5577
5578
5579
5580
5581
5582
5583
5584
5585
5586
5587
5588
5589
5590
5591
5592
5593
5594
5595
5596
5597
5598
5599
5600
5601
5602
5603
5604
5605
5606
5607
5608
5609
5610
5611
5612
5613
5614
5615
5616
5617
5618
5619
5620
5621
5622
5623
5624
5625
5626
5627
5628
5629
5630
5631
5632
5633
5634
5635
5636
5637
5638
5639
5640
5641
5642
5643
5644
5645
5646
5647
5648
5649
5650
5651
5652
5653
5654
5655
5656
5657
5658
5659
5660
5661
5662
5663
5664
5665
5666
5667
5668
5669
5670
5671
5672
5673
5674
5675
5676
5677
5678
5679
5680
5681
5682
5683
5684
5685
5686
5687
5688
5689
5690
5691
5692
5693
5694
5695
5696
5697
5698
5699
5700
5701
5702
5703
5704
5705
5706
5707
5708
5709
5710
5711
5712
5713
5714
5715
5716
5717
5718
5719
5720
5721
5722
5723
5724
5725
5726
5727
5728
5729
5730
5731
5732
5733
5734
5735
5736
5737
5738
5739
5740
5741
5742
5743
5744
5745
5746
5747
5748
5749
5750
5751
5752
5753
5754
5755
5756
5757
5758
5759
5760
5761
5762
5763
5764
5765
5766
5767
5768
5769
5770
5771
5772
5773
5774
5775
5776
5777
5778
5779
5780
5781
5782
5783
5784
5785
5786
5787
5788
5789
5790
5791
5792
5793
5794
5795
5796
5797
5798
5799
5800
5801
5802
5803
5804
5805
5806
5807
5808
5809
5810
5811
5812
5813
5814
5815
5816
5817
5818
5819
5820
5821
5822
5823
5824
5825
5826
5827
5828
5829
5830
5831
5832
5833
5834
5835
5836
5837
5838
5839
5840
5841
5842
5843
5844
5845
5846
5847
5848
5849
5850
5851
5852
5853
5854
5855
5856
5857
5858
5859
5860
5861
5862
5863
5864
5865
5866
5867
5868
5869
5870
5871
5872
5873
5874
5875
5876
5877
5878
5879
5880
5881
5882
5883
5884
5885
5886
5887
5888
5889
5890
5891
5892
5893
5894
5895
5896
5897
5898
5899
5900
5901
5902
5903
5904
5905
5906
5907
5908
5909
5910
5911
5912
5913
5914
5915
5916
5917
5918
5919
5920
5921
5922
5923
5924
5925
5926
5927
5928
5929
5930
5931
5932
5933
5934
5935
5936
5937
5938
5939
5940
5941
5942
5943
5944
5945
5946
5947
5948
5949
5950
5951
5952
5953
5954
5955
5956
5957
5958
5959
5960
5961
5962
5963
5964
5965
5966
5967
5968
5969
5970
5971
5972
5973
5974
5975
5976
5977
5978
5979
5980
5981
5982
5983
5984
5985
5986
5987
5988
5989
5990
5991
5992
5993
5994
5995
5996
5997
5998
5999
6000
6001
6002
6003
6004
6005
6006
6007
6008
6009
6010
6011
6012
6013
6014
6015
6016
6017
6018
6019
6020
6021
6022
6023
6024
6025
6026
6027
6028
6029
6030
6031
6032
6033
6034
6035
6036
6037
6038
6039
6040
6041
6042
6043
6044
6045
6046
6047
6048
6049
6050
6051
6052
6053
6054
6055
6056
6057
6058
6059
6060
6061
6062
6063
6064
6065
6066
6067
6068
6069
6070
6071
6072
6073
6074
6075
6076
6077
6078
6079
6080
6081
6082
6083
6084
6085
6086
6087
6088
6089
6090
6091
6092
6093
6094
6095
6096
6097
6098
6099
6100
6101
6102
6103
6104
6105
6106
6107
6108
6109
6110
6111
6112
6113
6114
6115
6116
6117
6118
6119
6120
6121
6122
6123
6124
6125
6126
6127
6128
6129
6130
6131
6132
6133
6134
6135
6136
6137
6138
6139
6140
6141
6142
6143
6144
6145
6146
6147
6148
6149
6150
6151
6152
6153
6154
6155
6156
6157
6158
6159
6160
6161
6162
6163
6164
6165
6166
6167
6168
6169
6170
6171
6172
6173
6174
6175
6176
6177
6178
6179
6180
6181
6182
6183
6184
6185
6186
6187
6188
6189
6190
6191
6192
6193
6194
6195
6196
6197
6198
6199
6200
6201
6202
6203
6204
6205
6206
6207
6208
6209
6210
6211
6212
6213
6214
6215
6216
6217
6218
6219
6220
6221
6222
6223
6224
6225
6226
6227
6228
6229
6230
6231
6232
6233
6234
6235
6236
6237
6238
6239
6240
6241
6242
6243
6244
6245
6246
6247
6248
6249
6250
6251
6252
6253
6254
6255
6256
6257
6258
6259
6260
6261
6262
6263import { readdirSync, writeFileSync } from "node:fs";
import { join } from "node:path";
import { extractJson } from "../util/json.ts";
import { stripSoftHyphens, stripStyleAttributes, tightenDigitGroups } from "../util/html.ts";
import { mapWithConcurrency } from "../util/concurrency.ts";
import { loadAgent, type AgentSpec } from "../agents/loader.ts";
import { feedbackPreamble, loadImage, type InputImage, type PipelineContext } from "./context.ts";
import { ACCESSIBILITY_REQUIREMENTS } from "./accessibility.ts";
import { unjudgedVerdict, verifyAgentOutput, type VerifyVerdict } from "./feedback.ts";
import {
  captionClaims,
  carriesContent,
  changedAnything,
  claimRecheck,
  correctionEffect,
  destroyedPage,
  recheckSampler,
  type RecheckSampler,
} from "./correction.ts";
import { examplesForPrompt } from "./memory.ts";
import { altTexts, genericAltProblem, genericAlts } from "./alt.ts";
import { missingLinkProblem, missingLinks, pageLinkContext, unexpectedHrefs } from "./links.ts";
import { duplicateIdProblem, duplicateIds, idAudit } from "./anchors.ts";
import { splitWordAudit, splitWordContradictions, splitWordProblem } from "./hyphens.ts";
import { STANDARD as STANDARD_AGENTS, isStandardType, logicalType } from "./contribute.ts";
import { isTruncatedResponseError, replyExcerpt, TruncatedResponseError } from "../providers/types.ts";
import type { Fragment } from "./fragment.ts";

const PAGE_AGENT = "page";

// Single coherent extraction: one vision call converts the WHOLE page into one
// accessible-HTML fragment. This replaces fanning the page out to many
// content agents that each re-rendered it (which produced duplicated output for
// nested structures like forms).
//
// The nine standard content agents that fan-out used to call are no longer in the
// repo. They were not merely unused: `dispatchSpecialist` declines every
// name in STANDARD_AGENTS below before `loadAgent` is reached, only `page.md` is
// ever trained, and `runContribution` filters the same names โ€” so no run could
// reach them by any path. What survives is the part that pays for itself:
// specialists for content a whole-page pass genuinely handles worse (see
// `chartDataAgent.md`), dispatched by name and merged in.
//
// The prompt now lives in agents/page.md so the page agent is a first-class,
// loadable, trainable, contributable agent (verified at build time, trained from
// feedback). This DEFAULT is used only when that file is absent, so the service
// still runs against a bare checkout. It also asks the model to flag a page that
// would benefit from a dedicated specialist agent (collected as `suggestions`).
//
// It therefore duplicates agents/page.md's "## System prompt" + "## Output
// contract" on purpose, and cannot be replaced by reading that file โ€” the whole
// point is the file being missing. `test/page-prompt.test.ts` asserts the two
// agree (word-for-word, whitespace-insensitive), because the file is what every
// real deployment runs while this copy is exercised only by bare checkouts: edit
// one and nothing here would otherwise notice. Exported for that test.
export const DEFAULT_PAGE_PROMPT = `You convert an ENTIRE document page (provided as an image) into a single, coherent,
accessible HTML fragment that meets WCAG 2.2 AA. You see the whole page and produce ONE
faithful representation of it. NEVER duplicate content or render the same thing two ways
(for example, do not output both a <form> and a <table> for the same fields) โ€” choose the
single structure that best matches the source.

Output ONLY the body content (no <html>, <head>, <body> or <main> wrapper). Use the most appropriate
semantic structure for what the page actually is: headings in correct nesting order,
paragraphs, lists, tables with <caption>/<thead>/<th scope>, forms with
<label>/<fieldset>/<legend>, figures with <figcaption>, footnotes, etc. Transcribe visible
text faithfully and do not invent content: apart from the accessibility scaffolding the rules
below ask for by name โ€” alt text, a placeholder src for a graphic you cannot embed, a <caption>
the page does not print, an accessible name on a marker the page prints as a symbol, the โ†ฉ that
returns from a footnote, a note about irregular numbering held to what the page shows, the page's
own words used to tell two headings it labels alike apart, a [not legible] marker where the marks
on the page do not resolve into characters, a [page not fully transcribed] marker where you could
not return all of it โ€” every word you emit is a word on the page. If content is cut off at a page
edge, note it in the "log" field.

A word the page gets wrong is still a word on the page. A misspelling, a letter the type broke, a
word the compositor set twice โ€” necessarv where the sense wants "necessary", Statistcs in the
title of a report โ€” is transcribed exactly as printed, and the fact goes in the "log" field.
Repairing it is the same act as inventing content, and it is harder to catch than an invention:
the delivered document reads as something the page says, no later pass can tell a word was changed,
and a reader checking it against the paper finds the two disagreeing with nothing to say which of
them is the paper's. The helpful reading is the wrong one in both directions โ€” supplying the word you
expected where the page prints a defective one, and substituting a familiar word for the unfamiliar
one the page really prints, Governmental for a printed "Governments" or Midwestern for a printed
"Mideastern" โ€” and the second is worse, because it makes right text wrong. Where the printing is so
damaged that you cannot tell which characters it is, that is the [not legible] case below and not
this one.

Letter case is transcribed as the page sets it, with one printed device excepted, because that
device is not case at all. Small capitals are a typeface: the first letter stands at cap height and
the rest are capital forms at x-height, so a line set that way prints "Table 11.", "Chapter 1." or a
name like "Ecker-Racz" in title case however capital its letters look, and emitting TABLE, CHAPTER or
ECKER-RACZ adds emphasis the page does not carry โ€” a run of capitals is also what a screen reader may
announce letter by letter as an initialism. Full capitals are the other device, every letter at one
height with no x-height form among them, and there the case is not the text by itself: what it means
depends on what the capitals are doing, which the display-capitals rule below decides. The two heights are what
tell them apart, and a document commonly settles it itself โ€” where the same words are set both ways,
a chapter title in small capitals and the same chapter named in mixed case a few pages on, the
mixed-case setting is what the small capitals mean. Where you cannot compare the two heights โ€” a scan
too coarse to resolve them, or a line with no letter of each kind in it to hold against the other โ€”
neither device has been identified, and an unidentified device is transcribed exactly as the page sets
it with a note in the "log" field saying the case could not be decided. That is the same answer the
uncertain-reading rule above gives, and it leaves a line of capitals standing as printed rather than
retyped on a guess. Neither device is carried as markup, because nothing you can write conveys a
typeface: a style attribute does not make small capitals reach a reader as small capitals, a <span>
does not, and neither does retyping the line in a case the page did not set โ€” which is why writing
"Table 11." for a line set in small capitals is the transcription of that line and not a case change of
your own. Typography you cannot transcribe is a note for the "log" field.

Capitals the page sets for weight are transcribed in title case; capitals that are how a word is
spelled are transcribed as printed. Those are the two things a run of full capitals can be, and the
test is which of them the capitals carry โ€” the word's own spelling, or emphasis the page has added to
the line. ACIR, HEW and U.S. are spelled that way: they have no lower-case form anywhere, so Acir and
Hew are text the page prints in no sense at all, and retyping them is the corruption this section
exists to prevent arriving by way of the fix. A heading, a running title, a table's stub head, the
name of a division โ€” PART I over a part of the report, GENERAL PROVISIONS over a run of sections โ€” is
the other kind: its words are ordinary words, written in mixed case wherever they are not being
emphasised, so emit Part I and General Provisions and record the printed casing in the "log" field.
The reason to down-case rather than keep the ink is the one given above: a run of capitals is what a
screen reader may announce letter by letter, which is right for ACIR and turns PART into P-A-R-T, and
the emphasis cannot be carried instead, because a style attribute, a class and text-transform are all
prohibited below and nothing you can write makes a line louder. Where the two cannot be told apart โ€”
a short run that may be an initialism you do not know, a line whose words appear nowhere else on the
page to compare โ€” the answer is the one an unidentified device gets: exactly as the page sets it,
with a note in the "log" field saying the case could not be decided.

No styling reaches the output at all: no style attribute, no class, no <style> element, no event
handler. A style attribute carries nothing a reader hears โ€” it is not announced, it does not survive
being read aloud, and it is dropped by anything that reformats the document โ€” so every use of one
here is either presentation that was never content, or content put where no reader can reach it. The
second is the case to watch, because removing the attribute is not the whole of the fix: padding-left
on forty row headings is a table's row groups and its scope attributes written in ink instead of in
markup, and an empty <span> given a coloured background is a legend swatch that paints nothing and
announces nothing. Where the indentation, the shading or the ink is carrying information โ€” which rank
a row belongs to, which band a state falls in, what a key's entry marks โ€” that information goes into
the markup that says so: a <tbody> per group with <th scope="rowgroup">, or the ink described in
words by the key rule below.

Everything the page shows reaches your output. A long page, a table of forty rows, a page carrying
three tables and a sidebar โ€” all of it is emitted, and none of it is summarised, abbreviated, or
handed back in part because the rest is more of the same. Two things leave the page, by rule and
not by judgement: a symbol the page itself explains as a navigational device is kept out of the text
and recorded in the "log" field, and the number the page prints on itself is carried by the name of
the page-break marker rather than transcribed beside it. Both rules are below, and both are narrow.
Nothing else leaves.
Nothing downstream marks what is missing: the document is assembled from what you return, so a row, an item or a section you leave
out is simply not in the document any reader gets, and no later pass can tell it was ever there.
Length is not a reason to stop. If the page truly holds more than you can return, emit it in
reading order, make [page not fully transcribed] the last thing you emit, and say in the "log" field
what you left. The marker is the part that matters: "log" is not delivered as the document, so a
page that stops without one reads as complete to every reader and to every later pass, while one
that says where it ends can be finished.

Read the page before deciding any of it is unreadable. Low contrast, small type, a watermark over
text, a lightly printed caution, the labels inside a diagram, the figures in a table cell: each of
those takes a second look rather than a first glance, and text a reader could make out with effort
is text you transcribe. Where marks do not resolve into characters even then, write [not legible]
where that word or phrase stands, keep the element it belongs to around it โ€” the <li>, the <td>,
the <p> of the caution box โ€” so the structure of the page survives, and say in the "log" field
which region it was. Mark only what you could not read: a placeholder standing for a paragraph you
could mostly read costs a reader the part you had. And put nothing else in its place โ€” not a
paraphrase, not a caution of your own that suits the picture, not an editorial note ("manual
transcription required", "insufficient contrast", "see the original manual"). Those are words no
reader can check against the page, and notes about the transcription belong in the "log" field,
which is not part of the document.
Where you can read the marks but not the word, what you emit is a reading OF those marks: "d :5["
is not a word, and where the shapes allow "disc" and the sentence is about an inserted disc, disc
is what the page says. That is not licence to write what the sentence ought to say. A word whose
letters are not on the page is invented content however well it fits, and a number, a part code, a
measurement or a model name is never settled this way, because nothing around it can confirm the
reading โ€” those are the strings a reader will act on, so an uncertain one is marked, not mended.
Where no reading of the marks is one you would stand behind, the placeholder is the honest answer,
and the "log" field is where you say what you could see of it.

A landmark names a part of the document, and a page is not one. You are shown one page at a time,
but a page is a unit of printing rather than a unit of meaning: never wrap what you emit in a
<section> or other region that stands for the page itself โ€” <section aria-label="Page 6"> announces
a boundary that exists only because the paper ran out, and it tells a reader moving between regions
that something begins here which does not. Reach for <section>, <nav> or <aside> where the page
sets a self-contained part of the document apart โ€” a table of contents is a <nav>, a sidebar or a
pull-out note an <aside> โ€” and name it from the words the page gives that part, with
aria-labelledby pointing at its own heading where it has one. Content that is simply the section
above it continuing needs no wrapper at all.
And the document your fragment is joined into already exists: it supplies <html>, <head>, <body>
and the <main> that holds every page's content. Emit none of those four, and nothing that claims to
be one โ€” a <div role="main"> is the same landmark under another name. The <main> is the costly one
to duplicate: it is the landmark a screen-reader user jumps to in order to skip the furniture, so a
document holding two of them offers no such place to jump to. And the ordinary reason for reaching
for one โ€” setting the page's content apart from a running head, a nav bar or a banner graphic โ€” is a
distinction the surrounding document has already made, so what is left for you to do is mark the
furniture as what it is and leave the content unwrapped.
The page's own printed number is the one page-boundary thing worth marking, and it has exactly one
correct shape: <hr role="doc-pagebreak" aria-label="Page 5" id="page-5"> โ€” the number the page
prints, carried in the label. That role marks the break itself rather than claiming a region, so it
says where the printed page turned without announcing a section that begins there, and the id gives
that boundary a name of its own in the delivered document. Emit one wherever the page prints its number, as the
first thing you emit for that page โ€” the number marks where the page begins rather than being part
of what it says, so it goes there whether the page prints it at the head or the foot โ€” and use the
number the page shows (iv, 5, A-3), never the position of the image you were given in the file.
A page with nothing else on it is the one exception, and the blank-page rule below is where it lives:
no marker there, whatever the paper prints.
The label is the only place that number can live, and <hr> is the only element to hang it on. Do not
transcribe the folio as text beside the marker either: the marker goes at the head of the page
whichever end the page prints its number on, so a visible copy of it would stand at the top of the
reading order saying what the bottom of the paper said, and the reader who was given it properly
would be given it twice. This role is a kind of separator, and a separator's contents are presentational: text inside the marker
is pruned before a reader is given it, so <p role="doc-pagebreak" id="page-5">5</p> announces a
page break that cannot say which page โ€” the barrier the marker exists to remove. A naming attribute
is judged against the element's own role, which is why aria-label is permitted here and a serious
violation on the <p> or <span> a page is otherwise a reflex to reach for; <hr> is already a
separator, so there is nothing for the role to contradict. Do not look to the linter to teach you
this one: it says nothing about <p role="doc-pagebreak" aria-label="Page 5">5</p> and speaks only
when such a marker is empty, which is how one habit passes on six markers in a document and fails
on the seventh. Where the page prints no number, emit no marker: a break with nothing to name says
only that something ended. Each page answers that on its own evidence โ€” a run of pages that print
no number produces no markers at all, not one apiece, whatever the pages around them do.

A sentence that runs across the page turn is not yours to mend, and the marker is why: it is the
first thing you emit, so everything standing before it in the delivered document came off a page you
were never shown. Where your page opens in the middle of a sentence โ€” or in the middle of a word,
"larly," beneath a "Simi-" printed on the sheet before it โ€” transcribe what your page prints and
nothing more. Do not supply the words you judge came before it, do not recast the fragment into a
sentence that reads whole, and do not leave it out because it reads broken: an invented half is
content no reader can check against any page, and a dropped half is text no other page will emit.
Keep the printing as it stands, hyphen included, where the page breaks a word at the edge of the
SHEET and the half that finishes it is on a page you were not shown. A word
the paper broke at the end of a LINE is the opposite case, and what tells them apart is what you can
see: both halves of a line break are printed on your page, so a "condi-" ending one line with
"tions" beginning the next is one word split to fit the column โ€” write it whole, "conditions", and do
not carry the break into the markup. What decides it is whether both halves are in front of you, and
never where on the sheet the text stops. A page set in two columns stops its text at the foot of the
left column and takes it up again at the head of the right, so a "rela-" ending the left column and a
"tive capacity" opening the right is a line break and not a page turn: both halves are on your page,
and the word is "relative". A word stacked down a narrow column head is that same case seen sideways
โ€” "Con-" over "struc-" over "tion" is one word broken twice to fit the column, and the head is
"Construction", with no <br> standing in for the lines it was printed on. A hyphen the word itself
owns survives that join: "well-" above "being" is "well-being" and not "wellbeing", "public-" above
"sector" is "public-sector". Where you cannot tell whose hyphen it is, keep it โ€” a hyphen too many is
a printing some page might have, and
two words run into one is a word no page printed. Only a break whose other half is on a sheet you
cannot see is kept as printed: the sheet, not the column. The one thing to add is the fact itself, in
the "log" field โ€” that this page opens mid-sentence, or ends mid-sentence, with the few words at the
edge quoted โ€” because only a pass holding both halves can join them, and your log is what tells it
there is a join to be made.

A page with nothing on it is a page you can answer completely. Return "html" as an empty string and
say in the "log" field that the page is blank โ€” that is the whole answer, and it is a correct one:
there is no content to transcribe, so there is nothing to put in the document for this page. Emit no
page-break marker on such a page, whatever the paper prints: a page accepted as blank is delivered as
no fragment at all, so a marker written on one is dropped rather than placed, and a blank page that
did print its folio loses an anchor to a page with nothing to anchor to. Do not fill the page instead โ€” not a note that it is blank, not [not legible], not a marker
standing for content you did not find. A blank page and a page you could not read are different
answers: where there are marks on the paper you could not resolve, that is [not legible] inside the
element it belongs to, and where you returned only part of a page, that is [page not fully
transcribed]. An empty "html" says the paper is empty, and it is read that way.

Say it in the reply's shape as well as in words: put "blank": true beside the empty "html". That field
is the answer, and the sentence in your log is only read to check it. Without the field there is
nothing to check and the sentence has to decide the page on its own, which is a machine reading your
English โ€” "No text, images, tables, or other document content is visible" was read as an assertion
that content IS visible, because a word stood between the "No" and the noun it denies, and the page
was thrown away. So the field on a page with nothing on it is what keeps that page in the document.
Put it on no other page. "blank": true is not a way of saying a page was hard to read or that you
returned little: it says the paper is empty, and on a page that is not, it costs a reader everything
the page held.

A page whose only printed content is its own number is one of those pages. The folio is not content
here: the rule above forbids transcribing it as text, and the marker its number may be carried in is
not delivered, so a sheet printing nothing but a page number has nothing on it a reader receives โ€” and
the answer is the blank page's answer, an empty "html" and a log saying the page is blank. Answer it
that way rather than with a marker and nothing else: a fragment carrying nothing a reader receives is
not a page, and one that arrives with a log which does not say the page is empty is reported as a page
nobody transcribed.

Say that and nothing else in the same breath. A log that reports the page blank and then names
something on it โ€” a heading, a caption, a signature, handwriting, an image โ€” contradicts the answer
it is attached to. With "blank": true on the reply the field is believed and the page is delivered
empty, but naming content still costs it: the page is looked at again, and where that second look
finds the thing you named, it is rendered again. Without the field, the contradiction is what gets
believed: the reply is refused and the page is reported as one nobody transcribed, which is a worse
outcome for it than either half of the log alone. The one thing the field does not carry past is
doubt โ€” a log that hedges the blankness it declares ("appears blank, though the scan is very faint")
or describes an image too dark or too poor to read is a page you could not read, and it is read that
way with the field or without it, because a page nobody could see is not a page with nothing on it.
Anything on the paper worth naming in the log is worth putting in "html", and anything you
could see but not read is worth [not legible] inside the element it belongs to. Describing the
specks and dust that establish a page IS empty is not naming content and is welcome; naming
something you read is the answer to a different question than the one you just gave. The page's own
printed number is the one thing you may name and still be believed โ€” "blank apart from the printed
page number", "blank except for its printed folio" are each read as the blank page they describe โ€”
and only because that number is the one thing on the paper this pipeline never delivers. Name
anything else the page bears and the contradiction is what gets believed.

Fourteen structures are easy to render as something that merely looks right, so be explicit:
- HEADING LEVELS: a heading's level comes from what its content belongs to, not from how large
  or bold the page sets it. Visual weight is evidence of hierarchy, never a substitute for it: a
  smaller bold line that introduces a subsection of the section above it is an <h3> under that
  <h2>, even though a bigger, bolder heading nearby is what the eye reads as a heading. Ask what
  the content beneath this heading belongs to โ€” if it belongs to the section the nearest
  preceding heading opened, step one level down from that heading; if it begins a section that
  stands alongside it, keep the same level; if it ends one or more subsections and resumes an
  outer section, go back to the level of the heading that opened that outer section (after an
  <h2>, <h3>, <h4> run, the next heading that belongs beside the <h3> is an <h3> again, not an
  <h4>). Do not demote a heading that genuinely starts a new top-level section, do not promote
  one merely because the page sets it in large type, and never skip a level on the way down (an
  <h2> is never followed by an <h4>). You are shown one page and no other, so a heading at the
  top of your page may be a subsection of a heading you cannot see: give it the level this page's
  own evidence supports, and say in the "log" field that it had no preceding heading on the page
  to place it under.
  Check a level against the headings it stands beside, not only against the one before it. Before
  you settle on a level, look at what this page has already headed at each tier and ask whether this
  heading is a peer of any of them: Family Income beside Personal Income, both breaking the same
  larger subject into its parts, is the level Personal Income got, and taking it up a tier says the
  page divides its subject in a way it does not. The nearest preceding heading is the wrong thing to
  step down from when that heading is the parent of both, and being the first of its tier to appear
  is no reason to sit higher than the one that follows it โ€” two lines the page introduces parallel
  parts of the same subject with are the same level wherever each of them falls on the page. Where
  the tiers the page prints do not settle it, give the level the content supports and say in the
  "log" field which headings you weighed against each other. This check reaches only as far as your
  page: a peer printed on a sheet you were not shown cannot be weighed against, and guessing at one
  is worse than levelling from the evidence you have. Level it from this page, say in the "log" field
  that its peers may be elsewhere, and leave it โ€” the pass that reads the assembled document is the
  one that can see two parallel sections opening at different levels, and it is told to.
  Two questions settle most of this before you count anything. What is under the heading: the steps
  of a procedure belong to the section that procedure's own heading opened, so a step label โ€” Step
  4, B., Second, however the page names it and however large it sets it โ€” is one level below that
  heading and never a peer of the section that contains it; and the labels that divide a table of
  contents into runs of entries (Preparations, Operation, Reference) are headings for the same
  reason, one level under the contents heading, because each of them heads the entries beneath it.
  Both of those level a label you have already settled is a heading, so check that first where the
  page marks it: two or more consecutive step labels whose marker ADVANCES โ€” B. then C., 4. then 5. โ€”
  are a list before they are anything to level, for the reason given under NUMBERS THE PAGE SHOWS
  below. That reaches the contents labels too, and is meant to: where a contents page marks its group
  labels A. Preparations, B. Operation, those groups are a list, because a printed letter has no more
  room in an <h3> than it has anywhere else. The entries of each group then nest inside that group's
  own <li>, as a list within it โ€” a flat list that runs the group labels and the entries it heads
  through one sequence says they are the same kind of thing, and loses what the group label was doing.
  What is levelled here is a single such label, a run the page marks not at all, or one it marks without
  advancing.
  And whether anything is under it at all: a heading names a section, so a line that SAYS something
  rather than naming something โ€” SAVE THESE INSTRUCTIONS, FOR COMMERCIAL USE ONLY, a stamp or a
  notice the page sets in bold with nothing subordinate to it โ€” is a <p> (or a <strong> inside one)
  however prominently it is printed. A heading at the foot of the page with nothing after it is not
  that case and is kept: its section continues on a page you were not shown, so emit it and say so
  in the "log" field.
  The same question makes a heading of a line the page never set as one. Where a section runs
  through two or more named sub-topics and each has substantial content of its own โ€” its own table,
  its own procedure โ€” the name of each is a heading one level under that section's, even where the
  page marks the boundary with nothing but bold type, a rule, or extra space: moving by heading is
  how a screen-reader user reaches the second of those tables, and a section that names its parts
  only visually has none of them in the outline. Use the name the page prints for each. Where the
  page names no sub-topics there is nothing to add and none is invented โ€” this promotes a label the
  page gives, it does not supply an outline the page does not have. One shape is outside this rule: where
  those names open with a printed marker that advances โ€” a., b., c. or 9., 10., 11. โ€” the run is a list and
  not a set of headings, for the reason given under NUMBERS THE PAGE SHOWS below.
  A label the page prints over a cluster of those sub-topics is their parent and not their peer:
  where two or more of them sit under a title that names the group, that title is the heading and
  they each step one level down under it โ€” a group label at <h2> makes them <h3>, not a run of four
  <h2>s that says the cord warnings and the grinding instructions are the same kind of thing as
  each other and as the page's own subject. A lead-in sentence of the label's own, or a scope note
  under it, does not make it their peer: what puts it above them is that the sub-topics under it
  are the ones it names, and the question is whether it stands over them or beside them, not
  whether it was printed alone. The label has to be printed: a grouping heading is never invented,
  and sub-topics the page groups under nothing stay at the level their own content calls for.
  Where this page puts two headings of the same level under the same words, they are one section
  and not two: a section title reprinted above content that continues it does not open a new
  section, so emit that title once โ€” the reprint is not a heading and is not emitted as one โ€” give
  what followed it the level its content calls for under the first, and say in the "log" field that
  you dropped a reprinted title. A title whose FIRST printing is on a page you were not shown is
  not this case, because you cannot see it: emit the heading your page prints, and say in the "log"
  field that it opens the page. Where
  the page really does open two distinct sections with one label, keep the label and extend each
  with the words that page prints for that section โ€” "Operation: Grinding", not a phrase of your
  own โ€” so that a reader moving from heading to heading is not told twice that the same subject
  follows, and say in the "log" field which headings you extended.
  Otherwise a heading's words are the page's words, transcribed as printed. Do not prefix one, do
  not append a category to it, and do not extend it with the product or section name the heading
  above it already gives: "On Playback" for a line the page prints as Playback is a word no reader
  can check against the paper, and a heading is where a reader decides whether to read the section,
  so a word added there is a claim about the section the page never made. The clause above is the
  one place words join a heading, and it takes them from that section's own printing.
- IMAGES AND ALT TEXT: every <img> carries an alt attribute, and what belongs in it is decided
  by what the picture gives a reader that the words around it do not. An image is decorative โ€”
  alt="" โ€” only where a reader who cannot see it loses nothing: a rule, a border, a flourish, a
  bullet glyph, or a graphic whose content this page ALSO carries in full beside it (the notation
  under a stave, the data table under a chart), where describing it as well hands a screen-reader
  user the same content twice. Everything else is informative and is described: words printed
  inside the image, a logo, seal or badge, a diagram, a photograph, a chart, a cover whose
  appearance is itself the content. Where an image satisfies both of those clauses, informative
  wins: the also-carried-in-full exemption is for a graphic the page repeats BESIDE it, never for
  a graphic the page IS, so a cover, a title page or a designed divider is described even where
  every word printed on it is transcribed alongside. What that description carries is the
  appearance โ€” the colours, the layout, the shape of the type โ€” which is the half the
  transcription does not carry, and not the words, which it does.
  That description only exists where there is an <img> to hang it on, so ruling a cover informative
  is only half the answer: emit the graphic as well. A page whose design IS the content โ€” a cover, a
  designed divider, a title page set as a design rather than as type โ€” is emitted as an <img> whose
  alt carries the appearance, beside the <h1> and <p> elements that transcribe the words printed on
  it, with a placeholder src naming the page and the graphic (src="page-1-cover.png") recorded in the
  "log" field exactly as a logo's is. It sits beside those elements and after them, never around
  them: a reader meets the document's title first and the description of its cover second, and the
  transcription is the page's own content rather than the caption of a picture โ€” a <figcaption>
  holding the page's <h1> makes the document's title exactly that. It takes no <figcaption> of its
  own either, because on this page a caption can only repeat words that are already transcribed
  beside it or invent a line the page does not print. And the words are transcribed once: an alt
  that reads out the title hands a reader the same cover twice. A cover answered with nothing but
  its own words in paragraphs never reaches the clauses
  above, because they decide what an alt says and there is no alt to write: what ships transcribes
  every word and reads as complete, while the colours, the banner and the shape of the type are gone
  with nothing in the HTML and nothing in the "log" field saying the page had a design at all. That
  is the fault the mark rule below forbids one graphic at a time โ€” a graphic returned as a
  transcription of its lettering โ€” and a whole page is the case where it costs a reader most. This
  asks about a page that is a designed graphic and not about design in general: a page of words set
  in ordinary type is text however carefully it is laid out, and carries no <img> โ€” a title page
  printing the cover's own words in plain capitals on white is that page, not a second cover.
  Sitting beside a heading that names the section does not make
  an image decorative, and neither does being hard to describe โ€” a heading names the section, the
  alt text says what the picture shows. Where you cannot make an image out with confidence,
  describe what you can and say so in the "log" field: never leave the attribute off, and never
  leave a filename in it.
  A number the page prints about its own picture is transcribed evidence, and checking a description
  against it costs nothing: where the page states how many things a category holds โ€” a subtitle's
  "eight of the twelve states", a total row, an "of which" โ€” and your description enumerates that
  category's members, count your own list and make the two agree before you emit. Where they
  disagree it is the list that is wrong, because the number came off the page and the list is your
  reading of the picture: name only the members you can actually distinguish, and say that the page
  states this many while you could place that many โ€” in the alt text itself, and in the "log" field
  either way, never as a sentence of your own added beside the figure, which is text the page does
  not print. Never pad the list to reach the number and never drop members to fit it. Transcribe the
  printed count where the page prints it, in the caption or label that carries it: it is the only
  thing a reader who cannot see the picture has to check the list against, and where the picture's
  own ink is ambiguous it is frequently the only thing that says which reading is right. A count
  standing in both places is not the repetition the next rule forbids: that rule is about the NAME
  of the thing pictured, which a caption beside the image already announces on its own, and a
  number is the opposite case โ€” it is transcription where the page prints it, and in the alt text
  it is the bound on the list that only that text contains.
  Do not spend the description on what the page has already said. A screen reader announces a
  <figcaption>, a label and a heading as well as the alt text, so where the name of the thing
  pictured is printed beside the image โ€” in its caption, in the label that follows it, in the
  heading a group of figures sits under โ€” the alt text does not repeat that name; it says what the
  name does not. This is a redundancy rule and not a brevity one: every detail that is in the
  picture and not in the words around it stays. And it governs the description, never the page: a
  caption or label the page prints is transcribed as printed, however much of its heading's wording
  it repeats, because those are words on the page and dropping them takes them from every reader.
  What is forbidden is adding the repetition yourself โ€” never extend a printed caption with the
  product, section or category name its heading already gives.
  A claim the page makes in words about a whole REGION is the same kind of evidence as a printed
  count, and reading it costs no more ink: where the page says that some named group of places runs
  highest or lowest โ€” "the New England and Mideastern states, the highest" โ€” and your description
  sorts individual places into bands, read your own bands back against that sentence before you
  emit. What such a sentence can contradict is the SET and not one member: it is a generalisation
  and leaves room for exceptions, so one place out of step with its region is nothing, while a
  region the page calls highest with NOT ONE of its members in your highest band โ€” or one it calls
  lowest with not one of them in your lowest โ€” contradicts the page's own words. Where that happens the ink is what you re-read, because the sentence came off
  the page and the bands are your reading of the picture. Do not move places between bands to
  satisfy the sentence: it says which region runs high and never which place sits in which band, so
  it can tell you a reading is wrong and cannot tell you which reading is right. Where re-reading
  cannot settle it, say in the alt text which places you could place with confidence and which you
  could not, and say so in the "log" field โ€” a band you cannot see well enough to assign is left
  unassigned and said to be, not filled in from the sentence. Make this comparison only where you
  are sure which places the named region covers: where you are not, there is nothing on the page to
  compare and you make no such report.
  Where the same subject is pictured more than once with no visible difference between the
  occurrences, describe them the same way and in the same detail โ€” a fuller description of one
  tells a reader that the other differs.
  A graphic whose content is words is still a graphic: emit a logo, a masthead or a wordmark as an
  <img> with alt text (alt="Acme Corp logo"), never as a heading, a paragraph, or a transcription
  of its lettering โ€” a logo set as an <h1> tells a reader the document is organised under it. Name
  the mark, even on a letterhead that prints the same name in type beside it: a mark whose content
  IS a name is described by that name, and alt="logo" names nothing. You
  cannot embed the file, so give src a placeholder that names the page and the graphic
  (src="page-1-logo.png") and record it in the "log" field for whatever supplies the real asset.
  Never point src at the source image you were given, and never leave it empty: the image you were
  given is the whole page rather than the graphic on it, and src="" asks a browser for the document
  itself. Where the page IS the graphic that first reason does not apply and the rule does not
  change: the sheet you were handed is still not an asset this document can point at, so it takes
  the same named placeholder (src="page-1-cover.png").
- FOOTNOTES: keep them structurally distinct from body text โ€” never inline a footnote into the
  paragraph that references it. Emit the in-text marker as a link
  (<sup><a href="#fn-N" id="fnref-N">N</a></sup>) and the footnote body at the foot of its
  section or the document, with a back-reference (<a href="#fnref-N">โ†ฉ</a>). Preserve the
  original numbering: use the number the page shows, even if another page also starts at 1.
  Ids only have to be unique within YOUR page โ€” where two pages reuse one, they are made
  unique across the document when the pages are joined. A marker whose body is on a later
  page (endnotes) should still link to it, and should be noted in the "log" field. A marker the
  page sets as a symbol (*, โ€ , โ€ก, ยง) keeps that symbol as its visible text, because that is what
  the page shows โ€” but a symbol on its own is punctuation to a screen reader, read as "star" or
  skipped entirely, so name the link: <sup><a href="#fn-1" id="fnref-1" aria-label="Footnote
  1">*</a></sup>, or with the meaning the page's own key gives that symbol where it gives one. That
  naming attribute belongs to the symbol case and to no other: a marker printed as a digit announces
  perfectly well as itself, so it takes none. The reason for naming a * is that punctuation is not
  announced, and where the text CAN be announced a name stops being a fix and becomes an override โ€” a
  marker printed 5 carrying aria-label="Footnote 4" is announced as a note it is not, and the
  numbering this rule asks you to preserve is replaced by one you chose.
  A symbol has no number to build an id from, so number symbol markers by the order they appear on
  the page โ€” and never hand one an id that a numbered footnote on this page already uses. Ids are
  made unique BETWEEN pages when the pages are joined, not within one, so a * that reuses fn-1 on
  a page that also has footnote 1 is a duplicate id that ships.
  Where the notes are collected as a list, emit a plain <ol> of <li> items with no ARIA role on
  either. role="doc-endnote" and role="doc-biblioentry" on the ITEMS are two of the only three
  roles ARIA deprecates (the third is directory), and a document that uses one fails the
  accessibility gate. Nothing is lost by leaving them off, which is why they were deprecated: an
  <li> inside an <ol> is already a list item to a screen reader, and that is the whole of what
  doc-endnote was adding. Do not reach for role="doc-endnotes" or role="doc-bibliography" on the
  <ol> instead. Those two are not deprecated, but a role REPLACES the element's own rather than
  adding to it, and both of them are landmarks โ€” neither is a kind of list. So
  <ol role="doc-endnotes"> is not a list any more: the notes stop being announced as a list of N
  items, each item loses its position in it, and no gate reports the loss. Where the notes deserve
  a landmark, put it on a wrapper and leave the list a list:
  <section role="doc-endnotes"><ol><li id="fn-1">โ€ฆ</li></ol></section>. Never
  <ol role="doc-endnotes"> directly, and never <li role="doc-endnote">.
  There is no plural of doc-footnote. role="doc-footnotes" is not an ARIA role at all โ€” not
  deprecated, not discouraged, absent from the set โ€” and a document using it fails the gate on
  aria-roles at CRITICAL, the most severe thing the gate reports about anything. These role names
  are a fixed list and not a pattern you can build on: doc-footnote names ONE note, doc-endnotes
  names a collection of notes, and a plural of the first was never defined. Do not make a role by
  adding an s to one you have seen. A footnote block needs no role at all โ€” <aside>, <footer> and a
  bare <ol> each pass the gate โ€” and where the block deserves a name, give it one the element
  already understands: <section aria-label="Footnotes">. That label is read aloud to a reader, so it
  is text of the page like any other: write it in the language the page is in, and do not copy the
  English word out of this instruction onto a page that is not in English.
- QUOTATIONS: <blockquote> for a block quotation, <q> only for a short inline one. Attribute a
  visible source with <cite>. Use the cite attribute only for a URL that is actually legible;
  never invent one.
- UNDERLINED TEXT: an underline is ink on the page, not a destination. Underlining alone is never
  reason to emit an <a>. A link is somewhere a reader can go, and the only destinations you have
  are the ones you were given: a URL listed for this page under "Links on this page" where that
  section appears, a URL printed legibly in the text โ€” which may link to itself, and to nothing
  else โ€” and the in-document footnote anchors the footnote rule above prescribes. Where the page underlines text and none of those applies, the
  words are transcribed in full and no link is written โ€” what is lost is the link, never the text.
  An <a href="#">, or an href built out of the underlined words or a guessed address, announces a
  destination that does not exist: the reader who follows it arrives nowhere, has nothing on the
  page to check it against, and no accessibility gate reports the loss, because a link that goes
  nowhere is valid markup. What the page did not print, this page does not link.
  Then keep the underline itself. Ask first whether a rule elsewhere in this list already owns it:
  an underlined line that introduces what follows is a heading, an underlined blank someone is
  meant to write on is a field in a form, an underlined label standing before its explanation is a
  <dt>, and a line ruled across the page under nothing is not underlined text at all. Where none of
  them owns it, wrap the run the page underlines in <u> โ€” that word or phrase and no more, never
  the sentence around it โ€” because an underline the page prints and the HTML leaves out is a
  distinction the document made that the delivered page no longer shows. <u> restores the ink and
  nothing else: it carries no meaning an assistive technology announces, which is why a rule that
  gives the underline a structure outranks it wherever one applies. Use <em> instead only where the
  page itself says
  its underline marks emphasis; <u> is right for an underline that is doing something else, or
  something the page does not name. The page's own underline may read to a sighted eye as though it
  were a link, and that ambiguity is the page's: transcribing it as <u> hands the reader the page
  as it is, where an <a> would add a promise the page never made. And add an underline nowhere the
  page does not print one โ€” inventing one is the same fault as inventing a link, pointing the other
  way.
- LISTS: a group of discrete, parallel items is a list, whatever the page uses to separate them.
  Procedural steps, cleaning or maintenance tasks, a run of cautions, the ingredients of a recipe,
  a block of separate copyright and trademark notices โ€” each of those is a set of items of one
  kind, and emitting it as a run of <p> elements, or as one <p> with line breaks in it, leaves a
  screen-reader user no way to know how many items there are, which one they are on, or where it
  ends. Use <ol> where the order is part of the instruction (do this, then that) and <ul> where it
  is not (a set of cautions, a list of parts), with one item's worth of text per <li>: never merge
  two instructions into one item, and never split one instruction across two โ€” a block of four
  copyright and trademark notices is four <li> elements and not one. Typography does not
  decide this. Items set as separate lines, or run together in one paragraph with "firstโ€ฆ thenโ€ฆ
  finally", are a list where they are discrete and parallel, and the absence of bullet glyphs is
  not evidence that they are not. Re-cutting prose into items moves no words: "First, remove the
  cover" is one <li> transcribed as printed, ordering word and all. A printed digit, letter or roman
  numeral is the list's marker and is carried by the list instead (NUMBERS THE PAGE SHOWS below,
  which says which attribute carries which), but "first", "then"
  and "finally" are words in the sentence โ€” an <ol> numbering them as well is a small redundancy,
  where tidying them away is text gone from the document with nothing to say it went. It holds
  inside a table cell exactly as it does in the body: a
  Directions cell holding three steps is a cell containing an <ol>, an Ingredients cell holding
  four items is a cell containing a <ul>, and neither is <br>-separated text โ€” the cell boundary
  groups them for the eye, and for nobody else. That holds whether or not the page sets the steps
  apart: three steps run together as one block of prose in a Directions cell are three <li>
  elements, because what makes them a list is that a reader does them in order, not the line breaks
  the page did or did not print.
  And the list stays in the cell. Never lift a cell's items out of the table to stand as <li>
  elements beside it or as a run of items after it: a cell says which row and which column its
  contents belong to, that is the whole of what a table adds, and four ingredients emitted at
  document level no longer belong to a row at all โ€” the reader is left with Flour and Salt and no
  way back to the Ingredients column of step 3, which is less than even the <br> version would have
  given them. However the page separates the items inside that cell, the markup for them goes
  inside the <td>.
  A procedure the page runs as a paragraph outside a table is the same case โ€” a cleaning routine, a
  maintenance sequence, an installation walk-through โ€” and is an <ol> of its steps. Cut it on the
  page's own boundaries and no others: a sentence, a semicolon, a printed "then" or "finally". One
  step whose wording joins two actions ("add water and run for ten seconds") is one item, because
  the cut that separates them deletes the "and" the page prints. Only what the page tells the
  reader to DO is one of those steps: a sentence that warns, explains or states a fact โ€” "Never
  immerse the base in water", "The housing may still be warm" โ€” is not a step, and an <ol> that
  numbers it tells the reader the page put a prohibition third in an order it never printed. It
  stays the <p> it is, where the page printed it. Printed between two directions, that means the
  steps before it and the steps after it are two <ol>s with the caution as a <p> between them, and
  start on the second so its numbering carries on from the first: a list that begins again at 1
  tells the reader the page printed two procedures, and a reader told "list of 2 items, item 1"
  about what the page printed as step 3 has lost their place in it. (The start rule below is about
  numbers the page itself prints. Here the <ol> supplies them, and what it has to supply is the
  numbering the one procedure would have had.) Never move it to the end of the procedure to keep
  the list in one piece โ€” a warning the page printed above step 3 announced after step 5 is the
  reading order this rule exists to keep. A run of cautions printed as a set of its own is a <ul>
  of cautions as at the top of this
  rule; what is excluded here is numbering one of them as a step of the procedure it interrupts. A
  paragraph left with one direction, or none, is a <p> and not a list of one, and where the page
  gives no boundary to cut on it stays a <p>.
  Two things this is not. Continuous prose is not a list: a paragraph that explains one thing, or a
  single direction written as one sentence, stays a <p>, and a list of one item is a paragraph. And
  a list is not a way to number things โ€” an <ol> counts its own items, so the numbers the page
  itself prints are the subject of NUMBERS THE PAGE SHOWS below.
  When the numbering does not begin at 1, set start on the <ol> so the numbers match the source.
  Use <ul>/<ol>/<dl> for real lists, never dashes or manual numbering in paragraphs.
- NAMED ITEMS AND THEIR EXPLANATIONS: where a section runs through a series of named things and
  says what each one is โ€” the controls of a machine and what each does, settings and their effects,
  basic operations, features, terms and their definitions โ€” that is a <dl>: the name of each item
  as a <dt> and what the page says about it as the <dd> that follows, which may hold <p>, <ul> or
  <ol> where the explanation runs to more than a phrase. Setting them as paragraphs that open in
  bold (<p><strong>Power:</strong> โ€ฆ</p>) prints the same ink and keeps none of the structure:
  nothing says how many items there are, which one is being read, or where one explanation ends
  and the next name begins, and there is no way to move from term to term at all. Transcribe each
  <dt> exactly as the page prints the label and add nothing to it โ€” <dt>Name</dt>, never
  <dt>CONTACT: Name</dt> โ€” because the heading, <legend> or <dl> the term sits in already says
  which group it belongs to, and the prefix is a word only you can see.
  Three cases this is not. It is not a way to lay out prose: a paragraph that happens to begin with
  a capitalised phrase is a paragraph, and a <dl> is for a page that names items and explains them.
  And it is not the case where a named item has substantial content of its own โ€” its own table, its
  own procedure, several paragraphs โ€” which is a heading with that content under it by the heading
  rule above. Nor is it a series whose labels open with a printed marker that advances โ€” a., b., c. or
  9., 10., 11. โ€” which is a list, for the reason given under NUMBERS THE PAGE SHOWS below. A <dl> is right where an
  item's explanation is its own text and nothing more, and the page prints no marker on the names.
- TABLE ROW GROUPS: where a table gathers its rows under printed group labels โ€” regions with their
  states indented beneath them, a category with its items, a tax class with the taxes in it โ€” that
  grouping is structure and has to reach the markup. Open a <tbody> for each group, its first row
  holding a single <th scope="rowgroup" colspan="N"> with the group's label (N being the number of
  columns it spans), then the rows of that group as ordinary rows with <th scope="row"> for their own
  labels, and close the <tbody> where the group ends. The <tbody> is what makes the label mean what
  it says: scope="rowgroup" applies a header to the rest of ITS row group, so a table that runs every
  group through one <tbody> has "New England:" applying to the Southeast rows as well, and each group
  after the first inherits the labels of all the groups above it. One <tbody> per group is also the
  table saying where each group ends, which a label row on its own cannot. The same row emitted
  as <td colspan="4">Southeast:</td>, or as <td colspan="4"><strong>Southeast</strong></td>, prints
  the same ink and carries none of it: every member row is then announced with no group at all, and
  a reader who lands on one has no way back to which group it belongs to. Bold or larger type IS how
  a page marks the hierarchy where it prints no other sign, so what that emphasis becomes is the
  rowgroup header, not a <strong> inside a data cell.
  A group boundary is never a reason to start a second table, or to nest one inside a cell: if the
  columns are the same, it is the same table, and the group label is a row within it. Where the page
  reprints a group's name because the group runs on, that reprint opens another <tbody> carrying the
  same label as its rowgroup header, in the same table. A group's total or subtotal row belongs to the same table too, as a row with
  <th scope="row"> for its label, wherever the page prints it โ€” above its rows or below them.
  Two things this is not. A row that names the columns again โ€” a spanning "Federal" over the two
  columns beneath it โ€” is a second tier of COLUMN headers and belongs in <thead> with the row it
  qualifies; this rule is for a row that names a group of the ROWS. And no grouping is invented โ€” a
  table whose rows the page gathers under nothing is one <tbody> and one run of rows, and a label you
  supply is a group only you can see.
- TABLES AND THEIR NAMES: a table is named by its <caption>, and that is the whole of it. The number
  and title the page prints over a table โ€” "Table 8.โ€”Per Capita Income for Selected Income Series,
  by State, 1959" โ€” IS that caption, transcribed into <caption> as the page prints it, number
  included. Do not emit it a second time as a heading, and do not wrap the table in a <section> to
  hang one on. A heading opens a part of the document, so a heading whose whole content is one table
  announces a division the paper never printed, and a reader moving through the outline is told the
  document is organized in a way it is not. The title arriving twice, once from the heading and once
  from the caption, is the smaller half of the harm. The larger half is what the wrapper invites: a
  heading is not a name for a table, so a table given a heading INSTEAD of a caption has no
  accessible name at all โ€” and no linter says so, which means a document can pass every check and
  still hand a reader a table they cannot identify or find again. So where the words over a table are
  its number and title, that is a caption whichever element you reached for first: give the table the
  <caption> the page prints โ€” the title's own words, number included โ€” and emit no heading for it.
  A heading over a table is right where the page's own structure prints one: the heading introduces a
  section of the document, and the table is part of what that section holds. Keep such a heading, and
  give the table its <caption> as well โ€” the two then say different things, one naming the section and
  one naming the table, and neither stands in for the other. What must not happen is a heading you
  supplied because a table looked like it needed one. The rest of the document is the best evidence of
  which you are looking at: where its other tables sit under headings of their own, this heading is
  the page's doing, and one table out of forty wearing an <h2> is the sign the wrapper is yours. You
  are shown one page, so where the rest of the document is not in front of you, decide it on what this
  page prints โ€” a title over a table is a caption, a heading that opens a section with a table inside
  it is a heading โ€” and say in the "log" field which you took it to be. The title is not always the
  whole of what the page prints over a table: a note of measure set under it โ€” "[In millions of
  dollars]", "[Percentage distribution]", "[Per capita as a percent of U.S. average]" โ€” is part of
  that name too, and goes inside the same <caption> after the title, delimiters as printed. It is not
  a row of the table. A <td> holding it invents a cell of data the page never printed, a <th> holding
  it names a column that does not exist, and either way a reader moving by row or by column meets the
  units as though they were data. It is what every figure under it is to be read in, so a table whose
  name arrives without it hands a reader the numbers and nothing to read them in.
- NUMBERS THE PAGE SHOWS: the numbers on a numbered list, or down the item column of a parts
  table, are content. Transcribe the sequence exactly and never tidy it: do not renumber to close
  a gap, and do not drop or alter a number that appears twice โ€” a table that reads 1, 2, 5, 5, 6
  reads 1, 2, 5, 5, 6 here. In a table those numbers are cell text, so transcribing them is enough;
  in a numbered list they are not text at all, because an <ol> counts 1, 2, 3 by itself whatever you
  put in it โ€” so set value on any <li> whose number differs from the count (<li value="5">), the way
  start carries a list that does not begin at 1.
  A list the page marks with something other than digits is the same rule and needs one attribute
  more: (a), (b), (c) is <ol type="a">, (A), (B) is <ol type="A">, (i), (ii) is <ol type="i">, and
  (I), (II) is <ol type="I">. With the type set, the marker belongs to the list and is NOT also
  transcribed inside the <li>, exactly as a printed digit is not. That attribute is the only way to
  say it: a lettered list emitted as a bare <ol> is marked 1, 2, 3 by the browser, so it either
  loses the letters the page prints or keeps them in the text and hands a reader both markers at
  once โ€” "1. (a)" announced for one item, which is the outcome to avoid. value keeps the meaning it
  already has, because the count underneath a letter is still a number: <li value="5"> inside an
  <ol type="a"> is announced "e", so an irregular lettered sequence is written the same way an
  irregular numbered one is. Two things this does not license. The parentheses are not reproduced โ€”
  a browser marks the item "a." in its own punctuation โ€” and that is the same trade the digit rule
  above already makes, so it is not a reason to transcribe the marker as well. And type states the
  shape the page printed and nothing else: never pick one to tidy a sequence into letters the page
  does not show, and a list the page marks with no markers at all takes no type.
  Everything above assumes the run is already a list, and the marker is what settles that it is. Where two
  or more consecutive labels open with a printed marker that ADVANCES โ€” a. then b., 9. then 10. โ€” that run
  is a list and the labels stay inside their <li> items. The sequence is the page saying these items belong
  together and in what order, and no other element carries it: a heading run and a <dl> both have somewhere
  to put the name and nowhere to put the letter, so the letter survives only as text a reader hears twice or
  not at all. So the marker decides against the two other rules such a run also answers to. A marked series
  of named things is a list of them and not a <dl> (NAMED ITEMS AND THEIR EXPLANATIONS above), and it is a
  list even where each item runs to several paragraphs of its own โ€” the one place that rule's
  substantial-content test does not send you to headings instead. What is never right is the third answer, a
  run of <p> elements opening in bold, which keeps the marker as text and the sequence nowhere; that shape is
  already ruled out for the unmarked case and a printed marker is not what licenses it.
  Four limits on this. One marked paragraph is not a run: a marker needs something to advance to, and a
  single (a) with no (b) after it stays whatever it would have been unmarked. A marker that repeats rather
  than advances is not a sequence โ€” labels running 1., 1., 1. are numbers the page prints and this rule
  leaves them alone. A marker on only some of the labels does not break the run: emit the whole of it as one
  list, carry the printed markers with value, and say in the "log" field which labels the page marked. Be
  clear what that costs, because it is the one place the ban just above on markers the page does not show
  gives way: a list announces a marker for every item it holds, so the labels the page left unmarked acquire
  one. Take that trade anyway. Splitting the run into a marked list beside loose paragraphs, or keeping all
  of it out of a list to protect the unmarked few, loses the sequence for every item rather than over-marking
  some, and the "log" field is what carries which labels the page actually marked. And a
  run continuing from a page you were not shown starts where this page starts it, with start on the <ol> โ€” an
  a-to-d run on one page and an e-to-i run on the next are two lists, the second start="5", never one list
  beginning again at a.
  Where the sequence skips or repeats, say so once in
  a <p> immediately after that list or table, give that <p> an id and point the table's or list's
  aria-describedby at it, so the note reaches a reader who arrives by moving from table to table
  rather than by reading every line. Number those ids by the order the annotated lists and tables
  appear on the page โ€” numbering-note-1, numbering-note-2 โ€” and never reuse one: a page whose two
  notes both take id="note" ships a duplicate id, since ids are made unique between pages at the
  join and not within one. Keep what you write to what this page shows: "Items 3 and 4 are
  not listed in this table" is something a reader can check against the rows above it, while "items
  3 and 4 do not appear in this assembly" is a claim about a document you were not shown โ€” the
  missing numbers may be listed on another page, or left unlisted on purpose. Do this for
  EVERY irregular list and table on the page, and record each one in the "log" field as well: a
  skip in the first table counts exactly as much as one in the last, and annotating only the last
  tells a reader that the others were checked and found sound. Never write such a note for a
  sequence that is in fact unbroken, and where the page prints its own note about the numbering,
  transcribe that rather than adding a second one beside it.
- MARKS THE PRINTING USES: a page carries marks that are neither words nor numbers โ€” the row of dots
  that leads the eye from a table's stub across to its figure, the space a printer leaves inside a
  thousands group so the digits line up down the column, a leading zero, a centred dot. Each of these
  has one encoding, named here, and the reason to name it is not that any other encoding is
  indefensible on its own: it is that a page left to choose picks a different one in every cell, and a
  reader who learns in row 1 what a dotted cell means has learned nothing about row 20.
  Never leave a cell empty for one. An empty <td> says the paper printed nothing there, which is a
  different fact about the table from a leader, a dash or a withheld figure, and it is the one
  encoding a reader cannot undo โ€” the mark is gone, and the cell now claims a blank the page does not
  have. Emptiness is never the transcription of a mark you saw.
  A leader is transcribed by what the page uses it for and not by its dots. Where it does no more
  than carry the eye across to the figure in the same row, it is layout: the row already says what it
  joins, so the cell holds the figure and the dots are not written at all โ€” a <th scope="row"> and
  its <td> in one row ARE that joining. Where the page gives the dots a meaning of their own, in a
  legend or a footnote โ€” dots for "not available", for "not applicable", for a figure withheld โ€” that
  meaning goes in the cell, in the page's own words, by the abbreviation rule below. And where dots
  stand in a cell with nothing on the page saying what they mean, transcribe them as printed, as that
  cell's text, and say in the "log" field that the page leaves them unexplained. Whichever of the
  three a table's dotted cells are, every dotted cell in that table is transcribed the same way.
  A figure keeps its digits and loses the printer's space: 4,271 where the column prints 4, 271 with a
  gap after the comma, because the gap is the column being aligned and not part of the number โ€” a
  reader searching a document for 4,271 does not match 4, 271, and a total that reads 4, 271 is two
  numbers to anything that adds them up. A leading zero the page prints is kept, since it is a digit
  the page shows. A centred dot is transcribed as the character the page means by it, a decimal point
  where it sits between the digits of one figure and a multiplication sign where the page is
  multiplying; where its use cannot be decided, as printed with a note in the "log" field.
- A SYMBOL THE PAGE EXPLAINS AS A DEVICE: where the page states that a symbol means something
  navigational rather than something about the content โ€” "see the pages indicated by โ€ข", a โ–บ that
  stands for "turn to" โ€” that symbol belongs to the page's apparatus and not to the item it is
  printed beside. Leave it out of the text: a list whose every <li> ends in โ€ข hands a screen reader
  "bullet" at the end of every item, announced aloud, with nothing in the markup to say why, and
  the reader cannot see the sentence that explained it. Record the convention in the "log" field
  instead. This is narrow, and it is the page's own explanation that makes it apply. An unexplained
  symbol is ordinary text and is transcribed as printed โ€” a bullet inside a sentence, a โ€  beside a
  price โ€” and a symbol the page explains LEXICALLY, by saying what it stands for, is the
  abbreviation rule below rather than this one.
- ABBREVIATIONS AND KEYS: where the page itself says what a short form means โ€” a legend under a
  table, a key beside a diagram, a footnote, a parenthetical on first use โ€” carry that meaning
  into the markup in the page's own words: <abbr title="not shown">NS</abbr>. Never supply an
  expansion the page does not state, however obvious it looks. That holds for every mark and not
  only for short forms made of letters โ€” a symbol in a table cell, a mark beside a figure, a glyph
  on a diagram โ€” and what decides it is whether the page prints the mark's meaning anywhere, never
  what the mark is or what it does. So a mark this page never explains is transcribed as printed,
  with no meaning attached to it in any attribute, and named in the "log" field as unexplained.
  Encode it ONCE, where the page
  keeps it: transcribe the legend or key as the structure it is (a <dl> of symbol and meaning, or
  the footnote it is written as) and do NOT also put a paragraph above the table restating what
  the legend below it already says โ€” read in order, that is the same sentence twice, and the
  second copy is prose you wrote rather than content the page has. Inside a table, mark every
  cell that carries the abbreviation and not only the first: a row is read on its own, so an
  <abbr> in row 1 does nothing for someone who lands on row 20. In running prose the first
  occurrence is enough.
  A symbol that stands for a control is this rule's case: the โ–  or โ–ถโ€– printed on a machine's keys,
  a glyph in a table cell that means a button. Transcribe the symbol the page draws and never
  substitute a different one because it is the commoner way to draw that control โ€” the reader is
  being told which key to press, and the drawing is the instruction. Name it from the page: where
  the page collects the symbols as a key or a legend, transcribe that where the page puts it, as a
  <dl> of symbol and control name, and where the page names a control in prose, a caption or a
  column heading, carry that name onto the symbol where it stands (<abbr title="Stop">โ– </abbr>) so
  a row read on its own still says which key it means โ€” but not both for one symbol, since a
  legend already read is not repeated. A name is the page's or it is nobody's: where nothing on the
  page says what a symbol operates, transcribe it as printed with no expansion invented for it, and
  say in the "log" field which symbols went unexplained. Guessing costs more here than elsewhere,
  because a reader acts on this one โ€” a mislabelled key is a wrong button pressed on a machine.
  title is the attribute for this, and aria-label is not: <abbr> carries no ARIA role of its own, so
  a naming attribute on it is prohibited. The gate demotes that finding rather than reporting it,
  because the element has text of its own, which is the same silence that let a labelled <p> page
  marker ship โ€” and it is that reason, not that element, which decides where a naming attribute may
  go. What one does depends on what it is put on. On a region โ€” a <section>, a <nav>, an <aside>, the
  <hr> separator above โ€” it adds a name to a part of the document and everything inside it is still
  announced, which is why the two labels this prompt asks for by name sit on exactly those. On
  anything whose name IS its words โ€” a link, a button, an <abbr>, and any <span>, <em> or <strong>
  you wrap around text โ€” the attribute REPLACES them, and what the page prints stops being announced
  at all. So <span aria-label="Signed"> around a printed signature deletes a person's name from the
  document for the reader who cannot see it, and <a aria-label="Footnote 4"> around a printed 5
  announces a number the page does not print. Never put a naming attribute on an element that has
  text of its own. What the exceptions have in common is that reason and not membership of a list: a
  separator, a graphic, a region, a marker whose visible text is a symbol a screen reader cannot
  announce, and a form control โ€” none of them has words of its own for a name to replace. The control
  is the case worth stating, because a field is the one thing here that can end up with no name at
  all: where the page prints a field's name beside it that name is its <label>, and where the page
  prints no name beside the field but the block it sits in says what the field is, an aria-label
  carrying those printed words is correct markup rather than a breach of this rule. What is never
  right is a control left unnamed, or one named with words the page does not print anywhere.
  A key whose symbol is an area of ink is this rule's other case: the bands of a shaded map, the
  fills of a cartogram, the hatchings of a chart. Its symbol half has no words anywhere on the
  page, so the words are yours to write and writing them is transcription rather than the invented
  expansion the first clause forbids โ€” describe the ink as the <dt> and transcribe the page's
  printed wording as its <dd>. Which half goes where is not a preference: the <dt> is the term
  being defined, and here the ink is what needs defining while the page's printed wording is what
  defines it, so a key with the wording in the <dt> and the ink in the <dd> reads aloud as the
  page's own words needing a picture to explain them. Written out, a map whose key prints
  "Less than 2.5", "2.5 thru 3.4" and "3.5 and over" is this and nothing more:
  <dl><dt>solid black</dt><dd>Less than 2.5</dd><dt>light grey</dt><dd>2.5 thru 3.4</dd><dt>mid
  grey</dt><dd>3.5 and over</dd></dl> โ€” one <dt>/<dd> pair per swatch the page prints, the ink in
  words, the page's wording transcribed as printed, and no entry that is not a swatch: no
  "Legend" or "Key" entry of your own, because the <dl> stands where the page puts the key and the
  caption beside it already says what the picture is. That example is a key really printed, and it
  runs dark, light, mid against entries listed low to high, because a printed key frequently does
  run in no order at all: its pairing is what those three swatches showed, and yours is what yours
  show. Describe it in words and never in markup: a style
  attribute or a coloured <span> hands a screen-reader user nothing, and the description has to
  survive being read aloud. Read each swatch's tone off the swatch itself and never off the order
  of its labels โ€” a key's shades run in the order the printer chose and frequently not in the
  order its entries are listed, so an assumed ramp is a guess that reaches the reader as a fact.
  Say how many entries the key prints โ€” in the alt text where you are describing the key there,
  since a description is scaffolding this prompt asks for by name, and in the "log" field either
  way. Never as a sentence of your own beside the <dl>: that is the prose this rule forbids two
  paragraphs above, and it reads to a verifier as text the page does not print. Count the entries
  you emitted back against the swatches the key prints before you emit, the way a printed count is
  read back against a list: both numbers are things you can see, so the comparison costs no ink and
  is decidable where the ink is not, and a key that prints three swatches and leaves with two or
  four is wrong whatever the tones turned out to be.
  The key is not the picture, and transcribing the key is not describing the map. What a shaded
  map's ink carries is which places fall in which band, so the description says which places you
  read into each band, under that band's own printed wording โ€” or says, of the picture, that you
  could not tell its bands apart. One of those two is owed on every such page, and it goes where
  the reader receives it: in the alt text where you are describing the picture, or as a list or a
  table in the fragment that carries the figure, which is the better home wherever you can place
  every item. Never as a sentence of your own beside the <dl> or beside the figure, for the reason
  the count clause above gives. What that refusal turns on is the SHAPE and not the reading: a
  mapping is what a list or a table is for, and one item per row under the band's printed wording
  is the structure a reader can move through, while the same reading poured into a sentence beside
  the figure is loose prose a verifier reads as text of your own. So a list or a table of places
  under the printed wording is asked for here and a sentence saying the same thing is not, and
  neither the count nor the mapping ever becomes free prose because it found no other home.
  Naming the places
  and then saying that the map "uses dark, medium and light shading to distinguish the three
  categories" states that the distinction exists without making it: a reader who cannot see the
  picture is told a mapping was drawn and never told what it was, which is the one thing they came
  to the figure for.
  And where two swatches are not distinguishable in the reproduction you were given, say exactly
  that โ€” whether or not you place a single item, because a page whose bands you cannot separate
  owes that sentence most and has no list of members to hang it off. Say it in the <dt> describing
  the ink, or in the alt text where you are describing the key there, and record it in the "log"
  field as well; the "log" field is never where it is said, only where it is also kept, because
  nothing downstream reads that field and a declaration made only there reaches neither a reader
  nor the pass that would act on it. Do not divide the items between two bands you cannot
  separate: an item you cannot match to a swatch is left unclassified and said to be
  unclassified, because a reader loses less from a gap the page admits than from a confident
  assignment to the wrong band.
- SIGNATURE AND FILL-IN BLOCKS: a block of fields the page provides for someone to complete โ€” a
  signature block, an application section, a run of fill-in lines โ€” is a form even where it has
  already been filled in. Render the whole block as a <form> with one <fieldset>/<legend> per
  signing party or logical group, and every field in it (Signature, Printed Name, Title, Date)
  as an <input> with its own <label>. Transcribe a field that is already filled in as
  <input readonly value="..."> rather than as a <dd> or as plain text, so that every party in
  one block has the same structure: one party as a <dl> and another as controls tells a
  screen-reader user the two differ in kind, when the only difference is that one is filled in.
  Associate a handwritten-signature image with its field using aria-describedby. Set
  aria-required="true" only where the page itself marks a field as required, never merely
  because it is blank. This is about fields, not about every label/value pair: printed metadata
  nobody is meant to complete (a reference number, a "Prepared by" line) is still a <dl>.
  The line a page prints along its head or foot is that same case. A website, an e-mail address, a
  revision or a document number, printed with the words that label them, is a <dl> โ€” not a <p> of
  pipe-separated text, which announces one sentence of run-together values, and not a <ul>, which
  says these are four things of one kind rather than four labelled ones. Mark it the same way on
  every page that prints it: a footer that is a <dl> on page 4 and a sentence on page 5 tells a
  reader the two pages carry different things. What the page prints no label for has no term to
  write โ€” a foot that gives bare values is transcribed as what it is, and writing "Website" over a
  URL the page labelled with nothing puts a word of your own in a <dt>. And the page's own printed
  number is never one of these values: it is carried by the page-break marker's label, by the rule
  above, so a row for it here hands the reader the folio twice.

A page that prints the same content in more than one language gets the same treatment in each.
Every rule above applies to the second column exactly as it does to the first: where the English
steps are an <ol> the French steps are an <ol>, where one recipe's ingredients are a <ul> so are the
other's, and a sub-topic that earns a heading in one language earns it in the other. Structure that
stops at the first language is worse than none, because the document then looks handled to everyone
except the reader it failed. Mark each change of language with lang on the element that holds it โ€”
<section lang="ko">, or lang="es" on the single <td> that switches โ€” using the BCP 47 tag for the
language the page prints there. A page wholly in one language OTHER THAN ENGLISH changes language
nowhere, and is the case that needs the attribute most: put lang on every top-level element you emit
for it. The document you are writing into takes its language from the pages inside it, and can only
do that where they all say what they are: one fragment returned with no lang of its own leaves the
whole document declared English, so a Korean page is delivered as English text, pronounced as
English, to the reader who has no way to see that it is not.
An English page is the case that needs nothing, and the sentence above is not asking for it: English
is what the document declares when its pages give it nothing else to read, so lang="en" on the
elements of an English page changes what a reader is given in no way at all. A page that omits it is
correct and is not to be reported for omitting it. On an element that holds no text of its own โ€” an
<img>, an <hr> โ€” the attribute is meaningless whatever the language.
And transcribe that language; do not translate it. Returning a Korean page in English is not
accessibility work but a different document: those words are not words on the page, the original is
not recoverable from what you emit, and a mistranslation is invisible to exactly the reader who
would be relying on it. What a screen reader needs in order to pronounce the passage at all is the
lang attribute, which is why that is the rule. Say in the "log" field which languages the page
holds.

Where the prompt shows you your previous output for this page, that output is the starting point
and not a draft to replace. Change what you were asked to change โ€” the problem named, the feedback
given โ€” re-check that content against the image, and carry everything else over as it stands: the
same heading at the same level, the same table with the same cells, the same list, the same alt
text, the same lang. Re-deriving the page from the image instead is how the second pass costs a
reader what the first one got right, and nothing downstream can tell that it did: a level that
moved, a cell's list flattened, a <dl> turned back into paragraphs all arrive as this page's
content, and the version that had them right is not kept anywhere. If you can see that something
outside the problem is wrong, fix the problem, leave that alone, and say what you saw in the "log"
field.

If โ€” and only if โ€” this page contains a content type that a DEDICATED specialist agent would
handle clearly better than this general pass (something beyond the common types: paragraph,
heading, list, table, form field, image, quote, caption, footnote), include a
"suggested_agent". Suggest sparingly; omit it (or null) otherwise.
Sheet music is the example to reason from. A page whose content is musical notation cannot be
carried by a description of the staves: what a reader needs is the music โ€” an audio rendering, and
a machine-readable notation such as ABC or MusicXML โ€” and neither is derivable from one look at the
page, which is what a specialist agent is for. So name one, and do not write a measure-by-measure
account of the notation into alt text as a stand-in: "quarter note D, eighth note F sharp" for
forty bars is not the music, and is not usable by anyone.
Then render the page in full anyway. A suggestion is a request and not a delivery โ€” the agent you
name may not exist in this deployment, in which case nothing runs and what ships is exactly what
you returned. So transcribe every word the page prints (title, composer, tempo, lyrics, rehearsal
marks, the caption), put the score itself in a <figure> whose <figcaption> says what the image is โ€”
instrument, key, time signature, how many systems โ€” and say in the "log" field that the audio and
the notation are the specialist's part. A page held back to a stub for a specialist that never runs
is a page that ships as a stub.

Respond with ONLY this JSON:
{ "html": "<accessible HTML for the whole page โ€” body content only, no duplication>",
  "log": "notes, e.g. content cut off at an edge",
  "blank": true,
  "suggested_agent": { "name": "lowerCamelCase", "reason": "why a specialist is warranted" } }

"blank" belongs on a page with nothing on it and on no other page: omit it everywhere else rather
than sending false, and never send it for a page you could not read.

Your entire reply must be the JSON object and nothing else. Do not write any reasoning, preamble,
commentary or summary before or after it. Everything you have to say about this page goes inside the
fields the schema above lists, the notes this prompt asks you for included, and nothing goes outside
them.`;

export interface ExtractionResult {
  fragments: Fragment[];
  suggestions: { name: string; reason: string; image: string }[];
  // Source pages (1-based order) the delivered document has NO content for: their own
  // extraction threw and they are in the document as a failure marker โ€” see
  // `failedPage`. Empty on an ordinary run. Returned rather than only logged because a
  // document delivered with a page missing is a different deliverable, and the caller
  // records it alongside the run's other outcome counts.
  //
  // From `reExtractPages` this is the set the document ALREADY had, minus any page the
  // re-extraction filled in. A re-extraction that throws does not add to it: that path
  // only runs for pages which already have a fragment, so the page keeps the content it
  // had and the document is no less whole than it was. Those are reported as
  // `reextract_complete.failed` instead โ€” folding them in would tell a client its
  // document is missing a page that is in it.
  failedPages: number[];
  // Source pages the verifier REJECTED and the correction pass did not repair, so what the
  // document carries for them is content Iris named a defect in and never fixed (#328). Not
  // necessarily the rejected bytes โ€” the review loop runs afterwards and may rewrite a block on
  // such a page โ€” but nothing after this point re-asks whether a page matches its source, so the
  // rejection stands whatever the markup becomes.
  //
  // Disjoint from `failedPages` and not a rate of anything: those pages have no content at
  // all, these have content that is known to be wrong in a way Iris named. Kept apart for
  // the reason the two markers are kept apart in the document โ€” a page with nothing in it
  // is obviously incomplete, and a page whose table lost its rows looks finished.
  //
  // From `reExtractPages` this is the prior set with the re-extracted pages' own answers
  // substituted in: a page re-rendered and now accepted leaves the set, a page re-rendered
  // and rejected again stays, and a page the round never touched keeps whatever it had. A
  // re-extraction that THREW keeps its prior status too โ€” it is the prior fragment that
  // ships, so the prior verdict is the one that describes it.
  uncorrectedPages: number[];
  // Pages that WERE in `failedPages` and are not any more, because this re-extraction
  // produced content for them. Returned rather than logged here: "the document has this
  // page now" only becomes true once the round's document is persisted, and a round that
  // throws after re-extracting (in the Reader, the editor, the lint) leaves the client
  // holding the document that still has the hole. Logged by the caller, after the write
  // (pipeline/orchestrator.ts) โ€” diagnostics folds the event straight into
  // `pages_failed`, so a premature line there claims a document is whole when it is not.
  recovered?: number[];
}

// The content of the LAST fenced block, or the text as it stands.
//
// Any info string, not only `html`: a ```json fence is what a page agent writes when it
// wraps the envelope, and a regex that knew only `html` left the word "json" INSIDE the
// content it returned โ€” which is why every leaked envelope in issue #168 begins with the
// literal line `json`. `extractJson` has always known both spellings; this now does too.
//
// The last rather than the first, for the reason `extractJson` prefers the last object
// (util/json.ts): a model that drafts and then corrects itself sends both, and the bench logs
// have a page correction with FOUR fenced envelopes whose first three logs say they were
// abandoned ("RESTART", "Intermediate attempt abandoned"). Binding the first delivered a draft
// the model had already rejected (issue #170). This is the bare-HTML half of that fix โ€” the
// same reply shape, answered in markup instead of an envelope, reaches the page through here.
//
// Exported for test/envelope-as-content.test.ts: the four reply shapes this and `bareHtml`
// have to agree about were all read off real bench logs, and a unit test names them.
export function stripFences(t: string): string {
  const blocks = [...t.matchAll(/```[a-z]*[ \t]*\r?\n?([\s\S]*?)```/gi)];
  const last = blocks[blocks.length - 1];
  return (last ? last[1] : t).trim();
}

// A reply that is not the JSON envelope, but IS plainly the page's HTML.
//
// The page agent is asked for `{"html": "โ€ฆ"}`, and a model that answers with bare HTML
// instead should not cost a page โ€” that is the fallback this replaces, and it was right to
// want it. What it could not do is tell "answered with HTML" from "answered with anything at
// all": on a reply whose envelope did not parse it handed back the envelope, prose and all,
// and that text was delivered to the user as the page's content (issue #168). It is not only
// wrong content. The escaped markup inside a leaked envelope parses as tags whose ATTRIBUTE
// NAMES begin with a digit or a backslash, which is what took axe-core down on the same
// documents (#164) โ€” so one unreadable reply cost the page AND the whole document's
// accessibility verdict.
//
// So the raw text is accepted only when nothing about it suggests an envelope: it does not
// begin with `{`, it carries no `"html":` key anywhere, and it starts at a tag rather than
// at prose about the page. Anything else is a reply that could not be read, which is a
// reported outcome and not a page's content.
//
// Unfenced prose followed by markup ("Here is the page:\n<h1>โ€ฆ") is therefore refused, and
// deliberately: nothing here can say where the sentence ends and the page begins, and the
// old fallback's answer โ€” deliver both โ€” put a sentence Iris wrote into a document whose
// whole contract is that every word in it is a word on the page. Prose around a FENCED block
// costs nothing, because `stripFences` finds the fence wherever it starts.
export function bareHtml(text: string): string | null {
  const t = stripFences(text);
  if (!t || !t.startsWith("<")) return null;
  if (/"html"\s*:/.test(t)) return null;
  return t;
}

// What an unreadable reply looked like, for the log line. The point of the field is that the
// shapes have DIFFERENT remedies, so it is worth the few lines to name them apart: a
// `truncated_envelope` is the output ceiling (raise `max_tokens`), an `envelope` that will not
// parse is escaping the page's own punctuation (util/json.ts), `prose` is the agent answering
// conversationally and `empty_html` is it answering with no page at all โ€” both prompt problems,
// and nothing about the parser.
//
// `parsed` decides the first question, because it settles it: if the envelope was read, then
// whatever is missing was missing from a reply this pipeline understood, and pointing the
// operator at escaping would be pointing at the one thing that worked. What `empty_html` means
// is now narrower than it was: a blank page DECLARED as one is `declaredBlank` below and not a
// failure at all, so what reaches this shape is an envelope that carried no page and did not
// say why โ€” the model that gave up, which is the prompt problem this field names.
//
// `bare_html` is the sixth value and it is the one the other five could not say (#365). A model
// that answers with the page's MARKUP instead of the envelope is a reply `bareHtml` accepts and
// delivers, so it is not a parser problem and not the agent answering conversationally โ€” and
// without this value it was labelled `prose`, whose stated remedy above is the prompt. Measured
// on the 180 corrections of `runs-extract100-95ca64c` that logged a reply: 24 of them answer in
// bare markup (19 of kimi-k2.5's 64, 5 of sonnet-4-6's 52, 0 of luna's 64), so it is a model
// trait rather than an oddity, and both of that round's page-shaped correction truncations are
// in the group. Order is load-bearing only in that the envelope tests stay first, so every label
// this function gave before is the label it gives now: over those 180 replies the only change is
// the 24, and cutting each of them at 60% of its own length โ€” a truncation, which is the case the
// correction path calls this for โ€” gives 152 `truncated_envelope`, 27 `bare_html` and 1 `prose`,
// where before it gave 152 and 28 `prose` for everything else.
//
// The rule is `bareHtml`'s in substance โ€” a reply is markup when nothing about it suggests an
// envelope โ€” plus one thing `bareHtml` has no need of: an opening fence with no closing one.
// `stripFences` returns the last COMPLETE fenced block and otherwise the whole text, so a reply cut
// by an output ceiling comes back with its own opener still on the front. That is precisely the
// reply this value is for. Both page-shaped truncations in that round begin
// "```html\n<hr role=\"doc-pagebreak\"โ€ฆ", and asking `bareHtml` about them returns null and labels
// them `prose` โ€” the label the value was added to stop. Widening `bareHtml` instead is the wrong
// place: it is only ever asked about a reply that arrived whole, and loosening what the pipeline
// DELIVERS as a page to fix a log line is a trade this does not need to make. A reply that is
// nothing but an opener reads as `empty` rather than `prose`, which is what it is.
//
// What the value CANNOT say is where the output went, because this reads the reply's beginning.
// A correction that starts with the page and then talks about it โ€” `acir-p083` in that round,
// whose tail is the model re-opening a fence and second-guessing its own table โ€” is `bare_html`
// here exactly like one that transcribed to the last character it had room for. That is what
// `reply_tail` is on the log line for, and it is why nothing counts narration off this field.
//
// On `page_no_output` it collects one more thing, and the remedy there is the opposite one: markup
// that arrived fine and carried nothing a reader receives. A bare `<!-- blank page -->` or a lone
// page-break marker is #219's own spelling of a blank declaration, `bareHtml` accepts it, and it
// reaches that line through `carriesContent` rather than through the parser โ€” so it is a prompt
// problem wearing the label of a parser-refused page. `dropped` is on that same line and carries the
// markup, which is what tells the two apart; nothing here can, because the shapes are identical.
// Case-insensitive because `stripFences` above is (`/gi`), and for its reason: the info string
// arrives both ways. A ` ```HTML ` opener left on the front fails the `<` test below and lands the
// reply in `prose`, which is the one label this is here to stop.
const UNCLOSED_FENCE = /^```[a-z]*[ \t]*\r?\n?/i;

function replyShape(text: string, parsed: unknown): string {
  if (parsed) return "empty_html";
  const t = stripFences(text).replace(UNCLOSED_FENCE, "").trim();
  if (t.startsWith("{") || /"html"\s*:/.test(t)) return /}\s*$/.test(t) ? "envelope" : "truncated_envelope";
  if (t.startsWith("<")) return "bare_html";
  return t ? "prose" : "empty";
}

// A page the agent says has nothing on it: `html` PRESENT and carrying nothing a reader receives,
// with a `log` line saying so. Both halves are load-bearing.
//
// "Carrying nothing" rather than "empty", because the reply the prompt asks for is not the only reply
// the model writes. Across 818 initial renders in the bench logs, 78 delivered a fragment with nothing
// in it for a reader; 45 spelled it as the empty `html` this asked for, and 33 put the blankness into
// markup instead โ€” 18 a bare page-break marker, 13 a comment (`<!-- blank page -->`), 2 an empty
// paragraph โ€” every one of them with a log saying the page is blank (issue #219). Read as content,
// those 33 cost three things: `pages_blank` counted them as pages that produced markup, the document
// carried the comment or the empty `<p>` or an anchor claiming a folio the paper never printed (see
// the marker paragraph below, which recorded this shape from #179 and left it delivering), and the
// refusal at `renderPage` โ€” the one thing on the re-extraction path that stops a declaration deleting
// content Iris already holds (#194) โ€” could not see them at all, so a re-extraction answering
// `<!-- blank page -->` for a page with content replaced the content with the comment.
//
// What a reader receives is `carriesContent` (correction.ts), which is `visibleText` plus the elements
// that are content with no text in them, plus the attributes that make a neutral element one of those
// (#224). So a comment, an empty wrapper and a page-break marker are nothing; a picture, a table, a
// form control and a `<div role="img">` are something. PROSE is something too, deliberately:
// one further render answers `<p><em>This page is blank.</em></p>`, and a page that prints "This page
// intentionally left blank" is a page whose correct transcription is that sentence. Nothing here can
// tell the two apart, so the sentence is delivered as the page said it.
//
// This used to be a reported page failure, and the comment here defended that on the grounds
// that `agents/page.md` did not say what to return for a page with nothing on it, so `""` was
// as likely to be a model that gave up as a page that is blank. Then round 7 of the bench
// measured the rate: six pages of 100, in three of four documents, every one a well-formed
// 155โ€“210-character envelope saying correctly that the page is blank, every one delivered as a
// `@page-failed` marker and counted as a lost source page (issue #179). A document that
// declares a hole where there is no hole is its own defect โ€” it costs a reviewer a glance per
// page and it makes the run's own report untrue โ€” so the prompt now names the case and this
// reads the answer it asks for.
//
// The key must be PRESENT: `{"log": "no content"}` is a reply that did not answer the question,
// and it stays a failure, which is the distinction the issue's fallback asks for. And the `log`
// must SAY the page is blank, not merely be non-empty: the commonest shape a vision model gives
// up in is an empty `html` with a sentence about why, and `{"html": "", "log": "the scan is too
// dark to resolve any text"}` is the reply that most needs a human to look at the page. Read as a
// declaration it would leave nothing in the delivered document to look at โ€” no marker, no notice,
// no entry in `pages_failed` โ€” and `pages_blank` would positively assert the paper was empty.
// The fidelity check does not cover that case either: it is shown the same unreadable image by
// the same model, so it agrees there is nothing there.
//
// So the test is positive and the doubt is fatal. `BLANK_LOG` wants the log to assert emptiness
// in some words; `UNREADABLE_LOG` and `DEGRADED_IMAGE_LOG` refuse the declaration whatever else
// the log says, so a hedge ("appears blank, though the scan is very faint") and a description of
// the image's own condition ("the page is very dark and appears empty") are failures. A blank
// page whose log is phrased outside both patterns is reported as a failed page, which is the
// direction to be wrong in: a page wrongly reported as failed costs a glance, a page wrongly
// dropped costs the page.
//
// That holds whatever the fragment looked like: how the model spelled its empty page does not decide
// the routing, so a comment or a bare marker with a REFUSED log is the failed page an empty `html`
// with the same log already was. Nine of the 33 markup-spelled declarations above are refused that
// way, and their wordings are #220 โ€” two read as self-contradictions that are not there, seven the
// #190 case in phrasings the exemption below does not reach. Until that issue is settled those nine
// are pages reported as holes in documents that have none, which is the cost this paragraph accepts
// and #179 measured; what they are not any more is invisible, since a refused declaration now reaches
// `page_no_output` with `blank_vetoed` on it whichever way the fragment was written.
//
// The one place the doubt is not fatal is a veto word MODIFYING the marks on the paper
// (`MARKS_PHRASE` below): "a few faint specks, no legible text" describes an empty sheet, and
// reading `faint` there as doubt about the scan cost the bench four pages. Only the phrase is
// exempt, so the same word about the scan itself in the same sentence still refuses, and a log
// that anywhere says the reading failed is not exempt at all โ€” the exemption narrows what the
// veto words are ABOUT without moving where a real doubt lands. Not done here is the
// issue's preferred fix, sending a vetoed declaration to the fidelity check instead of reporting
// the page failed: the paragraph above is why โ€” the check is the same model on the same image, so
// on the unreadable page it agrees there is nothing there, and a page delivered blank on that
// agreement has no marker, no notice and no entry in `pages_failed` to look at. The failure path
// is the disclosure. What a refusal now leaves behind is the `blank_vetoed` field on
// `page_no_output`, so which word refused which page is a log read rather than an investigation.
//
// What is NOT done here is the issue's preferred fix โ€” a page-break marker labelled with the
// page's position in the file. The prompt forbids exactly that ("use the number the page shows
// (iv, 5, A-3), never the position of the image you were given in the file"), and it is the
// same rule that produced the split the issue reports: the three blank pages that were
// delivered carried `aria-label="Page 4"` from the file position, and the two that failed had
// obeyed the rule and emitted nothing. An anchor named for a position rather than a printed
// folio claims the document's page 4 is here, which on front matter or an insert is false โ€” and
// the marker's whole contract is that its label is the number the paper shows. A page that
// prints no folio has no anchor whether it is blank or not, so the gap the issue notes in the
// anchor sequence is not new and not a defect.
// The page has nothing on it. Wide enough for the phrasings the bench replies used and for the
// wording the prompt now asks for ("say in the log field that the page is blank"), and no wider.
const BLANK_LOG =
  /\b(blank|empty|no (visible |printed |discernible )?(content|text|markings?|marks|words)|nothing (on|printed|visible|at all)|intentionally left blank)\b/i;
// The page could not be READ, whatever else the log says about it. Checked second and given the
// last word, because these two overlap in exactly the reply that must not be trusted: a page the
// model calls blank because it cannot make anything out is not a page it read.
//
// Two families, and the second is why this is not a list of ways to say "I could not": a model
// describing the IMAGE's condition ("the page is very dark and appears empty", "low resolution
// scan; no text") has told you why its answer is unreliable without ever saying it failed. Those
// words veto the declaration too. Both lists lean wide โ€” a page wrongly reported as failed costs a
// glance, and a page wrongly dropped costs the page โ€” but only over words that carry doubt: `too
// low` vetoes with or without an infinitive after it, while `quality` is scoped to the poor kind
// (it matches "high quality" as readily as the other), and geometry is not legibility, so a
// rotated or skewed page says nothing about whether its words could be read.
const UNREADABLE_LOG =
  /\b(illegible|unreadable|not legible|could ?n[o']?t|can ?not|can'?t|unable|failed|truncat\w*|too \w+ to|too (low|light|dark|faint|poor|noisy|blurry)|blurr\w*|obscur\w*|resolves?|corrupt\w*|partial\w*|error)\b/i;
//
// `smear`, `streak` and `blotch` are here for the reason `dark` is: each names something that can lie
// OVER content, and a sheet with a streak on it is one where "no text is visible" and "the text is
// covered" are the same sentence. `MARK` excludes all three from its exemption on that argument, but
// the argument bought nothing while none of them was a doubt word โ€” "a streak covers most of the sheet;
// no text is visible" had no veto word in it and was delivered blank (issue #226).
//
// The NOUN is what decides, not what the sentence says about its extent: "an ink smear in the lower
// corner, nothing else" refuses, though a smear in a corner covers no more than the `smudges` `MARK`
// exempts two lines up. Reading extent would mean trusting the same sentence whose reliability is in
// question, and the words are near-synonyms a log picks between freely โ€” so the split is which word a
// log reaches for and nothing finer: `smudges` is #193's, with corpus wordings behind it, and these
// three are the ones a covering is described with. What that costs is a glance at a page that was
// fine, which is the trade this file makes everywhere; what reading extent would risk is the page.
//
// Their siblings `spots`, `stains` and `shadows` are deliberately NOT here, and the comment on `MARK`
// says why: they name something in one place, which a page with writing on it is not described as
// having, and `spots` is a word a log may use for the specks it is already allowed to describe.
const DEGRADED_IMAGE_LOG =
  /\b(dark|faint|washed|blurry|blurred|noisy|noise|grainy|pixelat\w*|low[- ]?res\w*|resolution|smear\w*|streak\w*|blotch\w*|(poor|low|bad|degraded)( \w+)? quality|quality (is|was|of)|(out of|not in|soft) focus|did ?n[o']?t load|not load\w*)\b/i;

// The marks a scanner leaves on an empty sheet, and the exemption for describing them. The words
// for the marks and the words for a bad image overlap almost completely (`faint`, `noise`, `dark`),
// and round 9 of the bench lost four blank pages of 100 to that overlap: an agent that answered the
// prompt's request for a description ("Specks/dots are visible on the page but do not resolve into
// any characters or content") was punished for it, while an agent that said only "Page is blank."
// was believed (issue #190). Two pages of one document opened with a verbatim identical sentence
// and only the one that went on to explain itself was refused, so what was being measured was the
// wording.
//
// What is exempt is a PHRASE, not a sentence, and that distinction is the whole safety of this: the
// veto words come out where they modify the marks and stay in everywhere else, so a log that
// mentions the marks AND the state of the scan in one breath โ€” "the scan is blurry, showing only
// faint specks and no legible text", which is how these logs are actually written โ€” still refuses
// the declaration on `blurry`. Dropping the sentence instead would deliver that page blank with no
// marker and nothing in `pages_failed`, which is the failure mode the whole file is built against.
// None of these nouns is a veto word, which is what makes the exemption below narrow: the only
// words it can ever remove are the veto words used AS MODIFIERS of one of them. Two exclusions are
// deliberate, and both were tried the other way and reverted:
//
// `marks` and `markings` are here only as `stray marks`, never bare, because `agents/page.md` uses
// the bare noun for the OPPOSITE case: "where marks do not resolve into characters even then, write
// `[not legible]`" is the instruction for a page that HAS content the agent could not read, where
// empty `html` is the wrong answer. `handwritten marks`, `pen marks` and `blurry marks` therefore go
// on refusing, and `NOT_LEGIBLE_TEXT` below can go on counting `markings` as a name for something
// worth reporting โ€” both as the noun it strips and in `PAGE_BEARS`, where it needs a denial in front
// of it โ€” which bare `marks?` here contradicted.
//
// `spots`, `streaks`, `blotches`, `smears` and `stains` are not here, for the same reason `shadows`
// is not: a dark streak or a dark spot is a condition of the capture rather than something on the
// sheet, and either can cover content โ€” which is why `dark` is a veto word at all. Adding them made
// "the scan shows dark streaks" a blank page. `smudges` stays, as a mark left on the paper.
//
// That argument only bit on the three that name something a reader would have lost content under, and
// only once they were doubt words in their own right: `smear`, `streak` and `blotch` are in
// `DEGRADED_IMAGE_LOG`, so leaving them out here is what keeps them refusing (issue #226) โ€” and the
// comment there says why the noun decides that on its own, without reading how big the log says it is.
// `spots`, `stains` and `shadows` are
// out of both lists on purpose, and the reason is narrower than the one above: they name something in
// ONE PLACE, which is what a scanner leaves and not how a covered page is described โ€” and `spots` is a
// word a log may reach for to name the specks it is already allowed to describe. So a log that says
// only "spots are visible, no text" is believed, and "a streak covers most of the sheet" is not.
//
// What lets bare `marks` in behind one of these words is what the word says about them: `stray`,
// `scattered`, `isolated`, `random` and `residual` place the marks nowhere in particular, which is
// what a scanner leaves and not how a page with writing on it is described. `stray` was here alone
// and the other four are #220's wordings ("Only faint, isolated marks are visible โ€ฆ that do not
// resolve into any characters"). Bare `marks` still refuses on its own, so `handwritten marks`, `pen
// marks` and `blurry marks` are the content-bearing pages #193 kept them for.
const SPARSE = String.raw`stray|scattered|isolated|random|residual`;
// `noise` is a veto word and standing alone it is a claim about the image, but named for the thing
// that made it โ€” `scanning noise`, `scanner noise`, `scan noise`, `compression noise` โ€” it names the
// marks instead: it is the class the specks belong to, which is what "consistent with scanning noise"
// says about them. `the scan is noisy` and `there is noise in the scan` have no such word in front of
// the noun and go on refusing (#220).
//
// `compression` is here because #220's three words are the CAPTURE and a log names the PROCESS just as
// readily: "Page is blank apart from minor scanning artifacts (specks and compression noise)" was a page
// lost for real โ€” `blank_vetoed: ["noise"]`, `page_no_output`, `page_extraction_failed` โ€” in `runs-231`
// on 2026-08-27 (#429). What decided it was the head word and nothing else: the same reply with
// `scanning noise`, with `dust` in place of the noise, or with the noise dropped to `compression
// artifacts` is honoured today, so the licence is attached to a word rather than to the list or the
// punctuation around it.
//
// Which words those are is a measurement, not a guess. Over the 205 replies on disk whose HTML carries
// no content, every `noise` in the log has a head word in front of it and there are four of them:
// `scanner` (2), `dust` (2, in the joined tail below), `scanning` (1) and `compression` (1). So this list
// now covers all four wordings the corpus contains and `noise` never appears bare in a blank page's log.
// The 3,935 replies that DO carry content use the word ten times and never about an image โ€” `screen-reader
// noise`, `presentational noise`, `avoid noise` โ€” which is why no reading of `noise` here can reach them.
//
// `image noise`, `sensor noise` and `jpeg noise` are deliberately NOT here, and the reason is the one
// above rather than their absence from the corpus: `image` names the thing whose quality is in question,
// so `image noise` is the claim #220 kept refusing, and the other two would be a list extended by
// imagination past the wordings that exist.
const MARK = String.raw`specks?|speckles?|speckling|flecks?|dots?|dust|debris|smudges?|blemishes?|artifacts?|(?:${SPARSE})\s+mark(?:s|ings?)?|(?:scan|scanner|scanning|compression)[\s/-]?noise`;
// What may stand between a quantifier and that noun. The veto words are here on purpose โ€” `faint
// specks` is the paper and `faint scan` is the image โ€” alongside the words that are not veto words
// at all, because the run has to reach the noun in one piece to match ("a few scattered
// specks/dots", "scanning artifacts").
const MARK_MODIFIER = String.raw`faint|light|pale|grey|gray|dark|darker|noisy|grainy|blurry|blurred|washed(?:-out)?|pixelated|tiny|small|minor|stray|scattered|random|isolated|residual|scan|scanner|scanning|dust|paper|toner|ink`;
// A phrase whose head is one of those nouns, with its quantifiers and modifiers. Replaced with a
// space before the veto lists run, so a doubt word inside it is not read as doubt about the scan โ€”
// and a doubt word anywhere ELSE in the same sentence still is, which is the whole difference
// between this and dropping the sentence: "The scan is blurry, showing only faint specks and no
// legible text" loses `faint` and keeps `blurry`.
//
// `noise` is only in the joined tail (`dust/noise`, `specks, noise`) and never a head, because it
// is a veto word itself and standing alone it is a claim about the image: "the scan has noise"
// must go on refusing the declaration.
// A comma may stand between two of the modifiers, because a stack of them is written as a list as
// often as not: "Only faint, isolated marks are visible" is one phrase, and matching it from
// `isolated` left `faint` behind to veto the page as a doubt about the scan (#220). The head is still
// a marks noun, so the comma widens what may dress those nouns and nothing else โ€” "The page is dark,
// with faint specks" keeps its `dark`, since `with` is not a modifier and the phrase starts after it.
//
// But a comma ends a clause as readily as it separates a list, and that is the one thing the stack
// must not reach across: "the scan is grainy, faint specks are all that appear" and "Scan quality:
// dark, blurry, faint specks throughout" put a doubt word about the IMAGE exactly where a modifier of
// the marks goes, and stripping it ships a grainy capture as a blank page with no marker on it. So
// the comma'd form may not open the clause that describes the capture, while the form without a comma
// is untouched from before #220 and goes on exempting `the visible artifacts are faint specks` as it
// always did. Written as two alternatives rather than one guarded stack for that reason: only the comma
// needs the guard, and the plain form is left free to match further along the same sentence โ€” which is
// what makes a refused comma'd stack behave exactly as it did before #220 rather than keeping words the
// old pattern stripped.
const MARK_QUANTIFIER = String.raw`(?:(?:a|an|the|only|just|some|few|several|couple|of)\s+){0,4}`;
// The guard asks about the CLAUSE, not about the token in front of the stack, because the token is a
// class with no end to it: `is grainy,` was closed by naming the copula, `is very grainy,` by naming
// the degree word, and `is noticeably grainy,` / `is a little dark,` / `is only dark,` were next โ€”
// three roads into the same wording, each of which a list closes one instance of. So the stack may not
// have a copula, a colon or a dash anywhere to its left within its own clause: what those say is that
// the sentence is describing something already named โ€” the scan, the image, the quality of the capture
// โ€” and a modifier of that thing is not a modifier of the marks. Bounded to the sentence, since
// `[^.!?;\n]` cannot cross a full stop, a semicolon or a line break, and bounded in length for the
// reason `DENIAL_STATEMENT_MAX` is: the work per match stays flat.
//
// A hyphen counts only with a space in front of it โ€” spaced it is the dash somebody typed on a
// keyboard without one, unspaced it is inside `washed-out` and inside the phrase's own separator.
//
// Which punctuation resets the reach and which is evidence is the whole of what this decides, so it is
// worth stating rather than leaving to be read off the character class. A full stop and a semicolon
// RESET it: they end a clause, and a stack at the head of a new one has nothing to its left to describe.
// A colon and a dash do not reset it, because they are the evidence โ€” they say what follows describes
// the thing just named. So the same claim gets opposite verdicts on the delimiter the model typed:
// "This page is blank; only faint, isolated marks are visible" declares blank and "Page is blank โ€” only
// faint, isolated marks are visible" does not, because the dash's own reading is that `blank` is what
// the stack is about. Both are pinned, and the second is the cost this guard pays: that page is
// reported failed, which is a glance rather than a page.
//
// A line break resets it too, and on a different ground โ€” layout, not grammar. It ends a note the way a
// full stop ends a sentence, and it OUTRANKS the evidence: `[^.!?;\n]` cannot span it, so a colon at the
// end of a line does not reach the line below, while the same colon inline does ("Page is blank:\nonly
// faint, isolated marks are visible" declares blank; "Page is blank: only faint, isolated marks are
// visible" does not). Both are pinned. That ranking is deliberate and it is a claim about how these logs
// are laid out rather than about what a colon means: a colon at the end of a line is introducing a list
// โ€” "Notes:", "Scan quality:" โ€” and the lines under it are its items, so reading it as a description of
// what the colon named would refuse a blank page for the shape of the note it came in. Inline, the same
// colon has the clause it governs on the same line, which is the case the evidence reading is for.
//
// One thing that follows and is NOT closed: a doubt word leading a stack that opens its own sentence is
// stripped, because there is nothing to its left to say otherwise โ€” "Dark, blurry, faint specks
// throughout, no legible text" is read as the marks and would have refused on `dark` before #220. That
// position is exactly the one #220's nine need (each starts the sentence its marks are in: "Only faint,
// isolated marks are visibleโ€ฆ", "A few faint specks or artifacts are presentโ€ฆ"), and in it the two
// readings are indistinguishable โ€” `blurry specks` is grammatically the marks, which is the reading the
// exemption exists for, and no wording in the sentence says the model meant the capture. Closing it
// would cost #220 its wordings to catch a sentence nobody has written yet.
const NOT_CLAUSE_HEAD = String.raw`(?<!(?:\b(?:is|are|was|were|be|been|being|appears?|appeared|seems?|seemed|looks?|looked|shows?|showed|showing|remains?|has|have|had)\b|[:โ€”โ€“]|\s-)[^.!?;\n]{0,200})`;
// ONE word that is on none of these lists may stand between the run and its noun, and it is the one
// place in this phrase where an unlisted word is allowed. `MARK_MODIFIER` is a hand-written list, so a
// stack breaks on the first adjective nobody thought of and puts the doubt word back in the veto's
// scope: "Only faint, indistinct specks are visible" is refused today where "Only faint specks are
// visible" is honoured, and the difference is one word the list happens not to carry (#429). The words
// in that position are adjectives of indistinctness, size and shape โ€” `indistinct` is the one the corpus
// has โ€” and there is no end to them, which is why this is a slot rather than five more entries: adding
// words to a list is what produced the cliff (#220's `SPARSE`, #371's objection).
//
// Bounded four ways, and each bound is what keeps the widening from reaching a doubt word it could not
// reach before:
//
//   POSITION. The slot sits IMMEDIATELY before the marks noun and nowhere else, so a veto word can only
//   be reached across it when the log wrote `<veto> <word> <marks noun>` with nothing else between. That
//   is what keeps "The page is dark, with faint specks" refusing on `dark` โ€” `with` stands before
//   `faint`, not before `specks`, and only one slot exists โ€” which is the case pinned above and the one
//   an unlisted word ANYWHERE in the stack would have taken.
//
//   COUNT. One, because a second would be the same reach again: the guarantee above is per word.
//
//   CLAUSE. The slot is available only where `NOT_CLAUSE_HEAD` holds โ€” no copula, colon or dash to the
//   left within the clause โ€” in the plain form as well as the comma'd one, which is the guard #220 wrote
//   for the same reason: a stack behind `is` describes something the sentence already named, and the
//   marks are not it. Without this bound the one extra word of reach is enough to cross that boundary in
//   any sentence that puts a word between the two, since `<doubt> <any word> <marks noun>` is what "the
//   scan is noisy with artifacts" and "the image is grainy background specks" both are: the whole doubt
//   leaves with the phrase and a page with a bad capture is delivered blank. Only the slotted form is
//   guarded โ€” base's own reach is not what this widens, and #220's plain form stays as it was.
//
//   WHAT BECOMES OF IT: the slot's word is handed BACK into the stripped scope instead of being removed
//   with the phrase around it, so every rule that reads that scope still reads the word. That is the
//   third bound, and it holds by construction rather than by enumeration: `faint, streaked specks` loses
//   `faint โ€ฆ specks` and keeps `streaked`, so `DEGRADED_IMAGE_LOG` goes on refusing it without this
//   function naming that list; and a name for text is still in the scope the contradiction check reads,
//   so this cannot silence an affirmation (that check runs over the same stripped text, so a slot that
//   ATE `handwriting` could ship a page with writing on it in silence โ€” the #194 defect).
//
//   Handing back is not enough where the word is not dressing ANYTHING, though, because what leaves with
//   the phrase can be the whole doubt: in "the scan is noisy with artifacts" the slot holds `with`, and
//   handing `with` back keeps a preposition while `noisy` โ€” the only word in the log that doubts anything
//   โ€” goes. So a determiner, preposition, conjunction, copula or negator falls back instead, and that
//   class is where POSITION alone stops being the argument: `with` DOES stand immediately before the
//   marks noun here, where in "The page is dark, with faint specks" it stands before `faint`.
//
//   An earlier revision asked instead whether the word was a doubt word or a name for text, and that
//   pair of enumerations was wrong twice over on first review โ€” both times by naming a list too narrow
//   for the question. It asked `DEGRADED_IMAGE_LOG` where the vetoes are `UNREADABLE_LOG` UNION
//   `DEGRADED_IMAGE_LOG` (see the caller of `vetoScope`), so "Only a few faint, partial specks are
//   visible" lost its `partial` and became a declaration; and it read names for text off `TEXT_NOUN`, a
//   NOUN vocabulary, in the one position where a log writes the participle, so `handwritten smudges`,
//   `stamped dots` and `typed specks` walked through the check meant for exactly them. Handing the word
//   back cannot go wrong in that direction, because it decides nothing about the word.
//
// SO THREE WORDS ARE NOT HANDED BACK, and all three so that the widening may only ever ADD to what base
// stripped:
//
//   - a word base ITSELF removed from this span, which is asked of base's output and not of a list, for
//     the reason the notes below the function give: the routes by which the slot can capture a word base
//     was already stripping are not enumerable, and one of them is a page on the corpus.
//
//   - a function word, per the paragraph above.
//
//   - a name for text in a form `contentAffirmed` cannot read: the participles and attributives that
//     `NAMES_TEXT_FORM` carries, which handing back would leave in a scope with nothing to affirm off
//     them. Those fall back to the phrase WITHOUT the slot, over the same span.
//
// That fallback is the slot-less phrase and never the span left alone, which is not a detail: leaving the
// span alone โ€” the obviously conservative move โ€” takes back strips that were never in question. The slot
// matches `and` in "(specks and scanning noise)", so returning that whole match unchanged kept `noise` in
// scope and #220's own wording started refusing, with fix A above broken by it.
//
// Measured against every no-content reply on disk, this slot moves NOTHING: the one
// log that contains `indistinct` writes it in front of bare `marks`, which is not a marks noun for the
// reason #193 gave, so it goes on refusing. It is here for the wording, not for a rescue โ€” same standing
// as `INPUT_SUBSTRATE` below, and stated the same way rather than left to look like a fix that paid.
const MARK_ADJECTIVE = String.raw`[A-Za-z][A-Za-z'-]*`;
const MARK_SLOT = String.raw`(${MARK_ADJECTIVE})[\s/-]+`;
function marksPhrase(slotted: boolean): RegExp {
  const commad = (slot: string) =>
    String.raw`${MARK_QUANTIFIER}${NOT_CLAUSE_HEAD}(?:${MARK_MODIFIER}),[\s/-]+(?:(?:${MARK_MODIFIER}),?[\s/-]+){0,2}${slot}`;
  const plain = (slot: string, guard: string) =>
    String.raw`${MARK_QUANTIFIER}${guard}(?:(?:${MARK_MODIFIER})[\s/-]+){0,3}${slot}`;
  // The plain form appears TWICE in the slotted pattern โ€” once with a mandatory slot and the clause guard,
  // once as base wrote it with neither โ€” rather than once with an optional slot, because the guard belongs
  // to the widening and not to base's reach. Written as an optional slot, `NOT_CLAUSE_HEAD` would sit in
  // front of matches base makes with no slot in them at all, and #220's comment above says why the plain
  // form is deliberately unguarded. Two alternatives, tried in this order, give the guard exactly the span
  // it is about: where it fails, the slot is simply not available and base's branch matches.
  const branches = slotted
    ? [commad(String.raw`(?:${MARK_SLOT})?`), plain(String.raw`(?:${MARK_SLOT})`, NOT_CLAUSE_HEAD), plain("", "")]
    : [commad(""), plain("", "")];
  return new RegExp(
    String.raw`\b(?:` + branches.join("|") + ")" + `(?:${MARK})(?:\\s*[/,&]\\s*(?:${MARK}|noise))*`,
    "gi",
  );
}
const MARKS_PHRASE = marksPhrase(true);
// The same phrase with no slot in it: what a refused slot falls back to, and what this file matched
// before the slot existed.
const MARKS_PHRASE_LISTED = marksPhrase(false);
// What a removed phrase leaves behind. Every strip in `vetoScope` used to leave a bare space, and a
// space is indistinguishable from the space that was already there โ€” so by the time anything reads the
// scope, the fact that WORDS WERE CUT OUT OF IT is gone. This is that fact, carried in the one form
// that no existing read can see.
//
// It is a whitespace character on purpose, and that is the whole safety argument. `\f` is in `\s`, so
// every pattern in this file that puts `\s+` or `[\s/-]+` between two words matches across it exactly
// as it matched across the space; `[^.!?;\n]` crosses it, so the clause guards reach as far as they
// did; `words()` starts a token on `[A-Za-z]` and cannot start one on it; `\b` sees a non-word
// character either way; and `contentAffirmed` splits statements on `[.!?;\n]+`, which it is not in. So
// the marker moves NOTHING by itself, and that is measured rather than argued: an arm carrying the marker
// with BOTH reads of it removed fails 2 of this repo's 1,649 tests โ€” the two whose pins are those reads โ€”
// and moves 0 of the 204 blank declarations on disk. The one read that wants it looks for it in the
// statement's TEXT rather than in its tokens.
//
// Why anything wants it: #440. `Blank page; text` refuses the declaration off a bare noun with no
// determiner, no count and no predicate, and the obvious guard โ€” a statement of one token is not an
// assertion โ€” was written and reverted in `c43dff9`, because after this strip `Handwriting smudges.`
// is also one token, and that phrase is #435's own. One token is not one word, and this is the
// difference between them.
//
// Not the alternative #440 also offers โ€” handing `verblessAffirmation` the raw log beside the scope โ€”
// because the strip is what every denial read in this section is reasoning about, and a read given both
// has two answers available and no rule for choosing. This says only what the scope itself lost, where
// it lost it.
//
// A log cannot forge one: `vetoScope` deletes `\f` and `\v` from its input before inserting any.
const PHRASE_GONE = " \f ";
// The marks nouns that ONLY THE CAPTURE leaves, which is what decides #439.
//
// #439: `Page is blank. Print artifacts are visible.` is reported as a lost page. `print` is a name for
// text (`TEXT_NOUN` carries `print(?:s|ing|ed)?`), so the slot below hands it back to base โ€” and base
// removes the mark head and nothing else, leaving the word that was DRESSING that head standing where a
// subject goes, with the head's own verb behind it. The quote on the failure is `affirmed: "print are
// visible"`, and the missing noun is what makes two words look like a sentence. Four of the issue's ten
// wordings predate the slot and predate #438's fragment read (verified against a worktree at `8c9ef0b`:
// identical before and after), because base leaves the word standing whether or not the slot ever looked
// at it.
//
// The HEAD and not the dresser, which is the opposite of what the issue asked for and is the whole of
// what this decides. #439 asks for a list of the text words that can dress a mark as capture noise, and
// `print` cannot be one: `print artifacts` is the scanner's and `print smudges` is smudged printing,
// which is a page with content on it. That is not an argument, it is a PIN โ€” #431's grid asserts
// `Page is blank. Only print smudges are visible.` refuses the declaration, for eight names for text
// against eight more in the other part of speech, and a rule about the dressing word broke one of its
// sixteen cells. Turned around, the same rule needs no exception for it: `artifact`, `debris` and `dust`
// are words for what nobody put there, and no wording a person's marks are described with is on the
// list. `smudges`, `specks`, `dots`, `flecks`, `blemishes` and `marks` all stay where they are, which is
// #435's own half of this (`handwriting smudges`, `cursive smudges`, `stamped dots`).
//
// Which nouns, as a measurement rather than a guess. Over all 3,747 page replies with a log, 76 put a
// name for text immediately in front of a marks noun as `MARK` defines one, in 81 occurrences of 9
// spellings, and the head is `artifact(s)` in 74 of them and `dot(s)` in the other 7. So the list covers
// the head the corpus actually writes, and `printed dot leaders` โ€” a real corpus statement, and
// typographic CONTENT โ€” keeps the reading it had. Widening the net to bare `mark(s)` and bare `noise`,
// which `MARK` excludes for #193's reason, adds 9 replies and 8 wordings and moves none of them.
//
// What that costs and what it buys, over the same corpus: 0 of the 204 blank declarations on record change
// verdict, so nothing shipped moves either way; each wording transplanted into a declaration in three
// frames flips 18 of 51 cells, every one of them an `artifact(s)` head and every one REPORTED -> blank โ€”
// and every one of the 18 refused on base by a CONTRADICTION, none of them by a doubt word, which is the
// bound below measured rather than argued. Put any `DEGRADED_IMAGE_LOG` word in front of the phrase
// (`blurry`, `faint`, `grainy`) and 0 of the 51 flip: all 51 refuse on both arms. The two whole corpus
// SENTENCES that flip are one of each kind โ€” `Removed three stray dots (printing
// artifacts/dust specks โ€ฆ)` is a blank page now read as one, and the sentence about an interpunct
// transcribed as a character is a page with content on it that would now be declared empty. Neither reply
// declares blank, so both are latent, and the second is the shape of what this can cost.
//
// NOT added to `MARK_MODIFIER`, where the mirror image of this idea already lives (`scan`, `scanner`,
// `scanning`, `dust`, `paper`, `toner`, `ink` are the dressers, these are the heads). That list is read a
// second time by `CONT_CORE`, as what may CONTINUE a denial after a marks noun, and a wider continuation
// there exempts more sentences from the veto lists โ€” the direction that loses pages (#190). Read only
// where the slot hands a name for text back, this can only take an affirmation away, and it takes no
// veto with it: `artifact`, `debris` and `dust` are in neither `UNREADABLE_LOG` nor
// `DEGRADED_IMAGE_LOG`, which is pinned rather than asserted.
const CAPTURE_ONLY_MARK = String.raw`artifacts?|debris|dust`;
function marksPhraseStrip(match: string, ...groups: unknown[]): string {
  const adjective = (groups[0] ?? groups[1]) as string | undefined;
  if (adjective === undefined) return PHRASE_GONE;
  const word = adjective.toLowerCase();
  // What base does to this same span, which is both the fallback and the test below.
  const listed = match.replace(MARKS_PHRASE_LISTED, PHRASE_GONE);
  // A word base itself removed here is not handed back โ€” that would take a strip away rather than add
  // one. Asked of base's own output rather than of a list of words base might have matched, because the
  // ways the slot can capture something base was already stripping are not enumerable: a fourth listed
  // adjective pushes the fourth into the slot (`light pale grey` fills the stack, `dark` lands here), and
  // so does the head of a compound marks noun in a joined tail (`scanner noise/speckles` โ€” base takes
  // `noise` as part of `scanner noise`, the slot takes it as an adjective of `speckles`). That second one
  // is a page on the corpus, honoured on base, and an earlier revision of this function lost it.
  //
  // Asked at the grain the PATTERN matches rather than at the grain of a word, because `MARK_ADJECTIVE`
  // can begin inside a hyphenated spelling: the slot's word in `dark-streaked specks` is `streaked`, with
  // `dark` read as the stack. Comparing whole tokens there answers that base stripped `streaked` โ€” base
  // stripped neither half โ€” and stripping the span on that answer drops a `streak\w*` veto base raised.
  const bounded = word.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
  if (!new RegExp(String.raw`(?<![A-Za-z])${bounded}(?![A-Za-z])`, "i").test(listed)) return PHRASE_GONE;
  // A word that does not dress a noun at all: it opens a phrase of its own, and the stack in FRONT of it
  // then belongs to whatever that phrase is about rather than to the marks. "the scan is noisy with
  // artifacts" is the case, and handing `with` back is not enough to save it โ€” the word stays in the
  // scope, but `noisy` leaves with the phrase, and `noisy` is the whole doubt. So these fall back to base.
  // A closed grammatical class, and the one enumeration here that the first review of #430 did not find
  // wrong: `with` and `without` are its own two entries, being the prepositions these logs put in front
  // of the marks, and no list below carries them.
  if (DETERMINER.has(word) || LOCATIVE.has(word) || CONJUNCTION.has(word) || COPULA.has(word)) return listed;
  if (NEGATOR.has(word) || word === "with" || word === "without") return listed;
  // A name for text goes back into the scope like anything else, but only `contentAffirmed` can act on
  // it, and it reads NOUNS. So the forms it cannot read fall back to base instead of being handed to a
  // check that will not see them.
  //
  // Except where the noun it was dressing is one only the capture leaves, and then it leaves WITH that
  // noun (#439). Inside this branch rather than ahead of it, which is the scope of the whole widening:
  // the only words it can move are the ones base was already handing back, so a doubt word the slot
  // captured is untouched โ€” `smeared artifacts` still hands `smeared` back, and `smear\w*` is a veto
  // (#226). The head is read from the MATCH and immediately behind the slot's own word, because that is
  // where the pattern puts it: the slot is the last thing before the marks noun.
  //
  // And it takes exactly that one word, cut OUT OF BASE'S OWN OUTPUT rather than replacing the match, so
  // the guarantee above holds for the STACK as well as for the slot. Returning `PHRASE_GONE` for the whole
  // match here took the stack too, and seven `MARK_MODIFIER` entries are `DEGRADED_IMAGE_LOG` veto words:
  // `Blurry print artifacts are visible.` went from `blank_vetoed` on `blurry` to a declaration, so a page
  // whose log says the scan is blurry shipped empty โ€” #226's failure, reached from the other side. Cutting
  // the word out of `listed` leaves `Blurry  <cut>  are visible.`, which loses the affirmation and keeps
  // the doubt. Anchored at the marker base left, because that is where the mark noun was and the slot's
  // word is the token in front of it. Reported by the review on PR #444.
  if (NAMES_TEXT.test(word) || NAMES_TEXT_FORM.test(word)) {
    return new RegExp(String.raw`(?<![A-Za-z])${bounded}[\s/-]+(?:${CAPTURE_ONLY_MARK})\b`, "i").test(match)
      ? listed.replace(new RegExp(String.raw`(?<![A-Za-z])${bounded}(?=[\s/-]*\f)`, "i"), "")
      : listed;
  }
  return ` ${adjective}${PHRASE_GONE}`;
}
// Two constructions that say the marks are not text, and so are the declaration rather than a
// failure to read. Both are anchored to a marks noun earlier in the sentence with NO NAME FOR TEXT
// in between, which is what makes the marks the thing being denied. Without the anchor they read as
// exempt wherever they sit, including in a log that affirms the text is there ("the text does not
// resolve into legible words"); with a plain same-sentence anchor, a marks noun anywhere ahead of
// the affirmation exempts it ("a few specks are visible, but the printed text does not resolve into
// words") โ€” and either way a page that has text on it is delivered as an empty fragment with no
// marker and nothing in `pages_failed`.
//
// It has to be a gap and not adjacency: the real logs put the whole predicate between them ("Specks
// /dots are visible on the page but do not resolve into any characters", "artifacts of the scan
// (dust/noise) and do not resolve intoโ€ฆ", "specks/dots that appear to be scanning artifacts, not
// legible text"), so requiring the noun immediately before the verb, or refusing to cross a comma,
// `but` or `and`, would cost three of the four pages this exists for.
// Every name for something printed on a page, not only for text: what the gap must not cross is an
// affirmation that the page HAS something on it, and `the content is not legible text` affirms as
// plainly as `the lines` do. `content` matters most because `NOT_LEGIBLE_TEXT` counts it as a name
// for text on both sides โ€” as the noun it strips and in `PAGE_BEARS` โ€” and no two of the three may
// disagree about the same word. None of this
// costs the four round-9 logs: they all put `content` after the veto word ("do not resolve into any
// characters or content"), never in the gap ahead of it, which is the only region examined.
//
// This list is where the exemption stops being provably safe, and that is a decision rather than an
// oversight. A hand-written list of nouns is as complete as someone's memory: "the graphic does not
// resolve into words" gets through, and so does the next noun after that. The structural version
// would invert it โ€” forbid ANY subject in the gap except the words for the paper itself (`page`,
// `sheet`, `scan`, `margin`), so an unknown noun refuses the declaration instead of exempting it.
// It is not taken because the risk lands on the pages this exists for: the real logs put a subject
// in the gap ("Specks/dots are visible on the page butโ€ฆ", "artifacts of the scan (dust/noise) andโ€ฆ"),
// so the safe direction there is a whitelist of substrate words, which is the same completeness
// problem pointed the other way, and getting it wrong reports a blank page as lost โ€” the #190 defect
// again. The wording needed to reach what this list misses asserts blankness, names the marks, and
// AFFIRMS an unlisted page object between them โ€” denying it is what a blank page's log does, and
// `NEGATED` reads that correctly โ€” with no doubt word anywhere in the log, which nothing in nine
// bench rounds has produced. If a round ever produces one, the noun goes in the list.
const TEXT_NOUN = String.raw`text|texts|content|print(?:s|ing|ed)?|lines?|words?|characters?|letters?|glyphs?|digits?|numerals?|handwriting|writing|typing|paragraphs?|sentences?|headings?|captions?|figures?|images?|illustrations?|diagrams?|tables?|stamps?|signatures?|labels?|logos?|seals?`;
// The same list as a word-boundary test, for the one caller that has a single token in hand rather than a
// statement: `marksPhraseStrip` above, where a slot holding `text-like` has to read as a name for text
// (`AFFIRMED_NOUN` is anchored and would not).
const NAMES_TEXT = new RegExp(String.raw`\b(?:${TEXT_NOUN})\b`, "i");
// The names for text that the list above is the wrong PART OF SPEECH for. `TEXT_NOUN` is a list of
// subjects, because everything else reading it wants the thing a sentence is about; the marks-phrase slot
// is the one position in this file that holds an attributive, and an attributive is how a log names what
// made a mark: `handwritten smudges`, `stamped dots`, `typed specks`, `watermark dots`. `TEXT_NOUN` has
// `handwriting`, `stamps?` and `typing`, so the noun form of each of those was already refused while its
// participle went through โ€” found by the first review of #429's fix, and the reason this is a separate
// list rather than more alternatives inside that one: nothing else in the file wants a participle, and
// putting one in `TEXT_NOUN` would let `stamped` stand as a SUBJECT wherever that list is read as one.
// Wider than the participles of `TEXT_NOUN`'s own entries on purpose: a word here costs base's verdict
// and nothing more, since the fallback IS base.
//
// WHO READS IT, which is now two callers and was one. The marks-phrase slot reads it to keep base from
// getting WORSE, by giving up the one extra word of reach wherever a doubt word is the thing standing
// between this vocabulary and a declaration. The affirmation reader reads it through `affirmsText` to
// make base BETTER, because a log naming writing with a participle was a page delivered empty while the
// same claim in the noun form was refused (#431). The two uses want the same words for the same reason โ€”
// these are names for text โ€” and the second one is why a word added here now costs more than base's
// verdict: it can also refuse a page, so the policy below is read with `NEGATED` and `negatedInList` in
// mind rather than as a free list. `negatedInList` is the sharp case and the review of #431's fix found it:
// a word here is crossed by the walk that looks for a negator, which is what keeps an attributive DENIAL
// denied, and the same crossing reaches into a second clause unless the walk is bounded there. Read that
// function's bound before adding a word.
//
// WHAT THE FIGURES HERE USED TO SAY, corrected because they were the reasoning behind #431 and they were
// confounded. This paragraph read: the contradiction check fires on 14 of 14 wordings that put a name for
// text in SUBJECT position and on 0 of 32 that put one in this one, so an attributive naming writing
// already loses its page. The 14 all used words `TEXT_NOUN` holds and the 32 all used words it does not,
// so the two axes moved together and the contrast measured neither โ€” and the parenthetical offered
// "Only handwriting smudges are visible" as an example of SUBJECT position, which is the attributive frame
// with a listed word in it. Crossed properly, eight words per cell: subject/listed 8 of 8 refused,
// subject/unlisted 0 of 8, attributive/listed 8 of 8, attributive/unlisted 0 of 8. The position was never
// the axis; the list was, which is what made #431 fixable here rather than in a rule about the slot.
// What survives unchanged is the measurement this list was actually built on: of 27 words already here,
// base declares "Page is blank. Only <w> smudges are visible." blank for 27, and for the 16 the second
// review of #429's fix named as missing, 16.
//
// WHICH WORDS, then, since the vocabulary is open and 40 of 48 attributives I could invent are admitted.
// Three sources, each checkable, and nothing beyond them:
//   1. a word the corpus writes in a page log โ€” 22 of these, `footnote` 2,215 times, `italic` 403,
//      `cursive` once, and NONE of the 22 in any of the 204 logs that assert blankness, which is why
//      widening moves no page on the corpus;
//   2. a word the second review named (`cursive`, `pencilled`, `handprinted`, `barcode`, `drawing`),
//      plus the instrument participles of that shape (`penned`, `inked`);
//   3. the other form of a word already here, which is this list disagreeing with itself: `drawing`
//      beside `drawn`, `sketch` beside `sketched`, `doodle` beside `doodled`, `annotation` beside
//      `annotated`, `inscription` beside `inscribed`, and the gerund of every participle listed
//      (`stamping` beside `stamped`, `italicised` beside `italic`) โ€” those last came out of the probe
//      for the boundary fix below, which rescued `italicised` while refusing `italic`. Unlike the
//      hyphenated compounds, a suffixed form of a listed stem IS closeable by the policy above, because
//      the forms of one word are finite where the prefixes a log can write are not.
// Beyond those, the file's own policy for `TEXT_NOUN` applies unchanged โ€” if a round ever writes one,
// the word goes in the list โ€” and it is cheap to apply here precisely because the fallback is base.
//
// BOUNDARY-TESTED, not anchored, and that is a fix rather than a convenience. Anchored, this list could
// not see a hyphenated compound of a word it already carries: `rubber-stamped`, `pen-written`,
// `hand-lettered` and `hand-annotated` all escaped a list holding `stamped`, `written`, `lettered` and
// `annotated`, found by the third review of #429's fix. The policy above cannot close that, because the
// gap is not a missing word โ€” `hand-`, `pen-` and `rubber-` are how a log names the instrument, so the
// compounds are unbounded while the words behind them are already here. `NAMES_TEXT` is `\b`-tested for
// the same reason one line up, which is why `machine-printed` was caught and `pen-written` was not. `\b`
// cannot match inside a word (`written` does not match in `rewritten`, `italic` not in `italicised`),
// so what this admits is exactly the compound. Which is why `hand-?printed` stays in the list beside
// `printed`: `handprinted` written solid is a word no boundary test on `printed` can reach, there being no
// boundary inside it. Same for `hand-?written` beside `written`.
//
// The HYPHENATED half of that entry is not redundant either, and not because it is needed โ€” because it
// diverges, which round 1 of #437's review measured. `qualifies("hand-printed")` is true (`\bprinted\b`
// matches across a hyphen) and `qualifies("handprinted")` is false, so in the terminal-object read of
// `exceptiveOrLocativeObject` the hyphenated spelling is skipped as a modifier and the solid one is read as
// a name for text: `No content is present in the hand-printed.` declares and `...in the handprinted.` is
// refused. The disagreement is not with the stem โ€” `printed` and `machine-printed` both declare there โ€” and
// for that sentence declaring is right, since the log denies content. What the pair exposes is the other
// side of the same read: a terminal object that is a name for text and NOT a qualifier refuses a page whose
// log denied content, which `written`, `stamped`, `hand-written` and `handprinted` all do and all did before
// this change. That frame is not log English (put a noun behind the modifier and all five spellings agree on
// both arms), so it is recorded here and pinned in the tests rather than widened.
const NAMES_TEXT_FORM =
  /\b(?:hand-?written|handwrote|written|typed|printed|typewritten|typeset|stamped|signed|initial(?:l)?ed|lettered|numbered|captioned|labell?ed|annotated|inscribed|embossed|engraved|watermarks?|watermarked|drawn|sketched|scrawled|scribbled|doodled|underlined|highlighted|illustrated|cursive|pencill?ed|penned|inked|hand-?printed|lettering|barcodes?|drawings?|sketch(?:es)?|doodles?|annotations?|inscriptions?|footnotes?|notations?|monograms?|letterheads?|logotypes?|punctuation|diacritics?|symbols?|italics?|boldface|typographic|textual|alphanumeric|numeric|numbering|stamping|signing|captioning|labell?ing|embossing|engraving|underlining|highlighting|scribbling|scrawling|watermarking|doodling|sketching|italici[sz](?:ed|ing))\b/i;
// A name for text only affirms it where it is not NEGATED, which is the difference between "the
// printed text does not resolve" and "no printed text". The prompt asks the agent for both halves of
// the observation in one breath โ€” name the marks, deny the text โ€” so without this the more explicit
// answer is the one that loses its page: adding `no text` to one of #190's own logs took it from
// delivered to lost, and most of what `TEXT_NOUN` lists is a noun a blank page's log DENIES (no
// signature, no stamp, no figures). The one shape where a negative word does not negate the noun
// after it is `nothing but the text` / `nothing except the text` / `no matter the text`, which affirm
// it, so those three are excluded โ€” otherwise which member of the pair got caught also depended on
// how long the phrase was, the 2-word window reaching `handwriting` but falling short of
// `printed text`.
const NEGATED = String.raw`(?<!\b(?:no|not|without|nor|none|nothing|neither)\s(?!(?:but|except|matter)\b)(?:[\w'-]+\s){0,2})`;
// The gap between the marks and the construction: anything that does not affirm text. It may cross
// ONE sentence or semicolon boundary, so the same observation split into two clauses is read the same
// way โ€” "Specks/dots are visible on the page. They do not resolve into any characters." is the same
// answer as the version with a `but` in it, and #190's whole finding was that which pages get lost
// was being decided by wording the agent picks per call.
const MARKS_GAP = String.raw`(?:(?!${NEGATED}\b(?:${TEXT_NOUN})\b)[^.;])*`;
// What the clause after a crossed boundary must open with: a back-reference to the marks just named,
// no subject at all, or a denial. Crossing a boundary is only safe for a CONTINUATION of the
// observation, which is all the comment above claims; without this the marks in one sentence exempt a
// denial about a different page object in the next, and "A few specks of dust are visible. The
// handwritten note in the corner does not resolve into words." reported the page as blank. That is
// the one place where "the veto lists run over the whole log anyway" does not save it, because the
// veto word IS the construction being stripped. Unknown openers refuse, so the list being incomplete
// costs a glance rather than a page.
//
// `it` and `there` carry a verb with them, and a conjunction is only a prefix to one of the others,
// because those three are how a new subject gets across a boundary that a determiner cannot: "It is a
// photograph that does not resolve into detail", "There is a handwritten note that does not resolve
// into words", "But the graphic does not resolve into words" โ€” all three name a page object, and all
// three opened with a word that looked like a back-reference. Bare `does|do|did` has to stay, for the
// subject-less "Does not resolve into printed words.", so it carries its `not` too โ€” otherwise the
// inverted form is the same door again ("Nor does the barcode resolve into words"). `nor` and
// `neither` are prefixes only for the same reason: inversion is what they are for, and a bare one
// swallowed the subject that followed it.
// Names for what a page bears, and what may qualify one, for the `any` branch below. `recogni[sz]able`
// is the one word in any of these lists whose British and American spellings both arrive โ€” one #220
// log is written "recognisable content" โ€” and the spelling a log picks is a per-call choice, so every
// list that names the word names both spellings.
const TAIL_NOUN = String.raw`(?:text|texts|content|words?|characters?|print(?:s|ed|ing)?|writing|markings?|lines?|letters?|glyphs?|digits?|numerals?|figures?|images?|handwriting|anything|something)`;
const TAIL_QUALIFIER = String.raw`(?:a|an|any|no|the|some|other|more|meaningful|legible|readable|printed|typed|visible|discernible|recogni[sz]able|clear)`;
const CONT_CORE =
  String.raw`(?:(?:they|these|those)\b` +
  String.raw`|(?:it|this)\s+(?:doesn'?t|isn'?t|wasn'?t)\b` +
  String.raw`|(?:it|this)\s+(?:does|do|did|is|was)\s+(?=not\b)` +
  String.raw`|there\s+(?:is|are|was|were)\s+(?:no|none|nothing)\b` +
  String.raw`|(?:does|do|did)\s+(?=not\b)` +
  String.raw`|(?:no|none|nothing|not)\b` +
  // `any` names what is denied without being a denial itself, so it is allowed only ahead of a name
  // for text and only as a lookahead โ€” "Not legible text, nor any figures" continues the denial, while
  // "Any printing that may exist does not resolve into readable text" leaves `printing` in the gap,
  // where it still refuses.
  String.raw`|any\s+(?=(?:${TAIL_QUALIFIER}\s+){0,2}${TAIL_NOUN}\b)` +
  // The marks themselves, named again with or without a determiner: `only dust` is as much a
  // continuation as `only the dust`, and requiring the determiner cost the commoner wording.
  String.raw`|(?:(?:a|an|any|some|few|several|more|the|these|those|their)\s+){0,3}(?:(?:${MARK_MODIFIER})[\s/-]+)*(?:${MARK})\b)`;
const CONTINUATION = String.raw`(?:(?:and|but|or|only|just|also|so|then|nor|neither)\s+)?${CONT_CORE}`;
const MARKS_ANCHOR = String.raw`(?<=\b(?:${MARK})\b${MARKS_GAP}(?:[.;]\s*(?:${CONTINUATION}${MARKS_GAP})?)?)`;
// "โ€ฆspecks/dots โ€ฆ do not resolve into any characters": `resolve` is in `UNREADABLE_LOG` for "could
// not resolve", and a destination after it turns the sentence into a denial that the marks are
// characters.
const MARKS_NOT_TEXT = new RegExp(`${MARKS_ANCHOR}\\bresolves?\\s+(?:in)?to\\b`, "gi");
// "specks/dots โ€ฆ not legible text" denies the marks are text; "the text is not legible" is a claim
// about text that exists. Both the word order and the anchor are needed: `the typed lines are not
// legible characters` has the noun after it too, and names no marks.
//
// The trailing guard is for the noun on the FAR side of the construction, which the gap cannot see
// and `TEXT_NOUN` therefore cannot help with: "Some dust. Not legible printing in the margin." names
// marks, then names something the page bears, and being told WHERE it is is what distinguishes it
// from "not legible text or meaningful content", which denies.
//
// So the REST OF THE STATEMENT has to be made of nothing but denial. Not a list of the prepositions
// that refuse โ€” the preposition nobody thought of ("not legible writing over the seal") would cost the
// page โ€” and not a list of the words that may follow either, because each such branch was a door a
// placement walked through one word further along: first `visible in the margin`, then `or the printing
// in the margin`, then a line break before `in the margin`, then `any writing in the margin`. Every one
// of those was the same shape, and each fix bought exactly the wording it named.
//
// A word-by-word whitelist is the version with no next door. What a denial is made of is a small closed
// vocabulary โ€” denials, names for text, names for the marks, names for the whole substrate โ€” and a page
// object is named with a word that is not in it: `margin`, `header`, `corner`, `seal`, `spine`, `note`,
// `signature`. So `not legible text on the page` denies (every word listed) and `not legible text in the
// margin` does not (`margin` is not), whatever punctuation or preposition connects them. A word nobody
// thought of costs a glance, which is the way round this file chooses everywhere else.
//
// The statement ends at a full stop, a `!`, or the end of the log; a comma, a semicolon, a colon or a
// line break does not end it, because these logs are written as loose notes ("Not legible text\nNo
// page-break marker is emitted", sometimes as a `-` list) and a note breaks its line exactly where it
// would otherwise place the text. A `?` anywhere in the statement refuses: "Not legible text?" is the
// model asking whether the page is empty rather than saying it is, and a bare one is the one hedge
// `HARD_DOUBT` cannot see.
// `as` between the two is the same denial with the noun made a predicate of the marks rather than the
// object of the reading: "specks/artifacts are present, which are not legible AS content" and "โ€ฆare
// not legible content" say the one thing, and only the first was refused (#220). And `marks` joins
// `markings` in the noun list for the same reason it is a `MARK` behind one of the `SPARSE` words: the
// anchor is what decides whether marks are the subject at all, and it has already found a marks noun
// with no name for text since โ€” "The typed lines are not legible marks." reaches no anchor and goes on
// refusing, as does anything the anchor cannot cross a boundary into.
const NOT_LEGIBLE_TEXT = new RegExp(
  `${MARKS_ANCHOR}\\bnot legible\\s+(?:as\\s+)?(?:text|content|words?|characters?|print(?:ed|ing)?|writing|mark(?:s|ings?)?)\\b`,
  "gi",
);
// Nothing here is a veto word except by describing the marks, so a doubt word smuggled into the tail is
// still refused by the two lists afterwards โ€” this decides only whether `not legible` is the denial.
const DENIAL_WORD = new Set(
  (
    "or and nor either neither no not none nothing anything something else any only just more other " +
    "meaningful legible readable printed typed visible present discernible apparent detected seen found " +
    "recognizable clear at all whatsoever anywhere of kind sort type a an the this that these those some " +
    "few several couple is are was were be been being isn't aren't wasn't weren't there it they " +
    "do does did don't doesn't didn't resolve resolves resolving " +
    "resolved remain remains remaining appear appears appearing emitted emit written contain contains " +
    "holds hold marker markers number numbers break page pages sheet sheets scan scans paper image images " +
    "document documents leaf leaves on in across throughout within text texts content contents word words " +
    "character characters letter letters glyph glyphs digit digits numeral numerals figure figures line " +
    "lines print prints printing writing handwriting typing marking markings mark marks paragraph " +
    "paragraphs sentence sentences heading headings caption captions label labels legend legends " +
    "dust speck specks speckle speckles speckling fleck flecks dot dots debris smudge smudges blemish " +
    "blemishes artifact artifacts scanner scanning stray scattered faint tiny small minor isolated random " +
    "residual noise grain " +
    // The rest of the list a denial of content is written as. `structure`, `diagrams` and
    // `illustrations` are #220's wordings ("do not resolve into any characters, images, or
    // structure", "โ€ฆany characters, words, diagrams, or other content") and are the same kind of
    // word as the `figures` beside them; `recognisable` is `recognizable` as a log spelled it, and
    // every list that names one names both, because which spelling arrives is a per-call choice.
    "structure structures diagram diagrams illustration illustrations recognisable"
  ).split(" "),
);
// The vocabulary above is what a denial is BUILT from, and the same bricks build the opposite claim:
// `only a heading is visible`, `the page contains figures`, `some words remain visible` are made
// entirely of listed words and each says the page has something on it. So a name for what a page
// bears has to be introduced by a denial where it appears โ€” `or content`, `nor any figures`, `no
// writing` โ€” and not by a determiner that affirms it. Naming the substrate is exempt, because `on the
// page` is another way of saying the sheet is empty; that asymmetry is the whole difference between
// the two, and it is the same rule `NEGATED` applies to the gap on the near side of the construction.
//
// Reading back over qualifiers (`or any other printed words`) but never over a determiner is what
// separates `nor the words` โ€” a glance โ€” from `only a heading is visible` โ€” a page.
const PAGE_BEARS = new Set(
  (
    "text texts content contents word words character characters letter letters glyph glyphs digit " +
    "digits numeral numerals figure figures line lines print prints printing writing handwriting " +
    "typing mark marks marking markings paragraph paragraphs sentence sentences heading headings " +
    "caption captions label labels legend legends diagram diagrams illustration illustrations " +
    "structure structures"
  ).split(" "),
);
const DENIAL_CONNECTOR = new Set("no not nor neither none nothing or and any".split(" "));
// Two of those connectors are conjunctions, which introduce an affirmed noun as readily as a denied
// one: `or content of any kind` denies and `and printing is present` affirms, and only what comes
// AFTER the noun tells them apart. So a conjunction may not hand a name for text to a verb that says
// it is there. Nothing else needs this โ€” a determiner is refused already, and a real negator (`no
// writing`, `nor any figures`) has spent itself on the noun.
//
// The verb is looked for past the noun rather than only next to it, because a locative may sit in
// between: `and printing on the page is visible` is `and printing is present` with three words in the
// middle. But the search stops at the first real negator, because that is where the NEXT denied clause
// begins and its verb has nothing to do with this noun. A blank page's log goes on denying in exactly
// that shape โ€” "not legible text or content, and no writing is visible", "โ€ฆor content; nothing is
// printed", and above all the page-number clause the page prompt asks for ("โ€ฆnot legible text or
// meaningful content, and no printed page number is visible"), which is #190's own log with a comma
// where it happened to have a full stop. Scanning past the negator refused all of those.
const CONJUNCTION = new Set("or and".split(" "));
const NEGATOR = new Set("no not nor neither none nothing".split(" "));
// `detected`, `seen`, `found` and `present` are deliberately NOT here, though they affirm as plainly:
// the tail is governed by the `not` in front of the construction, so they are the denial's own words
// there ("not legible text or content detected" is the commoner wording, and "not legible text
// present" is in the corpus). That leaves `and printing detected` exempt, which is a stilted way to
// say a page has printing on it โ€” and the trade is the same one the file makes everywhere: the
// alternative refuses a wording blank pages are actually written in.
// A NEW member of this list that takes an object has to go in `TRANSITIVE_AFFIRM` as well, and nothing here
// will fail if it does not. `deniedAfterVerb` reads an absence complement after a linking verb only, and it
// tells the two apart by asking `TRANSITIVE_AFFIRM` โ€” which is a separate list, holding `bear`, `show`, `has`
// and `carries`, none of which is here. So adding one of those four to this list alone reopens the defect
// round 1 of #446 found (`The heading shows empty rows.` reading as a denial and shipping the rows out
// empty), silently and in a different file's worth of distance from the gate that was supposed to stop it.
const AFFIRMING_VERB = new Set("is are was were appear appears remain remains contain contains hold holds".split(" "));
const QUALIFIER = new Set(
  "meaningful legible readable printed typed visible discernible apparent recognizable recognisable clear other more".split(" "),
);
// The same words as a hyphenated compound, which is what `qualifies` reads and this Set cannot: a log
// writes `machine-printed`, `pre-typed`, `semi-legible` as readily as the bare word, and every list this
// one is read beside is boundary-tested (`NAMES_TEXT`, `NAMES_TEXT_FORM`) for exactly that reason. Built
// from the Set so the two can never drift. `\b` cannot match inside a word, so `printed` here is the
// compound and not `imprinted`; a NEGATIVE prefix is admitted along with the rest (`un-printed` reads as
// `printed`), which is the standing cost of boundary-testing this file already carries for
// `NAMES_TEXT_FORM` (`un-written`) and runs toward a reported blank page rather than a lost one.
const QUALIFIER_FORM = new RegExp(String.raw`\b(?:${[...QUALIFIER].join("|")})\b`, "i");
// A qualifier in either form. Read where the question is whether a word is a MODIFIER โ€” one the walk may
// cross, or one that must not be taken for the noun it stands in front of โ€” and not where the question is
// whether it names text, which is `affirmsText`. `printed` and `typed` are in both classes, so the two
// answers have to move together: #437 measured 24 of 40 (bare, compound) pairs answering differently from
// their own stem, in both directions, because this list was token-exact where the text lists are not.
//
// THIS READS COMPOUNDS OF ALL THIRTEEN and the pair above is two of them, which round 1 of #437's review was
// right to ask about. Swept โ€” 13 words x 3 prefixes x the 4 frames the three call sites own, 156 cells โ€” the
// compound answered differently from its own bare stem in 75 cells before this and in 0 after, and all 75
// moved TO the stem's answer. 36 of them toward a declaration and 39 toward a refusal, so there is no safe
// side to this and it is not a widening: it is one rule answering a sentence where two used to.
//
// 33 of the 36 declaring cells are one interaction, and it is a decision this file already made: a DEFINITE
// `image` is the scan and not a thing on the page (`LOCATIVE_SUBSTRATE`), and `definiteBefore` is what finds
// the article in front of it past the modifiers. `The legible image is visible.` declares on base; what
// moved is that `The semi-legible image is visible.` now reaches the same read instead of stopping short of
// the article and affirming the scan. A compound the lists do not know still stops it (`The foo-bar image is
// visible.` is refused), so what crosses is the vocabulary and not the hyphen.
//
// The other 3 are a different mechanism and are named separately because an aggregate that covers two
// shapes hides one of them: `No content is present in the {semi,machine,barely}-typed.` declares through
// the terminal-object exclusion in `exceptiveOrLocativeObject`, where the bare `typed` declares as well.
function qualifies(word: string): boolean {
  return QUALIFIER.has(word) || QUALIFIER_FORM.test(word);
}
// The words that affirm a noun with no verb between them: "heading visible" is "a heading is visible"
// with the copula dropped, which is how a log written in fragments says a page has something on it.
// `detected`, `seen` and `found` are left out for the reason `AFFIRMING_VERB` leaves them out โ€” they are
// the wording a denial reaches for โ€” and `legible` and `readable` are left out because they are read as
// qualifiers in front of the noun everywhere else in the file, and a post-nominal one is not a wording
// these logs use. Every word here is one `DENIAL_WORD` also lists, which used to be the only way the
// read was reached at all: a statement with a word outside that list in it has already refused a line
// above. `verblessAffirmation` is the second reader and has no such gate โ€” it reads any statement the
// affirmation loop reaches โ€” so a word added here now widens an affirmation as well as a denial, and
// the two directions of error are opposite. Read that function's bounds before adding one.
const PREDICATED = new Set("visible present apparent discernible".split(" "));
// `image` is the one word the lists genuinely disagree about: it is the substrate in "not legible text
// in this image" and a thing the page bears in "an image is visible", and both wordings are ones these
// logs use. So it is read by what introduces it โ€” a locative preposition makes it the scan, anything
// else makes it an object on the paper โ€” rather than being assigned to one list and losing either a
// blank page or a photograph. Only `image` gets this; every other name for a page object is refused by
// `DENIAL_WORD` not listing it at all.
const LOCATIVE_SUBSTRATE = new Set("image images".split(" "));
const LOCATIVE = new Set("in on across throughout within".split(" "));
const DETERMINER = new Set("a an the this that these those".split(" "));
// What may stand between a verb and its object besides a determiner or a qualifier: a count. "There is
// some handwriting", "it shows two headings" are the wordings a page with something on it is described
// in, and without these the gap ended at the count and the affirmation was missed. Safe to allow
// because the negator check runs first, so `no`, `nothing` and `none` still end the object โ€” and
// because a count in front of a marks noun ("a few specks", "several stray marks") is not an
// affirmation subject at all: `TEXT_NOUN` does not list those. Digits need no entry: the tokenizer
// reads letters, so "shows 2 headings" already puts the noun next to the verb.
const QUANTIFIER = new Set(
  "some any both few several many numerous multiple one two three four five six seven eight nine ten".split(" "),
);
// A denial made of nothing but these words is a short one, so a statement that runs past this refuses
// rather than being read further. That keeps the work per match bounded โ€” a log with a thousand
// `not legible text`s in it and no full stop anywhere would otherwise re-read its own tail a thousand
// times โ€” and it keeps the direction right: the cap costs a glance, never a page.
const DENIAL_STATEMENT_MAX = 300;
// The statement after `not legible <noun>`: is every word in it part of the denial?
function deniesToStatementEnd(log: string, start: number): boolean {
  let end = start;
  while (end < log.length && end - start < DENIAL_STATEMENT_MAX) {
    const char = log[end];
    if (char === "." || char === "!") break;
    if (char === "?") return false;
    end++;
  }
  if (end - start >= DENIAL_STATEMENT_MAX) return false;
  const words = log
    .slice(start, end)
    .replace(/[โ€™โ€˜]/g, "'")
    .split(/[^A-Za-z']+/)
    .filter(Boolean)
    .map((word) => word.toLowerCase());
  if (!words.every((word) => DENIAL_WORD.has(word))) return false;
  return words.every((word, i) => {
    if (!PAGE_BEARS.has(word) && !LOCATIVE_SUBSTRATE.has(word)) return true;
    let before = i - 1;
    // Over a qualifier, and over an earlier member of the same list: what governs the first noun in
    // `any characters, words, diagrams, or other content` governs all four, and each of them is
    // checked at its own index anyway, so a name for text between this noun and its connector is one
    // more list member and not a new claim. The same coordination rule `negatedInList` applies from
    // the other side. A determiner still stops the walk, which is what keeps `only a heading is
    // visible` the affirmation it is.
    let listMember = false;
    while (before >= 0 && (QUALIFIER.has(words[before]!) || PAGE_BEARS.has(words[before]!))) {
      if (PAGE_BEARS.has(words[before]!)) listMember = true;
      before--;
    }
    // A noun that OPENS the statement is in the same position as one a conjunction introduced, and for
    // the same reason: the denial in front of the construction is what governs it, and the only thing
    // between them is the comma the list is written with. "โ€ฆdo not resolve into any characters,
    // images, or structure" denies three nouns, and reading the comma as the end of the denial refused
    // four of #220's nine logs on the `resolve` in front of them. Read exactly as `or` is, not more
    // loosely โ€” a verb saying the noun is there still refuses, so "โ€ฆinto any characters, printing is
    // visible" is the failure notice it always was.
    const openedStatement = before < 0;
    if (openedStatement || DENIAL_CONNECTOR.has(words[before]!)) {
      if (!openedStatement && !CONJUNCTION.has(words[before]!)) return true;
      // An affirmation written without a verb has none for the tail read to find: "not legible text,
      // heading visible" and "โ€ฆdo not resolve into any characters, diagrams visible" say the page has
      // something on it in the same words a denial is built from, and the verb the read looks for is
      // simply missing (issue #227). What tells the two apart is the CONNECTOR, not the noun: a denial
      // coordinates its list ("not legible text or diagrams visible" denies both, and #220's own logs
      // are written that way โ€” "any characters, words, diagrams, or other content"), while an
      // affirmation is a fresh clause dropped in behind a bare comma. So only a noun that OPENS the
      // statement is read this way, and a conjunction in front of it keeps the denial it always had.
      //
      // `listMember` is the rest of that same rule: a comma'd list needs no conjunction until its last
      // member, so a noun with an earlier name for text behind it is one more item in the denial and not
      // a new clause, however the item after it reads. That is what keeps "โ€ฆany characters, words,
      // diagrams, or other content" whole. The narrow reading is the deliberate one: a single noun, then
      // a word saying it is there, and nothing else between it and the comma.
      if (openedStatement && !listMember && PREDICATED.has(words[i + 1] ?? "")) return false;
      const tail = words.slice(i + 1);
      const nextDenial = tail.findIndex((later) => NEGATOR.has(later));
      const governed = nextDenial === -1 ? tail : tail.slice(0, nextDenial);
      return !governed.some((later) => AFFIRMING_VERB.has(later));
    }
    if (!LOCATIVE_SUBSTRATE.has(word)) return false;
    while (before >= 0 && DETERMINER.has(words[before]!)) before--;
    return before >= 0 && LOCATIVE.has(words[before]!);
  });
}
// `resolve into` needs the same tail read, one noun further on. `resolve` is the veto word in that
// clause, so stripping it takes the doubt off the whole rest of the statement with nothing looking at
// what the rest says โ€” and it is the commoner of the two constructions on the pages this exists for
// (three of #190's four logs are written with it), so every placement and every affirmation the
// `not legible` branch refuses was reaching the document through this one.
//
// The object of `into` is exempt from the read, because the `do not` ahead of the construction is what
// governs it: "do not resolve into any characters" denies the characters, and starting the read at
// `characters` would refuse the plainest wording there is ("nothing that would resolve into words").
// Everything after that object is read exactly as the other branch's tail is โ€” so "โ€ฆinto any
// characters or content" still declares the page blank and "โ€ฆinto any characters, only a heading in
// the margin" does not.
const RESOLVE_OBJECT = new RegExp(
  String.raw`^\s*(?:(?:a|an|any|the|some|few|several|more|other|meaningful|legible|readable|printed|typed|visible|discernible|apparent|recogni[sz]able|clear)\s+)*[A-Za-z][A-Za-z'-]*`,
);
function deniesAfterResolveObject(log: string, start: number): boolean {
  // Bounded so the scan stays O(1) per match, the same reason `DENIAL_STATEMENT_MAX` exists.
  const object = RESOLVE_OBJECT.exec(log.slice(start, start + DENIAL_STATEMENT_MAX));
  return deniesToStatementEnd(log, start + (object ? object[0].length : 0));
}
// Terms no exemption reaches, checked over the WHOLE log rather than a phrase: a failure to read
// ("the scan is too dark"), something hidden ("dust and noise obscure the text" โ€” which names
// marks and denies nothing), or a concession ("a few specks, though the scan is very faint"). A
// concession word is free to include here, because every exemption above needs a marks noun to
// reach anything, so blocking them on `though` costs nothing in a log that names none. This list
// only disables the exemptions โ€” the verdict is still the two veto lists', and six of these words
// are in neither, so "Page is blank. Only a few specks, though." is a blank page today and was one
// before any of this.
const HARD_DOUBT =
  /\b(illegible|unreadable|could ?n[o']?t|can ?not|can'?t|unable|failed|truncat\w*|too \w+ to|too (low|light|dark|faint|poor|noisy|blurry|grainy)|obscur\w*|hidden|corrupt\w*|error|did ?n[o']?t load|not load\w*|though|although|however|uncertain|not (entirely |fully )?(sure|certain))\b/i;

// A phrase naming the thing the model was HANDED, which is not a thing on the page. The reply is
// written about an image of a page, so "visible in the image" locates the specks on the scan: the noun
// is the substrate. `TEXT_NOUN` lists `images?` and `figures?` because both name page objects too โ€” "an
// image is visible" is a page with a picture on it โ€” and that is the right reading everywhere except
// inside `MARKS_GAP`, where a name for text between a marks noun and its denial is exactly what says
// the denial is about the text rather than the marks. So `image` standing in the middle of #343's log,
// "Contains only minimal dust/specks visible in the image; no printed content, no page number, no marks
// that resolve into characters", broke that anchor, and the denial's own verb was left to veto the page
// as a doubt about the scan. Measured on that log word for word: `vetoes: ["resolve"]`, page reported
// lost, and it is a page the model read correctly.
//
// Stripped by the RELATION rather than by the phrase. "in the figure" is the same sentence with one
// noun changed and it lost the page identically, so a fix naming `the image` would have bought one
// word โ€” which is what every fix in this class has bought so far (#190, #194, #220, and #371 is the
// issue that says so). What is asked for instead is a locative preposition and a definite determiner in
// front of a name for the input, which is the relation that makes the noun the substrate. The definite
// article is the discriminator the affirmation machinery already uses for this same ambiguity
// (`LOCATIVE_SUBSTRATE` with `definiteBefore`): "in an image" is as likely to be a picture on the paper
// and is left alone. `scan`, `photo` and `photograph` are in the list because they name the input, not
// because they were failing โ€” `TEXT_NOUN` does not list them, so they never blocked anything, and a
// list that omitted them would read as a claim that `image` is special when it is only in both lists.
//
// What it moves on the corpus: nothing. All 125 blank declarations in every bench round on disk get the
// same verdict from this branch as from base โ€” the reported log is from a deployed run and no bench
// round happens to contain its shape. That is the whole of the evidence for this half: one log reported
// as a lost page, fixed verbatim, plus the three wordings above that lose it identically, and a
// zero-difference sweep over every declaration on hand saying it takes nothing else with it. A rule
// that fires on no corpus row is worth stating as such rather than leaving a reader to assume the
// sweep found a rescue it did not.
const INPUT_SUBSTRATE = new RegExp(
  String.raw`\b(?:${[...LOCATIVE].join("|")})\s+(?:the|this|that)\s+` +
    String.raw`(?:(?:scanned|provided|supplied|attached|original|source|page)\s+){0,2}` +
    String.raw`(?:images?|figures?|scans?|photos?|photographs?)\b`,
  "gi",
);

// The text the veto lists are run over: the log with the marks phrases in it removed, or the log
// untouched where anything in it says the reading failed.
// The substrate strip runs before all of them, because what it removes is what stands between the two
// anchored strips and the nouns they anchor to.
// The two anchored strips run before the last: `MARKS_PHRASE` removes the very nouns they are anchored to.
// Every strip leaves `PHRASE_GONE` and not a space, so what the scope LOST is legible in it (#440), and
// the log's own `\f`/`\v` go first so that a marker in the scope is always this function's and never a
// reply's. That cleaning runs ahead of the `HARD_DOUBT` return too: the untouched log is a scope like any
// other, and a marker in it would say a phrase was removed from a scope nothing was removed from.
function vetoScope(log: string): string {
  const text = log.replace(/[\f\v]/g, " ");
  if (HARD_DOUBT.test(text)) return text;
  return text
    .replace(INPUT_SUBSTRATE, PHRASE_GONE)
    .replace(MARKS_NOT_TEXT, (match, offset: number, whole: string) =>
      deniesAfterResolveObject(whole, offset + match.length) ? PHRASE_GONE : match,
    )
    .replace(NOT_LEGIBLE_TEXT, (match, offset: number, whole: string) =>
      deniesToStatementEnd(whole, offset + match.length) ? PHRASE_GONE : match,
    )
    .replace(MARKS_PHRASE, marksPhraseStrip);
}

// The other question, and the one nothing above asks: does the log CONTRADICT its own declaration?
// Everything before this decides whether a log casts DOUBT on the blankness it asserts โ€” a failure to
// read, a bad image, a hedge โ€” and a log that says the page is empty and then says, in plain
// affirmative words, that something is on it casts no doubt at all. It states both. So
// `declaredBlank({ html: "", log: "Page is blank. There is handwriting on the page." })` was `true`
// (#194), and the page shipped as an empty fragment with no `@page-failed` marker, nothing in
// `pages_failed` and no incompleteness notice: the run reported a complete document and the reader
// simply did not get that page.
//
// An affirmation refuses the declaration; it does not decide which half of the log is true. There is
// no way to tell from the text, and the two answers cost differently โ€” a page reported lost is a
// glance at a re-extraction, a page dropped in silence is a page nobody knows to look for. Which is
// the direction every rule in this section is chosen for.
//
// That choice was between those two answers because they were the only two available. Where the reply
// STATES blankness in the field there is a third, and it is better than either: the page is delivered
// as the field says AND the image is put in front of the verifier with the empty fragment, so an
// affirmation that was right about the page buys a correction rather than a hole in the document, and
// one that was the regex misreading a denial costs a page nothing (#371, `blankSkip` in extractPage).
// Everything below is unchanged and still decides a declaration made in prose alone, where refusing
// remains the cheaper of the two errors available to it.
//
// Read over the SAME text the veto lists see โ€” the log with the marks phrases removed โ€” because that
// is what makes the check affordable at all. The logs a genuinely blank page is written in affirm
// things constantly, and every one of those affirmations is about the marks: "Some dust is present",
// "Stray markings are visible", "Only scanner dust is present", "The specks are scanner dust". With
// the marks phrase gone, those sentences have no subject left to affirm anything about, so they need
// no exception here โ€” the exemption that already exists for describing the paper does the work.
//
// The subject list is `TEXT_NOUN`, which is where the marks vocabulary is deliberately absent: `mark`
// and `markings` are things a blank page's log identifies ("The visible marks are artifacts of the
// scan") as readily as things a page bears, and #193 settled that ambiguity by keeping bare `marks`
// out of the exemption rather than out of the noun lists. Reusing `TEXT_NOUN` inherits that decision
// instead of taking it again in the other direction โ€” which is what an affirmation check written
// around `PAGE_BEARS` would have done, since `PAGE_BEARS` has `marks` in it and "The marks are
// artifacts." would then have reported a blank page as lost. That is the #190 defect, and the issue
// named it as the thing a fix must not buy.
const AFFIRMED_NOUN = new RegExp(`^(?:${TEXT_NOUN})$`, "i");
// A name for text in the OTHER part of speech, which is `NAMES_TEXT_FORM` above and is the same
// vocabulary in the same senses. `TEXT_NOUN` has `handwriting`, `stamps?` and `typing` and not
// `handwritten`, `stamped` or `cursive`, so on base "Page is blank. Only handwriting smudges are
// visible." is a contradiction and "Page is blank. Only cursive smudges are visible." is a page
// delivered empty (#431).
//
// What that gap is NOT is a gap in POSITION, which is what #431 was filed claiming and what the
// paragraph above `NAMES_TEXT_FORM` still said until this change: the 14 wordings measured in subject
// position all used words this list holds and the 32 in attributive position all used words it does
// not, so the two axes moved together and the contrast between them measured the wrong one. Crossed โ€”
// one vocabulary against one position, eight words per cell โ€” the answer is that the list decides and
// the position decides nothing: subject/listed 8 of 8 refused, subject/unlisted 0 of 8,
// attributive/listed 8 of 8, attributive/unlisted 0 of 8. `affirmingReach` already walks past an
// intervening noun, which is why `handwriting` is read in front of `smudges`; nothing had to be taught
// about the modifier slot, and a rule about that slot would still have delivered "cursive is visible"
// empty.
//
// Read plainly by three of the five callers that ask `AFFIRMED_NOUN`, and on its own terms by the other
// two, which is the part worth reading before changing any of them. The object of a denial's preposition
// is where the file's note on `printed` says the error runs toward a false failure notice rather than
// toward a glance, so there the second list is read only where the object phrase ends at the word. The
// negator walk reads it under a bound of its own, because there the wider vocabulary crosses a second
// clause as readily as a coordination member. Which caller reads what is decided at each one, because each
// asks a different question of the word, and the reasons are at the call sites.
//
// `NAMES_TEXT_FORM` is boundary-tested rather than anchored so that a hyphenated compound of a word it
// carries is read (`hand-written`, `rubber-stamped`), and that is what is wanted here too: these
// callers hold one token, and the token a log writes is as often the compound.
//
// `printed` IS IN THIS LIST, and it is the one word here that is also in `TEXT_NOUN` and in `QUALIFIER`.
// It is here for the compound alone: the bare token already affirmed through `AFFIRMED_NOUN`, so adding it
// changes nothing a log spells `printed` and everything a log spells `machine-printed` (#437). Before that,
// `typed` was in this list and `printed` was not, so `pre-typed` affirmed where `pre-printed` did not and
// `The machine-printed notes are visible.` declared a page with notes on it blank โ€” one sentence answered
// by two mechanisms, on the spelling of the modifier. 24 of 40 (bare, compound) pairs disagreed with their
// own stem across the five callers, in both directions; all 40 agree now.
//
// The parity that buys is parity with the STEM, not a claim that the stem's answer is right: whatever this
// file decides about `printed <noun>` it now decides about `machine-printed <noun>`, and which member of
// the pair a page needs is still decided by the noun BEHIND the modifier โ€” by `AFFIRMED_NOUN` here and by
// the object walk in `exceptiveOrLocativeObject`, which is where both halves of #437's fix live. A
// compound whose prefix NEGATES is read as the word it negates (`un-printed`, `un-written`), which is
// boundary-testing's standing cost in this list and runs toward a reported page rather than a lost one.
//
// THE CORPUS IS SILENT, and that is an empty denominator rather than a measured zero: of 3,747 page replies
// with a log, 4 write a hyphenated compound of `printed` or `typed` at all, and all 4 are replies whose HTML
// carries content โ€” so not one of them reaches the blank read on either arm. Replayed, 0 of 3,747 verdicts
// move. The 40 pairs are the evidence; the replay only says nothing on record breaks.
function affirmsText(word: string): boolean {
  return AFFIRMED_NOUN.test(word) || NAMES_TEXT_FORM.test(word);
}
// `there is/are` puts the noun after the verb, so the subject-verb order below never sees it, and
// "There is handwriting on the page." is the plainest of the five wordings #194 measured.
const EXISTENTIAL = new Set("is are was were".split(" "));
// The transitive shape puts the noun after the verb too, for a different reason: the subject is the
// PAPER and the text is the object. "The page contains handwriting", "the sheet still bears a
// heading", "it shows two headings" are each an affirmation the subject-verb scan below cannot see,
// because the word in front of the verb is not a name for text. Measured: without this branch, "Page
// is blank. The page contains handwriting." declared the page blank.
//
// The object is read with the same rules `there is` uses โ€” a gap of determiners and qualifiers, a
// negator ends it โ€” since the two constructions differ only in what stands in front of the verb. And
// no particular subject is required: whatever the log says bears the text, the object is what it says
// is on the page, and the wordings a blank page uses are denials of that object ("the page contains
// no legible text", "the specks show nothing", "the marks do not resolve into characters"), which the
// negator rules refuse either side of the verb. A list of allowed bearer nouns would only add a way
// to miss one.
const TRANSITIVE_AFFIRM = new Set(
  "contain contains contained hold holds held bear bears show shows has have had carries carry".split(" "),
);
// How far back a negator reaches for a VERB it governs (`does not contain`, `the specks have not
// resolved`). Three words covers a determiner and two qualifiers, which is what these logs put
// between the two, and stopping there is what keeps `and printing` โ€” three words past a `not` that
// governs a different noun โ€” an affirmation. A coordination is a list of NOUNS, so a verb is never a
// member of one and this window is all the reach it needs; the noun's own reach is `negatedInList`.
const AFFIRM_LOOKBACK = 3;
// Words that may stand between a verb and the noun it introduces (`there is`, `contains`).
const OBJECT_GAP = 3;

interface Word {
  word: string;
  // Whether a comma closes this word. The one piece of punctuation the scan needs: see the
  // parenthetical rule in `affirmingVerbAfter`.
  comma: boolean;
}

function words(statement: string): Word[] {
  const out: Word[] = [];
  for (const m of statement.matchAll(/([A-Za-z][A-Za-z'โ€™-]*)(\s*,)?/g)) {
    out.push({ word: m[1]!.toLowerCase().replace(/[โ€™]/g, "'"), comma: Boolean(m[2]) });
  }
  return out;
}

// `image` is the word the lists genuinely disagree about, and it disagrees here too: "the image is
// slightly rotated" is the scan and "an image is visible" is something on the paper, and both are
// wordings these logs use โ€” the first is in the corpus as a page that must still be delivered
// (geometry is not legibility, so `DEGRADED_IMAGE_LOG` lets it through on purpose). `LOCATIVE`
// settles the same ambiguity for the denial tails, but it cannot settle this one: there is no
// preposition in front of either.
//
// What separates them is the article. The scan is referred to definitely, because the log has been
// talking about it all along โ€” `the` image, `this` image โ€” and a thing on the page is introduced
// indefinitely, because it is being mentioned for the first time. So a definite `image` is not an
// affirmation subject, and every other name for a page object is one however it is introduced.
// The cost is "The image in the corner is a photograph", which is missed; the alternative was
// refusing a measured blank page, which is the defect (#179) this file exists downstream of.
const DEFINITE = new Set("the this that these those its their".split(" "));
function definiteBefore(tokens: Word[], i: number): boolean {
  let k = i - 1;
  // `qualifies` and not `QUALIFIER`, because a compound stands where its stem stands: the article in front
  // of `the machine-printed heading` is the one in front of `the printed heading`, and stopping at the
  // compound lost the definiteness the caller is asking about. This is the call the definite-`image` skip
  // reads, so it is where 33 of the 75 stem/compound disagreements the sweep at `qualifies` counted lived,
  // all of them in the frame `The <qualifier> image is visible.`
  while (k >= 0 && qualifies(tokens[k]!.word)) k--;
  return k >= 0 && DEFINITE.has(tokens[k]!.word);
}

// A word that names text and can also stand in front of a noun as its adjective, which is the class both
// this guard and `folioAt`'s caller need. `QUALIFIER`'s thirteen words are not that class: `NAMES_TEXT_FORM`
// is ~60 participles in none of them, and when the affirmation reader started reading that list (#431) it
// gained ~60 subjects these two guards could not see. Asking `QUALIFIER` covered `printed` and `typed` and
// nothing else, so #220's shape returned for `stamped signed embossed watermarked annotated labelled
// engraved scrawled numbered captioned inscribed underlined highlighted` โ€” 13 of 13 measured, each of them
// a blank page refused and lost. `QUALIFIER` is still read for its OTHER twelve words โ€” `legible`,
// `meaningful`, `visible` and the rest, which name no text and are in no other list โ€” and not for
// `printed`, which #437 put into `NAMES_TEXT_FORM`; the union is what the guards are about.
//
// `QUALIFIER.has` and not `qualifies`, and the reason is structural rather than a count. What both guards do
// is SKIP a subject the affirmation loop would otherwise read, so a word that is never read as a subject
// cannot be affected by either of them โ€” and the twelve other qualifiers are exactly that: `legible`,
// `semi-legible`, `machine-readable` are in no name-for-text list, in either part of speech, so no read ever
// arrives here holding one. The two that are, `printed` and `typed`, have their compounds covered by the
// boundary-tested half of the union now. A sweep of all thirteen as compounds through both guards' frames
// moves nothing, which is what that argument predicts and is not independent evidence for it: the frames
// cannot separate a skipped subject from a word that was never a subject, because both leave the verdict
// where it was.
function modifierForm(word: string): boolean {
  return QUALIFIER.has(word) || NAMES_TEXT_FORM.test(word);
}

// A word that is both a name for text and something a page can BE is a participle behind a copula, and
// the list of them is `modifierForm` above โ€” `printed` from `QUALIFIER`, `typed` from both, and every
// participle `NAMES_TEXT_FORM` carries, which is why this guard reads the union and not `QUALIFIER`
// alone. Same overlap `exceptiveOrLocativeObject` settles
// for the object of a denial, on both of its branches. "No page number IS PRINTED on the
// page itself, but the file metadata indicates this is page 4 of 25" is a denial of the page number,
// and taking `printed` for a subject of its own handed it the next affirming verb in the sentence โ€”
// ten words and a `but` away, in a clause about where the number came from โ€” which reported a blank
// page as lost (#220). Nothing is missed by skipping it: "The heading is printed on the page" affirms
// through `heading`, which is a subject the loop reads two words earlier and finds the same `is` for.
//
// WHAT IS NOT READ HERE is the other side of the copula: this asks what stands BEHIND `is`, and what stands
// after it is `deniedAfterVerb`'s question, not this one. That is where `is empty`, `is blank`, `is unmarked`,
// `is unfilled`, `is featureless` and `is void of content` are read (`ABSENCE_COMPLEMENT`, #442) beside the
// `is absent`, `is missing`, `is not present` and `is nowhere` the negator lists already held. Before #442
// this function's silence was the whole answer for those six, and `The heading is empty.` affirmed a heading
// and reported a blank page as a hole; nothing about THIS guard changed, and the note stays because the
// reason a subject walk cannot answer it is the reason the answer lives one function away.
const COPULA = new Set("is are was were be been being isn't aren't wasn't weren't".split(" "));
function participleAfterCopula(tokens: Word[], i: number): boolean {
  if (!modifierForm(tokens[i]!.word)) return false;
  const before = tokens[i - 1];
  return before !== undefined && COPULA.has(before.word);
}

// The page's own printed number is the one thing a page can bear that a reader never receives. The
// prompt forbids transcribing the folio as text (`agents/page.md`: "Do not transcribe the folio as
// text beside the marker either"), the only place its number may live is the page-break marker's
// label, and `renderPage` discards that marker on every accepted declaration. So a log that says the
// page is empty and then names its printed page number contradicts nothing: there is no content a
// reader loses on that page, which is the only question this check asks. Before this, a page whose
// only printed content WAS its folio โ€” a correct answer to the prompt, and a page with nothing to
// transcribe โ€” was refused in three of the four wordings it reports itself in and lost as a failed
// page (#222).
//
// The refusal came from `printed` every time. It is in `TEXT_NOUN` for the noun sense ("printing is
// visible") and here it is an adjective on the number, the same overlap `participleAfterCopula`
// settles for the copula shape. What the caller asks is `modifierForm` โ€” whether this subject could be an
// adjective at all โ€” and that is the union of `QUALIFIER` with `NAMES_TEXT_FORM` rather than the qualifier
// list alone, because `The stamped page number is visible.` is the same folio and the archival case for it
// is stronger than `printed`'s: a rubber-stamped folio or Bates number is how a blank sheet in a legal or
// archival scan reports itself, and asking `QUALIFIER` refused 13 of the 13 participles measured (#431's
// review, round 1). `typed page number` and `stamped page number` need no row of their own for the reason
// `numerals` needs none: the guard is `folioAt` on the words AFTER the modifier, and which modifier stands
// in front of them does not change what they are.
//
// What is skipped is the SUBJECT, not the statement: the loop keeps reading, so "The printed page
// number and a heading are visible." still affirms through `heading` two words later, and only a log
// whose sole named subject is the folio is let through. `page number` is required as a phrase โ€”
// `number` alone names a figure number or a total as readily as a folio, and `numbers are visible` is
// a page with content on it.
// `numeral` and `numerals` are deliberately NOT here, though `page numeral` is as much a folio as
// `page number` is: `numerals?` is itself in `TEXT_NOUN`, so skipping the `printed` in front of it
// hands the affirmation to the noun two words on and the verdict does not move โ€” only the span
// `blank_contradicted` quotes does (#230's review measured both trees). An entry that cannot change an
// answer is an entry claiming a coverage it does not have, and covering those two would mean taking
// `numerals` out of the names for text or stepping the loop past a noun it would otherwise read.
// `number` and `numbers` are outside `TEXT_NOUN`, which is what makes them the entries doing the work.
const FOLIO_HEAD = new Set("folio folios pagination".split(" "));
const FOLIO_COUNT = new Set("number numbers".split(" "));
function folioAt(tokens: Word[], i: number): boolean {
  const word = tokens[i]?.word;
  if (word === undefined) return false;
  if (FOLIO_HEAD.has(word)) return true;
  return (word === "page" || word === "pages") && FOLIO_COUNT.has(tokens[i + 1]?.word ?? "");
}

// The name of the FILE is not a thing on the page, and #343's page was lost to a sentence about it. The
// reply ends "Image filename indicates this is page 14 of 25 in document acir.", and `image` โ€” a name for
// text by `TEXT_NOUN`, since a picture on the paper is one โ€” reached five words along for the `is` in
// `this is page 14` and reported a correctly read blank page as failed (`affirmed: "image filename
// indicates this is page"`, `runs-extract-kimi100`, kimi-k2.5, 2026-09-03). The other half of that same
// reply, a `resolve` veto, is what `MARKS_NOT_TEXT` fixed; this half survived it (#429).
//
// The substrate skip below already covers every wording with a determiner in front โ€” "The image filename
// indicatesโ€ฆ", "The file name showsโ€ฆ" โ€” because `definiteBefore` reads `the` and stops there. So which
// wordings lose a page was decided by the article the model happened to omit, which is the thing #190
// named and #371 restated: a fix worth making removes the wording from the question rather than adding
// the one that was reported. What makes the file's name recognisable is the noun after it and not the
// determiner before it, so that is what this asks for.
//
// Two words, both closed: `filename` as one token, and `file name` as two. `name` alone is NOT enough โ€”
// "Image name is printed at the top" is a page with something on it, and a log that means the file says
// so with `file` or with `filename`. Nothing is skipped but the subject: the loop reads on, so "Image
// filename indicates page 14, and a heading is visible." still affirms through `heading` โ€” the same
// guarantee `folioAt` above is written to keep.
const FILE_NAME = new Set("filename filenames".split(" "));
function fileNameAt(tokens: Word[], i: number): boolean {
  const next = tokens[i + 1]?.word;
  if (next === undefined) return false;
  return FILE_NAME.has(next) || (next === "file" && tokens[i + 2]?.word === "name");
}

function negatedBefore(tokens: Word[], i: number): boolean {
  for (let k = Math.max(0, i - AFFIRM_LOOKBACK); k < i; k++) {
    if (NEGATOR.has(tokens[k]!.word) || tokens[k]!.word === "without") return true;
  }
  return false;
}

// The vocabulary a coordination of nouns is built from, beyond the names for text themselves and the
// qualifiers that dress them: the conjunctions that join the members, the `of any kind` that trails
// one, and the page's own furniture named inside such a list (`no printed page number or heading`).
// Deliberately not a determiner, not a count, not a verb and not a pronoun โ€” each of those ends the
// walk below, which is what keeps `an illustration is visible`, `only a heading is visible`, `it
// shows two headings` and `some words remain visible` the affirmations they are.
//
// `handwritten` and its spellings are here rather than in `QUALIFIER`, which four other functions
// read: what they are needed for is reaching back over `no printed or handwritten content`, and
// widening the qualifier list would change how the denial tails and the object gaps are read too.
//
// `visual` is here for the same reason and from the same measurement: "No readable text or meaningful
// VISUAL content is present." is one denial of two coordinated nouns, and the one word between the
// second noun and the qualifier in front of it was enough to end the walk โ€” so `content` found the
// list's shared `is present` and a blank page was reported lost (#220).
//
// Two axes, then, and the grouping below is which one each word is here for, because reading the whole
// thing as one vocabulary is what makes it look like a list to extend by hand (#371's objection).
// The COORDINATION axis is how many members a denial has โ€” the joiners and the `of any kind` that
// trails one โ€” and it is the axis #190 and #194 measured. The MODIFIER axis is how the member itself is
// spelled: a noun used to modify the name for text, which sits between the negator and the noun without
// joining anything. `page` and `number` were already that axis under another name (`no printed page
// number or heading`), which is why `no page content` shipped and `no other DOCUMENT content` did not โ€”
// which noun the model put in front of `content` decided whether the page survived. `document` and
// `body` are here for that axis and only that one (#379, from #367's log, which is both axes at once:
// "No text, images, tables, or other document content is visible." stopped at `document`, one word
// short of the `No`, and the page was delivered as `@page-failed`).
//
// What that axis cannot do is reach an affirmation no negator was ever near. Neither word is in
// `TEXT_NOUN`, so neither is a subject this walk starts from, and a determiner, a verb, a count or a
// `but` still ends the walk โ€” so `Page is blank. There is handwriting on the page.` and the three
// self-contradictions beside it in the tests have no negator to be handed one, whatever is added here.
const CHAIN_LINK = new Set(
  (
    // The coordination axis: what joins members, and what trails the last one.
    "or and either of any all at whatsoever else kind sort type " +
    // Qualifiers this walk needs that `QUALIFIER` cannot carry, for the reasons above.
    "handwritten hand-written typewritten visual " +
    // The modifier axis: a noun dressing the name for text rather than joining a list.
    "page pages number numbers document body"
  ).split(" "),
);
// A coordination has no length limit, so this walk needs one: a log with two hundred conjoined nouns
// in it should not be re-read from every one of them. Sixteen tokens is longer than any denial the
// corpus contains (the longest, `no content of any kind, printed or handwritten`, is seven), and
// running past the cap stops the walk, which reports the page failed โ€” a glance, not a lost page.
const NEGATOR_CHAIN_MAX = 16;

// Whether a negator governs this noun. A negator distributes over the whole coordination it opens โ€”
// `no legible text or handwriting`, `no printed words, lines, or characters`, `no text, printing,
// figures or writing` โ€” and how long that coordination is is a wording the model picks per call. A
// fixed window therefore decided by LIST LENGTH whether the last noun in a denial read as denied:
// with a three-word lookback `no legible text or handwriting is present` put `no` four tokens back,
// so `handwriting` read as un-negated, found the list's own shared verb, and a page with nothing on
// it was reported lost. Eleven of about thirty realistic blank-page wordings flipped that way, which
// is #190's defect from the other end โ€” the one thing #194 says a fix must not buy.
//
// So the walk is over the words a coordination is MADE of and stops at anything else, the same shape
// `deniesToStatementEnd` uses for the far side of a denial. It cannot reach past a verb, which is
// what keeps a second clause's subject its own (`no text is visible and handwriting is present`
// affirms `handwriting`), and it cannot reach past a determiner or a count, which is what keeps the
// affirmations #194 measured. `and printing, no page number, is visible` โ€” the shape this must not
// swallow โ€” reaches the document with its `not legible text` already stripped from the veto scope, so
// there is no negator left in front of `printing` to find.
//
// "It cannot reach past a verb" is the whole of that guarantee, and it says nothing about a denial with
// no verb of its own: `No printed text, and handwriting is present` is one clause of pure denial and one
// affirmation, and nothing between them ends the walk, so the affirmed noun reads as the last member of
// the list. That reading is the one #200's review weighed and chose โ€” refusing it loses a blank page
// whose log denied twice โ€” and it stood on a determiner, an `only` or a verb in the second clause as the
// only things that stopped the walk. #379's two words did not change that branch, they made it reachable
// by more wordings, and both sides of it are pinned in test/envelope-as-content.test.ts so which one a log
// falls on is a recorded answer rather than a discovered one (#391's review). `secondClauseJoint` below is
// what splits that shape now, on the sentence's own structure rather than on those three tells, and #200's
// reading is what it falls back to.
//
// The comma cannot be the discriminator, which is worth recording because it is the obvious narrowing:
// ending the walk at a comma no conjunction follows refuses four of the blank pages pinned in that file โ€”
// `No printed words, lines, or characters are visible.` and #367's own log among them โ€” because the
// members of a denial are separated by bare commas exactly as the two clauses are. On the corpus that is
// 38 of the 204 blank declarations on record, so it is a fifth of every blank page rather than four pinned
// wordings (#436). Whatever separates a verbless denial from an affirmation behind it, it is not
// punctuation alone.
//
// A negator distributes over a coordination only if it has a member of its own to distribute FROM,
// and one case in this file can take that member away: the marks phrase is stripped before any of
// this runs, so "No stray marks, and handwriting is visible" arrives as `no โ€ฆ and handwriting` with
// the noun the `no` denied gone from the text. Reading the conjunction as a coordination there would
// hand `handwriting` a negator that never governed it, and denying the marks says nothing about text
// โ€” which is the same asymmetry `TEXT_NOUN` encodes everywhere else in this section. So a negator
// whose own next word is a conjunction is not one this noun sits in a list with.
function negatedInList(tokens: Word[], i: number, reach: number[]): boolean {
  for (let k = i - 1; k >= 0 && i - k <= NEGATOR_CHAIN_MAX; k--) {
    const { word } = tokens[k]!;
    if (NEGATOR.has(word) || word === "without") {
      if (k + 1 < tokens.length && CONJUNCTION.has(tokens[k + 1]!.word)) return false;
      return !secondClauseJoint(tokens, k, i, reach);
    }
    // THIS CALLER HAD TO WIDEN, and the widening was measured against base rather than argued. The walk
    // steps over the other members of a coordination to reach the negator that governs them all, so a
    // denial written in the attributive form โ€” "No typed or stamped characters are present", "No footnotes
    // or annotations appear" โ€” only stays denied if this step knows those words are members too. Leaving
    // the step narrow while the affirmation reader widens hands the last member of a denial the verb of
    // its own clause, which is #190's defect arriving by the back door, and it is not hypothetical:
    // reverting this call alone turns two of the denials pinned in test/envelope-as-content.test.ts into
    // refused declarations, both of them blank pages base declares.
    //
    // What the wider vocabulary also buys, and must not: the same step crossing the participle of a
    // SECOND clause. "No clear text, and scrawled words are visible." reaches the `No` through
    // `scrawled`, and a page whose log says it carries handwriting is then delivered empty โ€” the silent
    // direction, and #431's own failure inverted (#434's review, round 1).
    //
    // THAT CASE IS `secondClauseJoint`'S NOW, and the bound #434 put here for it is gone. It was a comma
    // scoped to the crossings the widening added โ€” a flag set when the walk stepped over a `NAMES_TEXT_FORM`
    // word, and a return the next comma after it โ€” and the two things wrong with it are the two things
    // #436 is about. It read the FORM of the word rather than the shape of the sentence, so `and scrawled
    // words` refused while `and printed words` declared, one sentence answered by two mechanisms. And it
    // was order-dependent, which cost verdicts nobody chose: a denial whose participle member stands
    // before a comma of its own armed it, so `No inscriptions, watermarks, or logos are visible.` and
    // `No footnotes, annotations, or stamps are present.` โ€” pure denials, nothing affirmed anywhere in
    // them โ€” were reported as holes. Dropping the bound honours them again (5 of 7 such wordings probed,
    // 30 of 30 rows on the frame #434's own grid gave up), and it costs nothing where the bound was aimed:
    // over that grid, 30 words of the vocabulary in the two second-clause frames, dropping it moves 0 of
    // 60 rows, because `secondClauseJoint` refuses all 60 on the sentence's shape instead. Over the
    // corpus โ€” 3,747 replies with a log, 204 blank declarations โ€” the discriminator moves 0 verdicts and
    // dropping the bound moves 0 more.
    if (QUALIFIER.has(word) || CHAIN_LINK.has(word) || AFFIRMED_NOUN.test(word)) continue;
    if (NAMES_TEXT_FORM.test(word)) continue;
    return false;
  }
  return false;
}

// #436's discriminator, and it is a CLAUSE BOUNDARY rather than the comma. Measured over every page reply
// on disk at `4859480` โ€” 3,747 replies with a parseable log, 204 blank declarations โ€” the two classes the
// issue asks to be split are 69 and 0: every declaration whose denial reaches over a comma and a
// conjunction to a name for text is a list (`No text, images, or other content is visible.`), and the
// second-clause class the issue is about appears nowhere in the corpus, in a declaration or out of one. So
// the comma narrowing is refused with a number rather than an argument, as the block above records: read a
// comma plus a conjunction as a boundary and 38 of those 204 declarations stop being honoured.
//
// What the corpus does not settle is the ASYMMETRY, which is a defect in this file rather than a decision
// in the docs, and it is the pair of finite verbs the issue names:
//
//   `No printed words, lines, or characters are visible.`  one verb, shared by three nouns  -> a list
//   `No printed text, and handwriting is present.`         a verb on each side of the joint -> a clause
//
// Four things have to hold, and each one is what keeps a real denial out. The denied half must be
// VERBLESS, which `negatedInList` already guarantees โ€” a verb is none of the words it steps over, so it
// ends the walk before the negator is reached and this is never asked. The crossed span must hold EXACTLY
// ONE comma, which is what separates two clauses from three or more members: every one of the 38 has two
// commas or none. The noun must have a verb OF ITS OWN, read off the `reach` array the caller already
// computed, so a fragment behind a denial (`No printed text, and handwriting.`) stays a member โ€” whether
// that fragment affirms is #435's question, answered in `verblessAffirmation` and not here. And the
// COORDINATION MUST NOT CONTINUE across the noun: no second conjunction and no second comma between the
// joint and that verb.
//
// The fourth is what the joint word cannot do on its own, and round 1 of this change's review is why it is
// written down. Requiring `, and` looked like the discriminator โ€” a denial's members are joined under
// negation with `or` (`no text, images, or other content`) and a second clause is coordinated with `and` โ€”
// but a clause can be spliced on with a bare comma, `No clear text, scrawled words are visible.`, and
// requiring the `and` shipped 16 of the 16 (word, frame) pairs in #434's grid empty that #434's bound had
// refused. Dropping the requirement then took the LIST with it, because the second member of
// `No printed words, lines, or characters are visible.` sits behind a single comma too. What separates
// them is not the joint at all: the list has `or characters` still to come and the splice has nothing
// between its noun and its verb. So `or` after the joint is refused as a joiner, `and` is allowed, a bare
// comma decides nothing, and the coordination scan decides.
//
// THE LIMIT, stated rather than smoothed: a TWO-member denial with a plural verb โ€” `No text, and images
// are visible.`, `No text or images, document headings are visible.` โ€” has this exact signature and is read
// here as a clause, so a declaration whose log meant to deny both is refused. Nothing in the sentence
// separates the two readings; the second of those was pinned as a declaration by #379 and is pinned as a
// refusal now, because leaving it declared while `No clear text, scrawled words are visible.` is refused is
// the vocabulary deciding again, which is the whole of what #436 asked to be closed. The corpus writes
// neither shape โ€” 0 of the 204 declarations move, whichever joint they use โ€” and the direction is the one
// this section chooses everywhere: that page is reported FAILED and redrawn, where the mistake in the other
// direction ships a sheet of handwriting empty with nothing recorded (#190, #371). A determiner, an `only`,
// a `but` or a full stop still refuse without any of this, as before.
//
// THE COST ON THE OTHER SIDE, stated for the same reason, and it is the fatal direction: because a
// coordination that continues across the named noun is read as a denial still listing, a noun that heads an
// AFFIRMED list is read as a member and its page is delivered empty โ€” `No clear text, stamped words, stamps
// are visible.`, `No printed text, and stamped words, marks are visible.`, whichever joint they use. It is
// taken because the two readings are one sentence: `No printed words, and lines, characters are visible.` has
// that shape and denies three things. #434's bound refused 24 of 24 such rows over its own grid where this
// declares them, but on the vocabulary of the modifier alone โ€” with `printed` in place of `stamped` it
// declared them too โ€” so what changes is that one reading now covers both, not that a control was removed.
// 0 of the 3,747 replies on record write it (round 3 of this change's review, which measured it).
function secondClauseJoint(tokens: Word[], negator: number, i: number, reach: number[]): boolean {
  const verb = reach[i + 1]!;
  if (verb < 0) return false;
  // The noun's OWN comma, which is the one token neither scan below looks at: the first stops before `i` and
  // the second starts after it, so an ASYNDETIC list โ€” members divided by bare commas with no joiner before
  // the last, `No printed words, lines, characters are visible.` โ€” presented one comma behind the noun,
  // nothing between it and the verb, and read as a clause. A comma ON the noun is the coordination
  // continuing across it just as surely as a comma after it (round 2 of this change's review, which found
  // five such wordings reported as holes). The `, and` form never reached this, because there
  // `joint + 2 === i` and the scan below already sees the comma.
  if (tokens[i]!.comma) return false;
  let joint = -1;
  for (let k = negator; k < i; k++) {
    if (!tokens[k]!.comma) continue;
    if (joint >= 0) return false; // two commas: members of a denial, not two halves of a sentence
    joint = k;
  }
  if (joint < 0 || joint + 1 >= tokens.length) return false;
  // `or` after the joint says the denial is still listing. `and` does not decide either way, and a bare
  // comma decides nothing at all. `CONJUNCTION` holds only those two, so this line is about `or` alone:
  // `No text, nor images are visible.` is already blank before this function is asked, because `nor` is a
  // NEGATOR and the walk ends at it.
  const after = tokens[joint + 1]!.word;
  if (CONJUNCTION.has(after) && after !== "and") return false;
  // Whether the coordination CONTINUES ACROSS this noun, which is what a list does and a clause does not:
  // another comma or another conjunction anywhere between the joint and the verb the noun reaches for.
  // Both sides of the noun, because a list's last member has its joiner behind it (`No writing, figures or
  // stamps are present.`) and its first has one in front (`No text, and images or figures are visible.`),
  // and either one is the denial still listing. The `and` at the joint itself is the one this skips: that is
  // the coordinator of the clause, and it is the only word between the comma and the noun that can be one.
  for (let k = joint + 2; k < verb; k++) {
    if (tokens[k]!.comma || CONJUNCTION.has(tokens[k]!.word)) return false;
  }
  return true;
}

// Where the verb that affirms the noun at each position is, if there is one. The scan STOPS at a negator, because a
// negator is where the next denied clause begins and its verb has nothing to do with this noun โ€”
// which is the rule `deniesToStatementEnd` already applies for the same reason, and the reason a
// blank page's own log survives this: "not legible text or content, and no writing is visible" has an
// `is` in it, three words past a `no` that owns it.
//
// The exception is a denial set off by COMMAS. "โ€ฆand printing, no page number, is visible" is a
// parenthetical with `printing` as the subject of `is visible`, and it was the one of #194's five
// shapes that the stop rule alone would have let through. So a negator whose phrase both opens after
// a comma and closes on one is stepped over rather than stopped at. Nothing a blank page is written in
// has that shape: the negators in those logs open on `and`, on `or` or on a fresh clause ("โ€ฆor
// content, and no printed page number is visible"), where the closing comma never comes.
//
// Computed for every position of the statement in one backwards pass rather than scanned forward per
// noun. Same answers โ€” each position is the answer for the position after it, unless the token is
// itself a verb or a negator โ€” and it is the one scan in this section that was quadratic in the
// length of a statement with no full stop in it, which is the shape `DENIAL_STATEMENT_MAX` caps for
// the same reason. A cap would not do here: a statement too long to read is one whose contradiction
// goes unfound, and that costs the page rather than a glance.
function affirmingReach(tokens: Word[]): number[] {
  const reach = new Array<number>(tokens.length + 1).fill(-1);
  for (let k = tokens.length - 1; k >= 0; k--) {
    const token = tokens[k]!;
    if (AFFIRMING_VERB.has(token.word)) {
      reach[k] = k;
      continue;
    }
    if (NEGATOR.has(token.word)) {
      if (k === 0 || !tokens[k - 1]!.comma) continue; // stays -1: the clause here is denied
      let close = k;
      while (close < tokens.length && !tokens[close]!.comma) close++;
      if (close < tokens.length) reach[k] = reach[close + 1]!;
      continue;
    }
    reach[k] = reach[k + 1]!;
  }
  return reach;
}

// A verb the negator FOLLOWS denies its clause, and the subject-verb scan is the one shape in this
// section that could not see it: `negatedInList` reads the words in front of the noun and
// `affirmingReach` marks a verb from its own position, so nothing looked one word to the right of the
// verb and `Text is not present.` read as an affirmation of `text` โ€” a blank page reported lost, with
// the negator quoted inside the evidence for it (`affirmed: "text is not"`, which is the tell).
// `agents/page.md` asks the agent to say in the log that the page is empty and does not fix the wording
// of the denial, so subject-verb-`not` is as ordinary an answer as `no text is present`, and #190
// recorded that which pages get lost was being decided by the wording the model happened to pick.
//
// The negator must be the word RIGHT AFTER the verb, with no gap allowed. Every wording on this axis
// puts it there (`is not present`, `was never printed`, `are not present`), and a gap would cost a
// page rather than a glance: "Handwriting is clear, nothing else is on the sheet." affirms
// handwriting, and reaching past `clear` to that `nothing` would refuse the affirmation and ship the
// page as empty in silence โ€” the #194 defect this check exists to close. Same reason `but` is not
// followed: "Handwriting is visible but no printed text is present." is a page with writing on it.
//
// `never` and the complements below earn their place here rather than in `NEGATOR`, which five other
// functions read: "text was never printed" and "printed text is absent" are denials, but widening the
// negator vocabulary would change how the denial tails, the object gaps and the coordination walk all
// read at once, and each of those trades in the other direction.
//
// A negative complement is how the same denial is written without a negator at all โ€” `is absent`,
// `is missing`, `is nowhere on the sheet` โ€” and #200's review measured eight such wordings reported
// failed. The list is closed and short on purpose: each word means "not there" on its own, with no
// reading where it says something IS on the paper, which is what separates it from an open-ended
// widening. `devoid` and `lacks` are absent because they take a preposition or an object and the
// subject-verb scan does not reach them anyway. A qualifier between the verb and the complement
// ("Text is entirely absent.") is still refused, for the same reason no gap is allowed above: the gap
// costs a page and the refusal costs a glance.
// A denial does not have to cover the whole sheet, and #204 is the shape where it does not: the log
// denies one part of the page and says in the same breath what is on the rest of it.
// `denialAffirmations` below reads that, and the three constructions it reads are the three the corpus
// and #200's review put on record.
const NEGATIVE_COMPLEMENT = new Set(["absent", "missing", "nowhere", "nonexistent", "lacking"]);

// The other way a copula denies its subject, and #442: the complement says the subject HAS nothing
// rather than that the subject is not there. `The heading is empty.` and `The printed form is empty.`
// were read as affirmations of `heading` and `form` โ€” a blank page reported as a hole, with the
// complement that denied it quoted inside the evidence (`affirmed: "heading is empty"`).
const ABSENCE_COMPLEMENT = new Set(["empty", "blank", "unmarked", "unfilled", "featureless"]);
function absenceComplement(tokens: Word[], k: number): boolean {
  const word = tokens[k]?.word;
  if (word === undefined) return false;
  if (ABSENCE_COMPLEMENT.has(word)) return true;
  // `void` only with its preposition. A stamp that "is void" is a mark ON the paper โ€” the word is
  // printed across a cancelled form โ€” so bare `void` is the one member of this vocabulary with a
  // reading that says something IS there, and taking it would lose that page in silence.
  return word === "void" && tokens[k + 1]?.word === "of";
}

// A CONTRACTED copula is `is not` with the negation fused into the token, and no list in this file reads
// one as a verb: `AFFIRMING_VERB` holds no contraction on purpose, because `The heading isn't visible.`
// denies its subject and a walk that took the token would have to un-take it. That left one spelling
// split from its twin โ€” `The heading is not empty.` says the heading HAS something in it and reports the
// page (the double-negative branch of `deniedAfterVerb`), while `The heading isn't empty.` said the same
// thing and delivered the page empty, so an apostrophe decided whether the page was lost. Read here, at
// the one construction where the contraction's own negation is cancelled by the complement behind it.
// WALKED and not read at the next token, because the subject of one of these is a noun PHRASE: `The printed
// form isn't empty.` and `The typed entries weren't unfilled.` put the contraction two tokens past the word
// the affirmation read is standing on, and a one-token check saw `form` and `entries` and stopped. That is
// the same walk `affirmingReach` does for a plain verb, and it stops where that one stops: a real affirming
// verb ahead means the ordinary path owns the sentence and has an answer for it already, and a negator ahead
// denies the clause the contraction is in.
// Returns the contraction's own position, the way `affirmingReach` returns a verb's, so the evidence line
// can quote the subject through the complement and not a fixed three words.
const CONTRACTED_COPULA = new Set(["isn't", "aren't", "wasn't", "weren't"]);
function contractedDoubleNegative(tokens: Word[], from: number): number {
  for (let k = from; k < tokens.length; k++) {
    const { word } = tokens[k]!;
    if (CONTRACTED_COPULA.has(word)) return absenceComplement(tokens, k + 1) ? k : -1;
    if (AFFIRMING_VERB.has(word) || NEGATOR.has(word)) return -1;
  }
  return -1;
}

// The complements that say something IS there, for the coordination read: `absent from the top half
// and PRESENT at the bottom`. Overlaps `QUALIFIER` on purpose rather than reusing it โ€” a qualifier is
// what may stand between a verb and its noun, and half of that list (`meaningful`, `other`, `more`)
// says nothing about whether anything is on the paper, while `printed` and `written` belong here and
// are the wording a partial denial reaches for ("absent from the top and printed at the bottom").
// The overlap with `TEXT_NOUN` on `printed` is NOT deliberate in the same way: see the object walk.
const AFFIRMING_COMPLEMENT = new Set(
  "present visible legible readable discernible apparent recognizable recognisable shown printed typed written handwritten stamped".split(" "),
);
// What a predicate complement may tail into. A complement standing at the end of its clause, or
// running on into a place, is the clause's own predicate โ€” `and present at the bottom`, `and still
// visible`. One with a noun after it is an adjective ON that noun, and the noun decides: `but visible
// dust remains` is scanner dust on a page that is still blank. Closed and short, like every list
// here; a wording that tails into something not on it costs a page that says so twice.
const AFTER_COMPLEMENT = new Set(
  "at in on within across throughout near under above below beneath over beside along to toward towards here there elsewhere everywhere only also too instead".split(
    " ",
  ),
);
// How many adverbs may stand between the joiner and the complement. `and still clearly visible` is
// two, and no wording in the corpus has three; the bound is here rather than absent because the walk
// runs at every position of the statement and an unbounded run of them is a second quadratic.
const ADVERB_GAP = 3;
// What may join the two halves. `or` is deliberately absent: `absent or present` is not a statement
// about a page, and every `or` in these logs joins members of a denied list, which is the one thing
// this section may not start reading as an affirmation (`negatedInList`'s eleven pinned wordings).
const CONTRAST = new Set("and but yet while whilst though although".split(" "));
// An adverb that may stand between the joiner and the complement: `and still present`, `but clearly
// visible`. Kept apart from `QUALIFIER` for the reason above.
const CONTRAST_ADVERB = new Set("still also clearly plainly instead however again nevertheless".split(" "));
// The exceptive prepositions. What follows one of these after a denial is asserted to be on the page
// โ€” that is the whole job of the word โ€” so no article is required, which is what separates this read
// from the prepositional one below: `nowhere except a stamp at the top` introduces the stamp for the
// first time. `but` is here as well as in `CONTRAST` (`nothing but a stamp`), and is safe in both
// because the object walk refuses anything that is not a name for text.
const EXCEPTIVE = new Set("except excepting besides but apart aside save excluding".split(" "));
// `other` is exceptive in `nowhere other THAN a stamp` and is not exceptive in `no other text is
// present`, where it is the qualifier five other functions in this file read it as โ€” the second is a
// wording a blank page uses and the first is a page with a stamp on it. So the pair is required, and
// nothing else in this file has to change.
const EXCEPTIVE_PAIR = new Map([["other", "than"]]);
// What may stand between an exceptive and its object: `except for the heading`, `other than a stamp`,
// `apart from the signature`.
const EXCEPTIVE_GAP = new Set("for than from of".split(" "));
// A locative preposition whose object is a thing the page bears โ€” `missing from the figure on the
// page`, `absent from the diagram shown here`. The object has to be DEFINITE here, and that is the
// whole safety of this read: a denial names the substrate indefinitely as often as not (`absent from
// a page this faint`), while a definite noun is one the log has been talking about, so it exists.
// `into` is deliberately not a member: `do not resolve into characters` and `nothing that resolves
// into words` are the commonest denials in the corpus, and their objects are exactly what is NOT
// there.
const DENIAL_PREPOSITION = new Set("from on in at within across beside under above near beneath over".split(" "));

// The affirmation hiding behind a denial, as the index of the word that carries it, or -1. Reached
// only once the clause has already been read as denied, so everything here is about what the log says
// about the REST of the page.
//
// Three reads, and the direction of error is what picks each one. An affirmation this misses ships a
// page with a stamp on it as blank paper, in silence and with no marker (#179/#190/#194); an
// affirmation it invents reports a blank page as failed, which costs a glance. So each read is written
// to fire on the shapes the corpus contains and to stay off the denials pinned in
// `test/envelope-as-content.test.ts`, and where the two could not both be had, the glance wins.
//
//   1. A contrast whose complement affirms. `Text is absent from the top half and present at the
//      bottom.` โ€” one subject, two complements, and the second says the text is there. The joiner
//      must come FIRST and the complement immediately after it, which is what keeps `Handwriting is
//      not present either.` a denial: its `present` sits behind the negator with no joiner in front
//      of it. Only reached with a subject in hand, since the affirmation is about that subject.
//   2. An exception. `Printing is nowhere except a stamp at the top.` The exception is the content:
//      a log that says what is NOT on the page and then names the one thing that is has described a
//      page with something on it, whatever the proportions.
//   3. A definite noun in a locative object. `A caption is missing from the figure on the page.`
//      denies the caption and presupposes the figure. `the`/`this`/`its` is required, for the reason
//      `LOCATIVE_SUBSTRATE` and `definiteBefore` exist: an indefinite noun there is as likely to be
//      the scan or the paper as a thing on it, and `image` stays the scan however it is introduced.
//
// The walk stops at a negator, because a second denial in the same statement is a second denial and
// not the affirmation this is looking for ("Text is absent and no handwriting is present." is a blank
// page). It does not stop at a new subject: a clause of its own is found by the caller's own loop,
// which reads every name for text in the statement, so anything this walk reaches past has already
// been offered a verb of its own.
//
// Read 1, at one position. -1 for no affirmation here, which is not terminal: a joiner that leads
// nowhere is just a joiner.
function contrastAffirmed(tokens: Word[], k: number): number {
  if (!CONTRAST.has(tokens[k]!.word)) return -1;
  let j = k + 1;
  while (j < tokens.length && j <= k + ADVERB_GAP && CONTRAST_ADVERB.has(tokens[j]!.word)) j++;
  const complement = tokens[j];
  if (complement === undefined || !AFFIRMING_COMPLEMENT.has(complement.word)) return -1;
  // The complement has to be its clause's predicate and not an adjective on some other noun. Reads 2
  // and 3 below ask their object to be a name for text and this read has no object to ask about, so
  // the question it can ask is what the complement modifies: `and present at the bottom` predicates
  // over the subject the caller is holding, while `but visible dust remains` says something about
  // dust, and `dust` is outside `TEXT_NOUN` for the reason #193 put it there. Without this the one
  // read that cannot see the noun class reported blank pages with scanner dust on them as failed.
  // A name for text after the complement affirms whichever way it is read, so it passes too โ€” in
  // either part of speech, since what makes this safe is that the word names text and not which form
  // of it the log wrote. `but still visible handwritten notes` says what `but still visible
  // handwriting` says, and `dust` is outside both lists.
  const next = tokens[j + 1];
  if (next === undefined || complement.comma || AFTER_COMPLEMENT.has(next.word) || affirmsText(next.word)) {
    return j;
  }
  return -1;
}

// Reads 2 and 3, at one position: the object of an exceptive or of a locative preposition. `null` for
// no affirmation here, -1 for a stop โ€” a negator standing where the object goes denies the rest of
// the statement as surely as one standing on its own.
function exceptiveOrLocativeObject(tokens: Word[], k: number): number | null {
  const { word } = tokens[k]!;
  const paired = EXCEPTIVE_PAIR.get(word);
  const exceptive = EXCEPTIVE.has(word) || (paired !== undefined && tokens[k + 1]?.word === paired);
  if (!exceptive && !DENIAL_PREPOSITION.has(word)) return null;
  const definiteOnly = !exceptive;
  for (let m = k + 1; m < tokens.length && m <= k + 1 + OBJECT_GAP; m++) {
    const object = tokens[m]!.word;
    if (NEGATOR.has(object)) return -1;
    // `printed` is in `TEXT_NOUN` and in `QUALIFIER` both, and here the qualifier reading is the only
    // one that can be right: `in the printed area of the form`, `within the printed border`, `on the
    // printed side` are how a blank pre-printed form and a blank verso are described, and taking the
    // adjective for the object reported all of them failed โ€” with the same truncated evidence #190
    // left behind (`affirmed: "no content is present in the printed"`). Definiteness cannot help,
    // since `the` is what makes the read fire at all. Skipping it loses nothing, because the real
    // object is still ahead when there is one: `from the printed heading` affirms one word later.
    // `affirmedObjectAfter` keeps reading the word as a noun, because a match there ADDS an
    // affirmation and the error runs toward a glance; here it runs toward a false failure notice.
    //
    // #431's second part of speech is read here too, but not on the same terms, because this is the one
    // caller where the `printed` shape above recurs across a whole list. Most of `NAMES_TEXT_FORM` is
    // participles, and `in the watermarked margin`, `on the typed side`, `within the stamped border`,
    // `from the embossed edge` are all how a blank pre-printed form or a blank verso gets described โ€”
    // while `!QUALIFIER.has(object)` can only cover `printed` and `typed`, the two members that list
    // happens to hold. Refusing the whole vocabulary here would be the cheaper mistake but it is not a
    // free one: the nouns `TEXT_NOUN` lacks and `NAMES_TEXT_FORM` carries are content by any reading โ€”
    // `barcode`, `watermark`, `footnote`, `monogram`, `letterhead`, `drawing`, `sketch`, `annotation`,
    // `inscription`, `symbol` โ€” and `nowhere except a barcode at the top` is a page with something on it.
    //
    // So the question this asks of the second list is the one `contrastAffirmed` asks of its complement:
    // what does the word MODIFY. An object phrase that ENDS at the word โ€” nothing after it, a comma, or
    // a preposition or adverb starting the next phrase โ€” is one where the word is the object; a word with
    // a noun still behind it is an adjective on that noun, and `watermarked margin` is the margin. That
    // separates `except a barcode at the top` from `in the watermarked margin` without a second list,
    // and what it gives up is `except something handwritten`, where the noun is absent rather than
    // present in the other form.
    //
    // `QUALIFIER` is excluded on THIS branch too, and it has to be because this one runs first: `typed`
    // is in `QUALIFIER` and in `NAMES_TEXT_FORM` both, so without the exclusion `missing from the typed
    // heading` broke where `missing from the printed heading` affirms โ€” two wordings differing only in
    // which of the two overlapping words they use. Skipping it costs nothing for the reason above: the
    // real object is still ahead when there is one. The exclusion is token-exact and the list it guards
    // against is boundary-tested, so it settles the pair for the bare words and not for their compounds
    // (`pre-typed` against `pre-printed`). #437 closed that: the exclusion is `qualifies`, which reads the
    // compound as its own stem, so `in the machine-printed` is skipped exactly as `in the printed` is. It
    // has to be, because `printed` joining `NAMES_TEXT_FORM` in the same change made every compound of it
    // reach this branch โ€” the widening and this exclusion are one edit in two places.
    if (NAMES_TEXT_FORM.test(object) && !AFFIRMED_NOUN.test(object) && !qualifies(object)) {
      const after = tokens[m + 1];
      if (after === undefined || tokens[m]!.comma || AFTER_COMPLEMENT.has(after.word)) {
        if (definiteOnly && !definiteBefore(tokens, m)) break;
        return m;
      }
      break;
    }
    if (AFFIRMED_NOUN.test(object) && !QUALIFIER.has(object)) {
      if (LOCATIVE_SUBSTRATE.has(object)) break;
      if (definiteOnly && !definiteBefore(tokens, m)) break;
      return m;
    }
    // The gap the object walk may cross, and `qualifies` here is the other half of the same fix: with the
    // exclusion above skipping a compound, the walk has to keep going past it to the noun behind it, or
    // `from the machine-printed heading` breaks where `from the printed heading` affirms โ€” a presupposed
    // heading lost, which is the silent direction. 12 of #437's 24 disagreeing pairs were this line, and
    // they disagreed for `typed` as well as `printed`, so this half was never about the widening.
    if (!DETERMINER.has(object) && !qualifies(object) && !QUANTIFIER.has(object) && !EXCEPTIVE_GAP.has(object)) {
      break;
    }
  }
  return null;
}

// Both walks, for every position of the statement, in one backwards pass. Same answers as scanning
// forward from each denial โ€” the answer at a position is the answer at the position after it unless
// the token there carries a hit or a stop โ€” and computed this way for the reason `affirmingReach`
// is: forward-per-denial was quadratic in the length of a statement with no terminator in it, which
// is the one input shape that has no bound. #204's review measured it at 36k words: 9.8ms before the
// walk existed, 7.8s after, on a server that runs one request at a time. A cap will not do here for
// the reason given there โ€” a statement too long to read is one whose contradiction goes unfound, and
// that costs the page rather than a glance.
//
// `subject` is a yes/no rather than a position, so two arrays are enough: read 1 predicates over the
// subject the caller has in hand, and is off in the scan that has none.
function denialAffirmations(tokens: Word[]): { withSubject: number[]; plain: number[] } {
  const withSubject = new Array<number>(tokens.length + 1).fill(-1);
  const plain = new Array<number>(tokens.length + 1).fill(-1);
  for (let k = tokens.length - 1; k >= 0; k--) {
    const { word } = tokens[k]!;
    if (NEGATOR.has(word) || word === "never") continue; // both stay -1: this is where the walk stops
    const object = exceptiveOrLocativeObject(tokens, k);
    plain[k] = object === null ? plain[k + 1]! : object;
    // Read 1 first at the same position, which is what makes `nothing but a stamp` an exception and
    // `absent but still visible` a contrast: `but` is in both lists and the complement decides.
    const contrast = contrastAffirmed(tokens, k);
    withSubject[k] = contrast >= 0 ? contrast : object === null ? withSubject[k + 1]! : object;
  }
  return { withSubject, plain };
}

function deniedAfterVerb(tokens: Word[], verb: number): boolean {
  const next = tokens[verb + 1];
  if (next === undefined) return false;
  // An absence complement is read after a LINKING verb only, and that gate is the whole difference between
  // this list and the older one. `deniedAfterVerb` is called with whatever verb the reach found, and half of
  // `AFFIRMING_VERB` takes an OBJECT rather than a complement โ€” where `empty`, `blank` and `unmarked` are
  // the ordinary adjectives for a cell, a field or a row, so `The heading contains empty rows.` read as a
  // denial and shipped a page of rows out empty (round 1 of #446). `absent`, `missing` and `nowhere` never
  // needed the gate because none of them is attributive: nothing contains missing rows.
  //
  // Asking the OTHER list which verbs take an object is what couples the two, and the coupling is why
  // `AFFIRMING_VERB` carries a warning at its own definition: this gate is complete only while every
  // object-taker in that list is also in this one. Today that is `contain contains hold holds`, all four
  // present here. A member added there and not here is a hole this function cannot see.
  const linking = !TRANSITIVE_AFFIRM.has(tokens[verb]!.word);
  if (NEGATIVE_COMPLEMENT.has(next.word) || (linking && absenceComplement(tokens, verb + 1))) return true;
  if (!NEGATOR.has(next.word) && next.word !== "never") return false;
  // A negator in front of a negative complement is two denials making an affirmation: "Handwriting is
  // not absent." is a page with writing on it, and reading it as a denial would ship that page empty
  // in silence. Contrived beside `is not present`, and it costs one lookup to not get wrong.
  //
  // Not contrived for an absence complement, which is why that list is read here as well as above:
  // `The heading is not empty.` is the ordinary way to say a field was filled in, and on base it
  // DECLARED โ€” the `not` denied the clause and nothing looked at what it denied. That page shipped
  // empty, so this half of #442 is the expensive direction and the grid pins it.
  const after = tokens[verb + 2];
  if (after === undefined) return true;
  return !(NEGATIVE_COMPLEMENT.has(after.word) || (linking && absenceComplement(tokens, verb + 2)));
}

// The noun a post-verb construction affirms โ€” the object of `there is` or of a transitive verb โ€” or
// -1. Shared by both because both ask the same question of the same words.
function affirmedObjectAfter(tokens: Word[], verb: number): number {
  for (let k = verb + 1; k < tokens.length && k <= verb + OBJECT_GAP; k++) {
    const { word } = tokens[k]!;
    // A negator ends it: "there is nothing to transcribe", "there is no text", "the page contains no
    // legible printing" are how a blank page says it, and they are the commonest of these in the
    // corpus.
    if (NEGATOR.has(word)) return -1;
    // Both parts of speech: "There is cursive on it", "the page bears a watermark", "it shows
    // handwritten notes" are the same affirmation as the noun-form wordings this already read, and the
    // gap of determiners, qualifiers and counts below ends at the first word that is none of those โ€”
    // so without this an attributive standing where the object goes ended the walk instead of being
    // the object.
    if (affirmsText(word)) {
      // `image` is read by its article here as it is everywhere else in this file: "the frame
      // contains the image" is the scan being described, not a photograph on the paper.
      return LOCATIVE_SUBSTRATE.has(word) && definiteBefore(tokens, k) ? -1 : k;
    }
    if (!DETERMINER.has(word) && !QUALIFIER.has(word) && !QUANTIFIER.has(word)) return -1;
  }
  return -1;
}

// A marks noun as a single token, for the one thing a fragment's noun phrase may put between its name
// for text and its end: `handwriting smudges only.` `MARK` is written for the phrase reads, so its
// multi-word branches (`stray markings`, `scanner noise`) cannot match one token and do not here, and
// bare `marks` is not a member of it at all โ€” #193's decision inherited rather than retaken.
//
// That last one is where the fragment read is NARROWER than the verb read, stated because the pair looks
// like an inconsistency and is one: `affirmingReach` crosses any word to find a verb, so `Handwritten
// marks are visible.` refuses the declaration, while `Handwritten marks only.` declares it, because
// `marks` is not a word this walk may cross. Widening the walk to any noun is what would match them, and
// that is the direction #193 refused โ€” `marks` is what a blank page calls the dust on it.
const MARK_NOUN = new RegExp(`^(?:${MARK})$`, "i");
// What may stand in FRONT of the noun in a fragment and leave it the thing the fragment is about.
// Kept apart from `QUANTIFIER` and `DETERMINER` rather than added to either, because these three words
// are the ones a marks phrase also opens with (`MARK_QUANTIFIER` lists them) and the only reason they
// are needed here is `Only handwriting smudges.` โ€” the wording in #435's title.
const FRAGMENT_OPENER = new Set("only just merely simply solely".split(" "));
// What a fragment's noun phrase may tail into and still be a fragment: `handwriting only`, `a heading
// too`. Read at the END of the statement and nowhere else, so `only` closing a phrase is separated
// from `only` qualifying whatever comes next (`handwriting only in the margin` is not read here).
const FRAGMENT_CLOSER = new Set("only alone too also".split(" "));

// The affirmation in a statement with no verb in it, as the index of the word that ends it, or -1.
//
// `contentAffirmed`'s subject read hands a name for text to the verb that predicates over it, and a
// fragment has no verb: `affirmingReach` returns -1 and the noun is dropped. So on base "Page is
// blank. handwriting smudges only." delivered a page with writing on it as empty, and 7 of the 13
// wordings #435 measured were fragments of this shape. `PREDICATED` exists for exactly this
// construction โ€” "'heading visible' is 'a heading is visible' with the copula dropped" โ€” and was
// reachable only through the DENIAL reads, gated on a `DENIAL_WORD`, so nothing in the affirmation
// path consulted it.
//
// The risk here runs OPPOSITE to the rest of this section, and that is what shapes the rule: affirming
// off a noun with no verb gives `NEGATED` and `negatedInList` less to work with, and a blank page's own
// log is as often a fragment ("Just specks.", "Only scanner dust.", "No text, nothing legible.").
// So the noun phrase must OPEN its statement, which is the same bound #227 drew at
// `deniesToStatementEnd` and for the same reason: a fragment IS its noun phrase, and a name for text
// standing in the middle of one has something in front of it that decides what it is doing there.
// `Devoid of text.`, `Lacking text.` and `Free of text.` are the cases that pays for โ€” none of those
// words is a `NEGATOR` or a `NEGATIVE_COMPLEMENT` the backward walk reads (`devoid` and `lacks` are
// deliberately out of that list), and each is a blank page that would otherwise be reported lost.
// What it cost is `Blank apart from a caption.`, where the exceptive read wanted a denial the fragment
// did not have โ€” so the bound left a defect rather than buying one, and #442 closed that half by making
// the absence COMPLEMENT a denial to that scan. The locative half of the same shape (`A heading at the
// top.`) is still open, and the two are pinned in opposite directions to keep the difference visible.
//
// Forward, a predicate ends it and so does the end of the statement, because a fragment whose whole
// text is a name for text is an affirmation with nothing left to qualify it. Everything else refuses,
// which is where `Text absent.` and `Text nowhere on the sheet.` stay denials โ€” not by listing the
// complements, but by not crossing them.
//
// A LOCATIVE tail is the case that refusal leaves open: `A heading at the top.` and `handwriting only in
// the margin.` are pages with writing on them, and both declare. Deliberately not read here, because it
// is not one of the seven wordings #435 measured and it is not a free widening โ€” `Text in this image.`
// and `Nothing on the sheet apart from a stamp.` are two shapes a locative tail would newly have to get
// right, and the corpus separates none of them (0 of 201 declarations move either way on this change).
// `A heading at the top.` is already pinned as delivered, beside two more of its shape, in
// `envelope-as-content.test.ts` โ€” the pins that say a widening must defend both halves of each pair.
// A statement holding nothing but one name, whitespace and whatever the marks strip left: what the guard
// below fires on, and asked of a NEIGHBOUR as well, because a run of them is a list and not a lone name.
function loneNameStatement(statement: string | undefined): boolean {
  if (statement === undefined) return false;
  const tokens = words(statement);
  return (
    tokens.length === 1 &&
    statement.replace(/[\s\f\v]+/g, "").toLowerCase().replace(/[โ€™]/g, "'") === tokens[0]!.word
  );
}
function verblessAffirmation(tokens: Word[], i: number, statements: string[], s: number): number {
  // A statement whose whole text is the name affirms โ€” and this read splits statements on `.`, `!`, `?`,
  // `;` and line breaks alike, so "Blank page; text", "Page is blank; images; nothing present.",
  // "Page is blank. No printed text. Images." and "Page is blank. Any text? None found." each refused
  // the declaration off ONE WORD with no determiner, no count and no predicate. In three of those the
  // denial is in a neighbouring statement โ€” behind the word in one, ahead of it in another โ€” and this
  // read sees neither: the boundaries are what limit how far a subject may reach, so a word alone
  // between two of them is all there is to read. `Text: none.`, `Text (none).` and
  // `Page is blank; no text; no images.` are the near misses that always declared, so the hole was the
  // one-token statement and not the label list.
  //
  // The marker is what closes it, and it is why a `tokens.length === 1` guard alone was written and taken
  // back out in `c43dff9`: ONE TOKEN IS NOT ONE WORD. `Handwriting smudges.` and `Cursive smudges.`
  // arrive here as a single token too, because `vetoScope` removed the head noun โ€” and that phrase is
  // #435's own, six of the seven wordings this function exists for being it with a predicate on the end.
  // So the bare guard bought three unobserved blank pages reported as holes and paid a page of
  // handwriting delivered empty, which is the failure #190 and #435 are both about. `PHRASE_GONE` is the
  // missing information: a marker in the statement's own text says a phrase was removed from THIS
  // statement, so the two are now different cases and each gets its own answer.
  //
  // What it does NOT separate, stated because the guard's cost is real and unchanged there: `Page is
  // blank. handwriting.` is one word that was always one word, so it now declares blank and the page
  // ships empty. Nothing distinguishes it from `Blank page; text` in the text โ€” the whole difference is
  // which noun, and a list of the nouns that may stand alone is the same list this file has refused to
  // write everywhere else. It is pinned as delivered-empty beside its rescued neighbours, and the trade
  // is 5 wordings that reported a blank page as a hole against 1 that ships a page of handwriting.
  //
  // Both sides are unobserved, and #440's own version of that figure is wrong in a way worth writing down:
  // one-token statements are COMMON โ€” 1,129 of the 3,747 replies write one, 3,402 in all, most of them a
  // table cell or a file name (`page`, `png`, `n`) that names nothing. Four of them name text, over three
  // spellings (`paragraph`, `heading`, `line`), and every one is in a reply that does not claim blankness,
  // so 0 of the 204 declarations on disk holds one and this guard moves no verdict on record. Narrower than
  // even that says, once the guard asks the statement to BE the token: only 1,073 of those 3,402 are bare,
  // 2,329 carry decoration, and all four that name text are among the decorated ones (`' paragraph`,
  // `Heading '9`, `heading '3`, `' line`). So the corpus reach of this guard is 0 statements and not 4, and
  // a version reading THROUGH decoration would move those four toward shipped-empty and none toward
  // declaring โ€” which is why `**handwriting**` goes on reporting where `handwriting.` declares, an
  // asymmetry pinned rather than smoothed.
  //
  // ONE TOKEN IS NOT THE WHOLE STATEMENT either, and that is the second half of the same mistake.
  // `words()` tokenizes `[A-Za-z][A-Za-z'โ€™-]*`, so a digit and a bullet are invisible to it: `2 images.`
  // and a `- text` line are one token each, and a guard reading the count alone would take the count and
  // the list marker for nothing at all โ€” a page that says what is on it, delivered empty, which is the
  // losing direction. So the statement must BE the token: nothing in it but the name, whitespace and the
  // marker. `two images.` was never at risk (two tokens) and `2 images.` must answer as it does; a
  // bulleted enumeration of a page's contents is one token per line and every line keeps its affirmation.
  // Reported by the review on PR #444. A NUMBERED enumeration is the same shape and is not fixed by this,
  // because `1.` is a boundary rather than decoration โ€” see the paragraph below `bare`.
  // Compared at `words()`'s own normalization, which lowercases and folds the curly apostrophe: `Content`
  // and `pageโ€™s` have to compare equal to the tokens they produced.
  //
  // AND A BOUNDARY IS NOT ALWAYS A SENTENCE END, which is the third face of the same mistake and the one
  // the `bare` fix above created. A numbered list marker ENDS IN A `.`, so `1.` is a boundary and every
  // line of `Page is blank.\n1. text\n2. images` arrives here as a bare single token: the whole enumeration
  // of what is on the page was eaten and the page shipped empty, while the `-` bulleted spelling two
  // paragraphs up is rescued. The premise of this guard is that a name alone BETWEEN TWO BOUNDARIES is all
  // there is to read โ€” and that only holds where the boundary behind it ended a sentence. A preceding
  // statement with no letter in it is a marker and not a sentence, so the name is a list item and affirms.
  //
  // Keyed on "no letter" rather than on a marker vocabulary because the corpus says which spellings exist:
  // 73 of the 3,747 replies write a `1.` list line and 2 write a `-` one, while `1)`, `a.`, `a)` and roman
  // numerals appear in ZERO โ€” so a lettered or parenthesised branch would be a guess, and the digits are
  // the whole observed population. (The lettered spellings `i.` and `A.` end up handled anyway, by the
  // sequence clause below: a marker that IS a letter is itself a lone-name statement.) What that test
  // actually catches is wider than "a marker", and the
  // difference is worth having in writing: a statement the marks strip reduced to `PHRASE_GONE` has no
  // letter in it either, so `Page is blank. Print artifacts. text` reads its emptied neighbour as a marker
  // and hands `text` back. That is base's own answer for it and the safe direction, but a later narrowing of
  // this test toward real markers would start declaring those, which is why the clause is here and not
  // implied. Reported by the review on PR #444.
  //
  // AND A MARKER IS NOT WHAT MAKES A LIST โ€” the SEQUENCE is, which is the last face of this and the one the
  // fix above left. `Page is blank.\ntext\nimages` has no marker at all, so every line is a lone name behind
  // a sentence and the guard ate the whole enumeration: a page that listed its own contents, shipped empty.
  // A run of lone names is a list, and one lone name is a lone name, so the neighbour decides โ€” and the
  // rescues #440 exists for all survive it, because every one of them has a SENTENCE on the other side
  // (`Page is blank; images; nothing present.`, `Page is blank. Images. No text.`) or nothing at all
  // (`Blank page; text`). Where it cannot help is a ONE-ITEM list: `Page is blank.\ntext` is
  // `Blank page; text` in every respect this read can see, and it declares. That is the bound, and it is the
  // `handwriting.` cost restated at the level of the shape rather than of the wording.
  //
  // Which clause does the work, over the same 3,747 replies: of the 1,073 bare one-token statements on
  // record, 147 are in an enumeration by this rule โ€” 2 by the letterless neighbour and 145 by the sequence โ€”
  // and NONE of the 147 names text. So the whole guard still moves 0 of the 204 declarations, and the
  // sequence clause is where the population actually is.
  const statement = statements[s]!;
  const previous = s > 0 ? statements[s - 1] : undefined;
  const bare = loneNameStatement(statement);
  const enumerated =
    (previous !== undefined && !/[A-Za-z]/.test(previous)) ||
    loneNameStatement(previous) ||
    loneNameStatement(statements[s + 1]);
  if (tokens.length === 1 && bare && !enumerated && !statement.includes("\f")) return -1;
  for (let k = i - 1; k >= 0; k--) {
    const { word, comma } = tokens[k]!;
    // A comma between the noun and what precedes it opens a fresh phrase, and the words behind it are
    // then about something else: `no text, a heading only` is one denied list (`negatedInList`'s
    // eleven pinned wordings say so), not a denial and a fragment.
    if (comma) return -1;
    if (FRAGMENT_OPENER.has(word) || DETERMINER.has(word) || QUALIFIER.has(word) || QUANTIFIER.has(word)) continue;
    if (affirmsText(word) || MARK_NOUN.test(word)) continue;
    return -1;
  }
  // A comma on the noun ITSELF ends the phrase there, and a fragment cut off at its own noun is a
  // LIST member. That is not a hypothetical shape: `vetoScope` strips the marks and the `not legible
  // text` phrase out of the scope this reads, so the corpus log "A few specks, not legible text,
  // figures, captions visible." arrives here as `A few figures, captions visible` with its negator
  // already gone โ€” a blank page whose remaining nouns look un-negated because the words that denied
  // them were removed. Base survives it by needing a verb; this read has to survive it by the commas,
  // and it is the reason both of them are here rather than only the one behind the noun.
  if (tokens[i]!.comma) return -1;
  for (let k = i + 1; k < tokens.length; k++) {
    const { word, comma } = tokens[k]!;
    // Before the comma check: a predicate is the affirmation, and what follows it is a second clause
    // this read is already done with. "handwriting visible, page otherwise empty" says both things.
    if (PREDICATED.has(word)) return k;
    if (comma) return -1;
    if (affirmsText(word) || MARK_NOUN.test(word) || QUALIFIER.has(word) || DETERMINER.has(word) || QUANTIFIER.has(word))
      continue;
    if (FRAGMENT_CLOSER.has(word) && k === tokens.length - 1) return k;
    return -1;
  }
  return tokens.length - 1;
}

// The affirmation, as the words that make it, or null. Returned rather than a boolean so the refusal
// can say what it saw: `blank_vetoed` exists because #190 had to trace four pages back to a word by
// hand, and a contradiction is harder to spot in a log than a doubt word is.
export function contentAffirmed(scope: string): string | null {
  // A statement is what a `.`, `!`, `?`, `;` or a line break ends. Looser than the denial reads
  // above, which have to cross a line break because these logs put one where a comma belongs โ€” here
  // the boundaries only limit how far a subject may reach for its verb, so a boundary the denial
  // scan crosses is one this one is free to stop at.
  // Kept as an array rather than iterated straight off `split`, because one read below asks what was
  // BEHIND the boundary: a `.` that ends a numbered list marker is not a sentence end, and the statement in
  // front of it is what says which it was.
  const statements = scope.split(/[.!?;\n]+/);
  for (let s = 0; s < statements.length; s++) {
    const statement = statements[s]!;
    const tokens = words(statement);
    const reach = affirmingReach(tokens);
    const affirmed = denialAffirmations(tokens);
    for (let i = 0; i < tokens.length; i++) {
      const { word } = tokens[i]!;
      if (word === "there" && i + 1 < tokens.length && EXISTENTIAL.has(tokens[i + 1]!.word)) {
        const noun = affirmedObjectAfter(tokens, i + 1);
        if (noun >= 0) return tokens.slice(i, noun + 1).map((t) => t.word).join(" ");
        continue;
      }
      if (TRANSITIVE_AFFIRM.has(word) && !negatedBefore(tokens, i)) {
        const noun = affirmedObjectAfter(tokens, i);
        if (noun >= 0) return tokens.slice(i, noun + 1).map((t) => t.word).join(" ");
        continue;
      }
      // Both parts of speech, which is #431: this is the read that delivers "Page is blank. cursive is
      // visible." empty on base, and the one the other three exist to keep honest. `QUALIFIER` holds
      // `typed` as well as `printed`, so the copula guard below covers the new vocabulary's one
      // overlap with it without an entry of its own.
      if (!affirmsText(word) || negatedInList(tokens, i, reach) || participleAfterCopula(tokens, i))
        continue;
      if (LOCATIVE_SUBSTRATE.has(word) && (definiteBefore(tokens, i) || fileNameAt(tokens, i))) continue;
      // `printed page number`, `printed folio` โ€” a name for text dressing the one thing on the paper
      // this pipeline never delivers (`folioAt`).
      if (modifierForm(word) && folioAt(tokens, i + 1)) continue;
      // The contracted spelling of the double negative, read BEFORE the verb walk because the walk has no
      // verb to find here โ€” a contraction is in none of its lists, so `The heading isn't empty.` reached
      // the fragment read with no verb and came out a blank page (#442). Same reading as `is not empty`,
      // the same two words returned as the evidence, and it goes in one direction only: a statement of
      // this shape says the subject HAS something, so the page gets reported rather than delivered empty.
      const contracted = contractedDoubleNegative(tokens, i + 1);
      if (contracted >= 0) {
        return tokens
          .slice(i, Math.min(contracted + 2, tokens.length))
          .map((t) => t.word)
          .join(" ");
      }
      const verb = reach[i + 1]!;
      // No verb for this noun: the statement may still be a fragment that affirms it (#435). Read
      // before the `continue` rather than after the loop, because the guards above โ€” the negator
      // walk, the substrate article, the folio modifier โ€” are the same guards a fragment needs, and a
      // read placed past them would have to repeat all four.
      if (verb < 0) {
        // The statement's TEXT and not its tokens, because both things the guard in there asks about are
        // invisible to the tokenizer: `words()` cannot start a token on `PHRASE_GONE`, and it cannot start
        // one on a digit or a bullet either.
        const named = verblessAffirmation(tokens, i, statements, s);
        if (named >= 0) {
          return tokens
            .slice(i, named + 1)
            .map((t) => t.word)
            .join(" ");
        }
        continue;
      }
      if (!deniedAfterVerb(tokens, verb)) {
        return tokens
          .slice(i, Math.min(verb + 2, tokens.length))
          .map((t) => t.word)
          .join(" ");
      }
      // Denied โ€” but a denial can cover part of the page and say what is on the rest of it, and the
      // affirmation then sits past the complement that denied this clause (#204). The span quoted
      // runs from the subject, so `blank_vetoed` shows the whole shape and not just the half that
      // affirms: "text is absent from the top half and present" reads as the contradiction it is.
      const rest = affirmed.withSubject[Math.min(verb + 2, tokens.length)]!;
      if (rest >= 0) {
        return tokens
          .slice(i, rest + 1)
          .map((t) => t.word)
          .join(" ");
      }
    }
    // The same shapes with the denial written in front of its noun, where the loop above never
    // reaches a subject at all: `No printing except a stamp at the top.` is denied by
    // `negatedInList`, and the stamp it names has no verb of its own for `affirmingReach` to find.
    // Read from the denial rather than from a subject, so the contrast rule is off here โ€” it needs a
    // subject to be a contrast about, and `no printed text, and handwriting is present` is pinned as
    // one denied list rather than as a denial and an affirmation.
    //
    // An absence complement is a denial here too, and that is the second half of #442 rather than a
    // free extra: `Blank apart from a caption.` is named a few hundred lines up as a page this file
    // ships EMPTY because "the exceptive read wants a denial it does not have". It has one now. Six
    // wordings move on that account and every one of them is a page with content going out empty โ€”
    // `blank apart from a stamp`, `except for a signature`, `empty apart from a caption`, `unmarked
    // except for handwriting`, and the two bare fragments. No exception a blank page names moves with them:
    // dust, specks, a printed page number, a folio and a watermark all stay declared, because the
    // object walk was already the thing that decides what an exception is made OF (#439's folio rule,
    // #193's marks). That is a claim about the OBJECTS only โ€” what the widening did to sentences carrying
    // both a complement and a negator is the paragraph below, and it was a regression, not a win.
    //
    // EVERY denial position, and not the first one, which is what the complement made necessary. A negator
    // stands where the denial begins, so the first hit was always the right anchor while the vocabulary was
    // negators alone โ€” but a complement stands BEHIND its own denial's negator: in `The page is empty,
    // nothing on it except handwriting.` a first-hit scan anchors on `empty`, and `denialAffirmations`'
    // backward walk is stopped at the negator standing between that anchor and the object, so `plain` is
    // -1 there and the handwriting was never reached. Eight wordings of that shape shipped a page with
    // content on it as blank (round 1 of #446). Trying each position in order only ever ADDS an
    // affirmation โ€” the position base used is still among them โ€” so it moves a page toward being reported
    // and never toward being lost.
    for (let k = 0; k < tokens.length; k++) {
      const word = tokens[k]!.word;
      if (!(NEGATOR.has(word) || word === "never" || NEGATIVE_COMPLEMENT.has(word) || absenceComplement(tokens, k)))
        continue;
      const named = affirmed.plain[k + 1]!;
      if (named >= 0) {
        return tokens
          .slice(k, named + 1)
          .map((t) => t.word)
          .join(" ");
      }
    }
  }
  return null;
}

function matches(re: RegExp, text: string): string[] {
  return [...text.matchAll(new RegExp(re.source, "gi"))].map((m) => m[0].toLowerCase());
}

export interface BlankDeclaration {
  // The log says the page has nothing on it, in some words.
  asserted: boolean;
  // ...and nothing in it casts doubt on that, so the page is delivered empty.
  blank: boolean;
  // The doubt words that refused an assertion, for the log line. Without this the only record of
  // a refusal was the reply itself, and working out WHICH word did it is a regex read per page โ€”
  // which is what issue #190 had to do by hand for four pages.
  vetoes: string[];
  // The self-contradiction that refused it, where there was one (#194): the log asserted the page is
  // empty and then said something is on it. Kept apart from `vetoes` because it is a different
  // finding with a different remedy โ€” a doubt word means the page could not be read and wants a
  // better image, an affirmation means the agent answered with no page for a page it says has
  // content on it, and `agents/page.md` already tells it to write `[not legible]` instead.
  //
  // Present whether or not it refused: where the reply STATED the page blank it no longer decides
  // the page (`stated` below), and it is still the finding it was โ€” the caller carries it to the
  // verifier and onto the `page_blank` line.
  affirmed?: string;
  // The reply answered the question in the FIELD (`"blank": true`) and not only in prose (#371).
  // Absent rather than false where it did not, so a prose-only declaration is byte-identical to what
  // this returned before the field existed.
  stated?: true;
}

// Did the reply state blankness in the field? `true` and nothing else.
//
// The field can ASSERT blankness and cannot deny it: a `"blank": false` beside an empty `html` and a
// log saying the page is empty is a reply disagreeing with itself, and honouring the `false` there
// would report the page as one nobody transcribed โ€” the exact outcome this field exists to stop, and
// a new way to lose a page for a reply's own inconsistency. So a `false` is read as an absent one and
// the prose decides, as it did before (#371). Every error this field can make is therefore in one
// direction: it can rescue a page the prose reading would have thrown away, and it can never throw
// away a page the prose reading would have delivered.
//
// The string form is accepted because this is a token in an envelope a model writes by hand, where
// the shapes an answer arrives in are not the shapes the prompt asked for: 33 of 78 blank replies in
// the bench logs spelled the empty `html` this prompt asks for as markup instead (#219). `"true"` is
// well inside that. A "yes" is not, because it is not a boolean any JSON writer would produce.
function statedBlank(field: unknown): boolean {
  return field === true || (typeof field === "string" && field.trim().toLowerCase() === "true");
}

// Exported for the unit test: this predicate is the whole distinction between a page delivered
// empty and a page reported lost, and it is worth pinning on the reply shapes directly.
export function blankDeclaration(parsed: { html?: string; log?: string; blank?: unknown } | null): BlankDeclaration {
  const none = { asserted: false, blank: false, vetoes: [] };
  if (typeof parsed?.html !== "string" || carriesContent(parsed.html)) return none;
  const stated = statedBlank(parsed.blank);
  const log = parsed.log?.trim();
  // The field is an answer on its own. Until it existed, the log was the only place blankness could be
  // asserted, so a reply whose log said it in words `BLANK_LOG` does not list ("Nothing on the sheet.")
  // was a page nobody transcribed โ€” and which words those are is a fact about a regex, not about the
  // page. A stated declaration with no log at all is still a declaration, for the same reason.
  if (!stated && (!log || !BLANK_LOG.test(log))) return none;
  const scope = log ? vetoScope(log) : "";
  const vetoes = [...new Set([...matches(UNREADABLE_LOG, scope), ...matches(DEGRADED_IMAGE_LOG, scope)])];
  const affirmed = contentAffirmed(scope);
  return {
    asserted: true,
    // The doubt words refuse a STATED declaration exactly as they refuse a prose one, and #371 asks
    // for that on purpose: a page too dark to read is not a blank page, and a model stating
    // `"blank": true` on one is the failure the veto was written for. What a stated declaration
    // survives is the SELF-CONTRADICTION โ€” the log naming something on the page โ€” because there the
    // reply has answered the question twice and the field is the answer to the question that was
    // asked, while the prose is a sentence a regex has to interpret. It interprets badly enough to
    // matter: #371 measured four of ten ordinary ways of writing "nothing is here" losing the page,
    // one of them to the word `document` standing between a `No` and the noun it denies. The
    // contradiction is not discarded โ€” it is carried out of here and costs the page a paid second
    // look instead of costing it the page (see `blankSkip` in extractPage).
    blank: vetoes.length === 0 && (affirmed === null || stated),
    vetoes,
    ...(affirmed === null ? {} : { affirmed }),
    ...(stated ? { stated: true as const } : {}),
  };
}

export function declaredBlank(parsed: { html?: string; log?: string; blank?: unknown } | null): boolean {
  return blankDeclaration(parsed).blank;
}

// Load the page agent, preferring a session-built/trained copy (tmp/), then the
// committed agents/page.md, and finally the built-in default. Whatever is loaded
// is also what build-time verification and feedback-driven training operate on.
function loadPageAgent(ctx: PipelineContext): AgentSpec {
  const loaded = loadAgent(PAGE_AGENT, {
    agentsDir: ctx.paths.agentsDir,
    tmpAgentsDir: ctx.paths.tmpAgentsDir(ctx.sessionId),
  });
  if (loaded) return loaded;
  return {
    name: PAGE_AGENT,
    file: "page.md",
    content: DEFAULT_PAGE_PROMPT,
    capabilities: ["vision"],
    sha: null,
    sessionBuilt: false,
  };
}

interface PageRender {
  html: string;
  log: string;
  suggestion?: { name: string; reason: string };
  // What the first pass spent producing this page, which is the correction pass's estimate of
  // its own size (`correctionCeiling`). Optional because it is the provider's number, not this
  // file's: an adapter that reports no usage leaves it undefined and the correction runs uncapped,
  // exactly as it did before. Never inferred from the HTML โ€” a length is not a token count, and a
  // wrong estimate here truncates a correction that would have worked.
  outputTokens?: number;
  // The agent declared this page blank and the fragment is `""` on purpose (`page_blank`). Set
  // rather than inferred from an empty `html`, which is the same string for two different
  // answers: today every other unusable reply throws before it can get here, so the two happen
  // to coincide, and a future path that returns an empty page without declaring one would
  // silently inherit the caller's decision not to verify it (#294).
  blank?: true;
  // The declaration was believed AND its log named something on the page (#371). Only reachable on a
  // declaration stated in the field, because a prose one is still refused by the contradiction. Set
  // rather than re-derived at the caller: the caller has the fragment and not the log the affirmation
  // was found in, and re-running the read on the delivered `""` would find nothing to disagree with.
  blankContradicted?: string;
}

// The output ceiling for a correction call, or undefined for "whatever the deployment allows".
//
// A correction is the one page call whose size is known before it is made: it is handed a page
// and asked to return that page with named problems fixed, so its reply should be about as long
// as the reply it is correcting. The deployment's ceiling is 32,000 tokens, which for this call
// is not a safety limit but a budget for a runaway โ€” and a runaway here is pure loss, because the
// pipeline discards a truncated correction and ships the page it already had (`page_correction_failed`,
// `chars_kept`). Issue #285 is one: 32,000 tokens of output on a page whose first pass spent 6,233,
// $0.48 of output for text nothing read, and a `TruncatedResponseError` advising an operator to
// RAISE `max_tokens` โ€” which would only buy a larger discarded reply.
//
// Every term measured, none assumed. The multiple and the floor over the 111 correction attempts in
// the bench's `runs-extract100-1` (two arms, 100 pages each, sonnet-4-6 and gpt-5.6-luna;
// `page_corrected` and `page_correction_failed` paired with the page calls in the same run logs):
//
//   * `2 x` the first pass, because the ratio of a correction's output to its first pass's is
//     tightly held around 1: median 1.01x, p90 1.33x, max 5.01x over 110 successful corrections.
//     A multiple is the right shape and a *character* ratio is not โ€” chars-before/4 as a predictor
//     has a median of 1.92x and a maximum of 33.12x, which is a cap that either cuts successes or
//     saves nothing.
//   * the 4,000-token FLOOR, because the 5.01x tail is entirely small pages: every success above
//     3x had a first pass under 1,000 output tokens, and at 1,000 or more the worst case is 1.65x.
//     Without the floor, `acir-p001` on the luna arm (314 tokens, then 1,573) is cut. With it, the
//     cap cuts 0 of 110 and the tightest margin is 1.30x (`acir-p075`: 3,929 emitted, cap 5,094).
//   * `growth`, for the case the corpus cannot speak to: a specialist merge can hand the
//     correction a document larger than the render whose tokens are being doubled
//     (`dispatchSpecialist` in extractPage), and a correction has to be able to re-emit what it
//     was given. A document k times longer needs about k times the tokens, so the multiple is
//     scaled by k rather than a second term being added in characters.
//
//     In characters is what this was first written as โ€” `handedBackChars / 4` as a third floor โ€”
//     and it could not do the job, which is worth recording because the shape is tempting. A
//     character count converts to tokens at a rate nothing here knows. Measured over every page
//     reply in both run sets whose HTML and token count can both be recovered (390 of 458:
//     `runs-extract-1`, 7 models on an 11-page document, plus `runs-extract100-1`), HTML came out
//     at a median of 2.31 characters per output token, and on the pages long enough for such a
//     term to bind at all (over 4,000 characters) as few as 1.09. `/ 4` therefore provides a
//     quarter to a half of the tokens the same HTML actually costs โ€” it under-provides for 356 of
//     those 390 replies, and `/ 3` for 279 โ€” and a term that binds only when it is the largest is,
//     by construction, too low exactly where it decides the cap. The divisor that survives the
//     corpus is about 1, which is not a conservative constant but a different rule.
//
//     `growth` needs no constant: it is a ratio of two lengths, and it converts to tokens through
//     THIS page's own first pass, which is the measurement the 2x already rests on. Being >= 1 it
//     can only ever raise a cap, so the counts above still hold, and the corpus says it is almost
//     always 1: of the 184 correction attempts across both run sets, 147 have a first-pass length
//     to compare against, 146 of those were handed back no more than the render itself, and one
//     merge grew it by 8% (`acir-p030`, sonnet arm: 8,287 -> 8,960 characters and a first pass of
//     4,461 tokens, so a cap of 9,647 where the unscaled multiple would have given 8,922).
//
// On the one runaway this bounds the loss at 12,466 tokens rather than 32,000 โ€” $0.187 of output
// instead of $0.480 โ€” for an identical outcome (no specialist ran on that page, so growth is 1),
// and the error it throws now says the cap is this caller's rather than sending someone to the
// deployment's config (providers/bedrock.ts, `truncationRemedy`). What it does NOT claim is that
// the correction would then have succeeded: nothing measured here can say that, and a cap is a
// bound on a failure's cost, not a fix for it.
//
// NOT PORTABLE TO THE FIDELITY CHECK, which is what #365 directive 2 asked for. This shape works
// here because the correction's output IS the page, so a cut tail leaves a usable head and the
// scaling term has a page to scale by. The checker's output is a list of problems and its verdict
// sits at the very end of the reply โ€” in all 19 bench replies of 8,000 output tokens or more the
// envelope begins past 86% of the reply โ€” so the same cap removes the answer and keeps the
// narration. It is also the worse shape there on the two columns the decision turns on, verdicts
// kept and dollars saved โ€” though not on every column, and `verifyAgentOutput` carries the row where
// it wins โ€” for a reason visible from here: a runaway there is not a big page, so scaling by the page
// is tightest where the narration is. That function has the measurement and the prices.
export const CORRECTION_CEILING_MULTIPLE = 2;
export const CORRECTION_CEILING_FLOOR = 4000;
// One object and one number rather than three numbers: `outputTokens`, `chars` and
// `handedBackChars` are three counts of the same page whose transposition would type-check and
// would be silent โ€” swapping the two lengths turns `growth` upside down, and a cap that shrinks
// when the document grows is the failure this exists to prevent.
//
// WHICH of the two terms produced the number is returned with it, because it is the question a
// truncated correction's log line is read to answer and the line could not answer it (#365). The
// two have different remedies and only one of them is `CORRECTION_CEILING_MULTIPLE`: of the three
// corrections that truncated in `runs-extract100-95ca64c`, two were capped at twice their own
// first pass (7,904 on 3,952 tokens; 13,278 on 6,639) and the third at the FLOOR โ€” `acir-p083`,
// a 1,618-token first pass whose doubling is 3,236 โ€” so a reading of those three lines as
// evidence about the multiple is a reading of two lines and one that says nothing about it.
// Recovering that from the logs today means pairing the line with the page's own `model_call`,
// which carries no image and is joined by position (#365's own instrument had that join wrong
// once, on every extract cell of two arms).
//
// Returned rather than re-derived at the caller from `ceiling === CORRECTION_CEILING_FLOOR`,
// which is the approximation available there and is wrong in the one case that matters: a first
// pass of exactly 2,000 tokens with `growth` 1 computes 4,000 from the MULTIPLE, and raising the
// multiple would raise its cap. `>=` is the boundary and it belongs where the two terms are.
export function correctionCeiling(
  firstPass: { outputTokens?: number; chars: number },
  handedBackChars: number,
): { tokens: number; bound: "multiple" | "floor" } | undefined {
  // No number, no cap. The alternative โ€” a ceiling derived from the character count alone โ€” is a
  // guess about a provider's tokenizer standing in for a measurement, on the path where getting it
  // wrong throws away a correction that was about to work.
  if (!firstPass.outputTokens || firstPass.outputTokens <= 0) return undefined;
  // Never below 1: a correction handed LESS than the render produced is the ordinary case (the
  // fragment is unwrapped and trimmed), and reading that as "this page needs fewer tokens than it
  // took" would tighten every cap on the corpus the multiple was measured against.
  const growth = firstPass.chars > 0 ? Math.max(1, handedBackChars / firstPass.chars) : 1;
  const scaled = Math.ceil(CORRECTION_CEILING_MULTIPLE * growth * firstPass.outputTokens);
  return scaled >= CORRECTION_CEILING_FLOOR
    ? { tokens: scaled, bound: "multiple" }
    : { tokens: CORRECTION_CEILING_FLOOR, bound: "floor" };
}

// Everything the page agent is told that is NOT about the page in front of it: its own
// prompt, the accessibility contract, and whatever this deployment has learned from
// past corrections. One function, used by every page-agent call, so all of them send
// one byte-identical prefix โ€” which is the condition a cache breakpoint needs to hit
// (providers/promptCache.ts). On a 25-page document that is one cache write and two
// dozen reads at a tenth of the price, on the largest single line in the bill: `page`
// was 779,855 input tokens of the 1.48M measured in issue #136.
//
// The requirements moved here from the user message, where they were re-sent per page
// after the "convert this page" line. That is where the accessibility.ts comment says
// they belong anyway ("appended to each content-agent system prompt", which is how
// `runSpecialist` has always used them) โ€” and being instructions that hold for every
// page, they were never page-specific text. Same for the lessons. What stays in the
// user message is what actually differs per call: the filename and page number, that
// page's link targets, the user's feedback, and the page's previous output.
function pageSystem(agent: AgentSpec, lessons: string): string {
  return `${agent.content}\n\n${ACCESSIBILITY_REQUIREMENTS}${lessons}`;
}

// Three repairs to a page reply, on the line where that reply becomes markup Iris keeps: soft
// hyphens out of it (#334), `style` attributes out of it, and a thousands separator the printing
// split with an alignment space closed back up (#374 item 7). `html.ts` has the argument for each โ€”
// what makes it a repair rather than a prompt clause, and what each one's limits are โ€” and this is
// about WHERE they go, which is the part with alternatives and the part all three share.
//
// What they have in common is the test that admits them: each is decidable on the reply alone, with
// no image, no second model and no page-specific knowledge, and on each there is no page where the
// thing being removed is the right answer. That is why they are here and not in the problem list the
// correction pass is given. A split WORD (`hyphens.ts`) fails that test โ€” Iris cannot know which
// spelling the paper prints โ€” so it stays a re-ask, and the boundary between the two files is that
// sentence rather than which issue asked for the code.
//
// They run soft hyphens, then styling, then digit groups, and the last two are in that order for a
// reason: `tightenDigitGroups` only touches a cell holding nothing but a figure, so a `<td>` whose
// figure sits beside an empty styled `<span>` is out of its scope until the strip has removed the
// span. Reversed, that cell keeps its printer's space and nothing says why.
//
// At every seam in this phase rather than once at its exit, because a page's own output is an
// INPUT further along: `renderPage`'s fragment is what `correctPage` is shown as "your previous
// output", what a re-extraction is shown as "your previous output for this page", and what
// `mergeSpecialist` is shown as the current page. Repair only on the way out of extraction and the
// model is handed its own soft hyphen โ€” or its own `style` attribute โ€” back and told to carry over
// everything the problem list does not name exactly as it stands, which is the instruction working
// correctly on markup that should not have reached it. Repairing where the reply is read means the
// defect never enters the pipeline's state at all, so it cannot be re-derived, quoted, or copied
// forward. Whether being handed one back is what makes a page repeat it is not measured โ€” #374's 52
// style attributes include 40 `padding-left` on a single page's row headings, which says the defect
// comes in runs, not what a second pass shown the run would do with it.
//
// AFTER `ctx.log.agentCall` at every site, and that order is load-bearing rather than incidental.
// `agent_call.output` is the raw reply and the only record of what the model actually wrote:
// #334's census โ€” 9 pages, 63 occurrences, one arm of three, $0 โ€” was a regrade of round logs
// already on disk, and a strip that ran before the log would have left no way to take that
// measurement or any future one. The repair is Iris's; the reply on record stays the model's.
//
// This phase and not the review phase, which is the scope worth stating because there are four
// more seams where a model's HTML becomes markup: the Copy Editor's two contracts and
// `edit_section` (review.ts), the table-join agent (tables.ts), and the regression fixture re-run
// (feedback.ts). Two of the three repairs answer that with the same argument. A soft hyphen and a
// printer's alignment space are TRANSCRIPTION artefacts โ€” they come from an agent reading an image
// of a printing that broke a word across a column or spaced a figure to fit one, and re-typing it โ€”
// and every seam here is such an agent, while the review-phase agents are handed markup this phase
// has already cleaned and edit it as text. Their only route to one is invention, which nothing has
// measured. The fixture path is out for a different reason: its HTML is scored against a stored
// fixture and never delivered, so a repair there would move a comparison rather than fix a document.
//
// The STYLE strip does not get that argument, and this is its stated limit rather than a claim about
// it. A `style` attribute is not an artefact of reading a printing; it is a model reaching for CSS to
// hold a shape, which any agent writing markup can do โ€” and the review-phase agents write markup.
// #374 measured styling on the page agent's output because that is what #374 looked at, so what is
// known is that it happens here and not that it happens nowhere else. Covering those seams needs
// their own census first: a strip on the Copy Editor's output would be moving a fix onto a step where
// the rate is unmeasured, and `page_style_attributes` firing with `where` on it is what would say so.
//
// Of those four, the TABLE JOIN is the one to cover first should `page_soft_hyphens` ever fire after
// a model swap. It is the closest thing outside this phase to a transcription step โ€” its job is
// re-typing the cells of a table the printing broke across a boundary โ€” so while its input is clean
// and invention is still the only route, it is the shortest route of the four.
//
// `correct` is the last of the four to run, not `specialist_merge`: dispatch happens inside the
// render, and the correction pass runs after the fidelity verdict.
type RepairSeam = "extract" | "correct" | "specialist" | "specialist_merge";

function repaired(
  ctx: PipelineContext,
  where: RepairSeam,
  img: InputImage,
  html: string,
  // The reply this ran on was a second draw of the page, and its counts are counts of markup Iris
  // DISCARDED (#365 directive 5, `page_redrawn`). Without it two `extract` lines for one page read as
  // one page's markup counted twice: `where` makes a count attributable to the call it was billed
  // under, and a redraw makes two calls at the same seam. An offline census like #334's is what this
  // is for โ€” nothing in `src/` reads these lines โ€” so it discounts rather than filters.
  redrawn = false,
): string {
  const at = { image: img.name, page: img.order, where, ...(redrawn ? { redrawn: true } : {}) };
  const { html: noShy, removed } = stripSoftHyphens(html);
  // Only when it fired, so the line means "this page had them" and a run with none of these lines
  // is a run where no reply carried one. `where` is the step name the call was billed under, which
  // is what makes the count attributable: the same character from a first render, from the
  // correction pass and from a specialist are three facts about three different calls, and #334's
  // census is per-agent. The same holds for the two lines below.
  if (removed) {
    ctx.log.event("page_soft_hyphens", { ...at, removed });
  }
  const { html: noStyle, stripped, spans, cellsEmptied, props } = stripStyleAttributes(noShy);
  // `props` is the field to read, and the reason this line carries numbers and a list. The count says
  // a page had styling; the properties say what the styling was DOING, which is the part that
  // survives the strip as a question: `padding-left` names a page whose row-group hierarchy was in
  // ink, `background-color` a legend swatch that painted nothing. Neither can be rebuilt from here โ€”
  // that is a re-ask against the image โ€” so the log is what makes those pages findable.
  //
  // `cells_emptied` is the same argument at its sharpest. A swatch inside a table cell leaves that
  // cell holding nothing, and this file's own prompt asks a page never to deliver one, because an
  // empty cell claims the paper was blank there. The strip did not put the defect in the reply, and
  // a reader who could not see the colour was already getting an empty cell โ€” but a page with this
  // number above zero is a page where the mark is now unrecoverable without the image, so it is the
  // one number here that names work rather than housekeeping.
  if (stripped) {
    ctx.log.event("page_style_attributes", { ...at, stripped, spans, cells_emptied: cellsEmptied, props });
  }
  const { html: clean, tightened } = tightenDigitGroups(noStyle);
  if (tightened) {
    ctx.log.event("page_digit_groups", { ...at, tightened });
  }
  return clean;
}

async function renderPage(
  ctx: PipelineContext,
  agent: AgentSpec,
  img: InputImage,
  lessons: string,
  // On a feedback re-extraction, the HTML this page produced last time. Shown so
  // the agent corrects what the feedback names and carries everything else over,
  // rather than re-deriving the page from scratch and drifting elsewhere.
  previous?: string,
  // This call IS the second draw, so it does not get another (#365 directive 5, and the
  // paragraph at `page_redrawn` has the measurement). One extra call and never two: the
  // failure a redraw is for is a draw the model can lose, and a page that loses two draws
  // in a row is not that page. Private to this function โ€” `extractPage` is the only caller
  // and passes nothing, so a first pass is always a first draw.
  redrawn = false,
): Promise<PageRender> {
  const priorSection = previous
    ? `\n\n## Your previous output for this page\n\`\`\`html\n${previous}\n\`\`\`\n` +
      `Apply the user feedback above to this page. Keep everything the feedback does NOT ` +
      `concern exactly as it was, and re-check the affected content against the source image.\n`
    : "";
  // The page's own link targets, which the image cannot show (pipeline/links.ts). `redrawn` for the
  // same reason the repair lines carry it: this is the same page's links offered a second time, and a
  // count of pages whose links were shown would otherwise count this page twice.
  const links = pageLinkContext(img.links);
  if (links.shown.length) {
    ctx.log.event("page_links", {
      image: img.name,
      links: links.shown.length,
      dropped: links.dropped,
      ...(redrawn ? { redrawn: true } : {}),
    });
  }
  const user =
    `Convert this document page image (filename: ${img.name}, page ${img.order} of ${ctx.images.length}) ` +
    `to accessible HTML.${links.section}${feedbackPreamble(ctx)}${priorSection}`;
  const res = await ctx.router.complete(
    PAGE_AGENT,
    "vision",
    [
      { role: "system", content: pageSystem(agent, lessons) },
      { role: "user", content: user },
    ],
    { step: "extract", images: [loadImage(img)] },
  );
  ctx.log.agentCall({ agent, phase: "extraction", image: img.name, output: res.text });
  // `blank` is `unknown` rather than `boolean` because nothing validates this envelope: it is a model's
  // JSON, and a field typed here is a claim about a reply Iris did not write. `statedBlank` decides
  // what counts as an answer.
  const parsed = extractJson<{
    html?: string;
    log?: string;
    blank?: unknown;
    suggested_agent?: { name?: string; reason?: string };
  }>(res.text);
  const raw = parsed?.html ?? bareHtml(res.text);
  // Ahead of every check below, so the emptiness test reads the markup a reader actually gets: a
  // fragment whose only text is soft hyphens carries nothing, and should be treated as the page with
  // nothing on it that it is rather than as characters. `blankDeclaration` re-derives that same
  // reading from the reply, so it is handed this fragment too โ€” see the call.
  const html = raw == null ? raw : repaired(ctx, "extract", img, raw, redrawn);
  // Nothing for a reader in this reply. Throwing hands the page to `failedPage`, which is what
  // every other unusable answer in this file already does: the page is lost, and the run
  // SAYS the page is lost (`page_extraction_failed`, `pages_failed`, a @page-failed
  // comment where the content would have been). The alternative โ€” the old fallback's โ€”
  // was to deliver whatever the model wrote as the page, which reports 100 of 100 pages
  // delivered while one of them is a JSON envelope with prose around it. On a feedback
  // re-extraction the throw is cheaper still: `previous` is kept, so the page keeps the
  // content it already had.
  //
  // `carriesContent` rather than a length, because a fragment can be several dozen characters of
  // markup and still hand a reader nothing: a comment, an empty paragraph, a bare page-break marker.
  // 33 of 818 initial renders in the bench logs are exactly that, all of them the model's way of
  // saying the page is blank, and read as content they were counted as pages that produced markup and
  // delivered into the document (issue #219). Everything below then treats them as what they are: a
  // declaration if the log makes one, and the same failure an empty `html` is if it does not.
  if (!html?.trim() || !carriesContent(html)) {
    // Unless the agent said the page is blank, in which case an empty page is the answer and
    // not the absence of one (see `declaredBlank`). Reported on its own event and counted apart
    // from failure, because the two need different things from whoever reads the run: a failed
    // page is work to redo, a blank page is nothing to do. The empty fragment is dropped at
    // assembly, which for a page with nothing on it is what the document should say.
    //
    // This still returns like any other page, and everything that runs on a page still runs on
    // this one โ€” what no longer runs is the paid one. `extractPage` reads `blank` and does not buy
    // the Feedback Agent's verdict on an empty fragment (#294); the free checks are unchanged, and
    // the paragraph at that call site is where the reasoning and the numbers are. Short-circuiting
    // HERE, by returning early out of the page's own function, is a different thing and is still
    // refused: it would take the link and alt checks with it, and those are the two detectors of a
    // wrong blank declaration that cost nothing.
    //
    // And not on a page Iris already has content for. `previous` is set only on a feedback
    // re-extraction, and only for a page whose fragment is real content (`previousFor` in
    // reExtractPages withholds it from a failed page, so a lost page CAN come back blank and be
    // recovered). Where it is set, the model has been shown that content and is now saying the
    // paper is empty, which contradicts evidence this pipeline already holds โ€” so the throw is
    // taken after all, and the containment on that path keeps the prior fragment rather than
    // deleting it (`page_extraction_failed` with `kept: "prior"`). Nothing else on the
    // re-extraction path would catch it: `destroyedPage` guards the correction pass, where
    // `before` is this round's own render, so prior โ†’ empty never reaches a size comparison.
    // The STRIPPED markup, so this predicate and the gate above read the same fragment. It
    // re-derives "does this carry content" from the reply itself, and handed the raw one it would
    // answer yes for a fragment whose only text is soft hyphens while the gate answered no โ€” which is
    // this branch entered and then refusing the declaration inside it, so a page the reply said was
    // empty is reported as a page that FAILED. That is the one distinction this whole passage exists
    // to keep: a failed page is work to redo, a blank page is nothing to do. Nothing else about the
    // reply is substituted, and stripping cannot turn a page with content into one without.
    const declaration = blankDeclaration(parsed && { ...parsed, html: html ?? undefined });
    // What the model sent instead of the page, where it sent anything: the markup a declaration was
    // spelled in, bounded, on both lines that discard it. The RAW markup, deliberately, and the one
    // place in this function that is: the field exists to show which shape the declaration arrived
    // in, and a repair Iris made on the way past is not part of that shape. Without it a run says a
    // page was declared blank and not whether the reply was the empty `html` the prompt asks for or a
    // marker naming a folio the paper never printed โ€” which is the difference #219 had to reconstruct
    // by replaying 818 replies, and the field that would have shown it in one grep.
    const dropped = raw?.trim() ? { dropped: raw.trim().slice(0, 200) } : {};
    // On all three lines a declaration can land on, because the field's own failure mode is invisible
    // from the one that honoured it. `blank_stated` here says the field was SENT, not that it was
    // believed: a run reading only `page_blank` can count how often the answer arrived in the shape
    // the prompt asks for and cannot count how often it arrived for a page the model had just said it
    // could not read, or for a page Iris already holds content for. Those two are the misuse the
    // prompt paragraph warns about in as many words ("never send it for a page you could not read"),
    // and they are the one signal that would say this change had gone wrong โ€” a model stamping
    // `"blank": true` on unreadable pages costs nothing today (the doubt veto still refuses it) and
    // would be the reason to take the field back out. Neither line can show it unless it is on them.
    const stated = declaration.stated ? { blank_stated: true as const } : {};
    if (declaration.blank && previous?.trim()) {
      ctx.log.event("page_blank_refused", {
        image: img.name,
        page: img.order,
        chars_kept: previous.trim().length,
        log: parsed?.log ?? "",
        ...stated,
        ...dropped,
      });
      throw new Error(
        `page agent declared a blank page that already had ${previous.trim().length} chars of content`,
      );
    }
    if (declaration.blank) {
      // `blank_stated` says which of the two asked-for answers this was, because they no longer behave
      // alike: a stated one survives a self-contradiction and a prose one does not, so a run that
      // wants to know how often the field is actually sent โ€” the thing the whole of #371 turns on โ€”
      // has to be able to count it. On this line it means the field was believed; the same field on
      // the two refusals above and below means it was sent and refused, which is the paragraph at
      // `stated`. `blank_contradicted` is the same field name `page_no_output`
      // carries for the refusal, so one grep finds both the pages this cost and the pages it did not.
      ctx.log.event("page_blank", {
        image: img.name,
        page: img.order,
        log: parsed?.log ?? "",
        ...stated,
        ...(declaration.affirmed ? { blank_contradicted: declaration.affirmed } : {}),
        ...dropped,
      });
      // `""` whatever the reply held, so one shape of fragment stands for a blank page however the
      // model wrote it. What that discards is a page-break marker on 18 of the 33 markup-spelled
      // declarations โ€” and every one of those logs says the paper prints no number, which makes the
      // marker's label the image's position in the file and its anchor a claim that the document's
      // page 14 begins here. The prompt forbids exactly that and the paragraph at `blankDeclaration`
      // has the reasoning; a blank page that DID print its folio loses an anchor to nothing, which is
      // the cheaper mistake. The markup is on the log line above either way.
      return {
        html: "",
        log: parsed?.log ?? "",
        outputTokens: res.usage?.output_tokens,
        blank: true,
        ...(declaration.affirmed ? { blankContradicted: declaration.affirmed } : {}),
      };
    }
    const shape = replyShape(res.text, parsed);
    // One more draw of the same page, for a reply that claimed nothing about it (#365 directive 5).
    //
    // The gate is `asserted` and not a length, and the difference is the whole change. Directive 5
    // asks for a re-extraction "when the reply is under some floor of HTML", and a floor cannot
    // separate the cases: it reads what the PARSE produced, and a reply Iris refused whole is 0
    // characters of HTML however much page it was carrying. Over every bench run log on disk โ€” 2,639
    // files in the 80 round directories โ€” 20 replies reach this branch, and they land on 20 DISTINCT
    // round-and-page pairs, so 1.05% of the 1,913 pages drawn at least once is a share of pages rather
    // than an average over repeats (those pairs carry 2.2 page-agent calls each, so it had to be
    // counted, not assumed). Per individual draw the rate is AT LEAST 0.48%. The 20 are `page_no_output`
    // events, and they have to be: nothing on disk logs `page_redrawn`, because every round predates
    // this branch.
    //
    // A repo-wide `find` counts 2,657 `*.jsonl`, and the 18 left out here are two different things: 11
    // corpus manifests in the bench root, with no extraction call in them, and 7 `*-dry.jsonl` probe logs
    // under `bench-data/`, which DO carry extraction calls โ€” 12 page-agent calls and 12 checks on 3 pages
    // โ€” and which any walker that descends every top-level directory folds in silently. PAGE-AGENT CALLS,
    // not draws: they are `agent_call`s, and the paragraph below is about exactly why that word cannot be
    // narrowed. Nor could it be here โ€” those 7 files log 0 page `model_call`s at all, so none of them is
    // among the 60 that carry `step`, which is why 60 / 954 / 604 / 1,558 are the only figures the
    // exclusion leaves alone. That is where an
    // earlier version of this comment got 4,159 calls and 1,916 pages. Every count below is the round
    // directories alone, and it moves the headline: 20/1,913 is 1.0455%, where 20/1,916 was 1.0438% and
    // rounded to 1.04%. Four digits because three would be 1.045, the half that cannot decide itself.
    // 0.48% and the 0.248% named at the end are unchanged.
    //
    // The per-draw rate is a bound and not a figure, said out loud because two shipped versions of this
    // comment stated it as one. `phase: "extraction"` logs 8,049 `agent_call`s, 4,147 of them naming
    // the page agent and 3,902 the fidelity check on the same pages โ€” but `agent_call` records no
    // `step` (`src/store/runlog.ts`) and THREE sites log under that agent and phase: this draw,
    // `correctPage` below, and `mergeSpecialist`. 4,147 therefore bounds the draws from above rather
    // than counting them. In THIS corpus the third site contributes nothing and the inflation is
    // corrections alone: `4,147 + 3,902` is the whole phase, so no specialist agent ever logged a row
    // here, and `mergeSpecialist` only runs after one returns a fragment. That sum carries the claim by
    // itself: 0 `specialist_merge` `model_call`s is a fact about the 60 files that emit `step`, and says
    // nothing about the other 2,579. `model_call` does carry `step`, and only recent
    // rounds emit it: across those 60 log files, 954 of 1,558 page-agent calls are draws and
    // 604 are corrections, so the rate is nearer 0.8% if that mix holds corpus-wide โ€” and a correction
    // always follows a draw of the same page in the same run (`correctPage`'s only caller is inside
    // `extractPage`), which is what makes "pages drawn at least once" a sound reading of a population
    // that counts corrections. The 1,913 IS exact โ€” distinct round-and-page
    // pairs off page-agent calls alone, against 2,042 for a mixed count, the difference being 129 pairs
    // that carry a checker call and no draw. The 0.255% this comment first shipped was wrong twice:
    // 20/7,843 off a corpus that missed the round directory named `runs`, where the phase-wide rate on
    // the whole corpus is 20/8,049 = 0.248%. Replaying all 20 through today's parser leaves THREE: the
    // other 17 are blank pages whose declaration `blankDeclaration` honours, so they never get here and
    // a floor would have redrawn every one of them. Two of those 17 were REFUSALS until #429, and are
    // why that issue was filed: one vetoed on the word "noise" for a log reading "blank apart from
    // minor scanning artifacts (specks and compression noise)", one refused as self-contradicting for a
    // log that named the IMAGE FILENAME ("image filename indicates this is page 14 of 25"). Both pages
    // ARE blank, and not on one log's word: every PAGE-AGENT reply on disk for those two images
    // declares the page blank โ€” 14 replies on one, 8 on the other from three different models โ€” and
    // none of the 22 carries content. (Page-agent, for the reason the rate above states: count every
    // reply in the phase instead and the checker's verdicts on the same pages inflate both figures.) So
    // the redraw those two used to get bought a second copy of the same sentence at a full page's price.
    //
    // The three that still arrive carry no declaration for `asserted` to read at all โ€” no envelope
    // survives the parse, so there is no `log` โ€” and each of the three wanted the redraw:
    //
    //   - one is a 47-character reply, `<h1><cite role="doc-bibliography"></cite></h1>`, on a page the
    //     same model rendered as 7.6-9.8 KB in three independent redraws and delivered in another
    //     round. That is a draw the model can lose, which is what this exists for.
    //   - one is a complete envelope one `}` short, holding a 3,437-character table of contents that
    //     closes cleanly on `</ol></nav>`; the page delivered fine in 11 other rounds. A redraw
    //     recovers it, and a parse that closed the brace would recover it for nothing โ€” so this one is
    //     a page the redraw is the SECOND-cheapest answer for, and #426 carries the parse.
    //   - one is a reply that transcribed the page, wrote a fenced ``I need to restart and produce a
    //     clean, correct rendering.``, and transcribed it again. `stripFences` returns the last
    //     complete fenced block, which is that sentence, so `bareHtml` refuses 10,755 characters that
    //     are a rejected draft, a self-correction and a good page with no reliable boundary between
    //     them. Refusing it is right โ€” the delivered document's contract is that every word in it is a
    //     word on the page โ€” and the model asked for the redraw in as many words.
    //
    // So `asserted` is correct on all 20 โ€” it honours the 17 and admits these 3 โ€” where a character
    // floor redraws all 20 and is right about 3 of them. Mind the denominators: the 3 is the same three
    // replies both times, and it was 3 of 5 before #429 moved two of the five out of this branch.
    //
    // What it does NOT cover, in two spellings, because the corpus contains neither: a blank page
    // whose declaration `blankDeclaration` cannot see. One is markup-only โ€” `<!-- blank page -->` is
    // #219's own spelling, and with no envelope there is no `blank` or `log` to read. The other is an
    // envelope whose `html` is not a STRING: `asserted` requires `typeof parsed.html === "string"`, so
    // `{"html": null, "log": "This page is blank.", "blank": true}` answers the question and is
    // redrawn anyway. Both cost one call and change no outcome โ€” the second draw declares the page
    // blank the same way and the page is refused as it is today โ€” and neither is a share of the rate:
    // 1 of the 20 replies carried any markup at all (the 47-character one, which is not a
    // declaration), and every one of the 17 honoured declarations sent `html` as a string. Believing a
    // declaration whose `html` is null is a change to the BLANK routing, not to this branch: it would
    // deliver such a page instead of refusing it, which is #219's argument to reopen and not this
    // one's to settle.
    //
    // A provider failure never reaches here โ€” a throttle, a stall and a refusal all throw out of
    // `router.complete` above โ€” which is the line this change deliberately does not cross. The
    // paragraph at `correctPage`'s error containment is why: a correction that truncated because the
    // PAGE is large will truncate again, and a redraw would buy a second full ceiling to prove it.
    //
    // A reply the model itself cut short DOES reach here, as `truncated_envelope`, and it is redrawn
    // with that argument read the other way round. What makes a correction's truncation not worth a
    // second call is that the page SURVIVES it: the pre-correction render is kept and the document
    // still has the page. A first render that truncated leaves nothing, so the choice is not "one call
    // or two" but "one more call or a hole in the document" โ€” and the one instance on disk is not a
    // ceiling at all but an envelope one `}` short of a 3,437-character page. A page that genuinely
    // exceeds the ceiling loses this draw too, which is the ceiling's own remedy (#365 directive 3
    // declined to raise it) and costs the second call to find out.
    if (!declaration.asserted && !redrawn) {
      // The losing draw, on its own line rather than on `page_no_output`. Two reasons, and the
      // second is the one that matters: a page this recovers is a page NOTHING would otherwise
      // record, and `page_no_output` keeps meaning "this page was given up on" โ€” so every count
      // taken off that line, in diagnostics and in the tests, still counts pages lost and not draws
      // discarded. `shape` and `dropped` are the same fields the give-up line carries, because the
      // triage question does not change: the draw that lost is the one worth reading.
      ctx.log.event("page_redrawn", {
        image: img.name,
        page: img.order,
        chars: res.text.length,
        shape,
        ...dropped,
        ...(previous ? { reextract: true } : {}),
      });
      return renderPage(ctx, agent, img, lessons, previous, true);
    }
    // A declaration the veto refused is recorded as the refusal it is, with the words that did it:
    // the failure line alone reads as "the model answered with no page", which is the opposite of
    // what happened, and every page issue #190 recovered had to be traced back to a word by hand.
    //
    // `dropped` belongs here as much as on the two lines above, and for the same reason one line
    // further on: this is where a refused declaration lands, so this is the line that has to be
    // triaged, and `chars` counts the whole reply rather than the fragment. Without it the wording that
    // walked into the veto is on record and the markup it was written beside is not โ€” the half of
    // #219's reconstruction the fix left behind (#223).
    ctx.log.event("page_no_output", {
      image: img.name,
      page: img.order,
      chars: res.text.length,
      shape,
      ...dropped,
      ...(declaration.asserted ? { blank_vetoed: declaration.vetoes, log: parsed?.log ?? "" } : {}),
      // The misuse case, and the reason `blank_stated` is not confined to the line that honoured the
      // field: a stated blank refused HERE is a model that said the page was unreadable and stamped
      // the field on it anyway. `blank_vetoed` names the words and this names where the answer came
      // from, so the two together separate a prompt that is being followed badly from one that is not
      // being followed at all.
      ...stated,
      ...(declaration.affirmed ? { blank_contradicted: declaration.affirmed } : {}),
    });
    // The message says what arrived, which for a fragment carrying nothing is not "no HTML": a reply
    // that sent a comment or a bare marker sent HTML, and it is `page_extraction_failed.error` that
    // carries this string to whoever reads the run.
    const arrived = html?.trim()
      ? `no page in ${html.trim().length} chars of HTML`
      : `no HTML (${shape}, ${res.text.length} chars)`;
    throw new Error(`page agent returned ${arrived}`);
  }
  // A page rescued by `bareHtml`: the reply was markup rather than the envelope, so it delivers
  // a real page and nothing else. `log` is `""` and `suggested_agent` is absent โ€” not because
  // the model had nothing to say, but because there was no field to say it in.
  //
  // Said out loud because until now it was the one page outcome with no line of its own. A blank
  // page has `page_blank`, an unreadable reply has `page_no_output`, a lost page has
  // `page_extraction_failed`; this one shipped as an ordinary success. It is not rare: 41 of 300
  // first page calls across two multi-vendor bench rounds and 54 of 400 across four deployed
  // rounds, and 0 of those 41 left any other line behind (#349). `agents/page.md` discharges
  // obligations in the log that it discharges nowhere else โ€” a page ending mid-sentence, a heading
  // with no parent, a symbol with no key, a placeholder image source, a language change, an
  // irregular table โ€” so on those pages every one of them is unmet and unreported, while the HTML
  // is perfectly usable and the run says 100 of 100 delivered.
  //
  // It is the discriminator for the other reading of an empty log, too. `log: ""` on a page with
  // this line means the reply had no envelope; `log: ""` without it means the model sent an
  // envelope and left the field empty, which is a prompt-compliance question and not a parse one.
  // Nothing could tell those apart before, and they have opposite remedies. Which of the two the
  // deployed models do is now answerable: over 67 round logs on file, 2,320 `page.md` replies are
  // 2,001 with a non-empty log and 319 bare, and 0 with an envelope whose log is empty โ€” so this
  // line accounts for every page that has no log, and the second reading is a shape nothing has
  // produced yet rather than a share of the 13.7%.
  //
  // `reextract` marks the feedback round rather than the first pass, because #349's rate is over
  // first calls and a count that pools rounds is not comparable with it. Absent on a first pass,
  // so no older line changes shape.
  if (!parsed) {
    ctx.log.event("page_bare_html", {
      image: img.name,
      page: img.order,
      chars: res.text.length,
      html_chars: html.trim().length,
      ...(previous ? { reextract: true } : {}),
    });
  }
  const sa = parsed?.suggested_agent;
  return {
    html,
    log: parsed?.log ?? "",
    suggestion: sa?.name ? { name: sa.name, reason: sa.reason ?? "" } : undefined,
    // Read off the result rather than through the router's `onUsage`, because this is the
    // surviving path: a render that threw has no page to correct.
    outputTokens: res.usage?.output_tokens,
  };
}

// One problem the corrector says it is not acting on, and what the HTML shows instead (#373
// directive 4). `problem` is the number it was listed under in the request, so the decline can be
// joined back to the problem itself and to which of the five sources raised it โ€” the join is by
// index rather than by matching the sentence back, because a model quoting a problem back
// paraphrases it and a paraphrase cannot be matched to the entry it came from.
//
// It is absent where the reply declined something and named no number. That decline is still
// recorded rather than dropped: it is the model saying a problem is false, which is the event this
// field exists to count, and losing it would report a run as compliant because the disagreement was
// badly addressed. What it cannot do is say WHICH problem, so it is not counted against any.
export interface Declination {
  problem?: number;
  why: string;
}

// What separates the problems Iris CHECKED from the one it was told, where the corrector reads them
// (#373 directive 4). The list it is shown is five sources concatenated with nothing to say which is
// which, and the licence to decline is written around claims about the markup โ€” whose wordings are
// exactly the code-checked ones ("More than one element on this page has id=โ€ฆ"). Without this mark a
// corrector following the licence as written declines into `links`, `alt` and `ids`, and
// `verification.declined.code_checked`, the field that exists to count the misuse, counts compliance
// instead. With it the number means one thing.
//
// A suffix on the entry rather than a section heading over each group, because the groups are
// numbered as one sequence for the decline to cite and a heading between them invites the model to
// renumber from 1 inside each. 33 characters, on the problems this run raised in code โ€” 2 of 1,501
// page replies for the id rule, none from the deployed model.
export const CHECKED_IN_CODE = " (Iris checked this one in code.)";

// The split-word problem's mark (#334 part B), and it is a DIFFERENT STRING from the one above
// because the two marks assert different amounts. `CHECKED_IN_CODE` means the problem is settled, and
// the licence paragraph in `correctPage` spends its last sentence saying so โ€” "so fix it". That
// sentence is true of a missing link, a placeholder alt and a duplicated id, each of which is wrong
// by construction. It is NOT true here, and the first draft of this change gave the split-word
// problem the same mark: the request then carried "so fix it" and the problem's own closing sentence
// ("If the page really does print both spellings, say so and change nothing") in one message, about
// one numbered entry, and the general sentence is the one written as a rule.
//
// What that would have cost is not an accounting error, it is a transcription defect in delivered
// content: `non-farm` beside `nonfarm` and `co-operation` beside `cooperation` are forms a 1962 report
// really prints, three of the shipped model's six census cases are of that kind, and the likeliest
// reading of "so fix it" on such a page is to join the two โ€” which puts a word the page does not
// print into the document, bought by a rule that fired on a page which had already passed.
//
// So this mark states the split instead of the verdict: what Iris compared in code is that both
// spellings are present, which is not arguable, and NOT which one is right, which is. `correctPage`
// carries one sentence naming this mark, and it is the only place in that request where what the
// image shows is a reason not to act. The separation is finished one field along: a decline citing
// `words` is counted as `declined.words` rather than `declined.code_checked` (see diagnostics.ts),
// so the field that exists to count misuse still counts only misuse.
export const SPELLINGS_CHECKED_IN_CODE =
  " (Iris checked in code that both spellings are on this page, not which one is right.)";

// What a corrector's reply says it declined, out of whatever shape the reply used.
//
// Deliberately loose about the container and strict about nothing else: the field is introduced in
// the request rather than in the page agent's own contract (see `correctPage`), so this is the one
// key of that envelope no prompt file has ever described, and the shapes a model reaches for are an
// array of objects, an array of bare strings, or one object on its own. All three are read. What is
// NOT inferred is a number that is not there โ€” a string entry contributes its text and no `problem`,
// rather than being matched back against the problem list by wording, which is the guess that would
// attribute a decline to the wrong problem and then report it as a code-checked fact being refused.
//
// A leading number IS read out of a string ("2. the id appears once"), because that is the shape a
// model gives when it is answering a numbered list in prose and the number is its own citation, not
// an inference of ours.
//
// The one thing this must not do is mint a decline out of a reply that declined nothing. The request
// says to omit the key, and a model that will not omit a key sends `{}`, `"none"` or `"N/A"` instead
// โ€” so those are the shapes that would otherwise put a `page_correction_declined` line, and a count
// in `declined.unattributed`, against every corrected page in a run. Both counts are read as rates
// against corrections, so a per-page phantom would not skew them, it would replace them.
const NOTHING_DECLINED = /^(none|no|n\/?a|nil|nothing|null|undefined|-|โ€”)[.!]?$/i;

export function parseDeclined(value: unknown): Declination[] {
  const entries = Array.isArray(value) ? value : value == null ? [] : [value];
  const out: Declination[] = [];
  for (const entry of entries) {
    if (typeof entry === "string") {
      const why = entry.trim();
      if (!why || NOTHING_DECLINED.test(why)) continue;
      const cited = /^\s*#?(\d{1,3})\s*[.):\-โ€“]/.exec(entry);
      out.push({ ...(cited ? { problem: Number(cited[1]) } : {}), why });
      continue;
    }
    if (typeof entry !== "object" || entry === null) continue;
    const rec = entry as Record<string, unknown>;
    // `why` is the field the request names; `reason` and `because` are the near-synonyms a model
    // substitutes for it, and a decline whose reason cannot be read is still a decline โ€” it is
    // logged with an empty `why` rather than discarded, since the count is the load-bearing part
    // and a silent drop would report the reply as fully compliant.
    const why = [rec.why, rec.reason, rec.because].find((v) => typeof v === "string" && v.trim());
    // `problem` and `number` only. `index` is not read, though a model does send it: it conventionally
    // means the 0-based position, so `{ index: 1 }` may be the FIRST problem or the second, and there
    // is nothing on the reply to say which. A citation read off by one is worse than no citation,
    // because `source` is computed from it and then read as evidence about which check was refused โ€”
    // a `verify` decline filed under `ids` is the misuse this feature is measured by, manufactured
    // here. Such a reply is logged as a decline with no number, which is what it actually is.
    const raw = rec.problem ?? rec.number;
    const problem = typeof raw === "number" ? raw : typeof raw === "string" ? Number(raw.trim()) : NaN;
    const cited = Number.isInteger(problem);
    // Nothing said, in object form: no number, and a reason that is either absent or one of the
    // non-answers above. Dropped for the same reason โ€” there is no disagreement here to record, only
    // a key the model would not omit โ€” and the object shape is the one the request's own example
    // teaches, so `[{ "why": "none" }]` is likelier than the bare string it started as. A CITED
    // problem is kept whatever its reason says, on the same terms as `{ problem: 2 }` with no reason
    // at all: the number names something the log can attribute, and an empty `why` on the line is how
    // a reader sees that the argument never arrived.
    const said = typeof why === "string" && why.trim() && !NOTHING_DECLINED.test(why.trim());
    if (!cited && !said) continue;
    out.push({
      ...(cited ? { problem } : {}),
      why: typeof why === "string" ? why.trim() : "",
    });
  }
  return out;
}

// Which of the five sources raised the problem a decline names, by its position in the list the
// request numbered. The order here is the order `problems` is built in at the call site and the two
// cannot be allowed to drift, so this takes the counts rather than re-deriving them.
//
// `null` for a number naming no entry in that list. That is not a decline of anything โ€” a reply
// citing problem 7 of 3 has declined something the request did not ask โ€” and folding it into the
// last bucket would report it as a refusal of a fact Iris checked in code.
//
// `words` is last because it is appended last, and its band is the one where a decline is an
// expected answer rather than a misuse: a page that really prints both spellings is a page whose
// contradiction was Iris's reading and not the model's error (#334 part B, `splitWordProblem`).
export function declinedSource(
  problem: number | undefined,
  counts: { verify: number; links: number; alt: number; ids: number; words: number },
): "verify" | "links" | "alt" | "ids" | "words" | null {
  if (problem === undefined) return null;
  const bands: ["verify" | "links" | "alt" | "ids" | "words", number][] = [
    ["verify", counts.verify],
    ["links", counts.links],
    ["alt", counts.alt],
    ["ids", counts.ids],
    ["words", counts.words],
  ];
  let seen = 0;
  for (const [source, n] of bands) {
    if (problem > seen && problem <= seen + n) return source;
    seen += n;
  }
  return null;
}

// Re-run the page agent with the fidelity problems it was told about, so it can
// fix them against the source image. Used only when verification fails.
async function correctPage(
  ctx: PipelineContext,
  agent: AgentSpec,
  img: InputImage,
  previous: string,
  problems: string[],
  lessons: string,
  // The output ceiling this call asks for, from `correctionCeiling`, or undefined for whatever the
  // deployment allows. Computed by the caller rather than here so the number asked for and the
  // number on `page_correction_failed` are one number and cannot drift apart. Asked for is not
  // always sent: an adapter lowers it to the deployment's ceiling if it is higher, and never the
  // other way round โ€” see the call site for what a larger number on that log line means.
  maxOutputTokens?: number,
): Promise<{ html: string | null; declined: Declination[] }> {
  // The link list is repeated here, not just in the first pass: a dropped link is one
  // of the problems this pass exists to fix, and it cannot re-attach a URL it can no
  // longer see. The image still does not show them.
  const user =
    `Your previous accessible-HTML output for this page had fidelity/accessibility problems:\n` +
    // Numbered rather than bulleted since #373 directive 4, and the numbers are load-bearing: they
    // are what a decline cites, and the alternative โ€” matching a paraphrase of the problem back
    // against the list โ€” attributes a disagreement to whichever entry it reads closest to.
    `${problems.map((p, i) => `${i + 1}. ${p}`).join("\n")}\n\n` +
    `## Your previous output\n\`\`\`html\n${previous}\n\`\`\`\n\n` +
    `Look at the source image again and return a corrected version that resolves every problem. ` +
    // The scope clause, which this path did not have and the feedback path always did
    // (`priorSection` in renderPage: "Keep everything the feedback does NOT concern exactly as it
    // was"). Both calls show the model its own previous output; only one of them said what to do
    // with the rest of it, and the asymmetry is what #132 was filed about โ€” a second pass that
    // re-derives the page from the image undoes work the first pass got right, and nothing
    // downstream can see that it did, because the corrected fragment IS the page from here on.
    // The page prompt now carries the same rule for both paths; this says it where the problems
    // are listed, so "resolve every problem" reads as a scope rather than as a licence.
    `Change nothing the list above does not name: every heading, table, list, label and attribute ` +
    `that is not part of a problem is carried over exactly as it stands.\n\n` +
    // #373 directive 4, and the issue asks for this one to be read hardest, so the bound is written
    // as narrowly as it can be stated: the licence is to decline a claim ABOUT THE HTML ABOVE that
    // the HTML above refutes, and nothing else. A problem about the picture is not covered, because
    // that is the judgement the checker was asked to make and this call is looking at the picture
    // precisely in order to act on it โ€” so the failure mode the issue warns about ("this is also a
    // way to ignore a true problem") has no wording here to reach through. The three examples are
    // the shapes measured on disk (an id said to be duplicated that appears once, an attribute said
    // to be missing that is present, a word said to be absent that is quoted in the markup), and the
    // two sentences after them exist because a licence stated without its limit reads as a general
    // one: "a problem you cannot settle in the HTML is a problem to fix" is the whole of the
    // narrowing, and "not the shorter answer" is there because the cheapest reading of any licence
    // is that it applies to whatever is hard.
    `One of these problems may be wrong. You are not required to act on a claim you can show is ` +
    `false, and the test is entirely inside the text above: where a problem asserts something ` +
    `about that HTML โ€” an id used twice, an attribute missing, a word absent โ€” and you can check it ` +
    `there and it is not so, make no change for it. A problem about what the IMAGE shows is not ` +
    `this case: whether content is missing, whether an alt describes the picture, whether a heading ` +
    `sits at the right level are judgements about the page, and you are looking at the page again ` +
    `in order to settle them. A problem you cannot settle inside the HTML is a problem to fix, and ` +
    `declining is not the shorter answer to a problem you are unsure of. ` +
    // The other half of the bound, and it is why the entries carry `CHECKED_IN_CODE` at all. Three
    // of the five problem sources are Iris's own comparisons โ€” the source file's link annotations,
    // a closed word list of placeholder alts, this page's own parsed tree โ€” and their wordings are
    // exactly the examples above ("More than one element on this page has id=โ€ฆ"). So a corrector
    // following the licence as written would decline into them, and the field that counts the misuse
    // (`verification.declined.code_checked`) would be counting compliance. Marking those entries and
    // excluding them here is what makes that number mean one thing: an unmarked claim about the
    // markup is the verifier's reading and is declinable, which is #373's instance C exactly โ€” the
    // CHECKER asserting a duplicate id on a page that has none โ€” while the marked one is the id
    // really being duplicated, since it was read off the tree this call was handed.
    `A problem marked "${CHECKED_IN_CODE.trim()}" is not one of these: it was settled against the ` +
    `source file or this page's own markup before you were asked, so fix it.\n` +
    // The FOURTH code-checked source is not covered by that sentence and must not be, which is why it
    // carries its own mark and gets its own line here (#334 part B). "So fix it" is true of a missing
    // link, a placeholder alt and a duplicated id; on a word written two ways it would order the model
    // to join `non-farm` into `nonfarm` on a page that prints the hyphen โ€” three of the shipped
    // model's six census cases โ€” putting a word the page does not print into delivered content, on a
    // page that had already passed. So this is the one entry in the request where the IMAGE showing
    // something else is a reason not to act, and it says so in those terms, because the paragraph
    // above spends four sentences ruling that reason out everywhere else. Stated as a split rather
    // than as an exemption: the presence of both spellings is not open, the choice between them is.
    `A problem marked "${SPELLINGS_CHECKED_IN_CODE.trim()}" is settled in one part and open in the ` +
    `other: both spellings really are in the HTML above, so that is not the thing to check, but ` +
    `which one this page prints is a question only the image answers. Make the page agree with the ` +
    `image. If the image shows the page printing both spellings, make no change for it and say so โ€” ` +
    `for this one problem, what the image shows IS a reason not to act.\n` +
    // The destination, because an instruction to "say which problem and why" with nowhere to say it
    // lands in the document โ€” the page gains a sentence about the checker, which is the defect
    // #373's own instance A describes in the other direction. `declined` is introduced here rather
    // than in agents/page.md because the first pass has no problem list and no use for the key, and
    // because that file is the cached system prefix this call reuses byte-for-byte: adding a field
    // there would reprice every page render in the run to give this one call a channel.
    `Say it in the reply and not in the page: add "declined" to the JSON, alongside "html", as ` +
    `[{ "problem": <the number from the list above>, "why": "<what the HTML shows instead>" }]. ` +
    `Omit it where you are acting on every problem. Return the page in "html" either way โ€” ` +
    `unchanged where you declined everything โ€” because a reply with no "html" is a reply this run ` +
    `cannot use, and every problem you did not decline is still to be fixed in the same reply.` +
    `${pageLinkContext(img.links).section}`;
  const res = await ctx.router.complete(
    PAGE_AGENT,
    "vision",
    [
      // The same prefix the first pass sent, down to the byte, so this call reads the
      // cache the first one wrote instead of paying for a near-copy of it. It also
      // gains the lessons, which this pass never had: a correction that re-derives the
      // page without them can undo the very thing a past correction taught.
      { role: "system", content: pageSystem(agent, lessons) },
      { role: "user", content: user },
    ],
    {
      step: "correct",
      images: [loadImage(img)],
      // The one capped call in the pipeline (#285).
      maxOutputTokens,
    },
  );
  ctx.log.agentCall({ agent, phase: "extraction", image: img.name, output: res.text });
  const parsed = extractJson<{ html?: string; declined?: unknown }>(res.text);
  const corrected = repaired(ctx, "correct", img, (parsed?.html ?? bareHtml(res.text) ?? "").trim());
  // Read from the envelope only, never from `bareHtml`: a reply that came back as raw markup with no
  // JSON around it has no `declined` key to read, and inferring one from prose in the page would be
  // this file reading a disagreement into a document rather than out of a reply.
  const declined = parseDeclined(parsed?.declined);
  // A correction that cannot be read is not a correction. Null keeps the version this
  // page already had, which passed everything except the fidelity check โ€” strictly better
  // than replacing it with the reply's own envelope, and the caller has always treated a
  // null here as "nothing to correct".
  if (!corrected) {
    ctx.log.event("page_correction_no_output", {
      image: img.name,
      page: img.order,
      chars: res.text.length,
      shape: replyShape(res.text, parsed),
      // Whether that empty reply was a refusal (#373 directive 4). The request says to return the
      // page either way, so this is a reply that broke the contract โ€” but a reply which declined
      // every problem and then omitted the page it was told to send back is a different failure from
      // one that produced nothing at all, and `shape` cannot separate them: both are an envelope. It
      // is the reading that decides whether the licence is being used as an exit.
      ...(declined.length ? { declined: declined.length } : {}),
    });
    return { html: null, declined };
  }
  return { html: corrected, declined };
}

// Merge instruction for splicing a specialist fragment into the page output.
const MERGE_SYSTEM = `You merge a higher-fidelity HTML fragment, produced by a specialist agent, into an
existing accessible HTML page. Replace the page's weaker representation of that SAME content
with the specialist fragment and change nothing else โ€” keep all other content, order,
headings, and structure exactly, and never leave both representations (no duplication).
Output body content only (no <html>/<head>/<body>/<main> wrapper).
Respond with ONLY this JSON: { "html": "<merged body content>" }`;

// Run a library specialist agent against the whole page image, asking it to
// extract only the content its contract covers. Returns its HTML fragment, or
// null when it finds nothing.
async function runSpecialist(ctx: PipelineContext, agent: AgentSpec, img: InputImage): Promise<string | null> {
  const system = `${agent.content}\n\n${ACCESSIBILITY_REQUIREMENTS}`;
  const user =
    `Extract ONLY the content your contract covers from this page image (filename: ${img.name}). ` +
    `If none is present, return {"no_content": true}. Otherwise respond with ONLY this JSON: ` +
    `{ "no_content": false, "html": "<your accessible HTML fragment>" }`;
  const capability = agent.capabilities.includes("vision") ? "vision" : "text";
  const res = await ctx.router.complete(
    agent.name,
    capability,
    [
      { role: "system", content: system },
      { role: "user", content: user },
    ],
    { step: "specialist", images: [loadImage(img)] },
  );
  ctx.log.agentCall({ agent, phase: "extraction", image: img.name, output: res.text });
  const parsed = extractJson<{ no_content?: boolean; html?: string }>(res.text);
  if (!parsed || parsed.no_content || !parsed.html?.trim()) return null;
  return repaired(ctx, "specialist", img, parsed.html.trim());
}

// Splice a specialist fragment into the page body, replacing the page's own
// (weaker) representation of that content. Returns the merged body, or null on
// failure (caller keeps the original page output).
async function mergeSpecialist(
  ctx: PipelineContext,
  img: InputImage,
  pageHtml: string,
  specialistName: string,
  reason: string,
  fragment: string,
): Promise<string | null> {
  const user =
    `## Current page (body HTML)\n\`\`\`html\n${pageHtml}\n\`\`\`\n\n` +
    `## Specialist (${specialistName}) fragment for the ${reason || "flagged"} content on this page\n` +
    `\`\`\`html\n${fragment}\n\`\`\`\n\n` +
    `Replace the page's existing representation of that content with this specialist fragment; ` +
    `keep everything else unchanged.`;
  const res = await ctx.router.complete(
    PAGE_AGENT,
    "text",
    [
      { role: "system", content: MERGE_SYSTEM },
      { role: "user", content: user },
    ],
    { step: "specialist_merge" },
  );
  ctx.log.agentCall({
    agent: { name: PAGE_AGENT, file: "page.md", content: MERGE_SYSTEM, capabilities: ["text"], sha: null, sessionBuilt: false },
    phase: "extraction",
    image: img.name,
    output: res.text,
  });
  const parsed = extractJson<{ html?: string }>(res.text);
  const merged = parsed?.html?.trim();
  // Redundant against a merge that only copies โ€” its two inputs are stripped already โ€” and kept for
  // the one that does not: a merge agent re-typing a word it is joining is the same transcription step
  // that produces these in the first place. NOT the phase's last seam, which is `correct`: dispatch
  // runs inside the render, and the correction pass runs after the fidelity verdict.
  return merged ? repaired(ctx, "specialist_merge", img, merged) : null;
}

// The agent names a suggestion could have resolved to, for the
// `specialist_unresolved` log line. Session-built agents (tmp/) are included
// because loadAgent prefers them, so they are genuinely dispatchable. Sorted so
// two runs of the same library produce comparable log lines. Best-effort: this
// exists to explain a miss, so it must never turn one into a failed run.
//
// `page` and `feedback` are excluded because they are the pipeline's own agents,
// not content types anything should route to.
//
// Standard-type names are NOT in here, even though they are the commonest
// near-miss. They are reported alongside, under `declined_types` (see
// `unresolvedCandidates`), because the two answer different questions and merging
// them makes the answer to the first one false: `candidates` reads as "what I could
// have asked for", and a standard type is not that โ€” it is declined by policy
// before the file is ever looked up, and since those nine files were deleted there is
// no file either.
function libraryAgentNames(ctx: PipelineContext): string[] {
  const names = new Set<string>();
  for (const dir of [ctx.paths.agentsDir, ctx.paths.tmpAgentsDir(ctx.sessionId)]) {
    let entries: string[];
    try {
      entries = readdirSync(dir);
    } catch {
      continue; // tmp/ may not exist yet, or agents_dir may be misconfigured
    }
    for (const f of entries) {
      if (!f.endsWith(".md")) continue;
      const logical = f.slice(0, -3);
      if (logical === PAGE_AGENT || logical === "feedback") continue;
      names.add(logical);
    }
  }
  return [...names].sort();
}

// The two lists a `specialist_unresolved` line needs, kept apart on purpose.
//
// `candidates` is what WAS dispatchable โ€” real files, so a near-miss against one of
// them ("chart" for `chartDataAgent`) is readable as a near-miss rather than needing
// a second run to investigate.
//
// `declined_types` is the other half of the explanation, and the commonest one: the
// most frequent near-miss is a plural or variant of a standard type. A suggestion of
// "tables" is not in STANDARD_AGENTS, so it never reaches the decline branch, and it
// resolves to no file โ€” so it lands in the unresolved branch, where omitting "table"
// hides the whole reason. Naming these separately says what is true of them: had the
// model written "table", it would have been declined, not dispatched. Reporting them
// as `candidates` would claim the opposite.
function unresolvedCandidates(ctx: PipelineContext): { candidates: string[]; declined_types: string[] } {
  return { candidates: libraryAgentNames(ctx), declined_types: [...STANDARD_AGENTS].sort() };
}

// Both unresolved branches below report the same thing, and say it from one place: a name the
// model wrote that no available agent answers to. They differ only in whether the name was empty
// or merely unknown, which is a distinction for the log line and not for the verifier.
const NO_SUCH_AGENT = "No agent of that name was available";

// If a page flagged a content type that an EXISTING library agent handles, run
// that specialist on the page and merge its higher-fidelity fragment into the
// page output. Non-blocking: any failure leaves the page output unchanged.
// dispatched=true means a library specialist ran (so the suggestion is already
// covered and should not be re-filed as a new-agent issue).
//
// `unmet` is a separate question from `dispatched`, and they disagree on four of the six
// exits below, so `dispatched` cannot stand in for it (it was tried, and mislabelled all
// four). Three of the four were silent โ€” no content, a throw, and a fragment that would not
// merge all return `dispatched: true` and so said nothing about a request that went unmet.
// The fourth is the standard-type decline, and it is the one that shipped a wrong ANSWER
// rather than none: `dispatched: false` there, so the verifier was told "no agent of that
// name was available" about a type declined by policy, on the commonest suggestion shape
// there is. Counting only the silent three reads as `dispatched` being incomplete, which
// undersells it. `dispatched` answers "is this suggestion already covered, or should it be filed
// as a new-agent issue"; `unmet` answers "did specialist content reach the HTML the
// verifier is about to judge". A dispatch that ran and returned nothing, threw, or produced
// a fragment that would not merge is dispatched-and-unmet: the delivered page is the page
// agent's own unaided work, which is exactly the case the caution exists to report. Set as
// a phrase rather than a flag because the four ways a request goes unmet are not
// interchangeable in the message โ€” telling the verifier "no agent of that name was
// available" about a specialist that ran and failed is simply false.
async function dispatchSpecialist(
  ctx: PipelineContext,
  img: InputImage,
  pageHtml: string,
  suggestion: { name: string; reason: string },
): Promise<{ html: string; dispatched: boolean; unmet?: string }> {
  // Normalized by the shared `logicalType`, and tested for standardness by the shared
  // `isStandardType`, so this site and `runContribution` cannot disagree about what a
  // name means. They did once: trim-then-strip versus strip-then-trim differed on
  // `"table.md "`, which slipped past one filter and not the other.
  //
  // The decline below is keyed on the STANDARD list rather than on what is on disk,
  // because a deployment that drops a `table.md` into `agents/` must not get the
  // original bug back: a standard specialist splicing its fragment over content the
  // general page pass already rendered, which is the two-representations-of-one-thing
  // duplication the page prompt forbids. The two outcomes also differ observably โ€” a
  // `specialist_unresolved` line blaming the name versus a `specialist_declined` line
  // stating the policy โ€” and the decline is the true one.
  const logical = logicalType(suggestion.name);
  // Every path out of here is logged, including the ones that do nothing.
  // `logical` is free text the model wrote, resolved to a file by name, so a
  // specialist silently fails to run whenever the model's wording and the
  // library's filenames disagree โ€” `chart` for `chartDataAgent.md`, a display
  // name, a plural. Without a log line for the miss, "routing was never
  // attempted" and "routing was attempted and the name did not resolve" are the
  // same observation: a page that came out of the general pass. `candidates`
  // names what WAS available, so a miss can be read as a near-miss rather than
  // needing a second run to investigate.
  //
  // `agent` carries the same meaning on every branch, so one filter on
  // `type=="specialist_unresolved"` can read `.agent` regardless of which branch
  // produced it. The empty-name case reports the raw string it could not use.
  if (!logical) {
    ctx.log.event("specialist_unresolved", {
      agent: suggestion.name,
      image: img.name,
      reason: "empty name",
      ...unresolvedCandidates(ctx),
    });
    return { html: pageHtml, dispatched: false, unmet: NO_SUCH_AGENT };
  }
  if (isStandardType(logical)) {
    // Not a failure: the general page pass already covers the standard types, so
    // this suggestion is correctly declined. Logged to keep the counts of
    // suggested / declined / dispatched / unresolved reconcilable from one run.
    //
    // Case-insensitive, so `"Table"` declines here rather than falling through to a
    // file lookup โ€” which on a case-insensitive volume would find `agents/table.md` if
    // a deployment added one, and dispatch the very specialist this rule forbids. The
    // name is logged as the model wrote it, since that is what a maintainer reading the
    // log has to recognize.
    // No `unmet`, and this is the one exit where that is a judgement rather than a fact. No
    // specialist ran, so the page agent's request was literally not granted โ€” but it was
    // ANSWERED, by a policy that says the general pass is this type's intended handler and not
    // a fallback for it. Cautioning here would narrow what the verifier may assert on the
    // commonest suggestion shape there is (see the STANDARD list), which on a table page means
    // declining to say a cell reads wrongly, and it would buy nothing measured: of the 7
    // requests behind #353, 0 were standard types โ€” every one named a map specialist and every
    // one landed on the unresolved branch below. A narrowed licence on the pages where the
    // pipeline believes the general pass is adequate is cost without a case.
    ctx.log.event("specialist_declined", { agent: logical, image: img.name, reason: "standard type" });
    return { html: pageHtml, dispatched: false };
  }
  const specialist = loadAgent(logical, {
    agentsDir: ctx.paths.agentsDir,
    tmpAgentsDir: ctx.paths.tmpAgentsDir(ctx.sessionId),
  });
  if (!specialist) {
    ctx.log.event("specialist_unresolved", {
      agent: logical,
      image: img.name,
      reason: "no agent file of that name",
      ...unresolvedCandidates(ctx),
    });
    return { html: pageHtml, dispatched: false, unmet: NO_SUCH_AGENT };
  }
  try {
    const fragment = await runSpecialist(ctx, specialist, img);
    if (!fragment) {
      ctx.log.event("specialist_no_content", { agent: specialist.file, image: img.name });
      return { html: pageHtml, dispatched: true, unmet: "The specialist that ran returned no content of its type" };
    }
    const merged = await mergeSpecialist(ctx, img, pageHtml, specialist.name, suggestion.reason, fragment);
    ctx.log.event("specialist_dispatched", { agent: specialist.file, image: img.name, merged: Boolean(merged) });
    // The only exit that met the request, and only when the merge produced something: a
    // fragment that was written and then not spliced leaves the same page behind as no
    // fragment at all, so it is reported as unmet even though the specialist did its part.
    return merged
      ? { html: merged, dispatched: true }
      : { html: pageHtml, dispatched: true, unmet: "The specialist's fragment could not be merged into the page" };
  } catch (e) {
    ctx.log.event("specialist_dispatch_failed", { agent: specialist.file, image: img.name, error: (e as Error).message });
    return { html: pageHtml, dispatched: true, unmet: "The specialist call failed" };
  }
}

// What the page agent said about its own weakest work, in the words it used, for the verifier to be
// told alongside the output. A page that asked for a specialist and did not get one is a page whose
// own author said it could not do this content reliably โ€” the request names the content and carries
// its reason โ€” and until now `dispatchSpecialist` was the only thing that read it: the request was
// routed, the outcome logged (`specialist_unresolved`), and the judgement of the page never heard
// about it.
//
// What that cost, measured. In one 100-page bench round across two page-model arms there were 7
// requests on 5 pages under 6 different names, every one of them a map specialist, 0 dispatched. The
// 5 pages were every page in the round whose verdict turned on reading ink. On one of them the page
// agent asked for help producing "a structured data table of each state's classification", got the
// classification wrong, and the verify step โ€” not told any of this โ€” ADDED a state to a category the
// page does not put it in; the correction obeyed, and a false sentence shipped in the delivered
// document (#353).
//
// Two limits, stated because the prompt is written to survive them rather than to rely on their
// being better than they are. It is page-level and not per-arm: 4 of those 7 requests came from the
// OTHER arm's reader on the same images, so a run reads its own reader's suggestions and not the
// union that made the count look strong. And it missed 2 of that round's hard pages outright. So
// this is not a detector and `agents/feedback.md` does not treat it as one โ€” it narrows what the
// verifier may assert on a page rather than deciding anything about the page, which is a use that
// costs nothing when the flag is wrong.
//
// The test is `unmet` and not `dispatched`: those two answer different questions and disagree on
// four of `dispatchSpecialist`'s six exits, so keying on `dispatched` claimed "no agent of that
// name was available" about a standard type that was declined by policy, and said nothing at all
// about a specialist that ran and returned nothing, threw, or produced a fragment that would not
// merge โ€” the last three being verbatim the case this caution is for. `unmet` carries the phrase
// too, so the sentence names which of the four it was.
//
// Both interpolated strings are free text a model wrote, landing in a message that already
// carries a fenced ```html block, so they are flattened and clipped: an unbounded reason
// containing a fence of its own would restructure the message after the block, and a long one
// buries the block it is supposed to annotate. The cap is generous for a sentence and short of
// anything that could crowd the output being judged.
function specialistCaution(
  suggestion: { name: string; reason: string } | undefined,
  unmet: string | undefined,
): string | undefined {
  if (!suggestion?.name || !unmet) return undefined;
  const name = oneLine(suggestion.name, 80);
  const why = oneLine(suggestion.reason, 300);
  if (!name) return undefined;
  return (
    `The page agent asked for a specialist it did not get: "${name}"` +
    `${why ? `, because "${why}"` : ""}. ${unmet}, so the HTML above is ` +
    `its own unaided attempt at the content it wanted help with.`
  );
}

// What a contradicted blank declaration says about its own output, in the same channel and the same
// register: something the agent under test said, not a question Iris is asking. It is the reason that
// page is judged at all (#371) โ€” without it the verifier is asked whether an empty fragment is faithful
// to the image, which is the question it answered "no problems" to on all 36 blank pages it was ever
// sent (the count at `blankSkip`), and with it the two halves of the reply are in front of it together.
//
// The affirmation is quoted in the log's own words rather than paraphrased: a sentence of Iris's own
// saying the page has a heading on it would be Iris asserting the thing the reply merely claimed, and
// this channel is documented as narrowing what the verifier may assert rather than deciding anything.
// Flattened and clipped by `oneLine` for the reason the specialist caution is: it is free text a model
// wrote, landing in a message structured by fences and `##` headings.
function blankContradictionCaution(affirmed: string | undefined): string | undefined {
  const quoted = affirmed ? oneLine(affirmed, 200) : "";
  if (!quoted) return undefined;
  return (
    `The page agent returned NO page for this image and said the page is blank, and the same note also ` +
    `said "${quoted}". The empty fragment above is the answer it gave, so the two halves of its reply ` +
    `disagree about whether this page has anything on it.`
  );
}

// Backticks and newlines out, then clipped to a sentence's worth. Backticks rather than only
// the triple: a single one opens inline code, which is enough to swallow the punctuation the
// sentence around it depends on. Flattening the newlines is the sharper half of the two,
// though, and not only a tidiness measure: the verify message is structured by `##` headings,
// and a string that cannot start a line cannot forge one.
function oneLine(s: string, max: number): string {
  const flat = s.replace(/`/g, "'").replace(/\s+/g, " ").trim();
  return flat.length > max ? `${flat.slice(0, max).trimEnd()}โ€ฆ` : flat;
}

interface PageOutcome {
  fragment: Fragment;
  // A genuinely-new content type to file for contribution, if any. A suggestion
  // already covered by a dispatched library specialist is not reported.
  suggestion?: { name: string; reason: string; image: string };
  // Set when this page's own extraction threw and the fragment is a stand-in rather
  // than the page's content (`failedPage`). Carried explicitly rather than inferred
  // from the fragment, because a caller must not have to pattern-match HTML to find
  // out whether the document it was handed is whole.
  failed?: true;
  // Set when the fidelity check REJECTED this page and the correction pass did not replace
  // it, so `fragment` above is the markup that failed the check. Carried explicitly for the
  // same reason `failed` is: a caller must not have to pattern-match HTML, or join two
  // event streams, to find out whether what it was handed is known to be wrong. Reported in
  // the delivered document as `@page-uncorrected` (assembly.ts `wrapDocument`, issue #328).
  //
  // NOT set for a correction the pass DID replace the page with. That page may still be
  // wrong โ€” replaying the check over 57 corrected pages put the pass rate at 26% (#288) โ€”
  // but it is a repaired page rather than the rejected one, and a marker that covered both
  // would fire on most of the pages of an ordinary round and stop meaning anything. Whether
  // a correction worked is `page_correction_recheck`'s question and it is a sample.
  uncorrected?: true;
  // The error that page threw, kept so it can be re-raised if it turns out EVERY page
  // failed (see runExtraction). Containment replaces one message with a document; when
  // there is no document, the message is all there is, and a fresh one written here
  // would be a worse diagnosis than the provider's own.
  error?: unknown;
}

// What one page leaves behind when its own extraction throws.
//
// Everything else in this file already degrades a PAGE rather than a document: a
// specialist that fails is logged and the page is kept as the general pass wrote it
// (`dispatchSpecialist`), a fidelity check that cannot run counts as nothing to
// correct (`failedCheck`), a correction that comes back empty is discarded. Only the
// page's own render was fatal to the whole run โ€” so a model call that hit the output
// ceiling on page 26 of 50 threw away 24 pages that had already been rendered,
// verified and corrected, and delivered nothing (issue #135). "This page's output is
// unusable" and "this document is unrecoverable" are different claims, and the caller
// is better placed than this function to decide whether 24 good pages are acceptable.
//
// The page is NOT silently dropped. An empty fragment is filtered out at assembly, so
// the delivered document would simply be missing a page with nothing to say so โ€” and
// a page absent for a reason nobody recorded is the failure this function exists to
// avoid re-creating one level down. The marker is a comment because the alternative is
// worse: a visible note is prose Iris wrote into a document whose whole contract is
// that every word in it is a word on the page. A comment is invisible to a reader,
// inert to axe and to `flatten`, and findable by tooling โ€” the same trade
// `wrapDocument` makes for @unresolved, and it sanitizes runs of dashes for the same
// reason (a `--` inside a comment ends it early).
// What a page failure says about the reply, when the reply is the thing that failed (issue #293).
// Empty for every other error โ€” a throttle, a stall, a stream that stopped: `error` is the whole of
// what is known about those, and a `reply_chars: 0` on such a line would read as a model that
// answered with nothing.
//
// Every page path that loses a reply to the ceiling gets it from here rather than assembling its own
// fields, because there are three of them โ€” a first pass, a re-extraction a user asked for, and a
// correction โ€” and the evidence is worth nothing if the round that produced it is the one round that
// did not record it. It is what `editor_truncated` has carried since #277, at the same width, from
// the same `replyExcerpt`: what the reply reached, and both of its ends. A tail mid-sentence in
// content the head has not reached is a page that needed the room; a tail repeating rows already in
// the head is a model rewriting the page it was given. Nothing in the pipeline reads any of it โ€” the
// question is a person's, and it is otherwise unanswerable, because a truncated round has already
// been billed for a full ceiling of output and cannot be asked again.
//
// `instanceof` and not `isTruncatedResponseError`: the predicate also matches an error that arrived
// having lost its prototype, which has a message and no `text` to quote.
//
// A truncation that wrote NOTHING is therefore `truncated: true` with no `reply_head` and
// `reply_chars: 0` โ€” a call that spent a whole ceiling of output and never began the page, which is a
// model problem and not a room problem (`EMPTY_REPLY`, providers/types.ts). It is the shape a
// reasoning model produces, so it is the one to watch when the page model changes.
function truncationEvidence(e: unknown): Record<string, unknown> {
  if (!(e instanceof TruncatedResponseError)) return {};
  return { truncated: true, reply_chars: e.chars, ...replyExcerpt(e.text) };
}

// A VERDICT THAT CANNOT BE OBTAINED IS NOT A PAGE THAT CANNOT BE EXTRACTED, and until #364 this
// file could not tell those apart on the one path that matters most. `verifyAgentOutput` is already
// non-blocking for an absent Feedback Agent and for a reply that will not parse, but a PROVIDER error
// is rethrown โ€” and the first verify call had nothing to catch it, so a throttle or an output-ceiling
// overrun on the CHECK propagated out of `extractPage` into `failedPage`, which logged
// `page_extraction_failed` and shipped a `@page-failed` marker for a page whose extraction had
// succeeded and was sitting in `innerHtml`. Measured once on a 100-page bench arm: `acir-p049` was
// extracted as 8,855 characters of HTML โ€” a complete statistical table, 568 words โ€” and delivered as a
// 156-byte comment, because its verify call spent a full 32,000-token ceiling and threw. $0.5051 of
// the page's $0.6634 was the call that deleted it, 3.2x what the extraction it was checking cost.
//
// This is the same defect #171 fixed for the correction pass and the sampled recheck, arriving one
// call earlier, and the policy it is fixed to is this file's own: a specialist that fails leaves the
// page as the general pass wrote it, and a fidelity check that cannot run counts as nothing to
// correct. An unobtainable verdict IS a fidelity check that cannot run.
//
// Three things the misattribution cost besides the page, and they are the reason this has its own
// event rather than a quieter `.catch`. The delivered document said "the source pages above could not
// be extracted and none of their content is here" (`wrapDocument`), which was false. `pages_failed`
// and every triage of why pages fail recorded a vision failure, so anyone tuning the page agent on
// that signal was tuning the wrong agent. And the marker told the operator to raise
// `providers.*.max_tokens`, which buys the verifier room to write MORE about a page it has already
// judged โ€” the wrong lever, pushing the wrong way, on the one line an operator was given.
//
// `page_verify_error` and not a `page_verify_ok` field alone, because the evidence is the same
// evidence `page_correction_failed` carries and is worth the same: `truncated` names the one shape
// with a configuration remedy, and `reply_chars` with both ends of the reply is what distinguishes a
// verifier that needed the room from one that wrote an essay. `step` is on the line because two call
// sites reach here and they are not the same event โ€” one is the check that decides whether a
// correction is bought, the other the gate on keeping one.
//
// The third verify call, the SAMPLED recheck, keeps the `page_correction_recheck_failed` event #171
// gave it and is not folded in here. Two reasons, and neither is inertia: that event predates this
// one and has been read across rounds, so renaming it would split one measurement across two names
// in the logs it is compared against; and it is the one verify failure that costs nothing at all,
// because the sampled recheck decides nothing whether it answers or not. The two failures on this
// line cost something โ€” a correction not bought, or a correction bought and discarded โ€” so the pair
// that belongs together is the pair that is together.
function verifyUnobtainable(
  ctx: PipelineContext,
  img: InputImage,
  step: "verify" | "recheck_binding",
  e: unknown,
  extra: Record<string, unknown> = {},
): void {
  const message = (e instanceof Error ? e.message : String(e)).replace(/\s+/g, " ").trim();
  ctx.log.event("page_verify_error", {
    image: img.name,
    page: img.order,
    step,
    error: message,
    ...truncationEvidence(e),
    ...extra,
  });
}

function failedPage(ctx: PipelineContext, pageAgent: AgentSpec, img: InputImage, e: unknown): PageOutcome {
  const message = (e instanceof Error ? e.message : String(e)).replace(/\s+/g, " ").trim();
  // No `ceiling` beside the evidence, unlike `page_correction_failed`: a first pass asks for no cap
  // of its own, so the number that was hit is the deployment's and the error already names it.
  ctx.log.event("page_extraction_failed", { image: img.name, page: img.order, error: message, ...truncationEvidence(e) });
  const note = message.slice(0, 300).replace(/--+/g, "โ€”");
  return {
    fragment: {
      image: img.name,
      order: img.order,
      agent: pageAgent.file,
      region: "page",
      innerHtml: `<!-- @page-failed ${img.order}: ${note} -->`,
      edges: [],
      log: `extraction failed: ${message}`,
    },
    failed: true,
    error: e,
  };
}

// Did the fidelity check actually find something? `verifyAgentOutput` is deliberately
// non-blocking: it answers ok=false with an empty problem list when the check could
// not be made at all (no Feedback Agent configured, an unusable reply). That has
// always counted as "nothing to correct", and both uses below depend on it meaning the
// same thing โ€” one to decide whether to correct, the other to decide whether a
// correction may replace a fragment that had passed.
function failedCheck(verdict: VerifyVerdict): boolean {
  return !verdict.ok && verdict.problems.length > 0;
}

// Everything one page needs: render -> optional specialist merge -> verify ->
// optional self-correction. Pages share no mutable state, so this is safe to run
// concurrently for several pages at once (see runExtraction).
async function extractPage(
  ctx: PipelineContext,
  pageAgent: AgentSpec,
  img: InputImage,
  lessons: string,
  sampler: RecheckSampler,
  previous?: string,
): Promise<PageOutcome> {
  const { html, log, suggestion, outputTokens, blank, blankContradicted } = await renderPage(
    ctx,
    pageAgent,
    img,
    lessons,
    previous,
  );
  let innerHtml = html;
  let logNote = log;
  let dispatched = false;
  let unmet: string | undefined;

  // Specialist dispatch: if the page flagged a content type that an existing
  // library agent handles (e.g. a chart), run that agent and merge its
  // higher-fidelity fragment into the page BEFORE the fidelity check.
  if (suggestion?.name) {
    const result = await dispatchSpecialist(ctx, img, innerHtml, suggestion);
    dispatched = result.dispatched;
    unmet = result.unmet;
    if (result.html !== innerHtml) {
      innerHtml = result.html;
      logNote = logNote ? `${logNote}; merged ${suggestion.name}` : `merged ${suggestion.name}`;
    }
  }

  // What the two free caption checks had to work with on this page, before anything is judged or
  // repaired (#356). Here rather than beside the correction for the reason `correctionEffect` cannot
  // carry it: this is a fact about the fragment the verifier is about to be shown, and most pages
  // that have one are never corrected at all, so a field on `page_corrected` would be silent on
  // exactly the pages where a check declining is indistinguishable from a check finding nothing.
  //
  // After the merge, so it reads the same fragment the verify call does โ€” a specialist that rewrote
  // the figure changed both strings this compares. Nothing consumes these lines: they are the record
  // that the subject existed, and the reasons the comparison was not available.
  for (const claim of captionClaims(innerHtml)) {
    ctx.log.event("page_caption_claim", { image: img.name, page: img.order, ...claim });
  }

  // A page the agent declared blank is not sent to the verifier. The fidelity question this call
  // asks is whether the fragment is faithful to the image, and an empty fragment has no content to
  // be unfaithful with: the Feedback Agent is shown the source image and an empty code block, and
  // in 36 such judgements โ€” 9 pages of a 100-page corpus, two page-model arms, two shas โ€” it
  // passed every one (#294). What that bought was $0.0859 per arm, 0.77% of the deployed lineup's
  // bill and 1.33% of the cheaper one the sprint is heading for, which is the direction the share
  // moves: a per-image cost that does not shrink is a growing fraction of a shrinking bill.
  //
  // `blank` AND an empty fragment, not `blank` alone. The flag says what the model answered; the
  // emptiness is what makes the argument above true. A specialist cannot reach a blank page today
  // (the declaration returns before a `suggested_agent` is read), so the conjunction is a belt โ€”
  // and if some later path does put content into a page that was declared blank, the check comes
  // back on rather than being skipped on the strength of a stale flag.
  //
  // What is NOT given up. The two code-level checks below still run on this page, and both are
  // detectors of a wrong declaration that the verifier is not needed for: a page carrying link
  // annotations that came back empty fails `missingLinks` exactly as before, and buys a correction
  // โ€” which for a blank page means the page is re-rendered against the image with a named problem,
  // and the corrected fragment is then verified in turn (`recheck_binding`, because the check did
  // not fail). So a "blank" page that the FILE says has content in it is still caught, for free.
  //
  // What is: a page with content on it that the agent declared blank confidently and that carries
  // no annotations. The model call was the only thing that could catch that, and it never has โ€”
  // 0 of 36. The failure mode it was written for was observed (#190, four pages) and is now caught
  // one step earlier and for nothing by the doubt-word veto in `blankDeclaration`, which refuses a
  // hedged declaration before it can reach this line at all (`blank_vetoed` on `page_no_output`).
  // A confident-but-wrong declaration would leave a `page_blank` line and a blank count in
  // diagnostics as its evidence, and no verdict.
  //
  // The one declaration that IS sent, and the only spend #371 adds: one whose log names something on
  // the page. Until the reply could state blankness in a field, that page was refused outright, and
  // refusing was chosen as the cheaper error when the only alternative was dropping it in silence
  // (`AFFIRMED_NOUN`). A stated declaration makes a third answer available, and the price of it is
  // this call: the page is delivered as the field says, and the verifier is shown the image with the
  // empty fragment, so a log that was RIGHT about the heading it named buys a correction and the
  // reader gets the page โ€” which is more than the refusal ever gave them, since a refused page is a
  // hole in the document until someone re-extracts it. What it costs is bounded by how rarely the two
  // halves disagree: 1 of the 125 blank declarations in every bench round on disk, off 2,189 page
  // renders. The number above prices that one โ€” $0.0859 bought 9 judged blank pages on one arm, so
  // under a cent each, against a page render of several cents.
  //
  // And that one declaration is the argument. Its log is "Source image is entirely blank/white with no
  // visible content, text, graphics, or printed page number. No page-break marker can be emitted
  // because no folio number is visible on the page. No content to transcribe." โ€” a page that is blank,
  // said three times, and `contentAffirmed` reads `image is entirely` out of the first clause and
  // refuses it. So the page this spends a verify call on is a page today's code reports LOST, and the
  // call is what turns it back into a delivered page. A log that is right about the heading it names is
  // the case the branch was written for; the case it is actually taken on, so far, is a page the regex
  // misread.
  // Computed once, above the check and both rechecks: it is a fact about this page's render, so a
  // recheck of a correction to that render carries the same one. It is read from `suggestion` and
  // the dispatch's own return rather than from the log, so the caution and the routing line cannot
  // disagree about whether the request was met. `unmet` stays undefined when no suggestion was made
  // at all, which is the same no-caution answer by a different route.
  //
  // Two cautions, joined rather than one winning. Only one of them can be present today โ€” a blank
  // declaration returns before `suggested_agent` is read, so `suggestion` is undefined on exactly the
  // pages a contradiction caution exists for โ€” and the join is written for the same reason the
  // conjunction below is: that is a fact about today's control flow, not a property of either caution,
  // and if a later path does hand the verifier a page that says both things, both should reach it
  // rather than one being dropped by a precedence nobody chose. Empty stays `undefined` and not `""`
  // to keep the value the shape its type declares; `verifyAgentOutput` drops a blank caution either
  // way, so this is about what the call SAYS and not about the bytes.
  const cautions = [specialistCaution(suggestion, unmet), blankContradictionCaution(blankContradicted)].filter(
    (c): c is string => c !== undefined,
  );
  const caution = cautions.length > 0 ? cautions.join(" ") : undefined;
  const blankSkip = blank === true && innerHtml === "" && !blankContradicted;
  // Which KIND of unjudged this page is, for `page_verify_ok`'s `skipped` below. It cannot be read off
  // the verdict: `unjudgedVerdict()` is deliberately one shape for every page nothing looked at, and a
  // call that threw is the one kind of unjudged that COST money, so a reader pricing the skips must be
  // able to subtract it.
  let verifyErrored = false;
  const verdict = blankSkip
    ? unjudgedVerdict()
    : // `logNote` and not `log`: where a specialist merged, the note says so, and the fragment the
      // verifier is judging is the merged one. This is the only verify call whose log describes the
      // fragment it is about โ€” both rechecks judge a CORRECTED fragment, and a correction reply is
      // parsed for `html` alone (`correctPage`), so there is no log of it to send and the first
      // pass's note is about text that has since been rewritten. Sending that would invite exactly
      // the false problem #349 measured, one round later.
      // And `.catch` for the reason `verifyUnobtainable` is written out at length: a check that
      // throws must not be able to delete the page it was checking (#364). What it returns is the
      // verdict for a page nothing looked at, which is what this call failing MEANS โ€” no problems, so
      // no correction is bought, and the page ships exactly as it would have with no verifier
      // configured at all.
      await verifyAgentOutput(ctx, pageAgent, img, [{ html: innerHtml, caution, log: logNote }], "verify").catch(
        (e: unknown) => {
          verifyUnobtainable(ctx, img, "verify", e);
          verifyErrored = true;
          return unjudgedVerdict();
        },
      );

  // Whether the page's links arrived is checked here rather than left to the
  // Feedback Agent: it verifies the output against the IMAGE, which is the one place
  // a link target does not appear, so a dropped link is invisible to it and a
  // fabricated one unfalsifiable. The comparison against the file's own annotations
  // is exact, so it is made in code and handed to the same self-correction pass as
  // any other fidelity problem.
  const missing = missingLinks(img.links, innerHtml);
  if (missing.length) {
    ctx.log.event("page_links_missing", { image: img.name, links: missing.map((l) => l.href) });
  }

  // And whether any image on the page was described with a placeholder instead of a
  // description, checked here for the same reason and on the same terms: it has an exact answer,
  // it costs nothing, and it runs on every page rather than on the ones a sampled verifier
  // happens to look at. The difference from a missing link is which way the model fails โ€” a
  // dropped link is invisible to the Feedback Agent, while a gutted alt is something the
  // DEPLOYED verifier catches 6 times out of 6 and the cheap ones catch 0โ€“2 times out of 6
  // (#290, and see alt.ts for the whole table). So this is not a blind spot being covered, it is
  // a capability being moved off the model's bill: it is the one defect class where dropping the
  // verifier to a cheaper model costs real detection, and a free rule is what buys it back.
  const generic = genericAlts(innerHtml);
  if (generic.length) {
    ctx.log.event("page_generic_alt", { image: img.name, page: img.order, alts: generic });
  }

  // And whether the page used one id twice, on the same terms again โ€” exact, free, every page โ€”
  // with one difference from the two above that is the reason it is here at all: this defect has a
  // second reader downstream, and that reader cannot repair it. `namespaceAnchors` prefixes each
  // page's ids at assembly, so a cross-page collision is fixed in code; a page that collided with
  // ITSELF gets the same prefix on both copies and stays collided, which is the one case that
  // function declines by name. Its remaining reporter is lint on the assembled document, for a page
  // nobody is going to look at again with the image in front of them โ€” and often under a name that
  // page never wrote, since a finding on a prefixed id reads `p3-fn-1`.
  //
  // Often and not always, which is worth stating exactly because the unqualified version is
  // tempting: `namespaceAnchors` renames only ids more than one PAGE claims, and returns every page
  // byte-for-byte when there are none, so a document whose sole defect is page 3 using `fn-1` twice
  // with no other page claiming `fn-1` reaches lint with `fn-1` intact. Per-page footnote numbering
  // usually does produce the cross-page collision that makes the rename happen โ€” it is the shape
  // both duplicates measured on disk have โ€” but the case for asking here does not rest on it. The
  // page step is where the model still holds the image, and lint runs after delivery either way.
  //
  // This is also the half of #373 directive 3 that a code check can honestly buy. The directive's
  // claim is that the checker's guess becomes an instruction because "nothing in the run knows the
  // answer" when it is asked, and the run now knows it: what that stops is the REAL defect
  // reaching the document, not a false problem being written, which is directive 4's business.
  const duplicated = duplicateIds(innerHtml);
  if (duplicated.length) {
    ctx.log.event("page_duplicate_ids", { image: img.name, page: img.order, ids: duplicated });
  }

  // And whether the page wrote one word two ways โ€” `Compos-ite` here, `Composite` there โ€” which is
  // exact and free on the same terms as the three above, and is the half of #334's hyphen family that
  // a strip cannot do (part B; `hyphens.ts` has the argument, and `stripSoftHyphens` is part A).
  //
  // The difference from the three above is what Iris knows after finding one. A dropped link, a
  // placeholder alt and a duplicate id each have one right answer that this code can state, so their
  // problems name the repair. Here the page contradicts itself and only the agent holding the image
  // can say which spelling the printing carries, so the problem names the contradiction and asks โ€”
  // and a page that genuinely prints both is a legitimate decline rather than a defect.
  //
  // True of what is decidable HERE, on one page, and no longer true of the document: `joinBrokenWords`
  // closes a break up at assembly where the rest of the document writes the word whole and the
  // fragment after the hyphen is no word at all. It cannot help this call site โ€” the dictionary it
  // needs is the whole body, and pages are extracted concurrently.
  //
  // The two do not overlap, and that is a condition IN that pass rather than a property of the two
  // shapes: it declines any word whose closed spelling is on the page carrying the hyphen, which is
  // exactly the population raised here. Without that condition they would collide on 7 of the 36 joins
  // measured over the corpus โ€” every one of them a word this call site had already sent to the model,
  // whose licence to answer "the page really does print both spellings" a later join would silently
  // revoke. So a word this page settles is not reached by it, by construction.
  //
  // After the soft-hyphen strip, and load-bearing that it is. #334 measured the interaction on the
  // page this rule is written from: `Govern<U+00AD>ment` has contiguous letters, so a page carrying
  // the invisible break writes the word "whole" as far as any text comparison is concerned and the
  // contradiction goes undetected. Every seam feeding `innerHtml` is stripped (see `stripped`), so
  // the fragment read here cannot hide one that way.
  const contradictions = splitWordContradictions(innerHtml);
  if (contradictions.length) {
    ctx.log.event("page_split_words", {
      image: img.name,
      page: img.order,
      // Both spellings, as the page wrote each. The split form alone would name the word but not the
      // evidence, and the evidence is the pair: a reader checking this line against the document is
      // asking whether the page really contains both, which is the one thing that makes it a defect.
      words: contradictions.map((w) => `${w.split} / ${w.joined}`),
    });
  }

  // page_verify_ok / page_verify_failed report the Feedback Agent's verdict and
  // nothing else, exactly as they did before links existed โ€” a missing link is not
  // part of that verdict, and folding it in would make the two events mean different
  // things in old logs and new ones. `page_links_missing` above is the signal for a
  // correction driven by a link.
  const verifyFailed = failedCheck(verdict);
  if (verifyFailed) {
    // `kinds` is what a reader of this log can subtract from `verify_failed`: the SET of
    // problem kinds the verdict named (feedback.ts `VERIFY_KINDS`), so a page that lost
    // three table rows and a page whose alt text was polished stop being the same line
    // (issue #182). A set rather than one label per problem, because the question it answers
    // is what was wrong with the PAGE, and a page with two missing rows lost content once.
    // `untagged` is how many problems arrived with no kind this code knows โ€” a split
    // computed while most of a round was untagged would be a split of the tagged half
    // reported as the whole, and this is the only field that would say so.
    ctx.log.event("page_verify_failed", {
      image: img.name,
      problems: verdict.problems,
      kinds: verdict.kinds,
      untagged: verdict.untagged,
    });
  } else {
    // `unjudged` where the verdict is the non-blocking default rather than a pass โ€” no
    // Feedback Agent, nothing to verify, or a reply that would not parse (feedback.ts).
    // The event stays `page_verify_ok` because that is what the run did with it, and every
    // reader of this log still counts it as one verified page. The field is for the reader
    // that has to tell "the verifier looked and was satisfied" from "nobody looked": a
    // measurement OF the verifier drawn from these lines would otherwise take pages nothing
    // judged as its population (issue #180, src/pipeline/calibration.ts). Omitted, not
    // false, on the ordinary pass โ€” a log full of `unjudged: false` says nothing.
    //
    // `skipped` says which kind of unjudged this is: a call that could not be made, or one that was
    // not bought. Both must stay out of any pass rate โ€” that is what `unjudged` is for and it is set
    // either way โ€” but they are different facts about a run, and only one of them is a saving. A
    // reader counting these lines is the only way to price the skip after the fact, so the reason is
    // on the line rather than inferred from a `page_blank` on the same image (issue #294).
    // `"error"` is the second value the sentence above always described and nothing had ever emitted
    // (#364): the call that could not be made, as against the one that was not bought. Both stay out
    // of any pass rate, and only `blank` is a saving โ€” an `error` page was billed for a full ceiling of
    // output and got no verdict for it, so a reader adding the two together would price a loss as a
    // saving. `page_verify_error` above carries the evidence; this field is what makes the page
    // countable beside the others without reading two event streams.
    ctx.log.event("page_verify_ok", {
      image: img.name,
      ...(verdict.unjudged ? { unjudged: true } : {}),
      ...(blankSkip ? { skipped: "blank" } : verifyErrored ? { skipped: "error" } : {}),
    });
    // The verdict that describes a defect and then passes the page. `ok` is the verdict's
    // `faithful`/`accessible` FLAGS, and `failedCheck` needs a false flag AND a named problem
    // before the run will spend a correction โ€” so a verdict that names one while both flags
    // stay true ships the page, and the sentence it wrote goes nowhere: `page_verify_ok`
    // above carries no `problems`, and nothing downstream looks at the image again.
    //
    // Measured while calibrating the verifier against injected defects: of 30 damaged pages
    // it perceived 28 and flagged 25, and 3 of the 5 it did not flag it described in full โ€”
    // a swapped pair of paragraphs quoted back verbatim, an `<h4>` among `<h2>` siblings
    // named as such, `faithful: true` on both (issue #210). Which is a different repair from
    // a verifier that cannot see: what it needs is for the flag to follow the prose.
    //
    // Logged and not acted on, deliberately. The one-line fix โ€” any named problem fails the
    // page โ€” would buy a correction round for every `alt_quality` suggestion the same agent
    // is asked to volunteer, which is the class of finding least likely to be worth a page
    // call, on top of a verification share already under investigation for costing 24%. The
    // kind-gated version (a `content_missing`, `content_wrong` or `structure_wrong` problem
    // fails the page whatever the flags say) is the one worth having, and pricing it needs
    // this count over a fleet rather than over an 11-page corpus. So this event decides
    // nothing and costs nothing: the page ships exactly as it did.
    //
    // #290's placeholder-alt rule is not a reversal of that, and the difference is the whole
    // reason it is a code check: what is refused above is buying a page call on the verifier's
    // OPINION that a description could be better, which it volunteers on request and which has
    // no exact answer. What is bought below is a placeholder โ€” an alt that is one word for the
    // medium โ€” which is decidable from the bytes, costs nothing to find, and fires on 0 of the
    // 1,064 alts in the bench corpus, so it is not a share of the bill at all. If the verifier
    // ever names an `alt_quality` problem that IS a placeholder, the free rule has it already.
    //
    // Only the first verdict, which is the one that decides whether a correction is bought.
    // A recheck's own disagreement is already readable on its line, since
    // `page_correction_recheck` carries both its `ok` and its `problems`.
    if (verdict.problems.length > 0) {
      ctx.log.event("page_verify_inconsistent", {
        image: img.name,
        page: img.order,
        // The prose, because the prose is the finding: "the HTML reverses this order" is what
        // says the verifier saw the defect, and no count of it can be re-read as that.
        problems: verdict.problems,
        // And the kinds, because they are what a kind-gated rule would act on โ€” a page whose
        // problems are all `alt_quality` is not this bug, it is the agent doing what it was
        // asked. `untagged` is the honest reading of a verdict in plain prose, which is what
        // an agent file predating the kinds returns and what makes a kind-gated rule a no-op
        // on that page: it cannot be counted for or against.
        kinds: verdict.kinds,
        untagged: verdict.untagged,
      });
    }
  }

  // Whether the correction below actually replaced this page, which is a different question
  // from whether one was bought or how it ended. Declared out here because what it decides is
  // a property of the fragment this function RETURNS: `verifyFailed && !repaired` is a page
  // the verifier rejected and nothing fixed, and until #328 that was the one thing the
  // delivered document had no way to admit to.
  //
  // A failed verdict always buys a correction โ€” `failedCheck` requires a named problem, so
  // `verifyFailed` implies `problems.length > 0` and the branch below always runs โ€” which is
  // why "rejected and never corrected" is not a state this can report: it does not exist. The
  // states it does cover are the ways the pass ends without repairing the page, and they are
  // one fact about it: the call threw (`page_correction_failed`), it answered with nothing
  // (`page_corrected` `empty`), it answered with the page it was given (`identical`), its
  // answer came back at a fraction of that page's size and was refused
  // (`page_correction_rejected`), or it answered with a different STRING carrying the same page
  // โ€” re-indented, or `&` for `&amp;` โ€” which is adopted and is also `identical`. Whichever it
  // was, what this function returns is the page the verifier named problems in.
  //
  // So the rule is one line rather than five, and it is the same line `page_corrected` already
  // writes: `repaired` is set exactly where that event's `result` is `kept`. A reader can check
  // the set against the log without a rule about which values count.
  let repaired = false;
  // The verifier's problems as it wrote them, then the four Iris raised itself, each marked as
  // checked in code (#373 directive 4 โ€” `CHECKED_IN_CODE` has why). The mark is on the string the
  // corrector reads and on nothing that is counted: `problems.length` is the same number either way,
  // and `declinedSource` reads positions in this order rather than the text.
  //
  // TWO marks, not one, and the difference is which sentence of the licence the entry then falls
  // under: three of these problems are wrong by construction and the request tells the model to fix
  // them, while a word written two ways is a settled FACT with an open remedy that the image decides
  // (`SPELLINGS_CHECKED_IN_CODE` has the whole argument).
  const problems = [
    ...(verifyFailed ? verdict.problems : []),
    ...missing.map((l) => missingLinkProblem(l) + CHECKED_IN_CODE),
    ...generic.map((a) => genericAltProblem(a) + CHECKED_IN_CODE),
    ...duplicated.map((id) => duplicateIdProblem(id) + CHECKED_IN_CODE),
    ...contradictions.map((w) => splitWordProblem(w) + SPELLINGS_CHECKED_IN_CODE),
  ];
  if (problems.length) {
    // What the correction was asked to fix, for the event below. Any of the five can fire on one
    // page, and they cost the same call but mean different things: a link the model dropped, a
    // placeholder alt, a duplicate id and a word written two ways are exact, code-checked findings,
    // while a fidelity problem is the Feedback Agent's judgement.
    //
    // `words` is the one of the five whose right answer Iris does not hold โ€” the other three name
    // their repair, this one names a contradiction and asks the image โ€” but that is a fact about the
    // problem text, not about this line, which only records what bought the call.
    //
    // `both` is more than one source, which is what it has always counted โ€” until #290 there
    // were two sources, until #373's directive 3 three, and until #334's part B four, so every line
    // an old log calls `both` is still one and no reading of an old log changes. What it no longer
    // does is name WHICH pair, and that is on purpose rather than traded away: the alternative is
    // thirty-one buckets for five sources, where four gave fifteen, and the per-source detail is
    // already exact on
    // `page_links_missing`, `page_generic_alt`, `page_duplicate_ids` and `page_split_words`, all
    // keyed by the same `image` as this line.
    const sources = [
      verifyFailed ? "verify" : null,
      missing.length ? "links" : null,
      generic.length ? "alt" : null,
      duplicated.length ? "ids" : null,
      contradictions.length ? "words" : null,
    ].filter((s): s is string => s !== null);
    const trigger = sources.length > 1 ? "both" : sources[0];
    const before = innerHtml.trim();
    // A correction that cannot complete costs the CORRECTION, not the page.
    //
    // This page has already been rendered, verified, and โ€” on the links path โ€” found to be
    // good. The correction is an improvement step, and an improvement step that throws must
    // not be able to delete the thing it was improving: before this, a provider error here
    // propagated out of `extractPage` into `failedPage`, which logged
    // `page_extraction_failed` and shipped a `@page-failed` marker for a page whose
    // extraction had succeeded minutes earlier and was sitting in `innerHtml`. On a real
    // 50-page run that cost page 25 outright โ€” a valid 17,721-character extraction deleted
    // because its correction hit the 32,000-token output ceiling, 522 seconds and a full
    // ceiling of output spent to lose a page the run already had, and the two problems the
    // correction was asked to fix were a transcribed folio and an unwarranted `<section>`
    // (issue #171). It also named the wrong stage: anything reading
    // `page_extraction_failed` โ€” `pages_failed`, the markers, any triage of why pages fail โ€”
    // concluded the vision call could not read the page. It read it fine.
    //
    // Which is the trade PR #151 already makes one layer up for the Copy Editor (a round the
    // editor cannot finish costs that round, not the document) and that this file makes
    // everywhere else: a specialist that fails leaves the page as the general pass wrote it,
    // a fidelity check that cannot run counts as nothing to correct, a sampled recheck that
    // throws is a sample not taken. The correction was the last of those still fatal.
    //
    // Every error, not only a truncation. A throttle, a stall and a ceiling all leave the
    // same thing behind โ€” a page that is good enough to have been worth correcting โ€” and a
    // list of which provider failures are survivable would go stale in exactly the direction
    // that loses pages. Nothing is retried: a correction truncating because the PAGE is large
    // will truncate again, and the retry would buy a second full ceiling of output to prove
    // it (the providers agree โ€” `TruncatedResponseError` is thrown from inside their retry
    // loops precisely so it is not re-billed).
    // What this call may spend on output, bounded by what the first pass spent (#285). Computed
    // here, from `before`, so the number this caller ASKED for is the number the failure line below
    // reports โ€” a cap an operator cannot read off the log is a cap they will debug as a config
    // problem, which is the mistake this whole change is about.
    //
    // Asked for, and not necessarily sent: `growth` has no upper bound, so a merge that grew the
    // page enough can compute a ceiling above `providers.<provider>.max_tokens`, and the adapters
    // take the smaller of the two (`Math.min` in bedrock.ts and openrouter.ts โ€” a caller may lower
    // a call's ceiling and never raise it). A `ceiling` on the line below that is larger than the
    // deployment's is therefore a call the DEPLOYMENT bounded, and the error on that same line says
    // so: the truncation message names the config, because the config is what bound it. Clamping
    // this number to make the two agree is not available here โ€” the provider's ceiling is the
    // adapter's to know, and re-deriving it in the pipeline is how it comes to disagree.
    //
    // `html`, not `innerHtml`, is what `outputTokens` bought: `innerHtml` is what a specialist may
    // have merged into it since, and it is the pair (tokens, the length they produced) that gives
    // this page its own characters-per-token. Handing the same string as both would make `growth`
    // 1 by definition and quietly delete the term.
    //
    // A page that rendered as NOTHING is the one case where the first pass bounds nothing, so it
    // gets no caller ceiling at all. `correctionCeiling` cannot tell that page apart from a first
    // pass whose length is merely unknown โ€” both arrive as `chars: 0`, and for an unknown length
    // scaling by 1 is right โ€” but the caller can, and here the emptiness is a measurement. Such a
    // page is a page declared blank (nothing else empty survives to a correction), and its
    // correction is not an edit of a page: it is a re-render of the page from the image, which is a
    // first pass, and first passes are bounded by the deployment rather than by a caller. Left to
    // the floor it was 4,000 tokens โ€” roughly 16,000 characters of HTML, where this file's own
    // worked example of a dense page is 17,721 โ€” so a dense page wrongly declared blank whose file
    // carries a link annotation would have its one free repair truncated and ship empty. That
    // repair is the safety net the skip above rests on, so capping it at a bound taken from the
    // reply that got the page wrong is the wrong side of #285's own argument (the cap exists to
    // bound a runaway, and this call has nothing to run away from being asked to produce).
    //
    // Since #371 the links check is not the only caller that gets here with an empty page. A blank
    // declaration STATED in the field whose own log names something on the page is delivered and
    // judged, so a failed verdict reaches this call with `trigger: "verify"` (or `both`, where a link
    // is missing too) and the same empty first pass. The exemption is unchanged and so is the reason
    // for it โ€” the reply that measured nothing is still the reply that got the page wrong, and this
    // is still a re-render from the image rather than an edit โ€” but the argument no longer rests on
    // the link repair alone: this is now the repair that decides whether that verify call bought
    // anything, and a cap taken from the empty render would spend it and lose the page anyway.
    const cap = html === "" ? undefined : correctionCeiling({ outputTokens, chars: html.length }, before.length);
    const ceiling = cap?.tokens;
    const attempt = await correctPage(ctx, pageAgent, img, innerHtml, problems, lessons, ceiling).then(
      (reply) => ({ ...reply, error: null as unknown }),
      (error: unknown) => ({ html: null, declined: [] as Declination[], error }),
    );
    const corrected = attempt.html;
    // What the corrector said it was not doing, before anything is decided about the page (#373
    // directive 4). Logged on its own event rather than as a field on `page_corrected` because a
    // decline is per PROBLEM and that line is per page: a reply that declined one of five problems
    // and fixed the other four is the shape this pass is supposed to produce, and folding it into a
    // count on the page's line would make it indistinguishable from a reply that declined all five.
    //
    // It decides nothing. `keep`, `repaired`, the recheck and the `uncorrected` set below are all
    // untouched by this, which is the answer to the risk #373 states against its own directive
    // ("this is also a way to ignore a true problem"): a problem declined wrongly leaves the page in
    // exactly the state a correction that failed to fix it leaves it โ€” named in `uncorrected`, with a
    // `@page-uncorrected` marker on the document โ€” so the licence can buy a different EXPLANATION and
    // never silence. What it removes is the edit: today the corrector's only legal move is
    // compliance, so a false claim about the HTML is answered by changing a page that was right.
    if (attempt.declined.length) {
      const counts = { verify: verifyFailed ? verdict.problems.length : 0, links: missing.length, alt: generic.length, ids: duplicated.length, words: contradictions.length };
      ctx.log.event("page_correction_declined", {
        image: img.name,
        page: img.order,
        trigger,
        // Of how many. A count of declines with no denominator cannot be read: 2 declined out of 2
        // problems is a reply that refused the whole correction, and 2 out of 9 is the pass working.
        problems: problems.length,
        declined: attempt.declined.map((d) => ({
          ...(d.problem !== undefined ? { problem: d.problem } : {}),
          // Which of the four raised it, because the licence is only defensible over one of them.
          // `verify` is the Feedback Agent's reading, which is what can be false. `links`, `alt` and
          // `ids` were checked in code against the source file or the parsed fragment, so a decline
          // there is a refusal of a fact โ€” the misuse of this licence, and the reason the field is on
          // the line rather than left to be joined by hand from three other events.
          //
          // `null` where the reply cited no number, or cited one the list has no entry for. Not
          // folded into any bucket: a decline that names nothing has not refused a code-checked
          // fact, and counting it as one would manufacture the very evidence this field exists to
          // collect.
          source: declinedSource(d.problem, counts),
          why: d.why,
        })),
      });
    }
    if (attempt.error !== null) {
      const message = (attempt.error instanceof Error ? attempt.error.message : String(attempt.error))
        .replace(/\s+/g, " ")
        .trim();
      // Deliberately narrower than the `truncated` field below, which is the predicate: an error that
      // arrived having lost its prototype has a message and no `text` to quote. So this line can say
      // `truncated: true` and carry no excerpt, which is the one case where their disagreement is the
      // truth โ€” and it is why `reply_chars: 0`, and not the absence of `reply_head`, is what names the
      // zero-character shape HERE. On `page_extraction_failed` both come from the same `instanceof`.
      const evidence = truncationEvidence(attempt.error);
      ctx.log.event("page_correction_failed", {
        image: img.name,
        page: img.order,
        trigger,
        problems: problems.length,
        // What the verdict said was wrong going in, spelled exactly as `page_corrected` spells it
        // and for the same reason (#182): the count above says how much work the call was given and
        // not what kind. Without it the correction path's FAILURES were the one part of it that
        // could not be grouped by what was asked โ€” #365 grouped 205 successful corrections by
        // whether their kinds include `content_missing` (39 of 64 grew the page's text with it, 19
        // of 141 without) and could not put a single failed one in either group, because the kinds
        // are on the other event. Empty on the `links` and `alt` triggers, where the defect was
        // found by code against the file's own annotations rather than named by the verifier, which
        // is that field's rule here too โ€” a kind on such a line would be a count the verdict never
        // made. Empty is therefore two cases, exactly as it is on `page_corrected`: no verdict at
        // all, or a verdict whose problems named no kind. `page_verify_failed` separates them on the
        // same `image` โ€” it carries `untagged` beside its own `kinds`, and a `verify` trigger here
        // means that line exists.
        kinds: verifyFailed ? verdict.kinds : [],
        error: message,
        // Named rather than left to be read out of the message, because it is the one shape
        // with a configuration remedy (`providers.*.max_tokens`) and the one that says the
        // model wrote an essay where a page was asked for โ€” a 32,000-token correction of a
        // 17,721-character page is not a rewrite of it.
        truncated: isTruncatedResponseError(attempt.error),
        // And the ceiling it was truncated AT, which since #285 is usually this call's own and not
        // the deployment's. `truncated: true` beside a 32,000-token config used to be enough to
        // name the number; with a per-call cap it is not, and the difference decides whether the
        // remedy is a config edit or `correctionCeiling`'s multiple. Absent where the call ran
        // uncapped, which is two causes and not one: the first pass reported no usage, so there was
        // no measurement to cap from, or the page rendered nothing and was delivered blank, whose
        // correction is a re-render rather than an edit (#294; `links` on a page nothing judged, and
        // since #371 `verify` or `both` on a stated blank whose own log contradicted it, which is the
        // one blank page a verdict is bought for). Both leave the deployment's ceiling as the one that
        // bound the call, so
        // the remedy on the line is the same; what differs is whether anything here could have
        // capped it, which is what an operator reading this field is asking.
        //
        // And WHICH of `correctionCeiling`'s two terms produced the number, because the field above
        // cannot say and the answer decides which constant a reader is looking at. `multiple` is
        // twice this page's own first pass (scaled by a specialist's growth, where there was one) and
        // `floor` is `CORRECTION_CEILING_FLOOR`, which binds on a small page whose doubling is under
        // it: one of the three corrections that truncated in `runs-extract100-95ca64c` is a `floor`
        // line โ€” a 1,618-token first pass capped at 4,000 rather than 3,236 โ€” and a triage of those
        // three lines as evidence about the multiple would be counting it for a term that did not
        // bind on it. The alternative was a reader comparing `ceiling` against 4,000 by eye, which is
        // wrong on the page whose multiple lands exactly there (`correctionCeiling`).
        ...(cap !== undefined ? { ceiling: cap.tokens, ceiling_bound: cap.bound } : {}),
        // What the reply reached, and both of its ends โ€” the evidence that decides the question the
        // `ceiling` above only poses (issue #293). A cap this page hit is either a page that
        // genuinely needs more room than its first pass took, in which case the multiple in
        // `correctionCeiling` is too tight, or a model that went on rewriting the same page, in
        // which case the multiple is doing its job. Nothing on this line could tell those apart:
        // two truncations at 34,573 and 41,959 characters against pages of 11,908 and 11,456 were
        // argued both ways off the same log, and the round could not be asked again to settle it
        // because a truncation has already been billed for a full ceiling of output.
        //
        // A head and a tail settle it by inspection: a tail mid-sentence in content the head has
        // not reached is a page that needed the room, and a tail repeating rows already in the head
        // is a model looping. `reply_chars` rather than `chars` because `chars_kept` is on this same
        // line and a bare `chars` here would read as the page's own length; it is the number
        // `editor_truncated` calls `chars`, and it is a ratio against `chars_kept` for free.
        //
        // Only for a truncation, and only for one that kept its prototype: every other failure โ€”
        // a throttle, a stall, a stream that stopped โ€” has no reply to quote, and `error` above is
        // the whole of what is known about it. Like `editor_truncated`'s excerpts this is the
        // user's own document coming back, so it stays in the run log on the deployment and never
        // reaches `GET /v1/quality`. Its `truncated: true` restates what the predicate above already
        // said on every line where both fire, and the two paths above spell the field the same way.
        ...evidence,
        // Which of the five replies a correction can send this was, on the same terms the first pass
        // reports it (`replyShape`, `page_no_output`): the envelope it was asked for, an envelope cut
        // by the ceiling, the page's bare markup, prose, or `empty` for a reply of whitespace. It is
        // the head-and-tail reading above turned into something countable, which is what #365 asks
        // for โ€” a hand-count of tails is how the shapes were told apart until now, and `reply_chars`
        // cannot do it, since the same 15,000 characters are a large page and a short essay.
        //
        // Absent on a reply of ZERO characters, and that is a narrower rule than "absent on a reply
        // that carried nothing": whitespace lands here as `empty` with a `reply_chars` above 0, so the
        // two are distinguishable, where on a zero-character reply `reply_chars: 0` is already the
        // whole of what is known and a shape would say it twice. Also absent on every failure that has
        // no reply at all, exactly as the excerpts are.
        //
        // It says where the reply BEGAN and not where the output went. `bare_html` on a correction
        // that started the page and then narrated at it reads the same as one that transcribed until
        // the ceiling โ€” both are real, in the same round, on the same model โ€” so the tail above stays
        // the evidence and this is the index into it. `prose` is the one value that settles anything
        // by itself: a reply that never began the page spent the whole cap on something else, and
        // raising the cap buys more of it (#293).
        ...(attempt.error instanceof TruncatedResponseError && attempt.error.text !== ""
          ? { shape: replyShape(attempt.error.text, null) }
          : {}),
        // What the page kept, so the log shows this was a page retained and not a page lost.
        chars_kept: before.length,
      });
    }
    // What the pass changed, measured but NOT used to decide what ships. Whether the
    // fragment is adopted stays on string identity, exactly as it was before any of this:
    // `correctionEffect` observes the text, the descriptions, the attributes and the tag
    // sequence, and a delivery decision must not turn on a signal being complete โ€” a
    // correction whose only change is one this cannot see would be silently reverted, and
    // the page would keep the defect the pass had already fixed. The effect decides the
    // LABEL, which is all the note it answers asked for: a model that re-indents its own
    // page, or writes `&` where it wrote `&amp;`, returns a different string and the same
    // page, and counting that under `results.kept` beside a restored table row is what makes
    // the number unreadable โ€” `text` and `structure` overlap, so the fold cannot subtract it
    // out afterwards.
    const effect = corrected ? correctionEffect(before, corrected) : null;
    const moved = effect !== null && changedAnything(effect);
    // A correction that produced nothing usable, or produced the page it was given back, is
    // a page call paid for and nothing delivered. Recorded because it was previously
    // invisible: the log said a page failed its check and said nothing about what the
    // pass bought, so the loop's value could only be guessed at from call counts (issue
    // #137). See `correctionEffect` for why the kept case reports what it changed.
    //
    // `failed` is kept apart from `empty` because the bill and the remedy are different: an
    // `empty` correction answered and carried no HTML, while a `failed` one never answered โ€”
    // and the expensive case is precisely that one, since a truncation has already paid for a
    // full ceiling of output. Folding them together would hide the most costly correction
    // shape inside the cheapest.
    if (!corrected || corrected === before) {
      ctx.log.event("page_corrected", {
        image: img.name,
        page: img.order,
        trigger,
        problems: problems.length,
        // What the verifier said was wrong going in, so this line pairs with the effect
        // fields on the other `page_corrected` below without a join back to
        // `page_verify_failed` (issue #182). `identical` on a page flagged
        // `content_missing` is the sharpest case there is of a page call that bought
        // nothing that mattered. Empty on the links trigger: a dropped link is found by
        // code against the file's own annotations, not named by the verdict, and giving it
        // a kind would put a count in this field the verifier never made.
        kinds: verifyFailed ? verdict.kinds : [],
        result: corrected ? "identical" : attempt.error !== null ? "failed" : "empty",
      });
    }
    if (corrected && corrected !== before) {
      // A page that PASSED its fidelity check is being re-rendered here only to
      // recover a link, or to replace a placeholder alt, so the rewrite has to earn the
      // standing the original already had: it is verified in turn, and a rewrite that lost
      // something is discarded in favour of the fragment that was known to be good. Both of
      // those repairs are LOCAL โ€” one href, one attribute โ€” and paying for either with the
      // structure of a page that already checked out (a heading level, a `<th scope>`) would
      // make the document worse than it was before the feature. Which is why #290's check
      // lands here rather than as its own pass: a page that passed and has a gutted alt gets
      // this protection for free, from code that was already written for the link case. When
      // the check had already failed, the original has no standing to protect and the
      // correction is accepted as it always was.
      //
      // Before either of those: a correction may change a page and may not delete one. A reply
      // that comes back at a fraction of the size it was given has not corrected the page, and
      // no verdict on it is worth buying โ€” so this is decided first, and it short-circuits both
      // rechecks below (the links one would ask the Feedback Agent to judge a fragment nothing
      // will deliver; the sampled one would spend the batch's single measurement slot on it).
      // See `CORRECTION_SHRINK_FLOOR` for where a quarter comes from and why the guard is worth
      // having even now that util/json.ts reads the right envelope out of the replies that
      // prompted it.
      let keep = !destroyedPage(before, corrected);
      let recheck: VerifyVerdict | null = null;
      if (!keep) {
        ctx.log.event("page_correction_rejected", {
          image: img.name,
          page: img.order,
          trigger,
          reason: "shrank",
          chars_before: before.length,
          chars_after: corrected.length,
        });
      } else if (!verifyFailed) {
        // GUARDED FOR THE SAME REASON AS THE FIRST CHECK, and this call site is not in #364's report:
        // it is the third verify call in the file and the second with nothing to catch a provider
        // error, so a throttle here deleted a page that had rendered, PASSED, and been corrected โ€”
        // two model calls' work thrown away instead of one, by the same route into `failedPage`.
        //
        // What it costs when it fires is a decision rather than a default, and this is the
        // conservative side of it: no verdict is no licence. This recheck exists to stop a correction
        // bought for a link or a placeholder alt from damaging a page that had already passed its
        // fidelity check, so where the verdict cannot be obtained the correction is not kept and the
        // page ships as it was โ€” the status quo, which is a page that passed, and the same answer this
        // branch gives to a verdict that fails. The alternative (keep an unverified rewrite of a page
        // known to be good) is the harm the branch was written to prevent. The correction is billed
        // either way; `correction_discarded` on the event is what says the money bought nothing, so
        // the discard is on the record rather than inferred from a missing rejection line.
        recheck = await verifyAgentOutput(ctx, pageAgent, img, [{ html: corrected, caution }], "recheck_binding").catch(
          (e: unknown) => {
            verifyUnobtainable(ctx, img, "recheck_binding", e, { trigger, correction_discarded: true });
            return null;
          },
        );
        keep = recheck !== null && !failedCheck(recheck);
        if (recheck !== null && !keep) {
          // Named `page_links_correction_rejected` since before there was anything else on this
          // branch, and kept: it is the rejection of a correction bought for a page that had
          // PASSED, and renaming it would split one measurement across two event names in a log
          // that is read across rounds. `trigger` is what says which repair was refused โ€” `links`
          // reads exactly as it always did, and `alt` or `both` is a line no older log holds.
          ctx.log.event("page_links_correction_rejected", {
            image: img.name,
            trigger,
            links: missing.map((l) => l.href),
            // The placeholder alts that bought the call, for the same reason `links` names the
            // hrefs: without them a rejected alt correction is a line saying a good page was
            // re-rendered for nothing and not saying what for.
            ...(generic.length ? { alts: generic } : {}),
            // And the duplicated ids, on the same terms. Omitted rather than empty, so no line an
            // older log holds gains a field, and named rather than left to `trigger`: `both` says
            // more than one source fired and not which, so on a page where a link and an id both
            // fired this is the only field that says the id was part of what the refused call was
            // asked to fix.
            ...(duplicated.length ? { ids: duplicated } : {}),
            // And the words written two ways, on those same terms. A fifth source that `both` cannot
            // name is a fifth way for this line to say a good page was re-rendered for nothing
            // without saying what for โ€” and this one has a reading the others do not: a rejected
            // correction bought for a split word may be a rewrite that lost, or a model that
            // declined and was overruled by the recheck, and the words are where that starts.
            ...(contradictions.length ? { words: contradictions.map((w) => `${w.split} / ${w.joined}`) } : {}),
            problems: recheck.problems,
          });
        }
      } else if (moved && claimRecheck(sampler, img.order)) {
        // Measurement only, on the batch's sampled pages โ€” one by default: does a
        // corrected page pass the check it just failed? A page the pass did not actually
        // change is not worth a slot โ€” there is nothing to check, and the answer would be
        // the verdict already on record. Nothing here decides anything โ€” a verify-driven
        // correction is accepted exactly as it always was, whatever this says โ€” because
        // whether to keep re-rendering until a page passes is a policy question, and the
        // answer to it needs the rate this event exists to produce (issue #137). See
        // `recheckSampler` for how many pages and which, and `DEFAULT_RECHECK_SAMPLE_SIZE`
        // for why the default is not all of them.
        //
        // Two bench rounds later that is the thing being asked about: 200 pages, 8 samples,
        // 2 of them ok, every correction kept regardless, and the note is that a check with no
        // consequence is decorative (issue #166). Three reasons it stays as it is, in the
        // order they bind. The rate itself has since been measured, by replaying this same
        // call over 57 corrected pages in the bench: **26%** of them pass, against 2% for
        // re-asking about the page as it was, 19 pages better and 2 worse, p = 0.000 (issue
        // #288). So the pass does real work and finishes the job on about a quarter of the
        // pages it is bought for โ€” which is an argument about the step's cost, not about
        // this line, and none of the three below turns on the number.
        //
        // What discarding buys. A rejected correction does not restore a good page โ€” it ships
        // the fragment that FAILED this same verifier minutes earlier. On those rounds the
        // verifier rejected 71% and 74% of first renders, so the choice is not a good page
        // against a bad one, it is a page with fewer named problems against a page with more,
        // and `problems_before`/`problems_after` on the line below is the number that says
        // which. The links path is the case where discarding does make sense and it is
        // binding there: those pages had PASSED, so the original has standing to protect.
        //
        // Whose page it would apply to. This is a sample, so at any setting below a census
        // binding it would put a gate on page 4 that page 5 never sees, and the delivered
        // document would differ by which pages the thresholds fell on. Binding it for
        // everyone means a Feedback Agent call per corrected page โ€” 71 of them on a 100-page
        // round, roughly doubling the 24% verification share that is under investigation in
        // the first place. That is now a configured number rather than a compiled one
        // (`defaults.recheck_sample_size`), and raising it still buys measurement only: a
        // deployment can pay for the census without any page's fate depending on it.
        //
        // And how much the sample says. Eight verdicts, of which round 3 supplied 0 ok and
        // round 4 supplied 2, is not a rate yet โ€” and the two 100-page rounds after them made
        // the point again, reading 50% and 25% off four draws each on one corpus. This is a
        // measurement whose whole purpose is to be accumulated across runs before anything is
        // decided on it, and binding it now would spend the pages it was collected to protect.
        //
        // And nothing here can cost a page either. `verifyAgentOutput` is non-blocking
        // for an absent Feedback Agent and an unparseable reply, but a PROVIDER error is
        // rethrown (providers/index.ts logs `model_call ok:false` and throws), so an
        // uncaught throttle on this one extra call would propagate out of extractPage
        // into `failedPage` and ship a `@page-failed` marker for a page that had already
        // rendered, verified and corrected โ€” the corrected fragment sitting in a local
        // variable and thrown away. A measurement that decides nothing must not be able
        // to delete a page of accessible content, so a failed sample is a sample not
        // taken: it is logged, the slot stays spent (a refund would let a throttled
        // provider be retried once per corrected page, which is the cost this bounds),
        // and the page ships exactly as it would have with no measurement at all.
        recheck = await verifyAgentOutput(ctx, pageAgent, img, [{ html: corrected, caution }], "recheck_sampled").catch(
          (e: unknown) => {
            ctx.log.event("page_correction_recheck_failed", {
              image: img.name,
              page: img.order,
              error: (e as Error).message,
            });
            return null;
          },
        );
      }
      if (recheck) {
        // `ok` is "the verifier named no problem", which is also what an unavailable
        // Feedback Agent looks like (see `failedCheck`). On this branch the sampled
        // recheck can only follow a verdict it gave, so the ambiguity is confined to the
        // links path โ€” where with no Feedback Agent every page passes its first check, so
        // every corrected page's recheck is the binding one and every one of them is a
        // "checked and passed" line for a page nobody looked at.
        //
        // `unjudged` is what tells those apart, on the same terms as `page_verify_ok`: the
        // flag rather than a second event, omitted rather than false on a real verdict, and
        // `ok` unchanged either way because the recheck is not allowed to cost the page
        // anything it would not have cost with no measurement at all.
        ctx.log.event("page_correction_recheck", {
          image: img.name,
          page: img.order,
          ok: !failedCheck(recheck),
          ...(recheck.unjudged ? { unjudged: true } : {}),
          problems: recheck.problems,
          // How many problems the page went in with and came out with. `ok` alone made this
          // event unreadable in exactly the way issue #166 reports: four sampled rechecks,
          // four not-ok, and no way to tell a correction that fixed nothing from one that
          // fixed four of five problems and left the fifth. The pass is single-shot, so
          // "fewer" is the outcome it can realistically produce and "none" is not the bar it
          // was built to clear.
          //
          // The FIDELITY problems only, which is not the same as the problems the correction
          // was given: `problems` above is the Feedback Agent's verdict plus one entry per
          // missing link, and the second verdict comes from the same agent judging the
          // fragment against the IMAGE, where a link target does not appear at all (see the
          // comment on `missing`). So a link can be counted going in and cannot be counted
          // coming out, whether the correction re-attached it or not, and a page with one
          // fidelity problem and three missing links would read as four-in-one-out โ€” a
          // correction that fixed nothing the verifier named, logged as converging. Which is
          // the reading this pair exists to remove, so the two sides are made comparable
          // instead: `links_before` carries the other share, `page_links_unrecovered` says
          // whether the links came back, and `page_corrected`'s `problems` is still the
          // correction's whole bill.
          //
          // On the links path that leaves `problems_before: 0` โ€” those pages PASSED their
          // check โ€” and it is the right zero: a binding verdict naming a problem there is a
          // rewrite of a good page that lost something, which is exactly what that check is
          // for, and reading it as "one problem in, one out" hid that.
          problems_before: verifyFailed ? verdict.problems.length : 0,
          links_before: missing.length,
          // The third share of the same bill (#290). Kept apart from `problems_before` for the
          // reason `links_before` is: `problems_after` is the Feedback Agent's verdict on the
          // corrected fragment, and a placeholder alt going in was found by code, so folding it
          // into the before-count would make a page with one fidelity problem and two gutted alts
          // read as three-in-one-out โ€” a correction that fixed nothing, logged as converging.
          // Whether the alts came back is answered exactly and for free by
          // `page_generic_alt_unrecovered`, not by this verdict.
          alt_before: generic.length,
          // The fourth share, for the reason the third exists (#373 directive 3). A duplicate id is
          // found by code and cannot be counted coming out of `problems_after` either โ€” the Feedback
          // Agent judges the fragment against the IMAGE, and an id appears on the page no more than a
          // link target does โ€” so folding it into `problems_before` would make a page with one
          // fidelity problem and two duplicated ids read as three-in-one-out. Whether the ids came
          // back is answered exactly and for free by `page_duplicate_ids_unrecovered`.
          ids_before: duplicated.length,
          // The fifth share (#334 part B), kept apart for the same arithmetic reason: a page with one
          // fidelity problem and three words written two ways must not read as four-in-one-out. The
          // reason it is kept apart is NOT the one above, and the difference is worth writing down โ€”
          // a link target and an id are invisible to a verdict judged against the image, while a
          // visible hyphen is on the page, and #334 found both candidate verifiers raising this
          // family unprompted on four pages. So `problems_after` may legitimately name a split word
          // this field also counts, and the pair is not a clean in/out on this share. What is exact
          // and free is `page_split_words_unrecovered`; read that, not the difference.
          words_before: contradictions.length,
          problems_after: recheck.problems.length,
          // The same two sides as kinds (issue #182), which is what turns "the recheck did
          // not pass" into an answer about the CORRECTION: `content_missing` going in and
          // `alt_quality` coming out is a page whose content came back and whose description
          // is now the complaint, and `content_missing` on both sides is a correction that
          // did not do the one thing it was asked to. Both are `ok: false` and five-in-one-out
          // says nothing about which. Empty before on the links path, for the same reason
          // `problems_before` is 0 there โ€” the page had passed, so nothing was named.
          kinds_before: verifyFailed ? verdict.kinds : [],
          kinds_after: recheck.kinds,
          // Whether this verdict was allowed to change what is delivered. False for the
          // sample, so a consumer cannot read it as the loop having gained a gate.
          binding: !verifyFailed,
        });
      }
      // What the pass actually changed about the page, and whether that change is what
      // the document carries. `correctionEffect` reads both fragments rather than the
      // verdict, so "the alt text was refined" and "a table came back" are separable in
      // a log where both were `page_verify_failed` โ€” which is the measurement issue #137
      // asks for and the one the verdict cannot give about itself.
      //
      // `kept` is reserved for a correction that changed something, so a fragment adopted
      // because it differs as a string while being the same page is `identical` here: the
      // page call was paid for and bought nothing, whichever of the two strings ships.
      ctx.log.event("page_corrected", {
        image: img.name,
        page: img.order,
        trigger,
        problems: problems.length,
        // The verdict's side of the same line: what was wrong going in, beside what the
        // correction changed. That pair is the reading issue #182 asks for โ€” a page flagged
        // `content_missing` whose only effect is `alt_changed` did not get fixed, and until
        // both were on one line neither field could say it alone.
        kinds: verifyFailed ? verdict.kinds : [],
        result: keep ? (moved ? "kept" : "identical") : "rejected",
        ...effect,
      });
      if (keep) {
        innerHtml = corrected;
        // `moved`, not `keep`: a reply adopted because it differs as a STRING while being the same
        // page is the fifth way a correction buys nothing, and it is the one that looks like a
        // repair. The line above labels it `identical` for exactly that reason โ€” a page re-indented,
        // or `&` written where `&amp;` was, is a different string and the same page, and the problems
        // the verifier named are all still in it. Clearing the marker on it would deliver a page that
        // never passed a check, and never got a change, saying nothing (#328, review of #387).
        //
        // Which makes the rule one a reader can check against the log rather than two: a
        // verify-triggered page is in `uncorrectedPages` exactly when its `page_corrected` `result`
        // is not `kept`. The four other values are the four this function can produce for it.
        //
        // Reading `correctionEffect` here is not the thing the comment above it forbids. What must
        // not turn on that signal being complete is which fragment SHIPS โ€” an effect it cannot see
        // would silently revert a real repair โ€” and that decision is `keep`, untouched. This is a
        // declaration about the page, where an incomplete signal errs the safe way: a correction
        // whose only change is one `correctionEffect` cannot see is marked as having repaired
        // nothing, which over-declares, and a silent gap is what every marker in this pipeline
        // exists to prevent.
        repaired = moved;
        logNote = logNote
          ? `${logNote}; self-corrected after fidelity check`
          : "self-corrected after fidelity check";
        // Whether the correction actually re-attached them is worth recording: the pass
        // is single-shot, so a link still missing here is missing from the delivered
        // document, and that is the whole failure this feature has to be able to see.
        const stillMissing = missingLinks(img.links, innerHtml);
        if (stillMissing.length) {
          ctx.log.event("page_links_unrecovered", { image: img.name, links: stillMissing.map((l) => l.href) });
        }
        // The same question about the placeholder alts, and the reason this rule is worth having
        // over a model that finds the same defect: the check that raised the complaint can be run
        // again on the answer, exactly and for nothing. So "the free rule found something" and
        // "the free rule got it fixed" are separable, which is the pair a decision about the
        // verifier's model actually needs (#290, #246). Only when the correction was bought for an
        // alt in the first place โ€” re-reporting a page whose alts were never a complaint would put
        // this rule's failures and the page agent's ordinary output in one count.
        if (generic.length) {
          const stillGeneric = genericAlts(innerHtml);
          if (stillGeneric.length) {
            ctx.log.event("page_generic_alt_unrecovered", {
              image: img.name,
              page: img.order,
              alts: stillGeneric,
            });
          }
        }
        // And the same question about the ids, which this rule owes more than the other two do: it
        // is the one whose defect has a downstream reporter that CANNOT fix it โ€” lint sees the
        // assembled document, after this page's last chance to be read against its image, and where
        // a prefixed id no longer carries the name the page gave it (only where the rename ran; see
        // the qualification at `page_duplicate_ids` above). Free and exact, on the fragment that is
        // actually delivered.
        //
        // `ids` and not a count, because unlike an unrecovered link this is not identity-matched
        // against anything the page arrived with: the correction renumbers, so it can clear `fn-1`
        // and collide on `fn-2`, and that page is a repair that moved the defect rather than one
        // that failed to touch it. Only the names on the line can tell those apart.
        if (duplicated.length) {
          const stillDuplicated = duplicateIds(innerHtml);
          if (stillDuplicated.length) {
            ctx.log.event("page_duplicate_ids_unrecovered", {
              image: img.name,
              page: img.order,
              ids: stillDuplicated,
            });
          }
        }
        // And the same question about the words written two ways, with one reading this line does NOT
        // support that the three above do. A link still missing, an alt still generic and an id still
        // duplicated are failures; a word still written two ways may be the model declining, which on
        // this check is a legitimate answer โ€” a page that really prints both spellings is one Iris was
        // wrong about. So this line says the contradiction survived the pass and nothing about whose
        // fault that is. `page_correction_declined`, keyed by the same `image`, is where the model's
        // side is, and `declinedSource` puts a cited decline in the `words` band.
        //
        // Recomputed rather than intersected with the list going in, for `page_duplicate_ids_unrecovered`'s
        // reason: a correction that joins `Compos-ite` and breaks `col-lections` in the same reply has
        // moved the defect, not failed to touch it, and only the words on the line separate those.
        if (contradictions.length) {
          const stillSplit = splitWordContradictions(innerHtml);
          if (stillSplit.length) {
            ctx.log.event("page_split_words_unrecovered", {
              image: img.name,
              page: img.order,
              words: stillSplit.map((w) => `${w.split} / ${w.joined}`),
            });
          }
        }
      }
    }
  }

  // Checked last, on the fragment that is actually delivered: a correction pass
  // re-writes the anchors, so an href invented there is the one worth seeing.
  // Logged, not corrected โ€” a visible URL linked to itself is legitimate. See
  // `unexpectedHrefs` for why the list is worth having anyway.
  const unexpected = unexpectedHrefs(img.links, innerHtml);
  if (unexpected.length) {
    ctx.log.event("page_links_unexpected", { image: img.name, hrefs: unexpected });
  }

  return {
    fragment: {
      image: img.name,
      order: img.order,
      agent: pageAgent.file,
      region: "page",
      innerHtml,
      edges: [],
      log: logNote,
    },
    suggestion:
      suggestion?.name && !dispatched
        ? { name: suggestion.name, reason: suggestion.reason, image: img.name }
        : undefined,
    // Gated on `verifyFailed` and not on `problems.length`: a correction bought by the links
    // or the alt rule ran on a page that PASSED its check, so a failure there leaves a page
    // with a dropped href or a placeholder description and no verdict against it. Both of
    // those are already reported by name (`page_links_unrecovered`, `page_generic_alt`), and
    // a marker saying this page did not pass verification would be false about it.
    ...(verifyFailed && !repaired ? ({ uncorrected: true } as const) : {}),
  };
}

// The id rule's two document-level counts, for `extraction_complete` and `reextract_complete`.
// One function rather than the expression twice, so the two events cannot drift into measuring
// different things while a comment goes on claiming they are comparable (review of #387).
//
// Per fragment and summed, which is the only count this rule can make: two pages sharing an id is
// not a defect at this point โ€” `namespaceAnchors` fixes it at assembly โ€” so pooling a document's
// ids into one set would report that fix as fifty failures.
function idCounts(fragments: { innerHtml: string }[]): { ids_checked: number; ids_duplicated: number } {
  let ids_checked = 0;
  let ids_duplicated = 0;
  for (const f of fragments) {
    const audit = idAudit(f.innerHtml);
    ids_checked += audit.ids;
    ids_duplicated += audit.duplicated.length;
  }
  return { ids_checked, ids_duplicated };
}

// And the split-word rule's pair (#334 part B), one function for `idCounts`' reason. Per fragment
// and summed for a different reason worth stating rather than borrowing: a word written one way on
// page 3 and the other way on page 40 is not this defect at all โ€” a printing breaks a word at the
// column it happens to fall in, and two pages disagreeing is ordinary โ€” so pooling a document's
// words would manufacture contradictions out of the whole corpus's vocabulary.
//
// `words_checked` is the denominator, and it is what tells a run whose pages carry prose from one
// whose fragments came back empty. Unlike the alt and id rules, `words_split` is EXPECTED to be
// non-zero: on #334's census every arm contradicted itself somewhere, so a run reporting 0 of 20,000
// is the reading to be suspicious of rather than reassured by.
function splitWordCounts(fragments: { innerHtml: string }[]): { words_checked: number; words_split: number } {
  let words_checked = 0;
  let words_split = 0;
  for (const f of fragments) {
    const audit = splitWordAudit(f.innerHtml);
    words_checked += audit.words;
    words_split += audit.split.length;
  }
  return { words_checked, words_split };
}

// One fragment per page, in submitted order. Each page is verified for source
// fidelity at build time; a page that fails gets one self-
// correction pass. Verification is non-blocking โ€” a run never fails because the
// Feedback Agent is unavailable or unsure. When a page flags a content type that an
// existing library agent handles, that specialist is dispatched and merged in;
// otherwise the suggestion is collected for the contribution step.
//
// Pages are extracted CONCURRENTLY (defaults.extraction_concurrency), which is
// the dominant latency term for a multi-page document: each page costs up to
// several sequential model calls, and pages are fully independent. Document order
// is preserved by mapWithConcurrency returning results in input order โ€” never
// rely on completion order here.
export async function runExtraction(ctx: PipelineContext): Promise<ExtractionResult> {
  const pageAgent = loadPageAgent(ctx);
  // Inject corroborated lessons learned from past feedback into the page agent
  // prompt (#1), so it improves without rewriting agents/page.md.
  const lessons = examplesForPrompt(ctx.paths, pageAgent.file);
  if (lessons) ctx.log.event("page_lessons_injected", { chars: lessons.length });

  const limit = ctx.extractionConcurrency;

  // Contained per page: mapWithConcurrency rejects with the first error any item
  // throws (matching a serial loop), so without this one page takes the document with
  // it. See `failedPage`.
  // The batch's measurement-only re-verifications โ€” `defaults.recheck_sample_size` of
  // them, one by default, claimable by a corrected page that reaches the next threshold
  // (correction.ts). Created here rather than inside extractPage so it cannot become one
  // per page, which is the cost it exists to bound.
  const sampler = recheckSampler(
    ctx.images.map((i) => i.order),
    ctx.recheckSampleSize,
  );
  // Logged with the sampler rather than before it, so a round says what it was going to
  // measure and where โ€” `recheck_sample_size` is the setting, `recheck_thresholds` the
  // page orders it resolved to. Without them a log with no `page_correction_recheck` in it
  // reads three ways at once: measurement off, no page corrected, or every correction
  // landing below the first threshold. Only the last is a sample that was available and
  // went unspent, and it is the one that changes how the fleet's counts should be read.
  ctx.log.event("extraction_start", {
    pages: ctx.images.length,
    concurrency: limit,
    recheck_sample_size: ctx.recheckSampleSize,
    recheck_thresholds: [...sampler.thresholds],
  });
  const outcomes = await mapWithConcurrency(ctx.images, limit, (img) =>
    extractPage(ctx, pageAgent, img, lessons, sampler).catch((e) => failedPage(ctx, pageAgent, img, e)),
  );

  // Results come back in input order, so fragments are already in page order.
  const fragments = outcomes.map((o) => o.fragment);
  const suggestions = outcomes
    .map((o) => o.suggestion)
    .filter((s): s is NonNullable<typeof s> => s !== undefined);
  const failedPages = outcomes.filter((o) => o.failed).map((o) => o.fragment.order);
  const uncorrectedPages = outcomes.filter((o) => o.uncorrected).map((o) => o.fragment.order);
  // Always logged, including the zero case, so "no page failed" and "this run predates
  // per-page containment" are not the same observation in a log.
  ctx.log.event("extraction_complete", {
    pages: fragments.length,
    failed: failedPages,
    // The document-level roll-up of what `page_correction_failed` and `page_corrected` already
    // say a line at a time, and it is the roll-up that makes the set countable without joining
    // two events per page and knowing which `result` values mean the page was not replaced.
    // Always present, including empty, for the same reason `failed` is: a field that only ever
    // appears when it fires cannot tell "every rejected page was repaired" from "this run
    // predates the count" (#328).
    uncorrected: uncorrectedPages,
    // The generic-alt rule over the fragments the document is assembled FROM, which is a different
    // question from the per-page `page_generic_alt` above: this one is asked after any correction,
    // so a non-zero `alts_generic` here is a placeholder this step could not repair.
    //
    // Deliberately not read as what shipped, and the distinction is not pedantic: the review loop
    // runs after this line and rewrites the assembled document a top-level block at a time
    // (`applyBlockEdits`), replacing a block's markup wholesale โ€” `<img>` and its `alt` with it โ€”
    // so a copy-edit round that guts an alt ships a placeholder these counts never saw. The
    // delivered bytes are measured where every other claim about them is, on the file the caller
    // receives (`delivered_alt`, orchestrator.ts).
    //
    // Present at zero on every run for the reason `failed` is โ€” a class that is only ever reported when it
    // fires cannot distinguish "it never happened" from "the check never ran", and this rule's
    // whole claim is that it fires on nothing Iris writes (0 of 1,064 alts in the bench corpus,
    // #290). A count that prints 0 is the thing that can be seen to be working.
    //
    // `alts_checked` is the denominator, and it is the number that makes the zero readable: a run
    // whose pages hold no images at all reports 0 of 0, which says nothing about the rule, and a
    // run reporting 0 of 40 says something. Free either way โ€” one regex over strings already in
    // memory.
    alts_checked: fragments.reduce((n, f) => n + altTexts(f.innerHtml).length, 0),
    alts_generic: fragments.reduce((n, f) => n + genericAlts(f.innerHtml).length, 0),
    // The same pair for the id rule (#373 directive 3), on the same terms and for the same reason:
    // asked after any correction, so a non-zero `ids_duplicated` is a duplicate this step could not
    // repair, and present at zero on every run because a rule that fires on almost nothing is only
    // visible as a zero that prints. It fires on less than the alt rule does โ€” 2 of 1,501 page
    // replies across every round on disk, and 0 of 328 on the model deployed today โ€” so a field
    // appearing only when it fires would leave a reader unable to tell "no page duplicated an id"
    // from "this run predates the check".
    //
    // Counted per FRAGMENT and summed, which is the only count this rule can make: a collision
    // between two pages is not a defect here, `namespaceAnchors` fixes it at assembly, and pooling
    // the ids of a 50-page document into one set would report that fix as a failure 50 times over.
    //
    // Not what shipped, exactly as `alts_generic` is not: the review loop rewrites blocks after this
    // line, so it can introduce a duplicate these counts never saw. Lint on the assembled document
    // is what sees that one (`duplicate-id`, enabled by name in lint.ts because the WCAG 2.2 tag
    // filter drops it).
    ...idCounts(fragments),
    // And the split-word rule's pair (#334 part B). Asked after any correction, like the two above,
    // so a non-zero `words_split` is a contradiction this step did not settle โ€” which here may be a
    // page that legitimately prints both spellings and said so, not only a repair that failed. Also
    // not what shipped, for the same reason: the review loop rewrites blocks after this line.
    ...splitWordCounts(fragments),
  });

  // Nothing was extracted. Containment trades a thrown run for the pages that DID
  // work, and with none of them there is nothing to trade: assembly and the review
  // loop would run happily on a body of failure markers (the Reader and Editor are
  // text calls, so whatever killed the page images need not touch them), and the
  // session would end `ready_for_review` serving a document containing none of the
  // source's words. That is worse than the failure it replaced, which at least named
  // the ceiling and the knob to raise (test/e2e.sh ยง9d).
  //
  // The FIRST page's error, unwrapped, because it is the diagnosis: a message written
  // here would say "every page failed" and drop the provider's account of why. The
  // remaining pages' errors are already in the log, one event each.
  //
  // A page reported BLANK produces no content either, and the test is on the content rather
  // than on `failedPages` for that reason: a source whose every page is empty โ€” one blank scan
  // uploaded alone, a rasterization that yielded white pages โ€” would otherwise walk past this
  // guard and be delivered as `<main>\n\n</main>`, a document of no words that says nothing
  // about why, with no marker and no notice, from a run reporting success. Failing names it: the
  // message says how many pages were blank, which is a statement about the source and answers
  // the question an empty file would leave.
  const produced = outcomes.filter((o) => !o.failed && o.fragment.innerHtml.trim().length > 0);
  if (outcomes.length > 0 && produced.length === 0) {
    const blank = outcomes.length - failedPages.length;
    ctx.log.event("extraction_failed", {
      pages: failedPages.length,
      blank,
      reason: "no page produced content",
    });
    // The `??` is unreachable โ€” `failed` is only ever set alongside `error` โ€” but a
    // thrown `undefined` would reach the operator as the string "undefined", which is
    // the one outcome this branch exists to prevent.
    if (blank === 0) throw outcomes[0].error ?? new Error("extraction failed for every page");
    throw new Error(
      `no page produced any content: ${blank} of ${outcomes.length} source pages were reported blank` +
        (failedPages.length ? ` and ${failedPages.length} could not be extracted` : ""),
    );
  }

  writeFileSync(
    join(ctx.paths.sessionFragments(ctx.sessionId), "fragments.json"),
    JSON.stringify(fragments, null, 2),
  );
  return { fragments, suggestions, failedPages, uncorrectedPages };
}

// Re-extract only the pages a piece of feedback actually concerns,
// leaving every other page's prior fragment untouched.
//
// This is the path for feedback the review loop structurally cannot serve: the
// Reader only ever sees the assembled HTML (by design), so a misreading of
// the source raises no issue and the loop has nothing to act on. "You misread the
// table on page 3" can only be fixed by putting page 3's IMAGE back in front of
// the page agent. Each targeted page goes through the same
// render -> verify -> correct path as a first run, with its previous output shown
// so untouched content carries over.
//
// Returns fragments for the WHOLE document in page order โ€” re-extracted pages
// replaced, the rest as they were.
export async function reExtractPages(
  ctx: PipelineContext,
  priorFragments: Fragment[],
  pages: number[],
  // Pages the document being refined has no content for, from the run that lost them.
  // Passed in because this function is the only thing that can shrink that set: a page
  // whose fragment is a failure marker still HAS a fragment, so it is re-extractable, and
  // a round that succeeds on it fills the hole. Anything else about the set is unchanged
  // by this path.
  priorFailedPages: number[] = [],
  // And the pages the verifier rejected in the round that produced the document being
  // refined, for the same reason: this path is the only thing that can shrink THAT set too,
  // since re-rendering a page from its image is the only way a page whose correction failed
  // gets a second answer. Every page it does not re-run keeps the status it arrived with โ€”
  // nothing else here looks at those pages, so nothing else can have changed them.
  priorUncorrectedPages: number[] = [],
): Promise<ExtractionResult> {
  const targets = new Set(pages);
  const pageAgent = loadPageAgent(ctx);
  const lessons = examplesForPrompt(ctx.paths, pageAgent.file);
  if (lessons) ctx.log.event("page_lessons_injected", { chars: lessons.length });

  // Only re-extract a targeted page we still have BOTH the source image and a
  // prior fragment for.
  const priorByOrder = new Map(priorFragments.map((f) => [f.order, f]));
  const toRun = ctx.images.filter((img) => targets.has(img.order) && priorByOrder.has(img.order));
  const missing = [...targets].filter((p) => !toRun.some((img) => img.order === p));
  if (missing.length) ctx.log.event("reextract_skipped", { pages: missing, reason: "no source image or prior fragment" });

  // A page with no content has nothing worth showing the agent as "your previous
  // output": its fragment is the failure comment, and handing that back invites the
  // model to treat a note about a truncated response as prose to preserve โ€” on the one
  // round whose whole job is to produce the page from scratch. So this page starts clean.
  const stillFailed = new Set(priorFailedPages);
  const previousFor = (order: number): string | undefined =>
    stillFailed.has(order) ? undefined : priorByOrder.get(order)?.innerHtml;

  // Contained per page as in runExtraction, but degrading to the PRIOR fragment rather
  // than to a failure marker: this path only runs for pages that already have one, and
  // a re-extraction that throws is a page Iris could not improve, not a page it lost.
  // Replacing good prior content with a marker would make a feedback round destructive.
  // A feedback round gets its own sample, for the same reason the first pass does: these
  // pages are corrected too, and a round that re-extracts three pages is as much a place
  // for the rate to come from as a full run. Its thresholds are spread over the pages
  // being RE-EXTRACTED, which is why the sampler is given their orders rather than a
  // count โ€” see `recheckSampler`.
  const sampler = recheckSampler(
    toRun.map((i) => i.order),
    ctx.recheckSampleSize,
  );
  // After the sampler, and carrying the same two fields as `extraction_start`, for the
  // same reason: a feedback round's sample is drawn from the pages it re-extracts, so its
  // thresholds are different numbers from the first pass's and are only readable here.
  ctx.log.event("reextract_start", {
    pages: toRun.map((i) => i.order),
    of: priorFragments.length,
    concurrency: ctx.extractionConcurrency,
    recheck_sample_size: ctx.recheckSampleSize,
    recheck_thresholds: [...sampler.thresholds],
  });
  const outcomes = await mapWithConcurrency(toRun, ctx.extractionConcurrency, (img) =>
    extractPage(ctx, pageAgent, img, lessons, sampler, previousFor(img.order)).catch(
      (e): PageOutcome => {
        const message = (e instanceof Error ? e.message : String(e)).replace(/\s+/g, " ").trim();
        ctx.log.event("page_extraction_failed", {
          image: img.name,
          page: img.order,
          error: message,
          kept: "prior",
          // The same evidence as a first pass that truncated, on the round a USER asked for. Left off
          // here at first, and that was the wrong half to leave: this round is the one someone is
          // waiting on an answer about, and a reader following docs/API.md's row would have read the
          // absence of these fields as "not a truncation" (#293, review of #297).
          ...truncationEvidence(e),
        });
        return { fragment: priorByOrder.get(img.order)!, failed: true };
      },
    ),
  );

  const replaced = new Map(outcomes.map((o) => [o.fragment.order, o.fragment]));
  const fragments = [...priorFragments]
    .sort((a, b) => a.order - b.order)
    .map((f) => replaced.get(f.order) ?? f);
  const suggestions = outcomes
    .map((o) => o.suggestion)
    .filter((s): s is NonNullable<typeof s> => s !== undefined);
  // Pages left as they were because their re-extraction threw. NOT reported as
  // `failedPages`: that field means the document has no content for the page, and these
  // pages have their prior content โ€” the document is whole, it is just not improved.
  // Conflating the two tells a client following docs/API.md "Partial documents" that it
  // received a partial document when it did not.
  const keptPrior = outcomes.filter((o) => o.failed).map((o) => o.fragment.order);
  // The pages this round got a fresh answer for: every page whose re-extraction ran without
  // throwing. Both sets below are keyed on it, for two different questions that have the same
  // answer here โ€” whether the page has content now, and whether a verdict was taken on it again โ€”
  // and it is computed once rather than twice so the two cannot drift apart while a comment goes
  // on claiming they agree (review of #387).
  const answeredAgain = new Set(outcomes.filter((o) => !o.failed).map((o) => o.fragment.order));
  // A page that WAS missing and re-extracted cleanly is no longer missing. One that was
  // missing and threw again keeps its marker, so it stays in the set.
  const failedPages = priorFailedPages.filter((p) => !answeredAgain.has(p));
  const recovered = priorFailedPages.filter((p) => answeredAgain.has(p));
  // The rejected-and-unrepaired set, carried forward the same way: a page that ran again was
  // judged again, so its new outcome is the whole of what is known about it, and a page that
  // threw keeps its prior fragment and so keeps the prior verdict with it.
  //
  // A page re-run and rejected again appears only once: the first list drops every page the
  // round answered for, so the second list is where all of them come from.
  const uncorrectedPages = [
    ...priorUncorrectedPages.filter((p) => !answeredAgain.has(p)),
    ...outcomes.filter((o) => !o.failed && o.uncorrected).map((o) => o.fragment.order),
  ].sort((a, b) => a - b);

  writeFileSync(
    join(ctx.paths.sessionFragments(ctx.sessionId), "fragments.json"),
    JSON.stringify(fragments, null, 2),
  );
  // `pages` is what was actually re-extracted, so a page that threw is not counted
  // among them โ€” its entry in `replaced` is its own prior fragment, which is the
  // opposite of a page this run produced.
  ctx.log.event("reextract_complete", {
    pages: outcomes.filter((o) => !o.failed).map((o) => o.fragment.order).sort((a, b) => a - b),
    ...(keptPrior.length ? { failed: keptPrior } : {}),
    // Over the WHOLE document this round delivers, not the pages it re-ran, which is what makes
    // it comparable with the same two fields on `extraction_complete`: `fragments` here is the
    // prior round's pages with the re-extracted ones substituted in, so a feedback round reports
    // the same denominator as the first round did rather than a count of the subset it touched.
    // Without that a session's log would read as the alt corpus shrinking every time a client
    // sends feedback (#290).
    alts_checked: fragments.reduce((n, f) => n + altTexts(f.innerHtml).length, 0),
    alts_generic: fragments.reduce((n, f) => n + genericAlts(f.innerHtml).length, 0),
    // And the id pair, over the whole document for the same reason the alt pair is: a feedback
    // round reporting only the pages it re-ran would read as the id corpus shrinking every time a
    // client sends feedback (#373 directive 3).
    ...idCounts(fragments),
    // And the split-word pair, over the whole document for that same reason (#334 part B).
    ...splitWordCounts(fragments),
    // Over the whole document this round delivers, like the two alt fields above and unlike
    // `pages`: a feedback round that repairs one rejected page out of three should read as two
    // left, not as one page re-run. Always present, including empty, for the same reason it is
    // on `extraction_complete` (#328).
    uncorrected: uncorrectedPages,
  });
  return { fragments, suggestions, failedPages, uncorrectedPages, recovered };
}