-
Notifications
You must be signed in to change notification settings - Fork 2
Expand file tree
/
Copy pathocr.js
More file actions
658 lines (604 loc) · 24.9 KB
/
Copy pathocr.js
File metadata and controls
658 lines (604 loc) · 24.9 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
// Reconstruct camp-name text from outlined glyph artwork.
//
// The 2026 placement map has no live text: every label was converted to
// outlines before export, so the "camp names" layer arrives as ~15k glyph-
// shaped paths. This module clusters those glyphs back into labels, rasterizes
// each label upright, OCRs it with tesseract, snaps the result to the camp
// roster, and emits text items in exactly the shape extract.js produces for a
// year whose PDF still has real text. Everything downstream is unchanged.
//
// Results are cached in <data dir>/ocr-cache.json keyed by glyph geometry, so
// only the first run pays for OCR.
import { createCanvas } from '@napi-rs/canvas';
import { execFile } from 'node:child_process';
import { mkdtempSync, rmSync, writeFileSync, readFileSync } from 'node:fs';
import { tmpdir, cpus } from 'node:os';
import path from 'node:path';
import { cmdsBbox, flattenCmds, isPageRect, CANONICAL } from './layers.js';
// Glyphs join the same run when the gap between their boxes is under this
// fraction of their size. Deliberately tight: it yields runs of a few words
// that are reliably one line of one label, which the merge pass below then
// assembles into whole labels using each run's measured baseline angle.
const JOIN_GAP = 0.45;
// Runs merge into one label when they are parallel to within this angle...
const MERGE_ANG_TOL = 8 * Math.PI / 180;
// ...and either continue the same line across a gap of at most this many glyph
// heights, or stack as the next line within this many glyph heights.
const MERGE_ALONG = 1.6;
const MERGE_STACK = 1.1;
// Glyphs render at roughly this many pixels tall for OCR.
const TARGET_GLYPH_PX = 34;
const RASTER_PAD_PX = 14;
// Cap-height to em ratio, used to turn measured glyph heights into a font size.
const CAP_RATIO = 0.7;
// Minimum similarity to accept a roster name for an OCR string.
const SNAP_THRESHOLD = 0.72;
// Below this mean confidence, and with no roster match, a reading is suspect
// enough to be worth re-reading at 90 degrees.
const RETRY_CONF = 70;
const normalizeName = s => String(s)
.toLowerCase()
.replace(/&/g, 'and')
.replace(/[^a-z0-9]/g, '');
// ---------------------------------------------------------------------------
// Clustering
function boxGap(a, b) {
const dx = Math.max(0, Math.max(a[0] - b[2], b[0] - a[2]));
const dy = Math.max(0, Math.max(a[1] - b[3], b[1] - a[3]));
return Math.hypot(dx, dy);
}
function clusterGlyphs(glyphs) {
const parent = glyphs.map((_, i) => i);
const find = x => { while (parent[x] !== x) x = parent[x] = parent[parent[x]]; return x; };
const union = (a, b) => { a = find(a); b = find(b); if (a !== b) parent[a] = b; };
// Bucket by centre so each glyph only tests its neighbourhood.
const CELL = 30;
const grid = new Map();
const cellKey = (x, y) => `${Math.floor(x / CELL)}:${Math.floor(y / CELL)}`;
glyphs.forEach((g, i) => {
const k = cellKey(g.cx, g.cy);
if (!grid.has(k)) grid.set(k, []);
grid.get(k).push(i);
});
for (let i = 0; i < glyphs.length; i++) {
const gi = glyphs[i];
const gx = Math.floor(gi.cx / CELL), gy = Math.floor(gi.cy / CELL);
for (let dx = -1; dx <= 1; dx++) {
for (let dy = -1; dy <= 1; dy++) {
const arr = grid.get(`${gx + dx}:${gy + dy}`);
if (!arr) continue;
for (const j of arr) {
if (j <= i) continue;
const gj = glyphs[j];
const ref = Math.max(gi.w, gi.h, gj.w, gj.h);
if (boxGap(gi.bbox, gj.bbox) <= JOIN_GAP * ref) union(i, j);
}
}
}
}
const byRoot = new Map();
for (let i = 0; i < glyphs.length; i++) {
const r = find(i);
if (!byRoot.has(r)) byRoot.set(r, []);
byRoot.get(r).push(glyphs[i]);
}
return [...byRoot.values()];
}
// ---------------------------------------------------------------------------
// Grouping runs by the camp they sit in
//
// The map draws each camp's label inside that camp's outline, so the outlines
// are a far better guide to which glyphs belong together than any proximity
// rule: two labels a few units apart across a street are unambiguous here.
// Only the labels of camps too small to hold their own text — the ones with a
// callout leader — fall outside every outline, and those go to mergeRuns().
function pointInRings(rings, x, y) {
let inside = false;
for (const ring of rings) {
for (let i = 0, j = ring.length - 1; i < ring.length; j = i++) {
const [xi, yi] = ring[i], [xj, yj] = ring[j];
if ((yi > y) !== (yj > y) && x < ((xj - xi) * (y - yi)) / (yj - yi) + xi) inside = !inside;
}
}
return inside;
}
function buildOutlineIndex(paths, width, height) {
const polys = [];
for (const p of paths) {
if (p.layer !== CANONICAL.CAMP_OUTLINES) continue;
const bbox = cmdsBbox(p.cmds);
if (!isFinite(bbox[0]) || isPageRect(bbox, width, height)) continue;
const w = bbox[2] - bbox[0], h = bbox[3] - bbox[1];
if (w <= 0 || h <= 0) continue;
polys.push({ bbox, rings: flattenCmds(p.cmds), area: w * h });
}
// Smallest first, so a camp nested inside another wins its own glyphs.
polys.sort((a, b) => a.area - b.area);
const CELL = 128;
const grid = new Map();
polys.forEach((poly, i) => {
for (let gx = Math.floor(poly.bbox[0] / CELL); gx <= Math.floor(poly.bbox[2] / CELL); gx++) {
for (let gy = Math.floor(poly.bbox[1] / CELL); gy <= Math.floor(poly.bbox[3] / CELL); gy++) {
const k = `${gx}:${gy}`;
if (!grid.has(k)) grid.set(k, []);
grid.get(k).push(i);
}
}
});
return {
count: polys.length,
// Index of the smallest camp outline containing (x, y), or -1.
at(x, y) {
const candidates = grid.get(`${Math.floor(x / CELL)}:${Math.floor(y / CELL)}`);
if (!candidates) return -1;
for (const i of candidates) { // already in ascending-area order
const poly = polys[i];
if (x < poly.bbox[0] || x > poly.bbox[2] || y < poly.bbox[1] || y > poly.bbox[3]) continue;
if (pointInRings(poly.rings, x, y)) return i;
}
return -1;
},
};
}
// Assign each run to the camp holding most of its glyphs. Runs that land in no
// camp at all are returned separately for proximity clustering.
function groupRunsByCamp(runs, outlines) {
const byCamp = new Map();
const loose = [];
for (const run of runs) {
const votes = new Map();
for (const g of run.members) {
const i = outlines.at(g.cx, g.cy);
if (i >= 0) votes.set(i, (votes.get(i) || 0) + 1);
}
let bestCamp = -1, bestVotes = 0;
for (const [i, n] of votes) if (n > bestVotes) { bestVotes = n; bestCamp = i; }
if (bestCamp < 0 || bestVotes * 2 < run.members.length) {
loose.push(run);
} else {
if (!byCamp.has(bestCamp)) byCamp.set(bestCamp, []);
byCamp.get(bestCamp).push(run);
}
}
return { byCamp, loose };
}
// Second pass: glue the runs of one label together. Camp labels wrap onto two
// or three lines and put wide spaces between words, both of which the tight
// glyph clustering above splits apart; neighbouring camps, by contrast, sit at
// noticeably different angles and don't line up. So a merge needs matching
// angles plus either same-line continuation or next-line stacking.
function mergeRuns(runs) {
const parent = runs.map((_, i) => i);
const find = x => { while (parent[x] !== x) x = parent[x] = parent[parent[x]]; return x; };
const CELL = 60;
const grid = new Map();
runs.forEach((r, i) => {
const k = `${Math.floor(r.cx / CELL)}:${Math.floor(r.cy / CELL)}`;
if (!grid.has(k)) grid.set(k, []);
grid.get(k).push(i);
});
for (let i = 0; i < runs.length; i++) {
const a = runs[i];
const gx = Math.floor(a.cx / CELL), gy = Math.floor(a.cy / CELL);
for (let dx = -1; dx <= 1; dx++) {
for (let dy = -1; dy <= 1; dy++) {
for (const j of grid.get(`${gx + dx}:${gy + dy}`) || []) {
if (j <= i) continue;
if (runsBelongTogether(a, runs[j])) {
const ra = find(i), rb = find(j);
if (ra !== rb) parent[ra] = rb;
}
}
}
}
}
const byRoot = new Map();
runs.forEach((r, i) => {
const k = find(i);
if (!byRoot.has(k)) byRoot.set(k, []);
byRoot.get(k).push(...r.members);
});
return [...byRoot.values()];
}
function angleDiffMod180(a, b) {
let d = Math.abs(a - b) % Math.PI;
return Math.min(d, Math.PI - d);
}
function runsBelongTogether(a, b) {
const hMax = Math.max(a.h, b.h), hMin = Math.min(a.h, b.h);
if (hMax / Math.max(0.01, hMin) > 1.8) return false;
if (a.reliable && b.reliable && angleDiffMod180(a.angle, b.angle) > MERGE_ANG_TOL) return false;
// Measure in the frame of whichever run gave the more trustworthy angle.
const lead = (a.reliable && a.members.length >= b.members.length) || !b.reliable ? a : b;
const fa = labelFrame(a.members, lead.angle);
const fb = labelFrame(b.members, lead.angle);
const span = (lo1, hi1, lo2, hi2) => ({
overlap: Math.min(hi1, hi2) - Math.max(lo1, lo2),
gap: Math.max(0, Math.max(lo1 - hi2, lo2 - hi1)),
});
const u = span(fa.uMin, fa.uMax, fb.uMin, fb.uMax);
const v = span(fa.vMin, fa.vMax, fb.vMin, fb.vMax);
// Same line, separated by a wide word space.
if (v.overlap >= 0.4 * hMin && u.gap <= MERGE_ALONG * hMax) return true;
// The next line of the same label, sitting directly above or below.
const uLen = Math.min(fa.uMax - fa.uMin, fb.uMax - fb.uMin);
if (v.gap <= MERGE_STACK * hMax && u.overlap >= Math.max(0.3 * uLen, 0.5 * hMax)) return true;
return false;
}
// Baseline direction of a label, mod pi, from the spread of its glyph centres.
function principalAngle(members) {
if (members.length < 2) return 0;
let sx = 0, sy = 0;
for (const g of members) { sx += g.cx; sy += g.cy; }
const mx = sx / members.length, my = sy / members.length;
let xx = 0, xy = 0, yy = 0;
for (const g of members) {
const dx = g.cx - mx, dy = g.cy - my;
xx += dx * dx; xy += dx * dy; yy += dy * dy;
}
return 0.5 * Math.atan2(2 * xy, xx - yy);
}
// ---------------------------------------------------------------------------
// Rasterizing
// Project every point of the cluster into the label frame for angle `phi`.
function labelFrame(members, phi) {
const c = Math.cos(phi), s = Math.sin(phi);
let uMin = Infinity, uMax = -Infinity, vMin = Infinity, vMax = -Infinity;
for (const g of members) {
for (const cmd of g.cmds) {
for (let i = 1; i < cmd.length; i += 2) {
const u = cmd[i] * c + cmd[i + 1] * s;
const v = -cmd[i] * s + cmd[i + 1] * c;
if (u < uMin) uMin = u;
if (u > uMax) uMax = u;
if (v < vMin) vMin = v;
if (v > vMax) vMax = v;
}
}
}
return { c, s, uMin, uMax, vMin, vMax };
}
function renderLabel(members, phi, glyphHeight) {
const f = labelFrame(members, phi);
const scale = Math.max(1, TARGET_GLYPH_PX / Math.max(0.5, glyphHeight));
const W = Math.ceil((f.uMax - f.uMin) * scale) + 2 * RASTER_PAD_PX;
const H = Math.ceil((f.vMax - f.vMin) * scale) + 2 * RASTER_PAD_PX;
if (W < 4 || H < 4 || W > 6000 || H > 6000) return null;
const canvas = createCanvas(W, H);
const ctx = canvas.getContext('2d');
ctx.fillStyle = '#fff';
ctx.fillRect(0, 0, W, H);
ctx.fillStyle = '#000';
// PDF space is y-up and the label frame is (u along the baseline, v up);
// the raster is y-down, hence vMax - v.
const px = (x, y) => [
((x * f.c + y * f.s) - f.uMin) * scale + RASTER_PAD_PX,
(f.vMax - (-x * f.s + y * f.c)) * scale + RASTER_PAD_PX,
];
ctx.beginPath();
for (const g of members) {
let cur = [0, 0];
for (const cmd of g.cmds) {
if (cmd[0] === 'M') {
const p = px(cmd[1], cmd[2]); ctx.moveTo(p[0], p[1]); cur = [cmd[1], cmd[2]];
} else if (cmd[0] === 'L') {
const p = px(cmd[1], cmd[2]); ctx.lineTo(p[0], p[1]); cur = [cmd[1], cmd[2]];
} else if (cmd[0] === 'C') {
const a = px(cmd[1], cmd[2]), b = px(cmd[3], cmd[4]), d = px(cmd[5], cmd[6]);
ctx.bezierCurveTo(a[0], a[1], b[0], b[1], d[0], d[1]); cur = [cmd[5], cmd[6]];
} else if (cmd[0] === 'V') {
// curveTo2: first control point is the current point.
const a = px(cur[0], cur[1]), b = px(cmd[1], cmd[2]), d = px(cmd[3], cmd[4]);
ctx.bezierCurveTo(a[0], a[1], b[0], b[1], d[0], d[1]); cur = [cmd[3], cmd[4]];
} else if (cmd[0] === 'Y') {
// curveTo3: second control point is the endpoint.
const a = px(cmd[1], cmd[2]), d = px(cmd[3], cmd[4]);
ctx.bezierCurveTo(a[0], a[1], d[0], d[1], d[0], d[1]); cur = [cmd[3], cmd[4]];
} else if (cmd[0] === 'Z') {
ctx.closePath();
}
}
}
// Nonzero, not even-odd: a letterform is often several overlapping contours,
// and even-odd knocks white notches out of every overlap. Counters are wound
// the other way round, so they still come out as holes.
ctx.fill();
return canvas.encode('png');
}
// ---------------------------------------------------------------------------
// tesseract
function runTesseract(file) {
return new Promise(resolve => {
execFile('tesseract', [file, 'stdout', '--psm', '6', '-l', 'eng', 'tsv'],
{ maxBuffer: 1 << 22 },
(err, stdout) => resolve(err ? '' : stdout));
});
}
// TSV columns: level page block par line word left top width height conf text
function parseTsv(tsv) {
const words = [];
for (const line of tsv.split('\n')) {
const f = line.split('\t');
if (f.length < 12 || f[0] !== '5') continue;
const conf = Number(f[10]);
const text = f[11].trim();
if (!text || !isFinite(conf) || conf < 0) continue;
words.push({ text, conf });
}
if (!words.length) return { text: '', conf: 0 };
let chars = 0, weighted = 0;
for (const w of words) { chars += w.text.length; weighted += w.conf * w.text.length; }
return { text: words.map(w => w.text).join(' '), conf: weighted / Math.max(1, chars) };
}
async function pooled(items, limit, worker) {
const out = new Array(items.length);
let next = 0;
const runners = Array.from({ length: Math.min(limit, items.length) }, async () => {
for (;;) {
const i = next++;
if (i >= items.length) return;
out[i] = await worker(items[i], i);
}
});
await Promise.all(runners);
return out;
}
// ---------------------------------------------------------------------------
// Roster snapping
// Letters this map's typeface and tesseract argue about. Its 't' in particular
// reads as an 'i' most of the time ("Waffles" -> "Waifles", "Yacht" -> "Yachi"),
// which no amount of rendering resolution fixes. Treat a confusion as a near
// miss rather than a wrong letter, and index names by the collapsed form so the
// right camp still turns up as a candidate.
const CONFUSIONS = ['itl1j', 'o0q', 's5', 'cev', 'g9q', 'b6', 'z2', 'nh', 'uv', 'ft'];
const COLLAPSE = new Map();
for (const group of CONFUSIONS) {
for (const ch of group) if (!COLLAPSE.has(ch)) COLLAPSE.set(ch, group[0]);
}
const collapse = s => s.replace(/./g, ch => COLLAPSE.get(ch) || ch);
const CONFUSION_COST = 0.35;
function substitutionCost(a, b) {
if (a === b) return 0;
return COLLAPSE.get(a) === COLLAPSE.get(b) && COLLAPSE.has(a) ? CONFUSION_COST : 1;
}
function levenshtein(a, b) {
const m = a.length, n = b.length;
if (!m) return n;
if (!n) return m;
let prev = new Array(n + 1);
let cur = new Array(n + 1);
for (let j = 0; j <= n; j++) prev[j] = j;
for (let i = 1; i <= m; i++) {
cur[0] = i;
for (let j = 1; j <= n; j++) {
const cost = substitutionCost(a[i - 1], b[j - 1]);
cur[j] = Math.min(prev[j] + 1, cur[j - 1] + 1, prev[j - 1] + cost);
}
[prev, cur] = [cur, prev];
}
return prev[n];
}
function trigrams(s) {
const padded = ` ${s} `;
const out = [];
for (let i = 0; i + 3 <= padded.length; i++) out.push(padded.slice(i, i + 3));
return out;
}
function buildRosterIndex(names) {
const entries = names.map(name => ({ name, key: normalizeName(name) })).filter(e => e.key);
const index = new Map();
entries.forEach((e, i) => {
for (const t of new Set(trigrams(collapse(e.key)))) {
if (!index.has(t)) index.set(t, []);
index.get(t).push(i);
}
});
return { entries, index };
}
// Trigram prescreen, then edit distance on the best handful of candidates.
function snapToRoster(text, roster) {
const key = normalizeName(text);
if (!key || !roster) return null;
const scores = new Map();
for (const t of new Set(trigrams(collapse(key)))) {
for (const i of roster.index.get(t) || []) scores.set(i, (scores.get(i) || 0) + 1);
}
const candidates = [...scores.entries()].sort((a, b) => b[1] - a[1]).slice(0, 25);
let best = null, bestSim = 0;
for (const [i] of candidates) {
const e = roster.entries[i];
const sim = 1 - levenshtein(key, e.key) / Math.max(key.length, e.key.length);
if (sim > bestSim) { bestSim = sim; best = e.name; }
}
return bestSim >= SNAP_THRESHOLD ? { name: best, similarity: bestSim } : null;
}
// ---------------------------------------------------------------------------
function hashKey(members) {
// Geometry is stable across runs, so the rounded glyph boxes identify a label.
const parts = members
.map(g => g.bbox.map(v => Math.round(v * 10)).join(','))
.sort();
let h = 0x811c9dc5;
const s = parts.join(';');
for (let i = 0; i < s.length; i++) {
h ^= s.charCodeAt(i);
h = Math.imul(h, 0x01000193) >>> 0;
}
return `${members.length}-${h.toString(16)}`;
}
/**
* Turn outlined glyph paths into text items.
*
* @param paths canonical-layer paths (only CANONICAL.CAMP_NAMES is read)
* @param opts.width/height page size, to drop the artboard clip rect
* @param opts.rosterNames camp names from camps.json, for snapping
* @param opts.cachePath where to persist OCR results
* @returns text items in extract.js's shape, flagged `fromGlyphs` so the client
* knows the real lettering is already in the paths and draws that
* instead of our reading of it.
*/
export async function textsFromOutlinedGlyphs(paths, opts) {
const { width, height, rosterNames = [], cachePath } = opts;
const glyphs = [];
for (const p of paths) {
if (p.layer !== CANONICAL.CAMP_NAMES) continue;
const bbox = cmdsBbox(p.cmds);
if (!isFinite(bbox[0]) || isPageRect(bbox, width, height)) continue;
glyphs.push({
bbox,
cmds: p.cmds,
w: bbox[2] - bbox[0],
h: bbox[3] - bbox[1],
cx: (bbox[0] + bbox[2]) / 2,
cy: (bbox[1] + bbox[3]) / 2,
});
}
if (!glyphs.length) return [];
const runs = clusterGlyphs(glyphs).map(makeRun);
const outlines = buildOutlineIndex(paths, width, height);
const { byCamp, loose } = groupRunsByCamp(runs, outlines);
// Sitting in the same camp is strong evidence but not proof: a plot can hold
// a second camp's label, set at its own angle. Run the merge rules inside each
// camp as well, so those stay apart and only genuine continuations join.
const clusters = [
...[...byCamp.values()].flatMap(rs => mergeRuns(rs)),
...mergeRuns(loose),
];
console.log(` OCR: ${glyphs.length} glyph outlines -> ${runs.length} runs -> ` +
`${clusters.length} labels (${byCamp.size} inside a camp outline, ${loose.length} runs loose)`);
const roster = rosterNames.length ? buildRosterIndex(rosterNames) : null;
let cache = {};
try { cache = JSON.parse(readFileSync(cachePath, 'utf8')); } catch { /* first run */ }
const jobs = [];
const labels = clusters.map(members => {
const heights = members.map(g => g.h).sort((a, b) => a - b);
const glyphHeight = heights[Math.floor(heights.length / 2)] || 1;
const theta = principalAngle(members);
const key = hashKey(members);
const label = { members, glyphHeight, theta, key, cached: cache[key] };
if (!label.cached) jobs.push(label);
return label;
});
if (jobs.length) {
const dir = mkdtempSync(path.join(tmpdir(), 'brc-ocr-'));
const t0 = Date.now();
console.log(` OCR: reading ${jobs.length} labels with tesseract (${labels.length - jobs.length} cached)...`);
try {
// Read each label both ways round: the glyph spread gives the baseline
// axis but not which end is the start, and BRC labels point every way.
await ocrPass(jobs, [0, Math.PI], dir, roster);
// Where that produced nothing convincing, the axis itself was probably
// wrong — a short or stacked label whose spread runs across the baseline
// rather than along it. Try the perpendicular before giving up.
const retry = jobs.filter(l => !l.best || (!l.best.snapped && l.best.conf < RETRY_CONF));
if (retry.length) {
console.log(` OCR: retrying ${retry.length} labels on the perpendicular axis`);
await ocrPass(retry, [Math.PI / 2, -Math.PI / 2], dir, roster);
}
for (const label of jobs) {
const best = label.best || { text: '', conf: 0, dAngle: 0 };
cache[label.key] = { text: best.text, conf: best.conf, dAngle: best.dAngle };
label.cached = cache[label.key];
}
writeFileSync(cachePath, JSON.stringify(cache));
console.log(` OCR: done in ${((Date.now() - t0) / 1000).toFixed(1)}s, cached to ${path.basename(cachePath)}`);
} finally {
rmSync(dir, { recursive: true, force: true });
}
} else {
console.log(` OCR: all ${labels.length} labels served from cache`);
}
const texts = [];
let snapped = 0, blank = 0;
for (const label of labels) {
const res = label.cached || { text: '', conf: 0, dAngle: 0 };
const raw = String(res.text || '').replace(/\s+/g, ' ').trim();
if (!raw) { blank++; continue; }
const hit = snapToRoster(raw, roster);
if (hit) snapped++;
texts.push(makeTextItem(label, label.theta + (res.dAngle || 0), hit ? hit.name : raw));
}
console.log(` OCR: ${texts.length} labels read (${snapped} matched a roster camp, ${blank} unreadable)`);
return texts;
}
function makeRun(members) {
const heights = members.map(g => g.h).sort((a, b) => a - b);
let sx = 0, sy = 0;
for (const g of members) { sx += g.cx; sy += g.cy; }
const angle = principalAngle(members);
// An angle is only worth trusting when the run really is a line: several
// glyphs, spread far further along the axis than across it.
let along = 0, across = 0;
if (members.length >= 3) {
const f = labelFrame(members, angle);
along = f.uMax - f.uMin;
across = f.vMax - f.vMin;
}
return {
members,
h: heights[Math.floor(heights.length / 2)] || 1,
cx: sx / members.length,
cy: sy / members.length,
angle,
reliable: members.length >= 3 && along > 1.8 * across,
};
}
// Render and read one batch of labels at the given angle offsets, keeping the
// best reading found for each. A reading that matches a roster camp always beats
// one that doesn't, however confident tesseract felt about it.
async function ocrPass(labels, offsets, dir, roster) {
const variants = [];
for (const [i, label] of labels.entries()) {
for (const [k, dAngle] of offsets.entries()) {
const png = await renderLabel(label.members, label.theta + dAngle, label.glyphHeight);
if (!png) continue;
// Set BRC_OCR_DEBUG_DIR to an existing directory to keep the crops around
// and see exactly what tesseract was shown. Names match ocr-cache.json keys.
const file = path.join(process.env.BRC_OCR_DEBUG_DIR || dir, `${label.key}_${i}_${k}.png`);
writeFileSync(file, png);
variants.push({ label, dAngle, file });
}
}
const results = await pooled(variants, Math.max(2, cpus().length), async v => parseTsv(await runTesseract(v.file)));
for (const [i, v] of variants.entries()) {
const r = results[i];
const reading = {
text: r.text,
conf: r.conf,
dAngle: v.dAngle,
snapped: !!snapToRoster(r.text, roster),
};
const prev = v.label.best;
const better = !prev
|| (reading.snapped && !prev.snapped)
|| (reading.snapped === prev.snapped && reading.conf > prev.conf);
if (better) v.label.best = reading;
}
}
// Build a text item indistinguishable, to the client, from one extract.js reads
// out of a PDF that still has real text.
//
// The client approximates a run's width as 0.5 em per character, so we set the
// advance column from the label's measured width to make that guess exact —
// which puts both the origin (baseline, left edge) and the client's derived
// centre where they really are. The up column carries the true glyph size, so
// size-based tolerances and on-screen rendering stay honest.
function makeTextItem(label, phi, str) {
const f = labelFrame(label.members, phi);
const size = label.glyphHeight / CAP_RATIO;
const advance = Math.max(1e-3, 2 * (f.uMax - f.uMin) / Math.max(1, str.length));
const cu = Math.cos(phi), su = Math.sin(phi);
// Centre of the label in page space.
const midU = (f.uMin + f.uMax) / 2, midV = (f.vMin + f.vMax) / 2;
const centreX = midU * cu - midV * su;
const centreY = midU * su + midV * cu;
const halfW = (f.uMax - f.uMin) / 2;
const ox = centreX - halfW * cu - (-su) * size / 2;
const oy = centreY - halfW * su - (cu) * size / 2;
return {
layer: CANONICAL.CAMP_NAMES,
str,
transform: [advance * cu, advance * su, -size * su, size * cu, ox, oy],
fromGlyphs: true,
};
}