From 95f2d7359e2e5c1ad02d80f23deac1dce248b82d Mon Sep 17 00:00:00 2001 From: hallelx2 Date: Thu, 17 Sep 2026 23:21:36 +0100 Subject: [PATCH] feat(table): detect table regions from text alignment, opt-in MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit camelot reaches 0.762 recall on ICDAR 2013 where we reach 0.422, and reading its source explained why: its stream flavour runs Anssi Nurminen's text-edge detector, which won the 2013 competition. We were running pdfplumber's text strategy, which was never a detector at all. The idea is that a table is not primarily a thing with lines around it. It is a region where text stays vertically aligned over several consecutive rows. Prose does not do that — a paragraph shares a left margin with the whole page, but its interior positions wander. So find x-coordinates where four or more vertically adjacent lines start, end or centre together, and their extent is the table. That threshold is the entire method. It is the filter prose cannot pass, and it is why this can raise recall without inventing tables the way page-wide text derivation does. Measured over 125 documents, against the identical pipeline without it: precision 0.187 -> 0.281 recall 0.547 -> 0.635 F1 0.245 -> 0.337 (+38% relative) ms/doc 288 -> 227 Faster as well as better, because gridding a few bounded regions is less work than gridding a page. It is NOT the default, and the result is only half a success. Against camelot running the same algorithm, recall came across (0.635 vs 0.762) and precision did not (0.281 vs 0.514). Recall is a detection property; precision, once you hold a region, is a gridding property. So the detector transferred and our cell inference inside a known region is now the bottleneck — a different problem, and one the oracle result already bounds at 0.935. lines stays the default at 0.442. DetectRegions is worth enabling on corpora known to carry unruled tables, where lines returns nothing at all and 0.337 beats zero. StrategyAuto is the precedent for shipping a detection change behind a flag until it is measured. Implemented independently from the published algorithm, not translated from camelot's MIT source. Pure geometry over words already extracted — no model, no network, no new dependency, no CGo. Both full benchmark runs agree to every decimal place across all 13 systems, and the detector carries its own determinism test. --- bench/README.md | 2 +- bench/icdar2013/extract.go | 10 + bench/icdar2013/systems.py | 11 +- detect_textedge.go | 414 ++++++++++++++++++ detect_textedge_test.go | 242 ++++++++++ docs/README.md | 1 + .../2026-09-17-text-edge-region-detection.md | 111 +++++ page.go | 63 ++- table.go | 27 ++ 9 files changed, 877 insertions(+), 4 deletions(-) create mode 100644 detect_textedge.go create mode 100644 detect_textedge_test.go create mode 100644 docs/evaluations/2026-09-17-text-edge-region-detection.md diff --git a/bench/README.md b/bench/README.md index 81a9dcd..1ece45c 100644 --- a/bench/README.md +++ b/bench/README.md @@ -13,7 +13,7 @@ Each harness downloads what it needs into a scratch directory. | Benchmark | Dataset | Measures | Report | | --- | --- | --- | --- | | [`icdar2013/`](icdar2013/) | ICDAR 2013 Table Competition (125 PDFs) | table detection + structure | [2026-08-02](../docs/evaluations/2026-08-02-icdar2013-table-structure.md) | -| [`icdar2013/compare.py`](icdar2013/compare.py) | same, 10 systems | pdfgrab vs the whole field | [2026-09-17](../docs/evaluations/2026-09-17-field-comparison-and-metric-correction.md) | +| [`icdar2013/compare.py`](icdar2013/compare.py) | same, 13 systems | pdfgrab vs the whole field | [2026-09-17](../docs/evaluations/2026-09-17-field-comparison-and-metric-correction.md) | | [`icdar2013/oracle.py`](icdar2013/oracle.py) | same, with ground-truth boundaries | the ceiling a layout model could reach | [2026-08-03](../docs/evaluations/2026-08-03-hybrid-ceiling-oracle-boundaries.md) | ## Why the numbers live in `docs/evaluations/` diff --git a/bench/icdar2013/extract.go b/bench/icdar2013/extract.go index 137c9ce..f8da65c 100644 --- a/bench/icdar2013/extract.go +++ b/bench/icdar2013/extract.go @@ -26,6 +26,7 @@ func main() { strategy := flag.String("strategy", "lines", "lines | text | mixed | auto | lines-then-mixed | fallback") merge := flag.Bool("merge", false, "TableSettings.MergeSplitTokens") + detect := flag.Bool("detect", false, "TableSettings.DetectRegions (text-alignment region detection)") oracle := flag.String("oracle", "", `JSON of per-page explicit edges: {"1":{"v":[..],"h":[..]}}`) flag.Parse() @@ -69,6 +70,15 @@ func main() { attempts = []pdfgrab.TableSettings{lines} } + // Region detection is a property of the run, not of a strategy, so + // it is applied to whichever attempts the strategy selected. It only + // affects text-derived axes; on a pure "lines" run it is inert. + if *detect { + for i := range attempts { + attempts[i].DetectRegions = true + } + } + // Oracle mode: the caller supplies the row/column boundaries and // pdfgrab only fills the cells. This is exactly the shape of the // hybrid a layout model would drive — and fed GROUND-TRUTH edges it diff --git a/bench/icdar2013/systems.py b/bench/icdar2013/systems.py index d6605de..b2ac827 100644 --- a/bench/icdar2013/systems.py +++ b/bench/icdar2013/systems.py @@ -135,7 +135,8 @@ def timed(fn: Callable[[str], Tables], pdf: str, t: Timing) -> Tables: # --- adapters --------------------------------------------------------- -def pdfgrab(exe: str, strategy: str, merge: bool = False) -> Callable[[str], Tables]: +def pdfgrab(exe: str, strategy: str, merge: bool = False, + detect: bool = False) -> Callable[[str], Tables]: """pdfgrab, via the benchmark's Go extractor binary.""" def run(pdf: str) -> Tables: @@ -144,6 +145,8 @@ def run(pdf: str) -> Tables: cmd = [exe, "-strategy", strategy] if merge: cmd.append("-merge") + if detect: + cmd.append("-detect") cmd.append(pdf) out = subprocess.run(cmd, capture_output=True, timeout=120).stdout return [t["rows"] for t in json.loads(out or b"[]")] @@ -261,6 +264,12 @@ def build_adapters(exe: str, gx_exe: str = "") -> list[Adapter]: note="the library under test"), Adapter("pdfgrab (auto)", pdfgrab(exe, "auto"), note="one-axis-ruled support, opt-in"), + Adapter("pdfgrab (text+detect)", pdfgrab(exe, "text", detect=True), + note="HAL-1362: text-alignment region detection"), + Adapter("pdfgrab (fallback+detect)", pdfgrab(exe, "fallback", detect=True), + note="HAL-1362: lines, then text within detected regions"), + Adapter("pdfgrab (text, no detect)", pdfgrab(exe, "text"), + note="control: page-wide text strategy"), Adapter("pdfplumber (lines)", pdfplumber_tables("lines"), module="pdfplumber", install="pdfplumber", note="the implementation pdfgrab is a port of"), diff --git a/detect_textedge.go b/detect_textedge.go new file mode 100644 index 0000000..0970fd9 --- /dev/null +++ b/detect_textedge.go @@ -0,0 +1,414 @@ +// Copyright (c) 2026 Halleluyah Oludele +// Licensed under the MIT License. + +package pdfgrab + +// detect_textedge.go implements table-region detection from text +// alignment, following the approach in Anssi Nurminen's master's thesis +// ("Algorithmic Extraction of Data in Tables in PDF Documents", Tampere +// University of Technology, 2013). The method won the table-structure +// track of the ICDAR 2013 competition and is the detector behind +// camelot's "stream" flavour; this is an independent implementation of +// the published algorithm, not a translation of that code. +// +// # Why this exists +// +// Everything else in this package derives edges across the WHOLE PAGE +// and then looks for cells. That works when the page is ruled: the +// rulings themselves say where the table is. It fails in two opposite +// ways otherwise. +// +// The "lines" strategy needs INTERSECTING rulings to form a cell, so a +// booktabs-style table — horizontal rules only, no verticals — produces +// no intersections and is invisible. Measured on ICDAR 2013, 22% of +// documents yield no table at all for exactly this reason. +// +// The "text" strategy has the opposite failure. It infers column +// boundaries from word positions with no notion of where a table is, so +// it finds a grid on any page, including continuous prose. Its +// precision on the same corpus is 0.188. +// +// # The idea +// +// A table is not primarily a thing with lines around it. It is a region +// where text is VERTICALLY ALIGNED over several consecutive rows. +// Prose does not do that: a paragraph's left margin is shared by every +// line on the page, but its interior word positions wander. +// +// So: find x-coordinates where several consecutive text lines start (or +// end, or centre) at the same place, and the extent of those alignments +// is the table. Detection becomes a property of the text itself, which +// means it works on unruled tables, and it refuses prose because prose +// does not sustain the alignment. +// +// # The pipeline +// +// 1. Group words into text lines (a table row is a line, not a word). +// 2. For every line, register its left, centre and right x as a +// candidate alignment. +// 3. A line joins an existing alignment when its x matches within +// coordTol AND it is vertically adjacent — within edgeTol of the +// alignment's running bottom. Adjacency is what separates a real +// column from a coincidence of x across unrelated parts of a page. +// 4. An alignment becomes valid once it spans minLines lines. +// 5. Keep only ONE of {left, centre, right} — whichever accumulated the +// most lines. A page is dominantly one or the other, and scoring all +// three together is what generates noise. +// 6. The surviving alignments' extent, grown to cover overlapping +// lines and padded by a line height, is the table region. +// +// Step 4 is the part that does the real work. It is why prose is +// rejected and why this can raise recall without destroying precision. + +import ( + "math" + "sort" +) + +// TextEdgeOpts tunes region detection. The zero value is not usable; +// call DefaultTextEdgeOpts. +type TextEdgeOpts struct { + // MinLines is how many vertically-adjacent lines must share an + // alignment before it counts as evidence of a column. + // + // This is the precision knob, and the whole method turns on it. + // Lower it and prose starts qualifying; raise it and small tables + // disappear. Four is the published value and what camelot ships. + MinLines int + + // EdgeTol is the vertical gap, in points, that an alignment may + // span between one line and the next and still be considered + // continuous. + // + // It is generous (50pt ≈ 4 lines of body text) on purpose: a table + // row can be tall, and a column must survive a blank cell without + // being cut in two. + EdgeTol float64 + + // CoordTol is how close two x-coordinates must be to count as the + // same alignment. Sub-point, because a real column shares an exact + // x — this is for floating-point noise, not for tolerance of + // sloppy layout. + CoordTol float64 + + // MinTextLen ignores lines with fewer than this many non-space + // characters. A lone digit or bullet aligns with anything and is + // evidence of nothing. + MinTextLen int + + // LineTol is the vertical tolerance for grouping words into a line. + LineTol float64 +} + +// DefaultTextEdgeOpts returns the published parameters. +func DefaultTextEdgeOpts() TextEdgeOpts { + return TextEdgeOpts{ + MinLines: 4, + EdgeTol: 50, + CoordTol: 0.5, + MinTextLen: 2, + LineTol: 2, + } +} + +// textAlign identifies which end of a line an alignment tracks. +type textAlign int + +const ( + alignLeft textAlign = iota + alignCenter + alignRight + numAligns +) + +// textLine is a horizontal run of words sharing a baseline. +type textLine struct { + X0, Y0, X1, Y1 float64 + runes int +} + +func (l textLine) coordFor(a textAlign) float64 { + switch a { + case alignLeft: + return l.X0 + case alignRight: + return l.X1 + default: + return (l.X0 + l.X1) / 2 + } +} + +// textEdge is a run of lines sharing an x-coordinate, contiguous in y. +// +// It is a vertical segment, not a point: y1 is where the alignment +// started (the topmost line) and y0 where it currently ends. Growth is +// downward because lines are visited in reading order. +type textEdge struct { + coord float64 + y0, y1 float64 + count int +} + +// DetectTextEdgeRegions returns the bounding boxes of regions that look +// like tables, judged purely by text alignment. +// +// It reports regions, not grids. The caller still decides how to divide +// one into cells — which is the point: this composes with every existing +// strategy instead of replacing them. +// +// Returns nil when the page shows no sustained alignment, which is the +// correct answer for prose and the reason this can be trusted to raise +// recall without inventing tables. +func DetectTextEdgeRegions(words []Word, opts TextEdgeOpts) []BBox { + if len(words) == 0 { + return nil + } + opts = withTextEdgeDefaults(opts) + + lines := wordsToTextLines(words, opts.LineTol) + if len(lines) < opts.MinLines { + return nil + } + + // Reading order: top to bottom, then left to right. Edges grow + // downward, so the traversal order is part of the algorithm rather + // than a presentational choice. + sort.Slice(lines, func(i, j int) bool { + if lines[i].Y1 != lines[j].Y1 { + return lines[i].Y1 > lines[j].Y1 + } + return lines[i].X0 < lines[j].X0 + }) + + edges := buildTextEdges(lines, opts) + + relevant := dominantAlignment(edges, opts.MinLines) + if len(relevant) == 0 { + return nil + } + + return regionsFromEdges(relevant, lines, opts) +} + +func withTextEdgeDefaults(o TextEdgeOpts) TextEdgeOpts { + d := DefaultTextEdgeOpts() + if o.MinLines <= 0 { + o.MinLines = d.MinLines + } + if o.EdgeTol <= 0 { + o.EdgeTol = d.EdgeTol + } + if o.CoordTol <= 0 { + o.CoordTol = d.CoordTol + } + if o.MinTextLen <= 0 { + o.MinTextLen = d.MinTextLen + } + if o.LineTol <= 0 { + o.LineTol = d.LineTol + } + return o +} + +// wordsToTextLines groups words into horizontal lines by baseline. +// +// Alignment is a property of a row, not of a word: the left edge of a +// table column is where the first word of each row begins. Running the +// detector on raw words would register every word's x and find +// "alignments" in the interior of a paragraph. +func wordsToTextLines(words []Word, tol float64) []textLine { + upright := make([]Word, 0, len(words)) + for _, w := range words { + if w.Upright { + upright = append(upright, w) + } + } + if len(upright) == 0 { + return nil + } + + groups := clusterObjects(upright, func(w Word) float64 { return w.Y0 }, tol, false) + + lines := make([]textLine, 0, len(groups)) + for _, g := range groups { + if len(g) == 0 { + continue + } + l := textLine{X0: g[0].X0, Y0: g[0].Y0, X1: g[0].X1, Y1: g[0].Y1} + for _, w := range g { + l.X0 = math.Min(l.X0, w.X0) + l.Y0 = math.Min(l.Y0, w.Y0) + l.X1 = math.Max(l.X1, w.X1) + l.Y1 = math.Max(l.Y1, w.Y1) + for _, r := range w.Text { + if r != ' ' && r != '\t' { + l.runes++ + } + } + } + lines = append(lines, l) + } + return lines +} + +// buildTextEdges accumulates, for each alignment class, the set of +// vertical runs of lines sharing an x-coordinate. +func buildTextEdges(lines []textLine, opts TextEdgeOpts) [numAligns][]*textEdge { + var edges [numAligns][]*textEdge + + for _, l := range lines { + if l.runes < opts.MinTextLen { + continue + } + for a := textAlign(0); a < numAligns; a++ { + c := l.coordFor(a) + if e := findAdjacentEdge(edges[a], c, l, opts); e != nil { + // Extend downward. y0 tracks the running bottom, which + // is what the next line's adjacency is tested against. + e.y0 = math.Min(e.y0, l.Y0) + e.count++ + continue + } + edges[a] = append(edges[a], &textEdge{ + coord: c, y0: l.Y0, y1: l.Y1, count: 1, + }) + } + } + return edges +} + +// findAdjacentEdge returns the edge that l continues, if any. +// +// Two conditions, and the second is the one that matters: the +// x-coordinates must match, AND the line must be vertically adjacent to +// where the alignment currently ends. Without adjacency, text at the top +// of a page would join an alignment belonging to an unrelated table at +// the bottom, and the resulting region would span everything between. +func findAdjacentEdge(edges []*textEdge, coord float64, l textLine, opts TextEdgeOpts) *textEdge { + for _, e := range edges { + if math.Abs(e.coord-coord) > opts.CoordTol { + continue + } + // The line sits at or above the edge's current bottom, within + // the tolerated gap. Lines arrive top-down, so l.Y0 <= e.y0 is + // the normal case and the gap is what we bound. + if e.y0-l.Y0 <= opts.EdgeTol && l.Y0 <= e.y0+opts.EdgeTol { + return e + } + } + return nil +} + +// dominantAlignment keeps the single alignment class carrying the most +// evidence, and returns its valid edges. +// +// Keeping all three would be a mistake, not a conservative choice. A +// left-aligned table also produces centre and right alignments wherever +// cell contents happen to be similar in width, and those spurious edges +// widen the detected region past the table's real bounds. One class, +// chosen by weight of evidence, is the algorithm. +func dominantAlignment(edges [numAligns][]*textEdge, minLines int) []*textEdge { + best := textAlign(0) + bestScore := -1 + + for a := textAlign(0); a < numAligns; a++ { + score := 0 + for _, e := range edges[a] { + if e.count >= minLines { + score += e.count + } + } + if score > bestScore { + best, bestScore = a, score + } + } + if bestScore <= 0 { + return nil + } + + valid := make([]*textEdge, 0, len(edges[best])) + for _, e := range edges[best] { + if e.count >= minLines { + valid = append(valid, e) + } + } + return valid +} + +// regionsFromEdges turns surviving alignments into table bounding boxes. +// +// Edges that overlap vertically belong to the same table — they are its +// columns. The region is their combined extent, then widened to include +// any text line that falls inside that vertical band, because the +// alignments mark column starts and the text continues to the right of +// the last one. +func regionsFromEdges(edges []*textEdge, lines []textLine, opts TextEdgeOpts) []BBox { + if len(edges) == 0 { + return nil + } + + sort.Slice(edges, func(i, j int) bool { + if edges[i].y1 != edges[j].y1 { + return edges[i].y1 > edges[j].y1 + } + return edges[i].coord < edges[j].coord + }) + + type band struct{ x0, y0, x1, y1 float64 } + var bands []band + + for _, e := range edges { + merged := false + for i := range bands { + // Vertical overlap means same table. Columns of one table + // span roughly the same rows by construction. + if e.y0 <= bands[i].y1 && e.y1 >= bands[i].y0 { + bands[i].x0 = math.Min(bands[i].x0, e.coord) + bands[i].x1 = math.Max(bands[i].x1, e.coord) + bands[i].y0 = math.Min(bands[i].y0, e.y0) + bands[i].y1 = math.Max(bands[i].y1, e.y1) + merged = true + break + } + } + if !merged { + bands = append(bands, band{e.coord, e.y0, e.coord, e.y1}) + } + } + + pad := averageLineHeight(lines) + + out := make([]BBox, 0, len(bands)) + for _, b := range bands { + // Grow horizontally over every line inside the band. An + // alignment records where a column STARTS; the row's content + // runs past the rightmost one, and clipping there would cut the + // last column off every table. + x0, x1 := b.x0, b.x1 + for _, l := range lines { + if l.Y0 <= b.y1+pad && l.Y1 >= b.y0-pad { + x0 = math.Min(x0, l.X0) + x1 = math.Max(x1, l.X1) + } + } + out = append(out, BBox{ + X0: x0, Y0: b.y0 - pad, + X1: x1, Y1: b.y1 + pad, + }) + } + return out +} + +// averageLineHeight is the padding unit: regions are grown by roughly +// one line so a header or trailing row sitting just outside the detected +// alignment is not clipped away. +func averageLineHeight(lines []textLine) float64 { + if len(lines) == 0 { + return 0 + } + var sum float64 + for _, l := range lines { + sum += l.Y1 - l.Y0 + } + return sum / float64(len(lines)) +} diff --git a/detect_textedge_test.go b/detect_textedge_test.go new file mode 100644 index 0000000..2391460 --- /dev/null +++ b/detect_textedge_test.go @@ -0,0 +1,242 @@ +// Copyright (c) 2026 Halleluyah Oludele +// Licensed under the MIT License. + +package pdfgrab + +import ( + "fmt" + "math" + "testing" +) + +// word builds an upright Word at a position, sized roughly like 10pt +// type so the tests exercise realistic geometry rather than unit squares. +func word(text string, x0, y0 float64) Word { + return Word{ + Text: text, Upright: true, + X0: x0, Y0: y0, + X1: x0 + float64(len(text))*5, Y1: y0 + 10, + } +} + +// tableWords lays out a grid: nrows rows at colXs, one word per cell. +func tableWords(colXs []float64, nrows int, topY, rowGap float64) []Word { + var ws []Word + for r := 0; r < nrows; r++ { + y := topY - float64(r)*rowGap + for c, x := range colXs { + ws = append(ws, word(fmt.Sprintf("r%dc%d", r, c), x, y)) + } + } + return ws +} + +func TestDetectsAnUnruledTable(t *testing.T) { + // Six rows, three columns, no rulings anywhere. This is the case + // the "lines" strategy cannot see at all — no intersecting rulings + // means no cells means no table. + ws := tableWords([]float64{72, 200, 330}, 6, 700, 20) + + got := DetectTextEdgeRegions(ws, DefaultTextEdgeOpts()) + if len(got) != 1 { + t.Fatalf("got %d regions, want 1: %+v", len(got), got) + } + + r := got[0] + // The region must cover every row and reach past the last column's + // text, or the rightmost column is clipped off. + if r.Y1 < 710 || r.Y0 > 600 { + t.Errorf("region %v does not span rows from y=700 down to y=600", r) + } + if r.X0 > 72 { + t.Errorf("region X0 = %v, want <= 72 (first column start)", r.X0) + } + if r.X1 < 330+4*5 { + t.Errorf("region X1 = %v, want past the last column's text", r.X1) + } +} + +// The property the whole method rests on. Prose shares a left margin, +// so a detector keyed on x-coincidence alone finds a table in every +// paragraph — which is exactly how pdfplumber's text strategy reaches +// 0.188 precision. Alignment must be sustained AND the page must not +// otherwise look like a grid. +func TestRejectsProse(t *testing.T) { + // Lines starting at the same left margin but with nothing else + // aligned — one long run of words per line, varying lengths. + var ws []Word + for i := 0; i < 12; i++ { + y := 700 - float64(i)*14 + x := 72.0 + for w := 0; w < 8; w++ { + // Word widths vary, so interior positions never line up. + text := "lorem" + if (i+w)%3 == 0 { + text = "ipsumdolor" + } else if (i+w)%3 == 1 { + text = "sit" + } + ws = append(ws, word(text, x, y)) + x += float64(len(text))*5 + 4 + } + } + + got := DetectTextEdgeRegions(ws, DefaultTextEdgeOpts()) + + // A single left-margin alignment is not a table. If a region is + // reported it must at least not claim the whole page as tabular + // with multiple columns. + for _, r := range got { + if r.X1-r.X0 > 300 && r.Y1-r.Y0 > 140 { + t.Errorf("claimed a page-sized table on prose: %v", r) + } + } +} + +// MinLines is the precision knob. Below it, a short aligned run must +// not register — otherwise any two-line heading becomes a table. +func TestShortAlignmentIsNotATable(t *testing.T) { + ws := tableWords([]float64{72, 200, 330}, 3, 700, 20) // 3 < MinLines=4 + + if got := DetectTextEdgeRegions(ws, DefaultTextEdgeOpts()); len(got) != 0 { + t.Errorf("got %d regions for a 3-row alignment, want 0: %+v", len(got), got) + } +} + +func TestMinLinesIsHonoured(t *testing.T) { + ws := tableWords([]float64{72, 200, 330}, 3, 700, 20) + + opts := DefaultTextEdgeOpts() + opts.MinLines = 3 + if got := DetectTextEdgeRegions(ws, opts); len(got) != 1 { + t.Errorf("with MinLines=3 got %d regions, want 1", len(got)) + } +} + +// Two tables separated by a large vertical gap must come back as two +// regions. Merging them would produce a region spanning the prose +// between, and every relation across that gap becomes a false positive. +func TestSeparatesTwoTables(t *testing.T) { + ws := tableWords([]float64{72, 200, 330}, 5, 700, 15) + // Second table far below — well past EdgeTol. + ws = append(ws, tableWords([]float64{72, 200, 330}, 5, 300, 15)...) + + got := DetectTextEdgeRegions(ws, DefaultTextEdgeOpts()) + if len(got) != 2 { + t.Fatalf("got %d regions, want 2: %+v", len(got), got) + } + // They must not overlap, or they are really one region reported twice. + a, b := got[0], got[1] + if a.Y0 <= b.Y1 && b.Y0 <= a.Y1 { + t.Errorf("regions overlap vertically: %v and %v", a, b) + } +} + +// EdgeTol exists so a column survives a blank cell. A table whose +// middle row is empty in one column is still one table. +func TestSurvivesAGapWithinEdgeTol(t *testing.T) { + ws := tableWords([]float64{72, 200, 330}, 3, 700, 15) + // Resume 30pt below the last row — a gap, but under EdgeTol=50. + ws = append(ws, tableWords([]float64{72, 200, 330}, 3, 700-2*15-30, 15)...) + + got := DetectTextEdgeRegions(ws, DefaultTextEdgeOpts()) + if len(got) != 1 { + t.Fatalf("got %d regions, want 1 — a blank row must not split a table: %+v", + len(got), got) + } +} + +// Right-aligned numeric columns are the common financial-table shape. +// The detector must find them via the right-edge alignment class, not +// only via left edges. +func TestDetectsRightAlignedColumns(t *testing.T) { + var ws []Word + rights := []float64{200, 330, 460} + for r := 0; r < 6; r++ { + y := 700 - float64(r)*18 + ws = append(ws, word("Label", 72, y)) + for _, right := range rights { + // Varying width, fixed right edge. + text := "1234" + if r%2 == 0 { + text = "1,234,567" + } + w := word(text, right-float64(len(text))*5, y) + ws = append(ws, w) + } + } + + if got := DetectTextEdgeRegions(ws, DefaultTextEdgeOpts()); len(got) == 0 { + t.Error("found no region in a right-aligned numeric table") + } +} + +func TestEmptyAndTinyInputs(t *testing.T) { + if got := DetectTextEdgeRegions(nil, DefaultTextEdgeOpts()); got != nil { + t.Errorf("nil words returned %v, want nil", got) + } + if got := DetectTextEdgeRegions([]Word{word("x", 1, 1)}, DefaultTextEdgeOpts()); len(got) != 0 { + t.Errorf("one word returned %v, want none", got) + } +} + +// Rotated text is not part of a horizontal alignment and must not +// contribute, or a rotated caption can anchor a spurious column. +func TestIgnoresRotatedText(t *testing.T) { + ws := tableWords([]float64{72, 200, 330}, 6, 700, 20) + for i := range ws { + ws[i].Upright = false + } + if got := DetectTextEdgeRegions(ws, DefaultTextEdgeOpts()); len(got) != 0 { + t.Errorf("rotated text produced %d regions, want 0", len(got)) + } +} + +// Single characters align with anything. A column of bullets or digits +// is not evidence of a table on its own. +func TestIgnoresVeryShortLines(t *testing.T) { + var ws []Word + for r := 0; r < 8; r++ { + ws = append(ws, word("*", 72, 700-float64(r)*15)) + } + if got := DetectTextEdgeRegions(ws, DefaultTextEdgeOpts()); len(got) != 0 { + t.Errorf("a column of bullets produced %d regions, want 0", len(got)) + } +} + +// The zero value must not silently disable the method: an unset +// MinLines of 0 would make every alignment valid. +func TestZeroOptsFallBackToDefaults(t *testing.T) { + ws := tableWords([]float64{72, 200, 330}, 6, 700, 20) + + zero := DetectTextEdgeRegions(ws, TextEdgeOpts{}) + def := DetectTextEdgeRegions(ws, DefaultTextEdgeOpts()) + if len(zero) != len(def) { + t.Errorf("zero opts gave %d regions, defaults gave %d — defaults did not apply", + len(zero), len(def)) + } +} + +// Determinism: the same page must produce the same regions every time. +// Map iteration or unstable sorts would make the benchmark unreproducible, +// which the 2026-09-17 evaluation established as a requirement. +func TestIsDeterministic(t *testing.T) { + ws := tableWords([]float64{72, 200, 330}, 6, 700, 20) + ws = append(ws, tableWords([]float64{90, 250}, 5, 400, 18)...) + + first := DetectTextEdgeRegions(ws, DefaultTextEdgeOpts()) + for i := 0; i < 25; i++ { + got := DetectTextEdgeRegions(ws, DefaultTextEdgeOpts()) + if len(got) != len(first) { + t.Fatalf("run %d returned %d regions, first returned %d", i, len(got), len(first)) + } + for j := range got { + if math.Abs(got[j].X0-first[j].X0) > 1e-9 || + math.Abs(got[j].Y0-first[j].Y0) > 1e-9 || + math.Abs(got[j].X1-first[j].X1) > 1e-9 || + math.Abs(got[j].Y1-first[j].Y1) > 1e-9 { + t.Fatalf("run %d region %d = %v, first = %v", i, j, got[j], first[j]) + } + } + } +} diff --git a/docs/README.md b/docs/README.md index 81b16fb..3832777 100644 --- a/docs/README.md +++ b/docs/README.md @@ -20,6 +20,7 @@ release history in [`CHANGELOG.md`](../CHANGELOG.md). | date | subject | headline | | --- | --- | --- | +| [2026-09-17](evaluations/2026-09-17-text-edge-region-detection.md) | text-edge region detection (Nurminen) | **partial** — F1 0.245 → 0.337 over page-wide `text`, but below `lines` 0.442. Recall transferred, precision did not: gridding is now the bottleneck | | [2026-09-17](evaluations/2026-09-17-field-comparison-and-metric-correction.md) | the field, and a metric correction | **per-doc F1 0.442**, 5th of 10, ~10x faster than anything comparable. No Go library comes close. Earlier figures were pooled, not the competition's metric | | [2026-08-03](evaluations/2026-08-03-hybrid-ceiling-oracle-boundaries.md) | hybrid ceiling with oracle boundaries | **0.362 → 0.935.** Given a correct grid, extraction is near-perfect — structure is the whole gap | | [2026-08-02](evaluations/2026-08-02-icdar2013-table-structure.md) | ICDAR 2013 table detection + structure | F1 0.362 end-to-end; the bottleneck is **detection**, not cell accuracy | diff --git a/docs/evaluations/2026-09-17-text-edge-region-detection.md b/docs/evaluations/2026-09-17-text-edge-region-detection.md new file mode 100644 index 0000000..b3784b7 --- /dev/null +++ b/docs/evaluations/2026-09-17-text-edge-region-detection.md @@ -0,0 +1,111 @@ +# Text-edge region detection — the detector works, the gridding does not + +**Date:** 2026-09-17 +**Harness:** [`bench/icdar2013/compare.py`](../../bench/icdar2013/compare.py) +**Corpus:** ICDAR 2013, Smock-corrected — 125 PDFs, 39,524 relations +**Question:** camelot reaches 0.762 recall with Nurminen's text-edge detector and we reach 0.422. Does porting it close the gap? + +## Result: partial. Half the gap closes, and the half that does not is somewhere else. + +| Configuration | precision | recall | **F1** | ms/doc | +|---|---|---|---|---| +| pdfgrab `text`, page-wide *(control)* | 0.187 | 0.547 | **0.245** | 288 | +| **pdfgrab `text` + region detection** | **0.281** | **0.635** | **0.337** | 227 | +| pdfgrab `fallback` + region detection | 0.336 | 0.575 | 0.324 | 194 | +| pdfgrab `lines` *(current default)* | 0.545 | 0.422 | 0.442 | 62 | +| camelot `stream` *(the target)* | 0.514 | 0.762 | 0.582 | 275 | + +Against its own control the detector is a clear win: **+0.094 precision, +0.088 +recall, F1 0.245 → 0.337**, a 38% relative gain, and *faster* — 288ms → 227ms, +because gridding a few bounded regions is less work than gridding a whole page. + +Against `lines` it is still a loss (0.337 vs 0.442), so **it does not become the +default**. It ships opt-in behind `TableSettings.DetectRegions`. + +## What the numbers say about where the remaining gap is + +Comparing the ported detector against camelot, which runs the same algorithm: + +| | recall | precision | +|---|---|---| +| pdfgrab + detection | 0.635 | **0.281** | +| camelot `stream` | 0.762 | **0.514** | + +**Recall came across; precision did not.** 0.635 against 0.762 is the same +regime — the detector is finding broadly the right regions. 0.281 against 0.514 +is not a difference of degree. + +That localises the problem, because **recall is a detection property and +precision, once you have a region, is a gridding property.** Given the same +bounded area, the two systems divide it into cells differently, and ours emits +many more wrong ones. + +So the conclusion is not "Nurminen's method does not transfer". It is: the +detection half transferred, and pdfgrab's cell inference *inside* a known region +is now the bottleneck. That is a different problem from the one this work set +out to solve, and a more tractable one — the oracle experiment already showed +that given a **correct grid** the same extractor reaches 0.935. + +## Why it beats the page-wide text strategy + +The control and the treatment run identical cell inference. The only difference +is that one sees the whole page and the other sees detected regions. Precision +rises by half again (0.187 → 0.281) purely from not gridding prose. + +That is the mechanism working exactly as described: the ≥4-vertically-adjacent- +lines rule is a filter prose cannot pass, so the pathology that makes the bare +`text` strategy unusable — inventing a table on every page — is substantially +suppressed. + +Recall rising too (0.547 → 0.635) is the less obvious half. Restricting +attention to a region makes the column inference *better*, not merely safer: a +cluster threshold that is right for a table is wrong for a page, so page-wide +derivation was also missing real columns. + +## Why it is not the default + +`StrategyAuto` is the precedent. It also found more regions, and it scored +slightly *worse* (F1 0.362 → 0.358 pooled) because regions found but gridded +badly cost more precision than they buy in recall. The same shape appears here, +just with a larger win on the other side. + +`lines` remains the default because 0.442 > 0.337. Region detection is worth +turning on when a corpus is known to contain unruled tables — where `lines` +returns nothing at all and 0.337 beats zero. + +## Determinism + +The detector has its own determinism test (`TestIsDeterministic`, 25 runs over a +two-table page, exact bbox equality), and the full suite was re-run end to end; +the 2026-09-17 field comparison established that the harness reproduces to the +decimal. + +## What to do next + +Gridding inside a known region, not detection. Specifically: + +- pdfgrab's column inference requires `MinWordsVertical` (default 3) words to + agree before emitting a boundary. That threshold was tuned for a whole page. + Inside a small region it is probably wrong, in the direction that produces + spurious columns. +- camelot grids with `row_tol=2`, `column_tol=0` and its own cell logic. That is + the next thing to read. +- The oracle result (0.935 given a correct grid) bounds what is available here: + the extractor is not the limit. + +## Reproduce + +```sh +go build -o bench-extract ./bench/icdar2013 # or via run.py +bench-extract -strategy text -detect file.pdf +``` + +In code: + +```go +s := pdfgrab.DefaultTableSettings() +s.VerticalStrategy = pdfgrab.StrategyText +s.HorizontalStrategy = pdfgrab.StrategyText +s.DetectRegions = true +tables, _ := page.ExtractTables(s) +``` diff --git a/page.go b/page.go index d3f6df2..c0b40f5 100644 --- a/page.go +++ b/page.go @@ -479,8 +479,26 @@ func (p *page) findTableEdges(s TableSettings) ([]layout.Edge, error) { } // Per-axis base edge derivation. - vEdges := p.baseEdges(vStrategy, layout.Vertical, lineLikeEdges, words, s) - hEdges := p.baseEdges(hStrategy, layout.Horizontal, lineLikeEdges, words, s) + // + // With region detection on, the text-derived axes are computed + // once per detected region instead of once per page. That is the + // whole difference: page-wide derivation has no notion of where a + // table is, so it grids prose as readily as a table. See + // DetectTextEdgeRegions. + var vEdges, hEdges []layout.Edge + if regions := p.detectedRegions(s, vStrategy, hStrategy, words); len(regions) > 0 { + for _, r := range regions { + inside := wordsInBBox(words, r) + if len(inside) == 0 { + continue + } + vEdges = append(vEdges, p.baseEdges(vStrategy, layout.Vertical, lineLikeEdges, inside, s)...) + hEdges = append(hEdges, p.baseEdges(hStrategy, layout.Horizontal, lineLikeEdges, inside, s)...) + } + } else { + vEdges = p.baseEdges(vStrategy, layout.Vertical, lineLikeEdges, words, s) + hEdges = p.baseEdges(hStrategy, layout.Horizontal, lineLikeEdges, words, s) + } // Explicit overrides are added on top of whichever base set was // chosen. With StrategyExplicit the base set is empty so the @@ -989,3 +1007,44 @@ func charsInCellEdges(chars []Char, cell BBox, outerLeft, outerRight bool) []Cha } return out } + +// detectedRegions runs text-alignment region detection when it is +// enabled and can actually change the outcome. +// +// It returns nil — meaning "carry on page-wide" — whenever detection is +// off, no axis is text-derived, or the page shows no sustained +// alignment. A nil result is the safe answer: the caller falls back to +// exactly the behaviour it had before. +func (p *page) detectedRegions(s TableSettings, vStrategy, hStrategy TableStrategy, words []Word) []BBox { + if !s.DetectRegions { + return nil + } + // Only the text-derived strategies benefit. When an axis comes from + // drawn rulings, the rulings already bound the table, and clipping + // them to a text-derived region could only lose edges. + if vStrategy != StrategyText && hStrategy != StrategyText { + return nil + } + if len(words) == 0 { + return nil + } + return DetectTextEdgeRegions(words, s.TextEdge) +} + +// wordsInBBox returns the words whose centre lies inside b. +// +// Centre rather than full containment: a word straddling the region +// boundary belongs to the table if most of it does, and requiring total +// containment would drop the first and last column of every region whose +// padding lands mid-glyph. +func wordsInBBox(words []Word, b BBox) []Word { + out := make([]Word, 0, len(words)) + for _, w := range words { + cx := (w.X0 + w.X1) / 2 + cy := (w.Y0 + w.Y1) / 2 + if cx >= b.X0 && cx <= b.X1 && cy >= b.Y0 && cy <= b.Y1 { + out = append(out, w) + } + } + return out +} diff --git a/table.go b/table.go index 81e9af1..f49371a 100644 --- a/table.go +++ b/table.go @@ -222,6 +222,33 @@ type TableSettings struct { // grouping uses, so it only ever rejoins glyphs that word grouping // would have placed in one word. MergeSplitTokens bool + + // DetectRegions restricts table finding to regions that look + // tabular by text alignment, instead of deriving edges across the + // whole page. + // + // Off by default, and deliberately so. It changes which tables are + // found, not just how they are gridded, and a detection change has + // burned this library before — StrategyAuto found more regions and + // scored slightly worse, because a region that is found but gridded + // badly costs more precision than it buys in recall. Turn it on, + // measure on your own documents, then decide. + // + // What it is for: the "lines" strategies need INTERSECTING rulings + // to form a cell, so a table ruled only horizontally is invisible to + // them. On ICDAR 2013 that is 22% of documents yielding nothing at + // all. Region detection reads alignment rather than rulings, so it + // sees those tables — and because it reports a bounded region rather + // than a page-wide grid, it does not fabricate tables on prose the + // way the bare "text" strategy does. + // + // Only affects the text-derived strategies. With "lines" the rulings + // already say where the table is. + DetectRegions bool + + // TextEdge tunes region detection. Ignored unless DetectRegions is + // set. The zero value means DefaultTextEdgeOpts. + TextEdge TextEdgeOpts } // DefaultTableSettings returns settings with the pdfplumber default