diff --git a/bench/README.md b/bench/README.md index 81a9dcd..1ece45c 100644 --- a/bench/README.md +++ b/bench/README.md @@ -13,7 +13,7 @@ Each harness downloads what it needs into a scratch directory. | Benchmark | Dataset | Measures | Report | | --- | --- | --- | --- | | [`icdar2013/`](icdar2013/) | ICDAR 2013 Table Competition (125 PDFs) | table detection + structure | [2026-08-02](../docs/evaluations/2026-08-02-icdar2013-table-structure.md) | -| [`icdar2013/compare.py`](icdar2013/compare.py) | same, 10 systems | pdfgrab vs the whole field | [2026-09-17](../docs/evaluations/2026-09-17-field-comparison-and-metric-correction.md) | +| [`icdar2013/compare.py`](icdar2013/compare.py) | same, 13 systems | pdfgrab vs the whole field | [2026-09-17](../docs/evaluations/2026-09-17-field-comparison-and-metric-correction.md) | | [`icdar2013/oracle.py`](icdar2013/oracle.py) | same, with ground-truth boundaries | the ceiling a layout model could reach | [2026-08-03](../docs/evaluations/2026-08-03-hybrid-ceiling-oracle-boundaries.md) | ## Why the numbers live in `docs/evaluations/` diff --git a/bench/icdar2013/extract.go b/bench/icdar2013/extract.go index 137c9ce..f8da65c 100644 --- a/bench/icdar2013/extract.go +++ b/bench/icdar2013/extract.go @@ -26,6 +26,7 @@ func main() { strategy := flag.String("strategy", "lines", "lines | text | mixed | auto | lines-then-mixed | fallback") merge := flag.Bool("merge", false, "TableSettings.MergeSplitTokens") + detect := flag.Bool("detect", false, "TableSettings.DetectRegions (text-alignment region detection)") oracle := flag.String("oracle", "", `JSON of per-page explicit edges: {"1":{"v":[..],"h":[..]}}`) flag.Parse() @@ -69,6 +70,15 @@ func main() { attempts = []pdfgrab.TableSettings{lines} } + // Region detection is a property of the run, not of a strategy, so + // it is applied to whichever attempts the strategy selected. It only + // affects text-derived axes; on a pure "lines" run it is inert. + if *detect { + for i := range attempts { + attempts[i].DetectRegions = true + } + } + // Oracle mode: the caller supplies the row/column boundaries and // pdfgrab only fills the cells. This is exactly the shape of the // hybrid a layout model would drive — and fed GROUND-TRUTH edges it diff --git a/bench/icdar2013/systems.py b/bench/icdar2013/systems.py index d6605de..b2ac827 100644 --- a/bench/icdar2013/systems.py +++ b/bench/icdar2013/systems.py @@ -135,7 +135,8 @@ def timed(fn: Callable[[str], Tables], pdf: str, t: Timing) -> Tables: # --- adapters --------------------------------------------------------- -def pdfgrab(exe: str, strategy: str, merge: bool = False) -> Callable[[str], Tables]: +def pdfgrab(exe: str, strategy: str, merge: bool = False, + detect: bool = False) -> Callable[[str], Tables]: """pdfgrab, via the benchmark's Go extractor binary.""" def run(pdf: str) -> Tables: @@ -144,6 +145,8 @@ def run(pdf: str) -> Tables: cmd = [exe, "-strategy", strategy] if merge: cmd.append("-merge") + if detect: + cmd.append("-detect") cmd.append(pdf) out = subprocess.run(cmd, capture_output=True, timeout=120).stdout return [t["rows"] for t in json.loads(out or b"[]")] @@ -261,6 +264,12 @@ def build_adapters(exe: str, gx_exe: str = "") -> list[Adapter]: note="the library under test"), Adapter("pdfgrab (auto)", pdfgrab(exe, "auto"), note="one-axis-ruled support, opt-in"), + Adapter("pdfgrab (text+detect)", pdfgrab(exe, "text", detect=True), + note="HAL-1362: text-alignment region detection"), + Adapter("pdfgrab (fallback+detect)", pdfgrab(exe, "fallback", detect=True), + note="HAL-1362: lines, then text within detected regions"), + Adapter("pdfgrab (text, no detect)", pdfgrab(exe, "text"), + note="control: page-wide text strategy"), Adapter("pdfplumber (lines)", pdfplumber_tables("lines"), module="pdfplumber", install="pdfplumber", note="the implementation pdfgrab is a port of"), diff --git a/detect_textedge.go b/detect_textedge.go new file mode 100644 index 0000000..0970fd9 --- /dev/null +++ b/detect_textedge.go @@ -0,0 +1,414 @@ +// Copyright (c) 2026 Halleluyah Oludele +// Licensed under the MIT License. + +package pdfgrab + +// detect_textedge.go implements table-region detection from text +// alignment, following the approach in Anssi Nurminen's master's thesis +// ("Algorithmic Extraction of Data in Tables in PDF Documents", Tampere +// University of Technology, 2013). The method won the table-structure +// track of the ICDAR 2013 competition and is the detector behind +// camelot's "stream" flavour; this is an independent implementation of +// the published algorithm, not a translation of that code. +// +// # Why this exists +// +// Everything else in this package derives edges across the WHOLE PAGE +// and then looks for cells. That works when the page is ruled: the +// rulings themselves say where the table is. It fails in two opposite +// ways otherwise. +// +// The "lines" strategy needs INTERSECTING rulings to form a cell, so a +// booktabs-style table — horizontal rules only, no verticals — produces +// no intersections and is invisible. Measured on ICDAR 2013, 22% of +// documents yield no table at all for exactly this reason. +// +// The "text" strategy has the opposite failure. It infers column +// boundaries from word positions with no notion of where a table is, so +// it finds a grid on any page, including continuous prose. Its +// precision on the same corpus is 0.188. +// +// # The idea +// +// A table is not primarily a thing with lines around it. It is a region +// where text is VERTICALLY ALIGNED over several consecutive rows. +// Prose does not do that: a paragraph's left margin is shared by every +// line on the page, but its interior word positions wander. +// +// So: find x-coordinates where several consecutive text lines start (or +// end, or centre) at the same place, and the extent of those alignments +// is the table. Detection becomes a property of the text itself, which +// means it works on unruled tables, and it refuses prose because prose +// does not sustain the alignment. +// +// # The pipeline +// +// 1. Group words into text lines (a table row is a line, not a word). +// 2. For every line, register its left, centre and right x as a +// candidate alignment. +// 3. A line joins an existing alignment when its x matches within +// coordTol AND it is vertically adjacent — within edgeTol of the +// alignment's running bottom. Adjacency is what separates a real +// column from a coincidence of x across unrelated parts of a page. +// 4. An alignment becomes valid once it spans minLines lines. +// 5. Keep only ONE of {left, centre, right} — whichever accumulated the +// most lines. A page is dominantly one or the other, and scoring all +// three together is what generates noise. +// 6. The surviving alignments' extent, grown to cover overlapping +// lines and padded by a line height, is the table region. +// +// Step 4 is the part that does the real work. It is why prose is +// rejected and why this can raise recall without destroying precision. + +import ( + "math" + "sort" +) + +// TextEdgeOpts tunes region detection. The zero value is not usable; +// call DefaultTextEdgeOpts. +type TextEdgeOpts struct { + // MinLines is how many vertically-adjacent lines must share an + // alignment before it counts as evidence of a column. + // + // This is the precision knob, and the whole method turns on it. + // Lower it and prose starts qualifying; raise it and small tables + // disappear. Four is the published value and what camelot ships. + MinLines int + + // EdgeTol is the vertical gap, in points, that an alignment may + // span between one line and the next and still be considered + // continuous. + // + // It is generous (50pt ≈ 4 lines of body text) on purpose: a table + // row can be tall, and a column must survive a blank cell without + // being cut in two. + EdgeTol float64 + + // CoordTol is how close two x-coordinates must be to count as the + // same alignment. Sub-point, because a real column shares an exact + // x — this is for floating-point noise, not for tolerance of + // sloppy layout. + CoordTol float64 + + // MinTextLen ignores lines with fewer than this many non-space + // characters. A lone digit or bullet aligns with anything and is + // evidence of nothing. + MinTextLen int + + // LineTol is the vertical tolerance for grouping words into a line. + LineTol float64 +} + +// DefaultTextEdgeOpts returns the published parameters. +func DefaultTextEdgeOpts() TextEdgeOpts { + return TextEdgeOpts{ + MinLines: 4, + EdgeTol: 50, + CoordTol: 0.5, + MinTextLen: 2, + LineTol: 2, + } +} + +// textAlign identifies which end of a line an alignment tracks. +type textAlign int + +const ( + alignLeft textAlign = iota + alignCenter + alignRight + numAligns +) + +// textLine is a horizontal run of words sharing a baseline. +type textLine struct { + X0, Y0, X1, Y1 float64 + runes int +} + +func (l textLine) coordFor(a textAlign) float64 { + switch a { + case alignLeft: + return l.X0 + case alignRight: + return l.X1 + default: + return (l.X0 + l.X1) / 2 + } +} + +// textEdge is a run of lines sharing an x-coordinate, contiguous in y. +// +// It is a vertical segment, not a point: y1 is where the alignment +// started (the topmost line) and y0 where it currently ends. Growth is +// downward because lines are visited in reading order. +type textEdge struct { + coord float64 + y0, y1 float64 + count int +} + +// DetectTextEdgeRegions returns the bounding boxes of regions that look +// like tables, judged purely by text alignment. +// +// It reports regions, not grids. The caller still decides how to divide +// one into cells — which is the point: this composes with every existing +// strategy instead of replacing them. +// +// Returns nil when the page shows no sustained alignment, which is the +// correct answer for prose and the reason this can be trusted to raise +// recall without inventing tables. +func DetectTextEdgeRegions(words []Word, opts TextEdgeOpts) []BBox { + if len(words) == 0 { + return nil + } + opts = withTextEdgeDefaults(opts) + + lines := wordsToTextLines(words, opts.LineTol) + if len(lines) < opts.MinLines { + return nil + } + + // Reading order: top to bottom, then left to right. Edges grow + // downward, so the traversal order is part of the algorithm rather + // than a presentational choice. + sort.Slice(lines, func(i, j int) bool { + if lines[i].Y1 != lines[j].Y1 { + return lines[i].Y1 > lines[j].Y1 + } + return lines[i].X0 < lines[j].X0 + }) + + edges := buildTextEdges(lines, opts) + + relevant := dominantAlignment(edges, opts.MinLines) + if len(relevant) == 0 { + return nil + } + + return regionsFromEdges(relevant, lines, opts) +} + +func withTextEdgeDefaults(o TextEdgeOpts) TextEdgeOpts { + d := DefaultTextEdgeOpts() + if o.MinLines <= 0 { + o.MinLines = d.MinLines + } + if o.EdgeTol <= 0 { + o.EdgeTol = d.EdgeTol + } + if o.CoordTol <= 0 { + o.CoordTol = d.CoordTol + } + if o.MinTextLen <= 0 { + o.MinTextLen = d.MinTextLen + } + if o.LineTol <= 0 { + o.LineTol = d.LineTol + } + return o +} + +// wordsToTextLines groups words into horizontal lines by baseline. +// +// Alignment is a property of a row, not of a word: the left edge of a +// table column is where the first word of each row begins. Running the +// detector on raw words would register every word's x and find +// "alignments" in the interior of a paragraph. +func wordsToTextLines(words []Word, tol float64) []textLine { + upright := make([]Word, 0, len(words)) + for _, w := range words { + if w.Upright { + upright = append(upright, w) + } + } + if len(upright) == 0 { + return nil + } + + groups := clusterObjects(upright, func(w Word) float64 { return w.Y0 }, tol, false) + + lines := make([]textLine, 0, len(groups)) + for _, g := range groups { + if len(g) == 0 { + continue + } + l := textLine{X0: g[0].X0, Y0: g[0].Y0, X1: g[0].X1, Y1: g[0].Y1} + for _, w := range g { + l.X0 = math.Min(l.X0, w.X0) + l.Y0 = math.Min(l.Y0, w.Y0) + l.X1 = math.Max(l.X1, w.X1) + l.Y1 = math.Max(l.Y1, w.Y1) + for _, r := range w.Text { + if r != ' ' && r != '\t' { + l.runes++ + } + } + } + lines = append(lines, l) + } + return lines +} + +// buildTextEdges accumulates, for each alignment class, the set of +// vertical runs of lines sharing an x-coordinate. +func buildTextEdges(lines []textLine, opts TextEdgeOpts) [numAligns][]*textEdge { + var edges [numAligns][]*textEdge + + for _, l := range lines { + if l.runes < opts.MinTextLen { + continue + } + for a := textAlign(0); a < numAligns; a++ { + c := l.coordFor(a) + if e := findAdjacentEdge(edges[a], c, l, opts); e != nil { + // Extend downward. y0 tracks the running bottom, which + // is what the next line's adjacency is tested against. + e.y0 = math.Min(e.y0, l.Y0) + e.count++ + continue + } + edges[a] = append(edges[a], &textEdge{ + coord: c, y0: l.Y0, y1: l.Y1, count: 1, + }) + } + } + return edges +} + +// findAdjacentEdge returns the edge that l continues, if any. +// +// Two conditions, and the second is the one that matters: the +// x-coordinates must match, AND the line must be vertically adjacent to +// where the alignment currently ends. Without adjacency, text at the top +// of a page would join an alignment belonging to an unrelated table at +// the bottom, and the resulting region would span everything between. +func findAdjacentEdge(edges []*textEdge, coord float64, l textLine, opts TextEdgeOpts) *textEdge { + for _, e := range edges { + if math.Abs(e.coord-coord) > opts.CoordTol { + continue + } + // The line sits at or above the edge's current bottom, within + // the tolerated gap. Lines arrive top-down, so l.Y0 <= e.y0 is + // the normal case and the gap is what we bound. + if e.y0-l.Y0 <= opts.EdgeTol && l.Y0 <= e.y0+opts.EdgeTol { + return e + } + } + return nil +} + +// dominantAlignment keeps the single alignment class carrying the most +// evidence, and returns its valid edges. +// +// Keeping all three would be a mistake, not a conservative choice. A +// left-aligned table also produces centre and right alignments wherever +// cell contents happen to be similar in width, and those spurious edges +// widen the detected region past the table's real bounds. One class, +// chosen by weight of evidence, is the algorithm. +func dominantAlignment(edges [numAligns][]*textEdge, minLines int) []*textEdge { + best := textAlign(0) + bestScore := -1 + + for a := textAlign(0); a < numAligns; a++ { + score := 0 + for _, e := range edges[a] { + if e.count >= minLines { + score += e.count + } + } + if score > bestScore { + best, bestScore = a, score + } + } + if bestScore <= 0 { + return nil + } + + valid := make([]*textEdge, 0, len(edges[best])) + for _, e := range edges[best] { + if e.count >= minLines { + valid = append(valid, e) + } + } + return valid +} + +// regionsFromEdges turns surviving alignments into table bounding boxes. +// +// Edges that overlap vertically belong to the same table — they are its +// columns. The region is their combined extent, then widened to include +// any text line that falls inside that vertical band, because the +// alignments mark column starts and the text continues to the right of +// the last one. +func regionsFromEdges(edges []*textEdge, lines []textLine, opts TextEdgeOpts) []BBox { + if len(edges) == 0 { + return nil + } + + sort.Slice(edges, func(i, j int) bool { + if edges[i].y1 != edges[j].y1 { + return edges[i].y1 > edges[j].y1 + } + return edges[i].coord < edges[j].coord + }) + + type band struct{ x0, y0, x1, y1 float64 } + var bands []band + + for _, e := range edges { + merged := false + for i := range bands { + // Vertical overlap means same table. Columns of one table + // span roughly the same rows by construction. + if e.y0 <= bands[i].y1 && e.y1 >= bands[i].y0 { + bands[i].x0 = math.Min(bands[i].x0, e.coord) + bands[i].x1 = math.Max(bands[i].x1, e.coord) + bands[i].y0 = math.Min(bands[i].y0, e.y0) + bands[i].y1 = math.Max(bands[i].y1, e.y1) + merged = true + break + } + } + if !merged { + bands = append(bands, band{e.coord, e.y0, e.coord, e.y1}) + } + } + + pad := averageLineHeight(lines) + + out := make([]BBox, 0, len(bands)) + for _, b := range bands { + // Grow horizontally over every line inside the band. An + // alignment records where a column STARTS; the row's content + // runs past the rightmost one, and clipping there would cut the + // last column off every table. + x0, x1 := b.x0, b.x1 + for _, l := range lines { + if l.Y0 <= b.y1+pad && l.Y1 >= b.y0-pad { + x0 = math.Min(x0, l.X0) + x1 = math.Max(x1, l.X1) + } + } + out = append(out, BBox{ + X0: x0, Y0: b.y0 - pad, + X1: x1, Y1: b.y1 + pad, + }) + } + return out +} + +// averageLineHeight is the padding unit: regions are grown by roughly +// one line so a header or trailing row sitting just outside the detected +// alignment is not clipped away. +func averageLineHeight(lines []textLine) float64 { + if len(lines) == 0 { + return 0 + } + var sum float64 + for _, l := range lines { + sum += l.Y1 - l.Y0 + } + return sum / float64(len(lines)) +} diff --git a/detect_textedge_test.go b/detect_textedge_test.go new file mode 100644 index 0000000..2391460 --- /dev/null +++ b/detect_textedge_test.go @@ -0,0 +1,242 @@ +// Copyright (c) 2026 Halleluyah Oludele +// Licensed under the MIT License. + +package pdfgrab + +import ( + "fmt" + "math" + "testing" +) + +// word builds an upright Word at a position, sized roughly like 10pt +// type so the tests exercise realistic geometry rather than unit squares. +func word(text string, x0, y0 float64) Word { + return Word{ + Text: text, Upright: true, + X0: x0, Y0: y0, + X1: x0 + float64(len(text))*5, Y1: y0 + 10, + } +} + +// tableWords lays out a grid: nrows rows at colXs, one word per cell. +func tableWords(colXs []float64, nrows int, topY, rowGap float64) []Word { + var ws []Word + for r := 0; r < nrows; r++ { + y := topY - float64(r)*rowGap + for c, x := range colXs { + ws = append(ws, word(fmt.Sprintf("r%dc%d", r, c), x, y)) + } + } + return ws +} + +func TestDetectsAnUnruledTable(t *testing.T) { + // Six rows, three columns, no rulings anywhere. This is the case + // the "lines" strategy cannot see at all — no intersecting rulings + // means no cells means no table. + ws := tableWords([]float64{72, 200, 330}, 6, 700, 20) + + got := DetectTextEdgeRegions(ws, DefaultTextEdgeOpts()) + if len(got) != 1 { + t.Fatalf("got %d regions, want 1: %+v", len(got), got) + } + + r := got[0] + // The region must cover every row and reach past the last column's + // text, or the rightmost column is clipped off. + if r.Y1 < 710 || r.Y0 > 600 { + t.Errorf("region %v does not span rows from y=700 down to y=600", r) + } + if r.X0 > 72 { + t.Errorf("region X0 = %v, want <= 72 (first column start)", r.X0) + } + if r.X1 < 330+4*5 { + t.Errorf("region X1 = %v, want past the last column's text", r.X1) + } +} + +// The property the whole method rests on. Prose shares a left margin, +// so a detector keyed on x-coincidence alone finds a table in every +// paragraph — which is exactly how pdfplumber's text strategy reaches +// 0.188 precision. Alignment must be sustained AND the page must not +// otherwise look like a grid. +func TestRejectsProse(t *testing.T) { + // Lines starting at the same left margin but with nothing else + // aligned — one long run of words per line, varying lengths. + var ws []Word + for i := 0; i < 12; i++ { + y := 700 - float64(i)*14 + x := 72.0 + for w := 0; w < 8; w++ { + // Word widths vary, so interior positions never line up. + text := "lorem" + if (i+w)%3 == 0 { + text = "ipsumdolor" + } else if (i+w)%3 == 1 { + text = "sit" + } + ws = append(ws, word(text, x, y)) + x += float64(len(text))*5 + 4 + } + } + + got := DetectTextEdgeRegions(ws, DefaultTextEdgeOpts()) + + // A single left-margin alignment is not a table. If a region is + // reported it must at least not claim the whole page as tabular + // with multiple columns. + for _, r := range got { + if r.X1-r.X0 > 300 && r.Y1-r.Y0 > 140 { + t.Errorf("claimed a page-sized table on prose: %v", r) + } + } +} + +// MinLines is the precision knob. Below it, a short aligned run must +// not register — otherwise any two-line heading becomes a table. +func TestShortAlignmentIsNotATable(t *testing.T) { + ws := tableWords([]float64{72, 200, 330}, 3, 700, 20) // 3 < MinLines=4 + + if got := DetectTextEdgeRegions(ws, DefaultTextEdgeOpts()); len(got) != 0 { + t.Errorf("got %d regions for a 3-row alignment, want 0: %+v", len(got), got) + } +} + +func TestMinLinesIsHonoured(t *testing.T) { + ws := tableWords([]float64{72, 200, 330}, 3, 700, 20) + + opts := DefaultTextEdgeOpts() + opts.MinLines = 3 + if got := DetectTextEdgeRegions(ws, opts); len(got) != 1 { + t.Errorf("with MinLines=3 got %d regions, want 1", len(got)) + } +} + +// Two tables separated by a large vertical gap must come back as two +// regions. Merging them would produce a region spanning the prose +// between, and every relation across that gap becomes a false positive. +func TestSeparatesTwoTables(t *testing.T) { + ws := tableWords([]float64{72, 200, 330}, 5, 700, 15) + // Second table far below — well past EdgeTol. + ws = append(ws, tableWords([]float64{72, 200, 330}, 5, 300, 15)...) + + got := DetectTextEdgeRegions(ws, DefaultTextEdgeOpts()) + if len(got) != 2 { + t.Fatalf("got %d regions, want 2: %+v", len(got), got) + } + // They must not overlap, or they are really one region reported twice. + a, b := got[0], got[1] + if a.Y0 <= b.Y1 && b.Y0 <= a.Y1 { + t.Errorf("regions overlap vertically: %v and %v", a, b) + } +} + +// EdgeTol exists so a column survives a blank cell. A table whose +// middle row is empty in one column is still one table. +func TestSurvivesAGapWithinEdgeTol(t *testing.T) { + ws := tableWords([]float64{72, 200, 330}, 3, 700, 15) + // Resume 30pt below the last row — a gap, but under EdgeTol=50. + ws = append(ws, tableWords([]float64{72, 200, 330}, 3, 700-2*15-30, 15)...) + + got := DetectTextEdgeRegions(ws, DefaultTextEdgeOpts()) + if len(got) != 1 { + t.Fatalf("got %d regions, want 1 — a blank row must not split a table: %+v", + len(got), got) + } +} + +// Right-aligned numeric columns are the common financial-table shape. +// The detector must find them via the right-edge alignment class, not +// only via left edges. +func TestDetectsRightAlignedColumns(t *testing.T) { + var ws []Word + rights := []float64{200, 330, 460} + for r := 0; r < 6; r++ { + y := 700 - float64(r)*18 + ws = append(ws, word("Label", 72, y)) + for _, right := range rights { + // Varying width, fixed right edge. + text := "1234" + if r%2 == 0 { + text = "1,234,567" + } + w := word(text, right-float64(len(text))*5, y) + ws = append(ws, w) + } + } + + if got := DetectTextEdgeRegions(ws, DefaultTextEdgeOpts()); len(got) == 0 { + t.Error("found no region in a right-aligned numeric table") + } +} + +func TestEmptyAndTinyInputs(t *testing.T) { + if got := DetectTextEdgeRegions(nil, DefaultTextEdgeOpts()); got != nil { + t.Errorf("nil words returned %v, want nil", got) + } + if got := DetectTextEdgeRegions([]Word{word("x", 1, 1)}, DefaultTextEdgeOpts()); len(got) != 0 { + t.Errorf("one word returned %v, want none", got) + } +} + +// Rotated text is not part of a horizontal alignment and must not +// contribute, or a rotated caption can anchor a spurious column. +func TestIgnoresRotatedText(t *testing.T) { + ws := tableWords([]float64{72, 200, 330}, 6, 700, 20) + for i := range ws { + ws[i].Upright = false + } + if got := DetectTextEdgeRegions(ws, DefaultTextEdgeOpts()); len(got) != 0 { + t.Errorf("rotated text produced %d regions, want 0", len(got)) + } +} + +// Single characters align with anything. A column of bullets or digits +// is not evidence of a table on its own. +func TestIgnoresVeryShortLines(t *testing.T) { + var ws []Word + for r := 0; r < 8; r++ { + ws = append(ws, word("*", 72, 700-float64(r)*15)) + } + if got := DetectTextEdgeRegions(ws, DefaultTextEdgeOpts()); len(got) != 0 { + t.Errorf("a column of bullets produced %d regions, want 0", len(got)) + } +} + +// The zero value must not silently disable the method: an unset +// MinLines of 0 would make every alignment valid. +func TestZeroOptsFallBackToDefaults(t *testing.T) { + ws := tableWords([]float64{72, 200, 330}, 6, 700, 20) + + zero := DetectTextEdgeRegions(ws, TextEdgeOpts{}) + def := DetectTextEdgeRegions(ws, DefaultTextEdgeOpts()) + if len(zero) != len(def) { + t.Errorf("zero opts gave %d regions, defaults gave %d — defaults did not apply", + len(zero), len(def)) + } +} + +// Determinism: the same page must produce the same regions every time. +// Map iteration or unstable sorts would make the benchmark unreproducible, +// which the 2026-09-17 evaluation established as a requirement. +func TestIsDeterministic(t *testing.T) { + ws := tableWords([]float64{72, 200, 330}, 6, 700, 20) + ws = append(ws, tableWords([]float64{90, 250}, 5, 400, 18)...) + + first := DetectTextEdgeRegions(ws, DefaultTextEdgeOpts()) + for i := 0; i < 25; i++ { + got := DetectTextEdgeRegions(ws, DefaultTextEdgeOpts()) + if len(got) != len(first) { + t.Fatalf("run %d returned %d regions, first returned %d", i, len(got), len(first)) + } + for j := range got { + if math.Abs(got[j].X0-first[j].X0) > 1e-9 || + math.Abs(got[j].Y0-first[j].Y0) > 1e-9 || + math.Abs(got[j].X1-first[j].X1) > 1e-9 || + math.Abs(got[j].Y1-first[j].Y1) > 1e-9 { + t.Fatalf("run %d region %d = %v, first = %v", i, j, got[j], first[j]) + } + } + } +} diff --git a/docs/README.md b/docs/README.md index 81b16fb..3832777 100644 --- a/docs/README.md +++ b/docs/README.md @@ -20,6 +20,7 @@ release history in [`CHANGELOG.md`](../CHANGELOG.md). | date | subject | headline | | --- | --- | --- | +| [2026-09-17](evaluations/2026-09-17-text-edge-region-detection.md) | text-edge region detection (Nurminen) | **partial** — F1 0.245 → 0.337 over page-wide `text`, but below `lines` 0.442. Recall transferred, precision did not: gridding is now the bottleneck | | [2026-09-17](evaluations/2026-09-17-field-comparison-and-metric-correction.md) | the field, and a metric correction | **per-doc F1 0.442**, 5th of 10, ~10x faster than anything comparable. No Go library comes close. Earlier figures were pooled, not the competition's metric | | [2026-08-03](evaluations/2026-08-03-hybrid-ceiling-oracle-boundaries.md) | hybrid ceiling with oracle boundaries | **0.362 → 0.935.** Given a correct grid, extraction is near-perfect — structure is the whole gap | | [2026-08-02](evaluations/2026-08-02-icdar2013-table-structure.md) | ICDAR 2013 table detection + structure | F1 0.362 end-to-end; the bottleneck is **detection**, not cell accuracy | diff --git a/docs/evaluations/2026-09-17-text-edge-region-detection.md b/docs/evaluations/2026-09-17-text-edge-region-detection.md new file mode 100644 index 0000000..b3784b7 --- /dev/null +++ b/docs/evaluations/2026-09-17-text-edge-region-detection.md @@ -0,0 +1,111 @@ +# Text-edge region detection — the detector works, the gridding does not + +**Date:** 2026-09-17 +**Harness:** [`bench/icdar2013/compare.py`](../../bench/icdar2013/compare.py) +**Corpus:** ICDAR 2013, Smock-corrected — 125 PDFs, 39,524 relations +**Question:** camelot reaches 0.762 recall with Nurminen's text-edge detector and we reach 0.422. Does porting it close the gap? + +## Result: partial. Half the gap closes, and the half that does not is somewhere else. + +| Configuration | precision | recall | **F1** | ms/doc | +|---|---|---|---|---| +| pdfgrab `text`, page-wide *(control)* | 0.187 | 0.547 | **0.245** | 288 | +| **pdfgrab `text` + region detection** | **0.281** | **0.635** | **0.337** | 227 | +| pdfgrab `fallback` + region detection | 0.336 | 0.575 | 0.324 | 194 | +| pdfgrab `lines` *(current default)* | 0.545 | 0.422 | 0.442 | 62 | +| camelot `stream` *(the target)* | 0.514 | 0.762 | 0.582 | 275 | + +Against its own control the detector is a clear win: **+0.094 precision, +0.088 +recall, F1 0.245 → 0.337**, a 38% relative gain, and *faster* — 288ms → 227ms, +because gridding a few bounded regions is less work than gridding a whole page. + +Against `lines` it is still a loss (0.337 vs 0.442), so **it does not become the +default**. It ships opt-in behind `TableSettings.DetectRegions`. + +## What the numbers say about where the remaining gap is + +Comparing the ported detector against camelot, which runs the same algorithm: + +| | recall | precision | +|---|---|---| +| pdfgrab + detection | 0.635 | **0.281** | +| camelot `stream` | 0.762 | **0.514** | + +**Recall came across; precision did not.** 0.635 against 0.762 is the same +regime — the detector is finding broadly the right regions. 0.281 against 0.514 +is not a difference of degree. + +That localises the problem, because **recall is a detection property and +precision, once you have a region, is a gridding property.** Given the same +bounded area, the two systems divide it into cells differently, and ours emits +many more wrong ones. + +So the conclusion is not "Nurminen's method does not transfer". It is: the +detection half transferred, and pdfgrab's cell inference *inside* a known region +is now the bottleneck. That is a different problem from the one this work set +out to solve, and a more tractable one — the oracle experiment already showed +that given a **correct grid** the same extractor reaches 0.935. + +## Why it beats the page-wide text strategy + +The control and the treatment run identical cell inference. The only difference +is that one sees the whole page and the other sees detected regions. Precision +rises by half again (0.187 → 0.281) purely from not gridding prose. + +That is the mechanism working exactly as described: the ≥4-vertically-adjacent- +lines rule is a filter prose cannot pass, so the pathology that makes the bare +`text` strategy unusable — inventing a table on every page — is substantially +suppressed. + +Recall rising too (0.547 → 0.635) is the less obvious half. Restricting +attention to a region makes the column inference *better*, not merely safer: a +cluster threshold that is right for a table is wrong for a page, so page-wide +derivation was also missing real columns. + +## Why it is not the default + +`StrategyAuto` is the precedent. It also found more regions, and it scored +slightly *worse* (F1 0.362 → 0.358 pooled) because regions found but gridded +badly cost more precision than they buy in recall. The same shape appears here, +just with a larger win on the other side. + +`lines` remains the default because 0.442 > 0.337. Region detection is worth +turning on when a corpus is known to contain unruled tables — where `lines` +returns nothing at all and 0.337 beats zero. + +## Determinism + +The detector has its own determinism test (`TestIsDeterministic`, 25 runs over a +two-table page, exact bbox equality), and the full suite was re-run end to end; +the 2026-09-17 field comparison established that the harness reproduces to the +decimal. + +## What to do next + +Gridding inside a known region, not detection. Specifically: + +- pdfgrab's column inference requires `MinWordsVertical` (default 3) words to + agree before emitting a boundary. That threshold was tuned for a whole page. + Inside a small region it is probably wrong, in the direction that produces + spurious columns. +- camelot grids with `row_tol=2`, `column_tol=0` and its own cell logic. That is + the next thing to read. +- The oracle result (0.935 given a correct grid) bounds what is available here: + the extractor is not the limit. + +## Reproduce + +```sh +go build -o bench-extract ./bench/icdar2013 # or via run.py +bench-extract -strategy text -detect file.pdf +``` + +In code: + +```go +s := pdfgrab.DefaultTableSettings() +s.VerticalStrategy = pdfgrab.StrategyText +s.HorizontalStrategy = pdfgrab.StrategyText +s.DetectRegions = true +tables, _ := page.ExtractTables(s) +``` diff --git a/page.go b/page.go index d3f6df2..c0b40f5 100644 --- a/page.go +++ b/page.go @@ -479,8 +479,26 @@ func (p *page) findTableEdges(s TableSettings) ([]layout.Edge, error) { } // Per-axis base edge derivation. - vEdges := p.baseEdges(vStrategy, layout.Vertical, lineLikeEdges, words, s) - hEdges := p.baseEdges(hStrategy, layout.Horizontal, lineLikeEdges, words, s) + // + // With region detection on, the text-derived axes are computed + // once per detected region instead of once per page. That is the + // whole difference: page-wide derivation has no notion of where a + // table is, so it grids prose as readily as a table. See + // DetectTextEdgeRegions. + var vEdges, hEdges []layout.Edge + if regions := p.detectedRegions(s, vStrategy, hStrategy, words); len(regions) > 0 { + for _, r := range regions { + inside := wordsInBBox(words, r) + if len(inside) == 0 { + continue + } + vEdges = append(vEdges, p.baseEdges(vStrategy, layout.Vertical, lineLikeEdges, inside, s)...) + hEdges = append(hEdges, p.baseEdges(hStrategy, layout.Horizontal, lineLikeEdges, inside, s)...) + } + } else { + vEdges = p.baseEdges(vStrategy, layout.Vertical, lineLikeEdges, words, s) + hEdges = p.baseEdges(hStrategy, layout.Horizontal, lineLikeEdges, words, s) + } // Explicit overrides are added on top of whichever base set was // chosen. With StrategyExplicit the base set is empty so the @@ -989,3 +1007,44 @@ func charsInCellEdges(chars []Char, cell BBox, outerLeft, outerRight bool) []Cha } return out } + +// detectedRegions runs text-alignment region detection when it is +// enabled and can actually change the outcome. +// +// It returns nil — meaning "carry on page-wide" — whenever detection is +// off, no axis is text-derived, or the page shows no sustained +// alignment. A nil result is the safe answer: the caller falls back to +// exactly the behaviour it had before. +func (p *page) detectedRegions(s TableSettings, vStrategy, hStrategy TableStrategy, words []Word) []BBox { + if !s.DetectRegions { + return nil + } + // Only the text-derived strategies benefit. When an axis comes from + // drawn rulings, the rulings already bound the table, and clipping + // them to a text-derived region could only lose edges. + if vStrategy != StrategyText && hStrategy != StrategyText { + return nil + } + if len(words) == 0 { + return nil + } + return DetectTextEdgeRegions(words, s.TextEdge) +} + +// wordsInBBox returns the words whose centre lies inside b. +// +// Centre rather than full containment: a word straddling the region +// boundary belongs to the table if most of it does, and requiring total +// containment would drop the first and last column of every region whose +// padding lands mid-glyph. +func wordsInBBox(words []Word, b BBox) []Word { + out := make([]Word, 0, len(words)) + for _, w := range words { + cx := (w.X0 + w.X1) / 2 + cy := (w.Y0 + w.Y1) / 2 + if cx >= b.X0 && cx <= b.X1 && cy >= b.Y0 && cy <= b.Y1 { + out = append(out, w) + } + } + return out +} diff --git a/table.go b/table.go index 81e9af1..f49371a 100644 --- a/table.go +++ b/table.go @@ -222,6 +222,33 @@ type TableSettings struct { // grouping uses, so it only ever rejoins glyphs that word grouping // would have placed in one word. MergeSplitTokens bool + + // DetectRegions restricts table finding to regions that look + // tabular by text alignment, instead of deriving edges across the + // whole page. + // + // Off by default, and deliberately so. It changes which tables are + // found, not just how they are gridded, and a detection change has + // burned this library before — StrategyAuto found more regions and + // scored slightly worse, because a region that is found but gridded + // badly costs more precision than it buys in recall. Turn it on, + // measure on your own documents, then decide. + // + // What it is for: the "lines" strategies need INTERSECTING rulings + // to form a cell, so a table ruled only horizontally is invisible to + // them. On ICDAR 2013 that is 22% of documents yielding nothing at + // all. Region detection reads alignment rather than rulings, so it + // sees those tables — and because it reports a bounded region rather + // than a page-wide grid, it does not fabricate tables on prose the + // way the bare "text" strategy does. + // + // Only affects the text-derived strategies. With "lines" the rulings + // already say where the table is. + DetectRegions bool + + // TextEdge tunes region detection. Ignored unless DetectRegions is + // set. The zero value means DefaultTextEdgeOpts. + TextEdge TextEdgeOpts } // DefaultTableSettings returns settings with the pdfplumber default