Compare commits
4
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d5a59d9eab | ||
|
|
a954df0c05 | ||
|
|
1555f96071 | ||
|
|
3782473d90 |
+53
-2
@@ -21,7 +21,6 @@ CMakeUserPresets.json
|
||||
/third_party/pdfium/checkout/
|
||||
/third_party/pdfium/install/
|
||||
/third_party/pdfium/.gclient*
|
||||
/corpus/
|
||||
|
||||
# Skia from-source build (depot_tools / GN / Ninja)
|
||||
/third_party/skia/depot_tools/
|
||||
@@ -68,8 +67,45 @@ gateway/*.pyd
|
||||
gateway/*.so
|
||||
gateway/*.dylib
|
||||
|
||||
# Downloaded fuzzing corpus (large; fetched via scripts/fetch_corpus.py)
|
||||
# Corpora — keep convert metadata; ignore large samples / archives
|
||||
/corpus/*
|
||||
!/corpus/convert/
|
||||
/corpus/convert/**
|
||||
!/corpus/convert/real/
|
||||
!/corpus/convert/public/
|
||||
!/corpus/convert/third_party/
|
||||
/corpus/convert/real/**
|
||||
!/corpus/convert/real/pack.json
|
||||
!/corpus/convert/real/*.md
|
||||
/corpus/convert/public/**
|
||||
!/corpus/convert/public/README.md
|
||||
!/corpus/convert/public/pack_public.json
|
||||
!/corpus/convert/public/manifests/
|
||||
!/corpus/convert/public/manifests/**
|
||||
!/corpus/convert/public/reports/
|
||||
!/corpus/convert/public/reports/**/*.md
|
||||
!/corpus/convert/public/reports/**/*.json
|
||||
/corpus/convert/public/samples/
|
||||
/corpus/convert/third_party/**
|
||||
!/corpus/convert/third_party/NOTICE
|
||||
!/corpus/convert/third_party/LICENSE_CHECKLIST.md
|
||||
!/corpus/convert/third_party/checksums.sha256
|
||||
corpus/fuzz/
|
||||
# Layout ML weights (download via scripts/convert/fetch_layout_onnx.py)
|
||||
/gateway/models/layout/**/*.onnx
|
||||
/gateway/models/layout/**/*.pdmodel
|
||||
/gateway/models/layout/**/*.pdiparams
|
||||
!/gateway/models/layout/**/README.md
|
||||
!/gateway/models/layout/**/.gitkeep
|
||||
# Arabic OCR rec weights (download via scripts/convert/fetch_ocr_arabic_onnx.py)
|
||||
/gateway/models/ocr/**/*.onnx
|
||||
!/gateway/models/ocr/**/README.md
|
||||
!/gateway/models/ocr/**/.gitkeep
|
||||
!/gateway/models/ocr/**/*.txt
|
||||
# DocLayNet / Pub* lab archives
|
||||
**/DocLayNet*.zip
|
||||
**/PubLayNet*
|
||||
**/PubTabNet*
|
||||
# Large-corpus regression baseline (derived from the gitignored corpus above)
|
||||
tests/regression/baseline-large/
|
||||
# Fuzzer working artifacts (crashes, leaks, coverage)
|
||||
@@ -79,6 +115,10 @@ leak-*
|
||||
timeout-*
|
||||
|
||||
PDF Editor Timeline.xlsx
|
||||
# Operator convert score log (Phase 0); never commit customer DOCX
|
||||
convert-scores.jsonl
|
||||
**/convert-scores.jsonl
|
||||
|
||||
# Local environment config / secrets
|
||||
/third_party/pdfium-wasm/
|
||||
|
||||
@@ -90,3 +130,14 @@ models/**/*.safetensors
|
||||
models/**/*.index
|
||||
!models/**/.gitkeep
|
||||
.github
|
||||
|
||||
# Runtime job store. gateway/data/jobs.sqlite3 is a live SQLite database the
|
||||
# service writes to on every job — including every test run — so a tracked copy
|
||||
# shows up as a modified file after any use of the gateway, and can carry job
|
||||
# rows (customer filenames among them) into the repository. It is state, not
|
||||
# source; the schema is created on startup.
|
||||
gateway/data/jobs.sqlite3
|
||||
gateway/data/jobs.sqlite3-wal
|
||||
gateway/data/jobs.sqlite3-shm
|
||||
gateway/data/jobs_spool/
|
||||
!gateway/data/jobs_spool/.gitkeep
|
||||
|
||||
+5
-4
@@ -73,9 +73,9 @@ find_package(freetype CONFIG REQUIRED)
|
||||
find_package(harfbuzz CONFIG REQUIRED)
|
||||
find_package(spdlog CONFIG REQUIRED)
|
||||
find_package(nlohmann_json CONFIG REQUIRED)
|
||||
if(PDFENGINE_WITH_QPDF)
|
||||
find_package(qpdf CONFIG REQUIRED)
|
||||
endif()
|
||||
find_package(qpdf CONFIG REQUIRED)
|
||||
find_package(ZLIB REQUIRED)
|
||||
find_package(JPEG REQUIRED)
|
||||
if(WIN32 AND DEFINED VCPKG_TARGET_TRIPLET)
|
||||
link_directories("${CMAKE_BINARY_DIR}/vcpkg_installed/${VCPKG_TARGET_TRIPLET}/lib")
|
||||
link_directories("${CMAKE_BINARY_DIR}/vcpkg_installed/${VCPKG_TARGET_TRIPLET}/debug/lib")
|
||||
@@ -110,4 +110,5 @@ message(STATUS " Build tests .......... ${PDFENGINE_BUILD_TESTS}")
|
||||
message(STATUS " Sanitizers ........... ${PDFENGINE_ENABLE_SANITIZERS}")
|
||||
message(STATUS " Warnings as errors ... ${PDFENGINE_WARNINGS_AS_ERRORS}")
|
||||
message(STATUS " Link PDFium .......... ${PDFENGINE_WITH_PDFIUM}")
|
||||
message(STATUS " Link Skia ............ ${PDFENGINE_WITH_SKIA}")
|
||||
message(STATUS " Link Skia ............ ${PDFENGINE_WITH_SKIA}")
|
||||
message(STATUS " Link QPDF ............ ${PDFENGINE_WITH_QPDF}")
|
||||
@@ -17,7 +17,10 @@
|
||||
"displayName": "Windows • RelWithDebInfo + PDFium (static CRT — build dir outside OneDrive/spaces)",
|
||||
"inherits": "windows-release",
|
||||
"binaryDir": "D:/pdfeng-build/win-local-pdfium",
|
||||
"cacheVariables": { "PDFENGINE_WITH_PDFIUM": "ON" }
|
||||
"cacheVariables": {
|
||||
"PDFENGINE_WITH_PDFIUM": "ON",
|
||||
"PDFENGINE_WITH_QPDF": "ON"
|
||||
}
|
||||
}
|
||||
],
|
||||
"buildPresets": [
|
||||
|
||||
@@ -0,0 +1,517 @@
|
||||
# Document Conversion Engine: End-to-End Deep Dive
|
||||
|
||||
**Investigation date:** 2026-09-09
|
||||
**Scope:** `pdf/` repository, with emphasis on `gateway/app/services/convert`, the
|
||||
PDF engine binding, OCR/layout models, writers, configuration, and conversion tests.
|
||||
**Evidence standard:** real documents were passed through the public
|
||||
`run_conversion` path and the resulting Office files were parsed and rendered.
|
||||
Numbers below are measurements on the Windows development host, not vendor
|
||||
claims or a declaration of production readiness.
|
||||
|
||||
## Current State
|
||||
|
||||
### Executive finding
|
||||
|
||||
The engine is a composed, mostly in-process conversion stack. It is effective for
|
||||
simple digital PDFs, text extraction, Office document generation, and searchable
|
||||
PDF creation. It is not a general PDF layout clone. A PDF has positioned drawing
|
||||
operators, while DOCX/XLSX require a semantic flow/grid model; the conversion
|
||||
therefore reconstructs an intermediate document model (IDM) and necessarily loses
|
||||
some semantics and positioning.
|
||||
|
||||
The most visible historical defect, interleaved text on multi-column PDFs, had a
|
||||
specific upstream cause and is fixed in the current tree. The remaining quality
|
||||
ceiling is dominated by table reconstruction, OCR cost/ambiguity, and the choice
|
||||
of ReportLab as the Office-to-PDF fallback. A successful HTTP response is not
|
||||
equivalent to faithful content or layout: the measured corpus still contains
|
||||
low-order similarity and three failing assertions.
|
||||
|
||||
### Execution topology
|
||||
|
||||
The current request path is:
|
||||
|
||||
```text
|
||||
HTTP /v1/convert/*
|
||||
-> filename/MIME/magic-byte/size validation
|
||||
-> converter registry and optional automatic PDF policy
|
||||
-> bounded worker admission (queue + shared conversion slots)
|
||||
-> conversion/document-cache scopes
|
||||
-> PDF route (digital, scanned, image-based, mixed)
|
||||
-> per-page geometry and content recovery
|
||||
C++ pdfengine ordered glyphs (preferred)
|
||||
pypdf visitor/content-stream geometry (fallback)
|
||||
line grouping -> paragraphs -> reading order -> tables/images
|
||||
optional ONNX layout regions and table separators
|
||||
-> OCR for scan/hybrid pages
|
||||
PDF render -> RapidOCR Latin pass -> optional Arabic pass
|
||||
confidence/script merge -> IDM blocks and metadata
|
||||
-> IDM optimization (headers/footers, wrapped rows, figures)
|
||||
-> format-specific formatter/writer
|
||||
-> OOXML/PDF validity checks, quality score and warnings
|
||||
-> content-addressed result cache
|
||||
```
|
||||
|
||||
`run_conversion` binds cancellation, deadlines, options, automatic-routing
|
||||
decisions, and a per-conversion read/raster/geometry cache. The public converter
|
||||
registry currently covers PDF, DOCX, XLSX, PPTX, HTML, Markdown, text, CSV,
|
||||
JSON, and image paths; the exact pair list is exposed by the convert info route.
|
||||
|
||||
### Component inventory
|
||||
|
||||
| Layer | Implementation in this tree | Observed role and limitation |
|
||||
| --- | --- | --- |
|
||||
| PDF parser/render core | C++23 `pdfengine` pybind module; PDFium, FreeType, HarfBuzz, optional Skia/QPDF | Gives authoritative glyphs, fonts, page geometry, rendering, and editing on this host. Native availability is not guaranteed on every Python deployment. |
|
||||
| Python PDF parser | `pypdf==5.3.0` | Text extraction, page count, content-stream fallback, XObject inspection. Its inferred whitespace and broken ToUnicode output cannot be treated as ground truth. |
|
||||
| Raster path | `pypdfium2==5.13.0` declared in `pyproject.toml` | Required for hosts without the C++ renderer. It is absent from the active venv, so raster-specific tests skip there; the C++ binding supplies rendering for the current audit. |
|
||||
| OCR | `rapidocr-onnxruntime==1.2.3`, `onnxruntime==1.28.0` | Shared Latin detector/recognizer; optional PP-OCRv5 Arabic recognizer. Inference is serialized because RapidOCR sessions mutate shared state. |
|
||||
| Layout ML | PP-DocLayoutV3-derived ONNX, 130,502,330 bytes FP32 or 34,694,580 bytes INT8 | Region labels narrow heuristic decisions; it does not reconstruct a document by itself. It runs on CPU in the measured environment. |
|
||||
| Table ML | Optional `structure.onnx` (Table Transformer/SLANet operator-supplied) | Not vendored (`shipped: false` in `models/MANIFEST.json`); heuristic lattice/stream/rectangle detectors remain the default. |
|
||||
| IDM | `app/services/convert/idm/model.py` and layout pipeline | Stores pages, BBoxes, paragraphs, headings, headers/footers, figures, and cell grids. It is a useful interchange contract, but not a complete PDF display list or OOXML style graph. |
|
||||
| Office writers | `python-docx==1.1.2`, `openpyxl==3.1.5`, `python-pptx` code, XML sanitization | Good for creating valid Office files; cannot infer all original PDF semantics (fonts, fields, anchored objects, section behavior). |
|
||||
| Office -> PDF | Gotenberg/LibreOffice attempt, then in-process ReportLab 4.2.5 | Gotenberg is best fidelity when reachable. The benchmark used the ReportLab fallback, which is concurrent-safe but not a Word/Excel print-layout clone. |
|
||||
| Images | Pillow 10.4.0, PDFium render/XObject extraction | Images are extracted, resized/compacted, or the complete page is embedded when reconstruction would lose visible content. |
|
||||
| Font recognition | `FontRecognitionService` facade, classifier and embedding store | The checked-in `FontClassifier` uses aspect ratio/dark-pixel heuristics and `EmbeddingStore` uses a deterministic 16x8 grayscale projection. `load_model()` reads metadata but does not load an OpenCLIP ONNX session; `model.onnx` and `fonts.index` are not present in the repository. This is metadata assistance, not a validated font-identification model. |
|
||||
|
||||
The declared dependency set is intentionally permissive (pypdf BSD, pypdfium2
|
||||
Apache/BSD, python-docx/openpyxl MIT, ReportLab BSD, Pillow HPND). `mammoth==1.8.0`
|
||||
is installed in this venv for DOCX/HTML fallback paths; `fitz`/PyMuPDF is absent,
|
||||
which avoids an AGPL transitive dependency but removes that optional path.
|
||||
|
||||
### Runtime configuration and resource policy
|
||||
|
||||
The startup script defaults to automatic reconstruction, layout ML (INT8 when
|
||||
available), Arabic OCR with adaptive gating, a 1,800 second maximum conversion
|
||||
timeout, 22 seconds per OCR page budget, 45 seconds per-page timeout, a 1,200
|
||||
second OCR wall budget, four conversion workers, four simultaneous conversion
|
||||
slots, and a queue of roughly two batches. Result caching uses namespace
|
||||
`conversion-engine-v2` and includes effective options and routing mode in the key.
|
||||
|
||||
These are protection mechanisms, not throughput guarantees. A single process can
|
||||
still hold multiple ONNX sessions and large page rasters, and OCR inference is
|
||||
deliberately serialized. Horizontal scaling or process isolation is needed for
|
||||
untrusted large-document workloads.
|
||||
|
||||
### Jobs, storage, and failure boundaries
|
||||
|
||||
Synchronous and asynchronous routes share a process-wide `ThreadPoolExecutor`
|
||||
and bounded semaphores. Asynchronous job state is persisted in SQLite with WAL
|
||||
and `synchronous=NORMAL`; results larger than 512 KB are spooled to a local
|
||||
`jobs_spool` directory, while smaller results are stored as BLOBs. Jobs older
|
||||
than three hours are purged, and jobs left queued/running after a process restart
|
||||
are marked failed. This gives useful single-node crash recovery, but SQLite and
|
||||
local spool paths are not a multi-instance object store: replicas need a shared
|
||||
database/object store or a queue service before horizontal scaling.
|
||||
|
||||
The timeout helper releases the HTTP caller on deadline by abandoning a worker
|
||||
thread (`ThreadPoolExecutor.shutdown(wait=False)`). Python cannot kill that
|
||||
thread, so a native render/OCR operation may continue in the background until it
|
||||
returns. Cooperative cancellation is checked at page/OCR boundaries, which
|
||||
limits normal work, but a hung native call can still consume CPU/RAM after a 504.
|
||||
Production deployments should put untrusted conversions behind process/container
|
||||
boundaries with an OS-level kill and cgroup limits.
|
||||
|
||||
Gotenberg/LibreOffice is attempted for Office-to-PDF with a 30-second client
|
||||
timeout and fails open to ReportLab. This makes the endpoint available when the
|
||||
microservice is absent, but it can silently change fidelity class unless the
|
||||
response warning is surfaced to callers and monitored.
|
||||
|
||||
## Root Cause
|
||||
|
||||
The table below ranks causes by customer impact and likelihood in the measured
|
||||
system. "Fixed" means the implementation changed and the targeted regression
|
||||
improved; it does not mean the entire class of failures is solved.
|
||||
|
||||
| Rank | Severity | Root cause | Evidence | State |
|
||||
| --- | --- | --- | --- | --- |
|
||||
| 1 | P0 | Precise glyph coverage was compared with `len(pypdf.extract_text())`, which includes inferred spaces/newlines, while the precise walker emits text-showing characters only. Pages near the 0.80 gate fell into `_glyphs_from_visitor`, whose width calibration measured the two-column pitch rather than font advances. | Mozilla pages had coverage 0.754-0.858; stripping whitespace raised it to 0.879-1.000. Estimated spans crossed the gutter; the visible sentence was interleaved. | Fixed by comparing non-whitespace characters and retaining the precise path. |
|
||||
| 2 | P0 | Baseline clustering was page-global. Different leading in adjacent columns let rows "bridge" into one bucket; splitting columns later could not undo the vertical weld. | On the paper page, 6.97 pt copyright rows and 8.97 pt body rows were interleaved; 4.6 pt and 3.3 pt hops were under the global tolerance. Reading-order inversions fell from 189 to 1 after column-aware bands. | Fixed with `_column_band` and a bounded baseline clip. |
|
||||
| 3 | P0 | OCR is CPU-heavy and each bilingual page can run Latin and Arabic recognition, plus per-line/per-word font metadata work. A shared RapidOCR session cannot safely run concurrently. | 20-page scan: 142.747 s wall, 655.016 CPU-s, 1,769.4 MB peak RSS. Warm one-page inference was 3.021 s/17.318 CPU-s with Arabic versus 1.370 s/7.943 CPU-s with adaptive skip. | Partially fixed: inference/model locks, adaptive Arabic, page caps, budgets, and raster reuse are present. |
|
||||
| 4 | P1 | Table reconstruction is heuristic by default; ruled headers and multi-line cells are represented as fragments before grid assignment. The optional structure model is not shipped. | W3C exact table gate produced recall 0.867 (<0.90). `Participants` became `s Participant`; `Time to complete` became `Results Time to complete`; `n=1` gained a space after `=`. | Open; targeted fix is required. |
|
||||
| 5 | P1 | PDF-to-editable-Office is a semantic reflow, not a pixel-preserving transform. Header/footer detection, style inference, anchored objects, columns, fields, and arbitrary positioned graphics have no one-to-one OOXML representation. | Pixel similarity varies from 0.784 (USGS) to 0.950 (govinfo) in the ReportLab-rendered comparison; order scores are 0.232-0.674 for several digital PDFs. | Fundamental limitation; use page-image fallback or a fidelity backend when editability is not required. |
|
||||
| 6 | P1 | OCR recognition is sensitive to source DPI, contrast, skew, script, and confidence thresholds. Low thresholds (0.08 box/text) preserve faint marks but also admit noise; post-processing must distinguish symbols from rules. | Scan recall is 1.000 against expected page tokens but precision is 0.077 because OCR also emits repeated body/table text. Arabic adaptive mode improves CPU cost but whole-run wall time is noisy. | Partially fixed; needs calibrated per-document confidence and preprocessing. |
|
||||
| 7 | P1 | Quality scoring historically rewarded agreement with a corrupted pypdf text layer and ignored duplicates/order. Structure bonus could be earned by a diagram label classified as a header. | Mozilla score moved 0.9475 -> 0.8998 while the corrupted duplicate text disappeared and real prose improved. New metrics expose precision/order and store `idm.meta["text_metrics"]`. | Fixed for diagnostics; corpus gates still need independent gold/layout checks. |
|
||||
| 8 | P2 | Repeated PDF reads, geometry extraction, rasterization, and model initialization inflated latency and memory. Unbounded cache keys also allowed stale results across option/routing changes. | Per-conversion document cache and canonical cache keys were added; regression suites cover collisions and queue admission. | Fixed/mitigated. |
|
||||
| 9 | P2 | API contract mismatch unrelated to conversion: unknown font endpoint currently returns 204 while its test expects 404. | Full suite failure in `tests/test_inplace_editing.py::test_font_endpoint_rejects_unknown`. | Open decision for API owner. |
|
||||
| 10 | P2 | Timeout release is thread-based, not process-based. A native renderer or OCR call that ignores cancellation can continue after the request receives 504; local SQLite/spool storage also limits multi-instance scaling. | `run_with_timeout()` explicitly uses `shutdown(wait=False)` because Python cannot kill a thread; job store uses one local SQLite file and local spool directory. No runaway crash was observed in the audit, but the resource behavior is a design risk. | Mitigate with process/container isolation and shared durable storage. |
|
||||
|
||||
### Content accuracy
|
||||
|
||||
PDF text is not necessarily a trustworthy source string. Embedded fonts can have
|
||||
missing or malformed ToUnicode maps, CID encodings, zero-width markers, and text
|
||||
show operators split in ways that do not correspond to words. The current pipeline
|
||||
therefore combines C++ glyph extraction, pypdf extraction, content-stream evidence,
|
||||
OCR, and text-quality heuristics. Any stage that chooses a wrong representation can
|
||||
duplicate, drop, or reorder characters before the writer sees them.
|
||||
|
||||
The old comparison metric made this harder to observe: ASCII-only unique-token
|
||||
intersection ignored Arabic, CJK, combining marks, symbols, repeated occurrences,
|
||||
and order. The replacement is Unicode NFKC normalization, line-break hyphen joining,
|
||||
duplicate-aware multiset matching, meaningful-symbol retention, and an ordered
|
||||
sequence score. It is still a diagnostic metric, not a semantic proof.
|
||||
|
||||
### Layout and structure fidelity
|
||||
|
||||
The IDM preserves BBoxes and reading order, but a flowing DOCX writer must choose
|
||||
paragraphs, columns, tables, and page breaks. Heuristics infer headings from size
|
||||
and text, detect headers/footers from repetition and position, and merge rows across
|
||||
pages. They cannot preserve every PDF transformation matrix, clipping path,
|
||||
baseline, text box, field, or z-order. Automatic page-raster fallback is the safety
|
||||
floor for complex positioned pages: visual content survives, but text is no longer
|
||||
editable or searchable as individual runs.
|
||||
|
||||
### Formatting preservation
|
||||
|
||||
The writers can carry bold, italic, colors, hyperlinks, bullets, numbering, and
|
||||
several paragraph attributes when those attributes exist in the source Office XML
|
||||
or can be inferred from PDF glyph metadata. PDF extraction usually lacks the
|
||||
original style names and theme inheritance. Font aliases and locally installed
|
||||
fonts determine metrics; missing fonts fall back to Helvetica/Liberation/DejaVu or
|
||||
script-specific Noto faces. Word-to-PDF currently applies a fixed paragraph
|
||||
`spaceAfter` in the ReportLab path rather than reproducing every source paragraph's
|
||||
spacing, so small but systematic vertical shifts are expected.
|
||||
|
||||
### Tables
|
||||
|
||||
The detector combines ruling-line (lattice), whitespace/stream, rectangle, and
|
||||
plausibility passes, then maps glyph lines into cells. A two-column spanning header
|
||||
or a row whose words sit on separate baselines can be split into fragments. Merged
|
||||
cells and nested content are represented approximately. The structure ONNX hook
|
||||
can supply separators, but the model is operator-supplied and fail-open; it is not
|
||||
part of the reproducible default deployment.
|
||||
|
||||
### Images and graphics
|
||||
|
||||
XObjects and rendered page images are retained where possible. Large figures are
|
||||
compacted rather than silently discarded. If a page reconstructs to no meaningful
|
||||
blocks, the pipeline renders and embeds that page as a figure, or emits an explicit
|
||||
warning. This preserves visible pixels but cannot preserve editable diagram
|
||||
objects, vector semantics, captions' relationships, or exact crop/ordering in all
|
||||
cases.
|
||||
|
||||
## Evidence
|
||||
|
||||
### Measurement method
|
||||
|
||||
`tools/audit_conversion.py` invokes the same in-process `run_conversion` entry point
|
||||
used by the gateway. For each case it records input/output bytes, source and output
|
||||
page counts, warnings/errors, duplicate-aware text recall/precision/order/F1, table
|
||||
payloads, and model status. A Windows process sampler records working-set peak and
|
||||
`GetProcessTimes` CPU seconds at 20 ms intervals. DOCX/XLSX outputs are parsed with
|
||||
`python-docx`/`openpyxl`; visual comparison renders both sides at 96 DPI and reports
|
||||
`1 - mean(abs(gray_a-gray_b))/255` over the compared pages.
|
||||
|
||||
The visual metric is intentionally simple and should be treated as a trend signal.
|
||||
It is not SSIM, does not align transformed pages, and in this run renders Office
|
||||
outputs with the in-process ReportLab fallback. It therefore does not measure what
|
||||
LibreOffice/Gotenberg would produce.
|
||||
|
||||
### Representative real-document run
|
||||
|
||||
The latest six-case digital run is the current `audit_results.json` artifact; the
|
||||
20-page scan row is from the preceding targeted run and will be appended after the
|
||||
final OCR run. Times are wall seconds; CPU is process CPU seconds; RSS is peak
|
||||
working set. "Order" is the ordered token similarity, not a page-layout score.
|
||||
Output size and output page count are taken directly from the audit JSON. Re-run
|
||||
the command in Appendix A after any code change to refresh all rows. Results can
|
||||
vary with cache/model warm-up and with whether the page-raster fallback is
|
||||
selected, so the harness should be run more than once when comparing changes.
|
||||
|
||||
| Document -> target | Source pages | Rendered output pages* | Page delta* | Wall s | CPU s | Peak RSS MB | Output bytes | Warnings | Recall | Precision | Order | Pixel similarity |
|
||||
| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |
|
||||
| govinfo -> DOCX | 1 | 2 | +1 | 4.707 | 12.250 | 717.0 | 39,779 | 7 | 0.941 | 0.888 | 0.443 | 0.950 |
|
||||
| IRS 1040 -> DOCX | 2 | 7 | +5 | 4.531 | 13.578 | 1118.7 | 43,430 | 9 | 0.936 | 0.944 | 0.674 | 0.890 |
|
||||
| Mozilla PDF spec (14 p) -> DOCX | 14 | 28 | +14 | 29.785 | 92.750 | 1141.1 | 77,386 | 24 | 0.916 | 0.937 | 0.149 | 0.895 |
|
||||
| USGS factsheet (4 p) -> DOCX | 4 | 7 | +3 | 14.897 | 30.594 | 1346.8 | 161,987 | 13 | 0.767 | 0.793 | 0.232 | 0.784 |
|
||||
| W3C ruled table -> XLSX | 1 | 1 | 0 | 1.732 | 5.391 | 1191.7 | 5,328 | 7 | 1.000 | 1.000 | 0.990 | 0.959 |
|
||||
| Multipage table (2 p) -> XLSX | 2 | 2 | 0 | 3.280 | 11.891 | 1195.6 | 5,593 | 7 | 0.600 | 1.000 | 0.750 | 0.998 |
|
||||
| Scan pack 1 (20 p) -> DOCX | 20 | 4 | -16 | 142.747 | 655.016 | 1769.4 | 58,244 | 102 | 1.000 | 0.077 | 0.143 | 0.957 |
|
||||
|
||||
Source characteristics for the digital cases were 3,487 chars/545 words (govinfo),
|
||||
10,416/2,223 (IRS), 82,714/13,439 (Mozilla), 14,022/1,887 (USGS), 375/65 (W3C),
|
||||
and 73/25 (multipage table) according to pypdf's source read. The scan pack has
|
||||
20 raster pages and only a tiny/invalid embedded text layer, so expected page-token
|
||||
recall was supplied independently (`ScanTokenpack1P000` ... `P019`).
|
||||
|
||||
\*Rendered output pages are the pages produced when the audit harness sends the
|
||||
Office bytes through its in-process ReportLab visual renderer. They are not a
|
||||
claim about a user's pagination in Word. For example, the scan DOCX contains 19
|
||||
explicit page breaks (20 intended pages) even though the fallback visual render
|
||||
reports four pages; both page-break metadata and rendered page count must be
|
||||
tracked.
|
||||
|
||||
The scan result is a useful warning: recall 1.000 alone would say "perfect," while
|
||||
precision 0.077 and order 0.143 show that the editable OCR reconstruction contains
|
||||
substantial extra/repeated material. Its pixel similarity is high because the page
|
||||
images are retained; visual success and editable-text success are different
|
||||
products.
|
||||
|
||||
### Multi-column defect A/B evidence
|
||||
|
||||
On the Mozilla document, the old path produced 189 reading-order inversions in the
|
||||
probe. After the coverage-gate and column-aware row fixes, the probe reports 1.
|
||||
The corrected prose sentence is no longer spliced. The old corpus quality number
|
||||
was 0.9475 and the new number 0.8998, but the output character count fell from
|
||||
83,136 to 79,223 because duplicate visitor text disappeared. A/B instrumentation
|
||||
showed recall 0.9806 -> 0.9608 and missing token *types* 41 -> 83, while missing
|
||||
real words changed 31 -> 32; the additional "missing" tokens were chart-axis
|
||||
labels that pypdf had decoded differently. This is a metric-baseline artifact,
|
||||
not evidence that the repaired sentence was lost.
|
||||
|
||||
### Table evidence
|
||||
|
||||
The hand-labelled table truth in `corpus/convert/table_truth.json` requires the W3C
|
||||
exact six-column grid to reach recall 0.90 and precision 0.55. The current failing
|
||||
output from `build_idm_from_pdf(table.pdf)` is:
|
||||
|
||||
```text
|
||||
["Disability Category", "s Participant", "Ballots Completed",
|
||||
"Ballots Incomplete/ Terminated", "Accuracy", "Results Time to complete"]
|
||||
["Blind", "5", "1", "4", "34.5%, n= 1", "1199 sec, n=1"]
|
||||
...
|
||||
["Mobility", "3", "3", "0", "95.4%, n= 3", "1416 sec, n=3"]
|
||||
```
|
||||
|
||||
Compiled glyph inspection found a zero-width-space marker at approximately
|
||||
`x=222.28` and the following `s` at `x=222.31`, with near-identical baseline;
|
||||
`lines_from_glyphs()` separated them. Header words were spread over multiple
|
||||
baselines (`Participant`/`s`, `Results`/`Time to complete`, and similar pairs),
|
||||
and some spans contained the punctuation artifact `"= "`. The two failing table
|
||||
tests are therefore reproducible extraction defects, not a threshold problem.
|
||||
|
||||
### OCR and model evidence
|
||||
|
||||
The active venv reports:
|
||||
|
||||
```text
|
||||
pypdf 5.3.0
|
||||
onnxruntime 1.28.0
|
||||
rapidocr-onnxruntime 1.2.3
|
||||
providers: AzureExecutionProvider, CPUExecutionProvider
|
||||
pypdfium2: unavailable
|
||||
```
|
||||
|
||||
No CUDA provider was available, so the measured conversion was CPU-bound. The
|
||||
layout model is 130.5 MB FP32 (34.7 MB INT8 alternative); the Arabic recognizer is
|
||||
8.0 MB. The root `models/font` tree contains only metadata/tokenizer placeholders,
|
||||
not `model.onnx` or `fonts.index`, and the classifier code does not create an ONNX
|
||||
session. Any font-confidence value from that service must therefore be treated as
|
||||
a heuristic, not a calibrated model probability.
|
||||
|
||||
Arabic warm-page measurements were stable enough to show the tradeoff:
|
||||
|
||||
| Mode | Mean wall/page | Mean CPU/page | Interpretation |
|
||||
| --- | ---: | ---: | --- |
|
||||
| Arabic pass enabled | 3.021 s | 17.318 CPU-s | Better script coverage, roughly double inference work |
|
||||
| Adaptive skip on clean Latin | 1.370 s | 7.943 CPU-s | About 55% lower inference cost; known Arabic/broken-cmap pages still force the pass |
|
||||
|
||||
Whole 20-page wall time was noisy (non-adaptive 150.389 s versus adaptive
|
||||
178.408 s in separate runs), so the warm-page measurements, not that pair alone,
|
||||
should guide capacity planning.
|
||||
|
||||
### Test evidence
|
||||
|
||||
Focused regression groups passed as follows: earlier cache/queue suite 96 tests;
|
||||
OCR budget/Arabic suite 32; CORS/model packaging 8; raster/quality/real-corpus
|
||||
focused run 34 passed with 16 raster-specific skips; streaming export plus
|
||||
cache/queue 7; corpus/phase/production smoke 36; and an additional metrics/OCR
|
||||
run 133. The full gateway suite completed 746 passed, 21 skipped, and 3 failed.
|
||||
|
||||
The three failures were:
|
||||
|
||||
1. `test_labelled_table_is_recovered[w3c_ruled_grid]`: recall 0.867, floor 0.90.
|
||||
2. `test_table_pdf_reading_order_and_spans`: the same `n= 1` and header-fragment
|
||||
defects.
|
||||
3. `test_font_endpoint_rejects_unknown`: endpoint returned 204; test expects 404.
|
||||
|
||||
The raster skips are environmental (`pypdfium2` is missing), not green evidence
|
||||
that image export/OCR works on a renderer-free host.
|
||||
|
||||
## Recommended Fix
|
||||
|
||||
### Immediate P0/P1 work
|
||||
|
||||
1. **Finish W3C/table normalization and fragment stitching.** In the table path
|
||||
only, remove whitespace immediately after `=` and normalize punctuation spacing
|
||||
around `%`, `=`, and commas. Merge one-character continuation fragments when
|
||||
their baseline and x-position overlap a neighboring fragment. Stitch header
|
||||
fragments by column anchor and vertical proximity, preserving true multi-line
|
||||
cells. Add a regression for the exact five bad cells and for ordinary prose so
|
||||
the normalization cannot alter sentence spacing.
|
||||
2. **Merge the two geometry walkers.** The precise C++ walker and pypdf visitor
|
||||
observe the same content stream. Use precise glyph bounds whenever available and
|
||||
visitor text only as a decoding fallback, instead of calibrating visitor width
|
||||
from cross-column span gaps. This removes the remaining unsound estimator rather
|
||||
than merely reducing how often it runs.
|
||||
3. **Add independent layout gates.** Keep hand-labelled table truth, and add page
|
||||
bounding-box/column anchors, header/footer repetition, page-break, image-count,
|
||||
and cross-column order fixtures. Report recall, precision, duplicates, order,
|
||||
and visual similarity separately; never gate only on pypdf-derived text.
|
||||
4. **Make the table structure model reproducible.** Either vendor a checked,
|
||||
permissively licensed Table Transformer/SLANet model with checksum and measured
|
||||
CPU/RAM cost, or explicitly document heuristic-only deployment and route
|
||||
high-risk tables to the page-image fallback.
|
||||
|
||||
### OCR and resource work
|
||||
|
||||
5. Keep one process-wide session per model, but use a dedicated OCR worker pool or
|
||||
process queue so serialized inference cannot block unrelated conversions. Bound
|
||||
image dimensions and decoded bytes before OCR; expose queue wait, model-load,
|
||||
render, inference, and rebuild timings per page.
|
||||
6. Calibrate DPI and preprocessing by document class (deskew, contrast, denoise,
|
||||
adaptive threshold, crop). Use confidence distributions and script detection to
|
||||
decide a second pass; preserve low-confidence text with a warning rather than
|
||||
silently replacing it. Evaluate RapidOCR against PaddleOCR PP-OCRv5, EasyOCR,
|
||||
and a CPU-friendly transformer on a labelled multilingual set before switching.
|
||||
7. Treat the current font service as a heuristic fallback until a real model is
|
||||
shipped. If font fidelity matters, load and checksum an actual ONNX model,
|
||||
batch line crops, and cache one prediction per style/span rather than running
|
||||
line and word inference for every OCR block. Otherwise disable it in the hot
|
||||
conversion path and use extracted PDF font metadata.
|
||||
|
||||
### Fidelity and operations
|
||||
|
||||
8. Preserve the existing page-raster fallback for complex positioned PDFs and make
|
||||
the choice explicit in response metadata (`editable` versus `visual`). For
|
||||
customer-facing editable PDF->Office jobs, use Gotenberg/LibreOffice or a
|
||||
commercial fidelity provider behind the same plugin interface; retain ReportLab
|
||||
for concurrent-safe low-cost fallback.
|
||||
9. Install `pypdfium2` in every supported runtime image and run the raster suite in
|
||||
CI. Keep the C++ engine path as the preferred renderer, but fail loudly when a
|
||||
host has neither renderer nor OCR capability.
|
||||
10. Resolve the font endpoint 204/404 contract, and expose structured error codes
|
||||
for page-cap, OCR-budget, table-confidence, and visual-fallback events.
|
||||
11. Run conversion workers in separate processes/containers for hostile or very
|
||||
large files. Apply per-tenant byte/page/time budgets, temporary-directory
|
||||
quotas, cancellation cleanup, and a dead-letter queue. Track p95/p99 latency,
|
||||
peak RSS, CPU saturation, cache hit rate, and output-fidelity distributions.
|
||||
|
||||
## Implementation
|
||||
|
||||
The following changes are present in the working tree and were validated by the
|
||||
focused suites:
|
||||
|
||||
- `layout/pipeline.py`, `pypdf_geometry.py`, `glyphs.py`, and paragraph/order
|
||||
helpers now use non-whitespace coverage, column-aware row bands, zero-width
|
||||
whitespace handling, improved paragraph grouping, and figure preservation.
|
||||
- `ocr.py` serializes model load/inference safely, supports adaptive Arabic
|
||||
recognition (`CONVERT_OCR_ARABIC_ADAPTIVE`), preserves meaningful symbols such
|
||||
as `$` and `%`, and records whether Arabic was used per call.
|
||||
- `ocr/rebuild.py` raises the page cap to 200, separates page-cap and time-budget
|
||||
warnings/metadata, compacts oversized figures instead of dropping them, and
|
||||
embeds unreconstructable pages as honest page images.
|
||||
- `result_cache.py` canonicalizes effective options, namespace, and automatic
|
||||
versus explicit resolution mode. `doc_cache.py` reuses readers, geometry,
|
||||
rasters, and engine handles inside one conversion.
|
||||
- `routers/convert/jobs.py` and conversion scopes bound admission and shared
|
||||
execution slots; cancellation and deadlines are checked at page/OCR boundaries.
|
||||
- `quality/text_metrics.py` and `quality/scorer.py` now expose Unicode-aware,
|
||||
duplicate-aware, order-sensitive metrics and warnings for low precision/order;
|
||||
the IDM retains component metrics for diagnostics.
|
||||
- PDF->PNG/JPEG/TIFF exports stream pages; TIFF is multi-frame LZW with DPI
|
||||
metadata. DOCX raster fallback is selected from measured complexity/content
|
||||
evidence rather than blindly reflowing every page.
|
||||
- `Dockerfile`, `models/MANIFEST.json`, and `tools/verify_models.py` package and
|
||||
checksum runtime layout/Arabic weights. CORS defaults are explicit and
|
||||
production-safe. `start_convert_gateway.ps1` documents OCR, model, timeout,
|
||||
queue, and cache settings.
|
||||
- `tools/audit_conversion.py` provides repeatable end-to-end measurements. New
|
||||
tests cover metrics, OCR resource guards, cache keys, queue admission, streaming
|
||||
exports, model packaging, and table truth.
|
||||
|
||||
No component was replaced solely because it was open source. The changes preserve
|
||||
the C++/PDFium core and existing plugin interfaces; alternatives are recommended
|
||||
only where the evidence shows a fundamental fidelity or reproducibility gap.
|
||||
|
||||
## Validation Results
|
||||
|
||||
The current state passes the broad focused checks listed in Evidence and runs the
|
||||
full gateway suite to completion without a crash or timeout. The multi-column
|
||||
order probe improved from 189 inversions to 1, and the OCR/page-cap/resource
|
||||
regressions are green. Streaming image export and OOXML validity checks are green.
|
||||
|
||||
The result is not yet release-clean:
|
||||
|
||||
- W3C exact table recall is 0.867 against a 0.90 floor.
|
||||
- Table reading-order/span assertions fail on the same punctuation/header
|
||||
fragments.
|
||||
- Unknown-font endpoint status disagrees with its test contract.
|
||||
- USGS and multipage-table order/recall remain low despite high pixel similarity.
|
||||
- Scan-pack editable text has perfect supplied-token recall but only 0.077
|
||||
precision and 0.143 order similarity; it must not be marketed as accurate OCR
|
||||
without a richer ground truth.
|
||||
|
||||
Expected gains from the recommended work are stated as targets, not measured
|
||||
claims: table stitching should raise W3C recall above the existing 0.90 gate and
|
||||
remove the four known header/punctuation errors; a merged geometry walker should
|
||||
keep the 1-inversion result while eliminating the remaining visitor-width risk;
|
||||
adaptive OCR already demonstrates approximately 55% lower warm-page inference
|
||||
cost on clean Latin scans; process-level isolation should make peak RSS roughly
|
||||
linear per worker instead of allowing simultaneous jobs to share one unbounded
|
||||
address space. Each target must be re-measured on the same corpus before release.
|
||||
|
||||
## Remaining Limitations
|
||||
|
||||
1. PDF->editable DOCX/XLSX cannot be pixel-perfect for arbitrary positioned PDF
|
||||
content. The page-image fallback preserves appearance by giving up editability;
|
||||
the reflow path preserves more semantics by accepting layout drift.
|
||||
2. The default table detector remains heuristic and the table structure model is
|
||||
not shipped. Merged cells, borderless grids, nested tables, and multi-line
|
||||
headers remain the largest content-structure risk.
|
||||
3. RapidOCR is CPU-bound in the measured host, and shared-session locking limits
|
||||
parallel OCR throughput. Whole-document timings vary with cache/model state and
|
||||
Windows scheduling; warm per-page numbers are more stable.
|
||||
4. The active venv lacks `pypdfium2`, so renderer-free Linux behavior and several
|
||||
raster tests were skipped. The Docker dependency declaration and local runtime
|
||||
are currently inconsistent until the venv is rebuilt from `pyproject.toml`.
|
||||
5. No CUDA provider was available in this audit. GPU utilization, GPU memory, and
|
||||
multi-GPU scaling are therefore unmeasured.
|
||||
6. The advertised OpenCLIP font model and FAISS index are metadata placeholders in
|
||||
this checkout. Font predictions are heuristic and should not be used as a
|
||||
fidelity guarantee.
|
||||
7. Visual similarity is a grayscale pixel heuristic over at most the compared
|
||||
pages and, for Office outputs in this audit, uses ReportLab rather than a real
|
||||
LibreOffice/Gotenberg render.
|
||||
8. The quality score is useful for triage but is not a single acceptance metric.
|
||||
Source text can be corrupt, OCR can add legitimate-looking alternatives, and
|
||||
order/layout require independent annotations.
|
||||
9. The repository has a `.git_disabled` metadata directory rather than a normal
|
||||
`.git` checkout, and much of the conversion subsystem is working-tree content.
|
||||
Before/after comparisons in this investigation therefore come from harness
|
||||
snapshots and controlled runs, not a pristine commit diff.
|
||||
|
||||
## Appendix A: Reproduction Commands and Evidence Files
|
||||
|
||||
Run from `pdf/` with the gateway venv:
|
||||
|
||||
```powershell
|
||||
gateway\.venv\Scripts\python.exe tools\audit_conversion.py
|
||||
gateway\.venv\Scripts\python.exe tools\audit_conversion.py --no-scan
|
||||
gateway\.venv\Scripts\python.exe -m pytest gateway\tests\convert -q
|
||||
gateway\.venv\Scripts\python.exe -m pytest gateway\tests -q
|
||||
```
|
||||
|
||||
Primary evidence and implementation references:
|
||||
|
||||
- `audit_results.json` - machine-readable audit output (the checked-in file may
|
||||
be targeted to one case; regenerate for the full matrix).
|
||||
- `tools/audit_conversion.py` - timing, resource, content, table, and visual
|
||||
measurement harness.
|
||||
- `corpus/convert/table_truth.json` and `corpus/convert/baselines.json` - external
|
||||
table labels and suite floors.
|
||||
- `gateway/app/services/convert/layout/pipeline.py` - routing, IDM construction,
|
||||
OCR budgeting, and page-image fallback.
|
||||
- `gateway/app/services/convert/layout/pypdf_geometry.py` and `glyphs.py` - glyph
|
||||
walkers, line grouping, and geometry evidence.
|
||||
- `gateway/app/services/convert/ocr/rebuild.py` and `gateway/app/services/ocr.py` -
|
||||
raster/OCR/rebuild path and Arabic adaptive policy.
|
||||
- `gateway/app/services/convert/formatters/docx_formatter.py`,
|
||||
`writers/pdf_from_docx.py`, and `backends/pdf_exporter.py` - Office writers and
|
||||
Gotenberg/ReportLab selection.
|
||||
- `gateway/pyproject.toml`, `gateway/Dockerfile`, `gateway/models/MANIFEST.json`,
|
||||
and `gateway/start_convert_gateway.ps1` - dependency, packaging, and runtime
|
||||
configuration evidence.
|
||||
|
||||
**Bottom line:** the engine now has honest diagnostics and materially better
|
||||
multi-column recovery, but it should be released with explicit fidelity classes,
|
||||
the W3C table defect fixed, renderer/OCR dependencies verified in CI, and separate
|
||||
content, structure, and visual acceptance gates.
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -1,6 +1,74 @@
|
||||
# Python bindings for PdfEngine using pybind11
|
||||
#
|
||||
|
||||
# The .pyd ABI tag (cp312 vs cp313) must match the interpreter that runs
|
||||
# the FastAPI gateway. Prefer gateway/.venv when it exists so `import pdfengine`
|
||||
# works in uvicorn without a second Python install.
|
||||
if(WIN32)
|
||||
set(_gateway_python "${CMAKE_SOURCE_DIR}/gateway/.venv/Scripts/python.exe")
|
||||
else()
|
||||
set(_gateway_python "${CMAKE_SOURCE_DIR}/gateway/.venv/bin/python")
|
||||
endif()
|
||||
if(EXISTS "${_gateway_python}")
|
||||
set(Python_EXECUTABLE "${_gateway_python}" CACHE FILEPATH "Python for pybind11" FORCE)
|
||||
set(Python3_EXECUTABLE "${_gateway_python}" CACHE FILEPATH "Python3 for pybind11" FORCE)
|
||||
set(PYTHON_EXECUTABLE "${_gateway_python}" CACHE FILEPATH "Python for pybind11 (legacy)" FORCE)
|
||||
message(STATUS "pybind11: using gateway venv ${_gateway_python}")
|
||||
endif()
|
||||
|
||||
# Drop stale FindPythonLibsNew cache from a previous configure (it can keep
|
||||
# PYTHON_MODULE_EXTENSION=.cp313-*.pyd and python313.lib even after we switch
|
||||
# the interpreter to the gateway 3.12 venv).
|
||||
unset(PYTHON_MODULE_EXTENSION CACHE)
|
||||
unset(PYTHON_LIBRARIES CACHE)
|
||||
unset(PYTHON_INCLUDE_DIRS CACHE)
|
||||
unset(PYTHON_VERSION CACHE)
|
||||
unset(PYTHON_VERSION_MAJOR CACHE)
|
||||
unset(PYTHON_VERSION_MINOR CACHE)
|
||||
unset(PYTHON_IS_DEBUG CACHE)
|
||||
unset(PYTHON_MODULE_PREFIX CACHE)
|
||||
unset(PYTHON_MODULE_DEBUG_POSTFIX CACHE)
|
||||
|
||||
if(Python_EXECUTABLE)
|
||||
execute_process(
|
||||
COMMAND "${Python_EXECUTABLE}" -c "import sys, sysconfig, pathlib; base=pathlib.Path(sys.base_prefix); ver=f'{sys.version_info.major}{sys.version_info.minor}'; print(sysconfig.get_config_var('EXT_SUFFIX') or ''); print(sysconfig.get_path('include')); print(base / 'libs' / f'python{ver}.lib')"
|
||||
OUTPUT_VARIABLE _pdfengine_py_info
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
ERROR_QUIET
|
||||
)
|
||||
string(REPLACE "\r" "" _pdfengine_py_info "${_pdfengine_py_info}")
|
||||
string(REPLACE "\n" ";" _pdfengine_py_info_lines "${_pdfengine_py_info}")
|
||||
list(LENGTH _pdfengine_py_info_lines _pdfengine_py_info_len)
|
||||
if(_pdfengine_py_info_len GREATER_EQUAL 3)
|
||||
list(GET _pdfengine_py_info_lines 0 _pdfengine_py_ext)
|
||||
list(GET _pdfengine_py_info_lines 1 _pdfengine_py_inc)
|
||||
list(GET _pdfengine_py_info_lines 2 _pdfengine_py_lib)
|
||||
file(TO_CMAKE_PATH "${_pdfengine_py_ext}" _pdfengine_py_ext)
|
||||
file(TO_CMAKE_PATH "${_pdfengine_py_inc}" _pdfengine_py_inc)
|
||||
file(TO_CMAKE_PATH "${_pdfengine_py_lib}" _pdfengine_py_lib)
|
||||
if(_pdfengine_py_ext)
|
||||
set(PYTHON_MODULE_EXTENSION "${_pdfengine_py_ext}" CACHE INTERNAL "Python extension suffix")
|
||||
message(STATUS "pybind11: extension suffix ${_pdfengine_py_ext}")
|
||||
endif()
|
||||
# vcpkg ships a static python312.lib; linking the .pyd against that
|
||||
# crashes when uvicorn loads it (no python312.dll import). Use the
|
||||
# interpreter's own import lib / headers instead.
|
||||
if(EXISTS "${_pdfengine_py_lib}" AND EXISTS "${_pdfengine_py_inc}/Python.h")
|
||||
set(Python_LIBRARY "${_pdfengine_py_lib}" CACHE FILEPATH "Python import library" FORCE)
|
||||
set(Python_LIBRARIES "${_pdfengine_py_lib}" CACHE FILEPATH "Python import library" FORCE)
|
||||
set(Python_INCLUDE_DIR "${_pdfengine_py_inc}" CACHE PATH "Python headers" FORCE)
|
||||
set(Python3_LIBRARY "${_pdfengine_py_lib}" CACHE FILEPATH "Python3 import library" FORCE)
|
||||
set(Python3_INCLUDE_DIR "${_pdfengine_py_inc}" CACHE PATH "Python3 headers" FORCE)
|
||||
get_filename_component(_pdfengine_py_root "${_pdfengine_py_inc}" DIRECTORY)
|
||||
set(Python_ROOT_DIR "${_pdfengine_py_root}" CACHE PATH "CPython install root" FORCE)
|
||||
message(STATUS "pybind11: CPython lib ${_pdfengine_py_lib}")
|
||||
message(STATUS "pybind11: CPython include ${_pdfengine_py_inc}")
|
||||
endif()
|
||||
endif()
|
||||
endif()
|
||||
|
||||
set(PYBIND11_FINDPYTHON ON)
|
||||
|
||||
find_package(pybind11 CONFIG REQUIRED)
|
||||
|
||||
# Declare the python module target. We name the target pdfengine_py to avoid
|
||||
@@ -12,9 +80,23 @@ set_target_properties(pdfengine_py PROPERTIES
|
||||
OUTPUT_NAME "pdfengine"
|
||||
ARCHIVE_OUTPUT_NAME "pdfengine_py_import"
|
||||
)
|
||||
if(_pdfengine_py_ext)
|
||||
set_target_properties(pdfengine_py PROPERTIES SUFFIX "${_pdfengine_py_ext}")
|
||||
endif()
|
||||
|
||||
target_link_libraries(pdfengine_py PRIVATE pdfengine::pdfengine)
|
||||
|
||||
# OBJECT-library usage requirements are not always enough on MSVC for the
|
||||
# python module (it also consumes $<TARGET_OBJECTS:pdfengine>). Repeat the
|
||||
# libraries the engine objects reference so the .pyd link line is complete.
|
||||
target_link_libraries(pdfengine_py PRIVATE qpdf::libqpdf ZLIB::ZLIB JPEG::JPEG)
|
||||
if(PDFENGINE_WITH_PDFIUM)
|
||||
target_link_libraries(pdfengine_py PRIVATE pdfium::pdfium)
|
||||
endif()
|
||||
if(_pdfengine_py_lib AND EXISTS "${_pdfengine_py_lib}")
|
||||
target_link_libraries(pdfengine_py PRIVATE "${_pdfengine_py_lib}")
|
||||
endif()
|
||||
|
||||
target_include_directories(pdfengine_py PRIVATE
|
||||
"${CMAKE_SOURCE_DIR}/engine/src"
|
||||
)
|
||||
@@ -23,6 +105,28 @@ if(MSVC)
|
||||
target_link_options(pdfengine_py PRIVATE "/FORCE:MULTIPLE")
|
||||
endif()
|
||||
|
||||
# vcpkg's find_package(Python) wrapper still injects its static python312.lib.
|
||||
# A pybind module must link only the host interpreter's import lib (python312.dll).
|
||||
foreach(_py_tgt IN ITEMS Python::Module Python::Python Python3::Module Python3::Python)
|
||||
if(TARGET ${_py_tgt} AND _pdfengine_py_lib)
|
||||
set_property(TARGET ${_py_tgt} PROPERTY INTERFACE_LINK_LIBRARIES "${_pdfengine_py_lib}")
|
||||
set_property(TARGET ${_py_tgt} PROPERTY IMPORTED_LOCATION "${_pdfengine_py_lib}")
|
||||
set_property(TARGET ${_py_tgt} PROPERTY IMPORTED_IMPLIB "${_pdfengine_py_lib}")
|
||||
endif()
|
||||
endforeach()
|
||||
get_target_property(_pdfengine_py_link pdfengine_py LINK_LIBRARIES)
|
||||
if(_pdfengine_py_link)
|
||||
set(_pdfengine_py_kept "")
|
||||
foreach(_lib IN LISTS _pdfengine_py_link)
|
||||
if(_lib MATCHES "vcpkg_installed" AND _lib MATCHES "[Pp]ython")
|
||||
message(STATUS "pybind11: dropping vcpkg CPython ${_lib}")
|
||||
continue()
|
||||
endif()
|
||||
list(APPEND _pdfengine_py_kept "${_lib}")
|
||||
endforeach()
|
||||
set_property(TARGET pdfengine_py PROPERTY LINK_LIBRARIES "${_pdfengine_py_kept}")
|
||||
endif()
|
||||
|
||||
# Set warnings and sanitizers for the bindings module
|
||||
pdfengine_set_warnings(pdfengine_py)
|
||||
pdfengine_enable_sanitizers(pdfengine_py)
|
||||
|
||||
+18
-4
@@ -35,9 +35,16 @@ This document serves as the single source of truth for commands across our devel
|
||||
```powershell
|
||||
powershell -ExecutionPolicy Bypass -File scripts/bootstrap.ps1
|
||||
cd gateway
|
||||
python -m venv .venv
|
||||
|
||||
# Fast path with uv (recommended: 10x faster)
|
||||
uv venv
|
||||
.venv\Scripts\Activate.ps1
|
||||
pip install -e ".[dev]"
|
||||
uv pip install -e ".[dev]"
|
||||
|
||||
# (Alternative fallback with standard pip)
|
||||
# python -m venv .venv
|
||||
# .venv\Scripts\Activate.ps1
|
||||
# pip install -e ".[dev]"
|
||||
cd ..
|
||||
```
|
||||
|
||||
@@ -92,9 +99,16 @@ npm run build
|
||||
```powershell
|
||||
powershell -ExecutionPolicy Bypass -File scripts/bootstrap.ps1
|
||||
cd gateway
|
||||
python -m venv .venv
|
||||
|
||||
# Fast path with uv (recommended: 10x faster)
|
||||
uv venv
|
||||
.venv\Scripts\Activate.ps1
|
||||
pip install -e ".[dev]"
|
||||
uv pip install -e ".[dev]"
|
||||
|
||||
# (Alternative fallback with standard pip)
|
||||
# python -m venv .venv
|
||||
# .venv\Scripts\Activate.ps1
|
||||
# pip install -e ".[dev]"
|
||||
cd ..
|
||||
```
|
||||
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
# Public free eval corpus (Wave-1)
|
||||
|
||||
Licensed third-party + DocQube self-generated smoke for PDF↔DOCX/XLSX, Path A, images, OCR.
|
||||
|
||||
## Quick start (Windows)
|
||||
|
||||
```powershell
|
||||
cd pdf
|
||||
..\gateway\.venv\Scripts\python.exe scripts\convert\generate_public_smoke_fixtures.py
|
||||
..\gateway\.venv\Scripts\python.exe scripts\convert\score_public_pack.py
|
||||
```
|
||||
|
||||
Optional third-party fetch (after LICENSE_CHECKLIST sign-off):
|
||||
|
||||
```powershell
|
||||
..\gateway\.venv\Scripts\python.exe scripts\convert\fetch_public_eval.py
|
||||
```
|
||||
|
||||
## Layout
|
||||
|
||||
- `manifests/` — per-family case definitions
|
||||
- `pack_public.json` — unified pytest pack
|
||||
- `samples/` — gitignored binaries (generated or fetched)
|
||||
- `reports/` — baseline / A/B JSON
|
||||
|
||||
## Licenses
|
||||
|
||||
See `../third_party/NOTICE` and `../third_party/LICENSE_CHECKLIST.md`.
|
||||
Default fixtures are **self-generated** (no third-party license). DocLayNet/CORD/SROIE/PubTabNet entries in manifests are placeholders until fetched.
|
||||
@@ -0,0 +1,23 @@
|
||||
{
|
||||
"cases": [
|
||||
{
|
||||
"id": "cord_placeholder",
|
||||
"family": "cord",
|
||||
"license": "CC-BY-4.0",
|
||||
"source": "public/samples/cord/.gitkeep",
|
||||
"pair": "pdf_txt",
|
||||
"bars": {
|
||||
"cell_match_min": null,
|
||||
"prose_tokens": [],
|
||||
"text_recall_min": 0.5
|
||||
},
|
||||
"gt": {
|
||||
"grid": null,
|
||||
"text_file": null,
|
||||
"regions_coco": null
|
||||
},
|
||||
"notes": "placeholder until fetch_public_eval.py",
|
||||
"optional": true
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,49 @@
|
||||
{
|
||||
"cases": [
|
||||
{
|
||||
"id": "synth_digital_001",
|
||||
"family": "doclaynet",
|
||||
"license": "DocQube",
|
||||
"source": "public/samples/doclaynet/synth_digital_001.pdf",
|
||||
"pair": "pdf_docx",
|
||||
"bars": {
|
||||
"cell_match_min": null,
|
||||
"prose_tokens": [],
|
||||
"text_recall_min": 0.95
|
||||
},
|
||||
"gt": {
|
||||
"grid": null,
|
||||
"text_file": null,
|
||||
"regions_coco": null
|
||||
},
|
||||
"expected_tokens": [
|
||||
"Digital",
|
||||
"Report",
|
||||
"AlphaBridge"
|
||||
],
|
||||
"notes": "self-gen digital layout stand-in"
|
||||
},
|
||||
{
|
||||
"id": "synth_multicol_001",
|
||||
"family": "doclaynet",
|
||||
"license": "DocQube",
|
||||
"source": "public/samples/doclaynet/synth_multicol_001.pdf",
|
||||
"pair": "pdf_docx",
|
||||
"bars": {
|
||||
"cell_match_min": null,
|
||||
"prose_tokens": [],
|
||||
"text_recall_min": 0.8
|
||||
},
|
||||
"gt": {
|
||||
"grid": null,
|
||||
"text_file": null,
|
||||
"regions_coco": null
|
||||
},
|
||||
"expected_tokens": [
|
||||
"ColumnLeftOne",
|
||||
"ColumnRightOne"
|
||||
],
|
||||
"notes": "two-column reading-order stress"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,22 @@
|
||||
{
|
||||
"cases": [
|
||||
{
|
||||
"id": "images_png_pdf",
|
||||
"family": "images",
|
||||
"license": "DocQube",
|
||||
"source": "public/samples/images/smoke.png",
|
||||
"pair": "png_pdf",
|
||||
"bars": {
|
||||
"cell_match_min": null,
|
||||
"prose_tokens": [],
|
||||
"text_recall_min": null
|
||||
},
|
||||
"gt": {
|
||||
"grid": null,
|
||||
"text_file": null,
|
||||
"regions_coco": null
|
||||
},
|
||||
"notes": "PNG\u2192PDF"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,45 @@
|
||||
{
|
||||
"cases": [
|
||||
{
|
||||
"id": "path_a_docx",
|
||||
"family": "path_a",
|
||||
"license": "DocQube",
|
||||
"source": "public/samples/path_a/path_a_sample.docx",
|
||||
"pair": "docx_pdf",
|
||||
"bars": {
|
||||
"cell_match_min": null,
|
||||
"prose_tokens": [
|
||||
"PathAParagraphToken"
|
||||
],
|
||||
"text_recall_min": null
|
||||
},
|
||||
"gt": {
|
||||
"grid": null,
|
||||
"text_file": null,
|
||||
"regions_coco": null
|
||||
},
|
||||
"notes": "DOCX\u2192PDF Path A"
|
||||
},
|
||||
{
|
||||
"id": "path_a_xlsx",
|
||||
"family": "path_a",
|
||||
"license": "DocQube",
|
||||
"source": "public/samples/path_a/path_a_sample.xlsx",
|
||||
"pair": "xlsx_pdf",
|
||||
"bars": {
|
||||
"cell_match_min": null,
|
||||
"prose_tokens": [
|
||||
"HeaderA",
|
||||
"CellAlpha"
|
||||
],
|
||||
"text_recall_min": null
|
||||
},
|
||||
"gt": {
|
||||
"grid": null,
|
||||
"text_file": null,
|
||||
"regions_coco": null
|
||||
},
|
||||
"notes": "XLSX\u2192PDF Path A"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,23 @@
|
||||
{
|
||||
"cases": [
|
||||
{
|
||||
"id": "sroie_placeholder",
|
||||
"family": "sroie",
|
||||
"license": "CC-BY-4.0",
|
||||
"source": "public/samples/sroie/.gitkeep",
|
||||
"pair": "pdf_txt",
|
||||
"bars": {
|
||||
"cell_match_min": null,
|
||||
"prose_tokens": [],
|
||||
"text_recall_min": 0.5
|
||||
},
|
||||
"gt": {
|
||||
"grid": null,
|
||||
"text_file": null,
|
||||
"regions_coco": null
|
||||
},
|
||||
"notes": "placeholder until fetch_public_eval.py",
|
||||
"optional": true
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,42 @@
|
||||
{
|
||||
"cases": [
|
||||
{
|
||||
"id": "synth_invoice_001",
|
||||
"family": "synthetic_invoice",
|
||||
"license": "DocQube",
|
||||
"source": "public/samples/invoices/synth_invoice_001.pdf",
|
||||
"pair": "pdf_xlsx",
|
||||
"bars": {
|
||||
"cell_match_min": 0.85,
|
||||
"prose_tokens": [
|
||||
"Invoice",
|
||||
"AcmeCorp",
|
||||
"Total"
|
||||
],
|
||||
"text_recall_min": null
|
||||
},
|
||||
"gt": {
|
||||
"grid": [
|
||||
[
|
||||
"Desc",
|
||||
"Qty",
|
||||
"Amount"
|
||||
],
|
||||
[
|
||||
"Service",
|
||||
"1",
|
||||
"100"
|
||||
],
|
||||
[
|
||||
"Tax",
|
||||
"1",
|
||||
"10"
|
||||
]
|
||||
],
|
||||
"text_file": null,
|
||||
"regions_coco": null
|
||||
},
|
||||
"notes": "self-gen invoice"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,218 @@
|
||||
{
|
||||
"cases": [
|
||||
{
|
||||
"id": "synth_invoice_001",
|
||||
"family": "synthetic_invoice",
|
||||
"license": "DocQube",
|
||||
"source": "public/samples/invoices/synth_invoice_001.pdf",
|
||||
"pair": "pdf_xlsx",
|
||||
"bars": {
|
||||
"cell_match_min": 0.85,
|
||||
"prose_tokens": [
|
||||
"Invoice",
|
||||
"AcmeCorp",
|
||||
"Total"
|
||||
],
|
||||
"text_recall_min": null
|
||||
},
|
||||
"gt": {
|
||||
"grid": [
|
||||
[
|
||||
"Desc",
|
||||
"Qty",
|
||||
"Amount"
|
||||
],
|
||||
[
|
||||
"Service",
|
||||
"1",
|
||||
"100"
|
||||
],
|
||||
[
|
||||
"Tax",
|
||||
"1",
|
||||
"10"
|
||||
]
|
||||
],
|
||||
"text_file": null,
|
||||
"regions_coco": null
|
||||
},
|
||||
"notes": "self-gen invoice"
|
||||
},
|
||||
{
|
||||
"id": "synth_table_001",
|
||||
"family": "pubtabnet",
|
||||
"license": "DocQube",
|
||||
"source": "public/samples/pubtab/synth_table_001.pdf",
|
||||
"pair": "pdf_xlsx",
|
||||
"bars": {
|
||||
"cell_match_min": 0.9,
|
||||
"prose_tokens": [],
|
||||
"text_recall_min": null
|
||||
},
|
||||
"gt": {
|
||||
"grid": [
|
||||
[
|
||||
"Item",
|
||||
"Qty",
|
||||
"Price"
|
||||
],
|
||||
[
|
||||
"Widget",
|
||||
"2",
|
||||
"10"
|
||||
],
|
||||
[
|
||||
"Gadget",
|
||||
"1",
|
||||
"25"
|
||||
]
|
||||
],
|
||||
"text_file": null,
|
||||
"regions_coco": null
|
||||
},
|
||||
"notes": "self-gen stand-in for PubTabNet until fetch"
|
||||
},
|
||||
{
|
||||
"id": "synth_digital_001",
|
||||
"family": "doclaynet",
|
||||
"license": "DocQube",
|
||||
"source": "public/samples/doclaynet/synth_digital_001.pdf",
|
||||
"pair": "pdf_docx",
|
||||
"bars": {
|
||||
"cell_match_min": null,
|
||||
"prose_tokens": [],
|
||||
"text_recall_min": 0.95
|
||||
},
|
||||
"gt": {
|
||||
"grid": null,
|
||||
"text_file": null,
|
||||
"regions_coco": null
|
||||
},
|
||||
"expected_tokens": [
|
||||
"Digital",
|
||||
"Report",
|
||||
"AlphaBridge"
|
||||
],
|
||||
"notes": "self-gen digital layout stand-in"
|
||||
},
|
||||
{
|
||||
"id": "synth_multicol_001",
|
||||
"family": "doclaynet",
|
||||
"license": "DocQube",
|
||||
"source": "public/samples/doclaynet/synth_multicol_001.pdf",
|
||||
"pair": "pdf_docx",
|
||||
"bars": {
|
||||
"cell_match_min": null,
|
||||
"prose_tokens": [],
|
||||
"text_recall_min": 0.8
|
||||
},
|
||||
"gt": {
|
||||
"grid": null,
|
||||
"text_file": null,
|
||||
"regions_coco": null
|
||||
},
|
||||
"expected_tokens": [
|
||||
"ColumnLeftOne",
|
||||
"ColumnRightOne"
|
||||
],
|
||||
"notes": "two-column reading-order stress"
|
||||
},
|
||||
{
|
||||
"id": "path_a_docx",
|
||||
"family": "path_a",
|
||||
"license": "DocQube",
|
||||
"source": "public/samples/path_a/path_a_sample.docx",
|
||||
"pair": "docx_pdf",
|
||||
"bars": {
|
||||
"cell_match_min": null,
|
||||
"prose_tokens": [
|
||||
"PathAParagraphToken"
|
||||
],
|
||||
"text_recall_min": null
|
||||
},
|
||||
"gt": {
|
||||
"grid": null,
|
||||
"text_file": null,
|
||||
"regions_coco": null
|
||||
},
|
||||
"notes": "DOCX\u2192PDF Path A"
|
||||
},
|
||||
{
|
||||
"id": "path_a_xlsx",
|
||||
"family": "path_a",
|
||||
"license": "DocQube",
|
||||
"source": "public/samples/path_a/path_a_sample.xlsx",
|
||||
"pair": "xlsx_pdf",
|
||||
"bars": {
|
||||
"cell_match_min": null,
|
||||
"prose_tokens": [
|
||||
"HeaderA",
|
||||
"CellAlpha"
|
||||
],
|
||||
"text_recall_min": null
|
||||
},
|
||||
"gt": {
|
||||
"grid": null,
|
||||
"text_file": null,
|
||||
"regions_coco": null
|
||||
},
|
||||
"notes": "XLSX\u2192PDF Path A"
|
||||
},
|
||||
{
|
||||
"id": "images_png_pdf",
|
||||
"family": "images",
|
||||
"license": "DocQube",
|
||||
"source": "public/samples/images/smoke.png",
|
||||
"pair": "png_pdf",
|
||||
"bars": {
|
||||
"cell_match_min": null,
|
||||
"prose_tokens": [],
|
||||
"text_recall_min": null
|
||||
},
|
||||
"gt": {
|
||||
"grid": null,
|
||||
"text_file": null,
|
||||
"regions_coco": null
|
||||
},
|
||||
"notes": "PNG\u2192PDF"
|
||||
},
|
||||
{
|
||||
"id": "cord_placeholder",
|
||||
"family": "cord",
|
||||
"license": "CC-BY-4.0",
|
||||
"source": "public/samples/cord/.gitkeep",
|
||||
"pair": "pdf_txt",
|
||||
"bars": {
|
||||
"cell_match_min": null,
|
||||
"prose_tokens": [],
|
||||
"text_recall_min": 0.5
|
||||
},
|
||||
"gt": {
|
||||
"grid": null,
|
||||
"text_file": null,
|
||||
"regions_coco": null
|
||||
},
|
||||
"notes": "placeholder until fetch_public_eval.py",
|
||||
"optional": true
|
||||
},
|
||||
{
|
||||
"id": "sroie_placeholder",
|
||||
"family": "sroie",
|
||||
"license": "CC-BY-4.0",
|
||||
"source": "public/samples/sroie/.gitkeep",
|
||||
"pair": "pdf_txt",
|
||||
"bars": {
|
||||
"cell_match_min": null,
|
||||
"prose_tokens": [],
|
||||
"text_recall_min": 0.5
|
||||
},
|
||||
"gt": {
|
||||
"grid": null,
|
||||
"text_file": null,
|
||||
"regions_coco": null
|
||||
},
|
||||
"notes": "placeholder until fetch_public_eval.py",
|
||||
"optional": true
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"ml_off_passed": 7,
|
||||
"ml_on_passed": 7,
|
||||
"deltas": [],
|
||||
"regressions": 0
|
||||
}
|
||||
@@ -0,0 +1,64 @@
|
||||
{
|
||||
"ml": "0",
|
||||
"results": [
|
||||
{
|
||||
"id": "synth_invoice_001",
|
||||
"pair": "pdf_xlsx",
|
||||
"ml": "0",
|
||||
"pass": true,
|
||||
"cell_match": 1.0
|
||||
},
|
||||
{
|
||||
"id": "synth_table_001",
|
||||
"pair": "pdf_xlsx",
|
||||
"ml": "0",
|
||||
"pass": true,
|
||||
"cell_match": 1.0
|
||||
},
|
||||
{
|
||||
"id": "synth_digital_001",
|
||||
"pair": "pdf_docx",
|
||||
"ml": "0",
|
||||
"pass": true,
|
||||
"text_recall": 1.0
|
||||
},
|
||||
{
|
||||
"id": "synth_multicol_001",
|
||||
"pair": "pdf_docx",
|
||||
"ml": "0",
|
||||
"pass": true,
|
||||
"text_recall": 1.0
|
||||
},
|
||||
{
|
||||
"id": "path_a_docx",
|
||||
"pair": "docx_pdf",
|
||||
"ml": "0",
|
||||
"pass": true
|
||||
},
|
||||
{
|
||||
"id": "path_a_xlsx",
|
||||
"pair": "xlsx_pdf",
|
||||
"ml": "0",
|
||||
"pass": true
|
||||
},
|
||||
{
|
||||
"id": "images_png_pdf",
|
||||
"pair": "png_pdf",
|
||||
"ml": "0",
|
||||
"pass": true
|
||||
},
|
||||
{
|
||||
"id": "cord_placeholder",
|
||||
"skipped": true,
|
||||
"reason": "missing_source"
|
||||
},
|
||||
{
|
||||
"id": "sroie_placeholder",
|
||||
"skipped": true,
|
||||
"reason": "missing_source"
|
||||
}
|
||||
],
|
||||
"passed": 7,
|
||||
"failed": 0,
|
||||
"skipped": 2
|
||||
}
|
||||
@@ -0,0 +1,728 @@
|
||||
[
|
||||
{
|
||||
"file": "digital_report_001.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36742,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.9025",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "digital_report_001.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 4947,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.9025",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "govinfo.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 38508,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.8932",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "govinfo.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 7086,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.8932",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "govinfo_small.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 38508,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.8932",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "govinfo_small.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 7085,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.8932",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "invoice_001.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36748,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.7275",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "invoice_001.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 4980,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.6132",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "invoice_002.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36742,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.7275",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "invoice_002.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 4963,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.6050",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "irs_f1040_sample.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 41114,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.8993",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "irs_f1040_sample.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 10851,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.8993",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "merged_001.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36833,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.7275",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "merged_001.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 5007,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.9550",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | layout_ml=onnx | Page 1: layout_ml=onnx regions=0 | Capped at medium: PDF-\u003eOffice "
|
||||
},
|
||||
{
|
||||
"file": "mozilla_pdf_spec_excerpt.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 74515,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.9076",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "mozilla_pdf_spec_excerpt.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 56039,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.9076",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "multicolumn_001.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36671,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.8875",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "multicolumn_001.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 4896,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.8875",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "multicolumn_002.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36671,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.8875",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "multicolumn_002.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 4896,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.8875",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "multipage_table_001.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36835,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.7125",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "multipage_table_001.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 5593,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.8000",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | layout_ml=onnx | Page 1: layout_ml=onnx regions=0 | Page 2: layout_ml=onnx regions"
|
||||
},
|
||||
{
|
||||
"file": "simple_table_001.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36701,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.7275",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "simple_table_001.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 4914,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.6050",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "table_gap_001.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36668,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.8875",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: glyph/engine text we"
|
||||
},
|
||||
{
|
||||
"file": "table_gap_001.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 4898,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.8875",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "table_pipe_001.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36847,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.7275",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "table_pipe_001.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 5033,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.9550",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | layout_ml=onnx | Page 1: layout_ml=onnx regions=0 | Capped at medium: PDF-\u003eOffice "
|
||||
},
|
||||
{
|
||||
"file": "table_simple_001.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36854,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.7275",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "table_simple_001.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 5051,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.9550",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | layout_ml=onnx | Page 1: layout_ml=onnx regions=0 | Capped at medium: PDF-\u003eOffice "
|
||||
},
|
||||
{
|
||||
"file": "usgs_factsheet.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 3065773,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.6825",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "usgs_factsheet.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 15077,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.5600",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "w3c_pdf_table.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 37075,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.6096",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "w3c_pdf_table.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 5335,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.5431",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | layout_ml=onnx | Page 1: layout_ml=onnx regions=2"
|
||||
},
|
||||
{
|
||||
"file": "w3c_table.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 37075,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.6096",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "w3c_table.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 5336,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.5431",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | layout_ml=onnx | Page 1: layout_ml=onnx regions=2"
|
||||
},
|
||||
{
|
||||
"file": "synth_digital_001.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36742,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.9025",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "synth_digital_001.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 4947,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.9025",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "synth_multicol_001.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36739,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.9175",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "synth_multicol_001.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 4933,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.9175",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "synth_invoice_001.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36748,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.7275",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "synth_invoice_001.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 4981,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.6132",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "synth_table_001.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36701,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.7275",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "synth_table_001.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 4915,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.6050",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "digital_report_001.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36742,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.9025",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "digital_report_001.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 4946,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.9025",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "invoice_001.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36748,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.7275",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "invoice_001.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 4981,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.6132",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "invoice_002.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36742,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.7275",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "invoice_002.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 4965,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.6050",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "merged_001.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36833,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.7275",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "merged_001.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 5006,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.9550",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | layout_ml=onnx | Page 1: layout_ml=onnx regions=0 | Capped at medium: PDF-\u003eOffice "
|
||||
},
|
||||
{
|
||||
"file": "multicolumn_001.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36671,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.8875",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "multicolumn_001.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 4895,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.8875",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "multicolumn_002.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36671,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.8875",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "multicolumn_002.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 4895,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.8875",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "multipage_table_001.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36835,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.7125",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "multipage_table_001.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 5593,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.8000",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | layout_ml=onnx | Page 1: layout_ml=onnx regions=0 | Page 2: layout_ml=onnx regions"
|
||||
},
|
||||
{
|
||||
"file": "simple_table_001.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36701,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.7275",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "simple_table_001.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 4914,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.6050",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "table_gap_001.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36668,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.8875",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: glyph/engine text we"
|
||||
},
|
||||
{
|
||||
"file": "table_gap_001.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 4898,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.8875",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | No table structure detected; exported paragraphs as a single column. | layout_ml=o"
|
||||
},
|
||||
{
|
||||
"file": "table_pipe_001.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36847,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.7275",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "table_pipe_001.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 5032,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.9550",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | layout_ml=onnx | Page 1: layout_ml=onnx regions=0 | Capped at medium: PDF-\u003eOffice "
|
||||
},
|
||||
{
|
||||
"file": "table_simple_001.pdf",
|
||||
"target": "docx",
|
||||
"status": 200,
|
||||
"bytes": 36854,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.7275",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Layout fidelity is lossy; headers/footers/fonts are not fully preserved. | layout_ml=onnx | Page 1: layout_ml=onnx regio"
|
||||
},
|
||||
{
|
||||
"file": "table_simple_001.pdf",
|
||||
"target": "xlsx",
|
||||
"status": 200,
|
||||
"bytes": 5051,
|
||||
"magic": "PK",
|
||||
"fidelity": "lossy",
|
||||
"quality": "0.9550",
|
||||
"ml": "PASS_ML",
|
||||
"warn": "Table detection is heuristic/lossy. | layout_ml=onnx | Page 1: layout_ml=onnx regions=0 | Capped at medium: PDF-\u003eOffice "
|
||||
}
|
||||
]
|
||||
@@ -0,0 +1,64 @@
|
||||
{
|
||||
"ml": "1",
|
||||
"results": [
|
||||
{
|
||||
"id": "synth_invoice_001",
|
||||
"pair": "pdf_xlsx",
|
||||
"ml": "1",
|
||||
"pass": true,
|
||||
"cell_match": 1.0
|
||||
},
|
||||
{
|
||||
"id": "synth_table_001",
|
||||
"pair": "pdf_xlsx",
|
||||
"ml": "1",
|
||||
"pass": true,
|
||||
"cell_match": 1.0
|
||||
},
|
||||
{
|
||||
"id": "synth_digital_001",
|
||||
"pair": "pdf_docx",
|
||||
"ml": "1",
|
||||
"pass": true,
|
||||
"text_recall": 1.0
|
||||
},
|
||||
{
|
||||
"id": "synth_multicol_001",
|
||||
"pair": "pdf_docx",
|
||||
"ml": "1",
|
||||
"pass": true,
|
||||
"text_recall": 1.0
|
||||
},
|
||||
{
|
||||
"id": "path_a_docx",
|
||||
"pair": "docx_pdf",
|
||||
"ml": "1",
|
||||
"pass": true
|
||||
},
|
||||
{
|
||||
"id": "path_a_xlsx",
|
||||
"pair": "xlsx_pdf",
|
||||
"ml": "1",
|
||||
"pass": true
|
||||
},
|
||||
{
|
||||
"id": "images_png_pdf",
|
||||
"pair": "png_pdf",
|
||||
"ml": "1",
|
||||
"pass": true
|
||||
},
|
||||
{
|
||||
"id": "cord_placeholder",
|
||||
"skipped": true,
|
||||
"reason": "missing_source"
|
||||
},
|
||||
{
|
||||
"id": "sroie_placeholder",
|
||||
"skipped": true,
|
||||
"reason": "missing_source"
|
||||
}
|
||||
],
|
||||
"passed": 7,
|
||||
"failed": 0,
|
||||
"skipped": 2
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
# Public pack triage (A/B)
|
||||
|
||||
- ML off passed: 7
|
||||
- ML on passed: 7
|
||||
- Regressions: 0
|
||||
|
||||
## Suggested climb owners
|
||||
|
||||
| Failure pattern | Climb | Files |
|
||||
| --- | --- | --- |
|
||||
| Invoice / grid cell | M2 | layout/tables.py, quality/table_metrics.py |
|
||||
| Multi-column order | M3 | reading_order.py, ml_merge.py |
|
||||
| Scan / OCR empty | M4 | ocr/rebuild.py, rapid_adapter.py |
|
||||
| Missing figures | M5 | layout/images.py |
|
||||
| Header/footer in body | S1 | headers_footers.py, docx_formatter.py |
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
# Wave-1 production smoke pack (curated)
|
||||
|
||||
Internal curated PDFs used for **shippable Wave-1** acceptance bars.
|
||||
These are synthetic/corpus-derived stand-ins for business docs until customer PDFs are added.
|
||||
|
||||
| File | Role |
|
||||
| --- | --- |
|
||||
| digital_report_001.pdf | Simple digital text → DOCX |
|
||||
| table_pipe_001.pdf / table_simple_001.pdf | Simple tables → XLSX |
|
||||
| invoice_001.pdf / invoice_002.pdf | Invoice grid + prose → XLSX |
|
||||
| merged_001.pdf | Merged-style table → XLSX |
|
||||
| multicolumn_*.pdf | Multi-column → DOCX text recall |
|
||||
| multipage_table_001.pdf | Multi-page tables |
|
||||
| table_gap_001.pdf | Gap-separated columns |
|
||||
|
||||
License: synthetic-internal. Not ConvertAPI goldens.
|
||||
@@ -0,0 +1,182 @@
|
||||
{
|
||||
"cases": [
|
||||
{
|
||||
"id": "real_digital_docx",
|
||||
"source": "real/digital_report_001.pdf",
|
||||
"pair": "pdf->docx",
|
||||
"expected_tokens": [
|
||||
"Digital",
|
||||
"Report",
|
||||
"AlphaBridge",
|
||||
"AcmeCorp"
|
||||
],
|
||||
"text_recall_min": 0.95
|
||||
},
|
||||
{
|
||||
"id": "real_table_xlsx",
|
||||
"source": "real/simple_table_001.pdf",
|
||||
"pair": "pdf->xlsx",
|
||||
"expected_cells": [
|
||||
[
|
||||
"Item",
|
||||
"Qty",
|
||||
"Price"
|
||||
],
|
||||
[
|
||||
"Widget",
|
||||
"2",
|
||||
"10"
|
||||
],
|
||||
[
|
||||
"Gadget",
|
||||
"1",
|
||||
"25"
|
||||
]
|
||||
],
|
||||
"cell_match_min": 0.9
|
||||
},
|
||||
{
|
||||
"id": "real_invoice_xlsx",
|
||||
"source": "real/invoice_001.pdf",
|
||||
"pair": "pdf->xlsx",
|
||||
"expected_cells": [
|
||||
[
|
||||
"Desc",
|
||||
"Qty",
|
||||
"Amount"
|
||||
],
|
||||
[
|
||||
"Service",
|
||||
"1",
|
||||
"100"
|
||||
]
|
||||
],
|
||||
"prose_tokens": [
|
||||
"Invoice",
|
||||
"AcmeCorp",
|
||||
"Vendor"
|
||||
],
|
||||
"cell_match_min": 0.85
|
||||
},
|
||||
{
|
||||
"id": "real_two_col_docx",
|
||||
"source": "real/multicolumn_001.pdf",
|
||||
"pair": "pdf->docx",
|
||||
"expected_tokens": [
|
||||
"LeftColA",
|
||||
"LeftColB",
|
||||
"RightColA",
|
||||
"RightColB"
|
||||
],
|
||||
"text_recall_min": 0.95
|
||||
},
|
||||
{
|
||||
"id": "real_three_col_docx",
|
||||
"source": "real/multicolumn_003.pdf",
|
||||
"pair": "pdf->docx",
|
||||
"expected_tokens": [
|
||||
"AlphaLeft",
|
||||
"BravoMid",
|
||||
"CharlieRight",
|
||||
"AlphaLast",
|
||||
"CharlieLast"
|
||||
],
|
||||
"text_recall_min": 0.9
|
||||
},
|
||||
{
|
||||
"id": "real_fourcol_xlsx",
|
||||
"source": "real/table_fourcol_001.pdf",
|
||||
"pair": "pdf->xlsx",
|
||||
"expected_cells": [
|
||||
[
|
||||
"Category",
|
||||
"Q1",
|
||||
"Q2",
|
||||
"Total"
|
||||
],
|
||||
[
|
||||
"Food",
|
||||
"10",
|
||||
"12",
|
||||
"22"
|
||||
],
|
||||
[
|
||||
"Travel",
|
||||
"5",
|
||||
"7",
|
||||
"12"
|
||||
]
|
||||
],
|
||||
"cell_match_min": 0.85
|
||||
},
|
||||
{
|
||||
"id": "real_arabic_docx",
|
||||
"source": "real/arabic_line_001.pdf",
|
||||
"pair": "pdf->docx",
|
||||
"expected_tokens": [
|
||||
"Hello",
|
||||
"World",
|
||||
"TRAIL_AR",
|
||||
"Bilingual"
|
||||
],
|
||||
"text_recall_min": 0.85
|
||||
},
|
||||
{
|
||||
"id": "real_scanlike_docx",
|
||||
"source": "real/scanlike_001.pdf",
|
||||
"pair": "pdf->docx",
|
||||
"expected_tokens": [
|
||||
"ScanLike",
|
||||
"ScanTokenSMOKEP000",
|
||||
"Widget"
|
||||
],
|
||||
"text_recall_min": 0.9
|
||||
},
|
||||
{
|
||||
"id": "real_spanning_docx",
|
||||
"source": "real/spanning_title_001.pdf",
|
||||
"pair": "pdf->docx",
|
||||
"expected_tokens": [
|
||||
"SpanningTitleAcrossTheFullPageWidthHere",
|
||||
"SpanLeftA",
|
||||
"SpanRightA"
|
||||
],
|
||||
"text_recall_min": 0.9
|
||||
},
|
||||
{
|
||||
"id": "real_rect_table_xlsx",
|
||||
"source": "real/rect_table_001.pdf",
|
||||
"pair": "pdf->xlsx",
|
||||
"expected_cells": [
|
||||
[
|
||||
"Name",
|
||||
"Qty",
|
||||
"Cost"
|
||||
],
|
||||
[
|
||||
"Alpha",
|
||||
"3",
|
||||
"9"
|
||||
],
|
||||
[
|
||||
"Beta",
|
||||
"4",
|
||||
"8"
|
||||
]
|
||||
],
|
||||
"cell_match_min": 0.7
|
||||
},
|
||||
{
|
||||
"id": "real_hyphen_md",
|
||||
"source": "real/hyphen_md_001.pdf",
|
||||
"pair": "pdf->md",
|
||||
"expected_tokens": [
|
||||
"HyphenSmoke",
|
||||
"word",
|
||||
"Chapter",
|
||||
"postprocess"
|
||||
],
|
||||
"text_recall_min": 0.75
|
||||
}
|
||||
]
|
||||
}
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
# License verification checklist (operator)
|
||||
|
||||
Sign off before fetching each asset into `corpus/convert` or `gateway/models/layout`.
|
||||
|
||||
| Asset | Upstream URL | License claimed | Verified by | Date | OK for prod train/eval? |
|
||||
| --- | --- | --- | --- | --- | --- |
|
||||
| DocLayNet | https://github.com/DS4SD/DocLayNet | CDLA-Permissive-1.0 | | | Y/N |
|
||||
| PubLayNet | https://github.com/ibm-aur-nlp/publaynet | CDLA-P + PMC CA images | | | Y/N |
|
||||
| PubTabNet | https://github.com/ibm-aur-nlp/PubTabNet | CDLA-P + PMC CA | | | Y/N |
|
||||
| CORD | https://github.com/clovaai/cord | CC-BY-4.0 | | | Y/N |
|
||||
| SROIE | https://rrc.cvc.uab.es/?ch=13 | CC-BY-4.0 (verify) | | | Y/N |
|
||||
| PP-DocLayout ONNX | PaddleOCR release / fetch script | Apache-2.0 | | | Y/N |
|
||||
| PP-OCRv5 Arabic rec ONNX | RapidAI ModelScope `arabic_PP-OCRv5_rec_mobile.onnx` + `ppocrv5_arabic_dict.txt` | Apache-2.0 (RapidAI/Paddle) | | | Y/N |
|
||||
| Self-gen fixtures | DocQube scripts | Proprietary/internal | | | Y |
|
||||
|
||||
**Never fetch for prod:** FUNSD (NC), LayoutLMv3 weights (NC), DocLayout-YOLO (AGPL).
|
||||
|
||||
Disk budgets: smoke ≤ 500 MB; full DocLayNet ~28 GB lab-only (`--full`).
|
||||
Vendored
+40
@@ -0,0 +1,40 @@
|
||||
DocQube convert — third-party NOTICE
|
||||
====================================
|
||||
|
||||
Append a block below whenever you fetch a public dataset or layout model.
|
||||
Keep license names and upstream URLs accurate.
|
||||
|
||||
How to append
|
||||
-------------
|
||||
1. Verify license on LICENSE_CHECKLIST.md
|
||||
2. Download via scripts/convert/fetch_public_eval.py or fetch_layout_onnx.py
|
||||
3. Add:
|
||||
|
||||
Name: <asset>
|
||||
License: <SPDX or common name>
|
||||
Source: <URL>
|
||||
Used for: <eval|train|inference>
|
||||
Date: <ISO date>
|
||||
|
||||
Self-generated fixtures under corpus/convert (real/ / public/samples/invoices etc.)
|
||||
are owned by DocQube and require no third-party attribution.
|
||||
|
||||
--- Upstream entries (filled by fetch scripts) ---
|
||||
|
||||
Name: PP-DocLayoutV3 ONNX (layout.onnx / inference.onnx)
|
||||
License: Apache-2.0
|
||||
Source: https://huggingface.co/PaddlePaddle/PP-DocLayoutV3_onnx
|
||||
Used for: inference
|
||||
Date: 2026-08-27
|
||||
SHA256: 45bf71750b00739a41fc209f132eb104a4d6b5bb29483c9078164d8b87cf28ba
|
||||
Path: gateway/models/layout/v1/layout.onnx (copy of inference.onnx)
|
||||
|
||||
Name: PP-OCRv5 Arabic recognition ONNX (rec.onnx)
|
||||
License: Apache-2.0
|
||||
Source: https://www.modelscope.cn/models/RapidAI/RapidOCR
|
||||
Used for: inference
|
||||
|
||||
Name: PP-OCRv5 Arabic charset (arabic_dict.txt)
|
||||
License: Apache-2.0
|
||||
Source: https://www.modelscope.cn/models/RapidAI/RapidOCR/resolve/v3.9.2/paddle/PP-OCRv5/rec/arabic_PP-OCRv5_rec_mobile/ppocrv5_arabic_dict.txt
|
||||
Used for: inference
|
||||
+16
@@ -0,0 +1,16 @@
|
||||
# SHA256 checksums for convert public samples and layout weights
|
||||
# Format: <sha256> <relative-path-from-pdf/>
|
||||
# Populated by fetch_public_eval.py / fetch_layout_onnx.py / generate fixtures.
|
||||
58a051ee4dcc59e49b1c379ed36e528e1d99dc6968c9a5c11bf6934e0a8a1f36 corpus/convert/real/digital_report_001.pdf
|
||||
e67e0040b551d674ed57097d71b1d11bb2c0b45de18da902b847053f9b00d6e8 corpus/convert/real/simple_table_001.pdf
|
||||
ba2bf716c44970858e298c340fbd5e8a31c40639315fa2ce42da6bbee0c1a279 corpus/convert/real/invoice_001.pdf
|
||||
ffad883045c91f860039352ad9343fad1eb958e6a11fd49a4a2e7f8d22102a5c corpus/convert/public/samples/invoices/synth_invoice_001.pdf
|
||||
0d9165e3d425d3782f4b41e38b890343c6f6cd032691d04f5e345f8ed1105211 corpus/convert/public/samples/pubtab/synth_table_001.pdf
|
||||
7fc1dc773d71db49bafdb0458d964e5ec68f76488f785d753f4575b1b685e40f corpus/convert/public/samples/doclaynet/synth_digital_001.pdf
|
||||
314e13e9e4f1178dace76985b10cb41aa5db9a45aaa2e319dc21aab37a336023 corpus/convert/public/samples/doclaynet/synth_multicol_001.pdf
|
||||
e2e2f4f6671e753955036f4a0ac2b56e3534bf3901fe9feeeca0e68950a87749 corpus/convert/public/samples/path_a/path_a_sample.docx
|
||||
eb203ee73bea31647892d0892605cb65a2aad9a0abdb7a76e8b2e2e60e625775 corpus/convert/public/samples/path_a/path_a_sample.xlsx
|
||||
a33ad10e0028c1f84fd1ce229a4de2d0bdb931bc68ac01e80222dd9772008d05 corpus/convert/public/samples/images/smoke.png
|
||||
45bf71750b00739a41fc209f132eb104a4d6b5bb29483c9078164d8b87cf28ba gateway/models/layout/v1/layout.onnx
|
||||
c1192e632d0baa9146ae5b756a0e635e3dc63c1733737ebfd1629e87144e9295 gateway/models/ocr/ar/v5/rec.onnx
|
||||
7f92f7dbb9b75a4787a83bfb4f6d14a8ab515525130c9d40a9036f61cf6999e9 gateway/models/ocr/ar/v5/arabic_dict.txt
|
||||
@@ -9,6 +9,10 @@ services:
|
||||
environment:
|
||||
PDFENGINE_ENVIRONMENT: dev
|
||||
PDFENGINE_ENGINE_AVAILABLE: "true"
|
||||
# Keep browser access explicit; production deployments should replace
|
||||
# this with the exact frontend origin(s), never a wildcard.
|
||||
PDFENGINE_CORS_ALLOWED_ORIGINS: http://localhost:5173,http://127.0.0.1:5173,https://pdf-dev.maskantech.in
|
||||
PDFENGINE_CORS_ALLOW_CREDENTIALS: "false"
|
||||
PORT: 8765
|
||||
ports:
|
||||
- "8765:8765"
|
||||
@@ -39,3 +43,10 @@ services:
|
||||
depends_on:
|
||||
- gateway
|
||||
restart: unless-stopped
|
||||
|
||||
gotenberg:
|
||||
image: gotenberg/gotenberg:8
|
||||
container_name: pdf-engine-gotenberg
|
||||
ports:
|
||||
- "3000:3000"
|
||||
restart: unless-stopped
|
||||
|
||||
+25
-13
@@ -111,11 +111,15 @@ target_link_libraries(pdfengine
|
||||
)
|
||||
|
||||
if(PDFENGINE_WITH_PDFIUM)
|
||||
target_link_libraries(pdfengine PRIVATE pdfium::pdfium)
|
||||
target_link_libraries(pdfengine PUBLIC pdfium::pdfium)
|
||||
target_compile_definitions(pdfengine PUBLIC PDFENGINE_WITH_PDFIUM)
|
||||
# PDFium statically bundles its own libjpeg, zlib, etc. which conflicts with vcpkg.
|
||||
# We use LLD, so we can safely allow multiple definitions to pick the first one.
|
||||
target_link_options(pdfengine PUBLIC "-Wl,--allow-multiple-definition")
|
||||
# PDFium statically bundles its own libjpeg, zlib, etc. which conflict with vcpkg.
|
||||
# MSVC ignores -Wl,--allow-multiple-definition (that is a GNU ld flag).
|
||||
if(MSVC)
|
||||
target_link_options(pdfengine PUBLIC "/FORCE:MULTIPLE")
|
||||
else()
|
||||
target_link_options(pdfengine PUBLIC "-Wl,--allow-multiple-definition")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if(EMSCRIPTEN)
|
||||
@@ -129,16 +133,24 @@ if(PDFENGINE_WITH_SKIA)
|
||||
target_compile_definitions(pdfengine PUBLIC PDFENGINE_WITH_SKIA)
|
||||
endif()
|
||||
|
||||
# qpdf / libjpeg are used by sources that compile even when PDFENGINE_WITH_QPDF is
|
||||
# off (resource_resolver.hpp, qpdf_resource_resolver.cpp, image_decoder.cpp,
|
||||
# filter_decoder.cpp). vcpkg puts the headers on the include path automatically,
|
||||
# so the objects compile — but MSVC then fails with LNK2019 unless we link here.
|
||||
# pdfengine is an OBJECT library: PUBLIC is required so consumers (the Python
|
||||
# module and smoke test) actually get these libs on the final link line.
|
||||
find_package(ZLIB REQUIRED)
|
||||
find_package(JPEG REQUIRED)
|
||||
find_package(qpdf CONFIG REQUIRED)
|
||||
if(NOT TARGET zs)
|
||||
add_library(zs ALIAS ZLIB::ZLIB)
|
||||
endif()
|
||||
if(NOT TARGET jpeg)
|
||||
add_library(jpeg ALIAS JPEG::JPEG)
|
||||
endif()
|
||||
target_link_libraries(pdfengine PUBLIC qpdf::libqpdf ZLIB::ZLIB JPEG::JPEG)
|
||||
|
||||
if(PDFENGINE_WITH_QPDF)
|
||||
find_package(ZLIB REQUIRED)
|
||||
find_package(JPEG REQUIRED)
|
||||
if(NOT TARGET zs)
|
||||
add_library(zs ALIAS ZLIB::ZLIB)
|
||||
endif()
|
||||
if(NOT TARGET jpeg)
|
||||
add_library(jpeg ALIAS JPEG::JPEG)
|
||||
endif()
|
||||
target_link_libraries(pdfengine PRIVATE qpdf::libqpdf ZLIB::ZLIB JPEG::JPEG)
|
||||
target_compile_definitions(pdfengine PUBLIC PDFENGINE_WITH_QPDF)
|
||||
endif()
|
||||
|
||||
|
||||
@@ -46,9 +46,7 @@ if(PDFENGINE_WITH_PDFIUM)
|
||||
target_link_libraries(pdfengine_smoke PRIVATE pdfium::pdfium)
|
||||
endif()
|
||||
|
||||
if(PDFENGINE_WITH_QPDF)
|
||||
target_link_libraries(pdfengine_smoke PRIVATE qpdf::libqpdf ZLIB::ZLIB JPEG::JPEG)
|
||||
endif()
|
||||
target_link_libraries(pdfengine_smoke PRIVATE qpdf::libqpdf ZLIB::ZLIB JPEG::JPEG)
|
||||
|
||||
if(MSVC)
|
||||
target_link_options(pdfengine_smoke PRIVATE "/FORCE:MULTIPLE")
|
||||
|
||||
+2
-1
@@ -1 +1,2 @@
|
||||
VITE_GATEWAY_URL=http://127.0.0.1:8765
|
||||
VITE_GATEWAY_URL=http://127.0.0.1:8080
|
||||
# VITE_GATEWAY_URL=https://pdfapi-dev.maskantech.in
|
||||
+32
-32
@@ -12,6 +12,7 @@ import { ExportPDFModal } from './components/ExportPDFModal';
|
||||
import { MergePDFModal } from './components/MergePDFModal';
|
||||
import { WatermarkModal, type WatermarkConfig } from './components/WatermarkModal';
|
||||
import { triggerPDFDownload } from './lib/pdfExport';
|
||||
import { convertFileMaybeAsync, downloadBlob } from './lib/convertService';
|
||||
|
||||
import { CustomConfirmationModal } from './components/custom/CustomConfirmationModal';
|
||||
import type { CustomConfirmationOptions } from './components/custom/CustomConfirmationModal';
|
||||
@@ -31,6 +32,7 @@ import { UnlockModal } from './components/UnlockModal';
|
||||
import type { UnlockModalState } from './components/UnlockModal';
|
||||
import { CompareModal } from './components/CompareModal';
|
||||
import { CompareWorkspace } from './components/CompareWorkspace';
|
||||
import { ProductHub } from './components/ProductHub';
|
||||
import type { DocumentInfo, SearchResult, EditOperation, DocumentMetadata, FontInfo, OutlineItem, PDFPermissions, CompareResponse, DocumentSummary } from './lib/gatewayService';
|
||||
import { viewportRectToPdf } from './lib/coordinateMapping';
|
||||
import type { Rect } from './lib/coordinateMapping';
|
||||
@@ -42,6 +44,7 @@ const rid = (p: string) => `${p}_${Math.random().toString(36).substring(2, 11)}`
|
||||
|
||||
function App() {
|
||||
const viewerRef = useRef<PDFViewerRef>(null);
|
||||
const [productView, setProductView] = useState<'chooser' | 'editor' | 'conversion'>('chooser');
|
||||
|
||||
const [documents, setDocuments] = useState<DocumentInfo[]>([]);
|
||||
const [activeDoc, setActiveDoc] = useState<DocumentInfo | null>(null);
|
||||
@@ -158,7 +161,7 @@ function App() {
|
||||
const [searchResults, setSearchResults] = useState<SearchResult[]>([]);
|
||||
const [searchCurrentMatch, setSearchCurrentMatch] = useState(0);
|
||||
|
||||
const [isOCRLoading, setIsOCRLoading] = useState(false);
|
||||
const [isOCRLoading] = useState(false);
|
||||
const [creatorPageCount, setCreatorPageCount] = useState<number>(1);
|
||||
const [creatorPages, setCreatorPages] = useState<PageLayout[]>([]);
|
||||
const [watermarkPreview, setWatermarkPreview] = useState<WatermarkConfig | null>(null);
|
||||
@@ -168,19 +171,6 @@ function App() {
|
||||
if (pages) setCreatorPages(pages);
|
||||
}, []);
|
||||
|
||||
const handleRunOCR = async () => {
|
||||
if (!activeDoc) return;
|
||||
setIsOCRLoading(true);
|
||||
try {
|
||||
await gatewayService.performPageOCR(activeDoc.id, currentPage);
|
||||
viewerRef.current?.refreshPageLayout(currentPage);
|
||||
} catch (err: any) {
|
||||
alert(`OCR processing failed: ${err.message || err}`);
|
||||
} finally {
|
||||
setIsOCRLoading(false);
|
||||
}
|
||||
};
|
||||
|
||||
const forceOpenDocument = useCallback((id: string) => {
|
||||
localStorage.removeItem('active_mode');
|
||||
setCreatePdfModalOpen(false);
|
||||
@@ -1115,10 +1105,10 @@ function App() {
|
||||
setExportModalOpen(true);
|
||||
};
|
||||
|
||||
const handleConfirmExport = async (targetFilename: string) => {
|
||||
const handleConfirmExport = async (targetFilename: string, targetFormat = 'pdf') => {
|
||||
const isCreatorActive = activeTool === 'create_pdf' || createPdfModalOpen;
|
||||
console.log(`[EXPORT_PDF] Export requested. Document ID: ${isCreatorActive ? 'new-blank-creator' : selectedDocId}`);
|
||||
console.log(`[EXPORT_PDF] Filename: ${targetFilename}`);
|
||||
console.log(`[EXPORT_PDF] Filename: ${targetFilename} target=${targetFormat}`);
|
||||
|
||||
let pdfBlob: Blob;
|
||||
if (isCreatorActive) {
|
||||
@@ -1137,7 +1127,17 @@ function App() {
|
||||
}
|
||||
|
||||
console.log(`[EXPORT_PDF] PDF size: ${pdfBlob.size} bytes`);
|
||||
await triggerPDFDownload(pdfBlob, targetFilename);
|
||||
if (targetFormat === 'pdf') {
|
||||
await triggerPDFDownload(pdfBlob, targetFilename);
|
||||
return;
|
||||
}
|
||||
|
||||
const sourceName = (isCreatorActive ? 'document.pdf' : activeDoc?.filename) || 'document.pdf';
|
||||
const file = new File([pdfBlob], sourceName.replace(/\.[^.]+$/, '.pdf'), {
|
||||
type: 'application/pdf',
|
||||
});
|
||||
const result = await convertFileMaybeAsync(file, 'pdf', targetFormat);
|
||||
downloadBlob(result.blob, targetFilename || result.filename);
|
||||
};
|
||||
|
||||
const handlePrint = async () => {
|
||||
@@ -1205,6 +1205,17 @@ function App() {
|
||||
}], 'Annotation updated');
|
||||
};
|
||||
|
||||
if (productView !== 'editor') {
|
||||
return (
|
||||
<ProductHub
|
||||
view={productView}
|
||||
onOpenEditor={() => setProductView('editor')}
|
||||
onOpenConversion={() => setProductView('conversion')}
|
||||
onBack={() => setProductView('chooser')}
|
||||
/>
|
||||
);
|
||||
}
|
||||
|
||||
return (
|
||||
<div className="flex h-full w-full flex-col overflow-hidden bg-[#f1f2f4] text-[#18212e] antialiased">
|
||||
<TopBar
|
||||
@@ -1218,7 +1229,7 @@ function App() {
|
||||
isSaving={isSaving}
|
||||
isDirtySaved={(activeTool === 'create_pdf' || createPdfModalOpen) ? (creatorActions?.canUndo ?? false) : hist.stack.length > 1}
|
||||
onRotate={handleRotate}
|
||||
onExport={activeTool === 'create_pdf' || createPdfModalOpen ? (() => creatorActions?.generate()) : handleExport}
|
||||
onExport={handleExport}
|
||||
onPrint={(activeTool === 'create_pdf' || createPdfModalOpen) ? (() => creatorActions?.print?.()) : handlePrint}
|
||||
onProtect={() => {
|
||||
const isCreatorActive = activeTool === 'create_pdf' || createPdfModalOpen;
|
||||
@@ -1248,6 +1259,7 @@ function App() {
|
||||
onGoToPage={(p) => viewerRef.current?.scrollToPage(p)}
|
||||
onSave={handleSave}
|
||||
isRemote={isRemote}
|
||||
onBackToHome={() => setProductView('chooser')}
|
||||
/>
|
||||
|
||||
<div className="flex min-h-0 flex-1">
|
||||
@@ -1452,16 +1464,6 @@ function App() {
|
||||
<p className="text-[14px] font-semibold text-[#18212e]">Import Error</p>
|
||||
<p className="text-[13px]">{importError}</p>
|
||||
</div>
|
||||
) : isRemote && !activeDoc && !importError ? (
|
||||
<div className="flex h-full flex-col items-center justify-center gap-4 text-[#98a1ad]">
|
||||
<div className="flex h-16 w-16 items-center justify-center rounded-2xl bg-blue-50/50">
|
||||
<div className="h-6 w-6 animate-spin rounded-full border-2 border-[#2563eb] border-t-transparent"></div>
|
||||
</div>
|
||||
<div className="text-center">
|
||||
<p className="text-[14px] font-semibold text-[#18212e]">Loading Document</p>
|
||||
<p className="text-[12px] mt-1">Preparing your document for editing...</p>
|
||||
</div>
|
||||
</div>
|
||||
) : (
|
||||
<div className="flex h-full flex-col items-center justify-center gap-3 text-[#98a1ad]">
|
||||
<div className="flex h-16 w-16 items-center justify-center rounded-2xl bg-[#edeff2]">
|
||||
@@ -1647,12 +1649,10 @@ function App() {
|
||||
|
||||
<MergePDFModal
|
||||
isOpen={mergeModalOpen}
|
||||
onClose={() => { setMergeModalOpen(false); setMergeInitialFile(null); }}
|
||||
onMergeComplete={(docInfo) => { setMergeModalOpen(false); forceOpenDocument(docInfo.id); }}
|
||||
onClose={() => setMergeModalOpen(false)}
|
||||
onMergeComplete={handleMergeCompleted}
|
||||
apiBaseUrl={import.meta.env.VITE_GATEWAY_URL || 'http://localhost:8000'}
|
||||
initialFile={mergeInitialFile}
|
||||
parentDocumentId={selectedDocId}
|
||||
isRemote={isRemote}
|
||||
/>
|
||||
</div>
|
||||
);
|
||||
|
||||
@@ -1,12 +1,23 @@
|
||||
import React, { useState, useEffect } from 'react';
|
||||
import React, { useEffect, useMemo, useRef, useState } from 'react';
|
||||
import { Modal } from './ui';
|
||||
import { CustomButton } from './custom/CustomButton';
|
||||
import {
|
||||
ENGINE_CONVERT_PAIRS,
|
||||
FORMAT_META,
|
||||
fetchConvertInfo,
|
||||
fileExtensionForTarget,
|
||||
pairDescription,
|
||||
pairDisplayName,
|
||||
replaceFilenameExtension,
|
||||
targetsFromSource,
|
||||
type ConverterInfo,
|
||||
} from '../lib/convertService';
|
||||
|
||||
interface ExportPDFModalProps {
|
||||
open: boolean;
|
||||
onClose: () => void;
|
||||
documentName?: string;
|
||||
onConfirmExport: (filename: string) => Promise<void>;
|
||||
onConfirmExport: (filename: string, target: string) => Promise<void>;
|
||||
}
|
||||
|
||||
export const ExportPDFModal: React.FC<ExportPDFModalProps> = ({
|
||||
@@ -16,40 +27,70 @@ export const ExportPDFModal: React.FC<ExportPDFModalProps> = ({
|
||||
onConfirmExport,
|
||||
}) => {
|
||||
const [filename, setFilename] = useState('');
|
||||
const [target, setTarget] = useState('pdf');
|
||||
const [isExporting, setIsExporting] = useState(false);
|
||||
const [errorMessage, setErrorMessage] = useState<string | null>(null);
|
||||
const [pairs, setPairs] = useState<ConverterInfo[]>(
|
||||
targetsFromSource(ENGINE_CONVERT_PAIRS, 'pdf').filter((p) => p.target !== 'pdf')
|
||||
);
|
||||
const skipExtSync = useRef(true);
|
||||
|
||||
const selected = useMemo(
|
||||
() => pairs.find((p) => p.target === target),
|
||||
[pairs, target]
|
||||
);
|
||||
|
||||
useEffect(() => {
|
||||
if (open) {
|
||||
const baseName = documentName.replace(/\.pdf$/i, '');
|
||||
const defaultExportName = baseName ? `${baseName}-final.pdf` : 'document-final.pdf';
|
||||
setFilename(defaultExportName);
|
||||
setIsExporting(false);
|
||||
setErrorMessage(null);
|
||||
}
|
||||
if (!open) return;
|
||||
let cancelled = false;
|
||||
void fetchConvertInfo().then((info) => {
|
||||
if (cancelled) return;
|
||||
setPairs(targetsFromSource(info, 'pdf').filter((p) => p.target !== 'pdf'));
|
||||
});
|
||||
return () => {
|
||||
cancelled = true;
|
||||
};
|
||||
}, [open]);
|
||||
|
||||
useEffect(() => {
|
||||
if (!open) return;
|
||||
skipExtSync.current = true;
|
||||
setTarget('pdf');
|
||||
setIsExporting(false);
|
||||
setErrorMessage(null);
|
||||
const stem = documentName.replace(/\.[^.]+$/i, '') || 'document';
|
||||
setFilename(`${stem}-final.pdf`);
|
||||
}, [open, documentName]);
|
||||
|
||||
useEffect(() => {
|
||||
if (!open) return;
|
||||
if (skipExtSync.current) {
|
||||
skipExtSync.current = false;
|
||||
return;
|
||||
}
|
||||
setFilename((prev) => replaceFilenameExtension(prev || documentName, target));
|
||||
}, [target, open, documentName]);
|
||||
|
||||
const handleDownload = async () => {
|
||||
if (isExporting) return;
|
||||
|
||||
let trimmed = filename.trim();
|
||||
if (!trimmed) {
|
||||
trimmed = 'document-final.pdf';
|
||||
trimmed = replaceFilenameExtension('document-final', target);
|
||||
}
|
||||
if (!trimmed.toLowerCase().endsWith('.pdf')) {
|
||||
trimmed += '.pdf';
|
||||
const ext = fileExtensionForTarget(target);
|
||||
if (!trimmed.toLowerCase().endsWith(`.${ext}`)) {
|
||||
trimmed = replaceFilenameExtension(trimmed, target);
|
||||
}
|
||||
|
||||
try {
|
||||
setIsExporting(true);
|
||||
setErrorMessage(null);
|
||||
console.log(`[EXPORT_PDF] Export requested. Filename: ${trimmed}`);
|
||||
await onConfirmExport(trimmed);
|
||||
console.log('[EXPORT_PDF] Export completed successfully');
|
||||
await onConfirmExport(trimmed, target);
|
||||
onClose();
|
||||
} catch (err: any) {
|
||||
console.error('[EXPORT_PDF] Export failed:', err);
|
||||
setErrorMessage(err.message || 'Export failed. Please try again.');
|
||||
} catch (err: unknown) {
|
||||
const message = err instanceof Error ? err.message : 'Export failed. Please try again.';
|
||||
setErrorMessage(message);
|
||||
} finally {
|
||||
setIsExporting(false);
|
||||
}
|
||||
@@ -57,14 +98,27 @@ export const ExportPDFModal: React.FC<ExportPDFModalProps> = ({
|
||||
|
||||
if (!open) return null;
|
||||
|
||||
const formats: { target: string; label: string; hint: string }[] = [
|
||||
{
|
||||
target: 'pdf',
|
||||
label: 'PDF',
|
||||
hint: 'Edited document as PDF',
|
||||
},
|
||||
...pairs.map((p) => ({
|
||||
target: p.target,
|
||||
label: pairDisplayName(p),
|
||||
hint: pairDescription(p),
|
||||
})),
|
||||
];
|
||||
|
||||
return (
|
||||
<Modal
|
||||
open={open}
|
||||
onClose={() => {
|
||||
if (!isExporting) onClose();
|
||||
}}
|
||||
title="Export PDF"
|
||||
width={440}
|
||||
title="Export / Convert"
|
||||
width={560}
|
||||
>
|
||||
<div className="flex flex-col gap-4 text-[13px]">
|
||||
{errorMessage && (
|
||||
@@ -92,7 +146,7 @@ export const ExportPDFModal: React.FC<ExportPDFModalProps> = ({
|
||||
onKeyDown={(e) => {
|
||||
if (e.key === 'Enter' && !isExporting) {
|
||||
e.preventDefault();
|
||||
handleDownload();
|
||||
void handleDownload();
|
||||
}
|
||||
}}
|
||||
placeholder="document-final.pdf"
|
||||
@@ -102,14 +156,41 @@ export const ExportPDFModal: React.FC<ExportPDFModalProps> = ({
|
||||
|
||||
<div className="flex flex-col gap-1.5">
|
||||
<label className="text-[12px] font-bold text-text-primary">
|
||||
Format:
|
||||
Format (conversion engine):
|
||||
</label>
|
||||
<div className="flex h-9 w-full items-center justify-between rounded-lg border border-border-primary bg-bg-secondary px-3 text-[13px] font-semibold text-text-primary">
|
||||
<span>PDF Document (.pdf)</span>
|
||||
<span className="rounded bg-brand-primary/10 px-2 py-0.5 text-[11px] font-bold text-brand-primary">
|
||||
PDF
|
||||
</span>
|
||||
<div className="grid max-h-[240px] grid-cols-2 gap-1.5 overflow-auto pr-0.5">
|
||||
{formats.map((fmt) => {
|
||||
const meta = FORMAT_META[fmt.target];
|
||||
const active = target === fmt.target;
|
||||
return (
|
||||
<button
|
||||
key={fmt.target}
|
||||
type="button"
|
||||
disabled={isExporting}
|
||||
onClick={() => setTarget(fmt.target)}
|
||||
className={`rounded-[10px] border px-3 py-2 text-left transition ${
|
||||
active
|
||||
? 'border-brand-primary bg-brand-primary/5'
|
||||
: 'border-border-primary bg-bg-primary hover:border-brand-primary/50'
|
||||
} disabled:opacity-50`}
|
||||
>
|
||||
<span className="flex items-center justify-between gap-2">
|
||||
<span className="text-[12px] font-extrabold text-text-primary">{fmt.label}</span>
|
||||
<span className="rounded bg-brand-primary/10 px-1.5 py-0.5 text-[10px] font-bold uppercase text-brand-primary">
|
||||
{meta?.extension ?? fmt.target}
|
||||
</span>
|
||||
</span>
|
||||
<span className="mt-0.5 block text-[10px] leading-4 text-text-secondary">{fmt.hint}</span>
|
||||
</button>
|
||||
);
|
||||
})}
|
||||
</div>
|
||||
{target !== 'pdf' && selected && (
|
||||
<p className="text-[11px] leading-5 text-text-secondary">
|
||||
Sends the current PDF through the document conversion engine
|
||||
({selected.fidelity} fidelity).
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
|
||||
<div className="mt-2 flex items-center justify-end gap-2.5 pt-3 border-t border-border-primary">
|
||||
@@ -122,7 +203,7 @@ export const ExportPDFModal: React.FC<ExportPDFModalProps> = ({
|
||||
</CustomButton>
|
||||
<CustomButton
|
||||
variant="primary"
|
||||
onClick={handleDownload}
|
||||
onClick={() => void handleDownload()}
|
||||
disabled={isExporting}
|
||||
>
|
||||
{isExporting ? (
|
||||
@@ -137,10 +218,12 @@ export const ExportPDFModal: React.FC<ExportPDFModalProps> = ({
|
||||
<circle cx="12" cy="12" r="10" strokeOpacity="0.25" />
|
||||
<path d="M12 2a10 10 0 0 1 10 10" />
|
||||
</svg>
|
||||
Generating PDF...
|
||||
{target === 'pdf' ? 'Generating PDF…' : `Converting to ${target.toUpperCase()}…`}
|
||||
</span>
|
||||
) : (
|
||||
) : target === 'pdf' ? (
|
||||
'Download PDF'
|
||||
) : (
|
||||
`Convert to ${FORMAT_META[target]?.label ?? target.toUpperCase()}`
|
||||
)}
|
||||
</CustomButton>
|
||||
</div>
|
||||
|
||||
@@ -16,8 +16,6 @@ interface MergePDFModalProps {
|
||||
onMergeComplete: (docInfo: any) => void;
|
||||
apiBaseUrl?: string;
|
||||
initialFile?: File | null;
|
||||
parentDocumentId?: string;
|
||||
isRemote?: boolean;
|
||||
}
|
||||
|
||||
export const MergePDFModal: React.FC<MergePDFModalProps> = ({
|
||||
@@ -26,12 +24,9 @@ export const MergePDFModal: React.FC<MergePDFModalProps> = ({
|
||||
onMergeComplete,
|
||||
apiBaseUrl = 'http://localhost:8000',
|
||||
initialFile = null,
|
||||
parentDocumentId,
|
||||
isRemote = false,
|
||||
}) => {
|
||||
const [files, setFiles] = useState<FileItem[]>([]);
|
||||
const [outputFilename, setOutputFilename] = useState<string>('merged_document.pdf');
|
||||
const [isDownloading, setIsDownloading] = useState(false);
|
||||
const [isMerging, setIsMerging] = useState<boolean>(false);
|
||||
const [error, setError] = useState<string | null>(null);
|
||||
const [mergedResult, setMergedResult] = useState<any | null>(null);
|
||||
@@ -56,7 +51,9 @@ export const MergePDFModal: React.FC<MergePDFModalProps> = ({
|
||||
}
|
||||
}, [isOpen, initialFile]);
|
||||
|
||||
const handleAddFiles = React.useCallback((selectedFiles: FileList | null) => {
|
||||
if (!isOpen) return null;
|
||||
|
||||
const handleAddFiles = (selectedFiles: FileList | null) => {
|
||||
if (!selectedFiles || selectedFiles.length === 0) return;
|
||||
setError(null);
|
||||
const newItems: FileItem[] = Array.from(selectedFiles).map((file) => ({
|
||||
@@ -66,38 +63,7 @@ export const MergePDFModal: React.FC<MergePDFModalProps> = ({
|
||||
customPages: '',
|
||||
}));
|
||||
setFiles((prev) => [...prev, ...newItems]);
|
||||
}, []);
|
||||
|
||||
React.useEffect(() => {
|
||||
if (!isRemote) return;
|
||||
|
||||
const handleMessage = async (event: MessageEvent) => {
|
||||
if (event.data?.type === 'DOCQUBE_FILE_SELECTED') {
|
||||
const { stream_url, token, filename } = event.data.payload;
|
||||
try {
|
||||
setIsDownloading(true);
|
||||
const response = await fetch(stream_url, {
|
||||
headers: {
|
||||
'Authorization': `Bearer ${token}`
|
||||
}
|
||||
});
|
||||
if (!response.ok) throw new Error('Failed to fetch file');
|
||||
const blob = await response.blob();
|
||||
const file = new File([blob], filename || 'document.pdf', { type: 'application/pdf' });
|
||||
handleAddFiles([file] as unknown as FileList);
|
||||
} catch (err) {
|
||||
console.error('Failed to download file from DocQube:', err);
|
||||
setError('Failed to download file from DocQube');
|
||||
} finally {
|
||||
setIsDownloading(false);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
window.addEventListener('message', handleMessage);
|
||||
return () => window.removeEventListener('message', handleMessage);
|
||||
}, [isRemote, handleAddFiles]);
|
||||
if (!isOpen) return null;
|
||||
};
|
||||
|
||||
const handleMoveUp = (index: number) => {
|
||||
if (index <= 0) return;
|
||||
@@ -166,9 +132,6 @@ export const MergePDFModal: React.FC<MergePDFModalProps> = ({
|
||||
});
|
||||
formData.append('manifest', JSON.stringify(manifest));
|
||||
formData.append('output_filename', outputFilename || 'merged_document.pdf');
|
||||
if (parentDocumentId) {
|
||||
formData.append('parent_document_id', parentDocumentId);
|
||||
}
|
||||
|
||||
const response = await fetch(`${apiBaseUrl}/documents/merge`, {
|
||||
method: 'POST',
|
||||
@@ -284,58 +247,29 @@ export const MergePDFModal: React.FC<MergePDFModalProps> = ({
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Upload Dropzones */}
|
||||
<div className="flex flex-col sm:flex-row gap-4 w-full">
|
||||
<div
|
||||
onClick={() => fileInputRef.current?.click()}
|
||||
onDragOver={(e) => e.preventDefault()}
|
||||
onDrop={(e) => {
|
||||
e.preventDefault();
|
||||
handleAddFiles(e.dataTransfer.files);
|
||||
}}
|
||||
className="flex flex-col items-center justify-center p-6 border-2 border-dashed border-border-primary hover:border-brand-primary rounded-xl cursor-pointer bg-bg-secondary/40 hover:bg-bg-secondary transition-all text-center group flex-1"
|
||||
>
|
||||
<input
|
||||
ref={fileInputRef}
|
||||
type="file"
|
||||
multiple
|
||||
accept=".pdf,image/*"
|
||||
className="hidden"
|
||||
onChange={(e) => handleAddFiles(e.target.files)}
|
||||
/>
|
||||
<div className="flex h-12 w-12 items-center justify-center rounded-full bg-brand-primary/10 text-brand-primary group-hover:scale-110 transition-transform">
|
||||
<UploadIcon size={22} />
|
||||
</div>
|
||||
<span className="mt-3 text-sm font-bold text-text-primary">Click or drop PDF files here</span>
|
||||
<span className="text-xs text-text-secondary mt-0.5">Select multiple PDF files to combine</span>
|
||||
{/* Upload Dropzone */}
|
||||
<div
|
||||
onClick={() => fileInputRef.current?.click()}
|
||||
onDragOver={(e) => e.preventDefault()}
|
||||
onDrop={(e) => {
|
||||
e.preventDefault();
|
||||
handleAddFiles(e.dataTransfer.files);
|
||||
}}
|
||||
className="flex flex-col items-center justify-center p-6 border-2 border-dashed border-border-primary hover:border-brand-primary rounded-xl cursor-pointer bg-bg-secondary/40 hover:bg-bg-secondary transition-all text-center group"
|
||||
>
|
||||
<input
|
||||
ref={fileInputRef}
|
||||
type="file"
|
||||
multiple
|
||||
accept=".pdf,image/*"
|
||||
className="hidden"
|
||||
onChange={(e) => handleAddFiles(e.target.files)}
|
||||
/>
|
||||
<div className="flex h-12 w-12 items-center justify-center rounded-full bg-brand-primary/10 text-brand-primary group-hover:scale-110 transition-transform">
|
||||
<UploadIcon size={22} />
|
||||
</div>
|
||||
|
||||
{isRemote && (
|
||||
<div
|
||||
onClick={() => {
|
||||
if (!isDownloading) {
|
||||
window.parent.postMessage({ type: 'DOCQUBE_REQUEST_FILE_PICKER' }, '*');
|
||||
}
|
||||
}}
|
||||
className={`flex flex-col items-center justify-center p-6 border-2 border-dashed border-border-primary hover:border-brand-primary rounded-xl bg-bg-secondary/40 hover:bg-bg-secondary transition-all text-center flex-1 ${isDownloading ? 'opacity-70 cursor-not-allowed' : 'cursor-pointer group'}`}
|
||||
>
|
||||
<div className="flex h-12 w-12 items-center justify-center rounded-full bg-brand-primary/10 text-brand-primary group-hover:scale-110 transition-transform">
|
||||
{isDownloading ? (
|
||||
<SpinnerIcon size={22} className="animate-spin" />
|
||||
) : (
|
||||
<svg width="22" height="22" viewBox="0 0 24 24" fill="none" stroke="currentColor" strokeWidth={2.5} strokeLinecap="round" strokeLinejoin="round">
|
||||
<path d="M17.5 19H9a7 7 0 1 1 6.71-9h1.79a4.5 4.5 0 1 1 0 9Z" />
|
||||
</svg>
|
||||
)}
|
||||
</div>
|
||||
<span className="mt-3 text-sm font-bold text-text-primary">
|
||||
{isDownloading ? 'Downloading...' : 'Pull from DocQube'}
|
||||
</span>
|
||||
<span className="text-xs text-text-secondary mt-0.5">
|
||||
{isDownloading ? 'Please wait' : 'Select a PDF from your Drive'}
|
||||
</span>
|
||||
</div>
|
||||
)}
|
||||
<span className="mt-3 text-sm font-bold text-text-primary">Click or drop PDF files here</span>
|
||||
<span className="text-xs text-text-secondary mt-0.5">Select multiple PDF files to combine</span>
|
||||
</div>
|
||||
|
||||
{/* File Queue List */}
|
||||
|
||||
@@ -0,0 +1,514 @@
|
||||
import React, { useEffect, useMemo, useRef, useState } from 'react';
|
||||
import {
|
||||
ASYNC_THRESHOLD_BYTES,
|
||||
ENGINE_CONVERT_PAIRS,
|
||||
FORMAT_META,
|
||||
convertFileMaybeAsync,
|
||||
downloadBlob,
|
||||
fetchConvertInfo,
|
||||
groupedPairs,
|
||||
inferSourceFormat,
|
||||
pairDescription,
|
||||
pairDisplayName,
|
||||
pairId,
|
||||
uploadAcceptFromPairs,
|
||||
type ConverterInfo,
|
||||
type LayoutMode,
|
||||
} from '../lib/convertService';
|
||||
|
||||
type ProductView = 'chooser' | 'conversion';
|
||||
|
||||
interface ProductHubProps {
|
||||
view: ProductView;
|
||||
onOpenEditor: () => void;
|
||||
onOpenConversion: () => void;
|
||||
onBack: () => void;
|
||||
}
|
||||
|
||||
const FileGlyph = ({ conversion = false }: { conversion?: boolean }) => (
|
||||
<div
|
||||
className={`flex h-12 w-12 shrink-0 items-center justify-center rounded-[14px] ${
|
||||
conversion ? 'bg-[#d9f5e8] text-[#087443]' : 'bg-[#e7edff] text-brand-primary'
|
||||
}`}
|
||||
>
|
||||
<svg
|
||||
width="24"
|
||||
height="24"
|
||||
viewBox="0 0 24 24"
|
||||
fill="none"
|
||||
stroke="currentColor"
|
||||
strokeWidth="1.8"
|
||||
strokeLinecap="round"
|
||||
strokeLinejoin="round"
|
||||
aria-hidden="true"
|
||||
>
|
||||
{conversion ? (
|
||||
<>
|
||||
<path d="M5 8h12" />
|
||||
<path d="m14 5 3 3-3 3" />
|
||||
<path d="M19 16H7" />
|
||||
<path d="m10 13-3 3 3 3" />
|
||||
</>
|
||||
) : (
|
||||
<>
|
||||
<path d="M6 3h8l4 4v14H6a2 2 0 0 1-2-2V5a2 2 0 0 1 2-2Z" />
|
||||
<path d="M14 3v5h5" />
|
||||
</>
|
||||
)}
|
||||
</svg>
|
||||
</div>
|
||||
);
|
||||
|
||||
const ArrowIcon = () => (
|
||||
<svg
|
||||
width="18"
|
||||
height="18"
|
||||
viewBox="0 0 24 24"
|
||||
fill="none"
|
||||
stroke="currentColor"
|
||||
strokeWidth="2"
|
||||
strokeLinecap="round"
|
||||
strokeLinejoin="round"
|
||||
aria-hidden="true"
|
||||
>
|
||||
<path d="M5 12h14" />
|
||||
<path d="m13 6 6 6-6 6" />
|
||||
</svg>
|
||||
);
|
||||
|
||||
const CheckIcon = () => (
|
||||
<svg
|
||||
width="16"
|
||||
height="16"
|
||||
viewBox="0 0 24 24"
|
||||
fill="none"
|
||||
stroke="currentColor"
|
||||
strokeWidth="2.5"
|
||||
strokeLinecap="round"
|
||||
strokeLinejoin="round"
|
||||
aria-hidden="true"
|
||||
>
|
||||
<path d="m5 12 4 4L19 6" />
|
||||
</svg>
|
||||
);
|
||||
|
||||
const BrandMark = () => (
|
||||
<div className="flex h-10 w-10 items-center justify-center rounded-[12px] bg-[#ff6b4a] text-white shadow-[0_8px_18px_rgba(255,107,74,0.25)]">
|
||||
<svg
|
||||
width="21"
|
||||
height="21"
|
||||
viewBox="0 0 24 24"
|
||||
fill="none"
|
||||
stroke="currentColor"
|
||||
strokeWidth="2"
|
||||
strokeLinecap="round"
|
||||
strokeLinejoin="round"
|
||||
aria-hidden="true"
|
||||
>
|
||||
<path d="M6 3h8l4 4v14H6a2 2 0 0 1-2-2V5a2 2 0 0 1 2-2Z" />
|
||||
<path d="M14 3v5h5" />
|
||||
<path d="M8 13h8M8 17h5" />
|
||||
</svg>
|
||||
</div>
|
||||
);
|
||||
|
||||
export const ProductHub: React.FC<ProductHubProps> = ({
|
||||
view,
|
||||
onOpenEditor,
|
||||
onOpenConversion,
|
||||
onBack,
|
||||
}) => {
|
||||
const fileRef = useRef<HTMLInputElement>(null);
|
||||
const [selectedFile, setSelectedFile] = useState<File | null>(null);
|
||||
const [converters, setConverters] = useState<ConverterInfo[]>(ENGINE_CONVERT_PAIRS);
|
||||
const [selectedPairId, setSelectedPairId] = useState(pairId('pdf', 'docx'));
|
||||
const [converting, setConverting] = useState(false);
|
||||
const [error, setError] = useState<string | null>(null);
|
||||
const [statusNote, setStatusNote] = useState<string | null>(null);
|
||||
|
||||
useEffect(() => {
|
||||
let cancelled = false;
|
||||
void fetchConvertInfo().then((info) => {
|
||||
if (!cancelled) setConverters(info);
|
||||
});
|
||||
return () => {
|
||||
cancelled = true;
|
||||
};
|
||||
}, []);
|
||||
|
||||
const sourceFormat = selectedFile ? inferSourceFormat(selectedFile) : null;
|
||||
const formatGroups = useMemo(() => groupedPairs(converters), [converters]);
|
||||
const uploadAccept = useMemo(() => uploadAcceptFromPairs(converters), [converters]);
|
||||
|
||||
const activePair = useMemo(
|
||||
() => converters.find((p) => pairId(p.source, p.target) === selectedPairId) ?? converters[0],
|
||||
[converters, selectedPairId]
|
||||
);
|
||||
|
||||
const formatEnabled = (format: ConverterInfo) => {
|
||||
if (!sourceFormat) return true;
|
||||
return format.source === sourceFormat;
|
||||
};
|
||||
|
||||
const formatBadge = (format: ConverterInfo) => {
|
||||
const ext = FORMAT_META[format.target]?.extension ?? format.target.toUpperCase();
|
||||
if (!sourceFormat) return ext;
|
||||
if (formatEnabled(format)) return ext;
|
||||
return `Need ${format.source.toUpperCase()}`;
|
||||
};
|
||||
|
||||
const destinationHint = (() => {
|
||||
if (!sourceFormat) {
|
||||
return 'Destinations come from the conversion engine. Upload a file to see which pairs unlock.';
|
||||
}
|
||||
const targets = converters
|
||||
.filter((p) => p.source === sourceFormat)
|
||||
.map((p) => (FORMAT_META[p.target]?.label ?? p.target.toUpperCase()));
|
||||
if (!targets.length) {
|
||||
return `Detected ${sourceFormat.toUpperCase()}, but the engine has no registered destinations for it.`;
|
||||
}
|
||||
return `${sourceFormat.toUpperCase()} selected → ${targets.join(', ')}.`;
|
||||
})();
|
||||
|
||||
const canRun =
|
||||
!!selectedFile &&
|
||||
!!sourceFormat &&
|
||||
!!activePair &&
|
||||
activePair.source === sourceFormat &&
|
||||
!converting;
|
||||
|
||||
const handleConvert = async () => {
|
||||
if (!selectedFile || !sourceFormat || !canRun || !activePair) return;
|
||||
setConverting(true);
|
||||
setError(null);
|
||||
setStatusNote(null);
|
||||
const useAsync = selectedFile.size >= ASYNC_THRESHOLD_BYTES;
|
||||
if (useAsync) {
|
||||
setStatusNote(
|
||||
'Large file — starting async job. Word rebuild OCR on scanned PDFs can take up to 30 minutes. Keep this tab open; the PDF editor is a different, faster path.'
|
||||
);
|
||||
}
|
||||
try {
|
||||
const result = await convertFileMaybeAsync(
|
||||
selectedFile,
|
||||
sourceFormat,
|
||||
activePair.target,
|
||||
{
|
||||
onProgress: (p) => {
|
||||
const elapsedMin = Math.max(1, Math.round(p.elapsedMs / 60_000));
|
||||
const budgetMin = Math.round(p.maxWaitMs / 60_000);
|
||||
setStatusNote(
|
||||
`Async job ${p.status}… ${elapsedMin}/${budgetMin} min (OCR + layout). Do not close this tab.`
|
||||
);
|
||||
},
|
||||
}
|
||||
);
|
||||
downloadBlob(result.blob, result.filename);
|
||||
const score =
|
||||
result.qualityScore != null
|
||||
? ` Quality score: ${(result.qualityScore * 100).toFixed(0)}%.`
|
||||
: '';
|
||||
const warn = result.warnings ? ` Warnings: ${result.warnings}` : '';
|
||||
const asyncHint = useAsync ? ' (async job path).' : '';
|
||||
setStatusNote(
|
||||
`Converted ${sourceFormat}→${activePair.target} with fidelity=${result.fidelity}.${asyncHint}${score}${warn}`
|
||||
);
|
||||
} catch (err) {
|
||||
setError(err instanceof Error ? err.message : 'Conversion failed');
|
||||
setStatusNote(null);
|
||||
} finally {
|
||||
setConverting(false);
|
||||
}
|
||||
};
|
||||
|
||||
const selectedLabel = activePair ? pairDisplayName(activePair) : 'Choose format';
|
||||
|
||||
if (view === 'chooser') {
|
||||
return (
|
||||
<main className="min-h-full overflow-auto bg-[#f5f7f9] text-[#16232d]">
|
||||
<div className="mx-auto flex min-h-screen w-full max-w-[1280px] flex-col px-5 py-5 sm:px-9 lg:px-12">
|
||||
<header className="flex items-center justify-between border-b border-[#dfe5e7] pb-5">
|
||||
<div className="flex items-center gap-3">
|
||||
<BrandMark />
|
||||
<div>
|
||||
<p className="text-[15px] font-extrabold tracking-tight">DocQube</p>
|
||||
<p className="text-[11px] font-bold uppercase tracking-[0.18em] text-[#82909a]">
|
||||
Your document desk
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
<div className="hidden items-center gap-6 text-[12px] font-semibold text-[#71808a] sm:flex">
|
||||
<span>All workspaces</span>
|
||||
<span className="h-2 w-2 rounded-full bg-[#53bd85]" /> <span>Secure & private</span>
|
||||
</div>
|
||||
</header>
|
||||
<section className="grid flex-1 items-center gap-12 py-12 lg:grid-cols-[0.8fr_1.2fr] lg:gap-20 lg:py-16">
|
||||
<div className="animate-[slideUp_400ms_ease-out]">
|
||||
<p className="mb-4 text-[11px] font-extrabold uppercase tracking-[0.2em] text-[#ff6b4a]">
|
||||
Good to see you
|
||||
</p>
|
||||
<h1 className="max-w-[500px] text-5xl font-extrabold leading-[1.02] tracking-[-0.04em] sm:text-7xl">
|
||||
Make your next document <span className="text-[#ff6b4a]">better.</span>
|
||||
</h1>
|
||||
<p className="mt-6 max-w-[430px] text-[16px] leading-7 text-[#63727c]">
|
||||
A focused space to polish PDFs, or move their contents into the tools you already use.
|
||||
</p>
|
||||
<div className="mt-8 flex items-center gap-3 text-[12px] font-bold text-[#63727c]">
|
||||
<span className="flex h-7 w-7 items-center justify-center rounded-full bg-[#d9f5e8] text-[#087443]">
|
||||
<CheckIcon />
|
||||
</span>{' '}
|
||||
Files stay on your device
|
||||
</div>
|
||||
</div>
|
||||
<div className="grid gap-4 sm:grid-cols-2">
|
||||
<button
|
||||
onClick={onOpenEditor}
|
||||
className="group relative flex min-h-[300px] flex-col justify-between overflow-hidden rounded-[22px] bg-[#172b3a] p-7 text-left text-white shadow-[0_20px_45px_rgba(23,43,58,0.16)] transition duration-200 hover:-translate-y-1 hover:shadow-[0_25px_55px_rgba(23,43,58,0.24)]"
|
||||
>
|
||||
<span className="absolute -right-10 -top-12 h-44 w-44 rounded-full border-[22px] border-white/5" />
|
||||
<div className="flex h-12 w-12 items-center justify-center rounded-[14px] bg-white/10 text-[#a8d5ff]">
|
||||
<FileGlyph />
|
||||
</div>
|
||||
<div>
|
||||
<div className="flex items-center justify-between gap-3">
|
||||
<h2 className="text-[22px] font-extrabold">Edit a PDF</h2>
|
||||
<ArrowIcon />
|
||||
</div>
|
||||
<p className="mt-3 text-[13px] leading-6 text-[#b8c8d1]">
|
||||
Annotate, rearrange, sign, protect, and convert your document.
|
||||
</p>
|
||||
<p className="mt-6 text-[10px] font-bold uppercase tracking-[0.18em] text-[#829aa8]">
|
||||
Open editor
|
||||
</p>
|
||||
</div>
|
||||
</button>
|
||||
<button
|
||||
onClick={onOpenConversion}
|
||||
className="group flex min-h-[300px] flex-col justify-between rounded-[22px] border border-[#bfe7d1] bg-[#d9f5e8] p-7 text-left text-[#12362a] shadow-[0_20px_45px_rgba(54,137,95,0.1)] transition duration-200 hover:-translate-y-1 hover:border-[#53bd85] hover:shadow-[0_25px_55px_rgba(54,137,95,0.18)]"
|
||||
>
|
||||
<div className="flex h-12 w-12 items-center justify-center rounded-[14px] bg-white/70 text-[#087443]">
|
||||
<FileGlyph conversion />
|
||||
</div>
|
||||
<div>
|
||||
<div className="flex items-center justify-between gap-3">
|
||||
<h2 className="text-[22px] font-extrabold">Convert a PDF</h2>
|
||||
<span className="text-[#087443] transition-transform group-hover:translate-x-1">
|
||||
<ArrowIcon />
|
||||
</span>
|
||||
</div>
|
||||
<p className="mt-3 text-[13px] leading-6 text-[#477363]">
|
||||
Turn pages into Word, spreadsheets, images, or text using the conversion engine.
|
||||
</p>
|
||||
<p className="mt-6 text-[10px] font-bold uppercase tracking-[0.18em] text-[#478c6e]">
|
||||
Choose a format
|
||||
</p>
|
||||
</div>
|
||||
</button>
|
||||
</div>
|
||||
</section>
|
||||
<footer className="flex items-center justify-between border-t border-[#dfe5e7] pt-5 text-[11px] font-semibold text-[#82909a]">
|
||||
<span>DocQube workspace</span>
|
||||
<span>Built for calm, careful work</span>
|
||||
</footer>
|
||||
</div>
|
||||
</main>
|
||||
);
|
||||
}
|
||||
|
||||
return (
|
||||
<main className="min-h-full overflow-auto bg-[#f5f7f9] text-[#16232d]">
|
||||
<div className="mx-auto min-h-screen w-full max-w-[1280px] px-5 py-5 sm:px-9 lg:px-12">
|
||||
<header className="flex items-center justify-between border-b border-[#dfe5e7] pb-5">
|
||||
<div className="flex items-center gap-3">
|
||||
<button
|
||||
onClick={onBack}
|
||||
className="flex h-9 w-9 items-center justify-center rounded-[10px] border border-[#d5dfe2] bg-white text-[#63727c] transition hover:border-[#ff6b4a] hover:text-[#ff6b4a]"
|
||||
aria-label="Back to workspace"
|
||||
>
|
||||
<
|
||||
</button>
|
||||
<div>
|
||||
<p className="text-[15px] font-extrabold">Conversion desk</p>
|
||||
<p className="text-[11px] font-bold uppercase tracking-[0.15em] text-[#82909a]">
|
||||
Engine-linked convert desk
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
<button
|
||||
onClick={onOpenEditor}
|
||||
className="hidden items-center gap-2 rounded-[10px] border border-[#d5dfe2] bg-white px-3.5 py-2.5 text-[12px] font-bold text-[#63727c] transition hover:border-[#172b3a] hover:text-[#172b3a] sm:flex"
|
||||
>
|
||||
<FileGlyph /> Open editor
|
||||
</button>
|
||||
</header>
|
||||
|
||||
<section className="py-10 sm:py-14">
|
||||
<div className="flex flex-col justify-between gap-5 sm:flex-row sm:items-end">
|
||||
<div>
|
||||
<p className="text-[11px] font-extrabold uppercase tracking-[0.2em] text-[#ff6b4a]">
|
||||
Conversion desk
|
||||
</p>
|
||||
<h1 className="mt-3 text-4xl font-extrabold leading-none tracking-[-0.04em] sm:text-6xl">
|
||||
Move it forward.
|
||||
</h1>
|
||||
<p className="mt-4 max-w-[520px] text-[15px] leading-7 text-[#63727c]">
|
||||
Formats listed here are the pairs registered in the document conversion engine.
|
||||
Office rebuilds are honest lossy/medium — not ConvertAPI/Word-print perfect.
|
||||
</p>
|
||||
</div>
|
||||
<div className="flex items-center gap-2 text-[11px] font-bold text-[#087443]">
|
||||
<span className="h-2 w-2 rounded-full bg-[#53bd85]" /> {converters.length} engine pairs
|
||||
</div>
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<section className="grid gap-5 lg:grid-cols-[0.75fr_1.25fr]">
|
||||
<div className="rounded-[22px] bg-[#172b3a] p-6 text-white shadow-[0_20px_45px_rgba(23,43,58,0.14)] sm:p-8">
|
||||
<div className="flex items-start justify-between">
|
||||
<div>
|
||||
<p className="text-[10px] font-extrabold uppercase tracking-[0.2em] text-[#a8d5ff]">
|
||||
Step 01
|
||||
</p>
|
||||
<h2 className="mt-3 text-2xl font-extrabold">Bring your file</h2>
|
||||
</div>
|
||||
<FileGlyph conversion />
|
||||
</div>
|
||||
<input
|
||||
ref={fileRef}
|
||||
type="file"
|
||||
accept={uploadAccept}
|
||||
className="hidden"
|
||||
onChange={(event) => {
|
||||
const file = event.target.files?.[0] ?? null;
|
||||
setSelectedFile(file);
|
||||
setError(null);
|
||||
setStatusNote(null);
|
||||
if (file) {
|
||||
const src = inferSourceFormat(file);
|
||||
const first = converters.find((f) => f.source === src);
|
||||
if (first) setSelectedPairId(pairId(first.source, first.target));
|
||||
}
|
||||
}}
|
||||
/>
|
||||
<button
|
||||
onClick={() => fileRef.current?.click()}
|
||||
className="mt-12 flex min-h-[145px] w-full flex-col items-center justify-center rounded-[16px] border border-dashed border-[#6b8796] bg-white/5 px-4 text-center transition hover:border-[#a8d5ff] hover:bg-white/10"
|
||||
>
|
||||
<span className="text-[14px] font-bold">
|
||||
{selectedFile ? selectedFile.name : 'Choose a file from your device'}
|
||||
</span>
|
||||
<span className="mt-2 text-[11px] text-[#aebfc8]">
|
||||
{selectedFile
|
||||
? `Detected: ${sourceFormat ?? 'unknown'} — click to replace`
|
||||
: 'PDF, DOCX, XLSX, PPTX, HTML, MD, PNG, JPEG, TIFF'}
|
||||
</span>
|
||||
</button>
|
||||
<div className="mt-6 flex items-center gap-2 text-[11px] text-[#aebfc8]">
|
||||
<CheckIcon /> Original file remains unchanged
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div className="rounded-[22px] border border-[#dce4e5] bg-white p-6 shadow-[0_16px_35px_rgba(35,55,65,0.06)] sm:p-8">
|
||||
<div className="flex items-start justify-between">
|
||||
<div>
|
||||
<p className="text-[10px] font-extrabold uppercase tracking-[0.2em] text-[#ff6b4a]">
|
||||
Step 02
|
||||
</p>
|
||||
<h2 className="mt-3 text-2xl font-extrabold">Choose a destination</h2>
|
||||
</div>
|
||||
<span className="rounded-full bg-[#fff0eb] px-3 py-1.5 text-[11px] font-extrabold text-[#d65338]">
|
||||
{selectedLabel}
|
||||
</span>
|
||||
</div>
|
||||
<p className="mt-4 text-[11px] leading-5 text-[#5f717c]">{destinationHint}</p>
|
||||
|
||||
<div className="mt-6 grid gap-6 sm:grid-cols-2">
|
||||
{formatGroups.map((group) => (
|
||||
<div key={group.label}>
|
||||
<h3 className="mb-3 text-[10px] font-extrabold uppercase tracking-[0.18em] text-[#82909a]">
|
||||
{group.label}
|
||||
</h3>
|
||||
<div className="grid gap-2">
|
||||
{group.formats.map((format) => {
|
||||
const enabled = formatEnabled(format);
|
||||
const id = pairId(format.source, format.target);
|
||||
return (
|
||||
<button
|
||||
key={id}
|
||||
disabled={!enabled}
|
||||
onClick={() => setSelectedPairId(id)}
|
||||
className={`flex min-h-[55px] items-center justify-between rounded-[12px] border px-3.5 text-left transition ${
|
||||
selectedPairId === id
|
||||
? 'border-[#ff6b4a] bg-[#fff7f4] shadow-[0_4px_12px_rgba(255,107,74,0.1)]'
|
||||
: enabled
|
||||
? 'border-[#e2e8e9] bg-white hover:border-[#ffb3a2]'
|
||||
: 'border-[#eef1f2] bg-[#f7f9f9]'
|
||||
} ${!enabled ? 'cursor-not-allowed opacity-55' : ''}`}
|
||||
>
|
||||
<span>
|
||||
<span className="block text-[12px] font-extrabold">{pairDisplayName(format)}</span>
|
||||
<span className="block text-[10px] text-[#82909a]">{pairDescription(format)}</span>
|
||||
</span>
|
||||
<span
|
||||
className={`ml-2 shrink-0 text-[10px] font-extrabold uppercase tracking-[0.08em] ${
|
||||
selectedPairId === id
|
||||
? 'text-[#ff6b4a]'
|
||||
: enabled
|
||||
? 'text-[#087443]'
|
||||
: 'text-[#9aa6ad]'
|
||||
}`}
|
||||
>
|
||||
{formatBadge(format)}
|
||||
</span>
|
||||
</button>
|
||||
);
|
||||
})}
|
||||
</div>
|
||||
</div>
|
||||
))}
|
||||
</div>
|
||||
|
||||
{sourceFormat === 'pdf' && (
|
||||
<p className="mt-6 rounded-[14px] border border-[#e7ecec] bg-[#f7f9f9] px-4 py-3 text-[11px] leading-5 text-[#63727c]">
|
||||
The engine chooses the best reconstruction for each page. Word and Excel are
|
||||
rebuilt from the document (lossy); PNG, JPEG and TIFF are page-accurate images.
|
||||
</p>
|
||||
)}
|
||||
|
||||
<div className="mt-8 flex flex-col gap-3 border-t border-[#e7ecec] pt-6 sm:flex-row sm:items-center sm:justify-between">
|
||||
<p className="text-[11px] leading-5 text-[#82909a]">
|
||||
Live pairs from GET /v1/convert/info. Large files (≥{Math.round(ASYNC_THRESHOLD_BYTES / (1024 * 1024))}
|
||||
MB) use async jobs. Destinations that need another source type show Need DOCX/XLSX.
|
||||
</p>
|
||||
<button
|
||||
disabled={!canRun}
|
||||
onClick={() => void handleConvert()}
|
||||
className="flex items-center justify-center gap-2 rounded-[10px] bg-[#ff6b4a] px-5 py-3 text-[12px] font-extrabold text-white shadow-[0_8px_18px_rgba(255,107,74,0.2)] transition enabled:hover:bg-[#e5573b] disabled:cursor-not-allowed disabled:bg-[#d7dede] disabled:shadow-none"
|
||||
>
|
||||
{converting ? 'Converting… keep tab open' : `Convert to ${selectedLabel}`}{' '}
|
||||
<ArrowIcon />
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{error ? (
|
||||
<p className="mt-4 rounded-[10px] border border-[#f5c2c0] bg-[#fff5f5] px-3 py-2 text-[12px] font-semibold text-[#b42318]">
|
||||
{error}
|
||||
</p>
|
||||
) : null}
|
||||
{statusNote ? (
|
||||
<p
|
||||
className={`mt-4 rounded-[10px] border px-3 py-2 text-[12px] font-semibold ${
|
||||
converting
|
||||
? 'border-[#c5d0d6] bg-[#f4f7f8] text-[#3d4f5a]'
|
||||
: 'border-[#b7e4c7] bg-[#f0fdf4] text-[#166534]'
|
||||
}`}
|
||||
>
|
||||
{statusNote}
|
||||
</p>
|
||||
) : null}
|
||||
</div>
|
||||
</section>
|
||||
</div>
|
||||
</main>
|
||||
);
|
||||
};
|
||||
@@ -17,6 +17,8 @@ interface SidebarProps {
|
||||
onOpenSignature: () => void;
|
||||
onOpenAbout: () => void;
|
||||
disabledTools?: Set<ToolId>;
|
||||
isOCRLoading?: boolean;
|
||||
isRemote?: boolean;
|
||||
}
|
||||
|
||||
interface GroupDef {
|
||||
@@ -59,7 +61,7 @@ const SidebarButton: React.FC<{
|
||||
);
|
||||
|
||||
export const Sidebar: React.FC<SidebarProps> = ({
|
||||
activeTool, activeGroup, onToolChange, onGroupChange, hasSignature, onOpenSignature, onOpenAbout, disabledTools
|
||||
activeTool, activeGroup, onToolChange, onGroupChange, onOpenAbout, disabledTools
|
||||
}) => {
|
||||
return (
|
||||
<nav className="flex h-full shrink-0 flex-col items-center gap-2 overflow-y-auto border-r border-border-primary bg-bg-primary scroll-micro" style={{ width: '92px', paddingTop: '16px', paddingBottom: '16px' }}>
|
||||
|
||||
@@ -116,7 +116,6 @@ export const Toolbar: React.FC<ToolbarProps> = ({
|
||||
pendingRedactionCount = 0, onApplyRedactions, onClearRedactions, onRedactPages, onOpenWatermark,
|
||||
selectedAnnotation, onUpdateAnnotation, onDeleteAnnotation, onDeselectAnnotation
|
||||
}) => {
|
||||
const meta = TOOL_META[activeTool];
|
||||
const isRedact = activeTool === 'redact';
|
||||
const toolsInGroup = GROUP_TOOLS[activeGroup] || [];
|
||||
|
||||
|
||||
@@ -40,6 +40,7 @@ interface TopBarProps {
|
||||
canExport?: boolean;
|
||||
canAssemble?: boolean;
|
||||
isRemote?: boolean;
|
||||
onBackToHome?: () => void;
|
||||
}
|
||||
|
||||
const ZOOM_PRESETS = [0.5, 0.75, 1, 1.25, 1.5, 2, 3];
|
||||
@@ -48,7 +49,7 @@ export const TopBar: React.FC<TopBarProps> = ({
|
||||
documentName, backendHealthy, engineReady, zoom, onZoomChange, onFitWidth,
|
||||
currentPage, totalPages, onGoToPage, canUndo, canRedo, onUndo, onRedo,
|
||||
isSaving, isDirtySaved, onRotate, onExport, onPrint, onProtect, onUnlock, onCompare, onMergePDF, isEncrypted, onUpload, onNewBlankPDF, onSave,
|
||||
canPrint = true, canExport = true, canAssemble = true, isRemote = false,
|
||||
canPrint = true, canExport = true, canAssemble = true, isRemote = false, onBackToHome,
|
||||
}) => {
|
||||
const fileRef = useRef<HTMLInputElement>(null);
|
||||
const handleFile = (e: React.ChangeEvent<HTMLInputElement>) => {
|
||||
@@ -63,6 +64,7 @@ export const TopBar: React.FC<TopBarProps> = ({
|
||||
style={{ height: '56px', paddingLeft: '24px', paddingRight: '24px' }}
|
||||
>
|
||||
<div className="flex shrink-0 items-center gap-2">
|
||||
{onBackToHome && <CustomButton variant="unstyled" className="flex h-8 items-center rounded-[8px] px-2 text-[12px] font-bold text-text-secondary hover:bg-bg-tertiary hover:text-text-primary" onClick={onBackToHome} title="Back to workspace">< Workspaces</CustomButton>}
|
||||
<div className="flex items-center gap-2">
|
||||
<div className="flex h-8 w-8 items-center justify-center rounded-[9px] bg-brand-primary text-white shadow-sm">
|
||||
<svg width="17" height="17" viewBox="0 0 24 24" fill="none" stroke="currentColor" strokeWidth={2} strokeLinecap="round" strokeLinejoin="round">
|
||||
@@ -85,7 +87,7 @@ export const TopBar: React.FC<TopBarProps> = ({
|
||||
{!isRemote && <MenuItem icon={<svg width="16" height="16" viewBox="0 0 24 24" fill="none" stroke="currentColor" strokeWidth={2} strokeLinecap="round" strokeLinejoin="round"><path d="M14 2H6a2 2 0 0 0-2 2v16a2 2 0 0 0 2 2h12a2 2 0 0 0 2-2V8z" /><polyline points="14 2 14 8 20 8" /><line x1="12" y1="18" x2="12" y2="12" /><line x1="9" y1="15" x2="15" y2="15" /></svg>} onClick={() => onNewBlankPDF?.()}>New Blank PDF…</MenuItem>}
|
||||
{!isRemote && <MenuItem icon={<UploadIcon size={16} />} onClick={() => fileRef.current?.click()}>Open PDF…</MenuItem>}
|
||||
<MenuItem icon={<svg width="16" height="16" viewBox="0 0 24 24" fill="none" stroke="currentColor" strokeWidth={2} strokeLinecap="round" strokeLinejoin="round"><path d="M8 7v10M12 4v16M16 7v10M4 11h16M4 15h16"/></svg>} onClick={() => onMergePDF?.()}>Merge PDFs…</MenuItem>
|
||||
<MenuItem icon={<DownloadIcon size={16} />} onClick={onExport} disabled={!documentName || !canExport}>Export / Download</MenuItem>
|
||||
<MenuItem icon={<DownloadIcon size={16} />} onClick={onExport} disabled={!documentName || !canExport}>Export / Convert</MenuItem>
|
||||
<MenuItem icon={<svg width="16" height="16" viewBox="0 0 24 24" fill="none" stroke="currentColor" strokeWidth={2} strokeLinecap="round" strokeLinejoin="round"><polyline points="6 9 6 2 18 2 18 9"></polyline><path d="M6 18H4a2 2 0 0 1-2-2v-5a2 2 0 0 1 2-2h16a2 2 0 0 1 2 2v5a2 2 0 0 1-2 2h-2"></path><rect x="6" y="14" width="12" height="8"></rect></svg>} onClick={onPrint} disabled={!documentName || !canPrint}>Print</MenuItem>
|
||||
<MenuItem icon={<svg width="16" height="16" viewBox="0 0 24 24" fill="none" stroke="currentColor" strokeWidth={2} strokeLinecap="round" strokeLinejoin="round"><path d="M9 5H7a2 2 0 0 0-2 2v12a2 2 0 0 0 2 2h10a2 2 0 0 0 2-2V7a2 2 0 0 0-2-2h-2"/><rect x="9" y="3" width="6" height="4" rx="1"/></svg>} onClick={() => onCompare?.()} disabled={!documentName}>Compare PDFs…</MenuItem>
|
||||
{isEncrypted ? (
|
||||
|
||||
@@ -0,0 +1,554 @@
|
||||
/**
|
||||
* DocQube conversion API client (Wave 1+ OSS converters).
|
||||
*/
|
||||
|
||||
export type ConvertFidelity = 'high' | 'medium' | 'lossy';
|
||||
|
||||
export type LayoutMode = 'flowing' | 'continuous' | 'nocolumns' | 'exact';
|
||||
export type OcrPolicy = 'auto' | 'force' | 'never';
|
||||
export type OcrEngineChoice = 'auto' | 'rapid' | 'mistral';
|
||||
export type HeaderFooterMode = 'detect' | 'ignore' | 'remove';
|
||||
export type RecognitionMode = 'enhanced_flow' | 'flow' | 'textbox';
|
||||
|
||||
export interface ConvertDeskOptions {
|
||||
layoutMode?: LayoutMode;
|
||||
ocrPolicy?: OcrPolicy;
|
||||
ocrEngine?: OcrEngineChoice;
|
||||
headerFooterMode?: HeaderFooterMode;
|
||||
recognitionMode?: RecognitionMode;
|
||||
detectTables?: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
* Reconstruction defaults, kept for internal callers and tests only.
|
||||
*
|
||||
* The conversion desk does NOT send these. Omitting every reconstruction
|
||||
* parameter is what tells the gateway to route the document itself — which
|
||||
* path suits a clean digital contract is not the same as what suits a
|
||||
* photographed invoice, and the customer should not have to know that.
|
||||
*/
|
||||
export const DEFAULT_CONVERT_OPTIONS: Required<ConvertDeskOptions> = {
|
||||
layoutMode: 'flowing',
|
||||
ocrPolicy: 'auto',
|
||||
ocrEngine: 'auto',
|
||||
headerFooterMode: 'detect',
|
||||
recognitionMode: 'enhanced_flow',
|
||||
detectTables: true,
|
||||
};
|
||||
|
||||
/**
|
||||
* Query string for a convert call.
|
||||
*
|
||||
* With no options this is empty, so the server applies automatic routing.
|
||||
* Only the keys a caller explicitly sets are sent: filling in the rest from
|
||||
* the defaults would read as "explicit" to the gateway and silently disable
|
||||
* the routing.
|
||||
*/
|
||||
function convertQuery(opts?: ConvertDeskOptions): string {
|
||||
if (!opts) return '';
|
||||
const q = new URLSearchParams();
|
||||
if (opts.layoutMode) q.set('layout_mode', opts.layoutMode);
|
||||
if (opts.ocrPolicy) q.set('ocr_policy', opts.ocrPolicy);
|
||||
if (opts.ocrEngine) q.set('ocr_engine', opts.ocrEngine);
|
||||
if (opts.headerFooterMode) q.set('header_footer_mode', opts.headerFooterMode);
|
||||
if (opts.recognitionMode) q.set('recognition_mode', opts.recognitionMode);
|
||||
if (opts.detectTables !== undefined) q.set('detect_tables', opts.detectTables ? 'true' : 'false');
|
||||
return q.toString();
|
||||
}
|
||||
|
||||
export interface ConvertJob {
|
||||
job_id: string;
|
||||
status: 'queued' | 'running' | 'completed' | 'failed';
|
||||
source?: string;
|
||||
target?: string;
|
||||
fidelity?: ConvertFidelity | null;
|
||||
warnings?: string[];
|
||||
filename?: string | null;
|
||||
size_bytes?: number | null;
|
||||
error?: string | null;
|
||||
download_path?: string | null;
|
||||
meta?: Record<string, unknown>;
|
||||
}
|
||||
|
||||
export interface ConvertResult {
|
||||
blob: Blob;
|
||||
fidelity: string;
|
||||
warnings: string;
|
||||
filename: string;
|
||||
/** 0..1 from X-Quality-Score when the gateway provides it */
|
||||
qualityScore: number | null;
|
||||
statusCode?: number;
|
||||
}
|
||||
|
||||
/** Sync convert size above which we prefer async job polling (bytes). */
|
||||
export const ASYNC_THRESHOLD_BYTES = 8 * 1024 * 1024;
|
||||
|
||||
function gatewayBaseUrl(): string {
|
||||
const url = import.meta.env.VITE_GATEWAY_URL as string | undefined;
|
||||
if (!url) {
|
||||
throw new Error('VITE_GATEWAY_URL is not configured');
|
||||
}
|
||||
return url.replace(/\/$/, '');
|
||||
}
|
||||
|
||||
export function extensionToTarget(extension: string): string {
|
||||
const ext = extension.toLowerCase().replace(/^\./, '');
|
||||
if (ext === 'jpg') return 'jpeg';
|
||||
if (ext === 'data') return 'json';
|
||||
return ext;
|
||||
}
|
||||
|
||||
/** Infer convert source from filename / MIME. */
|
||||
export function inferSourceFormat(file: File): string {
|
||||
const name = (file.name || '').toLowerCase();
|
||||
const ext = name.includes('.') ? name.split('.').pop() || '' : '';
|
||||
if (ext === 'jpg' || file.type === 'image/jpeg') return 'jpeg';
|
||||
if (ext === 'png' || file.type === 'image/png') return 'png';
|
||||
if (ext === 'tif' || ext === 'tiff' || file.type === 'image/tiff') return 'tiff';
|
||||
if (ext === 'docx') return 'docx';
|
||||
if (ext === 'xlsx') return 'xlsx';
|
||||
if (ext === 'csv' || file.type === 'text/csv') return 'csv';
|
||||
if (ext === 'pptx') return 'pptx';
|
||||
if (ext === 'html' || ext === 'htm') return 'html';
|
||||
if (ext === 'md' || ext === 'markdown') return 'md';
|
||||
if (ext === 'txt') return 'txt';
|
||||
if (ext === 'json' || file.type === 'application/json') return 'json';
|
||||
if (ext === 'pdf' || file.type === 'application/pdf') return 'pdf';
|
||||
return ext || 'pdf';
|
||||
}
|
||||
|
||||
export interface ConverterInfo {
|
||||
source: string;
|
||||
target: string;
|
||||
fidelity: ConvertFidelity;
|
||||
description?: string;
|
||||
gaps?: string[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Fallback pairs matching gateway `converters.all_plugins()`.
|
||||
* Used when `/v1/convert/info` is unreachable so the editor still lists
|
||||
* the engine's Wave-1 formats rather than inventing planned ones.
|
||||
*/
|
||||
export const ENGINE_CONVERT_PAIRS: ConverterInfo[] = [
|
||||
{ source: 'pdf', target: 'pdf', fidelity: 'high', description: 'PDF->PDF OCR Searchable rebuild' },
|
||||
{ source: 'pdf', target: 'docx', fidelity: 'lossy', description: 'PDF->DOCX via layout IDM' },
|
||||
{ source: 'docx', target: 'pdf', fidelity: 'medium', description: 'DOCX->PDF via reportlab' },
|
||||
{ source: 'pdf', target: 'xlsx', fidelity: 'lossy', description: 'PDF->XLSX via layout IDM' },
|
||||
{ source: 'xlsx', target: 'pdf', fidelity: 'medium', description: 'XLSX->PDF via reportlab' },
|
||||
{ source: 'pdf', target: 'csv', fidelity: 'lossy', description: 'PDF->CSV via layout IDM' },
|
||||
{ source: 'csv', target: 'pdf', fidelity: 'high', description: 'CSV->PDF via reportlab' },
|
||||
{ source: 'pdf', target: 'pptx', fidelity: 'medium', description: 'PDF->PPTX via layout IDM' },
|
||||
{ source: 'pptx', target: 'pdf', fidelity: 'high', description: 'PPTX->PDF via slide renderer' },
|
||||
{ source: 'pdf', target: 'txt', fidelity: 'medium', description: 'PDF->TXT via IDM' },
|
||||
{ source: 'txt', target: 'pdf', fidelity: 'high', description: 'TXT->PDF via reportlab' },
|
||||
{ source: 'json', target: 'pdf', fidelity: 'high', description: 'JSON->PDF via styled grid or inspector' },
|
||||
{ source: 'pdf', target: 'md', fidelity: 'medium', description: 'PDF->Markdown via IDM' },
|
||||
{ source: 'docx', target: 'md', fidelity: 'medium', description: 'DOCX->Markdown' },
|
||||
{ source: 'xlsx', target: 'md', fidelity: 'medium', description: 'XLSX->Markdown' },
|
||||
{ source: 'pptx', target: 'md', fidelity: 'medium', description: 'PPTX->Markdown' },
|
||||
{ source: 'pdf', target: 'json', fidelity: 'medium', description: 'PDF->JSON idm.v1' },
|
||||
{ source: 'pdf', target: 'html', fidelity: 'medium', description: 'PDF->HTML via IDM' },
|
||||
{ source: 'pdf', target: 'png', fidelity: 'high', description: 'PDF->PNG (ZIP if multi-page)' },
|
||||
{ source: 'pdf', target: 'jpeg', fidelity: 'high', description: 'PDF->JPEG (ZIP if multi-page)' },
|
||||
{ source: 'pdf', target: 'tiff', fidelity: 'high', description: 'PDF->TIFF' },
|
||||
{ source: 'png', target: 'pdf', fidelity: 'high', description: 'PNG->PDF' },
|
||||
{ source: 'jpeg', target: 'pdf', fidelity: 'high', description: 'JPEG->PDF' },
|
||||
{ source: 'tiff', target: 'pdf', fidelity: 'high', description: 'TIFF->PDF' },
|
||||
{ source: 'html', target: 'pdf', fidelity: 'medium', description: 'HTML->PDF via reportlab' },
|
||||
{ source: 'md', target: 'html', fidelity: 'high', description: 'Markdown->HTML' },
|
||||
{ source: 'md', target: 'pdf', fidelity: 'medium', description: 'Markdown->PDF' },
|
||||
{ source: 'html', target: 'md', fidelity: 'medium', description: 'HTML->Markdown' },
|
||||
{ source: 'docx', target: 'txt', fidelity: 'high', description: 'DOCX->TXT' },
|
||||
{ source: 'docx', target: 'json', fidelity: 'medium', description: 'DOCX->JSON' },
|
||||
{ source: 'docx', target: 'html', fidelity: 'medium', description: 'DOCX->HTML via mammoth' },
|
||||
{ source: 'html', target: 'docx', fidelity: 'medium', description: 'HTML->DOCX' },
|
||||
{ source: 'docx', target: 'docx', fidelity: 'high', description: 'DOCX passthrough' },
|
||||
];
|
||||
|
||||
export const FORMAT_META: Record<
|
||||
string,
|
||||
{ label: string; extension: string; group: string }
|
||||
> = {
|
||||
pdf: { label: 'PDF', extension: 'PDF', group: 'PDF' },
|
||||
docx: { label: 'Word', extension: 'DOCX', group: 'Office' },
|
||||
xlsx: { label: 'Excel', extension: 'XLSX', group: 'Office' },
|
||||
csv: { label: 'CSV', extension: 'CSV', group: 'Office' },
|
||||
pptx: { label: 'PowerPoint', extension: 'PPTX', group: 'Office' },
|
||||
html: { label: 'HTML', extension: 'HTML', group: 'Web and text' },
|
||||
md: { label: 'Markdown', extension: 'MD', group: 'Web and text' },
|
||||
txt: { label: 'Plain text', extension: 'TXT', group: 'Web and text' },
|
||||
json: { label: 'JSON', extension: 'JSON', group: 'Web and text' },
|
||||
png: { label: 'PNG', extension: 'PNG', group: 'Images' },
|
||||
jpeg: { label: 'JPEG', extension: 'JPG', group: 'Images' },
|
||||
tiff: { label: 'TIFF', extension: 'TIFF', group: 'Images' },
|
||||
};
|
||||
|
||||
export const CONVERT_GROUP_ORDER = ['Office', 'Web and text', 'Images', 'PDF'];
|
||||
|
||||
export function pairId(source: string, target: string): string {
|
||||
return `${source}->${target}`;
|
||||
}
|
||||
|
||||
export function fileExtensionForTarget(target: string): string {
|
||||
const t = extensionToTarget(target);
|
||||
return t === 'jpeg' ? 'jpg' : t;
|
||||
}
|
||||
|
||||
export function replaceFilenameExtension(name: string, target: string): string {
|
||||
const stem = name.replace(/\.[^.]+$/, '') || 'document';
|
||||
return `${stem}.${fileExtensionForTarget(target)}`;
|
||||
}
|
||||
|
||||
export function pairDisplayName(pair: ConverterInfo): string {
|
||||
const tgt = FORMAT_META[pair.target]?.label ?? pair.target.toUpperCase();
|
||||
if (pair.source === pair.target) return `${tgt} passthrough`;
|
||||
if (pair.target === 'pdf') return `PDF from ${pair.source.toUpperCase()}`;
|
||||
if (pair.source === 'pdf') return tgt;
|
||||
return `${tgt} from ${pair.source.toUpperCase()}`;
|
||||
}
|
||||
|
||||
export function pairDescription(pair: ConverterInfo): string {
|
||||
const fid = pair.fidelity;
|
||||
if (pair.description) return `${pair.description} (${fid})`;
|
||||
return `${pair.source.toUpperCase()} → ${pair.target.toUpperCase()} (${fid})`;
|
||||
}
|
||||
|
||||
let cachedConvertInfo: ConverterInfo[] | null = null;
|
||||
|
||||
/** Live converter registry from the document conversion engine. */
|
||||
export async function fetchConvertInfo(): Promise<ConverterInfo[]> {
|
||||
if (cachedConvertInfo?.length) return cachedConvertInfo;
|
||||
try {
|
||||
const res = await fetch(`${gatewayBaseUrl()}/v1/convert/info`);
|
||||
if (res.ok) {
|
||||
const data = (await res.json()) as { converters?: ConverterInfo[] };
|
||||
if (Array.isArray(data.converters) && data.converters.length > 0) {
|
||||
cachedConvertInfo = data.converters;
|
||||
return cachedConvertInfo;
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
/* fall through to engine fallback */
|
||||
}
|
||||
return ENGINE_CONVERT_PAIRS;
|
||||
}
|
||||
|
||||
export function targetsFromSource(pairs: ConverterInfo[], source: string): ConverterInfo[] {
|
||||
const src = extensionToTarget(source);
|
||||
return pairs.filter((p) => p.source === src);
|
||||
}
|
||||
|
||||
export function groupedPairs(pairs: ConverterInfo[]): { label: string; formats: ConverterInfo[] }[] {
|
||||
const buckets = new Map<string, ConverterInfo[]>();
|
||||
for (const pair of pairs) {
|
||||
const group = FORMAT_META[pair.target]?.group ?? 'More formats';
|
||||
const list = buckets.get(group) ?? [];
|
||||
list.push(pair);
|
||||
buckets.set(group, list);
|
||||
}
|
||||
const ordered = CONVERT_GROUP_ORDER.filter((g) => buckets.has(g)).map((label) => ({
|
||||
label,
|
||||
formats: buckets.get(label)!,
|
||||
}));
|
||||
for (const [label, formats] of buckets) {
|
||||
if (!CONVERT_GROUP_ORDER.includes(label)) ordered.push({ label, formats });
|
||||
}
|
||||
return ordered;
|
||||
}
|
||||
|
||||
export function uploadAcceptFromPairs(pairs: ConverterInfo[]): string {
|
||||
const extras: Record<string, string[]> = {
|
||||
pdf: ['.pdf', 'application/pdf'],
|
||||
docx: ['.docx'],
|
||||
xlsx: ['.xlsx'],
|
||||
pptx: ['.pptx'],
|
||||
html: ['.html', '.htm', 'text/html'],
|
||||
md: ['.md', '.markdown'],
|
||||
png: ['.png', 'image/png'],
|
||||
jpeg: ['.jpg', '.jpeg', 'image/jpeg'],
|
||||
tiff: ['.tif', '.tiff', 'image/tiff'],
|
||||
json: ['.json', 'application/json'],
|
||||
};
|
||||
const tokens = new Set<string>();
|
||||
for (const pair of pairs) {
|
||||
(extras[pair.source] ?? [`.${pair.source}`]).forEach((t) => tokens.add(t));
|
||||
}
|
||||
return [...tokens].join(',');
|
||||
}
|
||||
|
||||
export async function canConvert(source: string, target: string): Promise<boolean> {
|
||||
const res = await fetch(
|
||||
`${gatewayBaseUrl()}/v1/convert/canconvert/${source}/to/${target}`
|
||||
);
|
||||
if (!res.ok) return false;
|
||||
const data = (await res.json()) as { can_convert?: boolean };
|
||||
return Boolean(data.can_convert);
|
||||
}
|
||||
|
||||
function parseErrorDetail(status: number, bodyText: string): string {
|
||||
const statusHint =
|
||||
status === 504
|
||||
? 'Conversion timed out (504). Try a smaller file or raise CONVERT_TIMEOUT_SECONDS.'
|
||||
: status === 422
|
||||
? 'This conversion pair is not supported (422).'
|
||||
: status === 400
|
||||
? 'Invalid input (400).'
|
||||
: `Conversion failed (${status}).`;
|
||||
try {
|
||||
const body = JSON.parse(bodyText) as { detail?: unknown };
|
||||
if (typeof body?.detail === 'string' && body.detail.trim()) return body.detail;
|
||||
if (Array.isArray(body?.detail)) return JSON.stringify(body.detail);
|
||||
} catch {
|
||||
if (bodyText.trim()) return bodyText.slice(0, 400);
|
||||
}
|
||||
return statusHint;
|
||||
}
|
||||
|
||||
function parseFilenameFromDisposition(disposition: string, fallback: string): string {
|
||||
// Prefer RFC 5987 filename*=UTF-8''...
|
||||
const star = /filename\*\s*=\s*UTF-8''([^;]+)/i.exec(disposition);
|
||||
if (star?.[1]) {
|
||||
try {
|
||||
return decodeURIComponent(star[1].trim().replace(/^"+|"+$/g, ''));
|
||||
} catch {
|
||||
/* fall through */
|
||||
}
|
||||
}
|
||||
const plain = /filename\s*=\s*"([^"]+)"|filename\s*=\s*([^;]+)/i.exec(disposition);
|
||||
const raw = (plain?.[1] || plain?.[2] || '').trim();
|
||||
return raw || fallback;
|
||||
}
|
||||
|
||||
function friendlyNetworkError(err: unknown): Error {
|
||||
const msg = err instanceof Error ? err.message : String(err);
|
||||
if (/failed to fetch|networkerror|load failed/i.test(msg)) {
|
||||
return new Error(
|
||||
'Could not reach the convert gateway (or download crashed). ' +
|
||||
'Confirm VITE_GATEWAY_URL points at a running gateway on 8080, then retry.'
|
||||
);
|
||||
}
|
||||
return err instanceof Error ? err : new Error(msg);
|
||||
}
|
||||
|
||||
async function parseConvertResponse(res: Response, target: string): Promise<ConvertResult> {
|
||||
if (!res.ok) {
|
||||
const bodyText = await res.text();
|
||||
const err = new Error(parseErrorDetail(res.status, bodyText));
|
||||
(err as Error & { statusCode?: number }).statusCode = res.status;
|
||||
throw err;
|
||||
}
|
||||
|
||||
const blob = await res.blob();
|
||||
const disposition = res.headers.get('Content-Disposition') || '';
|
||||
const filename = parseFilenameFromDisposition(
|
||||
disposition,
|
||||
`converted.${target === 'jpeg' ? 'jpg' : target}`
|
||||
);
|
||||
const qsRaw = res.headers.get('X-Quality-Score');
|
||||
const qualityScore = qsRaw != null && qsRaw !== '' ? Number.parseFloat(qsRaw) : null;
|
||||
return {
|
||||
blob,
|
||||
fidelity: res.headers.get('X-Fidelity') || 'lossy',
|
||||
warnings: res.headers.get('X-Warnings') || '',
|
||||
filename,
|
||||
qualityScore: Number.isFinite(qualityScore) ? qualityScore : null,
|
||||
statusCode: res.status,
|
||||
};
|
||||
}
|
||||
|
||||
export async function convertFile(
|
||||
file: File,
|
||||
source: string,
|
||||
targetExtension: string,
|
||||
deskOptions?: ConvertDeskOptions
|
||||
): Promise<ConvertResult> {
|
||||
const target = extensionToTarget(targetExtension);
|
||||
const form = new FormData();
|
||||
form.append('file', file, file.name || `upload.${source}`);
|
||||
const qs = convertQuery(deskOptions);
|
||||
|
||||
let res: Response;
|
||||
try {
|
||||
res = await fetch(`${gatewayBaseUrl()}/v1/convert/${source}/to/${target}${qs ? `?${qs}` : ''}`, {
|
||||
method: 'POST',
|
||||
body: form,
|
||||
});
|
||||
} catch (err) {
|
||||
throw friendlyNetworkError(err);
|
||||
}
|
||||
return parseConvertResponse(res, target);
|
||||
}
|
||||
|
||||
/**
|
||||
* Client poll budget for async convert.
|
||||
*
|
||||
* Must stay at CONVERT_TIMEOUT_MAX_SECONDS (gateway default 1800s). A
|
||||
* size-scaled wait used to abort a 15 MB bilingual scan at ~13 min while
|
||||
* the job was still inside the 20 min OCR wall — that is a desk timeout,
|
||||
* not a convert failure. The PDF editor never hits this path: it only
|
||||
* renders pages.
|
||||
*/
|
||||
export const ASYNC_POLL_CAP_MS = 1_800_000;
|
||||
|
||||
export function asyncPollMaxWaitMs(_fileSizeBytes: number): number {
|
||||
return ASYNC_POLL_CAP_MS;
|
||||
}
|
||||
|
||||
function nextPollDelayMs(elapsedMs: number, fallback: number): number {
|
||||
if (elapsedMs < 60_000) return fallback;
|
||||
if (elapsedMs < 300_000) return Math.max(fallback, 3_000);
|
||||
return Math.max(fallback, 5_000);
|
||||
}
|
||||
|
||||
export type AsyncConvertProgress = {
|
||||
jobId: string;
|
||||
status: ConvertJob['status'];
|
||||
elapsedMs: number;
|
||||
maxWaitMs: number;
|
||||
};
|
||||
|
||||
async function downloadCompletedJob(
|
||||
jobId: string,
|
||||
target: string,
|
||||
status: ConvertJob
|
||||
): Promise<ConvertResult> {
|
||||
let dl: Response;
|
||||
try {
|
||||
dl = await fetch(`${gatewayBaseUrl()}/v1/jobs/${jobId}/download`);
|
||||
} catch (err) {
|
||||
throw friendlyNetworkError(err);
|
||||
}
|
||||
const result = await parseConvertResponse(dl, target);
|
||||
// Prefer UTF-8 original stem from job metadata when header fallback is ASCII-only
|
||||
if (status.filename && /[^\x00-\x7F]/.test(status.filename)) {
|
||||
result.filename = status.filename;
|
||||
}
|
||||
const metaScore = status.meta?.quality_score;
|
||||
if (result.qualityScore == null && typeof metaScore === 'number') {
|
||||
result.qualityScore = metaScore;
|
||||
}
|
||||
if (!result.fidelity && status.fidelity) result.fidelity = status.fidelity;
|
||||
if (!result.warnings && status.warnings?.length) {
|
||||
result.warnings = status.warnings.join(' | ');
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
/** Prefer async path for large uploads; poll until completed/failed. */
|
||||
export async function convertFileMaybeAsync(
|
||||
file: File,
|
||||
source: string,
|
||||
targetExtension: string,
|
||||
opts?: {
|
||||
forceAsync?: boolean;
|
||||
pollMs?: number;
|
||||
maxWaitMs?: number;
|
||||
onProgress?: (p: AsyncConvertProgress) => void;
|
||||
deskOptions?: ConvertDeskOptions;
|
||||
}
|
||||
): Promise<ConvertResult> {
|
||||
const target = extensionToTarget(targetExtension);
|
||||
const useAsync = opts?.forceAsync || file.size >= ASYNC_THRESHOLD_BYTES;
|
||||
if (!useAsync) {
|
||||
return convertFile(file, source, target, opts?.deskOptions);
|
||||
}
|
||||
|
||||
const form = new FormData();
|
||||
form.append('file', file, file.name || `upload.${source}`);
|
||||
const qs = convertQuery(opts?.deskOptions);
|
||||
let start: Response;
|
||||
try {
|
||||
start = await fetch(
|
||||
`${gatewayBaseUrl()}/v1/async/convert/${source}/to/${target}${qs ? `?${qs}` : ''}`,
|
||||
{
|
||||
method: 'POST',
|
||||
body: form,
|
||||
}
|
||||
);
|
||||
} catch (err) {
|
||||
throw friendlyNetworkError(err);
|
||||
}
|
||||
if (!start.ok) {
|
||||
const bodyText = await start.text();
|
||||
throw new Error(parseErrorDetail(start.status, bodyText));
|
||||
}
|
||||
const job = (await start.json()) as ConvertJob;
|
||||
const pollMs = opts?.pollMs ?? 1500;
|
||||
const maxWaitMs = opts?.maxWaitMs ?? asyncPollMaxWaitMs(file.size);
|
||||
const t0 = Date.now();
|
||||
|
||||
const emit = (status: ConvertJob['status']) => {
|
||||
opts?.onProgress?.({
|
||||
jobId: job.job_id,
|
||||
status,
|
||||
elapsedMs: Date.now() - t0,
|
||||
maxWaitMs,
|
||||
});
|
||||
};
|
||||
emit(job.status || 'queued');
|
||||
|
||||
while (Date.now() - t0 < maxWaitMs) {
|
||||
await new Promise((r) => setTimeout(r, nextPollDelayMs(Date.now() - t0, pollMs)));
|
||||
let st: Response;
|
||||
try {
|
||||
st = await fetch(`${gatewayBaseUrl()}/v1/jobs/${job.job_id}`);
|
||||
} catch (err) {
|
||||
throw friendlyNetworkError(err);
|
||||
}
|
||||
if (!st.ok) {
|
||||
throw new Error(parseErrorDetail(st.status, await st.text()));
|
||||
}
|
||||
const status = (await st.json()) as ConvertJob;
|
||||
emit(status.status);
|
||||
if (status.status === 'failed') {
|
||||
throw new Error(status.error || 'Async conversion failed');
|
||||
}
|
||||
if (status.status === 'completed') {
|
||||
return downloadCompletedJob(job.job_id, target, status);
|
||||
}
|
||||
}
|
||||
|
||||
// Final check: job may have completed in the last poll interval (race).
|
||||
try {
|
||||
const last = await fetch(`${gatewayBaseUrl()}/v1/jobs/${job.job_id}`);
|
||||
if (last.ok) {
|
||||
const status = (await last.json()) as ConvertJob;
|
||||
if (status.status === 'completed') {
|
||||
return downloadCompletedJob(job.job_id, target, status);
|
||||
}
|
||||
if (status.status === 'failed') {
|
||||
throw new Error(status.error || 'Async conversion failed');
|
||||
}
|
||||
}
|
||||
} catch (err) {
|
||||
if (err instanceof Error && !/failed to fetch|networkerror/i.test(err.message)) {
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
const mins = Math.round(maxWaitMs / 60_000);
|
||||
throw new Error(
|
||||
`Async conversion timed out after ~${mins} min while job ${job.job_id} was still running. ` +
|
||||
`Keep the gateway running and retry Convert — a finished job can still be downloaded. ` +
|
||||
`Do not close the tab while Converting is shown.`
|
||||
);
|
||||
}
|
||||
|
||||
export async function convertPdfFile(
|
||||
file: File,
|
||||
targetExtension: string
|
||||
): Promise<ConvertResult> {
|
||||
return convertFileMaybeAsync(file, 'pdf', targetExtension);
|
||||
}
|
||||
|
||||
export function downloadBlob(blob: Blob, filename: string): void {
|
||||
const url = URL.createObjectURL(blob);
|
||||
const a = document.createElement('a');
|
||||
a.href = url;
|
||||
a.download = filename;
|
||||
document.body.appendChild(a);
|
||||
a.click();
|
||||
a.remove();
|
||||
URL.revokeObjectURL(url);
|
||||
}
|
||||
@@ -9,6 +9,9 @@ export function useTheme() {
|
||||
if (savedTheme) {
|
||||
return savedTheme;
|
||||
}
|
||||
if (window.matchMedia && window.matchMedia('(prefers-color-scheme: dark)').matches) {
|
||||
return 'dark';
|
||||
}
|
||||
}
|
||||
return 'light';
|
||||
});
|
||||
|
||||
+18
-3
@@ -34,8 +34,11 @@ RUN git clone https://github.com/microsoft/vcpkg.git /opt/vcpkg \
|
||||
&& /opt/vcpkg/bootstrap-vcpkg.sh -disableMetrics
|
||||
ENV VCPKG_ROOT=/opt/vcpkg
|
||||
|
||||
# Grab uv binary for high-speed package management
|
||||
COPY --from=ghcr.io/astral-sh/uv:latest /uv /uvx /bin/
|
||||
|
||||
# Ensure a recent CMake is available (system cmake in slim images can be too old for vcpkg)
|
||||
RUN python -m pip install --upgrade pip cmake
|
||||
RUN uv pip install --system cmake
|
||||
|
||||
# Cache vcpkg dependencies in a separate layer
|
||||
COPY vcpkg.json ./
|
||||
@@ -102,13 +105,25 @@ RUN groupadd --system app \
|
||||
|
||||
WORKDIR /home/app
|
||||
|
||||
# Copy uv binary for high-speed package installation
|
||||
COPY --from=ghcr.io/astral-sh/uv:latest /uv /uvx /bin/
|
||||
|
||||
# Copy gateway Python code
|
||||
COPY gateway/pyproject.toml gateway/README.md ./
|
||||
RUN python -m pip install --upgrade pip \
|
||||
&& pip install --no-cache-dir -e ".[dev]"
|
||||
RUN uv pip install --system --no-cache -e ".[dev]"
|
||||
|
||||
COPY gateway/app ./app
|
||||
COPY gateway/tests ./tests
|
||||
# Model files are runtime inputs, not source-only documentation. Omitting
|
||||
# them makes the image start successfully while silently disabling Arabic OCR
|
||||
# and layout ML because all model lookups are intentionally fail-open.
|
||||
COPY gateway/models ./models
|
||||
COPY gateway/tools/verify_models.py ./tools/verify_models.py
|
||||
|
||||
# Fail the image build if a vendored model is missing, truncated, or has drifted
|
||||
# from the checked-in checksum manifest. Optional capabilities remain optional
|
||||
# according to MANIFEST.json, but a copied file can never be silently corrupt.
|
||||
RUN python tools/verify_models.py
|
||||
|
||||
RUN chown -R app:app /home/app
|
||||
USER app
|
||||
|
||||
@@ -40,7 +40,29 @@ for `import pdfengine` lazily and reports `engine_available: false` via
|
||||
|
||||
Requires Python 3.11+ (3.12 works). Build dirs are git-ignored.
|
||||
|
||||
### Fast path with `uv` (Recommended — 10x faster)
|
||||
|
||||
```powershell
|
||||
# Windows
|
||||
cd gateway
|
||||
uv venv
|
||||
.venv\Scripts\Activate.ps1
|
||||
uv pip install -e ".[dev]"
|
||||
```
|
||||
|
||||
```sh
|
||||
# Linux/macOS
|
||||
cd gateway
|
||||
uv venv
|
||||
source .venv/bin/activate
|
||||
uv pip install -e ".[dev]"
|
||||
```
|
||||
|
||||
<details>
|
||||
<summary>Standard <code>pip</code> fallback</summary>
|
||||
|
||||
```powershell
|
||||
# Windows
|
||||
cd gateway
|
||||
python -m venv .venv
|
||||
.venv\Scripts\Activate.ps1
|
||||
@@ -54,6 +76,7 @@ python -m venv .venv
|
||||
source .venv/bin/activate
|
||||
pip install -e ".[dev]"
|
||||
```
|
||||
</details>
|
||||
|
||||
Run the service:
|
||||
|
||||
@@ -77,6 +100,45 @@ uvicorn app.main:app --reload --port 8080
|
||||
# → http://127.0.0.1:<port>/health
|
||||
# → http://127.0.0.1:<port>/docs (OpenAPI UI)
|
||||
|
||||
## Document conversion API (OSS)
|
||||
|
||||
Wave 1+ converters live under `app/routers/convert` and `app/services/convert`.
|
||||
No ConvertAPI/Aspose/PyMuPDF — `python-docx`, `openpyxl`, `reportlab`, `pypdf`,
|
||||
and the C++ engine (when present) for text/layout/render.
|
||||
|
||||
| Method | Path | Notes |
|
||||
| --- | --- | --- |
|
||||
| GET | `/v1/convert/info` | Registered converter pairs + fidelity/gaps |
|
||||
| GET | `/v1/convert/canconvert/{from}/to/{to}` | Capability probe |
|
||||
| POST | `/v1/convert/{from}/to/{to}` | Sync convert (multipart `file`) |
|
||||
| POST | `/v1/async/convert/{from}/to/{to}` | Returns `job_id` |
|
||||
| GET | `/v1/jobs/{id}` | Job status |
|
||||
| GET | `/v1/jobs/{id}/download` | Result bytes |
|
||||
|
||||
Success responses include `X-Fidelity: high|medium|lossy`, optional `X-Quality-Score` (0–1), and `X-Warnings`.
|
||||
|
||||
**Honesty (M2 + prod readiness):** `X-Quality-Score` is output-based; live table score uses source-derived grids. Sync convert uses `asyncio.to_thread`; hard timeouts return **504**. Soft cancel is between-pages only. Wave-1 Office stays `lossy`/`medium` — **shippable**, not ConvertAPI-class. Prod smoke: `pdf/corpus/convert/real/` + `tests/convert/test_prod_smoke.py`. Frontend: `pdf/frontend` Conversion desk.
|
||||
|
||||
**Accuracy path (Wave-1 climb):** PDF→Office/text goes through layout → Intermediate Document Model (**idm.v1**) → formatters. Tables **v3** (shared column schema + rulings + X-gap; under-detect preferred), span/font runs into DOCX/XLSX, headers/footers, multi-column reading order, RapidOCR + live Mistral OCR when configured. Corpus CI under `pdf/corpus/convert/` — see [docs/accuracy_climb.md](docs/accuracy_climb.md). No ConvertAPI/Aspose; **no LibreOffice**.
|
||||
|
||||
**Office→PDF (Path A — reportlab max):** in-process, concurrent-safe, no LibreOffice.
|
||||
DOCX: BaseDocTemplate frames (incl. **multi-column**), font alias map, runs/lists/tables/nested
|
||||
tables/images/**text boxes**/HF. XLSX: number formats, fills, merges, print area/titles,
|
||||
freeze→repeat rows, column-band split, **chart→reportlab Drawing** (bar/line/pie). Fidelity
|
||||
stays **medium** (not Word/Excel print clone). SSIM harness is a structural **proxy only** (no
|
||||
raster SSIM in CI). Timeouts: `CONVERT_TIMEOUT_SECONDS`; OCR: `CONVERT_OCR_PAGE_CAP`.
|
||||
|
||||
**Implemented pairs (honest fidelity):** PDF↔DOCX/XLSX (lossy/medium), DOCX→PDF,
|
||||
PDF→TXT/MD/JSON/HTML/PNG/JPEG/TIFF, HTML→PDF, MD→HTML/PDF, images→PDF. `pdf→json` exports **idm.v1**.
|
||||
|
||||
**Rejected in v1:** legacy `.doc`, `.docm`, encrypted PDFs, unknown pairs (422).
|
||||
|
||||
**Windows / multi-user note:** DOCX/XLSX→PDF uses **reportlab** in-process so many concurrent converts stay light. Fidelity is medium (not Word/Excel print-identical).
|
||||
|
||||
```powershell
|
||||
pytest tests/convert -v
|
||||
```
|
||||
|
||||
|
||||
Lint, format, and test (the exact commands CI runs):
|
||||
|
||||
@@ -95,6 +157,8 @@ Environment variables are read by `app/config.py` with the prefix
|
||||
|------------------------------|---------|--------------------------------------------------|
|
||||
| `PDFENGINE_ENVIRONMENT` | `dev` | `dev` / `staging` / `prod` — echoed in `/health` |
|
||||
| `PDFENGINE_ENGINE_AVAILABLE` | `false` | Forces the engine-availability flag for testing |
|
||||
| `PDFENGINE_CORS_ALLOWED_ORIGINS` | `http://localhost:5173,http://127.0.0.1:5173` | Comma-separated browser origins allowed to call the gateway. |
|
||||
| `PDFENGINE_CORS_ALLOW_CREDENTIALS` | `false` | Enables cookies/credentials only for explicitly listed origins; wildcard is rejected. |
|
||||
|
||||
Additionally, the gateway startup scripts and frontend configuration support:
|
||||
|
||||
@@ -102,6 +166,67 @@ Additionally, the gateway startup scripts and frontend configuration support:
|
||||
|--------------------|-------------------------|--------------------------------------------------|
|
||||
| `PORT` | `8765` | Gateway listening port (used by `start_gateway.ps1`). |
|
||||
| `VITE_GATEWAY_URL` | `http://127.0.0.1:8765` | URL of the gateway API (used by the frontend). |
|
||||
| `CONVERT_TIMEOUT_SECONDS` | `120` | Hard convert timeout; exceeded runs return **504**. |
|
||||
| `CONVERT_OCR_PAGE_CAP` | `200` | Max pages OCR'd per document; capped runs report skipped pages in metadata/warnings. |
|
||||
| `CONVERT_OCR_IMAGE_MAX_BYTES` | `15000000` | Per-page OCR raster budget; oversized pages are downsampled/compressed, never dropped. |
|
||||
| `CONVERT_FIGURE_MAX_BYTES` | `4000000` | In-memory figure budget; large figures are compacted while retained. |
|
||||
| `CONVERT_OCR_PAGE_TIMEOUT` | `30` | Per-page OCR timeout (seconds). |
|
||||
| `CONVERT_LAYOUT_ML` | `0` | `1` enables optional Apache ONNX layout beside heuristics (fail-open). |
|
||||
| `CONVERT_LAYOUT_ML_WEIGHTS` | `models/layout/v1/layout.onnx` | ONNX path (relative to gateway cwd). |
|
||||
| `CONVERT_LAYOUT_ML_DEVICE` | `cpu` | `cpu` or `cuda`. |
|
||||
| `CONVERT_LAYOUT_ML_DPI` | `150` | Page raster DPI for layout ML. |
|
||||
| `CONVERT_LAYOUT_ML_PAGE_TIMEOUT` | `15` | Per-page ML soft timeout (seconds). |
|
||||
| `CONVERT_LAYOUT_ML_MIN_SCORE` | `0.5` | Drop weak detection boxes. |
|
||||
| `CONVERT_LAYOUT_ML_TABLE_STRUCTURE` | `0` | `1` enables optional table-structure ONNX (column separators only, fail-open). |
|
||||
| `CONVERT_LAYOUT_ML_TABLE_STRUCTURE_WEIGHTS` | `models/layout/table/structure.onnx` | Structure ONNX path; missing ⇒ heuristics, no error. |
|
||||
| `CONVERT_LAYOUT_ML_TABLE_STRUCTURE_MIN_SCORE` | `0.5` | Drop weak column/row boxes. |
|
||||
| `CONVERT_LAYOUT_ML_TABLE_STRUCTURE_TIMEOUT` | `10` | Per-region structure timeout (seconds). |
|
||||
| `CONVERT_OCR_ARABIC` | auto | Dual-pass Arabic rec; auto = on when `rec.onnx` exists. |
|
||||
| `CONVERT_OCR_ARABIC_ADAPTIVE` | 0 (startup script: 1) | When enabled, skip the Arabic pass for high-confidence Latin pages and retry it for empty/garbled/Arabic-looking EN output. Set 0/always for forced dual-pass behavior. |
|
||||
| `CONVERT_OCR_ARABIC_WEIGHTS` | `models/ocr/ar/v5/rec.onnx` | Arabic PP-OCRv5 rec ONNX path. |
|
||||
| `CONVERT_OCR_ARABIC_DICT` | `models/ocr/ar/v5/arabic_dict.txt` | Arabic charset dict. |
|
||||
| `CONVERT_OCR_REC_MODEL` | (unset) | Optional Latin/CJK **rec** ONNX. Do not reuse this path for det/cls. |
|
||||
| `CONVERT_OCR_DET_MODEL` | (unset) | Optional text **det** ONNX. |
|
||||
| `CONVERT_OCR_CLS_MODEL` | (unset) | Optional angle **cls** ONNX. |
|
||||
|
||||
Convert plugins default to **50 MB** max upload (`max_bytes`). The `pdf/frontend` Conversion desk uses async jobs for files ≥ 8 MB.
|
||||
|
||||
**Start gateway (recommended on Windows):** use the helper so port 8080 is freed and the **project venv** is used (avoids WinError 10048 from an orphan `python3.12` still listening):
|
||||
|
||||
```powershell
|
||||
cd pdf\gateway
|
||||
.\start_convert_gateway.ps1
|
||||
```
|
||||
|
||||
If you start manually and see `only one usage of each socket address` / errno 10048:
|
||||
|
||||
```powershell
|
||||
# Find and stop the holder
|
||||
Get-NetTCPConnection -LocalPort 8080 -State Listen | Select OwningProcess
|
||||
Stop-Process -Id <PID> -Force
|
||||
```
|
||||
|
||||
Always start with `.\.venv\Scripts\python.exe` from `pdf\gateway` — not Windows Store `python.exe`.
|
||||
|
||||
**Layout ML / public eval:** see [`docs/ml_layout_free_stack.md`](docs/ml_layout_free_stack.md). Generate smoke fixtures:
|
||||
|
||||
```powershell
|
||||
cd pdf
|
||||
.\gateway\.venv\Scripts\python.exe scripts\convert\generate_public_smoke_fixtures.py
|
||||
.\gateway\.venv\Scripts\python.exe -m pytest gateway\tests\convert -q -k "not scanned_page_count"
|
||||
```
|
||||
|
||||
Enable layout ML later (PowerShell — quote URLs; no `<placeholders>`):
|
||||
|
||||
```powershell
|
||||
cd pdf
|
||||
.\gateway\.venv\Scripts\python.exe scripts\convert\fetch_layout_onnx.py
|
||||
# when you have a real Apache ONNX URL:
|
||||
.\gateway\.venv\Scripts\python.exe scripts\convert\fetch_layout_onnx.py --url "https://YOUR-HOST/layout.onnx" --sha256 "YOUR_SHA256_HEX"
|
||||
cd gateway
|
||||
$env:CONVERT_LAYOUT_ML="1"
|
||||
$env:CONVERT_LAYOUT_ML_WEIGHTS="models/layout/v1/layout.onnx"
|
||||
```
|
||||
|
||||
A `.env` file in `gateway/` is auto-loaded if present (it is git-ignored
|
||||
via the repo-wide `.venv/` and Python rules — add `.env` to your local
|
||||
|
||||
@@ -0,0 +1,200 @@
|
||||
"""One-shot validator for an operator-owned reconverted DOCX. Not a test."""
|
||||
from __future__ import annotations
|
||||
|
||||
import collections
|
||||
import re
|
||||
import sys
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
|
||||
if sys.platform == "win32":
|
||||
try:
|
||||
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
|
||||
sys.stderr.reconfigure(encoding="utf-8", errors="replace")
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
from docx import Document
|
||||
from docx.oxml.ns import qn
|
||||
from lxml import etree
|
||||
|
||||
AR = re.compile(r"[\u0600-\u06FF]")
|
||||
GARBAGE = re.compile(
|
||||
r"Arob|AUTflOR|AUTHORIIY|ldentity|ctltztt|cu5I0M|Jil-riYl|IIOERAI|"
|
||||
r"IDINTITY|StCURtTy|@\s|Areb |Emlrates|Urted ",
|
||||
re.I,
|
||||
)
|
||||
# Generic OCR salad, not agency names
|
||||
SALAD = re.compile(r"\b(?:[A-Za-z]*[Il1]{2,}[A-Za-z]*){2,}\b")
|
||||
|
||||
|
||||
def all_text(doc: Document) -> list[str]:
|
||||
chunks = []
|
||||
for p in doc.paragraphs:
|
||||
t = p.text.strip()
|
||||
if t:
|
||||
chunks.append(t)
|
||||
for table in doc.tables:
|
||||
for row in table.rows:
|
||||
for cell in row.cells:
|
||||
t = cell.text.strip()
|
||||
if t:
|
||||
chunks.append(t)
|
||||
for section in doc.sections:
|
||||
for part in (section.header, section.footer):
|
||||
for p in part.paragraphs:
|
||||
t = p.text.strip()
|
||||
if t:
|
||||
chunks.append(f"[hf] {t}")
|
||||
for table in part.tables:
|
||||
for row in table.rows:
|
||||
for cell in row.cells:
|
||||
t = cell.text.strip()
|
||||
if t:
|
||||
chunks.append(f"[hf-table] {t}")
|
||||
return chunks
|
||||
|
||||
|
||||
def count_drawings(xml: bytes) -> int:
|
||||
root = etree.fromstring(xml)
|
||||
ns = {
|
||||
"w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main",
|
||||
"wp": "http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing",
|
||||
"a": "http://schemas.openxmlformats.org/drawingml/2006/main",
|
||||
"r": "http://schemas.openxmlformats.org/officeDocument/2006/relationships",
|
||||
}
|
||||
return len(root.xpath(".//w:drawing", namespaces=ns))
|
||||
|
||||
|
||||
def run_font_sizes(doc: Document) -> list[tuple[float, str]]:
|
||||
out = []
|
||||
for p in doc.paragraphs:
|
||||
for run in p.runs:
|
||||
t = (run.text or "").strip()
|
||||
if not t or run.font.size is None:
|
||||
continue
|
||||
pt = run.font.size.pt
|
||||
if pt and pt >= 20:
|
||||
out.append((pt, t[:80]))
|
||||
return out
|
||||
|
||||
|
||||
def page_breaks(doc: Document) -> int:
|
||||
n = 0
|
||||
body = doc.element.body
|
||||
for el in body.iter():
|
||||
if el.tag == qn("w:br") and el.get(qn("w:type")) == "page":
|
||||
n += 1
|
||||
if el.tag == qn("w:lastRenderedPageBreak"):
|
||||
n += 1
|
||||
return n
|
||||
|
||||
|
||||
def sect_page_fields(doc: Document) -> bool:
|
||||
xml = doc.element.xml
|
||||
return "PAGE" in xml or "w:instrText" in xml
|
||||
|
||||
|
||||
def main(path: Path) -> None:
|
||||
data = path.read_bytes()
|
||||
print(f"file={path.name}")
|
||||
print(f"bytes={len(data)}")
|
||||
doc = Document(str(path))
|
||||
paras = [p.text.strip() for p in doc.paragraphs if p.text.strip()]
|
||||
tables = doc.tables
|
||||
print(f"body_paras={len(paras)}")
|
||||
print(f"tables={len(tables)}")
|
||||
print(f"sections={len(doc.sections)}")
|
||||
|
||||
texts = all_text(doc)
|
||||
joined = "\n".join(texts)
|
||||
ar_chars = len(AR.findall(joined))
|
||||
print(f"arabic_chars={ar_chars}")
|
||||
print(f"latin_letters={sum(1 for c in joined if c.isascii() and c.isalpha())}")
|
||||
print(f"total_chars={len(joined)}")
|
||||
|
||||
hits = GARBAGE.findall(joined)
|
||||
print(f"known_garbage_hits={len(hits)} unique={sorted(set(hits))[:20]}")
|
||||
|
||||
logo_ph = sum(1 for t in texts if "[Logo]" in t or t.strip() == "Logo")
|
||||
print(f"logo_placeholders={logo_ph}")
|
||||
|
||||
# header/footer
|
||||
hdr_txt, ftr_txt, hdr_draw, ftr_draw = [], [], 0, 0
|
||||
page_field = False
|
||||
for i, section in enumerate(doc.sections):
|
||||
h, f = section.header, section.footer
|
||||
hdr_txt.extend(p.text.strip() for p in h.paragraphs if p.text.strip())
|
||||
ftr_txt.extend(p.text.strip() for p in f.paragraphs if p.text.strip())
|
||||
for t in h.tables:
|
||||
for row in t.rows:
|
||||
for c in row.cells:
|
||||
if c.text.strip():
|
||||
hdr_txt.append(c.text.strip())
|
||||
hdr_draw += count_drawings(h._element.xml.encode() if isinstance(h._element.xml, str) else h._element.xml)
|
||||
# python-docx returns str xml
|
||||
hx = etree.tostring(h._element)
|
||||
fx = etree.tostring(f._element)
|
||||
hdr_draw = count_drawings(hx) if i == 0 else hdr_draw + count_drawings(hx)
|
||||
ftr_draw += count_drawings(fx)
|
||||
fxml = f._element.xml
|
||||
if "PAGE" in fxml or "instrText" in fxml:
|
||||
page_field = True
|
||||
print("header_lines:")
|
||||
for t in hdr_txt[:15]:
|
||||
print(f" {t[:120]!r}")
|
||||
print("footer_lines:")
|
||||
for t in ftr_txt[:10]:
|
||||
print(f" {t[:120]!r}")
|
||||
print(f"header_drawings={hdr_draw} footer_drawings={ftr_draw} page_field={page_field}")
|
||||
|
||||
print("table_shapes:")
|
||||
for i, t in enumerate(tables):
|
||||
rows, cols = len(t.rows), len(t.columns)
|
||||
sample = " | ".join((c.text.strip()[:40] for c in t.rows[0].cells))
|
||||
print(f" t{i}: {rows}x{cols} :: {sample[:140]}")
|
||||
|
||||
sizes = run_font_sizes(doc)
|
||||
print(f"large_runs>={20}pt: {len(sizes)}")
|
||||
for pt, t in sizes[:12]:
|
||||
print(f" {pt}pt {t!r}")
|
||||
|
||||
# zip media
|
||||
with zipfile.ZipFile(path) as z:
|
||||
media = [n for n in z.namelist() if n.startswith("word/media/")]
|
||||
print(f"media={len(media)} {media}")
|
||||
for n in media:
|
||||
info = z.getinfo(n)
|
||||
print(f" {n} {info.file_size}b")
|
||||
rels = z.read("word/_rels/document.xml.rels").decode("utf-8", "replace")
|
||||
print(f"header_rels={'header' in ''.join(z.namelist())}")
|
||||
headers = [n for n in z.namelist() if "header" in n]
|
||||
footers = [n for n in z.namelist() if "footer" in n]
|
||||
print(f"header_parts={headers} footer_parts={footers}")
|
||||
|
||||
# first / last body paras
|
||||
print("first_paras:")
|
||||
for t in paras[:12]:
|
||||
print(f" {t[:160]!r}")
|
||||
print("sample_arabic_paras:")
|
||||
ar_paras = [t for t in paras if AR.search(t)]
|
||||
print(f" count={len(ar_paras)}")
|
||||
for t in ar_paras[:8]:
|
||||
print(f" {t[:160]!r}")
|
||||
|
||||
# garbage paragraphs
|
||||
bad = [t for t in paras if GARBAGE.search(t)]
|
||||
print(f"garbage_body_paras={len(bad)}")
|
||||
for t in bad[:15]:
|
||||
print(f" {t[:160]!r}")
|
||||
|
||||
# @ heading
|
||||
at = [t for t in paras if t.strip() in ("@",) or t.startswith("@ ")]
|
||||
print(f"at_paras={at[:5]}")
|
||||
|
||||
print(f"lastRendered_or_page_br_approx={page_breaks(doc)}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
p = Path(sys.argv[1])
|
||||
main(p)
|
||||
@@ -13,7 +13,7 @@ class BaseModelWrapper(ABC):
|
||||
self._is_loaded = False
|
||||
|
||||
@abstractmethod
|
||||
def load((self) -> bool:
|
||||
def load(self) -> bool:
|
||||
"""Load model weights into memory."""
|
||||
pass
|
||||
|
||||
|
||||
+46
-4
@@ -7,7 +7,7 @@ needed for `/health` and the app factory.
|
||||
|
||||
from functools import lru_cache
|
||||
|
||||
from pydantic import Field
|
||||
from pydantic import Field, model_validator
|
||||
from pydantic_settings import BaseSettings, SettingsConfigDict
|
||||
|
||||
|
||||
@@ -26,18 +26,60 @@ class Settings(BaseSettings):
|
||||
"Phase 0 default is False — routes that need the engine return 501."
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
render_cache_enabled: bool = Field(default=True)
|
||||
render_cache_max_entries: int = Field(default=256)
|
||||
render_cache_max_bytes: int = Field(
|
||||
default=536_870_912,
|
||||
description="Max total bytes for the tile render cache. Set to 0 to disable byte cap."
|
||||
description="Max total bytes for the tile render cache. Set to 0 to disable byte cap.",
|
||||
)
|
||||
render_cache_max_entry_bytes: int = Field(
|
||||
default=8_388_608,
|
||||
description="Max bytes for a single cache entry. Oversized tiles are rendered but not cached."
|
||||
description="Max bytes for a single cache entry. Oversized tiles are rendered but not cached.",
|
||||
)
|
||||
|
||||
gotenberg_url: str = Field(
|
||||
default="http://127.0.0.1:3000",
|
||||
description="URL for the Gotenberg headless microservice for high-fidelity Office->PDF.",
|
||||
)
|
||||
|
||||
# CORS is explicit by default. A comma-separated string keeps deployment
|
||||
# configuration ergonomic across Docker/PowerShell/systemd, while the
|
||||
# property below gives the middleware a normalized list.
|
||||
cors_allowed_origins: str = Field(
|
||||
default="http://localhost:5173,http://127.0.0.1:5173",
|
||||
description="Comma-separated browser origins allowed to call the gateway.",
|
||||
)
|
||||
cors_allow_credentials: bool = Field(
|
||||
default=False,
|
||||
description="Allow browser credentials only with an explicit origin allowlist.",
|
||||
)
|
||||
convert_max_concurrent: int = Field(
|
||||
default=4,
|
||||
description="Max concurrent conversion jobs executed simultaneously to prevent memory exhaustion.",
|
||||
)
|
||||
convert_queue_timeout_seconds: float = Field(
|
||||
default=60.0,
|
||||
description="Max seconds a request will wait in the queue before timing out with 504.",
|
||||
)
|
||||
|
||||
@property
|
||||
def cors_origins(self) -> list[str]:
|
||||
"""Return normalized CORS origins, preserving ``*`` when explicitly set."""
|
||||
return [origin.strip() for origin in self.cors_allowed_origins.split(",") if origin.strip()]
|
||||
|
||||
@model_validator(mode="after")
|
||||
def validate_cors(self) -> "Settings":
|
||||
"""Keep wildcard origins out of credentialed and production deployments."""
|
||||
if self.cors_allow_credentials and "*" in self.cors_origins:
|
||||
raise ValueError(
|
||||
"PDFENGINE_CORS_ALLOWED_ORIGINS cannot contain '*' when "
|
||||
"PDFENGINE_CORS_ALLOW_CREDENTIALS is enabled"
|
||||
)
|
||||
if self.environment.strip().lower() == "prod" and "*" in self.cors_origins:
|
||||
raise ValueError("PDFENGINE_CORS_ALLOWED_ORIGINS cannot contain '*' in production")
|
||||
return self
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def get_settings() -> Settings:
|
||||
|
||||
+27
-4
@@ -1,13 +1,26 @@
|
||||
"""FastAPI app factory."""
|
||||
|
||||
from contextlib import asynccontextmanager
|
||||
|
||||
from fastapi import FastAPI
|
||||
from fastapi.middleware.cors import CORSMiddleware
|
||||
|
||||
from app import __version__
|
||||
from app.routers import compare, documents, edits, health, info, internal, layout, ocr, render
|
||||
from app.services import ocr as ocr_service
|
||||
from app.ai.services.font_service import FontRecognitionService
|
||||
from app.config import get_settings
|
||||
from app.routers import (
|
||||
compare,
|
||||
convert,
|
||||
documents,
|
||||
edits,
|
||||
health,
|
||||
info,
|
||||
internal,
|
||||
layout,
|
||||
ocr,
|
||||
render,
|
||||
)
|
||||
from app.services import ocr as ocr_service
|
||||
|
||||
|
||||
@asynccontextmanager
|
||||
@@ -44,6 +57,7 @@ async def lifespan(app: FastAPI):
|
||||
|
||||
|
||||
def create_app() -> FastAPI:
|
||||
settings = get_settings()
|
||||
app = FastAPI(
|
||||
title="PDF Engine Gateway",
|
||||
version=__version__,
|
||||
@@ -53,14 +67,23 @@ def create_app() -> FastAPI:
|
||||
|
||||
app.add_middleware(
|
||||
CORSMiddleware,
|
||||
allow_origins=["*"],
|
||||
allow_credentials=True,
|
||||
allow_origins=settings.cors_origins,
|
||||
allow_credentials=settings.cors_allow_credentials,
|
||||
allow_methods=["*"],
|
||||
allow_headers=["*"],
|
||||
expose_headers=[
|
||||
"Content-Disposition",
|
||||
"X-Fidelity",
|
||||
"X-Warnings",
|
||||
"X-Quality-Score",
|
||||
"X-Convert-Source",
|
||||
"X-Convert-Target",
|
||||
],
|
||||
)
|
||||
|
||||
app.include_router(health.router)
|
||||
app.include_router(info.router)
|
||||
app.include_router(convert.router)
|
||||
app.include_router(documents.router)
|
||||
app.include_router(compare.router)
|
||||
app.include_router(render.router)
|
||||
|
||||
@@ -0,0 +1,9 @@
|
||||
"""Convert routers package."""
|
||||
|
||||
from fastapi import APIRouter
|
||||
|
||||
from . import info, jobs
|
||||
|
||||
router = APIRouter()
|
||||
router.include_router(info.router)
|
||||
router.include_router(jobs.router)
|
||||
@@ -0,0 +1,28 @@
|
||||
"""Convert info endpoints."""
|
||||
|
||||
from fastapi import APIRouter
|
||||
|
||||
from app.schemas.convert import CanConvertResponse, ConvertInfoResponse
|
||||
from app.services.convert.registry import registry
|
||||
|
||||
router = APIRouter(prefix="/v1/convert", tags=["convert"])
|
||||
|
||||
|
||||
@router.get("/info", response_model=ConvertInfoResponse)
|
||||
def convert_info() -> ConvertInfoResponse:
|
||||
return ConvertInfoResponse(converters=registry.list_info())
|
||||
|
||||
|
||||
@router.get("/canconvert/{source}/to/{target}", response_model=CanConvertResponse)
|
||||
def can_convert(source: str, target: str) -> CanConvertResponse:
|
||||
plugin = registry.get(source, target)
|
||||
if not plugin:
|
||||
return CanConvertResponse(can_convert=False, source=source, target=target)
|
||||
return CanConvertResponse(
|
||||
can_convert=True,
|
||||
source=plugin.source,
|
||||
target=plugin.target,
|
||||
fidelity=plugin.fidelity,
|
||||
engine=plugin.engine,
|
||||
gaps=list(plugin.gaps),
|
||||
)
|
||||
@@ -0,0 +1,502 @@
|
||||
"""Sync/async conversion job endpoints."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import os
|
||||
import threading
|
||||
import time
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from typing import Annotated
|
||||
|
||||
from fastapi import APIRouter, File, HTTPException, Query, UploadFile, status
|
||||
from fastapi.responses import Response
|
||||
|
||||
from app.schemas.convert import ConvertJobResponse, JobStatus
|
||||
from app.services.convert.http_headers import (
|
||||
ascii_header_value,
|
||||
content_disposition_attachment,
|
||||
)
|
||||
from app.services.convert.jobs import job_store
|
||||
from app.services.convert.options import (
|
||||
ConvertOptions,
|
||||
HeaderFooterMode,
|
||||
LayoutMode,
|
||||
OcrEngineChoice,
|
||||
OcrPolicy,
|
||||
RecognitionMode,
|
||||
)
|
||||
from app.services.convert.pipeline import ConversionError, run_conversion
|
||||
from app.services.convert.validate import ValidationError, media_type_for
|
||||
|
||||
router = APIRouter(prefix="/v1", tags=["convert"])
|
||||
|
||||
_MAX_WORKERS = max(1, int(os.environ.get("CONVERT_MAX_WORKERS", "4")))
|
||||
_MAX_CONCURRENT = max(1, int(os.environ.get("CONVERT_MAX_CONCURRENT", "4")))
|
||||
_QUEUE_TIMEOUT = float(os.environ.get("CONVERT_QUEUE_TIMEOUT_SECONDS", "60.0"))
|
||||
_MAX_QUEUED = max(
|
||||
0, int(os.environ.get("CONVERT_MAX_QUEUED", str(_MAX_WORKERS * 2)))
|
||||
)
|
||||
_executor = ThreadPoolExecutor(max_workers=_MAX_WORKERS, thread_name_prefix="convert-worker")
|
||||
_semaphore: asyncio.Semaphore | None = None
|
||||
# The executor itself has an unbounded work queue. This admission gate keeps
|
||||
# uploaded payloads and job records bounded at ``workers + queued`` jobs.
|
||||
_async_admission = threading.BoundedSemaphore(_MAX_WORKERS + _MAX_QUEUED)
|
||||
# Sync requests and async workers share the same actual conversion budget. A
|
||||
# separate asyncio semaphore cannot coordinate work running in executor
|
||||
# threads, so this process-wide slot guard is intentionally thread based.
|
||||
_execution_slots = threading.BoundedSemaphore(_MAX_CONCURRENT)
|
||||
|
||||
|
||||
def _get_semaphore() -> asyncio.Semaphore:
|
||||
global _semaphore
|
||||
if _semaphore is None:
|
||||
_semaphore = asyncio.Semaphore(_MAX_CONCURRENT)
|
||||
return _semaphore
|
||||
|
||||
|
||||
def _acquire_execution_slot(cancel_check=None) -> bool:
|
||||
"""Wait briefly at a time so queued async jobs can observe cancellation."""
|
||||
deadline = time.monotonic() + max(0.0, _QUEUE_TIMEOUT)
|
||||
while True:
|
||||
if cancel_check is not None and cancel_check():
|
||||
return False
|
||||
remaining = deadline - time.monotonic()
|
||||
if remaining <= 0:
|
||||
return False
|
||||
if _execution_slots.acquire(timeout=min(0.25, remaining)):
|
||||
return True
|
||||
|
||||
|
||||
def _run_sync_conversion(
|
||||
data: bytes,
|
||||
filename: str,
|
||||
source: str,
|
||||
target: str,
|
||||
cancel_check,
|
||||
options: ConvertOptions | None,
|
||||
on_resolved,
|
||||
):
|
||||
"""Run one synchronous conversion while holding a shared process slot."""
|
||||
if not _acquire_execution_slot(cancel_check):
|
||||
raise ConversionError(
|
||||
"Conversion queue timeout: server is at capacity. Please retry shortly.",
|
||||
status_code=504,
|
||||
)
|
||||
try:
|
||||
return run_conversion(
|
||||
data,
|
||||
filename,
|
||||
source,
|
||||
target,
|
||||
cancel_check,
|
||||
options,
|
||||
on_resolved,
|
||||
)
|
||||
finally:
|
||||
_execution_slots.release()
|
||||
|
||||
|
||||
def _options_metadata(
|
||||
requested: ConvertOptions | None,
|
||||
effective: ConvertOptions | None = None,
|
||||
) -> dict[str, object]:
|
||||
"""Expose the actual knobs while retaining whether routing was automatic."""
|
||||
if effective is None:
|
||||
if requested is None:
|
||||
return {"auto": True}
|
||||
return requested.as_dict()
|
||||
values: dict[str, object] = dict(effective.as_dict())
|
||||
if requested is None:
|
||||
values["auto"] = True
|
||||
return values
|
||||
|
||||
|
||||
def _options_from_query(
|
||||
layout_mode: LayoutMode | None,
|
||||
ocr_policy: OcrPolicy | None,
|
||||
ocr_engine: OcrEngineChoice | None,
|
||||
header_footer_mode: HeaderFooterMode | None,
|
||||
recognition_mode: RecognitionMode | None,
|
||||
detect_tables: bool | None,
|
||||
) -> ConvertOptions | None:
|
||||
"""Build options from the query, or None when the caller supplied none.
|
||||
|
||||
Every reconstruction parameter defaults to ``None`` rather than to its
|
||||
value, so "omitted" is distinguishable from "explicitly set to the
|
||||
default". The conversion desk sends none of them and gets automatic
|
||||
routing; CI and power users pass what they want and it is honoured
|
||||
exactly, including a value that happens to equal a default.
|
||||
"""
|
||||
supplied = (
|
||||
layout_mode,
|
||||
ocr_policy,
|
||||
ocr_engine,
|
||||
header_footer_mode,
|
||||
recognition_mode,
|
||||
detect_tables,
|
||||
)
|
||||
if all(v is None for v in supplied):
|
||||
return None
|
||||
|
||||
fallback = ConvertOptions()
|
||||
return ConvertOptions(
|
||||
layout_mode=layout_mode if layout_mode is not None else fallback.layout_mode,
|
||||
ocr_policy=ocr_policy if ocr_policy is not None else fallback.ocr_policy,
|
||||
ocr_engine=ocr_engine if ocr_engine is not None else fallback.ocr_engine,
|
||||
header_footer_mode=(
|
||||
header_footer_mode
|
||||
if header_footer_mode is not None
|
||||
else fallback.header_footer_mode
|
||||
),
|
||||
recognition_mode=(
|
||||
recognition_mode if recognition_mode is not None else fallback.recognition_mode
|
||||
),
|
||||
detect_tables=detect_tables if detect_tables is not None else fallback.detect_tables,
|
||||
)
|
||||
|
||||
|
||||
def _run_job(
|
||||
job_id: str,
|
||||
data: bytes,
|
||||
filename: str,
|
||||
source: str,
|
||||
target: str,
|
||||
options: ConvertOptions | None,
|
||||
) -> None:
|
||||
slot_acquired = False
|
||||
resolved: dict[str, ConvertOptions] = {}
|
||||
try:
|
||||
if job_store.is_cancelled(job_id):
|
||||
job_store.update(
|
||||
job_id,
|
||||
status=JobStatus.failed.value,
|
||||
error="cancelled",
|
||||
result_bytes=None,
|
||||
)
|
||||
return
|
||||
|
||||
# Executor workers may start before an execution slot is available
|
||||
# (for example while synchronous requests are using all slots). Poll
|
||||
# in short intervals so cancellation remains responsive.
|
||||
slot_acquired = _acquire_execution_slot(
|
||||
cancel_check=lambda: job_store.is_cancelled(job_id)
|
||||
)
|
||||
if not slot_acquired:
|
||||
if job_store.is_cancelled(job_id):
|
||||
job_store.update(
|
||||
job_id,
|
||||
status=JobStatus.failed.value,
|
||||
error="cancelled",
|
||||
result_bytes=None,
|
||||
)
|
||||
else:
|
||||
job_store.update(
|
||||
job_id,
|
||||
status=JobStatus.failed.value,
|
||||
error=(
|
||||
"Conversion queue timeout: server is at capacity. "
|
||||
"Please retry shortly."
|
||||
),
|
||||
)
|
||||
return
|
||||
|
||||
job_store.update(job_id, status=JobStatus.running.value)
|
||||
if job_store.is_cancelled(job_id):
|
||||
job_store.update(
|
||||
job_id,
|
||||
status=JobStatus.failed.value,
|
||||
error="cancelled",
|
||||
result_bytes=None,
|
||||
)
|
||||
return
|
||||
|
||||
out, fidelity, warnings, _media, out_name, quality_score = run_conversion(
|
||||
data,
|
||||
filename,
|
||||
source,
|
||||
target,
|
||||
cancel_check=lambda: job_store.is_cancelled(job_id),
|
||||
options=options,
|
||||
on_resolved=lambda opts: resolved.__setitem__("options", opts),
|
||||
)
|
||||
if job_store.is_cancelled(job_id):
|
||||
job_store.update(
|
||||
job_id,
|
||||
status=JobStatus.failed.value,
|
||||
error="cancelled",
|
||||
result_bytes=None,
|
||||
)
|
||||
return
|
||||
effective = resolved.get("options") or options
|
||||
meta: dict = {"convert_options": _options_metadata(options, effective)}
|
||||
if quality_score is not None:
|
||||
meta["quality_score"] = round(quality_score, 4)
|
||||
job_store.update(
|
||||
job_id,
|
||||
status=JobStatus.completed.value,
|
||||
fidelity=fidelity.value,
|
||||
warnings=warnings,
|
||||
size_bytes=len(out),
|
||||
result_bytes=out,
|
||||
result_filename=out_name,
|
||||
download_path=f"/v1/jobs/{job_id}/download",
|
||||
error=None,
|
||||
meta=meta,
|
||||
)
|
||||
except (ValidationError, ConversionError) as exc:
|
||||
code = getattr(exc, "status_code", 500)
|
||||
err_code = getattr(exc, "code", None)
|
||||
meta = {
|
||||
"status_code": code,
|
||||
"convert_options": _options_metadata(options, resolved.get("options")),
|
||||
}
|
||||
if err_code is not None:
|
||||
meta["code"] = getattr(err_code, "value", str(err_code))
|
||||
job_store.update(
|
||||
job_id,
|
||||
status=JobStatus.failed.value,
|
||||
error=str(exc),
|
||||
meta=meta,
|
||||
)
|
||||
except Exception as exc:
|
||||
job_store.update(job_id, status=JobStatus.failed.value, error=f"Unexpected error: {exc}")
|
||||
finally:
|
||||
if slot_acquired:
|
||||
_execution_slots.release()
|
||||
|
||||
|
||||
def _run_admitted_job(
|
||||
job_id: str,
|
||||
data: bytes,
|
||||
filename: str,
|
||||
source: str,
|
||||
target: str,
|
||||
options: ConvertOptions | None,
|
||||
) -> None:
|
||||
"""Run an admitted async job and always return its admission slot."""
|
||||
try:
|
||||
_run_job(job_id, data, filename, source, target, options)
|
||||
finally:
|
||||
_async_admission.release()
|
||||
|
||||
|
||||
@router.post("/convert/{source}/to/{target}")
|
||||
async def convert_sync(
|
||||
source: str,
|
||||
target: str,
|
||||
file: UploadFile = File(...),
|
||||
layout_mode: Annotated[LayoutMode | None, Query()] = None,
|
||||
ocr_policy: Annotated[OcrPolicy | None, Query()] = None,
|
||||
ocr_engine: Annotated[OcrEngineChoice | None, Query()] = None,
|
||||
header_footer_mode: Annotated[HeaderFooterMode | None, Query()] = None,
|
||||
recognition_mode: Annotated[RecognitionMode | None, Query()] = None,
|
||||
detect_tables: Annotated[bool | None, Query()] = None,
|
||||
) -> Response:
|
||||
data = await file.read()
|
||||
filename = file.filename or f"upload.{source}"
|
||||
options = _options_from_query(
|
||||
layout_mode,
|
||||
ocr_policy,
|
||||
ocr_engine,
|
||||
header_footer_mode,
|
||||
recognition_mode,
|
||||
detect_tables,
|
||||
)
|
||||
sem = _get_semaphore()
|
||||
acquired = False
|
||||
try:
|
||||
try:
|
||||
await asyncio.wait_for(sem.acquire(), timeout=_QUEUE_TIMEOUT)
|
||||
acquired = True
|
||||
except TimeoutError:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_504_GATEWAY_TIMEOUT,
|
||||
detail="Conversion queue timeout: server is at capacity. Please retry shortly.",
|
||||
)
|
||||
|
||||
resolved: dict[str, ConvertOptions] = {}
|
||||
out, fidelity, warnings, media, out_name, quality_score = await asyncio.to_thread(
|
||||
_run_sync_conversion,
|
||||
data,
|
||||
filename,
|
||||
source,
|
||||
target,
|
||||
None,
|
||||
options,
|
||||
lambda opts: resolved.__setitem__("options", opts),
|
||||
)
|
||||
effective = resolved.get("options") or options or ConvertOptions()
|
||||
except ValidationError as exc:
|
||||
raise HTTPException(status_code=exc.status_code, detail=exc.to_detail()) from exc
|
||||
except ConversionError as exc:
|
||||
raise HTTPException(status_code=exc.status_code, detail=str(exc)) from exc
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as exc:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
|
||||
detail=f"Convert failed: {type(exc).__name__}: {exc}",
|
||||
) from exc
|
||||
finally:
|
||||
if acquired:
|
||||
sem.release()
|
||||
|
||||
warn_header = ascii_header_value(" | ".join(warnings)) if warnings else ""
|
||||
headers = {
|
||||
"Content-Disposition": content_disposition_attachment(out_name),
|
||||
"X-Fidelity": fidelity.value,
|
||||
"X-Warnings": warn_header,
|
||||
"X-Convert-Source": source,
|
||||
"X-Convert-Target": target,
|
||||
# Report what the engine actually used, not the literal word "auto":
|
||||
# a caller debugging an output needs the path that ran.
|
||||
"X-Layout-Mode": effective.layout_mode.value,
|
||||
"X-Ocr-Policy": effective.ocr_policy.value,
|
||||
"X-Reconstruction": "auto" if options is None else "explicit",
|
||||
}
|
||||
if quality_score is not None:
|
||||
headers["X-Quality-Score"] = f"{quality_score:.4f}"
|
||||
return Response(content=out, media_type=media, headers=headers)
|
||||
|
||||
|
||||
@router.post("/async/convert/{source}/to/{target}", response_model=ConvertJobResponse)
|
||||
async def convert_async(
|
||||
source: str,
|
||||
target: str,
|
||||
file: UploadFile = File(...),
|
||||
layout_mode: Annotated[LayoutMode | None, Query()] = None,
|
||||
ocr_policy: Annotated[OcrPolicy | None, Query()] = None,
|
||||
ocr_engine: Annotated[OcrEngineChoice | None, Query()] = None,
|
||||
header_footer_mode: Annotated[HeaderFooterMode | None, Query()] = None,
|
||||
recognition_mode: Annotated[RecognitionMode | None, Query()] = None,
|
||||
detect_tables: Annotated[bool | None, Query()] = None,
|
||||
) -> ConvertJobResponse:
|
||||
# Acquire before reading the upload so rejected requests do not retain a
|
||||
# second in-memory copy while waiting behind a saturated executor.
|
||||
if not _async_admission.acquire(blocking=False):
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_503_SERVICE_UNAVAILABLE,
|
||||
detail=(
|
||||
"Async conversion queue is full. Please retry shortly."
|
||||
),
|
||||
headers={"Retry-After": "1"},
|
||||
)
|
||||
|
||||
submitted = False
|
||||
try:
|
||||
data = await file.read()
|
||||
filename = file.filename or f"upload.{source}"
|
||||
options = _options_from_query(
|
||||
layout_mode,
|
||||
ocr_policy,
|
||||
ocr_engine,
|
||||
header_footer_mode,
|
||||
recognition_mode,
|
||||
detect_tables,
|
||||
)
|
||||
job = job_store.create(source=source, target=target, filename=filename)
|
||||
try:
|
||||
_executor.submit(
|
||||
_run_admitted_job,
|
||||
job["job_id"],
|
||||
data,
|
||||
filename,
|
||||
source,
|
||||
target,
|
||||
options,
|
||||
)
|
||||
submitted = True
|
||||
except Exception as exc:
|
||||
job_store.update(
|
||||
job["job_id"],
|
||||
status=JobStatus.failed.value,
|
||||
error=f"Unable to queue conversion: {exc}",
|
||||
)
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_503_SERVICE_UNAVAILABLE,
|
||||
detail="Conversion queue is unavailable. Please retry shortly.",
|
||||
headers={"Retry-After": "1"},
|
||||
) from exc
|
||||
|
||||
return ConvertJobResponse(
|
||||
job_id=job["job_id"],
|
||||
status=JobStatus.queued,
|
||||
source=source,
|
||||
target=target,
|
||||
filename=filename,
|
||||
download_path=None,
|
||||
meta={"convert_options": _options_metadata(options)},
|
||||
)
|
||||
finally:
|
||||
# The worker owns the slot only after submit succeeds. Every path
|
||||
# before that point, including upload/read and DB failures, releases it.
|
||||
if not submitted:
|
||||
_async_admission.release()
|
||||
|
||||
|
||||
@router.get("/jobs/{job_id}", response_model=ConvertJobResponse)
|
||||
def get_job(job_id: str) -> ConvertJobResponse:
|
||||
job = job_store.get(job_id)
|
||||
if not job:
|
||||
raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="Job not found")
|
||||
fidelity = job.get("fidelity")
|
||||
return ConvertJobResponse(
|
||||
job_id=job["job_id"],
|
||||
status=JobStatus(job["status"]),
|
||||
source=job.get("source"),
|
||||
target=job.get("target"),
|
||||
fidelity=fidelity,
|
||||
warnings=job.get("warnings") or [],
|
||||
filename=job.get("result_filename") or job.get("filename"),
|
||||
size_bytes=job.get("size_bytes"),
|
||||
error=job.get("error"),
|
||||
download_path=job.get("download_path") if job["status"] == JobStatus.completed.value else None,
|
||||
meta=job.get("meta") or {},
|
||||
)
|
||||
|
||||
|
||||
@router.get("/jobs/{job_id}/download")
|
||||
def download_job(job_id: str) -> Response:
|
||||
job = job_store.get(job_id)
|
||||
if not job:
|
||||
raise HTTPException(status_code=404, detail="Job not found")
|
||||
if job["status"] != JobStatus.completed.value:
|
||||
raise HTTPException(status_code=409, detail=f"Job is {job['status']}")
|
||||
data = job_store.get_result_bytes(job_id)
|
||||
if not data:
|
||||
raise HTTPException(status_code=404, detail="Result unavailable")
|
||||
target = job.get("target") or "bin"
|
||||
filename = job.get("result_filename") or f"converted.{target}"
|
||||
warn_header = ascii_header_value(" | ".join(job.get("warnings") or []))
|
||||
headers = {
|
||||
"Content-Disposition": content_disposition_attachment(filename),
|
||||
"X-Fidelity": ascii_header_value(str(job.get("fidelity") or "")),
|
||||
}
|
||||
if warn_header:
|
||||
headers["X-Warnings"] = warn_header
|
||||
meta = job.get("meta") or {}
|
||||
qs = meta.get("quality_score")
|
||||
if qs is not None:
|
||||
headers["X-Quality-Score"] = f"{float(qs):.4f}"
|
||||
return Response(
|
||||
content=data,
|
||||
media_type=media_type_for(target),
|
||||
headers=headers,
|
||||
)
|
||||
|
||||
|
||||
@router.delete("/jobs/{job_id}", response_model=ConvertJobResponse)
|
||||
def delete_job(job_id: str) -> ConvertJobResponse:
|
||||
job = job_store.get(job_id)
|
||||
if not job:
|
||||
raise HTTPException(status_code=404, detail="Job not found")
|
||||
job_store.request_cancel(job_id)
|
||||
job_store.update(job_id, status=JobStatus.failed.value, error="cancelled", result_bytes=None)
|
||||
return ConvertJobResponse(
|
||||
job_id=job_id,
|
||||
status=JobStatus.failed,
|
||||
source=job.get("source"),
|
||||
target=job.get("target"),
|
||||
error="cancelled",
|
||||
)
|
||||
@@ -383,7 +383,6 @@ async def merge_documents(
|
||||
files: list[UploadFile] = File(...),
|
||||
manifest: str = Form(default="[]"),
|
||||
output_filename: str = Form(default="merged.pdf"),
|
||||
parent_document_id: str = Form(default=""),
|
||||
) -> DocumentInfoResponse:
|
||||
if not engine.is_available():
|
||||
raise HTTPException(
|
||||
@@ -417,14 +416,33 @@ async def merge_documents(
|
||||
)
|
||||
files_bytes.append(data)
|
||||
|
||||
# A manifest that could not be parsed used to fall through to "all pages of
|
||||
# every file". The merge then *succeeded*, with content the caller never
|
||||
# asked for and no indication anywhere that its page selection had been
|
||||
# discarded — the worst available failure mode. A manifest that was sent
|
||||
# and cannot be honoured is now refused.
|
||||
parsed_manifest: list[dict] = []
|
||||
if manifest and manifest.strip():
|
||||
try:
|
||||
parsed = json.loads(manifest)
|
||||
if isinstance(parsed, list):
|
||||
parsed_manifest = parsed
|
||||
except Exception:
|
||||
pass
|
||||
except json.JSONDecodeError as exc:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_400_BAD_REQUEST,
|
||||
detail=(
|
||||
f"Merge manifest is not valid JSON ({exc.msg} at position {exc.pos}). "
|
||||
"Omit the manifest to merge all pages of every file."
|
||||
),
|
||||
) from exc
|
||||
if not isinstance(parsed, list):
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_400_BAD_REQUEST,
|
||||
detail=(
|
||||
"Merge manifest must be a JSON array of "
|
||||
'{"fileIndex": int, "pages": "all" | "1,3-5"} entries; '
|
||||
f"received {type(parsed).__name__}."
|
||||
),
|
||||
)
|
||||
parsed_manifest = parsed
|
||||
|
||||
if not parsed_manifest:
|
||||
parsed_manifest = [{"fileIndex": idx, "pages": "all"} for idx in range(len(files))]
|
||||
@@ -444,12 +462,6 @@ async def merge_documents(
|
||||
pdfengine = engine.require()
|
||||
doc = pdfengine.PdfDocument.load_from_memory(merged_bytes, "")
|
||||
info = document_store.add_document(output_filename, merged_bytes, doc)
|
||||
|
||||
if parent_document_id:
|
||||
parent_doc = document_store.get_document(parent_document_id)
|
||||
if parent_doc and "remote_context" in parent_doc:
|
||||
info["remote_context"] = parent_doc["remote_context"]
|
||||
|
||||
return make_document_response(info)
|
||||
except Exception as e:
|
||||
raise HTTPException(
|
||||
|
||||
@@ -4,11 +4,12 @@ import uuid
|
||||
from fastapi import APIRouter, HTTPException, Response, status
|
||||
from pydantic import BaseModel
|
||||
|
||||
from app.services import engine
|
||||
from app.services import engine, logs
|
||||
from app.services.convert.http_headers import content_disposition_attachment
|
||||
from app.services.store import document_store
|
||||
|
||||
router = APIRouter(tags=["documents"])
|
||||
logger = logging.getLogger(__name__)
|
||||
logger = logs.get_logger(__name__)
|
||||
|
||||
|
||||
class ExportRemoteRequest(BaseModel):
|
||||
@@ -44,7 +45,7 @@ def export_document(document_id: str):
|
||||
content=bytes_data,
|
||||
media_type="application/pdf",
|
||||
headers={
|
||||
"Content-Disposition": f'attachment; filename="{filename}"',
|
||||
"Content-Disposition": content_disposition_attachment(filename),
|
||||
"Content-Length": str(len(bytes_data)),
|
||||
},
|
||||
)
|
||||
@@ -70,7 +71,15 @@ async def export_remote_document(document_id: str, body: ExportRemoteRequest | N
|
||||
|
||||
ctx = d.get("remote_context")
|
||||
if not ctx or not ctx.get("upload_url"):
|
||||
print(f"[export-remote] FAILING with 400: ctx={ctx}", flush=True)
|
||||
# The context is worth seeing when this fails, but it holds the bearer
|
||||
# token used at the upload below — logging it whole put a live
|
||||
# credential in plaintext on every misconfigured export.
|
||||
logs.warn(
|
||||
logger,
|
||||
"export_remote_missing_context",
|
||||
document_id=document_id,
|
||||
context=logs.redact_mapping(ctx),
|
||||
)
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail="Document does not have a registered remote upload context.",
|
||||
@@ -103,14 +112,26 @@ async def export_remote_document(document_id: str, body: ExportRemoteRequest | N
|
||||
data["file_id"] = str(ctx["resource_id"])
|
||||
|
||||
upload_url = ctx["upload_url"]
|
||||
print(f"[export-remote] POST {upload_url} | file_id={data.get('file_id')} | filename={filename} | size={len(bytes_data)}", flush=True)
|
||||
logs.info(
|
||||
logger,
|
||||
"export_remote_upload_start",
|
||||
document_id=document_id,
|
||||
file_id=data.get("file_id"),
|
||||
filename=filename,
|
||||
size_bytes=len(bytes_data),
|
||||
)
|
||||
|
||||
files = {"upload": (filename, bytes_data, "application/pdf")}
|
||||
|
||||
async with httpx.AsyncClient(timeout=120.0, follow_redirects=True) as client:
|
||||
res = await client.post(upload_url, headers=headers, data=data, files=files)
|
||||
|
||||
print(f"[export-remote] Response status={res.status_code} body={res.text[:500]}", flush=True)
|
||||
logs.info(
|
||||
logger,
|
||||
"export_remote_upload_done",
|
||||
document_id=document_id,
|
||||
upstream_status=res.status_code,
|
||||
)
|
||||
|
||||
if res.status_code not in (200, 201):
|
||||
logger.error("[export-remote] upstream %s: %s", res.status_code, res.text[:500])
|
||||
|
||||
@@ -54,18 +54,18 @@ def get_font_bytes(document_id: str, internal_font_id: str) -> Response:
|
||||
)
|
||||
|
||||
if not internal_font_id or len(internal_font_id) > 256:
|
||||
return Response(status_code=status.HTTP_204_NO_CONTENT)
|
||||
raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="Font not found")
|
||||
|
||||
doc_info = document_store.get_document(document_id)
|
||||
if not doc_info:
|
||||
return Response(status_code=status.HTTP_204_NO_CONTENT)
|
||||
raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="Document not found")
|
||||
|
||||
try:
|
||||
data = bytes(doc_info["doc_instance"].get_font_data(internal_font_id))
|
||||
except Exception:
|
||||
data = b""
|
||||
if not data:
|
||||
return Response(status_code=status.HTTP_204_NO_CONTENT)
|
||||
raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="Font not found")
|
||||
|
||||
magic = data[:4]
|
||||
if magic in _SFNT_TTF_MAGIC:
|
||||
@@ -73,7 +73,7 @@ def get_font_bytes(document_id: str, internal_font_id: str) -> Response:
|
||||
elif magic == _SFNT_OTF_MAGIC:
|
||||
media_type = "font/otf"
|
||||
else:
|
||||
return Response(status_code=status.HTTP_204_NO_CONTENT)
|
||||
raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="Font not found")
|
||||
|
||||
etag = '"' + hashlib.sha256(data).hexdigest()[:32] + '"'
|
||||
return Response(
|
||||
|
||||
@@ -12,6 +12,7 @@ from pydantic import BaseModel
|
||||
|
||||
from app import __version__
|
||||
from app.config import Settings, get_settings
|
||||
from app.services import engine as engine_service
|
||||
|
||||
router = APIRouter(tags=["health"])
|
||||
|
||||
@@ -23,6 +24,37 @@ class HealthResponse(BaseModel):
|
||||
version: str
|
||||
environment: str
|
||||
engine_available: bool
|
||||
# What this process can actually do. Every model path in the converter is
|
||||
# fail-open, so a deploy that forgot the weights serves happily with
|
||||
# Arabic recognition off and no page renderer. Without this an operator
|
||||
# has to read startup logs to find that out — usually after a customer does.
|
||||
capabilities: dict[str, bool] = {}
|
||||
|
||||
|
||||
def _capabilities() -> dict[str, bool]:
|
||||
"""Probe each optional capability, never raising: /health must always answer."""
|
||||
caps: dict[str, bool] = {}
|
||||
try:
|
||||
from app.services import raster
|
||||
|
||||
caps["page_raster"] = raster.is_available()
|
||||
except Exception:
|
||||
caps["page_raster"] = False
|
||||
try:
|
||||
from app.services import ocr
|
||||
|
||||
caps["ocr"] = ocr.is_ocr_available()
|
||||
caps["ocr_arabic"] = ocr.is_ocr_arabic_available()
|
||||
except Exception:
|
||||
caps["ocr"] = False
|
||||
caps["ocr_arabic"] = False
|
||||
try:
|
||||
from app.services.convert.layout import ml_regions
|
||||
|
||||
caps["layout_ml"] = ml_regions.layout_ml_enabled() and ml_regions._weights_path().is_file()
|
||||
except Exception:
|
||||
caps["layout_ml"] = False
|
||||
return caps
|
||||
|
||||
|
||||
@router.get("/health", response_model=HealthResponse)
|
||||
@@ -31,5 +63,6 @@ def health(settings: SettingsDep) -> HealthResponse:
|
||||
status="ok",
|
||||
version=__version__,
|
||||
environment=settings.environment,
|
||||
engine_available=settings.engine_available,
|
||||
engine_available=engine_service.is_available() or settings.engine_available,
|
||||
capabilities=_capabilities(),
|
||||
)
|
||||
|
||||
@@ -0,0 +1,87 @@
|
||||
"""Conversion API schemas."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from enum import Enum
|
||||
from typing import Any
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
|
||||
class Fidelity(str, Enum):
|
||||
high = "high"
|
||||
medium = "medium"
|
||||
lossy = "lossy"
|
||||
|
||||
|
||||
class JobStatus(str, Enum):
|
||||
queued = "queued"
|
||||
running = "running"
|
||||
completed = "completed"
|
||||
failed = "failed"
|
||||
|
||||
|
||||
class ConvertFormat(str, Enum):
|
||||
pdf = "pdf"
|
||||
docx = "docx"
|
||||
xlsx = "xlsx"
|
||||
pptx = "pptx"
|
||||
html = "html"
|
||||
md = "md"
|
||||
txt = "txt"
|
||||
json = "json"
|
||||
png = "png"
|
||||
jpeg = "jpeg"
|
||||
jpg = "jpg"
|
||||
tiff = "tiff"
|
||||
csv = "csv"
|
||||
xml = "xml"
|
||||
|
||||
|
||||
class ConverterInfo(BaseModel):
|
||||
source: str
|
||||
target: str
|
||||
fidelity: Fidelity
|
||||
sync_ok: bool = True
|
||||
max_bytes: int = 50 * 1024 * 1024
|
||||
description: str = ""
|
||||
gaps: list[str] = Field(default_factory=list)
|
||||
|
||||
|
||||
class ConvertInfoResponse(BaseModel):
|
||||
converters: list[ConverterInfo]
|
||||
|
||||
|
||||
class CanConvertResponse(BaseModel):
|
||||
can_convert: bool
|
||||
source: str
|
||||
target: str
|
||||
fidelity: Fidelity | None = None
|
||||
engine: str | None = None
|
||||
gaps: list[str] = Field(default_factory=list)
|
||||
|
||||
|
||||
class ConvertResultMeta(BaseModel):
|
||||
fidelity: Fidelity
|
||||
warnings: list[str] = Field(default_factory=list)
|
||||
source: str
|
||||
target: str
|
||||
filename: str
|
||||
size_bytes: int
|
||||
page_count: int | None = None
|
||||
engine: str = "oss"
|
||||
convert_options: dict[str, Any] | None = None
|
||||
|
||||
|
||||
class ConvertJobResponse(BaseModel):
|
||||
job_id: str
|
||||
status: JobStatus
|
||||
source: str | None = None
|
||||
target: str | None = None
|
||||
fidelity: Fidelity | None = None
|
||||
warnings: list[str] = Field(default_factory=list)
|
||||
filename: str | None = None
|
||||
size_bytes: int | None = None
|
||||
error: str | None = None
|
||||
download_path: str | None = None
|
||||
meta: dict[str, Any] = Field(default_factory=dict)
|
||||
@@ -0,0 +1,7 @@
|
||||
"""Conversion service package."""
|
||||
|
||||
from app.services.convert.pipeline import ConversionError, run_conversion
|
||||
from app.services.convert.registry import registry
|
||||
from app.services.convert.validate import ValidationError
|
||||
|
||||
__all__ = ["ConversionError", "ValidationError", "registry", "run_conversion"]
|
||||
@@ -0,0 +1,316 @@
|
||||
"""Automatic reconstruction policy — the engine picks the path, not the user.
|
||||
|
||||
The conversion desk asks for one thing only: the output format. Everything
|
||||
else — whether to OCR, whether to reflow or position, how to treat running
|
||||
headers, whether to look for tables — is decided here from what the document
|
||||
actually is.
|
||||
|
||||
Why route first rather than run one universal path: a clean digital contract, a
|
||||
photographed invoice and a bilingual RFP with a broken ToUnicode map need
|
||||
different treatment, and applying any one of those treatments to the other two
|
||||
makes the output worse. Routing costs a page scan; guessing costs the document.
|
||||
|
||||
Honest limits, stated the same way in the UI and the docs:
|
||||
|
||||
* Word and Excel are **reconstructions**. They stay labelled lossy.
|
||||
* PNG/JPEG/TIFF of pages are exact rasters of what was uploaded.
|
||||
* A page that cannot be reconstructed is embedded as a page image rather than
|
||||
having a layout invented for it.
|
||||
|
||||
Every decision is recorded on ``document.meta.convert_policy`` so a support
|
||||
question ("why did page 7 come out as a picture?") has an answer.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from collections.abc import Iterator
|
||||
from contextlib import contextmanager
|
||||
from contextvars import ContextVar
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
from app.services.convert.layout.pdf_router import PdfDocType, PdfRouteResult
|
||||
from app.services.convert.options import (
|
||||
ConvertOptions,
|
||||
HeaderFooterMode,
|
||||
LayoutMode,
|
||||
OcrPolicy,
|
||||
RecognitionMode,
|
||||
)
|
||||
|
||||
# Targets that are reconstructions of the document's *content*.
|
||||
OFFICE_TARGETS = frozenset({"docx", "xlsx", "csv", "pptx", "html", "md", "txt", "json"})
|
||||
# Targets that reproduce the page as pixels; nothing here is reconstructed.
|
||||
RASTER_TARGETS = frozenset({"png", "jpeg", "jpg", "tiff", "tif"})
|
||||
|
||||
# --- brochure detection -----------------------------------------------------
|
||||
# A brochure-like page is mostly artwork with captions placed around it.
|
||||
# Reflowing one produces a single mashed paragraph, so it is emitted positioned.
|
||||
BROCHURE_IMAGE_COVERAGE = 0.45
|
||||
# Below this many words a page is not carrying body prose.
|
||||
BROCHURE_MAX_WORDS = 120
|
||||
# Short positioned blocks per paragraph: a poster is many small labels, a report
|
||||
# is a few long paragraphs.
|
||||
BROCHURE_BLOCK_RATIO = 2.5
|
||||
# The share of pages that must look like this before the *document* is treated
|
||||
# as brochure-like. One art page in a report does not change the whole file.
|
||||
BROCHURE_PAGE_SHARE = 0.6
|
||||
|
||||
# A page with no body prose at all does not need table detection; looking for
|
||||
# grids in running prose is where soup tables come from.
|
||||
PROSE_ONLY_MAX_TABLE_HINT = 0.02
|
||||
|
||||
|
||||
def auto_enabled() -> bool:
|
||||
"""Whether omitted knobs should be filled in by this module."""
|
||||
raw = (os.environ.get("CONVERT_AUTO_RECONSTRUCT") or "1").strip().lower()
|
||||
return raw not in ("0", "false", "off", "no")
|
||||
|
||||
|
||||
@dataclass
|
||||
class DocumentSignals:
|
||||
"""What we know about the document before choosing a path."""
|
||||
|
||||
doc_type: PdfDocType = PdfDocType.text_based
|
||||
confidence: float = 0.0
|
||||
page_count: int = 1
|
||||
pages_needing_ocr: list[int] = field(default_factory=list)
|
||||
# Mean fraction of page area covered by images.
|
||||
image_coverage: float = 0.0
|
||||
# Pages that are one full-page raster and nothing else.
|
||||
full_page_raster_pages: list[int] = field(default_factory=list)
|
||||
# Words of usable digital text across the document.
|
||||
text_words: int = 0
|
||||
# Positioned blocks vs paragraphs, averaged per page.
|
||||
block_to_paragraph_ratio: float = 0.0
|
||||
brochure_pages: list[int] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def ocr_fraction(self) -> float:
|
||||
return len(self.pages_needing_ocr) / max(self.page_count, 1)
|
||||
|
||||
@property
|
||||
def brochure_fraction(self) -> float:
|
||||
return len(self.brochure_pages) / max(self.page_count, 1)
|
||||
|
||||
|
||||
@dataclass
|
||||
class PolicyDecision:
|
||||
"""The chosen options plus why, for logging and for X-Warnings."""
|
||||
|
||||
options: ConvertOptions
|
||||
reasons: list[str] = field(default_factory=list)
|
||||
signals: DocumentSignals = field(default_factory=DocumentSignals)
|
||||
|
||||
def as_dict(self) -> dict:
|
||||
return {
|
||||
"auto": True,
|
||||
"options": self.options.as_dict(),
|
||||
"reasons": list(self.reasons),
|
||||
"doc_type": self.signals.doc_type.value,
|
||||
"confidence": round(self.signals.confidence, 3),
|
||||
"pages_needing_ocr": len(self.signals.pages_needing_ocr),
|
||||
"page_count": self.signals.page_count,
|
||||
"image_coverage": round(self.signals.image_coverage, 3),
|
||||
"brochure_pages": len(self.signals.brochure_pages),
|
||||
}
|
||||
|
||||
def summary(self) -> str:
|
||||
"""One short line for X-Warnings — no internal jargon."""
|
||||
opts = self.options
|
||||
bits = [f"auto: {self.signals.doc_type.value}"]
|
||||
if opts.layout_mode != LayoutMode.flowing:
|
||||
bits.append(f"layout={opts.layout_mode.value}")
|
||||
if opts.ocr_policy != OcrPolicy.auto:
|
||||
bits.append(f"ocr={opts.ocr_policy.value}")
|
||||
if not opts.detect_tables:
|
||||
bits.append("tables=off")
|
||||
return "; ".join(bits)
|
||||
|
||||
|
||||
def _decide_ocr(signals: DocumentSignals, reasons: list[str]) -> OcrPolicy:
|
||||
"""OCR the pages that need it — never the whole document out of caution.
|
||||
|
||||
``auto`` already means "the rebuild step decides per page from
|
||||
pages_needing_ocr and the encoding-broken list", which is exactly the
|
||||
per-page behaviour wanted. It is escalated to ``force`` only when the
|
||||
document is so thoroughly broken that per-page detection would still leave
|
||||
corrupt text behind, and dropped to ``never`` when there is nothing to gain.
|
||||
"""
|
||||
if signals.doc_type == PdfDocType.text_based and not signals.pages_needing_ocr:
|
||||
reasons.append("clean digital text layer: no OCR")
|
||||
return OcrPolicy.never
|
||||
if signals.ocr_fraction >= 0.95 and signals.page_count > 1:
|
||||
reasons.append(f"every page needs OCR ({signals.doc_type.value})")
|
||||
return OcrPolicy.force
|
||||
if signals.pages_needing_ocr:
|
||||
reasons.append(f"OCR {len(signals.pages_needing_ocr)}/{signals.page_count} pages")
|
||||
return OcrPolicy.auto
|
||||
|
||||
|
||||
def _decide_layout(
|
||||
signals: DocumentSignals, target: str, reasons: list[str]
|
||||
) -> tuple[LayoutMode, RecognitionMode]:
|
||||
"""Flowing keeps the output editable; exact keeps a poster looking like one."""
|
||||
if target != "docx":
|
||||
# Only DOCX has a positioned emit. Everything else reflows by nature.
|
||||
return LayoutMode.flowing, RecognitionMode.enhanced_flow
|
||||
|
||||
brochure = signals.brochure_fraction >= BROCHURE_PAGE_SHARE
|
||||
image_doc = signals.doc_type == PdfDocType.image_based
|
||||
if brochure or (image_doc and signals.text_words < BROCHURE_MAX_WORDS * signals.page_count):
|
||||
why = "image-led layout" if image_doc else "brochure-like pages"
|
||||
reasons.append(f"{why}: positioned emit so the design survives")
|
||||
return LayoutMode.exact, RecognitionMode.textbox
|
||||
|
||||
return LayoutMode.flowing, RecognitionMode.enhanced_flow
|
||||
|
||||
|
||||
def _decide_tables(signals: DocumentSignals, target: str, reasons: list[str]) -> bool:
|
||||
"""Table detection off only where it can do nothing but harm."""
|
||||
if target in ("xlsx", "csv", "json"):
|
||||
# A spreadsheet target exists to carry grids.
|
||||
return True
|
||||
if signals.doc_type == PdfDocType.image_based and not signals.pages_needing_ocr:
|
||||
reasons.append("image-only document: table detection off")
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def decide(
|
||||
signals: DocumentSignals, *, target: str, base: ConvertOptions | None = None
|
||||
) -> PolicyDecision:
|
||||
"""Choose reconstruction options for this document and target."""
|
||||
target = (target or "").lower()
|
||||
reasons: list[str] = []
|
||||
|
||||
if target in RASTER_TARGETS:
|
||||
# Rasters reproduce the page; nothing below applies to them.
|
||||
options = base or ConvertOptions()
|
||||
return PolicyDecision(
|
||||
options=options,
|
||||
reasons=["raster target: page reproduced exactly, no reconstruction"],
|
||||
signals=signals,
|
||||
)
|
||||
|
||||
ocr_policy = _decide_ocr(signals, reasons)
|
||||
layout_mode, recognition_mode = _decide_layout(signals, target, reasons)
|
||||
detect_tables = _decide_tables(signals, target, reasons)
|
||||
|
||||
options = ConvertOptions(
|
||||
layout_mode=layout_mode,
|
||||
ocr_policy=ocr_policy,
|
||||
ocr_engine=(base.ocr_engine if base else ConvertOptions().ocr_engine),
|
||||
header_footer_mode=HeaderFooterMode.detect,
|
||||
recognition_mode=recognition_mode,
|
||||
detect_tables=detect_tables,
|
||||
)
|
||||
if not reasons:
|
||||
reasons.append("digital text: flowing reconstruction")
|
||||
return PolicyDecision(options=options, reasons=reasons, signals=signals)
|
||||
|
||||
|
||||
def signals_from_pdf(data: bytes, route: PdfRouteResult | None = None) -> DocumentSignals:
|
||||
"""Gather the document signals the policy needs, reusing cached work.
|
||||
|
||||
Deliberately cheap: the router result and the page text are already in the
|
||||
per-conversion cache by the time this runs, and image placement comes from
|
||||
the same content-stream walk the figure extractor uses.
|
||||
"""
|
||||
from app.services.convert import doc_cache
|
||||
from app.services.convert.layout.images import image_placements
|
||||
from app.services.convert.layout.page_classify import estimate_image_coverage
|
||||
from app.services.convert.layout.pdf_router import route_pdf
|
||||
|
||||
if route is None:
|
||||
try:
|
||||
route = route_pdf(data)
|
||||
except Exception: # a routing failure must not fail the conversion
|
||||
route = None
|
||||
|
||||
signals = DocumentSignals()
|
||||
if route is not None:
|
||||
signals.doc_type = route.doc_type
|
||||
signals.confidence = route.confidence
|
||||
signals.pages_needing_ocr = list(route.pages_needing_ocr)
|
||||
|
||||
try:
|
||||
reader = doc_cache.get_reader(data)
|
||||
except Exception:
|
||||
return signals
|
||||
|
||||
pages = reader.pages
|
||||
signals.page_count = len(pages)
|
||||
coverages: list[float] = []
|
||||
words = 0
|
||||
|
||||
for index, page in enumerate(pages):
|
||||
try:
|
||||
width = float(page.mediabox.width)
|
||||
height = float(page.mediabox.height)
|
||||
except Exception:
|
||||
width = height = 0.0
|
||||
try:
|
||||
blocks = [
|
||||
{"x": p.x, "y": p.y, "w": p.w, "h": p.h} for p in image_placements(page)
|
||||
]
|
||||
except Exception:
|
||||
blocks = []
|
||||
coverage = estimate_image_coverage(blocks, width, height) if blocks else 0.0
|
||||
coverages.append(coverage)
|
||||
if coverage >= 0.98:
|
||||
signals.full_page_raster_pages.append(index)
|
||||
|
||||
try:
|
||||
text = doc_cache.page_text(data, index)
|
||||
except Exception:
|
||||
text = ""
|
||||
page_words = len(text.split())
|
||||
words += page_words
|
||||
|
||||
# Brochure: a lot of artwork and very little running prose.
|
||||
if coverage >= BROCHURE_IMAGE_COVERAGE and page_words < BROCHURE_MAX_WORDS:
|
||||
signals.brochure_pages.append(index)
|
||||
|
||||
signals.image_coverage = sum(coverages) / max(len(coverages), 1)
|
||||
signals.text_words = words
|
||||
return signals
|
||||
|
||||
|
||||
def policy_for(data: bytes, *, target: str, base: ConvertOptions | None = None) -> PolicyDecision:
|
||||
"""Full auto decision for a PDF source."""
|
||||
return decide(signals_from_pdf(data), target=target, base=base)
|
||||
|
||||
|
||||
# The decision travels with the conversion so the formatter layer can record it
|
||||
# on the document without every plugin signature growing a parameter. A context
|
||||
# variable, not a module global: concurrent jobs must not see each other's.
|
||||
_active: ContextVar[PolicyDecision | None] = ContextVar("convert_auto_policy", default=None)
|
||||
|
||||
|
||||
def set_active_decision(decision: PolicyDecision | None) -> None:
|
||||
_active.set(decision)
|
||||
|
||||
|
||||
def get_active_decision() -> PolicyDecision | None:
|
||||
return _active.get()
|
||||
|
||||
|
||||
@contextmanager
|
||||
def decision_scope(decision: PolicyDecision | None = None) -> Iterator[None]:
|
||||
"""Bind the routing decision to one conversion and nothing else.
|
||||
|
||||
A ContextVar ``set`` persists in whatever context performed it. Conversions
|
||||
run on pooled worker threads whose context outlives the job, so a decision
|
||||
left behind by an automatic conversion was still readable by the next one —
|
||||
including an explicit conversion that did no routing at all, which would
|
||||
then report someone else's ``convert_policy`` as its own.
|
||||
|
||||
Entering resets the decision to *nothing*, so a job can only ever see a
|
||||
decision its own run made; leaving restores what was there before.
|
||||
"""
|
||||
token = _active.set(decision)
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
_active.reset(token)
|
||||
@@ -0,0 +1,5 @@
|
||||
"""Conversion backends (reportlab Office→PDF exporter)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
__all__ = ["pdf_exporter"]
|
||||
@@ -0,0 +1,95 @@
|
||||
"""Office → PDF via reportlab only (concurrent-safe, no LibreOffice).
|
||||
|
||||
LibreOffice was removed: heavy, poor multi-user concurrency, and out of
|
||||
primary stack policy. Improve fidelity in pdf_from_docx instead.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from typing import Callable
|
||||
import httpx
|
||||
|
||||
from app.services.convert.validation import convert_timeout_seconds, run_with_timeout
|
||||
from app.services.convert.writers import pdf_from_docx as reportlab_pdf
|
||||
|
||||
|
||||
def _gotenberg_url() -> str:
|
||||
return os.environ.get("GOTENBERG_URL", "http://127.0.0.1:3000").rstrip("/")
|
||||
|
||||
|
||||
def _try_gotenberg_convert(data: bytes, source_ext: str, timeout: float = 30.0) -> bytes | None:
|
||||
"""Attempt high-fidelity Office->PDF conversion via Gotenberg LibreOffice endpoint."""
|
||||
url = f"{_gotenberg_url()}/forms/libreoffice/convert"
|
||||
ext = source_ext.lstrip(".").lower()
|
||||
filename = f"document.{ext}"
|
||||
|
||||
mime_types = {
|
||||
"docx": "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
||||
"doc": "application/msword",
|
||||
"xlsx": "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
|
||||
"xls": "application/vnd.ms-excel",
|
||||
"pptx": "application/vnd.openxmlformats-officedocument.presentationml.presentation",
|
||||
"ppt": "application/vnd.ms-powerpoint",
|
||||
}
|
||||
content_type = mime_types.get(ext, "application/octet-stream")
|
||||
|
||||
try:
|
||||
with httpx.Client(timeout=timeout) as client:
|
||||
files = {"files": (filename, data, content_type)}
|
||||
response = client.post(url, files=files)
|
||||
if response.status_code == 200 and len(response.content) >= 10 and response.content.startswith(b"%PDF"):
|
||||
return response.content
|
||||
except Exception:
|
||||
# Fail-open contract: Gotenberg absence or network glitch must never crash conversion
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
def export_office_to_pdf(
|
||||
data: bytes,
|
||||
*,
|
||||
source_ext: str,
|
||||
reportlab_fn: Callable[[bytes], bytes] | None = None,
|
||||
timeout: float | None = None,
|
||||
) -> tuple[bytes, list[str]]:
|
||||
"""
|
||||
Convert DOCX/XLSX/PPTX bytes to PDF.
|
||||
First attempts high-fidelity Gotenberg if available; seamlessly falls back
|
||||
to in-process ReportLab.
|
||||
Enforces CONVERT_TIMEOUT_SECONDS (or explicit timeout).
|
||||
"""
|
||||
limit = timeout if timeout is not None else convert_timeout_seconds(120.0)
|
||||
ext = source_ext.lstrip(".").lower()
|
||||
|
||||
# 1. Try Gotenberg for maximum Office layout fidelity
|
||||
gotenberg_pdf = _try_gotenberg_convert(data, ext, timeout=min(limit, 30.0))
|
||||
if gotenberg_pdf is not None:
|
||||
return gotenberg_pdf, [
|
||||
"PDF converted via Gotenberg (high-fidelity LibreOffice microservice).",
|
||||
]
|
||||
|
||||
# 2. Seamless fallback to in-process ReportLab
|
||||
warnings = [
|
||||
"PDF via reportlab (in-process; concurrent-safe). Complex Word/Excel layout may differ.",
|
||||
]
|
||||
|
||||
def _run() -> tuple[bytes, list[str]]:
|
||||
if reportlab_fn is not None:
|
||||
return reportlab_fn(data), []
|
||||
if ext in ("docx", "doc"):
|
||||
return reportlab_pdf.docx_to_pdf(data), []
|
||||
if ext in ("xlsx", "xls"):
|
||||
out, extra = reportlab_pdf.xlsx_to_pdf(data)
|
||||
return out, list(extra)
|
||||
if ext in ("pptx", "ppt"):
|
||||
from app.services.convert.writers.pdf_from_pptx import pptx_to_pdf
|
||||
out, extra = pptx_to_pdf(data)
|
||||
return out, list(extra)
|
||||
raise ValueError(f"No PDF export path for .{ext}")
|
||||
|
||||
try:
|
||||
out, extra = run_with_timeout(_run, limit, label="Office→PDF reportlab")
|
||||
except TimeoutError:
|
||||
raise
|
||||
return out, warnings + list(extra)
|
||||
@@ -0,0 +1,178 @@
|
||||
"""Per-conversion cancellation and deadlines.
|
||||
|
||||
Two mechanisms, both scoped to the conversion that owns them:
|
||||
|
||||
``CancelToken``
|
||||
A caller-supplied "should I stop?" predicate, used to honour a job DELETE.
|
||||
|
||||
``Deadline``
|
||||
A wall-clock budget the pipeline checks at page boundaries, so a runaway
|
||||
document stops doing work rather than merely having its result discarded.
|
||||
|
||||
Both travel in :mod:`contextvars` rather than module globals. A module global
|
||||
is shared by every thread in the process, so with concurrent conversions one
|
||||
job would overwrite another's callback — and cancelling one job would abort an
|
||||
unrelated one. Context variables give each conversion its own value whether it
|
||||
runs in a thread, a task, or nested inside a worker.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
from collections.abc import Callable, Iterator
|
||||
from contextlib import contextmanager
|
||||
from contextvars import ContextVar
|
||||
from dataclasses import dataclass
|
||||
|
||||
from app.services import logs
|
||||
|
||||
logger = logs.get_logger(__name__)
|
||||
|
||||
|
||||
class ConversionCancelled(RuntimeError):
|
||||
"""Raised when the caller cancelled the conversion."""
|
||||
|
||||
|
||||
class ConversionDeadlineExceeded(TimeoutError):
|
||||
"""Raised when a conversion exhausted its wall-clock budget."""
|
||||
|
||||
|
||||
CancelToken = Callable[[], bool]
|
||||
|
||||
_cancel_check: ContextVar[CancelToken | None] = ContextVar("convert_cancel_check", default=None)
|
||||
_deadline: ContextVar[Deadline | None] = ContextVar("convert_deadline", default=None)
|
||||
# Whether this conversion has already reported a broken cancel token. The check
|
||||
# runs at every page boundary, so an unguarded log would produce one line per
|
||||
# page for a single underlying fault.
|
||||
_cancel_fault_reported: ContextVar[bool] = ContextVar(
|
||||
"convert_cancel_fault_reported", default=False
|
||||
)
|
||||
|
||||
|
||||
def _ask_cancel_token(fn: CancelToken) -> bool:
|
||||
"""Ask the caller's token whether to stop, surviving a token that raises.
|
||||
|
||||
Treating a raising token as "not cancelled" is the only safe answer — the
|
||||
alternative is aborting work nobody asked to abort — but doing it silently
|
||||
meant a broken token made every job **quietly uncancellable**, with no
|
||||
trace anywhere. The answer is unchanged; the silence is not.
|
||||
"""
|
||||
try:
|
||||
return bool(fn())
|
||||
except Exception as exc:
|
||||
if not _cancel_fault_reported.get():
|
||||
_cancel_fault_reported.set(True)
|
||||
logs.error(
|
||||
logger,
|
||||
"cancel_token_failed",
|
||||
error=type(exc).__name__,
|
||||
detail=str(exc)[:200],
|
||||
consequence="job cannot be cancelled",
|
||||
exc_info=True,
|
||||
)
|
||||
return False
|
||||
|
||||
|
||||
@dataclass
|
||||
class Deadline:
|
||||
"""A wall-clock budget for one conversion."""
|
||||
|
||||
seconds: float
|
||||
started: float
|
||||
|
||||
@classmethod
|
||||
def start(cls, seconds: float) -> Deadline:
|
||||
return cls(seconds=seconds, started=time.monotonic())
|
||||
|
||||
@property
|
||||
def elapsed(self) -> float:
|
||||
return time.monotonic() - self.started
|
||||
|
||||
@property
|
||||
def remaining(self) -> float:
|
||||
return self.seconds - self.elapsed
|
||||
|
||||
@property
|
||||
def expired(self) -> bool:
|
||||
return self.seconds > 0 and self.remaining <= 0
|
||||
|
||||
|
||||
@contextmanager
|
||||
def conversion_scope(
|
||||
*, cancel_check: CancelToken | None = None, timeout: float | None = None
|
||||
) -> Iterator[None]:
|
||||
"""Bind a cancel token and/or deadline for the duration of one conversion."""
|
||||
cancel_reset = _cancel_check.set(cancel_check)
|
||||
deadline_reset = _deadline.set(Deadline.start(timeout) if timeout and timeout > 0 else None)
|
||||
fault_reset = _cancel_fault_reported.set(False)
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
_cancel_check.reset(cancel_reset)
|
||||
_deadline.reset(deadline_reset)
|
||||
_cancel_fault_reported.reset(fault_reset)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def deadline_scope(seconds: float | None) -> Iterator[None]:
|
||||
"""Bind only a wall-clock budget, leaving any cancel token in place."""
|
||||
reset = _deadline.set(Deadline.start(seconds) if seconds and seconds > 0 else None)
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
_deadline.reset(reset)
|
||||
|
||||
|
||||
def set_cancel_check(fn: CancelToken | None) -> None:
|
||||
"""Bind a cancel token for the current context.
|
||||
|
||||
Prefer :func:`conversion_scope`; this exists for callers and tests that
|
||||
manage the lifetime themselves.
|
||||
"""
|
||||
_cancel_check.set(fn)
|
||||
|
||||
|
||||
def get_cancel_check() -> CancelToken | None:
|
||||
return _cancel_check.get()
|
||||
|
||||
|
||||
def get_deadline() -> Deadline | None:
|
||||
return _deadline.get()
|
||||
|
||||
|
||||
def remaining_seconds(default: float) -> float:
|
||||
"""Budget left for a sub-step, never more than the conversion's own budget."""
|
||||
deadline = _deadline.get()
|
||||
if deadline is None or deadline.seconds <= 0:
|
||||
return default
|
||||
return max(0.0, min(default, deadline.remaining))
|
||||
|
||||
|
||||
def check_cancelled(stage: str = "conversion") -> None:
|
||||
"""Raise only if the caller cancelled — never on an expired deadline.
|
||||
|
||||
For loops that handle an exhausted budget themselves by stopping early and
|
||||
returning what they have. Using :func:`check` there defeats the point: a
|
||||
single page overrunning the budget raises out of the loop and discards
|
||||
every page already recovered, which is exactly the failure the budget
|
||||
logic exists to prevent.
|
||||
"""
|
||||
fn = _cancel_check.get()
|
||||
if fn is None:
|
||||
return
|
||||
if _ask_cancel_token(fn):
|
||||
raise ConversionCancelled(f"{stage} cancelled.")
|
||||
|
||||
|
||||
def check(stage: str = "conversion") -> None:
|
||||
"""Raise if the conversion was cancelled or has run out of time.
|
||||
|
||||
Call at safe points — page boundaries, between OCR pages — so a runaway
|
||||
document stops working instead of merely having its result thrown away.
|
||||
"""
|
||||
fn = _cancel_check.get()
|
||||
if fn is not None and _ask_cancel_token(fn):
|
||||
raise ConversionCancelled("Job cancelled.")
|
||||
deadline = _deadline.get()
|
||||
if deadline is not None and deadline.expired:
|
||||
raise ConversionDeadlineExceeded(f"{stage} exceeded its {deadline.seconds:.0f}s budget")
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,293 @@
|
||||
"""Per-conversion cache for expensive PDF reads.
|
||||
|
||||
``PdfReader.extract_text`` is the dominant cost of a text-heavy conversion, and
|
||||
three independent stages need the same text: the document router deciding which
|
||||
pages need OCR, the layout pipeline building the IDM, and the quality scorer
|
||||
establishing a source baseline. Each stage opened its own reader, so a 14-page
|
||||
document ran extraction 42 times.
|
||||
|
||||
The cache is bound to a conversion scope, so entries are released when the
|
||||
conversion ends and two concurrent conversions never share state. Outside a
|
||||
scope every call falls through to a direct read, which keeps callers total and
|
||||
makes the cache a pure optimisation.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import io
|
||||
from collections import OrderedDict
|
||||
from collections.abc import Iterator
|
||||
from contextlib import contextmanager
|
||||
from contextvars import ContextVar
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any
|
||||
|
||||
from pypdf import PdfReader
|
||||
|
||||
# Hashing the whole payload is cheap next to a single page extraction
|
||||
# (~1ms/MB against ~130ms/page), and keying on content means a converter that
|
||||
# handles more than one document inside a scope stays correct.
|
||||
_DIGEST_BYTES = 8
|
||||
|
||||
|
||||
def _key(data: bytes) -> str:
|
||||
return hashlib.blake2b(data, digest_size=_DIGEST_BYTES).hexdigest()
|
||||
|
||||
|
||||
@dataclass
|
||||
class DocumentCache:
|
||||
"""Readers and extracted page text for the documents seen in one scope."""
|
||||
|
||||
readers: dict[str, PdfReader] = field(default_factory=dict)
|
||||
# Bounded like the other two, but by bytes rather than by page count — see
|
||||
# MAX_TEXT_CACHE_BYTES below for the measurement that settled it. Evicting
|
||||
# is always safe: a later ask recomputes.
|
||||
page_text: OrderedDict[tuple[str, int], str] = field(default_factory=OrderedDict)
|
||||
# Glyph boxes and vector path operators recovered from the content stream
|
||||
# when the C++ engine is absent. Tokenising a content stream and resolving
|
||||
# its fonts costs about as much as extracting the page's text, and the
|
||||
# layout pipeline, table detector and scorer all ask for the same answer.
|
||||
geometry: OrderedDict[tuple[str, int], tuple[list, list]] = field(default_factory=OrderedDict)
|
||||
# Rendered page images keyed by (document, page, dpi). A page is rasterised
|
||||
# once and reused by layout detection, OCR and image export.
|
||||
rasters: OrderedDict[tuple[str, int, int], bytes] = field(default_factory=OrderedDict)
|
||||
# The engine document handle, opened at most once per document.
|
||||
engines: dict[str, object] = field(default_factory=dict)
|
||||
hits: int = 0
|
||||
misses: int = 0
|
||||
raster_hits: int = 0
|
||||
raster_misses: int = 0
|
||||
raster_bytes: int = 0
|
||||
# Extraction failures, counted rather than cached. A non-zero value here is
|
||||
# the difference between "this document has blank pages" and "we could not
|
||||
# read this document", which the output alone cannot tell you.
|
||||
text_failures: int = 0
|
||||
geometry_failures: int = 0
|
||||
last_text_error: str = ""
|
||||
last_geometry_error: str = ""
|
||||
text_bytes: int = 0
|
||||
|
||||
def reader_for(self, data: bytes) -> PdfReader:
|
||||
key = _key(data)
|
||||
reader = self.readers.get(key)
|
||||
if reader is None:
|
||||
reader = PdfReader(io.BytesIO(data), strict=False)
|
||||
self.readers[key] = reader
|
||||
return reader
|
||||
|
||||
def text_for(self, data: bytes, index: int) -> str:
|
||||
"""One page's text, extracted at most once — unless extraction failed.
|
||||
|
||||
A failure used to be memoised as ``""``, which is the same value a
|
||||
genuinely blank page produces. Everything downstream — the OCR trigger,
|
||||
the table detector, the scorer — then read that empty string, agreed
|
||||
with each other, and produced a confidently blank page. A failure is
|
||||
now counted and *not* cached, so a later ask can succeed and the
|
||||
failure is visible in the counters instead of in the output.
|
||||
"""
|
||||
key = (_key(data), index)
|
||||
cached = self.page_text.get(key)
|
||||
if cached is not None:
|
||||
self.hits += 1
|
||||
self.page_text.move_to_end(key)
|
||||
return cached
|
||||
self.misses += 1
|
||||
reader = self.reader_for(data)
|
||||
try:
|
||||
text = reader.pages[index].extract_text() or ""
|
||||
except Exception as exc:
|
||||
self.text_failures += 1
|
||||
self.last_text_error = f"{type(exc).__name__}: {exc}"
|
||||
return ""
|
||||
self.page_text[key] = text
|
||||
self.text_bytes += len(text)
|
||||
while self.page_text and self.text_bytes > MAX_TEXT_CACHE_BYTES:
|
||||
_evicted, payload = self.page_text.popitem(last=False)
|
||||
self.text_bytes -= len(payload)
|
||||
return text
|
||||
|
||||
def geometry_for(self, data: bytes, index: int, extract) -> tuple[list, list]:
|
||||
"""``(glyphs, path_ops)`` for one page, computed at most once.
|
||||
|
||||
Bounded to the most recent pages. Every consumer of a page's glyphs —
|
||||
the layout pipeline, the table detectors, the scorer — asks for them
|
||||
while that page is being built and never again, so holding all 126
|
||||
pages' worth to the end of the conversion buys nothing and costs the
|
||||
peak. Evicting is always safe: a later ask recomputes.
|
||||
"""
|
||||
key = (_key(data), index)
|
||||
cached = self.geometry.get(key)
|
||||
if cached is not None:
|
||||
self.hits += 1
|
||||
# Refresh recency so a page still being worked on is not evicted.
|
||||
self.geometry.move_to_end(key)
|
||||
return cached
|
||||
self.misses += 1
|
||||
try:
|
||||
result = extract()
|
||||
except Exception as exc:
|
||||
# Not cached, for the same reason as ``text_for``: an empty glyph
|
||||
# list is indistinguishable from a page that genuinely has no
|
||||
# glyphs, and memoising the failure makes one transient error the
|
||||
# permanent truth about that page for the rest of the conversion.
|
||||
self.geometry_failures += 1
|
||||
self.last_geometry_error = f"{type(exc).__name__}: {exc}"
|
||||
return ([], [])
|
||||
self.geometry[key] = result
|
||||
while len(self.geometry) > MAX_CACHED_GEOMETRY_PAGES:
|
||||
self.geometry.popitem(last=False)
|
||||
return result
|
||||
|
||||
def raster_for(self, data: bytes, index: int, dpi: int, render) -> bytes:
|
||||
"""Return a rendered page, computing it at most once per (page, dpi).
|
||||
|
||||
Rasters are the largest thing here by an order of magnitude, and a
|
||||
page's raster is wanted by at most three consecutive stages — layout
|
||||
detection, OCR, image export. Once the cache is full the oldest page
|
||||
is dropped rather than the new one refused, so the pages being worked
|
||||
on now are the ones held.
|
||||
"""
|
||||
key = (_key(data), index, int(dpi))
|
||||
cached = self.rasters.get(key)
|
||||
if cached is not None:
|
||||
self.raster_hits += 1
|
||||
self.rasters.move_to_end(key)
|
||||
return cached
|
||||
self.raster_misses += 1
|
||||
png = render()
|
||||
if len(png) <= MAX_RASTER_CACHE_BYTES:
|
||||
self.rasters[key] = png
|
||||
self.raster_bytes += len(png)
|
||||
self.rasters.move_to_end(key)
|
||||
while self.raster_bytes > MAX_RASTER_CACHE_BYTES and len(self.rasters) > 1:
|
||||
_evicted, payload = self.rasters.popitem(last=False)
|
||||
self.raster_bytes -= len(payload)
|
||||
return png
|
||||
|
||||
def engine_for(self, data: bytes, open_document) -> object | None:
|
||||
"""Open the engine document once per conversion."""
|
||||
key = _key(data)
|
||||
if key in self.engines:
|
||||
return self.engines[key]
|
||||
doc = open_document()
|
||||
self.engines[key] = doc
|
||||
return doc
|
||||
|
||||
|
||||
# Page rasters at 200 dpi run 1-3 MB each. A page's raster is wanted by at
|
||||
# most three consecutive stages, so a cache large enough to hold a hundred of
|
||||
# them is a hundred pages of peak bought for nothing. 64 MB covers the working
|
||||
# set with room to spare and takes a third of a gigabyte off a long scan.
|
||||
MAX_RASTER_CACHE_BYTES = 64 * 1024 * 1024
|
||||
# Glyph lists are small next to a raster but not free: a dense page carries a
|
||||
# few thousand dicts, and a 126-page booklet held every one of them to the end
|
||||
# of the conversion. Twelve pages is far more than any stage looks back.
|
||||
MAX_CACHED_GEOMETRY_PAGES = 12
|
||||
# Page text is bounded by *bytes*, not by page count, and the difference is
|
||||
# measurable. A 64-page ceiling looked generous and was not: header/footer
|
||||
# detection samples the ends of a document, the scorer walks all of it, and the
|
||||
# layout pipeline walks it again, so on a 126-page booklet a page-count window
|
||||
# thrashed and re-extracted. Measured on the IRS 1040 instructions (126 pages,
|
||||
# pdf→docx): 212 s with a 64-page window against 161 s with none — a 32%
|
||||
# regression bought for nothing.
|
||||
#
|
||||
# A byte budget gets the memory guarantee without the thrash. Page text runs
|
||||
# tens of kilobytes even when dense, so 16 MB holds every page of any document
|
||||
# inside the 200-page input limit, while still refusing to grow without bound
|
||||
# if that limit is ever raised.
|
||||
MAX_TEXT_CACHE_BYTES = 16 * 1024 * 1024
|
||||
|
||||
_cache: ContextVar[DocumentCache | None] = ContextVar("convert_doc_cache", default=None)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def document_cache_scope() -> Iterator[DocumentCache]:
|
||||
"""Cache PDF reads for the duration of one conversion."""
|
||||
cache = DocumentCache()
|
||||
token = _cache.set(cache)
|
||||
try:
|
||||
yield cache
|
||||
finally:
|
||||
_cache.reset(token)
|
||||
|
||||
|
||||
def current() -> DocumentCache | None:
|
||||
return _cache.get()
|
||||
|
||||
|
||||
def get_reader(data: bytes) -> PdfReader:
|
||||
"""A ``PdfReader`` for *data*, reused within the current conversion."""
|
||||
cache = _cache.get()
|
||||
if cache is None:
|
||||
return PdfReader(io.BytesIO(data), strict=False)
|
||||
return cache.reader_for(data)
|
||||
|
||||
|
||||
def page_text(data: bytes, index: int) -> str:
|
||||
"""Extracted text for one page, computed at most once per conversion."""
|
||||
cache = _cache.get()
|
||||
if cache is None:
|
||||
try:
|
||||
reader = PdfReader(io.BytesIO(data), strict=False)
|
||||
return reader.pages[index].extract_text() or ""
|
||||
except Exception:
|
||||
return ""
|
||||
return cache.text_for(data, index)
|
||||
|
||||
|
||||
def page_geometry(data: bytes, index: int, extract) -> tuple[list, list]:
|
||||
"""Glyph boxes and path operators for one page, reused within a conversion."""
|
||||
cache = _cache.get()
|
||||
if cache is None:
|
||||
try:
|
||||
return extract()
|
||||
except Exception:
|
||||
return ([], [])
|
||||
return cache.geometry_for(data, index, extract)
|
||||
|
||||
|
||||
def page_raster(data: bytes, index: int, dpi: int, render) -> bytes:
|
||||
"""Rendered page image, reused across layout detection, OCR and export."""
|
||||
cache = _cache.get()
|
||||
if cache is None:
|
||||
return render()
|
||||
return cache.raster_for(data, index, dpi, render)
|
||||
|
||||
|
||||
def engine_document(data: bytes, open_document):
|
||||
"""Engine handle for *data*, opened at most once per conversion."""
|
||||
cache = _cache.get()
|
||||
if cache is None:
|
||||
return open_document()
|
||||
return cache.engine_for(data, open_document)
|
||||
|
||||
|
||||
def page_count(data: bytes) -> int:
|
||||
try:
|
||||
return len(get_reader(data).pages)
|
||||
except Exception:
|
||||
return 0
|
||||
|
||||
|
||||
def all_text(data: bytes) -> str:
|
||||
"""Whole-document plain text, reusing any per-page text already extracted."""
|
||||
count = page_count(data)
|
||||
return "\n".join(page_text(data, i) for i in range(count))
|
||||
|
||||
|
||||
def stats() -> dict[str, Any]:
|
||||
cache = _cache.get()
|
||||
if cache is None:
|
||||
return {}
|
||||
return {
|
||||
"documents": len(cache.readers),
|
||||
"pages_cached": len(cache.page_text),
|
||||
"geometry_cached": len(cache.geometry),
|
||||
"hits": cache.hits,
|
||||
"misses": cache.misses,
|
||||
"rasters_cached": len(cache.rasters),
|
||||
"raster_hits": cache.raster_hits,
|
||||
"raster_misses": cache.raster_misses,
|
||||
"raster_mb": round(cache.raster_bytes / 1024 / 1024, 1),
|
||||
}
|
||||
@@ -0,0 +1,56 @@
|
||||
"""Typed convert / validation error codes (Firecrawl anydoc-inspired taxonomy).
|
||||
|
||||
The codes exist so a caller can act on a failure instead of parsing prose. The
|
||||
first seven describe a bad *input* and were the only ones the engine had, which
|
||||
meant every failure after validation — engine bug, blown deadline, cancellation,
|
||||
an output that would not open — arrived as an untyped 500 with a sentence in it.
|
||||
The last four close that gap, so every error the engine can produce has a name.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from enum import Enum
|
||||
|
||||
|
||||
class ConvertErrorCode(str, Enum):
|
||||
# The input is unusable, and the caller can tell why.
|
||||
unsupported = "unsupported"
|
||||
encrypted = "encrypted"
|
||||
needs_ocr = "needs_ocr"
|
||||
malformed = "malformed"
|
||||
resource_limit = "resource_limit"
|
||||
missing_part = "missing_part"
|
||||
io = "io"
|
||||
# The conversion itself did not complete. The distinction matters to a
|
||||
# caller deciding whether to retry: `timeout` and `internal` may succeed on
|
||||
# a second attempt, `cancelled` was asked for, and `invalid_output` means
|
||||
# the engine produced something it will not stand behind.
|
||||
timeout = "timeout"
|
||||
cancelled = "cancelled"
|
||||
invalid_output = "invalid_output"
|
||||
internal = "internal"
|
||||
|
||||
|
||||
# Default HTTP status per code
|
||||
HTTP_STATUS: dict[ConvertErrorCode, int] = {
|
||||
ConvertErrorCode.unsupported: 415,
|
||||
ConvertErrorCode.encrypted: 400,
|
||||
ConvertErrorCode.needs_ocr: 422,
|
||||
ConvertErrorCode.malformed: 400,
|
||||
ConvertErrorCode.resource_limit: 413,
|
||||
ConvertErrorCode.missing_part: 400,
|
||||
ConvertErrorCode.io: 400,
|
||||
ConvertErrorCode.timeout: 504,
|
||||
ConvertErrorCode.cancelled: 409,
|
||||
ConvertErrorCode.invalid_output: 500,
|
||||
ConvertErrorCode.internal: 500,
|
||||
}
|
||||
|
||||
# Codes whose cause is transient: the same request may well succeed on a retry.
|
||||
# Published so a client does not have to hardcode the list.
|
||||
RETRYABLE_CODES = frozenset(
|
||||
{
|
||||
ConvertErrorCode.timeout,
|
||||
ConvertErrorCode.internal,
|
||||
}
|
||||
)
|
||||
@@ -0,0 +1 @@
|
||||
"""Formatters package."""
|
||||
@@ -0,0 +1,178 @@
|
||||
"""Positioned DOCX emit — Wave-2 ``layout_mode=exact`` / ``recognition_mode=textbox``.
|
||||
|
||||
Flowing reconstruction is right for a report: the text reflows, the reader can
|
||||
edit it, and the result is a document rather than a picture of one. It is wrong
|
||||
for a poster, a certificate or a brochure page, where the arrangement *is* the
|
||||
content. Reflowing one of those concatenates every caption into a single
|
||||
paragraph and destroys the thing the reader wanted.
|
||||
|
||||
This module emits each block into a floating text frame anchored at the
|
||||
block's own rectangle, in the page's own coordinate system. Word treats the
|
||||
result as a normal document — every frame stays selectable and editable — but
|
||||
the arrangement survives.
|
||||
|
||||
Chosen by :mod:`auto_policy`, never by a customer dropdown. Word output remains
|
||||
a reconstruction and stays labelled lossy: frame positions come from measured
|
||||
geometry, not from the PDF's own graphics state.
|
||||
|
||||
Blocks without usable geometry fall back to a flowing paragraph individually,
|
||||
with a warning naming the page, rather than dragging the whole page back to
|
||||
flowing or inventing a position for them.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from docx.enum.text import WD_ALIGN_PARAGRAPH
|
||||
from docx.oxml import OxmlElement
|
||||
from docx.oxml.ns import qn
|
||||
|
||||
from app.services.convert.idm.model import Block, BlockType, Page
|
||||
from app.services.convert.text.arabic_logical import is_rtl_dominant
|
||||
|
||||
# PDF user space is 72 points per inch; OOXML frames are in twips (1/20 pt).
|
||||
TWIPS_PER_POINT = 20
|
||||
# A frame narrower or shorter than this is measurement noise, not a text box.
|
||||
MIN_FRAME_POINTS = 6.0
|
||||
# Frames are given a little slack so a glyph that overhangs its measured box
|
||||
# does not wrap to a second line and push the layout apart.
|
||||
FRAME_PADDING_POINTS = 4.0
|
||||
|
||||
|
||||
def _twips(points: float) -> int:
|
||||
return round(points * TWIPS_PER_POINT)
|
||||
|
||||
|
||||
def has_usable_geometry(block: Block) -> bool:
|
||||
"""Whether this block can be positioned at all."""
|
||||
bbox = getattr(block, "bbox", None)
|
||||
if bbox is None:
|
||||
return False
|
||||
if getattr(block, "synthetic_geometry", False):
|
||||
# Fabricated from plain text: the numbers look real but describe
|
||||
# nothing on the page. Positioning by them would be inventing a layout.
|
||||
return False
|
||||
return bbox.w > MIN_FRAME_POINTS and bbox.h > MIN_FRAME_POINTS
|
||||
|
||||
|
||||
def _frame_properties(block: Block, page: Page) -> OxmlElement:
|
||||
"""``w:framePr`` placing a paragraph at the block's rectangle.
|
||||
|
||||
Word anchors frames to the page with y measured downwards from the top,
|
||||
while PDF measures upwards from the bottom, so the vertical coordinate is
|
||||
flipped here. Getting this wrong mirrors the page — the first thing to
|
||||
check if a positioned document comes out upside down.
|
||||
"""
|
||||
bbox = block.bbox
|
||||
page_height = float(page.height or 792.0)
|
||||
|
||||
x = max(0.0, float(bbox.x))
|
||||
top = max(0.0, page_height - (float(bbox.y) + float(bbox.h)))
|
||||
width = max(MIN_FRAME_POINTS, float(bbox.w) + FRAME_PADDING_POINTS)
|
||||
height = max(MIN_FRAME_POINTS, float(bbox.h) + FRAME_PADDING_POINTS)
|
||||
|
||||
frame = OxmlElement("w:framePr")
|
||||
frame.set(qn("w:w"), str(_twips(width)))
|
||||
frame.set(qn("w:h"), str(_twips(height)))
|
||||
frame.set(qn("w:hRule"), "atLeast")
|
||||
frame.set(qn("w:x"), str(_twips(x)))
|
||||
frame.set(qn("w:y"), str(_twips(top)))
|
||||
frame.set(qn("w:hAnchor"), "page")
|
||||
frame.set(qn("w:vAnchor"), "page")
|
||||
# Frames must not push each other around; each one owns its rectangle.
|
||||
frame.set(qn("w:wrap"), "none")
|
||||
return frame
|
||||
|
||||
|
||||
def position_paragraph(paragraph, block: Block, page: Page) -> None:
|
||||
"""Turn an ordinary paragraph into a frame at the block's rectangle."""
|
||||
pPr = paragraph._p.get_or_add_pPr()
|
||||
pPr.insert(0, _frame_properties(block, page))
|
||||
# Inside a frame the paragraph must not inherit body spacing, or the text
|
||||
# sits below the rectangle it was placed in.
|
||||
spacing = OxmlElement("w:spacing")
|
||||
spacing.set(qn("w:before"), "0")
|
||||
spacing.set(qn("w:after"), "0")
|
||||
pPr.append(spacing)
|
||||
if is_rtl_dominant(block.plain_text()):
|
||||
bidi = OxmlElement("w:bidi")
|
||||
bidi.set(qn("w:val"), "1")
|
||||
pPr.append(bidi)
|
||||
paragraph.alignment = WD_ALIGN_PARAGRAPH.RIGHT
|
||||
|
||||
|
||||
def position_figure(paragraph, block: Block, page: Page) -> None:
|
||||
"""Anchor a picture's paragraph at the block's own rectangle.
|
||||
|
||||
Before this, positioned pages put captions in frames and left the pictures
|
||||
flowing down the page underneath them — the arrangement half-applied, which
|
||||
reads worse than either approach on its own. A brochure's picture belongs
|
||||
where the brochure put it.
|
||||
"""
|
||||
position_paragraph(paragraph, block, page)
|
||||
|
||||
|
||||
def position_table(table, block: Block, page: Page) -> bool:
|
||||
"""Float a table at the block's rectangle. False if it has no usable geometry.
|
||||
|
||||
Word floats tables with ``w:tblpPr`` rather than ``w:framePr``. A table that
|
||||
cannot be placed is left flowing rather than dropped: a grid in the wrong
|
||||
position is still the customer's data, an absent grid is not.
|
||||
"""
|
||||
if not has_usable_geometry(block):
|
||||
return False
|
||||
bbox = block.bbox
|
||||
page_height = float(page.height or 792.0)
|
||||
top = max(0.0, page_height - (float(bbox.y) + float(bbox.h)))
|
||||
|
||||
tblPr = table._tbl.tblPr
|
||||
pos = OxmlElement("w:tblpPr")
|
||||
pos.set(qn("w:leftFromText"), "0")
|
||||
pos.set(qn("w:rightFromText"), "0")
|
||||
pos.set(qn("w:topFromText"), "0")
|
||||
pos.set(qn("w:bottomFromText"), "0")
|
||||
pos.set(qn("w:vertAnchor"), "page")
|
||||
pos.set(qn("w:horzAnchor"), "page")
|
||||
pos.set(qn("w:tblpX"), str(_twips(max(0.0, float(bbox.x)))))
|
||||
pos.set(qn("w:tblpY"), str(_twips(top)))
|
||||
tblPr.insert(0, pos)
|
||||
return True
|
||||
|
||||
|
||||
def page_is_positionable(page: Page, *, min_ratio: float = 0.5) -> bool:
|
||||
"""Whether enough of a page has real geometry for positioning to mean anything.
|
||||
|
||||
A page where most blocks would fall back to flowing is better emitted as a
|
||||
flowing page outright: a handful of floating frames interleaved with body
|
||||
text reads worse than either approach on its own.
|
||||
"""
|
||||
blocks = [
|
||||
b
|
||||
for b in page.blocks
|
||||
if b.type not in (BlockType.header, BlockType.footer) and (b.plain_text().strip() or b.image_png)
|
||||
]
|
||||
if not blocks:
|
||||
return False
|
||||
usable = sum(1 for b in blocks if has_usable_geometry(b))
|
||||
return usable / len(blocks) >= min_ratio
|
||||
|
||||
|
||||
def figure_size_points(block: Block, page: Page) -> tuple[float, float]:
|
||||
"""Width/height for a positioned figure, clamped to the page."""
|
||||
page_w = float(page.width or 612.0)
|
||||
page_h = float(page.height or 792.0)
|
||||
w = float(block.bbox.w or 0) or page_w * 0.4
|
||||
h = float(block.bbox.h or 0) or page_h * 0.3
|
||||
return min(w, page_w), min(h, page_h)
|
||||
|
||||
|
||||
__all__ = [
|
||||
"FRAME_PADDING_POINTS",
|
||||
"MIN_FRAME_POINTS",
|
||||
"TWIPS_PER_POINT",
|
||||
"figure_size_points",
|
||||
"has_usable_geometry",
|
||||
"page_is_positionable",
|
||||
"position_figure",
|
||||
"position_paragraph",
|
||||
"position_table",
|
||||
]
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,161 @@
|
||||
"""Format IDM document into native Microsoft PowerPoint (.pptx)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
from pptx import Presentation
|
||||
from pptx.util import Inches, Pt
|
||||
from pptx.dml.color import RGBColor
|
||||
from pptx.enum.text import PP_ALIGN
|
||||
from pptx.enum.shapes import MSO_SHAPE
|
||||
|
||||
from app.services.convert.idm.model import BlockType, Document, Page
|
||||
|
||||
|
||||
def format_pptx(doc: Document) -> bytes:
|
||||
"""Serialize canonical IDM document into a clean Microsoft PowerPoint presentation."""
|
||||
prs = Presentation()
|
||||
|
||||
# Determine presentation slide dimensions from first page (default 16:9 widescreen: 13.333 x 7.5 in)
|
||||
if doc.pages:
|
||||
p0 = doc.pages[0]
|
||||
w_pt = p0.width or 612.0
|
||||
h_pt = p0.height or 792.0
|
||||
# If landscape or wide, set 16:9 widescreen
|
||||
if w_pt >= h_pt:
|
||||
prs.slide_width = Inches(13.333)
|
||||
prs.slide_height = Inches(7.5)
|
||||
else:
|
||||
# 4:3 standard
|
||||
prs.slide_width = Inches(10.0)
|
||||
prs.slide_height = Inches(7.5)
|
||||
else:
|
||||
prs.slide_width = Inches(13.333)
|
||||
prs.slide_height = Inches(7.5)
|
||||
|
||||
blank_layout = prs.slide_layouts[6] # Blank slide
|
||||
|
||||
slide_w = prs.slide_width
|
||||
slide_h = prs.slide_height
|
||||
|
||||
for page in doc.pages:
|
||||
slide = prs.slides.add_slide(blank_layout)
|
||||
|
||||
# Scale factor from PDF points to PPTX slide dimensions
|
||||
pdf_w = page.width or 612.0
|
||||
pdf_h = page.height or 792.0
|
||||
sx = float(slide_w) / max(pdf_w, 1.0)
|
||||
sy = float(slide_h) / max(pdf_h, 1.0)
|
||||
|
||||
# Process blocks
|
||||
for block in page.blocks:
|
||||
if block.type == BlockType.table and block.cells:
|
||||
_render_table(slide, block, sx, sy, float(slide_w), float(slide_h), pdf_h)
|
||||
elif block.type == BlockType.heading:
|
||||
_render_heading(slide, block, sx, sy, float(slide_w), float(slide_h), pdf_h)
|
||||
elif block.type in (BlockType.paragraph, BlockType.list_item):
|
||||
_render_text(slide, block, sx, sy, float(slide_w), float(slide_h), pdf_h)
|
||||
|
||||
buf = io.BytesIO()
|
||||
prs.save(buf)
|
||||
return buf.getvalue()
|
||||
|
||||
|
||||
def _render_heading(
|
||||
slide, block, sx: float, sy: float, max_w: float, max_h: float, pdf_h: float
|
||||
) -> None:
|
||||
text = block.plain_text().strip()
|
||||
if not text:
|
||||
return
|
||||
|
||||
# PDF origin is bottom-left, PPTX origin is top-left
|
||||
top = int((pdf_h - (block.bbox.y + block.bbox.h)) * sy) if block.bbox.h else Inches(0.8)
|
||||
left = int(block.bbox.x * sx) if block.bbox.w else Inches(0.8)
|
||||
width = int(block.bbox.w * sx) if block.bbox.w else int(max_w - Inches(1.6))
|
||||
height = max(int(Inches(0.6)), int(block.bbox.h * sy))
|
||||
|
||||
# Clamp coordinates inside slide
|
||||
top = max(0, min(int(max_h - Inches(0.8)), top))
|
||||
left = max(0, min(int(max_w - Inches(1.5)), left))
|
||||
width = max(int(Inches(2.0)), min(int(max_w - left), width))
|
||||
|
||||
txBox = slide.shapes.add_textbox(left, top, width, height)
|
||||
tf = txBox.text_frame
|
||||
tf.word_wrap = True
|
||||
p = tf.paragraphs[0]
|
||||
p.text = text
|
||||
p.font.size = Pt(22)
|
||||
p.font.bold = True
|
||||
p.font.color.rgb = RGBColor(0x1E, 0x3A, 0x8A) # Navy corporate
|
||||
|
||||
|
||||
def _render_text(
|
||||
slide, block, sx: float, sy: float, max_w: float, max_h: float, pdf_h: float
|
||||
) -> None:
|
||||
text = block.plain_text().strip()
|
||||
if not text:
|
||||
return
|
||||
|
||||
top = int((pdf_h - (block.bbox.y + block.bbox.h)) * sy) if block.bbox.h else Inches(1.6)
|
||||
left = int(block.bbox.x * sx) if block.bbox.w else Inches(0.8)
|
||||
width = int(block.bbox.w * sx) if block.bbox.w else int(max_w - Inches(1.6))
|
||||
height = max(int(Inches(0.4)), int(block.bbox.h * sy))
|
||||
|
||||
top = max(0, min(int(max_h - Inches(0.6)), top))
|
||||
left = max(0, min(int(max_w - Inches(1.5)), left))
|
||||
width = max(int(Inches(2.0)), min(int(max_w - left), width))
|
||||
|
||||
txBox = slide.shapes.add_textbox(left, top, width, height)
|
||||
tf = txBox.text_frame
|
||||
tf.word_wrap = True
|
||||
p = tf.paragraphs[0]
|
||||
p.text = text
|
||||
p.font.size = Pt(13)
|
||||
p.font.color.rgb = RGBColor(0x1F, 0x29, 0x37) # Dark Charcoal
|
||||
|
||||
|
||||
def _render_table(
|
||||
slide, block, sx: float, sy: float, max_w: float, max_h: float, pdf_h: float
|
||||
) -> None:
|
||||
grid = block.cells
|
||||
if not grid or not grid[0]:
|
||||
return
|
||||
|
||||
rows = len(grid)
|
||||
cols = max(len(r) for r in grid)
|
||||
if rows < 1 or cols < 1:
|
||||
return
|
||||
|
||||
top = int((pdf_h - (block.bbox.y + block.bbox.h)) * sy) if block.bbox.h else Inches(2.0)
|
||||
left = int(block.bbox.x * sx) if block.bbox.w else Inches(0.8)
|
||||
width = int(block.bbox.w * sx) if block.bbox.w else int(max_w - Inches(1.6))
|
||||
height = max(int(Inches(1.0)), int(block.bbox.h * sy))
|
||||
|
||||
top = max(0, min(int(max_h - Inches(1.5)), top))
|
||||
left = max(0, min(int(max_w - Inches(2.0)), left))
|
||||
width = max(int(Inches(3.0)), min(int(max_w - left), width))
|
||||
|
||||
table_shape = slide.shapes.add_table(rows, cols, left, top, width, height)
|
||||
tbl = table_shape.table
|
||||
|
||||
for r_idx, row in enumerate(grid):
|
||||
is_header = r_idx == 0
|
||||
for c_idx in range(cols):
|
||||
cell_val = row[c_idx] if c_idx < len(row) else ""
|
||||
cell = tbl.cell(r_idx, c_idx)
|
||||
cell.text = str(cell_val).strip()
|
||||
|
||||
p = cell.text_frame.paragraphs[0]
|
||||
p.font.size = Pt(10)
|
||||
if is_header:
|
||||
p.font.bold = True
|
||||
p.font.color.rgb = RGBColor(0xFF, 0xFF, 0xFF)
|
||||
cell.fill.solid()
|
||||
cell.fill.fore_color.rgb = RGBColor(0x1E, 0x3A, 0x8A) # Navy
|
||||
else:
|
||||
p.font.color.rgb = RGBColor(0x11, 0x18, 0x27)
|
||||
cell.fill.solid()
|
||||
if r_idx % 2 == 1:
|
||||
cell.fill.fore_color.rgb = RGBColor(0xF9, 0xFA, 0xFB)
|
||||
else:
|
||||
cell.fill.fore_color.rgb = RGBColor(0xFF, 0xFF, 0xFF)
|
||||
@@ -0,0 +1,271 @@
|
||||
"""HTML / MD / TXT / JSON formatters from IDM."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
|
||||
from app.services.convert.formatters.text_postprocess import (
|
||||
postprocess_inline,
|
||||
postprocess_plain_text,
|
||||
)
|
||||
from app.services.convert.idm.model import Block, BlockType, Document
|
||||
from app.services.convert.idm.serialize import document_to_dict
|
||||
from app.services.convert.text.arabic_logical import contains_arabic, to_logical
|
||||
|
||||
|
||||
def _md_cell(text: str) -> str:
|
||||
"""Escape a value so it cannot break out of a Markdown table cell."""
|
||||
return (
|
||||
(text or "")
|
||||
.replace("\\", "\\\\")
|
||||
.replace("|", "\\|")
|
||||
.replace("\r\n", " ")
|
||||
.replace("\n", "<br>")
|
||||
.replace("\r", " ")
|
||||
.strip()
|
||||
)
|
||||
|
||||
|
||||
def _md_table(cells: list[list[str]], visual: bool) -> list[str]:
|
||||
"""Render a grid as a valid Markdown table.
|
||||
|
||||
Every row is padded to the widest row so the column count matches the
|
||||
delimiter row; ragged input otherwise produces a table no parser accepts.
|
||||
"""
|
||||
grid = [row for row in cells if row is not None]
|
||||
if not grid:
|
||||
return []
|
||||
width = max(len(row) for row in grid)
|
||||
if width == 0:
|
||||
return []
|
||||
out: list[str] = []
|
||||
for idx, row in enumerate(grid):
|
||||
values = [_md_cell(postprocess_inline(to_logical(c, visual=visual))) for c in row]
|
||||
values += [""] * (width - len(values))
|
||||
out.append("| " + " | ".join(values) + " |")
|
||||
if idx == 0:
|
||||
out.append("| " + " | ".join(["---"] * width) + " |")
|
||||
out.append("")
|
||||
return out
|
||||
|
||||
|
||||
def md_table_lines(cells: list[list[str]]) -> list[str]:
|
||||
"""Public entry to the Markdown table renderer, for non-IDM callers."""
|
||||
return _md_table(cells, False)
|
||||
|
||||
|
||||
def _escape(s: str) -> str:
|
||||
return (
|
||||
s.replace("&", "&")
|
||||
.replace("<", "<")
|
||||
.replace(">", ">")
|
||||
.replace('"', """)
|
||||
)
|
||||
|
||||
|
||||
def _block_visual(block) -> bool:
|
||||
"""True when a block's text is still in visual order and needs repairing.
|
||||
|
||||
Ordering is now resolved at extraction time — ``layout.glyphs`` and the OCR
|
||||
adapter sort each line by its own reading direction — so a block's source
|
||||
no longer implies visual order. Only an explicit span flag does, which is
|
||||
set when a producer genuinely could not determine direction.
|
||||
"""
|
||||
return any(getattr(s, "visual_order", False) for s in (getattr(block, "spans", None) or []))
|
||||
|
||||
|
||||
def _norm_text(block: Block) -> str:
|
||||
return postprocess_inline(to_logical(block.plain_text(), visual=_block_visual(block)))
|
||||
|
||||
|
||||
def format_txt(doc: Document) -> bytes:
|
||||
parts: list[str] = []
|
||||
for page in doc.pages:
|
||||
for block in sorted(page.blocks, key=lambda b: b.reading_order):
|
||||
t = _norm_text(block).strip()
|
||||
if t:
|
||||
parts.append(t)
|
||||
return postprocess_plain_text("\n".join(parts)).encode("utf-8")
|
||||
|
||||
|
||||
def format_md(doc: Document, *, page_markers: bool | None = None) -> bytes:
|
||||
"""IDM → Markdown.
|
||||
|
||||
``page_markers`` inserts a ``## Page N`` heading per page. It defaults to
|
||||
off because page headings fragment semantic chunking for search and LLM
|
||||
ingest; set ``CONVERT_MD_PAGE_MARKERS=1`` or pass True to restore them.
|
||||
"""
|
||||
if page_markers is None:
|
||||
page_markers = os.environ.get("CONVERT_MD_PAGE_MARKERS", "0").strip().lower() in (
|
||||
"1",
|
||||
"true",
|
||||
"yes",
|
||||
"on",
|
||||
)
|
||||
parts: list[str] = []
|
||||
for page in doc.pages:
|
||||
if page_markers:
|
||||
parts.append(f"## Page {page.index + 1}\n")
|
||||
in_list = False
|
||||
for block in sorted(page.blocks, key=lambda b: b.reading_order):
|
||||
if block.type == BlockType.heading:
|
||||
if in_list:
|
||||
parts.append("")
|
||||
in_list = False
|
||||
level = min(max(block.level, 1), 3)
|
||||
parts.append("#" * level + f" {_norm_text(block)}\n")
|
||||
elif block.type == BlockType.table and block.cells:
|
||||
if in_list:
|
||||
parts.append("")
|
||||
in_list = False
|
||||
parts.extend(_md_table(block.cells, _block_visual(block)))
|
||||
elif block.type == BlockType.list_item:
|
||||
in_list = True
|
||||
marker = "1." if block.list_style == "number" else "-"
|
||||
parts.append(f"{marker} {_norm_text(block)}")
|
||||
else:
|
||||
if in_list:
|
||||
parts.append("")
|
||||
in_list = False
|
||||
t = _norm_text(block).strip()
|
||||
if t:
|
||||
parts.append(t)
|
||||
parts.append("")
|
||||
if in_list:
|
||||
parts.append("")
|
||||
# Document-level passes for cross-block hyphenation / page# / leaders
|
||||
return postprocess_plain_text("\n".join(parts)).encode("utf-8")
|
||||
|
||||
|
||||
def format_html(doc: Document) -> bytes:
|
||||
chunks = [
|
||||
'<!DOCTYPE html><html><head><meta charset="utf-8">'
|
||||
"<title>Converted PDF</title></head><body>"
|
||||
]
|
||||
for page in doc.pages:
|
||||
chunks.append(f'<section data-page="{page.index + 1}">')
|
||||
chunks.append(f"<h2>Page {page.index + 1}</h2>")
|
||||
open_list: str | None = None # "ul" | "ol"
|
||||
|
||||
def _close_list() -> None:
|
||||
nonlocal open_list
|
||||
if open_list:
|
||||
chunks.append(f"</{open_list}>")
|
||||
open_list = None
|
||||
|
||||
for block in sorted(page.blocks, key=lambda b: b.reading_order):
|
||||
if block.type == BlockType.heading:
|
||||
_close_list()
|
||||
level = min(max(block.level, 1), 3)
|
||||
txt = _escape(_norm_text(block))
|
||||
rtl = ' dir="rtl" lang="ar"' if contains_arabic(block.plain_text()) else ""
|
||||
chunks.append(f"<h{level}{rtl}>{txt}</h{level}>")
|
||||
elif block.type == BlockType.table and block.cells:
|
||||
_close_list()
|
||||
visual = _block_visual(block)
|
||||
chunks.append('<table border="1" cellpadding="4">')
|
||||
for ri, row in enumerate(block.cells):
|
||||
tag = "th" if ri == 0 else "td"
|
||||
cells_html = []
|
||||
for c in row:
|
||||
cell_txt = _escape(to_logical(c, visual=visual))
|
||||
rtl = ' dir="rtl"' if contains_arabic(c) else ""
|
||||
cells_html.append(f"<{tag}{rtl}>{cell_txt}</{tag}>")
|
||||
chunks.append("<tr>" + "".join(cells_html) + "</tr>")
|
||||
chunks.append("</table>")
|
||||
elif block.type == BlockType.list_item:
|
||||
want = "ol" if block.list_style == "number" else "ul"
|
||||
if open_list != want:
|
||||
_close_list()
|
||||
chunks.append(f"<{want}>")
|
||||
open_list = want
|
||||
txt = _escape(_norm_text(block))
|
||||
rtl = ' dir="rtl"' if contains_arabic(block.plain_text()) else ""
|
||||
chunks.append(f"<li{rtl}>{txt}</li>")
|
||||
else:
|
||||
_close_list()
|
||||
t = _norm_text(block).strip()
|
||||
if t:
|
||||
rtl = ' dir="rtl" lang="ar"' if contains_arabic(block.plain_text()) else ""
|
||||
chunks.append(f"<p{rtl}>{_escape(t)}</p>")
|
||||
_close_list()
|
||||
chunks.append("</section>")
|
||||
chunks.append("</body></html>")
|
||||
return "\n".join(chunks).encode("utf-8")
|
||||
|
||||
|
||||
def format_json(doc: Document) -> bytes:
|
||||
payload = document_to_dict(doc)
|
||||
# Compatibility fields for consumers expecting page.text
|
||||
for page_dict, page in zip(payload["pages"], doc.pages, strict=True):
|
||||
page_dict["text"] = page.all_text()
|
||||
page_dict["paragraphs"] = [
|
||||
_norm_text(b)
|
||||
for b in page.blocks
|
||||
if b.type in (BlockType.paragraph, BlockType.heading, BlockType.list_item)
|
||||
]
|
||||
page_dict["tables"] = [b.cells for b in page.blocks if b.type == BlockType.table]
|
||||
return json.dumps(payload, indent=2, ensure_ascii=False).encode("utf-8")
|
||||
|
||||
|
||||
def format_csv(doc: Document, *, dialect: str = "excel") -> bytes:
|
||||
"""IDM Document -> RFC 4180 compliant CSV bytes with UTF-8 BOM.
|
||||
|
||||
Extracts all detected tables across document pages.
|
||||
- Normalizes ragged rows to the maximum column width per table.
|
||||
- Normalizes Arabic/RTL text to logical order.
|
||||
- Preserves embedded commas, quotes, and newlines per RFC 4180.
|
||||
- Emits UTF-8 BOM (\\xef\\xbb\\xbf) for seamless Excel display on Windows.
|
||||
- If multiple tables exist, separates them with clean section headers.
|
||||
- If zero tables exist, safely exports non-empty paragraphs as single-column records.
|
||||
"""
|
||||
import codecs
|
||||
import csv
|
||||
import io
|
||||
|
||||
stream = io.StringIO()
|
||||
writer = csv.writer(stream, dialect=dialect, quoting=csv.QUOTE_MINIMAL, lineterminator="\r\n")
|
||||
|
||||
# Collect all tables across pages
|
||||
tables_found: list[tuple[int, list[list[str]], bool]] = []
|
||||
for page in doc.pages:
|
||||
for block in sorted(page.blocks, key=lambda b: b.reading_order):
|
||||
if block.type == BlockType.table and block.cells:
|
||||
tables_found.append((page.index + 1, block.cells, _block_visual(block)))
|
||||
|
||||
if tables_found:
|
||||
for idx, (page_num, raw_cells, visual) in enumerate(tables_found, 1):
|
||||
if idx > 1:
|
||||
writer.writerow([])
|
||||
writer.writerow([f"# --- Table {idx} (Page {page_num}) ---"])
|
||||
elif len(tables_found) > 1:
|
||||
writer.writerow([f"# --- Table 1 (Page {page_num}) ---"])
|
||||
|
||||
grid = [row for row in raw_cells if row is not None]
|
||||
if not grid:
|
||||
continue
|
||||
width = max(len(row) for row in grid)
|
||||
if width == 0:
|
||||
continue
|
||||
|
||||
for row in grid:
|
||||
cleaned_row = []
|
||||
for c in row:
|
||||
val = postprocess_inline(to_logical(str(c or ""), visual=visual)).strip()
|
||||
val = val.replace("\r\n", "\n").replace("\r", "\n").replace("<br>", "\n")
|
||||
cleaned_row.append(val)
|
||||
if len(cleaned_row) < width:
|
||||
cleaned_row.extend([""] * (width - len(cleaned_row)))
|
||||
writer.writerow(cleaned_row)
|
||||
else:
|
||||
# Fallback for documents with no detected tables: export paragraphs
|
||||
for page in doc.pages:
|
||||
for block in sorted(page.blocks, key=lambda b: b.reading_order):
|
||||
t = _norm_text(block).strip()
|
||||
if t:
|
||||
writer.writerow([t])
|
||||
|
||||
csv_text = stream.getvalue()
|
||||
return codecs.BOM_UTF8 + csv_text.encode("utf-8")
|
||||
|
||||
@@ -0,0 +1,108 @@
|
||||
"""MD/HTML/TXT postprocess — hyphenation, TOC leaders, page numbers, drop-caps.
|
||||
|
||||
Every pass here deletes or rewrites extracted content, so each one is scoped as
|
||||
narrowly as the signal allows. A pass that removes a running page number must
|
||||
not also remove a standalone figure in a financial table, and a pass that
|
||||
merges a drop cap must not merge the "A." of a lettered list.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
_HYPHEN_BREAK_RE = re.compile(r"(\w)[\-‐‑\x02]\n+(\w)")
|
||||
_DOT_LEADER_RE = re.compile(r"(\S)\s*\.{2,}\s*(\S)")
|
||||
# A page-number line: an optional "page" word, digits, optional dash decoration.
|
||||
# Roman numerals are included because front matter uses them.
|
||||
_PAGE_NUM_LINE_RE = re.compile(
|
||||
r"^\s*(?:page\s+)?\d{1,4}\s*$"
|
||||
r"|^\s*[-–—]\s*\d{1,4}\s*[-–—]\s*$"
|
||||
r"|^\s*(?:page\s+)?\d{1,4}\s*(?:of|/)\s*\d{1,4}\s*$"
|
||||
r"|^\s*[ivxlcdm]{1,7}\s*$",
|
||||
re.I,
|
||||
)
|
||||
# A drop cap is a lone capital followed by lowercase continuation text. Require
|
||||
# the continuation to look like prose so "A." / "B." list markers are untouched.
|
||||
_DROP_CAP_RE = re.compile(r"(?m)^([A-ZÀ-ÖØ-Þ])\n+([a-zà-öø-ÿ][a-zà-öø-ÿ ,;]{8,})")
|
||||
|
||||
# Do not treat a lone number as a page number when it carries a decimal point,
|
||||
# a thousands separator, a sign or a currency symbol — those are data.
|
||||
_DATA_NUMBER_RE = re.compile(r"[.,%$€£¥₹+\-]")
|
||||
|
||||
|
||||
def rejoin_hyphenated_breaks(text: str) -> str:
|
||||
"""Join end-of-line hyphenation: ``end-\\nword`` → ``endword``."""
|
||||
if not text:
|
||||
return text
|
||||
prev = None
|
||||
out = text
|
||||
while prev != out:
|
||||
prev = out
|
||||
out = _HYPHEN_BREAK_RE.sub(r"\1\2", out)
|
||||
return out
|
||||
|
||||
|
||||
def collapse_dot_leaders(text: str) -> str:
|
||||
"""Collapse TOC-style dot leaders to ``' ... '``."""
|
||||
if not text:
|
||||
return text
|
||||
return _DOT_LEADER_RE.sub(r"\1 ... \2", text)
|
||||
|
||||
|
||||
def _looks_like_page_number(line: str) -> bool:
|
||||
stripped = line.strip()
|
||||
if not _PAGE_NUM_LINE_RE.match(line):
|
||||
return False
|
||||
# "1,234" or "12.5" or "-40" are values, not page numbers.
|
||||
return not _DATA_NUMBER_RE.search(stripped)
|
||||
|
||||
|
||||
def filter_page_number_lines(text: str, *, min_repeats: int = 2) -> str:
|
||||
"""Drop running page-number lines.
|
||||
|
||||
A single standalone number is ambiguous — it may be a page number or a
|
||||
value from a table that lost its row. The line is removed only when the
|
||||
document shows the *pattern* repeatedly (at least ``min_repeats`` such
|
||||
lines), which is what a running folio looks like and what an isolated data
|
||||
point does not.
|
||||
"""
|
||||
if not text:
|
||||
return text
|
||||
lines = text.splitlines()
|
||||
if len(lines) < 2:
|
||||
return text
|
||||
candidates = [i for i, ln in enumerate(lines) if _looks_like_page_number(ln)]
|
||||
if len(candidates) < min_repeats:
|
||||
return text
|
||||
drop = set(candidates)
|
||||
return "\n".join(ln for i, ln in enumerate(lines) if i not in drop)
|
||||
|
||||
|
||||
def merge_drop_caps(text: str) -> str:
|
||||
"""Merge single-letter drop-cap lines into the following paragraph."""
|
||||
if not text:
|
||||
return text
|
||||
return _DROP_CAP_RE.sub(r"\1\2", text)
|
||||
|
||||
|
||||
def postprocess_inline(text: str) -> str:
|
||||
"""Safe per-block polish (no page-number stripping)."""
|
||||
if not text:
|
||||
return text
|
||||
out = rejoin_hyphenated_breaks(text)
|
||||
out = collapse_dot_leaders(out)
|
||||
return out
|
||||
|
||||
|
||||
def postprocess_plain_text(text: str) -> str:
|
||||
"""Apply all MD/TXT document-level polish passes."""
|
||||
if not text:
|
||||
return text
|
||||
out = rejoin_hyphenated_breaks(text)
|
||||
out = collapse_dot_leaders(out)
|
||||
out = filter_page_number_lines(out)
|
||||
out = merge_drop_caps(out)
|
||||
out = re.sub(r"\n{3,}", "\n\n", out)
|
||||
if text.endswith("\n"):
|
||||
return out.strip() + "\n"
|
||||
return out.strip()
|
||||
@@ -0,0 +1,276 @@
|
||||
"""XLSX formatter from IDM."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import re
|
||||
from datetime import date
|
||||
|
||||
from openpyxl import Workbook
|
||||
from openpyxl.styles import Font
|
||||
from openpyxl.utils import get_column_letter
|
||||
|
||||
from app.services.convert.formatters.xml_sanitize import sanitize_ooxml_text
|
||||
from app.services.convert.idm.model import BlockType, Document
|
||||
from app.services.convert.text.arabic_logical import to_logical_cell
|
||||
from app.services.convert.text.numbers import (
|
||||
excel_number_format,
|
||||
parse_iso_date,
|
||||
parse_number,
|
||||
)
|
||||
|
||||
|
||||
def _block_visual(block) -> bool:
|
||||
"""True when a block's text is still in visual order and needs repairing.
|
||||
|
||||
Ordering is now resolved at extraction time — ``layout.glyphs`` and the OCR
|
||||
adapter sort each line by its own reading direction — so a block's source
|
||||
no longer implies visual order. Only an explicit span flag does, which is
|
||||
set when a producer genuinely could not determine direction.
|
||||
"""
|
||||
return any(getattr(s, "visual_order", False) for s in (getattr(block, "spans", None) or []))
|
||||
|
||||
|
||||
# Excel grid limits (xlsx). Exceeding either raises inside openpyxl.
|
||||
MAX_XLSX_ROWS = 1_048_576
|
||||
MAX_XLSX_COLS = 16_384
|
||||
|
||||
|
||||
def _safe_sheet_name(name: str, used: set[str]) -> str:
|
||||
cleaned = re.sub(r"[\[\]\*\/\\\?\:]", "_", sanitize_ooxml_text(name))[:31] or "Sheet"
|
||||
base = cleaned
|
||||
n = 1
|
||||
while cleaned in used:
|
||||
suffix = f"_{n}"
|
||||
cleaned = base[: 31 - len(suffix)] + suffix
|
||||
n += 1
|
||||
used.add(cleaned)
|
||||
return cleaned
|
||||
|
||||
|
||||
def coerce_cell(value: str, *, visual: bool = False) -> tuple[object, str | None]:
|
||||
"""Return ``(python value, excel number format)`` for a cell string.
|
||||
|
||||
Numbers are written as numbers so ``=SUM()`` works — storing them as text
|
||||
is the most common complaint about PDF-to-Excel output. The display format
|
||||
travels with the value so "4.50", "1,234" and "12%" still look the same in
|
||||
Excel while being numeric underneath.
|
||||
|
||||
Labels that merely contain digits ("R1", "Q3", "00123") stay text: coercing
|
||||
them would rewrite row identifiers into values.
|
||||
"""
|
||||
raw = sanitize_ooxml_text(to_logical_cell(value, visual=visual)).strip()
|
||||
if not raw:
|
||||
return "", None
|
||||
|
||||
iso = parse_iso_date(raw)
|
||||
if iso:
|
||||
return date(*iso), "yyyy-mm-dd"
|
||||
|
||||
parsed = parse_number(raw)
|
||||
if parsed is None:
|
||||
return raw, None
|
||||
|
||||
fmt = excel_number_format(parsed)
|
||||
if parsed.is_integral and not parsed.percent and not parsed.currency:
|
||||
return int(parsed.value), fmt
|
||||
return parsed.value, fmt
|
||||
|
||||
|
||||
def _coerce_cell(value: str, *, visual: bool = False):
|
||||
"""Backwards-compatible single-value form used by existing callers/tests."""
|
||||
return coerce_cell(value, visual=visual)[0]
|
||||
|
||||
|
||||
def _write_prose(ws, blocks, start_row: int = 1) -> int:
|
||||
r_idx = start_row
|
||||
for block in sorted(blocks, key=lambda b: b.reading_order):
|
||||
if block.type in (BlockType.header, BlockType.footer, BlockType.table, BlockType.figure):
|
||||
continue
|
||||
text = sanitize_ooxml_text(
|
||||
to_logical_cell(block.plain_text(), visual=_block_visual(block))
|
||||
).strip()
|
||||
if not text:
|
||||
continue
|
||||
value, number_format = coerce_cell(text)
|
||||
cell = ws.cell(row=r_idx, column=1, value=value)
|
||||
if number_format:
|
||||
cell.number_format = number_format
|
||||
if block.type == BlockType.heading or any(s.bold for s in block.spans):
|
||||
cell.font = Font(bold=True)
|
||||
r_idx += 1
|
||||
return r_idx
|
||||
|
||||
|
||||
def format_xlsx(doc: Document) -> tuple[bytes, list[str]]:
|
||||
warnings: list[str] = []
|
||||
wb = Workbook()
|
||||
default = wb.active
|
||||
any_table = False
|
||||
used_names: set[str] = set()
|
||||
first_sheet = True
|
||||
|
||||
all_blocks = [b for page in doc.pages for b in page.blocks]
|
||||
high_conf_tables = [
|
||||
b
|
||||
for b in all_blocks
|
||||
if b.type == BlockType.table and b.cells and b.table_confidence >= 0.65
|
||||
]
|
||||
|
||||
# Always keep a Content sheet with non-table prose (invoice KV, etc.)
|
||||
content_needed = any(
|
||||
b.type not in (BlockType.table, BlockType.header, BlockType.footer, BlockType.figure)
|
||||
and b.plain_text().strip()
|
||||
for b in all_blocks
|
||||
)
|
||||
|
||||
if len(high_conf_tables) > 1:
|
||||
for ti, table_block in enumerate(high_conf_tables):
|
||||
any_table = True
|
||||
name = _safe_sheet_name(f"Table{ti + 1}", used_names)
|
||||
if first_sheet:
|
||||
ws = default
|
||||
ws.title = name
|
||||
first_sheet = False
|
||||
else:
|
||||
ws = wb.create_sheet(name)
|
||||
_write_table(
|
||||
ws,
|
||||
table_block.cells,
|
||||
table_block.merges,
|
||||
freeze=table_block.table_confidence >= 0.75,
|
||||
warnings=warnings,
|
||||
visual=_block_visual(table_block),
|
||||
)
|
||||
if content_needed:
|
||||
name = _safe_sheet_name("Content", used_names)
|
||||
ws = wb.create_sheet(name)
|
||||
_write_prose(ws, all_blocks)
|
||||
else:
|
||||
for page in doc.pages:
|
||||
name = _safe_sheet_name(f"Page{page.index + 1}", used_names)
|
||||
if first_sheet:
|
||||
ws = default
|
||||
ws.title = name
|
||||
first_sheet = False
|
||||
else:
|
||||
ws = wb.create_sheet(name)
|
||||
|
||||
tables = [b for b in page.blocks if b.type == BlockType.table and b.cells]
|
||||
prose_blocks = [b for b in page.blocks if b.type != BlockType.table]
|
||||
r_next = 1
|
||||
if prose_blocks:
|
||||
r_next = _write_prose(ws, prose_blocks, start_row=1)
|
||||
if tables:
|
||||
r_next += 1
|
||||
if tables:
|
||||
any_table = True
|
||||
# Write first table starting at r_next
|
||||
_write_table_at(
|
||||
ws,
|
||||
tables[0].cells,
|
||||
tables[0].merges,
|
||||
start_row=r_next,
|
||||
freeze=tables[0].table_confidence >= 0.75,
|
||||
warnings=warnings,
|
||||
visual=_block_visual(tables[0]),
|
||||
)
|
||||
col_offset = max(len(r) for r in tables[0].cells) + 2
|
||||
for extra in tables[1:]:
|
||||
_write_table(
|
||||
ws,
|
||||
extra.cells,
|
||||
extra.merges,
|
||||
col_offset=col_offset,
|
||||
warnings=warnings,
|
||||
visual=_block_visual(extra),
|
||||
)
|
||||
col_offset += max(len(r) for r in extra.cells) + 2
|
||||
|
||||
if not any_table:
|
||||
warnings.append("No table structure detected; exported paragraphs as a single column.")
|
||||
|
||||
buf = io.BytesIO()
|
||||
wb.save(buf)
|
||||
return buf.getvalue(), warnings
|
||||
|
||||
|
||||
def _write_table(
|
||||
ws,
|
||||
cells: list[list[str]],
|
||||
merges: list[tuple[int, int, int, int]] | None = None,
|
||||
*,
|
||||
col_offset: int = 0,
|
||||
freeze: bool = False,
|
||||
warnings: list[str] | None = None,
|
||||
visual: bool = False,
|
||||
) -> None:
|
||||
_write_table_at(
|
||||
ws,
|
||||
cells,
|
||||
merges,
|
||||
start_row=1,
|
||||
col_offset=col_offset,
|
||||
freeze=freeze,
|
||||
warnings=warnings,
|
||||
visual=visual,
|
||||
)
|
||||
|
||||
|
||||
def _write_table_at(
|
||||
ws,
|
||||
cells: list[list[str]],
|
||||
merges: list[tuple[int, int, int, int]] | None = None,
|
||||
*,
|
||||
start_row: int = 1,
|
||||
col_offset: int = 0,
|
||||
freeze: bool = False,
|
||||
warnings: list[str] | None = None,
|
||||
visual: bool = False,
|
||||
) -> None:
|
||||
for r_idx, row in enumerate(cells):
|
||||
if start_row + r_idx > MAX_XLSX_ROWS:
|
||||
if warnings is not None:
|
||||
warnings.append(
|
||||
f"Table truncated at {MAX_XLSX_ROWS} rows (Excel sheet limit)."
|
||||
)
|
||||
break
|
||||
for c_idx, cell in enumerate(row):
|
||||
if col_offset + c_idx + 1 > MAX_XLSX_COLS:
|
||||
if warnings is not None:
|
||||
warnings.append(
|
||||
f"Table truncated at {MAX_XLSX_COLS} columns (Excel sheet limit)."
|
||||
)
|
||||
break
|
||||
number_format: str | None = None
|
||||
if isinstance(cell, str):
|
||||
val, number_format = coerce_cell(cell, visual=visual)
|
||||
elif cell is None:
|
||||
val = ""
|
||||
else:
|
||||
val = cell
|
||||
xl = ws.cell(row=start_row + r_idx, column=col_offset + c_idx + 1, value=val)
|
||||
if number_format:
|
||||
xl.number_format = number_format
|
||||
if r_idx == 0:
|
||||
xl.font = Font(bold=True)
|
||||
# Column width heuristic
|
||||
try:
|
||||
letter = get_column_letter(col_offset + c_idx + 1)
|
||||
cur = ws.column_dimensions[letter].width or 8
|
||||
ws.column_dimensions[letter].width = max(cur, min(28, len(str(val)) + 2))
|
||||
except Exception:
|
||||
pass
|
||||
for r0, c0, r1, c1 in merges or []:
|
||||
try:
|
||||
start = f"{get_column_letter(col_offset + c0 + 1)}{start_row + r0}"
|
||||
end = f"{get_column_letter(col_offset + c1 + 1)}{start_row + r1}"
|
||||
ws.merge_cells(f"{start}:{end}")
|
||||
except Exception as exc:
|
||||
if warnings is not None:
|
||||
warnings.append(f"XLSX merge failed ({r0},{c0})-({r1},{c1}): {exc}")
|
||||
if freeze and start_row >= 1:
|
||||
try:
|
||||
ws.freeze_panes = f"A{start_row + 1}"
|
||||
except Exception:
|
||||
pass
|
||||
@@ -0,0 +1,33 @@
|
||||
"""Strip characters illegal in XML 1.0 / OOXML text nodes.
|
||||
|
||||
PDF extractors sometimes emit C0 controls (e.g. \\x02 soft-break markers)
|
||||
that lxml (python-docx) and openpyxl reject on write. Also strip unpaired
|
||||
UTF-16 surrogates which can appear in broken PDF ToUnicode maps.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
# XML 1.0 Char production excludes C0 controls except TAB/LF/CR.
|
||||
# Also drop soft hyphen (U+00AD), BOM, non-chars, and lone surrogates.
|
||||
_ILLEGAL_OOXML_RE = re.compile(
|
||||
r"[\x00-\x08\x0b\x0c\x0e-\x1f\x7f-\x9f\u00ad\ufeff\ufffe\uffff"
|
||||
r"\ud800-\udfff]"
|
||||
)
|
||||
|
||||
|
||||
def sanitize_ooxml_text(value: object | None) -> str:
|
||||
"""Return text safe for OOXML / SpreadsheetML string cells."""
|
||||
if value is None:
|
||||
return ""
|
||||
if not isinstance(value, str):
|
||||
value = str(value)
|
||||
if not value:
|
||||
return ""
|
||||
# Drop unpaired surrogates that PDF extractors may leave in strings
|
||||
try:
|
||||
value.encode("utf-8")
|
||||
except UnicodeEncodeError:
|
||||
value = value.encode("utf-8", errors="surrogatepass").decode("utf-8", errors="ignore")
|
||||
return _ILLEGAL_OOXML_RE.sub("", value)
|
||||
@@ -0,0 +1,34 @@
|
||||
"""HTTP header helpers for convert responses (latin-1 safe)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from urllib.parse import quote
|
||||
|
||||
|
||||
def ascii_fallback_filename(filename: str) -> str:
|
||||
"""Strip/replace non-ASCII so Starlette can encode headers as latin-1."""
|
||||
raw = (filename or "download").replace('"', "_").replace("\r", "").replace("\n", "")
|
||||
ascii_name = raw.encode("ascii", "replace").decode("ascii")
|
||||
# '?' from replace looks ugly in downloads — use underscore
|
||||
ascii_name = ascii_name.replace("?", "_").strip() or "download"
|
||||
return ascii_name
|
||||
|
||||
|
||||
def content_disposition_attachment(filename: str) -> str:
|
||||
"""
|
||||
RFC 6266 / 5987 Content-Disposition.
|
||||
|
||||
HTTP header values must be latin-1. Arabic/CJK stems crash Starlette unless
|
||||
we provide an ASCII ``filename=`` fallback plus UTF-8 ``filename*=``.
|
||||
"""
|
||||
name = (filename or "download").replace("\r", "").replace("\n", "").strip() or "download"
|
||||
ascii_name = ascii_fallback_filename(name)
|
||||
if ascii_name == name and all(ord(c) < 128 for c in name):
|
||||
return f'attachment; filename="{name}"'
|
||||
encoded = quote(name, safe="")
|
||||
return f"attachment; filename=\"{ascii_name}\"; filename*=UTF-8''{encoded}"
|
||||
|
||||
|
||||
def ascii_header_value(value: str) -> str:
|
||||
"""Force a header value to ascii (warnings, fidelity extras)."""
|
||||
return (value or "").encode("ascii", "replace").decode("ascii")
|
||||
@@ -0,0 +1,24 @@
|
||||
"""IDM package."""
|
||||
|
||||
from app.services.convert.idm.model import (
|
||||
BBox,
|
||||
Block,
|
||||
BlockType,
|
||||
Document,
|
||||
Page,
|
||||
PageKind,
|
||||
TextSpan,
|
||||
)
|
||||
from app.services.convert.idm.serialize import dumps_idm, maybe_dump_idm
|
||||
|
||||
__all__ = [
|
||||
"BBox",
|
||||
"Block",
|
||||
"BlockType",
|
||||
"Document",
|
||||
"Page",
|
||||
"PageKind",
|
||||
"TextSpan",
|
||||
"dumps_idm",
|
||||
"maybe_dump_idm",
|
||||
]
|
||||
@@ -0,0 +1,119 @@
|
||||
"""Intermediate Document Model (IDM) for conversion."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from enum import Enum
|
||||
from typing import Any
|
||||
|
||||
|
||||
class PageKind(str, Enum):
|
||||
digital = "digital"
|
||||
scan = "scan"
|
||||
hybrid = "hybrid"
|
||||
blank = "blank"
|
||||
|
||||
|
||||
class BlockType(str, Enum):
|
||||
heading = "heading"
|
||||
paragraph = "paragraph"
|
||||
list_item = "list"
|
||||
table = "table"
|
||||
figure = "figure"
|
||||
header = "header"
|
||||
footer = "footer"
|
||||
|
||||
|
||||
# ``Block.source`` is usually ``digital`` or ``ocr``. This value marks a
|
||||
# whole source page that was deliberately embedded as a raster in a DOCX. It
|
||||
# is a fidelity fallback, not an OCR result: the writer must keep the picture
|
||||
# at page size and must not turn its placeholder text into a visible caption.
|
||||
PAGE_RASTER_SOURCE = "page-raster"
|
||||
|
||||
|
||||
@dataclass
|
||||
class BBox:
|
||||
x: float = 0.0
|
||||
y: float = 0.0
|
||||
w: float = 0.0
|
||||
h: float = 0.0
|
||||
|
||||
def as_list(self) -> list[float]:
|
||||
return [self.x, self.y, self.w, self.h]
|
||||
|
||||
|
||||
@dataclass
|
||||
class TextSpan:
|
||||
text: str
|
||||
font_name: str = ""
|
||||
font_size: float = 0.0
|
||||
bold: bool = False
|
||||
italic: bool = False
|
||||
x: float | None = None
|
||||
w: float | None = None
|
||||
url: str | None = None
|
||||
# Text colour as uppercase ``RRGGBB``, or None for Word's "automatic".
|
||||
# None is not the same as "000000": automatic follows the reader's theme,
|
||||
# which is what an author who never set a colour expects.
|
||||
color: str | None = None
|
||||
# True when OCR emitted visual LTR Arabic (needs reverse before OOXML storage)
|
||||
visual_order: bool = False
|
||||
|
||||
|
||||
@dataclass
|
||||
class Block:
|
||||
type: BlockType
|
||||
text: str = ""
|
||||
level: int = 0 # heading level 1-3 OR list nest depth
|
||||
bbox: BBox = field(default_factory=BBox)
|
||||
spans: list[TextSpan] = field(default_factory=list)
|
||||
cells: list[list[str]] = field(default_factory=list) # table rows
|
||||
# Merged cell ranges as (r0, c0, r1, c1) inclusive, 0-based
|
||||
merges: list[tuple[int, int, int, int]] = field(default_factory=list)
|
||||
table_confidence: float = 0.0
|
||||
image_png: bytes | None = None
|
||||
reading_order: int = 0
|
||||
align: str = "left" # left|center|right|justify
|
||||
list_style: str = "" # bullet|number|""
|
||||
# "ocr" | "digital" | PAGE_RASTER_SOURCE | "" — drives Arabic
|
||||
# visual→logical handling and the faithful DOCX page-raster fallback.
|
||||
source: str = ""
|
||||
# True when this block's bbox was fabricated from plain text rather than
|
||||
# measured from the page. The numbers look real — evenly spaced lines of
|
||||
# uniform height — so any consumer that reasons about proximity must check
|
||||
# this first, or it will read a column of table cells as wrapped prose.
|
||||
synthetic_geometry: bool = False
|
||||
|
||||
def plain_text(self) -> str:
|
||||
if self.type == BlockType.table:
|
||||
return "\n".join("\t".join(row) for row in self.cells)
|
||||
if self.spans:
|
||||
return "".join(s.text for s in self.spans)
|
||||
return self.text
|
||||
|
||||
|
||||
@dataclass
|
||||
class Page:
|
||||
index: int
|
||||
width: float
|
||||
height: float
|
||||
kind: PageKind = PageKind.digital
|
||||
blocks: list[Block] = field(default_factory=list)
|
||||
|
||||
def all_text(self) -> str:
|
||||
parts = [b.plain_text() for b in self.blocks if b.plain_text().strip()]
|
||||
return "\n".join(parts)
|
||||
|
||||
|
||||
@dataclass
|
||||
class Document:
|
||||
pages: list[Page] = field(default_factory=list)
|
||||
warnings: list[str] = field(default_factory=list)
|
||||
meta: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
@property
|
||||
def page_count(self) -> int:
|
||||
return len(self.pages)
|
||||
|
||||
def all_text(self) -> str:
|
||||
return "\n\n".join(p.all_text() for p in self.pages)
|
||||
@@ -0,0 +1,70 @@
|
||||
"""IDM JSON serialization / debug dump."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from app.services.convert.idm.model import Block, Document, Page
|
||||
|
||||
|
||||
def block_to_dict(block: Block) -> dict[str, Any]:
|
||||
return {
|
||||
"type": block.type.value,
|
||||
"text": block.text,
|
||||
"level": block.level,
|
||||
"bbox": block.bbox.as_list(),
|
||||
"spans": [
|
||||
{
|
||||
"text": s.text,
|
||||
"font_name": s.font_name,
|
||||
"font_size": s.font_size,
|
||||
"bold": s.bold,
|
||||
"italic": s.italic,
|
||||
}
|
||||
for s in block.spans
|
||||
],
|
||||
"cells": block.cells,
|
||||
"merges": [list(m) for m in block.merges],
|
||||
"table_confidence": block.table_confidence,
|
||||
"reading_order": block.reading_order,
|
||||
"has_image": block.image_png is not None,
|
||||
}
|
||||
|
||||
|
||||
def page_to_dict(page: Page) -> dict[str, Any]:
|
||||
return {
|
||||
"index": page.index,
|
||||
"width": page.width,
|
||||
"height": page.height,
|
||||
"kind": page.kind.value,
|
||||
"blocks": [block_to_dict(b) for b in page.blocks],
|
||||
}
|
||||
|
||||
|
||||
def document_to_dict(doc: Document) -> dict[str, Any]:
|
||||
return {
|
||||
"version": "idm.v1",
|
||||
"page_count": doc.page_count,
|
||||
"warnings": list(doc.warnings),
|
||||
"meta": dict(doc.meta),
|
||||
"pages": [page_to_dict(p) for p in doc.pages],
|
||||
}
|
||||
|
||||
|
||||
def dumps_idm(doc: Document, *, indent: int = 2) -> str:
|
||||
return json.dumps(document_to_dict(doc), indent=indent, ensure_ascii=False)
|
||||
|
||||
|
||||
def maybe_dump_idm(doc: Document, stem: str = "convert") -> str | None:
|
||||
"""If CONVERT_DEBUG_IDM=1, write IDM JSON under temp/cwd and return path."""
|
||||
flag = os.environ.get("CONVERT_DEBUG_IDM", "").strip().lower()
|
||||
if flag not in ("1", "true", "yes"):
|
||||
return None
|
||||
out_dir = Path(os.environ.get("CONVERT_DEBUG_DIR", "."))
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
path = out_dir / f"{stem}-idm.json"
|
||||
path.write_text(dumps_idm(doc), encoding="utf-8")
|
||||
return str(path)
|
||||
@@ -0,0 +1,308 @@
|
||||
"""Persistent SQLite-backed conversion job store with WAL mode and spooling."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import sqlite3
|
||||
import threading
|
||||
import time
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from app.schemas.convert import JobStatus
|
||||
|
||||
from contextlib import contextmanager
|
||||
|
||||
_DEFAULT_SPOOL_THRESHOLD = 512 * 1024 # 512 KB
|
||||
|
||||
|
||||
class JobStore:
|
||||
"""Persistent SQLite job store with WAL mode, disk spooling, and crash recovery."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
ttl_seconds: int = 3 * 60 * 60,
|
||||
db_path: str | Path | None = None,
|
||||
spool_dir: str | Path | None = None,
|
||||
):
|
||||
self._lock = threading.RLock()
|
||||
self._ttl = ttl_seconds
|
||||
|
||||
base_data = Path(__file__).resolve().parents[3] / "data"
|
||||
if db_path is None:
|
||||
db_path = os.environ.get("CONVERT_JOBS_DB", str(base_data / "jobs.sqlite3"))
|
||||
if spool_dir is None:
|
||||
spool_dir = os.environ.get("CONVERT_JOBS_SPOOL", str(base_data / "jobs_spool"))
|
||||
|
||||
self._db_path = Path(db_path)
|
||||
self._spool_dir = Path(spool_dir)
|
||||
|
||||
self._db_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
self._spool_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
self._init_db()
|
||||
self._recover_crashed_jobs()
|
||||
|
||||
@contextmanager
|
||||
def _conn(self):
|
||||
conn = sqlite3.connect(
|
||||
str(self._db_path),
|
||||
timeout=30.0,
|
||||
check_same_thread=False,
|
||||
)
|
||||
conn.row_factory = sqlite3.Row
|
||||
try:
|
||||
with conn:
|
||||
yield conn
|
||||
finally:
|
||||
try:
|
||||
conn.close()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
def _init_db(self) -> None:
|
||||
with self._lock, self._conn() as conn:
|
||||
conn.execute("PRAGMA journal_mode=WAL;")
|
||||
conn.execute("PRAGMA synchronous=NORMAL;")
|
||||
conn.execute(
|
||||
"""
|
||||
CREATE TABLE IF NOT EXISTS jobs (
|
||||
job_id TEXT PRIMARY KEY,
|
||||
status TEXT NOT NULL,
|
||||
source TEXT NOT NULL,
|
||||
target TEXT NOT NULL,
|
||||
filename TEXT NOT NULL,
|
||||
fidelity TEXT,
|
||||
warnings TEXT,
|
||||
size_bytes INTEGER,
|
||||
error TEXT,
|
||||
result_path TEXT,
|
||||
result_blob BLOB,
|
||||
result_filename TEXT,
|
||||
download_path TEXT,
|
||||
created_at REAL NOT NULL,
|
||||
updated_at REAL NOT NULL,
|
||||
meta TEXT,
|
||||
cancel_requested INTEGER DEFAULT 0
|
||||
);
|
||||
"""
|
||||
)
|
||||
conn.execute("CREATE INDEX IF NOT EXISTS idx_jobs_created_at ON jobs (created_at);")
|
||||
conn.execute("CREATE INDEX IF NOT EXISTS idx_jobs_status ON jobs (status);")
|
||||
|
||||
def _recover_crashed_jobs(self) -> None:
|
||||
"""Mark any jobs left queued or running from a crashed server as failed."""
|
||||
with self._lock, self._conn() as conn:
|
||||
conn.execute(
|
||||
"""
|
||||
UPDATE jobs
|
||||
SET status = ?, error = ?, updated_at = ?
|
||||
WHERE status IN (?, ?)
|
||||
""",
|
||||
(
|
||||
JobStatus.failed.value,
|
||||
"Server process restarted during execution",
|
||||
time.time(),
|
||||
JobStatus.queued.value,
|
||||
JobStatus.running.value,
|
||||
),
|
||||
)
|
||||
|
||||
def create(self, source: str, target: str, filename: str) -> dict[str, Any]:
|
||||
self._purge_expired()
|
||||
job_id = str(uuid.uuid4())
|
||||
now = time.time()
|
||||
job = {
|
||||
"job_id": job_id,
|
||||
"status": JobStatus.queued.value,
|
||||
"source": source,
|
||||
"target": target,
|
||||
"filename": filename,
|
||||
"fidelity": None,
|
||||
"warnings": [],
|
||||
"size_bytes": None,
|
||||
"error": None,
|
||||
"result_bytes": None,
|
||||
"result_filename": None,
|
||||
"download_path": None,
|
||||
"created_at": now,
|
||||
"updated_at": now,
|
||||
"meta": {},
|
||||
"cancel_requested": False,
|
||||
}
|
||||
with self._lock, self._conn() as conn:
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO jobs (
|
||||
job_id, status, source, target, filename, fidelity, warnings,
|
||||
size_bytes, error, result_path, result_blob, result_filename,
|
||||
download_path, created_at, updated_at, meta, cancel_requested
|
||||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
""",
|
||||
(
|
||||
job_id,
|
||||
job["status"],
|
||||
source,
|
||||
target,
|
||||
filename,
|
||||
None,
|
||||
json.dumps([]),
|
||||
None,
|
||||
None,
|
||||
None,
|
||||
None,
|
||||
None,
|
||||
None,
|
||||
now,
|
||||
now,
|
||||
json.dumps({}),
|
||||
0,
|
||||
),
|
||||
)
|
||||
return job
|
||||
|
||||
def update(self, job_id: str, **fields: Any) -> dict[str, Any] | None:
|
||||
with self._lock:
|
||||
existing = self.get(job_id)
|
||||
if not existing:
|
||||
return None
|
||||
|
||||
now = time.time()
|
||||
set_clauses = ["updated_at = ?"]
|
||||
params: list[Any] = [now]
|
||||
|
||||
# Spool large result bytes to disk
|
||||
if "result_bytes" in fields:
|
||||
rb = fields["result_bytes"]
|
||||
if rb is not None and len(rb) > _DEFAULT_SPOOL_THRESHOLD:
|
||||
spool_file = self._spool_dir / f"{job_id}.bin"
|
||||
spool_file.write_bytes(rb)
|
||||
set_clauses.extend(["result_path = ?", "result_blob = ?"])
|
||||
params.extend([str(spool_file), None])
|
||||
elif rb is not None:
|
||||
set_clauses.extend(["result_blob = ?", "result_path = ?"])
|
||||
params.extend([rb, None])
|
||||
else:
|
||||
set_clauses.extend(["result_blob = ?", "result_path = ?"])
|
||||
params.extend([None, None])
|
||||
|
||||
for k, v in fields.items():
|
||||
if k in ("result_bytes", "updated_at"):
|
||||
continue
|
||||
if k == "warnings":
|
||||
set_clauses.append("warnings = ?")
|
||||
params.append(json.dumps(v if isinstance(v, list) else []))
|
||||
elif k == "meta":
|
||||
set_clauses.append("meta = ?")
|
||||
params.append(json.dumps(v if isinstance(v, dict) else {}))
|
||||
elif k == "cancel_requested":
|
||||
set_clauses.append("cancel_requested = ?")
|
||||
params.append(1 if v else 0)
|
||||
elif k in (
|
||||
"status",
|
||||
"fidelity",
|
||||
"size_bytes",
|
||||
"error",
|
||||
"result_filename",
|
||||
"download_path",
|
||||
):
|
||||
set_clauses.append(f"{k} = ?")
|
||||
params.append(v)
|
||||
|
||||
params.append(job_id)
|
||||
query = f"UPDATE jobs SET {', '.join(set_clauses)} WHERE job_id = ?"
|
||||
with self._conn() as conn:
|
||||
conn.execute(query, params)
|
||||
|
||||
return self.get(job_id)
|
||||
|
||||
def request_cancel(self, job_id: str) -> None:
|
||||
"""Soft-cancel: checked between pages when cancel_check is wired."""
|
||||
with self._lock, self._conn() as conn:
|
||||
conn.execute(
|
||||
"UPDATE jobs SET cancel_requested = 1, updated_at = ? WHERE job_id = ?",
|
||||
(time.time(), job_id),
|
||||
)
|
||||
|
||||
def is_cancelled(self, job_id: str) -> bool:
|
||||
with self._lock, self._conn() as conn:
|
||||
cur = conn.execute("SELECT cancel_requested FROM jobs WHERE job_id = ?", (job_id,))
|
||||
row = cur.fetchone()
|
||||
return bool(row and row["cancel_requested"])
|
||||
|
||||
def get(self, job_id: str) -> dict[str, Any] | None:
|
||||
self._purge_expired()
|
||||
with self._lock, self._conn() as conn:
|
||||
cur = conn.execute("SELECT * FROM jobs WHERE job_id = ?", (job_id,))
|
||||
row = cur.fetchone()
|
||||
if not row:
|
||||
return None
|
||||
return self._row_to_dict(row)
|
||||
|
||||
def get_result_bytes(self, job_id: str) -> bytes | None:
|
||||
with self._lock, self._conn() as conn:
|
||||
cur = conn.execute(
|
||||
"SELECT result_blob, result_path FROM jobs WHERE job_id = ?",
|
||||
(job_id,),
|
||||
)
|
||||
row = cur.fetchone()
|
||||
if not row:
|
||||
return None
|
||||
if row["result_path"] and Path(row["result_path"]).is_file():
|
||||
try:
|
||||
return Path(row["result_path"]).read_bytes()
|
||||
except OSError:
|
||||
return None
|
||||
return row["result_blob"]
|
||||
|
||||
def _row_to_dict(self, row: sqlite3.Row) -> dict[str, Any]:
|
||||
warnings = []
|
||||
if row["warnings"]:
|
||||
try:
|
||||
warnings = json.loads(row["warnings"])
|
||||
except Exception:
|
||||
warnings = []
|
||||
|
||||
meta = {}
|
||||
if row["meta"]:
|
||||
try:
|
||||
meta = json.loads(row["meta"])
|
||||
except Exception:
|
||||
meta = {}
|
||||
|
||||
return {
|
||||
"job_id": row["job_id"],
|
||||
"status": row["status"],
|
||||
"source": row["source"],
|
||||
"target": row["target"],
|
||||
"filename": row["filename"],
|
||||
"fidelity": row["fidelity"],
|
||||
"warnings": warnings,
|
||||
"size_bytes": row["size_bytes"],
|
||||
"error": row["error"],
|
||||
"result_filename": row["result_filename"],
|
||||
"download_path": row["download_path"],
|
||||
"created_at": row["created_at"],
|
||||
"updated_at": row["updated_at"],
|
||||
"meta": meta,
|
||||
"cancel_requested": bool(row["cancel_requested"]),
|
||||
}
|
||||
|
||||
def _purge_expired(self) -> None:
|
||||
now = time.time()
|
||||
cutoff = now - self._ttl
|
||||
with self._lock, self._conn() as conn:
|
||||
cur = conn.execute(
|
||||
"SELECT job_id, result_path FROM jobs WHERE created_at < ?",
|
||||
(cutoff,),
|
||||
)
|
||||
expired = cur.fetchall()
|
||||
for row in expired:
|
||||
if row["result_path"]:
|
||||
Path(row["result_path"]).unlink(missing_ok=True)
|
||||
conn.execute("DELETE FROM jobs WHERE created_at < ?", (cutoff,))
|
||||
|
||||
|
||||
job_store = JobStore()
|
||||
@@ -0,0 +1,5 @@
|
||||
"""Layout package."""
|
||||
|
||||
from app.services.convert.layout.pipeline import build_idm_from_pdf
|
||||
|
||||
__all__ = ["build_idm_from_pdf"]
|
||||
@@ -0,0 +1,186 @@
|
||||
"""Block typing: heading, list, paragraph."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
from app.services.convert.idm.model import BBox, Block, BlockType, TextSpan
|
||||
from app.services.convert.layout.glyphs import Line
|
||||
from app.services.convert.layout.styles import heading_level, is_caption
|
||||
|
||||
_BULLET_RE = re.compile(r"""^(\u2022|\-|\*|•|○|●|◦|·|▪|['"`])\s+""")
|
||||
_NUMBER_RE = re.compile(r"^(\d+[\.\)]|\(\d+\))\s+")
|
||||
_LETTER_RE = re.compile(r"^([A-Za-z][\.\)]|\([A-Za-z]\))\s+")
|
||||
# OCR often turns a bullet into a lone I / l before the sentence
|
||||
_OCR_BULLET_RE = re.compile(r"^[Il]\s+[A-ZÀ-ÖØ-Þ]")
|
||||
_URI_RE = re.compile(r"(https?://[^\s<>]+|www\.[^\s<>]+)", re.I)
|
||||
|
||||
|
||||
def _annotate_uris(spans: list[TextSpan]) -> list[TextSpan]:
|
||||
out: list[TextSpan] = []
|
||||
for s in spans:
|
||||
m = _URI_RE.search(s.text or "")
|
||||
if m and not s.url:
|
||||
out.append(
|
||||
TextSpan(
|
||||
text=s.text,
|
||||
font_name=s.font_name,
|
||||
font_size=s.font_size,
|
||||
bold=s.bold,
|
||||
italic=s.italic,
|
||||
x=s.x,
|
||||
w=s.w,
|
||||
url=m.group(1),
|
||||
)
|
||||
)
|
||||
else:
|
||||
out.append(s)
|
||||
return out
|
||||
|
||||
|
||||
def _line_is_bold(line: Line) -> bool:
|
||||
"""Whether essentially the whole line is bold, weighted by character count.
|
||||
|
||||
A part-bold line is emphasis inside prose -- "**Note:** the rest of the
|
||||
sentence" -- and must not read as a heading, so the test is share of
|
||||
characters rather than presence of any bold span.
|
||||
"""
|
||||
total = bold = 0
|
||||
for span in line.spans or []:
|
||||
n = len((span.text or "").strip())
|
||||
if not n:
|
||||
continue
|
||||
total += n
|
||||
if span.bold:
|
||||
bold += n
|
||||
return total > 0 and (bold / total) >= 0.8
|
||||
|
||||
|
||||
def line_to_block(line: Line, order: int, body_size: float) -> Block:
|
||||
text = line.text.strip()
|
||||
list_style = ""
|
||||
nest = 0
|
||||
# M2: prefer list detection over heading for bullet/number markers
|
||||
if _BULLET_RE.match(text) or _OCR_BULLET_RE.match(text):
|
||||
btype = BlockType.list_item
|
||||
list_style = "bullet"
|
||||
level = 0
|
||||
elif _NUMBER_RE.match(text):
|
||||
btype = BlockType.list_item
|
||||
list_style = "number"
|
||||
level = 0
|
||||
elif _LETTER_RE.match(text):
|
||||
btype = BlockType.list_item
|
||||
list_style = "letter"
|
||||
level = 0
|
||||
elif is_caption(text):
|
||||
btype = BlockType.paragraph
|
||||
level = 0
|
||||
else:
|
||||
level = heading_level(text, line.font_size, body_size, bold=_line_is_bold(line))
|
||||
if level:
|
||||
btype = BlockType.heading
|
||||
else:
|
||||
btype = BlockType.paragraph
|
||||
level = 0
|
||||
# Indent heuristic → shallow nest
|
||||
if btype == BlockType.list_item and line.x0 > 100:
|
||||
nest = 1
|
||||
spans = line.spans or [TextSpan(text=text, font_size=line.font_size, font_name=line.font_name)]
|
||||
spans = _annotate_uris(spans)
|
||||
return Block(
|
||||
type=btype,
|
||||
text=text,
|
||||
level=level or nest,
|
||||
bbox=BBox(x=line.x0, y=line.y, w=max(line.x1 - line.x0, 1), h=line.font_size),
|
||||
spans=spans,
|
||||
reading_order=order,
|
||||
list_style=list_style,
|
||||
align="left",
|
||||
synthetic_geometry=bool(getattr(line, "synthetic_geometry", False)),
|
||||
)
|
||||
|
||||
|
||||
def lines_to_block(
|
||||
lines: list[Line],
|
||||
order: int,
|
||||
body_size: float,
|
||||
*,
|
||||
lexical_pairs: set[tuple[str, str]] | None = None,
|
||||
) -> Block:
|
||||
"""Build one block from a paragraph run of wrapped lines.
|
||||
|
||||
Single-line runs delegate to :func:`line_to_block` so heading, caption and
|
||||
list detection behave identically to the per-line path. Multi-line runs are
|
||||
joined with hyphenation resolved, and carry the union bbox plus the
|
||||
concatenated spans so run-level styling survives into the formatters.
|
||||
"""
|
||||
if not lines:
|
||||
raise ValueError("lines_to_block requires at least one line")
|
||||
if len(lines) == 1:
|
||||
return line_to_block(lines[0], order, body_size)
|
||||
|
||||
from app.services.convert.layout.paragraphs import join_line_texts
|
||||
|
||||
text = join_line_texts([ln.text for ln in lines], lexical_pairs=lexical_pairs)
|
||||
first = lines[0]
|
||||
|
||||
spans: list[TextSpan] = []
|
||||
for idx, ln in enumerate(lines):
|
||||
line_spans = ln.spans or [
|
||||
TextSpan(text=ln.text, font_size=ln.font_size, font_name=ln.font_name)
|
||||
]
|
||||
if idx > 0 and spans and line_spans:
|
||||
# Wrapped lines are separated by a space unless the previous line
|
||||
# was hyphenated, which join_line_texts has already resolved.
|
||||
prev_text = spans[-1].text or ""
|
||||
if prev_text and not prev_text.endswith((" ", "-", "")):
|
||||
spans.append(
|
||||
TextSpan(
|
||||
text=" ",
|
||||
font_size=line_spans[0].font_size,
|
||||
font_name=line_spans[0].font_name,
|
||||
)
|
||||
)
|
||||
elif prev_text.endswith(("-", "")) and not text.count(prev_text[-1]):
|
||||
# join_line_texts removed the hyphen; mirror that in the spans.
|
||||
spans[-1] = TextSpan(
|
||||
text=prev_text[:-1],
|
||||
font_name=spans[-1].font_name,
|
||||
font_size=spans[-1].font_size,
|
||||
bold=spans[-1].bold,
|
||||
italic=spans[-1].italic,
|
||||
x=spans[-1].x,
|
||||
w=spans[-1].w,
|
||||
url=spans[-1].url,
|
||||
color=spans[-1].color,
|
||||
visual_order=spans[-1].visual_order,
|
||||
)
|
||||
spans.extend(line_spans)
|
||||
spans = _annotate_uris(spans)
|
||||
|
||||
x0 = min(ln.x0 for ln in lines)
|
||||
x1 = max(ln.x1 for ln in lines)
|
||||
y_top = max(ln.y for ln in lines)
|
||||
y_bot = min(ln.y for ln in lines)
|
||||
height = max(y_top - y_bot + (first.font_size or body_size), first.font_size or body_size)
|
||||
|
||||
return Block(
|
||||
type=BlockType.paragraph,
|
||||
text=text,
|
||||
level=0,
|
||||
bbox=BBox(x=x0, y=y_bot, w=max(x1 - x0, 1), h=height),
|
||||
spans=spans,
|
||||
reading_order=order,
|
||||
align="left",
|
||||
synthetic_geometry=any(
|
||||
getattr(ln, "synthetic_geometry", False) for ln in lines
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def estimate_body_size(lines: list[Line]) -> float:
|
||||
if not lines:
|
||||
return 12.0
|
||||
sizes = sorted(ln.font_size for ln in lines if ln.font_size > 0)
|
||||
return sizes[len(sizes) // 2]
|
||||
@@ -0,0 +1,62 @@
|
||||
"""PDF page points ↔ raster pixel coordinate transforms."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
from app.services.convert.idm.model import BBox
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PageGeom:
|
||||
page_width: float
|
||||
page_height: float
|
||||
dpi: float
|
||||
|
||||
@property
|
||||
def scale(self) -> float:
|
||||
return self.dpi / 72.0
|
||||
|
||||
@property
|
||||
def pixel_w(self) -> int:
|
||||
return max(1, int(round(self.page_width * self.scale)))
|
||||
|
||||
@property
|
||||
def pixel_h(self) -> int:
|
||||
return max(1, int(round(self.page_height * self.scale)))
|
||||
|
||||
|
||||
def pdf_to_pixel(x: float, y: float, geom: PageGeom) -> tuple[float, float]:
|
||||
"""PDF origin bottom-left → image origin top-left."""
|
||||
px = x * geom.scale
|
||||
py = (geom.page_height - y) * geom.scale
|
||||
return px, py
|
||||
|
||||
|
||||
def pixel_to_pdf(px: float, py: float, geom: PageGeom) -> tuple[float, float]:
|
||||
"""Image top-left → PDF bottom-left."""
|
||||
x = px / geom.scale
|
||||
y = geom.page_height - (py / geom.scale)
|
||||
return x, y
|
||||
|
||||
|
||||
def pixel_bbox_to_pdf(x0: float, y0: float, x1: float, y1: float, geom: PageGeom) -> BBox:
|
||||
"""Axis-aligned pixel box (top-left origin) → PDF BBox (x,y,w,h bottom-left)."""
|
||||
# corners
|
||||
corners = [
|
||||
pixel_to_pdf(x0, y0, geom),
|
||||
pixel_to_pdf(x1, y0, geom),
|
||||
pixel_to_pdf(x0, y1, geom),
|
||||
pixel_to_pdf(x1, y1, geom),
|
||||
]
|
||||
xs = [c[0] for c in corners]
|
||||
ys = [c[1] for c in corners]
|
||||
x_min, x_max = min(xs), max(xs)
|
||||
y_min, y_max = min(ys), max(ys)
|
||||
return BBox(x=x_min, y=y_min, w=max(0.0, x_max - x_min), h=max(0.0, y_max - y_min))
|
||||
|
||||
|
||||
def pdf_bbox_to_pixel(bbox: BBox, geom: PageGeom) -> tuple[float, float, float, float]:
|
||||
x0, y_top = pdf_to_pixel(bbox.x, bbox.y + bbox.h, geom)
|
||||
x1, y_bot = pdf_to_pixel(bbox.x + bbox.w, bbox.y, geom)
|
||||
return min(x0, x1), min(y_top, y_bot), max(x0, x1), max(y_top, y_bot)
|
||||
@@ -0,0 +1,293 @@
|
||||
"""Document-wide analysis (ABBYY ADRT / Adobe Extract patterns, in-house)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import Counter
|
||||
|
||||
from app.services.convert.idm.model import Block, BlockType, Document
|
||||
from app.services.convert.layout.headers_footers import tag_headers_footers
|
||||
from app.services.convert.layout.styles import (
|
||||
TERMINAL_PUNCT_RE,
|
||||
dominant_body_size,
|
||||
heading_level,
|
||||
section_number_level,
|
||||
)
|
||||
from app.services.convert.options import HeaderFooterMode, get_options
|
||||
|
||||
|
||||
def _norm(text: str) -> str:
|
||||
return " ".join((text or "").lower().split())
|
||||
|
||||
|
||||
def _untag_headers_footers(document: Document) -> None:
|
||||
for page in document.pages:
|
||||
for b in page.blocks:
|
||||
if b.type in (BlockType.header, BlockType.footer):
|
||||
b.type = BlockType.paragraph
|
||||
b.level = 0
|
||||
|
||||
|
||||
def _drop_header_footer_blocks(document: Document) -> None:
|
||||
for page in document.pages:
|
||||
page.blocks = [
|
||||
b for b in page.blocks if b.type not in (BlockType.header, BlockType.footer)
|
||||
]
|
||||
for i, b in enumerate(page.blocks):
|
||||
b.reading_order = i
|
||||
|
||||
|
||||
def _repeating_heading_once(document: Document) -> None:
|
||||
"""Adobe Extract: repeating headings are reported once — retag rest as header.
|
||||
|
||||
A banner promoted to H1 on every page produces an unusable outline.
|
||||
"""
|
||||
if document.page_count < 2:
|
||||
return
|
||||
heading_texts: list[str] = []
|
||||
for page in document.pages:
|
||||
for b in page.blocks:
|
||||
if b.type == BlockType.heading:
|
||||
t = _norm(b.plain_text())
|
||||
if t:
|
||||
heading_texts.append(t)
|
||||
if len(heading_texts) < 2:
|
||||
return
|
||||
counts = Counter(heading_texts)
|
||||
repeats = {t for t, n in counts.items() if n >= 2 and len(t) >= 8}
|
||||
if not repeats:
|
||||
return
|
||||
seen: set[str] = set()
|
||||
for page in document.pages:
|
||||
for b in page.blocks:
|
||||
if b.type != BlockType.heading:
|
||||
continue
|
||||
key = _norm(b.plain_text())
|
||||
if key not in repeats:
|
||||
continue
|
||||
if key in seen:
|
||||
b.type = BlockType.header
|
||||
b.level = 0
|
||||
else:
|
||||
seen.add(key)
|
||||
|
||||
|
||||
def _block_is_bold(block: Block) -> bool:
|
||||
"""Whether essentially the whole block is bold, weighted by character count.
|
||||
|
||||
Mirrors ``layout.blocks._line_is_bold`` so the two passes classify the same
|
||||
text the same way.
|
||||
"""
|
||||
total = bold = 0
|
||||
for span in block.spans or []:
|
||||
n = len((span.text or "").strip())
|
||||
if not n:
|
||||
continue
|
||||
total += n
|
||||
if span.bold:
|
||||
bold += n
|
||||
return total > 0 and (bold / total) >= 0.8
|
||||
|
||||
|
||||
def _block_color(block: Block) -> str | None:
|
||||
"""The colour this block is *set* in, weighted by characters, or None.
|
||||
|
||||
Weighted rather than "any coloured span" for the same reason
|
||||
``_block_is_bold`` is: one coloured word inside a black sentence is emphasis,
|
||||
not a coloured line. A block only counts as coloured when essentially all of
|
||||
it is, and all of it the same colour.
|
||||
"""
|
||||
counts: dict[str, int] = {}
|
||||
total = 0
|
||||
for span in block.spans or []:
|
||||
n = len((span.text or "").strip())
|
||||
if not n:
|
||||
continue
|
||||
total += n
|
||||
colour = getattr(span, "color", None)
|
||||
if colour:
|
||||
counts[str(colour).upper()] = counts.get(str(colour).upper(), 0) + n
|
||||
if not total or not counts:
|
||||
return None
|
||||
colour, n = max(counts.items(), key=lambda kv: kv[1])
|
||||
return colour if (n / total) >= 0.8 else None
|
||||
|
||||
|
||||
# A coloured line is only heading evidence when it is at least this tall relative
|
||||
# to body text. Word's default Heading 3 is 12pt on an 11pt body -- 1.09x -- so
|
||||
# the bar has to sit below that, and equal-sized coloured text is admitted too:
|
||||
# a coloured run-in heading at body size is a real convention. Size is not doing
|
||||
# the work here, colour is; this only rejects coloured *small print*.
|
||||
_COLOUR_HEADING_MIN_RATIO = 0.98
|
||||
# Longest a coloured line may be to still read as a heading. Deliberately tighter
|
||||
# than the size-based path's 14: colour is weaker evidence than a size jump, and
|
||||
# a coloured pull-quote or a warning paragraph is exactly what would otherwise be
|
||||
# swept up.
|
||||
_COLOUR_HEADING_MAX_WORDS = 12
|
||||
|
||||
|
||||
def _promote_coloured_headings(document: Document) -> None:
|
||||
"""Promote blocks whose colour sets them apart from the document's body text.
|
||||
|
||||
Word's built-in heading styles are *coloured, not bold*: in the default
|
||||
template Heading 1 is 365F91 and Headings 2 and 3 are 4F81BD, all on black
|
||||
body text. Heading 3 is 12pt on an 11pt body -- a ratio of 1.09, below the
|
||||
smallest size tier -- and it is not bold, so neither the size path nor the
|
||||
weight path in ``styles.heading_level`` can see it. Colour is the only
|
||||
evidence the document offers, and until now it was thrown away at extraction.
|
||||
The effect was that every Heading 3 in a default-styled Word document came
|
||||
back from PDF as ordinary body text, which silently flattens the outline that
|
||||
Word's navigation pane and any generated table of contents are built from.
|
||||
|
||||
The comparison is against the document's *own* dominant text colour, not
|
||||
against black. A document set entirely in dark blue has no colour signal at
|
||||
all, and treating every short line in it as a heading would be far worse than
|
||||
missing a few: this pass returns immediately in that case. That is also why
|
||||
the body colour is counted over the whole document rather than per page --
|
||||
a page that happens to contain nothing but headings would otherwise conclude
|
||||
that its heading colour *is* the body colour and promote nothing.
|
||||
"""
|
||||
body_counts: dict[str | None, int] = {}
|
||||
for page in document.pages:
|
||||
for b in page.blocks:
|
||||
if b.type not in (BlockType.paragraph, BlockType.heading):
|
||||
continue
|
||||
n = len((b.plain_text() or "").strip())
|
||||
if n:
|
||||
key = _block_color(b)
|
||||
body_counts[key] = body_counts.get(key, 0) + n
|
||||
if not body_counts:
|
||||
return
|
||||
body_colour = max(body_counts.items(), key=lambda kv: kv[1])[0]
|
||||
|
||||
sizes: list[float] = []
|
||||
for page in document.pages:
|
||||
for b in page.blocks:
|
||||
if b.spans:
|
||||
sizes.extend(s.font_size for s in b.spans if s.font_size and s.font_size > 0)
|
||||
body_size = dominant_body_size(sizes) if sizes else 12.0
|
||||
|
||||
for page in document.pages:
|
||||
for b in page.blocks:
|
||||
if b.type != BlockType.paragraph:
|
||||
continue
|
||||
colour = _block_color(b)
|
||||
if not colour or colour == body_colour:
|
||||
continue
|
||||
text = (b.plain_text() or "").strip()
|
||||
words = text.split()
|
||||
if not words or len(words) > _COLOUR_HEADING_MAX_WORDS:
|
||||
continue
|
||||
# A line that ends a sentence is prose, however it is coloured --
|
||||
# the same test the weight-based path uses, and for the same reason.
|
||||
if TERMINAL_PUNCT_RE.search(text):
|
||||
continue
|
||||
if _heading_size(b) < body_size * _COLOUR_HEADING_MIN_RATIO:
|
||||
continue
|
||||
b.type = BlockType.heading
|
||||
# Provisional; ``_rank_heading_levels`` assigns the real depth below
|
||||
# once every heading size in the document is known.
|
||||
b.level = 3
|
||||
|
||||
|
||||
def _style_catalog_demote(document: Document) -> None:
|
||||
"""DetectStyles: body-sized 'headings' are body once the doc median is known."""
|
||||
sizes: list[float] = []
|
||||
for page in document.pages:
|
||||
for b in page.blocks:
|
||||
if b.type in (BlockType.paragraph, BlockType.heading, BlockType.list_item):
|
||||
if b.spans:
|
||||
sizes.extend(s.font_size for s in b.spans if s.font_size and s.font_size > 0)
|
||||
body = dominant_body_size(sizes) if sizes else 12.0
|
||||
for page in document.pages:
|
||||
for b in page.blocks:
|
||||
if b.type != BlockType.heading:
|
||||
continue
|
||||
text = b.plain_text().strip()
|
||||
size = 12.0
|
||||
if b.spans:
|
||||
ss = [s.font_size for s in b.spans if s.font_size and s.font_size > 0]
|
||||
if ss:
|
||||
size = sum(ss) / len(ss)
|
||||
# Trust numbered sections; demote size-only H1 that matches body.
|
||||
# The bold signal must be passed here too: this pass re-runs the same
|
||||
# classifier, so without it a heading recognised on weight would be
|
||||
# demoted straight back to body text and the fix upstream would have
|
||||
# no visible effect.
|
||||
if heading_level(text, size, body, bold=_block_is_bold(b)) == 0:
|
||||
b.type = BlockType.paragraph
|
||||
b.level = 0
|
||||
|
||||
|
||||
def _heading_size(block: Block) -> float:
|
||||
if block.spans:
|
||||
ss = [s.font_size for s in block.spans if s.font_size and s.font_size > 0]
|
||||
if ss:
|
||||
return sum(ss) / len(ss)
|
||||
return 12.0
|
||||
|
||||
|
||||
# Heading sizes within this many points of each other are one tier. Half a point
|
||||
# of drift is normal between a heading and its own continuation line.
|
||||
_TIER_SNAP_PT = 0.75
|
||||
|
||||
|
||||
def _rank_heading_levels(document: Document) -> None:
|
||||
"""Assign heading levels by rank among the sizes this document actually uses.
|
||||
|
||||
``styles.heading_level`` scores each line in isolation against fixed multiples
|
||||
of body size, needing 1.55x for level 1. Word's own default template sets
|
||||
Heading 1/2/3 at 16/13/12pt on an 11pt body -- ratios of 1.45, 1.18 and 1.09 --
|
||||
so a PDF printed from a default-styled Word document came back with its H1
|
||||
demoted to Heading 2, its H2 to Heading 3 and its H3 dropped to body text.
|
||||
The document then has no level-1 heading at all, which empties Word's
|
||||
navigation pane and any generated table of contents. Documents set with
|
||||
dramatic contrast scored correctly, which is why the fixed tiers survived:
|
||||
they encode one house style rather than a general rule.
|
||||
|
||||
Rank is the general rule. Whatever contrast a document chose, its largest
|
||||
heading size is its top level and each smaller size is one level down -- which
|
||||
is how a reader infers an outline, and it needs no assumption about how much
|
||||
larger a heading "should" be. Levels from explicit section numbers are left
|
||||
alone: ``1.2.3`` states its own depth, and that beats any inference from size.
|
||||
"""
|
||||
ranked: list[tuple[Block, float]] = []
|
||||
for page in document.pages:
|
||||
for b in page.blocks:
|
||||
if b.type != BlockType.heading:
|
||||
continue
|
||||
if section_number_level(b.plain_text().strip()):
|
||||
continue # authoritative depth, do not override
|
||||
ranked.append((b, round(_heading_size(b) * 2) / 2.0))
|
||||
if not ranked:
|
||||
return
|
||||
|
||||
tiers: list[float] = []
|
||||
for size in sorted({s for _b, s in ranked}, reverse=True):
|
||||
if not tiers or (tiers[-1] - size) > _TIER_SNAP_PT:
|
||||
tiers.append(size)
|
||||
for block, size in ranked:
|
||||
rank = min(range(len(tiers)), key=lambda i: abs(tiers[i] - size))
|
||||
block.level = min(rank + 1, 3)
|
||||
|
||||
|
||||
def apply_document_wide(document: Document) -> None:
|
||||
"""Cluster HF, style catalog, repeating-heading-once. Honour header_footer_mode."""
|
||||
opts = get_options()
|
||||
mode = opts.header_footer_mode
|
||||
|
||||
if mode == HeaderFooterMode.ignore:
|
||||
_untag_headers_footers(document)
|
||||
else:
|
||||
tag_headers_footers(document)
|
||||
if mode == HeaderFooterMode.remove:
|
||||
_drop_header_footer_blocks(document)
|
||||
|
||||
_style_catalog_demote(document)
|
||||
# After the demote, not before: the demote pass re-runs the size/weight
|
||||
# classifier and would take a colour-only heading straight back out again.
|
||||
# Before the ranking, so the promoted blocks are ranked with the rest.
|
||||
_promote_coloured_headings(document)
|
||||
_rank_heading_levels(document)
|
||||
_repeating_heading_once(document)
|
||||
if mode == HeaderFooterMode.remove:
|
||||
_drop_header_footer_blocks(document)
|
||||
@@ -0,0 +1,604 @@
|
||||
"""Glyph / run clustering into lines."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
from app.services.convert.idm.model import TextSpan
|
||||
from app.services.convert.layout.styles import infer_font_flags, style_key
|
||||
from app.services.convert.text.arabic_logical import is_rtl_dominant
|
||||
|
||||
# How many *consecutive* text rows a band must stay blank down before it counts
|
||||
# as a gutter. Five is enough to rule out a stretched space in justified text
|
||||
# and short enough that a full-width figure dropped between two columns still
|
||||
# leaves a usable run of rows above and below itself.
|
||||
MIN_GUTTER_ROWS = 5
|
||||
|
||||
# A PDF text walker can place a continuation glyph a fraction of a point on
|
||||
# the far side of a detected gutter. This is common when a zero-width space
|
||||
# marker and its following character are emitted with separate text matrices:
|
||||
# the marker/character pair is one word, not a new page column. Keep the
|
||||
# tolerance below ordinary inter-column spacing so genuine columns remain
|
||||
# isolated.
|
||||
GUTTER_EDGE_CONTINUATION_PT = 2.0
|
||||
|
||||
|
||||
@dataclass
|
||||
class Line:
|
||||
y: float
|
||||
x0: float
|
||||
x1: float
|
||||
text: str
|
||||
font_size: float = 12.0
|
||||
font_name: str = ""
|
||||
spans: list[TextSpan] = field(default_factory=list)
|
||||
# True when x/y were fabricated by lines_from_plain_text rather than measured
|
||||
# from glyph geometry. Consumers must not draw geometric conclusions
|
||||
# (columns, indents, line width) from a line carrying this flag.
|
||||
synthetic_geometry: bool = False
|
||||
|
||||
|
||||
def _span_from_glyph(g: dict, default_size: float) -> TextSpan | None:
|
||||
t = str(g.get("text", ""))
|
||||
if not t:
|
||||
return None
|
||||
w = float(g.get("w", 0) or 0)
|
||||
# PDF producers commonly encode a word space as a zero-width glyph whose
|
||||
# advance is carried by the following text matrix. Dropping that marker
|
||||
# forces the line builder to infer every boundary from tiny x gaps; in
|
||||
# tightly set fonts those gaps are indistinguishable from kerning and
|
||||
# ordinary words get welded together ("Tracebased" / "Justintime").
|
||||
# Keep the marker with zero width so the explicit source signal wins.
|
||||
fname = str(g.get("fontName") or g.get("font_name") or "")
|
||||
h = float(g.get("h", 0) or 0)
|
||||
raw_size = float(g.get("fontSize") or 0)
|
||||
if raw_size <= 2.0 and h > 2.0:
|
||||
fsize = h
|
||||
elif raw_size > 0:
|
||||
fsize = raw_size
|
||||
elif h > 0:
|
||||
fsize = h
|
||||
else:
|
||||
fsize = default_size
|
||||
bold = bool(g.get("bold")) if "bold" in g else False
|
||||
italic = bool(g.get("italic")) if "italic" in g else False
|
||||
if not bold and not italic:
|
||||
ib, ii = infer_font_flags(fname)
|
||||
bold = bold or ib
|
||||
italic = italic or ii
|
||||
return TextSpan(
|
||||
text=t,
|
||||
font_name=fname,
|
||||
font_size=fsize,
|
||||
bold=bold,
|
||||
italic=italic,
|
||||
x=float(g.get("x", 0)),
|
||||
w=w or len(t) * 6,
|
||||
color=(str(g["color"]) if g.get("color") else None),
|
||||
)
|
||||
|
||||
|
||||
def _runs_on(cur: TextSpan, nxt: TextSpan) -> bool:
|
||||
"""Whether *nxt* continues *cur* without an intervening gap.
|
||||
|
||||
A merged span keeps the first span's ``x`` and the *sum* of the widths, so
|
||||
merging across a gap yields a span that claims to start where the first one
|
||||
did and to be far narrower than the ground it covers. Every consumer that
|
||||
asks "which column does this text sit in?" is then told the wrong answer —
|
||||
which is how a table row's six cells collapsed into its first one.
|
||||
"""
|
||||
if cur.x is None or nxt.x is None:
|
||||
return True
|
||||
gap = float(nxt.x) - (float(cur.x) + float(cur.w or 0.0))
|
||||
if gap <= 0:
|
||||
return True
|
||||
# An inter-word space runs about a third of an em. Anything appreciably
|
||||
# wider is column spacing, a tab stop or a dot leader.
|
||||
return gap <= max(2.0, float(cur.font_size or 12.0) * 0.8)
|
||||
|
||||
|
||||
def coalesce_spans(spans: list[TextSpan]) -> list[TextSpan]:
|
||||
"""Merge adjacent spans that share a style *and* run on without a gap."""
|
||||
if not spans:
|
||||
return []
|
||||
out: list[TextSpan] = []
|
||||
cur = TextSpan(
|
||||
text=spans[0].text,
|
||||
font_name=spans[0].font_name,
|
||||
font_size=spans[0].font_size,
|
||||
bold=spans[0].bold,
|
||||
italic=spans[0].italic,
|
||||
x=spans[0].x,
|
||||
w=spans[0].w,
|
||||
color=spans[0].color,
|
||||
)
|
||||
cur_right = float(cur.x or 0) + float(cur.w or 0)
|
||||
for s in spans[1:]:
|
||||
sx = float(s.x or 0)
|
||||
sw = float(s.w or 0)
|
||||
gap = sx - cur_right
|
||||
same_style = (
|
||||
(
|
||||
style_key(cur.font_name, cur.font_size, cur.bold, cur.italic)
|
||||
== style_key(s.font_name, s.font_size, s.bold, s.italic)
|
||||
or (
|
||||
cur.font_name.lower().strip() == s.font_name.lower().strip()
|
||||
and cur.bold == s.bold
|
||||
and cur.italic == s.italic
|
||||
and abs(float(cur.font_size or 0) - float(s.font_size or 0)) <= 2.0
|
||||
)
|
||||
)
|
||||
and cur.color == s.color
|
||||
)
|
||||
max_gap = max(2.0, float(cur.font_size or 12.0) * 0.8)
|
||||
if same_style and (gap <= max_gap or gap <= 0):
|
||||
cur.text += s.text
|
||||
cur_right = max(cur_right, sx + sw)
|
||||
if cur.x is not None:
|
||||
cur.w = cur_right - float(cur.x)
|
||||
elif s.w is not None:
|
||||
cur.w = s.w
|
||||
else:
|
||||
out.append(cur)
|
||||
cur = TextSpan(
|
||||
text=s.text,
|
||||
font_name=s.font_name,
|
||||
font_size=s.font_size,
|
||||
bold=s.bold,
|
||||
italic=s.italic,
|
||||
x=s.x,
|
||||
w=s.w,
|
||||
color=s.color,
|
||||
)
|
||||
cur_right = sx + sw
|
||||
out.append(cur)
|
||||
return out
|
||||
|
||||
|
||||
def page_gutters(
|
||||
glyphs: list[dict], *, min_width: float = 12.0, row_height: float = 12.0
|
||||
) -> list[tuple[float, float, float]]:
|
||||
"""Vertical bands almost no row of the page writes in — its column gutters.
|
||||
|
||||
A projection profile over the whole page, not a per-line gap test. The
|
||||
gutter of a two-column paper is often only 20pt wide, so a fixed
|
||||
"gap > 36pt means two columns" rule welds both columns of every academic
|
||||
paper into single lines; and a *per-line* threshold cannot tell a stretched
|
||||
space in justified text from a real gutter, because on one line they look
|
||||
identical. Across the page they do not: a gutter is empty on nearly every
|
||||
row, a stretched space is not.
|
||||
|
||||
The profile is *local*, computed over a sliding window of neighbouring
|
||||
rows rather than the whole page. A page is rarely one thing throughout: a
|
||||
paper puts a full-width title and author block above two columns of
|
||||
abstract, and a full-width figure in the middle of them. A page-wide
|
||||
profile finds no gutter on such a page at all, because the title crosses
|
||||
every candidate band — and then both columns of the abstract come out
|
||||
interleaved line by line.
|
||||
|
||||
Returns ``(x, y_low, y_high)`` for each gutter band, so a caller can apply
|
||||
a gutter only to the rows it actually separates.
|
||||
"""
|
||||
if len(glyphs) < 12:
|
||||
return []
|
||||
xs0 = [float(g.get("x", 0.0)) for g in glyphs]
|
||||
xs1 = [x + float(g.get("w", 0.0) or 0.0) for x, g in zip(xs0, glyphs, strict=False)]
|
||||
left, right = min(xs0), max(xs1)
|
||||
span = right - left
|
||||
if span <= 0:
|
||||
return []
|
||||
|
||||
step = max(row_height, 1.0)
|
||||
bin_pt = 2.0
|
||||
bins = max(8, min(2048, int(span / bin_pt) + 1))
|
||||
per_bin = span / bins
|
||||
need = max(2, int(min_width / per_bin))
|
||||
|
||||
# Cluster glyphs into baseline rows so descenders (p, g, y) do not fall
|
||||
# into a separate row and create false gutter detections.
|
||||
ordered_y = sorted(glyphs, key=lambda g: -float(g.get("y", 0.0)))
|
||||
row_clusters: list[list[dict]] = []
|
||||
row_ys: list[float] = []
|
||||
cluster_tol = max(step * 0.7, 4.0)
|
||||
for g in ordered_y:
|
||||
gy = float(g.get("y", 0.0))
|
||||
placed = False
|
||||
for idx in range(len(row_clusters) - 1, -1, -1):
|
||||
if abs(gy - row_ys[idx]) <= cluster_tol:
|
||||
row_clusters[idx].append(g)
|
||||
placed = True
|
||||
break
|
||||
if row_ys[idx] - gy > cluster_tol:
|
||||
break
|
||||
if not placed:
|
||||
row_clusters.append([g])
|
||||
row_ys.append(gy)
|
||||
|
||||
# One occupancy bitmap per text row, top of page first.
|
||||
occ: dict[int, int] = {}
|
||||
for cl, ry in zip(row_clusters, row_ys, strict=False):
|
||||
row_key = int(round(ry / step))
|
||||
bits = 0
|
||||
for g in cl:
|
||||
gx0 = float(g.get("x", 0.0))
|
||||
gw = float(g.get("w", 0.0) or 0.0)
|
||||
i = max(0, min(bins - 1, int((gx0 - left) / span * bins)))
|
||||
j = max(0, min(bins - 1, int((gx0 + gw - left) / span * bins)))
|
||||
bits |= (((1 << (j - i + 1)) - 1) << i)
|
||||
occ[row_key] = occ.get(row_key, 0) | bits
|
||||
|
||||
order = sorted(occ, reverse=True)
|
||||
if len(order) < MIN_GUTTER_ROWS:
|
||||
return []
|
||||
|
||||
full = (1 << bins) - 1
|
||||
# Bins each row leaves blank, top of page first.
|
||||
blank = [full ^ occ[r] for r in order]
|
||||
|
||||
# A column must be substantially wider than the gutter beside it. Without
|
||||
# this, any persistently blank band qualifies -- and a form is full of them:
|
||||
# the numbered stub down the left of an IRS 1040 leaves a real, unbroken gap
|
||||
# between the item number and its label, so "24 Add lines 22 and 23" was
|
||||
# split into "24" and "Add lines 22 and 23", detaching every line number
|
||||
# from the line it numbers. A 10pt-wide column of two-digit numbers is a
|
||||
# stub, not a text column.
|
||||
min_side = max(min_width * 3.0, span * 0.12)
|
||||
|
||||
out: list[tuple[float, float, float]] = []
|
||||
# A gutter is a band that stays blank down a *run* of consecutive rows.
|
||||
#
|
||||
# This used to be a window centred on each row that required unanimity,
|
||||
# which inverted the intent described above: the three rows below a
|
||||
# full-width title, author block or figure caption still had that full-width
|
||||
# row inside their window, so the gutter was suppressed for exactly the
|
||||
# first rows of the column region -- the rows where the columns begin.
|
||||
# Measured on the TraceMonkey paper, the real x=305 gutter was found only
|
||||
# from y=421 down while the two-column abstract starts at y=445, so its
|
||||
# first rows were never split, both columns welded into single lines, and
|
||||
# "...more difficult to com-" was joined to "applications such as Google
|
||||
# Mail..." to yield "Zimbra Colpile". Recall cannot see that -- every word is
|
||||
# still present, in the wrong order and partly re-spelled -- which is how it
|
||||
# survived.
|
||||
#
|
||||
# Runs carry the same evidence without the boundary artefact: a band is a
|
||||
# gutter at every row of any run of MIN_GUTTER_ROWS consecutive rows it
|
||||
# stays blank down. A stretched space in justified text is blank on one or
|
||||
# two rows and still cannot reach the run length, so the false positive the
|
||||
# projection profile exists to avoid is still avoided; and a full-width
|
||||
# figure between two columns now ends one run and starts another instead of
|
||||
# erasing three rows on each side of itself.
|
||||
for s in range(len(order) - MIN_GUTTER_ROWS + 1):
|
||||
rows = order[s : s + MIN_GUTTER_ROWS]
|
||||
empty = full
|
||||
union = 0
|
||||
for i in range(s, s + MIN_GUTTER_ROWS):
|
||||
empty &= blank[i]
|
||||
union |= occ[order[i]]
|
||||
if not empty or not union:
|
||||
continue
|
||||
bits = bin(empty)[2:].zfill(bins)[::-1]
|
||||
# The ground the run's own rows cover. Judged over the run rather than a
|
||||
# neighbourhood, because the question is what *this* band separates: a
|
||||
# row that writes straight through the band is not one of its columns.
|
||||
used = bin(union)[2:].zfill(bins)[::-1]
|
||||
first, last = used.find("1"), used.rfind("1")
|
||||
run = 0
|
||||
for k in range(bins + 1):
|
||||
if k < bins and bits[k] == "1":
|
||||
run += 1
|
||||
continue
|
||||
if run >= need:
|
||||
start = k - run
|
||||
# Interior only: a gutter has content on both sides of it, and
|
||||
# enough of it on each side to be a column.
|
||||
lo_end, hi_start = used.rfind("1", 0, start), used.find("1", k)
|
||||
if lo_end >= 0 and hi_start >= 0:
|
||||
left_w = (lo_end - first + 1) * per_bin
|
||||
right_w = (last - hi_start + 1) * per_bin
|
||||
if left_w >= min_side and right_w >= min_side:
|
||||
x = left + (start + k) / 2.0 * per_bin
|
||||
out.extend((x, r * step, r * step) for r in rows)
|
||||
run = 0
|
||||
|
||||
return _merge_gutter_spans(out, step)
|
||||
|
||||
|
||||
def _merge_gutter_spans(
|
||||
spans: list[tuple[float, float, float]], step: float
|
||||
) -> list[tuple[float, float, float]]:
|
||||
"""Collapse per-row gutter hits into ``(x, y_low, y_high)`` bands."""
|
||||
if not spans:
|
||||
return []
|
||||
merged: list[list[float]] = []
|
||||
for x, y0, y1 in sorted(spans, key=lambda s: (round(s[0], 0), s[1])):
|
||||
for band in merged:
|
||||
if abs(band[0] - x) <= 4.0 and y0 <= band[2] + step * 2.5 and y1 >= band[1] - step * 2.5:
|
||||
band[0] = (band[0] + x) / 2.0
|
||||
band[1] = min(band[1], y0)
|
||||
band[2] = max(band[2], y1)
|
||||
break
|
||||
else:
|
||||
merged.append([x, y0, y1])
|
||||
return [(b[0], b[1], b[2]) for b in merged]
|
||||
|
||||
|
||||
def _split_group_at(
|
||||
group: list[dict], gutters: list[tuple[float, float, float]], *, slack: float = 6.0
|
||||
) -> list[list[dict]]:
|
||||
"""Split one y-group wherever it straddles a gutter that spans its rows."""
|
||||
if not gutters or len(group) < 2:
|
||||
return [group]
|
||||
ys = [float(g.get("y", 0.0)) for g in group]
|
||||
y = sum(ys) / len(ys)
|
||||
here = sorted(x for x, lo, hi in gutters if lo - slack <= y <= hi + slack)
|
||||
if not here:
|
||||
return [group]
|
||||
group = sorted(group, key=lambda g: float(g.get("x", 0)))
|
||||
parts: list[list[dict]] = []
|
||||
current: list[dict] = []
|
||||
idx = 0
|
||||
for g in group:
|
||||
x = float(g.get("x", 0))
|
||||
cur_right = max(
|
||||
(float(c.get("x", 0)) + float(c.get("w", 0) or 0) for c in current),
|
||||
default=0.0,
|
||||
)
|
||||
while idx < len(here) and x >= here[idx]:
|
||||
# Only split if there is an actual gap separating columns across the gutter
|
||||
if current and (x - cur_right >= 4.0) and (cur_right <= here[idx] + 2.0):
|
||||
parts.append(current)
|
||||
current = []
|
||||
idx += 1
|
||||
current.append(g)
|
||||
if current:
|
||||
parts.append(current)
|
||||
return parts or [group]
|
||||
|
||||
|
||||
def _split_group_on_gutter(group: list[dict], *, min_gap: float = 36.0) -> list[list[dict]]:
|
||||
"""Recursively split a y-clustered glyph group on large horizontal gutters (2+ columns)."""
|
||||
if len(group) < 2:
|
||||
return [group]
|
||||
group = sorted(group, key=lambda g: float(g.get("x", 0)))
|
||||
best_i = -1
|
||||
best_gap = 0.0
|
||||
for i in range(len(group) - 1):
|
||||
x0 = float(group[i].get("x", 0))
|
||||
w0 = float(group[i].get("w", 0) or 0)
|
||||
x1 = float(group[i + 1].get("x", 0))
|
||||
gap = x1 - (x0 + w0)
|
||||
if gap > best_gap:
|
||||
best_gap = gap
|
||||
best_i = i
|
||||
if best_i < 0 or best_gap < min_gap:
|
||||
return [group]
|
||||
left, right = group[: best_i + 1], group[best_i + 1 :]
|
||||
if not left or not right:
|
||||
return [group]
|
||||
out: list[list[dict]] = []
|
||||
out.extend(_split_group_on_gutter(left, min_gap=min_gap))
|
||||
out.extend(_split_group_on_gutter(right, min_gap=min_gap))
|
||||
return out
|
||||
|
||||
|
||||
def _group_is_rtl(group: list[dict], *, threshold: float = 0.5) -> bool:
|
||||
"""True when a line's letters are predominantly right-to-left."""
|
||||
text = "".join(str(g.get("text", "")) for g in group)
|
||||
return is_rtl_dominant(text, threshold=threshold)
|
||||
|
||||
|
||||
def _sort_group_by_reading_order(group: list[dict]) -> list[dict]:
|
||||
"""Order a line's glyphs in logical reading order for its dominant script."""
|
||||
if _group_is_rtl(group):
|
||||
# Right-to-left: the logical first token sits furthest right. Sort by
|
||||
# the right edge descending so a wide glyph does not overtake a narrow
|
||||
# neighbour that starts further right.
|
||||
return sorted(
|
||||
group,
|
||||
key=lambda g: -(float(g.get("x", 0)) + float(g.get("w", 0) or 0)),
|
||||
)
|
||||
return sorted(group, key=lambda g: float(g.get("x", 0)))
|
||||
|
||||
|
||||
def _column_band(
|
||||
g: dict,
|
||||
gutters: list[tuple[float, float, float]],
|
||||
*,
|
||||
slack: float = 6.0,
|
||||
at_y: float | None = None,
|
||||
) -> tuple[int, int]:
|
||||
"""Which column *g* sits in: whether gutters are active, and its column side index."""
|
||||
if not gutters:
|
||||
return (0, 0)
|
||||
y = at_y if at_y is not None else float(g.get("y", 0.0))
|
||||
active = tuple(sorted(gx for gx, lo, hi in gutters if lo - slack <= y <= hi + slack))
|
||||
if not active:
|
||||
return (0, 0)
|
||||
x = float(g.get("x", 0.0))
|
||||
return (
|
||||
1,
|
||||
sum(1 for gx in active if x >= gx + GUTTER_EDGE_CONTINUATION_PT),
|
||||
)
|
||||
|
||||
|
||||
def lines_from_glyphs(glyphs: list[dict]) -> list[Line]:
|
||||
"""Cluster glyph dicts {text,x,y,w,h,fontSize?,fontName?} into reading lines."""
|
||||
# Keep zero-width whitespace glyphs: they are real word-boundary markers
|
||||
# in PDFs that encode spacing through text-matrix advances.
|
||||
clean_glyphs = [g for g in glyphs if str(g.get("text", ""))]
|
||||
if not clean_glyphs:
|
||||
return []
|
||||
sizes = [float(g.get("fontSize") or g.get("h") or 12.0) for g in clean_glyphs]
|
||||
median = sorted(sizes)[len(sizes) // 2] if sizes else 12.0
|
||||
# Generous tol so descenders (g,p,y) stay on the same line as the baseline
|
||||
tol = max(median * 0.7, 4.0)
|
||||
# 12pt is wider than the gutter of the most densely set two-column pages --
|
||||
# the Federal Register CFR pages in the corpus leave only 9pt -- so those
|
||||
# still weld their columns together. Lowering the floor to about one em does
|
||||
# fix them, and the two tests in ``page_gutters`` are strong enough to keep
|
||||
# an inter-word space from passing at that width. It is not landed because it
|
||||
# is not yet *safe*: splitting the IRS 1040 address block into more lines
|
||||
# makes something downstream of table reconstruction drop about fifteen
|
||||
# words, both "Last name" labels among them, and losing a label from a tax
|
||||
# form is worse than leaving a known welding case unfixed. Raise this only
|
||||
# together with a fix for that.
|
||||
gutters = page_gutters(
|
||||
clean_glyphs, min_width=max(12.0, median * 1.4), row_height=max(median * 0.9, 4.0)
|
||||
)
|
||||
|
||||
# Baselines, descending. Two costs used to hide here: the mean of every
|
||||
# bucket was recomputed for every glyph, and every bucket was tried even
|
||||
# though the input is sorted. Carrying a running total and stopping once a
|
||||
# bucket's baseline is out of reach turns a quadratic scan — 1.6M
|
||||
# arithmetic operations on four dense pages — into a linear one.
|
||||
ordered = sorted(clean_glyphs, key=lambda g: -float(g.get("y", 0)))
|
||||
buckets: list[list[dict]] = []
|
||||
totals: list[float] = []
|
||||
# One band per bucket, so a bucket's running baseline is never an average of
|
||||
# two columns. See ``_column_band`` for what goes wrong without this.
|
||||
bands: list[tuple[int, int]] = []
|
||||
for g in ordered:
|
||||
y = float(g.get("y", 0))
|
||||
placed = False
|
||||
for idx in range(len(buckets) - 1, -1, -1):
|
||||
by = totals[idx] / len(buckets[idx])
|
||||
if abs(y - by) <= tol:
|
||||
band = _column_band(g, gutters, at_y=by)
|
||||
if bands[idx] != band:
|
||||
# Another column's row at this height. Keep looking for one
|
||||
# of *this* column's rows; do not join and do not stop, or a
|
||||
# glyph would be pulled into the wrong column.
|
||||
continue
|
||||
buckets[idx].append(g)
|
||||
totals[idx] += y
|
||||
placed = True
|
||||
break
|
||||
if by - y > tol:
|
||||
# Buckets before this one sit higher still; none can match.
|
||||
break
|
||||
if not placed:
|
||||
buckets.append([g])
|
||||
totals.append(y)
|
||||
bands.append(_column_band(g, gutters, at_y=y))
|
||||
|
||||
lines: list[Line] = []
|
||||
for group in buckets:
|
||||
for band in _split_group_at(group, gutters):
|
||||
for subgroup in _split_group_on_gutter(band):
|
||||
# Reading order within a line follows the script, not the page.
|
||||
# Sorting a right-to-left line by ascending x yields *visual* order,
|
||||
# which reverses the words: "مشروع منصة هويتي" comes out as
|
||||
# "هويتي منصة مشروع". Ordering by descending x recovers logical
|
||||
# order at the source, which is far more reliable than trying to
|
||||
# detect and undo the reversal downstream.
|
||||
rtl_line = _group_is_rtl(subgroup)
|
||||
subgroup = _sort_group_by_reading_order(subgroup)
|
||||
spans: list[TextSpan] = []
|
||||
parts: list[str] = []
|
||||
prev_x1: float | None = None
|
||||
for g in subgroup:
|
||||
span = _span_from_glyph(g, median)
|
||||
if span is None:
|
||||
continue
|
||||
x0 = float(g.get("x", 0))
|
||||
# Insert inter-word space when engine omits space glyphs.
|
||||
# Conservative: avoid splitting glued tokens like Apple10 / LeftCol1A.
|
||||
if (
|
||||
prev_x1 is not None
|
||||
and parts
|
||||
and not parts[-1].endswith((" ", "\t"))
|
||||
and not (span.text or "").startswith((" ", "\t"))
|
||||
):
|
||||
# In a right-to-left line the next glyph lies to the
|
||||
# *left*, so the inter-glyph gap runs the other way.
|
||||
gap = (prev_x1 - (x0 + float(g.get("w", 0) or 0))) if rtl_line else (x0 - prev_x1)
|
||||
prev_ch = parts[-1][-1]
|
||||
next_ch = (span.text or " ")[0]
|
||||
glued_alnum = prev_ch.isalnum() and next_ch.isalnum()
|
||||
# The gap is font-relative. Where the extractor reports
|
||||
# the font's own space advance for this text state, use
|
||||
# it: typeset text positions words with TJ offsets
|
||||
# rather than space glyphs, and a fraction-of-median
|
||||
# rule welds them together — "Trace-based Just-in-Time"
|
||||
# arrived as one token at 9pt. Fall back to the median
|
||||
# heuristic when no space width is reported.
|
||||
# A reported space width is trusted only inside the
|
||||
# band real fonts occupy — roughly 0.15em to 0.7em.
|
||||
# Outside it the number is in the wrong unit (text
|
||||
# space rather than page space) or the font lies, and
|
||||
# believing it either welds words together or, worse,
|
||||
# splits every letter of a word onto its own token.
|
||||
size = float(g.get("fontSize") or g.get("h") or median or 12.0)
|
||||
space_w = float(g.get("spaceWidth") or 0.0)
|
||||
lo, hi = size * 0.15, size * 0.70
|
||||
if lo <= space_w <= hi:
|
||||
min_word_gap = space_w * 0.55
|
||||
else:
|
||||
min_word_gap = max(3.0, median * 0.40)
|
||||
if glued_alnum:
|
||||
min_word_gap = max(min_word_gap, median * 0.55)
|
||||
if gap > min_word_gap:
|
||||
parts.append(" ")
|
||||
spans.append(
|
||||
TextSpan(
|
||||
text=" ",
|
||||
font_size=span.font_size,
|
||||
font_name=span.font_name,
|
||||
x=prev_x1,
|
||||
)
|
||||
)
|
||||
parts.append(span.text)
|
||||
spans.append(span)
|
||||
w = float(g.get("w", 0) or 0)
|
||||
# Prefer reported width; floor so under-reported w does not inflate next gap
|
||||
advance = w if w > 0.5 else max(len(span.text) * median * 0.45, median * 0.35)
|
||||
# The "trailing edge" in reading order is the left edge for RTL.
|
||||
prev_x1 = x0 if rtl_line else x0 + advance
|
||||
text = "".join(parts).strip()
|
||||
if not text:
|
||||
continue
|
||||
spans = coalesce_spans(spans)
|
||||
xs = [float(g.get("x", 0)) for g in subgroup]
|
||||
ws = [float(g.get("w", 0)) for g in subgroup]
|
||||
ys = [float(g.get("y", 0)) for g in subgroup]
|
||||
first = spans[0] if spans else None
|
||||
lines.append(
|
||||
Line(
|
||||
y=sum(ys) / len(ys),
|
||||
x0=min(xs),
|
||||
x1=max(x + w for x, w in zip(xs, ws, strict=False)),
|
||||
text=text,
|
||||
font_size=float(first.font_size if first else median),
|
||||
font_name=str(first.font_name if first else ""),
|
||||
spans=spans,
|
||||
)
|
||||
)
|
||||
return lines
|
||||
|
||||
|
||||
def lines_from_plain_text(text: str, page_height: float = 792.0) -> list[Line]:
|
||||
"""Fallback when only plain extract_text is available (no glyph X geometry).
|
||||
|
||||
The x/y values here are fabricated to keep downstream code total; they carry
|
||||
no information about the page. Lines are flagged ``synthetic_geometry`` so
|
||||
consumers can choose text-based heuristics over geometric ones.
|
||||
"""
|
||||
lines: list[Line] = []
|
||||
y = page_height - 72
|
||||
for raw in text.splitlines():
|
||||
t = raw.strip()
|
||||
if not t:
|
||||
y -= 14
|
||||
continue
|
||||
lines.append(
|
||||
Line(
|
||||
y=y,
|
||||
x0=72,
|
||||
x1=72 + len(t) * 6,
|
||||
text=t,
|
||||
spans=[TextSpan(text=t)],
|
||||
synthetic_geometry=True,
|
||||
)
|
||||
)
|
||||
y -= 14
|
||||
return lines
|
||||
@@ -0,0 +1,876 @@
|
||||
"""Header/footer band detection across pages."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import threading
|
||||
from difflib import SequenceMatcher
|
||||
from functools import lru_cache
|
||||
|
||||
from app.services.convert.idm.model import BlockType, Document, Page, PageKind, TextSpan
|
||||
|
||||
# Punctuation, whitespace and digits are exactly what varies between repeats of
|
||||
# the same running header, so they are removed before comparing.
|
||||
_SIGNATURE_STRIP_RE = re.compile(r"[^a-z\u0600-\u06ff]+")
|
||||
# Two signatures this similar are treated as the same banner.
|
||||
SIGNATURE_SIMILARITY = 0.88
|
||||
# Two texts sharing this share of their words are the same banner even when
|
||||
# the word order differs. Set high enough that a body sentence quoting the
|
||||
# agency name does not match the banner it mentions.
|
||||
TOKEN_OVERLAP_SIMILARITY = 0.72
|
||||
# Threshold after confusable folding. Higher than the raw one, because folding
|
||||
# has already removed the differences that were only glyph shape — what is
|
||||
# left must genuinely be the same line.
|
||||
FOLDED_SIMILARITY = 0.86
|
||||
# Containment only applies between texts this short — a running header is a
|
||||
# line, not a sentence — so body prose quoting the agency name cannot match.
|
||||
BANNER_MAX_WORDS = 12
|
||||
# ...and only for a signature this long, so a common short phrase inside a
|
||||
# longer banner is not treated as the banner itself.
|
||||
MIN_CONTAINMENT_CHARS = 16
|
||||
# Below this length a signature is too generic to identify a header.
|
||||
MIN_SIGNATURE_CHARS = 4
|
||||
# Members of a banner family kept as comparison probes for single linkage.
|
||||
MAX_LINKAGE_PROBES = 12
|
||||
|
||||
|
||||
def _norm(text: str) -> str:
|
||||
return " ".join(text.lower().split())
|
||||
|
||||
|
||||
@lru_cache(maxsize=8192)
|
||||
def _signature(text: str) -> str:
|
||||
"""Comparison key robust to OCR noise and per-page variation.
|
||||
|
||||
Running headers rarely repeat byte-for-byte across a scanned document:
|
||||
OCR renders the same banner as "FEDERAL AUTHORITY FOR IDENTITY," on one
|
||||
page and "FEDERALAUTHORITY FOR IDENTITY." on the next, and footers carry a
|
||||
page number that changes by design. Folding away punctuation, spacing and
|
||||
digits leaves the part that actually repeats.
|
||||
"""
|
||||
lowered = (text or "").lower()
|
||||
return _SIGNATURE_STRIP_RE.sub("", lowered)
|
||||
|
||||
|
||||
# Glyph shapes a recogniser confuses, folded to one representative before any
|
||||
# comparison. These are shape collisions, not language facts, so the same table
|
||||
# works for a Latin letterhead, a Cyrillic one or a serial number.
|
||||
_CONFUSABLE_MAP = str.maketrans(
|
||||
{
|
||||
"0": "o", "1": "l", "5": "s", "8": "b", "6": "b", "9": "g", "2": "z",
|
||||
"i": "l", "j": "l", "t": "l",
|
||||
"u": "v", "n": "m", "r": "m",
|
||||
"e": "o", "a": "o", "c": "o", "q": "o", "d": "o",
|
||||
"f": "t", "y": "v", "w": "v",
|
||||
# Arabic shape collisions the recogniser makes on the same printed line.
|
||||
"ص": "س", "ض": "د", "ط": "ت", "ظ": "ز",
|
||||
"ة": "ه", "ى": "ي", "أ": "ا", "إ": "ا", "آ": "ا", "ؤ": "و", "ئ": "ي",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
@lru_cache(maxsize=8192)
|
||||
def _fold_confusables(signature: str) -> str:
|
||||
"""Collapse visually confusable glyphs so OCR drift compares as itself.
|
||||
|
||||
``Urted Areb Emlrates`` and ``United Arab Emirates`` are the same printed
|
||||
line read twice. Character-for-character they are only 80% alike, which is
|
||||
below any threshold safe enough to use on raw text — but the differences
|
||||
are all shape collisions a recogniser makes, and folding those first lets
|
||||
the pair be recognised without loosening the threshold for genuinely
|
||||
different lines.
|
||||
"""
|
||||
return signature.translate(_CONFUSABLE_MAP)
|
||||
|
||||
|
||||
# Vowels, and the glyphs OCR emits when it fails. A candidate spelling with
|
||||
# more of the first and fewer of the second is the better reading.
|
||||
_VOWELS = frozenset("aeiou")
|
||||
_JUNK_GLYPHS = frozenset("0123456789|\\/~^`{}[]<>@#*_")
|
||||
|
||||
|
||||
def _reading_quality(text: str) -> float:
|
||||
"""How *undamaged* a string looks. A tiebreak, not a spell-checker.
|
||||
|
||||
This ranks a reading with digits and symbols stuck through it below a
|
||||
reading made of letters — ``U1t3d Ar@b Em1r@t3s`` below ``United Arab
|
||||
Emirates``. It deliberately does not try to choose between two plausible
|
||||
letter strings: without a dictionary, ``Urted Areb Emlrates`` and the
|
||||
correct spelling are equally word-shaped, and any character statistic
|
||||
claiming otherwise is inventing a signal.
|
||||
|
||||
Choosing between plausible readings is the majority vote's job in
|
||||
:func:`_canonical_variant`; OCR errors vary from page to page while the
|
||||
correct reading repeats, so the mode is the real evidence. This only breaks
|
||||
ties the vote leaves open.
|
||||
"""
|
||||
stripped = (text or "").strip()
|
||||
if not stripped:
|
||||
return 0.0
|
||||
letters = [ch for ch in stripped.lower() if ch.isalpha()]
|
||||
if not letters:
|
||||
return 0.0
|
||||
junk_share = sum(1 for ch in stripped if ch in _JUNK_GLYPHS) / len(stripped)
|
||||
alpha_share = len(letters) / len(stripped)
|
||||
# A run of letters with no vowel at all is a failed reading in every
|
||||
# alphabet that has vowels; the check is skipped where there are none.
|
||||
has_vowel = any(ch in _VOWELS for ch in letters)
|
||||
vowel_floor = 1.0 if (has_vowel or not letters) else 0.0
|
||||
return round(0.45 * alpha_share + 0.35 * (1.0 - junk_share) + 0.20 * vowel_floor, 4)
|
||||
|
||||
|
||||
# One matcher per right-hand string, per thread. ``SequenceMatcher`` indexes
|
||||
# its second sequence on construction, and banner matching compares hundreds of
|
||||
# candidate lines against the same handful of variants — so that index was
|
||||
# rebuilt for every comparison. The cache is thread-local because a matcher
|
||||
# carries the current left-hand sequence as state and conversions run
|
||||
# concurrently in a thread pool.
|
||||
_MATCHERS = threading.local()
|
||||
MAX_CACHED_MATCHERS = 512
|
||||
|
||||
|
||||
def _matcher_for(b: str) -> SequenceMatcher:
|
||||
cache = getattr(_MATCHERS, "by_seq2", None)
|
||||
if cache is None:
|
||||
cache = {}
|
||||
_MATCHERS.by_seq2 = cache
|
||||
matcher = cache.get(b)
|
||||
if matcher is None:
|
||||
if len(cache) >= MAX_CACHED_MATCHERS:
|
||||
cache.clear()
|
||||
matcher = SequenceMatcher(None, "", b)
|
||||
cache[b] = matcher
|
||||
return matcher
|
||||
|
||||
|
||||
def _similar(a: str, b: str) -> float:
|
||||
if not a or not b:
|
||||
return 0.0
|
||||
return SequenceMatcher(None, a, b).ratio()
|
||||
|
||||
|
||||
def _similar_at_least(a: str, b: str, floor: float) -> bool:
|
||||
"""``_similar(a, b) >= floor``, without paying for the full comparison.
|
||||
|
||||
Banner grouping compares every candidate line against every variant
|
||||
collected so far, and ``SequenceMatcher.ratio`` is quadratic in the string
|
||||
length. On a 126-page form that came to 67,000 comparisons and 83% of the
|
||||
conversion's wall-clock.
|
||||
|
||||
``real_quick_ratio`` and ``quick_ratio`` are difflib's own *upper bounds*
|
||||
on ``ratio`` — one from the lengths alone, one from the multiset of
|
||||
characters — so rejecting on them cannot change any verdict. Most pairs
|
||||
are nothing alike and die on the first, which costs two calls to ``len``.
|
||||
"""
|
||||
if not a or not b:
|
||||
return False
|
||||
matcher = _matcher_for(b)
|
||||
matcher.set_seq1(a)
|
||||
if matcher.real_quick_ratio() < floor:
|
||||
return False
|
||||
if matcher.quick_ratio() < floor:
|
||||
return False
|
||||
return matcher.ratio() >= floor
|
||||
|
||||
|
||||
def _tokens(text: str) -> frozenset[str]:
|
||||
"""Word set of a banner, folded the same way as its signature."""
|
||||
folded = (_SIGNATURE_STRIP_RE.sub("", w) for w in (text or "").lower().split())
|
||||
return frozenset(t for t in folded if len(t) >= 3)
|
||||
|
||||
|
||||
def _token_overlap(a: str, b: str) -> float:
|
||||
"""Order-independent similarity between two banner texts.
|
||||
|
||||
A bilingual running header comes back from OCR with its halves in
|
||||
different orders on different pages — "FEDERAL AUTHORITY FOR IDENTITY,
|
||||
CITIZENSHIP…" on one and "CITIZENSHIP, CUSTOMS & PORT SECURITY FEDERAL
|
||||
AUTHORITY FOR IDENTITY," on the next. Those share every word and almost no
|
||||
character *sequence*, so ``SequenceMatcher`` scores them far apart and the
|
||||
same banner splits into variants that each look too rare to be furniture.
|
||||
"""
|
||||
ta, tb = _tokens(a), _tokens(b)
|
||||
if not ta or not tb:
|
||||
return 0.0
|
||||
return len(ta & tb) / len(ta | tb)
|
||||
|
||||
|
||||
# Short letterhead stamps (a country name, a three-word agency line) drift
|
||||
# further than a long banner: almost every glyph can be a near-miss, so the
|
||||
# long-line folded threshold would split each page into its own family.
|
||||
SHORT_BANNER_WORDS = 5
|
||||
SHORT_FOLDED_SIMILARITY = 0.64
|
||||
# A stamp glued into a longer cell is recognised when a window of the cell
|
||||
# matches the stamp this closely after folding.
|
||||
STAMP_WINDOW_SIMILARITY = 0.78
|
||||
|
||||
|
||||
_DIGITS_RE = re.compile(r"\d+")
|
||||
|
||||
# An identifier is a token this many letters long or more. Below it the token
|
||||
# is a folio marker — "p", "no", "pg" — not content.
|
||||
IDENTIFIER_LETTERS = 4
|
||||
|
||||
|
||||
def _varies_by_an_identifier(a: str, b: str) -> bool:
|
||||
"""Whether two lines differ only in digits that belong to an identifier.
|
||||
|
||||
A running header may legitimately change from page to page — "Page 1 of
|
||||
126" — and the signature drops digits so those pages group together. But a
|
||||
*per-page identifier* also survives that stripping: "ScanTokenpack1P000"
|
||||
and "ScanTokenpack1P001" have identical signatures, so a scanned document
|
||||
stamped with a document, case or invoice number had all its numbers read as
|
||||
one running banner, hoisted into the Word header, and every page's but the
|
||||
first deleted from the body.
|
||||
|
||||
The difference is where the digits sit. A folio is a bare number, or a
|
||||
number after a short marker word. Digits embedded in a long alphanumeric
|
||||
token are an identifier, and an identifier is content.
|
||||
"""
|
||||
if a == b:
|
||||
return False
|
||||
if _DIGITS_RE.sub("", a) != _DIGITS_RE.sub("", b):
|
||||
return False
|
||||
ta, tb = a.split(), b.split()
|
||||
if len(ta) != len(tb):
|
||||
return False
|
||||
for left, right in zip(ta, tb, strict=False):
|
||||
if left == right:
|
||||
continue
|
||||
letters = sum(1 for ch in left if ch.isalpha())
|
||||
if letters >= IDENTIFIER_LETTERS:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _same_banner(a: str, b: str, *, allow_containment: bool = True) -> bool:
|
||||
"""Whether two texts are the same running banner.
|
||||
|
||||
Three tests, because OCR breaks the banner in three different ways across
|
||||
a scanned document: it misspells it (sequence similarity), it reorders the
|
||||
halves of a bilingual line (token overlap), and it splits the banner over
|
||||
two blocks on one page while gluing its words together on another
|
||||
(containment).
|
||||
|
||||
The containment test compares signatures, which have spacing stripped, so
|
||||
``FEDERALAUTHORITY FOR IDENTITY,`` is recognised inside the full
|
||||
``FEDERAL AUTHORITY FOR IDENTITY, CITIZENSHIP, CUSTOMS & PORT SECURITY``.
|
||||
It is applied only between two banner-length texts, so a body sentence
|
||||
that quotes the agency is never swallowed by it.
|
||||
"""
|
||||
if _varies_by_an_identifier(a, b):
|
||||
return False
|
||||
sig_a, sig_b = _signature(a), _signature(b)
|
||||
if _similar_at_least(sig_a, sig_b, SIGNATURE_SIMILARITY):
|
||||
return True
|
||||
# Heavy drift: fold the glyph shapes a recogniser confuses and compare
|
||||
# again. "Urted Areb Emlrates" is 0.80 against "United Arab Emirates" raw
|
||||
# — below any threshold that is safe on unfolded text — and comfortably
|
||||
# above it once o/0, l/1/i and the rest stop counting as differences.
|
||||
fold_a, fold_b = _fold_confusables(sig_a), _fold_confusables(sig_b)
|
||||
if _similar_at_least(fold_a, fold_b, FOLDED_SIMILARITY):
|
||||
return True
|
||||
wa, wb = len(a.split()), len(b.split())
|
||||
if (
|
||||
2 <= wa <= SHORT_BANNER_WORDS
|
||||
and 2 <= wb <= SHORT_BANNER_WORDS
|
||||
and abs(wa - wb) <= 1
|
||||
and _similar_at_least(fold_a, fold_b, SHORT_FOLDED_SIMILARITY)
|
||||
):
|
||||
return True
|
||||
if _token_overlap(a, b) >= TOKEN_OVERLAP_SIMILARITY:
|
||||
return True
|
||||
if not allow_containment:
|
||||
# Containment is the one relation that legitimately joins *different*
|
||||
# banner lines: a page that read the whole two-line banner as one block
|
||||
# contains each line. Chaining families through it would merge the
|
||||
# lines into a single family and lose one of them, so grouping asks for
|
||||
# the strict relations only.
|
||||
return False
|
||||
if len(a.split()) > BANNER_MAX_WORDS or len(b.split()) > BANNER_MAX_WORDS:
|
||||
return False
|
||||
short, long_ = sorted((sig_a, sig_b), key=len)
|
||||
if not (len(short) >= MIN_CONTAINMENT_CHARS and short in long_):
|
||||
return False
|
||||
# Containment is for a banner split vs glued, not for a journal title that
|
||||
# sits inside a longer issue line ("Perspectives" inside "Number 4, Winter…").
|
||||
if _containment_has_distinct_payload(a, b):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
_ISSUE_LINE_RE = re.compile(
|
||||
r"\b(?:vol(?:ume)?|no\.?|number|issue|pp\.?)\b|\d{2,}",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def _containment_has_distinct_payload(a: str, b: str) -> bool:
|
||||
"""True when the longer string carries a different line, not a fuller OCR."""
|
||||
wa, wb = a.split(), b.split()
|
||||
if abs(len(wa) - len(wb)) <= 2:
|
||||
return False
|
||||
short_w, long_w = (wa, wb) if len(wa) <= len(wb) else (wb, wa)
|
||||
short_set = {w.lower() for w in short_w}
|
||||
extra = [w for w in long_w if w.lower() not in short_set]
|
||||
extra_joined = " ".join(extra)
|
||||
return bool(_ISSUE_LINE_RE.search(extra_joined))
|
||||
|
||||
|
||||
# A signature must appear on at least this share of pages to count as running
|
||||
# furniture. Two pages is never enough on a long document, and requiring every
|
||||
# page is too strict for scans where OCR loses the banner occasionally.
|
||||
MIN_BAND_PAGE_SHARE = 0.25
|
||||
# The band is never shallower than this many text lines, so a two- or
|
||||
# three-line banner is not cut in half on a sparse page.
|
||||
BAND_MIN_LINES = 3.0
|
||||
|
||||
|
||||
def _page_bands(page, body_blocks: list, band_ratio: float) -> tuple[list, list]:
|
||||
"""Top and bottom band blocks, in whatever coordinate system the page uses.
|
||||
|
||||
Bands are measured against the *observed* extent of the page's own blocks
|
||||
rather than ``page.height``. OCR-rebuilt pages carry pixel-derived
|
||||
coordinates (negative y, hundreds of units tall) that share no origin or
|
||||
scale with the PDF point height, so comparing them against ``page.height``
|
||||
classified every block on every scanned page as "bottom band" and never
|
||||
found the banner at all.
|
||||
"""
|
||||
tops = [b.bbox.y + b.bbox.h for b in body_blocks]
|
||||
bots = [b.bbox.y for b in body_blocks]
|
||||
hi, lo = max(tops), min(bots)
|
||||
span = hi - lo
|
||||
if span <= 0:
|
||||
# Degenerate geometry — fall back to reading order.
|
||||
ordered = sorted(body_blocks, key=lambda b: b.reading_order)
|
||||
return ordered[:1], ordered[-1:]
|
||||
|
||||
# A running header is a couple of lines, so the band is at least that deep
|
||||
# however short the page is. Using the span fraction alone made the band
|
||||
# narrower than the banner itself on sparse pages — a title page with six
|
||||
# blocks — and caught only the banner's first line.
|
||||
heights = sorted(b.bbox.h for b in body_blocks if b.bbox.h > 0)
|
||||
line_h = heights[len(heights) // 2] if heights else 0.0
|
||||
depth = max(span * band_ratio, line_h * BAND_MIN_LINES)
|
||||
|
||||
top_band = [b for b in body_blocks if b.bbox.y + b.bbox.h >= hi - depth]
|
||||
bot_band = [b for b in body_blocks if b.bbox.y <= lo + depth]
|
||||
return top_band, bot_band
|
||||
|
||||
|
||||
def _repeating_band_variants(per_page: list[list[str]], page_count: int) -> list[str]:
|
||||
"""Representative texts of every banner that repeats across enough pages.
|
||||
|
||||
Counted once per page, so a banner split into two blocks on one page does
|
||||
not out-vote a banner that appears on twenty. The return value is a list of
|
||||
*representatives*, not an exhaustive set of exact strings: callers compare
|
||||
against them with :func:`_same_banner`, so a page whose spelling drifted
|
||||
beyond anything already collected is still recognised.
|
||||
"""
|
||||
pages_with_text = [texts for texts in per_page if texts]
|
||||
if len(pages_with_text) < 2:
|
||||
return []
|
||||
|
||||
groups: list[dict] = []
|
||||
# A running header usually repeats byte-for-byte. Resolving each distinct
|
||||
# string once and remembering where it landed turns 126 pages of
|
||||
# comparisons into one, which is most of the cost of this pass on a long
|
||||
# document.
|
||||
resolved: dict[str, dict] = {}
|
||||
for page_i, texts in enumerate(pages_with_text):
|
||||
for text in dict.fromkeys(texts):
|
||||
if len(_signature(text)) < MIN_SIGNATURE_CHARS:
|
||||
continue
|
||||
known = resolved.get(text)
|
||||
if known is not None:
|
||||
known["pages"].add(page_i)
|
||||
continue
|
||||
# Single linkage: a variant joins a family if it matches *any*
|
||||
# member, not just the first one seen. OCR drift forms a chain —
|
||||
# a badly damaged reading can be far from the clean spelling while
|
||||
# sitting close to a middling one — and comparing only against a
|
||||
# fixed representative split one banner into several families, each
|
||||
# too rare on its own to clear the page-share threshold.
|
||||
group = next(
|
||||
(
|
||||
g
|
||||
for g in groups
|
||||
if any(
|
||||
_same_banner(v, text, allow_containment=False)
|
||||
for v in g["probes"]
|
||||
)
|
||||
or _same_banner(g["text"], text)
|
||||
),
|
||||
None,
|
||||
)
|
||||
if group is None:
|
||||
group = {
|
||||
"text": text,
|
||||
"pages": {page_i},
|
||||
"variants": [text],
|
||||
"probes": [text],
|
||||
}
|
||||
groups.append(group)
|
||||
else:
|
||||
group["pages"].add(page_i)
|
||||
group["variants"].append(text)
|
||||
# Chaining needs several members to link through, but not all
|
||||
# of them: past a handful the extra probes are near-duplicates
|
||||
# of ones the family already holds and only cost comparisons.
|
||||
if len(group["probes"]) < MAX_LINKAGE_PROBES:
|
||||
group["probes"].append(text)
|
||||
resolved[text] = group
|
||||
|
||||
threshold = max(2, round(len(pages_with_text) * MIN_BAND_PAGE_SHARE))
|
||||
return [
|
||||
_canonical_variant(g["variants"]) for g in groups if len(g["pages"]) >= threshold
|
||||
]
|
||||
|
||||
|
||||
def _canonical_variant(variants: list[str]) -> str:
|
||||
"""The cleanest spelling of a banner that OCR read differently each page.
|
||||
|
||||
A recogniser reading the same printed line on twenty pages returns twenty
|
||||
near-misses — a zero for an O, an l for an I, a stray full stop. Taking
|
||||
whichever one happened to come first put page one's mistakes into the Word
|
||||
header of the whole document. Language-like spellings beat letter-salad
|
||||
even when the salad repeats; among equally clean readings the majority
|
||||
(then the shorter running title) wins.
|
||||
"""
|
||||
if not variants:
|
||||
return ""
|
||||
counts: dict[str, int] = {}
|
||||
for text in variants:
|
||||
counts[text] = counts.get(text, 0) + 1
|
||||
|
||||
from app.services.convert.layout.text_quality import ocr_text_quality_score
|
||||
|
||||
return max(
|
||||
counts,
|
||||
key=lambda t: (
|
||||
ocr_text_quality_score([t]),
|
||||
counts[t],
|
||||
-len(t.split()),
|
||||
-len(t),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
# A bare number, "Page 4", "4 of 12", "- 4 -". Page numbers differ on every
|
||||
# page by definition, so they never group as a repeating banner and were left
|
||||
# behind in the body as orphans, one stray digit per page.
|
||||
_PAGE_NUMBER_RE = re.compile(
|
||||
r"^(?:[-–—(\[]?\s*)"
|
||||
r"(?:(?:page|pg|p\.?|seite|página|صفحة)\s*)?"
|
||||
r"\d{1,4}"
|
||||
r"(?:\s*(?:/|of|de|von|من)\s*\d{1,4})?"
|
||||
r"(?:\s*[-–—)\]]?)$",
|
||||
re.I,
|
||||
)
|
||||
# Longer than this and it is a sentence that happens to start with a number.
|
||||
PAGE_NUMBER_MAX_CHARS = 24
|
||||
|
||||
|
||||
def looks_like_page_number(text: str) -> bool:
|
||||
"""Whether a line is a page-number stamp rather than content.
|
||||
|
||||
Deliberately shape-based: a short line that is essentially a number, with
|
||||
an optional label and an optional "of N". The few label words listed are a
|
||||
convenience, not the test — a bare "7" is recognised in any language.
|
||||
"""
|
||||
stripped = (text or "").strip()
|
||||
if not stripped or len(stripped) > PAGE_NUMBER_MAX_CHARS:
|
||||
return False
|
||||
if not any(ch.isdigit() for ch in stripped):
|
||||
return False
|
||||
return bool(_PAGE_NUMBER_RE.match(stripped))
|
||||
|
||||
|
||||
def tag_headers_footers(
|
||||
document: Document,
|
||||
*,
|
||||
band_ratio: float = 0.12,
|
||||
header_bboxes: list | None = None,
|
||||
footer_bboxes: list | None = None,
|
||||
) -> None:
|
||||
"""
|
||||
Detect repeating top/bottom text across ≥2 pages and retag those blocks
|
||||
as header/footer. Prefer high/low bbox Y when available.
|
||||
Optional ML header/footer bboxes boost single-page retag.
|
||||
"""
|
||||
# ML hints: retag blocks whose centers fall in hint boxes (works on 1-page docs too)
|
||||
if header_bboxes or footer_bboxes:
|
||||
for page in document.pages:
|
||||
for b in page.blocks:
|
||||
if not b.plain_text().strip() or b.type in (BlockType.table, BlockType.figure):
|
||||
continue
|
||||
cx = b.bbox.x + b.bbox.w / 2
|
||||
cy = b.bbox.y + b.bbox.h / 2
|
||||
for hb in header_bboxes or []:
|
||||
if hb.x <= cx <= hb.x + hb.w and hb.y <= cy <= hb.y + hb.h:
|
||||
b.type = BlockType.header
|
||||
break
|
||||
for fb in footer_bboxes or []:
|
||||
if fb.x <= cx <= fb.x + fb.w and fb.y <= cy <= fb.y + fb.h:
|
||||
if b.type != BlockType.header:
|
||||
b.type = BlockType.footer
|
||||
break
|
||||
|
||||
if document.page_count < 2:
|
||||
return
|
||||
|
||||
top_texts: list[list[str]] = []
|
||||
bottom_texts: list[list[str]] = []
|
||||
bands: dict[int, tuple[list, list]] = {}
|
||||
for page in document.pages:
|
||||
body_blocks = [b for b in page.blocks if b.plain_text().strip()]
|
||||
if not body_blocks:
|
||||
continue
|
||||
top_band, bot_band = _page_bands(page, body_blocks, band_ratio)
|
||||
bands[page.index] = (top_band, bot_band)
|
||||
# Every block in the band is a candidate, not just the outermost one.
|
||||
# A running banner is a *region*: this document repeats an agency line
|
||||
# and "United Arab Emirates" beneath it, and taking one block per page
|
||||
# left the second line in the body on all 26 pages.
|
||||
top_texts.append([_norm(b.plain_text()) for b in top_band])
|
||||
bottom_texts.append([_norm(b.plain_text()) for b in bot_band])
|
||||
|
||||
header_variants = _repeating_band_variants(top_texts, document.page_count)
|
||||
footer_variants = _repeating_band_variants(bottom_texts, document.page_count)
|
||||
# A doc can repeat the same line top and bottom; prefer header.
|
||||
footer_variants = [
|
||||
f for f in footer_variants if not any(_same_banner(f, h) for h in header_variants)
|
||||
]
|
||||
|
||||
numbered = False
|
||||
for page in document.pages:
|
||||
top_band, bot_band = bands.get(page.index, ([], []))
|
||||
_retag_page(page, header_variants, footer_variants, top_band, bot_band)
|
||||
numbered |= _tag_page_numbers(bot_band)
|
||||
|
||||
# Once a family is established, strip it wherever it appears — the cover
|
||||
# page, a mid-document divider, a cell. A variant that shows up on only two
|
||||
# pages never clears the page-share threshold on its own, but it is still
|
||||
# the same letterhead, and leaving it in the body is how a country line
|
||||
# ended up as the document's opening paragraph.
|
||||
_retag_stragglers(document, header_variants, footer_variants)
|
||||
furniture = header_variants + footer_variants
|
||||
# Stamp trimming searches every cell and paragraph for a fuzzy substring of
|
||||
# the banner. It exists because a recogniser reads the letterhead and the
|
||||
# first clause as one line; a digital text layer never does that — the
|
||||
# banner is its own text object at its own position, and retagging has
|
||||
# already moved it. Running the search anyway cost 4s of the 8s spent on
|
||||
# four pages of a tax table, comparing "1,234" against the running head
|
||||
# thousands of times.
|
||||
if _has_recognised_text(document):
|
||||
_strip_banners_from_tables(document, furniture)
|
||||
_strip_stamps_from_prose(document, furniture)
|
||||
|
||||
if numbered:
|
||||
# The writer emits one PAGE field instead of the literal numbers, which
|
||||
# is both correct in Word and the only way a per-page stamp can live in
|
||||
# a single header definition.
|
||||
meta = document.meta if document.meta is not None else {}
|
||||
meta["page_number_footer"] = True
|
||||
document.meta = meta
|
||||
|
||||
|
||||
# A straggler is only retagged if it is short enough to be furniture. Body
|
||||
# prose that quotes the organisation is longer than a banner line.
|
||||
STRAGGLER_MAX_WORDS = 12
|
||||
# ...and it must sit within this share of the page's own vertical extent,
|
||||
# measured from whichever end the family belongs to.
|
||||
STRAGGLER_BAND_SHARE = 0.30
|
||||
|
||||
|
||||
def _retag_stragglers(
|
||||
document: Document, header_variants: list[str], footer_variants: list[str]
|
||||
) -> None:
|
||||
"""Retag banner-family lines that fell outside a detected band.
|
||||
|
||||
The band vote answers "does this repeat enough to be furniture". Once it
|
||||
has answered yes, every other appearance of the same family is furniture
|
||||
too, wherever it sits: a cover page puts the letterhead lower than a body
|
||||
page does, and a divider page puts it alone in the middle. Those copies
|
||||
never joined the vote and so stayed in the body — the first paragraph of
|
||||
the document being a misread country line is exactly this.
|
||||
"""
|
||||
if not header_variants and not footer_variants:
|
||||
return
|
||||
for page in document.pages:
|
||||
candidates = [
|
||||
b
|
||||
for b in page.blocks
|
||||
if b.plain_text().strip()
|
||||
and b.type not in (BlockType.table, BlockType.figure)
|
||||
and not b.image_png
|
||||
]
|
||||
if not candidates:
|
||||
continue
|
||||
tops = [b.bbox.y + b.bbox.h for b in candidates]
|
||||
lo, hi = min(b.bbox.y for b in candidates), max(tops)
|
||||
span = max(hi - lo, 1.0)
|
||||
|
||||
for block in candidates:
|
||||
if block.type in (BlockType.header, BlockType.footer):
|
||||
continue
|
||||
text = block.plain_text().strip()
|
||||
if len(text.split()) > STRAGGLER_MAX_WORDS:
|
||||
continue
|
||||
# Position decides as much as wording. A line that repeats the
|
||||
# banner but sits in the middle of the page is a mention — the
|
||||
# document referring to its own author — while the same words at
|
||||
# the top of a cover page are the letterhead that the band vote
|
||||
# simply did not see enough times to count.
|
||||
near_top = (block.bbox.y + block.bbox.h - lo) / span >= 1.0 - STRAGGLER_BAND_SHARE
|
||||
near_bottom = (block.bbox.y - lo) / span <= STRAGGLER_BAND_SHARE
|
||||
# OCR pages often have unusable Y, so a letterhead line in the
|
||||
# "middle" of the reconstructed page is still furniture.
|
||||
scanish = page.kind in (PageKind.scan, PageKind.hybrid) or any(
|
||||
getattr(b, "source", "") == "ocr" for b in page.blocks
|
||||
)
|
||||
if (scanish or near_top) and any(_same_banner(text, h) for h in header_variants):
|
||||
block.type = BlockType.header
|
||||
block.level = 0
|
||||
elif near_bottom and any(_same_banner(text, f) for f in footer_variants):
|
||||
block.type = BlockType.footer
|
||||
block.level = 0
|
||||
|
||||
|
||||
def _tag_page_numbers(bot_band: list) -> bool:
|
||||
"""Move page-number stamps out of the body and into the footer."""
|
||||
found = False
|
||||
for block in bot_band or []:
|
||||
if block.type in (BlockType.table, BlockType.figure, BlockType.header):
|
||||
continue
|
||||
if looks_like_page_number(block.plain_text()):
|
||||
block.type = BlockType.footer
|
||||
block.level = 0
|
||||
found = True
|
||||
return found
|
||||
|
||||
|
||||
def _shared_character_budget(needle: str, haystack: str) -> int:
|
||||
"""Size of the multiset intersection of two strings.
|
||||
|
||||
No window of *haystack* can match *needle* more closely than the share of
|
||||
needle's characters that appear in haystack at all — a bound that costs one
|
||||
pass over each string, against the hundreds of full comparisons the sliding
|
||||
search would otherwise run.
|
||||
"""
|
||||
counts: dict[str, int] = {}
|
||||
for ch in needle:
|
||||
counts[ch] = counts.get(ch, 0) + 1
|
||||
shared = 0
|
||||
for ch in haystack:
|
||||
left = counts.get(ch, 0)
|
||||
if left:
|
||||
counts[ch] = left - 1
|
||||
shared += 1
|
||||
return shared
|
||||
|
||||
|
||||
@lru_cache(maxsize=4096)
|
||||
def _folded_window_match(needle: str, haystack: str) -> bool:
|
||||
"""Whether ``needle`` appears inside ``haystack`` after OCR-shape folding.
|
||||
|
||||
Used to find a short running stamp that was concatenated onto a real cell
|
||||
("Capability Area" + a misspelt country line). Whole-string equality fails
|
||||
because of the extra words; an un-folded substring fails because of the
|
||||
misspelling.
|
||||
|
||||
The sliding search is the most expensive comparison in the package — six
|
||||
window widths across every offset, each a fresh ``SequenceMatcher`` — so it
|
||||
is entered only after a character-budget check rules out the overwhelming
|
||||
majority of pairs, and the matcher's index over the needle is built once
|
||||
and reused across every window.
|
||||
"""
|
||||
n = _fold_confusables(_signature(needle))
|
||||
h = _fold_confusables(_signature(haystack))
|
||||
if len(n) < 10 or not h:
|
||||
return False
|
||||
if n in h:
|
||||
return True
|
||||
if _shared_character_budget(n, h) < STAMP_WINDOW_SIMILARITY * len(n):
|
||||
return False
|
||||
lo = max(8, len(n) - 2)
|
||||
hi = min(len(h), len(n) + 3)
|
||||
matcher = SequenceMatcher(None, "", n)
|
||||
for width in range(lo, hi + 1):
|
||||
if width > len(h):
|
||||
break
|
||||
for i in range(0, len(h) - width + 1):
|
||||
matcher.set_seq1(h[i : i + width])
|
||||
if matcher.real_quick_ratio() < STAMP_WINDOW_SIMILARITY:
|
||||
continue
|
||||
if matcher.quick_ratio() < STAMP_WINDOW_SIMILARITY:
|
||||
continue
|
||||
if matcher.ratio() >= STAMP_WINDOW_SIMILARITY:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _trim_stamp_from_text(text: str, variants: list[str]) -> str:
|
||||
"""Drop a running stamp glued onto a cell or a body line; keep the rest."""
|
||||
raw = (text or "").strip()
|
||||
if not raw:
|
||||
return raw
|
||||
words = raw.split()
|
||||
if len(words) <= SHORT_BANNER_WORDS and any(_same_banner(raw, v) for v in variants):
|
||||
return ""
|
||||
bullets = {"·", "•", "-", "—", "*", "|"}
|
||||
for v in variants:
|
||||
vn = len(v.split())
|
||||
if vn < 2:
|
||||
continue
|
||||
for n in (vn, vn + 1, vn - 1):
|
||||
if n < 2 or n > len(words):
|
||||
continue
|
||||
for i in range(0, len(words) - n + 1):
|
||||
window = " ".join(words[i : i + n])
|
||||
if not (_same_banner(window, v) or _folded_window_match(v, window)):
|
||||
continue
|
||||
at_edge = i == 0 or i + n == len(words)
|
||||
next_to_mark = (i > 0 and words[i - 1] in bullets) or (
|
||||
i + n < len(words) and words[i + n] in bullets
|
||||
)
|
||||
if not (at_edge or next_to_mark):
|
||||
continue
|
||||
kept = words[:i] + words[i + n :]
|
||||
return " ".join(kept).strip()
|
||||
return raw
|
||||
|
||||
|
||||
def _has_recognised_text(document: Document) -> bool:
|
||||
"""Whether any of this document's text came from OCR rather than the PDF."""
|
||||
for page in document.pages:
|
||||
if page.kind in (PageKind.scan, PageKind.hybrid):
|
||||
return True
|
||||
for block in page.blocks:
|
||||
if getattr(block, "source", "") == "ocr":
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _strip_banners_from_tables(document: Document, variants: list[str]) -> None:
|
||||
"""Remove running-banner rows that were swallowed into a grid.
|
||||
|
||||
On a scan the banner and the top of a table sit close together, and grid
|
||||
assembly sometimes takes the banner in as a first row. Retagging skips
|
||||
tables, so the banner then appeared in the Word header *and* as the table's
|
||||
header row — the same text twice, one of them wrong.
|
||||
"""
|
||||
if not variants:
|
||||
return
|
||||
banner_only: list = []
|
||||
for page in document.pages:
|
||||
for block in page.blocks:
|
||||
if block.type != BlockType.table or not block.cells:
|
||||
continue
|
||||
rows = []
|
||||
for row in block.cells:
|
||||
cleaned = [_trim_stamp_from_text(c or "", variants) for c in row]
|
||||
joined = " ".join(c for c in cleaned if (c or "").strip()).strip()
|
||||
if not joined:
|
||||
continue
|
||||
if any(_same_banner(joined, v) for v in variants):
|
||||
continue
|
||||
rows.append(cleaned)
|
||||
if rows == block.cells:
|
||||
continue
|
||||
if rows:
|
||||
block.cells = rows
|
||||
continue
|
||||
# Every row was banner. This is not a table with a banner in it —
|
||||
# it is the banner, mistaken for a table, and emitting it as a
|
||||
# one-row grid puts the letterhead in the document twice, once as
|
||||
# furniture and once as a box around nothing.
|
||||
block.cells = []
|
||||
block.type = BlockType.paragraph
|
||||
block.text = ""
|
||||
block.spans = []
|
||||
banner_only.append(block)
|
||||
|
||||
for page in document.pages:
|
||||
page.blocks = [b for b in page.blocks if id(b) not in {id(x) for x in banner_only}]
|
||||
|
||||
|
||||
def _strip_stamps_from_prose(document: Document, variants: list[str]) -> None:
|
||||
"""Trim a running stamp that was concatenated onto a body paragraph.
|
||||
|
||||
Tables are handled separately. Requirement lines often carry the letterhead
|
||||
in the same OCR line as the clause; retagging cannot move that line into
|
||||
the header without losing the clause.
|
||||
"""
|
||||
if not variants:
|
||||
return
|
||||
for page in document.pages:
|
||||
for block in page.blocks:
|
||||
if block.type in (BlockType.table, BlockType.figure, BlockType.header, BlockType.footer):
|
||||
continue
|
||||
raw = block.plain_text().strip()
|
||||
if not raw:
|
||||
continue
|
||||
cleaned = _trim_stamp_from_text(raw, variants)
|
||||
if cleaned == raw:
|
||||
continue
|
||||
block.text = cleaned
|
||||
if block.spans:
|
||||
sample = block.spans[0]
|
||||
block.spans = [
|
||||
TextSpan(
|
||||
text=cleaned,
|
||||
font_name=sample.font_name,
|
||||
font_size=sample.font_size,
|
||||
bold=sample.bold,
|
||||
italic=sample.italic,
|
||||
)
|
||||
]
|
||||
|
||||
|
||||
def _retag_page(
|
||||
page: Page,
|
||||
header_variants: list[str],
|
||||
footer_variants: list[str],
|
||||
top_band: list | None = None,
|
||||
bot_band: list | None = None,
|
||||
) -> None:
|
||||
"""Retag banner blocks inside the page's top/bottom band.
|
||||
|
||||
Matching within the band rather than walking a leading run matters for two
|
||||
real cases in scanned documents:
|
||||
|
||||
* a masthead logo sits above the banner, and a run-walk stops dead at the
|
||||
figure — leaving every banner line in the body of page one;
|
||||
* on a dense page the banner is not the first block in reading order, so a
|
||||
run starting at index 0 never reaches it.
|
||||
|
||||
The band already confines the search to the top or bottom of the page, and
|
||||
only text matching a document-wide banner is retagged, so body prose that
|
||||
merely mentions the agency keeps its place.
|
||||
"""
|
||||
if top_band is None or bot_band is None:
|
||||
ordered = sorted(
|
||||
[b for b in page.blocks if b.plain_text().strip()],
|
||||
key=lambda b: b.reading_order,
|
||||
)
|
||||
top_band = top_band if top_band is not None else ordered[:1]
|
||||
bot_band = bot_band if bot_band is not None else ordered[-1:]
|
||||
|
||||
skip = (BlockType.table, BlockType.figure)
|
||||
tagged: set[int] = set()
|
||||
|
||||
for block in top_band:
|
||||
if block.type in skip or not header_variants:
|
||||
continue
|
||||
if any(_same_banner(block.plain_text(), h) for h in header_variants):
|
||||
block.type = BlockType.header
|
||||
block.level = 0
|
||||
tagged.add(id(block))
|
||||
|
||||
for block in bot_band:
|
||||
if block.type in skip or id(block) in tagged or not footer_variants:
|
||||
continue
|
||||
if any(_same_banner(block.plain_text(), f) for f in footer_variants):
|
||||
block.type = BlockType.footer
|
||||
block.level = 0
|
||||
@@ -0,0 +1,982 @@
|
||||
"""Post-IDM optimize: ConvertAPI/Aspose reconstruction modes (in-house)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import replace
|
||||
|
||||
from app.services.convert.idm.model import (
|
||||
Block,
|
||||
BlockType,
|
||||
Document,
|
||||
Page,
|
||||
PageKind,
|
||||
TextSpan,
|
||||
)
|
||||
from app.services.convert.layout.paragraphs import SHORT_LINE_RATIO, SIZE_TOLERANCE
|
||||
from app.services.convert.layout.styles import (
|
||||
dominant_body_size,
|
||||
split_glued_section_number,
|
||||
)
|
||||
from app.services.convert.options import (
|
||||
LayoutMode,
|
||||
RecognitionMode,
|
||||
get_options,
|
||||
)
|
||||
|
||||
# If more than this fraction of body blocks are headings, demote body-sized ones.
|
||||
HEADING_DENSITY_CAP = 0.35
|
||||
# A table block carrying at least this confidence was assembled from measured
|
||||
# column alignment, so it is a detected region and is never soup-flattened.
|
||||
DETECTED_TABLE_CONFIDENCE = 0.7
|
||||
# Continuous mode: merge adjacent paragraphs whose font sizes are this close.
|
||||
# Only ever used when the caller explicitly asks for continuous layout; the
|
||||
# default flowing mode uses ``paragraphs.SIZE_TOLERANCE`` so that this pass and
|
||||
# ``breaks_paragraph`` agree on what counts as the same size. At 0.22 this value
|
||||
# treated 11pt body and a 12pt heading as one size and welded them.
|
||||
CONTINUOUS_SIZE_TOL = 0.22
|
||||
# Blocks whose left edges fall within this many points are treated as one
|
||||
# column when measuring where that column's right margin is.
|
||||
_WRAP_COLUMN_SNAP = 24.0
|
||||
|
||||
# How far right of ``prev`` a continuation block's left edge may sit and still be
|
||||
# the next line of the same paragraph. Wrapped lines share a left edge; this is
|
||||
# only slack for glyph-extent rounding.
|
||||
_WRAP_LEFT_TOL = 3.0
|
||||
|
||||
|
||||
def _block_size(block: Block) -> float:
|
||||
"""The size this block is *set* in: its dominant size, weighted by characters.
|
||||
|
||||
A plain mean over spans lets one large run misreport the whole block. The
|
||||
fixture's character-formatting line is 11pt prose containing a single 22pt
|
||||
word, which averaged to 12.6pt -- close enough to the 12pt heading beneath it
|
||||
that the two were judged the same size and welded together. Weighting by
|
||||
character count answers the question actually being asked, which is what size
|
||||
the reader sees this block as.
|
||||
"""
|
||||
if block.spans:
|
||||
weights: dict[float, int] = {}
|
||||
for s in block.spans:
|
||||
size = s.font_size
|
||||
if not size or size <= 0:
|
||||
continue
|
||||
n = len((s.text or "").strip())
|
||||
if not n:
|
||||
continue
|
||||
weights[round(size * 2) / 2.0] = weights.get(round(size * 2) / 2.0, 0) + n
|
||||
if weights:
|
||||
return max(weights.items(), key=lambda kv: (kv[1], kv[0]))[0]
|
||||
sizes = [s.font_size for s in block.spans if s.font_size and s.font_size > 0]
|
||||
if sizes:
|
||||
return sum(sizes) / len(sizes)
|
||||
return 12.0
|
||||
|
||||
|
||||
def _block_font(block: Block) -> str:
|
||||
"""Dominant font family of a block, normalised for comparison.
|
||||
|
||||
Weight and slant suffixes are stripped so ``Calibri`` and ``Calibri-Bold``
|
||||
read as one family: a bold phrase inside a sentence must not split the
|
||||
paragraph. ``Cambria`` and ``Calibri`` stay distinct. Normalisation matches
|
||||
``layout.paragraphs._style_changed`` so the two passes cannot disagree.
|
||||
"""
|
||||
counts: dict[str, int] = {}
|
||||
for span in block.spans or []:
|
||||
text = span.text or ""
|
||||
if not text.strip():
|
||||
continue
|
||||
name = (span.font_name or "").split("+")[-1].split("-")[0].strip().lower()
|
||||
if name:
|
||||
counts[name] = counts.get(name, 0) + len(text)
|
||||
return max(counts, key=lambda key: counts[key]) if counts else ""
|
||||
|
||||
|
||||
def _is_bodyish(block: Block) -> bool:
|
||||
return block.type in (
|
||||
BlockType.paragraph,
|
||||
BlockType.heading,
|
||||
BlockType.list_item,
|
||||
) and bool(block.plain_text().strip())
|
||||
|
||||
|
||||
def _latin_label_row(row: list[str]) -> bool:
|
||||
texts = [c.strip() for c in row if c and str(c).strip()]
|
||||
if not (2 <= len(texts) <= 4):
|
||||
return False
|
||||
if any(any(ch.isdigit() for ch in t) for t in texts):
|
||||
return False
|
||||
if any(len(t.split()) > 4 or len(t) >= 48 for t in texts):
|
||||
return False
|
||||
if any("\u0600" <= ch <= "\u06FF" for t in texts for ch in t):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def _row_is_empty_or_arabic_masthead(row: list[str]) -> bool:
|
||||
texts = [c.strip() for c in row if c and str(c).strip()]
|
||||
if not texts:
|
||||
return True
|
||||
ar = sum(1 for t in texts for ch in t if "\u0600" <= ch <= "\u06FF")
|
||||
latin = sum(1 for t in texts for ch in t if ch.isascii() and ch.isalpha())
|
||||
return ar >= 8 and ar >= latin
|
||||
|
||||
|
||||
def _table_is_mid_sentence_fragment(cells: list[list[str]] | None) -> bool:
|
||||
"""OCR leftover of a larger grid: starts mid-phrase or with a hyphen break."""
|
||||
rows = cells or []
|
||||
if not (1 <= len(rows) <= 4):
|
||||
return False
|
||||
ncols = max((len(r) for r in rows), default=0)
|
||||
if ncols > 4 or ncols < 2:
|
||||
return False
|
||||
first_row = [c.strip() for c in rows[0] if c and str(c).strip()]
|
||||
if not first_row:
|
||||
return False
|
||||
# Real tables start with short title-case headers, not "in structured 2-".
|
||||
if all(
|
||||
1 <= len(t.split()) <= 4
|
||||
and t[:1].isupper()
|
||||
and not t.endswith("-")
|
||||
for t in first_row
|
||||
):
|
||||
return False
|
||||
lead = first_row[0]
|
||||
if lead[:1].islower() and lead[:1].isascii():
|
||||
return True
|
||||
return any(t.endswith("-") for t in first_row)
|
||||
|
||||
|
||||
def _table_is_stamp_remnant(cells: list[list[str]] | None) -> bool:
|
||||
"""Tiny grids that are only leftover header stamps, not data tables."""
|
||||
rows = cells or []
|
||||
if not (1 <= len(rows) <= 2):
|
||||
return False
|
||||
ncols = max((len(r) for r in rows), default=0)
|
||||
if ncols > 4:
|
||||
return False
|
||||
if len(rows) == 2 and ncols > 3:
|
||||
return False
|
||||
if (
|
||||
len(rows) == 2
|
||||
and _latin_label_row(rows[0])
|
||||
and _row_is_empty_or_arabic_masthead(rows[1])
|
||||
):
|
||||
return True
|
||||
texts = [c.strip() for row in rows for c in row if c and str(c).strip()]
|
||||
if not texts:
|
||||
return True
|
||||
if any(ch.isdigit() for t in texts for ch in t):
|
||||
return False
|
||||
unique = {t.lower() for t in texts}
|
||||
max_words = max(len(t.split()) for t in texts)
|
||||
if not (
|
||||
len(unique) <= 4
|
||||
and max_words <= 4
|
||||
and all(len(t) < 48 for t in texts)
|
||||
):
|
||||
return False
|
||||
# Stamp leftover: the same labels repeat. A 2×N of distinct cells
|
||||
# (two-column body, Name/Qty/Cost + data) must stay.
|
||||
if len(rows) == 2:
|
||||
r0 = [str(c).strip().lower() for c in rows[0] if c and str(c).strip()]
|
||||
r1 = [str(c).strip().lower() for c in rows[1] if c and str(c).strip()]
|
||||
if r0 and r0 == r1:
|
||||
return True
|
||||
return len(unique) <= max(1, len(texts) // 2)
|
||||
lows = [t.lower().replace(" ", "") for t in texts]
|
||||
smashed = any(
|
||||
i != j and b in a and a != b and len(b) >= 6
|
||||
for i, a in enumerate(lows)
|
||||
for j, b in enumerate(lows)
|
||||
)
|
||||
return len(unique) <= 3 or smashed
|
||||
|
||||
|
||||
_VALUE_LIKE = re.compile(r"^[\s(\[]*[-+±]?[$€£¥₹]?\s*\d[\d,. :/–—-]*\s*[%°]?\s*[)\]]?$")
|
||||
|
||||
|
||||
def _is_value_like(text: str) -> bool:
|
||||
"""A number, amount, date, range or percentage — a cell, not a word.
|
||||
|
||||
``ocr_text_quality_score`` measures how much a string reads like language.
|
||||
A tax table's "5,600" reads like nothing at all and scores near zero, so a
|
||||
grid of numbers looked exactly like recogniser salad and was deleted:
|
||||
64 rows of the earned-income tables, every financial statement, every
|
||||
price list. Values have to be recognised as legitimate before that
|
||||
judgement is made.
|
||||
"""
|
||||
stripped = (text or "").strip()
|
||||
if not stripped:
|
||||
return False
|
||||
return bool(_VALUE_LIKE.match(stripped)) and any(ch.isdigit() for ch in stripped)
|
||||
|
||||
|
||||
def _table_cells_are_ocr_junk(cells: list[list[str]] | None) -> bool:
|
||||
"""A grid whose cells are almost all cmap-salad is not a table to keep.
|
||||
|
||||
Flattening it would dump the same salad into body paragraphs. Drop it.
|
||||
"""
|
||||
from app.services.convert.layout.text_quality import (
|
||||
is_cmap_garbled_line,
|
||||
is_ocr_noise_line,
|
||||
ocr_text_quality_score,
|
||||
)
|
||||
|
||||
texts = [c.strip() for row in (cells or []) for c in row if c and str(c).strip()]
|
||||
if not texts:
|
||||
return True
|
||||
bad = 0
|
||||
for t in texts:
|
||||
if _is_value_like(t):
|
||||
continue
|
||||
if is_ocr_noise_line(t) or is_cmap_garbled_line(t) or ocr_text_quality_score([t]) <= 0.15:
|
||||
bad += 1
|
||||
return (bad / len(texts)) >= 0.75
|
||||
|
||||
|
||||
def flatten_table_to_paragraphs(block: Block) -> list[Block]:
|
||||
"""Turn a rejected/soup table into flowing paragraphs (EnhancedFlow)."""
|
||||
out: list[Block] = []
|
||||
order = block.reading_order
|
||||
for row in block.cells or []:
|
||||
text = " ".join(c.strip() for c in row if c and c.strip()).strip()
|
||||
if not text:
|
||||
continue
|
||||
out.append(
|
||||
Block(
|
||||
type=BlockType.paragraph,
|
||||
text=text,
|
||||
spans=[TextSpan(text=text)],
|
||||
bbox=block.bbox,
|
||||
reading_order=order,
|
||||
source=block.source,
|
||||
)
|
||||
)
|
||||
order += 1
|
||||
return out or [
|
||||
Block(
|
||||
type=BlockType.paragraph,
|
||||
text=block.plain_text(),
|
||||
spans=block.spans or [TextSpan(text=block.plain_text())],
|
||||
bbox=block.bbox,
|
||||
reading_order=block.reading_order,
|
||||
source=block.source,
|
||||
)
|
||||
]
|
||||
|
||||
|
||||
def _flatten_soup_tables(document: Document) -> None:
|
||||
"""OCR word-soup grids → paragraphs. Digital real grids stay tables."""
|
||||
from app.services.convert.layout.tables import (
|
||||
is_fragmented_prose_grid,
|
||||
is_logo_or_wordmark_grid,
|
||||
)
|
||||
|
||||
for page in document.pages:
|
||||
rebuilt: list[Block] = []
|
||||
for block in page.blocks:
|
||||
if block.type != BlockType.table or not block.cells:
|
||||
rebuilt.append(block)
|
||||
continue
|
||||
soup = is_fragmented_prose_grid(block.cells)
|
||||
logo = is_logo_or_wordmark_grid(block.cells)
|
||||
ocr_only = (block.source or "") == "ocr"
|
||||
# A grid the detector assembled from aligned column positions is a
|
||||
# *detected table region*, not word soup, whatever its cells read
|
||||
# like. Word-soup flattening exists for the opposite case — prose
|
||||
# that only looked like a grid — and must never undo a real
|
||||
# detection: a two-column scanned table of short entries is
|
||||
# textually indistinguishable from soup and is not soup.
|
||||
detected = float(getattr(block, "table_confidence", 0.0) or 0.0) >= DETECTED_TABLE_CONFIDENCE
|
||||
if _table_cells_are_ocr_junk(block.cells):
|
||||
continue
|
||||
if logo or (soup and ocr_only and not detected):
|
||||
rebuilt.extend(flatten_table_to_paragraphs(block))
|
||||
else:
|
||||
rebuilt.append(block)
|
||||
for i, b in enumerate(rebuilt):
|
||||
b.reading_order = i
|
||||
page.blocks = rebuilt
|
||||
|
||||
|
||||
def _drop_all_tables(document: Document) -> None:
|
||||
for page in document.pages:
|
||||
rebuilt: list[Block] = []
|
||||
for block in page.blocks:
|
||||
if block.type == BlockType.table and block.cells:
|
||||
rebuilt.extend(flatten_table_to_paragraphs(block))
|
||||
else:
|
||||
rebuilt.append(block)
|
||||
for i, b in enumerate(rebuilt):
|
||||
b.reading_order = i
|
||||
page.blocks = rebuilt
|
||||
|
||||
|
||||
_SENTENCE_END = ".!?۔؟:;،"
|
||||
|
||||
|
||||
def _reads_as_wrapped(prev_text: str, next_text: str) -> bool:
|
||||
"""Whether two lines are one sentence split across a line break.
|
||||
|
||||
The only merge evidence available when a page has no geometry. A wrapped
|
||||
line leaves its sentence unfinished and the next line continues it in
|
||||
lower case (or in a script with no case, continues without a bullet or
|
||||
number). Table cells and list items fail both halves.
|
||||
"""
|
||||
prev_text, next_text = prev_text.strip(), next_text.strip()
|
||||
if not prev_text or not next_text:
|
||||
return False
|
||||
if prev_text[-1] in _SENTENCE_END:
|
||||
return False
|
||||
if prev_text.endswith("-"):
|
||||
return True
|
||||
head = next_text[0]
|
||||
if head.isdigit() or head in "•-*–—([":
|
||||
return False
|
||||
# Cased scripts: a capital starts a new line, not a continuation.
|
||||
if head.isupper():
|
||||
return False
|
||||
# A single bare token on either side is a cell or label, not prose.
|
||||
return len(prev_text.split()) > 1 and len(next_text.split()) > 1
|
||||
|
||||
|
||||
def _reached_wrap_margin(blocks: list[Block], prev: Block) -> bool:
|
||||
"""Whether ``prev`` runs to the right margin of the column it sits in.
|
||||
|
||||
A line that wrapped ends because it ran out of room, so it reaches the
|
||||
margin; a line the author ended stops short of it. That difference is the
|
||||
only evidence that two consecutive one-line blocks are really one paragraph,
|
||||
and it is the same signal ``paragraphs.SHORT_LINE_RATIO`` uses -- shared
|
||||
deliberately, so this pass and ``breaks_paragraph`` cannot reach opposite
|
||||
conclusions about the same pair of lines.
|
||||
|
||||
The margin is measured from blocks sharing ``prev``'s left edge rather than
|
||||
from the whole page. Measuring page-wide would put every left-column line of
|
||||
a two-column layout far short of the right column's margin and block every
|
||||
legitimate merge there.
|
||||
|
||||
When too few peers exist to measure, the answer is ``False`` -- do not merge.
|
||||
Being alone at a left edge is not missing information, it is information: a
|
||||
block that no other block lines up with is not part of a column of prose, so
|
||||
it is a centred line, a right-aligned line or a one-off inset, and none of
|
||||
those wrap into the block below. The opposite default let a centred line
|
||||
swallow the right-aligned line beneath it, since neither had peers.
|
||||
|
||||
``prev`` is excluded from the vote. A line cannot be evidence for its own
|
||||
margin: when it was the widest of its peers it set the margin it was then
|
||||
measured against, so the test returned "reached the margin" unconditionally.
|
||||
That is how a right-aligned line -- necessarily the widest thing at its own
|
||||
left edge -- absorbed the paragraph beneath it.
|
||||
"""
|
||||
x0 = prev.bbox.x if prev.bbox else None
|
||||
if x0 is None:
|
||||
return False
|
||||
peers = [
|
||||
b
|
||||
for b in blocks
|
||||
if b is not prev
|
||||
and b.type == BlockType.paragraph
|
||||
and b.bbox
|
||||
and b.bbox.w
|
||||
and not b.synthetic_geometry
|
||||
and abs((b.bbox.x or 0.0) - x0) <= _WRAP_COLUMN_SNAP
|
||||
]
|
||||
if len(peers) < 2:
|
||||
return False
|
||||
rights = sorted((b.bbox.x or 0.0) + (b.bbox.w or 0.0) for b in peers)
|
||||
col_right = rights[min(len(rights) - 1, int(len(rights) * 0.90))]
|
||||
col_width = max(col_right - x0, 1.0)
|
||||
prev_right = (prev.bbox.x or 0.0) + (prev.bbox.w or 0.0)
|
||||
return (col_right - prev_right) <= col_width * SHORT_LINE_RATIO
|
||||
|
||||
|
||||
def _coalesce_short_line_paragraphs(page: Page, *, aggressive: bool) -> None:
|
||||
blocks = sorted(page.blocks, key=lambda b: b.reading_order)
|
||||
if len(blocks) < 2:
|
||||
return
|
||||
out: list[Block] = []
|
||||
i = 0
|
||||
while i < len(blocks):
|
||||
cur = blocks[i]
|
||||
if cur.type != BlockType.paragraph:
|
||||
out.append(cur)
|
||||
i += 1
|
||||
continue
|
||||
run = [cur]
|
||||
i += 1
|
||||
while i < len(blocks):
|
||||
nxt = blocks[i]
|
||||
if nxt.type != BlockType.paragraph:
|
||||
break
|
||||
prev = run[-1]
|
||||
prev_t = prev.plain_text().strip()
|
||||
nxt_t = nxt.plain_text().strip()
|
||||
if not prev_t or not nxt_t:
|
||||
break
|
||||
# Don't glue a new sentence that already looks complete+complete
|
||||
# unless continuous mode (harder merge).
|
||||
size_ok = abs(_block_size(prev) - _block_size(nxt)) / max(
|
||||
_block_size(prev), 1.0
|
||||
) <= (CONTINUOUS_SIZE_TOL if aggressive else SIZE_TOLERANCE)
|
||||
# A change of typeface is a change of role. Without this the pass
|
||||
# undid the geometry path's own decision: ``breaks_paragraph``
|
||||
# splits on a font change, and this loop then welded the halves
|
||||
# back together, so a bold Cambria heading and the Calibri line
|
||||
# under it reached Word as a single run-on paragraph. Size alone
|
||||
# cannot catch it -- a 14pt heading above 16pt body is within
|
||||
# CONTINUOUS_SIZE_TOL, and the heading is the *smaller* of the two.
|
||||
prev_font, nxt_font = _block_font(prev), _block_font(nxt)
|
||||
font_ok = not (prev_font and nxt_font and prev_font != nxt_font)
|
||||
short_prev = len(prev_t.split()) <= (28 if aggressive else 14)
|
||||
short_nxt = len(nxt_t.split()) <= (28 if aggressive else 16)
|
||||
has_geometry = bool(
|
||||
prev.bbox
|
||||
and nxt.bbox
|
||||
and prev.bbox.h
|
||||
and nxt.bbox.h
|
||||
and not prev.synthetic_geometry
|
||||
and not nxt.synthetic_geometry
|
||||
)
|
||||
if has_geometry:
|
||||
dy = abs(prev.bbox.y - nxt.bbox.y)
|
||||
y_close = dy < max(prev.bbox.h, nxt.bbox.h, 12) * (3.2 if aggressive else 2.2)
|
||||
# Wrapped lines of one paragraph share a left edge. The old
|
||||
# tolerance was max(80pt, prev_width * 1.4) -- on a 310pt-wide
|
||||
# block that permitted a 434pt jump, so a centred line and the
|
||||
# right-aligned line under it were welded despite starting 94pt
|
||||
# apart. A first-line indent moves the *second* line left, never
|
||||
# right, so a rightward jump beyond a couple of points means the
|
||||
# two blocks are not one wrapping paragraph.
|
||||
nxt_x = nxt.bbox.x or 0.0
|
||||
prev_x = prev.bbox.x or 0.0
|
||||
if nxt_x - prev_x > _WRAP_LEFT_TOL or prev_x - nxt_x > max(
|
||||
80.0, (prev.bbox.w or 40) * 1.4
|
||||
):
|
||||
break
|
||||
# Proximity alone is not wrap evidence, and treating it as such
|
||||
# is what made this pass overturn the geometry path wholesale.
|
||||
# ``breaks_paragraph`` separates a line that stops short of its
|
||||
# column margin, because that is what the *last* line of a
|
||||
# paragraph looks like; this loop then rejoined it to whatever
|
||||
# followed within ~2 line heights, which on any document written
|
||||
# one line per paragraph is everything. A 7-line test page came
|
||||
# back as 3 run-on blobs grouped only by point size, and the
|
||||
# heading among them was absorbed and lost. Requiring the line
|
||||
# to have actually reached its margin is the same wrap signal
|
||||
# ``paragraphs.SHORT_LINE_RATIO`` encodes, so the two passes now
|
||||
# agree instead of cancelling out.
|
||||
if not aggressive and not _reached_wrap_margin(blocks, prev):
|
||||
break
|
||||
else:
|
||||
# No geometry at all — the page came through the plain-text
|
||||
# fallback, where every extracted token lands on its own
|
||||
# fabricated line. Absence of geometry is not evidence that two
|
||||
# lines sit close together, and treating it as such merged a
|
||||
# whole ruled table ("Name / Qty / Cost / Alpha / 3 / 9") into
|
||||
# one paragraph. With nothing to measure, only a textual wrap
|
||||
# signal justifies joining: the previous line must not finish a
|
||||
# sentence and the next must continue one.
|
||||
y_close = _reads_as_wrapped(prev_t, nxt_t)
|
||||
if not (size_ok and font_ok and short_prev and short_nxt and y_close):
|
||||
break
|
||||
if (not aggressive) and prev_t[-1:] in ".!?۔؟" and nxt_t[:1].isupper():
|
||||
break
|
||||
run.append(nxt)
|
||||
i += 1
|
||||
if len(run) == 1:
|
||||
out.append(run[0])
|
||||
continue
|
||||
texts = [b.plain_text().strip() for b in run]
|
||||
joined = " ".join(texts)
|
||||
spans: list[TextSpan] = []
|
||||
for j, b in enumerate(run):
|
||||
if j and spans:
|
||||
spans.append(TextSpan(text=" "))
|
||||
if b.spans:
|
||||
spans.extend(b.spans)
|
||||
else:
|
||||
spans.append(TextSpan(text=b.plain_text()))
|
||||
first = run[0]
|
||||
out.append(
|
||||
Block(
|
||||
type=BlockType.paragraph,
|
||||
text=joined,
|
||||
spans=spans,
|
||||
bbox=first.bbox,
|
||||
reading_order=first.reading_order,
|
||||
source=first.source,
|
||||
align=first.align,
|
||||
)
|
||||
)
|
||||
for i, b in enumerate(out):
|
||||
b.reading_order = i
|
||||
page.blocks = out
|
||||
|
||||
|
||||
def _demote_heading_density(page: Page, body_size: float) -> None:
|
||||
bodyish = [b for b in page.blocks if _is_bodyish(b)]
|
||||
headings = [b for b in bodyish if b.type == BlockType.heading]
|
||||
if not bodyish or not headings:
|
||||
return
|
||||
if len(headings) / len(bodyish) <= HEADING_DENSITY_CAP and len(headings) < 8:
|
||||
return
|
||||
# Demote headings whose size is at/under body * 1.15 (banner / OCR bbox).
|
||||
for b in headings:
|
||||
if b.level >= 1 and _block_size(b) <= body_size * 1.18:
|
||||
b.type = BlockType.paragraph
|
||||
b.level = 0
|
||||
|
||||
|
||||
def _single_column_order(page: Page) -> None:
|
||||
"""nocolumns: reading order is top-to-bottom, then left-to-right (or RTL)."""
|
||||
from app.services.convert.text.arabic_logical import contains_arabic
|
||||
|
||||
def key(b: Block) -> tuple:
|
||||
y = -(b.bbox.y if b.bbox else 0.0)
|
||||
x = b.bbox.x if b.bbox else 0.0
|
||||
rtl = contains_arabic(b.plain_text())
|
||||
return (round(y / 8.0), -x if rtl else x)
|
||||
|
||||
page.blocks.sort(key=key)
|
||||
for i, b in enumerate(page.blocks):
|
||||
b.reading_order = i
|
||||
|
||||
|
||||
|
||||
def _unglue_section_numbers(document: Document) -> None:
|
||||
"""Restore the lost space in ``4.4.1.3United Arab Emirates``.
|
||||
|
||||
The number and the title are separate text runs in the source, and a run
|
||||
boundary carries no space, so a numbered clause heading arrives glued. It
|
||||
then fails section-number detection, is not recognised as a heading, and
|
||||
reads wrong in every output format.
|
||||
|
||||
The block's own text and its first span are both repaired: formatters emit
|
||||
spans when a block has them and fall back to ``text`` when it does not, so
|
||||
fixing only one leaves the other stale.
|
||||
"""
|
||||
for page in document.pages:
|
||||
for block in page.blocks:
|
||||
if block.type not in (BlockType.heading, BlockType.paragraph, BlockType.list_item):
|
||||
continue
|
||||
spans = getattr(block, "spans", None) or []
|
||||
if spans and spans[0].text:
|
||||
fixed = split_glued_section_number(spans[0].text)
|
||||
if fixed != spans[0].text:
|
||||
spans[0].text = fixed
|
||||
if block.text:
|
||||
block.text = split_glued_section_number(block.text)
|
||||
|
||||
|
||||
|
||||
# A run of this many consecutive body-size lines opening with a bare integer is
|
||||
# a numbered list or a table column, not a section hierarchy.
|
||||
MIN_NUMBERED_RUN = 3
|
||||
# Sizes within this fraction of each other count as "the same size".
|
||||
NUMBERED_RUN_SIZE_TOL = 0.12
|
||||
|
||||
_BARE_NUMBER_RE = re.compile(r"^(\d{1,3})[.)]?\s+\S")
|
||||
|
||||
|
||||
def _bare_row_number(text: str) -> int | None:
|
||||
"""The leading integer of ``1 Mooring inspection``, or None.
|
||||
|
||||
Only a flat integer counts. Hierarchical numbering (``4.2``, ``4.2.1``) is
|
||||
genuine section numbering and is left alone.
|
||||
"""
|
||||
m = _BARE_NUMBER_RE.match((text or "").strip())
|
||||
if not m:
|
||||
return None
|
||||
return int(m.group(1))
|
||||
|
||||
|
||||
def _demote_numbered_row_runs(document: Document, body: float) -> None:
|
||||
"""Stop a numbered column of table rows becoming a column of headings.
|
||||
|
||||
Almost every scanned table has a narrow leading number column — S.No, line
|
||||
number, item, ref — and each row then reads exactly like a numbered
|
||||
section: ``1 Mooring inspection``. The text rule that recognises
|
||||
``1 Introduction`` as a heading cannot tell them apart from one line alone,
|
||||
and on a scan there is no font change to separate them either, because the
|
||||
whole page is one type size.
|
||||
|
||||
What does separate them is that headings do not arrive three-in-a-row at
|
||||
body size with consecutive numbers. When they do, they are rows.
|
||||
"""
|
||||
for page in document.pages:
|
||||
blocks = sorted(page.blocks, key=lambda b: b.reading_order)
|
||||
run: list[Block] = []
|
||||
|
||||
def _flush(run: list[Block]) -> None:
|
||||
if len(run) < MIN_NUMBERED_RUN:
|
||||
return
|
||||
for b in run:
|
||||
b.type = BlockType.paragraph
|
||||
b.level = 0
|
||||
|
||||
for block in blocks:
|
||||
number = (
|
||||
_bare_row_number(block.plain_text())
|
||||
if block.type == BlockType.heading
|
||||
else None
|
||||
)
|
||||
size = _block_size(block)
|
||||
body_sized = abs(size - body) / max(body, 1.0) <= NUMBERED_RUN_SIZE_TOL
|
||||
previous = _bare_row_number(run[-1].plain_text()) if run else None
|
||||
if number is None or not body_sized:
|
||||
_flush(run)
|
||||
run = []
|
||||
continue
|
||||
if previous is not None and number != previous + 1:
|
||||
_flush(run)
|
||||
run = [block]
|
||||
continue
|
||||
run.append(block)
|
||||
_flush(run)
|
||||
|
||||
|
||||
def _collapsed_span(spans: list[TextSpan], text: str) -> TextSpan:
|
||||
"""``text`` as one span, keeping the run style it is replacing.
|
||||
|
||||
The longest contributing span wins: after an echo collapse the surviving
|
||||
text is the content that span carried, so its font is the honest answer.
|
||||
``w`` is dropped because the old measured width describes the doubled text.
|
||||
"""
|
||||
lead = max(spans, key=lambda s: len(s.text or ""))
|
||||
return replace(lead, text=text, w=None)
|
||||
|
||||
|
||||
def _collapse_echoed_text(document: Document) -> None:
|
||||
from app.services.convert.layout.text_quality import collapse_repeated_span
|
||||
|
||||
for page in document.pages:
|
||||
for block in page.blocks:
|
||||
if block.cells:
|
||||
block.cells = [
|
||||
[collapse_repeated_span(c) for c in row] for row in block.cells
|
||||
]
|
||||
if block.text:
|
||||
block.text = collapse_repeated_span(block.text)
|
||||
if block.spans:
|
||||
original = "".join(s.text for s in block.spans)
|
||||
joined = collapse_repeated_span(original)
|
||||
# Compare stripped: ``collapse_repeated_span`` also trims, and
|
||||
# glyph runs recovered from a content stream almost always end
|
||||
# in a trailing space. Comparing raw made this branch fire on
|
||||
# essentially every paragraph of every PDF, and the replacement
|
||||
# span below carried no font -- so all font, size, bold and
|
||||
# italic information was discarded document-wide. It then read
|
||||
# as size 0.0 to _coalesce_short_line_paragraphs, whose
|
||||
# same-size test compares 0.0 against 0.0, so unrelated
|
||||
# neighbours were also welded into one paragraph: a 14pt
|
||||
# heading and the 16pt line beneath it came out as one run.
|
||||
if joined.strip() == original.strip():
|
||||
continue
|
||||
block.spans = [_collapsed_span(block.spans, joined)]
|
||||
block.text = joined
|
||||
|
||||
|
||||
def _drop_ocr_noise_blocks(document: Document) -> None:
|
||||
"""Remove recogniser specks that survived as their own paragraphs."""
|
||||
from app.services.convert.layout.text_quality import is_ocr_noise_line
|
||||
|
||||
for page in document.pages:
|
||||
kept: list[Block] = []
|
||||
for b in page.blocks:
|
||||
t = b.plain_text().strip()
|
||||
noisy = is_ocr_noise_line(t)
|
||||
if b.type == BlockType.table:
|
||||
kept.append(b)
|
||||
continue
|
||||
if b.type == BlockType.figure and b.image_png:
|
||||
# A picture with *no* caption is a picture, not a speck. The
|
||||
# OCR rebuild gives a recovered letterhead emblem an empty
|
||||
# caption deliberately — printing "[Logo]" would put a word in
|
||||
# the document nobody wrote — and ``is_ocr_noise_line("")`` is
|
||||
# True, so the emblem was deleted right here while the
|
||||
# warnings still reported it as placed in the Word header.
|
||||
# Only a figure whose caption is *itself* recogniser noise
|
||||
# ("@", ".r-\\^") is dropped.
|
||||
if t and noisy:
|
||||
continue
|
||||
kept.append(b)
|
||||
continue
|
||||
if noisy:
|
||||
continue
|
||||
kept.append(b)
|
||||
page.blocks = kept
|
||||
for i, b in enumerate(page.blocks):
|
||||
b.reading_order = i
|
||||
|
||||
|
||||
_ROW_KEY_RE = re.compile(
|
||||
r"^(section\s+\d+|\d{1,2}([.)]|[\s:])|total\b)",
|
||||
re.I,
|
||||
)
|
||||
|
||||
|
||||
def _is_table_row_key(text: str) -> bool:
|
||||
t = (text or "").strip()
|
||||
if not t:
|
||||
return False
|
||||
return bool(_ROW_KEY_RE.match(t))
|
||||
|
||||
|
||||
def _merge_wrapped_table_rows(cells: list[list[str]] | None) -> list[list[str]]:
|
||||
"""Join OCR line-wraps inside a grid (Section title split across rows).
|
||||
|
||||
The third test below used to be "the row starts with a lowercase letter",
|
||||
which is true of any table whose first column holds units, ids, or plain
|
||||
lowercase labels — a four-row grid of them collapsed into a single row of
|
||||
four run-together cells. A wrapped line is not merely lowercase: it is
|
||||
*incomplete*, and the visible sign of that is that it fills fewer cells
|
||||
than the row it continues. A row carrying a value in every column is a
|
||||
data row whatever letter it starts with.
|
||||
"""
|
||||
rows = [list(r) for r in (cells or [])]
|
||||
if len(rows) < 2:
|
||||
return rows
|
||||
width = max(len(r) for r in rows)
|
||||
norm = [r + [""] * (width - len(r)) for r in rows]
|
||||
|
||||
def _populated(row: list[str]) -> int:
|
||||
return sum(1 for c in row if (c or "").strip())
|
||||
|
||||
out = [norm[0]]
|
||||
for row in norm[1:]:
|
||||
left = (row[0] or "").strip()
|
||||
prev = out[-1]
|
||||
prev_left = (prev[0] or "").strip()
|
||||
wrap = (
|
||||
(not left)
|
||||
or (_is_table_row_key(prev_left) and not _is_table_row_key(left))
|
||||
or (
|
||||
left[:1].islower()
|
||||
and left[:1].isascii()
|
||||
and prev_left
|
||||
and _populated(row) < _populated(prev)
|
||||
)
|
||||
)
|
||||
if wrap:
|
||||
for i in range(width):
|
||||
a, b = (prev[i] or "").strip(), (row[i] or "").strip()
|
||||
prev[i] = (a + " " + b).strip() if b else a
|
||||
else:
|
||||
out.append(row)
|
||||
return out
|
||||
|
||||
|
||||
def _repair_wrapped_tables(document: Document) -> None:
|
||||
for page in document.pages:
|
||||
for block in page.blocks:
|
||||
if block.type == BlockType.table and block.cells and len(block.cells) >= 3:
|
||||
block.cells = _merge_wrapped_table_rows(block.cells)
|
||||
|
||||
|
||||
def _drop_repeated_short_stamps(document: Document) -> None:
|
||||
"""Drop leftover stamp/banner lines that repeat as body paragraphs."""
|
||||
from collections import Counter
|
||||
|
||||
texts: list[str] = []
|
||||
for page in document.pages:
|
||||
for b in page.blocks:
|
||||
if b.type in (BlockType.table, BlockType.figure, BlockType.header, BlockType.footer):
|
||||
continue
|
||||
t = b.plain_text().strip()
|
||||
if t:
|
||||
texts.append(t)
|
||||
drop = set()
|
||||
for t, n in Counter(texts).items():
|
||||
words = t.split()
|
||||
if n < 3 or not (2 <= len(words) <= 6):
|
||||
continue
|
||||
if any(ch.isdigit() for ch in t):
|
||||
continue
|
||||
if all(w[:1].isupper() or ("\u0600" <= w[:1] <= "\u06FF") for w in words):
|
||||
drop.add(t)
|
||||
if not drop:
|
||||
return
|
||||
for page in document.pages:
|
||||
page.blocks = [
|
||||
b
|
||||
for b in page.blocks
|
||||
if b.type in (BlockType.table, BlockType.figure, BlockType.header, BlockType.footer)
|
||||
or b.plain_text().strip() not in drop
|
||||
]
|
||||
for i, b in enumerate(page.blocks):
|
||||
b.reading_order = i
|
||||
|
||||
|
||||
def _drop_junk_ocr_tables(document: Document) -> None:
|
||||
"""Remove grids that are recogniser debris; demote the ones that are prose.
|
||||
|
||||
Three shapes used to be deleted outright. Two of them should be: cells that
|
||||
are pure recogniser junk, and a stamp remnant whose text is already in the
|
||||
document's header. The third — a grid whose first cell starts mid-phrase —
|
||||
is not debris at all. It is a *paragraph* the table detector cut into
|
||||
cells, and deleting it took real sentences out of the document before the
|
||||
enhanced-flow pass downstream could turn them back into prose. It is
|
||||
demoted to paragraphs here instead, so the words survive in every
|
||||
recognition mode.
|
||||
"""
|
||||
for page in document.pages:
|
||||
recognised = page.kind in (PageKind.scan, PageKind.hybrid)
|
||||
rebuilt: list[Block] = []
|
||||
for b in page.blocks:
|
||||
if b.type != BlockType.table:
|
||||
rebuilt.append(b)
|
||||
continue
|
||||
# Junk detection is about recogniser output. On a digital page
|
||||
# read from the text layer there is no recogniser, so its verdict
|
||||
# is a guess about content it was never designed to judge.
|
||||
from_ocr = (b.source or "") == "ocr" or recognised
|
||||
if from_ocr and (
|
||||
_table_cells_are_ocr_junk(b.cells) or _table_is_stamp_remnant(b.cells)
|
||||
):
|
||||
continue
|
||||
if _table_is_mid_sentence_fragment(b.cells):
|
||||
rebuilt.extend(flatten_table_to_paragraphs(b))
|
||||
continue
|
||||
rebuilt.append(b)
|
||||
page.blocks = rebuilt
|
||||
for i, b in enumerate(page.blocks):
|
||||
b.reading_order = i
|
||||
|
||||
|
||||
def optimize_document(document: Document) -> None:
|
||||
"""Apply reconstruction / recognition modes after per-page IDM + OCR."""
|
||||
opts = get_options()
|
||||
# ``exact`` is emitted as real OOXML frames by formatters.docx_exact. It is
|
||||
# no longer downgraded to flowing here, and the optimisations that assume a
|
||||
# reflowing document are skipped for it: on a positioned page every block
|
||||
# owns a rectangle, so merging neighbouring blocks into one paragraph
|
||||
# destroys the arrangement that was the whole reason to position them.
|
||||
exact = opts.layout_mode == LayoutMode.exact
|
||||
|
||||
meta = document.meta if document.meta is not None else {}
|
||||
meta["convert_options"] = opts.as_dict()
|
||||
document.meta = meta
|
||||
|
||||
_unglue_section_numbers(document)
|
||||
_collapse_echoed_text(document)
|
||||
|
||||
sizes = [_block_size(b) for p in document.pages for b in p.blocks if _is_bodyish(b)]
|
||||
body = dominant_body_size(sizes) if sizes else 12.0
|
||||
|
||||
for page in document.pages:
|
||||
if opts.layout_mode == LayoutMode.nocolumns:
|
||||
_single_column_order(page)
|
||||
if not exact:
|
||||
_coalesce_short_line_paragraphs(
|
||||
page, aggressive=opts.layout_mode == LayoutMode.continuous
|
||||
)
|
||||
_demote_heading_density(page, body)
|
||||
|
||||
_demote_numbered_row_runs(document, body)
|
||||
_drop_ocr_noise_blocks(document)
|
||||
_drop_repeated_short_stamps(document)
|
||||
_drop_junk_ocr_tables(document)
|
||||
_repair_wrapped_tables(document)
|
||||
|
||||
if not opts.detect_tables:
|
||||
_drop_all_tables(document)
|
||||
elif opts.recognition_mode == RecognitionMode.enhanced_flow:
|
||||
# Word-soup grids from OCR become paragraphs. This must not run on a
|
||||
# positioned page: a poster's panel grid is deliberate arrangement, not
|
||||
# a misread paragraph, and flattening it is how a brochure turns to mush.
|
||||
_flatten_soup_tables(document)
|
||||
# textbox / flow: keep grids as-is (no soup flatten)
|
||||
|
||||
if opts.detect_tables:
|
||||
_demote_implausible_tables(document)
|
||||
|
||||
|
||||
def _table_has_figure_caption(page: Page, table: Block) -> bool:
|
||||
"""Whether a visually table-like panel is explicitly captioned as a figure.
|
||||
|
||||
Layout models regularly label a legend, state diagram, or annotated code
|
||||
panel as a table because the labels form clean rows and columns. The
|
||||
document author provides stronger semantic evidence when a nearby caption
|
||||
explicitly says "Figure N". Preserve the panel's text as paragraphs in
|
||||
that case instead of exporting a misleading editable data table.
|
||||
|
||||
This deliberately does *not* match "Table N" captions, generic prose, or
|
||||
distant page text. The caption must sit directly below the candidate and
|
||||
overlap its horizontal footprint.
|
||||
"""
|
||||
box = table.bbox
|
||||
if box is None or box.w <= 0 or box.h <= 0:
|
||||
return False
|
||||
table_right = box.x + box.w
|
||||
minimum_overlap = min(72.0, box.w * 0.40)
|
||||
for candidate in page.blocks:
|
||||
if candidate is table or candidate.type is BlockType.table or candidate.bbox is None:
|
||||
continue
|
||||
caption = " ".join(candidate.plain_text().split())
|
||||
folded = caption.casefold()
|
||||
if folded.startswith("figure "):
|
||||
rest = folded[len("figure ") :]
|
||||
elif folded.startswith("fig. "):
|
||||
rest = folded[len("fig. ") :]
|
||||
elif folded.startswith("fig "):
|
||||
rest = folded[len("fig ") :]
|
||||
else:
|
||||
continue
|
||||
# Requiring an identifier avoids treating ordinary prose such as
|
||||
# "Figure out..." as a caption.
|
||||
if not rest or not (rest[0].isdigit() or rest[0].isalpha()):
|
||||
continue
|
||||
caption_box = candidate.bbox
|
||||
caption_top = caption_box.y + caption_box.h
|
||||
gap = box.y - caption_top
|
||||
if not -6.0 <= gap <= 54.0:
|
||||
continue
|
||||
overlap = min(table_right, caption_box.x + caption_box.w) - max(box.x, caption_box.x)
|
||||
if overlap >= minimum_overlap:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _demote_implausible_tables(document: Document) -> None:
|
||||
"""Last word on every grid in the document, however it got here.
|
||||
|
||||
Each detector checks its own output, but the passes that run afterwards
|
||||
build tables the detectors never saw: a table continued across a page
|
||||
boundary is merged with its continuation, a wrapped one is repaired, an
|
||||
ML region is forced into a grid. Those transforms leave shapes no detector
|
||||
would have accepted — a single row left behind by a merge, a worksheet
|
||||
paragraph gathered into one 485-character cell — and they reached Word
|
||||
looking like tables because nothing asked again.
|
||||
|
||||
Demotion, not deletion: the text becomes paragraphs, so a wrong answer
|
||||
here costs formatting rather than content.
|
||||
"""
|
||||
from app.services.convert.layout import table_plausibility
|
||||
|
||||
for page in document.pages:
|
||||
rebuilt: list[Block] = []
|
||||
changed = False
|
||||
figure_caption_demotions = 0
|
||||
for block in page.blocks:
|
||||
if block.type is not BlockType.table or not block.cells:
|
||||
rebuilt.append(block)
|
||||
continue
|
||||
figure_caption = _table_has_figure_caption(page, block)
|
||||
if not figure_caption and table_plausibility.assess(block.cells):
|
||||
rebuilt.append(block)
|
||||
continue
|
||||
changed = True
|
||||
figure_caption_demotions += int(figure_caption)
|
||||
rebuilt.extend(flatten_table_to_paragraphs(block))
|
||||
if changed:
|
||||
page.blocks = rebuilt
|
||||
for i, b in enumerate(page.blocks):
|
||||
b.reading_order = i
|
||||
if figure_caption_demotions:
|
||||
message = (
|
||||
f"Page {page.index + 1}: {figure_caption_demotions} table-like figure panel(s) "
|
||||
"were kept as text because of a Figure caption."
|
||||
)
|
||||
if message not in document.warnings:
|
||||
document.warnings.append(message)
|
||||
@@ -0,0 +1,503 @@
|
||||
"""Page image / display-list helpers for classification and table rulings."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
from contextlib import suppress
|
||||
from dataclasses import dataclass
|
||||
from typing import Any
|
||||
|
||||
# Identity transform, in PDF's [a b c d e f] form.
|
||||
_IDENTITY = (1.0, 0.0, 0.0, 1.0, 0.0, 0.0)
|
||||
|
||||
# Below this on-page size (points) an image is furniture — a rule, a bullet, a
|
||||
# spacer — not a figure worth carrying into the output.
|
||||
MIN_FIGURE_POINTS = 12.0
|
||||
|
||||
|
||||
def _mat_mul(m: tuple, n: tuple) -> tuple:
|
||||
"""PDF matrix product m x n (row-vector convention)."""
|
||||
a1, b1, c1, d1, e1, f1 = m
|
||||
a2, b2, c2, d2, e2, f2 = n
|
||||
return (
|
||||
a1 * a2 + b1 * c2,
|
||||
a1 * b2 + b1 * d2,
|
||||
c1 * a2 + d1 * c2,
|
||||
c1 * b2 + d1 * d2,
|
||||
e1 * a2 + f1 * c2 + e2,
|
||||
e1 * b2 + f1 * d2 + f2,
|
||||
)
|
||||
|
||||
|
||||
def _apply(m: tuple, x: float, y: float) -> tuple[float, float]:
|
||||
a, b, c, d, e, f = m
|
||||
return (a * x + c * y + e, b * x + d * y + f)
|
||||
|
||||
|
||||
@dataclass
|
||||
class ImagePlacement:
|
||||
"""Where an image XObject is actually painted on the page.
|
||||
|
||||
``name`` is the XObject resource name, so the placement can be matched back
|
||||
to the image data. Coordinates are PDF user space (origin bottom-left).
|
||||
"""
|
||||
|
||||
name: str
|
||||
x: float
|
||||
y: float
|
||||
w: float
|
||||
h: float
|
||||
px_w: int = 0
|
||||
px_h: int = 0
|
||||
|
||||
@property
|
||||
def area(self) -> float:
|
||||
return max(self.w, 0.0) * max(self.h, 0.0)
|
||||
|
||||
|
||||
def image_placements(page) -> list[ImagePlacement]:
|
||||
"""True on-page rectangles for every image drawn by this page.
|
||||
|
||||
Memoised on the page object: the coverage estimate and the figure
|
||||
extractor both need it, and walking the content stream twice per page is
|
||||
pure waste.
|
||||
|
||||
An image XObject is painted into the unit square, so the CTM in effect at
|
||||
its ``Do`` operator *is* its placement. Walking ``q``/``Q``/``cm``/``Do``
|
||||
recovers that exactly.
|
||||
|
||||
This replaces guessing placement from pixel dimensions, which produced a
|
||||
fabricated rectangle: a 1200x1200 logo and a full-page scan have similar
|
||||
pixel counts but completely different footprints, and treating the logo as
|
||||
a full-page raster silently dropped it from the output.
|
||||
"""
|
||||
cached = getattr(page, "_dq_image_placements", None)
|
||||
if cached is not None:
|
||||
return cached
|
||||
|
||||
out: list[ImagePlacement] = []
|
||||
try:
|
||||
from pypdf.generic import ContentStream
|
||||
|
||||
resources = page.get("/Resources")
|
||||
xobjects = {}
|
||||
if resources is not None:
|
||||
xo = resources.get_object().get("/XObject")
|
||||
if xo is not None:
|
||||
xobjects = xo.get_object()
|
||||
if not xobjects:
|
||||
return out
|
||||
|
||||
contents = page.get_contents()
|
||||
if contents is None:
|
||||
return out
|
||||
stream = ContentStream(contents, getattr(page, "pdf", None))
|
||||
|
||||
ctm = _IDENTITY
|
||||
stack: list[tuple] = []
|
||||
for operands, operator in stream.operations:
|
||||
if operator == b"q":
|
||||
stack.append(ctm)
|
||||
elif operator == b"Q":
|
||||
ctm = stack.pop() if stack else _IDENTITY
|
||||
elif operator == b"cm" and len(operands) >= 6:
|
||||
try:
|
||||
ctm = _mat_mul(tuple(float(v) for v in operands[:6]), ctm)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
elif operator == b"Do" and operands:
|
||||
name = str(operands[0])
|
||||
obj = xobjects.get(name)
|
||||
if obj is None:
|
||||
continue
|
||||
try:
|
||||
obj = obj.get_object()
|
||||
except Exception:
|
||||
continue
|
||||
if obj.get("/Subtype") != "/Image":
|
||||
continue
|
||||
corners = [_apply(ctm, cx, cy) for cx, cy in ((0, 0), (1, 0), (1, 1), (0, 1))]
|
||||
xs = [c[0] for c in corners]
|
||||
ys = [c[1] for c in corners]
|
||||
out.append(
|
||||
ImagePlacement(
|
||||
name=name,
|
||||
x=min(xs),
|
||||
y=min(ys),
|
||||
w=max(xs) - min(xs),
|
||||
h=max(ys) - min(ys),
|
||||
px_w=int(obj.get("/Width", 0) or 0),
|
||||
px_h=int(obj.get("/Height", 0) or 0),
|
||||
)
|
||||
)
|
||||
except Exception:
|
||||
return out
|
||||
with suppress(Exception):
|
||||
page._dq_image_placements = out
|
||||
return out
|
||||
|
||||
|
||||
def image_blocks_from_pypdf_page(page) -> list[dict]:
|
||||
"""Image footprints on a pypdf page, in PDF user-space units.
|
||||
|
||||
Uses the real content-stream placement, falling back to the old pixel-ratio
|
||||
approximation only when the content stream cannot be walked.
|
||||
"""
|
||||
placements = image_placements(page)
|
||||
if placements:
|
||||
return [
|
||||
{"x": p.x, "y": p.y, "w": p.w, "h": p.h, "width": p.w, "height": p.h, "name": p.name}
|
||||
for p in placements
|
||||
]
|
||||
|
||||
blocks: list[dict] = []
|
||||
try:
|
||||
if "/Resources" not in page or "/XObject" not in page["/Resources"]:
|
||||
return blocks
|
||||
xobjects = page["/Resources"]["/XObject"].get_object()
|
||||
mediabox = page.mediabox
|
||||
page_w = float(mediabox.width)
|
||||
page_h = float(mediabox.height)
|
||||
for _name in xobjects:
|
||||
obj = xobjects[_name]
|
||||
if obj.get("/Subtype") != "/Image":
|
||||
continue
|
||||
w = float(obj.get("/Width", page_w))
|
||||
h = float(obj.get("/Height", page_h))
|
||||
# Without CTM we approximate: large images cover most of the page
|
||||
scale = min(page_w / max(w, 1), page_h / max(h, 1))
|
||||
blocks.append({"w": w * scale, "h": h * scale, "width": w * scale, "height": h * scale})
|
||||
except Exception:
|
||||
return blocks
|
||||
return blocks
|
||||
|
||||
|
||||
def image_blocks_from_engine_page(page: Any) -> list[dict]:
|
||||
blocks: list[dict] = []
|
||||
for meth in ("extract_images", "list_images", "get_images"):
|
||||
if not hasattr(page, meth):
|
||||
continue
|
||||
try:
|
||||
raw = getattr(page, meth)()
|
||||
except Exception:
|
||||
continue
|
||||
if not raw:
|
||||
continue
|
||||
for item in raw:
|
||||
if isinstance(item, dict):
|
||||
blocks.append(item)
|
||||
else:
|
||||
blocks.append(
|
||||
{
|
||||
"w": float(getattr(item, "w", getattr(item, "width", 0)) or 0),
|
||||
"h": float(getattr(item, "h", getattr(item, "height", 0)) or 0),
|
||||
}
|
||||
)
|
||||
if blocks:
|
||||
return blocks
|
||||
return blocks
|
||||
|
||||
|
||||
def display_list_ops_from_engine_page(page: Any) -> list[dict]:
|
||||
if not hasattr(page, "extract_display_list"):
|
||||
return []
|
||||
try:
|
||||
raw = page.extract_display_list()
|
||||
except Exception:
|
||||
return []
|
||||
if isinstance(raw, str):
|
||||
raw_str = raw.strip()
|
||||
if raw_str.startswith(("[", "{")):
|
||||
try:
|
||||
import json
|
||||
raw = json.loads(raw_str)
|
||||
except Exception:
|
||||
pass
|
||||
if isinstance(raw, list):
|
||||
ops: list[dict] = []
|
||||
for x in raw:
|
||||
if isinstance(x, dict):
|
||||
op_dict = dict(x)
|
||||
if "args" in op_dict and isinstance(op_dict["args"], (list, tuple)):
|
||||
args = op_dict["args"]
|
||||
if len(args) >= 4 and str(op_dict.get("op") or "").lower() in ("re", "rect"):
|
||||
try:
|
||||
rx, ry, rw, rh = float(args[0]), float(args[1]), float(args[2]), float(args[3])
|
||||
op_dict.setdefault("x", rx)
|
||||
op_dict.setdefault("y", ry)
|
||||
op_dict.setdefault("w", rw)
|
||||
op_dict.setdefault("h", rh)
|
||||
op_dict.setdefault("x0", rx)
|
||||
op_dict.setdefault("y0", ry)
|
||||
op_dict.setdefault("x1", rx + rw)
|
||||
op_dict.setdefault("y1", ry + rh)
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
ops.append(op_dict)
|
||||
else:
|
||||
ops.append({"type": "line"})
|
||||
return ops
|
||||
if isinstance(raw, str):
|
||||
# Best-effort parse of simple "line x0 y0 x1 y1" lines
|
||||
ops: list[dict] = []
|
||||
for line in raw.splitlines():
|
||||
parts = line.strip().split()
|
||||
if len(parts) >= 5 and parts[0].lower() in ("line", "vline"):
|
||||
try:
|
||||
ops.append(
|
||||
{
|
||||
"type": parts[0].lower(),
|
||||
"x0": float(parts[1]),
|
||||
"y0": float(parts[2]),
|
||||
"x1": float(parts[3]),
|
||||
"y1": float(parts[4]),
|
||||
}
|
||||
)
|
||||
except ValueError:
|
||||
continue
|
||||
return ops
|
||||
return []
|
||||
|
||||
|
||||
def _colorspace_mode(cs) -> str | None:
|
||||
"""Map PDF ColorSpace to Pillow mode when possible."""
|
||||
if cs is None:
|
||||
return "RGB"
|
||||
name = str(cs)
|
||||
if "DeviceGray" in name or name.endswith("/DeviceGray"):
|
||||
return "L"
|
||||
if "DeviceRGB" in name or name.endswith("/DeviceRGB"):
|
||||
return "RGB"
|
||||
if "DeviceCMYK" in name or name.endswith("/DeviceCMYK"):
|
||||
return "CMYK"
|
||||
# ICCBased / Indexed: try RGB bytes if Length matches later
|
||||
if "ICCBased" in name:
|
||||
return "RGB"
|
||||
return None
|
||||
|
||||
|
||||
def _decode_xobject_png(obj) -> bytes | None:
|
||||
"""One image XObject as PNG bytes, or None when it cannot be decoded."""
|
||||
from PIL import Image
|
||||
from pypdf.generic import NameObject
|
||||
|
||||
width = int(obj.get("/Width", 0))
|
||||
height = int(obj.get("/Height", 0))
|
||||
try:
|
||||
data = obj.get_data()
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
filt = obj.get("/Filter")
|
||||
img = None
|
||||
try:
|
||||
if filt == "/DCTDecode" or (isinstance(filt, list) and NameObject("/DCTDecode") in filt):
|
||||
img = Image.open(io.BytesIO(data))
|
||||
elif filt == "/FlateDecode" or str(filt) == "/FlateDecode" or (
|
||||
isinstance(filt, list) and any("Flate" in str(f) for f in filt)
|
||||
):
|
||||
mode = _colorspace_mode(obj.get("/ColorSpace")) or "RGB"
|
||||
expected = width * height * (1 if mode == "L" else 4 if mode == "CMYK" else 3)
|
||||
if len(data) < expected:
|
||||
# Sometimes get_data already expands; try open as encoded
|
||||
try:
|
||||
img = Image.open(io.BytesIO(data))
|
||||
except Exception:
|
||||
return None
|
||||
else:
|
||||
img = Image.frombytes(mode, (width, height), data[:expected])
|
||||
else:
|
||||
img = Image.open(io.BytesIO(data))
|
||||
except Exception:
|
||||
return None
|
||||
if img is None:
|
||||
return None
|
||||
|
||||
buf = io.BytesIO()
|
||||
img.convert("RGB").save(buf, format="PNG")
|
||||
return buf.getvalue()
|
||||
|
||||
|
||||
def decode_page_images(
|
||||
page, *, warnings: list[str] | None = None, only: set[str] | None = None
|
||||
) -> dict[str, bytes]:
|
||||
"""Decodable image XObjects on the page, keyed by resource name.
|
||||
|
||||
Keyed by name so a placement recovered from the content stream can be
|
||||
matched to its pixels: a page with a chart and a photo needs both, and the
|
||||
old "largest image only" extraction returned one of them.
|
||||
|
||||
``only`` restricts decoding to the named XObjects. Decoding is the
|
||||
expensive part, so a caller that already knows which images it will keep
|
||||
should say so rather than decoding a full-page raster to discard it.
|
||||
"""
|
||||
out: dict[str, bytes] = {}
|
||||
try:
|
||||
if "/Resources" not in page or "/XObject" not in page["/Resources"]:
|
||||
return out
|
||||
xobjects = page["/Resources"]["/XObject"].get_object()
|
||||
image_count = 0
|
||||
for name in xobjects:
|
||||
if only is not None and str(name) not in only:
|
||||
continue
|
||||
obj = xobjects[name]
|
||||
if obj.get("/Subtype") != "/Image":
|
||||
continue
|
||||
image_count += 1
|
||||
png = _decode_xobject_png(obj)
|
||||
if png:
|
||||
out[str(name)] = png
|
||||
if not out and image_count and warnings is not None:
|
||||
warnings.append(
|
||||
f"page has {image_count} image XObject(s) but decode failed "
|
||||
f"({image_count} failure(s))."
|
||||
)
|
||||
except Exception as exc:
|
||||
if warnings is not None:
|
||||
warnings.append(f"embedded image extract failed: {exc}")
|
||||
return out
|
||||
|
||||
|
||||
def embedded_page_image_png(page, *, warnings: list[str] | None = None) -> bytes | None:
|
||||
"""Largest embedded image on a pypdf page as PNG bytes, if any.
|
||||
|
||||
"Largest" is by pixel area, as the OCR path needs the highest-resolution
|
||||
raster — not by encoded size, which favours noisy images over big ones.
|
||||
"""
|
||||
images = decode_page_images(page, warnings=warnings)
|
||||
if not images:
|
||||
return None
|
||||
|
||||
def _area(png: bytes) -> int:
|
||||
try:
|
||||
from PIL import Image
|
||||
|
||||
with Image.open(io.BytesIO(png)) as im:
|
||||
return im.width * im.height
|
||||
except Exception:
|
||||
return len(png)
|
||||
|
||||
return max(images.values(), key=_area)
|
||||
|
||||
|
||||
def figure_blocks_from_pypdf_page(
|
||||
page,
|
||||
*,
|
||||
reading_order: int = 0,
|
||||
warnings: list[str] | None = None,
|
||||
max_page_coverage: float = 0.38,
|
||||
max_bytes: int = 450_000,
|
||||
) -> list:
|
||||
"""Extract embedded images as IDM figure blocks (digital path).
|
||||
|
||||
Every image on the page is emitted, at the rectangle the content stream
|
||||
actually paints it into. Two things this fixes:
|
||||
|
||||
*multiple figures per page*
|
||||
The previous implementation extracted only the single largest image, so
|
||||
a page with a chart and a photograph lost one of them outright.
|
||||
|
||||
*real placement instead of a fabricated box*
|
||||
Position and size came from a hard-coded guess, which destroyed both
|
||||
reading order relative to the text and the aspect ratio. Coverage was
|
||||
likewise inferred from a pixel-count ladder, so a high-resolution logo
|
||||
was mistaken for a full-page scan and dropped.
|
||||
|
||||
Near-full-page rasters are still skipped: those belong to OCR/scan handling
|
||||
and would otherwise embed a multi-MB page image per page.
|
||||
"""
|
||||
from app.services.convert.idm.model import BBox, Block, BlockType
|
||||
|
||||
out: list = []
|
||||
try:
|
||||
mediabox = page.mediabox
|
||||
pw, ph = float(mediabox.width), float(mediabox.height)
|
||||
page_area = max(pw * ph, 1.0)
|
||||
|
||||
placements = image_placements(page)
|
||||
|
||||
if not placements:
|
||||
# No usable content stream: fall back to the single largest image
|
||||
# with a conservative centred box rather than dropping everything.
|
||||
png = embedded_page_image_png(page, warnings=warnings)
|
||||
if not png or len(png) > max_bytes:
|
||||
return out
|
||||
out.append(
|
||||
Block(
|
||||
type=BlockType.figure,
|
||||
text="[Image]",
|
||||
bbox=BBox(x=72, y=ph * 0.3, w=min(pw - 144, 400), h=min(ph * 0.4, 300)),
|
||||
image_png=png,
|
||||
reading_order=reading_order,
|
||||
)
|
||||
)
|
||||
return out
|
||||
|
||||
# Decide from geometry *before* decoding. Decoding a full-page raster
|
||||
# only to discard it costs seconds per page on an image-heavy document,
|
||||
# and the placement already says it will be discarded.
|
||||
keep: list[ImagePlacement] = []
|
||||
skipped_large = 0
|
||||
for placement in placements:
|
||||
if placement.w < MIN_FIGURE_POINTS or placement.h < MIN_FIGURE_POINTS:
|
||||
# Rules, bullets and spacer pixels are not figures.
|
||||
continue
|
||||
if placement.area / page_area >= max_page_coverage:
|
||||
skipped_large += 1
|
||||
continue
|
||||
keep.append(placement)
|
||||
|
||||
if keep:
|
||||
wanted = {p.name for p in keep}
|
||||
images = decode_page_images(page, warnings=warnings, only=wanted)
|
||||
# Top-to-bottom, then left-to-right: the order a reader meets them.
|
||||
keep.sort(key=lambda p: (-(p.y + p.h), p.x))
|
||||
for offset, placement in enumerate(keep):
|
||||
png = images.get(placement.name)
|
||||
if not png:
|
||||
continue
|
||||
if len(png) > max_bytes:
|
||||
skipped_large += 1
|
||||
continue
|
||||
out.append(
|
||||
Block(
|
||||
type=BlockType.figure,
|
||||
text="[Image]",
|
||||
bbox=BBox(x=placement.x, y=placement.y, w=placement.w, h=placement.h),
|
||||
image_png=png,
|
||||
reading_order=reading_order + offset,
|
||||
)
|
||||
)
|
||||
if skipped_large and warnings is not None:
|
||||
warnings.append(
|
||||
f"Skipped {skipped_large} large/full-page embedded image(s); prefer OCR text."
|
||||
)
|
||||
except Exception as exc:
|
||||
if warnings is not None:
|
||||
warnings.append(f"figure block extract failed: {exc}")
|
||||
return out
|
||||
return out
|
||||
|
||||
|
||||
def figure_blocks_from_engine_images(
|
||||
img_meta: list[dict], png_blobs: list[bytes], *, reading_order: int = 0
|
||||
) -> list:
|
||||
from app.services.convert.idm.model import BBox, Block, BlockType
|
||||
|
||||
out: list = []
|
||||
for i, png in enumerate(png_blobs):
|
||||
meta = img_meta[i] if i < len(img_meta) else {}
|
||||
w = float(meta.get("w") or meta.get("width") or 300)
|
||||
h = float(meta.get("h") or meta.get("height") or 200)
|
||||
x = float(meta.get("x", 72))
|
||||
y = float(meta.get("y", 200))
|
||||
out.append(
|
||||
Block(
|
||||
type=BlockType.figure,
|
||||
text=f"[Image {i + 1}]",
|
||||
bbox=BBox(x=x, y=y, w=w, h=h),
|
||||
image_png=png,
|
||||
reading_order=reading_order + i,
|
||||
)
|
||||
)
|
||||
return out
|
||||
@@ -0,0 +1,415 @@
|
||||
"""Masthead logo segmentation for pages that embed only full-page rasters.
|
||||
|
||||
Some scanned documents carry no separate image XObject at all: every page is
|
||||
one full-page bitmap, so the figure extractor correctly declines to embed any
|
||||
of it (a page-sized raster per page is not a figure, it is the page). The
|
||||
emblem at the top of page one is then lost entirely, even though it is the one
|
||||
piece of artwork a reader expects to survive.
|
||||
|
||||
This module recovers it by segmenting the raster instead of the PDF: ink in the
|
||||
masthead band that OCR did *not* claim as text is, by elimination, artwork.
|
||||
|
||||
Deliberately dependency-light — Pillow and numpy only, both already required by
|
||||
the OCR path. No OpenCV, no paid SDK.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
from dataclasses import dataclass
|
||||
|
||||
# At most this many emblems per masthead. A letterhead carries an emblem and
|
||||
# perhaps a wordmark or seal; more clusters than this means the band is text
|
||||
# the OCR mask failed to claim, not artwork.
|
||||
MAX_LOGOS = 3
|
||||
# A second mark must be at least this share of the largest one's area to be a
|
||||
# mark rather than a speck the text mask missed.
|
||||
MIN_RELATIVE_MARK_AREA = 0.08
|
||||
# Masthead band as a fraction of page height, measured from the top.
|
||||
BAND_RATIO = 0.22
|
||||
# A pixel this much darker than the page's paper white counts as ink.
|
||||
INK_DELTA = 42
|
||||
# Side strips of the masthead, as a fraction of page width. Letterhead marks
|
||||
# live here; a full-width OCR envelope for the banner must not decide them.
|
||||
SIDE_STRIP_RATIO = 0.36
|
||||
# Fainter ink still counts as a mark in a side strip — gold/grey emblems sit
|
||||
# closer to paper than black body text.
|
||||
SIDE_INK_DELTA = 18
|
||||
# Text boxes are grown by this fraction of their size before masking, so
|
||||
# antialiased edges of glyphs do not survive as stray ink.
|
||||
TEXT_PAD_RATIO = 0.14
|
||||
# A logo must occupy at least this fraction of the band's width and height.
|
||||
MIN_SIDE_RATIO = 0.025
|
||||
# ...and no more than this, or it is the whole banner rather than an emblem.
|
||||
MAX_WIDTH_RATIO = 0.55
|
||||
MAX_HEIGHT_RATIO = 0.95
|
||||
# A leftover banner stroke is a wide, short bar. Emblems are compact.
|
||||
MAX_MARK_ASPECT = 2.0
|
||||
# Below this ink density the crop is speckle, not artwork.
|
||||
MIN_INK_DENSITY = 0.04
|
||||
MIN_FAINT_INK_DENSITY = 0.022
|
||||
# Columns with less than this share of the band's peak ink are gaps.
|
||||
COLUMN_GAP_RATIO = 0.06
|
||||
# Output is scaled down to keep the DOCX small; a masthead needs no more.
|
||||
MAX_OUTPUT_SIDE = 420
|
||||
|
||||
|
||||
@dataclass
|
||||
class LogoCrop:
|
||||
"""One emblem cropped from a masthead, with where it sat on the raster."""
|
||||
|
||||
png: bytes
|
||||
# Pixel box on the page raster: (x, y, w, h), top-left origin.
|
||||
bbox_px: tuple[int, int, int, int]
|
||||
raster_size: tuple[int, int]
|
||||
|
||||
def bbox_points(self, page_width: float, page_height: float) -> tuple[float, float, float, float]:
|
||||
"""The same box in PDF points on a page of the given size.
|
||||
|
||||
Without this the recovered artwork reached the writer with no geometry
|
||||
at all, and every figure with no geometry was emitted at a hard-coded
|
||||
5.5 inches — which is how a two-centimetre emblem became a full-width
|
||||
banner across the top of the Word document.
|
||||
"""
|
||||
rw, rh = self.raster_size
|
||||
if rw <= 0 or rh <= 0:
|
||||
return 0.0, 0.0, 0.0, 0.0
|
||||
sx = float(page_width) / float(rw)
|
||||
sy = float(page_height) / float(rh)
|
||||
x, y, w, h = self.bbox_px
|
||||
# Raster y grows downwards, PDF y grows upwards.
|
||||
return (x * sx, page_height - (y + h) * sy, w * sx, h * sy)
|
||||
|
||||
|
||||
def extract_masthead_logos(
|
||||
page_png: bytes,
|
||||
text_boxes: list[tuple[float, float, float, float]] | None = None,
|
||||
*,
|
||||
band_ratio: float = BAND_RATIO,
|
||||
max_logos: int = MAX_LOGOS,
|
||||
) -> list[LogoCrop]:
|
||||
"""Crop every emblem from the top band of a page raster, left to right.
|
||||
|
||||
``text_boxes`` are OCR boxes in the same pixel space as ``page_png``, as
|
||||
``(x, y, w, h)``. Anything they cover is text and is masked out; the compact
|
||||
ink regions left in the band are artwork.
|
||||
|
||||
Letterheads very often carry two marks — an emblem on one side and a
|
||||
wordmark or seal on the other. Returning only the largest lost one of them
|
||||
and, worse, stretched the survivor across the page. Every qualifying cluster
|
||||
is returned, in reading order.
|
||||
|
||||
An empty list is the common case and must stay cheap and silent.
|
||||
"""
|
||||
try:
|
||||
import numpy as np
|
||||
from PIL import Image
|
||||
except Exception:
|
||||
return []
|
||||
|
||||
try:
|
||||
with Image.open(io.BytesIO(page_png)) as im:
|
||||
page = im.convert("L")
|
||||
width, height = page.size
|
||||
band_h = max(1, int(height * band_ratio))
|
||||
band = np.asarray(page.crop((0, 0, width, band_h)), dtype="int16")
|
||||
|
||||
if band.size == 0:
|
||||
return []
|
||||
|
||||
# Paper white is the band's brightest common value, not 255: a scan is
|
||||
# grey, and thresholding against pure white marks the whole page as ink.
|
||||
paper = int(np.percentile(band, 90))
|
||||
ink = band < (paper - INK_DELTA)
|
||||
ink_raw = ink.copy()
|
||||
faint = band < (paper - SIDE_INK_DELTA)
|
||||
|
||||
_mask_text(ink, text_boxes, band_h, page_width=width)
|
||||
clusters = _ink_regions(ink, np) if ink.any() else []
|
||||
out = _crops_from_boxes(clusters, page_png, width, height, band_h, ink, max_logos)
|
||||
|
||||
# Prefer one mark in each side strip from ink that the banner envelope
|
||||
# has not been allowed to erase. A letterhead is those two corners.
|
||||
out = _prefer_side_marks(
|
||||
out,
|
||||
_side_strip_crops(
|
||||
page_png, ink_raw, faint, width, height, band_h, np, text_boxes
|
||||
),
|
||||
width,
|
||||
max_logos,
|
||||
)
|
||||
return out
|
||||
except Exception:
|
||||
return []
|
||||
|
||||
|
||||
def extract_masthead_logo(
|
||||
page_png: bytes,
|
||||
text_boxes: list[tuple[float, float, float, float]] | None = None,
|
||||
*,
|
||||
band_ratio: float = BAND_RATIO,
|
||||
) -> bytes | None:
|
||||
"""The single most prominent emblem, as PNG bytes. Prefer the plural form."""
|
||||
crops = extract_masthead_logos(page_png, text_boxes, band_ratio=band_ratio, max_logos=MAX_LOGOS)
|
||||
if not crops:
|
||||
return None
|
||||
return max(crops, key=lambda c: c.bbox_px[2] * c.bbox_px[3]).png
|
||||
|
||||
|
||||
def _qualifies_as_mark(w: int, h: int, width: int, band_h: int) -> bool:
|
||||
if w < width * MIN_SIDE_RATIO or h < band_h * MIN_SIDE_RATIO:
|
||||
return False
|
||||
if w > width * MAX_WIDTH_RATIO or h > band_h * MAX_HEIGHT_RATIO:
|
||||
return False
|
||||
if h > 0 and (w / h) > MAX_MARK_ASPECT and h < band_h * 0.40:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def _crop_area(crop: LogoCrop) -> int:
|
||||
return crop.bbox_px[2] * crop.bbox_px[3]
|
||||
|
||||
|
||||
def _crops_from_boxes(boxes, page_png, width, height, band_h, ink, max_logos) -> list[LogoCrop]:
|
||||
out: list[LogoCrop] = []
|
||||
for x0, y0, x1, y1 in boxes:
|
||||
w, h = x1 - x0, y1 - y0
|
||||
if not _qualifies_as_mark(w, h, width, band_h):
|
||||
continue
|
||||
if ink[y0:y1, x0:x1].mean() < MIN_INK_DENSITY:
|
||||
continue
|
||||
png = _crop_png(page_png, (x0, y0, x1, y1))
|
||||
if not png:
|
||||
continue
|
||||
out.append(
|
||||
LogoCrop(png=png, bbox_px=(x0, y0, w, h), raster_size=(width, height))
|
||||
)
|
||||
if len(out) >= max_logos:
|
||||
break
|
||||
return out
|
||||
|
||||
|
||||
def _best_in_strip(
|
||||
ink, x0: int, x1: int, page_png, width, height, band_h, np, *, min_density: float = MIN_INK_DENSITY
|
||||
) -> LogoCrop | None:
|
||||
if x1 <= x0 or ink.size == 0:
|
||||
return None
|
||||
strip = ink[:, x0:x1]
|
||||
if not strip.any():
|
||||
return None
|
||||
boxes = _ink_regions(strip, np)
|
||||
if not boxes:
|
||||
return None
|
||||
bx0, by0, bx1, by1 = max(boxes, key=lambda b: (b[2] - b[0]) * (b[3] - b[1]))
|
||||
abs0, abs1 = bx0 + x0, bx1 + x0
|
||||
w, h = abs1 - abs0, by1 - by0
|
||||
if not _qualifies_as_mark(w, h, width, band_h):
|
||||
return None
|
||||
if ink[by0:by1, abs0:abs1].mean() < min_density:
|
||||
return None
|
||||
png = _crop_png(page_png, (abs0, by0, abs1, by1))
|
||||
if not png:
|
||||
return None
|
||||
return LogoCrop(png=png, bbox_px=(abs0, by0, w, h), raster_size=(width, height))
|
||||
|
||||
|
||||
def _mask_strip_text(ink, text_boxes, x0: int, x1: int, band_h: int) -> None:
|
||||
"""Mask OCR boxes inside a side strip, with no centre carve-out.
|
||||
|
||||
Page-wide banner envelopes are skipped: those are the case the side pass
|
||||
exists to recover. A compact box in the strip is real text and must go.
|
||||
"""
|
||||
if not text_boxes:
|
||||
return
|
||||
rows, cols = ink.shape
|
||||
wide = cols * 0.55
|
||||
outer = cols * 0.16
|
||||
for bx, by, bw, bh in text_boxes:
|
||||
if bw <= 0 or bh <= 0 or bw >= wide:
|
||||
continue
|
||||
if bh < band_h * 0.28 and bw > cols * 0.35:
|
||||
continue
|
||||
cx = bx + bw / 2
|
||||
if cx < outer or cx > cols - outer:
|
||||
continue
|
||||
sx0 = max(x0, 0, int(bx))
|
||||
sx1 = min(x1, cols, int(bx + bw))
|
||||
y0 = max(0, int(by))
|
||||
y1 = min(rows, int(by + bh), band_h)
|
||||
if sx1 > sx0 and y1 > y0:
|
||||
ink[y0:y1, sx0:sx1] = False
|
||||
|
||||
|
||||
def _side_strip_crops(
|
||||
page_png, ink_raw, faint, width, height, band_h, np, text_boxes
|
||||
) -> list[LogoCrop]:
|
||||
left_end = max(1, int(width * SIDE_STRIP_RATIO))
|
||||
right_start = min(width - 1, int(width * (1.0 - SIDE_STRIP_RATIO)))
|
||||
found: list[LogoCrop] = []
|
||||
for x0, x1 in ((0, left_end), (right_start, width)):
|
||||
masked = ink_raw.copy()
|
||||
_mask_strip_text(masked, text_boxes, x0, x1, band_h)
|
||||
crop = _best_in_strip(masked, x0, x1, page_png, width, height, band_h, np)
|
||||
if crop is None:
|
||||
# Do not mask OCR boxes on the faint pass: letterhead artwork is
|
||||
# often boxed as text and the second mark disappears.
|
||||
crop = _best_in_strip(
|
||||
faint,
|
||||
x0,
|
||||
x1,
|
||||
page_png,
|
||||
width,
|
||||
height,
|
||||
band_h,
|
||||
np,
|
||||
min_density=MIN_FAINT_INK_DENSITY,
|
||||
)
|
||||
if crop is not None:
|
||||
found.append(crop)
|
||||
return found
|
||||
|
||||
|
||||
def _centres_overlap(a: LogoCrop, b: LogoCrop) -> bool:
|
||||
ax, ay, aw, ah = a.bbox_px
|
||||
bx, by, bw, bh = b.bbox_px
|
||||
acx, acy = ax + aw / 2, ay + ah / 2
|
||||
return bx <= acx <= bx + bw and by <= acy <= by + bh
|
||||
|
||||
|
||||
def _prefer_side_marks(
|
||||
primary: list[LogoCrop],
|
||||
sides: list[LogoCrop],
|
||||
width: int,
|
||||
max_logos: int,
|
||||
) -> list[LogoCrop]:
|
||||
"""Keep at most one mark per third of the masthead, sides first.
|
||||
|
||||
A leftover banner fragment in the centre is often larger than the real
|
||||
emblems. Ranking by area then dropped the letterhead. The corners are the
|
||||
letterhead; the middle is only kept when it is comparable to them.
|
||||
"""
|
||||
combined: list[LogoCrop] = []
|
||||
for crop in list(sides) + list(primary):
|
||||
if any(_centres_overlap(crop, existing) or _centres_overlap(existing, crop) for existing in combined):
|
||||
continue
|
||||
combined.append(crop)
|
||||
|
||||
def _bucket(crop: LogoCrop) -> str:
|
||||
cx = crop.bbox_px[0] + crop.bbox_px[2] / 2
|
||||
if cx < width * 0.40:
|
||||
return "left"
|
||||
if cx > width * 0.60:
|
||||
return "right"
|
||||
return "mid"
|
||||
|
||||
buckets: dict[str, list[LogoCrop]] = {"left": [], "mid": [], "right": []}
|
||||
for crop in combined:
|
||||
buckets[_bucket(crop)].append(crop)
|
||||
|
||||
kept: list[LogoCrop] = []
|
||||
for key in ("left", "right"):
|
||||
if buckets[key]:
|
||||
kept.append(max(buckets[key], key=_crop_area))
|
||||
if buckets["mid"] and len(kept) < max_logos:
|
||||
mid = max(buckets["mid"], key=_crop_area)
|
||||
floor = max((_crop_area(c) for c in kept), default=1) * MIN_RELATIVE_MARK_AREA
|
||||
if _crop_area(mid) >= floor:
|
||||
kept.append(mid)
|
||||
kept.sort(key=lambda c: c.bbox_px[0])
|
||||
return kept[:max_logos]
|
||||
|
||||
|
||||
def _mask_text(ink, text_boxes, band_h: int, page_width: int | None = None) -> None:
|
||||
"""Zero every pixel a text box covers, padded for antialiasing."""
|
||||
if not text_boxes:
|
||||
return
|
||||
rows, cols = ink.shape
|
||||
page_w = float(page_width or cols)
|
||||
wide = page_w * 0.55
|
||||
# Letterhead marks sit in the outer sixths. OCR often boxes them as if they
|
||||
# were glyphs; masking those boxes is how one emblem survived and the other
|
||||
# vanished. Never erase that margin.
|
||||
margin = int(page_w * 0.16)
|
||||
side_keep = int(page_w * 0.20)
|
||||
for bx, by, bw, bh in text_boxes:
|
||||
if bw <= 0 or bh <= 0:
|
||||
continue
|
||||
if bw >= page_w * 0.35 and bh < band_h * 0.28:
|
||||
continue
|
||||
pad_x = bw * TEXT_PAD_RATIO
|
||||
pad_y = bh * TEXT_PAD_RATIO
|
||||
x0 = max(0, int(bx - pad_x))
|
||||
y0 = max(0, int(by - pad_y))
|
||||
x1 = min(cols, int(bx + bw + pad_x))
|
||||
y1 = min(rows, int(by + bh + pad_y))
|
||||
if bw >= wide:
|
||||
x0 = max(x0, side_keep)
|
||||
x1 = min(x1, cols - side_keep)
|
||||
x0 = max(x0, margin)
|
||||
x1 = min(x1, cols - margin)
|
||||
if y0 >= band_h:
|
||||
continue
|
||||
if x1 > x0 and y1 > y0:
|
||||
ink[y0:y1, x0:x1] = False
|
||||
|
||||
|
||||
def _ink_regions(ink, np) -> list[tuple[int, int, int, int]]:
|
||||
"""Bounding boxes of the contiguous column runs of ink, left to right.
|
||||
|
||||
Column projection rather than connected components: an emblem is one
|
||||
horizontal cluster separated from the rest of the masthead by whitespace,
|
||||
and projection needs no scipy and no arbitrary structuring element.
|
||||
"""
|
||||
col_ink = ink.sum(axis=0)
|
||||
peak = col_ink.max()
|
||||
if peak <= 0:
|
||||
return []
|
||||
gap = max(1.0, peak * COLUMN_GAP_RATIO)
|
||||
|
||||
spans: list[tuple[int, int]] = []
|
||||
start = None
|
||||
for x, value in enumerate(col_ink):
|
||||
if value >= gap:
|
||||
if start is None:
|
||||
start = x
|
||||
elif start is not None:
|
||||
spans.append((start, x))
|
||||
start = None
|
||||
if start is not None:
|
||||
spans.append((start, len(col_ink)))
|
||||
|
||||
boxes: list[tuple[int, int, int, int]] = []
|
||||
for x0, x1 in spans:
|
||||
rows = np.where(ink[:, x0:x1].any(axis=1))[0]
|
||||
if rows.size == 0:
|
||||
continue
|
||||
boxes.append((x0, int(rows[0]), x1, int(rows[-1]) + 1))
|
||||
return boxes
|
||||
|
||||
|
||||
def _largest_ink_region(ink, np):
|
||||
"""Bounding box of the densest contiguous column run. Kept for callers."""
|
||||
boxes = _ink_regions(ink, np)
|
||||
if not boxes:
|
||||
return None
|
||||
return max(boxes, key=lambda b: (b[2] - b[0]) * (b[3] - b[1]))
|
||||
|
||||
|
||||
def _crop_png(page_png: bytes, box: tuple[int, int, int, int]) -> bytes | None:
|
||||
from PIL import Image
|
||||
|
||||
x0, y0, x1, y1 = box
|
||||
with Image.open(io.BytesIO(page_png)) as im:
|
||||
crop = im.convert("RGB").crop((x0, y0, x1, y1))
|
||||
if crop.width < 2 or crop.height < 2:
|
||||
return None
|
||||
longest = max(crop.size)
|
||||
if longest > MAX_OUTPUT_SIDE:
|
||||
scale = MAX_OUTPUT_SIDE / longest
|
||||
crop = crop.resize(
|
||||
(max(1, int(crop.width * scale)), max(1, int(crop.height * scale))),
|
||||
Image.LANCZOS,
|
||||
)
|
||||
buf = io.BytesIO()
|
||||
crop.save(buf, format="PNG", optimize=True)
|
||||
return buf.getvalue()
|
||||
@@ -0,0 +1,396 @@
|
||||
"""Merge ML layout regions with heuristic glyph lines into IDM blocks."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from app.services.convert.idm.model import BBox, Block, BlockType, PageKind
|
||||
from app.services.convert.layout.blocks import (
|
||||
estimate_body_size,
|
||||
lines_to_block,
|
||||
)
|
||||
from app.services.convert.layout.glyphs import Line
|
||||
from app.services.convert.layout.ml_regions import Region
|
||||
from app.services.convert.layout.paragraphs import group_lines as group_paragraph_lines
|
||||
from app.services.convert.layout.paragraphs import lexical_hyphen_pairs
|
||||
from app.services.convert.layout.paragraphs import measure as measure_page
|
||||
from app.services.convert.layout.reading_order import order_lines
|
||||
from app.services.convert.layout.tables import coalesce_lines_by_y, extract_tables
|
||||
from app.services.convert.options import get_options
|
||||
|
||||
_LABEL_TO_BLOCK = {
|
||||
"title": BlockType.heading,
|
||||
"text": BlockType.paragraph,
|
||||
"list": BlockType.list_item,
|
||||
"table": BlockType.table,
|
||||
"figure": BlockType.figure,
|
||||
"header": BlockType.header,
|
||||
"footer": BlockType.footer,
|
||||
}
|
||||
|
||||
|
||||
def _line_center(line: Line) -> tuple[float, float]:
|
||||
return ((line.x0 + line.x1) / 2.0, line.y)
|
||||
|
||||
|
||||
def _contains(bbox: BBox, x: float, y: float, pad: float = 2.0) -> bool:
|
||||
return (
|
||||
bbox.x - pad <= x <= bbox.x + bbox.w + pad
|
||||
and bbox.y - pad <= y <= bbox.y + bbox.h + pad
|
||||
)
|
||||
|
||||
|
||||
def _overlap_line(bbox: BBox, line: Line, *, y_pad: float = 0.0) -> bool:
|
||||
cx, cy = _line_center(line)
|
||||
if _contains(bbox, cx, cy, pad=max(2.0, y_pad)):
|
||||
return True
|
||||
# horizontal overlap + y near region vertical span
|
||||
lx0, lx1 = line.x0, line.x1
|
||||
y0, y1 = bbox.y - y_pad, bbox.y + bbox.h + y_pad
|
||||
return not (lx1 < bbox.x or lx0 > bbox.x + bbox.w) and y0 <= cy <= y1
|
||||
|
||||
|
||||
def lines_in_region(lines: list[Line], region: Region) -> list[Line]:
|
||||
return [ln for ln in lines if _overlap_line(region.bbox_pdf, ln, y_pad=8.0)]
|
||||
|
||||
|
||||
def _paragraph_blocks(
|
||||
lines: list[Line],
|
||||
start_order: int,
|
||||
body: float,
|
||||
page_width: float,
|
||||
*,
|
||||
force_type: BlockType | None = None,
|
||||
) -> list[Block]:
|
||||
"""Group ordered lines into paragraph blocks.
|
||||
|
||||
The ML path previously emitted one block per visual line, so a document
|
||||
routed through layout detection came out with every line as its own
|
||||
``<w:p>`` even though the deterministic path reconstructed paragraphs.
|
||||
Both paths now share :mod:`layout.paragraphs`.
|
||||
|
||||
``force_type`` applies a region's label to the resulting blocks. Headings
|
||||
and list items are never merged across lines: a heading region holds one
|
||||
heading, and each list item is its own block.
|
||||
"""
|
||||
if not lines:
|
||||
return []
|
||||
|
||||
unmergeable = force_type in (
|
||||
BlockType.heading,
|
||||
BlockType.list_item,
|
||||
BlockType.header,
|
||||
BlockType.footer,
|
||||
)
|
||||
metrics = measure_page(lines, body_size=body, page_width=page_width)
|
||||
pairs = lexical_hyphen_pairs("\n".join(ln.text for ln in lines))
|
||||
runs = [[ln] for ln in lines] if unmergeable else group_paragraph_lines(lines, metrics)
|
||||
|
||||
out: list[Block] = []
|
||||
order = start_order
|
||||
for run in runs:
|
||||
block = lines_to_block(run, order, body, lexical_pairs=pairs)
|
||||
if force_type is BlockType.heading:
|
||||
block.type = BlockType.heading
|
||||
block.level = block.level or 1
|
||||
elif force_type is BlockType.paragraph:
|
||||
# Adobe-style: a text region is body, not a size-based H1.
|
||||
if block.type == BlockType.heading:
|
||||
block.type = BlockType.paragraph
|
||||
block.level = 0
|
||||
elif force_type is not None:
|
||||
block.type = force_type
|
||||
out.append(block)
|
||||
order += 1
|
||||
return out
|
||||
|
||||
|
||||
def _structure_rulings(
|
||||
reg: Region,
|
||||
*,
|
||||
page_png: bytes | None,
|
||||
page_width: float,
|
||||
page_height: float,
|
||||
page_index: int,
|
||||
vertical_rulings: list[float] | None,
|
||||
warnings: list[str],
|
||||
) -> list[float] | None:
|
||||
"""Column separators for one ML table region, or None to keep the heuristics.
|
||||
|
||||
Every failure path here returns None and leaves ``vertical_rulings``
|
||||
untouched: the optional structure model may sharpen a grid, never break one.
|
||||
"""
|
||||
from app.services.convert.layout import ml_table_structure as mts
|
||||
|
||||
if not mts.table_structure_enabled() or not page_png or not page_height:
|
||||
return None
|
||||
try:
|
||||
structure, err = mts.structure_for_region(
|
||||
page_png, reg.bbox_pdf, page_width, page_height
|
||||
)
|
||||
except Exception as exc: # pragma: no cover - defensive; hook must never raise
|
||||
warnings.append(f"Page {page_index + 1}: table_structure_fallback=heuristic ({exc})")
|
||||
return None
|
||||
if structure is None:
|
||||
if err:
|
||||
warnings.append(f"Page {page_index + 1}: table_structure_fallback=heuristic ({err})")
|
||||
return None
|
||||
merged = mts.merge_rulings(vertical_rulings, structure.column_x)
|
||||
warnings.append(
|
||||
f"Page {page_index + 1}: table_structure=onnx cols={structure.columns} "
|
||||
f"score={structure.score:.2f}."
|
||||
)
|
||||
return merged
|
||||
|
||||
|
||||
def merge_regions_with_lines(
|
||||
lines: list[Line],
|
||||
regions: list[Region],
|
||||
*,
|
||||
page_width: float,
|
||||
page_index: int,
|
||||
kind: PageKind,
|
||||
warnings: list[str],
|
||||
vertical_rulings: list[float] | None = None,
|
||||
page_rects: list | None = None,
|
||||
path_ops: list[dict] | None = None,
|
||||
page_png: bytes | None = None,
|
||||
page_height: float | None = None,
|
||||
) -> list[Block]:
|
||||
"""
|
||||
Build blocks using ML region order + heuristic tables.
|
||||
|
||||
Order of operations (best practice fusion):
|
||||
1. Coalesce gutter-split column fragments into logical rows.
|
||||
2. Extract grids from explicit ML *table* regions (relaxed / ml_hinted).
|
||||
3. Run page-level heuristic extract_tables on leftovers (catches pipe
|
||||
tables that ML mislabeled as per-row ``text`` regions).
|
||||
4. Type remaining lines via non-table ML regions (heading/list/header…).
|
||||
5. Leftover lines → paragraphs.
|
||||
"""
|
||||
if not regions:
|
||||
return []
|
||||
|
||||
working = coalesce_lines_by_y(lines) if lines else []
|
||||
if not any(ln.text.strip() for ln in working):
|
||||
# A full-page bitmap: the detector found rectangles, but there is no
|
||||
# text layer for them to describe. Emitting region blocks here — empty
|
||||
# tables in particular — puts structure on the page that nothing has
|
||||
# read yet, and the OCR rebuild that follows has the actual detection
|
||||
# boxes and assembles far better grids from them. The regions are not
|
||||
# wasted: the OCR pass still uses the table rectangles as crop hints.
|
||||
warnings.append(
|
||||
f"Page {page_index + 1}: layout_ml regions on a raster page; "
|
||||
"deferring structure to OCR."
|
||||
)
|
||||
return []
|
||||
assigned: set[int] = set()
|
||||
blocks: list[Block] = []
|
||||
order = 0
|
||||
body = estimate_body_size(working) if working else 12.0
|
||||
detect_tables = get_options().detect_tables
|
||||
|
||||
table_regs = sorted(
|
||||
[r for r in regions if r.label == "table"] if detect_tables else [],
|
||||
key=lambda r: r.reading_index,
|
||||
)
|
||||
other_regs = sorted(
|
||||
[r for r in regions if r.label != "table"],
|
||||
key=lambda r: r.reading_index,
|
||||
)
|
||||
|
||||
# --- Pass A: explicit ML table regions (relaxed grid builder) ---
|
||||
for reg in table_regs:
|
||||
idxs = [
|
||||
i
|
||||
for i, ln in enumerate(working)
|
||||
if i not in assigned and _overlap_line(reg.bbox_pdf, ln, y_pad=10.0)
|
||||
]
|
||||
# Expand vertically: include nearby unassigned lines inside region Y span
|
||||
y0, y1 = reg.bbox_pdf.y - 6.0, reg.bbox_pdf.y + reg.bbox_pdf.h + 6.0
|
||||
for i, ln in enumerate(working):
|
||||
if i in assigned or i in idxs:
|
||||
continue
|
||||
if y0 <= ln.y <= y1 and not (ln.x1 < reg.bbox_pdf.x or ln.x0 > reg.bbox_pdf.x + reg.bbox_pdf.w):
|
||||
idxs.append(i)
|
||||
idxs = sorted(set(idxs))
|
||||
reg_lines = [working[i] for i in idxs]
|
||||
for i in idxs:
|
||||
assigned.add(i)
|
||||
|
||||
if not reg_lines:
|
||||
# An empty table is not a table. It reached the document as a
|
||||
# bordered box around nothing, and it displaced the grid that OCR
|
||||
# would have built from the same rectangle.
|
||||
warnings.append(
|
||||
f"Page {page_index + 1}: ML table region with no text lines; skipped."
|
||||
)
|
||||
continue
|
||||
|
||||
# Optional structure model: column separators for THIS region only.
|
||||
# None (disabled, no weights, bad output) leaves the heuristics alone.
|
||||
region_rulings = _structure_rulings(
|
||||
reg,
|
||||
page_png=page_png,
|
||||
page_width=page_width,
|
||||
page_height=float(page_height or 0.0),
|
||||
page_index=page_index,
|
||||
vertical_rulings=vertical_rulings,
|
||||
warnings=warnings,
|
||||
)
|
||||
table_blocks, remaining, conf = extract_tables(
|
||||
order_lines(reg_lines, page_width),
|
||||
start_order=order,
|
||||
vertical_rulings=region_rulings if region_rulings is not None else vertical_rulings,
|
||||
page_rects=page_rects,
|
||||
path_ops=path_ops,
|
||||
ml_hinted=True,
|
||||
)
|
||||
# Require a real multi-row grid; weak 1-row hits release lines for Pass B.
|
||||
strong = bool(
|
||||
table_blocks
|
||||
and table_blocks[0].cells
|
||||
and len(table_blocks[0].cells) >= 2
|
||||
and max(len(r) for r in table_blocks[0].cells) >= 2
|
||||
)
|
||||
if strong:
|
||||
for tb in table_blocks:
|
||||
tb.reading_order = order
|
||||
tb.bbox = reg.bbox_pdf
|
||||
if tb.table_confidence < 0.5:
|
||||
tb.table_confidence = max(tb.table_confidence, float(reg.score or 0.5))
|
||||
blocks.append(tb)
|
||||
order += 1
|
||||
para_blocks = _paragraph_blocks(remaining, order, body, page_width)
|
||||
blocks.extend(para_blocks)
|
||||
order += len(para_blocks)
|
||||
warnings.append(
|
||||
f"Page {page_index + 1}: ML table->grid rows={len(table_blocks[0].cells)} "
|
||||
f"cols={len(table_blocks[0].cells[0])} conf={conf:.2f}."
|
||||
)
|
||||
else:
|
||||
# Release region lines so page-level heuristics can still build a grid
|
||||
for i in idxs:
|
||||
assigned.discard(i)
|
||||
warnings.append(
|
||||
f"Page {page_index + 1}: ML table region grid_failed conf={conf:.2f}; "
|
||||
"deferring to heuristic."
|
||||
)
|
||||
|
||||
# --- Pass B: page-level heuristic tables on unassigned lines ---
|
||||
# Fixes ML that labels each table row as a separate "text" region.
|
||||
leftover = [working[i] for i in range(len(working)) if i not in assigned]
|
||||
if leftover and detect_tables:
|
||||
table_blocks, remaining, conf = extract_tables(
|
||||
order_lines(leftover, page_width),
|
||||
start_order=order,
|
||||
vertical_rulings=vertical_rulings,
|
||||
page_rects=page_rects,
|
||||
path_ops=path_ops,
|
||||
ml_hinted=False,
|
||||
)
|
||||
if table_blocks and conf >= 0.60 and table_blocks[0].cells:
|
||||
remaining_ids = {id(ln) for ln in remaining}
|
||||
leftover_ids = {id(ln) for ln in leftover}
|
||||
for i, ln in enumerate(working):
|
||||
if id(ln) in leftover_ids and id(ln) not in remaining_ids:
|
||||
assigned.add(i)
|
||||
for tb in table_blocks:
|
||||
tb.reading_order = order
|
||||
blocks.append(tb)
|
||||
order += 1
|
||||
warnings.append(
|
||||
f"Page {page_index + 1}: heuristic table under ML regions "
|
||||
f"rows={len(table_blocks[0].cells)} conf={conf:.2f}."
|
||||
)
|
||||
|
||||
# --- Pass C: non-table ML regions for typing ---
|
||||
for reg in other_regs:
|
||||
idxs = [
|
||||
i
|
||||
for i, ln in enumerate(working)
|
||||
if i not in assigned and _overlap_line(reg.bbox_pdf, ln, y_pad=4.0)
|
||||
]
|
||||
reg_lines = [working[i] for i in idxs]
|
||||
for i in idxs:
|
||||
assigned.add(i)
|
||||
|
||||
btype = _LABEL_TO_BLOCK.get(reg.label, BlockType.paragraph)
|
||||
if not reg_lines:
|
||||
if btype in (BlockType.figure, BlockType.header, BlockType.footer):
|
||||
blocks.append(
|
||||
Block(type=btype, text="", bbox=reg.bbox_pdf, reading_order=order)
|
||||
)
|
||||
order += 1
|
||||
continue
|
||||
region_blocks = _paragraph_blocks(
|
||||
order_lines(reg_lines, page_width),
|
||||
order,
|
||||
body,
|
||||
page_width,
|
||||
force_type=btype,
|
||||
)
|
||||
blocks.extend(region_blocks)
|
||||
order += len(region_blocks)
|
||||
|
||||
# --- Pass D: true leftovers ---
|
||||
leftover = [working[i] for i in range(len(working)) if i not in assigned]
|
||||
if leftover:
|
||||
ordered = order_lines(leftover, page_width)
|
||||
table_blocks, remaining, conf = ([], ordered, 0.0)
|
||||
if detect_tables:
|
||||
table_blocks, remaining, conf = extract_tables(
|
||||
ordered,
|
||||
start_order=order,
|
||||
vertical_rulings=vertical_rulings,
|
||||
page_rects=page_rects,
|
||||
path_ops=path_ops,
|
||||
)
|
||||
if table_blocks and conf >= 0.65 and table_blocks[0].cells:
|
||||
for tb in table_blocks:
|
||||
tb.reading_order = order
|
||||
blocks.append(tb)
|
||||
order += 1
|
||||
para_blocks = _paragraph_blocks(remaining, order, body, page_width)
|
||||
blocks.extend(para_blocks)
|
||||
order += len(para_blocks)
|
||||
else:
|
||||
para_blocks = _paragraph_blocks(ordered, order, body, page_width)
|
||||
blocks.extend(para_blocks)
|
||||
order += len(para_blocks)
|
||||
|
||||
if kind == PageKind.blank and not blocks:
|
||||
warnings.append(f"Page {page_index + 1}: ML regions produced no blocks.")
|
||||
|
||||
# Order blocks by page reading order (XY-cut of text lines) and geometry,
|
||||
# rather than the order of detector passes (which previously put tables before titles).
|
||||
ordered_page_lines = order_lines(lines, page_width, page_height=float(page_height or 792.0))
|
||||
|
||||
def _block_sort_key(b: Block) -> tuple:
|
||||
if b.type == BlockType.header:
|
||||
return (0, 0, 0.0, 0.0)
|
||||
if b.type == BlockType.footer:
|
||||
return (2, 0, 0.0, 0.0)
|
||||
|
||||
box = b.bbox
|
||||
if box is not None:
|
||||
inside = [
|
||||
i
|
||||
for i, ln in enumerate(ordered_page_lines)
|
||||
if box.x - 3.0 <= (ln.x0 + ln.x1) / 2.0 <= box.x + box.w + 3.0
|
||||
and box.y - 3.0 <= ln.y <= box.y + box.h + 3.0
|
||||
]
|
||||
if inside:
|
||||
return (1, 0, min(inside), 0.0)
|
||||
top_y = box.y + box.h
|
||||
return (1, 1, -top_y, box.x)
|
||||
return (1, 2, b.reading_order, 0.0)
|
||||
|
||||
blocks.sort(key=_block_sort_key)
|
||||
for i, b in enumerate(blocks):
|
||||
b.reading_order = i
|
||||
return blocks
|
||||
|
||||
|
||||
def ml_header_footer_hints(regions: list[Region]) -> tuple[list[BBox], list[BBox]]:
|
||||
headers = [r.bbox_pdf for r in regions if r.label == "header"]
|
||||
footers = [r.bbox_pdf for r in regions if r.label == "footer"]
|
||||
return headers, footers
|
||||
@@ -0,0 +1,431 @@
|
||||
"""Optional Apache ONNX layout region detector (fail-open)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import os
|
||||
import threading
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
from app.services import model_paths
|
||||
from app.services.convert.idm.model import BBox
|
||||
from app.services.convert.layout.coords import PageGeom, pixel_bbox_to_pdf
|
||||
from app.services.convert.validation import run_with_timeout
|
||||
|
||||
_lock = threading.Lock()
|
||||
_session = None
|
||||
_session_path: str | None = None
|
||||
_load_error: str | None = None
|
||||
|
||||
# PP-DocLayoutV3 (Paddle) class list — see models/layout inference.yml / HF card.
|
||||
_PP_DOCLAYOUT_V3_LABELS = (
|
||||
"abstract",
|
||||
"algorithm",
|
||||
"aside_text",
|
||||
"chart",
|
||||
"content",
|
||||
"display_formula",
|
||||
"doc_title",
|
||||
"figure_title",
|
||||
"footer",
|
||||
"footer_image",
|
||||
"footnote",
|
||||
"formula_number",
|
||||
"header",
|
||||
"header_image",
|
||||
"image",
|
||||
"inline_formula",
|
||||
"number",
|
||||
"paragraph_title",
|
||||
"reference",
|
||||
"reference_content",
|
||||
"seal",
|
||||
"table",
|
||||
"text",
|
||||
"vertical_text",
|
||||
"vision_footnote",
|
||||
)
|
||||
|
||||
# Map raw detector labels → DocQube Region.label
|
||||
_LABEL_MAP = {
|
||||
0: "text",
|
||||
1: "title",
|
||||
2: "list",
|
||||
3: "table",
|
||||
4: "figure",
|
||||
5: "header",
|
||||
6: "footer",
|
||||
"text": "text",
|
||||
"title": "title",
|
||||
"list": "list",
|
||||
"table": "table",
|
||||
"figure": "figure",
|
||||
"header": "header",
|
||||
"footer": "footer",
|
||||
"Text": "text",
|
||||
"Title": "title",
|
||||
"List": "list",
|
||||
"Table": "table",
|
||||
"Figure": "figure",
|
||||
"Header": "header",
|
||||
"Footer": "footer",
|
||||
# PP-DocLayoutV3 names
|
||||
"abstract": "text",
|
||||
"algorithm": "text",
|
||||
"aside_text": "text",
|
||||
"chart": "figure",
|
||||
"content": "text",
|
||||
"display_formula": "text",
|
||||
"doc_title": "title",
|
||||
"figure_title": "title",
|
||||
"footer_image": "figure",
|
||||
"footnote": "footer",
|
||||
"formula_number": "text",
|
||||
"header_image": "figure",
|
||||
"image": "figure",
|
||||
"inline_formula": "text",
|
||||
"number": "text",
|
||||
"paragraph_title": "title",
|
||||
"reference": "text",
|
||||
"reference_content": "text",
|
||||
"seal": "figure",
|
||||
"vertical_text": "text",
|
||||
"vision_footnote": "footer",
|
||||
}
|
||||
|
||||
# Default input size for PP-DocLayoutV3 ONNX (inference.yml: Resize 800x800, keep_ratio false)
|
||||
_DEFAULT_TARGET = 800
|
||||
|
||||
|
||||
@dataclass
|
||||
class Region:
|
||||
label: str
|
||||
bbox_pdf: BBox
|
||||
score: float
|
||||
reading_index: int = 0
|
||||
|
||||
|
||||
def layout_ml_enabled() -> bool:
|
||||
return os.environ.get("CONVERT_LAYOUT_ML", "0").strip().lower() in ("1", "true", "yes", "on")
|
||||
|
||||
|
||||
def _weights_path() -> Path:
|
||||
explicit = (os.environ.get("CONVERT_LAYOUT_ML_WEIGHTS") or "").strip()
|
||||
if explicit:
|
||||
return model_paths.resolve(explicit)
|
||||
int8_cand = model_paths.resolve("models/layout/v1/layout_int8.onnx")
|
||||
if int8_cand.is_file():
|
||||
return int8_cand
|
||||
return model_paths.resolve("models/layout/v1/layout.onnx")
|
||||
|
||||
|
||||
def _min_score() -> float:
|
||||
try:
|
||||
return float(os.environ.get("CONVERT_LAYOUT_ML_MIN_SCORE", "0.5"))
|
||||
except ValueError:
|
||||
return 0.5
|
||||
|
||||
|
||||
def _device() -> str:
|
||||
return os.environ.get("CONVERT_LAYOUT_ML_DEVICE", "cpu").strip().lower()
|
||||
|
||||
|
||||
def _page_timeout() -> float:
|
||||
try:
|
||||
return float(os.environ.get("CONVERT_LAYOUT_ML_PAGE_TIMEOUT", "15"))
|
||||
except ValueError:
|
||||
return 15.0
|
||||
|
||||
|
||||
def _target_size() -> int:
|
||||
try:
|
||||
return max(64, int(os.environ.get("CONVERT_LAYOUT_ML_INPUT_SIZE", str(_DEFAULT_TARGET))))
|
||||
except ValueError:
|
||||
return _DEFAULT_TARGET
|
||||
|
||||
|
||||
def _get_session():
|
||||
global _session, _session_path, _load_error
|
||||
path = str(_weights_path())
|
||||
with _lock:
|
||||
if _session is not None and _session_path == path:
|
||||
return _session
|
||||
if _load_error and _session_path == path:
|
||||
return None
|
||||
try:
|
||||
import onnxruntime as ort
|
||||
|
||||
if not Path(path).is_file():
|
||||
_load_error = f"weights missing: {path}"
|
||||
_session = None
|
||||
_session_path = path
|
||||
return None
|
||||
|
||||
sess_options = ort.SessionOptions()
|
||||
sess_options.graph_optimization_level = ort.GraphOptimizationLevel.ORT_ENABLE_ALL
|
||||
# Limit ONNX intra-op threads to prevent CPU starvation for concurrent requests
|
||||
max_threads = max(1, min(4, os.cpu_count() or 2))
|
||||
sess_options.intra_op_num_threads = max_threads
|
||||
|
||||
providers = ["CPUExecutionProvider"]
|
||||
if _device() == "cuda":
|
||||
providers = ["CUDAExecutionProvider", "CPUExecutionProvider"]
|
||||
_session = ort.InferenceSession(path, sess_options=sess_options, providers=providers)
|
||||
_session_path = path
|
||||
_load_error = None
|
||||
return _session
|
||||
except Exception as exc:
|
||||
_load_error = str(exc)
|
||||
_session = None
|
||||
_session_path = path
|
||||
return None
|
||||
|
||||
|
||||
def reset_session_for_tests() -> None:
|
||||
global _session, _session_path, _load_error
|
||||
with _lock:
|
||||
_session = None
|
||||
_session_path = None
|
||||
_load_error = None
|
||||
|
||||
|
||||
def last_load_error() -> str | None:
|
||||
return _load_error
|
||||
|
||||
|
||||
def _map_label(cls: int | str) -> str:
|
||||
if isinstance(cls, (int, float)):
|
||||
idx = int(cls)
|
||||
if 0 <= idx < len(_PP_DOCLAYOUT_V3_LABELS):
|
||||
name = _PP_DOCLAYOUT_V3_LABELS[idx]
|
||||
return _LABEL_MAP.get(name, "text")
|
||||
return _LABEL_MAP.get(idx, "text")
|
||||
return _LABEL_MAP.get(str(cls), _LABEL_MAP.get(cls, "text"))
|
||||
|
||||
|
||||
def preprocess_pp_doclayout(
|
||||
rgb_hwc: np.ndarray,
|
||||
*,
|
||||
target_size: int = _DEFAULT_TARGET,
|
||||
) -> tuple[np.ndarray, np.ndarray, np.ndarray]:
|
||||
"""
|
||||
PP-DocLayoutV3 preprocess (matches official inference.yml):
|
||||
Resize 800x800 keep_ratio=false, NormalizeImage mean=0 std=1 (÷255), Permute NCHW.
|
||||
|
||||
Returns (image NCHW float32, im_shape [1,2], scale_factor [1,2] as [scale_h, scale_w]).
|
||||
"""
|
||||
import numpy as np
|
||||
from PIL import Image
|
||||
|
||||
if rgb_hwc.ndim != 3 or rgb_hwc.shape[2] != 3:
|
||||
raise ValueError(f"expected HWC RGB, got shape {getattr(rgb_hwc, 'shape', None)}")
|
||||
orig_h, orig_w = int(rgb_hwc.shape[0]), int(rgb_hwc.shape[1])
|
||||
target = int(target_size)
|
||||
if rgb_hwc.dtype != np.uint8:
|
||||
# Accept float 0-1 or 0-255
|
||||
scaled = rgb_hwc * 255.0 if float(np.nanmax(rgb_hwc)) <= 1.5 else rgb_hwc
|
||||
img = Image.fromarray(np.clip(scaled, 0, 255).astype(np.uint8))
|
||||
else:
|
||||
img = Image.fromarray(rgb_hwc)
|
||||
resized = img.resize((target, target), Image.BILINEAR)
|
||||
blob = np.asarray(resized).astype("float32") / 255.0
|
||||
image = np.transpose(blob, (2, 0, 1))[None, ...].astype("float32")
|
||||
scale_h = float(target) / float(orig_h)
|
||||
scale_w = float(target) / float(orig_w)
|
||||
im_shape = np.array([[float(target), float(target)]], dtype=np.float32)
|
||||
scale_factor = np.array([[scale_h, scale_w]], dtype=np.float32)
|
||||
return image, im_shape, scale_factor
|
||||
|
||||
|
||||
def build_onnx_feed(sess, rgb_hwc: np.ndarray, *, target_size: int | None = None) -> dict:
|
||||
"""
|
||||
Build ORT feed dict from session input metadata.
|
||||
|
||||
PP-DocLayoutV3 expects: im_shape, image, scale_factor.
|
||||
Simpler single-input models get NCHW (or NHWC if shape[-1]==3) image only.
|
||||
"""
|
||||
import numpy as np
|
||||
|
||||
t = _target_size() if target_size is None else int(target_size)
|
||||
image, im_shape, scale_factor = preprocess_pp_doclayout(rgb_hwc, target_size=t)
|
||||
inputs = list(sess.get_inputs())
|
||||
names = {i.name for i in inputs}
|
||||
feed: dict = {}
|
||||
|
||||
# Named PP-DocLayout / Paddle Detection I/O
|
||||
if "image" in names or "scale_factor" in names or "im_shape" in names:
|
||||
if "image" in names:
|
||||
feed["image"] = image
|
||||
if "im_shape" in names:
|
||||
feed["im_shape"] = im_shape
|
||||
if "scale_factor" in names:
|
||||
feed["scale_factor"] = scale_factor
|
||||
# Fill any remaining unknown inputs best-effort
|
||||
for inp in inputs:
|
||||
if inp.name in feed:
|
||||
continue
|
||||
shape = inp.shape
|
||||
if len(shape) == 4:
|
||||
if shape[-1] == 3:
|
||||
feed[inp.name] = np.transpose(image, (0, 2, 3, 1))
|
||||
else:
|
||||
feed[inp.name] = image
|
||||
elif len(shape) == 2:
|
||||
feed[inp.name] = im_shape
|
||||
return feed
|
||||
|
||||
# Single (or primary) image tensor models
|
||||
if not inputs:
|
||||
return feed
|
||||
name = inputs[0].name
|
||||
shape = inputs[0].shape
|
||||
if len(shape) == 4 and shape[-1] == 3:
|
||||
feed[name] = np.transpose(image, (0, 2, 3, 1))
|
||||
else:
|
||||
feed[name] = image
|
||||
return feed
|
||||
|
||||
|
||||
def _parse_pp_doclayout_boxes(boxes: np.ndarray, geom: PageGeom, min_score: float) -> list[Region]:
|
||||
"""Parse PP-DocLayoutV3 rows: [cls, score, x1, y1, x2, y2, read_order] (pixel space)."""
|
||||
import numpy as np
|
||||
|
||||
regions: list[Region] = []
|
||||
a = np.asarray(boxes)
|
||||
if a.ndim != 2 or a.shape[-1] < 6:
|
||||
return regions
|
||||
for row in a:
|
||||
row = [float(x) for x in row]
|
||||
n = len(row)
|
||||
# 7-col: cls, score, xyxy, order | 8-col: img_idx, cls, score, xyxy, order
|
||||
if n >= 8 and row[2] <= 1.0 and row[1] < 50:
|
||||
cls, sc, x1, y1, x2, y2 = int(row[1]), row[2], row[3], row[4], row[5], row[6]
|
||||
order = int(row[7]) if n > 7 else len(regions)
|
||||
elif n >= 7 and row[1] <= 1.5:
|
||||
cls, sc, x1, y1, x2, y2 = int(row[0]), row[1], row[2], row[3], row[4], row[5]
|
||||
order = int(row[6]) if n > 6 else len(regions)
|
||||
elif n >= 6 and row[4] <= 1.0 and row[5] < 50:
|
||||
x1, y1, x2, y2, sc, cls = row[0], row[1], row[2], row[3], row[4], int(row[5])
|
||||
order = len(regions)
|
||||
else:
|
||||
continue
|
||||
if sc < min_score:
|
||||
continue
|
||||
if x2 < x1 or y2 < y1:
|
||||
x2, y2 = x1 + max(0.0, x2), y1 + max(0.0, y2)
|
||||
if max(x1, y1, x2, y2) <= 1.5:
|
||||
x1 *= geom.pixel_w
|
||||
x2 *= geom.pixel_w
|
||||
y1 *= geom.pixel_h
|
||||
y2 *= geom.pixel_h
|
||||
label = _map_label(cls)
|
||||
bbox = pixel_bbox_to_pdf(float(x1), float(y1), float(x2), float(y2), geom)
|
||||
regions.append(Region(label=label, bbox_pdf=bbox, score=float(sc), reading_index=order))
|
||||
regions.sort(key=lambda r: (r.reading_index, -(r.bbox_pdf.y + r.bbox_pdf.h), r.bbox_pdf.x))
|
||||
for i, r in enumerate(regions):
|
||||
r.reading_index = i
|
||||
return regions
|
||||
|
||||
|
||||
def _parse_outputs(outs: list, geom: PageGeom, min_score: float) -> list[Region]:
|
||||
"""Best-effort parse of common detection outputs (boxes, scores, labels)."""
|
||||
import numpy as np
|
||||
|
||||
# Prefer PP-DocLayout-style (N, 6|7|8) float matrix
|
||||
for arr in outs:
|
||||
a = np.asarray(arr)
|
||||
if a.ndim == 2 and a.shape[-1] in (6, 7, 8) and a.dtype.kind == "f":
|
||||
parsed = _parse_pp_doclayout_boxes(a, geom, min_score)
|
||||
if parsed:
|
||||
return parsed
|
||||
|
||||
boxes = scores = labels = None
|
||||
for arr in outs:
|
||||
a = np.asarray(arr)
|
||||
if a.ndim >= 2 and a.shape[-1] >= 4 and boxes is None and a.dtype.kind == "f":
|
||||
boxes = a.reshape(-1, a.shape[-1])
|
||||
elif a.ndim == 1 and a.dtype != object and scores is None and a.size > 0:
|
||||
if a.dtype.kind == "f":
|
||||
scores = a.reshape(-1)
|
||||
else:
|
||||
labels = a.reshape(-1)
|
||||
elif a.ndim == 2 and a.shape[0] == 1 and a.shape[1] > 4 and a.dtype.kind == "f":
|
||||
boxes = a.reshape(-1, a.shape[-1])
|
||||
|
||||
if boxes is None:
|
||||
return []
|
||||
|
||||
# Reuse PP parser when shape matches; else legacy path
|
||||
if boxes.shape[-1] in (6, 7, 8):
|
||||
return _parse_pp_doclayout_boxes(boxes, geom, min_score)
|
||||
|
||||
regions: list[Region] = []
|
||||
for i, row in enumerate(boxes):
|
||||
row = list(row)
|
||||
if len(row) < 4:
|
||||
continue
|
||||
if len(row) >= 6:
|
||||
if row[4] <= 1.0 and row[5] < 20:
|
||||
x1, y1, x2, y2, sc, cls = row[0], row[1], row[2], row[3], row[4], int(row[5])
|
||||
else:
|
||||
sc, cls, x1, y1, x2, y2 = row[0], int(row[1]), row[2], row[3], row[4], row[5]
|
||||
else:
|
||||
x1, y1, x2, y2 = row[0], row[1], row[2], row[3]
|
||||
sc = float(scores[i]) if scores is not None and i < len(scores) else 1.0
|
||||
cls = int(labels[i]) if labels is not None and i < len(labels) else 0
|
||||
|
||||
if sc < min_score:
|
||||
continue
|
||||
if x2 < x1 or y2 < y1:
|
||||
x2, y2 = x1 + max(0.0, x2), y1 + max(0.0, y2)
|
||||
if max(x1, y1, x2, y2) <= 1.5:
|
||||
x1 *= geom.pixel_w
|
||||
x2 *= geom.pixel_w
|
||||
y1 *= geom.pixel_h
|
||||
y2 *= geom.pixel_h
|
||||
label = _map_label(cls)
|
||||
bbox = pixel_bbox_to_pdf(float(x1), float(y1), float(x2), float(y2), geom)
|
||||
regions.append(Region(label=label, bbox_pdf=bbox, score=float(sc), reading_index=len(regions)))
|
||||
regions.sort(key=lambda r: (-(r.bbox_pdf.y + r.bbox_pdf.h), r.bbox_pdf.x))
|
||||
for i, r in enumerate(regions):
|
||||
r.reading_index = i
|
||||
return regions
|
||||
|
||||
|
||||
def detect_regions(
|
||||
png_bytes: bytes,
|
||||
page_width: float,
|
||||
page_height: float,
|
||||
*,
|
||||
dpi: int | None = None,
|
||||
) -> tuple[list[Region], str | None]:
|
||||
"""
|
||||
Returns (regions, error_or_none).
|
||||
Empty regions + error ⇒ caller must fail-open to heuristics.
|
||||
"""
|
||||
from app.services.convert.layout.page_raster import layout_ml_dpi
|
||||
|
||||
d = dpi if dpi is not None else layout_ml_dpi()
|
||||
geom = PageGeom(page_width=page_width, page_height=page_height, dpi=float(d))
|
||||
sess = _get_session()
|
||||
if sess is None:
|
||||
return [], _load_error or "layout ML session unavailable"
|
||||
|
||||
try:
|
||||
import numpy as np
|
||||
from PIL import Image
|
||||
|
||||
img = Image.open(io.BytesIO(png_bytes)).convert("RGB")
|
||||
arr = np.asarray(img)
|
||||
feed = build_onnx_feed(sess, arr)
|
||||
|
||||
def _run():
|
||||
with _lock:
|
||||
return sess.run(None, feed)
|
||||
|
||||
outs = run_with_timeout(_run, _page_timeout(), label="layout ML infer")
|
||||
regions = _parse_outputs(list(outs), geom, _min_score())
|
||||
return regions, None
|
||||
except Exception as exc:
|
||||
return [], str(exc)
|
||||
@@ -0,0 +1,401 @@
|
||||
"""Optional table-structure model (fail-open) — column/row separators only.
|
||||
|
||||
The heuristics in :mod:`layout.tables` recover a grid from token x-clustering
|
||||
and from ruled rectangles. They are good at ruled tables and at tables whose
|
||||
columns are separated by real gutters. They are weak exactly where a human
|
||||
also has to squint: a borderless table whose columns are close together, or one
|
||||
where a long cell in row 3 closes the gutter that rows 1, 2 and 4 make obvious.
|
||||
|
||||
A structure model sees the whole table at once and answers the one question the
|
||||
heuristics get wrong: *where are the column boundaries*. That answer feeds the
|
||||
existing grid builder as ``vertical_rulings`` — the same input a ruled table
|
||||
already supplies — so the model never builds cells itself and can never invent a
|
||||
table where the heuristics found none. It narrows an existing decision; it does
|
||||
not make a new one.
|
||||
|
||||
Licensing (the reason this is a hook rather than a bundled model): the intended
|
||||
weights are Microsoft's Table Transformer structure model (MIT) or PaddleOCR's
|
||||
SLANet (Apache-2.0). Nothing is downloaded and nothing is vendored — the
|
||||
operator points ``CONVERT_LAYOUT_ML_TABLE_STRUCTURE_WEIGHTS`` at a file they
|
||||
chose.
|
||||
|
||||
**Fail-open is the contract.** Disabled, weights missing, onnxruntime missing,
|
||||
inference error, timeout, or output that fails the plausibility check — every
|
||||
one of those returns "no rulings" and the heuristic path runs exactly as it does
|
||||
today. This module can make table detection sharper; it can never be the reason
|
||||
a conversion is worse or fails.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import os
|
||||
import threading
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
from app.services import model_paths
|
||||
from app.services.convert.idm.model import BBox
|
||||
from app.services.convert.layout.coords import PageGeom, pdf_bbox_to_pixel
|
||||
from app.services.convert.validation import run_with_timeout
|
||||
|
||||
_lock = threading.Lock()
|
||||
_session = None
|
||||
_session_path: str | None = None
|
||||
_load_error: str | None = None
|
||||
|
||||
# Table Transformer (TATR) structure classes, in label order.
|
||||
_TATR_LABELS = (
|
||||
"table",
|
||||
"table column",
|
||||
"table row",
|
||||
"table column header",
|
||||
"table projected row header",
|
||||
"table spanning cell",
|
||||
)
|
||||
|
||||
# A table with more columns than this is a parse error, not a table.
|
||||
_MAX_COLUMNS = 40
|
||||
_MAX_ROWS = 200
|
||||
# Two separators closer together than this are the same separator seen twice.
|
||||
_MIN_SEPARATOR_GAP_POINTS = 4.0
|
||||
# A column strip thinner than this is noise from a low-confidence box.
|
||||
_MIN_COLUMN_WIDTH_POINTS = 6.0
|
||||
|
||||
|
||||
@dataclass
|
||||
class TableStructure:
|
||||
"""Column and row separator positions, in PDF points on the page."""
|
||||
|
||||
column_x: list[float] = field(default_factory=list)
|
||||
row_y: list[float] = field(default_factory=list)
|
||||
score: float = 0.0
|
||||
|
||||
@property
|
||||
def columns(self) -> int:
|
||||
# n separators between the outer edges ⇒ n-1 columns.
|
||||
return max(0, len(self.column_x) - 1)
|
||||
|
||||
@property
|
||||
def rows(self) -> int:
|
||||
return max(0, len(self.row_y) - 1)
|
||||
|
||||
|
||||
def table_structure_enabled() -> bool:
|
||||
return os.environ.get("CONVERT_LAYOUT_ML_TABLE_STRUCTURE", "0").strip().lower() in (
|
||||
"1",
|
||||
"true",
|
||||
"yes",
|
||||
"on",
|
||||
)
|
||||
|
||||
|
||||
def _weights_path() -> Path:
|
||||
return model_paths.from_env(
|
||||
"CONVERT_LAYOUT_ML_TABLE_STRUCTURE_WEIGHTS", "models/layout/table/structure.onnx"
|
||||
)
|
||||
|
||||
|
||||
def _min_score() -> float:
|
||||
try:
|
||||
return float(os.environ.get("CONVERT_LAYOUT_ML_TABLE_STRUCTURE_MIN_SCORE", "0.5"))
|
||||
except ValueError:
|
||||
return 0.5
|
||||
|
||||
|
||||
def _timeout() -> float:
|
||||
try:
|
||||
return float(os.environ.get("CONVERT_LAYOUT_ML_TABLE_STRUCTURE_TIMEOUT", "10"))
|
||||
except ValueError:
|
||||
return 10.0
|
||||
|
||||
|
||||
def _input_size() -> int:
|
||||
try:
|
||||
return max(64, int(os.environ.get("CONVERT_LAYOUT_ML_TABLE_STRUCTURE_INPUT_SIZE", "800")))
|
||||
except ValueError:
|
||||
return 800
|
||||
|
||||
|
||||
def _device() -> str:
|
||||
return os.environ.get("CONVERT_LAYOUT_ML_DEVICE", "cpu").strip().lower()
|
||||
|
||||
|
||||
def _get_session():
|
||||
"""ONNX session, or None with :func:`last_load_error` explaining why."""
|
||||
global _session, _session_path, _load_error
|
||||
path = str(_weights_path())
|
||||
with _lock:
|
||||
if _session is not None and _session_path == path:
|
||||
return _session
|
||||
if _load_error and _session_path == path:
|
||||
return None
|
||||
try:
|
||||
if not Path(path).is_file():
|
||||
_load_error = f"table structure weights missing: {path}"
|
||||
_session = None
|
||||
_session_path = path
|
||||
return None
|
||||
import onnxruntime as ort
|
||||
|
||||
providers = ["CPUExecutionProvider"]
|
||||
if _device() == "cuda":
|
||||
providers = ["CUDAExecutionProvider", "CPUExecutionProvider"]
|
||||
_session = ort.InferenceSession(path, providers=providers)
|
||||
_session_path = path
|
||||
_load_error = None
|
||||
return _session
|
||||
except Exception as exc:
|
||||
_load_error = str(exc)
|
||||
_session = None
|
||||
_session_path = path
|
||||
return None
|
||||
|
||||
|
||||
def reset_session_for_tests() -> None:
|
||||
global _session, _session_path, _load_error
|
||||
with _lock:
|
||||
_session = None
|
||||
_session_path = None
|
||||
_load_error = None
|
||||
|
||||
|
||||
def last_load_error() -> str | None:
|
||||
return _load_error
|
||||
|
||||
|
||||
def _label_name(cls: int | str) -> str:
|
||||
if isinstance(cls, (int, float)):
|
||||
idx = int(cls)
|
||||
if 0 <= idx < len(_TATR_LABELS):
|
||||
return _TATR_LABELS[idx]
|
||||
return "unknown"
|
||||
return str(cls).strip().lower().replace("_", " ")
|
||||
|
||||
|
||||
def _crop_png(png_bytes: bytes, bbox: BBox, geom: PageGeom) -> tuple[bytes, float, float]:
|
||||
"""Crop the table's rectangle out of the page raster.
|
||||
|
||||
Returns the crop plus its top-left pixel offset on the page, so boxes the
|
||||
model reports inside the crop can be put back on the page.
|
||||
"""
|
||||
from PIL import Image
|
||||
|
||||
img = Image.open(io.BytesIO(png_bytes)).convert("RGB")
|
||||
x0, y0, x1, y1 = pdf_bbox_to_pixel(bbox, geom)
|
||||
x0 = max(0.0, min(float(img.width), x0))
|
||||
x1 = max(0.0, min(float(img.width), x1))
|
||||
y0 = max(0.0, min(float(img.height), y0))
|
||||
y1 = max(0.0, min(float(img.height), y1))
|
||||
if x1 - x0 < 8 or y1 - y0 < 8:
|
||||
raise ValueError("table region too small to crop")
|
||||
left, top = int(x0), int(y0)
|
||||
crop = img.crop((left, top, int(x1), int(y1)))
|
||||
buf = io.BytesIO()
|
||||
crop.save(buf, format="PNG")
|
||||
# The offset returned is the one the crop actually used, not the unrounded
|
||||
# float: a half-pixel disagreement here shifts every separator on the page.
|
||||
return buf.getvalue(), float(left), float(top)
|
||||
|
||||
|
||||
def _build_feed(sess, png_bytes: bytes) -> dict:
|
||||
"""Feed dict from session input metadata — DETR-style and plain-NCHW both."""
|
||||
import numpy as np
|
||||
from PIL import Image
|
||||
|
||||
target = _input_size()
|
||||
img = Image.open(io.BytesIO(png_bytes)).convert("RGB").resize((target, target), Image.BILINEAR)
|
||||
blob = (np.asarray(img).astype("float32") / 255.0).transpose(2, 0, 1)[None, ...]
|
||||
|
||||
feed: dict = {}
|
||||
for inp in sess.get_inputs():
|
||||
shape = list(inp.shape)
|
||||
if len(shape) == 4:
|
||||
feed[inp.name] = (
|
||||
np.transpose(blob, (0, 2, 3, 1)) if shape[-1] == 3 else blob
|
||||
).astype("float32")
|
||||
elif len(shape) == 2:
|
||||
# DETR exports take the padding mask or the original size here.
|
||||
feed[inp.name] = np.array([[float(target), float(target)]], dtype="float32")
|
||||
return feed
|
||||
|
||||
|
||||
def _boxes_from_outputs(outs: list, min_score: float) -> list[tuple[str, float, float, float, float, float]]:
|
||||
"""(label, score, x0, y0, x1, y1) in crop-normalised or crop-pixel units."""
|
||||
import numpy as np
|
||||
|
||||
rows: list[tuple[str, float, float, float, float, float]] = []
|
||||
for arr in outs:
|
||||
a = np.asarray(arr)
|
||||
if a.dtype.kind != "f" or a.ndim < 2:
|
||||
continue
|
||||
m = a.reshape(-1, a.shape[-1])
|
||||
if m.shape[-1] < 6:
|
||||
continue
|
||||
for row in m:
|
||||
row = [float(v) for v in row]
|
||||
# cls, score, xyxy | xyxy, score, cls
|
||||
if row[1] <= 1.0 and row[0] < len(_TATR_LABELS) + 1:
|
||||
cls, sc, x0, y0, x1, y1 = row[0], row[1], row[2], row[3], row[4], row[5]
|
||||
elif row[4] <= 1.0 and row[5] < len(_TATR_LABELS) + 1:
|
||||
x0, y0, x1, y1, sc, cls = row[0], row[1], row[2], row[3], row[4], row[5]
|
||||
else:
|
||||
continue
|
||||
if sc < min_score:
|
||||
continue
|
||||
if x1 < x0 or y1 < y0:
|
||||
continue
|
||||
rows.append((_label_name(cls), float(sc), x0, y0, x1, y1))
|
||||
if rows:
|
||||
break
|
||||
return rows
|
||||
|
||||
|
||||
def _separators_from_strips(
|
||||
strips: list[tuple[float, float]], *, min_gap: float, min_size: float
|
||||
) -> list[float]:
|
||||
"""Column/row *strips* → the separator positions between them.
|
||||
|
||||
The model reports each column as a rectangle. What the grid builder wants is
|
||||
the boundaries, so adjacent strips are bridged at the middle of the space
|
||||
between them and the outer edges are kept.
|
||||
"""
|
||||
usable = [(a, b) for a, b in sorted(strips) if b - a >= min_size]
|
||||
if len(usable) < 2:
|
||||
return []
|
||||
seps = [usable[0][0]]
|
||||
for (_, prev_end), (next_start, _) in zip(usable, usable[1:], strict=False):
|
||||
seps.append((prev_end + next_start) / 2.0 if next_start >= prev_end else next_start)
|
||||
seps.append(usable[-1][1])
|
||||
|
||||
out: list[float] = []
|
||||
for s in seps:
|
||||
if not out or s - out[-1] >= min_gap:
|
||||
out.append(s)
|
||||
else:
|
||||
out[-1] = (out[-1] + s) / 2.0
|
||||
return out
|
||||
|
||||
|
||||
def structure_for_region(
|
||||
png_bytes: bytes,
|
||||
bbox: BBox,
|
||||
page_width: float,
|
||||
page_height: float,
|
||||
*,
|
||||
dpi: int | None = None,
|
||||
) -> tuple[TableStructure | None, str | None]:
|
||||
"""Column/row separators for one table region, in PDF points.
|
||||
|
||||
Returns ``(None, reason)`` whenever the caller must fall back to the
|
||||
heuristics — which is every failure mode, by design.
|
||||
"""
|
||||
if not table_structure_enabled():
|
||||
return None, None
|
||||
if not png_bytes:
|
||||
return None, "no page raster"
|
||||
|
||||
from app.services.convert.layout.page_raster import layout_ml_dpi
|
||||
|
||||
d = float(dpi if dpi is not None else layout_ml_dpi())
|
||||
geom = PageGeom(page_width=page_width, page_height=page_height, dpi=d)
|
||||
|
||||
sess = _get_session()
|
||||
if sess is None:
|
||||
return None, _load_error or "table structure session unavailable"
|
||||
|
||||
try:
|
||||
crop, off_x, off_y = _crop_png(png_bytes, bbox, geom)
|
||||
feed = _build_feed(sess, crop)
|
||||
if not feed:
|
||||
return None, "table structure model has no image input"
|
||||
|
||||
def _run():
|
||||
with _lock:
|
||||
return sess.run(None, feed)
|
||||
|
||||
outs = run_with_timeout(_run, _timeout(), label="table structure infer")
|
||||
boxes = _boxes_from_outputs(list(outs), _min_score())
|
||||
if not boxes:
|
||||
return None, "table structure produced no boxes"
|
||||
except Exception as exc:
|
||||
return None, str(exc)
|
||||
|
||||
# The model works in the crop's own frame; put it back on the page.
|
||||
from PIL import Image
|
||||
|
||||
crop_img = Image.open(io.BytesIO(crop))
|
||||
crop_w, crop_h = float(crop_img.width), float(crop_img.height)
|
||||
scale = geom.scale
|
||||
|
||||
col_strips: list[tuple[float, float]] = []
|
||||
row_strips: list[tuple[float, float]] = []
|
||||
best = 0.0
|
||||
for label, score, x0, y0, x1, y1 in boxes:
|
||||
if max(x0, y0, x1, y1) <= 1.5: # normalised
|
||||
x0, x1 = x0 * crop_w, x1 * crop_w
|
||||
y0, y1 = y0 * crop_h, y1 * crop_h
|
||||
if label in ("table column", "table column header"):
|
||||
if label == "table column":
|
||||
col_strips.append((x0, x1))
|
||||
best = max(best, score)
|
||||
elif label == "table row":
|
||||
row_strips.append((y0, y1))
|
||||
best = max(best, score)
|
||||
|
||||
col_px = _separators_from_strips(
|
||||
col_strips,
|
||||
min_gap=_MIN_SEPARATOR_GAP_POINTS * scale,
|
||||
min_size=_MIN_COLUMN_WIDTH_POINTS * scale,
|
||||
)
|
||||
row_px = _separators_from_strips(
|
||||
row_strips,
|
||||
min_gap=_MIN_SEPARATOR_GAP_POINTS * scale,
|
||||
min_size=_MIN_SEPARATOR_GAP_POINTS * scale,
|
||||
)
|
||||
if len(col_px) < 3:
|
||||
# Fewer than three separators means fewer than two columns: nothing the
|
||||
# heuristics could not already see.
|
||||
return None, "table structure found no usable columns"
|
||||
if len(col_px) - 1 > _MAX_COLUMNS or len(row_px) - 1 > _MAX_ROWS:
|
||||
return None, "table structure output implausible"
|
||||
|
||||
# Crop pixels → page pixels → PDF points. x needs no flip; y is handled by
|
||||
# the callers that need rows, and is reported top-down from the page top.
|
||||
column_x = [(off_x + px) / scale for px in col_px]
|
||||
row_y = [page_height - ((off_y + py) / scale) for py in row_px]
|
||||
|
||||
left, right = bbox.x - 4.0, bbox.x + bbox.w + 4.0
|
||||
column_x = [x for x in column_x if left <= x <= right]
|
||||
if len(column_x) < 3:
|
||||
return None, "table structure columns fell outside the region"
|
||||
|
||||
return TableStructure(column_x=column_x, row_y=row_y, score=best), None
|
||||
|
||||
|
||||
def merge_rulings(
|
||||
heuristic: list[float] | None, structure: list[float] | None, *, tolerance: float = 3.0
|
||||
) -> list[float]:
|
||||
"""Union of ruled-line x positions and model separators, near-duplicates fused.
|
||||
|
||||
A real ruling is measured from the page and is always right; a model
|
||||
separator is a guess about a gutter. Where they agree the ruling wins the
|
||||
position, so enabling the model can never move a line that was actually
|
||||
drawn on the page.
|
||||
"""
|
||||
out = sorted(float(x) for x in (heuristic or []))
|
||||
for x in sorted(float(x) for x in (structure or [])):
|
||||
if not any(abs(x - kept) <= tolerance for kept in out):
|
||||
out.append(x)
|
||||
out.sort()
|
||||
return out
|
||||
|
||||
|
||||
__all__ = [
|
||||
"TableStructure",
|
||||
"last_load_error",
|
||||
"merge_rulings",
|
||||
"reset_session_for_tests",
|
||||
"structure_for_region",
|
||||
"table_structure_enabled",
|
||||
]
|
||||
@@ -0,0 +1,129 @@
|
||||
"""Page classification: digital | scan | hybrid | blank (hardened)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from itertools import pairwise
|
||||
|
||||
from app.services.convert.idm.model import PageKind
|
||||
|
||||
|
||||
def classify_page(
|
||||
*,
|
||||
glyph_count: int,
|
||||
text_chars: int,
|
||||
width: float,
|
||||
height: float,
|
||||
image_coverage: float = 0.0,
|
||||
glyph_size_variance: float = 0.0,
|
||||
) -> PageKind:
|
||||
"""
|
||||
Classify a page.
|
||||
|
||||
image_coverage: fraction of page area covered by full-page-ish images (0..1).
|
||||
glyph_size_variance: variance of glyph font sizes (high + sparse text → hybrid/scan).
|
||||
"""
|
||||
area = max(width * height, 1.0)
|
||||
density = text_chars / area
|
||||
|
||||
if text_chars < 8 and glyph_count < 8:
|
||||
if image_coverage >= 0.35:
|
||||
return PageKind.scan
|
||||
return PageKind.blank if text_chars == 0 else PageKind.scan
|
||||
|
||||
# Large image backdrop with some digital text → hybrid (OCR empty regions only)
|
||||
if image_coverage >= 0.45 and text_chars >= 20:
|
||||
return PageKind.hybrid
|
||||
if image_coverage >= 0.55 and text_chars < 40:
|
||||
return PageKind.scan
|
||||
|
||||
if glyph_count >= 40 or text_chars >= 80 or density > 0.00015:
|
||||
# High size variance with modest glyphs can still be hybrid
|
||||
if glyph_size_variance > 40.0 and image_coverage >= 0.25:
|
||||
return PageKind.hybrid
|
||||
return PageKind.digital
|
||||
|
||||
# Short digital pages (title-only / few lines) with real glyphs and no backdrop
|
||||
if glyph_count >= 8 and image_coverage < 0.25:
|
||||
return PageKind.digital
|
||||
|
||||
if text_chars < 40:
|
||||
return PageKind.scan
|
||||
return PageKind.hybrid
|
||||
|
||||
|
||||
def estimate_image_coverage(image_blocks: list[dict] | None, width: float, height: float) -> float:
|
||||
"""Fraction of the page covered by images, in [0, 1].
|
||||
|
||||
When the blocks carry real positions the *union* area is measured, not the
|
||||
sum: a scan is often drawn as several overlapping strips, and summing them
|
||||
exceeds the page area and pins coverage at 1.0. Since coverage decides
|
||||
whether a page is classified as a scan — and therefore whether it takes the
|
||||
expensive OCR path — over-counting sends text pages to OCR needlessly.
|
||||
|
||||
Blocks without positions fall back to the summed estimate.
|
||||
"""
|
||||
if not image_blocks or width <= 0 or height <= 0:
|
||||
return 0.0
|
||||
page_area = width * height
|
||||
|
||||
rects: list[tuple[float, float, float, float]] = []
|
||||
summed = 0.0
|
||||
positioned = True
|
||||
for img in image_blocks:
|
||||
w = float(img.get("w") or img.get("width") or 0)
|
||||
h = float(img.get("h") or img.get("height") or 0)
|
||||
w, h = max(0.0, w), max(0.0, h)
|
||||
summed += w * h
|
||||
if "x" not in img or "y" not in img:
|
||||
positioned = False
|
||||
continue
|
||||
x, y = float(img["x"]), float(img["y"])
|
||||
if w > 0 and h > 0:
|
||||
rects.append((x, y, x + w, y + h))
|
||||
|
||||
if not positioned or not rects:
|
||||
return min(1.0, summed / page_area)
|
||||
return min(1.0, _union_area(rects) / page_area)
|
||||
|
||||
|
||||
def _union_area(rects: list[tuple[float, float, float, float]]) -> float:
|
||||
"""Area of the union of axis-aligned rectangles, by coordinate sweep.
|
||||
|
||||
Exact rather than approximate, and cheap at the handful of images a PDF
|
||||
page carries.
|
||||
"""
|
||||
xs = sorted({v for r in rects for v in (r[0], r[2])})
|
||||
if len(xs) < 2:
|
||||
return 0.0
|
||||
total = 0.0
|
||||
for left, right in pairwise(xs):
|
||||
strip_w = right - left
|
||||
if strip_w <= 0:
|
||||
continue
|
||||
spans = sorted(
|
||||
(r[1], r[3]) for r in rects if r[0] <= left and r[2] >= right and r[3] > r[1]
|
||||
)
|
||||
covered = 0.0
|
||||
cur_lo = cur_hi = None
|
||||
for lo, hi in spans:
|
||||
if cur_hi is None or lo > cur_hi:
|
||||
if cur_hi is not None:
|
||||
covered += cur_hi - cur_lo
|
||||
cur_lo, cur_hi = lo, hi
|
||||
elif hi > cur_hi:
|
||||
cur_hi = hi
|
||||
if cur_hi is not None:
|
||||
covered += cur_hi - cur_lo
|
||||
total += strip_w * covered
|
||||
return total
|
||||
|
||||
|
||||
def glyph_size_variance(glyphs: list[dict] | None) -> float:
|
||||
if not glyphs:
|
||||
return 0.0
|
||||
sizes = [float(g.get("fontSize") or g.get("h") or 0) for g in glyphs if g]
|
||||
sizes = [s for s in sizes if s > 0]
|
||||
if len(sizes) < 2:
|
||||
return 0.0
|
||||
mean = sum(sizes) / len(sizes)
|
||||
return sum((s - mean) ** 2 for s in sizes) / len(sizes)
|
||||
@@ -0,0 +1,31 @@
|
||||
"""Rasterize a PDF page to PNG for layout ML / OCR crops."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
from app.services.convert.pdf_bridge import render_page_png
|
||||
from app.services.convert.validation import run_with_timeout
|
||||
|
||||
|
||||
def layout_ml_dpi(default: float = 150.0) -> int:
|
||||
try:
|
||||
return max(72, int(float(os.environ.get("CONVERT_LAYOUT_ML_DPI", str(default)))))
|
||||
except ValueError:
|
||||
return int(default)
|
||||
|
||||
|
||||
def raster_page(
|
||||
pdf_bytes: bytes,
|
||||
page_index: int,
|
||||
*,
|
||||
dpi: int | None = None,
|
||||
timeout: float = 15.0,
|
||||
) -> bytes:
|
||||
"""Return PNG bytes. Raises on hard failure (caller should fail-open)."""
|
||||
d = dpi if dpi is not None else layout_ml_dpi()
|
||||
return run_with_timeout(
|
||||
lambda: render_page_png(pdf_bytes, page_index, dpi=d, allow_blank=True),
|
||||
timeout,
|
||||
label=f"layout raster page {page_index + 1}",
|
||||
)
|
||||
@@ -0,0 +1,485 @@
|
||||
"""Line → paragraph reconstruction for the deterministic (non-ML) layout path.
|
||||
|
||||
A PDF has no paragraph concept: it has positioned glyphs that ``glyphs.py``
|
||||
clusters into visual lines. Emitting one block per visual line produces a
|
||||
DOCX where every line is its own ``<w:p>`` — text is present but the document
|
||||
does not reflow, so it cannot be edited. This module groups consecutive lines
|
||||
back into paragraphs using the geometric signals a typesetter used to create
|
||||
them.
|
||||
|
||||
A line ends its paragraph when any of these holds:
|
||||
|
||||
* the vertical gap to the next line is materially larger than the body leading
|
||||
(paragraph spacing);
|
||||
* the line is *short* — its right edge stops well before the column's right
|
||||
margin, which is how the last line of a paragraph looks;
|
||||
* the next line is indented relative to the current one (first-line indent
|
||||
marks a new paragraph);
|
||||
* the next line starts a list item, or either line is a heading;
|
||||
* the style changes materially (font family, or size beyond a tolerance);
|
||||
* the flow leaves the column (the next line sits higher on the page, or its
|
||||
left edge jumps by more than a column's width).
|
||||
|
||||
All thresholds are relative to measured page statistics — body font size,
|
||||
modal leading, column right margin — so the module does not assume a page
|
||||
size, a font, or a language.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import statistics
|
||||
from dataclasses import dataclass
|
||||
from itertools import pairwise
|
||||
|
||||
from app.services.convert.layout.glyphs import Line
|
||||
from app.services.convert.layout.styles import TERMINAL_PUNCT_RE
|
||||
|
||||
# A line whose right edge falls short of the column margin by more than
|
||||
# SHORT_LINE_RATIO of the column width is treated as a paragraph's last line.
|
||||
SHORT_LINE_RATIO = 0.12
|
||||
# Next line indented by more than this fraction of column width starts a paragraph.
|
||||
INDENT_RATIO = 0.035
|
||||
# Vertical gap beyond leading * PARA_GAP_RATIO separates paragraphs.
|
||||
PARA_GAP_RATIO = 1.42
|
||||
# Font size change beyond this fraction ends the paragraph. A line's size is
|
||||
# taken from its first span (see ``glyphs``), so it is a stable number and the
|
||||
# tolerance only needs to absorb text-matrix scaling drift, which is well under
|
||||
# 1%. It used to be 0.18, which permitted a two-point change on 11pt body text --
|
||||
# the exact distance from Word's default body to its default Heading 3 -- so a
|
||||
# heading following a paragraph was absorbed into it. Type size changing between
|
||||
# two lines means their roles differ; that is what this test is for. At 0.06 a
|
||||
# one-point change on 11pt or 12pt text (8-9%) separates, with ample headroom
|
||||
# left for scaling drift. ``idm_optimize`` imports this so the two passes that
|
||||
# decide paragraph boundaries cannot disagree about what "the same size" means.
|
||||
SIZE_TOLERANCE = 0.06
|
||||
# Left edges within this fraction of column width count as aligned.
|
||||
ALIGN_RATIO = 0.02
|
||||
# ``prev`` inset from the column's left margin by more than this is not
|
||||
# left-aligned: it is centred, right-aligned or a block inset. Set above any
|
||||
# plausible first-line indent (a one-inch indent on a 6.5in column is 15%) so
|
||||
# ordinary indented paragraphs are untouched.
|
||||
INSET_LINE_RATIO = 0.20
|
||||
|
||||
# ruff: noqa: RUF001 — the non-ASCII characters in these patterns are the
|
||||
# subject matter: CJK and Arabic sentence punctuation, and the hyphen
|
||||
# variants (soft, non-breaking, unicode) that typesetters use to wrap lines.
|
||||
_BULLET_RE = re.compile(r"^\s*(•|[-*–—]|○|●|◦|▪)\s+")
|
||||
_NUMBER_RE = re.compile(r"^\s*(\(?\d{1,3}[.)]|\(\d{1,3}\))\s+")
|
||||
_LETTER_RE = re.compile(r"^\s*(\(?[A-Za-z][.)]|\([A-Za-z]\))\s+")
|
||||
_ROMAN_RE = re.compile(r"^\s*\(?(?=[ivxlcdm]{1,7}[.)])[ivxlcdm]+[.)]\s+", re.I)
|
||||
# Sentence-final punctuation lives in ``styles`` -- see TERMINAL_PUNCT_RE there
|
||||
# for why. ``styles`` used it without defining it, which raised NameError and
|
||||
# failed the whole conversion on any document containing a bold, larger-than-body,
|
||||
# short line that does not end a sentence (the IRS 1040 and Mozilla PDF-spec
|
||||
# corpus samples both do).
|
||||
_HYPHEN_END_RE = re.compile(r"(\w)[\-‐‑]\s*$")
|
||||
|
||||
|
||||
def is_list_start(text: str) -> bool:
|
||||
"""True when the line opens with a bullet, number, letter or roman marker."""
|
||||
t = text or ""
|
||||
return bool(
|
||||
_BULLET_RE.match(t) or _NUMBER_RE.match(t) or _LETTER_RE.match(t) or _ROMAN_RE.match(t)
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class PageMetrics:
|
||||
"""Measured page statistics that thresholds are expressed relative to."""
|
||||
|
||||
body_size: float
|
||||
leading: float
|
||||
col_left: float
|
||||
col_right: float
|
||||
|
||||
@property
|
||||
def col_width(self) -> float:
|
||||
return max(self.col_right - self.col_left, 1.0)
|
||||
|
||||
|
||||
def measure(lines: list[Line], *, body_size: float, page_width: float) -> PageMetrics:
|
||||
"""Derive body leading and column margins from the lines themselves."""
|
||||
if not lines:
|
||||
return PageMetrics(body_size or 12.0, (body_size or 12.0) * 1.2, 0.0, page_width or 612.0)
|
||||
|
||||
gaps: list[float] = []
|
||||
for prev, cur in pairwise(lines):
|
||||
dy = prev.y - cur.y
|
||||
# Ignore column jumps and pathological spacing when measuring leading.
|
||||
if 0.5 < dy < (body_size or 12.0) * 3.5:
|
||||
gaps.append(dy)
|
||||
leading = statistics.median(gaps) if gaps else max((body_size or 12.0) * 1.2, 1.0)
|
||||
|
||||
lefts = sorted(ln.x0 for ln in lines)
|
||||
rights = sorted(ln.x1 for ln in lines)
|
||||
# Robust margins: ignore outliers from stray marks and over-wide rules.
|
||||
col_left = lefts[max(0, int(len(lefts) * 0.10))]
|
||||
col_right = rights[min(len(rights) - 1, int(len(rights) * 0.90))]
|
||||
if col_right <= col_left:
|
||||
col_left, col_right = 0.0, page_width or 612.0
|
||||
return PageMetrics(
|
||||
body_size=body_size or 12.0, leading=leading, col_left=col_left, col_right=col_right
|
||||
)
|
||||
|
||||
|
||||
def _style_changed(a: Line, b: Line, metrics: PageMetrics) -> bool:
|
||||
fa, fb = a.font_size or metrics.body_size, b.font_size or metrics.body_size
|
||||
if fa > 0 and abs(fa - fb) / fa > SIZE_TOLERANCE:
|
||||
return True
|
||||
na = (a.font_name or "").split("+")[-1].split("-")[0].lower()
|
||||
nb = (b.font_name or "").split("+")[-1].split("-")[0].lower()
|
||||
return bool(na and nb and na != nb)
|
||||
|
||||
|
||||
def _column_lefts(lines: list[Line], metrics: PageMetrics) -> list[float]:
|
||||
"""Modal left edges of two (or three) text columns, or empty if one column.
|
||||
|
||||
First-line indents must not create a fake column: they sit a few points
|
||||
inward of a real column's left edge and are absorbed by the cluster snap.
|
||||
"""
|
||||
if len(lines) < 4:
|
||||
return []
|
||||
xs = sorted(ln.x0 for ln in lines)
|
||||
snap = max(18.0, metrics.col_width * 0.10)
|
||||
clusters: list[list[float]] = [[xs[0]]]
|
||||
for x in xs[1:]:
|
||||
if x - clusters[-1][-1] <= snap:
|
||||
clusters[-1].append(x)
|
||||
else:
|
||||
clusters.append([x])
|
||||
min_n = max(2, int(len(lines) * 0.12))
|
||||
lefts = sorted(sum(c) / len(c) for c in clusters if len(c) >= min_n)
|
||||
if len(lefts) < 2:
|
||||
return []
|
||||
sep = metrics.col_width * 0.18
|
||||
kept = [lefts[0]]
|
||||
for x in lefts[1:]:
|
||||
if x - kept[-1] >= sep:
|
||||
kept.append(x)
|
||||
return kept if len(kept) >= 2 else []
|
||||
|
||||
|
||||
def _line_column(ln: Line, lefts: list[float]) -> int:
|
||||
return min(range(len(lefts)), key=lambda i: abs(ln.x0 - lefts[i]))
|
||||
|
||||
|
||||
def _peer_column_edges(
|
||||
prev: Line,
|
||||
peers: list[Line] | None,
|
||||
metrics: PageMetrics,
|
||||
*,
|
||||
column_lefts: list[float] | None = None,
|
||||
) -> tuple[float, float]:
|
||||
"""Left/right margins of the column ``prev`` actually sits in.
|
||||
|
||||
Measuring one pair of margins across a two-column page makes every line in
|
||||
the left column look short of the right column's edge, so each typeset
|
||||
line became its own paragraph. Peers in the same column vote their own
|
||||
right edge; first-line indents still belong to that column.
|
||||
|
||||
That vote is only taken when columns were actually detected. The old code
|
||||
fell back to "peers sharing ``prev``'s left edge" whenever they were not,
|
||||
which is unsound on a single-column page: selecting peers by left edge
|
||||
selects for headings, list labels and table cells, so the 90th percentile of
|
||||
their right edges sits far inside the real margin. On the fidelity fixture it
|
||||
put the margin at 359.9 when the text column runs to 521.2, and a caption
|
||||
ending at 334.2 -- a clear 41% short of the true margin -- was read as a full
|
||||
line and welded onto the paragraph above it. The bias runs one way: it makes
|
||||
short lines look full, and a short line is the main evidence a paragraph
|
||||
ended. When no columns are detected the page-level margins are the honest
|
||||
answer, because then the page *is* the column.
|
||||
"""
|
||||
if not peers:
|
||||
return metrics.col_left, metrics.col_right
|
||||
lefts = column_lefts if column_lefts is not None else _column_lefts(peers, metrics)
|
||||
if not lefts:
|
||||
return metrics.col_left, metrics.col_right
|
||||
cid = _line_column(prev, lefts)
|
||||
col = [ln for ln in peers if _line_column(ln, lefts) == cid]
|
||||
if len(col) < 2:
|
||||
return metrics.col_left, metrics.col_right
|
||||
lefts_px = sorted(ln.x0 for ln in col)
|
||||
rights = sorted(ln.x1 for ln in col)
|
||||
col_left = lefts_px[max(0, int(len(lefts_px) * 0.10))]
|
||||
col_right = rights[min(len(rights) - 1, int(len(rights) * 0.90))]
|
||||
if col_right <= col_left:
|
||||
return metrics.col_left, metrics.col_right
|
||||
return col_left, col_right
|
||||
|
||||
|
||||
def breaks_paragraph(
|
||||
prev: Line,
|
||||
cur: Line,
|
||||
metrics: PageMetrics,
|
||||
*,
|
||||
peers: list[Line] | None = None,
|
||||
column_lefts: list[float] | None = None,
|
||||
) -> bool:
|
||||
"""True when ``cur`` starts a new paragraph rather than continuing ``prev``."""
|
||||
dy = prev.y - cur.y
|
||||
col_left, col_right = _peer_column_edges(
|
||||
prev, peers, metrics, column_lefts=column_lefts
|
||||
)
|
||||
col_width = max(col_right - col_left, 1.0)
|
||||
|
||||
# Flow left the column: reading order moved up the page or across a gutter.
|
||||
if dy <= -metrics.leading * 0.5:
|
||||
return True
|
||||
if abs(cur.x0 - prev.x0) > col_width * 0.55:
|
||||
return True
|
||||
|
||||
# Explicit list markers always start their own block.
|
||||
if is_list_start(cur.text) or is_list_start(prev.text):
|
||||
return True
|
||||
|
||||
# A trailing hyphen is the typesetter's own wrap marker — always continue,
|
||||
# provided the flow has not left the column (checked above).
|
||||
if _HYPHEN_END_RE.search(prev.text or ""):
|
||||
return False
|
||||
|
||||
if _style_changed(prev, cur, metrics):
|
||||
return True
|
||||
|
||||
# Paragraph spacing.
|
||||
if dy > metrics.leading * PARA_GAP_RATIO:
|
||||
return True
|
||||
|
||||
# Short previous line = last line of a paragraph. A line that ends with
|
||||
# sentence-final punctuation and stops short is a strong signal; one that
|
||||
# merely stops short is weaker, so require a wider shortfall.
|
||||
shortfall = col_right - prev.x1
|
||||
terminal = bool(TERMINAL_PUNCT_RE.search(prev.text or ""))
|
||||
if terminal and shortfall > col_width * SHORT_LINE_RATIO:
|
||||
return True
|
||||
if not terminal and shortfall > col_width * (SHORT_LINE_RATIO * 2.2):
|
||||
return True
|
||||
|
||||
# First-line indent on the incoming line.
|
||||
if cur.x0 - prev.x0 > col_width * INDENT_RATIO:
|
||||
return True
|
||||
|
||||
# Alignment returned to the margin: ``prev`` starts well inside the column
|
||||
# while ``cur`` starts at its left edge. Lines of one left-aligned paragraph
|
||||
# share a left edge, and a centred or right-aligned run keeps every line
|
||||
# inset, so a line dropping back to the margin under an inset line belongs to
|
||||
# a new paragraph. Without this a right-aligned line swallowed the paragraph
|
||||
# after it: right alignment puts prev.x1 *at* the margin, which defeats the
|
||||
# short-line test above -- the usual evidence that a paragraph ended.
|
||||
if (
|
||||
prev.x0 - col_left > col_width * INSET_LINE_RATIO
|
||||
and abs(cur.x0 - col_left) <= col_width * ALIGN_RATIO
|
||||
):
|
||||
return True
|
||||
|
||||
# Centred standalone line (both edges inset) reads as a caption or title.
|
||||
left_inset = prev.x0 - col_left
|
||||
right_inset = col_right - prev.x1
|
||||
return (
|
||||
left_inset > col_width * 0.15
|
||||
and right_inset > col_width * 0.15
|
||||
and abs(left_inset - right_inset) < col_width * 0.08
|
||||
)
|
||||
|
||||
|
||||
# --- text-only path (fabricated geometry) ----------------------------------
|
||||
# Minimum characters before a line is considered wide enough to have wrapped.
|
||||
MIN_WRAP_CHARS = 24
|
||||
# A line must reach this fraction of the measured column width to count as full.
|
||||
FULL_LINE_RATIO = 0.85
|
||||
# Percentile of line lengths taken as the column width. The maximum is a poor
|
||||
# estimator — a single long table row or URL skews it by 2x on real documents.
|
||||
FULL_LINE_PERCENTILE = 0.90
|
||||
|
||||
|
||||
# A line this many times longer than the median is a spanning title or banner,
|
||||
# not a wrapped body line, and stands as its own block.
|
||||
SPANNING_LINE_RATIO = 1.8
|
||||
|
||||
|
||||
def measure_full_line_chars(lines: list[Line]) -> int:
|
||||
"""Estimate the column width in characters from the line-length distribution."""
|
||||
lengths = sorted(len((ln.text or "").strip()) for ln in lines if (ln.text or "").strip())
|
||||
if not lengths:
|
||||
return 0
|
||||
idx = min(len(lengths) - 1, int(len(lengths) * FULL_LINE_PERCENTILE))
|
||||
return lengths[idx]
|
||||
|
||||
|
||||
def measure_median_line_chars(lines: list[Line]) -> int:
|
||||
lengths = sorted(len((ln.text or "").strip()) for ln in lines if (ln.text or "").strip())
|
||||
if not lengths:
|
||||
return 0
|
||||
return lengths[len(lengths) // 2]
|
||||
|
||||
|
||||
def _breaks_paragraph_text_only(
|
||||
prev: Line, cur: Line, full_width_chars: int, median_chars: int = 0
|
||||
) -> bool:
|
||||
"""Paragraph break decision when line coordinates are fabricated.
|
||||
|
||||
Without real geometry the only trustworthy signals are the text itself and
|
||||
the blank lines the extractor preserved. The rule is deliberately
|
||||
conservative: two lines join only when the first is a *full* line with no
|
||||
sentence-final punctuation, which is what a wrapped line looks like.
|
||||
Anything else stays separate, because a wrong merge destroys structure
|
||||
while a missed merge merely leaves it as found.
|
||||
"""
|
||||
prev_text = (prev.text or "").strip()
|
||||
cur_text = (cur.text or "").strip()
|
||||
if not prev_text or not cur_text:
|
||||
return True
|
||||
|
||||
# Blank line in the source is encoded as a wider vertical step.
|
||||
if (prev.y - cur.y) > 20.0:
|
||||
return True
|
||||
if is_list_start(cur_text) or is_list_start(prev_text):
|
||||
return True
|
||||
# A trailing hyphen is the typesetter's own wrap marker — always continue.
|
||||
if _HYPHEN_END_RE.search(prev_text):
|
||||
return False
|
||||
if TERMINAL_PUNCT_RE.search(prev_text):
|
||||
return True
|
||||
# A line far longer than the page median is a spanning title or banner —
|
||||
# the percentile column-width estimate is skewed by exactly such lines, so
|
||||
# guard against absorbing the body text that follows one.
|
||||
if median_chars and len(prev_text) > median_chars * SPANNING_LINE_RATIO:
|
||||
return True
|
||||
if len(prev_text) < MIN_WRAP_CHARS:
|
||||
return True
|
||||
return bool(full_width_chars and len(prev_text) < full_width_chars * FULL_LINE_RATIO)
|
||||
|
||||
|
||||
def group_lines(lines: list[Line], metrics: PageMetrics) -> list[list[Line]]:
|
||||
"""Group ordered lines into paragraph runs.
|
||||
|
||||
Lines whose geometry was fabricated (``synthetic_geometry``) are grouped by
|
||||
the conservative text-only rule; measured lines use the full geometric one.
|
||||
"""
|
||||
if not lines:
|
||||
return []
|
||||
|
||||
synthetic = sum(1 for ln in lines if getattr(ln, "synthetic_geometry", False))
|
||||
text_only = synthetic > len(lines) / 2
|
||||
full_width_chars = measure_full_line_chars(lines)
|
||||
median_chars = measure_median_line_chars(lines)
|
||||
|
||||
column_lefts = [] if text_only else _column_lefts(lines, metrics)
|
||||
if not text_only and len(column_lefts) >= 2:
|
||||
return _group_lines_by_column(lines, metrics, column_lefts)
|
||||
groups: list[list[Line]] = [[lines[0]]]
|
||||
for prev, cur in pairwise(lines):
|
||||
if text_only:
|
||||
brk = _breaks_paragraph_text_only(prev, cur, full_width_chars, median_chars)
|
||||
else:
|
||||
brk = breaks_paragraph(
|
||||
prev, cur, metrics, peers=lines, column_lefts=column_lefts
|
||||
)
|
||||
if brk:
|
||||
groups.append([cur])
|
||||
else:
|
||||
groups[-1].append(cur)
|
||||
return groups
|
||||
|
||||
|
||||
def _group_consecutive(lines: list[Line], metrics: PageMetrics, column_lefts: list[float]) -> list[list[Line]]:
|
||||
if not lines:
|
||||
return []
|
||||
groups: list[list[Line]] = [[lines[0]]]
|
||||
for prev, cur in pairwise(lines):
|
||||
if breaks_paragraph(prev, cur, metrics, peers=lines, column_lefts=column_lefts):
|
||||
groups.append([cur])
|
||||
else:
|
||||
groups[-1].append(cur)
|
||||
return groups
|
||||
|
||||
|
||||
def _group_lines_by_column(
|
||||
lines: list[Line], metrics: PageMetrics, column_lefts: list[float]
|
||||
) -> list[list[Line]]:
|
||||
"""Group wrap inside each column, then concatenate (titles that span first)."""
|
||||
spanning: list[Line] = []
|
||||
buckets: list[list[Line]] = [[] for _ in column_lefts]
|
||||
for ln in lines:
|
||||
if (ln.x1 - ln.x0) >= metrics.col_width * 0.72:
|
||||
spanning.append(ln)
|
||||
else:
|
||||
buckets[_line_column(ln, column_lefts)].append(ln)
|
||||
groups: list[list[Line]] = []
|
||||
for bucket in [spanning, *buckets]:
|
||||
bucket = sorted(bucket, key=lambda ln: -ln.y)
|
||||
groups.extend(_group_consecutive(bucket, metrics, column_lefts))
|
||||
return groups
|
||||
|
||||
|
||||
_MIDLINE_HYPHEN_RE = re.compile(r"(?<![\w-])([A-Za-z]{2,})-([A-Za-z]{2,})(?![\w-])")
|
||||
|
||||
|
||||
def lexical_hyphen_pairs(text: str) -> set[tuple[str, str]]:
|
||||
"""Hyphenated word pairs the document itself uses *mid-line*.
|
||||
|
||||
A compound the author writes as ``well-known`` in running text is lexical,
|
||||
not a typesetter's line break. Collecting those pairs from the document
|
||||
gives a reliable, language-independent keep-list at no cost and without a
|
||||
dictionary dependency.
|
||||
"""
|
||||
pairs: set[tuple[str, str]] = set()
|
||||
for raw_line in (text or "").splitlines():
|
||||
# Only mid-line hyphens count; a hyphen at end of line is the ambiguous case.
|
||||
line = raw_line.rstrip()
|
||||
if line.endswith("-"):
|
||||
line = line[:-1]
|
||||
for m in _MIDLINE_HYPHEN_RE.finditer(line):
|
||||
pairs.add((m.group(1).lower(), m.group(2).lower()))
|
||||
return pairs
|
||||
|
||||
|
||||
def join_line_texts(
|
||||
texts: list[str], *, lexical_pairs: set[tuple[str, str]] | None = None
|
||||
) -> str:
|
||||
"""Join wrapped line texts, resolving end-of-line hyphenation.
|
||||
|
||||
End-of-line hyphens are ambiguous: they may be a typesetter's break
|
||||
(``inter-`` / ``national`` → ``international``) or part of a real compound
|
||||
(``well-`` / ``known`` → ``well-known``). The hyphen is kept when:
|
||||
|
||||
* the pair appears hyphenated mid-line elsewhere in the document, or
|
||||
* the following fragment starts with a capital or a digit, or
|
||||
* either fragment is too short to be a plausible split.
|
||||
|
||||
Otherwise the hyphen is removed, which is the common case in wrapped prose.
|
||||
"""
|
||||
if not texts:
|
||||
return ""
|
||||
pairs = lexical_pairs or set()
|
||||
out = texts[0].strip()
|
||||
for raw in texts[1:]:
|
||||
nxt = raw.strip()
|
||||
if not nxt:
|
||||
continue
|
||||
if not out:
|
||||
out = nxt
|
||||
continue
|
||||
|
||||
# Soft hyphen: always a break marker, never part of the word.
|
||||
if out.endswith(""):
|
||||
out = out[:-1] + nxt
|
||||
continue
|
||||
|
||||
m = _HYPHEN_END_RE.search(out)
|
||||
if not m:
|
||||
out = f"{out} {nxt}"
|
||||
continue
|
||||
|
||||
head = out[: m.end(1)] # everything through the letter, hyphen excluded
|
||||
left_frag = re.search(r"([A-Za-z]+)[\-‐‑]\s*$", out)
|
||||
right_frag = re.match(r"([A-Za-z0-9]+)", nxt)
|
||||
keep_hyphen = True
|
||||
if left_frag and right_frag:
|
||||
lf, rf = left_frag.group(1), right_frag.group(1)
|
||||
if (lf.lower(), rf.lower()) in pairs or rf[:1].isupper() or rf[:1].isdigit() or len(lf) < 2 or len(rf) < 2:
|
||||
keep_hyphen = True
|
||||
else:
|
||||
keep_hyphen = False
|
||||
out = f"{head}-{nxt}" if keep_hyphen else f"{head}{nxt}"
|
||||
return out
|
||||
@@ -0,0 +1,224 @@
|
||||
"""Document-level PDF routing (pdf-inspector-inspired, in-house).
|
||||
|
||||
Classifies TextBased / Scanned / ImageBased / Mixed and lists pages needing OCR.
|
||||
Does not call Firecrawl or pdf-inspector — pure heuristics over pypdf + optional engine.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from dataclasses import dataclass, field
|
||||
from enum import Enum
|
||||
|
||||
from pypdf import PdfReader
|
||||
|
||||
from app.services.convert import doc_cache
|
||||
|
||||
|
||||
class PdfDocType(str, Enum):
|
||||
text_based = "text_based"
|
||||
scanned = "scanned"
|
||||
image_based = "image_based"
|
||||
mixed = "mixed"
|
||||
|
||||
|
||||
class ScanStrategy(str, Enum):
|
||||
early_exit = "early_exit"
|
||||
full = "full"
|
||||
sample = "sample"
|
||||
|
||||
|
||||
@dataclass
|
||||
class PdfRouteResult:
|
||||
doc_type: PdfDocType
|
||||
confidence: float
|
||||
pages_needing_ocr: list[int] = field(default_factory=list)
|
||||
pages_sampled: list[int] = field(default_factory=list)
|
||||
strategy: str = "full"
|
||||
|
||||
def as_dict(self) -> dict:
|
||||
return {
|
||||
"doc_type": self.doc_type.value,
|
||||
"confidence": round(self.confidence, 4),
|
||||
"pages_needing_ocr": list(self.pages_needing_ocr),
|
||||
"pages_sampled": list(self.pages_sampled),
|
||||
"strategy": self.strategy,
|
||||
}
|
||||
|
||||
|
||||
def _strategy_from_env() -> tuple[ScanStrategy, int]:
|
||||
raw = (os.environ.get("CONVERT_PDF_SCAN_STRATEGY") or "full").strip().lower()
|
||||
try:
|
||||
n = int(os.environ.get("CONVERT_PDF_SCAN_SAMPLE", "5"))
|
||||
except ValueError:
|
||||
n = 5
|
||||
n = max(2, min(n, 50))
|
||||
if raw in ("early_exit", "early-exit", "early"):
|
||||
return ScanStrategy.early_exit, n
|
||||
if raw in ("sample", "sampled"):
|
||||
return ScanStrategy.sample, n
|
||||
return ScanStrategy.full, n
|
||||
|
||||
|
||||
def _page_needs_ocr(reader: PdfReader, index: int, data: bytes | None = None) -> tuple[bool, str]:
|
||||
"""Heuristic: sparse / garbled extractable text → OCR. Returns (needs_ocr, reason)."""
|
||||
from app.services.convert.layout.text_quality import is_garbled_text, page_text_needs_ocr
|
||||
|
||||
try:
|
||||
page = reader.pages[index]
|
||||
text = (
|
||||
doc_cache.page_text(data, index) if data is not None else (page.extract_text() or "")
|
||||
).strip()
|
||||
except Exception:
|
||||
return True, "extract_failed"
|
||||
|
||||
# Image XObjects hint
|
||||
has_image = False
|
||||
try:
|
||||
resources = page.get("/Resources") or {}
|
||||
if hasattr(resources, "get_object"):
|
||||
resources = resources.get_object()
|
||||
xobj = resources.get("/XObject") if resources else None
|
||||
if xobj is not None:
|
||||
if hasattr(xobj, "get_object"):
|
||||
xobj = xobj.get_object()
|
||||
for _name, ref in (xobj.items() if hasattr(xobj, "items") else []):
|
||||
try:
|
||||
obj = ref.get_object() if hasattr(ref, "get_object") else ref
|
||||
if obj.get("/Subtype") == "/Image":
|
||||
has_image = True
|
||||
break
|
||||
except Exception:
|
||||
continue
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
chars = len(text)
|
||||
# Image-only / nearly empty → OCR. Decorative images beside real text → keep digital.
|
||||
if chars < 12:
|
||||
return True, "image_only" if has_image else "sparse_text"
|
||||
# Broken ToUnicode / CID garbage (Huwiyati-class) must not count as "text_ok"
|
||||
if is_garbled_text(text) or page_text_needs_ocr(text, image_coverage=0.2 if has_image else 0.0):
|
||||
return True, "garbled_text"
|
||||
return False, "text_ok"
|
||||
|
||||
|
||||
def _sample_indices(n_pages: int, strategy: ScanStrategy, sample_n: int) -> list[int]:
|
||||
if n_pages <= 0:
|
||||
return []
|
||||
if strategy == ScanStrategy.full or strategy == ScanStrategy.early_exit:
|
||||
return list(range(n_pages))
|
||||
# sample: first, last, evenly spaced middle
|
||||
if n_pages <= sample_n:
|
||||
return list(range(n_pages))
|
||||
idxs = {0, n_pages - 1}
|
||||
mid_slots = sample_n - 2
|
||||
if mid_slots > 0:
|
||||
step = (n_pages - 1) / (mid_slots + 1)
|
||||
for i in range(1, mid_slots + 1):
|
||||
idxs.add(min(n_pages - 1, max(0, int(round(i * step)))))
|
||||
return sorted(idxs)
|
||||
|
||||
|
||||
def route_pdf(data: bytes, *, strategy: ScanStrategy | None = None) -> PdfRouteResult:
|
||||
"""Classify PDF and list 0-based page indices that need OCR."""
|
||||
strat, sample_n = _strategy_from_env()
|
||||
if strategy is not None:
|
||||
strat = strategy
|
||||
|
||||
try:
|
||||
reader = doc_cache.get_reader(data)
|
||||
except Exception:
|
||||
return PdfRouteResult(
|
||||
doc_type=PdfDocType.scanned,
|
||||
confidence=0.3,
|
||||
pages_needing_ocr=[],
|
||||
pages_sampled=[],
|
||||
strategy=strat.value,
|
||||
)
|
||||
|
||||
n_pages = len(reader.pages)
|
||||
indices = _sample_indices(n_pages, strat, sample_n)
|
||||
needing: list[int] = []
|
||||
sampled: list[int] = []
|
||||
text_ok = 0
|
||||
image_heavy = 0
|
||||
garbled_heavy = 0
|
||||
|
||||
for i in indices:
|
||||
sampled.append(i)
|
||||
needs, reason = _page_needs_ocr(reader, i, data)
|
||||
if needs:
|
||||
needing.append(i)
|
||||
if reason in ("image_only", "hybrid_sparse"):
|
||||
image_heavy += 1
|
||||
if reason == "garbled_text":
|
||||
garbled_heavy += 1
|
||||
else:
|
||||
text_ok += 1
|
||||
if strat == ScanStrategy.early_exit and needing and not text_ok:
|
||||
# First page already needs OCR and no prior text → treat as scanned early
|
||||
break
|
||||
|
||||
# Expand early_exit / sample results to full page lists when mixed
|
||||
if strat == ScanStrategy.early_exit and needing and text_ok == 0 and len(sampled) < n_pages:
|
||||
# Likely fully scanned — mark all pages
|
||||
needing = list(range(n_pages))
|
||||
sampled = list(range(n_pages))
|
||||
elif strat == ScanStrategy.sample and needing:
|
||||
# Conservative: any sampled OCR page → include that page; do not invent others
|
||||
pass
|
||||
|
||||
total = max(len(sampled), 1)
|
||||
ocr_frac = len(needing) / total
|
||||
text_frac = text_ok / total
|
||||
garbled_frac = garbled_heavy / total
|
||||
|
||||
# Majority garbled / OCR-needed → promote whole document (hybrid RFP class)
|
||||
if ocr_frac >= 0.5 and len(sampled) < n_pages:
|
||||
needing = list(range(n_pages))
|
||||
sampled = list(range(n_pages))
|
||||
ocr_frac = 1.0
|
||||
text_frac = 0.0
|
||||
elif ocr_frac >= 0.5 and n_pages > 0 and len(needing) < n_pages:
|
||||
needing = list(range(n_pages))
|
||||
ocr_frac = 1.0
|
||||
text_frac = 0.0
|
||||
|
||||
if ocr_frac >= 0.95:
|
||||
if garbled_frac >= 0.4:
|
||||
doc_type = PdfDocType.mixed if text_ok > 0 else PdfDocType.scanned
|
||||
else:
|
||||
doc_type = PdfDocType.image_based if image_heavy >= text_ok else PdfDocType.scanned
|
||||
conf = 0.55 + 0.4 * ocr_frac
|
||||
elif ocr_frac <= 0.05:
|
||||
doc_type = PdfDocType.text_based
|
||||
conf = 0.55 + 0.4 * text_frac
|
||||
else:
|
||||
doc_type = PdfDocType.mixed
|
||||
conf = 0.5 + 0.3 * (1.0 - abs(0.5 - ocr_frac))
|
||||
|
||||
# Unique sorted
|
||||
needing = sorted(set(needing))
|
||||
return PdfRouteResult(
|
||||
doc_type=doc_type,
|
||||
confidence=min(1.0, conf),
|
||||
pages_needing_ocr=needing,
|
||||
pages_sampled=sampled,
|
||||
strategy=strat.value,
|
||||
)
|
||||
|
||||
|
||||
def page_kind_hint(doc_type: PdfDocType, page_index: int, pages_needing_ocr: list[int]):
|
||||
"""Map route into PageKind-compatible hint for a page."""
|
||||
from app.services.convert.idm.model import PageKind
|
||||
|
||||
if page_index in pages_needing_ocr:
|
||||
if doc_type == PdfDocType.mixed:
|
||||
return PageKind.hybrid
|
||||
return PageKind.scan
|
||||
if doc_type == PdfDocType.text_based:
|
||||
return PageKind.digital
|
||||
if doc_type == PdfDocType.mixed:
|
||||
return PageKind.digital
|
||||
return PageKind.scan
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,591 @@
|
||||
"""Column detection and reading order via recursive XY-Cut (Ha et al. 1995)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import bisect
|
||||
|
||||
from app.services.convert.layout.glyphs import Line
|
||||
|
||||
MIN_COL_LINES = 2
|
||||
MIN_GUTTER_RATIO = 0.05
|
||||
MIN_BLOCK_LINES = 2
|
||||
|
||||
# A band is read as a column only when it is not simply the left-hand cells of
|
||||
# rows that continue to its right. Above this share of shared baselines the
|
||||
# two sides are rows of one structure — a table, a definition list, a form —
|
||||
# and cutting between them interleaves the document.
|
||||
MAX_SHARED_BASELINES = 0.6
|
||||
|
||||
# ...and only when the candidate side does not fill its own width. Real prose
|
||||
# columns run to their margin; table cells and row labels do not.
|
||||
COLUMN_FILL_RATIO = 0.55
|
||||
|
||||
# Share of body lines that must sit clear of a candidate margin band for it to be
|
||||
# read as a sidebar. Not all of them: a footnote or an indented quote may reach
|
||||
# into the margin without making the band beside it a sidebar.
|
||||
SIDEBAR_CLEAR_RATIO = 0.9
|
||||
|
||||
|
||||
def _mid_x(ln: Line) -> float:
|
||||
return (ln.x0 + ln.x1) / 2.0
|
||||
|
||||
|
||||
def _shared_baseline_ratio(a: list[Line], b: list[Line]) -> float:
|
||||
"""Share of the smaller group whose lines sit on a baseline the other uses.
|
||||
|
||||
Two prose columns have independent line rhythms. The label column of a
|
||||
table sits on exactly the same baselines as the values beside it, which is
|
||||
what tells the two apart without having to detect the table first.
|
||||
"""
|
||||
if not a or not b:
|
||||
return 0.0
|
||||
small, large = (a, b) if len(a) <= len(b) else (b, a)
|
||||
others = sorted(ln.y for ln in large)
|
||||
if not others:
|
||||
return 0.0
|
||||
hits = 0
|
||||
for ln in small:
|
||||
tol = max(3.0, float(ln.font_size or 12.0) * 0.35)
|
||||
i = bisect.bisect_left(others, ln.y - tol)
|
||||
if i < len(others) and others[i] <= ln.y + tol:
|
||||
hits += 1
|
||||
return hits / len(small)
|
||||
|
||||
|
||||
def _fills_its_width(lines: list[Line]) -> bool:
|
||||
"""Whether a group's lines are all about as wide as its widest line.
|
||||
|
||||
Set text is justified or near-justified: every line but the last runs to
|
||||
the same right margin, so mean width sits close to maximum width. Table
|
||||
cells do not — a column holding "5" beside one holding "1199 sec, n=1"
|
||||
has a mean far below its maximum. Measuring against the widest *line*
|
||||
rather than the group's total extent keeps the test meaningful when the
|
||||
group is itself several columns, which is exactly the case a three-column
|
||||
page presents on the first cut.
|
||||
"""
|
||||
if len(lines) < 2:
|
||||
return True
|
||||
widths = [max(0.0, ln.x1 - ln.x0) for ln in lines]
|
||||
widest = max(widths)
|
||||
if widest <= 0:
|
||||
return True
|
||||
return (sum(widths) / len(widths)) / widest >= COLUMN_FILL_RATIO
|
||||
|
||||
|
||||
def _is_row_structure(left: list[Line], right: list[Line]) -> bool:
|
||||
"""True when a candidate vertical cut runs *through* rows, not between columns."""
|
||||
if len(left) < 2 or len(right) < 2:
|
||||
return False
|
||||
if _shared_baseline_ratio(left, right) < MAX_SHARED_BASELINES:
|
||||
return False
|
||||
return not (_fills_its_width(left) and _fills_its_width(right))
|
||||
|
||||
|
||||
def _mid_y(ln: Line) -> float:
|
||||
return ln.y
|
||||
|
||||
|
||||
def _outside_text_column(
|
||||
cand: list[Line], rest: list[Line], *, on_left: bool, gutter: float
|
||||
) -> bool:
|
||||
"""Whether ``cand`` sits outside the body column rather than inside it.
|
||||
|
||||
This is the property that actually makes a margin note a margin note, and
|
||||
without it the sidebar test is a trap. The test used to ask only whether a
|
||||
line's midpoint fell in the outer fifth of the page and whether the line was
|
||||
narrow. On US Letter with a one-inch margin the body itself starts at x0=72,
|
||||
so every line shorter than about 125pt -- roughly twenty characters -- passes
|
||||
both. Three such lines anywhere on a page were enough to relocate them all to
|
||||
the end of it, and short lines at the left margin are not unusual: they are
|
||||
headings, the first column of a table, one-word labels, dates, signature
|
||||
blocks. On the fidelity fixture this moved a Heading 3, a Heading 2 and two
|
||||
table cells out of the middle of the page and appended them after the page
|
||||
number, from x0=67.2 -- the body's own margin.
|
||||
|
||||
A real sidebar is *separated*: the body column begins to its right (or ends
|
||||
to its left) with clear space between. Requiring that separation is what
|
||||
distinguishes the two, and it cannot be inferred from width alone. The share
|
||||
is not 100% because an indented quotation or a footnote may reach into the
|
||||
margin without turning the band beside it into a sidebar.
|
||||
"""
|
||||
if not cand or not rest:
|
||||
return False
|
||||
if on_left:
|
||||
edge = max(ln.x1 for ln in cand) + gutter
|
||||
clear = sum(1 for ln in rest if ln.x0 >= edge)
|
||||
else:
|
||||
edge = min(ln.x0 for ln in cand) - gutter
|
||||
clear = sum(1 for ln in rest if ln.x1 <= edge)
|
||||
return (clear / len(rest)) >= SIDEBAR_CLEAR_RATIO
|
||||
|
||||
|
||||
def _arabic_ratio(lines: list[Line]) -> float:
|
||||
chars = 0
|
||||
ar = 0
|
||||
for ln in lines:
|
||||
for c in ln.text or "":
|
||||
chars += 1
|
||||
if "\u0600" <= c <= "\u06FF":
|
||||
ar += 1
|
||||
return (ar / chars) if chars else 0.0
|
||||
|
||||
|
||||
def _min_gutter_pt(page_width: float, page_height: float) -> float:
|
||||
"""Narrowest horizontal gap that may be read as a column gutter.
|
||||
|
||||
Scaled from the page *width*, not its longest side. Using the longest side
|
||||
tied the required gap to the page height on portrait pages -- 792 * 0.03 =
|
||||
23.76pt on US Letter -- which is wider than the gutter most two-column
|
||||
documents actually use. Measured on page 13 of the TraceMonkey paper, whose
|
||||
columns are cleanly separated by 21.66pt (left text ends at x=295.4, right
|
||||
begins at x=317.0): the cut was rejected, the page fell through to a plain
|
||||
top-to-bottom sort, and the two columns were interleaved line by line. That
|
||||
spliced the right column's "...sponsor Sun Microsystems under Project No.
|
||||
07-127." into the middle of a left-column sentence, corrupting both.
|
||||
Nothing detects this downstream, because every word is still present -- only
|
||||
the order is wrong, and recall cannot see order.
|
||||
|
||||
3% of the width is 18.36pt on Letter and 17.5pt on A4, which admits a normal
|
||||
two-column gutter while staying far above an inter-word space (about 3pt at
|
||||
10pt type). The floor keeps small or unusual page boxes sane.
|
||||
"""
|
||||
# The gutter is a horizontal distance, so scale it from the page width.
|
||||
# Using the longer page side made portrait pages demand a 23.8pt gap on a
|
||||
# 612pt sheet and rejected legitimate 21-24pt academic-paper columns.
|
||||
page = max(page_width, 1.0)
|
||||
return max(8.0, page * 0.03)
|
||||
|
||||
|
||||
def _best_x_gutter(
|
||||
lines: list[Line],
|
||||
x0: float,
|
||||
x1: float,
|
||||
*,
|
||||
min_gutter: float,
|
||||
) -> float | None:
|
||||
"""Return split X at largest empty vertical gutter, or None."""
|
||||
width = x1 - x0
|
||||
if width <= 0 or len(lines) < MIN_COL_LINES * 2:
|
||||
return None
|
||||
# Resolution has to be finer than the gutter being looked for, and 48 bins
|
||||
# across a 612pt page is 12.75pt per bin -- coarser than the 24pt gutter of a
|
||||
# standard two-column paper. Because a bin counts as occupied when text
|
||||
# covers any part of it, that gutter could never register as more than one
|
||||
# empty bin, and one bin measures 12.75pt, which is below every sane
|
||||
# ``min_gutter``. So ``_best_x_gutter`` returned None for pages with a wide,
|
||||
# perfectly clean column gap: measured on page 13 of the TraceMonkey paper,
|
||||
# where every line respects a 23.9pt gap (left column ends at x=293.1, right
|
||||
# begins at x=317.0) and the detector still saw no columns.
|
||||
#
|
||||
# The page then fell through to a plain top-to-bottom sort, which interleaves
|
||||
# the columns line by line. That is what spliced the right column's
|
||||
# "...sponsor Sun Microsystems under Project No. 07-127." through the middle
|
||||
# of the left column's sentence and destroyed both. Recall barely moves --
|
||||
# every word is still on the page -- which is why this survived so long.
|
||||
#
|
||||
# 4pt bins resolve any gutter worth finding. The scan below is linear in the
|
||||
# bin count, so this is not a meaningful cost.
|
||||
bins = 48
|
||||
hist = [0] * bins
|
||||
# Mark every bin a line *covers*, not the single bin its midpoint lands in.
|
||||
# A gutter is a vertical strip with no text in it, so the emptiness test has
|
||||
# to look at extents: scoring midpoints made a line invisible everywhere but
|
||||
# its centre, and a strip crossed by running text was still read as empty.
|
||||
# On the fidelity fixture that put the page's only column cut at x=337.9 with
|
||||
# four full lines passing through it, which sent a right-aligned paragraph
|
||||
# and the last column of a table to the end of the page. Column detection is
|
||||
# the first decision made about a page and everything downstream inherits it.
|
||||
# Membership still goes by midpoint -- that is what decides which side of a
|
||||
# real gutter a line belongs to, and it is a separate question from whether
|
||||
# the gutter is there at all.
|
||||
for ln in lines:
|
||||
m = _mid_x(ln)
|
||||
if m < x0 or m >= x1:
|
||||
continue
|
||||
lo = min(bins - 1, max(0, int((max(ln.x0, x0) - x0) / width * bins)))
|
||||
hi = min(bins - 1, max(0, int((min(ln.x1, x1) - x0) / width * bins)))
|
||||
for idx in range(lo, hi + 1):
|
||||
hist[idx] += 1
|
||||
|
||||
# One crossing is a rule, a stray figure label or a wide caption inside an
|
||||
# otherwise two-column region; a handful is running text.
|
||||
empty_max = max(1, int(len(lines) * 0.05))
|
||||
mid_lo, mid_hi = int(bins * 0.08), int(bins * 0.92)
|
||||
empty_runs: list[tuple[int, int]] = []
|
||||
i = mid_lo
|
||||
while i < mid_hi:
|
||||
if hist[i] <= empty_max:
|
||||
j = i
|
||||
while j < mid_hi and hist[j] <= empty_max:
|
||||
j += 1
|
||||
empty_runs.append((i, j))
|
||||
i = j
|
||||
else:
|
||||
i += 1
|
||||
|
||||
candidates: list[tuple[float, float, float, float]] = []
|
||||
for start, end in empty_runs:
|
||||
estimated_gap = (end - start) / bins * width
|
||||
if estimated_gap < min_gutter * 0.55:
|
||||
continue
|
||||
split = x0 + ((start + end) / 2.0) / bins * width
|
||||
left = [ln for ln in lines if _mid_x(ln) < split]
|
||||
right = [ln for ln in lines if _mid_x(ln) >= split]
|
||||
if len(left) < MIN_COL_LINES or len(right) < MIN_COL_LINES:
|
||||
continue
|
||||
left_edge = max((ln.x1 for ln in left), default=0.0)
|
||||
right_edge = min((ln.x0 for ln in right), default=0.0)
|
||||
gutter_pt = right_edge - left_edge
|
||||
if gutter_pt < min_gutter:
|
||||
continue
|
||||
left_span = max((ln.x1 for ln in left), default=0.0) - min(
|
||||
(ln.x0 for ln in left), default=0.0
|
||||
)
|
||||
right_span = max((ln.x1 for ln in right), default=0.0) - min(
|
||||
(ln.x0 for ln in right), default=0.0
|
||||
)
|
||||
# Reject if either side still spans most of the page (false gutter)
|
||||
if min(left_span, right_span) < width * 0.12:
|
||||
continue
|
||||
if left_span > width * 0.75 or right_span > width * 0.75:
|
||||
continue
|
||||
# The gap between a table's first column and its values is a wide, empty,
|
||||
# full-height band — indistinguishable from a column gutter by geometry
|
||||
# alone. Reading it as a gutter emits every row label first and then every
|
||||
# value, which is how "Blind 5 1 4" became "Blind / Low Vision / ... / 5".
|
||||
if _is_row_structure(left, right):
|
||||
continue
|
||||
candidates.append(
|
||||
(
|
||||
min(left_span, right_span),
|
||||
gutter_pt,
|
||||
-abs(split - (x0 + x1) / 2.0),
|
||||
split,
|
||||
)
|
||||
)
|
||||
if not candidates:
|
||||
return None
|
||||
return max(candidates)[-1]
|
||||
|
||||
|
||||
def _best_y_gutter(
|
||||
lines: list[Line],
|
||||
y0: float,
|
||||
y1: float,
|
||||
*,
|
||||
min_gutter: float,
|
||||
) -> float | None:
|
||||
"""Return horizontal split Y (PDF y-up: higher = top) at largest empty band."""
|
||||
height = y1 - y0
|
||||
if height <= 0 or len(lines) < MIN_BLOCK_LINES * 2:
|
||||
return None
|
||||
bins = 48
|
||||
hist = [0] * bins
|
||||
for ln in lines:
|
||||
m = _mid_y(ln)
|
||||
if m < y0 or m > y1:
|
||||
continue
|
||||
idx = min(bins - 1, max(0, int((m - y0) / height * bins)))
|
||||
hist[idx] += 1
|
||||
|
||||
mid_lo, mid_hi = int(bins * 0.08), int(bins * 0.92)
|
||||
best_gap: tuple[int, int] | None = None
|
||||
best_width = 0
|
||||
i = mid_lo
|
||||
while i < mid_hi:
|
||||
if hist[i] <= 1:
|
||||
j = i
|
||||
while j < mid_hi and hist[j] <= 1:
|
||||
j += 1
|
||||
w = j - i
|
||||
if w > best_width:
|
||||
best_width = w
|
||||
best_gap = (i, j)
|
||||
i = j
|
||||
else:
|
||||
i += 1
|
||||
|
||||
if not best_gap or best_width < max(1, int(bins * MIN_GUTTER_RATIO)):
|
||||
return None
|
||||
split = y0 + ((best_gap[0] + best_gap[1]) / 2.0) / bins * height
|
||||
below = [ln for ln in lines if _mid_y(ln) < split]
|
||||
above = [ln for ln in lines if _mid_y(ln) >= split]
|
||||
if len(below) < MIN_BLOCK_LINES or len(above) < MIN_BLOCK_LINES:
|
||||
return None
|
||||
gutter_pt = best_width / bins * height
|
||||
if gutter_pt < min_gutter:
|
||||
return None
|
||||
return split
|
||||
|
||||
|
||||
def _xy_cut(
|
||||
lines: list[Line],
|
||||
*,
|
||||
x0: float,
|
||||
x1: float,
|
||||
y0: float,
|
||||
y1: float,
|
||||
page_width: float,
|
||||
page_height: float,
|
||||
rtl: bool,
|
||||
depth: int = 0,
|
||||
) -> list[Line]:
|
||||
if not lines or depth > 12:
|
||||
return sorted(lines, key=lambda ln: (-ln.y, ln.x0 if not rtl else -ln.x0))
|
||||
|
||||
min_gutter = _min_gutter_pt(page_width, page_height)
|
||||
# Prefer largest empty gutter: try X then Y (or vice-versa by gap size)
|
||||
x_split = _best_x_gutter(lines, x0, x1, min_gutter=min_gutter)
|
||||
y_split = _best_y_gutter(lines, y0, y1, min_gutter=min_gutter)
|
||||
|
||||
# Score gutters by relative width if both exist
|
||||
use_x = False
|
||||
use_y = False
|
||||
if x_split is not None and y_split is not None:
|
||||
# Prefer vertical (column) cuts when both are viable — reading-order focus
|
||||
use_x = True
|
||||
elif x_split is not None:
|
||||
use_x = True
|
||||
elif y_split is not None:
|
||||
use_y = True
|
||||
else:
|
||||
return sorted(lines, key=lambda ln: (-ln.y, ln.x0 if not rtl else -ln.x0))
|
||||
|
||||
ordered: list[Line] = []
|
||||
if use_x and x_split is not None:
|
||||
left = [ln for ln in lines if _mid_x(ln) < x_split]
|
||||
right = [ln for ln in lines if _mid_x(ln) >= x_split]
|
||||
# Vertical cuts: LTR normally; RTL page → right column first
|
||||
first, second = (right, left) if rtl else (left, right)
|
||||
fx0, fx1 = (x_split, x1) if rtl else (x0, x_split)
|
||||
sx0, sx1 = (x0, x_split) if rtl else (x_split, x1)
|
||||
if rtl:
|
||||
# first is right band
|
||||
fx0, fx1 = x_split, x1
|
||||
sx0, sx1 = x0, x_split
|
||||
ordered.extend(
|
||||
_xy_cut(
|
||||
first,
|
||||
x0=fx0,
|
||||
x1=fx1,
|
||||
y0=y0,
|
||||
y1=y1,
|
||||
page_width=page_width,
|
||||
page_height=page_height,
|
||||
rtl=rtl,
|
||||
depth=depth + 1,
|
||||
)
|
||||
)
|
||||
ordered.extend(
|
||||
_xy_cut(
|
||||
second,
|
||||
x0=sx0,
|
||||
x1=sx1,
|
||||
y0=y0,
|
||||
y1=y1,
|
||||
page_width=page_width,
|
||||
page_height=page_height,
|
||||
rtl=rtl,
|
||||
depth=depth + 1,
|
||||
)
|
||||
)
|
||||
return ordered
|
||||
|
||||
if use_y and y_split is not None:
|
||||
# PDF y-up: process top (higher y) first
|
||||
top = [ln for ln in lines if _mid_y(ln) >= y_split]
|
||||
bottom = [ln for ln in lines if _mid_y(ln) < y_split]
|
||||
ordered.extend(
|
||||
_xy_cut(
|
||||
top,
|
||||
x0=x0,
|
||||
x1=x1,
|
||||
y0=y_split,
|
||||
y1=y1,
|
||||
page_width=page_width,
|
||||
page_height=page_height,
|
||||
rtl=rtl,
|
||||
depth=depth + 1,
|
||||
)
|
||||
)
|
||||
ordered.extend(
|
||||
_xy_cut(
|
||||
bottom,
|
||||
x0=x0,
|
||||
x1=x1,
|
||||
y0=y0,
|
||||
y1=y_split,
|
||||
page_width=page_width,
|
||||
page_height=page_height,
|
||||
rtl=rtl,
|
||||
depth=depth + 1,
|
||||
)
|
||||
)
|
||||
return ordered
|
||||
|
||||
return sorted(lines, key=lambda ln: (-ln.y, ln.x0 if not rtl else -ln.x0))
|
||||
|
||||
|
||||
def _column_bands(lines: list[Line], page_width: float) -> list[tuple[float, float]]:
|
||||
"""Legacy helper kept for callers/tests that inspect bands; single best X cut."""
|
||||
if not lines or page_width <= 0:
|
||||
return [(0.0, page_width or 612.0)]
|
||||
ys = [ln.y for ln in lines]
|
||||
y0, y1 = min(ys) - 1, max(ys) + 1
|
||||
split = _best_x_gutter(lines, 0.0, page_width, min_gutter=_min_gutter_pt(page_width, 792.0))
|
||||
if split is None:
|
||||
return [(0.0, page_width)]
|
||||
left = [ln for ln in lines if _mid_x(ln) < split]
|
||||
right = [ln for ln in lines if _mid_x(ln) >= split]
|
||||
if len(left) < MIN_COL_LINES or len(right) < MIN_COL_LINES:
|
||||
return [(0.0, page_width)]
|
||||
return [(0.0, split), (split, page_width)]
|
||||
|
||||
|
||||
def _cut_band(
|
||||
band: list[Line],
|
||||
*,
|
||||
page_width: float,
|
||||
page_height: float,
|
||||
rtl: bool,
|
||||
) -> list[Line]:
|
||||
"""Order one horizontal band, resolving any columns inside it."""
|
||||
if len(band) <= 1:
|
||||
return list(band)
|
||||
ys = [ln.y for ln in band]
|
||||
out = _xy_cut(
|
||||
list(band),
|
||||
x0=0.0,
|
||||
x1=page_width,
|
||||
y0=min(ys) - 1.0,
|
||||
y1=max(ys) + 1.0,
|
||||
page_width=page_width,
|
||||
page_height=page_height,
|
||||
rtl=rtl,
|
||||
)
|
||||
return out or sorted(band, key=lambda ln: -ln.y)
|
||||
|
||||
|
||||
def _order_with_spanning(
|
||||
spanning: list[Line],
|
||||
main: list[Line],
|
||||
*,
|
||||
page_width: float,
|
||||
page_height: float,
|
||||
rtl: bool,
|
||||
) -> list[Line]:
|
||||
"""Interleave full-width lines with the columnar bands between them.
|
||||
|
||||
A wide line is only a *banner* relative to what sits below it. Emitting every
|
||||
wide line before all narrower lines -- which is what concatenating the two
|
||||
groups does -- reorders any single-column page whose line lengths vary, and
|
||||
line lengths always vary: headings, list items, captions and the last line of
|
||||
each paragraph are all short, while body prose runs to the margin. The result
|
||||
is that prose floats to the top of the page and everything else sinks,
|
||||
measured at 4,497 line inversions across this corpus and concentrated in the
|
||||
structured real-world documents (IRS 1040, govinfo CFR, USGS, Mozilla spec).
|
||||
|
||||
Banding on the wide lines keeps the original intent -- a true banner above a
|
||||
multi-column region still precedes those columns -- while a wide line in the
|
||||
middle of running text keeps its place. A page that is entirely one column
|
||||
degenerates to a plain top-to-bottom sort, which is correct for it.
|
||||
"""
|
||||
ordered: list[Line] = []
|
||||
body_desc = sorted(main, key=lambda ln: -ln.y)
|
||||
i = 0
|
||||
for sp in sorted(spanning, key=lambda ln: -ln.y):
|
||||
band: list[Line] = []
|
||||
while i < len(body_desc) and body_desc[i].y > sp.y:
|
||||
band.append(body_desc[i])
|
||||
i += 1
|
||||
if band:
|
||||
ordered.extend(
|
||||
_cut_band(band, page_width=page_width, page_height=page_height, rtl=rtl)
|
||||
)
|
||||
ordered.append(sp)
|
||||
if i < len(body_desc):
|
||||
ordered.extend(
|
||||
_cut_band(
|
||||
body_desc[i:], page_width=page_width, page_height=page_height, rtl=rtl
|
||||
)
|
||||
)
|
||||
return ordered
|
||||
|
||||
|
||||
def order_lines(
|
||||
lines: list[Line],
|
||||
page_width: float,
|
||||
*,
|
||||
page_height: float = 792.0,
|
||||
) -> list[Line]:
|
||||
"""Return lines in reading order via recursive XY-Cut.
|
||||
|
||||
Spanning titles (wide lines) are emitted first in Y order, then body columns
|
||||
LTR (RTL when majority Arabic). Optional narrow sidebars are read after body (LTR).
|
||||
"""
|
||||
if not lines:
|
||||
return []
|
||||
pw = page_width or 612.0
|
||||
ph = page_height or 792.0
|
||||
rtl = _arabic_ratio(lines) >= 0.55
|
||||
|
||||
# 1) Spanning lines (titles / banners) — process before column cuts
|
||||
spanning: list[Line] = []
|
||||
body: list[Line] = []
|
||||
for ln in lines:
|
||||
span = max(0.0, ln.x1 - ln.x0)
|
||||
if span >= pw * 0.55 and len((ln.text or "").strip()) >= 3:
|
||||
spanning.append(ln)
|
||||
else:
|
||||
body.append(ln)
|
||||
spanning.sort(key=lambda ln: -ln.y)
|
||||
|
||||
if not body:
|
||||
return spanning
|
||||
|
||||
# 2) Sidebar: narrow tall band at left or right edge
|
||||
sidebar: list[Line] = []
|
||||
main = list(body)
|
||||
mids = [_mid_x(ln) for ln in body]
|
||||
if len(body) >= 4 and mids:
|
||||
left_edge = [ln for ln in body if _mid_x(ln) < pw * 0.22 and (ln.x1 - ln.x0) < pw * 0.28]
|
||||
right_edge = [ln for ln in body if _mid_x(ln) > pw * 0.78 and (ln.x1 - ln.x0) < pw * 0.28]
|
||||
# Prefer the denser sidebar candidate if it has ≥3 lines and body remains
|
||||
for cand, on_left in ((left_edge, True), (right_edge, False)):
|
||||
if len(cand) >= 3 and len(body) - len(cand) >= 2:
|
||||
ys = [ln.y for ln in cand]
|
||||
if max(ys) - min(ys) <= ph * 0.15:
|
||||
continue
|
||||
rest = [ln for ln in body if id(ln) not in {id(c) for c in cand}]
|
||||
# A margin note occupies vertical space of its own. A table's
|
||||
# label column shares every baseline with the values beside
|
||||
# it, and moving it to the end of the page destroys the table.
|
||||
if _shared_baseline_ratio(cand, rest) >= MAX_SHARED_BASELINES:
|
||||
continue
|
||||
# ...and it sits outside the text column, not inside it. Narrow
|
||||
# is not the same as separate.
|
||||
if not _outside_text_column(
|
||||
cand, rest, on_left=on_left, gutter=_min_gutter_pt(pw, ph)
|
||||
):
|
||||
continue
|
||||
sidebar = cand
|
||||
main = rest
|
||||
break
|
||||
|
||||
ordered_main = _order_with_spanning(
|
||||
spanning,
|
||||
main,
|
||||
page_width=pw,
|
||||
page_height=ph,
|
||||
rtl=rtl,
|
||||
)
|
||||
if not ordered_main:
|
||||
ordered_main = sorted(list(main) + list(spanning), key=lambda ln: -ln.y)
|
||||
|
||||
# Sidebars after body for LTR; before body for RTL (margin notes often outside)
|
||||
sidebar_ordered = sorted(sidebar, key=lambda ln: -ln.y)
|
||||
if rtl:
|
||||
out = sidebar_ordered + ordered_main
|
||||
else:
|
||||
out = ordered_main + sidebar_ordered
|
||||
return out if out else sorted(lines, key=lambda ln: -ln.y)
|
||||
@@ -0,0 +1,196 @@
|
||||
"""Heading detection and font-flag inference.
|
||||
|
||||
Heading detection drives the document outline, the navigation pane and any
|
||||
downstream table of contents, so a false positive is expensive: a page banner
|
||||
promoted to Heading 1 on every page produces an outline that is unusable.
|
||||
|
||||
The ordering of signals below is deliberate. An explicit section number is
|
||||
strong and language-independent evidence and is trusted on its own. Everything
|
||||
else — capitalisation, font size — is weak on its own and is required to
|
||||
corroborate.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from collections import Counter
|
||||
|
||||
_BOLD_RE = re.compile(r"(bold|black|heavy|extrabold|semibold|demi|bd\b|boldmt)", re.I)
|
||||
_ITALIC_RE = re.compile(r"(italic|oblique|it\b|ital)", re.I)
|
||||
_CAPTION_RE = re.compile(r"^(figure|table|source|fig\.|tab\.)\b", re.I)
|
||||
# Sentence-final punctuation, including CJK and Arabic full stops. Defined here
|
||||
# rather than in ``paragraphs`` because both modules need it and the dependency
|
||||
# only runs one way: ``glyphs`` imports this module, and ``paragraphs`` imports
|
||||
# ``glyphs``, so importing ``paragraphs`` from here would close a cycle. One
|
||||
# definition matters -- the bold-heading rule below and the paragraph-break rule
|
||||
# in ``paragraphs`` must agree on what ends a sentence, or a line can be both a
|
||||
# heading and the continuation of the prose above it.
|
||||
TERMINAL_PUNCT_RE = re.compile(r"[.!?;:。!?۔؟]['\"”’)\]]*\s*$") # noqa: RUF001
|
||||
|
||||
# "1 Introduction", "2.1 Invitation", "3.4.2 Scope" — depth sets the level.
|
||||
_SECTION_NUM_RE = re.compile(r"^(\d{1,2}(?:\.\d{1,2}){0,3})[.)]?\s+([A-Za-z\u0600-\u06ff].*)$")
|
||||
# The same heading with the space lost: "4.4.1.3United Arab Emirates". Bilingual
|
||||
# PDFs produce these constantly — the number and the title are separate text
|
||||
# runs, and a run boundary is not a space. At least one dot is required, so a
|
||||
# year or a quantity ("2024Report", "5Items") is left alone: those are far more
|
||||
# often a genuine token than a section number.
|
||||
_GLUED_SECTION_RE = re.compile(r"^(\d{1,2}(?:\.\d{1,2}){1,3})[.)]?(?=[A-Za-z\u0600-\u06ff])")
|
||||
# Dates open with digits and contain letters, so they look like numbered
|
||||
# sections. "22 -JUN -2026" is not a heading.
|
||||
_DATE_LIKE_RE = re.compile(
|
||||
r"^\s*\d{1,4}\s*[-/ ]\s*(?:\d{1,2}|jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)"
|
||||
r"[a-z]*\s*[-/ ]\s*\d{2,4}\s*$",
|
||||
re.I,
|
||||
)
|
||||
# "Chapter 4", "Appendix B", "Part II", "Annex 3"
|
||||
_NAMED_SECTION_RE = re.compile(
|
||||
r"^(chapter|section|appendix|annex|part|schedule)\s+([0-9]+|[ivxlcdm]+|[a-z])\b",
|
||||
re.I,
|
||||
)
|
||||
# A line that is mostly punctuation or digits is not a heading.
|
||||
_MOSTLY_WORDS_RE = re.compile(r"[A-Za-z-ۿ一-鿿]")
|
||||
|
||||
MAX_HEADING_CHARS = 120
|
||||
MAX_HEADING_WORDS = 14
|
||||
# An all-caps line must be at least this much larger than body text to count as
|
||||
# a heading on capitalisation alone. Page banners are set at or below body size.
|
||||
CAPS_SIZE_RATIO = 1.05
|
||||
# A wholly bold line must be at least this much larger than body text to be read
|
||||
# as a heading on weight. Bold set *at* body size is emphasis, not structure.
|
||||
BOLD_HEADING_RATIO = 1.04
|
||||
|
||||
|
||||
def _tier_edges(body_size: float) -> list[float]:
|
||||
"""Font-size tier boundaries relative to body text."""
|
||||
b = max(body_size, 8.0)
|
||||
return [b * 1.15, b * 1.30, b * 1.55, b * 1.90]
|
||||
|
||||
|
||||
def split_glued_section_number(text: str) -> str:
|
||||
"""Restore the space in ``4.4.1.3United Arab Emirates``.
|
||||
|
||||
The section number is never dropped — it is how the reader and the
|
||||
document's own cross-references identify the clause. Only the missing
|
||||
separator is added, and only when what follows the number is real words.
|
||||
"""
|
||||
stripped = (text or "").strip()
|
||||
if not stripped or _DATE_LIKE_RE.match(stripped):
|
||||
return text
|
||||
m = _GLUED_SECTION_RE.match(stripped)
|
||||
if not m:
|
||||
return text
|
||||
number = m.group(0)
|
||||
rest = stripped[len(number):]
|
||||
if not _MOSTLY_WORDS_RE.search(rest):
|
||||
return text
|
||||
return f"{number} {rest}"
|
||||
|
||||
|
||||
def section_number_level(text: str) -> int:
|
||||
"""Heading level implied by an explicit section number, or 0."""
|
||||
stripped = (text or "").strip()
|
||||
if _DATE_LIKE_RE.match(stripped):
|
||||
return 0
|
||||
m = _SECTION_NUM_RE.match(split_glued_section_number(stripped).strip())
|
||||
if m and _MOSTLY_WORDS_RE.search(m.group(2)):
|
||||
depth = m.group(1).count(".") + 1
|
||||
return min(depth, 3)
|
||||
if _NAMED_SECTION_RE.match((text or "").strip()):
|
||||
return 1
|
||||
return 0
|
||||
|
||||
|
||||
def heading_level(
|
||||
text: str, font_size: float, body_size: float, *, bold: bool = False
|
||||
) -> int:
|
||||
"""Heading level 1-3, or 0 for body text."""
|
||||
t = (text or "").strip()
|
||||
if not t or len(t) > MAX_HEADING_CHARS:
|
||||
return 0
|
||||
from app.services.convert.layout.text_quality import is_ocr_noise_line
|
||||
|
||||
if is_ocr_noise_line(t):
|
||||
return 0
|
||||
if _CAPTION_RE.match(t):
|
||||
return 0
|
||||
# Quoted / OCR-bullet prose is not a heading (Huwiyati-class false H1s).
|
||||
if t[:1] in "'\"`" or t.startswith(("'", '"', "`")):
|
||||
return 0
|
||||
if not _MOSTLY_WORDS_RE.search(t):
|
||||
# Pure numbers, dates and rule characters are not headings.
|
||||
return 0
|
||||
|
||||
words = t.split()
|
||||
if len(words) > MAX_HEADING_WORDS:
|
||||
return 0
|
||||
|
||||
# Strongest signal: an explicit section number. Language-independent and
|
||||
# carries its own depth, so it is trusted without a size check.
|
||||
numbered = section_number_level(t)
|
||||
if numbered:
|
||||
return numbered
|
||||
|
||||
ratio = (font_size / body_size) if body_size > 0 else 0.0
|
||||
|
||||
# Capitalisation alone is weak. A running page banner is often set in caps
|
||||
# at body size or smaller — requiring a size increase keeps it out of the
|
||||
# outline while still catching genuine caps headings.
|
||||
if t.isupper() and len(words) <= 12 and ratio >= CAPS_SIZE_RATIO:
|
||||
return 1 if ratio >= 1.30 else 2
|
||||
|
||||
if body_size > 0:
|
||||
edges = _tier_edges(body_size)
|
||||
fs = round(font_size * 2) / 2.0 # snap to half points
|
||||
if fs >= edges[3]:
|
||||
return 1
|
||||
if fs >= edges[2]:
|
||||
return 1
|
||||
if fs >= edges[1]:
|
||||
return 2
|
||||
if fs >= edges[0]:
|
||||
return 3
|
||||
|
||||
# Weight is the other half of the convention, and size alone cannot see it.
|
||||
# Word's default Heading 3 is 12pt bold on an 11pt body -- 1.09x, below the
|
||||
# smallest size tier -- so every H3 in a default-styled document arrived as
|
||||
# ordinary body text. Bold *and* larger than body *and* short *and* not
|
||||
# ending a sentence is a heading; requiring all four keeps bold emphasis
|
||||
# inside prose out, since an emphasised phrase is part of a longer line that
|
||||
# is not wholly bold and usually ends in punctuation. Level 3 is provisional
|
||||
# -- ``doc_wide._rank_heading_levels`` sets the real depth once every heading
|
||||
# size in the document is known.
|
||||
if (
|
||||
bold
|
||||
and body_size > 0
|
||||
and (font_size / body_size) >= BOLD_HEADING_RATIO
|
||||
and len(words) <= 12
|
||||
and not TERMINAL_PUNCT_RE.search(t)
|
||||
):
|
||||
return 3
|
||||
return 0
|
||||
|
||||
|
||||
def is_caption(text: str) -> bool:
|
||||
return bool(text and _CAPTION_RE.match(text.strip()))
|
||||
|
||||
|
||||
def infer_font_flags(font_name: str | None) -> tuple[bool, bool]:
|
||||
"""Infer (bold, italic) from a PDF font name when the engine omits flags."""
|
||||
name = (font_name or "").strip()
|
||||
if not name:
|
||||
return False, False
|
||||
return bool(_BOLD_RE.search(name)), bool(_ITALIC_RE.search(name))
|
||||
|
||||
|
||||
def style_key(font_name: str, font_size: float, bold: bool, italic: bool) -> tuple:
|
||||
return (font_name.lower().strip(), round(font_size, 1), bold, italic)
|
||||
|
||||
|
||||
def dominant_body_size(sizes: list[float]) -> float:
|
||||
"""Most common half-point size — the body size of a page."""
|
||||
if not sizes:
|
||||
return 12.0
|
||||
snapped = [round(s * 2) / 2.0 for s in sizes if s > 0]
|
||||
if not snapped:
|
||||
return 12.0
|
||||
return Counter(snapped).most_common(1)[0][0]
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user