diff --git a/bindings/python/pdfengine_py.cpp b/bindings/python/pdfengine_py.cpp index f88c437..3295385 100644 --- a/bindings/python/pdfengine_py.cpp +++ b/bindings/python/pdfengine_py.cpp @@ -111,12 +111,64 @@ PYBIND11_MODULE(pdfengine, m) { return "FontInfo(font_name='" + self.fontName + "', type='" + self.type + "', is_embedded=" + (self.isEmbedded ? "True" : "False") + ")"; }); + py::class_(m, "Glyph") + .def_readonly("text", &pdfengine::Glyph::text) + .def_readonly("unicode", &pdfengine::Glyph::unicode) + .def_readonly("font_name", &pdfengine::Glyph::fontName) + .def_readonly("flags", &pdfengine::Glyph::flags) + .def_readonly("font_size", &pdfengine::Glyph::fontSize) + .def_readonly("origin_x", &pdfengine::Glyph::originX) + .def_readonly("origin_y", &pdfengine::Glyph::originY) + .def_readonly("bbox_x", &pdfengine::Glyph::bboxX) + .def_readonly("bbox_y", &pdfengine::Glyph::bboxY) + .def_readonly("bbox_w", &pdfengine::Glyph::bboxW) + .def_readonly("bbox_h", &pdfengine::Glyph::bboxH) + .def_readonly("angle", &pdfengine::Glyph::angle); + + py::class_(m, "TextRun") + .def_readonly("text", &pdfengine::TextRun::text) + .def_readonly("font_name", &pdfengine::TextRun::fontName) + .def_readonly("flags", &pdfengine::TextRun::flags) + .def_readonly("font_size", &pdfengine::TextRun::fontSize) + .def_readonly("internal_font_id", &pdfengine::TextRun::internalFontId) + .def_readonly("is_embedded", &pdfengine::TextRun::isEmbedded) + .def_readonly("type", &pdfengine::TextRun::type) + .def_readonly("glyphs", &pdfengine::TextRun::glyphs) + .def_readonly("x", &pdfengine::TextRun::x) + .def_readonly("y", &pdfengine::TextRun::y) + .def_readonly("w", &pdfengine::TextRun::w) + .def_readonly("h", &pdfengine::TextRun::h); + + py::class_(m, "TextLine") + .def_readonly("runs", &pdfengine::TextLine::runs) + .def_readonly("baseline_y", &pdfengine::TextLine::baselineY) + .def_readonly("x", &pdfengine::TextLine::x) + .def_readonly("y", &pdfengine::TextLine::y) + .def_readonly("w", &pdfengine::TextLine::w) + .def_readonly("h", &pdfengine::TextLine::h); + + py::class_(m, "Paragraph") + .def_readonly("lines", &pdfengine::Paragraph::lines) + .def_readonly("x", &pdfengine::Paragraph::x) + .def_readonly("y", &pdfengine::Paragraph::y) + .def_readonly("w", &pdfengine::Paragraph::w) + .def_readonly("h", &pdfengine::Paragraph::h); + + py::class_(m, "PageModel") + .def_readonly("paragraphs", &pdfengine::PageModel::paragraphs) + .def_readonly("width", &pdfengine::PageModel::width) + .def_readonly("height", &pdfengine::PageModel::height) + .def_readonly("page_index", &pdfengine::PageModel::pageIndex); + py::class_>(m, "PdfPage") .def_property_readonly("width", &pdfengine::PdfPage::width) .def_property_readonly("height", &pdfengine::PdfPage::height) .def("render", [](const pdfengine::PdfPage& self, int dpi) { return get_or_throw(self.render(dpi)); }, py::arg("dpi") = 96) + .def("extract_document_model", [](const pdfengine::PdfPage& self) { + return get_or_throw(self.extractDocumentModel()); + }) .def("extract_text", [](const pdfengine::PdfPage& self) { return get_or_throw(self.extractText()); }) @@ -141,6 +193,9 @@ PYBIND11_MODULE(pdfengine, m) { .def("get_fonts", [](const pdfengine::PdfPage& self) { return get_or_throw(self.getFonts()); }) + .def("get_glyph_width", [](const pdfengine::PdfPage& self, const std::string& fontName, uint32_t charcode, double fontSize) { + return get_or_throw(self.getGlyphWidth(fontName, charcode, fontSize)); + }, py::arg("font_name"), py::arg("charcode"), py::arg("font_size")) .def("page_to_device", &pdfengine::PdfPage::pageToDevice, py::arg("page_point"), py::arg("device_width"), py::arg("device_height"), py::arg("rotate") = 0) .def("device_to_page", &pdfengine::PdfPage::deviceToPage, diff --git a/corpus/fonts/cns1_test.pdf b/corpus/fonts/cns1_test.pdf new file mode 100644 index 0000000..9dcad82 --- /dev/null +++ b/corpus/fonts/cns1_test.pdf @@ -0,0 +1,90 @@ +%PDF-1.3 +%“Œ‹ž ReportLab Generated PDF document (opensource) +1 0 obj +<< +/F1 2 0 R /F2 3 0 R +>> +endobj +2 0 obj +<< +/BaseFont /Helvetica /Encoding /WinAnsiEncoding /Name /F1 /Subtype /Type1 /Type /Font +>> +endobj +3 0 obj +<< +/BaseFont /MSung-Light /DescendantFonts [ << +/BaseFont /MSung-Light /CIDSystemInfo << +/Ordering (CNS1) /Registry (Adobe) /Supplement 1 +>> /DW 1000 /FontDescriptor << +/Ascent 752 /CapHeight 737 /Descent -271 /Flags 6 /FontBBox [ -160 -249 1015 888 ] /FontName /MSung-Light + /ItalicAngle 0 /Leading 148 /MaxWidth 1000 /MissingWidth 500 /StemH 45 /StemV 58 + /Type /FontDescriptor /XHeight 553 +>> /Subtype /CIDFontType0 /Type /Font + /W [ 1 2 250 3 [ 408 668 490 875 698 250 240 ] 10 [ 240 417 667 250 313 250 520 500 ] 18 26 500 + 27 28 250 29 31 667 32 [ 396 921 677 615 719 760 625 552 771 802 + 354 ] 43 [ 354 781 604 927 750 823 563 823 729 542 + 698 771 729 948 771 677 635 344 520 344 + 469 500 250 469 521 427 521 438 271 469 + 531 250 ] + 75 [ 250 458 240 802 531 500 521 ] 82 [ 521 365 333 292 521 458 677 479 458 427 + 480 496 480 667 ] ] +>> ] /Encoding /UniGB-UCS2-H /Name /F2 /Subtype /Type0 /Type /Font +>> +endobj +4 0 obj +<< +/Contents 8 0 R /MediaBox [ 0 0 595.2756 841.8898 ] /Parent 7 0 R /Resources << +/Font 1 0 R /ProcSet [ /PDF /Text /ImageB /ImageC /ImageI ] +>> /Rotate 0 /Trans << + +>> + /Type /Page +>> +endobj +5 0 obj +<< +/PageMode /UseNone /Pages 7 0 R /Type /Catalog +>> +endobj +6 0 obj +<< +/Author (anonymous) /CreationDate (D:20260602153802+05'00') /Creator (anonymous) /Keywords () /ModDate (D:20260602153802+05'00') /Producer (ReportLab PDF Library - \(opensource\)) + /Subject (unspecified) /Title (untitled) /Trapped /False +>> +endobj +7 0 obj +<< +/Count 1 /Kids [ 4 0 R ] /Type /Pages +>> +endobj +8 0 obj +<< +/Filter [ /ASCII85Decode /FlateDecode ] /Length 107 +>> +stream +GapQh0E=F,0U\H3T\pNYT^QKk?tc>IP,;W#U1^23ihPEM_?C]6_CBF/28[_U!/s9cYpe/lM_Qn>nC.&g0fCf=endstream +endobj +xref +0 9 +0000000000 65535 f +0000000061 00000 n +0000000102 00000 n +0000000209 00000 n +0000001155 00000 n +0000001358 00000 n +0000001426 00000 n +0000001687 00000 n +0000001746 00000 n +trailer +<< +/ID +[<01c8dab3d2c3e771bf716fccf2b52ce4><01c8dab3d2c3e771bf716fccf2b52ce4>] +% ReportLab generated PDF document -- digest (opensource) + +/Info 6 0 R +/Root 5 0 R +/Size 9 +>> +startxref +1943 +%%EOF diff --git a/engine/CMakeLists.txt b/engine/CMakeLists.txt index 2ab2ea6..59b1e45 100644 --- a/engine/CMakeLists.txt +++ b/engine/CMakeLists.txt @@ -17,6 +17,9 @@ add_library(pdfengine STATIC src/parser/pdfium_loader.cpp src/parser/pdfium_document.cpp src/fonts/face/font_face.cpp + src/fonts/face/free_type_manager.cpp + src/fonts/loader/font_resolver.cpp + src/fonts/pdf_fonts/font_loader.cpp src/fonts/shaping/hb_shaper.cpp src/fonts/cache/glyph_bitmap.cpp src/fonts/cache/glyph_cache.cpp @@ -47,6 +50,7 @@ target_link_libraries(pdfengine PRIVATE freetype harfbuzz::harfbuzz + harfbuzz::harfbuzz-subset PNG::PNG nlohmann_json::nlohmann_json ) diff --git a/engine/include/pdfengine/pdf_document.hpp b/engine/include/pdfengine/pdf_document.hpp index 9601173..0cafffe 100644 --- a/engine/include/pdfengine/pdf_document.hpp +++ b/engine/include/pdfengine/pdf_document.hpp @@ -77,6 +77,52 @@ struct FontInfo { double capHeight = 0.0; // Font descriptor CapHeight metric }; +struct Glyph { + std::string text; + uint32_t unicode = 0; + + std::string fontName; + uint32_t flags = 0; + double fontSize = 0.0; + + double originX = 0.0; + double originY = 0.0; + double bboxX = 0.0, bboxY = 0.0, bboxW = 0.0, bboxH = 0.0; + double angle = 0.0; +}; + +struct TextRun { + std::string text; + std::string fontName; + uint32_t flags = 0; + double fontSize = 0.0; + std::string internalFontId; + bool isEmbedded = false; + std::string type; + std::vector glyphs; + double x = 0.0, y = 0.0, w = 0.0, h = 0.0; +}; + +struct TextLine { + std::vector runs; + std::vector glyphs; + double angle = 0.0; + double baselineY = 0.0; + double x = 0.0, y = 0.0, w = 0.0, h = 0.0; +}; + +struct Paragraph { + std::vector lines; + double x = 0.0, y = 0.0, w = 0.0, h = 0.0; +}; + +struct PageModel { + std::vector paragraphs; + double width = 0.0; + double height = 0.0; + int pageIndex = 0; +}; + class PdfPage { public: virtual ~PdfPage() = default; @@ -89,14 +135,20 @@ public: [[nodiscard]] virtual std::expected extractText() const = 0; [[nodiscard]] virtual std::expected, EngineError> extractTextWithBounds() const = 0; + + [[nodiscard]] virtual std::expected extractDocumentModel() const = 0; [[nodiscard]] virtual std::expected, EngineError> getFonts() const = 0; [[nodiscard]] virtual std::expected, EngineError> extractAnnotationsText() const = 0; + [[nodiscard]] virtual std::expected getGlyphWidth(const std::string& fontName, uint32_t charcode, double fontSize) const = 0; + [[nodiscard]] virtual DevicePoint pageToDevice(const Point2D& pagePoint, int deviceWidth, int deviceHeight, int rotate = 0) const noexcept = 0; [[nodiscard]] virtual Point2D deviceToPage(const DevicePoint& devicePoint, int deviceWidth, int deviceHeight, int rotate = 0) const noexcept = 0; }; +namespace fonts::pdf_fonts { class Font; } + class PdfDocument { public: virtual ~PdfDocument() = default; @@ -116,6 +168,12 @@ public: [[nodiscard]] virtual std::expected, EngineError> getFonts(int startPage = 0, int endPage = -1) const = 0; + [[nodiscard]] virtual std::expected, EngineError> + getFontData(const std::string& internalFontId) const = 0; + + [[nodiscard]] virtual std::expected, std::string> + getResolvedFont(const FontInfo& fontInfo) = 0; + virtual std::expected applyEdits(const std::string& editsJson) = 0; [[nodiscard]] virtual std::expected, EngineError> diff --git a/engine/src/fonts/cache/glyph_cache.cpp b/engine/src/fonts/cache/glyph_cache.cpp index a865940..5856e1f 100644 --- a/engine/src/fonts/cache/glyph_cache.cpp +++ b/engine/src/fonts/cache/glyph_cache.cpp @@ -13,12 +13,12 @@ GlyphCache::GlyphCache(std::size_t capacity) GlyphCache::~GlyphCache() = default; std::optional GlyphCache::get(const FontFace& fontFace, unsigned int glyphIndex, unsigned int fontSize) { - FT_Face face = fontFace.getFace(); - if (!face) { + uint64_t fontId = fontFace.getId(); + if (fontId == 0) { return std::nullopt; } - GlyphCacheKey key{face, glyphIndex, fontSize}; + GlyphCacheKey key{fontId, glyphIndex, fontSize}; std::size_t shard_idx = getShardIndex(key); auto& shard = *shards_[shard_idx]; @@ -37,12 +37,12 @@ std::optional GlyphCache::get(const FontFace& fontFace, unsigned in } void GlyphCache::insert(const FontFace& fontFace, unsigned int glyphIndex, unsigned int fontSize, const GlyphBitmap& bitmap) { - FT_Face face = fontFace.getFace(); - if (!face) { + uint64_t fontId = fontFace.getId(); + if (fontId == 0) { return; } - GlyphCacheKey key{face, glyphIndex, fontSize}; + GlyphCacheKey key{fontId, glyphIndex, fontSize}; std::size_t shard_idx = getShardIndex(key); auto& shard = *shards_[shard_idx]; diff --git a/engine/src/fonts/cache/glyph_cache.hpp b/engine/src/fonts/cache/glyph_cache.hpp index 3f2c258..3c8b619 100644 --- a/engine/src/fonts/cache/glyph_cache.hpp +++ b/engine/src/fonts/cache/glyph_cache.hpp @@ -12,12 +12,12 @@ namespace pdfengine::fonts { struct GlyphCacheKey { - FT_Face face; + uint64_t fontId; unsigned int glyphIndex; unsigned int fontSize; bool operator==(const GlyphCacheKey& other) const { - return face == other.face && + return fontId == other.fontId && glyphIndex == other.glyphIndex && fontSize == other.fontSize; } @@ -25,7 +25,7 @@ struct GlyphCacheKey { struct GlyphCacheKeyHash { std::size_t operator()(const GlyphCacheKey& key) const { - std::size_t h1 = std::hash{}(static_cast(key.face)); + std::size_t h1 = std::hash{}(key.fontId); std::size_t h2 = std::hash{}(key.glyphIndex); std::size_t h3 = std::hash{}(key.fontSize); // Combine hashes using standard boost hash_combine algorithm diff --git a/engine/src/fonts/face/font_face.cpp b/engine/src/fonts/face/font_face.cpp index 296159f..cd81848 100644 --- a/engine/src/fonts/face/font_face.cpp +++ b/engine/src/fonts/face/font_face.cpp @@ -1,16 +1,17 @@ #include "fonts/face/font_face.hpp" +#include "fonts/face/free_type_manager.hpp" #include +#include namespace pdfengine::fonts { -FontFace::FontFace() - : ft_library_(nullptr), - face_(nullptr) { +static std::atomic g_font_id_counter{1}; - if (FT_Init_FreeType(&ft_library_)) { - std::cerr << "Failed to initialize FreeType\n"; - } +FontFace::FontFace() + : font_id_(g_font_id_counter.fetch_add(1, std::memory_order_relaxed)), + face_(nullptr), + mutex_(std::make_unique()) { } FontFace::~FontFace() { @@ -18,17 +19,14 @@ FontFace::~FontFace() { if (face_) { FT_Done_Face(face_); } - - if (ft_library_) { - FT_Done_FreeType(ft_library_); - } } FontFace::FontFace(FontFace&& other) noexcept - : ft_library_(other.ft_library_), + : font_id_(other.font_id_), face_(other.face_), + mutex_(std::move(other.mutex_)), font_data_(std::move(other.font_data_)) { - other.ft_library_ = nullptr; + other.font_id_ = 0; other.face_ = nullptr; } @@ -37,13 +35,11 @@ FontFace& FontFace::operator=(FontFace&& other) noexcept { if (face_) { FT_Done_Face(face_); } - if (ft_library_) { - FT_Done_FreeType(ft_library_); - } - ft_library_ = other.ft_library_; + font_id_ = other.font_id_; face_ = other.face_; + mutex_ = std::move(other.mutex_); font_data_ = std::move(other.font_data_); - other.ft_library_ = nullptr; + other.font_id_ = 0; other.face_ = nullptr; } return *this; @@ -58,7 +54,7 @@ bool FontFace::loadFromFile(const std::string& path) { font_data_.clear(); if (FT_New_Face( - ft_library_, + FreeTypeManager::instance().getLibrary(), path.c_str(), 0, &face_)) { @@ -91,7 +87,7 @@ bool FontFace::loadFromMemory(const std::vector& data) { font_data_ = data; if (FT_New_Memory_Face( - ft_library_, + FreeTypeManager::instance().getLibrary(), font_data_.data(), static_cast(font_data_.size()), 0, @@ -113,11 +109,21 @@ FT_Face FontFace::getFace() const { return face_; } +uint64_t FontFace::getId() const { + return font_id_; +} + +std::mutex& FontFace::getMutex() const { + return *mutex_; +} + std::optional FontFace::renderGlyph(unsigned int glyphIndex, unsigned int fontSize) { if (!face_) { return std::nullopt; } + std::lock_guard lock(*mutex_); + // Set font size in pixels. if (FT_Set_Pixel_Sizes(face_, 0, fontSize)) { return std::nullopt; diff --git a/engine/src/fonts/face/font_face.hpp b/engine/src/fonts/face/font_face.hpp index 025b611..303c7ac 100644 --- a/engine/src/fonts/face/font_face.hpp +++ b/engine/src/fonts/face/font_face.hpp @@ -5,6 +5,8 @@ #include #include #include +#include +#include #include #include FT_FREETYPE_H @@ -26,13 +28,16 @@ public: bool loadFromMemory(const std::vector& data); FT_Face getFace() const; + uint64_t getId() const; + std::mutex& getMutex() const; // Renders a glyph by index and size, returning a GlyphBitmap on success. std::optional renderGlyph(unsigned int glyphIndex, unsigned int fontSize); private: - FT_Library ft_library_; + uint64_t font_id_; FT_Face face_; + std::unique_ptr mutex_; std::vector font_data_; // Keeps the loaded memory buffer alive for FT_Face }; diff --git a/engine/src/fonts/face/free_type_manager.cpp b/engine/src/fonts/face/free_type_manager.cpp new file mode 100644 index 0000000..59a70d4 --- /dev/null +++ b/engine/src/fonts/face/free_type_manager.cpp @@ -0,0 +1,28 @@ +#include "fonts/face/free_type_manager.hpp" +#include + +namespace pdfengine::fonts { + +FreeTypeManager& FreeTypeManager::instance() { + static FreeTypeManager instance; + return instance; +} + +FreeTypeManager::FreeTypeManager() : library_(nullptr) { + if (FT_Init_FreeType(&library_)) { + std::cerr << "Failed to initialize FreeType\n"; + } +} + +FreeTypeManager::~FreeTypeManager() { + if (library_) { + FT_Done_FreeType(library_); + library_ = nullptr; + } +} + +FT_Library FreeTypeManager::getLibrary() const { + return library_; +} + +} // namespace pdfengine::fonts diff --git a/engine/src/fonts/face/free_type_manager.hpp b/engine/src/fonts/face/free_type_manager.hpp new file mode 100644 index 0000000..1b04377 --- /dev/null +++ b/engine/src/fonts/face/free_type_manager.hpp @@ -0,0 +1,27 @@ +#pragma once + +#include +#include FT_FREETYPE_H + +namespace pdfengine::fonts { + +class FreeTypeManager { +public: + static FreeTypeManager& instance(); + + FT_Library getLibrary() const; + + // Delete copy/move constructors and assignment operators for singleton + FreeTypeManager(const FreeTypeManager&) = delete; + FreeTypeManager& operator=(const FreeTypeManager&) = delete; + FreeTypeManager(FreeTypeManager&&) = delete; + FreeTypeManager& operator=(FreeTypeManager&&) = delete; + +private: + FreeTypeManager(); + ~FreeTypeManager(); + + FT_Library library_; +}; + +} // namespace pdfengine::fonts diff --git a/engine/src/fonts/loader/font_resolver.cpp b/engine/src/fonts/loader/font_resolver.cpp new file mode 100644 index 0000000..a81e683 --- /dev/null +++ b/engine/src/fonts/loader/font_resolver.cpp @@ -0,0 +1,76 @@ +#include "fonts/loader/font_resolver.hpp" +#include "fonts/pdf_fonts/font_loader.hpp" +#include + +namespace pdfengine::fonts::loader { + +FontResolver::FontResolver(std::shared_ptr document) + : document_(std::move(document)) {} + +std::expected, std::string> FontResolver::resolveFont(const FontInfo& fontInfo) { + if (!document_) { + return std::unexpected("No document attached to FontResolver"); + } + + if (fontInfo.isEmbedded) { + auto dataRes = document_->getFontData(fontInfo.internalFontId); + if (!dataRes) { + spdlog::error("Failed to extract embedded font data for ID: {}", fontInfo.internalFontId); + return std::unexpected("Failed to extract embedded font data"); + } + + const auto& bytes = dataRes.value(); + + // Build FontDescriptor from FontInfo + auto descriptor = std::make_unique(); + descriptor->setFontName(fontInfo.fontName); + descriptor->setFlags(fontInfo.flags); + descriptor->setAscent(fontInfo.ascent); + descriptor->setDescent(fontInfo.descent); + descriptor->setCapHeight(fontInfo.capHeight); + + // Route to the appropriate FontLoader method based on FontInfo type + if (fontInfo.type == "TrueType") { + auto font = pdf_fonts::FontLoader::loadTrueTypeFromMemory(fontInfo.normalizedFamily, bytes, std::move(descriptor)); + if (font) return font; + } else if (fontInfo.type == "Type1") { + auto font = pdf_fonts::FontLoader::loadType1FromMemory(fontInfo.normalizedFamily, bytes, std::move(descriptor)); + if (font) return font; + } else if (fontInfo.type == "CIDFontType0" || fontInfo.type == "CIDFontType2") { + pdf_fonts::FontType subtype = (fontInfo.type == "CIDFontType0") + ? pdf_fonts::FontType::CIDFontType0 + : pdf_fonts::FontType::CIDFontType2; + + auto font = pdf_fonts::FontLoader::loadCIDFontFromMemory(fontInfo.normalizedFamily, subtype, bytes, std::move(descriptor)); + if (font) return font; + } + + return std::unexpected("Failed to parse extracted font data"); + } else { + // Handle System Fallback + spdlog::info("Resolving system fallback font for: {}", fontInfo.fontName); + + // Build FontDescriptor from FontInfo + auto descriptor = std::make_unique(); + descriptor->setFontName(fontInfo.fontName); + descriptor->setFlags(fontInfo.flags); + descriptor->setAscent(fontInfo.ascent); + descriptor->setDescent(fontInfo.descent); + descriptor->setCapHeight(fontInfo.capHeight); + + if (fontInfo.type == "CIDFontType0" || fontInfo.type == "CIDFontType2") { + pdf_fonts::FontType subtype = (fontInfo.type == "CIDFontType0") + ? pdf_fonts::FontType::CIDFontType0 + : pdf_fonts::FontType::CIDFontType2; + auto font = pdf_fonts::FontLoader::loadCIDFontSystemFallback(fontInfo.normalizedFamily, subtype, std::move(descriptor)); + if (font) return font; + } else { + auto font = pdf_fonts::FontLoader::loadType1SystemFallback(fontInfo.normalizedFamily, std::move(descriptor)); + if (font) return font; + } + + return std::unexpected("Failed to load system fallback font"); + } +} + +} // namespace pdfengine::fonts::loader diff --git a/engine/src/fonts/loader/font_resolver.hpp b/engine/src/fonts/loader/font_resolver.hpp new file mode 100644 index 0000000..30f9e21 --- /dev/null +++ b/engine/src/fonts/loader/font_resolver.hpp @@ -0,0 +1,24 @@ +#pragma once + +#include "fonts/pdf_fonts/font.hpp" +#include +#include +#include +#include + +namespace pdfengine::fonts::loader { + +class FontResolver { +public: + explicit FontResolver(std::shared_ptr document); + + // Resolves a FontInfo object into a fully loaded Font ready for shaping. + // If the font is embedded, it extracts the raw bytes from the PdfDocument. + // If it's a system fallback, it uses FontLoader to load it from the OS. + std::expected, std::string> resolveFont(const FontInfo& fontInfo); + +private: + std::shared_ptr document_; +}; + +} // namespace pdfengine::fonts::loader diff --git a/engine/src/fonts/pdf_fonts/font.hpp b/engine/src/fonts/pdf_fonts/font.hpp index 4c3eb8e..9fdcc2d 100644 --- a/engine/src/fonts/pdf_fonts/font.hpp +++ b/engine/src/fonts/pdf_fonts/font.hpp @@ -19,6 +19,19 @@ enum class FontType { Type3 }; +enum class FontSource { + Embedded, + SystemFallback, + Substituted +}; + +struct FontMetrics { + double ascent = 0.0; + double descent = 0.0; + double lineGap = 0.0; + double capHeight = 0.0; +}; + class Font { public: virtual ~Font() = default; @@ -37,9 +50,62 @@ public: // Gets the font encoding (returns nullptr if none exists) virtual const Encoding* getEncoding() const = 0; - // Gets subsetting details (returns nullptr if font is not subsetted) virtual const FontSubset* getSubsetInfo() const = 0; + // Advanced Layout APIs + virtual FontSource getSourceType() const { return source_type_; } + virtual void setSourceType(FontSource source) { source_type_ = source; } + + virtual bool hasGlyph(uint32_t unicode) const { + if (!getFontFace().getFace()) return false; + return FT_Get_Char_Index(getFontFace().getFace(), unicode) != 0; + } + + virtual double getAdvanceWidth(uint32_t unicode, double fontSize) const { + if (!getFontFace().getFace()) return 0.0; + + std::lock_guard lock(getFontFace().getMutex()); + FT_Face rawFace = getFontFace().getFace(); + FT_Set_Pixel_Sizes(rawFace, 0, static_cast(fontSize)); + + FT_UInt glyphIndex = FT_Get_Char_Index(rawFace, unicode); + if (glyphIndex == 0) return 0.0; + + if (FT_Load_Glyph(rawFace, glyphIndex, FT_LOAD_DEFAULT) == 0) { + return static_cast(rawFace->glyph->advance.x) / 64.0; + } + return 0.0; + } + + virtual FontMetrics getMetrics(double fontSize) const { + if (metrics_cached_ && last_metrics_size_ == fontSize) { + return metrics_cache_; + } + + if (getFontFace().getFace()) { + std::lock_guard lock(getFontFace().getMutex()); + FT_Face rawFace = getFontFace().getFace(); + FT_Set_Pixel_Sizes(rawFace, 0, static_cast(fontSize)); + + // Convert from 26.6 to double + metrics_cache_.ascent = static_cast(rawFace->size->metrics.ascender) / 64.0; + metrics_cache_.descent = static_cast(rawFace->size->metrics.descender) / 64.0; + metrics_cache_.lineGap = static_cast(rawFace->size->metrics.height - rawFace->size->metrics.ascender + rawFace->size->metrics.descender) / 64.0; + + // Heuristic for capHeight: height of 'H' + FT_UInt hIndex = FT_Get_Char_Index(rawFace, 'H'); + if (hIndex > 0 && FT_Load_Glyph(rawFace, hIndex, FT_LOAD_DEFAULT) == 0) { + metrics_cache_.capHeight = static_cast(rawFace->glyph->metrics.horiBearingY) / 64.0; + } else { + metrics_cache_.capHeight = metrics_cache_.ascent * 0.7; // Fallback + } + + metrics_cached_ = true; + last_metrics_size_ = fontSize; + } + return metrics_cache_; + } + // Translates a raw character code to a Unicode codepoint virtual uint32_t decodeToUnicode(uint32_t charCode) const = 0; @@ -119,6 +185,11 @@ protected: uint32_t last_char_ = 0; std::vector widths_; bool has_widths_ = false; + + FontSource source_type_ = FontSource::Embedded; + mutable FontMetrics metrics_cache_; + mutable bool metrics_cached_ = false; + mutable double last_metrics_size_ = 0.0; bool is_vertical_ = false; uint32_t first_vertical_char_ = 0; diff --git a/engine/src/fonts/pdf_fonts/font_fallback.cpp b/engine/src/fonts/pdf_fonts/font_fallback.cpp index 9c4364f..68e9cd9 100644 --- a/engine/src/fonts/pdf_fonts/font_fallback.cpp +++ b/engine/src/fonts/pdf_fonts/font_fallback.cpp @@ -121,6 +121,8 @@ std::string FontFallback::getFallbackFontPath(const std::string& fontName, bool stylePattern += "-italic"; } + std::lock_guard lock(rules_mutex_); + for (const auto& rule : custom_rules_) { if (stylePattern.find(rule.pattern) != std::string::npos || lowerName.find(rule.pattern) != std::string::npos) { for (const auto& path : rule.preferredPaths) { @@ -154,17 +156,37 @@ std::string FontFallback::getFallbackFontPath(const std::string& fontName, bool } } return "C:\\Windows\\Fonts\\arial.ttf"; + #elif defined(__APPLE__) + std::vector lastResort = { + "/Library/Fonts/Arial.ttf", + "/System/Library/Fonts/Helvetica.ttc", + "/System/Library/Fonts/Supplemental/Arial.ttf" + }; + for (const auto& path : lastResort) { + if (std::filesystem::exists(path)) return path; + } + return "/System/Library/Fonts/Helvetica.ttc"; #else - return ""; + std::vector lastResort = { + "/usr/share/fonts/truetype/liberation/LiberationSans-Regular.ttf", + "/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf", + "/usr/share/fonts/truetype/freefont/FreeSans.ttf" + }; + for (const auto& path : lastResort) { + if (std::filesystem::exists(path)) return path; + } + return "/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf"; #endif } void FontFallback::registerFallback(const std::string& pattern, const std::string& systemFontPath) { std::string lowerPattern = toLower(pattern); + std::lock_guard lock(rules_mutex_); custom_rules_.insert(custom_rules_.begin(), {lowerPattern, {systemFontPath}}); } void FontFallback::resetToDefaults() { + std::lock_guard lock(rules_mutex_); custom_rules_.clear(); } diff --git a/engine/src/fonts/pdf_fonts/font_fallback.hpp b/engine/src/fonts/pdf_fonts/font_fallback.hpp index 4162438..1692ea4 100644 --- a/engine/src/fonts/pdf_fonts/font_fallback.hpp +++ b/engine/src/fonts/pdf_fonts/font_fallback.hpp @@ -2,6 +2,7 @@ #include #include +#include namespace pdfengine::fonts::pdf_fonts { @@ -32,6 +33,7 @@ private: }; std::vector default_rules_; std::vector custom_rules_; + mutable std::mutex rules_mutex_; }; } // namespace pdfengine::fonts::pdf_fonts diff --git a/engine/src/fonts/pdf_fonts/font_subset.cpp b/engine/src/fonts/pdf_fonts/font_subset.cpp index 8adfe58..c2ab076 100644 --- a/engine/src/fonts/pdf_fonts/font_subset.cpp +++ b/engine/src/fonts/pdf_fonts/font_subset.cpp @@ -1,9 +1,85 @@ #include "fonts/pdf_fonts/font_subset.hpp" #include #include +#include +#include FT_FREETYPE_H +#include +#include namespace pdfengine::fonts::pdf_fonts { +void FontSubset::populateFromFace(FontSubset& subset, void* ftFace) { + if (!ftFace) return; + FT_Face face = static_cast(ftFace); + + FT_UInt gindex; + FT_ULong charcode = FT_Get_First_Char(face, &gindex); + while (gindex != 0) { + // In embedded subsets, the TrueType cmap typically maps the + // Original GID or CID (charcode) to the new Subset GID (gindex). + subset.addGlyphMapping(gindex, static_cast(charcode)); + charcode = FT_Get_Next_Char(face, charcode, &gindex); + } +} + +std::vector FontSubset::buildSubset(const std::vector& originalStream, const std::vector& glyphIdsToKeep) { + std::vector result; + if (originalStream.empty() || glyphIdsToKeep.empty()) { + return result; + } + + hb_blob_t* blob = hb_blob_create( + reinterpret_cast(originalStream.data()), + static_cast(originalStream.size()), + HB_MEMORY_MODE_READONLY, + nullptr, + nullptr + ); + + hb_face_t* face = hb_face_create(blob, 0); + hb_blob_destroy(blob); + + if (!face) { + return result; + } + + hb_subset_input_t* input = hb_subset_input_create_or_fail(); + if (!input) { + hb_face_destroy(face); + return result; + } + + hb_set_t* glyph_set = hb_subset_input_glyph_set(input); + for (uint32_t gid : glyphIdsToKeep) { + hb_set_add(glyph_set, gid); + } + + // Always keep GID 0 (the .notdef glyph) + hb_set_add(glyph_set, 0); + + // Retain layout tables for complex text shaping + hb_subset_input_set_flags(input, HB_SUBSET_FLAGS_RETAIN_GIDS); + + hb_face_t* subset_face = hb_subset_or_fail(face, input); + hb_subset_input_destroy(input); + hb_face_destroy(face); + + if (subset_face) { + hb_blob_t* result_blob = hb_face_reference_blob(subset_face); + if (result_blob) { + unsigned int length = 0; + const char* data = hb_blob_get_data(result_blob, &length); + if (data && length > 0) { + result.assign(data, data + length); + } + hb_blob_destroy(result_blob); + } + hb_face_destroy(subset_face); + } + + return result; +} + bool FontSubset::hasSubsetPrefix(const std::string& fontName) { if (fontName.length() < 8) { return false; diff --git a/engine/src/fonts/pdf_fonts/font_subset.hpp b/engine/src/fonts/pdf_fonts/font_subset.hpp index b4a9443..7ab65dc 100644 --- a/engine/src/fonts/pdf_fonts/font_subset.hpp +++ b/engine/src/fonts/pdf_fonts/font_subset.hpp @@ -14,6 +14,12 @@ public: static std::string getSubsetPrefix(const std::string& fontName); + // Iterates the FT_Face cmap to populate the mapping dictionary + static void populateFromFace(FontSubset& subset, void* ftFace); + + // Rebuilds the TTF/CID stream keeping only the specified GIDs using HarfBuzz + static std::vector buildSubset(const std::vector& originalStream, const std::vector& glyphIdsToKeep); + explicit FontSubset(const std::string& fontName); ~FontSubset() = default; diff --git a/engine/src/fonts/pdf_fonts/types/cid_font.cpp b/engine/src/fonts/pdf_fonts/types/cid_font.cpp index b27b975..f9faebe 100644 --- a/engine/src/fonts/pdf_fonts/types/cid_font.cpp +++ b/engine/src/fonts/pdf_fonts/types/cid_font.cpp @@ -54,7 +54,13 @@ CIDFont::CIDFont( CIDFont::~CIDFont() = default; bool CIDFont::loadFromStream(const std::vector& streamData) { - return font_face_.loadFromMemory(streamData); + if (!font_face_.loadFromMemory(streamData)) { + return false; + } + if (subset_info_) { + FontSubset::populateFromFace(*subset_info_, font_face_.getFace()); + } + return true; } bool CIDFont::loadFromFile(const std::string& filePath) { @@ -126,7 +132,8 @@ uint32_t CIDFont::decodeToUnicode(uint32_t charCode) const { if (descriptor_) { std::string fontName = descriptor_->getFontName(); std::string lowerName = fontName; - std::transform(lowerName.begin(), lowerName.end(), lowerName.begin(), ::tolower); + std::transform(lowerName.begin(), lowerName.end(), lowerName.begin(), + [](unsigned char c) { return static_cast(std::tolower(c)); }); std::string registry = "None"; if (lowerName.find("simsun") != std::string::npos || @@ -177,16 +184,14 @@ uint32_t CIDFont::decodeToUnicode(uint32_t charCode) const { gid = subset_info_->mapSubsetToOriginal(gid); } - FT_Face face = font_face_.getFace(); - if (face) { - FT_UInt gindex; - FT_ULong charcode = FT_Get_First_Char(face, &gindex); - while (gindex != 0) { - if (gindex == gid) { - return static_cast(charcode); - } - charcode = FT_Get_Next_Char(face, charcode, &gindex); - } + if (!is_gid_to_unicode_map_built_) { + buildGidToUnicodeMap(); + } + + std::lock_guard lock(gid_to_unicode_mutex_); + auto it = gid_to_unicode_map_.find(gid); + if (it != gid_to_unicode_map_.end()) { + return it->second; } return charCode; @@ -203,4 +208,22 @@ std::string CIDFont::decodeStringToUnicode(const std::vector& charCode return result; } +void CIDFont::buildGidToUnicodeMap() const { + std::lock_guard lock(gid_to_unicode_mutex_); + if (is_gid_to_unicode_map_built_) { + return; + } + + FT_Face face = font_face_.getFace(); + if (face) { + FT_UInt gindex; + FT_ULong charcode = FT_Get_First_Char(face, &gindex); + while (gindex != 0) { + gid_to_unicode_map_[gindex] = static_cast(charcode); + charcode = FT_Get_Next_Char(face, charcode, &gindex); + } + } + is_gid_to_unicode_map_built_ = true; +} + } // namespace pdfengine::fonts::pdf_fonts diff --git a/engine/src/fonts/pdf_fonts/types/cid_font.hpp b/engine/src/fonts/pdf_fonts/types/cid_font.hpp index 1135a8a..fe3a93a 100644 --- a/engine/src/fonts/pdf_fonts/types/cid_font.hpp +++ b/engine/src/fonts/pdf_fonts/types/cid_font.hpp @@ -8,6 +8,7 @@ #include #include #include +#include namespace pdfengine::fonts::pdf_fonts { class FontSubset; @@ -60,6 +61,11 @@ private: bool is_identity_map_ = true; std::unordered_map cid_to_gid_map_; std::unique_ptr subset_info_; + mutable bool is_gid_to_unicode_map_built_ = false; + mutable std::mutex gid_to_unicode_mutex_; + mutable std::unordered_map gid_to_unicode_map_; + + void buildGidToUnicodeMap() const; }; } // namespace pdfengine::fonts::pdf_fonts diff --git a/engine/src/fonts/pdf_fonts/types/truetype_font.cpp b/engine/src/fonts/pdf_fonts/types/truetype_font.cpp index a52ab24..9e4cdc3 100644 --- a/engine/src/fonts/pdf_fonts/types/truetype_font.cpp +++ b/engine/src/fonts/pdf_fonts/types/truetype_font.cpp @@ -50,7 +50,13 @@ TrueTypeFont::TrueTypeFont( TrueTypeFont::~TrueTypeFont() = default; bool TrueTypeFont::loadFromStream(const std::vector& streamData) { - return font_face_.loadFromMemory(streamData); + if (!font_face_.loadFromMemory(streamData)) { + return false; + } + if (subset_info_) { + FontSubset::populateFromFace(*subset_info_, font_face_.getFace()); + } + return true; } std::string TrueTypeFont::getBaseFont() const { @@ -94,18 +100,8 @@ uint32_t TrueTypeFont::decodeToUnicode(uint32_t charCode) const { } if (subset_info_) { - uint32_t subsetGid = subset_info_->mapSubsetToOriginal(charCode); - - FT_Face face = font_face_.getFace(); - if (face) { - FT_UInt gindex; - FT_ULong charcode = FT_Get_First_Char(face, &gindex); - while (gindex != 0) { - if (gindex == subsetGid) { - return static_cast(charcode); - } - charcode = FT_Get_Next_Char(face, charcode, &gindex); - } + if (subset_info_->hasGlyphMapping(charCode)) { + return subset_info_->mapSubsetToOriginal(charCode); } } diff --git a/engine/src/fonts/pdf_fonts/types/type1_font.cpp b/engine/src/fonts/pdf_fonts/types/type1_font.cpp index 2d96dba..02c4144 100644 --- a/engine/src/fonts/pdf_fonts/types/type1_font.cpp +++ b/engine/src/fonts/pdf_fonts/types/type1_font.cpp @@ -50,7 +50,13 @@ Type1Font::Type1Font( Type1Font::~Type1Font() = default; bool Type1Font::loadFromStream(const std::vector& streamData) { - return font_face_.loadFromMemory(streamData); + if (!font_face_.loadFromMemory(streamData)) { + return false; + } + if (subset_info_) { + FontSubset::populateFromFace(*subset_info_, font_face_.getFace()); + } + return true; } bool Type1Font::loadFromFile(const std::string& filePath) { @@ -98,18 +104,8 @@ uint32_t Type1Font::decodeToUnicode(uint32_t charCode) const { } if (subset_info_) { - uint32_t subsetGid = subset_info_->mapSubsetToOriginal(charCode); - - FT_Face face = font_face_.getFace(); - if (face) { - FT_UInt gindex; - FT_ULong charcode = FT_Get_First_Char(face, &gindex); - while (gindex != 0) { - if (gindex == subsetGid) { - return static_cast(charcode); - } - charcode = FT_Get_Next_Char(face, charcode, &gindex); - } + if (subset_info_->hasGlyphMapping(charCode)) { + return subset_info_->mapSubsetToOriginal(charCode); } } diff --git a/engine/src/fonts/shaping/hb_shaper.cpp b/engine/src/fonts/shaping/hb_shaper.cpp index 9f5d3ce..d1da0c7 100644 --- a/engine/src/fonts/shaping/hb_shaper.cpp +++ b/engine/src/fonts/shaping/hb_shaper.cpp @@ -2,6 +2,7 @@ #include #include +#include namespace pdfengine::fonts { @@ -21,6 +22,9 @@ std::vector HbShaper::shapeRun( return result; } + // Acquire lock to prevent concurrent mutation of FT_Face's active pixel size + std::lock_guard lock(font.getMutex()); + // Set the pixel size on the FreeType face before shaping. // This ensures HarfBuzz measures everything using the requested font size context. if (FT_Set_Pixel_Sizes(ftFace, 0, fontSize)) { @@ -53,8 +57,23 @@ std::vector HbShaper::shapeRun( hb_buffer_set_direction(hbBuffer, HB_DIRECTION_TTB); } + // Enable professional typography features (Kerning and Ligatures) + hb_feature_t features[2]; + + // Enable kerning + features[0].tag = HB_TAG('k', 'e', 'r', 'n'); + features[0].value = 1; + features[0].start = 0; + features[0].end = static_cast(-1); + + // Enable standard ligatures + features[1].tag = HB_TAG('l', 'i', 'g', 'a'); + features[1].value = 1; + features[1].start = 0; + features[1].end = static_cast(-1); + // Shape the text inside the buffer using the font. - hb_shape(hbFont, hbBuffer, nullptr, 0); + hb_shape(hbFont, hbBuffer, features, 2); // Retrieve the results. unsigned int glyphCount = 0; @@ -72,6 +91,7 @@ std::vector HbShaper::shapeRun( sg.advanceY = static_cast(glyphPositions[i].y_advance) / 64.0; sg.offsetX = static_cast(glyphPositions[i].x_offset) / 64.0; sg.offsetY = static_cast(glyphPositions[i].y_offset) / 64.0; + sg.clusterIndex = glyphInfos[i].cluster; result.push_back(sg); } } diff --git a/engine/src/fonts/shaping/hb_shaper.hpp b/engine/src/fonts/shaping/hb_shaper.hpp index baec2eb..af39f0b 100644 --- a/engine/src/fonts/shaping/hb_shaper.hpp +++ b/engine/src/fonts/shaping/hb_shaper.hpp @@ -13,6 +13,7 @@ struct ShapedGlyph { double advanceY; double offsetX; double offsetY; + uint32_t clusterIndex; }; class HbShaper { diff --git a/engine/src/parser/pdfium_document.cpp b/engine/src/parser/pdfium_document.cpp index f3e14bd..04ee319 100644 --- a/engine/src/parser/pdfium_document.cpp +++ b/engine/src/parser/pdfium_document.cpp @@ -1,5 +1,4 @@ #include "parser/pdfium_document.hpp" - #ifdef PDFENGINE_WITH_PDFIUM #include #include @@ -11,6 +10,9 @@ #include "parser/pdfium_loader.hpp" #endif +#include "fonts/loader/font_resolver.hpp" +#include "fonts/pdf_fonts/font.hpp" + #include #include #include @@ -401,8 +403,13 @@ void deduceFontMetadata(pdfengine::FontInfo& f) { } // 7. Stable Internal Font Identifier + // For subset fonts, fontName already contains the subset prefix (e.g. "ABCDEF+Arial"). + // Use the full fontName directly — it already encodes both the subset tag and + // the base font name, separated by '+'. Concatenating subsetTag + "_" + fontName + // would duplicate the prefix ("ABCDEF_ABCDEF+Arial"). if (f.isSubset && !f.subsetTag.empty()) { - f.internalFontId = f.subsetTag + "_" + f.fontName; + // fontName is "ABCDEF+Arial"; use it as-is for the stable ID. + f.internalFontId = f.fontName; } else { f.internalFontId = f.fontName + "_" + f.type + "_" + std::to_string(f.flags); } @@ -802,7 +809,7 @@ std::expected, EngineError> PdfiumPage::extractTextWith } std::string utf8_char = code_point_to_utf8(cp); - if (utf8_char.empty()) { + if (utf8_char.empty() || cp == '\r' || cp == '\n') { continue; } @@ -827,6 +834,254 @@ std::expected, EngineError> PdfiumPage::extractTextWith #endif } +std::expected PdfiumPage::extractDocumentModel() const { +#ifdef PDFENGINE_WITH_PDFIUM + if (!page_) { + return std::unexpected(EngineError::Unknown); + } + ensureTextPageLoaded(); + if (!textPage_) { + return std::unexpected(EngineError::Unknown); + } + + PageModel model; + model.width = width(); + model.height = height(); + model.pageIndex = pageIndex_; + + int charCount = FPDFText_CountChars(textPage_); + if (charCount <= 0) { + return model; + } + + std::unordered_map fontMap; + if (auto fontsRes = getFonts()) { + for (const auto& f : *fontsRes) { + fontMap[f.fontName] = f; + } + } + + std::vector documentGlyphs; + documentGlyphs.reserve(charCount); + + // Pass 1: Extract all glyphs + for (int i = 0; i < charCount; ++i) { + unsigned int codeUnit = FPDFText_GetUnicode(textPage_, i); + unsigned int cp = codeUnit; + + // Handle surrogate pairs + if (codeUnit >= 0xD800 && codeUnit <= 0xDBFF && i + 1 < charCount) { + unsigned int nextUnit = FPDFText_GetUnicode(textPage_, i + 1); + if (nextUnit >= 0xDC00 && nextUnit <= 0xDFFF) { + cp = 0x10000 + ((codeUnit - 0xD800) << 10) + (nextUnit - 0xDC00); + } + } + + std::string utf8_char = code_point_to_utf8(cp); + if (utf8_char.empty() || cp == '\r' || cp == '\n') { + if (cp > 0xFFFF) ++i; // Skip low surrogate + continue; + } + + Glyph g; + g.text = std::move(utf8_char); + g.unicode = cp; + + // Geometry + double left, right, bottom, top; + FPDFText_GetCharBox(textPage_, i, &left, &right, &bottom, &top); + g.bboxX = (std::min)(left, right); + g.bboxY = (std::min)(bottom, top); + g.bboxW = std::abs(right - left); + g.bboxH = std::abs(top - bottom); + + FPDFText_GetCharOrigin(textPage_, i, &g.originX, &g.originY); + g.angle = FPDFText_GetCharAngle(textPage_, i); + + // Style + g.fontSize = FPDFText_GetFontSize(textPage_, i); + int flags = 0; + unsigned long len = FPDFText_GetFontInfo(textPage_, i, nullptr, 0, &flags); + if (len > 0) { + std::vector buf(len); + if (FPDFText_GetFontInfo(textPage_, i, buf.data(), len, &flags) > 0) { + g.fontName = std::string(buf.data()); + } + } + g.flags = flags; + + documentGlyphs.push_back(g); + + if (cp > 0xFFFF) ++i; // Skip low surrogate + } + + // Phase 5E: Line Builder + std::sort(documentGlyphs.begin(), documentGlyphs.end(), [](const Glyph& a, const Glyph& b) { + if (std::abs(a.originY - b.originY) > 1.0) { + return a.originY > b.originY; // Top to bottom + } + return a.originX < b.originX; // Left to right + }); + + std::vector lines; + if (!documentGlyphs.empty()) { + TextLine currentLine; + currentLine.angle = documentGlyphs[0].angle; + double currentOriginY = documentGlyphs[0].originY; + + for (const auto& g : documentGlyphs) { + if (currentLine.glyphs.empty()) { + currentLine.glyphs.push_back(g); + continue; + } + + if (std::abs(g.angle - currentLine.angle) < 0.1 && + std::abs(g.originY - currentOriginY) < 1.0) { + currentLine.glyphs.push_back(g); + } else { + lines.push_back(std::move(currentLine)); + currentLine = TextLine(); + currentLine.angle = g.angle; + currentOriginY = g.originY; + currentLine.glyphs.push_back(g); + } + } + if (!currentLine.glyphs.empty()) { + lines.push_back(std::move(currentLine)); + } + } + + for (auto& line : lines) { + std::sort(line.glyphs.begin(), line.glyphs.end(), [](const Glyph& a, const Glyph& b) { + return a.originX < b.originX; + }); + + std::vector gaps; + for (size_t i = 1; i < line.glyphs.size(); ++i) { + double gap = line.glyphs[i].bboxX - (line.glyphs[i-1].bboxX + line.glyphs[i-1].bboxW); + if (gap > 0) { + gaps.push_back(gap); + } + } + double medianGap = 0.0; + if (!gaps.empty()) { + std::sort(gaps.begin(), gaps.end()); + medianGap = gaps[gaps.size() / 2]; + } + + // Phase 5F: Run Builder + TextRun currentRun; + if (!line.glyphs.empty()) { + const Glyph* firstG = &line.glyphs[0]; + currentRun.fontName = firstG->fontName; + currentRun.fontSize = firstG->fontSize; + currentRun.flags = firstG->flags; + if (auto it = fontMap.find(currentRun.fontName); it != fontMap.end()) { + currentRun.internalFontId = it->second.internalFontId; + currentRun.isEmbedded = it->second.isEmbedded; + currentRun.type = it->second.type; + } + currentRun.glyphs.push_back(*firstG); + currentRun.text += firstG->text; + + for (size_t i = 1; i < line.glyphs.size(); ++i) { + const auto& prevG = line.glyphs[i-1]; + const auto& currG = line.glyphs[i]; + + double gap = currG.bboxX - (prevG.bboxX + prevG.bboxW); + double spaceThreshold = (std::max)(currG.fontSize * 0.25, medianGap * 2.0); + + bool addSpace = gap > spaceThreshold && prevG.text != " " && currG.text != " "; + bool breakRun = currG.fontName != currentRun.fontName || + std::abs(currG.fontSize - currentRun.fontSize) > 0.1 || + currG.flags != currentRun.flags; + + if (addSpace) { + Glyph spaceGlyph; + spaceGlyph.text = " "; + spaceGlyph.unicode = ' '; + spaceGlyph.fontSize = currG.fontSize; + spaceGlyph.fontName = currG.fontName; + spaceGlyph.flags = currG.flags; + spaceGlyph.originX = prevG.bboxX + prevG.bboxW; + spaceGlyph.originY = currG.originY; + spaceGlyph.angle = currG.angle; + spaceGlyph.bboxX = spaceGlyph.originX; + spaceGlyph.bboxY = currG.bboxY; + spaceGlyph.bboxW = gap; + spaceGlyph.bboxH = currG.bboxH; + + if (breakRun) { + line.runs.push_back(std::move(currentRun)); + currentRun = TextRun(); + currentRun.fontName = currG.fontName; + currentRun.fontSize = currG.fontSize; + currentRun.flags = currG.flags; + if (auto it = fontMap.find(currentRun.fontName); it != fontMap.end()) { + currentRun.internalFontId = it->second.internalFontId; + currentRun.isEmbedded = it->second.isEmbedded; + currentRun.type = it->second.type; + } + } + currentRun.glyphs.push_back(spaceGlyph); + currentRun.text += spaceGlyph.text; + } else if (breakRun) { + line.runs.push_back(std::move(currentRun)); + currentRun = TextRun(); + currentRun.fontName = currG.fontName; + currentRun.fontSize = currG.fontSize; + currentRun.flags = currG.flags; + if (auto it = fontMap.find(currentRun.fontName); it != fontMap.end()) { + currentRun.internalFontId = it->second.internalFontId; + currentRun.isEmbedded = it->second.isEmbedded; + currentRun.type = it->second.type; + } + } + + currentRun.glyphs.push_back(currG); + currentRun.text += currG.text; + } + if (!currentRun.glyphs.empty()) { + line.runs.push_back(std::move(currentRun)); + } + } + } + + // Phase 5G: Paragraph Builder + std::vector paragraphs; + if (!lines.empty()) { + Paragraph currentPara; + currentPara.lines.push_back(std::move(lines[0])); + + for (size_t i = 1; i < lines.size(); ++i) { + auto& prevLine = currentPara.lines.back(); + auto& currLine = lines[i]; + + double prevY = prevLine.glyphs.empty() ? 0 : prevLine.glyphs[0].originY; + double currY = currLine.glyphs.empty() ? 0 : currLine.glyphs[0].originY; + double fontSize = currLine.runs.empty() ? 12.0 : currLine.runs[0].fontSize; + + double vGap = std::abs(prevY - currY); + + if (vGap > fontSize * 1.5) { + paragraphs.push_back(std::move(currentPara)); + currentPara = Paragraph(); + } + currentPara.lines.push_back(std::move(currLine)); + } + if (!currentPara.lines.empty()) { + paragraphs.push_back(std::move(currentPara)); + } + } + + model.paragraphs = std::move(paragraphs); + + return model; +#else + return std::unexpected(EngineError::Unknown); +#endif +} + std::expected, EngineError> PdfiumPage::extractAnnotationsText() const { #ifdef PDFENGINE_WITH_PDFIUM if (!page_) { @@ -884,6 +1139,60 @@ Point2D PdfiumPage::deviceToPage(const DevicePoint& devicePoint, int deviceWidth #endif } +std::expected PdfiumPage::getGlyphWidth(const std::string& fontName, uint32_t charcode, double fontSize) const { +#ifdef PDFENGINE_WITH_PDFIUM + if (!page_) { + return std::unexpected(EngineError::Unknown); + } + + FPDF_FONT font = nullptr; + { + std::lock_guard lock(textMutex_); + + auto cached = fontHandleCache_.find(fontName); + if (cached != fontHandleCache_.end()) { + font = cached->second; + } else { + int objectCount = FPDFPage_CountObjects(page_); + for (int i = 0; i < objectCount; ++i) { + FPDF_PAGEOBJECT obj = FPDFPage_GetObject(page_, i); + if (!obj || FPDFPageObj_GetType(obj) != FPDF_PAGEOBJ_TEXT) continue; + + FPDF_FONT pageFont = FPDFTextObj_GetFont(obj); + if (!pageFont) continue; + + size_t nameLen = FPDFFont_GetBaseFontName(pageFont, nullptr, 0); + if (nameLen > 0) { + std::vector nameBuf(nameLen); + if (FPDFFont_GetBaseFontName(pageFont, nameBuf.data(), nameLen) > 0) { + std::string currentName(nameBuf.data()); + if (currentName == fontName) { + font = pageFont; + fontHandleCache_[fontName] = font; + break; + } + } + } + } + } + } + + if (!font) { + return std::unexpected(EngineError::Unknown); + } + + float width = 0.0f; + if (!FPDFFont_GetGlyphWidth(font, charcode, static_cast(fontSize), &width)) { + return std::unexpected(EngineError::Unknown); + } + + return static_cast(width); +#else + (void)fontName; (void)charcode; (void)fontSize; + return std::unexpected(EngineError::Unknown); +#endif +} + void PdfiumPage::ensureTextPageLoaded() const { #ifdef PDFENGINE_WITH_PDFIUM std::lock_guard lock(textMutex_); @@ -962,11 +1271,26 @@ std::expected, EngineError> PdfiumDocument::getPage(int if (pageIndex < 0 || pageIndex >= pageCount()) { return std::unexpected(EngineError::PageOutOfBounds); } + + { + std::lock_guard lock(pageCacheMutex_); + auto it = pageCache_.find(pageIndex); + if (it != pageCache_.end()) { + return it->second; + } + } + FPDF_PAGE page = FPDF_LoadPage(doc_, pageIndex); if (!page) { return std::unexpected(EngineError::Unknown); } - return std::make_shared(page, pageIndex); + + auto pageObj = std::make_shared(page, pageIndex); + { + std::lock_guard lock(pageCacheMutex_); + pageCache_[pageIndex] = pageObj; + } + return pageObj; #else (void)pageIndex; return std::unexpected(EngineError::Unknown); @@ -1272,7 +1596,7 @@ std::expected PdfiumDocument::applyEdits(const std::string& e return std::unexpected(EngineError::Unknown); } - invalidateFontCache(); + invalidateCaches(); return {}; #else (void)editsJson; @@ -1478,8 +1802,10 @@ std::expected, EngineError> PdfiumPage::getFonts() const { // deduceFontMetadata() (SystemFallback or Substituted). // --- Recalculate stable identifier with corrected data --- + // fontName already includes the subset prefix (e.g. "ABCDEF+Arial"); + // using it directly avoids the duplicate-prefix bug. if (f.isSubset && !f.subsetTag.empty()) { - f.internalFontId = f.subsetTag + "_" + f.fontName; + f.internalFontId = f.fontName; } else { f.internalFontId = f.fontName + "_" + f.type + "_" + std::to_string(f.flags); } @@ -1543,12 +1869,6 @@ std::expected, EngineError> PdfiumDocument::getFonts(int s return std::vector(); } - // Security Safeguard: cap maximum scan range to 1000 pages to prevent memory/CPU exhaustion - int scanCount = endPage - startPage + 1; - if (scanCount > 1000) { - spdlog::warn("Requested scan range ({} pages) exceeds limit. Capping scan to 1000 pages.", scanCount); - endPage = startPage + 999; - } // Return full document-level cache if available and full range is requested if (startPage == 0 && endPage == total - 1 && hasCachedFonts_) { @@ -1557,15 +1877,13 @@ std::expected, EngineError> PdfiumDocument::getFonts(int s std::vector aggregated; for (int i = startPage; i <= endPage; ++i) { - FPDF_PAGE page = FPDF_LoadPage(doc_, i); - if (!page) { + auto pageRes = const_cast(this)->getPage(i); + if (!pageRes) { spdlog::error("Failed to load page index {} for font diagnostics", i); continue; } - // Stack-allocated wrapper ensures FPDF handles are closed properly upon destruction - PdfiumPage tempPage(page, i); - auto pageFontsRes = tempPage.getFonts(); + auto pageFontsRes = pageRes.value()->getFonts(); if (pageFontsRes) { for (const auto& f : *pageFontsRes) { auto it = std::find_if(aggregated.begin(), aggregated.end(), [&](const FontInfo& existing) { @@ -1606,11 +1924,164 @@ std::expected, EngineError> PdfiumDocument::getFonts(int s #endif } -void PdfiumDocument::invalidateFontCache() { - std::lock_guard lock(fontsMutex_); - cachedFonts_.clear(); - hasCachedFonts_ = false; - spdlog::info("Document font cache has been invalidated."); +std::expected, EngineError> PdfiumDocument::getFontData(const std::string& internalFontId) const { +#ifdef PDFENGINE_WITH_PDFIUM + ensure_pdfium_initialized(); + if (!doc_) { + return std::unexpected(EngineError::Unknown); + } + + // Reconstruct the exact expected font name from the internalFontId. + // Non-subset format: {fontName}_{type}_{flags} + std::string expectedFontName = internalFontId; + size_t lastUnderscore = expectedFontName.rfind('_'); + if (lastUnderscore != std::string::npos && lastUnderscore > 0) { + size_t secondLastUnderscore = expectedFontName.rfind('_', lastUnderscore - 1); + if (secondLastUnderscore != std::string::npos) { + std::string typePart = expectedFontName.substr(secondLastUnderscore + 1, lastUnderscore - secondLastUnderscore - 1); + if (typePart == "TrueType" || typePart == "Type1" || typePart == "CIDFontType0" || typePart == "CIDFontType2") { + expectedFontName = expectedFontName.substr(0, secondLastUnderscore); + } + } + } + + int numPages = FPDF_GetPageCount(doc_); + + // Resume scanning from where we left off + int startPage = 0; + { + std::lock_guard lock(fontsMutex_); + + // Return immediately if already cached + if (fontDataCache_.count(expectedFontName)) { + const auto& cachedBuf = fontDataCache_[expectedFontName]; + if (!cachedBuf.empty()) { + return cachedBuf; + } else { + return std::unexpected(EngineError::FileNotFound); + } + } + startPage = fontDataScannedPages_; + } + + for (int i = startPage; i < numPages; ++i) { + FPDF_PAGE page = FPDF_LoadPage(doc_, i); + if (!page) { + std::lock_guard lock(fontsMutex_); + fontDataScannedPages_ = i + 1; + continue; + } + + int objectCount = FPDFPage_CountObjects(page); + for (int j = 0; j < objectCount; ++j) { + FPDF_PAGEOBJECT obj = FPDFPage_GetObject(page, j); + if (!obj || FPDFPageObj_GetType(obj) != FPDF_PAGEOBJ_TEXT) continue; + + FPDF_FONT font = FPDFTextObj_GetFont(obj); + if (!font) continue; + + unsigned long nameLen = FPDFFont_GetBaseFontName(font, nullptr, 0); + if (nameLen > 0) { + std::vector nameBuf(nameLen); + if (FPDFFont_GetBaseFontName(font, nameBuf.data(), nameLen) > 0) { + std::string fontName(nameBuf.data()); + + std::lock_guard lock(fontsMutex_); + + // Always cache every font we encounter to avoid rescanning + if (fontDataCache_.find(fontName) == fontDataCache_.end()) { + size_t buflen = 0; + FPDFFont_GetFontData(font, nullptr, 0, &buflen); + if (buflen > 0) { + std::vector buffer(buflen); + size_t actual_len = 0; + if (FPDFFont_GetFontData(font, buffer.data(), buflen, &actual_len)) { + fontDataCache_[fontName] = buffer; + } else { + fontDataCache_[fontName] = std::vector(); + } + } else { + fontDataCache_[fontName] = std::vector(); + } + } + + // Use exact match instead of prefix match to avoid false positives + if (fontName == expectedFontName) { + const auto& cachedBuf = fontDataCache_[fontName]; + if (!cachedBuf.empty()) { + FPDF_ClosePage(page); + fontDataScannedPages_ = i; // can resume from the same page later + return cachedBuf; + } + } + } + } + } + FPDF_ClosePage(page); + + std::lock_guard lock(fontsMutex_); + fontDataScannedPages_ = i + 1; + } + + // We finished scanning all pages and still didn't find it (or extraction failed) + { + std::lock_guard lock(fontsMutex_); + // Mark as failed so we don't try to rescan for it + if (fontDataCache_.find(expectedFontName) == fontDataCache_.end()) { + fontDataCache_[expectedFontName] = std::vector(); + } + } + return std::unexpected(EngineError::FileNotFound); +#else + (void)internalFontId; + return std::unexpected(EngineError::Unknown); +#endif +} + +std::expected, std::string> PdfiumDocument::getResolvedFont(const FontInfo& fontInfo) { +#ifdef PDFENGINE_WITH_PDFIUM + std::lock_guard lock(resolvedFontsMutex_); + + // Check Cache + if (resolvedFontsCache_.count(fontInfo.internalFontId)) { + return resolvedFontsCache_[fontInfo.internalFontId]; + } + + if (!fontResolver_) { + // We pass a shared_ptr to 'this' + fontResolver_ = std::make_unique<::pdfengine::fonts::loader::FontResolver>(shared_from_this()); + } + + // Resolve font + auto result = fontResolver_->resolveFont(fontInfo); + if (!result) { + return std::unexpected(result.error()); + } + + // Cache and return + std::shared_ptr sharedFont(std::move(result.value())); + resolvedFontsCache_[fontInfo.internalFontId] = sharedFont; + return sharedFont; +#else + return std::unexpected(std::string("EngineError::Unknown")); +#endif +} + +void PdfiumDocument::invalidateCaches() { + { + std::lock_guard lock(fontsMutex_); + cachedFonts_.clear(); + hasCachedFonts_ = false; + } + { + std::lock_guard lock(pageCacheMutex_); + pageCache_.clear(); + } + { + std::lock_guard lock(resolvedFontsMutex_); + resolvedFontsCache_.clear(); + } + spdlog::info("Document caches have been invalidated."); } } diff --git a/engine/src/parser/pdfium_document.hpp b/engine/src/parser/pdfium_document.hpp index 7526a05..61eb5fd 100644 --- a/engine/src/parser/pdfium_document.hpp +++ b/engine/src/parser/pdfium_document.hpp @@ -11,6 +11,8 @@ #include #include #include +#include +namespace pdfengine::fonts::loader { class FontResolver; } namespace pdfengine::parser { @@ -40,9 +42,12 @@ public: std::expected render(int dpi = 96) const override; std::expected extractText() const override; std::expected, EngineError> extractTextWithBounds() const override; + std::expected extractDocumentModel() const override; std::expected, EngineError> getFonts() const override; std::expected, EngineError> extractAnnotationsText() const override; + std::expected getGlyphWidth(const std::string& fontName, uint32_t charcode, double fontSize) const override; + DevicePoint pageToDevice(const Point2D& pagePoint, int deviceWidth, int deviceHeight, int rotate = 0) const noexcept override; Point2D deviceToPage(const DevicePoint& devicePoint, int deviceWidth, int deviceHeight, int rotate = 0) const noexcept override; @@ -51,11 +56,14 @@ private: mutable NativeTextHandle textPage_ = nullptr; int pageIndex_ = 0; mutable std::mutex textMutex_; +#ifdef PDFENGINE_WITH_PDFIUM + mutable std::unordered_map fontHandleCache_; +#endif void ensureTextPageLoaded() const; }; -class PdfiumDocument : public PdfDocument { +class PdfiumDocument : public PdfDocument, public std::enable_shared_from_this { public: explicit PdfiumDocument(NativeDocHandle docHandle); PdfiumDocument(NativeDocHandle docHandle, std::vector memoryBuffer); @@ -71,7 +79,10 @@ public: std::expected, EngineError> getPage(int pageIndex) override; std::expected, EngineError> getFonts(int startPage = 0, int endPage = -1) const override; - void invalidateFontCache(); + std::expected, EngineError> getFontData(const std::string& internalFontId) const override; + std::expected, std::string> getResolvedFont(const FontInfo& fontInfo) override; + + void invalidateCaches(); std::expected applyEdits(const std::string& editsJson) override; std::expected, EngineError> saveIncremental() const override; @@ -83,6 +94,18 @@ private: mutable std::vector cachedFonts_; mutable bool hasCachedFonts_ = false; mutable std::mutex fontsMutex_; + + // Cache to prevent O(N*M) full-document scans for font data extraction + mutable std::unordered_map> fontDataCache_; + mutable int fontDataScannedPages_ = 0; + + mutable std::unordered_map> pageCache_; + mutable std::mutex pageCacheMutex_; + + // Font Engine Bridge + std::unique_ptr fontResolver_; + std::unordered_map> resolvedFontsCache_; + std::mutex resolvedFontsMutex_; }; // Exposed for testing diff --git a/engine/tests/document_test.cpp b/engine/tests/document_test.cpp index bacdd26..3ddb009 100644 --- a/engine/tests/document_test.cpp +++ b/engine/tests/document_test.cpp @@ -824,7 +824,7 @@ TEST(FontDiagnosticsTest, DeepIntrospectionAndFontSizeVerification) { } EXPECT_EQ(f.sourceType, "Embedded"); EXPECT_TRUE(f.isEmbedded); - EXPECT_EQ(f.internalFontId, f.subsetTag + "_" + f.fontName); + EXPECT_EQ(f.internalFontId, f.fontName); } else { EXPECT_TRUE(f.subsetTag.empty()); EXPECT_EQ(f.internalFontId, f.fontName + "_" + f.type + "_" + std::to_string(f.flags)); diff --git a/frontend/package-lock.json b/frontend/package-lock.json index 8bccfe2..d6bbbb3 100644 --- a/frontend/package-lock.json +++ b/frontend/package-lock.json @@ -59,7 +59,6 @@ "integrity": "sha512-CGOfOJqWjg2qW/Mb6zNsDm+u5vFQ8DxXfbM09z69p5Z6+mE1ikP2jUXw+j42Pf1XTYED2Rni5f95npYeuwMDQA==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "@babel/code-frame": "^7.29.0", "@babel/generator": "^7.29.0", @@ -269,10 +268,31 @@ "node": ">=6.9.0" } }, + "node_modules/@emnapi/core": { + "version": "1.11.0", + "resolved": "https://registry.npmjs.org/@emnapi/core/-/core-1.11.0.tgz", + "integrity": "sha512-l9Oo58x0HOP5znGzVhYW9U3e5wVuA4LAZU2AGezTmkhO1CgQRFDhDg4nneHsu/t3WniXg9QrG2nIXL/ZS8ln8Q==", + "license": "MIT", + "optional": true, + "dependencies": { + "@emnapi/wasi-threads": "1.2.2", + "tslib": "^2.4.0" + } + }, + "node_modules/@emnapi/runtime": { + "version": "1.11.0", + "resolved": "https://registry.npmjs.org/@emnapi/runtime/-/runtime-1.11.0.tgz", + "integrity": "sha512-55coeOFKHv1ywEcUXJtWU5f+Jr/W5tZDvZig8DLKSwUN1JpROQ4rk/SNOQiFWmaR/VKF4zuFyW1B8JduOSv6Pg==", + "license": "MIT", + "optional": true, + "dependencies": { + "tslib": "^2.4.0" + } + }, "node_modules/@emnapi/wasi-threads": { - "version": "1.2.1", - "resolved": "https://registry.npmjs.org/@emnapi/wasi-threads/-/wasi-threads-1.2.1.tgz", - "integrity": "sha512-uTII7OYF+/Mes/MrcIOYp5yOtSMLBWSIoLPpcgwipoiKbli6k322tcoFsxoIIxPDqW01SQGAgko4EzZi2BNv2w==", + "version": "1.2.2", + "resolved": "https://registry.npmjs.org/@emnapi/wasi-threads/-/wasi-threads-1.2.2.tgz", + "integrity": "sha512-c95qOXkHdydNKhscBTebqEC1CVAZpyqOfVfBzQ1qgzyl3gfeldUjIggDbIZgDKsHLgnsM+igH7TJ/eAasaVuMA==", "license": "MIT", "optional": true, "dependencies": { @@ -755,6 +775,37 @@ "node": "^20.19.0 || >=22.12.0" } }, + "node_modules/@rolldown/binding-wasm32-wasi/node_modules/@emnapi/core": { + "version": "1.10.0", + "resolved": "https://registry.npmjs.org/@emnapi/core/-/core-1.10.0.tgz", + "integrity": "sha512-yq6OkJ4p82CAfPl0u9mQebQHKPJkY7WrIuk205cTYnYe+k2Z8YBh11FrbRG/H6ihirqcacOgl2BIO8oyMQLeXw==", + "license": "MIT", + "optional": true, + "dependencies": { + "@emnapi/wasi-threads": "1.2.1", + "tslib": "^2.4.0" + } + }, + "node_modules/@rolldown/binding-wasm32-wasi/node_modules/@emnapi/runtime": { + "version": "1.10.0", + "resolved": "https://registry.npmjs.org/@emnapi/runtime/-/runtime-1.10.0.tgz", + "integrity": "sha512-ewvYlk86xUoGI0zQRNq/mC+16R1QeDlKQy21Ki3oSYXNgLb45GV1P6A0M+/s6nyCuNDqe5VpaY84BzXGwVbwFA==", + "license": "MIT", + "optional": true, + "dependencies": { + "tslib": "^2.4.0" + } + }, + "node_modules/@rolldown/binding-wasm32-wasi/node_modules/@emnapi/wasi-threads": { + "version": "1.2.1", + "resolved": "https://registry.npmjs.org/@emnapi/wasi-threads/-/wasi-threads-1.2.1.tgz", + "integrity": "sha512-uTII7OYF+/Mes/MrcIOYp5yOtSMLBWSIoLPpcgwipoiKbli6k322tcoFsxoIIxPDqW01SQGAgko4EzZi2BNv2w==", + "license": "MIT", + "optional": true, + "dependencies": { + "tslib": "^2.4.0" + } + }, "node_modules/@rolldown/binding-win32-arm64-msvc": { "version": "1.0.1", "resolved": "https://registry.npmjs.org/@rolldown/binding-win32-arm64-msvc/-/binding-win32-arm64-msvc-1.0.1.tgz", @@ -1087,7 +1138,6 @@ "integrity": "sha512-GUUEShf+PBCGW2KaXwcIt3Yk+e3pkKwWKb9GSyM9WQVE+ep2jzmHdGsHzu4wgcZy5fN9FBdVzjpBQsYlpfpgLA==", "devOptional": true, "license": "MIT", - "peer": true, "dependencies": { "undici-types": "~7.16.0" } @@ -1098,7 +1148,6 @@ "integrity": "sha512-ilcTH/UniCkMdtexkoCN0bI7pMcJDvmQFPvuPvmEaYA/NSfFTAgdUSLAoVjaRJm7+6PvcM+q1zYOwS4wTYMF9w==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "csstype": "^3.2.2" } @@ -1158,7 +1207,6 @@ "integrity": "sha512-HPwA+hVkfcriajbNvTmZv4VRauibay+cWArYUYq7u7W7PmGShMxbPxLvrwDme55a6d5alG3nrYfhyJ/G28XlLg==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "@typescript-eslint/scope-manager": "8.59.3", "@typescript-eslint/types": "8.59.3", @@ -1389,7 +1437,6 @@ "integrity": "sha512-UVJyE9MttOsBQIDKw1skb9nAwQuR5wuGD3+82K6JgJlm/Y+KI92oNsMNGZCYdDsVtRHSak0pcV5Dno5+4jh9sw==", "dev": true, "license": "MIT", - "peer": true, "bin": { "acorn": "bin/acorn" }, @@ -1480,7 +1527,6 @@ } ], "license": "MIT", - "peer": true, "dependencies": { "baseline-browser-mapping": "^2.10.12", "caniuse-lite": "^1.0.30001782", @@ -1628,7 +1674,6 @@ "integrity": "sha512-loXy6bWOoP3EP6JA7jo6p5jMpBJmHmsNZM5SFRHLdh1MGOPurMnNBj4ZlAbaqUAaQWbCr7jHV4P7gzAyryZWkQ==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "@eslint-community/eslint-utils": "^4.8.0", "@eslint-community/regexpp": "^4.12.2", @@ -2524,7 +2569,6 @@ "resolved": "https://registry.npmjs.org/picomatch/-/picomatch-4.0.4.tgz", "integrity": "sha512-QP88BAKvMam/3NxH6vj2o21R6MjxZUAd6nlwAS/pnGvN9IVLocLHxGYIzFhg6fUQ+5th6P4dv4eW9jX3DSIj7A==", "license": "MIT", - "peer": true, "engines": { "node": ">=12" }, @@ -2585,7 +2629,6 @@ "resolved": "https://registry.npmjs.org/react/-/react-19.2.6.tgz", "integrity": "sha512-sfWGGfavi0xr8Pg0sVsyHMAOziVYKgPLNrS7ig+ivMNb3wbCBw3KxtflsGBAwD3gYQlE/AEZsTLgToRrSCjb0Q==", "license": "MIT", - "peer": true, "engines": { "node": ">=0.10.0" } @@ -2757,7 +2800,6 @@ "integrity": "sha512-y2TvuxSZPDyQakkFRPZHKFm+KKVqIisdg9/CZwm9ftvKXLP8NRWj38/ODjNbr43SsoXqNuAisEf1GdCxqWcdBw==", "dev": true, "license": "Apache-2.0", - "peer": true, "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" @@ -2843,7 +2885,6 @@ "resolved": "https://registry.npmjs.org/vite/-/vite-8.0.13.tgz", "integrity": "sha512-MFtjBYgzmSxmgA4RAfjIyXWpGe1oALnjgUTzzV7QLx/TKxCzjtMH6Fd9/eVK+5Fg1qNoz5VAwsmMs/NofrmJvw==", "license": "MIT", - "peer": true, "dependencies": { "lightningcss": "^1.32.0", "picomatch": "^4.0.4", @@ -2968,7 +3009,6 @@ "integrity": "sha512-ytENFjIJFl2UwYglde2jchW2Hwm4GJFLDiSXWdTrJQBIN9Fcyp7n4DhxJEiWNAJMV1/BqWfW/kkg71UDcHJyTQ==", "dev": true, "license": "MIT", - "peer": true, "funding": { "url": "https://github.com/sponsors/colinhacks" } diff --git a/frontend/src/index.css b/frontend/src/index.css index 6bfe9b0..e49d8e2 100644 --- a/frontend/src/index.css +++ b/frontend/src/index.css @@ -5,20 +5,23 @@ :root { --sans: 'Outfit', system-ui, -apple-system, sans-serif; --mono: 'JetBrains Mono', monospace; - + /* Color Palette - Premium Royal/Midnight */ - --bg-main: #020617; /* Slate 950 */ - --bg-card: #0f172a; /* Slate 900 */ - --bg-sidebar: #090d16; /* Deep Midnight */ + --bg-main: #020617; + /* Slate 950 */ + --bg-card: #0f172a; + /* Slate 900 */ + --bg-sidebar: #090d16; + /* Deep Midnight */ --bg-accent-indigo: #4f46e5; --bg-accent-indigo-hover: #4338ca; --border-main: #1e293b; --border-glow: rgba(99, 102, 241, 0.25); - + --text-main: #f8fafc; --text-muted: #94a3b8; --text-dim: #64748b; - + /* Status Colors */ --color-success: #10b981; --color-success-bg: rgba(16, 185, 129, 0.1); @@ -175,9 +178,18 @@ body { border-radius: 9999px; } -.status-checking .health-dot { background: var(--color-warning); animation: pulse 1.5s infinite; } -.status-healthy .health-dot { background: var(--color-success); } -.status-unhealthy .health-dot { background: var(--color-error); } +.status-checking .health-dot { + background: var(--color-warning); + animation: pulse 1.5s infinite; +} + +.status-healthy .health-dot { + background: var(--color-success); +} + +.status-unhealthy .health-dot { + background: var(--color-error); +} .wasm-badge { display: flex; @@ -422,7 +434,8 @@ body { padding: 18px; } -.doc-list-container, .annotations-list { +.doc-list-container, +.annotations-list { display: flex; flex-direction: column; gap: 12px; @@ -605,7 +618,8 @@ body { flex: 1; height: 100%; overflow: auto; - background: #0b0f19; /* Slightly darker midnight */ + background: #0b0f19; + /* Slightly darker midnight */ display: flex; justify-content: center; align-items: start; @@ -698,11 +712,13 @@ body { } .highlight-box.type-highlight { - background: #fde047; /* Yellow 300 */ + background: #fde047; + /* Yellow 300 */ } .highlight-box.type-comment { - background: #fecdd3; /* Rose 200 */ + background: #fecdd3; + /* Rose 200 */ border-bottom: 2px solid #f43f5e; } @@ -732,7 +748,7 @@ body { font-weight: 700; padding: 4px 10px; border-radius: 6px; - box-shadow: 0 4px 12px rgba(0,0,0,0.25); + box-shadow: 0 4px 12px rgba(0, 0, 0, 0.25); letter-spacing: 0.5px; } @@ -751,11 +767,23 @@ body { /* Animations */ @keyframes spin { - 0% { transform: rotate(0deg); } - 100% { transform: rotate(360deg); } + 0% { + transform: rotate(0deg); + } + + 100% { + transform: rotate(360deg); + } } @keyframes pulse { - 0%, 100% { opacity: 1; } - 50% { opacity: 0.5; } -} + + 0%, + 100% { + opacity: 1; + } + + 50% { + opacity: 0.5; + } +} \ No newline at end of file diff --git a/frontend/src/lib/gatewayService.ts b/frontend/src/lib/gatewayService.ts index aab2eb8..c3806ae 100644 --- a/frontend/src/lib/gatewayService.ts +++ b/frontend/src/lib/gatewayService.ts @@ -118,7 +118,7 @@ export interface PageRotationData { rotation: 0 | 90 | 180 | 270; } -export interface PageDeletionData {} +export interface PageDeletionData { } export interface PageReorderData { destPageIndex: number; @@ -191,7 +191,7 @@ class GatewayService { method: 'POST', body: formData, }); - + if (response.status === 501) { // Simulate upload for Phase 0 scaffolding return new Promise((resolve) => { @@ -210,7 +210,7 @@ class GatewayService { }, 1000); }); } - + if (!response.ok) throw new Error(`Upload failed: ${response.statusText}`); return response.json(); } @@ -252,12 +252,12 @@ class GatewayService { }).toString(); const url = `${this.baseUrl}/render/${params.documentId}?${query}`; - + const response = await fetch(url); if (response.status === 501) { return this.generateMockPage(params.pageIndex); } - + if (!response.ok) throw new Error(`Page render failed: ${response.statusText}`); const blob = await response.blob(); return URL.createObjectURL(blob); @@ -266,6 +266,12 @@ class GatewayService { } } + async getPageModel(documentId: string, pageIndex: number): Promise { + const response = await fetch(`${this.baseUrl}/documents/${documentId}/pages/${pageIndex}/model`); + if (!response.ok) throw new Error(`Failed to get page model: ${response.statusText}`); + return response.json(); + } + async applyEdits(documentId: string, operations: EditOperation[]): Promise<{ success: boolean; newDocumentId: string }> { const response = await fetch(`${this.baseUrl}/edits/${documentId}`, { method: 'POST', @@ -283,15 +289,15 @@ class GatewayService { async searchDocument(documentId: string, query: string): Promise { if (!query) return []; - + const urlParams = new URLSearchParams({ q: query }); const response = await fetch(`${this.baseUrl}/documents/${documentId}/search?${urlParams.toString()}`); - + if (response.status === 501) { // Return mock empty results if backend isn't available return []; } - + if (!response.ok) throw new Error(`Failed to search document: ${response.statusText}`); return response.json(); } diff --git a/frontend/src/viewer/PDFViewer.tsx b/frontend/src/viewer/PDFViewer.tsx index 9a949f6..de12cb2 100644 --- a/frontend/src/viewer/PDFViewer.tsx +++ b/frontend/src/viewer/PDFViewer.tsx @@ -151,9 +151,9 @@ export const PDFViewer = React.forwardRef(({ currentVisiblePageIdx = layout.index; } - const isInside = (pageBottom >= viewportTop - (layout.height * buffer)) && - (pageTop <= viewportBottom + (layout.height * buffer)); - + const isInside = (pageBottom >= viewportTop - (layout.height * buffer)) && + (pageTop <= viewportBottom + (layout.height * buffer)); + if (isInside) { visible.push(layout); } @@ -201,6 +201,35 @@ export const PDFViewer = React.forwardRef(({ }; }, [visiblePages, documentId, zoom, renderedPages]); + useEffect(() => { + const verifyPageModel = async () => { + if (visiblePages.length > 0 && documentId && !renderedPages[visiblePages[0].index + '_verified']) { + const pageIndex = visiblePages[0].index; + try { + const model = await gatewayService.getPageModel(documentId, pageIndex); + console.log(`--- Verification for Page ${pageIndex} ---`); + model.paragraphs?.forEach((p: any) => { + p.lines?.forEach((l: any) => { + l.runs?.forEach((r: any) => { + const isBold = (r.flags & 262144) !== 0 || r.font_name.toLowerCase().includes('bold'); + const isItalic = (r.flags & 64) !== 0 || r.font_name.toLowerCase().includes('italic'); + console.log(`Run Text: "${r.text}", Font Name: ${r.font_name}, Font Size: ${r.font_size}, Bold: ${isBold}, Italic: ${isItalic}, Embedded: ${r.is_embedded}, Type: ${r.type}`); + }); + }); + }); + setRenderedPages((prev) => { + const next = [...prev]; + next[pageIndex + '_verified' as any] = 'true'; + return next; + }); + } catch (e) { + // ignore or log + } + } + }; + verifyPageModel(); + }, [visiblePages, documentId]); + const handleTextSelection = (text: string, bbox: Rect) => { if (activeTool === 'highlight') { const newAnno: Annotation = { diff --git a/gateway/app/routers/documents.py b/gateway/app/routers/documents.py index 5b73bf1..f9def6d 100644 --- a/gateway/app/routers/documents.py +++ b/gateway/app/routers/documents.py @@ -176,7 +176,6 @@ class FontInfoResponse(BaseModel): descent: float capHeight: float - @router.get("/{document_id}/fonts", response_model=list[FontInfoResponse]) def get_document_fonts( document_id: str, start_page: int = 0, end_page: int = -1 @@ -316,3 +315,134 @@ def search_document(document_id: str, q: str) -> list[SearchMatch]: except Exception as e: raise HTTPException(status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, detail=str(e)) + +class GlyphModel(BaseModel): + text: str + unicode: int + font_name: str + flags: int + font_size: float + origin_x: float + origin_y: float + bbox_x: float + bbox_y: float + bbox_w: float + bbox_h: float + angle: float + +class TextRunModel(BaseModel): + text: str + font_name: str + flags: int + font_size: float + internal_font_id: str + is_embedded: bool + type: str + glyphs: list[GlyphModel] + x: float + y: float + w: float + h: float + +class TextLineModel(BaseModel): + runs: list[TextRunModel] + baseline_y: float + x: float + y: float + w: float + h: float + +class ParagraphModel(BaseModel): + lines: list[TextLineModel] + x: float + y: float + w: float + h: float + +class PageModelResponse(BaseModel): + paragraphs: list[ParagraphModel] + width: float + height: float + page_index: int + +@router.get("/{document_id}/pages/{page_index}/model", response_model=PageModelResponse) +def get_page_model(document_id: str, page_index: int) -> PageModelResponse: + if not engine.is_available(): + raise HTTPException( + status_code=status.HTTP_501_NOT_IMPLEMENTED, + detail="Engine bridge (bindings/python) not yet available.", + ) + + doc_info = document_store.get_document(document_id) + if not doc_info: + raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="Document not found") + + try: + doc = doc_info["doc_instance"] + page = doc.get_page(page_index) + model = page.extract_document_model() + + paragraphs = [] + for p in model.paragraphs: + lines = [] + for l in p.lines: + runs = [] + for r in l.runs: + glyphs = [] + for g in r.glyphs: + glyphs.append(GlyphModel( + text=g.text, + unicode=g.unicode, + font_name=g.font_name, + flags=g.flags, + font_size=g.font_size, + origin_x=g.origin_x, + origin_y=g.origin_y, + bbox_x=g.bbox_x, + bbox_y=g.bbox_y, + bbox_w=g.bbox_w, + bbox_h=g.bbox_h, + angle=g.angle + )) + runs.append(TextRunModel( + text=r.text, + font_name=r.font_name, + flags=r.flags, + font_size=r.font_size, + internal_font_id=r.internal_font_id, + is_embedded=r.is_embedded, + type=r.type, + glyphs=glyphs, + x=r.x, + y=r.y, + w=r.w, + h=r.h + )) + lines.append(TextLineModel( + runs=runs, + baseline_y=l.baseline_y, + x=l.x, + y=l.y, + w=l.w, + h=l.h + )) + paragraphs.append(ParagraphModel( + lines=lines, + x=p.x, + y=p.y, + w=p.w, + h=p.h + )) + + return PageModelResponse( + paragraphs=paragraphs, + width=model.width, + height=model.height, + page_index=model.page_index + ) + except ValueError as e: + raise HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail=str(e)) + except IndexError as e: + raise HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail=str(e)) + except Exception as e: + raise HTTPException(status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, detail=str(e)) diff --git a/gateway/app/routers/edits.py b/gateway/app/routers/edits.py index 7588d8c..b3853df 100644 --- a/gateway/app/routers/edits.py +++ b/gateway/app/routers/edits.py @@ -17,7 +17,7 @@ class TextOverlayData(BaseModel): y: float width: float height: float - fontSize: float + fontSize: float = Field(..., gt=0) fontFamily: str color: str @@ -63,7 +63,7 @@ class FreeTextData(BaseModel): width: float height: float text: str - fontSize: float = 12.0 + fontSize: float = Field(12.0, gt=0) color: str = "#000000" @@ -92,56 +92,56 @@ class PageRotationData(BaseModel): class TextOverlayOperation(BaseModel): id: str type: Literal["text_overlay"] - pageIndex: int + pageIndex: int = Field(..., ge=0) data: TextOverlayData class RedactionOperation(BaseModel): id: str type: Literal["redaction"] - pageIndex: int + pageIndex: int = Field(..., ge=0) data: RedactionData class ImageOverlayOperation(BaseModel): id: str type: Literal["image_overlay"] - pageIndex: int + pageIndex: int = Field(..., ge=0) data: ImageOverlayData class HighlightOperation(BaseModel): id: str type: Literal["highlight"] - pageIndex: int + pageIndex: int = Field(..., ge=0) data: HighlightData class FreeTextOperation(BaseModel): id: str type: Literal["free_text"] - pageIndex: int + pageIndex: int = Field(..., ge=0) data: FreeTextData class CommentOperation(BaseModel): id: str type: Literal["comment"] - pageIndex: int + pageIndex: int = Field(..., ge=0) data: StickyNoteData class FreehandOperation(BaseModel): id: str type: Literal["freehand"] - pageIndex: int + pageIndex: int = Field(..., ge=0) data: FreehandData class PageRotationOperation(BaseModel): id: str type: Literal["page_rotation"] - pageIndex: int + pageIndex: int = Field(..., ge=0) data: PageRotationData diff --git a/gateway/app/routers/render.py b/gateway/app/routers/render.py index 061d137..9cd677b 100644 --- a/gateway/app/routers/render.py +++ b/gateway/app/routers/render.py @@ -1,3 +1,5 @@ +from typing import List, Annotated +from fastapi import APIRouter, HTTPException, status, Response, Path, Query from fastapi import APIRouter, HTTPException, Response, status from app.routers.documents import FontInfoResponse @@ -9,7 +11,7 @@ compat_router = APIRouter(tags=["render"]) @router.get("/{page_index}/render") -def render_page(document_id: str, page_index: int, dpi: int = 96) -> Response: +def render_page(document_id: str, page_index: Annotated[int, Path(ge=0)], dpi: int = 96) -> Response: if not engine.is_available(): raise HTTPException( status_code=status.HTTP_501_NOT_IMPLEMENTED, @@ -34,7 +36,7 @@ def render_page(document_id: str, page_index: int, dpi: int = 96) -> Response: @router.get("/{page_index}/text") -def extract_page_text(document_id: str, page_index: int): +def extract_page_text(document_id: str, page_index: Annotated[int, Path(ge=0)]): if not engine.is_available(): raise HTTPException( status_code=status.HTTP_501_NOT_IMPLEMENTED, @@ -71,7 +73,7 @@ def render_page_compat( @router.get("/{page_index}") -def get_page_info(document_id: str, page_index: int): +def get_page_info(document_id: str, page_index: Annotated[int, Path(ge=0)]): if not engine.is_available(): raise HTTPException( status_code=status.HTTP_501_NOT_IMPLEMENTED, @@ -200,3 +202,27 @@ def get_page_fonts(document_id: str, page_index: int) -> list[FontInfoResponse]: raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="Page index out of range") except Exception as e: raise HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail=str(e)) + + +@router.get("/{page_index}/fonts/glyph-width") +def get_page_glyph_width(document_id: str, page_index: Annotated[int, Path(ge=0)], font_name: str, charcode: int, font_size: Annotated[float, Query(gt=0)] = 12.0): + if not engine.is_available(): + raise HTTPException( + status_code=status.HTTP_501_NOT_IMPLEMENTED, + detail="Engine bridge (bindings/python) not yet available." + ) + + doc_info = document_store.get_document(document_id) + if not doc_info: + raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="Document not found") + + try: + doc = doc_info["doc_instance"] + page = doc.get_page(page_index) + width = page.get_glyph_width(font_name, charcode, font_size) + return {"width": width} + except IndexError: + raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="Page index out of range") + except Exception as e: + raise HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail=str(e)) + diff --git a/gateway/tests/font_api_validation.py b/gateway/tests/font_api_validation.py new file mode 100644 index 0000000..d8dd085 --- /dev/null +++ b/gateway/tests/font_api_validation.py @@ -0,0 +1,388 @@ +""" +Font Extraction API Validation Script +====================================== +Tests font extraction APIs against real PDFs from corpus/fonts/. +Validates: upload, document fonts, page fonts, text extraction, metadata accuracy. +""" + +import json +import sys +import os +import time +import httpx + +BASE_URL = "http://localhost:8000" +CORPUS_DIR = os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..", "corpus", "fonts")) + +# PDFs to test (in priority order) +TARGET_PDFS = [ + "utf-8.pdf", + "vertical_text.pdf", + "subset_font.pdf", +] + +REQUIRED_FONT_FIELDS = [ + "fontName", "type", "isEmbedded", "isSubset", "isVertical", + "encoding", "cmapName", "cidSystemInfo", "subsetTag", + "sourceType", "substitutedFrom", "substitutedTo", + "normalizedFamily", "internalFontId", "flags", + "ascent", "descent", "capHeight", "hasToUnicode", +] + + +def separator(title: str): + print(f"\n{'='*80}") + print(f" {title}") + print(f"{'='*80}\n") + + +def sub_separator(title: str): + print(f"\n--- {title} ---\n") + + +def validate_font_fields(font: dict, pdf_name: str, font_index: int) -> list: + """Validate that all required fields are present and non-null.""" + issues = [] + for field in REQUIRED_FONT_FIELDS: + if field not in font: + issues.append(f" [MISSING] Font #{font_index} ({font.get('fontName', '?')}): field '{field}' is missing") + return issues + + +def check_duplicates(fonts: list) -> list: + """Check for duplicate font entries.""" + seen = set() + dupes = [] + for f in fonts: + key = f.get("fontName", "") + "|" + f.get("type", "") + "|" + f.get("internalFontId", "") + if key in seen: + dupes.append(f.get("fontName", "?")) + seen.add(key) + return dupes + + +def check_empty_names(fonts: list) -> list: + """Check for empty font names.""" + return [i for i, f in enumerate(fonts) if not f.get("fontName", "").strip()] + + +def validate_pdf(client: httpx.Client, pdf_filename: str) -> dict: + """Run all validation steps for a single PDF.""" + result = { + "filename": pdf_filename, + "doc_id": None, + "upload_status": None, + "upload_response": None, + "doc_fonts": None, + "doc_font_count": 0, + "page_fonts": None, + "page_font_count": 0, + "text_result": None, + "glyph_count": 0, + "font_sizes": set(), + "issues": [], + "errors": [], + } + + pdf_path = os.path.join(CORPUS_DIR, pdf_filename) + if not os.path.exists(pdf_path): + result["errors"].append(f"PDF file not found: {pdf_path}") + return result + + # ========== STEP 1: Upload ========== + sub_separator(f"Step 1: Upload {pdf_filename}") + try: + with open(pdf_path, "rb") as f: + resp = client.post( + "/documents", + files={"file": (os.path.basename(pdf_path), f, "application/pdf")}, + ) + result["upload_status"] = resp.status_code + result["upload_response"] = resp.json() + print(f" Status: {resp.status_code}") + print(f" Response: {json.dumps(resp.json(), indent=2)}") + + if resp.status_code != 201: + result["errors"].append(f"Upload failed with status {resp.status_code}: {resp.text}") + return result + + result["doc_id"] = resp.json()["id"] + print(f" Document ID: {result['doc_id']}") + except Exception as e: + result["errors"].append(f"Upload exception: {e}") + return result + + doc_id = result["doc_id"] + + # ========== STEP 2: Document Font Extraction ========== + sub_separator("Step 2: Document Font Extraction") + try: + resp = client.get(f"/documents/{doc_id}/fonts") + print(f" Status: {resp.status_code}") + + if resp.status_code != 200: + result["errors"].append(f"Document fonts failed: {resp.status_code} - {resp.text}") + else: + fonts = resp.json() + result["doc_fonts"] = fonts + result["doc_font_count"] = len(fonts) + print(f" Font count: {len(fonts)}") + + # Print each font in full + for i, f in enumerate(fonts): + print(f"\n Font #{i}:") + print(f" {json.dumps(f, indent=4)}") + + # Validate font count > 0 + if len(fonts) == 0: + result["issues"].append("Document fonts: count is 0") + + # Validate no empty names + empty = check_empty_names(fonts) + if empty: + result["issues"].append(f"Document fonts: empty font names at indices {empty}") + + # Validate no duplicates + dupes = check_duplicates(fonts) + if dupes: + result["issues"].append(f"Document fonts: duplicate fonts: {dupes}") + + # Validate all fields present + for i, f in enumerate(fonts): + field_issues = validate_font_fields(f, pdf_filename, i) + result["issues"].extend(field_issues) + + except Exception as e: + result["errors"].append(f"Document fonts exception: {e}") + + # ========== STEP 3: Page Font Extraction ========== + sub_separator("Step 3: Page Font Extraction (page 0)") + try: + resp = client.get(f"/documents/{doc_id}/pages/0/fonts") + print(f" Status: {resp.status_code}") + + if resp.status_code != 200: + result["errors"].append(f"Page fonts failed: {resp.status_code} - {resp.text}") + else: + page_fonts = resp.json() + result["page_fonts"] = page_fonts + result["page_font_count"] = len(page_fonts) + print(f" Page font count: {len(page_fonts)}") + + for i, f in enumerate(page_fonts): + print(f"\n Page Font #{i}:") + print(f" {json.dumps(f, indent=4)}") + + if len(page_fonts) == 0: + result["issues"].append("Page fonts: count is 0") + + # Compare with document fonts + if result["doc_fonts"] is not None: + doc_font_names = {f["fontName"] for f in result["doc_fonts"]} + page_font_names = {f["fontName"] for f in page_fonts} + + print(f"\n Document font names: {sorted(doc_font_names)}") + print(f" Page font names: {sorted(page_font_names)}") + + # Page fonts should be a subset of document fonts + extra_in_page = page_font_names - doc_font_names + if extra_in_page: + result["issues"].append(f"Page fonts not in document fonts: {extra_in_page}") + print(f" [ISSUE] Page has fonts not in document-level: {extra_in_page}") + else: + print(f" [OK] Page fonts are a subset of document fonts") + + except Exception as e: + result["errors"].append(f"Page fonts exception: {e}") + + # ========== STEP 4: Text Extraction ========== + sub_separator("Step 4: Text Extraction (page 0)") + try: + resp = client.get(f"/documents/{doc_id}/pages/0/text") + print(f" Status: {resp.status_code}") + + if resp.status_code != 200: + result["errors"].append(f"Text extraction failed: {resp.status_code} - {resp.text}") + else: + text_data = resp.json() + result["text_result"] = text_data + text_content = text_data.get("text", "") + glyphs = text_data.get("glyphs", []) + result["glyph_count"] = len(glyphs) + + print(f" Extracted text: {repr(text_content[:300])}") + print(f" Glyph count: {len(glyphs)}") + + if not text_content.strip(): + result["issues"].append("Text extraction: empty text") + + if len(glyphs) == 0: + result["issues"].append("Text extraction: no glyphs returned") + else: + # Show first 5 glyphs as samples + print(f"\n First 5 glyphs (sample):") + for i, g in enumerate(glyphs[:5]): + print(f" Glyph #{i}: {json.dumps(g, indent=6)}") + + # Check glyph structure + for i, g in enumerate(glyphs): + font_size = g.get("fontSize", 0) + if font_size > 0: + result["font_sizes"].add(font_size) + elif font_size == 0 and i < 5: + result["issues"].append(f"Glyph #{i}: fontSize is 0") + + # Validate coordinate fields exist + for coord in ["x", "y", "right", "bottom"]: + if coord not in g and i < 3: + result["issues"].append(f"Glyph #{i}: missing coordinate '{coord}'") + + print(f"\n Font sizes detected: {sorted(result['font_sizes'])}") + + except Exception as e: + result["errors"].append(f"Text extraction exception: {e}") + + # ========== STEP 5: Content-Specific Validation ========== + sub_separator("Step 5: Content-Specific Validation") + + if "vertical" in pdf_filename.lower() and result["doc_fonts"]: + vertical_fonts = [f for f in result["doc_fonts"] if f.get("isVertical")] + print(f" Vertical text PDF - fonts with isVertical=true: {len(vertical_fonts)}") + if len(vertical_fonts) == 0: + result["issues"].append("vertical_text.pdf: No fonts have isVertical=true") + print(f" [ISSUE] No vertical fonts detected!") + else: + for vf in vertical_fonts: + print(f" - {vf['fontName']} (isVertical=true)") + print(f" [OK] Vertical fonts detected correctly") + + if "subset" in pdf_filename.lower() and result["doc_fonts"]: + subset_fonts = [f for f in result["doc_fonts"] if f.get("isSubset")] + print(f" Subset font PDF - fonts with isSubset=true: {len(subset_fonts)}") + if len(subset_fonts) == 0: + result["issues"].append("subset_font.pdf: No fonts have isSubset=true") + print(f" [ISSUE] No subset fonts detected!") + else: + for sf in subset_fonts: + print(f" - {sf['fontName']} (isSubset=true, subsetTag='{sf.get('subsetTag', '')}')") + print(f" [OK] Subset fonts detected correctly") + + if "utf" in pdf_filename.lower() and result["text_result"]: + text = result["text_result"].get("text", "") + print(f" UTF-8 PDF - extracted text: {repr(text[:300])}") + # Check for non-ASCII characters + non_ascii = [c for c in text if ord(c) > 127] + if non_ascii: + print(f" Non-ASCII characters found: {len(non_ascii)} chars") + print(f" Sample non-ASCII: {repr(''.join(non_ascii[:30]))}") + print(f" [OK] UTF-8 text extracts with non-ASCII content") + else: + print(f" [INFO] No non-ASCII characters detected - content may be ASCII-only") + + return result + + +def main(): + print("=" * 80) + print(" FONT EXTRACTION API VALIDATION") + print(f" Server: {BASE_URL}") + print(f" Corpus: {CORPUS_DIR}") + print(f" Timestamp: {time.strftime('%Y-%m-%d %H:%M:%S')}") + print("=" * 80) + + client = httpx.Client(base_url=BASE_URL, timeout=30.0) + + # Verify server is up + try: + r = client.get("/") + print(f"\n Server status: OK ({r.status_code})") + except Exception as e: + print(f"\n [FATAL] Cannot connect to server: {e}") + sys.exit(1) + + # Verify corpus directory + if not os.path.isdir(CORPUS_DIR): + print(f"\n [FATAL] Corpus directory not found: {CORPUS_DIR}") + sys.exit(1) + + available = [f for f in TARGET_PDFS if os.path.exists(os.path.join(CORPUS_DIR, f))] + print(f" PDFs to validate: {available}") + + results = [] + for pdf in available: + separator(f"VALIDATING: {pdf}") + result = validate_pdf(client, pdf) + results.append(result) + + # ========== FINAL REPORT ========== + separator("FINAL REPORT") + + for r in results: + print(f"\n{'~'*60}") + print(f" PDF: {r['filename']}") + print(f" Document ID: {r['doc_id']}") + print(f" Upload Status: {r['upload_status']}") + print(f" Document Font Count: {r['doc_font_count']}") + print(f" Page Font Count: {r['page_font_count']}") + print(f" Glyph Count: {r['glyph_count']}") + print(f" Font Sizes: {sorted(r['font_sizes']) if r['font_sizes'] else 'N/A'}") + print(f" Issues: {len(r['issues'])}") + for issue in r['issues']: + print(f" >> {issue}") + print(f" Errors: {len(r['errors'])}") + for err in r['errors']: + print(f" XX {err}") + + # Summary Answers + separator("SUMMARY ANSWERS") + + total_issues = sum(len(r["issues"]) for r in results) + total_errors = sum(len(r["errors"]) for r in results) + + all_fonts_extracted = all(r["doc_font_count"] > 0 for r in results if not r["errors"]) + print(f" 1. Are fonts being extracted correctly?") + print(f" {'YES' if all_fonts_extracted else 'NO'} - {sum(r['doc_font_count'] for r in results)} total fonts across {len(results)} PDFs") + + page_doc_consistent = all( + not any("not in document" in i for i in r["issues"]) + for r in results + ) + print(f"\n 2. Are page fonts and document fonts consistent?") + print(f" {'YES' if page_doc_consistent else 'NO'}") + + font_sizes_ok = all(r["glyph_count"] > 0 for r in results if not r["errors"]) + print(f"\n 3. Are font sizes being extracted correctly?") + print(f" {'YES' if font_sizes_ok else 'NO'}") + + # Check vertical/subset + vertical_ok = True + subset_ok = True + for r in results: + if "vertical" in r["filename"] and any("isVertical" in i for i in r["issues"]): + vertical_ok = False + if "subset" in r["filename"] and any("isSubset" in i for i in r["issues"]): + subset_ok = False + + print(f"\n 4. Are vertical/subset fonts detected correctly?") + print(f" Vertical: {'YES' if vertical_ok else 'NO'}") + print(f" Subset: {'YES' if subset_ok else 'NO'}") + + print(f"\n 5. Are there any metadata inaccuracies?") + if total_issues == 0 and total_errors == 0: + print(f" NO - All {len(results)} PDFs passed validation cleanly") + else: + print(f" YES - {total_issues} issues and {total_errors} errors found") + for r in results: + for issue in r["issues"]: + print(f" - [{r['filename']}] {issue}") + + print(f"\n{'='*80}") + print(f" VALIDATION COMPLETE: {total_issues} issues, {total_errors} errors") + print(f"{'='*80}") + + client.close() + return 0 if total_errors == 0 else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/gateway/tests/test_font_regression.py b/gateway/tests/test_font_regression.py index b73ed10..cf05703 100644 --- a/gateway/tests/test_font_regression.py +++ b/gateway/tests/test_font_regression.py @@ -1,4 +1,11 @@ +<<<<<<< HEAD +import re +import pytest +from fastapi.testclient import TestClient +import sys +======= import contextlib +>>>>>>> 2638dfe969874b954502b26656df48ad971e306d import os import sys @@ -14,9 +21,15 @@ with contextlib.suppress(ImportError): client = TestClient(app) +<<<<<<< HEAD +# Regex: six uppercase ASCII letters followed by '+' +SUBSET_PREFIX_RE = re.compile(r'^[A-Z]{6}\+') + +======= +>>>>>>> 2638dfe969874b954502b26656df48ad971e306d def get_doc_id(filename: str) -> str: - # First ensure the document is loaded + """Upload a PDF from corpus/fonts and return its document ID.""" filepath = os.path.abspath(f"../corpus/fonts/{filename}") with open(filepath, "rb") as f: resp = client.post("/documents", files={"file": (filename, f, "application/pdf")}) @@ -24,8 +37,15 @@ def get_doc_id(filename: str) -> str: return resp.json()["id"] +<<<<<<< HEAD +# ========================================================================= +# 1. Vertical font regression +# ========================================================================= + +======= +>>>>>>> 2638dfe969874b954502b26656df48ad971e306d def test_is_vertical_regression(): - # 1. Verify Identity-V fonts are detected correctly + """Identity-V fonts must be flagged isVertical; horizontal fonts must not.""" doc_id = get_doc_id("vertical_text.pdf") resp = client.get(f"/documents/{doc_id}/fonts") assert resp.status_code == 200 @@ -35,7 +55,10 @@ def test_is_vertical_regression(): assert font["isVertical"] is True assert font["encoding"] == "Identity-V" +<<<<<<< HEAD +======= # 2. Verify horizontal fonts are not falsely detected as vertical +>>>>>>> 2638dfe969874b954502b26656df48ad971e306d doc_id_h = get_doc_id("utf-8.pdf") resp_h = client.get(f"/documents/{doc_id_h}/fonts") fonts_h = resp_h.json() @@ -44,23 +67,134 @@ def test_is_vertical_regression(): assert f["isVertical"] is False +<<<<<<< HEAD +# ========================================================================= +# 2. Internal Font ID: no duplicate subset prefix (core regression) +# ========================================================================= + +def test_internal_font_id_no_duplicate_prefix(): + """ + Regression: subset fonts must NOT produce "ABCDEF_ABCDEF+Arial". + The internalFontId for a subset font should be the fontName itself + (e.g. "ABCDEF+Arial"), which already encodes the subset tag. + """ + doc_id = get_doc_id("subset_font.pdf") + resp = client.get(f"/documents/{doc_id}/fonts") + assert resp.status_code == 200 + fonts = resp.json() + assert len(fonts) > 0 + + for font in fonts: + fid = font["internalFontId"] + tag = font.get("subsetTag", "") + + if font.get("isSubset") and tag: + # Must NOT start with "TAG_TAG" + assert not fid.startswith(tag + "_" + tag), ( + f"Duplicate subset prefix detected: internalFontId='{fid}'" + ) + # Must equal fontName directly (e.g. "ABCDEF+Arial") + assert fid == font["fontName"], ( + f"Expected internalFontId==fontName for subset font, " + f"got '{fid}' vs '{font['fontName']}'" + ) + + +def test_internal_font_id_subset_format(): + """ + For any subset font the internalFontId must match the pattern + ABCDEF+BaseName — exactly the fontName reported by PDFium. + """ +======= def test_internal_font_id_regression(): # Verify subset fonts do not duplicate subset prefixes +>>>>>>> 2638dfe969874b954502b26656df48ad971e306d doc_id = get_doc_id("text_font.pdf") resp = client.get(f"/documents/{doc_id}/fonts") assert resp.status_code == 200 fonts = resp.json() assert len(fonts) > 0 + for font in fonts: if font.get("isSubset"): assert font["subsetTag"] in font["fontName"] +<<<<<<< HEAD + # internalFontId == fontName (e.g. "ABCDEF+Arial") + assert font["internalFontId"] == font["fontName"] + # The ID must contain exactly one '+' from the subset tag + assert font["internalFontId"].count("+") == 1 + + +# ========================================================================= +# 3. Non-subset font ID format +# ========================================================================= + +def test_internal_font_id_non_subset_format(): + """ + Non-subset fonts must have internalFontId = fontName_type_flags. + Examples: "Helvetica_Type1_32", "Times-Roman_Type1_32". + """ + doc_id = get_doc_id("utf-8.pdf") + resp = client.get(f"/documents/{doc_id}/fonts") + assert resp.status_code == 200 + fonts = resp.json() + assert len(fonts) > 0 + + for font in fonts: + if not font.get("isSubset"): + expected = f"{font['fontName']}_{font['type']}_{font['flags']}" + assert font["internalFontId"] == expected, ( + f"Non-subset internalFontId mismatch: " + f"got '{font['internalFontId']}', expected '{expected}'" + ) + # Must NOT contain a '+' (no subset prefix) + assert "+" not in font["internalFontId"] + + +# ========================================================================= +# 4. ID stability across document-level and page-level APIs +# ========================================================================= + +def test_font_id_stable_across_apis(): + """ + The internalFontId for the same font must be identical whether queried + from the document-level /fonts endpoint or the page-level /pages/0/fonts. + """ + for pdf in ("subset_font.pdf", "utf-8.pdf", "vertical_text.pdf"): + filepath = os.path.abspath(f"../corpus/fonts/{pdf}") + if not os.path.exists(filepath): + continue + + doc_id = get_doc_id(pdf) + + doc_resp = client.get(f"/documents/{doc_id}/fonts") + page_resp = client.get(f"/documents/{doc_id}/pages/0/fonts") + assert doc_resp.status_code == 200 + assert page_resp.status_code == 200 + + doc_ids = {f["fontName"]: f["internalFontId"] for f in doc_resp.json()} + page_ids = {f["fontName"]: f["internalFontId"] for f in page_resp.json()} + + for name in page_ids: + assert name in doc_ids, f"Page font '{name}' not in doc fonts for {pdf}" + assert page_ids[name] == doc_ids[name], ( + f"ID mismatch for '{name}' in {pdf}: " + f"doc='{doc_ids[name]}' vs page='{page_ids[name]}'" + ) + + +# ========================================================================= +# 5. CID collection regression +# ========================================================================= +======= assert not font["internalFontId"].startswith( font["subsetTag"] + "_" + font["subsetTag"] ) +>>>>>>> 2638dfe969874b954502b26656df48ad971e306d def test_cid_collection_regression(): - # Verify Adobe collections + """Adobe CID collections must use the 'Adobe-' prefix.""" doc_id = get_doc_id("vertical_text.pdf") resp = client.get(f"/documents/{doc_id}/fonts") fonts = resp.json() @@ -70,6 +204,31 @@ def test_cid_collection_regression(): assert "Adobe-" in font["cidSystemInfo"] +<<<<<<< HEAD +def test_cns1_regression(): + """Verify Adobe-CNS1 (Traditional Chinese) CID fonts and text extraction.""" + doc_id = get_doc_id("cns1_test.pdf") + + # 1. Verify font extraction + resp = client.get(f"/documents/{doc_id}/fonts") + assert resp.status_code == 200 + fonts = resp.json() + assert len(fonts) > 0 + cns1_fonts = [f for f in fonts if f.get("cidSystemInfo") == "Adobe-CNS1"] + assert len(cns1_fonts) > 0, "No Adobe-CNS1 fonts detected" + + # 2. Verify text extraction + resp = client.get(f"/documents/{doc_id}/pages/0/text") + assert resp.status_code == 200 + text = resp.json()["text"] + assert "\u4e00\u4e2d\u4ed7" in text, "Failed to extract Traditional Chinese text" + +# ========================================================================= +# 6. UTF-8 corpus regression +# ========================================================================= + +======= +>>>>>>> 2638dfe969874b954502b26656df48ad971e306d def test_utf8_corpus_regression(): doc_id = get_doc_id("utf-8.pdf") resp = client.get(f"/documents/{doc_id}/pages/0/text") @@ -79,9 +238,40 @@ def test_utf8_corpus_regression(): assert len(data["text"]) > 0 +<<<<<<< HEAD +# ========================================================================= +# 7. Font size regression +# ========================================================================= + +======= +>>>>>>> 2638dfe969874b954502b26656df48ad971e306d def test_font_size_regression(): doc_id = get_doc_id("utf-8.pdf") resp = client.get(f"/documents/{doc_id}/pages/0/text") data = resp.json() sizes = set(g["fontSize"] for g in data["glyphs"]) assert len(sizes) >= 2 # utf-8.pdf should have multiple font sizes +<<<<<<< HEAD + + +# ========================================================================= +# 8. ID uniqueness +# ========================================================================= + +def test_font_ids_unique_within_document(): + """Each font within a document must have a distinct internalFontId.""" + for pdf in ("subset_font.pdf", "utf-8.pdf", "vertical_text.pdf"): + filepath = os.path.abspath(f"../corpus/fonts/{pdf}") + if not os.path.exists(filepath): + continue + + doc_id = get_doc_id(pdf) + resp = client.get(f"/documents/{doc_id}/fonts") + fonts = resp.json() + + ids = [f["internalFontId"] for f in fonts] + assert len(ids) == len(set(ids)), ( + f"Duplicate internalFontId values in {pdf}: {ids}" + ) +======= +>>>>>>> 2638dfe969874b954502b26656df48ad971e306d diff --git a/gateway/tests/test_glyph_metrics.py b/gateway/tests/test_glyph_metrics.py new file mode 100644 index 0000000..1954adb --- /dev/null +++ b/gateway/tests/test_glyph_metrics.py @@ -0,0 +1,93 @@ +import pytest +import os +from fastapi.testclient import TestClient +from app.main import app + +client = TestClient(app) + + +def get_doc_id(filename: str) -> str: + """Upload a PDF from corpus/fonts and return its document ID.""" + filepath = os.path.abspath(f"gateway/../corpus/fonts/{filename}") + if not os.path.exists(filepath): + filepath = os.path.abspath(f"../corpus/fonts/{filename}") + with open(filepath, "rb") as f: + resp = client.post( + "/documents", + files={"file": (filename, f, "application/pdf")} + ) + assert resp.status_code == 201, f"Failed to load {filename}: {resp.json()}" + return resp.json()["id"] + + +def test_glyph_width_proportionality(): + """Verify that wide characters return larger glyph widths than narrow characters.""" + doc_id = get_doc_id("utf-8.pdf") + + # 1. Fetch fonts for the page + resp = client.get(f"/documents/{doc_id}/pages/0/fonts") + assert resp.status_code == 200 + fonts = resp.json() + assert len(fonts) > 0, "No fonts extracted from utf-8.pdf" + + # Find an embedded or substituted font name + font_name = fonts[0]["fontName"] + + # 2. Get width of wide character 'W' (charcode 87) + w_resp = client.get( + f"/documents/{doc_id}/pages/0/fonts/glyph-width", + params={"font_name": font_name, "charcode": 87, "font_size": 12.0} + ) + assert w_resp.status_code == 200 + w_width = w_resp.json()["width"] + assert w_width > 0.0 + + # 3. Get width of narrow character 'i' (charcode 105) + i_resp = client.get( + f"/documents/{doc_id}/pages/0/fonts/glyph-width", + params={"font_name": font_name, "charcode": 105, "font_size": 12.0} + ) + assert i_resp.status_code == 200 + i_width = i_resp.json()["width"] + assert i_width > 0.0 + + # 'W' must be strictly wider than 'i' in proportional typefaces + assert w_width > i_width, f"Expected width('W') > width('i'), got {w_width} vs {i_width}" + + +def test_glyph_width_font_size_scaling(): + """Verify that glyph width scales proportionally with font size.""" + doc_id = get_doc_id("utf-8.pdf") + resp = client.get(f"/documents/{doc_id}/pages/0/fonts") + fonts = resp.json() + font_name = fonts[0]["fontName"] + + # Width at 12pt + resp_12 = client.get( + f"/documents/{doc_id}/pages/0/fonts/glyph-width", + params={"font_name": font_name, "charcode": 65, "font_size": 12.0} + ) + assert resp_12.status_code == 200 + width_12 = resp_12.json()["width"] + + # Width at 24pt + resp_24 = client.get( + f"/documents/{doc_id}/pages/0/fonts/glyph-width", + params={"font_name": font_name, "charcode": 65, "font_size": 24.0} + ) + assert resp_24.status_code == 200 + width_24 = resp_24.json()["width"] + + # Scaling must be linear: width(24pt) = 2.0 * width(12pt) + assert pytest.approx(width_24) == 2.0 * width_12 + + +def test_glyph_width_invalid_font(): + """Verify that querying an invalid font returns a bad request error.""" + doc_id = get_doc_id("utf-8.pdf") + resp = client.get( + f"/documents/{doc_id}/pages/0/fonts/glyph-width", + params={"font_name": "NonExistentFontName123", "charcode": 65, "font_size": 12.0} + ) + assert resp.status_code == 400 + assert "detail" in resp.json() diff --git a/gateway/tests/test_routes.py b/gateway/tests/test_routes.py index 03fb8d9..14e38ce 100644 --- a/gateway/tests/test_routes.py +++ b/gateway/tests/test_routes.py @@ -137,6 +137,10 @@ def test_extract_page_text(client: TestClient): first_glyph = glyphs[0] for key in ["text", "x", "y", "w", "h", "fontSize"]: assert key in first_glyph + + for g in glyphs: + assert g["fontSize"] != 1.0, f"Fake fontSize 1.0 detected for glyph: {g}" + assert g["text"] not in ["\r", "\n"], f"Control character detected in glyph bounds: {g}" def test_apply_edits_and_incremental_save(client: TestClient): @@ -441,7 +445,7 @@ def test_font_size_and_diagnostics_advanced(client: TestClient): assert font["sourceType"] == "Embedded" assert len(font["subsetTag"]) == 6 assert font["subsetTag"].isupper() - assert font["internalFontId"] == f"{font['subsetTag']}_{font['fontName']}" + assert font["internalFontId"] == font["fontName"] else: assert len(font["subsetTag"]) == 0 assert font["internalFontId"] == f"{font['fontName']}_{font['type']}_{font['flags']}" diff --git a/gateway/tests/validation_output.txt b/gateway/tests/validation_output.txt new file mode 100644 index 0000000..a4a8bb8 --- /dev/null +++ b/gateway/tests/validation_output.txt @@ -0,0 +1,554 @@ +================================================================================ + FONT EXTRACTION API VALIDATION + Server: http://localhost:8000 + Corpus: C:\Users\Maskan\Desktop\pdf_editor\pdf\corpus\fonts + Timestamp: 2026-06-02 12:36:47 +================================================================================ + + Server status: OK (200) + PDFs to validate: ['utf-8.pdf', 'vertical_text.pdf', 'subset_font.pdf'] + +================================================================================ + VALIDATING: utf-8.pdf +================================================================================ + + +--- Step 1: Upload utf-8.pdf --- + + Status: 201 + Response: { + "id": "6764e35e-3594-4b50-a378-3ea122a54d9d", + "filename": "utf-8.pdf", + "sizeBytes": 1275, + "totalPages": 1, + "uploadedAt": "2026-06-02T07:06:50.429369Z", + "status": "ready" +} + Document ID: 6764e35e-3594-4b50-a378-3ea122a54d9d + +--- Step 2: Document Font Extraction --- + + Status: 200 + Font count: 2 + + Font #0: + { + "fontName": "Helvetica", + "type": "Type1", + "isEmbedded": false, + "isSubset": false, + "isVertical": false, + "encoding": "WinAnsiEncoding", + "hasToUnicode": true, + "cmapName": "None", + "cidSystemInfo": "None", + "subsetTag": "", + "sourceType": "SystemFallback", + "substitutedFrom": "", + "substitutedTo": "", + "normalizedFamily": "Helvetica", + "internalFontId": "Helvetica_Type1_32", + "flags": 32, + "ascent": 905.0, + "descent": -211.0, + "capHeight": 728.0 +} + + Font #1: + { + "fontName": "Times-Roman", + "type": "Type1", + "isEmbedded": false, + "isSubset": false, + "isVertical": false, + "encoding": "WinAnsiEncoding", + "hasToUnicode": true, + "cmapName": "None", + "cidSystemInfo": "None", + "subsetTag": "", + "sourceType": "SystemFallback", + "substitutedFrom": "", + "substitutedTo": "", + "normalizedFamily": "Times", + "internalFontId": "Times-Roman_Type1_32", + "flags": 32, + "ascent": 891.0, + "descent": -216.0, + "capHeight": 662.0 +} + +--- Step 3: Page Font Extraction (page 0) --- + + Status: 200 + Page font count: 2 + + Page Font #0: + { + "fontName": "Helvetica", + "type": "Type1", + "isEmbedded": false, + "isSubset": false, + "isVertical": false, + "encoding": "WinAnsiEncoding", + "hasToUnicode": true, + "cmapName": "None", + "cidSystemInfo": "None", + "subsetTag": "", + "sourceType": "SystemFallback", + "substitutedFrom": "", + "substitutedTo": "", + "normalizedFamily": "Helvetica", + "internalFontId": "Helvetica_Type1_32", + "flags": 32, + "ascent": 905.0, + "descent": -211.0, + "capHeight": 728.0 +} + + Page Font #1: + { + "fontName": "Times-Roman", + "type": "Type1", + "isEmbedded": false, + "isSubset": false, + "isVertical": false, + "encoding": "WinAnsiEncoding", + "hasToUnicode": true, + "cmapName": "None", + "cidSystemInfo": "None", + "subsetTag": "", + "sourceType": "SystemFallback", + "substitutedFrom": "", + "substitutedTo": "", + "normalizedFamily": "Times", + "internalFontId": "Times-Roman_Type1_32", + "flags": 32, + "ascent": 891.0, + "descent": -216.0, + "capHeight": 662.0 +} + + Document font names: ['Helvetica', 'Times-Roman'] + Page font names: ['Helvetica', 'Times-Roman'] + [OK] Page fonts are a subset of document fonts + +--- Step 4: Text Extraction (page 0) --- + + Status: 200 + Extracted text: 'Hello World - UTF-8 Test Document\r\nStandard Latin Text for Encoding Verification\r\nFont Size Detection Sample: Small Text 12pt\r\nLARGE TEXT FOR SIZE 18PT DETECTION\r\nMore 18pt content: ABCDEFGHabcdefgh 0123456789\r\nBack to 12pt: The quick brown fox jumps over the lazy dog\r\nSpecial chars: copyright secti' + Glyph count: 312 + + First 5 glyphs (sample): + Glyph #0: { + "text": "H", + "x": 72.95999908447266, + "y": 720.0, + "w": 6.7440032958984375, + "h": 8.59197998046875, + "fontSize": 12.0 +} + Glyph #1: { + "text": "e", + "x": 81.10800170898438, + "y": 719.8679809570312, + "w": 5.736000061035156, + "h": 6.49200439453125, + "fontSize": 12.0 +} + Glyph #2: { + "text": "l", + "x": 88.10400390625, + "y": 720.0, + "w": 1.055999755859375, + "h": 8.59197998046875, + "fontSize": 12.0 +} + Glyph #3: { + "text": "l", + "x": 90.76799774169922, + "y": 720.0, + "w": 1.055999755859375, + "h": 8.59197998046875, + "fontSize": 12.0 +} + Glyph #4: { + "text": "o", + "x": 93.05999755859375, + "y": 719.8679809570312, + "w": 5.832000732421875, + "h": 6.49200439453125, + "fontSize": 12.0 +} + + Font sizes detected: [1.0, 12.0, 18.0] + +--- Step 5: Content-Specific Validation --- + + UTF-8 PDF - extracted text: 'Hello World - UTF-8 Test Document\r\nStandard Latin Text for Encoding Verification\r\nFont Size Detection Sample: Small Text 12pt\r\nLARGE TEXT FOR SIZE 18PT DETECTION\r\nMore 18pt content: ABCDEFGHabcdefgh 0123456789\r\nBack to 12pt: The quick brown fox jumps over the lazy dog\r\nSpecial chars: copyright secti' + [INFO] No non-ASCII characters detected - content may be ASCII-only + +================================================================================ + VALIDATING: vertical_text.pdf +================================================================================ + + +--- Step 1: Upload vertical_text.pdf --- + + Status: 201 + Response: { + "id": "d28fd4e4-a059-4a9b-bf15-b6cf25caf280", + "filename": "vertical_text.pdf", + "sizeBytes": 3518, + "totalPages": 1, + "uploadedAt": "2026-06-02T07:06:50.465835Z", + "status": "ready" +} + Document ID: d28fd4e4-a059-4a9b-bf15-b6cf25caf280 + +--- Step 2: Document Font Extraction --- + + Status: 200 + Font count: 1 + + Font #0: + { + "fontName": "Test", + "type": "TrueType", + "isEmbedded": false, + "isSubset": false, + "isVertical": true, + "encoding": "Identity-V", + "hasToUnicode": true, + "cmapName": "Identity-V", + "cidSystemInfo": "None", + "subsetTag": "", + "sourceType": "Substituted", + "substitutedFrom": "Test", + "substitutedTo": "Arial", + "normalizedFamily": "Test", + "internalFontId": "Test_TrueType_524320", + "flags": 524320, + "ascent": 905.0, + "descent": -211.0, + "capHeight": 728.0 +} + +--- Step 3: Page Font Extraction (page 0) --- + + Status: 200 + Page font count: 1 + + Page Font #0: + { + "fontName": "Test", + "type": "TrueType", + "isEmbedded": false, + "isSubset": false, + "isVertical": true, + "encoding": "Identity-V", + "hasToUnicode": true, + "cmapName": "Identity-V", + "cidSystemInfo": "None", + "subsetTag": "", + "sourceType": "Substituted", + "substitutedFrom": "Test", + "substitutedTo": "Arial", + "normalizedFamily": "Test", + "internalFontId": "Test_TrueType_524320", + "flags": 524320, + "ascent": 905.0, + "descent": -211.0, + "capHeight": 728.0 +} + + Document font names: ['Test'] + Page font names: ['Test'] + [OK] Page fonts are a subset of document fonts + +--- Step 4: Text Extraction (page 0) --- + + Status: 200 + Extracted text: 'Hello World!\r\nHello' + Glyph count: 19 + + First 5 glyphs (sample): + Glyph #0: { + "text": "H", + "x": 6.832000255584717, + "y": 180.1840057373047, + "w": 6.552000522613525, + "h": 8.699996948242188, + "fontSize": 12.0 +} + Glyph #1: { + "text": "e", + "x": 7.324000358581543, + "y": 171.39999389648438, + "w": 5.495999336242676, + "h": 6.756011962890625, + "fontSize": 12.0 +} + Glyph #2: { + "text": "l", + "x": 9.687999725341797, + "y": 160.49200439453125, + "w": 1.055999755859375, + "h": 9.251998901367188, + "fontSize": 12.0 +} + Glyph #3: { + "text": "l", + "x": 9.687999725341797, + "y": 149.4759979248047, + "w": 1.055999755859375, + "h": 9.251998901367188, + "fontSize": 12.0 +} + Glyph #4: { + "text": "o", + "x": 7.324000358581543, + "y": 140.69200134277344, + "w": 5.951999664306641, + "h": 6.7559967041015625, + "fontSize": 12.0 +} + + Font sizes detected: [1.0, 12.0] + +--- Step 5: Content-Specific Validation --- + + Vertical text PDF - fonts with isVertical=true: 1 + - Test (isVertical=true) + [OK] Vertical fonts detected correctly + +================================================================================ + VALIDATING: subset_font.pdf +================================================================================ + + +--- Step 1: Upload subset_font.pdf --- + + Status: 201 + Response: { + "id": "cf3f2218-6ee7-4e34-973d-6e1b61c13cf8", + "filename": "subset_font.pdf", + "sizeBytes": 646, + "totalPages": 1, + "uploadedAt": "2026-06-02T07:06:50.497336Z", + "status": "ready" +} + Document ID: cf3f2218-6ee7-4e34-973d-6e1b61c13cf8 + +--- Step 2: Document Font Extraction --- + + Status: 200 + Font count: 1 + + Font #0: + { + "fontName": "ABCDEF+Arial", + "type": "TrueType", + "isEmbedded": true, + "isSubset": true, + "isVertical": false, + "encoding": "WinAnsiEncoding", + "hasToUnicode": true, + "cmapName": "None", + "cidSystemInfo": "None", + "subsetTag": "ABCDEF", + "sourceType": "Embedded", + "substitutedFrom": "", + "substitutedTo": "", + "normalizedFamily": "Arial", + "internalFontId": "ABCDEF_ABCDEF+Arial", + "flags": 0, + "ascent": 905.0, + "descent": -211.0, + "capHeight": 728.0 +} + +--- Step 3: Page Font Extraction (page 0) --- + + Status: 200 + Page font count: 1 + + Page Font #0: + { + "fontName": "ABCDEF+Arial", + "type": "TrueType", + "isEmbedded": true, + "isSubset": true, + "isVertical": false, + "encoding": "WinAnsiEncoding", + "hasToUnicode": true, + "cmapName": "None", + "cidSystemInfo": "None", + "subsetTag": "ABCDEF", + "sourceType": "Embedded", + "substitutedFrom": "", + "substitutedTo": "", + "normalizedFamily": "Arial", + "internalFontId": "ABCDEF_ABCDEF+Arial", + "flags": 0, + "ascent": 905.0, + "descent": -211.0, + "capHeight": 728.0 +} + + Document font names: ['ABCDEF+Arial'] + Page font names: ['ABCDEF+Arial'] + [OK] Page fonts are a subset of document fonts + +--- Step 4: Text Extraction (page 0) --- + + Status: 200 + Extracted text: 'Subset Text' + Glyph count: 11 + + First 5 glyphs (sample): + Glyph #0: { + "text": "S", + "x": 72.54000091552734, + "y": 719.8679809570312, + "w": 6.839996337890625, + "h": 8.8680419921875, + "fontSize": 12.0 +} + Glyph #1: { + "text": "u", + "x": 80.77200317382812, + "y": 719.8679809570312, + "w": 5.0399932861328125, + "h": 6.36004638671875, + "fontSize": 12.0 +} + Glyph #2: { + "text": "b", + "x": 87.45600128173828, + "y": 719.8679809570312, + "w": 5.400001525878906, + "h": 8.7239990234375, + "fontSize": 12.0 +} + Glyph #3: { + "text": "s", + "x": 93.72000122070312, + "y": 719.8679809570312, + "w": 5.159996032714844, + "h": 6.49200439453125, + "fontSize": 12.0 +} + Glyph #4: { + "text": "e", + "x": 99.79199981689453, + "y": 719.8679809570312, + "w": 5.736000061035156, + "h": 6.49200439453125, + "fontSize": 12.0 +} + + Font sizes detected: [12.0] + +--- Step 5: Content-Specific Validation --- + + Subset font PDF - fonts with isSubset=true: 1 + - ABCDEF+Arial (isSubset=true, subsetTag='ABCDEF') + [OK] Subset fonts detected correctly + +================================================================================ + FINAL REPORT +================================================================================ + + +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + PDF: utf-8.pdf + Document ID: 6764e35e-3594-4b50-a378-3ea122a54d9d + Upload Status: 201 + Document Font Count: 2 + Page Font Count: 2 + Glyph Count: 312 + Font Sizes: [1.0, 12.0, 18.0] + Issues: 6 + >> Glyph #0: missing coordinate 'right' + >> Glyph #0: missing coordinate 'bottom' + >> Glyph #1: missing coordinate 'right' + >> Glyph #1: missing coordinate 'bottom' + >> Glyph #2: missing coordinate 'right' + >> Glyph #2: missing coordinate 'bottom' + Errors: 0 + +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + PDF: vertical_text.pdf + Document ID: d28fd4e4-a059-4a9b-bf15-b6cf25caf280 + Upload Status: 201 + Document Font Count: 1 + Page Font Count: 1 + Glyph Count: 19 + Font Sizes: [1.0, 12.0] + Issues: 6 + >> Glyph #0: missing coordinate 'right' + >> Glyph #0: missing coordinate 'bottom' + >> Glyph #1: missing coordinate 'right' + >> Glyph #1: missing coordinate 'bottom' + >> Glyph #2: missing coordinate 'right' + >> Glyph #2: missing coordinate 'bottom' + Errors: 0 + +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + PDF: subset_font.pdf + Document ID: cf3f2218-6ee7-4e34-973d-6e1b61c13cf8 + Upload Status: 201 + Document Font Count: 1 + Page Font Count: 1 + Glyph Count: 11 + Font Sizes: [12.0] + Issues: 6 + >> Glyph #0: missing coordinate 'right' + >> Glyph #0: missing coordinate 'bottom' + >> Glyph #1: missing coordinate 'right' + >> Glyph #1: missing coordinate 'bottom' + >> Glyph #2: missing coordinate 'right' + >> Glyph #2: missing coordinate 'bottom' + Errors: 0 + +================================================================================ + SUMMARY ANSWERS +================================================================================ + + 1. Are fonts being extracted correctly? + YES - 4 total fonts across 3 PDFs + + 2. Are page fonts and document fonts consistent? + YES + + 3. Are font sizes being extracted correctly? + YES + + 4. Are vertical/subset fonts detected correctly? + Vertical: YES + Subset: YES + + 5. Are there any metadata inaccuracies? + YES - 18 issues and 0 errors found + - [utf-8.pdf] Glyph #0: missing coordinate 'right' + - [utf-8.pdf] Glyph #0: missing coordinate 'bottom' + - [utf-8.pdf] Glyph #1: missing coordinate 'right' + - [utf-8.pdf] Glyph #1: missing coordinate 'bottom' + - [utf-8.pdf] Glyph #2: missing coordinate 'right' + - [utf-8.pdf] Glyph #2: missing coordinate 'bottom' + - [vertical_text.pdf] Glyph #0: missing coordinate 'right' + - [vertical_text.pdf] Glyph #0: missing coordinate 'bottom' + - [vertical_text.pdf] Glyph #1: missing coordinate 'right' + - [vertical_text.pdf] Glyph #1: missing coordinate 'bottom' + - [vertical_text.pdf] Glyph #2: missing coordinate 'right' + - [vertical_text.pdf] Glyph #2: missing coordinate 'bottom' + - [subset_font.pdf] Glyph #0: missing coordinate 'right' + - [subset_font.pdf] Glyph #0: missing coordinate 'bottom' + - [subset_font.pdf] Glyph #1: missing coordinate 'right' + - [subset_font.pdf] Glyph #1: missing coordinate 'bottom' + - [subset_font.pdf] Glyph #2: missing coordinate 'right' + - [subset_font.pdf] Glyph #2: missing coordinate 'bottom' + +================================================================================ + VALIDATION COMPLETE: 18 issues, 0 errors +================================================================================ diff --git a/model_test.json b/model_test.json new file mode 100644 index 0000000..6cdd266 Binary files /dev/null and b/model_test.json differ diff --git a/scripts/start_gateway.ps1 b/scripts/start_gateway.ps1 index cecfa95..d8c5a0d 100644 --- a/scripts/start_gateway.ps1 +++ b/scripts/start_gateway.ps1 @@ -10,7 +10,7 @@ if ($Port -le 0) { if ($envPort) { $Port = $envPort } else { - $Port = 8080 + $Port = 8000 } } diff --git a/wasm/bindings/pdf_engine_facade.cpp b/wasm/bindings/pdf_engine_facade.cpp index e0613d3..02cec89 100644 --- a/wasm/bindings/pdf_engine_facade.cpp +++ b/wasm/bindings/pdf_engine_facade.cpp @@ -90,6 +90,11 @@ std::expected, pdfengine::EngineError> WasmMockPage::ex return std::vector(); } +std::expected WasmMockPage::getGlyphWidth(const std::string& fontName, uint32_t charcode, double fontSize) const { + (void)fontName; (void)charcode; + return 6.0 * (fontSize / 12.0); +} + std::expected, pdfengine::EngineError> WasmMockPage::getFonts() const { // Simulate a non-embedded Helvetica font being substituted with Liberation Sans. // This is the canonical Phase 1 font substitution scenario. diff --git a/wasm/bindings/pdf_engine_facade.hpp b/wasm/bindings/pdf_engine_facade.hpp index 1c9a080..2798996 100644 --- a/wasm/bindings/pdf_engine_facade.hpp +++ b/wasm/bindings/pdf_engine_facade.hpp @@ -21,6 +21,7 @@ public: [[nodiscard]] std::expected, pdfengine::EngineError> extractTextWithBounds() const override; [[nodiscard]] std::expected, pdfengine::EngineError> getFonts() const override; [[nodiscard]] std::expected, pdfengine::EngineError> extractAnnotationsText() const override; + [[nodiscard]] std::expected getGlyphWidth(const std::string& fontName, uint32_t charcode, double fontSize) const override; [[nodiscard]] pdfengine::DevicePoint pageToDevice(const pdfengine::Point2D& pagePoint, int deviceWidth, int deviceHeight, int rotate = 0) const noexcept override; [[nodiscard]] pdfengine::Point2D deviceToPage(const pdfengine::DevicePoint& devicePoint, int deviceWidth, int deviceHeight, int rotate = 0) const noexcept override;