Merge pull request 'saqib' (#33) from saqib into dev

Reviewed-on: https://gitea.maskantech.in/gitea_admin/pdf/pulls/33
This commit is contained in:
furqan
2026-06-08 10:59:22 +00:00
42 changed files with 2739 additions and 159 deletions
+55
View File
@@ -111,12 +111,64 @@ PYBIND11_MODULE(pdfengine, m) {
return "FontInfo(font_name='" + self.fontName + "', type='" + self.type + "', is_embedded=" + (self.isEmbedded ? "True" : "False") + ")";
});
py::class_<pdfengine::Glyph>(m, "Glyph")
.def_readonly("text", &pdfengine::Glyph::text)
.def_readonly("unicode", &pdfengine::Glyph::unicode)
.def_readonly("font_name", &pdfengine::Glyph::fontName)
.def_readonly("flags", &pdfengine::Glyph::flags)
.def_readonly("font_size", &pdfengine::Glyph::fontSize)
.def_readonly("origin_x", &pdfengine::Glyph::originX)
.def_readonly("origin_y", &pdfengine::Glyph::originY)
.def_readonly("bbox_x", &pdfengine::Glyph::bboxX)
.def_readonly("bbox_y", &pdfengine::Glyph::bboxY)
.def_readonly("bbox_w", &pdfengine::Glyph::bboxW)
.def_readonly("bbox_h", &pdfengine::Glyph::bboxH)
.def_readonly("angle", &pdfengine::Glyph::angle);
py::class_<pdfengine::TextRun>(m, "TextRun")
.def_readonly("text", &pdfengine::TextRun::text)
.def_readonly("font_name", &pdfengine::TextRun::fontName)
.def_readonly("flags", &pdfengine::TextRun::flags)
.def_readonly("font_size", &pdfengine::TextRun::fontSize)
.def_readonly("internal_font_id", &pdfengine::TextRun::internalFontId)
.def_readonly("is_embedded", &pdfengine::TextRun::isEmbedded)
.def_readonly("type", &pdfengine::TextRun::type)
.def_readonly("glyphs", &pdfengine::TextRun::glyphs)
.def_readonly("x", &pdfengine::TextRun::x)
.def_readonly("y", &pdfengine::TextRun::y)
.def_readonly("w", &pdfengine::TextRun::w)
.def_readonly("h", &pdfengine::TextRun::h);
py::class_<pdfengine::TextLine>(m, "TextLine")
.def_readonly("runs", &pdfengine::TextLine::runs)
.def_readonly("baseline_y", &pdfengine::TextLine::baselineY)
.def_readonly("x", &pdfengine::TextLine::x)
.def_readonly("y", &pdfengine::TextLine::y)
.def_readonly("w", &pdfengine::TextLine::w)
.def_readonly("h", &pdfengine::TextLine::h);
py::class_<pdfengine::Paragraph>(m, "Paragraph")
.def_readonly("lines", &pdfengine::Paragraph::lines)
.def_readonly("x", &pdfengine::Paragraph::x)
.def_readonly("y", &pdfengine::Paragraph::y)
.def_readonly("w", &pdfengine::Paragraph::w)
.def_readonly("h", &pdfengine::Paragraph::h);
py::class_<pdfengine::PageModel>(m, "PageModel")
.def_readonly("paragraphs", &pdfengine::PageModel::paragraphs)
.def_readonly("width", &pdfengine::PageModel::width)
.def_readonly("height", &pdfengine::PageModel::height)
.def_readonly("page_index", &pdfengine::PageModel::pageIndex);
py::class_<pdfengine::PdfPage, std::shared_ptr<pdfengine::PdfPage>>(m, "PdfPage")
.def_property_readonly("width", &pdfengine::PdfPage::width)
.def_property_readonly("height", &pdfengine::PdfPage::height)
.def("render", [](const pdfengine::PdfPage& self, int dpi) {
return get_or_throw(self.render(dpi));
}, py::arg("dpi") = 96)
.def("extract_document_model", [](const pdfengine::PdfPage& self) {
return get_or_throw(self.extractDocumentModel());
})
.def("extract_text", [](const pdfengine::PdfPage& self) {
return get_or_throw(self.extractText());
})
@@ -141,6 +193,9 @@ PYBIND11_MODULE(pdfengine, m) {
.def("get_fonts", [](const pdfengine::PdfPage& self) {
return get_or_throw(self.getFonts());
})
.def("get_glyph_width", [](const pdfengine::PdfPage& self, const std::string& fontName, uint32_t charcode, double fontSize) {
return get_or_throw(self.getGlyphWidth(fontName, charcode, fontSize));
}, py::arg("font_name"), py::arg("charcode"), py::arg("font_size"))
.def("page_to_device", &pdfengine::PdfPage::pageToDevice,
py::arg("page_point"), py::arg("device_width"), py::arg("device_height"), py::arg("rotate") = 0)
.def("device_to_page", &pdfengine::PdfPage::deviceToPage,
+90
View File
@@ -0,0 +1,90 @@
%PDF-1.3
%“Œ‹ž ReportLab Generated PDF document (opensource)
1 0 obj
<<
/F1 2 0 R /F2 3 0 R
>>
endobj
2 0 obj
<<
/BaseFont /Helvetica /Encoding /WinAnsiEncoding /Name /F1 /Subtype /Type1 /Type /Font
>>
endobj
3 0 obj
<<
/BaseFont /MSung-Light /DescendantFonts [ <<
/BaseFont /MSung-Light /CIDSystemInfo <<
/Ordering (CNS1) /Registry (Adobe) /Supplement 1
>> /DW 1000 /FontDescriptor <<
/Ascent 752 /CapHeight 737 /Descent -271 /Flags 6 /FontBBox [ -160 -249 1015 888 ] /FontName /MSung-Light
/ItalicAngle 0 /Leading 148 /MaxWidth 1000 /MissingWidth 500 /StemH 45 /StemV 58
/Type /FontDescriptor /XHeight 553
>> /Subtype /CIDFontType0 /Type /Font
/W [ 1 2 250 3 [ 408 668 490 875 698 250 240 ] 10 [ 240 417 667 250 313 250 520 500 ] 18 26 500
27 28 250 29 31 667 32 [ 396 921 677 615 719 760 625 552 771 802
354 ] 43 [ 354 781 604 927 750 823 563 823 729 542
698 771 729 948 771 677 635 344 520 344
469 500 250 469 521 427 521 438 271 469
531 250 ]
75 [ 250 458 240 802 531 500 521 ] 82 [ 521 365 333 292 521 458 677 479 458 427
480 496 480 667 ] ]
>> ] /Encoding /UniGB-UCS2-H /Name /F2 /Subtype /Type0 /Type /Font
>>
endobj
4 0 obj
<<
/Contents 8 0 R /MediaBox [ 0 0 595.2756 841.8898 ] /Parent 7 0 R /Resources <<
/Font 1 0 R /ProcSet [ /PDF /Text /ImageB /ImageC /ImageI ]
>> /Rotate 0 /Trans <<
>>
/Type /Page
>>
endobj
5 0 obj
<<
/PageMode /UseNone /Pages 7 0 R /Type /Catalog
>>
endobj
6 0 obj
<<
/Author (anonymous) /CreationDate (D:20260602153802+05'00') /Creator (anonymous) /Keywords () /ModDate (D:20260602153802+05'00') /Producer (ReportLab PDF Library - \(opensource\))
/Subject (unspecified) /Title (untitled) /Trapped /False
>>
endobj
7 0 obj
<<
/Count 1 /Kids [ 4 0 R ] /Type /Pages
>>
endobj
8 0 obj
<<
/Filter [ /ASCII85Decode /FlateDecode ] /Length 107
>>
stream
GapQh0E=F,0U\H3T\pNYT^QKk?tc>IP,;W#U1^23ihPEM_?C]6_CBF/28[_U!/s9cYpe/lM_Qn>nC.&g0fCf=<!^TD#gi_<=5X,[c-mU(~>endstream
endobj
xref
0 9
0000000000 65535 f
0000000061 00000 n
0000000102 00000 n
0000000209 00000 n
0000001155 00000 n
0000001358 00000 n
0000001426 00000 n
0000001687 00000 n
0000001746 00000 n
trailer
<<
/ID
[<01c8dab3d2c3e771bf716fccf2b52ce4><01c8dab3d2c3e771bf716fccf2b52ce4>]
% ReportLab generated PDF document -- digest (opensource)
/Info 6 0 R
/Root 5 0 R
/Size 9
>>
startxref
1943
%%EOF
+4
View File
@@ -17,6 +17,9 @@ add_library(pdfengine STATIC
src/parser/pdfium_loader.cpp
src/parser/pdfium_document.cpp
src/fonts/face/font_face.cpp
src/fonts/face/free_type_manager.cpp
src/fonts/loader/font_resolver.cpp
src/fonts/pdf_fonts/font_loader.cpp
src/fonts/shaping/hb_shaper.cpp
src/fonts/cache/glyph_bitmap.cpp
src/fonts/cache/glyph_cache.cpp
@@ -47,6 +50,7 @@ target_link_libraries(pdfengine
PRIVATE
freetype
harfbuzz::harfbuzz
harfbuzz::harfbuzz-subset
PNG::PNG
nlohmann_json::nlohmann_json
)
+58
View File
@@ -77,6 +77,52 @@ struct FontInfo {
double capHeight = 0.0; // Font descriptor CapHeight metric
};
struct Glyph {
std::string text;
uint32_t unicode = 0;
std::string fontName;
uint32_t flags = 0;
double fontSize = 0.0;
double originX = 0.0;
double originY = 0.0;
double bboxX = 0.0, bboxY = 0.0, bboxW = 0.0, bboxH = 0.0;
double angle = 0.0;
};
struct TextRun {
std::string text;
std::string fontName;
uint32_t flags = 0;
double fontSize = 0.0;
std::string internalFontId;
bool isEmbedded = false;
std::string type;
std::vector<Glyph> glyphs;
double x = 0.0, y = 0.0, w = 0.0, h = 0.0;
};
struct TextLine {
std::vector<TextRun> runs;
std::vector<Glyph> glyphs;
double angle = 0.0;
double baselineY = 0.0;
double x = 0.0, y = 0.0, w = 0.0, h = 0.0;
};
struct Paragraph {
std::vector<TextLine> lines;
double x = 0.0, y = 0.0, w = 0.0, h = 0.0;
};
struct PageModel {
std::vector<Paragraph> paragraphs;
double width = 0.0;
double height = 0.0;
int pageIndex = 0;
};
class PdfPage {
public:
virtual ~PdfPage() = default;
@@ -89,14 +135,20 @@ public:
[[nodiscard]] virtual std::expected<std::string, EngineError> extractText() const = 0;
[[nodiscard]] virtual std::expected<std::vector<GlyphBounds>, EngineError> extractTextWithBounds() const = 0;
[[nodiscard]] virtual std::expected<PageModel, EngineError> extractDocumentModel() const = 0;
[[nodiscard]] virtual std::expected<std::vector<FontInfo>, EngineError> getFonts() const = 0;
[[nodiscard]] virtual std::expected<std::vector<std::string>, EngineError> extractAnnotationsText() const = 0;
[[nodiscard]] virtual std::expected<double, EngineError> getGlyphWidth(const std::string& fontName, uint32_t charcode, double fontSize) const = 0;
[[nodiscard]] virtual DevicePoint pageToDevice(const Point2D& pagePoint, int deviceWidth, int deviceHeight, int rotate = 0) const noexcept = 0;
[[nodiscard]] virtual Point2D deviceToPage(const DevicePoint& devicePoint, int deviceWidth, int deviceHeight, int rotate = 0) const noexcept = 0;
};
namespace fonts::pdf_fonts { class Font; }
class PdfDocument {
public:
virtual ~PdfDocument() = default;
@@ -116,6 +168,12 @@ public:
[[nodiscard]] virtual std::expected<std::vector<FontInfo>, EngineError>
getFonts(int startPage = 0, int endPage = -1) const = 0;
[[nodiscard]] virtual std::expected<std::vector<uint8_t>, EngineError>
getFontData(const std::string& internalFontId) const = 0;
[[nodiscard]] virtual std::expected<std::shared_ptr<fonts::pdf_fonts::Font>, std::string>
getResolvedFont(const FontInfo& fontInfo) = 0;
virtual std::expected<void, EngineError> applyEdits(const std::string& editsJson) = 0;
[[nodiscard]] virtual std::expected<std::vector<uint8_t>, EngineError>
+6 -6
View File
@@ -13,12 +13,12 @@ GlyphCache::GlyphCache(std::size_t capacity)
GlyphCache::~GlyphCache() = default;
std::optional<GlyphBitmap> GlyphCache::get(const FontFace& fontFace, unsigned int glyphIndex, unsigned int fontSize) {
FT_Face face = fontFace.getFace();
if (!face) {
uint64_t fontId = fontFace.getId();
if (fontId == 0) {
return std::nullopt;
}
GlyphCacheKey key{face, glyphIndex, fontSize};
GlyphCacheKey key{fontId, glyphIndex, fontSize};
std::size_t shard_idx = getShardIndex(key);
auto& shard = *shards_[shard_idx];
@@ -37,12 +37,12 @@ std::optional<GlyphBitmap> GlyphCache::get(const FontFace& fontFace, unsigned in
}
void GlyphCache::insert(const FontFace& fontFace, unsigned int glyphIndex, unsigned int fontSize, const GlyphBitmap& bitmap) {
FT_Face face = fontFace.getFace();
if (!face) {
uint64_t fontId = fontFace.getId();
if (fontId == 0) {
return;
}
GlyphCacheKey key{face, glyphIndex, fontSize};
GlyphCacheKey key{fontId, glyphIndex, fontSize};
std::size_t shard_idx = getShardIndex(key);
auto& shard = *shards_[shard_idx];
+3 -3
View File
@@ -12,12 +12,12 @@
namespace pdfengine::fonts {
struct GlyphCacheKey {
FT_Face face;
uint64_t fontId;
unsigned int glyphIndex;
unsigned int fontSize;
bool operator==(const GlyphCacheKey& other) const {
return face == other.face &&
return fontId == other.fontId &&
glyphIndex == other.glyphIndex &&
fontSize == other.fontSize;
}
@@ -25,7 +25,7 @@ struct GlyphCacheKey {
struct GlyphCacheKeyHash {
std::size_t operator()(const GlyphCacheKey& key) const {
std::size_t h1 = std::hash<void*>{}(static_cast<void*>(key.face));
std::size_t h1 = std::hash<uint64_t>{}(key.fontId);
std::size_t h2 = std::hash<unsigned int>{}(key.glyphIndex);
std::size_t h3 = std::hash<unsigned int>{}(key.fontSize);
// Combine hashes using standard boost hash_combine algorithm
+25 -19
View File
@@ -1,16 +1,17 @@
#include "fonts/face/font_face.hpp"
#include "fonts/face/free_type_manager.hpp"
#include <iostream>
#include <atomic>
namespace pdfengine::fonts {
FontFace::FontFace()
: ft_library_(nullptr),
face_(nullptr) {
static std::atomic<uint64_t> g_font_id_counter{1};
if (FT_Init_FreeType(&ft_library_)) {
std::cerr << "Failed to initialize FreeType\n";
}
FontFace::FontFace()
: font_id_(g_font_id_counter.fetch_add(1, std::memory_order_relaxed)),
face_(nullptr),
mutex_(std::make_unique<std::mutex>()) {
}
FontFace::~FontFace() {
@@ -18,17 +19,14 @@ FontFace::~FontFace() {
if (face_) {
FT_Done_Face(face_);
}
if (ft_library_) {
FT_Done_FreeType(ft_library_);
}
}
FontFace::FontFace(FontFace&& other) noexcept
: ft_library_(other.ft_library_),
: font_id_(other.font_id_),
face_(other.face_),
mutex_(std::move(other.mutex_)),
font_data_(std::move(other.font_data_)) {
other.ft_library_ = nullptr;
other.font_id_ = 0;
other.face_ = nullptr;
}
@@ -37,13 +35,11 @@ FontFace& FontFace::operator=(FontFace&& other) noexcept {
if (face_) {
FT_Done_Face(face_);
}
if (ft_library_) {
FT_Done_FreeType(ft_library_);
}
ft_library_ = other.ft_library_;
font_id_ = other.font_id_;
face_ = other.face_;
mutex_ = std::move(other.mutex_);
font_data_ = std::move(other.font_data_);
other.ft_library_ = nullptr;
other.font_id_ = 0;
other.face_ = nullptr;
}
return *this;
@@ -58,7 +54,7 @@ bool FontFace::loadFromFile(const std::string& path) {
font_data_.clear();
if (FT_New_Face(
ft_library_,
FreeTypeManager::instance().getLibrary(),
path.c_str(),
0,
&face_)) {
@@ -91,7 +87,7 @@ bool FontFace::loadFromMemory(const std::vector<uint8_t>& data) {
font_data_ = data;
if (FT_New_Memory_Face(
ft_library_,
FreeTypeManager::instance().getLibrary(),
font_data_.data(),
static_cast<FT_Long>(font_data_.size()),
0,
@@ -113,11 +109,21 @@ FT_Face FontFace::getFace() const {
return face_;
}
uint64_t FontFace::getId() const {
return font_id_;
}
std::mutex& FontFace::getMutex() const {
return *mutex_;
}
std::optional<GlyphBitmap> FontFace::renderGlyph(unsigned int glyphIndex, unsigned int fontSize) {
if (!face_) {
return std::nullopt;
}
std::lock_guard<std::mutex> lock(*mutex_);
// Set font size in pixels.
if (FT_Set_Pixel_Sizes(face_, 0, fontSize)) {
return std::nullopt;
+6 -1
View File
@@ -5,6 +5,8 @@
#include <string>
#include <vector>
#include <cstdint>
#include <mutex>
#include <memory>
#include <ft2build.h>
#include FT_FREETYPE_H
@@ -26,13 +28,16 @@ public:
bool loadFromMemory(const std::vector<uint8_t>& data);
FT_Face getFace() const;
uint64_t getId() const;
std::mutex& getMutex() const;
// Renders a glyph by index and size, returning a GlyphBitmap on success.
std::optional<GlyphBitmap> renderGlyph(unsigned int glyphIndex, unsigned int fontSize);
private:
FT_Library ft_library_;
uint64_t font_id_;
FT_Face face_;
std::unique_ptr<std::mutex> mutex_;
std::vector<uint8_t> font_data_; // Keeps the loaded memory buffer alive for FT_Face
};
@@ -0,0 +1,28 @@
#include "fonts/face/free_type_manager.hpp"
#include <iostream>
namespace pdfengine::fonts {
FreeTypeManager& FreeTypeManager::instance() {
static FreeTypeManager instance;
return instance;
}
FreeTypeManager::FreeTypeManager() : library_(nullptr) {
if (FT_Init_FreeType(&library_)) {
std::cerr << "Failed to initialize FreeType\n";
}
}
FreeTypeManager::~FreeTypeManager() {
if (library_) {
FT_Done_FreeType(library_);
library_ = nullptr;
}
}
FT_Library FreeTypeManager::getLibrary() const {
return library_;
}
} // namespace pdfengine::fonts
@@ -0,0 +1,27 @@
#pragma once
#include <ft2build.h>
#include FT_FREETYPE_H
namespace pdfengine::fonts {
class FreeTypeManager {
public:
static FreeTypeManager& instance();
FT_Library getLibrary() const;
// Delete copy/move constructors and assignment operators for singleton
FreeTypeManager(const FreeTypeManager&) = delete;
FreeTypeManager& operator=(const FreeTypeManager&) = delete;
FreeTypeManager(FreeTypeManager&&) = delete;
FreeTypeManager& operator=(FreeTypeManager&&) = delete;
private:
FreeTypeManager();
~FreeTypeManager();
FT_Library library_;
};
} // namespace pdfengine::fonts
+76
View File
@@ -0,0 +1,76 @@
#include "fonts/loader/font_resolver.hpp"
#include "fonts/pdf_fonts/font_loader.hpp"
#include <spdlog/spdlog.h>
namespace pdfengine::fonts::loader {
FontResolver::FontResolver(std::shared_ptr<PdfDocument> document)
: document_(std::move(document)) {}
std::expected<std::unique_ptr<pdf_fonts::Font>, std::string> FontResolver::resolveFont(const FontInfo& fontInfo) {
if (!document_) {
return std::unexpected("No document attached to FontResolver");
}
if (fontInfo.isEmbedded) {
auto dataRes = document_->getFontData(fontInfo.internalFontId);
if (!dataRes) {
spdlog::error("Failed to extract embedded font data for ID: {}", fontInfo.internalFontId);
return std::unexpected("Failed to extract embedded font data");
}
const auto& bytes = dataRes.value();
// Build FontDescriptor from FontInfo
auto descriptor = std::make_unique<pdf_fonts::FontDescriptor>();
descriptor->setFontName(fontInfo.fontName);
descriptor->setFlags(fontInfo.flags);
descriptor->setAscent(fontInfo.ascent);
descriptor->setDescent(fontInfo.descent);
descriptor->setCapHeight(fontInfo.capHeight);
// Route to the appropriate FontLoader method based on FontInfo type
if (fontInfo.type == "TrueType") {
auto font = pdf_fonts::FontLoader::loadTrueTypeFromMemory(fontInfo.normalizedFamily, bytes, std::move(descriptor));
if (font) return font;
} else if (fontInfo.type == "Type1") {
auto font = pdf_fonts::FontLoader::loadType1FromMemory(fontInfo.normalizedFamily, bytes, std::move(descriptor));
if (font) return font;
} else if (fontInfo.type == "CIDFontType0" || fontInfo.type == "CIDFontType2") {
pdf_fonts::FontType subtype = (fontInfo.type == "CIDFontType0")
? pdf_fonts::FontType::CIDFontType0
: pdf_fonts::FontType::CIDFontType2;
auto font = pdf_fonts::FontLoader::loadCIDFontFromMemory(fontInfo.normalizedFamily, subtype, bytes, std::move(descriptor));
if (font) return font;
}
return std::unexpected("Failed to parse extracted font data");
} else {
// Handle System Fallback
spdlog::info("Resolving system fallback font for: {}", fontInfo.fontName);
// Build FontDescriptor from FontInfo
auto descriptor = std::make_unique<pdf_fonts::FontDescriptor>();
descriptor->setFontName(fontInfo.fontName);
descriptor->setFlags(fontInfo.flags);
descriptor->setAscent(fontInfo.ascent);
descriptor->setDescent(fontInfo.descent);
descriptor->setCapHeight(fontInfo.capHeight);
if (fontInfo.type == "CIDFontType0" || fontInfo.type == "CIDFontType2") {
pdf_fonts::FontType subtype = (fontInfo.type == "CIDFontType0")
? pdf_fonts::FontType::CIDFontType0
: pdf_fonts::FontType::CIDFontType2;
auto font = pdf_fonts::FontLoader::loadCIDFontSystemFallback(fontInfo.normalizedFamily, subtype, std::move(descriptor));
if (font) return font;
} else {
auto font = pdf_fonts::FontLoader::loadType1SystemFallback(fontInfo.normalizedFamily, std::move(descriptor));
if (font) return font;
}
return std::unexpected("Failed to load system fallback font");
}
}
} // namespace pdfengine::fonts::loader
+24
View File
@@ -0,0 +1,24 @@
#pragma once
#include "fonts/pdf_fonts/font.hpp"
#include <pdfengine/pdf_document.hpp>
#include <memory>
#include <string>
#include <expected>
namespace pdfengine::fonts::loader {
class FontResolver {
public:
explicit FontResolver(std::shared_ptr<PdfDocument> document);
// Resolves a FontInfo object into a fully loaded Font ready for shaping.
// If the font is embedded, it extracts the raw bytes from the PdfDocument.
// If it's a system fallback, it uses FontLoader to load it from the OS.
std::expected<std::unique_ptr<pdf_fonts::Font>, std::string> resolveFont(const FontInfo& fontInfo);
private:
std::shared_ptr<PdfDocument> document_;
};
} // namespace pdfengine::fonts::loader
+72 -1
View File
@@ -19,6 +19,19 @@ enum class FontType {
Type3
};
enum class FontSource {
Embedded,
SystemFallback,
Substituted
};
struct FontMetrics {
double ascent = 0.0;
double descent = 0.0;
double lineGap = 0.0;
double capHeight = 0.0;
};
class Font {
public:
virtual ~Font() = default;
@@ -37,9 +50,62 @@ public:
// Gets the font encoding (returns nullptr if none exists)
virtual const Encoding* getEncoding() const = 0;
// Gets subsetting details (returns nullptr if font is not subsetted)
virtual const FontSubset* getSubsetInfo() const = 0;
// Advanced Layout APIs
virtual FontSource getSourceType() const { return source_type_; }
virtual void setSourceType(FontSource source) { source_type_ = source; }
virtual bool hasGlyph(uint32_t unicode) const {
if (!getFontFace().getFace()) return false;
return FT_Get_Char_Index(getFontFace().getFace(), unicode) != 0;
}
virtual double getAdvanceWidth(uint32_t unicode, double fontSize) const {
if (!getFontFace().getFace()) return 0.0;
std::lock_guard<std::mutex> lock(getFontFace().getMutex());
FT_Face rawFace = getFontFace().getFace();
FT_Set_Pixel_Sizes(rawFace, 0, static_cast<FT_UInt>(fontSize));
FT_UInt glyphIndex = FT_Get_Char_Index(rawFace, unicode);
if (glyphIndex == 0) return 0.0;
if (FT_Load_Glyph(rawFace, glyphIndex, FT_LOAD_DEFAULT) == 0) {
return static_cast<double>(rawFace->glyph->advance.x) / 64.0;
}
return 0.0;
}
virtual FontMetrics getMetrics(double fontSize) const {
if (metrics_cached_ && last_metrics_size_ == fontSize) {
return metrics_cache_;
}
if (getFontFace().getFace()) {
std::lock_guard<std::mutex> lock(getFontFace().getMutex());
FT_Face rawFace = getFontFace().getFace();
FT_Set_Pixel_Sizes(rawFace, 0, static_cast<FT_UInt>(fontSize));
// Convert from 26.6 to double
metrics_cache_.ascent = static_cast<double>(rawFace->size->metrics.ascender) / 64.0;
metrics_cache_.descent = static_cast<double>(rawFace->size->metrics.descender) / 64.0;
metrics_cache_.lineGap = static_cast<double>(rawFace->size->metrics.height - rawFace->size->metrics.ascender + rawFace->size->metrics.descender) / 64.0;
// Heuristic for capHeight: height of 'H'
FT_UInt hIndex = FT_Get_Char_Index(rawFace, 'H');
if (hIndex > 0 && FT_Load_Glyph(rawFace, hIndex, FT_LOAD_DEFAULT) == 0) {
metrics_cache_.capHeight = static_cast<double>(rawFace->glyph->metrics.horiBearingY) / 64.0;
} else {
metrics_cache_.capHeight = metrics_cache_.ascent * 0.7; // Fallback
}
metrics_cached_ = true;
last_metrics_size_ = fontSize;
}
return metrics_cache_;
}
// Translates a raw character code to a Unicode codepoint
virtual uint32_t decodeToUnicode(uint32_t charCode) const = 0;
@@ -119,6 +185,11 @@ protected:
uint32_t last_char_ = 0;
std::vector<double> widths_;
bool has_widths_ = false;
FontSource source_type_ = FontSource::Embedded;
mutable FontMetrics metrics_cache_;
mutable bool metrics_cached_ = false;
mutable double last_metrics_size_ = 0.0;
bool is_vertical_ = false;
uint32_t first_vertical_char_ = 0;
+23 -1
View File
@@ -121,6 +121,8 @@ std::string FontFallback::getFallbackFontPath(const std::string& fontName, bool
stylePattern += "-italic";
}
std::lock_guard<std::mutex> lock(rules_mutex_);
for (const auto& rule : custom_rules_) {
if (stylePattern.find(rule.pattern) != std::string::npos || lowerName.find(rule.pattern) != std::string::npos) {
for (const auto& path : rule.preferredPaths) {
@@ -154,17 +156,37 @@ std::string FontFallback::getFallbackFontPath(const std::string& fontName, bool
}
}
return "C:\\Windows\\Fonts\\arial.ttf";
#elif defined(__APPLE__)
std::vector<std::string> lastResort = {
"/Library/Fonts/Arial.ttf",
"/System/Library/Fonts/Helvetica.ttc",
"/System/Library/Fonts/Supplemental/Arial.ttf"
};
for (const auto& path : lastResort) {
if (std::filesystem::exists(path)) return path;
}
return "/System/Library/Fonts/Helvetica.ttc";
#else
return "";
std::vector<std::string> lastResort = {
"/usr/share/fonts/truetype/liberation/LiberationSans-Regular.ttf",
"/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf",
"/usr/share/fonts/truetype/freefont/FreeSans.ttf"
};
for (const auto& path : lastResort) {
if (std::filesystem::exists(path)) return path;
}
return "/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf";
#endif
}
void FontFallback::registerFallback(const std::string& pattern, const std::string& systemFontPath) {
std::string lowerPattern = toLower(pattern);
std::lock_guard<std::mutex> lock(rules_mutex_);
custom_rules_.insert(custom_rules_.begin(), {lowerPattern, {systemFontPath}});
}
void FontFallback::resetToDefaults() {
std::lock_guard<std::mutex> lock(rules_mutex_);
custom_rules_.clear();
}
@@ -2,6 +2,7 @@
#include <string>
#include <vector>
#include <mutex>
namespace pdfengine::fonts::pdf_fonts {
@@ -32,6 +33,7 @@ private:
};
std::vector<FallbackRule> default_rules_;
std::vector<FallbackRule> custom_rules_;
mutable std::mutex rules_mutex_;
};
} // namespace pdfengine::fonts::pdf_fonts
@@ -1,9 +1,85 @@
#include "fonts/pdf_fonts/font_subset.hpp"
#include <algorithm>
#include <cctype>
#include <ft2build.h>
#include FT_FREETYPE_H
#include <hb.h>
#include <hb-subset.h>
namespace pdfengine::fonts::pdf_fonts {
void FontSubset::populateFromFace(FontSubset& subset, void* ftFace) {
if (!ftFace) return;
FT_Face face = static_cast<FT_Face>(ftFace);
FT_UInt gindex;
FT_ULong charcode = FT_Get_First_Char(face, &gindex);
while (gindex != 0) {
// In embedded subsets, the TrueType cmap typically maps the
// Original GID or CID (charcode) to the new Subset GID (gindex).
subset.addGlyphMapping(gindex, static_cast<uint32_t>(charcode));
charcode = FT_Get_Next_Char(face, charcode, &gindex);
}
}
std::vector<uint8_t> FontSubset::buildSubset(const std::vector<uint8_t>& originalStream, const std::vector<uint32_t>& glyphIdsToKeep) {
std::vector<uint8_t> result;
if (originalStream.empty() || glyphIdsToKeep.empty()) {
return result;
}
hb_blob_t* blob = hb_blob_create(
reinterpret_cast<const char*>(originalStream.data()),
static_cast<unsigned int>(originalStream.size()),
HB_MEMORY_MODE_READONLY,
nullptr,
nullptr
);
hb_face_t* face = hb_face_create(blob, 0);
hb_blob_destroy(blob);
if (!face) {
return result;
}
hb_subset_input_t* input = hb_subset_input_create_or_fail();
if (!input) {
hb_face_destroy(face);
return result;
}
hb_set_t* glyph_set = hb_subset_input_glyph_set(input);
for (uint32_t gid : glyphIdsToKeep) {
hb_set_add(glyph_set, gid);
}
// Always keep GID 0 (the .notdef glyph)
hb_set_add(glyph_set, 0);
// Retain layout tables for complex text shaping
hb_subset_input_set_flags(input, HB_SUBSET_FLAGS_RETAIN_GIDS);
hb_face_t* subset_face = hb_subset_or_fail(face, input);
hb_subset_input_destroy(input);
hb_face_destroy(face);
if (subset_face) {
hb_blob_t* result_blob = hb_face_reference_blob(subset_face);
if (result_blob) {
unsigned int length = 0;
const char* data = hb_blob_get_data(result_blob, &length);
if (data && length > 0) {
result.assign(data, data + length);
}
hb_blob_destroy(result_blob);
}
hb_face_destroy(subset_face);
}
return result;
}
bool FontSubset::hasSubsetPrefix(const std::string& fontName) {
if (fontName.length() < 8) {
return false;
@@ -14,6 +14,12 @@ public:
static std::string getSubsetPrefix(const std::string& fontName);
// Iterates the FT_Face cmap to populate the mapping dictionary
static void populateFromFace(FontSubset& subset, void* ftFace);
// Rebuilds the TTF/CID stream keeping only the specified GIDs using HarfBuzz
static std::vector<uint8_t> buildSubset(const std::vector<uint8_t>& originalStream, const std::vector<uint32_t>& glyphIdsToKeep);
explicit FontSubset(const std::string& fontName);
~FontSubset() = default;
+35 -12
View File
@@ -54,7 +54,13 @@ CIDFont::CIDFont(
CIDFont::~CIDFont() = default;
bool CIDFont::loadFromStream(const std::vector<uint8_t>& streamData) {
return font_face_.loadFromMemory(streamData);
if (!font_face_.loadFromMemory(streamData)) {
return false;
}
if (subset_info_) {
FontSubset::populateFromFace(*subset_info_, font_face_.getFace());
}
return true;
}
bool CIDFont::loadFromFile(const std::string& filePath) {
@@ -126,7 +132,8 @@ uint32_t CIDFont::decodeToUnicode(uint32_t charCode) const {
if (descriptor_) {
std::string fontName = descriptor_->getFontName();
std::string lowerName = fontName;
std::transform(lowerName.begin(), lowerName.end(), lowerName.begin(), ::tolower);
std::transform(lowerName.begin(), lowerName.end(), lowerName.begin(),
[](unsigned char c) { return static_cast<char>(std::tolower(c)); });
std::string registry = "None";
if (lowerName.find("simsun") != std::string::npos ||
@@ -177,16 +184,14 @@ uint32_t CIDFont::decodeToUnicode(uint32_t charCode) const {
gid = subset_info_->mapSubsetToOriginal(gid);
}
FT_Face face = font_face_.getFace();
if (face) {
FT_UInt gindex;
FT_ULong charcode = FT_Get_First_Char(face, &gindex);
while (gindex != 0) {
if (gindex == gid) {
return static_cast<uint32_t>(charcode);
}
charcode = FT_Get_Next_Char(face, charcode, &gindex);
}
if (!is_gid_to_unicode_map_built_) {
buildGidToUnicodeMap();
}
std::lock_guard<std::mutex> lock(gid_to_unicode_mutex_);
auto it = gid_to_unicode_map_.find(gid);
if (it != gid_to_unicode_map_.end()) {
return it->second;
}
return charCode;
@@ -203,4 +208,22 @@ std::string CIDFont::decodeStringToUnicode(const std::vector<uint32_t>& charCode
return result;
}
void CIDFont::buildGidToUnicodeMap() const {
std::lock_guard<std::mutex> lock(gid_to_unicode_mutex_);
if (is_gid_to_unicode_map_built_) {
return;
}
FT_Face face = font_face_.getFace();
if (face) {
FT_UInt gindex;
FT_ULong charcode = FT_Get_First_Char(face, &gindex);
while (gindex != 0) {
gid_to_unicode_map_[gindex] = static_cast<uint32_t>(charcode);
charcode = FT_Get_Next_Char(face, charcode, &gindex);
}
}
is_gid_to_unicode_map_built_ = true;
}
} // namespace pdfengine::fonts::pdf_fonts
@@ -8,6 +8,7 @@
#include <memory>
#include <unordered_map>
#include <string>
#include <mutex>
namespace pdfengine::fonts::pdf_fonts {
class FontSubset;
@@ -60,6 +61,11 @@ private:
bool is_identity_map_ = true;
std::unordered_map<uint32_t, uint32_t> cid_to_gid_map_;
std::unique_ptr<FontSubset> subset_info_;
mutable bool is_gid_to_unicode_map_built_ = false;
mutable std::mutex gid_to_unicode_mutex_;
mutable std::unordered_map<uint32_t, uint32_t> gid_to_unicode_map_;
void buildGidToUnicodeMap() const;
};
} // namespace pdfengine::fonts::pdf_fonts
@@ -50,7 +50,13 @@ TrueTypeFont::TrueTypeFont(
TrueTypeFont::~TrueTypeFont() = default;
bool TrueTypeFont::loadFromStream(const std::vector<uint8_t>& streamData) {
return font_face_.loadFromMemory(streamData);
if (!font_face_.loadFromMemory(streamData)) {
return false;
}
if (subset_info_) {
FontSubset::populateFromFace(*subset_info_, font_face_.getFace());
}
return true;
}
std::string TrueTypeFont::getBaseFont() const {
@@ -94,18 +100,8 @@ uint32_t TrueTypeFont::decodeToUnicode(uint32_t charCode) const {
}
if (subset_info_) {
uint32_t subsetGid = subset_info_->mapSubsetToOriginal(charCode);
FT_Face face = font_face_.getFace();
if (face) {
FT_UInt gindex;
FT_ULong charcode = FT_Get_First_Char(face, &gindex);
while (gindex != 0) {
if (gindex == subsetGid) {
return static_cast<uint32_t>(charcode);
}
charcode = FT_Get_Next_Char(face, charcode, &gindex);
}
if (subset_info_->hasGlyphMapping(charCode)) {
return subset_info_->mapSubsetToOriginal(charCode);
}
}
@@ -50,7 +50,13 @@ Type1Font::Type1Font(
Type1Font::~Type1Font() = default;
bool Type1Font::loadFromStream(const std::vector<uint8_t>& streamData) {
return font_face_.loadFromMemory(streamData);
if (!font_face_.loadFromMemory(streamData)) {
return false;
}
if (subset_info_) {
FontSubset::populateFromFace(*subset_info_, font_face_.getFace());
}
return true;
}
bool Type1Font::loadFromFile(const std::string& filePath) {
@@ -98,18 +104,8 @@ uint32_t Type1Font::decodeToUnicode(uint32_t charCode) const {
}
if (subset_info_) {
uint32_t subsetGid = subset_info_->mapSubsetToOriginal(charCode);
FT_Face face = font_face_.getFace();
if (face) {
FT_UInt gindex;
FT_ULong charcode = FT_Get_First_Char(face, &gindex);
while (gindex != 0) {
if (gindex == subsetGid) {
return static_cast<uint32_t>(charcode);
}
charcode = FT_Get_Next_Char(face, charcode, &gindex);
}
if (subset_info_->hasGlyphMapping(charCode)) {
return subset_info_->mapSubsetToOriginal(charCode);
}
}
+21 -1
View File
@@ -2,6 +2,7 @@
#include <hb.h>
#include <hb-ft.h>
#include <mutex>
namespace pdfengine::fonts {
@@ -21,6 +22,9 @@ std::vector<ShapedGlyph> HbShaper::shapeRun(
return result;
}
// Acquire lock to prevent concurrent mutation of FT_Face's active pixel size
std::lock_guard<std::mutex> lock(font.getMutex());
// Set the pixel size on the FreeType face before shaping.
// This ensures HarfBuzz measures everything using the requested font size context.
if (FT_Set_Pixel_Sizes(ftFace, 0, fontSize)) {
@@ -53,8 +57,23 @@ std::vector<ShapedGlyph> HbShaper::shapeRun(
hb_buffer_set_direction(hbBuffer, HB_DIRECTION_TTB);
}
// Enable professional typography features (Kerning and Ligatures)
hb_feature_t features[2];
// Enable kerning
features[0].tag = HB_TAG('k', 'e', 'r', 'n');
features[0].value = 1;
features[0].start = 0;
features[0].end = static_cast<unsigned int>(-1);
// Enable standard ligatures
features[1].tag = HB_TAG('l', 'i', 'g', 'a');
features[1].value = 1;
features[1].start = 0;
features[1].end = static_cast<unsigned int>(-1);
// Shape the text inside the buffer using the font.
hb_shape(hbFont, hbBuffer, nullptr, 0);
hb_shape(hbFont, hbBuffer, features, 2);
// Retrieve the results.
unsigned int glyphCount = 0;
@@ -72,6 +91,7 @@ std::vector<ShapedGlyph> HbShaper::shapeRun(
sg.advanceY = static_cast<double>(glyphPositions[i].y_advance) / 64.0;
sg.offsetX = static_cast<double>(glyphPositions[i].x_offset) / 64.0;
sg.offsetY = static_cast<double>(glyphPositions[i].y_offset) / 64.0;
sg.clusterIndex = glyphInfos[i].cluster;
result.push_back(sg);
}
}
+1
View File
@@ -13,6 +13,7 @@ struct ShapedGlyph {
double advanceY;
double offsetX;
double offsetY;
uint32_t clusterIndex;
};
class HbShaper {
+493 -22
View File
@@ -1,5 +1,4 @@
#include "parser/pdfium_document.hpp"
#ifdef PDFENGINE_WITH_PDFIUM
#include <fpdfview.h>
#include <fpdf_text.h>
@@ -11,6 +10,9 @@
#include "parser/pdfium_loader.hpp"
#endif
#include "fonts/loader/font_resolver.hpp"
#include "fonts/pdf_fonts/font.hpp"
#include <nlohmann/json.hpp>
#include <spdlog/spdlog.h>
#include <fstream>
@@ -401,8 +403,13 @@ void deduceFontMetadata(pdfengine::FontInfo& f) {
}
// 7. Stable Internal Font Identifier
// For subset fonts, fontName already contains the subset prefix (e.g. "ABCDEF+Arial").
// Use the full fontName directly — it already encodes both the subset tag and
// the base font name, separated by '+'. Concatenating subsetTag + "_" + fontName
// would duplicate the prefix ("ABCDEF_ABCDEF+Arial").
if (f.isSubset && !f.subsetTag.empty()) {
f.internalFontId = f.subsetTag + "_" + f.fontName;
// fontName is "ABCDEF+Arial"; use it as-is for the stable ID.
f.internalFontId = f.fontName;
} else {
f.internalFontId = f.fontName + "_" + f.type + "_" + std::to_string(f.flags);
}
@@ -802,7 +809,7 @@ std::expected<std::vector<GlyphBounds>, EngineError> PdfiumPage::extractTextWith
}
std::string utf8_char = code_point_to_utf8(cp);
if (utf8_char.empty()) {
if (utf8_char.empty() || cp == '\r' || cp == '\n') {
continue;
}
@@ -827,6 +834,254 @@ std::expected<std::vector<GlyphBounds>, EngineError> PdfiumPage::extractTextWith
#endif
}
std::expected<PageModel, EngineError> PdfiumPage::extractDocumentModel() const {
#ifdef PDFENGINE_WITH_PDFIUM
if (!page_) {
return std::unexpected(EngineError::Unknown);
}
ensureTextPageLoaded();
if (!textPage_) {
return std::unexpected(EngineError::Unknown);
}
PageModel model;
model.width = width();
model.height = height();
model.pageIndex = pageIndex_;
int charCount = FPDFText_CountChars(textPage_);
if (charCount <= 0) {
return model;
}
std::unordered_map<std::string, FontInfo> fontMap;
if (auto fontsRes = getFonts()) {
for (const auto& f : *fontsRes) {
fontMap[f.fontName] = f;
}
}
std::vector<Glyph> documentGlyphs;
documentGlyphs.reserve(charCount);
// Pass 1: Extract all glyphs
for (int i = 0; i < charCount; ++i) {
unsigned int codeUnit = FPDFText_GetUnicode(textPage_, i);
unsigned int cp = codeUnit;
// Handle surrogate pairs
if (codeUnit >= 0xD800 && codeUnit <= 0xDBFF && i + 1 < charCount) {
unsigned int nextUnit = FPDFText_GetUnicode(textPage_, i + 1);
if (nextUnit >= 0xDC00 && nextUnit <= 0xDFFF) {
cp = 0x10000 + ((codeUnit - 0xD800) << 10) + (nextUnit - 0xDC00);
}
}
std::string utf8_char = code_point_to_utf8(cp);
if (utf8_char.empty() || cp == '\r' || cp == '\n') {
if (cp > 0xFFFF) ++i; // Skip low surrogate
continue;
}
Glyph g;
g.text = std::move(utf8_char);
g.unicode = cp;
// Geometry
double left, right, bottom, top;
FPDFText_GetCharBox(textPage_, i, &left, &right, &bottom, &top);
g.bboxX = (std::min)(left, right);
g.bboxY = (std::min)(bottom, top);
g.bboxW = std::abs(right - left);
g.bboxH = std::abs(top - bottom);
FPDFText_GetCharOrigin(textPage_, i, &g.originX, &g.originY);
g.angle = FPDFText_GetCharAngle(textPage_, i);
// Style
g.fontSize = FPDFText_GetFontSize(textPage_, i);
int flags = 0;
unsigned long len = FPDFText_GetFontInfo(textPage_, i, nullptr, 0, &flags);
if (len > 0) {
std::vector<char> buf(len);
if (FPDFText_GetFontInfo(textPage_, i, buf.data(), len, &flags) > 0) {
g.fontName = std::string(buf.data());
}
}
g.flags = flags;
documentGlyphs.push_back(g);
if (cp > 0xFFFF) ++i; // Skip low surrogate
}
// Phase 5E: Line Builder
std::sort(documentGlyphs.begin(), documentGlyphs.end(), [](const Glyph& a, const Glyph& b) {
if (std::abs(a.originY - b.originY) > 1.0) {
return a.originY > b.originY; // Top to bottom
}
return a.originX < b.originX; // Left to right
});
std::vector<TextLine> lines;
if (!documentGlyphs.empty()) {
TextLine currentLine;
currentLine.angle = documentGlyphs[0].angle;
double currentOriginY = documentGlyphs[0].originY;
for (const auto& g : documentGlyphs) {
if (currentLine.glyphs.empty()) {
currentLine.glyphs.push_back(g);
continue;
}
if (std::abs(g.angle - currentLine.angle) < 0.1 &&
std::abs(g.originY - currentOriginY) < 1.0) {
currentLine.glyphs.push_back(g);
} else {
lines.push_back(std::move(currentLine));
currentLine = TextLine();
currentLine.angle = g.angle;
currentOriginY = g.originY;
currentLine.glyphs.push_back(g);
}
}
if (!currentLine.glyphs.empty()) {
lines.push_back(std::move(currentLine));
}
}
for (auto& line : lines) {
std::sort(line.glyphs.begin(), line.glyphs.end(), [](const Glyph& a, const Glyph& b) {
return a.originX < b.originX;
});
std::vector<double> gaps;
for (size_t i = 1; i < line.glyphs.size(); ++i) {
double gap = line.glyphs[i].bboxX - (line.glyphs[i-1].bboxX + line.glyphs[i-1].bboxW);
if (gap > 0) {
gaps.push_back(gap);
}
}
double medianGap = 0.0;
if (!gaps.empty()) {
std::sort(gaps.begin(), gaps.end());
medianGap = gaps[gaps.size() / 2];
}
// Phase 5F: Run Builder
TextRun currentRun;
if (!line.glyphs.empty()) {
const Glyph* firstG = &line.glyphs[0];
currentRun.fontName = firstG->fontName;
currentRun.fontSize = firstG->fontSize;
currentRun.flags = firstG->flags;
if (auto it = fontMap.find(currentRun.fontName); it != fontMap.end()) {
currentRun.internalFontId = it->second.internalFontId;
currentRun.isEmbedded = it->second.isEmbedded;
currentRun.type = it->second.type;
}
currentRun.glyphs.push_back(*firstG);
currentRun.text += firstG->text;
for (size_t i = 1; i < line.glyphs.size(); ++i) {
const auto& prevG = line.glyphs[i-1];
const auto& currG = line.glyphs[i];
double gap = currG.bboxX - (prevG.bboxX + prevG.bboxW);
double spaceThreshold = (std::max)(currG.fontSize * 0.25, medianGap * 2.0);
bool addSpace = gap > spaceThreshold && prevG.text != " " && currG.text != " ";
bool breakRun = currG.fontName != currentRun.fontName ||
std::abs(currG.fontSize - currentRun.fontSize) > 0.1 ||
currG.flags != currentRun.flags;
if (addSpace) {
Glyph spaceGlyph;
spaceGlyph.text = " ";
spaceGlyph.unicode = ' ';
spaceGlyph.fontSize = currG.fontSize;
spaceGlyph.fontName = currG.fontName;
spaceGlyph.flags = currG.flags;
spaceGlyph.originX = prevG.bboxX + prevG.bboxW;
spaceGlyph.originY = currG.originY;
spaceGlyph.angle = currG.angle;
spaceGlyph.bboxX = spaceGlyph.originX;
spaceGlyph.bboxY = currG.bboxY;
spaceGlyph.bboxW = gap;
spaceGlyph.bboxH = currG.bboxH;
if (breakRun) {
line.runs.push_back(std::move(currentRun));
currentRun = TextRun();
currentRun.fontName = currG.fontName;
currentRun.fontSize = currG.fontSize;
currentRun.flags = currG.flags;
if (auto it = fontMap.find(currentRun.fontName); it != fontMap.end()) {
currentRun.internalFontId = it->second.internalFontId;
currentRun.isEmbedded = it->second.isEmbedded;
currentRun.type = it->second.type;
}
}
currentRun.glyphs.push_back(spaceGlyph);
currentRun.text += spaceGlyph.text;
} else if (breakRun) {
line.runs.push_back(std::move(currentRun));
currentRun = TextRun();
currentRun.fontName = currG.fontName;
currentRun.fontSize = currG.fontSize;
currentRun.flags = currG.flags;
if (auto it = fontMap.find(currentRun.fontName); it != fontMap.end()) {
currentRun.internalFontId = it->second.internalFontId;
currentRun.isEmbedded = it->second.isEmbedded;
currentRun.type = it->second.type;
}
}
currentRun.glyphs.push_back(currG);
currentRun.text += currG.text;
}
if (!currentRun.glyphs.empty()) {
line.runs.push_back(std::move(currentRun));
}
}
}
// Phase 5G: Paragraph Builder
std::vector<Paragraph> paragraphs;
if (!lines.empty()) {
Paragraph currentPara;
currentPara.lines.push_back(std::move(lines[0]));
for (size_t i = 1; i < lines.size(); ++i) {
auto& prevLine = currentPara.lines.back();
auto& currLine = lines[i];
double prevY = prevLine.glyphs.empty() ? 0 : prevLine.glyphs[0].originY;
double currY = currLine.glyphs.empty() ? 0 : currLine.glyphs[0].originY;
double fontSize = currLine.runs.empty() ? 12.0 : currLine.runs[0].fontSize;
double vGap = std::abs(prevY - currY);
if (vGap > fontSize * 1.5) {
paragraphs.push_back(std::move(currentPara));
currentPara = Paragraph();
}
currentPara.lines.push_back(std::move(currLine));
}
if (!currentPara.lines.empty()) {
paragraphs.push_back(std::move(currentPara));
}
}
model.paragraphs = std::move(paragraphs);
return model;
#else
return std::unexpected(EngineError::Unknown);
#endif
}
std::expected<std::vector<std::string>, EngineError> PdfiumPage::extractAnnotationsText() const {
#ifdef PDFENGINE_WITH_PDFIUM
if (!page_) {
@@ -884,6 +1139,60 @@ Point2D PdfiumPage::deviceToPage(const DevicePoint& devicePoint, int deviceWidth
#endif
}
std::expected<double, EngineError> PdfiumPage::getGlyphWidth(const std::string& fontName, uint32_t charcode, double fontSize) const {
#ifdef PDFENGINE_WITH_PDFIUM
if (!page_) {
return std::unexpected(EngineError::Unknown);
}
FPDF_FONT font = nullptr;
{
std::lock_guard<std::mutex> lock(textMutex_);
auto cached = fontHandleCache_.find(fontName);
if (cached != fontHandleCache_.end()) {
font = cached->second;
} else {
int objectCount = FPDFPage_CountObjects(page_);
for (int i = 0; i < objectCount; ++i) {
FPDF_PAGEOBJECT obj = FPDFPage_GetObject(page_, i);
if (!obj || FPDFPageObj_GetType(obj) != FPDF_PAGEOBJ_TEXT) continue;
FPDF_FONT pageFont = FPDFTextObj_GetFont(obj);
if (!pageFont) continue;
size_t nameLen = FPDFFont_GetBaseFontName(pageFont, nullptr, 0);
if (nameLen > 0) {
std::vector<char> nameBuf(nameLen);
if (FPDFFont_GetBaseFontName(pageFont, nameBuf.data(), nameLen) > 0) {
std::string currentName(nameBuf.data());
if (currentName == fontName) {
font = pageFont;
fontHandleCache_[fontName] = font;
break;
}
}
}
}
}
}
if (!font) {
return std::unexpected(EngineError::Unknown);
}
float width = 0.0f;
if (!FPDFFont_GetGlyphWidth(font, charcode, static_cast<float>(fontSize), &width)) {
return std::unexpected(EngineError::Unknown);
}
return static_cast<double>(width);
#else
(void)fontName; (void)charcode; (void)fontSize;
return std::unexpected(EngineError::Unknown);
#endif
}
void PdfiumPage::ensureTextPageLoaded() const {
#ifdef PDFENGINE_WITH_PDFIUM
std::lock_guard<std::mutex> lock(textMutex_);
@@ -962,11 +1271,26 @@ std::expected<std::shared_ptr<PdfPage>, EngineError> PdfiumDocument::getPage(int
if (pageIndex < 0 || pageIndex >= pageCount()) {
return std::unexpected(EngineError::PageOutOfBounds);
}
{
std::lock_guard<std::mutex> lock(pageCacheMutex_);
auto it = pageCache_.find(pageIndex);
if (it != pageCache_.end()) {
return it->second;
}
}
FPDF_PAGE page = FPDF_LoadPage(doc_, pageIndex);
if (!page) {
return std::unexpected(EngineError::Unknown);
}
return std::make_shared<PdfiumPage>(page, pageIndex);
auto pageObj = std::make_shared<PdfiumPage>(page, pageIndex);
{
std::lock_guard<std::mutex> lock(pageCacheMutex_);
pageCache_[pageIndex] = pageObj;
}
return pageObj;
#else
(void)pageIndex;
return std::unexpected(EngineError::Unknown);
@@ -1272,7 +1596,7 @@ std::expected<void, EngineError> PdfiumDocument::applyEdits(const std::string& e
return std::unexpected(EngineError::Unknown);
}
invalidateFontCache();
invalidateCaches();
return {};
#else
(void)editsJson;
@@ -1478,8 +1802,10 @@ std::expected<std::vector<FontInfo>, EngineError> PdfiumPage::getFonts() const {
// deduceFontMetadata() (SystemFallback or Substituted).
// --- Recalculate stable identifier with corrected data ---
// fontName already includes the subset prefix (e.g. "ABCDEF+Arial");
// using it directly avoids the duplicate-prefix bug.
if (f.isSubset && !f.subsetTag.empty()) {
f.internalFontId = f.subsetTag + "_" + f.fontName;
f.internalFontId = f.fontName;
} else {
f.internalFontId = f.fontName + "_" + f.type + "_" + std::to_string(f.flags);
}
@@ -1543,12 +1869,6 @@ std::expected<std::vector<FontInfo>, EngineError> PdfiumDocument::getFonts(int s
return std::vector<FontInfo>();
}
// Security Safeguard: cap maximum scan range to 1000 pages to prevent memory/CPU exhaustion
int scanCount = endPage - startPage + 1;
if (scanCount > 1000) {
spdlog::warn("Requested scan range ({} pages) exceeds limit. Capping scan to 1000 pages.", scanCount);
endPage = startPage + 999;
}
// Return full document-level cache if available and full range is requested
if (startPage == 0 && endPage == total - 1 && hasCachedFonts_) {
@@ -1557,15 +1877,13 @@ std::expected<std::vector<FontInfo>, EngineError> PdfiumDocument::getFonts(int s
std::vector<FontInfo> aggregated;
for (int i = startPage; i <= endPage; ++i) {
FPDF_PAGE page = FPDF_LoadPage(doc_, i);
if (!page) {
auto pageRes = const_cast<PdfiumDocument*>(this)->getPage(i);
if (!pageRes) {
spdlog::error("Failed to load page index {} for font diagnostics", i);
continue;
}
// Stack-allocated wrapper ensures FPDF handles are closed properly upon destruction
PdfiumPage tempPage(page, i);
auto pageFontsRes = tempPage.getFonts();
auto pageFontsRes = pageRes.value()->getFonts();
if (pageFontsRes) {
for (const auto& f : *pageFontsRes) {
auto it = std::find_if(aggregated.begin(), aggregated.end(), [&](const FontInfo& existing) {
@@ -1606,11 +1924,164 @@ std::expected<std::vector<FontInfo>, EngineError> PdfiumDocument::getFonts(int s
#endif
}
void PdfiumDocument::invalidateFontCache() {
std::lock_guard<std::mutex> lock(fontsMutex_);
cachedFonts_.clear();
hasCachedFonts_ = false;
spdlog::info("Document font cache has been invalidated.");
std::expected<std::vector<uint8_t>, EngineError> PdfiumDocument::getFontData(const std::string& internalFontId) const {
#ifdef PDFENGINE_WITH_PDFIUM
ensure_pdfium_initialized();
if (!doc_) {
return std::unexpected(EngineError::Unknown);
}
// Reconstruct the exact expected font name from the internalFontId.
// Non-subset format: {fontName}_{type}_{flags}
std::string expectedFontName = internalFontId;
size_t lastUnderscore = expectedFontName.rfind('_');
if (lastUnderscore != std::string::npos && lastUnderscore > 0) {
size_t secondLastUnderscore = expectedFontName.rfind('_', lastUnderscore - 1);
if (secondLastUnderscore != std::string::npos) {
std::string typePart = expectedFontName.substr(secondLastUnderscore + 1, lastUnderscore - secondLastUnderscore - 1);
if (typePart == "TrueType" || typePart == "Type1" || typePart == "CIDFontType0" || typePart == "CIDFontType2") {
expectedFontName = expectedFontName.substr(0, secondLastUnderscore);
}
}
}
int numPages = FPDF_GetPageCount(doc_);
// Resume scanning from where we left off
int startPage = 0;
{
std::lock_guard<std::mutex> lock(fontsMutex_);
// Return immediately if already cached
if (fontDataCache_.count(expectedFontName)) {
const auto& cachedBuf = fontDataCache_[expectedFontName];
if (!cachedBuf.empty()) {
return cachedBuf;
} else {
return std::unexpected(EngineError::FileNotFound);
}
}
startPage = fontDataScannedPages_;
}
for (int i = startPage; i < numPages; ++i) {
FPDF_PAGE page = FPDF_LoadPage(doc_, i);
if (!page) {
std::lock_guard<std::mutex> lock(fontsMutex_);
fontDataScannedPages_ = i + 1;
continue;
}
int objectCount = FPDFPage_CountObjects(page);
for (int j = 0; j < objectCount; ++j) {
FPDF_PAGEOBJECT obj = FPDFPage_GetObject(page, j);
if (!obj || FPDFPageObj_GetType(obj) != FPDF_PAGEOBJ_TEXT) continue;
FPDF_FONT font = FPDFTextObj_GetFont(obj);
if (!font) continue;
unsigned long nameLen = FPDFFont_GetBaseFontName(font, nullptr, 0);
if (nameLen > 0) {
std::vector<char> nameBuf(nameLen);
if (FPDFFont_GetBaseFontName(font, nameBuf.data(), nameLen) > 0) {
std::string fontName(nameBuf.data());
std::lock_guard<std::mutex> lock(fontsMutex_);
// Always cache every font we encounter to avoid rescanning
if (fontDataCache_.find(fontName) == fontDataCache_.end()) {
size_t buflen = 0;
FPDFFont_GetFontData(font, nullptr, 0, &buflen);
if (buflen > 0) {
std::vector<uint8_t> buffer(buflen);
size_t actual_len = 0;
if (FPDFFont_GetFontData(font, buffer.data(), buflen, &actual_len)) {
fontDataCache_[fontName] = buffer;
} else {
fontDataCache_[fontName] = std::vector<uint8_t>();
}
} else {
fontDataCache_[fontName] = std::vector<uint8_t>();
}
}
// Use exact match instead of prefix match to avoid false positives
if (fontName == expectedFontName) {
const auto& cachedBuf = fontDataCache_[fontName];
if (!cachedBuf.empty()) {
FPDF_ClosePage(page);
fontDataScannedPages_ = i; // can resume from the same page later
return cachedBuf;
}
}
}
}
}
FPDF_ClosePage(page);
std::lock_guard<std::mutex> lock(fontsMutex_);
fontDataScannedPages_ = i + 1;
}
// We finished scanning all pages and still didn't find it (or extraction failed)
{
std::lock_guard<std::mutex> lock(fontsMutex_);
// Mark as failed so we don't try to rescan for it
if (fontDataCache_.find(expectedFontName) == fontDataCache_.end()) {
fontDataCache_[expectedFontName] = std::vector<uint8_t>();
}
}
return std::unexpected(EngineError::FileNotFound);
#else
(void)internalFontId;
return std::unexpected(EngineError::Unknown);
#endif
}
std::expected<std::shared_ptr<fonts::pdf_fonts::Font>, std::string> PdfiumDocument::getResolvedFont(const FontInfo& fontInfo) {
#ifdef PDFENGINE_WITH_PDFIUM
std::lock_guard<std::mutex> lock(resolvedFontsMutex_);
// Check Cache
if (resolvedFontsCache_.count(fontInfo.internalFontId)) {
return resolvedFontsCache_[fontInfo.internalFontId];
}
if (!fontResolver_) {
// We pass a shared_ptr to 'this'
fontResolver_ = std::make_unique<::pdfengine::fonts::loader::FontResolver>(shared_from_this());
}
// Resolve font
auto result = fontResolver_->resolveFont(fontInfo);
if (!result) {
return std::unexpected(result.error());
}
// Cache and return
std::shared_ptr<fonts::pdf_fonts::Font> sharedFont(std::move(result.value()));
resolvedFontsCache_[fontInfo.internalFontId] = sharedFont;
return sharedFont;
#else
return std::unexpected(std::string("EngineError::Unknown"));
#endif
}
void PdfiumDocument::invalidateCaches() {
{
std::lock_guard<std::mutex> lock(fontsMutex_);
cachedFonts_.clear();
hasCachedFonts_ = false;
}
{
std::lock_guard<std::mutex> lock(pageCacheMutex_);
pageCache_.clear();
}
{
std::lock_guard<std::mutex> lock(resolvedFontsMutex_);
resolvedFontsCache_.clear();
}
spdlog::info("Document caches have been invalidated.");
}
}
+25 -2
View File
@@ -11,6 +11,8 @@
#include <string>
#include <vector>
#include <mutex>
#include <unordered_map>
namespace pdfengine::fonts::loader { class FontResolver; }
namespace pdfengine::parser {
@@ -40,9 +42,12 @@ public:
std::expected<PageImage, EngineError> render(int dpi = 96) const override;
std::expected<std::string, EngineError> extractText() const override;
std::expected<std::vector<GlyphBounds>, EngineError> extractTextWithBounds() const override;
std::expected<PageModel, EngineError> extractDocumentModel() const override;
std::expected<std::vector<FontInfo>, EngineError> getFonts() const override;
std::expected<std::vector<std::string>, EngineError> extractAnnotationsText() const override;
std::expected<double, EngineError> getGlyphWidth(const std::string& fontName, uint32_t charcode, double fontSize) const override;
DevicePoint pageToDevice(const Point2D& pagePoint, int deviceWidth, int deviceHeight, int rotate = 0) const noexcept override;
Point2D deviceToPage(const DevicePoint& devicePoint, int deviceWidth, int deviceHeight, int rotate = 0) const noexcept override;
@@ -51,11 +56,14 @@ private:
mutable NativeTextHandle textPage_ = nullptr;
int pageIndex_ = 0;
mutable std::mutex textMutex_;
#ifdef PDFENGINE_WITH_PDFIUM
mutable std::unordered_map<std::string, FPDF_FONT> fontHandleCache_;
#endif
void ensureTextPageLoaded() const;
};
class PdfiumDocument : public PdfDocument {
class PdfiumDocument : public PdfDocument, public std::enable_shared_from_this<PdfiumDocument> {
public:
explicit PdfiumDocument(NativeDocHandle docHandle);
PdfiumDocument(NativeDocHandle docHandle, std::vector<uint8_t> memoryBuffer);
@@ -71,7 +79,10 @@ public:
std::expected<std::shared_ptr<PdfPage>, EngineError> getPage(int pageIndex) override;
std::expected<std::vector<FontInfo>, EngineError> getFonts(int startPage = 0, int endPage = -1) const override;
void invalidateFontCache();
std::expected<std::vector<uint8_t>, EngineError> getFontData(const std::string& internalFontId) const override;
std::expected<std::shared_ptr<fonts::pdf_fonts::Font>, std::string> getResolvedFont(const FontInfo& fontInfo) override;
void invalidateCaches();
std::expected<void, EngineError> applyEdits(const std::string& editsJson) override;
std::expected<std::vector<uint8_t>, EngineError> saveIncremental() const override;
@@ -83,6 +94,18 @@ private:
mutable std::vector<FontInfo> cachedFonts_;
mutable bool hasCachedFonts_ = false;
mutable std::mutex fontsMutex_;
// Cache to prevent O(N*M) full-document scans for font data extraction
mutable std::unordered_map<std::string, std::vector<uint8_t>> fontDataCache_;
mutable int fontDataScannedPages_ = 0;
mutable std::unordered_map<int, std::shared_ptr<PdfPage>> pageCache_;
mutable std::mutex pageCacheMutex_;
// Font Engine Bridge
std::unique_ptr<pdfengine::fonts::loader::FontResolver> fontResolver_;
std::unordered_map<std::string, std::shared_ptr<fonts::pdf_fonts::Font>> resolvedFontsCache_;
std::mutex resolvedFontsMutex_;
};
// Exposed for testing
+1 -1
View File
@@ -824,7 +824,7 @@ TEST(FontDiagnosticsTest, DeepIntrospectionAndFontSizeVerification) {
}
EXPECT_EQ(f.sourceType, "Embedded");
EXPECT_TRUE(f.isEmbedded);
EXPECT_EQ(f.internalFontId, f.subsetTag + "_" + f.fontName);
EXPECT_EQ(f.internalFontId, f.fontName);
} else {
EXPECT_TRUE(f.subsetTag.empty());
EXPECT_EQ(f.internalFontId, f.fontName + "_" + f.type + "_" + std::to_string(f.flags));
+55 -15
View File
@@ -59,7 +59,6 @@
"integrity": "sha512-CGOfOJqWjg2qW/Mb6zNsDm+u5vFQ8DxXfbM09z69p5Z6+mE1ikP2jUXw+j42Pf1XTYED2Rni5f95npYeuwMDQA==",
"dev": true,
"license": "MIT",
"peer": true,
"dependencies": {
"@babel/code-frame": "^7.29.0",
"@babel/generator": "^7.29.0",
@@ -269,10 +268,31 @@
"node": ">=6.9.0"
}
},
"node_modules/@emnapi/core": {
"version": "1.11.0",
"resolved": "https://registry.npmjs.org/@emnapi/core/-/core-1.11.0.tgz",
"integrity": "sha512-l9Oo58x0HOP5znGzVhYW9U3e5wVuA4LAZU2AGezTmkhO1CgQRFDhDg4nneHsu/t3WniXg9QrG2nIXL/ZS8ln8Q==",
"license": "MIT",
"optional": true,
"dependencies": {
"@emnapi/wasi-threads": "1.2.2",
"tslib": "^2.4.0"
}
},
"node_modules/@emnapi/runtime": {
"version": "1.11.0",
"resolved": "https://registry.npmjs.org/@emnapi/runtime/-/runtime-1.11.0.tgz",
"integrity": "sha512-55coeOFKHv1ywEcUXJtWU5f+Jr/W5tZDvZig8DLKSwUN1JpROQ4rk/SNOQiFWmaR/VKF4zuFyW1B8JduOSv6Pg==",
"license": "MIT",
"optional": true,
"dependencies": {
"tslib": "^2.4.0"
}
},
"node_modules/@emnapi/wasi-threads": {
"version": "1.2.1",
"resolved": "https://registry.npmjs.org/@emnapi/wasi-threads/-/wasi-threads-1.2.1.tgz",
"integrity": "sha512-uTII7OYF+/Mes/MrcIOYp5yOtSMLBWSIoLPpcgwipoiKbli6k322tcoFsxoIIxPDqW01SQGAgko4EzZi2BNv2w==",
"version": "1.2.2",
"resolved": "https://registry.npmjs.org/@emnapi/wasi-threads/-/wasi-threads-1.2.2.tgz",
"integrity": "sha512-c95qOXkHdydNKhscBTebqEC1CVAZpyqOfVfBzQ1qgzyl3gfeldUjIggDbIZgDKsHLgnsM+igH7TJ/eAasaVuMA==",
"license": "MIT",
"optional": true,
"dependencies": {
@@ -755,6 +775,37 @@
"node": "^20.19.0 || >=22.12.0"
}
},
"node_modules/@rolldown/binding-wasm32-wasi/node_modules/@emnapi/core": {
"version": "1.10.0",
"resolved": "https://registry.npmjs.org/@emnapi/core/-/core-1.10.0.tgz",
"integrity": "sha512-yq6OkJ4p82CAfPl0u9mQebQHKPJkY7WrIuk205cTYnYe+k2Z8YBh11FrbRG/H6ihirqcacOgl2BIO8oyMQLeXw==",
"license": "MIT",
"optional": true,
"dependencies": {
"@emnapi/wasi-threads": "1.2.1",
"tslib": "^2.4.0"
}
},
"node_modules/@rolldown/binding-wasm32-wasi/node_modules/@emnapi/runtime": {
"version": "1.10.0",
"resolved": "https://registry.npmjs.org/@emnapi/runtime/-/runtime-1.10.0.tgz",
"integrity": "sha512-ewvYlk86xUoGI0zQRNq/mC+16R1QeDlKQy21Ki3oSYXNgLb45GV1P6A0M+/s6nyCuNDqe5VpaY84BzXGwVbwFA==",
"license": "MIT",
"optional": true,
"dependencies": {
"tslib": "^2.4.0"
}
},
"node_modules/@rolldown/binding-wasm32-wasi/node_modules/@emnapi/wasi-threads": {
"version": "1.2.1",
"resolved": "https://registry.npmjs.org/@emnapi/wasi-threads/-/wasi-threads-1.2.1.tgz",
"integrity": "sha512-uTII7OYF+/Mes/MrcIOYp5yOtSMLBWSIoLPpcgwipoiKbli6k322tcoFsxoIIxPDqW01SQGAgko4EzZi2BNv2w==",
"license": "MIT",
"optional": true,
"dependencies": {
"tslib": "^2.4.0"
}
},
"node_modules/@rolldown/binding-win32-arm64-msvc": {
"version": "1.0.1",
"resolved": "https://registry.npmjs.org/@rolldown/binding-win32-arm64-msvc/-/binding-win32-arm64-msvc-1.0.1.tgz",
@@ -1087,7 +1138,6 @@
"integrity": "sha512-GUUEShf+PBCGW2KaXwcIt3Yk+e3pkKwWKb9GSyM9WQVE+ep2jzmHdGsHzu4wgcZy5fN9FBdVzjpBQsYlpfpgLA==",
"devOptional": true,
"license": "MIT",
"peer": true,
"dependencies": {
"undici-types": "~7.16.0"
}
@@ -1098,7 +1148,6 @@
"integrity": "sha512-ilcTH/UniCkMdtexkoCN0bI7pMcJDvmQFPvuPvmEaYA/NSfFTAgdUSLAoVjaRJm7+6PvcM+q1zYOwS4wTYMF9w==",
"dev": true,
"license": "MIT",
"peer": true,
"dependencies": {
"csstype": "^3.2.2"
}
@@ -1158,7 +1207,6 @@
"integrity": "sha512-HPwA+hVkfcriajbNvTmZv4VRauibay+cWArYUYq7u7W7PmGShMxbPxLvrwDme55a6d5alG3nrYfhyJ/G28XlLg==",
"dev": true,
"license": "MIT",
"peer": true,
"dependencies": {
"@typescript-eslint/scope-manager": "8.59.3",
"@typescript-eslint/types": "8.59.3",
@@ -1389,7 +1437,6 @@
"integrity": "sha512-UVJyE9MttOsBQIDKw1skb9nAwQuR5wuGD3+82K6JgJlm/Y+KI92oNsMNGZCYdDsVtRHSak0pcV5Dno5+4jh9sw==",
"dev": true,
"license": "MIT",
"peer": true,
"bin": {
"acorn": "bin/acorn"
},
@@ -1480,7 +1527,6 @@
}
],
"license": "MIT",
"peer": true,
"dependencies": {
"baseline-browser-mapping": "^2.10.12",
"caniuse-lite": "^1.0.30001782",
@@ -1628,7 +1674,6 @@
"integrity": "sha512-loXy6bWOoP3EP6JA7jo6p5jMpBJmHmsNZM5SFRHLdh1MGOPurMnNBj4ZlAbaqUAaQWbCr7jHV4P7gzAyryZWkQ==",
"dev": true,
"license": "MIT",
"peer": true,
"dependencies": {
"@eslint-community/eslint-utils": "^4.8.0",
"@eslint-community/regexpp": "^4.12.2",
@@ -2524,7 +2569,6 @@
"resolved": "https://registry.npmjs.org/picomatch/-/picomatch-4.0.4.tgz",
"integrity": "sha512-QP88BAKvMam/3NxH6vj2o21R6MjxZUAd6nlwAS/pnGvN9IVLocLHxGYIzFhg6fUQ+5th6P4dv4eW9jX3DSIj7A==",
"license": "MIT",
"peer": true,
"engines": {
"node": ">=12"
},
@@ -2585,7 +2629,6 @@
"resolved": "https://registry.npmjs.org/react/-/react-19.2.6.tgz",
"integrity": "sha512-sfWGGfavi0xr8Pg0sVsyHMAOziVYKgPLNrS7ig+ivMNb3wbCBw3KxtflsGBAwD3gYQlE/AEZsTLgToRrSCjb0Q==",
"license": "MIT",
"peer": true,
"engines": {
"node": ">=0.10.0"
}
@@ -2757,7 +2800,6 @@
"integrity": "sha512-y2TvuxSZPDyQakkFRPZHKFm+KKVqIisdg9/CZwm9ftvKXLP8NRWj38/ODjNbr43SsoXqNuAisEf1GdCxqWcdBw==",
"dev": true,
"license": "Apache-2.0",
"peer": true,
"bin": {
"tsc": "bin/tsc",
"tsserver": "bin/tsserver"
@@ -2843,7 +2885,6 @@
"resolved": "https://registry.npmjs.org/vite/-/vite-8.0.13.tgz",
"integrity": "sha512-MFtjBYgzmSxmgA4RAfjIyXWpGe1oALnjgUTzzV7QLx/TKxCzjtMH6Fd9/eVK+5Fg1qNoz5VAwsmMs/NofrmJvw==",
"license": "MIT",
"peer": true,
"dependencies": {
"lightningcss": "^1.32.0",
"picomatch": "^4.0.4",
@@ -2968,7 +3009,6 @@
"integrity": "sha512-ytENFjIJFl2UwYglde2jchW2Hwm4GJFLDiSXWdTrJQBIN9Fcyp7n4DhxJEiWNAJMV1/BqWfW/kkg71UDcHJyTQ==",
"dev": true,
"license": "MIT",
"peer": true,
"funding": {
"url": "https://github.com/sponsors/colinhacks"
}
+47 -19
View File
@@ -5,20 +5,23 @@
:root {
--sans: 'Outfit', system-ui, -apple-system, sans-serif;
--mono: 'JetBrains Mono', monospace;
/* Color Palette - Premium Royal/Midnight */
--bg-main: #020617; /* Slate 950 */
--bg-card: #0f172a; /* Slate 900 */
--bg-sidebar: #090d16; /* Deep Midnight */
--bg-main: #020617;
/* Slate 950 */
--bg-card: #0f172a;
/* Slate 900 */
--bg-sidebar: #090d16;
/* Deep Midnight */
--bg-accent-indigo: #4f46e5;
--bg-accent-indigo-hover: #4338ca;
--border-main: #1e293b;
--border-glow: rgba(99, 102, 241, 0.25);
--text-main: #f8fafc;
--text-muted: #94a3b8;
--text-dim: #64748b;
/* Status Colors */
--color-success: #10b981;
--color-success-bg: rgba(16, 185, 129, 0.1);
@@ -175,9 +178,18 @@ body {
border-radius: 9999px;
}
.status-checking .health-dot { background: var(--color-warning); animation: pulse 1.5s infinite; }
.status-healthy .health-dot { background: var(--color-success); }
.status-unhealthy .health-dot { background: var(--color-error); }
.status-checking .health-dot {
background: var(--color-warning);
animation: pulse 1.5s infinite;
}
.status-healthy .health-dot {
background: var(--color-success);
}
.status-unhealthy .health-dot {
background: var(--color-error);
}
.wasm-badge {
display: flex;
@@ -422,7 +434,8 @@ body {
padding: 18px;
}
.doc-list-container, .annotations-list {
.doc-list-container,
.annotations-list {
display: flex;
flex-direction: column;
gap: 12px;
@@ -605,7 +618,8 @@ body {
flex: 1;
height: 100%;
overflow: auto;
background: #0b0f19; /* Slightly darker midnight */
background: #0b0f19;
/* Slightly darker midnight */
display: flex;
justify-content: center;
align-items: start;
@@ -698,11 +712,13 @@ body {
}
.highlight-box.type-highlight {
background: #fde047; /* Yellow 300 */
background: #fde047;
/* Yellow 300 */
}
.highlight-box.type-comment {
background: #fecdd3; /* Rose 200 */
background: #fecdd3;
/* Rose 200 */
border-bottom: 2px solid #f43f5e;
}
@@ -732,7 +748,7 @@ body {
font-weight: 700;
padding: 4px 10px;
border-radius: 6px;
box-shadow: 0 4px 12px rgba(0,0,0,0.25);
box-shadow: 0 4px 12px rgba(0, 0, 0, 0.25);
letter-spacing: 0.5px;
}
@@ -751,11 +767,23 @@ body {
/* Animations */
@keyframes spin {
0% { transform: rotate(0deg); }
100% { transform: rotate(360deg); }
0% {
transform: rotate(0deg);
}
100% {
transform: rotate(360deg);
}
}
@keyframes pulse {
0%, 100% { opacity: 1; }
50% { opacity: 0.5; }
}
0%,
100% {
opacity: 1;
}
50% {
opacity: 0.5;
}
}
+14 -8
View File
@@ -118,7 +118,7 @@ export interface PageRotationData {
rotation: 0 | 90 | 180 | 270;
}
export interface PageDeletionData {}
export interface PageDeletionData { }
export interface PageReorderData {
destPageIndex: number;
@@ -191,7 +191,7 @@ class GatewayService {
method: 'POST',
body: formData,
});
if (response.status === 501) {
// Simulate upload for Phase 0 scaffolding
return new Promise((resolve) => {
@@ -210,7 +210,7 @@ class GatewayService {
}, 1000);
});
}
if (!response.ok) throw new Error(`Upload failed: ${response.statusText}`);
return response.json();
}
@@ -252,12 +252,12 @@ class GatewayService {
}).toString();
const url = `${this.baseUrl}/render/${params.documentId}?${query}`;
const response = await fetch(url);
if (response.status === 501) {
return this.generateMockPage(params.pageIndex);
}
if (!response.ok) throw new Error(`Page render failed: ${response.statusText}`);
const blob = await response.blob();
return URL.createObjectURL(blob);
@@ -266,6 +266,12 @@ class GatewayService {
}
}
async getPageModel(documentId: string, pageIndex: number): Promise<any> {
const response = await fetch(`${this.baseUrl}/documents/${documentId}/pages/${pageIndex}/model`);
if (!response.ok) throw new Error(`Failed to get page model: ${response.statusText}`);
return response.json();
}
async applyEdits(documentId: string, operations: EditOperation[]): Promise<{ success: boolean; newDocumentId: string }> {
const response = await fetch(`${this.baseUrl}/edits/${documentId}`, {
method: 'POST',
@@ -283,15 +289,15 @@ class GatewayService {
async searchDocument(documentId: string, query: string): Promise<SearchResult[]> {
if (!query) return [];
const urlParams = new URLSearchParams({ q: query });
const response = await fetch(`${this.baseUrl}/documents/${documentId}/search?${urlParams.toString()}`);
if (response.status === 501) {
// Return mock empty results if backend isn't available
return [];
}
if (!response.ok) throw new Error(`Failed to search document: ${response.statusText}`);
return response.json();
}
+32 -3
View File
@@ -151,9 +151,9 @@ export const PDFViewer = React.forwardRef<PDFViewerRef, PDFViewerProps>(({
currentVisiblePageIdx = layout.index;
}
const isInside = (pageBottom >= viewportTop - (layout.height * buffer)) &&
(pageTop <= viewportBottom + (layout.height * buffer));
const isInside = (pageBottom >= viewportTop - (layout.height * buffer)) &&
(pageTop <= viewportBottom + (layout.height * buffer));
if (isInside) {
visible.push(layout);
}
@@ -201,6 +201,35 @@ export const PDFViewer = React.forwardRef<PDFViewerRef, PDFViewerProps>(({
};
}, [visiblePages, documentId, zoom, renderedPages]);
useEffect(() => {
const verifyPageModel = async () => {
if (visiblePages.length > 0 && documentId && !renderedPages[visiblePages[0].index + '_verified']) {
const pageIndex = visiblePages[0].index;
try {
const model = await gatewayService.getPageModel(documentId, pageIndex);
console.log(`--- Verification for Page ${pageIndex} ---`);
model.paragraphs?.forEach((p: any) => {
p.lines?.forEach((l: any) => {
l.runs?.forEach((r: any) => {
const isBold = (r.flags & 262144) !== 0 || r.font_name.toLowerCase().includes('bold');
const isItalic = (r.flags & 64) !== 0 || r.font_name.toLowerCase().includes('italic');
console.log(`Run Text: "${r.text}", Font Name: ${r.font_name}, Font Size: ${r.font_size}, Bold: ${isBold}, Italic: ${isItalic}, Embedded: ${r.is_embedded}, Type: ${r.type}`);
});
});
});
setRenderedPages((prev) => {
const next = [...prev];
next[pageIndex + '_verified' as any] = 'true';
return next;
});
} catch (e) {
// ignore or log
}
}
};
verifyPageModel();
}, [visiblePages, documentId]);
const handleTextSelection = (text: string, bbox: Rect) => {
if (activeTool === 'highlight') {
const newAnno: Annotation = {
+131 -1
View File
@@ -176,7 +176,6 @@ class FontInfoResponse(BaseModel):
descent: float
capHeight: float
@router.get("/{document_id}/fonts", response_model=list[FontInfoResponse])
def get_document_fonts(
document_id: str, start_page: int = 0, end_page: int = -1
@@ -316,3 +315,134 @@ def search_document(document_id: str, q: str) -> list[SearchMatch]:
except Exception as e:
raise HTTPException(status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, detail=str(e))
class GlyphModel(BaseModel):
text: str
unicode: int
font_name: str
flags: int
font_size: float
origin_x: float
origin_y: float
bbox_x: float
bbox_y: float
bbox_w: float
bbox_h: float
angle: float
class TextRunModel(BaseModel):
text: str
font_name: str
flags: int
font_size: float
internal_font_id: str
is_embedded: bool
type: str
glyphs: list[GlyphModel]
x: float
y: float
w: float
h: float
class TextLineModel(BaseModel):
runs: list[TextRunModel]
baseline_y: float
x: float
y: float
w: float
h: float
class ParagraphModel(BaseModel):
lines: list[TextLineModel]
x: float
y: float
w: float
h: float
class PageModelResponse(BaseModel):
paragraphs: list[ParagraphModel]
width: float
height: float
page_index: int
@router.get("/{document_id}/pages/{page_index}/model", response_model=PageModelResponse)
def get_page_model(document_id: str, page_index: int) -> PageModelResponse:
if not engine.is_available():
raise HTTPException(
status_code=status.HTTP_501_NOT_IMPLEMENTED,
detail="Engine bridge (bindings/python) not yet available.",
)
doc_info = document_store.get_document(document_id)
if not doc_info:
raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="Document not found")
try:
doc = doc_info["doc_instance"]
page = doc.get_page(page_index)
model = page.extract_document_model()
paragraphs = []
for p in model.paragraphs:
lines = []
for l in p.lines:
runs = []
for r in l.runs:
glyphs = []
for g in r.glyphs:
glyphs.append(GlyphModel(
text=g.text,
unicode=g.unicode,
font_name=g.font_name,
flags=g.flags,
font_size=g.font_size,
origin_x=g.origin_x,
origin_y=g.origin_y,
bbox_x=g.bbox_x,
bbox_y=g.bbox_y,
bbox_w=g.bbox_w,
bbox_h=g.bbox_h,
angle=g.angle
))
runs.append(TextRunModel(
text=r.text,
font_name=r.font_name,
flags=r.flags,
font_size=r.font_size,
internal_font_id=r.internal_font_id,
is_embedded=r.is_embedded,
type=r.type,
glyphs=glyphs,
x=r.x,
y=r.y,
w=r.w,
h=r.h
))
lines.append(TextLineModel(
runs=runs,
baseline_y=l.baseline_y,
x=l.x,
y=l.y,
w=l.w,
h=l.h
))
paragraphs.append(ParagraphModel(
lines=lines,
x=p.x,
y=p.y,
w=p.w,
h=p.h
))
return PageModelResponse(
paragraphs=paragraphs,
width=model.width,
height=model.height,
page_index=model.page_index
)
except ValueError as e:
raise HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail=str(e))
except IndexError as e:
raise HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail=str(e))
except Exception as e:
raise HTTPException(status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, detail=str(e))
+10 -10
View File
@@ -17,7 +17,7 @@ class TextOverlayData(BaseModel):
y: float
width: float
height: float
fontSize: float
fontSize: float = Field(..., gt=0)
fontFamily: str
color: str
@@ -63,7 +63,7 @@ class FreeTextData(BaseModel):
width: float
height: float
text: str
fontSize: float = 12.0
fontSize: float = Field(12.0, gt=0)
color: str = "#000000"
@@ -92,56 +92,56 @@ class PageRotationData(BaseModel):
class TextOverlayOperation(BaseModel):
id: str
type: Literal["text_overlay"]
pageIndex: int
pageIndex: int = Field(..., ge=0)
data: TextOverlayData
class RedactionOperation(BaseModel):
id: str
type: Literal["redaction"]
pageIndex: int
pageIndex: int = Field(..., ge=0)
data: RedactionData
class ImageOverlayOperation(BaseModel):
id: str
type: Literal["image_overlay"]
pageIndex: int
pageIndex: int = Field(..., ge=0)
data: ImageOverlayData
class HighlightOperation(BaseModel):
id: str
type: Literal["highlight"]
pageIndex: int
pageIndex: int = Field(..., ge=0)
data: HighlightData
class FreeTextOperation(BaseModel):
id: str
type: Literal["free_text"]
pageIndex: int
pageIndex: int = Field(..., ge=0)
data: FreeTextData
class CommentOperation(BaseModel):
id: str
type: Literal["comment"]
pageIndex: int
pageIndex: int = Field(..., ge=0)
data: StickyNoteData
class FreehandOperation(BaseModel):
id: str
type: Literal["freehand"]
pageIndex: int
pageIndex: int = Field(..., ge=0)
data: FreehandData
class PageRotationOperation(BaseModel):
id: str
type: Literal["page_rotation"]
pageIndex: int
pageIndex: int = Field(..., ge=0)
data: PageRotationData
+29 -3
View File
@@ -1,3 +1,5 @@
from typing import List, Annotated
from fastapi import APIRouter, HTTPException, status, Response, Path, Query
from fastapi import APIRouter, HTTPException, Response, status
from app.routers.documents import FontInfoResponse
@@ -9,7 +11,7 @@ compat_router = APIRouter(tags=["render"])
@router.get("/{page_index}/render")
def render_page(document_id: str, page_index: int, dpi: int = 96) -> Response:
def render_page(document_id: str, page_index: Annotated[int, Path(ge=0)], dpi: int = 96) -> Response:
if not engine.is_available():
raise HTTPException(
status_code=status.HTTP_501_NOT_IMPLEMENTED,
@@ -34,7 +36,7 @@ def render_page(document_id: str, page_index: int, dpi: int = 96) -> Response:
@router.get("/{page_index}/text")
def extract_page_text(document_id: str, page_index: int):
def extract_page_text(document_id: str, page_index: Annotated[int, Path(ge=0)]):
if not engine.is_available():
raise HTTPException(
status_code=status.HTTP_501_NOT_IMPLEMENTED,
@@ -71,7 +73,7 @@ def render_page_compat(
@router.get("/{page_index}")
def get_page_info(document_id: str, page_index: int):
def get_page_info(document_id: str, page_index: Annotated[int, Path(ge=0)]):
if not engine.is_available():
raise HTTPException(
status_code=status.HTTP_501_NOT_IMPLEMENTED,
@@ -200,3 +202,27 @@ def get_page_fonts(document_id: str, page_index: int) -> list[FontInfoResponse]:
raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="Page index out of range")
except Exception as e:
raise HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail=str(e))
@router.get("/{page_index}/fonts/glyph-width")
def get_page_glyph_width(document_id: str, page_index: Annotated[int, Path(ge=0)], font_name: str, charcode: int, font_size: Annotated[float, Query(gt=0)] = 12.0):
if not engine.is_available():
raise HTTPException(
status_code=status.HTTP_501_NOT_IMPLEMENTED,
detail="Engine bridge (bindings/python) not yet available."
)
doc_info = document_store.get_document(document_id)
if not doc_info:
raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="Document not found")
try:
doc = doc_info["doc_instance"]
page = doc.get_page(page_index)
width = page.get_glyph_width(font_name, charcode, font_size)
return {"width": width}
except IndexError:
raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="Page index out of range")
except Exception as e:
raise HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail=str(e))
+388
View File
@@ -0,0 +1,388 @@
"""
Font Extraction API Validation Script
======================================
Tests font extraction APIs against real PDFs from corpus/fonts/.
Validates: upload, document fonts, page fonts, text extraction, metadata accuracy.
"""
import json
import sys
import os
import time
import httpx
BASE_URL = "http://localhost:8000"
CORPUS_DIR = os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..", "corpus", "fonts"))
# PDFs to test (in priority order)
TARGET_PDFS = [
"utf-8.pdf",
"vertical_text.pdf",
"subset_font.pdf",
]
REQUIRED_FONT_FIELDS = [
"fontName", "type", "isEmbedded", "isSubset", "isVertical",
"encoding", "cmapName", "cidSystemInfo", "subsetTag",
"sourceType", "substitutedFrom", "substitutedTo",
"normalizedFamily", "internalFontId", "flags",
"ascent", "descent", "capHeight", "hasToUnicode",
]
def separator(title: str):
print(f"\n{'='*80}")
print(f" {title}")
print(f"{'='*80}\n")
def sub_separator(title: str):
print(f"\n--- {title} ---\n")
def validate_font_fields(font: dict, pdf_name: str, font_index: int) -> list:
"""Validate that all required fields are present and non-null."""
issues = []
for field in REQUIRED_FONT_FIELDS:
if field not in font:
issues.append(f" [MISSING] Font #{font_index} ({font.get('fontName', '?')}): field '{field}' is missing")
return issues
def check_duplicates(fonts: list) -> list:
"""Check for duplicate font entries."""
seen = set()
dupes = []
for f in fonts:
key = f.get("fontName", "") + "|" + f.get("type", "") + "|" + f.get("internalFontId", "")
if key in seen:
dupes.append(f.get("fontName", "?"))
seen.add(key)
return dupes
def check_empty_names(fonts: list) -> list:
"""Check for empty font names."""
return [i for i, f in enumerate(fonts) if not f.get("fontName", "").strip()]
def validate_pdf(client: httpx.Client, pdf_filename: str) -> dict:
"""Run all validation steps for a single PDF."""
result = {
"filename": pdf_filename,
"doc_id": None,
"upload_status": None,
"upload_response": None,
"doc_fonts": None,
"doc_font_count": 0,
"page_fonts": None,
"page_font_count": 0,
"text_result": None,
"glyph_count": 0,
"font_sizes": set(),
"issues": [],
"errors": [],
}
pdf_path = os.path.join(CORPUS_DIR, pdf_filename)
if not os.path.exists(pdf_path):
result["errors"].append(f"PDF file not found: {pdf_path}")
return result
# ========== STEP 1: Upload ==========
sub_separator(f"Step 1: Upload {pdf_filename}")
try:
with open(pdf_path, "rb") as f:
resp = client.post(
"/documents",
files={"file": (os.path.basename(pdf_path), f, "application/pdf")},
)
result["upload_status"] = resp.status_code
result["upload_response"] = resp.json()
print(f" Status: {resp.status_code}")
print(f" Response: {json.dumps(resp.json(), indent=2)}")
if resp.status_code != 201:
result["errors"].append(f"Upload failed with status {resp.status_code}: {resp.text}")
return result
result["doc_id"] = resp.json()["id"]
print(f" Document ID: {result['doc_id']}")
except Exception as e:
result["errors"].append(f"Upload exception: {e}")
return result
doc_id = result["doc_id"]
# ========== STEP 2: Document Font Extraction ==========
sub_separator("Step 2: Document Font Extraction")
try:
resp = client.get(f"/documents/{doc_id}/fonts")
print(f" Status: {resp.status_code}")
if resp.status_code != 200:
result["errors"].append(f"Document fonts failed: {resp.status_code} - {resp.text}")
else:
fonts = resp.json()
result["doc_fonts"] = fonts
result["doc_font_count"] = len(fonts)
print(f" Font count: {len(fonts)}")
# Print each font in full
for i, f in enumerate(fonts):
print(f"\n Font #{i}:")
print(f" {json.dumps(f, indent=4)}")
# Validate font count > 0
if len(fonts) == 0:
result["issues"].append("Document fonts: count is 0")
# Validate no empty names
empty = check_empty_names(fonts)
if empty:
result["issues"].append(f"Document fonts: empty font names at indices {empty}")
# Validate no duplicates
dupes = check_duplicates(fonts)
if dupes:
result["issues"].append(f"Document fonts: duplicate fonts: {dupes}")
# Validate all fields present
for i, f in enumerate(fonts):
field_issues = validate_font_fields(f, pdf_filename, i)
result["issues"].extend(field_issues)
except Exception as e:
result["errors"].append(f"Document fonts exception: {e}")
# ========== STEP 3: Page Font Extraction ==========
sub_separator("Step 3: Page Font Extraction (page 0)")
try:
resp = client.get(f"/documents/{doc_id}/pages/0/fonts")
print(f" Status: {resp.status_code}")
if resp.status_code != 200:
result["errors"].append(f"Page fonts failed: {resp.status_code} - {resp.text}")
else:
page_fonts = resp.json()
result["page_fonts"] = page_fonts
result["page_font_count"] = len(page_fonts)
print(f" Page font count: {len(page_fonts)}")
for i, f in enumerate(page_fonts):
print(f"\n Page Font #{i}:")
print(f" {json.dumps(f, indent=4)}")
if len(page_fonts) == 0:
result["issues"].append("Page fonts: count is 0")
# Compare with document fonts
if result["doc_fonts"] is not None:
doc_font_names = {f["fontName"] for f in result["doc_fonts"]}
page_font_names = {f["fontName"] for f in page_fonts}
print(f"\n Document font names: {sorted(doc_font_names)}")
print(f" Page font names: {sorted(page_font_names)}")
# Page fonts should be a subset of document fonts
extra_in_page = page_font_names - doc_font_names
if extra_in_page:
result["issues"].append(f"Page fonts not in document fonts: {extra_in_page}")
print(f" [ISSUE] Page has fonts not in document-level: {extra_in_page}")
else:
print(f" [OK] Page fonts are a subset of document fonts")
except Exception as e:
result["errors"].append(f"Page fonts exception: {e}")
# ========== STEP 4: Text Extraction ==========
sub_separator("Step 4: Text Extraction (page 0)")
try:
resp = client.get(f"/documents/{doc_id}/pages/0/text")
print(f" Status: {resp.status_code}")
if resp.status_code != 200:
result["errors"].append(f"Text extraction failed: {resp.status_code} - {resp.text}")
else:
text_data = resp.json()
result["text_result"] = text_data
text_content = text_data.get("text", "")
glyphs = text_data.get("glyphs", [])
result["glyph_count"] = len(glyphs)
print(f" Extracted text: {repr(text_content[:300])}")
print(f" Glyph count: {len(glyphs)}")
if not text_content.strip():
result["issues"].append("Text extraction: empty text")
if len(glyphs) == 0:
result["issues"].append("Text extraction: no glyphs returned")
else:
# Show first 5 glyphs as samples
print(f"\n First 5 glyphs (sample):")
for i, g in enumerate(glyphs[:5]):
print(f" Glyph #{i}: {json.dumps(g, indent=6)}")
# Check glyph structure
for i, g in enumerate(glyphs):
font_size = g.get("fontSize", 0)
if font_size > 0:
result["font_sizes"].add(font_size)
elif font_size == 0 and i < 5:
result["issues"].append(f"Glyph #{i}: fontSize is 0")
# Validate coordinate fields exist
for coord in ["x", "y", "right", "bottom"]:
if coord not in g and i < 3:
result["issues"].append(f"Glyph #{i}: missing coordinate '{coord}'")
print(f"\n Font sizes detected: {sorted(result['font_sizes'])}")
except Exception as e:
result["errors"].append(f"Text extraction exception: {e}")
# ========== STEP 5: Content-Specific Validation ==========
sub_separator("Step 5: Content-Specific Validation")
if "vertical" in pdf_filename.lower() and result["doc_fonts"]:
vertical_fonts = [f for f in result["doc_fonts"] if f.get("isVertical")]
print(f" Vertical text PDF - fonts with isVertical=true: {len(vertical_fonts)}")
if len(vertical_fonts) == 0:
result["issues"].append("vertical_text.pdf: No fonts have isVertical=true")
print(f" [ISSUE] No vertical fonts detected!")
else:
for vf in vertical_fonts:
print(f" - {vf['fontName']} (isVertical=true)")
print(f" [OK] Vertical fonts detected correctly")
if "subset" in pdf_filename.lower() and result["doc_fonts"]:
subset_fonts = [f for f in result["doc_fonts"] if f.get("isSubset")]
print(f" Subset font PDF - fonts with isSubset=true: {len(subset_fonts)}")
if len(subset_fonts) == 0:
result["issues"].append("subset_font.pdf: No fonts have isSubset=true")
print(f" [ISSUE] No subset fonts detected!")
else:
for sf in subset_fonts:
print(f" - {sf['fontName']} (isSubset=true, subsetTag='{sf.get('subsetTag', '')}')")
print(f" [OK] Subset fonts detected correctly")
if "utf" in pdf_filename.lower() and result["text_result"]:
text = result["text_result"].get("text", "")
print(f" UTF-8 PDF - extracted text: {repr(text[:300])}")
# Check for non-ASCII characters
non_ascii = [c for c in text if ord(c) > 127]
if non_ascii:
print(f" Non-ASCII characters found: {len(non_ascii)} chars")
print(f" Sample non-ASCII: {repr(''.join(non_ascii[:30]))}")
print(f" [OK] UTF-8 text extracts with non-ASCII content")
else:
print(f" [INFO] No non-ASCII characters detected - content may be ASCII-only")
return result
def main():
print("=" * 80)
print(" FONT EXTRACTION API VALIDATION")
print(f" Server: {BASE_URL}")
print(f" Corpus: {CORPUS_DIR}")
print(f" Timestamp: {time.strftime('%Y-%m-%d %H:%M:%S')}")
print("=" * 80)
client = httpx.Client(base_url=BASE_URL, timeout=30.0)
# Verify server is up
try:
r = client.get("/")
print(f"\n Server status: OK ({r.status_code})")
except Exception as e:
print(f"\n [FATAL] Cannot connect to server: {e}")
sys.exit(1)
# Verify corpus directory
if not os.path.isdir(CORPUS_DIR):
print(f"\n [FATAL] Corpus directory not found: {CORPUS_DIR}")
sys.exit(1)
available = [f for f in TARGET_PDFS if os.path.exists(os.path.join(CORPUS_DIR, f))]
print(f" PDFs to validate: {available}")
results = []
for pdf in available:
separator(f"VALIDATING: {pdf}")
result = validate_pdf(client, pdf)
results.append(result)
# ========== FINAL REPORT ==========
separator("FINAL REPORT")
for r in results:
print(f"\n{'~'*60}")
print(f" PDF: {r['filename']}")
print(f" Document ID: {r['doc_id']}")
print(f" Upload Status: {r['upload_status']}")
print(f" Document Font Count: {r['doc_font_count']}")
print(f" Page Font Count: {r['page_font_count']}")
print(f" Glyph Count: {r['glyph_count']}")
print(f" Font Sizes: {sorted(r['font_sizes']) if r['font_sizes'] else 'N/A'}")
print(f" Issues: {len(r['issues'])}")
for issue in r['issues']:
print(f" >> {issue}")
print(f" Errors: {len(r['errors'])}")
for err in r['errors']:
print(f" XX {err}")
# Summary Answers
separator("SUMMARY ANSWERS")
total_issues = sum(len(r["issues"]) for r in results)
total_errors = sum(len(r["errors"]) for r in results)
all_fonts_extracted = all(r["doc_font_count"] > 0 for r in results if not r["errors"])
print(f" 1. Are fonts being extracted correctly?")
print(f" {'YES' if all_fonts_extracted else 'NO'} - {sum(r['doc_font_count'] for r in results)} total fonts across {len(results)} PDFs")
page_doc_consistent = all(
not any("not in document" in i for i in r["issues"])
for r in results
)
print(f"\n 2. Are page fonts and document fonts consistent?")
print(f" {'YES' if page_doc_consistent else 'NO'}")
font_sizes_ok = all(r["glyph_count"] > 0 for r in results if not r["errors"])
print(f"\n 3. Are font sizes being extracted correctly?")
print(f" {'YES' if font_sizes_ok else 'NO'}")
# Check vertical/subset
vertical_ok = True
subset_ok = True
for r in results:
if "vertical" in r["filename"] and any("isVertical" in i for i in r["issues"]):
vertical_ok = False
if "subset" in r["filename"] and any("isSubset" in i for i in r["issues"]):
subset_ok = False
print(f"\n 4. Are vertical/subset fonts detected correctly?")
print(f" Vertical: {'YES' if vertical_ok else 'NO'}")
print(f" Subset: {'YES' if subset_ok else 'NO'}")
print(f"\n 5. Are there any metadata inaccuracies?")
if total_issues == 0 and total_errors == 0:
print(f" NO - All {len(results)} PDFs passed validation cleanly")
else:
print(f" YES - {total_issues} issues and {total_errors} errors found")
for r in results:
for issue in r["issues"]:
print(f" - [{r['filename']}] {issue}")
print(f"\n{'='*80}")
print(f" VALIDATION COMPLETE: {total_issues} issues, {total_errors} errors")
print(f"{'='*80}")
client.close()
return 0 if total_errors == 0 else 1
if __name__ == "__main__":
sys.exit(main())
+193 -3
View File
@@ -1,4 +1,11 @@
<<<<<<< HEAD
import re
import pytest
from fastapi.testclient import TestClient
import sys
=======
import contextlib
>>>>>>> 2638dfe969874b954502b26656df48ad971e306d
import os
import sys
@@ -14,9 +21,15 @@ with contextlib.suppress(ImportError):
client = TestClient(app)
<<<<<<< HEAD
# Regex: six uppercase ASCII letters followed by '+'
SUBSET_PREFIX_RE = re.compile(r'^[A-Z]{6}\+')
=======
>>>>>>> 2638dfe969874b954502b26656df48ad971e306d
def get_doc_id(filename: str) -> str:
# First ensure the document is loaded
"""Upload a PDF from corpus/fonts and return its document ID."""
filepath = os.path.abspath(f"../corpus/fonts/{filename}")
with open(filepath, "rb") as f:
resp = client.post("/documents", files={"file": (filename, f, "application/pdf")})
@@ -24,8 +37,15 @@ def get_doc_id(filename: str) -> str:
return resp.json()["id"]
<<<<<<< HEAD
# =========================================================================
# 1. Vertical font regression
# =========================================================================
=======
>>>>>>> 2638dfe969874b954502b26656df48ad971e306d
def test_is_vertical_regression():
# 1. Verify Identity-V fonts are detected correctly
"""Identity-V fonts must be flagged isVertical; horizontal fonts must not."""
doc_id = get_doc_id("vertical_text.pdf")
resp = client.get(f"/documents/{doc_id}/fonts")
assert resp.status_code == 200
@@ -35,7 +55,10 @@ def test_is_vertical_regression():
assert font["isVertical"] is True
assert font["encoding"] == "Identity-V"
<<<<<<< HEAD
=======
# 2. Verify horizontal fonts are not falsely detected as vertical
>>>>>>> 2638dfe969874b954502b26656df48ad971e306d
doc_id_h = get_doc_id("utf-8.pdf")
resp_h = client.get(f"/documents/{doc_id_h}/fonts")
fonts_h = resp_h.json()
@@ -44,23 +67,134 @@ def test_is_vertical_regression():
assert f["isVertical"] is False
<<<<<<< HEAD
# =========================================================================
# 2. Internal Font ID: no duplicate subset prefix (core regression)
# =========================================================================
def test_internal_font_id_no_duplicate_prefix():
"""
Regression: subset fonts must NOT produce "ABCDEF_ABCDEF+Arial".
The internalFontId for a subset font should be the fontName itself
(e.g. "ABCDEF+Arial"), which already encodes the subset tag.
"""
doc_id = get_doc_id("subset_font.pdf")
resp = client.get(f"/documents/{doc_id}/fonts")
assert resp.status_code == 200
fonts = resp.json()
assert len(fonts) > 0
for font in fonts:
fid = font["internalFontId"]
tag = font.get("subsetTag", "")
if font.get("isSubset") and tag:
# Must NOT start with "TAG_TAG"
assert not fid.startswith(tag + "_" + tag), (
f"Duplicate subset prefix detected: internalFontId='{fid}'"
)
# Must equal fontName directly (e.g. "ABCDEF+Arial")
assert fid == font["fontName"], (
f"Expected internalFontId==fontName for subset font, "
f"got '{fid}' vs '{font['fontName']}'"
)
def test_internal_font_id_subset_format():
"""
For any subset font the internalFontId must match the pattern
ABCDEF+BaseName — exactly the fontName reported by PDFium.
"""
=======
def test_internal_font_id_regression():
# Verify subset fonts do not duplicate subset prefixes
>>>>>>> 2638dfe969874b954502b26656df48ad971e306d
doc_id = get_doc_id("text_font.pdf")
resp = client.get(f"/documents/{doc_id}/fonts")
assert resp.status_code == 200
fonts = resp.json()
assert len(fonts) > 0
for font in fonts:
if font.get("isSubset"):
assert font["subsetTag"] in font["fontName"]
<<<<<<< HEAD
# internalFontId == fontName (e.g. "ABCDEF+Arial")
assert font["internalFontId"] == font["fontName"]
# The ID must contain exactly one '+' from the subset tag
assert font["internalFontId"].count("+") == 1
# =========================================================================
# 3. Non-subset font ID format
# =========================================================================
def test_internal_font_id_non_subset_format():
"""
Non-subset fonts must have internalFontId = fontName_type_flags.
Examples: "Helvetica_Type1_32", "Times-Roman_Type1_32".
"""
doc_id = get_doc_id("utf-8.pdf")
resp = client.get(f"/documents/{doc_id}/fonts")
assert resp.status_code == 200
fonts = resp.json()
assert len(fonts) > 0
for font in fonts:
if not font.get("isSubset"):
expected = f"{font['fontName']}_{font['type']}_{font['flags']}"
assert font["internalFontId"] == expected, (
f"Non-subset internalFontId mismatch: "
f"got '{font['internalFontId']}', expected '{expected}'"
)
# Must NOT contain a '+' (no subset prefix)
assert "+" not in font["internalFontId"]
# =========================================================================
# 4. ID stability across document-level and page-level APIs
# =========================================================================
def test_font_id_stable_across_apis():
"""
The internalFontId for the same font must be identical whether queried
from the document-level /fonts endpoint or the page-level /pages/0/fonts.
"""
for pdf in ("subset_font.pdf", "utf-8.pdf", "vertical_text.pdf"):
filepath = os.path.abspath(f"../corpus/fonts/{pdf}")
if not os.path.exists(filepath):
continue
doc_id = get_doc_id(pdf)
doc_resp = client.get(f"/documents/{doc_id}/fonts")
page_resp = client.get(f"/documents/{doc_id}/pages/0/fonts")
assert doc_resp.status_code == 200
assert page_resp.status_code == 200
doc_ids = {f["fontName"]: f["internalFontId"] for f in doc_resp.json()}
page_ids = {f["fontName"]: f["internalFontId"] for f in page_resp.json()}
for name in page_ids:
assert name in doc_ids, f"Page font '{name}' not in doc fonts for {pdf}"
assert page_ids[name] == doc_ids[name], (
f"ID mismatch for '{name}' in {pdf}: "
f"doc='{doc_ids[name]}' vs page='{page_ids[name]}'"
)
# =========================================================================
# 5. CID collection regression
# =========================================================================
=======
assert not font["internalFontId"].startswith(
font["subsetTag"] + "_" + font["subsetTag"]
)
>>>>>>> 2638dfe969874b954502b26656df48ad971e306d
def test_cid_collection_regression():
# Verify Adobe collections
"""Adobe CID collections must use the 'Adobe-' prefix."""
doc_id = get_doc_id("vertical_text.pdf")
resp = client.get(f"/documents/{doc_id}/fonts")
fonts = resp.json()
@@ -70,6 +204,31 @@ def test_cid_collection_regression():
assert "Adobe-" in font["cidSystemInfo"]
<<<<<<< HEAD
def test_cns1_regression():
"""Verify Adobe-CNS1 (Traditional Chinese) CID fonts and text extraction."""
doc_id = get_doc_id("cns1_test.pdf")
# 1. Verify font extraction
resp = client.get(f"/documents/{doc_id}/fonts")
assert resp.status_code == 200
fonts = resp.json()
assert len(fonts) > 0
cns1_fonts = [f for f in fonts if f.get("cidSystemInfo") == "Adobe-CNS1"]
assert len(cns1_fonts) > 0, "No Adobe-CNS1 fonts detected"
# 2. Verify text extraction
resp = client.get(f"/documents/{doc_id}/pages/0/text")
assert resp.status_code == 200
text = resp.json()["text"]
assert "\u4e00\u4e2d\u4ed7" in text, "Failed to extract Traditional Chinese text"
# =========================================================================
# 6. UTF-8 corpus regression
# =========================================================================
=======
>>>>>>> 2638dfe969874b954502b26656df48ad971e306d
def test_utf8_corpus_regression():
doc_id = get_doc_id("utf-8.pdf")
resp = client.get(f"/documents/{doc_id}/pages/0/text")
@@ -79,9 +238,40 @@ def test_utf8_corpus_regression():
assert len(data["text"]) > 0
<<<<<<< HEAD
# =========================================================================
# 7. Font size regression
# =========================================================================
=======
>>>>>>> 2638dfe969874b954502b26656df48ad971e306d
def test_font_size_regression():
doc_id = get_doc_id("utf-8.pdf")
resp = client.get(f"/documents/{doc_id}/pages/0/text")
data = resp.json()
sizes = set(g["fontSize"] for g in data["glyphs"])
assert len(sizes) >= 2 # utf-8.pdf should have multiple font sizes
<<<<<<< HEAD
# =========================================================================
# 8. ID uniqueness
# =========================================================================
def test_font_ids_unique_within_document():
"""Each font within a document must have a distinct internalFontId."""
for pdf in ("subset_font.pdf", "utf-8.pdf", "vertical_text.pdf"):
filepath = os.path.abspath(f"../corpus/fonts/{pdf}")
if not os.path.exists(filepath):
continue
doc_id = get_doc_id(pdf)
resp = client.get(f"/documents/{doc_id}/fonts")
fonts = resp.json()
ids = [f["internalFontId"] for f in fonts]
assert len(ids) == len(set(ids)), (
f"Duplicate internalFontId values in {pdf}: {ids}"
)
=======
>>>>>>> 2638dfe969874b954502b26656df48ad971e306d
+93
View File
@@ -0,0 +1,93 @@
import pytest
import os
from fastapi.testclient import TestClient
from app.main import app
client = TestClient(app)
def get_doc_id(filename: str) -> str:
"""Upload a PDF from corpus/fonts and return its document ID."""
filepath = os.path.abspath(f"gateway/../corpus/fonts/{filename}")
if not os.path.exists(filepath):
filepath = os.path.abspath(f"../corpus/fonts/{filename}")
with open(filepath, "rb") as f:
resp = client.post(
"/documents",
files={"file": (filename, f, "application/pdf")}
)
assert resp.status_code == 201, f"Failed to load {filename}: {resp.json()}"
return resp.json()["id"]
def test_glyph_width_proportionality():
"""Verify that wide characters return larger glyph widths than narrow characters."""
doc_id = get_doc_id("utf-8.pdf")
# 1. Fetch fonts for the page
resp = client.get(f"/documents/{doc_id}/pages/0/fonts")
assert resp.status_code == 200
fonts = resp.json()
assert len(fonts) > 0, "No fonts extracted from utf-8.pdf"
# Find an embedded or substituted font name
font_name = fonts[0]["fontName"]
# 2. Get width of wide character 'W' (charcode 87)
w_resp = client.get(
f"/documents/{doc_id}/pages/0/fonts/glyph-width",
params={"font_name": font_name, "charcode": 87, "font_size": 12.0}
)
assert w_resp.status_code == 200
w_width = w_resp.json()["width"]
assert w_width > 0.0
# 3. Get width of narrow character 'i' (charcode 105)
i_resp = client.get(
f"/documents/{doc_id}/pages/0/fonts/glyph-width",
params={"font_name": font_name, "charcode": 105, "font_size": 12.0}
)
assert i_resp.status_code == 200
i_width = i_resp.json()["width"]
assert i_width > 0.0
# 'W' must be strictly wider than 'i' in proportional typefaces
assert w_width > i_width, f"Expected width('W') > width('i'), got {w_width} vs {i_width}"
def test_glyph_width_font_size_scaling():
"""Verify that glyph width scales proportionally with font size."""
doc_id = get_doc_id("utf-8.pdf")
resp = client.get(f"/documents/{doc_id}/pages/0/fonts")
fonts = resp.json()
font_name = fonts[0]["fontName"]
# Width at 12pt
resp_12 = client.get(
f"/documents/{doc_id}/pages/0/fonts/glyph-width",
params={"font_name": font_name, "charcode": 65, "font_size": 12.0}
)
assert resp_12.status_code == 200
width_12 = resp_12.json()["width"]
# Width at 24pt
resp_24 = client.get(
f"/documents/{doc_id}/pages/0/fonts/glyph-width",
params={"font_name": font_name, "charcode": 65, "font_size": 24.0}
)
assert resp_24.status_code == 200
width_24 = resp_24.json()["width"]
# Scaling must be linear: width(24pt) = 2.0 * width(12pt)
assert pytest.approx(width_24) == 2.0 * width_12
def test_glyph_width_invalid_font():
"""Verify that querying an invalid font returns a bad request error."""
doc_id = get_doc_id("utf-8.pdf")
resp = client.get(
f"/documents/{doc_id}/pages/0/fonts/glyph-width",
params={"font_name": "NonExistentFontName123", "charcode": 65, "font_size": 12.0}
)
assert resp.status_code == 400
assert "detail" in resp.json()
+5 -1
View File
@@ -137,6 +137,10 @@ def test_extract_page_text(client: TestClient):
first_glyph = glyphs[0]
for key in ["text", "x", "y", "w", "h", "fontSize"]:
assert key in first_glyph
for g in glyphs:
assert g["fontSize"] != 1.0, f"Fake fontSize 1.0 detected for glyph: {g}"
assert g["text"] not in ["\r", "\n"], f"Control character detected in glyph bounds: {g}"
def test_apply_edits_and_incremental_save(client: TestClient):
@@ -441,7 +445,7 @@ def test_font_size_and_diagnostics_advanced(client: TestClient):
assert font["sourceType"] == "Embedded"
assert len(font["subsetTag"]) == 6
assert font["subsetTag"].isupper()
assert font["internalFontId"] == f"{font['subsetTag']}_{font['fontName']}"
assert font["internalFontId"] == font["fontName"]
else:
assert len(font["subsetTag"]) == 0
assert font["internalFontId"] == f"{font['fontName']}_{font['type']}_{font['flags']}"
+554
View File
@@ -0,0 +1,554 @@
================================================================================
FONT EXTRACTION API VALIDATION
Server: http://localhost:8000
Corpus: C:\Users\Maskan\Desktop\pdf_editor\pdf\corpus\fonts
Timestamp: 2026-06-02 12:36:47
================================================================================
Server status: OK (200)
PDFs to validate: ['utf-8.pdf', 'vertical_text.pdf', 'subset_font.pdf']
================================================================================
VALIDATING: utf-8.pdf
================================================================================
--- Step 1: Upload utf-8.pdf ---
Status: 201
Response: {
"id": "6764e35e-3594-4b50-a378-3ea122a54d9d",
"filename": "utf-8.pdf",
"sizeBytes": 1275,
"totalPages": 1,
"uploadedAt": "2026-06-02T07:06:50.429369Z",
"status": "ready"
}
Document ID: 6764e35e-3594-4b50-a378-3ea122a54d9d
--- Step 2: Document Font Extraction ---
Status: 200
Font count: 2
Font #0:
{
"fontName": "Helvetica",
"type": "Type1",
"isEmbedded": false,
"isSubset": false,
"isVertical": false,
"encoding": "WinAnsiEncoding",
"hasToUnicode": true,
"cmapName": "None",
"cidSystemInfo": "None",
"subsetTag": "",
"sourceType": "SystemFallback",
"substitutedFrom": "",
"substitutedTo": "",
"normalizedFamily": "Helvetica",
"internalFontId": "Helvetica_Type1_32",
"flags": 32,
"ascent": 905.0,
"descent": -211.0,
"capHeight": 728.0
}
Font #1:
{
"fontName": "Times-Roman",
"type": "Type1",
"isEmbedded": false,
"isSubset": false,
"isVertical": false,
"encoding": "WinAnsiEncoding",
"hasToUnicode": true,
"cmapName": "None",
"cidSystemInfo": "None",
"subsetTag": "",
"sourceType": "SystemFallback",
"substitutedFrom": "",
"substitutedTo": "",
"normalizedFamily": "Times",
"internalFontId": "Times-Roman_Type1_32",
"flags": 32,
"ascent": 891.0,
"descent": -216.0,
"capHeight": 662.0
}
--- Step 3: Page Font Extraction (page 0) ---
Status: 200
Page font count: 2
Page Font #0:
{
"fontName": "Helvetica",
"type": "Type1",
"isEmbedded": false,
"isSubset": false,
"isVertical": false,
"encoding": "WinAnsiEncoding",
"hasToUnicode": true,
"cmapName": "None",
"cidSystemInfo": "None",
"subsetTag": "",
"sourceType": "SystemFallback",
"substitutedFrom": "",
"substitutedTo": "",
"normalizedFamily": "Helvetica",
"internalFontId": "Helvetica_Type1_32",
"flags": 32,
"ascent": 905.0,
"descent": -211.0,
"capHeight": 728.0
}
Page Font #1:
{
"fontName": "Times-Roman",
"type": "Type1",
"isEmbedded": false,
"isSubset": false,
"isVertical": false,
"encoding": "WinAnsiEncoding",
"hasToUnicode": true,
"cmapName": "None",
"cidSystemInfo": "None",
"subsetTag": "",
"sourceType": "SystemFallback",
"substitutedFrom": "",
"substitutedTo": "",
"normalizedFamily": "Times",
"internalFontId": "Times-Roman_Type1_32",
"flags": 32,
"ascent": 891.0,
"descent": -216.0,
"capHeight": 662.0
}
Document font names: ['Helvetica', 'Times-Roman']
Page font names: ['Helvetica', 'Times-Roman']
[OK] Page fonts are a subset of document fonts
--- Step 4: Text Extraction (page 0) ---
Status: 200
Extracted text: 'Hello World - UTF-8 Test Document\r\nStandard Latin Text for Encoding Verification\r\nFont Size Detection Sample: Small Text 12pt\r\nLARGE TEXT FOR SIZE 18PT DETECTION\r\nMore 18pt content: ABCDEFGHabcdefgh 0123456789\r\nBack to 12pt: The quick brown fox jumps over the lazy dog\r\nSpecial chars: copyright secti'
Glyph count: 312
First 5 glyphs (sample):
Glyph #0: {
"text": "H",
"x": 72.95999908447266,
"y": 720.0,
"w": 6.7440032958984375,
"h": 8.59197998046875,
"fontSize": 12.0
}
Glyph #1: {
"text": "e",
"x": 81.10800170898438,
"y": 719.8679809570312,
"w": 5.736000061035156,
"h": 6.49200439453125,
"fontSize": 12.0
}
Glyph #2: {
"text": "l",
"x": 88.10400390625,
"y": 720.0,
"w": 1.055999755859375,
"h": 8.59197998046875,
"fontSize": 12.0
}
Glyph #3: {
"text": "l",
"x": 90.76799774169922,
"y": 720.0,
"w": 1.055999755859375,
"h": 8.59197998046875,
"fontSize": 12.0
}
Glyph #4: {
"text": "o",
"x": 93.05999755859375,
"y": 719.8679809570312,
"w": 5.832000732421875,
"h": 6.49200439453125,
"fontSize": 12.0
}
Font sizes detected: [1.0, 12.0, 18.0]
--- Step 5: Content-Specific Validation ---
UTF-8 PDF - extracted text: 'Hello World - UTF-8 Test Document\r\nStandard Latin Text for Encoding Verification\r\nFont Size Detection Sample: Small Text 12pt\r\nLARGE TEXT FOR SIZE 18PT DETECTION\r\nMore 18pt content: ABCDEFGHabcdefgh 0123456789\r\nBack to 12pt: The quick brown fox jumps over the lazy dog\r\nSpecial chars: copyright secti'
[INFO] No non-ASCII characters detected - content may be ASCII-only
================================================================================
VALIDATING: vertical_text.pdf
================================================================================
--- Step 1: Upload vertical_text.pdf ---
Status: 201
Response: {
"id": "d28fd4e4-a059-4a9b-bf15-b6cf25caf280",
"filename": "vertical_text.pdf",
"sizeBytes": 3518,
"totalPages": 1,
"uploadedAt": "2026-06-02T07:06:50.465835Z",
"status": "ready"
}
Document ID: d28fd4e4-a059-4a9b-bf15-b6cf25caf280
--- Step 2: Document Font Extraction ---
Status: 200
Font count: 1
Font #0:
{
"fontName": "Test",
"type": "TrueType",
"isEmbedded": false,
"isSubset": false,
"isVertical": true,
"encoding": "Identity-V",
"hasToUnicode": true,
"cmapName": "Identity-V",
"cidSystemInfo": "None",
"subsetTag": "",
"sourceType": "Substituted",
"substitutedFrom": "Test",
"substitutedTo": "Arial",
"normalizedFamily": "Test",
"internalFontId": "Test_TrueType_524320",
"flags": 524320,
"ascent": 905.0,
"descent": -211.0,
"capHeight": 728.0
}
--- Step 3: Page Font Extraction (page 0) ---
Status: 200
Page font count: 1
Page Font #0:
{
"fontName": "Test",
"type": "TrueType",
"isEmbedded": false,
"isSubset": false,
"isVertical": true,
"encoding": "Identity-V",
"hasToUnicode": true,
"cmapName": "Identity-V",
"cidSystemInfo": "None",
"subsetTag": "",
"sourceType": "Substituted",
"substitutedFrom": "Test",
"substitutedTo": "Arial",
"normalizedFamily": "Test",
"internalFontId": "Test_TrueType_524320",
"flags": 524320,
"ascent": 905.0,
"descent": -211.0,
"capHeight": 728.0
}
Document font names: ['Test']
Page font names: ['Test']
[OK] Page fonts are a subset of document fonts
--- Step 4: Text Extraction (page 0) ---
Status: 200
Extracted text: 'Hello World!\r\nHello'
Glyph count: 19
First 5 glyphs (sample):
Glyph #0: {
"text": "H",
"x": 6.832000255584717,
"y": 180.1840057373047,
"w": 6.552000522613525,
"h": 8.699996948242188,
"fontSize": 12.0
}
Glyph #1: {
"text": "e",
"x": 7.324000358581543,
"y": 171.39999389648438,
"w": 5.495999336242676,
"h": 6.756011962890625,
"fontSize": 12.0
}
Glyph #2: {
"text": "l",
"x": 9.687999725341797,
"y": 160.49200439453125,
"w": 1.055999755859375,
"h": 9.251998901367188,
"fontSize": 12.0
}
Glyph #3: {
"text": "l",
"x": 9.687999725341797,
"y": 149.4759979248047,
"w": 1.055999755859375,
"h": 9.251998901367188,
"fontSize": 12.0
}
Glyph #4: {
"text": "o",
"x": 7.324000358581543,
"y": 140.69200134277344,
"w": 5.951999664306641,
"h": 6.7559967041015625,
"fontSize": 12.0
}
Font sizes detected: [1.0, 12.0]
--- Step 5: Content-Specific Validation ---
Vertical text PDF - fonts with isVertical=true: 1
- Test (isVertical=true)
[OK] Vertical fonts detected correctly
================================================================================
VALIDATING: subset_font.pdf
================================================================================
--- Step 1: Upload subset_font.pdf ---
Status: 201
Response: {
"id": "cf3f2218-6ee7-4e34-973d-6e1b61c13cf8",
"filename": "subset_font.pdf",
"sizeBytes": 646,
"totalPages": 1,
"uploadedAt": "2026-06-02T07:06:50.497336Z",
"status": "ready"
}
Document ID: cf3f2218-6ee7-4e34-973d-6e1b61c13cf8
--- Step 2: Document Font Extraction ---
Status: 200
Font count: 1
Font #0:
{
"fontName": "ABCDEF+Arial",
"type": "TrueType",
"isEmbedded": true,
"isSubset": true,
"isVertical": false,
"encoding": "WinAnsiEncoding",
"hasToUnicode": true,
"cmapName": "None",
"cidSystemInfo": "None",
"subsetTag": "ABCDEF",
"sourceType": "Embedded",
"substitutedFrom": "",
"substitutedTo": "",
"normalizedFamily": "Arial",
"internalFontId": "ABCDEF_ABCDEF+Arial",
"flags": 0,
"ascent": 905.0,
"descent": -211.0,
"capHeight": 728.0
}
--- Step 3: Page Font Extraction (page 0) ---
Status: 200
Page font count: 1
Page Font #0:
{
"fontName": "ABCDEF+Arial",
"type": "TrueType",
"isEmbedded": true,
"isSubset": true,
"isVertical": false,
"encoding": "WinAnsiEncoding",
"hasToUnicode": true,
"cmapName": "None",
"cidSystemInfo": "None",
"subsetTag": "ABCDEF",
"sourceType": "Embedded",
"substitutedFrom": "",
"substitutedTo": "",
"normalizedFamily": "Arial",
"internalFontId": "ABCDEF_ABCDEF+Arial",
"flags": 0,
"ascent": 905.0,
"descent": -211.0,
"capHeight": 728.0
}
Document font names: ['ABCDEF+Arial']
Page font names: ['ABCDEF+Arial']
[OK] Page fonts are a subset of document fonts
--- Step 4: Text Extraction (page 0) ---
Status: 200
Extracted text: 'Subset Text'
Glyph count: 11
First 5 glyphs (sample):
Glyph #0: {
"text": "S",
"x": 72.54000091552734,
"y": 719.8679809570312,
"w": 6.839996337890625,
"h": 8.8680419921875,
"fontSize": 12.0
}
Glyph #1: {
"text": "u",
"x": 80.77200317382812,
"y": 719.8679809570312,
"w": 5.0399932861328125,
"h": 6.36004638671875,
"fontSize": 12.0
}
Glyph #2: {
"text": "b",
"x": 87.45600128173828,
"y": 719.8679809570312,
"w": 5.400001525878906,
"h": 8.7239990234375,
"fontSize": 12.0
}
Glyph #3: {
"text": "s",
"x": 93.72000122070312,
"y": 719.8679809570312,
"w": 5.159996032714844,
"h": 6.49200439453125,
"fontSize": 12.0
}
Glyph #4: {
"text": "e",
"x": 99.79199981689453,
"y": 719.8679809570312,
"w": 5.736000061035156,
"h": 6.49200439453125,
"fontSize": 12.0
}
Font sizes detected: [12.0]
--- Step 5: Content-Specific Validation ---
Subset font PDF - fonts with isSubset=true: 1
- ABCDEF+Arial (isSubset=true, subsetTag='ABCDEF')
[OK] Subset fonts detected correctly
================================================================================
FINAL REPORT
================================================================================
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
PDF: utf-8.pdf
Document ID: 6764e35e-3594-4b50-a378-3ea122a54d9d
Upload Status: 201
Document Font Count: 2
Page Font Count: 2
Glyph Count: 312
Font Sizes: [1.0, 12.0, 18.0]
Issues: 6
>> Glyph #0: missing coordinate 'right'
>> Glyph #0: missing coordinate 'bottom'
>> Glyph #1: missing coordinate 'right'
>> Glyph #1: missing coordinate 'bottom'
>> Glyph #2: missing coordinate 'right'
>> Glyph #2: missing coordinate 'bottom'
Errors: 0
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
PDF: vertical_text.pdf
Document ID: d28fd4e4-a059-4a9b-bf15-b6cf25caf280
Upload Status: 201
Document Font Count: 1
Page Font Count: 1
Glyph Count: 19
Font Sizes: [1.0, 12.0]
Issues: 6
>> Glyph #0: missing coordinate 'right'
>> Glyph #0: missing coordinate 'bottom'
>> Glyph #1: missing coordinate 'right'
>> Glyph #1: missing coordinate 'bottom'
>> Glyph #2: missing coordinate 'right'
>> Glyph #2: missing coordinate 'bottom'
Errors: 0
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
PDF: subset_font.pdf
Document ID: cf3f2218-6ee7-4e34-973d-6e1b61c13cf8
Upload Status: 201
Document Font Count: 1
Page Font Count: 1
Glyph Count: 11
Font Sizes: [12.0]
Issues: 6
>> Glyph #0: missing coordinate 'right'
>> Glyph #0: missing coordinate 'bottom'
>> Glyph #1: missing coordinate 'right'
>> Glyph #1: missing coordinate 'bottom'
>> Glyph #2: missing coordinate 'right'
>> Glyph #2: missing coordinate 'bottom'
Errors: 0
================================================================================
SUMMARY ANSWERS
================================================================================
1. Are fonts being extracted correctly?
YES - 4 total fonts across 3 PDFs
2. Are page fonts and document fonts consistent?
YES
3. Are font sizes being extracted correctly?
YES
4. Are vertical/subset fonts detected correctly?
Vertical: YES
Subset: YES
5. Are there any metadata inaccuracies?
YES - 18 issues and 0 errors found
- [utf-8.pdf] Glyph #0: missing coordinate 'right'
- [utf-8.pdf] Glyph #0: missing coordinate 'bottom'
- [utf-8.pdf] Glyph #1: missing coordinate 'right'
- [utf-8.pdf] Glyph #1: missing coordinate 'bottom'
- [utf-8.pdf] Glyph #2: missing coordinate 'right'
- [utf-8.pdf] Glyph #2: missing coordinate 'bottom'
- [vertical_text.pdf] Glyph #0: missing coordinate 'right'
- [vertical_text.pdf] Glyph #0: missing coordinate 'bottom'
- [vertical_text.pdf] Glyph #1: missing coordinate 'right'
- [vertical_text.pdf] Glyph #1: missing coordinate 'bottom'
- [vertical_text.pdf] Glyph #2: missing coordinate 'right'
- [vertical_text.pdf] Glyph #2: missing coordinate 'bottom'
- [subset_font.pdf] Glyph #0: missing coordinate 'right'
- [subset_font.pdf] Glyph #0: missing coordinate 'bottom'
- [subset_font.pdf] Glyph #1: missing coordinate 'right'
- [subset_font.pdf] Glyph #1: missing coordinate 'bottom'
- [subset_font.pdf] Glyph #2: missing coordinate 'right'
- [subset_font.pdf] Glyph #2: missing coordinate 'bottom'
================================================================================
VALIDATION COMPLETE: 18 issues, 0 errors
================================================================================
BIN
View File
Binary file not shown.
+1 -1
View File
@@ -10,7 +10,7 @@ if ($Port -le 0) {
if ($envPort) {
$Port = $envPort
} else {
$Port = 8080
$Port = 8000
}
}
+5
View File
@@ -90,6 +90,11 @@ std::expected<std::vector<std::string>, pdfengine::EngineError> WasmMockPage::ex
return std::vector<std::string>();
}
std::expected<double, pdfengine::EngineError> WasmMockPage::getGlyphWidth(const std::string& fontName, uint32_t charcode, double fontSize) const {
(void)fontName; (void)charcode;
return 6.0 * (fontSize / 12.0);
}
std::expected<std::vector<pdfengine::FontInfo>, pdfengine::EngineError> WasmMockPage::getFonts() const {
// Simulate a non-embedded Helvetica font being substituted with Liberation Sans.
// This is the canonical Phase 1 font substitution scenario.
+1
View File
@@ -21,6 +21,7 @@ public:
[[nodiscard]] std::expected<std::vector<pdfengine::GlyphBounds>, pdfengine::EngineError> extractTextWithBounds() const override;
[[nodiscard]] std::expected<std::vector<pdfengine::FontInfo>, pdfengine::EngineError> getFonts() const override;
[[nodiscard]] std::expected<std::vector<std::string>, pdfengine::EngineError> extractAnnotationsText() const override;
[[nodiscard]] std::expected<double, pdfengine::EngineError> getGlyphWidth(const std::string& fontName, uint32_t charcode, double fontSize) const override;
[[nodiscard]] pdfengine::DevicePoint pageToDevice(const pdfengine::Point2D& pagePoint, int deviceWidth, int deviceHeight, int rotate = 0) const noexcept override;
[[nodiscard]] pdfengine::Point2D deviceToPage(const pdfengine::DevicePoint& devicePoint, int deviceWidth, int deviceHeight, int rotate = 0) const noexcept override;