Merge pull request 'furqan' (#53) from furqan into dev
Reviewed-on: https://gitea.maskantech.in/gitea_admin/pdf/pulls/53
This commit is contained in:
@@ -81,6 +81,65 @@ std::vector<uint8_t> FontSubset::buildSubset(const std::vector<uint8_t>& origina
|
|||||||
return result;
|
return result;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
std::vector<uint8_t> FontSubset::buildSubsetByUnicode(const std::vector<uint8_t>& originalStream,
|
||||||
|
const std::vector<uint32_t>& codepoints) {
|
||||||
|
// Subset a font down to only the glyphs needed for `codepoints`, for embedding
|
||||||
|
// a substitute font compactly (Acrobat-style). Unlike buildSubset (GID-based,
|
||||||
|
// RETAIN_GIDS for existing subset streams), this drives the subset by UNICODE
|
||||||
|
// and lets HarfBuzz compact glyph ids + rebuild the cmap, so PDFium's
|
||||||
|
// FPDFText_SetText (unicode → glyph via cmap) maps correctly into the result.
|
||||||
|
std::vector<uint8_t> result;
|
||||||
|
if (originalStream.empty() || codepoints.empty()) {
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
|
hb_blob_t* blob = hb_blob_create(
|
||||||
|
reinterpret_cast<const char*>(originalStream.data()),
|
||||||
|
static_cast<unsigned int>(originalStream.size()),
|
||||||
|
HB_MEMORY_MODE_READONLY,
|
||||||
|
nullptr,
|
||||||
|
nullptr
|
||||||
|
);
|
||||||
|
hb_face_t* face = hb_face_create(blob, 0);
|
||||||
|
hb_blob_destroy(blob);
|
||||||
|
if (!face) {
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
|
hb_subset_input_t* input = hb_subset_input_create_or_fail();
|
||||||
|
if (!input) {
|
||||||
|
hb_face_destroy(face);
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
|
hb_set_t* unicodes = hb_subset_input_unicode_set(input);
|
||||||
|
for (uint32_t cp : codepoints) {
|
||||||
|
if (cp != 0) {
|
||||||
|
hb_set_add(unicodes, cp);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// No RETAIN_GIDS flag: compact the glyph ids and rebuild the cmap.
|
||||||
|
|
||||||
|
hb_face_t* subset_face = hb_subset_or_fail(face, input);
|
||||||
|
hb_subset_input_destroy(input);
|
||||||
|
hb_face_destroy(face);
|
||||||
|
|
||||||
|
if (subset_face) {
|
||||||
|
hb_blob_t* result_blob = hb_face_reference_blob(subset_face);
|
||||||
|
if (result_blob) {
|
||||||
|
unsigned int length = 0;
|
||||||
|
const char* data = hb_blob_get_data(result_blob, &length);
|
||||||
|
if (data && length > 0) {
|
||||||
|
result.assign(data, data + length);
|
||||||
|
}
|
||||||
|
hb_blob_destroy(result_blob);
|
||||||
|
}
|
||||||
|
hb_face_destroy(subset_face);
|
||||||
|
}
|
||||||
|
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
bool FontSubset::hasSubsetPrefix(const std::string& fontName) {
|
bool FontSubset::hasSubsetPrefix(const std::string& fontName) {
|
||||||
if (fontName.length() < 8) {
|
if (fontName.length() < 8) {
|
||||||
return false;
|
return false;
|
||||||
|
|||||||
@@ -22,6 +22,10 @@ public:
|
|||||||
// Rebuilds the TTF/CID stream keeping only the specified GIDs using HarfBuzz
|
// Rebuilds the TTF/CID stream keeping only the specified GIDs using HarfBuzz
|
||||||
static std::vector<uint8_t> buildSubset(const std::vector<uint8_t>& originalStream, const std::vector<uint32_t>& glyphIdsToKeep);
|
static std::vector<uint8_t> buildSubset(const std::vector<uint8_t>& originalStream, const std::vector<uint32_t>& glyphIdsToKeep);
|
||||||
|
|
||||||
|
// Subset a font by the UNICODE codepoints used (compacts GIDs + rebuilds cmap).
|
||||||
|
// Use when embedding a substitute font for newly-typed text.
|
||||||
|
static std::vector<uint8_t> buildSubsetByUnicode(const std::vector<uint8_t>& originalStream, const std::vector<uint32_t>& codepoints);
|
||||||
|
|
||||||
explicit FontSubset(const std::string& fontName);
|
explicit FontSubset(const std::string& fontName);
|
||||||
~FontSubset() = default;
|
~FontSubset() = default;
|
||||||
|
|
||||||
|
|||||||
@@ -15,6 +15,7 @@
|
|||||||
#include "fonts/pdf_fonts/font.hpp"
|
#include "fonts/pdf_fonts/font.hpp"
|
||||||
#include "pdfengine/hardened_limits.h"
|
#include "pdfengine/hardened_limits.h"
|
||||||
#include "fonts/pdf_fonts/font_fallback.hpp"
|
#include "fonts/pdf_fonts/font_fallback.hpp"
|
||||||
|
#include "fonts/pdf_fonts/font_subset.hpp"
|
||||||
#include "fonts/shaping/hb_shaper.hpp"
|
#include "fonts/shaping/hb_shaper.hpp"
|
||||||
#include "decoration_builder.hpp"
|
#include "decoration_builder.hpp"
|
||||||
|
|
||||||
@@ -2044,7 +2045,16 @@ std::expected<void, EngineError> PdfiumDocument::applyEdits(const std::string& e
|
|||||||
} else if (matchedFontInfo->isEmbedded && isSubsetFont && !subsetLacksGlyphs) {
|
} else if (matchedFontInfo->isEmbedded && isSubsetFont && !subsetLacksGlyphs) {
|
||||||
cacheKey = matchedFontInfo->internalFontId;
|
cacheKey = matchedFontInfo->internalFontId;
|
||||||
useEmbedded = true;
|
useEmbedded = true;
|
||||||
|
} else if (!matchedFontInfo->isEmbedded) {
|
||||||
|
// Tier 1: the original font is NOT embedded — by definition it's
|
||||||
|
// one of the standard base-14 fonts the viewer provides. Keep it
|
||||||
|
// that way: use the standard PDF-14 font, embed nothing (exactly
|
||||||
|
// what Acrobat does). Avoids ballooning the file with a full
|
||||||
|
// substitute font for a base-14 edit.
|
||||||
|
cacheKey = "standard_" + fontName;
|
||||||
} else {
|
} else {
|
||||||
|
// Embedded subset that can't render the new glyphs — substitute,
|
||||||
|
// but embed only a SUBSET (Tier 2, in the load path below).
|
||||||
cacheKey = "system_embed_" + matchedFontInfo->fontName + "_" + (bold ? "B" : "") + (italic ? "I" : "");
|
cacheKey = "system_embed_" + matchedFontInfo->fontName + "_" + (bold ? "B" : "") + (italic ? "I" : "");
|
||||||
useSystem = true;
|
useSystem = true;
|
||||||
}
|
}
|
||||||
@@ -2083,12 +2093,25 @@ std::expected<void, EngineError> PdfiumDocument::applyEdits(const std::string& e
|
|||||||
if (fs) {
|
if (fs) {
|
||||||
std::vector<uint8_t> fileBytes((std::istreambuf_iterator<char>(fs)), std::istreambuf_iterator<char>());
|
std::vector<uint8_t> fileBytes((std::istreambuf_iterator<char>(fs)), std::istreambuf_iterator<char>());
|
||||||
if (!fileBytes.empty()) {
|
if (!fileBytes.empty()) {
|
||||||
|
// Tier 2: embed only a SUBSET covering the glyphs actually
|
||||||
|
// used, not the whole multi-MB font file (Acrobat-style).
|
||||||
|
std::vector<uint8_t> subsetBytes =
|
||||||
|
fonts::pdf_fonts::FontSubset::buildSubsetByUnicode(fileBytes, unicodeCodepoints);
|
||||||
|
const size_t fullSize = fileBytes.size();
|
||||||
std::lock_guard<std::mutex> lock(loadedFontsMutex_);
|
std::lock_guard<std::mutex> lock(loadedFontsMutex_);
|
||||||
loadedFontDataBuffers_[cacheKey] = std::move(fileBytes);
|
loadedFontDataBuffers_[cacheKey] =
|
||||||
|
(!subsetBytes.empty()) ? std::move(subsetBytes) : std::move(fileBytes);
|
||||||
const auto& bytes = loadedFontDataBuffers_[cacheKey];
|
const auto& bytes = loadedFontDataBuffers_[cacheKey];
|
||||||
font = FPDFText_LoadFont(doc_, bytes.data(), static_cast<uint32_t>(bytes.size()), FPDF_FONT_TRUETYPE, false);
|
// Load as a composite CID font (cid=true): 2-byte codes
|
||||||
|
// + auto-generated ToUnicode, so full-Unicode edited text
|
||||||
|
// (€, ™, accents, CJK) stays searchable/extractable. A
|
||||||
|
// simple 1-byte font can only encode 256 codes and breaks
|
||||||
|
// text extraction for anything outside that range.
|
||||||
|
font = FPDFText_LoadFont(doc_, bytes.data(), static_cast<uint32_t>(bytes.size()), FPDF_FONT_TRUETYPE, true);
|
||||||
if (font) {
|
if (font) {
|
||||||
spdlog::info("Font Engine: Embedded system font '{}' from path '{}' (cache key: {})", matchedFontInfo->fontName, fontPath, cacheKey);
|
spdlog::info("Font Engine: Embedded {} CID system font '{}' ({} -> {} bytes) from '{}'",
|
||||||
|
bytes.size() < fullSize ? "SUBSET" : "FULL",
|
||||||
|
matchedFontInfo->fontName, fullSize, bytes.size(), fontPath);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
@@ -2375,182 +2398,6 @@ std::expected<void, EngineError> PdfiumDocument::applyEdits(const std::string& e
|
|||||||
return std::unexpected(EngineError::Unknown);
|
return std::unexpected(EngineError::Unknown);
|
||||||
}
|
}
|
||||||
|
|
||||||
FPDF_ClosePage(page);
|
|
||||||
} else if (type == "edit_text") {
|
|
||||||
// Rewrite an EXISTING text object in place (true content editing).
|
|
||||||
// Locate the target text object by bbox (no stable object id exists),
|
|
||||||
// replace its text, and horizontally squeeze it to stay within the
|
|
||||||
// original line bounds (line-level reflow only).
|
|
||||||
if (!op.contains("data") || !op["data"].is_object()) {
|
|
||||||
spdlog::error("edit_text operation missing 'data' object");
|
|
||||||
return std::unexpected(EngineError::InvalidFormat);
|
|
||||||
}
|
|
||||||
auto data = op["data"];
|
|
||||||
double x = data.value("x", 0.0);
|
|
||||||
double y = data.value("y", 0.0);
|
|
||||||
double width = data.value("width", 0.0);
|
|
||||||
double height = data.value("height", 0.0);
|
|
||||||
std::string newText = data.value("newText", "");
|
|
||||||
double fontSize = data.value("fontSize", 0.0);
|
|
||||||
const bool hasColor = data.contains("color") && data["color"].is_string();
|
|
||||||
std::string color = hasColor ? data["color"].get<std::string>() : "#000000";
|
|
||||||
std::string fallbackFont = data.value("fallbackFont", "");
|
|
||||||
|
|
||||||
if (newText.empty()) {
|
|
||||||
spdlog::warn("edit_text: empty newText — skipping (use redaction to delete text)");
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
if (width <= 0.0 || height <= 0.0) {
|
|
||||||
spdlog::error("edit_text: invalid target bbox ({}x{})", width, height);
|
|
||||||
return std::unexpected(EngineError::InvalidFormat);
|
|
||||||
}
|
|
||||||
|
|
||||||
FPDF_PAGE page = FPDF_LoadPage(doc_, pageIndex);
|
|
||||||
if (!page) {
|
|
||||||
spdlog::error("Failed to load page index {} for edit_text", pageIndex);
|
|
||||||
return std::unexpected(EngineError::Unknown);
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- locate best-matching text object by bbox overlap ---
|
|
||||||
const double tl = x, tb = y, tr = x + width, tt = y + height;
|
|
||||||
const double targetArea = width * height;
|
|
||||||
FPDF_PAGEOBJECT best = nullptr;
|
|
||||||
double bestScore = 0.0, secondScore = 0.0;
|
|
||||||
float bL = 0, bB = 0, bR = 0, bT = 0;
|
|
||||||
const int count = FPDFPage_CountObjects(page);
|
|
||||||
for (int i = 0; i < count; ++i) {
|
|
||||||
FPDF_PAGEOBJECT obj = FPDFPage_GetObject(page, i);
|
|
||||||
if (!obj || FPDFPageObj_GetType(obj) != FPDF_PAGEOBJ_TEXT) continue;
|
|
||||||
float l = 0, b = 0, r = 0, t = 0;
|
|
||||||
if (!FPDFPageObj_GetBounds(obj, &l, &b, &r, &t)) continue;
|
|
||||||
const double ix = (std::max)(0.0, (std::min)(static_cast<double>(r), tr) - (std::max)(static_cast<double>(l), tl));
|
|
||||||
const double iy = (std::max)(0.0, (std::min)(static_cast<double>(t), tt) - (std::max)(static_cast<double>(b), tb));
|
|
||||||
const double inter = ix * iy;
|
|
||||||
if (inter <= 0.0) continue;
|
|
||||||
const double objArea = (std::max)(1e-6, static_cast<double>(r - l) * static_cast<double>(t - b));
|
|
||||||
const double score = (std::max)(inter / (targetArea + objArea - inter), inter / objArea);
|
|
||||||
if (score > bestScore) {
|
|
||||||
secondScore = bestScore;
|
|
||||||
bestScore = score;
|
|
||||||
best = obj; bL = l; bB = b; bR = r; bT = t;
|
|
||||||
} else if (score > secondScore) {
|
|
||||||
secondScore = score;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (!best || bestScore < 0.30) {
|
|
||||||
spdlog::error("edit_text: no text object matches the target bbox (best score {:.3f})", bestScore);
|
|
||||||
FPDF_ClosePage(page);
|
|
||||||
return std::unexpected(EngineError::InvalidFormat);
|
|
||||||
}
|
|
||||||
if (bestScore - secondScore < 0.10) {
|
|
||||||
spdlog::error("edit_text: ambiguous target — overlapping text objects (best {:.3f}, second {:.3f})",
|
|
||||||
bestScore, secondScore);
|
|
||||||
FPDF_ClosePage(page);
|
|
||||||
return std::unexpected(EngineError::InvalidFormat);
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- capture original geometry BEFORE mutating ---
|
|
||||||
const double origLeft = bL, origBottom = bB, origWidth = static_cast<double>(bR - bL);
|
|
||||||
if (fontSize <= 0.0) fontSize = static_cast<double>(bT - bB);
|
|
||||||
if (fontSize <= 0.0) fontSize = 12.0;
|
|
||||||
FS_MATRIX m0{1, 0, 0, 1, 0, 0};
|
|
||||||
FPDFPageObj_GetMatrix(best, &m0);
|
|
||||||
const bool axisAligned = (std::abs(m0.b) < 1e-6 && std::abs(m0.c) < 1e-6);
|
|
||||||
|
|
||||||
// --- glyph-coverage check on the object's own font ---
|
|
||||||
auto decodeUtf8 = [](const std::string& s) {
|
|
||||||
std::vector<uint32_t> cps;
|
|
||||||
for (size_t i = 0; i < s.size();) {
|
|
||||||
unsigned char c = s[i];
|
|
||||||
uint32_t cp = 0; size_t extra = 0;
|
|
||||||
if (c < 0x80) { cp = c; extra = 0; }
|
|
||||||
else if ((c & 0xE0) == 0xC0) { cp = c & 0x1F; extra = 1; }
|
|
||||||
else if ((c & 0xF0) == 0xE0) { cp = c & 0x0F; extra = 2; }
|
|
||||||
else if ((c & 0xF8) == 0xF0) { cp = c & 0x07; extra = 3; }
|
|
||||||
else { i++; continue; }
|
|
||||||
if (i + extra >= s.size()) break;
|
|
||||||
for (size_t j = 1; j <= extra; ++j) cp = (cp << 6) | (s[i + j] & 0x3F);
|
|
||||||
cps.push_back(cp); i += extra + 1;
|
|
||||||
}
|
|
||||||
return cps;
|
|
||||||
};
|
|
||||||
bool needFallback = false;
|
|
||||||
FPDF_FONT objFont = FPDFTextObj_GetFont(best);
|
|
||||||
if (!fallbackFont.empty() && objFont) {
|
|
||||||
for (uint32_t cp : decodeUtf8(newText)) {
|
|
||||||
if (cp == ' ' || cp == '\t' || cp == '\n' || cp == '\r') continue;
|
|
||||||
float w = 0.0f;
|
|
||||||
if (!FPDFFont_GetGlyphWidth(objFont, cp, static_cast<float>(fontSize), &w) || w <= 0.0f) {
|
|
||||||
needFallback = true;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
FPDF_PAGEOBJECT target = best;
|
|
||||||
auto utf16 = utf8_to_utf16le(newText);
|
|
||||||
|
|
||||||
if (needFallback) {
|
|
||||||
// Original font can't render the new glyphs — recreate in a
|
|
||||||
// standard font at the same position (Acrobat-style substitution).
|
|
||||||
spdlog::info("edit_text: glyph coverage gap, substituting font '{}'", fallbackFont);
|
|
||||||
FPDFPage_RemoveObject(page, best);
|
|
||||||
FPDFPageObj_Destroy(best);
|
|
||||||
FPDF_FONT font = FPDFText_LoadStandardFont(doc_, fallbackFont.c_str());
|
|
||||||
if (!font) font = FPDFText_LoadStandardFont(doc_, "Helvetica");
|
|
||||||
target = FPDFPageObj_CreateTextObj(doc_, font, static_cast<float>(fontSize));
|
|
||||||
if (!target) {
|
|
||||||
FPDF_ClosePage(page);
|
|
||||||
return std::unexpected(EngineError::Unknown);
|
|
||||||
}
|
|
||||||
unsigned int r = 0, g = 0, b = 0;
|
|
||||||
parseHexColor(color, r, g, b);
|
|
||||||
FPDFPageObj_SetFillColor(target, r, g, b, 255);
|
|
||||||
if (!FPDFText_SetText(target, reinterpret_cast<FPDF_WIDESTRING>(utf16.data()))) {
|
|
||||||
FPDFPageObj_Destroy(target);
|
|
||||||
FPDF_ClosePage(page);
|
|
||||||
return std::unexpected(EngineError::Unknown);
|
|
||||||
}
|
|
||||||
FPDFPageObj_Transform(target, 1.0, 0.0, 0.0, 1.0, origLeft, origBottom);
|
|
||||||
FPDFPage_InsertObject(page, target);
|
|
||||||
} else {
|
|
||||||
if (hasColor) {
|
|
||||||
unsigned int r = 0, g = 0, b = 0;
|
|
||||||
parseHexColor(color, r, g, b);
|
|
||||||
FPDFPageObj_SetFillColor(target, r, g, b, 255);
|
|
||||||
}
|
|
||||||
if (!FPDFText_SetText(target, reinterpret_cast<FPDF_WIDESTRING>(utf16.data()))) {
|
|
||||||
spdlog::error("edit_text: FPDFText_SetText failed on the existing object");
|
|
||||||
FPDF_ClosePage(page);
|
|
||||||
return std::unexpected(EngineError::Unknown);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Regenerate so the new text's bounds are accurate, then squeeze to fit.
|
|
||||||
if (!FPDFPage_GenerateContent(page)) {
|
|
||||||
spdlog::error("edit_text: FPDFPage_GenerateContent failed");
|
|
||||||
FPDF_ClosePage(page);
|
|
||||||
return std::unexpected(EngineError::Unknown);
|
|
||||||
}
|
|
||||||
if (axisAligned && origWidth > 0.0) {
|
|
||||||
float nl = 0, nb = 0, nr = 0, nt = 0;
|
|
||||||
if (FPDFPageObj_GetBounds(target, &nl, &nb, &nr, &nt)) {
|
|
||||||
const double newWidth = static_cast<double>(nr - nl);
|
|
||||||
if (newWidth > origWidth && newWidth > 0.0) {
|
|
||||||
const double scaleX = origWidth / newWidth;
|
|
||||||
// Horizontal compression about the original left edge —
|
|
||||||
// keeps the line start, baseline, and font size, never overflows.
|
|
||||||
FPDFPageObj_Transform(target, scaleX, 0.0, 0.0, 1.0,
|
|
||||||
origLeft * (1.0 - scaleX), 0.0);
|
|
||||||
if (!FPDFPage_GenerateContent(page)) {
|
|
||||||
spdlog::error("edit_text: FPDFPage_GenerateContent failed after squeeze");
|
|
||||||
FPDF_ClosePage(page);
|
|
||||||
return std::unexpected(EngineError::Unknown);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
FPDF_ClosePage(page);
|
FPDF_ClosePage(page);
|
||||||
} else if (type == "update_field") {
|
} else if (type == "update_field") {
|
||||||
if (!op.contains("data") || !op["data"].is_object()) {
|
if (!op.contains("data") || !op["data"].is_object()) {
|
||||||
|
|||||||
@@ -295,14 +295,16 @@ function App() {
|
|||||||
setActiveTool('select');
|
setActiveTool('select');
|
||||||
};
|
};
|
||||||
|
|
||||||
// Rewrite existing page text in place. `run` is in PDF bottom-left points (the
|
// Rewrite existing page text in place via replace_text, which targets stable
|
||||||
// space getPageModel / the engine use), so no coordinate flip is needed here.
|
// page-object indices (no coordinates) and reflows the rest of the line.
|
||||||
const handleEditText = (pageIndex: number, run: EditableRun, newText: string) => {
|
const handleEditText = (pageIndex: number, run: EditableRun, newText: string) => {
|
||||||
applyOps([{
|
applyOps([{
|
||||||
id: rid('edit'), type: 'edit_text', pageIndex,
|
id: rid('edit'), type: 'replace_text', pageIndex,
|
||||||
data: {
|
data: {
|
||||||
x: run.x, y: run.y, width: run.w, height: run.h,
|
objectIndices: run.objectIndices,
|
||||||
newText, originalText: run.text, fontSize: run.fontSize, fallbackFont: 'Helvetica',
|
text: newText,
|
||||||
|
internalFontId: run.internalFontId,
|
||||||
|
fontSize: run.fontSize,
|
||||||
},
|
},
|
||||||
}], 'Text updated');
|
}], 'Text updated');
|
||||||
setActiveTool('select');
|
setActiveTool('select');
|
||||||
|
|||||||
@@ -193,20 +193,17 @@ export type EditOperationDataMap = {
|
|||||||
page_reorder: PageReorderData;
|
page_reorder: PageReorderData;
|
||||||
delete_annotation: DeleteAnnotationData;
|
delete_annotation: DeleteAnnotationData;
|
||||||
update_annotation: UpdateAnnotationData;
|
update_annotation: UpdateAnnotationData;
|
||||||
edit_text: EditTextData;
|
replace_text: ReplaceTextData;
|
||||||
};
|
};
|
||||||
|
|
||||||
export interface EditTextData {
|
// In-place rewrite of existing page text. Targets stable page-object indices
|
||||||
// Target run bbox in PDF bottom-left page space (same space getPageModel returns).
|
// (from getPageModel's run.object_indices) — no coordinates, so no Y-flip and no
|
||||||
x: number;
|
// bbox ambiguity. The engine reflows subsequent same-line text by the width delta.
|
||||||
y: number;
|
export interface ReplaceTextData {
|
||||||
width: number;
|
objectIndices: number[];
|
||||||
height: number;
|
text: string;
|
||||||
newText: string;
|
internalFontId: string;
|
||||||
originalText?: string;
|
fontSize: number;
|
||||||
fontSize?: number;
|
|
||||||
color?: string;
|
|
||||||
fallbackFont?: string;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
export interface DeleteAnnotationData {
|
export interface DeleteAnnotationData {
|
||||||
|
|||||||
@@ -1,7 +1,9 @@
|
|||||||
import React, { useEffect, useMemo, useRef, useState } from 'react';
|
import React, { useEffect, useMemo, useRef, useState } from 'react';
|
||||||
import { gatewayService } from '../lib/gatewayService';
|
import { gatewayService } from '../lib/gatewayService';
|
||||||
|
|
||||||
// A single editable run, in PDF BOTTOM-LEFT page space (as getPageModel returns).
|
// A single editable run. Geometry (x/y/w/h) is PDF BOTTOM-LEFT (as getPageModel
|
||||||
|
// returns) and is used only to position the inline editor. The replace_text op
|
||||||
|
// itself targets the stable objectIndices, not coordinates.
|
||||||
export interface EditableRun {
|
export interface EditableRun {
|
||||||
text: string;
|
text: string;
|
||||||
x: number;
|
x: number;
|
||||||
@@ -9,6 +11,8 @@ export interface EditableRun {
|
|||||||
w: number;
|
w: number;
|
||||||
h: number;
|
h: number;
|
||||||
fontSize: number;
|
fontSize: number;
|
||||||
|
objectIndices: number[];
|
||||||
|
internalFontId: string;
|
||||||
}
|
}
|
||||||
|
|
||||||
interface TextEditLayerProps {
|
interface TextEditLayerProps {
|
||||||
@@ -27,8 +31,16 @@ function flattenRuns(model: any): EditableRun[] {
|
|||||||
for (const p of paragraphs) {
|
for (const p of paragraphs) {
|
||||||
for (const line of p.lines ?? []) {
|
for (const line of p.lines ?? []) {
|
||||||
for (const r of line.runs ?? []) {
|
for (const r of line.runs ?? []) {
|
||||||
if (typeof r.text === 'string' && r.text.trim() && r.w > 0 && r.h > 0) {
|
const objectIndices: number[] = Array.isArray(r.object_indices) ? r.object_indices : [];
|
||||||
runs.push({ text: r.text, x: r.x, y: r.y, w: r.w, h: r.h, fontSize: r.font_size ?? r.h });
|
// Only runs backed by real page objects are editable (replace_text targets them).
|
||||||
|
if (typeof r.text === 'string' && r.text.trim() && r.w > 0 && r.h > 0 && objectIndices.length > 0) {
|
||||||
|
runs.push({
|
||||||
|
text: r.text,
|
||||||
|
x: r.x, y: r.y, w: r.w, h: r.h,
|
||||||
|
fontSize: r.font_size ?? r.h,
|
||||||
|
objectIndices,
|
||||||
|
internalFontId: r.internal_font_id ?? '',
|
||||||
|
});
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -168,26 +168,6 @@ class PageReorderOperation(BaseModel):
|
|||||||
data: PageReorderData
|
data: PageReorderData
|
||||||
|
|
||||||
|
|
||||||
class EditTextData(BaseModel):
|
|
||||||
# Target line/run bbox in PDF bottom-left page space (frontend does the Y-flip).
|
|
||||||
x: float
|
|
||||||
y: float
|
|
||||||
width: float = Field(..., gt=0)
|
|
||||||
height: float = Field(..., gt=0)
|
|
||||||
newText: str
|
|
||||||
originalText: str | None = None
|
|
||||||
fontSize: float | None = Field(default=None, gt=0)
|
|
||||||
color: str | None = None
|
|
||||||
fallbackFont: str = "Helvetica"
|
|
||||||
|
|
||||||
|
|
||||||
class EditTextOperation(BaseModel):
|
|
||||||
id: str
|
|
||||||
type: Literal["edit_text"]
|
|
||||||
pageIndex: int = Field(..., ge=0)
|
|
||||||
data: EditTextData
|
|
||||||
|
|
||||||
|
|
||||||
class UpdateFieldData(BaseModel):
|
class UpdateFieldData(BaseModel):
|
||||||
value: str | bool
|
value: str | bool
|
||||||
annotationId: str | None = None
|
annotationId: str | None = None
|
||||||
@@ -282,7 +262,6 @@ EditOperation = Annotated[
|
|||||||
| UpdateFieldOperation
|
| UpdateFieldOperation
|
||||||
| DeleteAnnotationOperation
|
| DeleteAnnotationOperation
|
||||||
| UpdateAnnotationOperation
|
| UpdateAnnotationOperation
|
||||||
| EditTextOperation
|
|
||||||
| ReplaceTextOperation
|
| ReplaceTextOperation
|
||||||
| UnderlineOperation
|
| UnderlineOperation
|
||||||
| StrikeoutOperation
|
| StrikeoutOperation
|
||||||
@@ -362,7 +341,7 @@ def apply_edits_impl(document_id: str, request: EditsRequest):
|
|||||||
|
|
||||||
# Redaction and in-place text rewrites mutate existing objects, which do
|
# Redaction and in-place text rewrites mutate existing objects, which do
|
||||||
# not round-trip cleanly through an incremental save — force a full save.
|
# not round-trip cleanly through an incremental save — force a full save.
|
||||||
full_save_types = {"redaction", "edit_text"}
|
full_save_types = {"redaction", "replace_text"}
|
||||||
needs_full = any(op.get("type") in full_save_types for op in req_dict.get("operations", []))
|
needs_full = any(op.get("type") in full_save_types for op in req_dict.get("operations", []))
|
||||||
new_bytes = doc.save_full() if needs_full else doc.save_incremental()
|
new_bytes = doc.save_full() if needs_full else doc.save_incremental()
|
||||||
|
|
||||||
|
|||||||
@@ -1,130 +0,0 @@
|
|||||||
#!/usr/bin/env python3
|
|
||||||
"""Round-trip tests for the `edit_text` op (line-level text rewriting).
|
|
||||||
|
|
||||||
Verifies that editing an existing text object: (1) actually changes the rendered
|
|
||||||
text, (2) keeps the result within the original line's horizontal bounds (the
|
|
||||||
acceptance criterion), and (3) fails closed when the target bbox matches nothing.
|
|
||||||
|
|
||||||
Run directly: gateway/.venv/Scripts/python.exe tests/edits/test_edit_text.py
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import json
|
|
||||||
import sys
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
ROOT = Path(__file__).resolve().parents[2]
|
|
||||||
sys.path.insert(0, str(ROOT / "gateway"))
|
|
||||||
import pdfengine # noqa: E402
|
|
||||||
|
|
||||||
CORPUS = ROOT / "corpus" / "fonts" / "utf-8.pdf"
|
|
||||||
EPS = 2.0 # points; tolerance for rounding / glyph metrics
|
|
||||||
|
|
||||||
|
|
||||||
def _load():
|
|
||||||
return pdfengine.PdfDocument.load_from_memory(CORPUS.read_bytes(), "")
|
|
||||||
|
|
||||||
|
|
||||||
def _runs(model):
|
|
||||||
return [r for p in model.paragraphs for l in p.lines for r in l.runs]
|
|
||||||
|
|
||||||
|
|
||||||
def _first_run(model):
|
|
||||||
return model.paragraphs[0].lines[0].runs[0]
|
|
||||||
|
|
||||||
|
|
||||||
def _edit_op(model, run, new_text: str):
|
|
||||||
# Engine model coords are PDFium-native BOTTOM-LEFT (y = bbox bottom), the same
|
|
||||||
# space the op + FPDFPageObj_GetBounds use — so no flip here.
|
|
||||||
return {
|
|
||||||
"version": "1.0",
|
|
||||||
"operations": [{
|
|
||||||
"id": "e1", "type": "edit_text", "pageIndex": 0,
|
|
||||||
"data": {
|
|
||||||
"x": run.x,
|
|
||||||
"y": run.y,
|
|
||||||
"width": run.w,
|
|
||||||
"height": run.h,
|
|
||||||
"newText": new_text,
|
|
||||||
"originalText": run.text,
|
|
||||||
"fontSize": run.font_size,
|
|
||||||
"fallbackFont": "Helvetica",
|
|
||||||
},
|
|
||||||
}],
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def test_roundtrip_replaces_text():
|
|
||||||
doc = _load()
|
|
||||||
model = doc.get_page(0).extract_document_model()
|
|
||||||
run = _first_run(model)
|
|
||||||
orig_x, orig_w, orig_y = run.x, run.w, run.y
|
|
||||||
doc.apply_edits(json.dumps(_edit_op(model, run, "REPLACED")))
|
|
||||||
out = doc.save_full()
|
|
||||||
|
|
||||||
m2 = pdfengine.PdfDocument.load_from_memory(out, "").get_page(0).extract_document_model()
|
|
||||||
edited = next((r for r in _runs(m2) if "REPLACED" in r.text), None)
|
|
||||||
assert edited is not None, "edited text 'REPLACED' not found after round-trip"
|
|
||||||
# stays within the original horizontal bounds
|
|
||||||
assert edited.x >= orig_x - EPS, f"left moved out of bounds: {edited.x} < {orig_x}"
|
|
||||||
assert edited.x + edited.w <= orig_x + orig_w + EPS, \
|
|
||||||
f"right overflowed bounds: {edited.x + edited.w} > {orig_x + orig_w}"
|
|
||||||
# baseline / vertical position preserved
|
|
||||||
assert abs(edited.y - orig_y) < EPS + run.h, f"baseline drifted: {edited.y} vs {orig_y}"
|
|
||||||
print(f" ok roundtrip: '{run.text[:20]}' -> '{edited.text[:20]}' "
|
|
||||||
f"(x {orig_x:.0f}..{orig_x + orig_w:.0f} -> {edited.x:.0f}..{edited.x + edited.w:.0f})")
|
|
||||||
|
|
||||||
|
|
||||||
def test_long_text_squeezes_within_bounds():
|
|
||||||
doc = _load()
|
|
||||||
model = doc.get_page(0).extract_document_model()
|
|
||||||
run = _first_run(model)
|
|
||||||
orig_right = run.x + run.w
|
|
||||||
long_text = "This is a deliberately very long replacement string to force squeeze"
|
|
||||||
doc.apply_edits(json.dumps(_edit_op(model, run, long_text)))
|
|
||||||
out = doc.save_full()
|
|
||||||
|
|
||||||
m2 = pdfengine.PdfDocument.load_from_memory(out, "").get_page(0).extract_document_model()
|
|
||||||
edited = next((r for r in _runs(m2) if "deliberately" in r.text), None)
|
|
||||||
assert edited is not None, "long edited text not found"
|
|
||||||
assert edited.x + edited.w <= orig_right + EPS, \
|
|
||||||
f"long text overflowed: {edited.x + edited.w:.1f} > {orig_right:.1f}"
|
|
||||||
print(f" ok squeeze: {len(long_text)} chars fit in {run.w:.0f}pt "
|
|
||||||
f"(right {edited.x + edited.w:.0f} <= {orig_right:.0f})")
|
|
||||||
|
|
||||||
|
|
||||||
def test_no_match_fails():
|
|
||||||
doc = _load()
|
|
||||||
model = doc.get_page(0).extract_document_model()
|
|
||||||
op = {
|
|
||||||
"version": "1.0",
|
|
||||||
"operations": [{
|
|
||||||
"id": "e1", "type": "edit_text", "pageIndex": 0,
|
|
||||||
"data": {"x": 5000.0, "y": 5000.0, "width": 50.0, "height": 10.0,
|
|
||||||
"newText": "X", "fallbackFont": "Helvetica"},
|
|
||||||
}],
|
|
||||||
}
|
|
||||||
try:
|
|
||||||
doc.apply_edits(json.dumps(op))
|
|
||||||
except Exception:
|
|
||||||
print(" ok no-match correctly raised")
|
|
||||||
return
|
|
||||||
raise AssertionError("edit_text with an off-page bbox should have failed, but succeeded")
|
|
||||||
|
|
||||||
|
|
||||||
def main() -> int:
|
|
||||||
tests = [test_roundtrip_replaces_text, test_long_text_squeezes_within_bounds, test_no_match_fails]
|
|
||||||
failed = 0
|
|
||||||
for t in tests:
|
|
||||||
try:
|
|
||||||
t()
|
|
||||||
except AssertionError as exc:
|
|
||||||
print(f" FAIL {t.__name__}: {exc}")
|
|
||||||
failed += 1
|
|
||||||
print(f"\n{len(tests) - failed}/{len(tests)} passed.")
|
|
||||||
return 1 if failed else 0
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
raise SystemExit(main())
|
|
||||||
@@ -0,0 +1,103 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Round-trip tests for the `replace_text` op (line-level text rewriting).
|
||||||
|
|
||||||
|
`replace_text` targets stable page-object indices (from the document model's
|
||||||
|
run.object_indices) rather than coordinates, copies the original styling, and
|
||||||
|
reflows the rest of the line by the width delta. Verifies that an edit actually
|
||||||
|
replaces the text and round-trips through save/reload.
|
||||||
|
|
||||||
|
Run: gateway/.venv/Scripts/python.exe tests/edits/test_replace_text.py
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
ROOT = Path(__file__).resolve().parents[2]
|
||||||
|
sys.path.insert(0, str(ROOT / "gateway"))
|
||||||
|
import pdfengine # noqa: E402
|
||||||
|
|
||||||
|
CORPUS = ROOT / "corpus" / "fonts" / "utf-8.pdf"
|
||||||
|
|
||||||
|
|
||||||
|
def _load():
|
||||||
|
return pdfengine.PdfDocument.load_from_memory(CORPUS.read_bytes(), "")
|
||||||
|
|
||||||
|
|
||||||
|
def _runs(model):
|
||||||
|
return [r for p in model.paragraphs for l in p.lines for r in l.runs]
|
||||||
|
|
||||||
|
|
||||||
|
def _alltext(model):
|
||||||
|
return " ".join(r.text for r in _runs(model))
|
||||||
|
|
||||||
|
|
||||||
|
def _first_editable(model):
|
||||||
|
for r in _runs(model):
|
||||||
|
if list(r.object_indices):
|
||||||
|
return r
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _op(run, new_text):
|
||||||
|
return {"version": "1.0", "operations": [{
|
||||||
|
"id": "r1", "type": "replace_text", "pageIndex": 0,
|
||||||
|
"data": {"objectIndices": list(run.object_indices), "text": new_text,
|
||||||
|
"internalFontId": run.internal_font_id, "fontSize": run.font_size},
|
||||||
|
}]}
|
||||||
|
|
||||||
|
|
||||||
|
def test_roundtrip_replaces_text():
|
||||||
|
doc = _load()
|
||||||
|
run = _first_editable(doc.get_page(0).extract_document_model())
|
||||||
|
assert run is not None, "no run with object_indices to edit"
|
||||||
|
original_fragment = run.text.strip().split()[0]
|
||||||
|
doc.apply_edits(json.dumps(_op(run, "REPLACED LINE")))
|
||||||
|
out = doc.save_full()
|
||||||
|
text2 = _alltext(pdfengine.PdfDocument.load_from_memory(out, "").get_page(0).extract_document_model())
|
||||||
|
assert "REPLACED LINE" in text2, "new text not found after round-trip"
|
||||||
|
assert original_fragment not in text2, f"original text '{original_fragment}' should be gone"
|
||||||
|
print(f" ok roundtrip: replaced run (objs {list(run.object_indices)}) -> 'REPLACED LINE'")
|
||||||
|
|
||||||
|
|
||||||
|
def test_reflow_keeps_other_text():
|
||||||
|
# Replacing the first run must not destroy text on other lines.
|
||||||
|
doc = _load()
|
||||||
|
model = doc.get_page(0).extract_document_model()
|
||||||
|
run = _first_editable(model)
|
||||||
|
other_lines = [r.text for r in _runs(model) if list(r.object_indices) != list(run.object_indices)]
|
||||||
|
doc.apply_edits(json.dumps(_op(run, "X")))
|
||||||
|
text2 = _alltext(pdfengine.PdfDocument.load_from_memory(doc.save_full(), "").get_page(0).extract_document_model())
|
||||||
|
survived = sum(1 for t in other_lines if t.strip() and t.strip() in text2)
|
||||||
|
assert survived >= max(1, len(other_lines) // 2), "too much surrounding text lost"
|
||||||
|
print(f" ok reflow: {survived}/{len(other_lines)} other runs preserved")
|
||||||
|
|
||||||
|
|
||||||
|
def test_empty_indices_is_noop():
|
||||||
|
doc = _load()
|
||||||
|
before = _alltext(doc.get_page(0).extract_document_model())
|
||||||
|
doc.apply_edits(json.dumps({"version": "1.0", "operations": [{
|
||||||
|
"id": "r1", "type": "replace_text", "pageIndex": 0,
|
||||||
|
"data": {"objectIndices": [], "text": "X", "internalFontId": "", "fontSize": 12.0}}]}))
|
||||||
|
after = _alltext(pdfengine.PdfDocument.load_from_memory(doc.save_full(), "").get_page(0).extract_document_model())
|
||||||
|
assert before == after, "empty objectIndices should be a no-op"
|
||||||
|
print(" ok empty objectIndices is a safe no-op")
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> int:
|
||||||
|
tests = [test_roundtrip_replaces_text, test_reflow_keeps_other_text, test_empty_indices_is_noop]
|
||||||
|
failed = 0
|
||||||
|
for t in tests:
|
||||||
|
try:
|
||||||
|
t()
|
||||||
|
except AssertionError as exc:
|
||||||
|
print(f" FAIL {t.__name__}: {exc}")
|
||||||
|
failed += 1
|
||||||
|
print(f"\n{len(tests) - failed}/{len(tests)} passed.")
|
||||||
|
return 1 if failed else 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
raise SystemExit(main())
|
||||||
Reference in New Issue
Block a user