Files
pdf/tests/edits/forensic_bullet_edit_entry.py
T
2026-07-31 10:50:37 +05:30

255 lines
8.8 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Edit-entry identity check for the voice-search bullet item (real resume).
Compares:
1) Overlay CSS construction (mirrors ParagraphEditor Fixes 1-4 + buildBulletItem)
2) Identity reflow region vs original region (pixel / glyph)
"""
from __future__ import annotations
import json
import re
import sys
from pathlib import Path
ROOT = Path(__file__).resolve().parents[2]
sys.path.insert(0, str(ROOT / "gateway"))
sys.path.insert(0, r"C:\Users\Maskan\pdfeng-build\win-local\lib")
import pdfengine # type: ignore
sys.path.insert(0, str(Path(__file__).resolve().parent))
from forensic_extract import compute_layout, extract_flat_runs, build_reflow_data # type: ignore
PDF = Path(r"C:\Users\Maskan\Downloads\Saqib_Ali_Mir_Resume.pdf")
OUT = Path(__file__).resolve().parent / "forensic_real" / "bullet_edit_entry_report.json"
TARGET = "voice search"
def is_bullet(t: str) -> bool:
t = (t or "").strip()
return t in {"•", "●", "○", "◆", "■", "-", "", "—", "*"} or (len(t) <= 3 and t[:1].isdigit() and t.endswith("."))
def build_bullet_item(para, run_line_index: int):
lines = para.lines
col_left = min(l.x for l in lines)
col_right = max(l.x + l.w for l in lines)
def lead_font(l):
for r in l.runs:
if (r.text or "").strip() and not is_bullet(r.text):
return r.internal_font_id or ""
return ""
hang = min((l.x for l in lines if l.x > col_left + 1), default=col_left + 8)
flush_thresh = col_left + (hang - col_left) * 0.5
def flush_left(l):
return l.x <= flush_thresh
def is_start(idx):
l = lines[idx]
if is_bullet((l.runs[0].text if l.runs else "") or ""):
return True
if idx == 0:
return True
return flush_left(l) and lead_font(l) and lead_font(l) != lead_font(lines[idx - 1])
start = run_line_index
while start > 0 and not is_start(start):
start -= 1
end = run_line_index + 1
while end < len(lines) and not is_start(end):
end += 1
item_lines = lines[start:end]
deltas = []
for i in range(start, end - 1):
if hasattr(lines[i], "baseline_y") and hasattr(lines[i + 1], "baseline_y"):
deltas.append(abs(lines[i].baseline_y - lines[i + 1].baseline_y))
leading = sorted(deltas)[len(deltas) // 2] if deltas else 12.0
first_runs = list(item_lines[0].runs)
sub_lines = item_lines
if first_runs and is_bullet(first_runs[0].text):
ti = 1
while ti < len(first_runs) and not (first_runs[ti].text or "").strip():
ti += 1
text_runs = first_runs[ti:]
text_indent = text_runs[0].x
class LW:
pass
new_lines = []
for idx, l in enumerate(item_lines):
w = LW()
if idx == 0:
w.runs = text_runs
w.x = text_indent
w.w = (l.x + l.w) - text_indent
else:
w.runs = l.runs
w.x = l.x
w.w = l.w
w.y = l.y
w.h = l.h
w.baseline_y = l.baseline_y
new_lines.append(w)
sub_lines = new_lines
class SP:
pass
sp = SP()
sp.lines = sub_lines
return sp, col_left, leading, col_right, start, end
def glyph_union(lines):
minx = miny = 1e18
maxx = maxy = -1e18
ascent = descent = 0.0
for ln in lines:
for r in ln.runs:
for g in r.glyphs:
minx = min(minx, g.bbox_x)
miny = min(miny, g.bbox_y)
maxx = max(maxx, g.bbox_x + g.bbox_w)
maxy = max(maxy, g.bbox_y + g.bbox_h)
gbl = g.origin_y if g.origin_y else ln.baseline_y
ascent = max(ascent, (g.bbox_y + g.bbox_h) - gbl)
descent = max(descent, gbl - g.bbox_y)
return {
"x": minx,
"y": miny,
"w": maxx - minx,
"h": maxy - miny,
"ascent": ascent,
"descent": descent,
"line_h": max(l.h for l in lines),
}
def main():
doc = pdfengine.PdfDocument.load_from_file(str(PDF), "")
page = doc.get_page(0)
model = page.extract_document_model()
para = None
pi = -1
for i, p in enumerate(model.paragraphs):
text = "".join(r.text or "" for ln in p.lines for r in ln.runs)
if TARGET.lower() in text.lower():
para, pi = p, i
break
assert para is not None
# line index of voice-search bullet start
run_line = 0
for li, ln in enumerate(para.lines):
t = "".join(r.text or "" for r in ln.runs)
if "voice search" in t.lower() or (li and "Speech API" in "".join(r.text or "" for r in para.lines[li - 1].runs)):
# find bullet start
pass
for li, ln in enumerate(para.lines):
t = "".join(r.text or "" for r in ln.runs)
if "Implemented a voice" in t:
run_line = li
break
sub, push_left, leading, col_right, start, end = build_bullet_item(para, run_line)
box = glyph_union(sub.lines)
layout = compute_layout(sub)
# dominant run
dom = None
for r in layout["seedRuns"]:
if (r.get("text") or "").strip() and r.get("fid"):
dom = r
break
font_name = (dom or {}).get("fontName") or ""
extracted = re.sub(r"^[A-Z]{6}\+", "", font_name).strip() or "sans-serif"
weight = 700 if re.search(r"bold|black|heavy", extracted, re.I) else 400
line_height_pt = leading # leadingOverride ?? paraBox.lineHeight — override wins
overlay = {
"font_family": extracted,
"font_weight": weight,
"font_size": (dom or {}).get("size"),
"line_height_pt": line_height_pt,
"width_pt": box["w"],
"height_pt": max(box["h"], line_height_pt),
"left_pt": box["x"],
"ascent": box["ascent"],
"pdf_line_h": box["line_h"],
"leading_override": leading,
"column_left": push_left,
"column_right": col_right,
"seed_text": "".join(r["text"] for r in layout["seedRuns"]),
"seed_has_bullet": any(is_bullet(r["text"]) for r in layout["seedRuns"]),
"n_lines": len(sub.lines),
"line_range": [start, end],
}
checks = []
checks.append(("font-family", font_name, extracted, extracted.lower() in (font_name or "").lower().replace("bcdjee+", "") or "arialmt" in extracted.lower()))
checks.append(("font-weight", weight, weight, True))
checks.append(("font-size", overlay["font_size"], overlay["font_size"], True))
# Multi-line: CSS line-height should be baseline delta (leading), NOT ink line.h
checks.append(("line-height(leading)", leading, line_height_pt, abs(leading - line_height_pt) < 0.01))
checks.append(("width(ink)", box["w"], overlay["width_pt"], abs(box["w"] - overlay["width_pt"]) < 0.01))
checks.append(("height(ink)", box["h"], overlay["height_pt"], abs(max(box["h"], line_height_pt) - overlay["height_pt"]) < 0.01))
# Identity reflow
fid = (dom or {}).get("fid") or ""
flat = extract_flat_runs(layout["seedRuns"], fid, (dom or {}).get("size") or 10, "#000000")
data = build_reflow_data(layout, flat, layout["origLines"], "x")
data["columnRight"] = col_right
data["pushColumnLeft"] = push_left
data["leading"] = leading
data["columnLeft"] = layout["columnLeft"]
op = {"version": "1.0", "operations": [{"id": "f", "type": "reflow_paragraph", "pageIndex": 0, "data": data}]}
# Original region crop
y_top = box["y"] - 2
h = box["h"] + 4
dpi = 144
orig_img = page.render_region_raw(dpi, y_top, h)
r = doc.apply_edits(json.dumps(op))
page2 = doc.get_page(0)
prev_img = page2.render_region_raw(dpi, y_top, h)
def sha(img):
import hashlib
return hashlib.sha256(bytes(img.data)).hexdigest()[:16] if img else None
pixel_match = False
if orig_img and prev_img and orig_img.width == prev_img.width and orig_img.height == prev_img.height:
a = bytes(orig_img.data)
b = bytes(prev_img.data)
pixel_match = a == b
diff = sum(1 for i in range(0, len(a), 4) if a[i : i + 3] != b[i : i + 3])
else:
diff = -1
report = {
"para_index": pi,
"overlay": overlay,
"checks": [{"property": a, "pdf": b, "overlay": c, "match": d} for a, b, c, d in checks],
"identity_reflow": {
"apply_ok": bool(r),
"orig_sha": sha(orig_img),
"prev_sha": sha(prev_img),
"pixel_identical": pixel_match,
"diff_pixels": diff,
"region": {"y_top": y_top, "h": h, "dpi": dpi},
},
}
OUT.parent.mkdir(parents=True, exist_ok=True)
OUT.write_text(json.dumps(report, indent=2), encoding="utf-8")
print(json.dumps(report, indent=2))
print("OVERLAY", "ALL MATCH" if all(c[3] for c in checks) else "DIFFS")
print("IDENTITY PIXELS", "MATCH" if pixel_match else f"DIFF ({diff})")
if __name__ == "__main__":
main()