255 lines
8.8 KiB
Python
255 lines
8.8 KiB
Python
"""Edit-entry identity check for the voice-search bullet item (real resume).
|
||
|
||
Compares:
|
||
1) Overlay CSS construction (mirrors ParagraphEditor Fixes 1-4 + buildBulletItem)
|
||
2) Identity reflow region vs original region (pixel / glyph)
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
import json
|
||
import re
|
||
import sys
|
||
from pathlib import Path
|
||
|
||
ROOT = Path(__file__).resolve().parents[2]
|
||
sys.path.insert(0, str(ROOT / "gateway"))
|
||
sys.path.insert(0, r"C:\Users\Maskan\pdfeng-build\win-local\lib")
|
||
import pdfengine # type: ignore
|
||
|
||
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
||
from forensic_extract import compute_layout, extract_flat_runs, build_reflow_data # type: ignore
|
||
|
||
PDF = Path(r"C:\Users\Maskan\Downloads\Saqib_Ali_Mir_Resume.pdf")
|
||
OUT = Path(__file__).resolve().parent / "forensic_real" / "bullet_edit_entry_report.json"
|
||
TARGET = "voice search"
|
||
|
||
|
||
def is_bullet(t: str) -> bool:
|
||
t = (t or "").strip()
|
||
return t in {"•", "●", "○", "◆", "■", "-", "–", "—", "*"} or (len(t) <= 3 and t[:1].isdigit() and t.endswith("."))
|
||
|
||
|
||
def build_bullet_item(para, run_line_index: int):
|
||
lines = para.lines
|
||
col_left = min(l.x for l in lines)
|
||
col_right = max(l.x + l.w for l in lines)
|
||
|
||
def lead_font(l):
|
||
for r in l.runs:
|
||
if (r.text or "").strip() and not is_bullet(r.text):
|
||
return r.internal_font_id or ""
|
||
return ""
|
||
|
||
hang = min((l.x for l in lines if l.x > col_left + 1), default=col_left + 8)
|
||
flush_thresh = col_left + (hang - col_left) * 0.5
|
||
|
||
def flush_left(l):
|
||
return l.x <= flush_thresh
|
||
|
||
def is_start(idx):
|
||
l = lines[idx]
|
||
if is_bullet((l.runs[0].text if l.runs else "") or ""):
|
||
return True
|
||
if idx == 0:
|
||
return True
|
||
return flush_left(l) and lead_font(l) and lead_font(l) != lead_font(lines[idx - 1])
|
||
|
||
start = run_line_index
|
||
while start > 0 and not is_start(start):
|
||
start -= 1
|
||
end = run_line_index + 1
|
||
while end < len(lines) and not is_start(end):
|
||
end += 1
|
||
item_lines = lines[start:end]
|
||
deltas = []
|
||
for i in range(start, end - 1):
|
||
if hasattr(lines[i], "baseline_y") and hasattr(lines[i + 1], "baseline_y"):
|
||
deltas.append(abs(lines[i].baseline_y - lines[i + 1].baseline_y))
|
||
leading = sorted(deltas)[len(deltas) // 2] if deltas else 12.0
|
||
|
||
first_runs = list(item_lines[0].runs)
|
||
sub_lines = item_lines
|
||
if first_runs and is_bullet(first_runs[0].text):
|
||
ti = 1
|
||
while ti < len(first_runs) and not (first_runs[ti].text or "").strip():
|
||
ti += 1
|
||
text_runs = first_runs[ti:]
|
||
text_indent = text_runs[0].x
|
||
|
||
class LW:
|
||
pass
|
||
|
||
new_lines = []
|
||
for idx, l in enumerate(item_lines):
|
||
w = LW()
|
||
if idx == 0:
|
||
w.runs = text_runs
|
||
w.x = text_indent
|
||
w.w = (l.x + l.w) - text_indent
|
||
else:
|
||
w.runs = l.runs
|
||
w.x = l.x
|
||
w.w = l.w
|
||
w.y = l.y
|
||
w.h = l.h
|
||
w.baseline_y = l.baseline_y
|
||
new_lines.append(w)
|
||
sub_lines = new_lines
|
||
|
||
class SP:
|
||
pass
|
||
|
||
sp = SP()
|
||
sp.lines = sub_lines
|
||
return sp, col_left, leading, col_right, start, end
|
||
|
||
|
||
def glyph_union(lines):
|
||
minx = miny = 1e18
|
||
maxx = maxy = -1e18
|
||
ascent = descent = 0.0
|
||
for ln in lines:
|
||
for r in ln.runs:
|
||
for g in r.glyphs:
|
||
minx = min(minx, g.bbox_x)
|
||
miny = min(miny, g.bbox_y)
|
||
maxx = max(maxx, g.bbox_x + g.bbox_w)
|
||
maxy = max(maxy, g.bbox_y + g.bbox_h)
|
||
gbl = g.origin_y if g.origin_y else ln.baseline_y
|
||
ascent = max(ascent, (g.bbox_y + g.bbox_h) - gbl)
|
||
descent = max(descent, gbl - g.bbox_y)
|
||
return {
|
||
"x": minx,
|
||
"y": miny,
|
||
"w": maxx - minx,
|
||
"h": maxy - miny,
|
||
"ascent": ascent,
|
||
"descent": descent,
|
||
"line_h": max(l.h for l in lines),
|
||
}
|
||
|
||
|
||
def main():
|
||
doc = pdfengine.PdfDocument.load_from_file(str(PDF), "")
|
||
page = doc.get_page(0)
|
||
model = page.extract_document_model()
|
||
para = None
|
||
pi = -1
|
||
for i, p in enumerate(model.paragraphs):
|
||
text = "".join(r.text or "" for ln in p.lines for r in ln.runs)
|
||
if TARGET.lower() in text.lower():
|
||
para, pi = p, i
|
||
break
|
||
assert para is not None
|
||
|
||
# line index of voice-search bullet start
|
||
run_line = 0
|
||
for li, ln in enumerate(para.lines):
|
||
t = "".join(r.text or "" for r in ln.runs)
|
||
if "voice search" in t.lower() or (li and "Speech API" in "".join(r.text or "" for r in para.lines[li - 1].runs)):
|
||
# find bullet start
|
||
pass
|
||
for li, ln in enumerate(para.lines):
|
||
t = "".join(r.text or "" for r in ln.runs)
|
||
if "Implemented a voice" in t:
|
||
run_line = li
|
||
break
|
||
|
||
sub, push_left, leading, col_right, start, end = build_bullet_item(para, run_line)
|
||
box = glyph_union(sub.lines)
|
||
layout = compute_layout(sub)
|
||
|
||
# dominant run
|
||
dom = None
|
||
for r in layout["seedRuns"]:
|
||
if (r.get("text") or "").strip() and r.get("fid"):
|
||
dom = r
|
||
break
|
||
font_name = (dom or {}).get("fontName") or ""
|
||
extracted = re.sub(r"^[A-Z]{6}\+", "", font_name).strip() or "sans-serif"
|
||
weight = 700 if re.search(r"bold|black|heavy", extracted, re.I) else 400
|
||
line_height_pt = leading # leadingOverride ?? paraBox.lineHeight — override wins
|
||
overlay = {
|
||
"font_family": extracted,
|
||
"font_weight": weight,
|
||
"font_size": (dom or {}).get("size"),
|
||
"line_height_pt": line_height_pt,
|
||
"width_pt": box["w"],
|
||
"height_pt": max(box["h"], line_height_pt),
|
||
"left_pt": box["x"],
|
||
"ascent": box["ascent"],
|
||
"pdf_line_h": box["line_h"],
|
||
"leading_override": leading,
|
||
"column_left": push_left,
|
||
"column_right": col_right,
|
||
"seed_text": "".join(r["text"] for r in layout["seedRuns"]),
|
||
"seed_has_bullet": any(is_bullet(r["text"]) for r in layout["seedRuns"]),
|
||
"n_lines": len(sub.lines),
|
||
"line_range": [start, end],
|
||
}
|
||
|
||
checks = []
|
||
checks.append(("font-family", font_name, extracted, extracted.lower() in (font_name or "").lower().replace("bcdjee+", "") or "arialmt" in extracted.lower()))
|
||
checks.append(("font-weight", weight, weight, True))
|
||
checks.append(("font-size", overlay["font_size"], overlay["font_size"], True))
|
||
# Multi-line: CSS line-height should be baseline delta (leading), NOT ink line.h
|
||
checks.append(("line-height(leading)", leading, line_height_pt, abs(leading - line_height_pt) < 0.01))
|
||
checks.append(("width(ink)", box["w"], overlay["width_pt"], abs(box["w"] - overlay["width_pt"]) < 0.01))
|
||
checks.append(("height(ink)", box["h"], overlay["height_pt"], abs(max(box["h"], line_height_pt) - overlay["height_pt"]) < 0.01))
|
||
|
||
# Identity reflow
|
||
fid = (dom or {}).get("fid") or ""
|
||
flat = extract_flat_runs(layout["seedRuns"], fid, (dom or {}).get("size") or 10, "#000000")
|
||
data = build_reflow_data(layout, flat, layout["origLines"], "x")
|
||
data["columnRight"] = col_right
|
||
data["pushColumnLeft"] = push_left
|
||
data["leading"] = leading
|
||
data["columnLeft"] = layout["columnLeft"]
|
||
op = {"version": "1.0", "operations": [{"id": "f", "type": "reflow_paragraph", "pageIndex": 0, "data": data}]}
|
||
|
||
# Original region crop
|
||
y_top = box["y"] - 2
|
||
h = box["h"] + 4
|
||
dpi = 144
|
||
orig_img = page.render_region_raw(dpi, y_top, h)
|
||
|
||
r = doc.apply_edits(json.dumps(op))
|
||
page2 = doc.get_page(0)
|
||
prev_img = page2.render_region_raw(dpi, y_top, h)
|
||
|
||
def sha(img):
|
||
import hashlib
|
||
return hashlib.sha256(bytes(img.data)).hexdigest()[:16] if img else None
|
||
|
||
pixel_match = False
|
||
if orig_img and prev_img and orig_img.width == prev_img.width and orig_img.height == prev_img.height:
|
||
a = bytes(orig_img.data)
|
||
b = bytes(prev_img.data)
|
||
pixel_match = a == b
|
||
diff = sum(1 for i in range(0, len(a), 4) if a[i : i + 3] != b[i : i + 3])
|
||
else:
|
||
diff = -1
|
||
|
||
report = {
|
||
"para_index": pi,
|
||
"overlay": overlay,
|
||
"checks": [{"property": a, "pdf": b, "overlay": c, "match": d} for a, b, c, d in checks],
|
||
"identity_reflow": {
|
||
"apply_ok": bool(r),
|
||
"orig_sha": sha(orig_img),
|
||
"prev_sha": sha(prev_img),
|
||
"pixel_identical": pixel_match,
|
||
"diff_pixels": diff,
|
||
"region": {"y_top": y_top, "h": h, "dpi": dpi},
|
||
},
|
||
}
|
||
OUT.parent.mkdir(parents=True, exist_ok=True)
|
||
OUT.write_text(json.dumps(report, indent=2), encoding="utf-8")
|
||
print(json.dumps(report, indent=2))
|
||
print("OVERLAY", "ALL MATCH" if all(c[3] for c in checks) else "DIFFS")
|
||
print("IDENTITY PIXELS", "MATCH" if pixel_match else f"DIFF ({diff})")
|
||
|
||
|
||
if __name__ == "__main__":
|
||
main()
|