449 lines
14 KiB
Python
449 lines
14 KiB
Python
from fastapi import APIRouter, File, HTTPException, UploadFile, status
|
|
from pydantic import BaseModel
|
|
|
|
from app.services import engine
|
|
from app.services.store import document_store
|
|
|
|
router = APIRouter(prefix="/documents", tags=["documents"])
|
|
|
|
|
|
class PageInfoResponse(BaseModel):
|
|
index: int
|
|
width: float
|
|
height: float
|
|
|
|
|
|
class DocumentInfoResponse(BaseModel):
|
|
id: str
|
|
filename: str
|
|
sizeBytes: int
|
|
totalPages: int
|
|
pageWidth: float
|
|
pageHeight: float
|
|
uploadedAt: str
|
|
status: str
|
|
pages: list[PageInfoResponse] = []
|
|
|
|
def make_document_response(d: dict) -> DocumentInfoResponse:
|
|
pages_list = []
|
|
if "doc_instance" in d:
|
|
doc = d["doc_instance"]
|
|
for i in range(doc.page_count):
|
|
try:
|
|
page = doc.get_page(i)
|
|
pages_list.append(PageInfoResponse(index=i, width=page.width, height=page.height))
|
|
except Exception:
|
|
pass
|
|
return DocumentInfoResponse(
|
|
id=d["id"],
|
|
filename=d["filename"],
|
|
sizeBytes=d["sizeBytes"],
|
|
totalPages=d["totalPages"],
|
|
pageWidth=d.get("pageWidth", 612.0),
|
|
pageHeight=d.get("pageHeight", 792.0),
|
|
uploadedAt=d["uploadedAt"],
|
|
status=d["status"],
|
|
pages=pages_list,
|
|
)
|
|
|
|
|
|
@router.post("", response_model=DocumentInfoResponse, status_code=status.HTTP_201_CREATED)
|
|
async def upload_document(file: UploadFile = File(...), password: str = "") -> DocumentInfoResponse:
|
|
if not engine.is_available():
|
|
raise HTTPException(
|
|
status_code=status.HTTP_501_NOT_IMPLEMENTED,
|
|
detail="Engine bridge (bindings/python) not yet available.",
|
|
)
|
|
|
|
bytes_data = await file.read()
|
|
try:
|
|
pdfengine = engine.require()
|
|
doc = pdfengine.PdfDocument.load_from_memory(bytes_data, password)
|
|
info = document_store.add_document(file.filename, bytes_data, doc)
|
|
return make_document_response(info)
|
|
except ValueError as e:
|
|
detail = str(e)
|
|
if "Password required" in detail:
|
|
raise HTTPException(
|
|
status_code=status.HTTP_401_UNAUTHORIZED, detail="Password required"
|
|
)
|
|
elif "Invalid password" in detail:
|
|
raise HTTPException(status_code=status.HTTP_401_UNAUTHORIZED, detail="Invalid password")
|
|
else:
|
|
raise HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail=detail)
|
|
except Exception as e:
|
|
raise HTTPException(
|
|
status_code=status.HTTP_400_BAD_REQUEST, detail=f"Failed to load PDF: {e!s}"
|
|
)
|
|
|
|
|
|
@router.get("", response_model=list[DocumentInfoResponse])
|
|
def list_documents() -> list[DocumentInfoResponse]:
|
|
if not engine.is_available():
|
|
raise HTTPException(
|
|
status_code=status.HTTP_501_NOT_IMPLEMENTED,
|
|
detail="Engine bridge (bindings/python) not yet available.",
|
|
)
|
|
|
|
docs = document_store.list_documents()
|
|
return [make_document_response(d) for d in docs]
|
|
|
|
|
|
@router.get("/{document_id}", response_model=DocumentInfoResponse)
|
|
def get_document(document_id: str) -> DocumentInfoResponse:
|
|
if not engine.is_available():
|
|
raise HTTPException(
|
|
status_code=status.HTTP_501_NOT_IMPLEMENTED,
|
|
detail="Engine bridge (bindings/python) not yet available.",
|
|
)
|
|
|
|
d = document_store.get_document(document_id)
|
|
if not d:
|
|
raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="Document not found")
|
|
|
|
return make_document_response(d)
|
|
|
|
|
|
@router.delete("/{document_id}")
|
|
def delete_document(document_id: str):
|
|
if not engine.is_available():
|
|
raise HTTPException(
|
|
status_code=status.HTTP_501_NOT_IMPLEMENTED,
|
|
detail="Engine bridge (bindings/python) not yet available.",
|
|
)
|
|
|
|
deleted = document_store.delete_document(document_id)
|
|
if not deleted:
|
|
raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="Document not found")
|
|
|
|
return {"success": True}
|
|
|
|
|
|
class DocumentMetadataResponse(BaseModel):
|
|
title: str
|
|
author: str
|
|
creator: str
|
|
producer: str
|
|
creation_date: str
|
|
modification_date: str
|
|
|
|
|
|
@router.get("/{document_id}/metadata", response_model=DocumentMetadataResponse)
|
|
def get_document_metadata(document_id: str) -> DocumentMetadataResponse:
|
|
if not engine.is_available():
|
|
raise HTTPException(
|
|
status_code=status.HTTP_501_NOT_IMPLEMENTED,
|
|
detail="Engine bridge (bindings/python) not yet available.",
|
|
)
|
|
|
|
d = document_store.get_document(document_id)
|
|
if not d:
|
|
raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="Document not found")
|
|
|
|
try:
|
|
doc = d["doc_instance"]
|
|
meta = doc.metadata
|
|
return DocumentMetadataResponse(
|
|
title=meta.title,
|
|
author=meta.author,
|
|
creator=meta.creator,
|
|
producer=meta.producer,
|
|
creation_date=meta.creation_date,
|
|
modification_date=meta.modification_date,
|
|
)
|
|
except Exception as e:
|
|
raise HTTPException(status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, detail=str(e))
|
|
|
|
|
|
class FontInfoResponse(BaseModel):
|
|
fontName: str
|
|
type: str
|
|
isEmbedded: bool
|
|
isSubset: bool
|
|
isVertical: bool
|
|
encoding: str
|
|
hasToUnicode: bool
|
|
cmapName: str
|
|
cidSystemInfo: str
|
|
subsetTag: str
|
|
sourceType: str
|
|
substitutedFrom: str
|
|
substitutedTo: str
|
|
normalizedFamily: str
|
|
internalFontId: str
|
|
flags: int
|
|
ascent: float
|
|
descent: float
|
|
capHeight: float
|
|
|
|
@router.get("/{document_id}/fonts", response_model=list[FontInfoResponse])
|
|
def get_document_fonts(
|
|
document_id: str, start_page: int = 0, end_page: int = -1
|
|
) -> list[FontInfoResponse]:
|
|
if not engine.is_available():
|
|
raise HTTPException(
|
|
status_code=status.HTTP_501_NOT_IMPLEMENTED,
|
|
detail="Engine bridge (bindings/python) not yet available.",
|
|
)
|
|
|
|
doc_info = document_store.get_document(document_id)
|
|
if not doc_info:
|
|
raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="Document not found")
|
|
|
|
try:
|
|
doc = doc_info["doc_instance"]
|
|
fonts = doc.get_fonts(start_page, end_page)
|
|
return [
|
|
FontInfoResponse(
|
|
fontName=f.font_name,
|
|
type=f.type,
|
|
isEmbedded=f.is_embedded,
|
|
isSubset=f.is_subset,
|
|
isVertical=f.is_vertical,
|
|
encoding=f.encoding,
|
|
hasToUnicode=f.has_to_unicode,
|
|
cmapName=f.cmap_name,
|
|
cidSystemInfo=f.cid_system_info,
|
|
subsetTag=f.subset_tag,
|
|
sourceType=f.source_type,
|
|
substitutedFrom=f.substituted_from,
|
|
substitutedTo=f.substituted_to,
|
|
normalizedFamily=f.normalized_family,
|
|
internalFontId=f.internal_font_id,
|
|
flags=f.flags,
|
|
ascent=f.ascent,
|
|
descent=f.descent,
|
|
capHeight=f.cap_height,
|
|
)
|
|
for f in fonts
|
|
]
|
|
except ValueError as e:
|
|
raise HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail=str(e))
|
|
except IndexError as e:
|
|
raise HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail=str(e))
|
|
except Exception as e:
|
|
raise HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail=str(e))
|
|
|
|
class SearchRect(BaseModel):
|
|
x: float
|
|
y: float
|
|
w: float
|
|
h: float
|
|
|
|
class SearchMatch(BaseModel):
|
|
pageIndex: int
|
|
rects: list[SearchRect]
|
|
text: str
|
|
|
|
@router.get("/{document_id}/search", response_model=list[SearchMatch])
|
|
def search_document(document_id: str, q: str) -> list[SearchMatch]:
|
|
if not engine.is_available():
|
|
raise HTTPException(
|
|
status_code=status.HTTP_501_NOT_IMPLEMENTED,
|
|
detail="Engine bridge (bindings/python) not yet available."
|
|
)
|
|
|
|
if not q:
|
|
return []
|
|
|
|
doc_info = document_store.get_document(document_id)
|
|
if not doc_info:
|
|
raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="Document not found")
|
|
|
|
try:
|
|
doc = doc_info["doc_instance"]
|
|
matches = []
|
|
lower_query = q.lower()
|
|
query_len = len(lower_query)
|
|
|
|
for page_idx in range(doc.page_count):
|
|
page = doc.get_page(page_idx)
|
|
glyphs = page.extract_text_with_bounds()
|
|
if not glyphs:
|
|
continue
|
|
|
|
text_str = ""
|
|
char_to_glyph = []
|
|
for i, g in enumerate(glyphs):
|
|
s = g.get("text", "")
|
|
start_len = len(text_str)
|
|
text_str += s
|
|
for _ in range(len(text_str) - start_len):
|
|
char_to_glyph.append(i)
|
|
|
|
lower_text = text_str.lower()
|
|
idx = 0
|
|
while True:
|
|
idx = lower_text.find(lower_query, idx)
|
|
if idx == -1:
|
|
break
|
|
|
|
start_glyph_idx = char_to_glyph[idx]
|
|
end_glyph_idx = char_to_glyph[idx + query_len - 1]
|
|
|
|
rects = []
|
|
current_rect = None
|
|
|
|
for g_idx in range(start_glyph_idx, end_glyph_idx + 1):
|
|
g = glyphs[g_idx]
|
|
dom_y = page.height - (g["y"] + g["h"])
|
|
|
|
if current_rect is None:
|
|
current_rect = {"x": g["x"], "y": dom_y, "w": g["w"], "h": g["h"]}
|
|
else:
|
|
if abs(dom_y - current_rect["y"]) < g.get("fontSize", 12) * 0.5:
|
|
max_x = max(current_rect["x"] + current_rect["w"], g["x"] + g["w"])
|
|
current_rect["w"] = max_x - current_rect["x"]
|
|
current_rect["y"] = min(current_rect["y"], dom_y)
|
|
current_rect["h"] = max(current_rect["h"], g["h"])
|
|
else:
|
|
rects.append(SearchRect(**current_rect))
|
|
current_rect = {"x": g["x"], "y": dom_y, "w": g["w"], "h": g["h"]}
|
|
|
|
if current_rect:
|
|
rects.append(SearchRect(**current_rect))
|
|
|
|
matches.append(SearchMatch(
|
|
pageIndex=page_idx,
|
|
rects=rects,
|
|
text=text_str[idx:idx + query_len]
|
|
))
|
|
|
|
idx += 1
|
|
|
|
return matches
|
|
|
|
except Exception as e:
|
|
raise HTTPException(status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, detail=str(e))
|
|
|
|
class GlyphModel(BaseModel):
|
|
text: str
|
|
unicode: int
|
|
font_name: str
|
|
flags: int
|
|
font_size: float
|
|
origin_x: float
|
|
origin_y: float
|
|
bbox_x: float
|
|
bbox_y: float
|
|
bbox_w: float
|
|
bbox_h: float
|
|
angle: float
|
|
|
|
class TextRunModel(BaseModel):
|
|
text: str
|
|
font_name: str
|
|
flags: int
|
|
font_size: float
|
|
internal_font_id: str
|
|
is_embedded: bool
|
|
type: str
|
|
glyphs: list[GlyphModel]
|
|
x: float
|
|
y: float
|
|
w: float
|
|
h: float
|
|
|
|
class TextLineModel(BaseModel):
|
|
runs: list[TextRunModel]
|
|
baseline_y: float
|
|
x: float
|
|
y: float
|
|
w: float
|
|
h: float
|
|
|
|
class ParagraphModel(BaseModel):
|
|
lines: list[TextLineModel]
|
|
x: float
|
|
y: float
|
|
w: float
|
|
h: float
|
|
|
|
class PageModelResponse(BaseModel):
|
|
paragraphs: list[ParagraphModel]
|
|
width: float
|
|
height: float
|
|
page_index: int
|
|
|
|
@router.get("/{document_id}/pages/{page_index}/model", response_model=PageModelResponse)
|
|
def get_page_model(document_id: str, page_index: int) -> PageModelResponse:
|
|
if not engine.is_available():
|
|
raise HTTPException(
|
|
status_code=status.HTTP_501_NOT_IMPLEMENTED,
|
|
detail="Engine bridge (bindings/python) not yet available.",
|
|
)
|
|
|
|
doc_info = document_store.get_document(document_id)
|
|
if not doc_info:
|
|
raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="Document not found")
|
|
|
|
try:
|
|
doc = doc_info["doc_instance"]
|
|
page = doc.get_page(page_index)
|
|
model = page.extract_document_model()
|
|
|
|
paragraphs = []
|
|
for p in model.paragraphs:
|
|
lines = []
|
|
for l in p.lines:
|
|
runs = []
|
|
for r in l.runs:
|
|
glyphs = []
|
|
for g in r.glyphs:
|
|
glyphs.append(GlyphModel(
|
|
text=g.text,
|
|
unicode=g.unicode,
|
|
font_name=g.font_name,
|
|
flags=g.flags,
|
|
font_size=g.font_size,
|
|
origin_x=g.origin_x,
|
|
origin_y=g.origin_y,
|
|
bbox_x=g.bbox_x,
|
|
bbox_y=g.bbox_y,
|
|
bbox_w=g.bbox_w,
|
|
bbox_h=g.bbox_h,
|
|
angle=g.angle
|
|
))
|
|
runs.append(TextRunModel(
|
|
text=r.text,
|
|
font_name=r.font_name,
|
|
flags=r.flags,
|
|
font_size=r.font_size,
|
|
internal_font_id=r.internal_font_id,
|
|
is_embedded=r.is_embedded,
|
|
type=r.type,
|
|
glyphs=glyphs,
|
|
x=r.x,
|
|
y=r.y,
|
|
w=r.w,
|
|
h=r.h
|
|
))
|
|
lines.append(TextLineModel(
|
|
runs=runs,
|
|
baseline_y=l.baseline_y,
|
|
x=l.x,
|
|
y=l.y,
|
|
w=l.w,
|
|
h=l.h
|
|
))
|
|
paragraphs.append(ParagraphModel(
|
|
lines=lines,
|
|
x=p.x,
|
|
y=p.y,
|
|
w=p.w,
|
|
h=p.h
|
|
))
|
|
|
|
return PageModelResponse(
|
|
paragraphs=paragraphs,
|
|
width=model.width,
|
|
height=model.height,
|
|
page_index=model.page_index
|
|
)
|
|
except ValueError as e:
|
|
raise HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail=str(e))
|
|
except IndexError as e:
|
|
raise HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail=str(e))
|
|
except Exception as e:
|
|
raise HTTPException(status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, detail=str(e))
|