Spaces:
Running
Running
File size: 1,952 Bytes
5b56da4 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 | """
pdf_pipeline
============
PDF -> markdown extraction pipeline:
pdfplumber (words+bboxes, image)
-> [empty words? -> OCR] -> match against pp-doclayout boxes
-> per-region markdown, with table regions routed through TableFormerONNX -> OTSL -> markdown
See README.md for the full architecture diagram and a usage walkthrough.
"""
from .layout import (
BBox,
DocLayoutResult,
DocLayoutV3,
LayoutBox,
PlacedWord,
RegionText,
align_ocr_to_layout,
align_tokens_to_layout,
align_words_to_layout,
log_layout_result,
pdfplumber_tokens_in_image_space,
print_layout_result,
)
from .logging_config import setup_pipeline_logging
from .ocr_backends import (
OCRBackend,
PytesseractBackend,
RapidOCRBackend,
get_ocr_backend,
)
from .pipeline import (
IMAGE_EXTENSIONS,
crop_and_save_image,
process_document,
process_images,
process_image_page,
process_pdf,
process_pdf_page,
region_to_markdown,
regions_to_markdown,
)
from .table_extraction import (
TableFormerONNX,
ocr_anchor_cells,
otsl_to_markdown,
seq_to_otsl,
table_image_to_otsl,
)
__all__ = [
"BBox",
"DocLayoutResult",
"DocLayoutV3",
"LayoutBox",
"PlacedWord",
"RegionText",
"align_ocr_to_layout",
"align_tokens_to_layout",
"align_words_to_layout",
"pdfplumber_tokens_in_image_space",
"log_layout_result",
"print_layout_result",
"OCRBackend",
"PaddleOCRBackend",
"PytesseractBackend",
"RapidOCRBackend",
"get_ocr_backend",
"crop_and_save_image",
"IMAGE_EXTENSIONS",
"process_document",
"process_images",
"process_image_page",
"process_pdf",
"process_pdf_page",
"region_to_markdown",
"regions_to_markdown",
"TableFormerONNX",
"ocr_anchor_cells",
"otsl_to_markdown",
"seq_to_otsl",
"table_image_to_otsl",
"setup_pipeline_logging",
]
|