EADST

Image2Text: Automating Document Layout Analysis with Python and LayoutParser

Image2Text: Automating Document Layout Analysis with Python and LayoutParser

import cv2
import layoutparser as lp
import os
import json
from PIL import Image
import numpy as np

def to_serializable(obj):
    if isinstance(obj, (np.float32, np.float64)):
        return float(obj)
    elif isinstance(obj, np.ndarray):
        return obj.tolist()
    else:
        return obj

def process_image(image_path, model):
    # Read and preprocess the image
    image = cv2.imread(image_path)
    image = image[..., ::-1]  # Convert from BGR to RGB

    # Use the model to detect layout
    layout = model.detect(image)

    # Convert layout objects to a serializable format
    layout_data = []
    for obj in layout:
        obj_dict = obj.to_dict()
        # Iterate through the dictionary, converting all numpy data types to serializable types
        obj_dict_serializable = {key: to_serializable(value) for key, value in obj_dict.items()}
        layout_data.append(obj_dict_serializable)

    return layout_data

def save_layout_to_json(layout_data, json_path):
    # Save layout data to a JSON file
    with open(json_path, 'w') as json_file:
        json.dump(layout_data, json_file)

# Load the model
model = lp.PaddleDetectionLayoutModel(
    config_path="lp://PubLayNet/ppyolov2_r50vd_dcn_365e_publaynet/config",
    threshold=0.5,
    label_map={0: "Text", 1: "Title", 2: "List", 3: "Table", 4: "Figure"},
    enforce_cpu=False,
    enable_mkldnn=True
)

def process_folder(folder_path):
    # Iterate through all files and subfolders in the folder
    for root, dirs, files in os.walk(folder_path):
        for file in files:
            if file.lower().endswith('.jpg'):  # Check if it's a JPG file
                file_path = os.path.join(root, file)
                layout_data = process_image(file_path, model)  # Process the image

                # Create JSON file path
                json_path = os.path.splitext(file_path)[0] + '.json'
                save_layout_to_json(layout_data, json_path)  # Save layout data as JSON


# Specify the folder path to process
folder_path = '/your_folder_path/'
process_folder(folder_path)
相关标签
About Me
XD
Goals determine what you are going to be.
Category
标签云
AI DeepSeek Ptyhon Hungarian Quantize scipy 域名 Bin RL Pickle Docker Ubuntu UNIX git SPIE Bert Paper LLAMA Miniforge Zip IndexTTS2 PDB SAM 强化学习 YOLO Nginx Paddle 图形思考法 FP64 SQLite Bitcoin MD5 公式 mmap Hotel torchinfo 净利润 v0.dev OpenCV 论文速读 音频 GGML Qwen2.5 ms-swift 多进程 Attention 搞笑 git-lfs Anaconda Streamlit GPTQ tqdm Web Pillow Firewall GoogLeNet 报税 Datetime Rebuttal Freesound 顶会 VSCode XML CLAP FP16 Land ModelScope BeautifulSoup Plate Use Github Qwen2 Agent Pytorch Michelin PIP PDF CSV Breakpoint RAR QWEN Django Algorithm Statistics ChatGPT GPT4 TTS EXCEL 第一性原理 飞书 Sklearn Jetson diffusers HaggingFace UI OCR 云服务器 Conda Disk Git Template COCO Transformers Markdown BTC NLTK printf 阿里云 Claude Math hf Quantization 版权 Excel FlashAttention Bipartite Python Plotly Augmentation Baidu Random TensorRT Vim CTC JSON Knowledge Jupyter Clash icon News llama.cpp TSV TensorFlow PyTorch Website SQL Animate Qwen 财报 LaTeX CEIR Gemma Dataset FP8 WebCrawler tar XGBoost InvalidArgumentError Distillation Google Vmess VGG-16 Food FastAPI uWSGI Logo Tensor Numpy Tiktoken Hilton OpenAI Crawler Card VPN LLM 继承 Tracking GIT Cloudreve 多线程 Translation Mixtral Shortcut CC HuggingFace 关于博主 Diagram FP32 ResNet-50 Heatmap WAN 证件照 DeepStream Password 论文 v2ray PyCharm Review 腾讯云 签证 图标 Linux Magnet CV uwsgi SVR BF16 transformers NLP Proxy Search Interview Windows ONNX Input 算法题 Image2Text RGB CUDA Pandas LeetCode C++ LoRA API Data logger Safetensors Color NameSilo 递归学习法 Llama CAM Permission Domain Base64 Video
站点统计

本站现有博文333篇,共被浏览922383

本站已经建立2628天!

热门文章
文章归档
回到顶部