EADST

Image2Text: Automating Document Layout Analysis with Python and LayoutParser

Image2Text: Automating Document Layout Analysis with Python and LayoutParser

import cv2
import layoutparser as lp
import os
import json
from PIL import Image
import numpy as np

def to_serializable(obj):
    if isinstance(obj, (np.float32, np.float64)):
        return float(obj)
    elif isinstance(obj, np.ndarray):
        return obj.tolist()
    else:
        return obj

def process_image(image_path, model):
    # Read and preprocess the image
    image = cv2.imread(image_path)
    image = image[..., ::-1]  # Convert from BGR to RGB

    # Use the model to detect layout
    layout = model.detect(image)

    # Convert layout objects to a serializable format
    layout_data = []
    for obj in layout:
        obj_dict = obj.to_dict()
        # Iterate through the dictionary, converting all numpy data types to serializable types
        obj_dict_serializable = {key: to_serializable(value) for key, value in obj_dict.items()}
        layout_data.append(obj_dict_serializable)

    return layout_data

def save_layout_to_json(layout_data, json_path):
    # Save layout data to a JSON file
    with open(json_path, 'w') as json_file:
        json.dump(layout_data, json_file)

# Load the model
model = lp.PaddleDetectionLayoutModel(
    config_path="lp://PubLayNet/ppyolov2_r50vd_dcn_365e_publaynet/config",
    threshold=0.5,
    label_map={0: "Text", 1: "Title", 2: "List", 3: "Table", 4: "Figure"},
    enforce_cpu=False,
    enable_mkldnn=True
)

def process_folder(folder_path):
    # Iterate through all files and subfolders in the folder
    for root, dirs, files in os.walk(folder_path):
        for file in files:
            if file.lower().endswith('.jpg'):  # Check if it's a JPG file
                file_path = os.path.join(root, file)
                layout_data = process_image(file_path, model)  # Process the image

                # Create JSON file path
                json_path = os.path.splitext(file_path)[0] + '.json'
                save_layout_to_json(layout_data, json_path)  # Save layout data as JSON


# Specify the folder path to process
folder_path = '/your_folder_path/'
process_folder(folder_path)
相关标签
About Me
XD
Goals determine what you are going to be.
Category
标签云
CUDA UNIX 财报 Nginx Quantization IndexTTS2 Food Magnet v0.dev Bin Input NLTK PDB ModelScope Dataset Python Claude hf Permission VGG-16 printf ResNet-50 第一性原理 Qwen2.5 Google Pytorch CTC 递归学习法 Translation torchinfo Hotel Bitcoin Disk 域名 Safetensors icon CLAP 阿里云 Bipartite Use TensorRT FP32 RGB Search LLAMA Crawler 云服务器 证件照 Random transformers 多线程 OpenCV BeautifulSoup Hungarian BF16 Transformers Color Distillation Plotly Bert QWEN Freesound Base64 搞笑 Heatmap uwsgi EXCEL FastAPI UI HaggingFace git 多进程 Paper Agent 强化学习 Interview LeetCode LaTeX uWSGI GGML Breakpoint Diagram Rebuttal 图形思考法 YOLO API Plate CV CSV OpenAI CC Streamlit Jetson Knowledge Vmess TTS llama.cpp Website 公式 WebCrawler PyCharm Animate Image2Text git-lfs Video Shortcut LoRA DeepSeek 关于博主 Numpy 算法题 Template 顶会 Michelin Algorithm Quantize Clash SAM OCR 腾讯云 Datetime Linux SQL JSON mmap MD5 RAR ONNX VPN Anaconda PyTorch News Pandas Qwen Review logger COCO Web Windows 签证 BTC scipy Git Conda CEIR Ptyhon Gemma 音频 Cloudreve AI Pillow 净利润 GIT ms-swift RL HuggingFace ChatGPT GPT4 Excel FlashAttention 飞书 SQLite NLP VSCode Proxy Firewall Logo DeepStream diffusers Paddle WAN Tiktoken Django FP16 CAM Land XGBoost XML Password Attention Augmentation SPIE FP64 继承 Github 论文速读 tar Mixtral C++ TSV InvalidArgumentError Pickle Ubuntu Baidu 版权 Statistics Hilton SVR Llama Tensor Math Miniforge Jupyter Docker PDF Sklearn GPTQ Data Markdown Card FP8 v2ray 图标 TensorFlow GoogLeNet 报税 PIP Qwen2 Tracking LLM NameSilo Zip tqdm Domain 论文 Vim
站点统计

本站现有博文333篇,共被浏览917643

本站已经建立2622天!

热门文章
文章归档
回到顶部