Merge pull request #1063 from opendatalab/release-0.10.0

Release 0.10.0

Merge pull request #1063 from opendatalab/release-0.10.0
Release 0.10.0
158e556b · Xiaomeng Zhao · GitHub · 038f48d3 · 30be5017 · 038f48d3
Unverified Commit 158e556b authored Nov 22, 2024 by Xiaomeng Zhao Committed by GitHub Nov 22, 2024
20 changed files
--- a/magic_pdf/libs/drop_tag.py
+++ b/magic_pdf/libs/drop_tag.py
-
-COLOR_BG_HEADER_TXT_BLOCK = "color_background_header_txt_block"
-PAGE_NO = "page-no" # 页码
-CONTENT_IN_FOOT_OR_HEADER = 'in-foot-header-area' # 页眉页脚内的文本
-VERTICAL_TEXT = 'vertical-text' # 垂直文本
-ROTATE_TEXT = 'rotate-text' # 旋转文本
-EMPTY_SIDE_BLOCK = 'empty-side-block' # 边缘上的空白没有任何内容的block
-ON_IMAGE_TEXT = 'on-image-text' # 文本在图片上
-ON_TABLE_TEXT = 'on-table-text' # 文本在表格上
-
-
-class DropTag:
-    PAGE_NUMBER = "page_no"
-    HEADER = "header"
-    FOOTER = "footer"
-    FOOTNOTE = "footnote"
-    NOT_IN_LAYOUT = "not_in_layout"
-    SPAN_OVERLAP = "span_overlap"
-    BLOCK_OVERLAP = "block_overlap"
--- a/magic_pdf/libs/pdf_image_tools.py
+++ b/magic_pdf/libs/pdf_image_tools.py
-
-from magic_pdf.rw.AbsReaderWriter import AbsReaderWriter
-from magic_pdf.libs.commons import fitz
-from magic_pdf.libs.commons import join_path
+from io import BytesIO
+import cv2
+import numpy as np
+from PIL import Image
+from magic_pdf.data.data_reader_writer import DataWriter
+from magic_pdf.libs.commons import fitz, join_path
 from magic_pdf.libs.hash_utils import compute_sha256


-def cut_image(bbox: tuple, page_num: int, page: fitz.Page, return_path, imageWriter: AbsReaderWriter):
-    """
-    从第page_num页的page中，根据bbox进行裁剪出一张jpg图片，返回图片路径
-    save_path：需要同时支持s3和本地, 图片存放在save_path下，文件名是: {page_num}_{bbox[0]}_{bbox[1]}_{bbox[2]}_{bbox[3]}.jpg , bbox内数字取整。
-    """
+def cut_image(bbox: tuple, page_num: int, page: fitz.Page, return_path, imageWriter: DataWriter):
+    """从第page_num页的page中，根据bbox进行裁剪出一张jpg图片，返回图片路径 save_path：需要同时支持s3和本地,
+    图片存放在save_path下，文件名是:
+    {page_num}_{bbox[0]}_{bbox[1]}_{bbox[2]}_{bbox[3]}.jpg , bbox内数字取整。"""
    # 拼接文件名
-    filename = f"{page_num}_{int(bbox[0])}_{int(bbox[1])}_{int(bbox[2])}_{int(bbox[3])}"
+    filename = f'{page_num}_{int(bbox[0])}_{int(bbox[1])}_{int(bbox[2])}_{int(bbox[3])}'

    # 老版本返回不带bucket的路径
    img_path = join_path(return_path, filename) if return_path is not None else None

    # 新版本生成平铺路径
-    img_hash256_path = f"{compute_sha256(img_path)}.jpg"
+    img_hash256_path = f'{compute_sha256(img_path)}.jpg'

    # 将坐标转换为fitz.Rect对象
    rect = fitz.Rect(*bbox)
@@ -28,6 +29,29 @@ def cut_image(bbox: tuple, page_num: int, page: fitz.Page, return_path, imageWri

    byte_data = pix.tobytes(output='jpeg', jpg_quality=95)

-    imageWriter.write(byte_data, img_hash256_path, AbsReaderWriter.MODE_BIN)
+    imageWriter.write(img_hash256_path, byte_data)

    return img_hash256_path
+
+
+def cut_image_to_pil_image(bbox: tuple, page: fitz.Page, mode="pillow"):
+
+    # 将坐标转换为fitz.Rect对象
+    rect = fitz.Rect(*bbox)
+    # 配置缩放倍数为3倍
+    zoom = fitz.Matrix(3, 3)
+    # 截取图片
+    pix = page.get_pixmap(clip=rect, matrix=zoom)
+
+    # 将字节数据转换为文件对象
+    image_file = BytesIO(pix.tobytes(output='png'))
+    # 使用 Pillow 打开图像
+    pil_image = Image.open(image_file)
+    if mode == "cv2":
+        image_result = cv2.cvtColor(np.asarray(pil_image), cv2.COLOR_RGB2BGR)
+    elif mode == "pillow":
+        image_result = pil_image
+    else:
+        raise ValueError(f"mode: {mode} is not supported.")
+
+    return image_result
\ No newline at end of file
--- a/magic_pdf/model/doc_analyze_by_custom_model.py
+++ b/magic_pdf/model/doc_analyze_by_custom_model.py
@@ -163,7 +163,9 @@ def doc_analyze(pdf_bytes: bytes, ocr: bool = False, show_log: bool = False,
        page_width = img_dict["width"]
        page_height = img_dict["height"]
        if start_page_id <= index <= end_page_id:
+            page_start = time.time()
            result = custom_model(img)
+            logger.info(f'-----page_id : {index}, page total time: {round(time.time() - page_start, 2)}-----')
        else:
            result = []
        page_info = {"page_no": index, "height": page_height, "width": page_width}

--- a/magic_pdf/model/magic_model.py
+++ b/magic_pdf/model/magic_model.py
 import enum
 import json

+from magic_pdf.config.model_block_type import ModelBlockTypeEnum
+from magic_pdf.config.ocr_content_type import CategoryId, ContentType
+from magic_pdf.data.data_reader_writer import (FileBasedDataReader,
+                                               FileBasedDataWriter)
 from magic_pdf.data.dataset import Dataset
 from magic_pdf.libs.boxbase import (_is_in, _is_part_overlap, bbox_distance,
                                    bbox_relative_pos, box_area, calculate_iou,
@@ -9,11 +13,7 @@ from magic_pdf.libs.boxbase import (_is_in, _is_part_overlap, bbox_distance,
 from magic_pdf.libs.commons import fitz, join_path
 from magic_pdf.libs.coordinate_transform import get_scale_ratio
 from magic_pdf.libs.local_math import float_gt
-from magic_pdf.libs.ModelBlockTypeEnum import ModelBlockTypeEnum
-from magic_pdf.libs.ocr_content_type import CategoryId, ContentType
 from magic_pdf.pre_proc.remove_bbox_overlap import _remove_overlap_between_bbox
-from magic_pdf.rw.AbsReaderWriter import AbsReaderWriter
-from magic_pdf.rw.DiskReaderWriter import DiskReaderWriter

 CAPATION_OVERLAP_AREA_RATIO = 0.6
 MERGE_BOX_OVERLAP_AREA_RATIO = 1.1
@@ -1050,27 +1050,27 @@ class MagicModel:


 if __name__ == '__main__':
-    drw = DiskReaderWriter(r'D:/project/20231108code-clean')
+    drw = FileBasedDataReader(r'D:/project/20231108code-clean')
    if 0:
        pdf_file_path = r'linshixuqiu\19983-00.pdf'
        model_file_path = r'linshixuqiu\19983-00_new.json'
-        pdf_bytes = drw.read(pdf_file_path, AbsReaderWriter.MODE_BIN)
-        model_json_txt = drw.read(model_file_path, AbsReaderWriter.MODE_TXT)
+        pdf_bytes = drw.read(pdf_file_path)
+        model_json_txt = drw.read(model_file_path).decode()
        model_list = json.loads(model_json_txt)
        write_path = r'D:\project\20231108code-clean\linshixuqiu\19983-00'
        img_bucket_path = 'imgs'
-        img_writer = DiskReaderWriter(join_path(write_path, img_bucket_path))
+        img_writer = FileBasedDataWriter(join_path(write_path, img_bucket_path))
        pdf_docs = fitz.open('pdf', pdf_bytes)
        magic_model = MagicModel(model_list, pdf_docs)

    if 1:
+        from magic_pdf.data.dataset import PymuDocDataset
+
        model_list = json.loads(
            drw.read('/opt/data/pdf/20240418/j.chroma.2009.03.042.json')
        )
-        pdf_bytes = drw.read(
-            '/opt/data/pdf/20240418/j.chroma.2009.03.042.pdf', AbsReaderWriter.MODE_BIN
-        )
-        pdf_docs = fitz.open('pdf', pdf_bytes)
-        magic_model = MagicModel(model_list, pdf_docs)
+        pdf_bytes = drw.read('/opt/data/pdf/20240418/j.chroma.2009.03.042.pdf')
+
+        magic_model = MagicModel(model_list, PymuDocDataset(pdf_bytes))
        for i in range(7):
            print(magic_model.get_imgs(i))
--- a/magic_pdf/model/pdf_extract_kit.py
+++ b/magic_pdf/model/pdf_extract_kit.py
-import numpy as np
-import torch
-from loguru import logger
+# flake8: noqa
 import os
 import time
+
 import cv2
+import numpy as np
+import torch
 import yaml
+from loguru import logger
 from PIL import Image

 os.environ['NO_ALBUMENTATIONS_UPDATE'] = '1'  # 禁止albumentations检查更新
@@ -13,16 +15,18 @@ os.environ['YOLO_VERBOSE'] = 'False'  # disable yolo logger
 try:
    import torchtext

-    if torchtext.__version__ >= "0.18.0":
+    if torchtext.__version__ >= '0.18.0':
        torchtext.disable_torchtext_deprecation_warning()
 except ImportError:
    pass

-from magic_pdf.libs.Constants import *
+from magic_pdf.config.constants import *
 from magic_pdf.model.model_list import AtomicModel
 from magic_pdf.model.sub_modules.model_init import AtomModelSingleton
-from magic_pdf.model.sub_modules.model_utils import get_res_list_from_layout_res, crop_img, clean_vram
-from magic_pdf.model.sub_modules.ocr.paddleocr.ocr_utils import get_adjusted_mfdetrec_res, get_ocr_result_list
+from magic_pdf.model.sub_modules.model_utils import (
+    clean_vram, crop_img, get_res_list_from_layout_res)
+from magic_pdf.model.sub_modules.ocr.paddleocr.ocr_utils import (
+    get_adjusted_mfdetrec_res, get_ocr_result_list)


 class CustomPEKModel:
@@ -41,42 +45,54 @@ class CustomPEKModel:
        model_config_dir = os.path.join(root_dir, 'resources', 'model_config')
        # 构建 model_configs.yaml 文件的完整路径
        config_path = os.path.join(model_config_dir, 'model_configs.yaml')
-        with open(config_path, "r", encoding='utf-8') as f:
+        with open(config_path, 'r', encoding='utf-8') as f:
            self.configs = yaml.load(f, Loader=yaml.FullLoader)
        # 初始化解析配置

        # layout config
-        self.layout_config = kwargs.get("layout_config")
-        self.layout_model_name = self.layout_config.get("model", MODEL_NAME.DocLayout_YOLO)
+        self.layout_config = kwargs.get('layout_config')
+        self.layout_model_name = self.layout_config.get(
+            'model', MODEL_NAME.DocLayout_YOLO
+        )

        # formula config
-        self.formula_config = kwargs.get("formula_config")
-        self.mfd_model_name = self.formula_config.get("mfd_model", MODEL_NAME.YOLO_V8_MFD)
-        self.mfr_model_name = self.formula_config.get("mfr_model", MODEL_NAME.UniMerNet_v2_Small)
-        self.apply_formula = self.formula_config.get("enable", True)
+        self.formula_config = kwargs.get('formula_config')
+        self.mfd_model_name = self.formula_config.get(
+            'mfd_model', MODEL_NAME.YOLO_V8_MFD
+        )
+        self.mfr_model_name = self.formula_config.get(
+            'mfr_model', MODEL_NAME.UniMerNet_v2_Small
+        )
+        self.apply_formula = self.formula_config.get('enable', True)

        # table config
-        self.table_config = kwargs.get("table_config")
-        self.apply_table = self.table_config.get("enable", False)
-        self.table_max_time = self.table_config.get("max_time", TABLE_MAX_TIME_VALUE)
-        self.table_model_name = self.table_config.get("model", MODEL_NAME.RAPID_TABLE)
+        self.table_config = kwargs.get('table_config')
+        self.apply_table = self.table_config.get('enable', False)
+        self.table_max_time = self.table_config.get('max_time', TABLE_MAX_TIME_VALUE)
+        self.table_model_name = self.table_config.get('model', MODEL_NAME.RAPID_TABLE)

        # ocr config
        self.apply_ocr = ocr
-        self.lang = kwargs.get("lang", None)
+        self.lang = kwargs.get('lang', None)

        logger.info(
-            "DocAnalysis init, this may take some times, layout_model: {}, apply_formula: {}, apply_ocr: {}, "
-            "apply_table: {}, table_model: {}, lang: {}".format(
-                self.layout_model_name, self.apply_formula, self.apply_ocr, self.apply_table, self.table_model_name,
-                self.lang
+            'DocAnalysis init, this may take some times, layout_model: {}, apply_formula: {}, apply_ocr: {}, '
+            'apply_table: {}, table_model: {}, lang: {}'.format(
+                self.layout_model_name,
+                self.apply_formula,
+                self.apply_ocr,
+                self.apply_table,
+                self.table_model_name,
+                self.lang,
            )
        )
        # 初始化解析方案
-        self.device = kwargs.get("device", "cpu")
-        logger.info("using device: {}".format(self.device))
-        models_dir = kwargs.get("models_dir", os.path.join(root_dir, "resources", "models"))
-        logger.info("using models_dir: {}".format(models_dir))
+        self.device = kwargs.get('device', 'cpu')
+        logger.info('using device: {}'.format(self.device))
+        models_dir = kwargs.get(
+            'models_dir', os.path.join(root_dir, 'resources', 'models')
+        )
+        logger.info('using models_dir: {}'.format(models_dir))

        atom_model_manager = AtomModelSingleton()

@@ -85,18 +101,24 @@ class CustomPEKModel:
            # 初始化公式检测模型
            self.mfd_model = atom_model_manager.get_atom_model(
                atom_model_name=AtomicModel.MFD,
-                mfd_weights=str(os.path.join(models_dir, self.configs["weights"][self.mfd_model_name])),
-                device=self.device
+                mfd_weights=str(
+                    os.path.join(
+                        models_dir, self.configs['weights'][self.mfd_model_name]
+                    )
+                ),
+                device=self.device,
            )

            # 初始化公式解析模型
-            mfr_weight_dir = str(os.path.join(models_dir, self.configs["weights"][self.mfr_model_name]))
-            mfr_cfg_path = str(os.path.join(model_config_dir, "UniMERNet", "demo.yaml"))
+            mfr_weight_dir = str(
+                os.path.join(models_dir, self.configs['weights'][self.mfr_model_name])
+            )
+            mfr_cfg_path = str(os.path.join(model_config_dir, 'UniMERNet', 'demo.yaml'))
            self.mfr_model = atom_model_manager.get_atom_model(
                atom_model_name=AtomicModel.MFR,
                mfr_weight_dir=mfr_weight_dir,
                mfr_cfg_path=mfr_cfg_path,
-                device=self.device
+                device=self.device,
            )

        # 初始化layout模型
@@ -104,42 +126,51 @@ class CustomPEKModel:
            self.layout_model = atom_model_manager.get_atom_model(
                atom_model_name=AtomicModel.Layout,
                layout_model_name=MODEL_NAME.LAYOUTLMv3,
-                layout_weights=str(os.path.join(models_dir, self.configs['weights'][self.layout_model_name])),
-                layout_config_file=str(os.path.join(model_config_dir, "layoutlmv3", "layoutlmv3_base_inference.yaml")),
-                device=self.device
+                layout_weights=str(
+                    os.path.join(
+                        models_dir, self.configs['weights'][self.layout_model_name]
+                    )
+                ),
+                layout_config_file=str(
+                    os.path.join(
+                        model_config_dir, 'layoutlmv3', 'layoutlmv3_base_inference.yaml'
+                    )
+                ),
+                device=self.device,
            )
        elif self.layout_model_name == MODEL_NAME.DocLayout_YOLO:
            self.layout_model = atom_model_manager.get_atom_model(
                atom_model_name=AtomicModel.Layout,
                layout_model_name=MODEL_NAME.DocLayout_YOLO,
-                doclayout_yolo_weights=str(os.path.join(models_dir, self.configs['weights'][self.layout_model_name])),
-                device=self.device
+                doclayout_yolo_weights=str(
+                    os.path.join(
+                        models_dir, self.configs['weights'][self.layout_model_name]
+                    )
+                ),
+                device=self.device,
            )
        # 初始化ocr
-        if self.apply_ocr:
-            self.ocr_model = atom_model_manager.get_atom_model(
-                atom_model_name=AtomicModel.OCR,
-                ocr_show_log=show_log,
-                det_db_box_thresh=0.3,
-                lang=self.lang
-            )
+        self.ocr_model = atom_model_manager.get_atom_model(
+            atom_model_name=AtomicModel.OCR,
+            ocr_show_log=show_log,
+            det_db_box_thresh=0.3,
+            lang=self.lang
+        )
        # init table model
        if self.apply_table:
-            table_model_dir = self.configs["weights"][self.table_model_name]
+            table_model_dir = self.configs['weights'][self.table_model_name]
            self.table_model = atom_model_manager.get_atom_model(
                atom_model_name=AtomicModel.Table,
                table_model_name=self.table_model_name,
                table_model_path=str(os.path.join(models_dir, table_model_dir)),
                table_max_time=self.table_max_time,
-                device=self.device
+                device=self.device,
            )

        logger.info('DocAnalysis init done!')

    def __call__(self, image):

-        page_start = time.time()
-
        # layout检测
        layout_start = time.time()
        layout_res = []
@@ -150,7 +181,7 @@ class CustomPEKModel:
            # doclayout_yolo
            layout_res = self.layout_model.predict(image)
        layout_cost = round(time.time() - layout_start, 2)
-        logger.info(f"layout detection time: {layout_cost}")
+        logger.info(f'layout detection time: {layout_cost}')

        pil_img = Image.fromarray(image)

@@ -158,40 +189,47 @@ class CustomPEKModel:
            # 公式检测
            mfd_start = time.time()
            mfd_res = self.mfd_model.predict(image)
-            logger.info(f"mfd time: {round(time.time() - mfd_start, 2)}")
+            logger.info(f'mfd time: {round(time.time() - mfd_start, 2)}')

            # 公式识别
            mfr_start = time.time()
            formula_list = self.mfr_model.predict(mfd_res, image)
            layout_res.extend(formula_list)
            mfr_cost = round(time.time() - mfr_start, 2)
-            logger.info(f"formula nums: {len(formula_list)}, mfr time: {mfr_cost}")
+            logger.info(f'formula nums: {len(formula_list)}, mfr time: {mfr_cost}')

        # 清理显存
        clean_vram(self.device, vram_threshold=8)

        # 从layout_res中获取ocr区域、表格区域、公式区域
-        ocr_res_list, table_res_list, single_page_mfdetrec_res = get_res_list_from_layout_res(layout_res)
+        ocr_res_list, table_res_list, single_page_mfdetrec_res = (
+            get_res_list_from_layout_res(layout_res)
+        )

        # ocr识别
-        if self.apply_ocr:
-            ocr_start = time.time()
-            # Process each area that requires OCR processing
-            for res in ocr_res_list:
-                new_image, useful_list = crop_img(res, pil_img, crop_paste_x=50, crop_paste_y=50)
-                adjusted_mfdetrec_res = get_adjusted_mfdetrec_res(single_page_mfdetrec_res, useful_list)
-
-                # OCR recognition
-                new_image = cv2.cvtColor(np.asarray(new_image), cv2.COLOR_RGB2BGR)
+        ocr_start = time.time()
+        # Process each area that requires OCR processing
+        for res in ocr_res_list:
+            new_image, useful_list = crop_img(res, pil_img, crop_paste_x=50, crop_paste_y=50)
+            adjusted_mfdetrec_res = get_adjusted_mfdetrec_res(single_page_mfdetrec_res, useful_list)
+
+            # OCR recognition
+            new_image = cv2.cvtColor(np.asarray(new_image), cv2.COLOR_RGB2BGR)
+            if self.apply_ocr:
                ocr_res = self.ocr_model.ocr(new_image, mfd_res=adjusted_mfdetrec_res)[0]
+            else:
+                ocr_res = self.ocr_model.ocr(new_image, mfd_res=adjusted_mfdetrec_res, rec=False)[0]

-                # Integration results
-                if ocr_res:
-                    ocr_result_list = get_ocr_result_list(ocr_res, useful_list)
-                    layout_res.extend(ocr_result_list)
+            # Integration results
+            if ocr_res:
+                ocr_result_list = get_ocr_result_list(ocr_res, useful_list)
+                layout_res.extend(ocr_result_list)

-            ocr_cost = round(time.time() - ocr_start, 2)
+        ocr_cost = round(time.time() - ocr_start, 2)
+        if self.apply_ocr:
            logger.info(f"ocr time: {ocr_cost}")
+        else:
+            logger.info(f"det time: {ocr_cost}")

        # 表格识别 table recognition
        if self.apply_table:
@@ -202,27 +240,35 @@ class CustomPEKModel:
                html_code = None
                if self.table_model_name == MODEL_NAME.STRUCT_EQTABLE:
                    with torch.no_grad():
-                        table_result = self.table_model.predict(new_image, "html")
+                        table_result = self.table_model.predict(new_image, 'html')
                        if len(table_result) > 0:
                            html_code = table_result[0]
                elif self.table_model_name == MODEL_NAME.TABLE_MASTER:
                    html_code = self.table_model.img2html(new_image)
                elif self.table_model_name == MODEL_NAME.RAPID_TABLE:
-                    html_code, table_cell_bboxes, elapse = self.table_model.predict(new_image)
+                    html_code, table_cell_bboxes, elapse = self.table_model.predict(
+                        new_image
+                    )
                run_time = time.time() - single_table_start_time
                if run_time > self.table_max_time:
-                    logger.warning(f"table recognition processing exceeds max time {self.table_max_time}s")
+                    logger.warning(
+                        f'table recognition processing exceeds max time {self.table_max_time}s'
+                    )
                # 判断是否返回正常
                if html_code:
-                    expected_ending = html_code.strip().endswith('</html>') or html_code.strip().endswith('</table>')
+                    expected_ending = html_code.strip().endswith(
+                        '</html>'
+                    ) or html_code.strip().endswith('</table>')
                    if expected_ending:
-                        res["html"] = html_code
+                        res['html'] = html_code
                    else:
-                        logger.warning(f"table recognition processing fails, not found expected HTML table end")
+                        logger.warning(
+                            'table recognition processing fails, not found expected HTML table end'
+                        )
                else:
-                    logger.warning(f"table recognition processing fails, not get html return")
-            logger.info(f"table time: {round(time.time() - table_start, 2)}")
-
-        logger.info(f"-----page total time: {round(time.time() - page_start, 2)}-----")
+                    logger.warning(
+                        'table recognition processing fails, not get html return'
+                    )
+            logger.info(f'table time: {round(time.time() - table_start, 2)}')

        return layout_res
--- a/magic_pdf/model/sub_modules/model_init.py
+++ b/magic_pdf/model/sub_modules/model_init.py
 from loguru import logger

-from magic_pdf.libs.Constants import MODEL_NAME
+from magic_pdf.config.constants import MODEL_NAME
 from magic_pdf.model.model_list import AtomicModel
-from magic_pdf.model.sub_modules.layout.doclayout_yolo.DocLayoutYOLO import DocLayoutYOLOModel
-from magic_pdf.model.sub_modules.layout.layoutlmv3.model_init import Layoutlmv3_Predictor
+from magic_pdf.model.sub_modules.layout.doclayout_yolo.DocLayoutYOLO import \
+    DocLayoutYOLOModel
+from magic_pdf.model.sub_modules.layout.layoutlmv3.model_init import \
+    Layoutlmv3_Predictor
 from magic_pdf.model.sub_modules.mfd.yolov8.YOLOv8 import YOLOv8MFDModel
-
 from magic_pdf.model.sub_modules.mfr.unimernet.Unimernet import UnimernetModel
-from magic_pdf.model.sub_modules.ocr.paddleocr.ppocr_273_mod import ModifiedPaddleOCR
+from magic_pdf.model.sub_modules.ocr.paddleocr.ppocr_273_mod import \
+    ModifiedPaddleOCR
+from magic_pdf.model.sub_modules.table.rapidtable.rapid_table import \
+    RapidTableModel
 # from magic_pdf.model.sub_modules.ocr.paddleocr.ppocr_291_mod import ModifiedPaddleOCR
-from magic_pdf.model.sub_modules.table.structeqtable.struct_eqtable import StructTableModel
-from magic_pdf.model.sub_modules.table.tablemaster.tablemaster_paddle import TableMasterPaddleModel
-from magic_pdf.model.sub_modules.table.rapidtable.rapid_table import RapidTableModel
+from magic_pdf.model.sub_modules.table.structeqtable.struct_eqtable import \
+    StructTableModel
+from magic_pdf.model.sub_modules.table.tablemaster.tablemaster_paddle import \
+    TableMasterPaddleModel


 def table_model_init(table_model_type, model_path, max_time, _device_='cpu'):
@@ -19,14 +24,14 @@ def table_model_init(table_model_type, model_path, max_time, _device_='cpu'):
        table_model = StructTableModel(model_path, max_new_tokens=2048, max_time=max_time)
    elif table_model_type == MODEL_NAME.TABLE_MASTER:
        config = {
-            "model_dir": model_path,
-            "device": _device_
+            'model_dir': model_path,
+            'device': _device_
        }
        table_model = TableMasterPaddleModel(config)
    elif table_model_type == MODEL_NAME.RAPID_TABLE:
        table_model = RapidTableModel()
    else:
-        logger.error("table model type not allow")
+        logger.error('table model type not allow')
        exit(1)

    return table_model
@@ -58,7 +63,7 @@ def ocr_model_init(show_log: bool = False,
                   use_dilation=True,
                   det_db_unclip_ratio=1.8,
                   ):
-    if lang is not None:
+    if lang is not None and lang != '':
        model = ModifiedPaddleOCR(
            show_log=show_log,
            det_db_box_thresh=det_db_box_thresh,
@@ -87,8 +92,8 @@ class AtomModelSingleton:
        return cls._instance

    def get_atom_model(self, atom_model_name: str, **kwargs):
-        lang = kwargs.get("lang", None)
-        layout_model_name = kwargs.get("layout_model_name", None)
+        lang = kwargs.get('lang', None)
+        layout_model_name = kwargs.get('layout_model_name', None)
        key = (atom_model_name, layout_model_name, lang)
        if key not in self._models:
            self._models[key] = atom_model_init(model_name=atom_model_name, **kwargs)
@@ -98,47 +103,47 @@ class AtomModelSingleton:
 def atom_model_init(model_name: str, **kwargs):
    atom_model = None
    if model_name == AtomicModel.Layout:
-        if kwargs.get("layout_model_name") == MODEL_NAME.LAYOUTLMv3:
+        if kwargs.get('layout_model_name') == MODEL_NAME.LAYOUTLMv3:
            atom_model = layout_model_init(
-                kwargs.get("layout_weights"),
-                kwargs.get("layout_config_file"),
-                kwargs.get("device")
+                kwargs.get('layout_weights'),
+                kwargs.get('layout_config_file'),
+                kwargs.get('device')
            )
-        elif kwargs.get("layout_model_name") == MODEL_NAME.DocLayout_YOLO:
+        elif kwargs.get('layout_model_name') == MODEL_NAME.DocLayout_YOLO:
            atom_model = doclayout_yolo_model_init(
-                kwargs.get("doclayout_yolo_weights"),
-                kwargs.get("device")
+                kwargs.get('doclayout_yolo_weights'),
+                kwargs.get('device')
            )
    elif model_name == AtomicModel.MFD:
        atom_model = mfd_model_init(
-            kwargs.get("mfd_weights"),
-            kwargs.get("device")
+            kwargs.get('mfd_weights'),
+            kwargs.get('device')
        )
    elif model_name == AtomicModel.MFR:
        atom_model = mfr_model_init(
-            kwargs.get("mfr_weight_dir"),
-            kwargs.get("mfr_cfg_path"),
-            kwargs.get("device")
+            kwargs.get('mfr_weight_dir'),
+            kwargs.get('mfr_cfg_path'),
+            kwargs.get('device')
        )
    elif model_name == AtomicModel.OCR:
        atom_model = ocr_model_init(
-            kwargs.get("ocr_show_log"),
-            kwargs.get("det_db_box_thresh"),
-            kwargs.get("lang")
+            kwargs.get('ocr_show_log'),
+            kwargs.get('det_db_box_thresh'),
+            kwargs.get('lang')
        )
    elif model_name == AtomicModel.Table:
        atom_model = table_model_init(
-            kwargs.get("table_model_name"),
-            kwargs.get("table_model_path"),
-            kwargs.get("table_max_time"),
-            kwargs.get("device")
+            kwargs.get('table_model_name'),
+            kwargs.get('table_model_path'),
+            kwargs.get('table_max_time'),
+            kwargs.get('device')
        )
    else:
-        logger.error("model name not allow")
+        logger.error('model name not allow')
        exit(1)

    if atom_model is None:
-        logger.error("model init failed")
+        logger.error('model init failed')
        exit(1)
    else:
        return atom_model
--- a/magic_pdf/model/sub_modules/ocr/paddleocr/ocr_utils.py
+++ b/magic_pdf/model/sub_modules/ocr/paddleocr/ocr_utils.py
@@ -71,7 +71,13 @@ def remove_intervals(original, masks):

 def update_det_boxes(dt_boxes, mfd_res):
    new_dt_boxes = []
+    angle_boxes_list = []
    for text_box in dt_boxes:
+
+        if calculate_is_angle(text_box):
+            angle_boxes_list.append(text_box)
+            continue
+
        text_bbox = points_to_bbox(text_box)
        masks_list = []
        for mf_box in mfd_res:
@@ -85,6 +91,9 @@ def update_det_boxes(dt_boxes, mfd_res):
            temp_dt_box.append(bbox_to_points([text_remove_mask[0], text_bbox[1], text_remove_mask[1], text_bbox[3]]))
        if len(temp_dt_box) > 0:
            new_dt_boxes.extend(temp_dt_box)
+
+    new_dt_boxes.extend(angle_boxes_list)
+
    return new_dt_boxes


@@ -143,9 +152,11 @@ def merge_det_boxes(dt_boxes):
    angle_boxes_list = []
    for text_box in dt_boxes:
        text_bbox = points_to_bbox(text_box)
-        if text_bbox[2] <= text_bbox[0] or text_bbox[3] <= text_bbox[1]:
+
+        if calculate_is_angle(text_box):
            angle_boxes_list.append(text_box)
            continue
+
        text_box_dict = {
            'bbox': text_bbox,
            'type': 'text',
@@ -200,15 +211,21 @@ def get_ocr_result_list(ocr_res, useful_list):
    ocr_result_list = []
    for box_ocr_res in ocr_res:

-        p1, p2, p3, p4 = box_ocr_res[0]
-        text, score = box_ocr_res[1]
-        average_angle_degrees = calculate_angle_degrees(box_ocr_res[0])
-        if average_angle_degrees > 0.5:
+        if len(box_ocr_res) == 2:
+            p1, p2, p3, p4 = box_ocr_res[0]
+            text, score = box_ocr_res[1]
+        else:
+            p1, p2, p3, p4 = box_ocr_res
+            text, score = "", 1
+        # average_angle_degrees = calculate_angle_degrees(box_ocr_res[0])
+        # if average_angle_degrees > 0.5:
+        poly = [p1, p2, p3, p4]
+        if calculate_is_angle(poly):
            # logger.info(f"average_angle_degrees: {average_angle_degrees}, text: {text}")
            # 与x轴的夹角超过0.5度，对边界做一下矫正
            # 计算几何中心
-            x_center = sum(point[0] for point in box_ocr_res[0]) / 4
-            y_center = sum(point[1] for point in box_ocr_res[0]) / 4
+            x_center = sum(point[0] for point in poly) / 4
+            y_center = sum(point[1] for point in poly) / 4
            new_height = ((p4[1] - p1[1]) + (p3[1] - p2[1])) / 2
            new_width = p3[0] - p1[0]
            p1 = [x_center - new_width / 2, y_center - new_height / 2]
@@ -257,3 +274,12 @@ def calculate_angle_degrees(poly):
    # logger.info(f"average_angle_degrees: {average_angle_degrees}")
    return average_angle_degrees

+
+def calculate_is_angle(poly):
+    p1, p2, p3, p4 = poly
+    height = ((p4[1] - p1[1]) + (p3[1] - p2[1])) / 2
+    if 0.8 * height <= (p3[1] - p1[1]) <= 1.2 * height:
+        return False
+    else:
+        # logger.info((p3[1] - p1[1])/height)
+        return True
\ No newline at end of file
--- a/magic_pdf/model/sub_modules/ocr/paddleocr/ppocr_273_mod.py
+++ b/magic_pdf/model/sub_modules/ocr/paddleocr/ppocr_273_mod.py
@@ -78,9 +78,18 @@ class ModifiedPaddleOCR(PaddleOCR):
            for idx, img in enumerate(imgs):
                img = preprocess_image(img)
                dt_boxes, elapse = self.text_detector(img)
-                if not dt_boxes:
+                if dt_boxes is None:
                    ocr_res.append(None)
                    continue
+                dt_boxes = sorted_boxes(dt_boxes)
+                # merge_det_boxes 和 update_det_boxes 都会把poly转成bbox再转回poly，因此需要过滤所有倾斜程度较大的文本框
+                dt_boxes = merge_det_boxes(dt_boxes)
+                if mfd_res:
+                    bef = time.time()
+                    dt_boxes = update_det_boxes(dt_boxes, mfd_res)
+                    aft = time.time()
+                    logger.debug("split text box by formula, new dt_boxes num : {}, elapsed : {}".format(
+                        len(dt_boxes), aft - bef))
                tmp_res = [box.tolist() for box in dt_boxes]
                ocr_res.append(tmp_res)
            return ocr_res
@@ -125,9 +134,8 @@ class ModifiedPaddleOCR(PaddleOCR):

        dt_boxes = sorted_boxes(dt_boxes)

-        # @todo 目前是在bbox层merge，对倾斜文本行的兼容性不佳，需要修改成支持poly的merge
-        # dt_boxes = merge_det_boxes(dt_boxes)
-
+        # merge_det_boxes 和 update_det_boxes 都会把poly转成bbox再转回poly，因此需要过滤所有倾斜程度较大的文本框
+        dt_boxes = merge_det_boxes(dt_boxes)

        if mfd_res:
            bef = time.time()

--- a/magic_pdf/model/sub_modules/table/rapidtable/rapid_table.py
+++ b/magic_pdf/model/sub_modules/table/rapidtable/rapid_table.py
@@ -10,5 +10,7 @@ class RapidTableModel(object):

    def predict(self, image):
        ocr_result, _ = self.ocr_engine(np.asarray(image))
+        if ocr_result is None:
+            return None, None, None
        html_code, table_cell_bboxes, elapse = self.table_model(np.asarray(image), ocr_result)
        return html_code, table_cell_bboxes, elapse
\ No newline at end of file
--- a/magic_pdf/model/sub_modules/table/tablemaster/tablemaster_paddle.py
+++ b/magic_pdf/model/sub_modules/table/tablemaster/tablemaster_paddle.py
+import os
+
 import cv2
+import numpy as np
 from paddleocr.ppstructure.table.predict_table import TableSystem
 from paddleocr.ppstructure.utility import init_args
-from magic_pdf.libs.Constants import *
-import os
 from PIL import Image
-import numpy as np
+
+from magic_pdf.config.constants import *  # noqa: F403


 class TableMasterPaddleModel(object):
-    """
-        This class is responsible for converting image of table into HTML format using a pre-trained model.
+    """This class is responsible for converting image of table into HTML format
+    using a pre-trained model.

-        Attributes:
-        - table_sys: An instance of TableSystem initialized with parsed arguments.
+    Attributes:
+    - table_sys: An instance of TableSystem initialized with parsed arguments.

-        Methods:
-        - __init__(config): Initializes the model with configuration parameters.
-        - img2html(image): Converts a PIL Image or NumPy array to HTML string.
-        - parse_args(**kwargs): Parses configuration arguments.
+    Methods:
+    - __init__(config): Initializes the model with configuration parameters.
+    - img2html(image): Converts a PIL Image or NumPy array to HTML string.
+    - parse_args(**kwargs): Parses configuration arguments.
    """

    def __init__(self, config):
@@ -40,30 +42,30 @@ class TableMasterPaddleModel(object):
            image = np.asarray(image)
            image = cv2.cvtColor(image, cv2.COLOR_RGB2BGR)
        pred_res, _ = self.table_sys(image)
-        pred_html = pred_res["html"]
+        pred_html = pred_res['html']
        # res = '<td><table  border="1">' + pred_html.replace("<html><body><table>", "").replace(
        # "</table></body></html>","") + "</table></td>\n"
        return pred_html

    def parse_args(self, **kwargs):
        parser = init_args()
-        model_dir = kwargs.get("model_dir")
-        table_model_dir = os.path.join(model_dir, TABLE_MASTER_DIR)
-        table_char_dict_path = os.path.join(model_dir, TABLE_MASTER_DICT)
-        det_model_dir = os.path.join(model_dir, DETECT_MODEL_DIR)
-        rec_model_dir = os.path.join(model_dir, REC_MODEL_DIR)
-        rec_char_dict_path = os.path.join(model_dir, REC_CHAR_DICT)
-        device = kwargs.get("device", "cpu")
-        use_gpu = True if device.startswith("cuda") else False
+        model_dir = kwargs.get('model_dir')
+        table_model_dir = os.path.join(model_dir, TABLE_MASTER_DIR)  # noqa: F405
+        table_char_dict_path = os.path.join(model_dir, TABLE_MASTER_DICT)  # noqa: F405
+        det_model_dir = os.path.join(model_dir, DETECT_MODEL_DIR)  # noqa: F405
+        rec_model_dir = os.path.join(model_dir, REC_MODEL_DIR)  # noqa: F405
+        rec_char_dict_path = os.path.join(model_dir, REC_CHAR_DICT)  # noqa: F405
+        device = kwargs.get('device', 'cpu')
+        use_gpu = True if device.startswith('cuda') else False
        config = {
-            "use_gpu": use_gpu,
-            "table_max_len": kwargs.get("table_max_len", TABLE_MAX_LEN),
-            "table_algorithm": "TableMaster",
-            "table_model_dir": table_model_dir,
-            "table_char_dict_path": table_char_dict_path,
-            "det_model_dir": det_model_dir,
-            "rec_model_dir": rec_model_dir,
-            "rec_char_dict_path": rec_char_dict_path,
+            'use_gpu': use_gpu,
+            'table_max_len': kwargs.get('table_max_len', TABLE_MAX_LEN),  # noqa: F405
+            'table_algorithm': 'TableMaster',
+            'table_model_dir': table_model_dir,
+            'table_char_dict_path': table_char_dict_path,
+            'det_model_dir': det_model_dir,
+            'rec_model_dir': rec_model_dir,
+            'rec_char_dict_path': rec_char_dict_path,
        }
        parser.set_defaults(**config)
        return parser.parse_args([])
--- a/magic_pdf/para/para_pipeline.py
+++ b/magic_pdf/para/para_pipeline.py
-import os
-import json
-
-from magic_pdf.para.commons import *
-
-from magic_pdf.para.raw_processor import RawBlockProcessor
-from magic_pdf.para.layout_match_processor import LayoutFilterProcessor
-from magic_pdf.para.stats import BlockStatisticsCalculator
-from magic_pdf.para.stats import DocStatisticsCalculator
-from magic_pdf.para.title_processor import TitleProcessor
-from magic_pdf.para.block_termination_processor import BlockTerminationProcessor
-from magic_pdf.para.block_continuation_processor import BlockContinuationProcessor
-from magic_pdf.para.draw import DrawAnnos
-from magic_pdf.para.exceptions import (
-    DenseSingleLineBlockException,
-    TitleDetectionException,
-    TitleLevelException,
-    ParaSplitException,
-    ParaMergeException,
-    DiscardByException,
-)
-
-
-if sys.version_info[0] >= 3:
-    sys.stdout.reconfigure(encoding="utf-8")  # type: ignore
-
-
-class ParaProcessPipeline:
-    def __init__(self) -> None:
-        pass
-
-    def para_process_pipeline(self, pdf_info_dict, para_debug_mode=None, input_pdf_path=None, output_pdf_path=None):
-        """
-        This function processes the paragraphs, including:
-        1. Read raw input json file into pdf_dic
-        2. Detect and replace equations
-        3. Combine spans into a natural line
-        4. Check if the paragraphs are inside bboxes passed from "layout_bboxes" key
-        5. Compute statistics for each block
-        6. Detect titles in the document
-        7. Detect paragraphs inside each block
-        8. Divide the level of the titles
-        9. Detect and combine paragraphs from different blocks into one paragraph
-        10. Check whether the final results after checking headings, dividing paragraphs within blocks, and merging paragraphs between blocks are plausible and reasonable.
-        11. Draw annotations on the pdf file
-
-        Parameters
-        ----------
-        pdf_dic_json_fpath : str
-            path to the pdf dictionary json file.
-            Notice: data noises, including overlap blocks, header, footer, watermark, vertical margin note have been removed already.
-        input_pdf_doc : str
-            path to the input pdf file
-        output_pdf_path : str
-            path to the output pdf file
-
-        Returns
-        -------
-        pdf_dict : dict
-            result dictionary
-        """
-
-        error_info = None
-
-        output_json_file = ""
-        output_dir = ""
-
-        if input_pdf_path is not None:
-            input_pdf_path = os.path.abspath(input_pdf_path)
-
-            # print_green_on_red(f">>>>>>>>>>>>>>>>>>> Process the paragraphs of {input_pdf_path}")
-
-        if output_pdf_path is not None:
-            output_dir = os.path.dirname(output_pdf_path)
-            output_json_file = f"{output_dir}/pdf_dic.json"
-
-        def __save_pdf_dic(pdf_dic, output_pdf_path, stage="0", para_debug_mode=para_debug_mode):
-            """
-            Save the pdf_dic to a json file
-            """
-            output_pdf_file_name = os.path.basename(output_pdf_path)
-            # output_dir = os.path.dirname(output_pdf_path)
-            output_dir = "\\tmp\\pdf_parse"
-            output_pdf_file_name = output_pdf_file_name.replace(".pdf", f"_stage_{stage}.json")
-            pdf_dic_json_fpath = os.path.join(output_dir, output_pdf_file_name)
-
-            if not os.path.exists(output_dir):
-                os.makedirs(output_dir)
-
-            if para_debug_mode == "full":
-                with open(pdf_dic_json_fpath, "w", encoding="utf-8") as f:
-                    json.dump(pdf_dic, f, indent=2, ensure_ascii=False)
-
-            # Validate the output already exists
-            if not os.path.exists(pdf_dic_json_fpath):
-                print_red(f"Failed to save the pdf_dic to {pdf_dic_json_fpath}")
-                return None
-            else:
-                print_green(f"Succeed to save the pdf_dic to {pdf_dic_json_fpath}")
-
-            return pdf_dic_json_fpath
-
-        """
-        Preprocess the lines of block
-        """
-        # Find and replace the interline and inline equations, should be better done before the paragraph processing
-        # Create "para_blocks" for each page.
-        # equationProcessor = EquationsProcessor()
-        # pdf_dic = equationProcessor.batch_process_blocks(pdf_info_dict)
-
-        # Combine spans into a natural line
-        rawBlockProcessor = RawBlockProcessor()
-        pdf_dic = rawBlockProcessor.batch_process_blocks(pdf_info_dict)
-        # print(f"pdf_dic['page_0']['para_blocks'][0]: {pdf_dic['page_0']['para_blocks'][0]}", end="\n\n")
-
-        # Check if the paragraphs are inside bboxes passed from "layout_bboxes" key
-        layoutFilter = LayoutFilterProcessor()
-        pdf_dic = layoutFilter.batch_process_blocks(pdf_dic)
-
-        # Compute statistics for each block
-        blockStatisticsCalculator = BlockStatisticsCalculator()
-        pdf_dic = blockStatisticsCalculator.batch_process_blocks(pdf_dic)
-        # print(f"pdf_dic['page_0']['para_blocks'][0]: {pdf_dic['page_0']['para_blocks'][0]}", end="\n\n")
-
-        # Compute statistics for all blocks(namely this pdf document)
-        docStatisticsCalculator = DocStatisticsCalculator()
-        pdf_dic = docStatisticsCalculator.calc_stats_of_doc(pdf_dic)
-        # print(f"pdf_dic['statistics']: {pdf_dic['statistics']}", end="\n\n")
-
-        # Dump the first three stages of pdf_dic to a json file
-        if para_debug_mode == "full":
-            pdf_dic_json_fpath = __save_pdf_dic(pdf_dic, output_pdf_path, stage="0", para_debug_mode=para_debug_mode)
-
-        """
-        Detect titles in the document
-        """
-        doc_statistics = pdf_dic["statistics"]
-        titleProcessor = TitleProcessor(doc_statistics)
-        pdf_dic = titleProcessor.batch_process_blocks_detect_titles(pdf_dic)
-
-        if para_debug_mode == "full":
-            pdf_dic_json_fpath = __save_pdf_dic(pdf_dic, output_pdf_path, stage="1", para_debug_mode=para_debug_mode)
-
-        """
-        Detect and divide the level of the titles
-        """
-        titleProcessor = TitleProcessor()
-
-        pdf_dic = titleProcessor.batch_process_blocks_recog_title_level(pdf_dic)
-
-        if para_debug_mode == "full":
-            pdf_dic_json_fpath = __save_pdf_dic(pdf_dic, output_pdf_path, stage="2", para_debug_mode=para_debug_mode)
-
-        """
-        Detect and split paragraphs inside each block
-        """
-        blockInnerParasProcessor = BlockTerminationProcessor()
-
-        pdf_dic = blockInnerParasProcessor.batch_process_blocks(pdf_dic)
-
-        if para_debug_mode == "full":
-            pdf_dic_json_fpath = __save_pdf_dic(pdf_dic, output_pdf_path, stage="3", para_debug_mode=para_debug_mode)
-
-        # pdf_dic_json_fpath = __save_pdf_dic(pdf_dic, output_pdf_path, stage="3", para_debug_mode="full")
-        # print_green(f"pdf_dic_json_fpath: {pdf_dic_json_fpath}")
-
-        """
-        Detect and combine paragraphs from different blocks into one paragraph
-        """
-        blockContinuationProcessor = BlockContinuationProcessor()
-
-        pdf_dic = blockContinuationProcessor.batch_tag_paras(pdf_dic)
-        pdf_dic = blockContinuationProcessor.batch_merge_paras(pdf_dic)
-
-        if para_debug_mode == "full":
-            pdf_dic_json_fpath = __save_pdf_dic(pdf_dic, output_pdf_path, stage="4", para_debug_mode=para_debug_mode)
-
-        # pdf_dic_json_fpath = __save_pdf_dic(pdf_dic, output_pdf_path, stage="4", para_debug_mode="full")
-        # print_green(f"pdf_dic_json_fpath: {pdf_dic_json_fpath}")
-
-        """
-        Discard pdf files by checking exceptions and return the error info to the caller
-        """
-        discardByException = DiscardByException()
-
-        is_discard_by_single_line_block = discardByException.discard_by_single_line_block(
-            pdf_dic, exception=DenseSingleLineBlockException()
-        )
-        is_discard_by_title_detection = discardByException.discard_by_title_detection(
-            pdf_dic, exception=TitleDetectionException()
-        )
-        is_discard_by_title_level = discardByException.discard_by_title_level(pdf_dic, exception=TitleLevelException())
-        is_discard_by_split_para = discardByException.discard_by_split_para(pdf_dic, exception=ParaSplitException())
-        is_discard_by_merge_para = discardByException.discard_by_merge_para(pdf_dic, exception=ParaMergeException())
-
-        """
-        if any(
-            info is not None
-            for info in [
-                is_discard_by_single_line_block,
-                is_discard_by_title_detection,
-                is_discard_by_title_level,
-                is_discard_by_split_para,
-                is_discard_by_merge_para,
-            ]
-        ):
-            error_info = next(
-                (
-                    info
-                    for info in [
-                        is_discard_by_single_line_block,
-                        is_discard_by_title_detection,
-                        is_discard_by_title_level,
-                        is_discard_by_split_para,
-                        is_discard_by_merge_para,
-                    ]
-                    if info is not None
-                ),
-                None,
-            )
-            return pdf_dic, error_info
-
-        if any(
-            info is not None
-            for info in [
-                is_discard_by_single_line_block,
-                is_discard_by_title_detection,
-                is_discard_by_title_level,
-                is_discard_by_split_para,
-                is_discard_by_merge_para,
-            ]
-        ):
-            error_info = next(
-                (
-                    info
-                    for info in [
-                        is_discard_by_single_line_block,
-                        is_discard_by_title_detection,
-                        is_discard_by_title_level,
-                        is_discard_by_split_para,
-                        is_discard_by_merge_para,
-                    ]
-                    if info is not None
-                ),
-                None,
-            )
-            return pdf_dic, error_info
-        """
-
-        """
-        Dump the final pdf_dic to a json file
-        """
-        if para_debug_mode is not None:
-            with open(output_json_file, "w", encoding="utf-8") as f:
-                json.dump(pdf_info_dict, f, ensure_ascii=False, indent=4)
-
-        """
-        Draw the annotations
-        """
-
-        if is_discard_by_single_line_block is not None:
-            error_info = is_discard_by_single_line_block
-        elif is_discard_by_title_detection is not None:
-            error_info = is_discard_by_title_detection
-        elif is_discard_by_title_level is not None:
-            error_info = is_discard_by_title_level
-        elif is_discard_by_split_para is not None:
-            error_info = is_discard_by_split_para
-        elif is_discard_by_merge_para is not None:
-            error_info = is_discard_by_merge_para
-
-        if error_info is not None:
-            return pdf_dic, error_info
-
-        """
-        Dump the final pdf_dic to a json file
-        """
-        if para_debug_mode is not None:
-            with open(output_json_file, "w", encoding="utf-8") as f:
-                json.dump(pdf_info_dict, f, ensure_ascii=False, indent=4)
-
-        """
-        Draw the annotations
-        """
-        if para_debug_mode is not None:
-            drawAnnos = DrawAnnos()
-            drawAnnos.draw_annos(input_pdf_path, pdf_dic, output_pdf_path)
-
-        """
-        Remove the intermediate files which are generated in the process of paragraph processing if debug_mode is simple
-        """
-        if para_debug_mode is not None:
-            for fpath in os.listdir(output_dir):
-                if fpath.endswith(".json") and "stage" in fpath:
-                    os.remove(os.path.join(output_dir, fpath))
-
-        return pdf_dic, error_info
--- a/magic_pdf/para/para_split.py
+++ b/magic_pdf/para/para_split.py
--- a/magic_pdf/para/para_split_v2.py
+++ b/magic_pdf/para/para_split_v2.py
--- a/magic_pdf/para/para_split_v3.py
+++ b/magic_pdf/para/para_split_v3.py
 import copy

-from loguru import logger
-
-from magic_pdf.libs.Constants import LINES_DELETED, CROSS_PAGE
-from magic_pdf.libs.ocr_content_type import BlockType, ContentType
-
-LINE_STOP_FLAG = ('.', '!', '?', '。', '！', '？', ')', '）', '"', '”', ':', '：', ';', '；')
+from magic_pdf.config.constants import CROSS_PAGE, LINES_DELETED
+from magic_pdf.config.ocr_content_type import BlockType, ContentType
+
+LINE_STOP_FLAG = (
+    '.',
+    '!',
+    '?',
+    '。',
+    '！',
+    '？',
+    ')',
+    '）',
+    '"',
+    '”',
+    ':',
+    '：',
+    ';',
+    '；',
+)
 LIST_END_FLAG = ('.', '。', ';', '；')


 class ListLineTag:
-    IS_LIST_START_LINE = "is_list_start_line"
-    IS_LIST_END_LINE = "is_list_end_line"
+    IS_LIST_START_LINE = 'is_list_start_line'
+    IS_LIST_END_LINE = 'is_list_end_line'


 def __process_blocks(blocks):
@@ -27,12 +40,14 @@ def __process_blocks(blocks):

        # 如果当前块是 text 类型
        if current_block['type'] == 'text':
-            current_block["bbox_fs"] = copy.deepcopy(current_block["bbox"])
-            if 'lines' in current_block and len(current_block["lines"]) > 0:
-                current_block['bbox_fs'] = [min([line['bbox'][0] for line in current_block['lines']]),
-                                            min([line['bbox'][1] for line in current_block['lines']]),
-                                            max([line['bbox'][2] for line in current_block['lines']]),
-                                            max([line['bbox'][3] for line in current_block['lines']])]
+            current_block['bbox_fs'] = copy.deepcopy(current_block['bbox'])
+            if 'lines' in current_block and len(current_block['lines']) > 0:
+                current_block['bbox_fs'] = [
+                    min([line['bbox'][0] for line in current_block['lines']]),
+                    min([line['bbox'][1] for line in current_block['lines']]),
+                    max([line['bbox'][2] for line in current_block['lines']]),
+                    max([line['bbox'][3] for line in current_block['lines']]),
+                ]
            current_group.append(current_block)

        # 检查下一个块是否存在
@@ -64,6 +79,7 @@ def __is_list_or_index_block(block):
        line_height = first_line['bbox'][3] - first_line['bbox'][1]
        block_weight = block['bbox_fs'][2] - block['bbox_fs'][0]
        block_height = block['bbox_fs'][3] - block['bbox_fs'][1]
+        page_weight, page_height = block['page_size']

        left_close_num = 0
        left_not_close_num = 0
@@ -75,10 +91,17 @@ def __is_list_or_index_block(block):
        multiple_para_flag = False
        last_line = block['lines'][-1]

+        if page_weight == 0:
+            block_weight_radio = 0
+        else:
+            block_weight_radio = block_weight / page_weight
+        # logger.info(f"block_weight_radio: {block_weight_radio}")
+
        # 如果首行左边不顶格而右边顶格,末行左边顶格而右边不顶格 （第一行可能可以右边不顶格）
-        if (first_line['bbox'][0] - block['bbox_fs'][0] > line_height / 2 and
-                abs(last_line['bbox'][0] - block['bbox_fs'][0]) < line_height / 2 and
-                block['bbox_fs'][2] - last_line['bbox'][2] > line_height
+        if (
+            first_line['bbox'][0] - block['bbox_fs'][0] > line_height / 2
+            and abs(last_line['bbox'][0] - block['bbox_fs'][0]) < line_height / 2
+            and block['bbox_fs'][2] - last_line['bbox'][2] > line_height
        ):
            multiple_para_flag = True

@@ -86,14 +109,14 @@ def __is_list_or_index_block(block):
            line_mid_x = (line['bbox'][0] + line['bbox'][2]) / 2
            block_mid_x = (block['bbox_fs'][0] + block['bbox_fs'][2]) / 2
            if (
-                    line['bbox'][0] - block['bbox_fs'][0] > 0.8 * line_height and
-                    block['bbox_fs'][2] - line['bbox'][2] > 0.8 * line_height
+                line['bbox'][0] - block['bbox_fs'][0] > 0.8 * line_height
+                and block['bbox_fs'][2] - line['bbox'][2] > 0.8 * line_height
            ):
                external_sides_not_close_num += 1
            if abs(line_mid_x - block_mid_x) < line_height / 2:
                center_close_num += 1

-            line_text = ""
+            line_text = ''

            for span in line['spans']:
                span_type = span['type']
@@ -114,7 +137,12 @@ def __is_list_or_index_block(block):
                right_close_num += 1
            else:
                # 右侧不顶格情况下是否有一段距离，拍脑袋用0.3block宽度做阈值
-                closed_area = 0.26 * block_weight
+                # block宽的阈值可以小些，block窄的阈值要大
+
+                if block_weight_radio >= 0.5:
+                    closed_area = 0.26 * block_weight
+                else:
+                    closed_area = 0.36 * block_weight
                if block['bbox_fs'][2] - line['bbox'][2] > closed_area:
                    right_not_close_num += 1

@@ -136,15 +164,19 @@ def __is_list_or_index_block(block):
                    if line_text[-1].isdigit():
                        num_end_count += 1

-            if num_start_count / len(lines_text_list) >= 0.8 or num_end_count / len(lines_text_list) >= 0.8:
+            if (
+                num_start_count / len(lines_text_list) >= 0.8
+                or num_end_count / len(lines_text_list) >= 0.8
+            ):
                line_num_flag = True
            if flag_end_count / len(lines_text_list) >= 0.8:
                line_end_flag = True

        # 有的目录右侧不贴边, 目前认为左边或者右边有一边全贴边，且符合数字规则极为index
-        if ((left_close_num / len(block['lines']) >= 0.8 or right_close_num / len(block['lines']) >= 0.8)
-                and line_num_flag
-        ):
+        if (
+            left_close_num / len(block['lines']) >= 0.8
+            or right_close_num / len(block['lines']) >= 0.8
+        ) and line_num_flag:
            for line in block['lines']:
                line[ListLineTag.IS_LIST_START_LINE] = True
            return BlockType.Index
@@ -152,17 +184,21 @@ def __is_list_or_index_block(block):
        # 全部line都居中的特殊list识别，每行都需要换行，特征是多行，且大多数行都前后not_close,每line中点x坐标接近
        # 补充条件block的长宽比有要求
        elif (
-                external_sides_not_close_num >= 2 and
-                center_close_num == len(block['lines']) and
-                external_sides_not_close_num / len(block['lines']) >= 0.5 and
-                block_height / block_weight > 0.4
+            external_sides_not_close_num >= 2
+            and center_close_num == len(block['lines'])
+            and external_sides_not_close_num / len(block['lines']) >= 0.5
+            and block_height / block_weight > 0.4
        ):
            for line in block['lines']:
                line[ListLineTag.IS_LIST_START_LINE] = True
            return BlockType.List

-        elif left_close_num >= 2 and (
-                right_not_close_num >= 2 or line_end_flag or left_not_close_num >= 2) and not multiple_para_flag:
+        elif (
+            left_close_num >= 2
+            and (right_not_close_num >= 2 or line_end_flag or left_not_close_num >= 2)
+            and not multiple_para_flag
+            # and block_weight_radio > 0.27
+        ):
            # 处理一种特殊的没有缩进的list，所有行都贴左边，通过右边的空隙判断是否是item尾
            if left_close_num / len(block['lines']) > 0.8:
                # 这种是每个item只有一行，且左边都贴边的短item list
@@ -173,10 +209,15 @@ def __is_list_or_index_block(block):
                # 这种是大部分line item 都有结束标识符的情况，按结束标识符区分不同item
                elif line_end_flag:
                    for i, line in enumerate(block['lines']):
-                        if len(lines_text_list[i]) > 0 and lines_text_list[i][-1] in LIST_END_FLAG:
+                        if (
+                            len(lines_text_list[i]) > 0
+                            and lines_text_list[i][-1] in LIST_END_FLAG
+                        ):
                            line[ListLineTag.IS_LIST_END_LINE] = True
                            if i + 1 < len(block['lines']):
-                                block['lines'][i + 1][ListLineTag.IS_LIST_START_LINE] = True
+                                block['lines'][i + 1][
+                                    ListLineTag.IS_LIST_START_LINE
+                                ] = True
                # line item基本没有结束标识符，而且也没有缩进，按右侧空隙判断哪些是item end
                else:
                    line_start_flag = False
@@ -185,7 +226,10 @@ def __is_list_or_index_block(block):
                            line[ListLineTag.IS_LIST_START_LINE] = True
                            line_start_flag = False

-                        if abs(block['bbox_fs'][2] - line['bbox'][2]) > 0.1 * block_weight:
+                        if (
+                            abs(block['bbox_fs'][2] - line['bbox'][2])
+                            > 0.1 * block_weight
+                        ):
                            line[ListLineTag.IS_LIST_END_LINE] = True
                            line_start_flag = True
            # 一种有缩进的特殊有序list,start line 左侧不贴边且以数字开头，end line 以 IS_LIST_END_FLAG 结尾且数量和start line 一致
@@ -223,18 +267,25 @@ def __merge_2_text_blocks(block1, block2):
            if len(last_line['spans']) > 0:
                last_span = last_line['spans'][-1]
                line_height = last_line['bbox'][3] - last_line['bbox'][1]
-                if (abs(block2['bbox_fs'][2] - last_line['bbox'][2]) < line_height and
-                        not last_span['content'].endswith(LINE_STOP_FLAG) and
-                        # 两个block宽度差距超过2倍也不合并
-                        abs(block1_weight - block2_weight) < min_block_weight
-                ):
-                    if block1['page_num'] != block2['page_num']:
-                        for line in block1['lines']:
-                            for span in line['spans']:
-                                span[CROSS_PAGE] = True
-                    block2['lines'].extend(block1['lines'])
-                    block1['lines'] = []
-                    block1[LINES_DELETED] = True
+                if len(first_line['spans']) > 0:
+                    first_span = first_line['spans'][0]
+                    if len(first_span['content']) > 0:
+                        span_start_with_num = first_span['content'][0].isdigit()
+                        if (
+                            abs(block2['bbox_fs'][2] - last_line['bbox'][2])
+                            < line_height
+                            and not last_span['content'].endswith(LINE_STOP_FLAG)
+                            # 两个block宽度差距超过2倍也不合并
+                            and abs(block1_weight - block2_weight) < min_block_weight
+                            and not span_start_with_num
+                        ):
+                            if block1['page_num'] != block2['page_num']:
+                                for line in block1['lines']:
+                                    for span in line['spans']:
+                                        span[CROSS_PAGE] = True
+                            block2['lines'].extend(block1['lines'])
+                            block1['lines'] = []
+                            block1[LINES_DELETED] = True

    return block1, block2

@@ -263,7 +314,6 @@ def __is_list_group(text_blocks_group):
 def __para_merge_page(blocks):
    page_text_blocks_groups = __process_blocks(blocks)
    for text_blocks_group in page_text_blocks_groups:
-
        if len(text_blocks_group) > 0:
            # 需要先在合并前对所有block判断是否为list or index block
            for block in text_blocks_group:
@@ -272,7 +322,6 @@ def __para_merge_page(blocks):
                # logger.info(f"{block['type']}:{block}")

        if len(text_blocks_group) > 1:
-
            # 在合并前判断这个group 是否是一个 list group
            is_list_group = __is_list_group(text_blocks_group)

@@ -284,11 +333,18 @@ def __para_merge_page(blocks):
                if i - 1 >= 0:
                    prev_block = text_blocks_group[i - 1]

-                    if current_block['type'] == 'text' and prev_block['type'] == 'text' and not is_list_group:
+                    if (
+                        current_block['type'] == 'text'
+                        and prev_block['type'] == 'text'
+                        and not is_list_group
+                    ):
                        __merge_2_text_blocks(current_block, prev_block)
                    elif (
-                            (current_block['type'] == BlockType.List and prev_block['type'] == BlockType.List) or
-                            (current_block['type'] == BlockType.Index and prev_block['type'] == BlockType.Index)
+                        current_block['type'] == BlockType.List
+                        and prev_block['type'] == BlockType.List
+                    ) or (
+                        current_block['type'] == BlockType.Index
+                        and prev_block['type'] == BlockType.Index
                    ):
                        __merge_2_list_blocks(current_block, prev_block)

@@ -296,12 +352,13 @@ def __para_merge_page(blocks):
            continue


-def para_split(pdf_info_dict, debug_mode=False):
+def para_split(pdf_info_dict):
    all_blocks = []
    for page_num, page in pdf_info_dict.items():
        blocks = copy.deepcopy(page['preproc_blocks'])
        for block in blocks:
            block['page_num'] = page_num
+            block['page_size'] = page['page_size']
        all_blocks.extend(blocks)

    __para_merge_page(all_blocks)
@@ -317,4 +374,4 @@ if __name__ == '__main__':
    # 调用函数
    groups = __process_blocks(input_blocks)
    for group_index, group in enumerate(groups):
-        print(f"Group {group_index}: {group}")
+        print(f'Group {group_index}: {group}')
--- a/magic_pdf/pdf_parse_by_ocr.py
+++ b/magic_pdf/pdf_parse_by_ocr.py
@@ -9,6 +9,7 @@ def parse_pdf_by_ocr(pdf_bytes,
                     start_page_id=0,
                     end_page_id=None,
                     debug_mode=False,
+                     lang=None,
                     ):
    dataset = PymuDocDataset(pdf_bytes)
    return pdf_parse_union(dataset,
@@ -18,4 +19,5 @@ def parse_pdf_by_ocr(pdf_bytes,
                           start_page_id=start_page_id,
                           end_page_id=end_page_id,
                           debug_mode=debug_mode,
+                           lang=lang,
                           )
--- a/magic_pdf/pdf_parse_by_txt.py
+++ b/magic_pdf/pdf_parse_by_txt.py
@@ -10,6 +10,7 @@ def parse_pdf_by_txt(
    start_page_id=0,
    end_page_id=None,
    debug_mode=False,
+    lang=None,
 ):
    dataset = PymuDocDataset(pdf_bytes)
    return pdf_parse_union(dataset,
@@ -19,4 +20,5 @@ def parse_pdf_by_txt(
                           start_page_id=start_page_id,
                           end_page_id=end_page_id,
                           debug_mode=debug_mode,
+                           lang=lang,
                           )
--- a/magic_pdf/pdf_parse_union_core.py
+++ b/magic_pdf/pdf_parse_union_core.py
--- a/magic_pdf/pdf_parse_union_core_v2.py
+++ b/magic_pdf/pdf_parse_union_core_v2.py
@@ -7,18 +7,32 @@ from typing import List
 import torch
 from loguru import logger

+from magic_pdf.config.drop_reason import DropReason
 from magic_pdf.config.enums import SupportedPdfParseMethod
+from magic_pdf.config.ocr_content_type import BlockType, ContentType
 from magic_pdf.data.dataset import Dataset, PageableData
 from magic_pdf.libs.boxbase import calculate_overlap_area_in_bbox1_area_ratio
 from magic_pdf.libs.clean_memory import clean_memory
 from magic_pdf.libs.commons import fitz, get_delta_time
 from magic_pdf.libs.config_reader import get_local_layoutreader_model_dir
 from magic_pdf.libs.convert_utils import dict_to_list
-from magic_pdf.libs.drop_reason import DropReason
 from magic_pdf.libs.hash_utils import compute_md5
 from magic_pdf.libs.local_math import float_equal
-from magic_pdf.libs.ocr_content_type import ContentType, BlockType
+from magic_pdf.libs.pdf_image_tools import cut_image_to_pil_image
 from magic_pdf.model.magic_model import MagicModel
+
+os.environ['NO_ALBUMENTATIONS_UPDATE'] = '1'  # 禁止albumentations检查更新
+os.environ['YOLO_VERBOSE'] = 'False'  # disable yolo logger
+
+try:
+    import torchtext
+
+    if torchtext.__version__ >= "0.18.0":
+        torchtext.disable_torchtext_deprecation_warning()
+except ImportError:
+    pass
+from magic_pdf.model.sub_modules.model_init import AtomModelSingleton
+
 from magic_pdf.para.para_split_v3 import para_split
 from magic_pdf.pre_proc.citationmarker_remove import remove_citation_marker
 from magic_pdf.pre_proc.construct_page_dict import \
@@ -30,8 +44,8 @@ from magic_pdf.pre_proc.equations_replace import (
 from magic_pdf.pre_proc.ocr_detect_all_bboxes import \
    ocr_prepare_bboxes_for_layout_split_v2
 from magic_pdf.pre_proc.ocr_dict_merge import (fill_spans_in_blocks,
-                                               fix_discarded_block,
-                                               fix_block_spans_v2)
+                                               fix_block_spans_v2,
+                                               fix_discarded_block)
 from magic_pdf.pre_proc.ocr_span_list_modify import (
    get_qa_need_list_v2, remove_overlaps_low_confidence_spans,
    remove_overlaps_min_spans)
@@ -74,7 +88,151 @@ def __replace_STX_ETX(text_str: str):
    return text_str


-def txt_spans_extract(pdf_page, inline_equations, interline_equations):
+def chars_to_content(span):
+        # # 先给chars按char['bbox']的x坐标排序
+        # span['chars'] = sorted(span['chars'], key=lambda x: x['bbox'][0])
+
+        # 先给chars按char['bbox']的中心点的x坐标排序
+        span['chars'] = sorted(span['chars'], key=lambda x: (x['bbox'][0] + x['bbox'][2]) / 2)
+        content = ''
+
+        # 求char的平均宽度
+        if len(span['chars']) == 0:
+            span['content'] = content
+            del span['chars']
+            return
+        else:
+            char_width_sum = sum([char['bbox'][2] - char['bbox'][0] for char in span['chars']])
+            char_avg_width = char_width_sum / len(span['chars'])
+
+        for char in span['chars']:
+            # 如果下一个char的x0和上一个char的x1距离超过一个字符宽度，则需要在中间插入一个空格
+            if char['bbox'][0] - span['chars'][span['chars'].index(char) - 1]['bbox'][2] > char_avg_width:
+                content += ' '
+            content += char['c']
+        span['content'] = __replace_STX_ETX(content)
+        del span['chars']
+
+
+LINE_STOP_FLAG = ('.', '!', '?', '。', '！', '？', ')', '）', '"', '”', ':', '：', ';', '；', ']', '】', '}', '}', '>', '》', '、', ',', '，', '-', '—', '–',)
+def fill_char_in_spans(spans, all_chars):
+
+    for char in all_chars:
+        for span in spans:
+            # 判断char是否属于LINE_STOP_FLAG
+            if char['c'] in LINE_STOP_FLAG:
+                char_is_line_stop_flag = True
+            else:
+                char_is_line_stop_flag = False
+            if calculate_char_in_span(char['bbox'], span['bbox'], char_is_line_stop_flag):
+                span['chars'].append(char)
+                break
+
+    for span in spans:
+        chars_to_content(span)
+
+
+# 使用鲁棒性更强的中心点坐标判断
+def calculate_char_in_span(char_bbox, span_bbox, char_is_line_stop_flag):
+    char_center_x = (char_bbox[0] + char_bbox[2]) / 2
+    char_center_y = (char_bbox[1] + char_bbox[3]) / 2
+    span_center_y = (span_bbox[1] + span_bbox[3]) / 2
+    span_height = span_bbox[3] - span_bbox[1]
+
+    if (
+        span_bbox[0] < char_center_x < span_bbox[2]
+        and span_bbox[1] < char_center_y < span_bbox[3]
+        and abs(char_center_y - span_center_y) < span_height / 4  # 字符的中轴和span的中轴高度差不能超过1/4span高度
+    ):
+        return True
+    else:
+        # 如果char是LINE_STOP_FLAG，就不用中心点判定，换一种方案（左边界在span区域内，高度判定和之前逻辑一致）
+        # 主要是给结尾符号一个进入span的机会，这个char还应该离span右边界较近
+        if char_is_line_stop_flag:
+            if (
+                (span_bbox[2] - span_height) < char_bbox[0] < span_bbox[2]
+                and char_center_x > span_bbox[0]
+                and span_bbox[1] < char_center_y < span_bbox[3]
+                and abs(char_center_y - span_center_y) < span_height / 4
+            ):
+                return True
+        else:
+            return False
+
+
+def txt_spans_extract_v2(pdf_page, spans, all_bboxes, all_discarded_blocks, lang):
+
+    useful_spans = []
+    unuseful_spans = []
+    for span in spans:
+        for block in all_bboxes:
+            if block[7] in [BlockType.ImageBody, BlockType.TableBody, BlockType.InterlineEquation]:
+                continue
+            else:
+                if calculate_overlap_area_in_bbox1_area_ratio(span['bbox'], block[0:4]) > 0.5:
+                    useful_spans.append(span)
+                    break
+        for block in all_discarded_blocks:
+            if calculate_overlap_area_in_bbox1_area_ratio(span['bbox'], block[0:4]) > 0.5:
+                unuseful_spans.append(span)
+                break
+
+    text_blocks = pdf_page.get_text('rawdict', flags=fitz.TEXTFLAGS_TEXT)['blocks']
+
+    # @todo: 拿到char之后把倾斜角度较大的先删一遍
+    all_pymu_chars = []
+    for block in text_blocks:
+        for line in block['lines']:
+            for span in line['spans']:
+                all_pymu_chars.extend(span['chars'])
+
+    new_spans = []
+
+    for span in useful_spans:
+        if span['type'] in [ContentType.Text]:
+            span['chars'] = []
+            new_spans.append(span)
+
+    for span in unuseful_spans:
+        if span['type'] in [ContentType.Text]:
+            span['chars'] = []
+            new_spans.append(span)
+
+    fill_char_in_spans(new_spans, all_pymu_chars)
+
+    empty_spans = []
+    for span in new_spans:
+        if len(span['content']) == 0:
+            empty_spans.append(span)
+    if len(empty_spans) > 0:
+
+        # 初始化ocr模型
+        atom_model_manager = AtomModelSingleton()
+        ocr_model = atom_model_manager.get_atom_model(
+            atom_model_name="ocr",
+            ocr_show_log=False,
+            det_db_box_thresh=0.3,
+            lang=lang
+        )
+
+        for span in empty_spans:
+            spans.remove(span)
+            # 对span的bbox截图
+            span_img = cut_image_to_pil_image(span['bbox'], pdf_page, mode="cv2")
+            ocr_res = ocr_model.ocr(span_img, det=False)
+            # logger.info(f"ocr_res: {ocr_res}")
+            # logger.info(f"empty_span: {span}")
+            if ocr_res and len(ocr_res) > 0:
+                if len(ocr_res[0]) > 0:
+                    ocr_text, ocr_score = ocr_res[0][0]
+                    if ocr_score > 0.5 and len(ocr_text) > 0:
+                            span['content'] = ocr_text
+                            spans.append(span)
+
+    return spans
+
+
+def txt_spans_extract_v1(pdf_page, inline_equations, interline_equations):
    text_raw_blocks = pdf_page.get_text('dict', flags=fitz.TEXTFLAGS_TEXT)['blocks']
    char_level_text_blocks = pdf_page.get_text('rawdict', flags=fitz.TEXTFLAGS_TEXT)[
        'blocks'
@@ -164,8 +322,8 @@ class ModelSingleton:


 def do_predict(boxes: List[List[int]], model) -> List[int]:
-    from magic_pdf.model.sub_modules.reading_oreder.layoutreader.helpers import (boxes2inputs, parse_logits,
-                                                                                 prepare_inputs)
+    from magic_pdf.model.sub_modules.reading_oreder.layoutreader.helpers import (
+        boxes2inputs, parse_logits, prepare_inputs)

    inputs = boxes2inputs(boxes)
    inputs = prepare_inputs(inputs, model)
@@ -206,7 +364,9 @@ def cal_block_index(fix_blocks, sorted_bboxes):
                del block['real_lines']

        import numpy as np
-        from magic_pdf.model.sub_modules.reading_oreder.layoutreader.xycut import recursive_xy_cut
+
+        from magic_pdf.model.sub_modules.reading_oreder.layoutreader.xycut import \
+            recursive_xy_cut

        random_boxes = np.array(block_bboxes)
        np.random.shuffle(random_boxes)
@@ -291,7 +451,7 @@ def sort_lines_by_model(fix_blocks, page_w, page_h, line_height):
                    page_line_list.append(bbox)
        elif block['type'] in [BlockType.ImageBody, BlockType.TableBody]:
            bbox = block['bbox']
-            block["real_lines"] = copy.deepcopy(block['lines'])
+            block['real_lines'] = copy.deepcopy(block['lines'])
            lines = insert_lines_into_block(bbox, line_height, page_w, page_h)
            block['lines'] = []
            for line in lines:
@@ -462,18 +622,16 @@ def remove_outside_spans(spans, all_bboxes, all_discarded_blocks):


 def parse_page_core(
-    page_doc: PageableData, magic_model, page_id, pdf_bytes_md5, imageWriter, parse_mode
+    page_doc: PageableData, magic_model, page_id, pdf_bytes_md5, imageWriter, parse_mode, lang
 ):
    need_drop = False
    drop_reason = []

    """从magic_model对象中获取后面会用到的区块信息"""
-    # img_blocks = magic_model.get_imgs(page_id)
-    # table_blocks = magic_model.get_tables(page_id)
-
    img_groups = magic_model.get_imgs_v2(page_id)
    table_groups = magic_model.get_tables_v2(page_id)

+    """对image和table的区块分组"""
    img_body_blocks, img_caption_blocks, img_footnote_blocks = process_groups(
        img_groups, 'image_body', 'image_caption_list', 'image_footnote_list'
    )
@@ -517,38 +675,20 @@ def parse_page_core(
            page_h,
        )

+    """获取所有的spans信息"""
    spans = magic_model.get_all_spans(page_id)

-    """根据parse_mode，构造spans"""
-    if parse_mode == SupportedPdfParseMethod.TXT:
-        """ocr 中文本类的 span 用 pymu spans 替换！"""
-        pymu_spans = txt_spans_extract(page_doc, inline_equations, interline_equations)
-        spans = replace_text_span(pymu_spans, spans)
-    elif parse_mode == SupportedPdfParseMethod.OCR:
-        pass
-    else:
-        raise Exception('parse_mode must be txt or ocr')
-
    """在删除重复span之前，应该通过image_body和table_body的block过滤一下image和table的span"""
    """顺便删除大水印并保留abandon的span"""
    spans = remove_outside_spans(spans, all_bboxes, all_discarded_blocks)

-    """删除重叠spans中置信度较低的那些"""
-    spans, dropped_spans_by_confidence = remove_overlaps_low_confidence_spans(spans)
-    """删除重叠spans中较小的那些"""
-    spans, dropped_spans_by_span_overlap = remove_overlaps_min_spans(spans)
-    """对image和table截图"""
-    spans = ocr_cut_image_and_table(
-        spans, page_doc, page_id, pdf_bytes_md5, imageWriter
-    )
-
    """先处理不需要排版的discarded_blocks"""
    discarded_block_with_spans, spans = fill_spans_in_blocks(
        all_discarded_blocks, spans, 0.4
    )
    fix_discarded_blocks = fix_discarded_block(discarded_block_with_spans)

-    """如果当前页面没有bbox则跳过"""
+    """如果当前页面没有有效的bbox则跳过"""
    if len(all_bboxes) == 0:
        logger.warning(f'skip this page, not found useful bbox, page_id: {page_id}')
        return ocr_construct_page_component_v2(
@@ -566,7 +706,32 @@ def parse_page_core(
            drop_reason,
        )

-    """将span填入blocks中"""
+    """删除重叠spans中置信度较低的那些"""
+    spans, dropped_spans_by_confidence = remove_overlaps_low_confidence_spans(spans)
+    """删除重叠spans中较小的那些"""
+    spans, dropped_spans_by_span_overlap = remove_overlaps_min_spans(spans)
+
+    """根据parse_mode，构造spans，主要是文本类的字符填充"""
+    if parse_mode == SupportedPdfParseMethod.TXT:
+
+        """之前的公式替换方案"""
+        # pymu_spans = txt_spans_extract_v1(page_doc, inline_equations, interline_equations)
+        # spans = replace_text_span(pymu_spans, spans)
+
+        """ocr 中文本类的 span 用 pymu spans 替换！"""
+        spans = txt_spans_extract_v2(page_doc, spans, all_bboxes, all_discarded_blocks, lang)
+
+    elif parse_mode == SupportedPdfParseMethod.OCR:
+        pass
+    else:
+        raise Exception('parse_mode must be txt or ocr')
+
+    """对image和table截图"""
+    spans = ocr_cut_image_and_table(
+        spans, page_doc, page_id, pdf_bytes_md5, imageWriter
+    )
+
+    """span填充进block"""
    block_with_spans, spans = fill_spans_in_blocks(all_bboxes, spans, 0.5)

    """对block进行fix操作"""
@@ -616,6 +781,7 @@ def pdf_parse_union(
    start_page_id=0,
    end_page_id=None,
    debug_mode=False,
+    lang=None,
 ):
    pdf_bytes_md5 = compute_md5(dataset.data_bits())

@@ -652,7 +818,7 @@ def pdf_parse_union(
        """解析pdf中的每一页"""
        if start_page_id <= page_id <= end_page_id:
            page_info = parse_page_core(
-                page, magic_model, page_id, pdf_bytes_md5, imageWriter, parse_mode
+                page, magic_model, page_id, pdf_bytes_md5, imageWriter, parse_mode, lang
            )
        else:
            page_info = page.get_page_info()
@@ -664,7 +830,7 @@ def pdf_parse_union(
        pdf_info_dict[f'page_{page_id}'] = page_info

    """分段"""
-    para_split(pdf_info_dict, debug_mode=debug_mode)
+    para_split(pdf_info_dict)

    """dict转list"""
    pdf_info_list = dict_to_list(pdf_info_dict)

--- a/magic_pdf/pipe/AbsPipe.py
+++ b/magic_pdf/pipe/AbsPipe.py
--- a/magic_pdf/pipe/OCRPipe.py
+++ b/magic_pdf/pipe/OCRPipe.py
 from loguru import logger

-from magic_pdf.libs.MakeContentConfig import DropMode, MakeMode
+from magic_pdf.config.make_content_config import DropMode, MakeMode
+from magic_pdf.data.data_reader_writer import DataWriter
 from magic_pdf.model.doc_analyze_by_custom_model import doc_analyze
-from magic_pdf.rw.AbsReaderWriter import AbsReaderWriter
 from magic_pdf.pipe.AbsPipe import AbsPipe
 from magic_pdf.user_api import parse_ocr_pdf


 class OCRPipe(AbsPipe):

-    def __init__(self, pdf_bytes: bytes, model_list: list, image_writer: AbsReaderWriter, is_debug: bool = False,
+    def __init__(self, pdf_bytes: bytes, model_list: list, image_writer: DataWriter, is_debug: bool = False,
                 start_page_id=0, end_page_id=None, lang=None,
                 layout_model=None, formula_enable=None, table_enable=None):
        super().__init__(pdf_bytes, model_list, image_writer, is_debug, start_page_id, end_page_id, lang,
@@ -32,10 +32,10 @@ class OCRPipe(AbsPipe):

    def pipe_mk_uni_format(self, img_parent_path: str, drop_mode=DropMode.WHOLE_PDF):
        result = super().pipe_mk_uni_format(img_parent_path, drop_mode)
-        logger.info("ocr_pipe mk content list finished")
+        logger.info('ocr_pipe mk content list finished')
        return result

    def pipe_mk_markdown(self, img_parent_path: str, drop_mode=DropMode.WHOLE_PDF, md_make_mode=MakeMode.MM_MD):
        result = super().pipe_mk_markdown(img_parent_path, drop_mode, md_make_mode)
-        logger.info(f"ocr_pipe mk {md_make_mode} finished")
+        logger.info(f'ocr_pipe mk {md_make_mode} finished')
        return result