DiskReaderWriter.py 2.17 KB
Newer Older
liukaiwen's avatar
liukaiwen committed
1
2
import os
from magic_pdf.io.AbsReaderWriter import AbsReaderWriter
liukaiwen's avatar
liukaiwen committed
3
from loguru import logger
liukaiwen's avatar
liukaiwen committed
4
5
6
7


MODE_TXT = "text"
MODE_BIN = "binary"
许瑞's avatar
许瑞 committed
8
9


liukaiwen's avatar
liukaiwen committed
10
class DiskReaderWriter(AbsReaderWriter):
liukaiwen's avatar
liukaiwen committed
11

许瑞's avatar
许瑞 committed
12
    def __init__(self, parent_path, encoding="utf-8"):
liukaiwen's avatar
liukaiwen committed
13
14
15
        self.path = parent_path
        self.encoding = encoding

liukaiwen's avatar
liukaiwen committed
16
17
18
19
20
21
22
23
24
    def read(self, path, mode=MODE_TXT):
        if os.path.isabs(path):
            abspath = path
        else:
            abspath = os.path.join(self.path, path)
        if not os.path.exists(abspath):
            logger.error(f"文件 {abspath} 不存在")
            raise Exception(f"文件 {abspath} 不存在")
        if mode == MODE_TXT:
许瑞's avatar
许瑞 committed
25
            with open(abspath, "r", encoding=self.encoding) as f:
liukaiwen's avatar
liukaiwen committed
26
                return f.read()
liukaiwen's avatar
liukaiwen committed
27
        elif mode == MODE_BIN:
许瑞's avatar
许瑞 committed
28
            with open(abspath, "rb") as f:
liukaiwen's avatar
liukaiwen committed
29
30
31
32
                return f.read()
        else:
            raise ValueError("Invalid mode. Use 'text' or 'binary'.")

liukaiwen's avatar
liukaiwen committed
33
34
35
36
37
    def write(self, content, path, mode=MODE_TXT):
        if os.path.isabs(path):
            abspath = path
        else:
            abspath = os.path.join(self.path, path)
liukaiwen's avatar
liukaiwen committed
38
39
        directory_path = os.path.dirname(abspath)
        if not os.path.exists(directory_path):
liukaiwen's avatar
liukaiwen committed
40
            os.makedirs(directory_path)
liukaiwen's avatar
liukaiwen committed
41
        if mode == MODE_TXT:
许瑞's avatar
许瑞 committed
42
            with open(abspath, "w", encoding=self.encoding) as f:
liukaiwen's avatar
liukaiwen committed
43
44
                f.write(content)
                logger.info(f"内容已成功写入 {abspath}")
liukaiwen's avatar
liukaiwen committed
45

liukaiwen's avatar
liukaiwen committed
46
        elif mode == MODE_BIN:
许瑞's avatar
许瑞 committed
47
            with open(abspath, "wb") as f:
liukaiwen's avatar
liukaiwen committed
48
49
                f.write(content)
                logger.info(f"内容已成功写入 {abspath}")
liukaiwen's avatar
liukaiwen committed
50
51
52
        else:
            raise ValueError("Invalid mode. Use 'text' or 'binary'.")

许瑞's avatar
许瑞 committed
53
    def read_jsonl(self, path: str, byte_start=0, byte_end=None, encoding="utf-8"):
liukaiwen's avatar
liukaiwen committed
54
        return self.read(path)
liukaiwen's avatar
liukaiwen committed
55

许瑞's avatar
许瑞 committed
56

liukaiwen's avatar
liukaiwen committed
57
58
# 使用示例
if __name__ == "__main__":
liukaiwen's avatar
liukaiwen committed
59
    file_path = "io/test/example.txt"
liukaiwen's avatar
liukaiwen committed
60
    drw = DiskReaderWriter("D:\projects\papayfork\Magic-PDF\magic_pdf")
liukaiwen's avatar
liukaiwen committed
61
62

    # 写入内容到文件
liukaiwen's avatar
liukaiwen committed
63
    drw.write(b"Hello, World!", path="io/test/example.txt", mode="binary")
liukaiwen's avatar
liukaiwen committed
64
65

    # 从文件读取内容
liukaiwen's avatar
liukaiwen committed
66
    content = drw.read(path=file_path)
liukaiwen's avatar
liukaiwen committed
67
    if content:
liukaiwen's avatar
liukaiwen committed
68
        logger.info(f"从 {file_path} 读取的内容: {content}")