Python处理文档

txt文件

读取

python 复制代码
# 方法1:read() 一次性读取整个文件
def read_txt_basic(filepath) -> str:
    with open(filepath, 'r', encoding='utf-8') as file:
        content = file.read()
    return content


# 方法2:readlines() 按行读取
def read_txt_line(filepath) -> list[str]:
    with open(filepath, 'r', encoding='utf-8') as file:
        content = file.readlines()
    return content

# 方法3:逐行读取(适合大文件),使用迭代器方法
def read_txt_line_by_line(filepath):
    with open(filepath, 'r', encoding='utf-8') as file:
        for line in file:
            noblank = line.replace('\n', '').replace('\t','').strip()
            if noblank != '':
                yield noblank

# 方法4:使用exception做好保证
def read_txt_file_exception(filepath)->str:
    try:
        with open(filepath, 'r', encoding='utf-8') as file:
            return file.read()
    except FileNotFoundError:
        print(f'文件{filepath}不存在')
        return ''
    except UnicodeDecodeError:
        with open(filepath, 'r', encoding='gbk',errors='ignore') as file:
            return file.read()

# 高级:自动检测编码
import chardet
def read_txt_encoding_detection(filepath)->str:
    with open(filepath, 'rb') as file:
        raw_data = file.read()
        encoding = chardet.detect(raw_data)['encoding']

    with open(filepath, 'r', encoding=encoding) as file:
        return file.read()

写入

python 复制代码
# 方法1:write() 写入
def write_txt_basic(content, filepath):
    with open(filepath, 'w', encoding='utf-8') as file:
        file.write(content)


# 方法2:writelines() 写入多行
def write_txt_lines(lines, filepath):
    with open(filepath, 'w', encoding='utf-8') as file:
        # 确保每行都有换行符
        new_lines = [line + '\n' if not line.endswith('\n') else line for line in lines]
        file.writelines(new_lines)


# 多行读入,使用迭代器
def read_txt_lines(filepath):
    with open(filepath, 'r', encoding='utf-8') as file:
        # 避开文件名输出
        print(file.readline())
        for line in file:
            yield line

# 方法3:追加写入
def write_txt_append(content,filepath):
    with open(filepath, 'a', encoding='utf-8') as file:
        file.write(content + '\n')

# 方法4:高性能批量写入
def write_txt_efficient(datalist,filepath,batch_size=1000):
    with open(filepath,'w',encoding='utf-8') as file:
        buffer=[]
        for i,item in enumerate(datalist,1):
            cleanstr = item.replace('\n', '').replace('\t', '').strip()
            if not cleanstr:
                continue
            buffer.append(str(cleanstr) + '\n')
            if i % batch_size == 0:
                file.writelines(buffer)
                buffer=[]
        if buffer:
            file.writelines(buffer)
相关推荐
weixin_460443562 小时前
2026年新修订《网络安全法》实施:Java私有化考试系统如何做好SBOM、漏洞治理与安全升级?
java·开发语言·在线考试系统·企业培训考试系统·宏远培训考试系统·国企培训考试系统·国产化培训系统
子兮曰2 小时前
把 649 种文件格式塞进一个网页里:一个「文件不出浏览器」的预览器
前端·javascript·后端
kobe_OKOK_4 小时前
__getattr__和__getattribute__如何用
python
全栈弄潮儿5 小时前
Python实战第2期: 变量与数据类型
后端·python·agent
长沙三为智能科技5 小时前
家政小程序门店运营模块设计:会员、下单、派单、结算四段链路的落地拆解
python·小程序·内容运营
Cc.Y5 小时前
Java零基础入门:可变字符串与包装类:StringBuilder、StringBuffer 与 通讯录管理系统实战
java·开发语言·python
启观川5 小时前
数据结构与算法 -第 3 章 常用算法-动态规划
数据结构·笔记·python·算法
码爸5 小时前
排序算法介绍
python·算法·排序算法
sm_926787055 小时前
RFID 标签打印的技术实现要点与二次开发实践
java·大数据·前端·c++·编辑器
阿森_dev5 小时前
2027 届毕设选题怎么选:30 个 Java 方向选题的实现难度排序
java·开发语言·spring boot·毕业设计·课程设计