目录
一、文件操作基础概念 文件操作是编程中最常见的 I/O 操作之一。Python 通过内置的 open() 函数提供了统一、优雅的文件处理接口。
核心三步骤 打开文件 - 建立程序与文件的连接
读写操作 - 进行数据交换
关闭文件 - 释放系统资源
二、打开文件的正确姿势 2.1 基本用法
# 语法:open(file, mode='r', buffering=-1, encoding=None, ...)
file = open('example.txt', 'r', encoding='utf-8')
content = file.read()
file.close()
2.2 推荐使用上下文管理器(with语句) 这是 Python 最优雅的特性之一,自动管理资源释放:
# 离开 with 代码块时,文件会自动关闭
with open('example.txt', 'r', encoding='utf-8') as f:
content = f.read()
# 处理内容...
# 此处文件已自动关闭
2.3 文件打开模式详解
模式 含义 文件不存在 文件存在 指针位置 'r' 只读 报错 打开 开头 'w' 只写 创建 清空 开头 'a' 追加 创建 打开 末尾 'x' 独占创建 创建 报错 开头 'r+' 读写 报错 打开 开头 'w+' 写读 创建 清空 开头 'a+' 追加读 创建 打开 末尾 二进制模式:在模式后加 'b',如 'rb'、'wb'
# 文本模式 vs 二进制模式
with open('data.txt', 'r', encoding='utf-8') as f: # 返回字符串
text = f.read()
with open('image.jpg', 'rb') as f: # 返回字节串
binary_data = f.read()
三、读取文件的方法对比
# 准备测试文件
with open('sample.txt', 'w', encoding='utf-8') as f:
f.write("第一行内容\n第二行内容\n第三行内容\n")
3.1 四种读取方式
# 1. read() - 一次性读取全部内容
with open('sample.txt', 'r', encoding='utf-8') as f:
all_content = f.read() # 返回字符串
print("read()结果:", repr(all_content))
# 2. read(size) - 按字符数读取(文本模式)或字节数(二进制模式)
with open('sample.txt', 'r', encoding='utf-8') as f:
chunk = f.read(5) # 读取5个字符
print(f"前5个字符:{chunk}")
# 3. readline() - 逐行读取(包括换行符)
with open('sample.txt', 'r', encoding='utf-8') as f:
line1 = f.readline()
line2 = f.readline()
print(f"第一行:{repr(line1)}")
print(f"第二行:{repr(line2)}")
# 4. readlines() - 返回所有行的列表
with open('sample.txt', 'r', encoding='utf-8') as f:
lines = f.readlines()
print(f"所有行:{lines}")
3.2 大文件处理的最佳实践
# 错误示范:一次性读取大文件可能导致内存溢出
# with open('huge_file.txt', 'r') as f:
# data = f.read() # 危险!
# 正确做法:逐行迭代(内存友好)
with open('huge_file.txt', 'r', encoding='utf-8') as f:
for line in f: # 文件对象本身是可迭代的
process_line(line.strip())
# 高级技巧:使用生成器按块读取
def read_in_chunks(file_object, chunk_size=8192):
"""分块读取文件的生成器"""
while True:
data = file_object.read(chunk_size)
if not data:
break
yield data
with open('large_file.dat', 'rb') as f:
for chunk in read_in_chunks(f):
process_chunk(chunk)
四、写入文件的方法
# 1. write() - 写入字符串
with open('output.txt', 'w', encoding='utf-8') as f:
f.write("Hello, ")
f.write("World!\n")
# 2. writelines() - 写入字符串序列
lines = ["第一行\n", "第二行\n", "第三行\n"]
with open('output.txt', 'w', encoding='utf-8') as f:
f.writelines(lines) # 注意:不会自动添加换行符
# 3. print() 写入文件
with open('output.txt', 'w', encoding='utf-8') as f:
print("自动添加换行", file=f)
print("还可以", "用多个", "参数", sep="-", file=f)
五、文件指针操作
# 创建测试文件
with open('pointer_test.txt', 'w', encoding='utf-8') as f:
f.write("0123456789ABCDEF")
with open('pointer_test.txt', 'r+', encoding='utf-8') as f:
# tell() - 获取当前指针位置
print(f"初始位置:{f.tell()}") # 0
f.read(5)
print(f"读取5个字符后:{f.tell()}") # 5
# seek(offset, whence) - 移动指针
# whence: 0-开头 1-当前位置 2-末尾
f.seek(0) # 回到开头
print(f"seek(0)后:{f.tell()}")
f.seek(3, 0) # 从开头偏移3
print(f"seek(3,0)后:{f.tell()}")
f.seek(0, 2) # 移动到末尾
print(f"末尾位置:{f.tell()}")
# 注意:文本模式下只能相对于开头移动(whence=0)
# 二进制模式支持所有 whence 值
六、高级特性与技巧 6.1 多种编码处理
# 处理不同编码的文件
encodings_to_try = ['utf-8', 'gbk', 'gb2312', 'latin-1']
def read_file_safely(filename):
"""安全读取文件,自动尝试多种编码"""
for encoding in encodings_to_try:
try:
with open(filename, 'r', encoding=encoding) as f:
return f.read()
except UnicodeDecodeError:
continue
# 最后手段:忽略无法解码的字符
with open(filename, 'r', encoding='utf-8', errors='ignore') as f:
return f.read()
# errors 参数的其他选项:
# 'strict' - 抛出异常(默认)
# 'ignore' - 忽略错误字符
# 'replace' - 用 � 替换错误字符
# 'backslashreplace' - 用 \x 转义序列替换
6.2 文件锁定(防止并发写入冲突)
import fcntl # Unix/Linux/Mac
import msvcrt # Windows
import os
def lock_file(file_handle):
"""跨平台文件锁定"""
if os.name == 'nt': # Windows
msvcrt.locking(file_handle.fileno(), msvcrt.LK_LOCK, 1)
else: # Unix/Linux/Mac
fcntl.flock(file_handle.fileno(), fcntl.LOCK_EX)
def unlock_file(file_handle):
"""跨平台文件解锁"""
if os.name == 'nt':
msvcrt.locking(file_handle.fileno(), msvcrt.LK_UNLCK, 1)
else:
fcntl.flock(file_handle.fileno(), fcntl.LOCK_UN)
# 使用示例
with open('shared_file.txt', 'a', encoding='utf-8') as f:
lock_file(f)
f.write("安全写入的数据\n")
unlock_file(f)
6.3 临时文件处理
import tempfile
# 1. 创建临时文件(自动删除)
with tempfile.TemporaryFile(mode='w+', encoding='utf-8') as tmp:
tmp.write("临时数据\n")
tmp.seek(0)
print(tmp.read())
# 文件已自动删除
# 2. 创建带名称的临时文件
with tempfile.NamedTemporaryFile(mode='w+', suffix='.txt',
delete=False) as tmp:
tmp.write("持久化临时数据\n")
tmp_name = tmp.name
print(f"临时文件路径:{tmp_name}")
# 手动删除
import os
os.unlink(tmp_name)
# 3. 临时目录
with tempfile.TemporaryDirectory() as tmp_dir:
file_path = os.path.join(tmp_dir, 'test.txt')
with open(file_path, 'w') as f:
f.write("临时目录中的文件")
# 目录和文件都已自动删除
6.4 内存映射文件(处理超大文件)
import mmap
# 高效处理超大文件,不加载全部内容到内存
def search_in_huge_file(filename, pattern):
"""在超大文件中搜索字符串"""
with open(filename, 'r+b') as f:
# 创建内存映射
with mmap.mmap(f.fileno(), 0) as mmapped_file:
# 像操作字节串一样搜索
if pattern.encode() in mmapped_file:
return True
return False
# 修改超大文件的特定部分
def modify_large_file(filename, offset, new_data):
"""修改文件特定位置的内容"""
with open(filename, 'r+b') as f:
with mmap.mmap(f.fileno(), 0) as mmapped_file:
mmapped_file[offset:offset+len(new_data)] = new_data
七、pathlib - 现代化的文件路径操作 Python 3.4+ 推荐使用 pathlib 代替 os.path:
from pathlib import Path
# 创建路径对象
p = Path('/home/user/documents/report.txt')
# 路径信息
print(f"文件名:{p.name}")
print(f"后缀:{p.suffix}")
print(f"父目录:{p.parent}")
print(f"不带后缀:{p.stem}")
# 路径操作
new_path = p.with_name('new_report.txt')
new_path = p.with_suffix('.md')
# 文件操作(无需 open)
content = p.read_text(encoding='utf-8')
p.write_text("新的内容", encoding='utf-8')
data = p.read_bytes()
p.write_bytes(b'binary data')
# 目录遍历
for file in Path('.').glob('*.py'):
print(f"找到 Python 文件:{file}")
# 递归搜索
for py_file in Path('.').rglob('*.py'):
print(f"递归找到:{py_file}")
八、实战案例:日志文件轮转
import os
from datetime import datetime
from pathlib import Path
class RotatingFileWriter:
"""简单的日志轮转实现"""
def __init__(self, base_filename, max_size_mb=10, backup_count=5):
self.base_path = Path(base_filename)
self.max_size = max_size_mb * 1024 * 1024
self.backup_count = backup_count
def write(self, message):
"""写入消息,自动检查是否需要轮转"""
# 确保目录存在
self.base_path.parent.mkdir(parents=True, exist_ok=True)
# 检查当前文件大小
if self.base_path.exists() and self.base_path.stat().st_size >= self.max_size:
self._rotate()
# 写入消息
with open(self.base_path, 'a', encoding='utf-8') as f:
timestamp = datetime.now().isoformat()
f.write(f"[{timestamp}] {message}\n")
def _rotate(self):
"""执行文件轮转"""
for i in range(self.backup_count - 1, 0, -1):
old_file = self.base_path.with_suffix(f'.{i}.log')
new_file = self.base_path.with_suffix(f'.{i+1}.log')
if old_file.exists():
if new_file.exists():
new_file.unlink()
old_file.rename(new_file)
# 重命名当前文件
if self.base_path.exists():
backup = self.base_path.with_suffix('.1.log')
if backup.exists():
backup.unlink()
self.base_path.rename(backup)
# 使用示例
logger = RotatingFileWriter('app.log', max_size_mb=1, backup_count=3)
for i in range(10000):
logger.write(f"这是第 {i} 条日志消息")
九、常见陷阱与最佳实践 9.1 必须注意的问题
# 1. 总是指定编码
# 错误:跨平台可能出现乱码
# with open('file.txt', 'w') as f:
# f.write("中文")
# 正确
with open('file.txt', 'w', encoding='utf-8') as f:
f.write("中文")
# 2. 注意换行符转换
# Windows下 '\n' 会自动转换为 '\r\n'
# 如需保持原样,使用二进制模式或指定 newline
with open('file.txt', 'w', newline='\n', encoding='utf-8') as f:
f.write("Line1\nLine2\n")
# 3. 异常处理
try:
with open('maybe_not_exist.txt', 'r', encoding='utf-8') as f:
content = f.read()
except FileNotFoundError:
print("文件不存在")
except PermissionError:
print("没有权限读取")
except UnicodeDecodeError:
print("编码错误")
9.2 性能优化建议
# 1. 使用缓冲(默认已启用)
with open('file.txt', 'w', buffering=8192, encoding='utf-8') as f:
for i in range(1000000):
f.write(f"Line {i}\n")
# 2. 批量写入 vs 单次写入
# 慢(多次系统调用)
with open('test.txt', 'w', encoding='utf-8') as f:
for item in data:
f.write(item + '\n')
# 快(一次系统调用)
with open('test.txt', 'w', encoding='utf-8') as f:
f.write('\n'.join(data))
# 3. 使用生成器表达式而非列表推导
# 内存占用大
lines = [line.upper() for line in open('input.txt', encoding='utf-8')]
with open('output.txt', 'w', encoding='utf-8') as f:
f.writelines(lines)
# 内存友好
with open('output.txt', 'w', encoding='utf-8') as outf:
with open('input.txt', 'r', encoding='utf-8') as inf:
outf.writelines(line.upper() for line in inf)
十、总结 场景 推荐方案 小文件读写 pathlib.Path.read_text() / write_text() 逐行处理 for line in file: 迭代 二进制文件 'rb' / 'wb' 模式 超大文件 内存映射 mmap 或分块读取 临时文件 tempfile 模块 并发写入 文件锁 fcntl / msvcrt 路径操作 pathlib.Path 代替 os.path Python 的文件操作体系设计精妙,从简单的文本读写到复杂的内存映射都提供了优雅的接口。掌握这些技术,能让你的代码既高效又健壮。记住最重要的原则:永远使用 with 语句,永远指定编码。
本文由 尚先生 原创,转载请注明出处。
评论
0