预计阅读时间:37 分钟
Python 正则表达式完全指南
一、正则表达式的本质:模式匹配的艺术
正则表达式(Regular Expression)是一种描述字符串模式的语言。它就像一个精密的过滤器,能从海量文本中准确提取你需要的信息。掌握正则表达式,你就拥有了处理文本的超能力。
1.1 为什么需要正则表达式?
# 没有正则表达式的生活:繁琐且易错
def validate_email_manual(email):
"""手动验证邮箱(痛苦的方式)"""
if '@' not in email:
return False
parts = email.split('@')
if len(parts) != 2:
return False
local, domain = parts
if not local or not domain:
return False
if '.' not in domain:
return False
# 还有很多规则要检查...
return True
# 使用正则表达式:简洁且强大
import re
def validate_email_regex(email):
"""使用正则验证邮箱(优雅的方式)"""
pattern = r'^[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}$'
return bool(re.match(pattern, email))
# 测试
test_emails = [
'user@example.com',
'invalid.email',
'user@domain',
'user.name+tag@sub.domain.co.uk'
]
for email in test_emails:
print(f"{email:30s} -> {'✓' if validate_email_regex(email) else '✗'}")
1.2 Python 的 re 模块基础
import re
# re 模块的主要函数
text = "Python is awesome! Python is powerful!"
# 1. re.search() - 搜索第一个匹配
match = re.search(r'Python', text)
if match:
print(f"search: 找到 '{match.group()}' 在位置 {match.start()}-{match.end()}")
# 2. re.match() - 从开头匹配
match = re.match(r'Python', text)
print(f"match: {'成功' if match else '失败'}(从开头匹配)")
match = re.match(r'awesome', text)
print(f"match: {'成功' if match else '失败'}(不在开头)")
# 3. re.findall() - 查找所有匹配
matches = re.findall(r'Python', text)
print(f"findall: 找到 {len(matches)} 个匹配: {matches}")
# 4. re.finditer() - 返回迭代器
for match in re.finditer(r'Python', text):
print(f"finditer: 在位置 {match.start()} 找到 '{match.group()}'")
# 5. re.sub() - 替换
new_text = re.sub(r'Python', 'JavaScript', text)
print(f"sub: {new_text}")
# 6. re.split() - 分割
parts = re.split(r'\s+', text)
print(f"split: {parts}")
二、正则表达式基础语法
2.1 元字符
# 元字符是正则表达式的基石
"""
. - 匹配任意单个字符(除换行符)
^ - 匹配字符串开头
$ - 匹配字符串结尾
* - 匹配0次或多次
+ - 匹配1次或多次
? - 匹配0次或1次
{m} - 匹配精确m次
{m,n}- 匹配m到n次
[] - 字符类,匹配其中的任意字符
| - 或运算符
() - 分组
\ - 转义字符
"""
# 示例:演示每个元字符
test_strings = [
"cat",
"caaat",
"ct",
"dog",
"cBt"
]
print("元字符示例:")
for s in test_strings:
# . 匹配任意字符
if re.match(r'c.t', s):
print(f" c.t 匹配: {s}")
# * 匹配0次或多次
if re.match(r'ca*t', s):
print(f" ca*t 匹配: {s}")
# + 匹配1次或多次
if re.match(r'ca+t', s):
print(f" ca+t 匹配: {s}")
# 字符类 []
print("\n字符类示例:")
text = "The cat sat on the mat"
# 匹配 cat, sat, mat
words = re.findall(r'[csm]at', text)
print(f" [csm]at 匹配: {words}")
# 范围
print(f" [0-9] 匹配数字: {re.findall(r'[0-9]', 'a1b2c3')}")
print(f" [a-z] 匹配小写: {re.findall(r'[a-z]', 'AbC123')}")
print(f" [A-Z] 匹配大写: {re.findall(r'[A-Z]', 'AbC123')}")
print(f" [^0-9] 匹配非数字: {re.findall(r'[^0-9]', 'a1b2c3')}")
# 量词
print("\n量词示例:")
print(f" a{{3}} 精确3个a: {re.findall(r'a{3}', 'aaaaa')}")
print(f" a{{2,4}} 2-4个a: {re.findall(r'a{2,4}', 'aaaaa')}")
print(f" a{{2,}} 至少2个a: {re.findall(r'a{2,}', 'aaaaa')}")
2.2 特殊字符类
# 预定义的字符类
"""
\d - 数字 [0-9]
\D - 非数字 [^0-9]
\w - 单词字符 [a-zA-Z0-9_]
\W - 非单词字符 [^a-zA-Z0-9_]
\s - 空白字符 [ \t\n\r\f\v]
\S - 非空白字符 [^ \t\n\r\f\v]
\b - 单词边界
\B - 非单词边界
"""
text = "User123@email.com has 2 cats and 3 dogs!"
print("预定义字符类:")
print(f" 数字: {re.findall(r'\d+', text)}")
print(f" 单词: {re.findall(r'\w+', text)}")
print(f" 非单词字符: {re.findall(r'\W+', text)}")
# 单词边界
print("\n单词边界示例:")
text = "cat category educate"
print(f" 'cat' 单词: {re.findall(r'\bcat\b', text)}")
print(f" 包含'cat': {re.findall(r'cat', text)}")
print(f" 以'cat'开头: {re.findall(r'\bcat', text)}")
print(f" 以'cat'结尾: {re.findall(r'cat\b', text)}")
2.3 分组与捕获
# 分组是正则表达式最强大的特性之一
text = "John Doe, 30 years old, email: john@example.com"
text2 = "Jane Smith, 25 years old, email: jane.smith@company.co.uk"
# 1. 普通分组 ()
pattern = r'(\w+)\s+(\w+),\s+(\d+)\s+years old,\s+email:\s+(\S+@\S+)'
for t in [text, text2]:
match = re.search(pattern, t)
if match:
first_name = match.group(1)
last_name = match.group(2)
age = match.group(3)
email = match.group(4)
print(f"姓名: {first_name} {last_name}, 年龄: {age}, 邮箱: {email}")
print(f" 所有分组: {match.groups()}")
# 2. 命名分组 (?P<name>...)
pattern = r'(?P<first>\w+)\s+(?P<last>\w+),\s+(?P<age>\d+)\s+years old,\s+email:\s+(?P<email>\S+@\S+)'
match = re.search(pattern, text)
if match:
print(f"\n命名分组:")
print(f" 姓名: {match.group('first')} {match.group('last')}")
print(f" 年龄: {match.group('age')}")
print(f" 邮箱: {match.group('email')}")
print(f" 分组字典: {match.groupdict()}")
# 3. 非捕获分组 (?:...)
# 用于分组但不捕获,提高性能
text = "https://www.example.com/path/to/page"
pattern = r'(?:https?://)?(?:www\.)?([^/]+)'
match = re.search(pattern, text)
if match:
print(f"\n域名(非捕获分组): {match.group(1)}")
# 4. 反向引用
# 查找重复单词
text = "The the cat sat on on the mat mat"
duplicates = re.findall(r'\b(\w+)\s+\1\b', text, re.IGNORECASE)
print(f"\n重复单词: {duplicates}")
# 查找成对的HTML标签
html = "<div>Content</div> <p>Paragraph</p> <div>Mismatch</p>"
pattern = r'<(\w+)>.*?</\1>'
tags = re.findall(pattern, html)
print(f"成对标签: {tags}")
2.4 断言(零宽断言)
# 断言匹配位置,不消耗字符
"""
(?=...) - 正向前瞻:后面跟着...
(?!...) - 负向前瞻:后面不跟着...
(?<=...) - 正向后顾:前面是...
(?<!...) - 负向后顾:前面不是...
"""
text = "apple price: $5.99, orange price: $3.49, banana price: $2.99"
# 1. 正向前瞻:只匹配后面跟着数字的价格
prices_with_dollar = re.findall(r'\$\d+\.\d{2}', text)
print(f"所有价格: {prices_with_dollar}")
# 只提取数字部分(不包含$)
prices_only = re.findall(r'(?<=\$)\d+\.\d{2}', text)
print(f"价格数字(后顾): {prices_only}")
# 2. 负向前瞻:匹配不以特定模式结尾的单词
words = "color colour colur"
# 匹配 colo 后面不是 r 的
pattern = r'colo(?!u)r'
print(f"负向前瞻: {re.findall(pattern, words)}")
# 3. 复杂示例:密码验证
def validate_password(password):
"""
密码要求:
- 至少8个字符
- 包含至少一个大写字母
- 包含至少一个小写字母
- 包含至少一个数字
- 包含至少一个特殊字符
"""
# 使用正向前瞻同时检查多个条件
pattern = r'^(?=.*[A-Z])(?=.*[a-z])(?=.*\d)(?=.*[!@#$%^&*])[A-Za-z\d!@#$%^&*]{8,}$'
return bool(re.match(pattern, password))
test_passwords = [
"Weak",
"weakpassword",
"WEAKPASSWORD",
"NoNumbers!",
"StrongP@ss123",
"Short1!"
]
print("\n密码验证:")
for pwd in test_passwords:
print(f" {pwd:15s} -> {'✓' if validate_password(pwd) else '✗'}")
# 4. 更复杂的断言示例
text = "The quick brown fox jumps over the lazy dog"
# 找出所有后面跟着空格和"fox"的单词
words_before_fox = re.findall(r'\b\w+\b(?=\s+fox)', text)
print(f"\n'fox'前面的单词: {words_before_fox}")
# 找出所有前面是"the"的单词(不区分大小写)
words_after_the = re.findall(r'(?<=\b[Tt]he\s+)\w+', text)
print(f"'the'后面的单词: {words_after_the}")
三、re 模块高级功能
3.1 编译正则表达式
# 编译正则表达式可以提高性能(多次使用时)
import timeit
# 未编译
def uncompiled_search(text):
return re.findall(r'\b\w{4,}\b', text)
# 编译
pattern = re.compile(r'\b\w{4,}\b')
def compiled_search(text):
return pattern.findall(text)
text = "Python is an awesome programming language" * 1000
# 性能对比
uncompiled_time = timeit.timeit(lambda: uncompiled_search(text), number=100)
compiled_time = timeit.timeit(lambda: compiled_search(text), number=100)
print(f"未编译: {uncompiled_time:.4f}秒")
print(f"已编译: {compiled_time:.4f}秒")
print(f"编译后快 {uncompiled_time/compiled_time:.2f}倍")
# 编译时指定标志
# re.IGNORECASE 或 re.I - 忽略大小写
# re.MULTILINE 或 re.M - 多行模式
# re.DOTALL 或 re.S - . 匹配包括换行符
# re.VERBOSE 或 re.X - 详细模式(可添加注释)
pattern = re.compile(r"""
^ # 开头
(?P<username> # 用户名分组
[a-zA-Z0-9._%+-]+
)
@ # @ 符号
(?P<domain> # 域名分组
[a-zA-Z0-9.-]+
\.
[a-zA-Z]{2,}
)
$ # 结尾
""", re.VERBOSE | re.IGNORECASE)
emails = ['user@example.com', 'invalid.email', 'USER@DOMAIN.COM']
for email in emails:
match = pattern.match(email)
if match:
print(f"有效邮箱: {email}")
print(f" 用户名: {match.group('username')}")
print(f" 域名: {match.group('domain')}")
3.2 替换的高级用法
# 1. 使用函数进行替换
def format_currency(match):
"""格式化货币"""
amount = float(match.group(1))
return f"${amount:,.2f}"
text = "Prices: USD 1234.56, EUR 789.12, GBP 345.67"
# 将货币数字格式化为带千位分隔符
formatted = re.sub(r'(\d+\.\d{2})', format_currency, text)
print(f"格式化货币: {formatted}")
# 2. 处理日期格式
def convert_date_format(match):
"""转换日期格式:MM/DD/YYYY -> YYYY-MM-DD"""
month, day, year = match.groups()
return f"{year}-{month.zfill(2)}-{day.zfill(2)}"
text = "Dates: 1/15/2024, 12/25/2023, 6/1/2024"
converted = re.sub(r'(\d{1,2})/(\d{1,2})/(\d{4})', convert_date_format, text)
print(f"日期格式转换: {converted}")
# 3. 使用命名分组进行替换
text = "Name: John Doe, Email: john@example.com"
pattern = r'Name: (?P<name>\w+\s+\w+), Email: (?P<email>\S+@\S+)'
# 使用 \g<name> 引用命名分组
result = re.sub(pattern, r'Contact: \g<name> <\g<email>>', text)
print(f"格式化联系人: {result}")
# 4. 条件替换
def mask_sensitive_data(text):
"""遮蔽敏感数据"""
# 遮蔽邮箱
text = re.sub(
r'(\w{1,3})[\w._%+-]*(\w{1,3}@[\w.-]+\.[a-zA-Z]{2,})',
r'\1***\2',
text
)
# 遮蔽电话号码(保留后4位)
text = re.sub(
r'\b(\d{3})[-.]?(\d{3})[-.]?(\d{4})\b',
r'***-***-\3',
text
)
# 遮蔽信用卡号(只显示后4位)
text = re.sub(
r'\b\d{4}[ -]?\d{4}[ -]?\d{4}[ -]?(\d{4})\b',
r'****-****-****-\1',
text
)
return text
sensitive_text = """
Contact: john.doe@example.com
Phone: 123-456-7890
Card: 4111-1111-1111-1234
"""
masked = mask_sensitive_data(sensitive_text)
print(f"遮蔽后:\n{masked}")
3.3 贪婪与非贪婪匹配
# 贪婪匹配(默认):尽可能多地匹配
# 非贪婪匹配:尽可能少地匹配(使用 ?)
html = "<div>First div</div> <div>Second div</div>"
# 贪婪匹配
greedy = re.findall(r'<div>.*</div>', html)
print(f"贪婪匹配: {greedy}")
# 输出:['<div>First div</div> <div>Second div</div>'] - 匹配了整个字符串
# 非贪婪匹配
non_greedy = re.findall(r'<div>.*?</div>', html)
print(f"非贪婪匹配: {non_greedy}")
# 输出:['<div>First div</div>', '<div>Second div</div>'] - 分别匹配每个div
# 量词的非贪婪版本
"""
*? - 0次或多次,尽可能少
+? - 1次或多次,尽可能少
?? - 0次或1次,尽可能少
{m,n}? - m到n次,尽可能少
"""
# 实际应用:提取HTML标签内容
def extract_tag_content(html, tag):
"""提取指定标签的内容"""
pattern = f'<{tag}.*?>(.*?)</{tag}>'
return re.findall(pattern, html, re.DOTALL)
html_content = """
<html>
<title>My Page</title>
<body>
<h1>Welcome</h1>
<p>This is the first paragraph.</p>
<p>This is the second paragraph.</p>
</body>
</html>
"""
titles = extract_tag_content(html_content, 'title')
paragraphs = extract_tag_content(html_content, 'p')
print(f"标题: {titles}")
print(f"段落: {paragraphs}")
四、实战应用
4.1 日志分析
import re
from collections import Counter
from datetime import datetime
# 示例日志格式
sample_log = """
2024-01-15 10:23:45 INFO User alice logged in from 192.168.1.100
2024-01-15 10:24:12 ERROR Database connection failed: timeout
2024-01-15 10:25:01 INFO User bob logged in from 10.0.0.50
2024-01-15 10:25:45 WARNING High memory usage: 85%
2024-01-15 10:26:30 ERROR API request failed: 500 Internal Server Error
2024-01-15 10:27:15 INFO User alice performed action: delete_file
2024-01-15 10:28:00 ERROR Database connection failed: timeout
2024-01-15 10:28:45 INFO User charlie logged in from 172.16.0.20
"""
class LogAnalyzer:
"""日志分析器"""
# 日志格式正则表达式
LOG_PATTERN = re.compile(r"""
^(\d{4}-\d{2}-\d{2}\s+\d{2}:\d{2}:\d{2})\s+ # 时间戳
(\w+)\s+ # 日志级别
(.*)$ # 消息
""", re.VERBOSE | re.MULTILINE)
# IP地址提取模式
IP_PATTERN = re.compile(r'\b(?:\d{1,3}\.){3}\d{1,3}\b')
# 用户提取模式
USER_PATTERN = re.compile(r'User\s+(\w+)')
def __init__(self, log_text):
self.log_text = log_text
self.entries = self._parse_logs()
def _parse_logs(self):
"""解析日志条目"""
entries = []
for match in self.LOG_PATTERN.finditer(self.log_text):
timestamp_str, level, message = match.groups()
timestamp = datetime.strptime(timestamp_str, '%Y-%m-%d %H:%M:%S')
# 提取IP
ip_match = self.IP_PATTERN.search(message)
ip = ip_match.group() if ip_match else None
# 提取用户
user_match = self.USER_PATTERN.search(message)
user = user_match.group(1) if user_match else None
entries.append({
'timestamp': timestamp,
'level': level,
'message': message.strip(),
'ip': ip,
'user': user
})
return entries
def get_level_stats(self):
"""统计日志级别"""
levels = Counter(entry['level'] for entry in self.entries)
return dict(levels)
def get_user_activity(self):
"""统计用户活动"""
users = Counter(entry['user'] for entry in self.entries if entry['user'])
return dict(users)
def get_ip_stats(self):
"""统计IP地址"""
ips = Counter(entry['ip'] for entry in self.entries if entry['ip'])
return dict(ips)
def get_errors(self):
"""获取所有错误"""
return [entry for entry in self.entries if entry['level'] == 'ERROR']
def find_pattern(self, pattern):
"""搜索特定模式"""
return [
entry for entry in self.entries
if re.search(pattern, entry['message'], re.IGNORECASE)
]
# 使用日志分析器
analyzer = LogAnalyzer(sample_log)
print("=== 日志分析结果 ===")
print(f"总日志条目: {len(analyzer.entries)}")
print(f"级别统计: {analyzer.get_level_stats()}")
print(f"用户活动: {analyzer.get_user_activity()}")
print(f"IP统计: {analyzer.get_ip_stats()}")
print("\n错误日志:")
for error in analyzer.get_errors():
print(f" [{error['timestamp']}] {error['message']}")
print("\n包含'database'的日志:")
for entry in analyzer.find_pattern(r'database'):
print(f" [{entry['level']}] {entry['message']}")
4.2 数据提取与清洗
class DataExtractor:
"""数据提取器"""
# 常见模式
PATTERNS = {
'email': re.compile(r'\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Z|a-z]{2,}\b'),
'url': re.compile(r'https?://(?:[-\w.]|(?:%[\da-fA-F]{2}))+[^\s]*'),
'phone': re.compile(r'\b\d{3}[-.]?\d{3}[-.]?\d{4}\b'),
'date': re.compile(r'\b\d{4}-\d{2}-\d{2}\b|\b\d{2}/\d{2}/\d{4}\b'),
'ip': re.compile(r'\b(?:\d{1,3}\.){3}\d{1,3}\b'),
'hashtag': re.compile(r'#\w+'),
'mention': re.compile(r'@\w+'),
'price': re.compile(r'\$\d+(?:\.\d{2})?'),
'time': re.compile(r'\b\d{2}:\d{2}(?::\d{2})?\b'),
}
@classmethod
def extract(cls, text, pattern_type):
"""提取指定类型的数据"""
if pattern_type not in cls.PATTERNS:
raise ValueError(f"未知模式: {pattern_type}")
pattern = cls.PATTERNS[pattern_type]
return pattern.findall(text)
@classmethod
def extract_all(cls, text):
"""提取所有类型的数据"""
results = {}
for name, pattern in cls.PATTERNS.items():
matches = pattern.findall(text)
if matches:
results[name] = matches
return results
@classmethod
def clean_html(cls, html):
"""清理HTML标签"""
# 移除所有HTML标签
text = re.sub(r'<[^>]+>', '', html)
# 替换HTML实体
entities = {
' ': ' ',
'<': '<',
'>': '>',
'&': '&',
'"': '"',
''': "'"
}
for entity, char in entities.items():
text = text.replace(entity, char)
# 压缩空白字符
text = re.sub(r'\s+', ' ', text)
return text.strip()
@classmethod
def normalize_phone(cls, phone):
"""标准化电话号码"""
# 移除所有非数字字符
digits = re.sub(r'\D', '', phone)
# 格式化为 (XXX) XXX-XXXX
if len(digits) == 10:
return f"({digits[:3]}) {digits[3:6]}-{digits[6:]}"
elif len(digits) == 11 and digits[0] == '1':
return f"+1 ({digits[1:4]}) {digits[4:7]}-{digits[7:]}"
else:
return phone
# 测试数据提取器
sample_text = """
Contact us at support@example.com or visit https://www.example.com/help
Call us at 123-456-7890 or (987) 654-3210
Follow us on Twitter @example and use #ExampleTag
Prices start at $19.99 and go up to $299.00
Server IP: 192.168.1.1
Meeting at 14:30 on 2024-01-15
"""
print("=== 数据提取示例 ===")
for pattern_name in ['email', 'url', 'phone', 'ip', 'hashtag', 'mention', 'price', 'date', 'time']:
matches = DataExtractor.extract(sample_text, pattern_name)
if matches:
print(f"{pattern_name:10s}: {matches}")
# HTML清理
html_sample = """
<div class="content">
<h1>Welcome to our <site></h1>
<p>This is a paragraph with <strong>bold</strong> text.</p>
<p>Another paragraph with multiple spaces.</p>
</div>
"""
cleaned = DataExtractor.clean_html(html_sample)
print(f"\n清理后的HTML:\n{cleaned}")
# 电话号码标准化
phones = ['123-456-7890', '987.654.3210', '1234567890', '11234567890']
print("\n电话号码标准化:")
for phone in phones:
print(f" {phone:15s} -> {DataExtractor.normalize_phone(phone)}")
4.3 代码分析工具
class CodeAnalyzer:
"""Python 代码分析器"""
# Python 代码模式
PATTERNS = {
'import': re.compile(r'^import\s+(\w+)|^from\s+(\w+)\s+import', re.MULTILINE),
'function': re.compile(r'^def\s+(\w+)\s*\([^)]*\):', re.MULTILINE),
'class': re.compile(r'^class\s+(\w+)[:(]', re.MULTILINE),
'variable': re.compile(r'^(\w+)\s*=\s*', re.MULTILINE),
'decorator': re.compile(r'^@(\w+)', re.MULTILINE),
'comment': re.compile(r'#.*$', re.MULTILINE),
'docstring': re.compile(r'""".*?"""|\'\'\'.*?\'\'\'', re.DOTALL),
'todo': re.compile(r'#\s*TODO:?\s*(.*)$', re.MULTILINE | re.IGNORECASE),
'fixme': re.compile(r'#\s*FIXME:?\s*(.*)$', re.MULTILINE | re.IGNORECASE),
}
def __init__(self, code):
self.code = code
self.lines = code.split('\n')
def count_lines(self):
"""统计代码行数"""
total = len(self.lines)
empty = sum(1 for line in self.lines if not line.strip())
comments = len(self.PATTERNS['comment'].findall(self.code))
return {
'total': total,
'empty': empty,
'comments': comments,
'code': total - empty - comments
}
def extract_imports(self):
"""提取所有导入"""
imports = set()
for match in self.PATTERNS['import'].finditer(self.code):
module = match.group(1) or match.group(2)
if module:
imports.add(module)
return sorted(imports)
def extract_functions(self):
"""提取所有函数"""
functions = []
for match in self.PATTERNS['function'].finditer(self.code):
func_name = match.group(1)
# 尝试提取参数
line = match.group(0)
params_match = re.search(r'\((.*?)\)', line)
params = params_match.group(1) if params_match else ''
functions.append({
'name': func_name,
'params': params,
'line': self.code[:match.start()].count('\n') + 1
})
return functions
def extract_classes(self):
"""提取所有类"""
classes = []
for match in self.PATTERNS['class'].finditer(self.code):
class_name = match.group(1)
# 提取父类
line = match.group(0)
inheritance_match = re.search(r'class\s+\w+\s*\(([^)]*)\)', line)
parents = inheritance_match.group(1) if inheritance_match else ''
classes.append({
'name': class_name,
'parents': parents,
'line': self.code[:match.start()].count('\n') + 1
})
return classes
def extract_todos(self):
"""提取所有TODO注释"""
todos = []
for match in self.PATTERNS['todo'].finditer(self.code):
todos.append({
'task': match.group(1).strip(),
'line': self.code[:match.start()].count('\n') + 1
})
return todos
def analyze_complexity(self):
"""简单的复杂度分析"""
# 计算圈复杂度(简化版)
complexity = 1 # 基础复杂度
complexity += len(self.PATTERNS['function'].findall(self.code))
complexity += len(re.findall(r'\bif\b', self.code))
complexity += len(re.findall(r'\belif\b', self.code))
complexity += len(re.findall(r'\bfor\b', self.code))
complexity += len(re.findall(r'\bwhile\b', self.code))
complexity += len(re.findall(r'\band\b', self.code))
complexity += len(re.findall(r'\bor\b', self.code))
complexity += len(re.findall(r'\bexcept\b', self.code))
return complexity
# 示例代码
sample_code = """
#!/usr/bin/env python
# -*- coding: utf-8 -*-
import os
import sys
from datetime import datetime
from typing import List, Optional
# TODO: Add error handling
class DataProcessor:
\"\"\"A class for processing data.\"\"\"
def __init__(self, data: List):
self.data = data
self.processed = False
def process(self) -> List:
\"\"\"Process the data.\"\"\"
# FIXME: This is inefficient for large datasets
result = []
for item in self.data:
if item is not None:
result.append(self._transform(item))
self.processed = True
return result
def _transform(self, item):
# TODO: Implement better transformation
return str(item).upper()
def main():
# Main function
processor = DataProcessor([1, 2, 3, None, 5])
result = processor.process()
print(f"Result: {result}")
if __name__ == '__main__':
main()
"""
# 分析代码
analyzer = CodeAnalyzer(sample_code)
print("=== 代码分析报告 ===")
print(f"行数统计: {analyzer.count_lines()}")
print(f"\n导入的模块: {analyzer.extract_imports()}")
print("\n函数列表:")
for func in analyzer.extract_functions():
print(f" {func['name']}({func['params']}) - 第{func['line']}行")
print("\n类列表:")
for cls in analyzer.extract_classes():
parents = f"继承自 {cls['parents']}" if cls['parents'] else "无继承"
print(f" {cls['name']} ({parents}) - 第{cls['line']}行")
print("\nTODO列表:")
for todo in analyzer.extract_todos():
print(f" 第{todo['line']}行: {todo['task']}")
print(f"\n圈复杂度: {analyzer.analyze_complexity()}")
五、性能优化与最佳实践
5.1 正则表达式优化技巧
import timeit
class RegexOptimizer:
"""正则表达式优化示例"""
@staticmethod
def compare_performance():
"""比较不同写法的性能"""
text = "a" * 1000 + "b"
# 1. 回溯问题:嵌套量词
patterns = {
'bad': r'(a+)+b', # 灾难性回溯
'good': r'a+b', # 简单直接
'better': re.compile(r'a+b') # 编译后
}
print("性能对比:")
for name, pattern in patterns.items():
if name == 'better':
def test(): return pattern.match(text)
else:
def test(): return re.match(pattern, text)
try:
time = timeit.timeit(test, number=1000)
print(f" {name:8s}: {time:.6f}秒")
except Exception as e:
print(f" {name:8s}: 错误 - {e}")
@staticmethod
def optimize_pattern(pattern):
"""优化正则表达式模式"""
tips = []
# 检查灾难性回溯
if re.search(r'\([^)]*[+*][^)]*\)[+*]', pattern):
tips.append("警告:可能存在嵌套量词,可能导致灾难性回溯")
# 检查是否使用了 .* (贪婪匹配)
if '.*' in pattern and '.*?' not in pattern:
tips.append("建议:考虑使用 .*? 非贪婪匹配,或更具体的字符类")
# 检查字符类优化
if '[0-9]' in pattern:
tips.append("建议:使用 \\d 代替 [0-9]")
if '[a-zA-Z0-9_]' in pattern:
tips.append("建议:使用 \\w 代替 [a-zA-Z0-9_]")
# 检查是否应该使用非捕获组
if '(' in pattern and '(?' not in pattern:
tips.append("建议:如果不需要捕获,使用 (?:...) 非捕获组提高性能")
# 检查锚点使用
if pattern.startswith('.*'):
tips.append("建议:在开头使用 .* 可能效率低下,考虑使用更具体的模式")
return tips if tips else ["模式看起来不错!"]
# 运行性能对比
RegexOptimizer.compare_performance()
# 检查模式优化建议
patterns_to_check = [
r'(a+)+b',
r'.*keyword.*',
r'[0-9]{3}-[0-9]{4}',
r'(https?://)?(www\.)?([a-z]+\.com)'
]
print("\n优化建议:")
for pattern in patterns_to_check:
print(f"\n模式: {pattern}")
for tip in RegexOptimizer.optimize_pattern(pattern):
print(f" - {tip}")
5.2 常用正则表达式模式库
class RegexLibrary:
"""常用正则表达式库"""
PATTERNS = {
# 网络相关
'email': r'^[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}$',
'url': r'^https?://(?:www\.)?[-a-zA-Z0-9@:%._+~#=]{1,256}\.[a-zA-Z0-9()]{1,6}\b(?:[-a-zA-Z0-9()@:%_+.~#?&/=]*)$',
'domain': r'^[a-zA-Z0-9][a-zA-Z0-9-]{0,61}[a-zA-Z0-9]?(?:\.[a-zA-Z]{2,})+$',
'ipv4': r'^(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?)\.){3}(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?)$',
'ipv6': r'^(?:[A-F0-9]{1,4}:){7}[A-F0-9]{1,4}$',
'mac_address': r'^([0-9A-Fa-f]{2}[:-]){5}([0-9A-Fa-f]{2})$',
# 个人信息
'phone_us': r'^\(?([0-9]{3})\)?[-. ]?([0-9]{3})[-. ]?([0-9]{4})$',
'phone_cn': r'^1[3-9]\d{9}$',
'zipcode_us': r'^\d{5}(?:-\d{4})?$',
'zipcode_cn': r'^\d{6}$',
'ssn': r'^\d{3}-\d{2}-\d{4}$',
# 金融相关
'credit_card': r'^(?:4[0-9]{12}(?:[0-9]{3})?|5[1-5][0-9]{14}|3[47][0-9]{13}|6(?:011|5[0-9][0-9])[0-9]{12})$',
'currency': r'^\$?\d{1,3}(?:,?\d{3})*(?:\.\d{2})?$',
'bitcoin_address': r'^[13][a-km-zA-HJ-NP-Z1-9]{25,34}$',
# 日期时间
'date_iso': r'^\d{4}-\d{2}-\d{2}$',
'date_us': r'^(0[1-9]|1[0-2])/(0[1-9]|[12][0-9]|3[01])/\d{4}$',
'time_24h': r'^([01]?[0-9]|2[0-3]):[0-5][0-9](?::[0-5][0-9])?$',
'time_12h': r'^(0?[1-9]|1[0-2]):[0-5][0-9](?::[0-5][0-9])?\s?[APap][Mm]$',
'datetime_iso': r'^\d{4}-\d{2}-\d{2}[ T]\d{2}:\d{2}:\d{2}(?:\.\d+)?(?:Z|[+-]\d{2}:?\d{2})?$',
# 编程相关
'identifier': r'^[a-zA-Z_][a-zA-Z0-9_]*$',
'semver': r'^\d+\.\d+\.\d+(?:-[0-9A-Za-z-]+(?:\.[0-9A-Za-z-]+)*)?(?:\+[0-9A-Za-z-]+)?$',
'uuid': r'^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$',
# 文件路径
'filename_windows': r'^[^<>:"/\\|?*]+$',
'filepath_unix': r'^(/[^/ ]*)+/?$',
'file_extension': r'^.*\.([^.]+)$',
# 其他
'hex_color': r'^#?([a-fA-F0-9]{6}|[a-fA-F0-9]{3})$',
'strong_password': r'^(?=.*[A-Z])(?=.*[a-z])(?=.*\d)(?=.*[!@#$%^&*])[A-Za-z\d!@#$%^&*]{8,}$',
'username': r'^[a-zA-Z0-9._-]{3,16}$',
'slug': r'^[a-z0-9]+(?:-[a-z0-9]+)*$',
}
@classmethod
def validate(cls, pattern_name, value):
"""验证值是否符合指定模式"""
if pattern_name not in cls.PATTERNS:
raise ValueError(f"未知模式: {pattern_name}")
pattern = cls.PATTERNS[pattern_name]
return bool(re.match(pattern, str(value), re.IGNORECASE))
@classmethod
def get_pattern(cls, pattern_name):
"""获取模式字符串"""
return cls.PATTERNS.get(pattern_name)
@classmethod
def list_patterns(cls):
"""列出所有可用模式"""
return sorted(cls.PATTERNS.keys())
# 使用正则库
print("=== 正则表达式库测试 ===\n")
test_cases = [
('email', 'user@example.com'),
('email', 'invalid-email'),
('url', 'https://www.example.com'),
('ipv4', '192.168.1.1'),
('ipv4', '256.256.256.256'),
('phone_cn', '13812345678'),
('date_iso', '2024-01-15'),
('uuid', '550e8400-e29b-41d4-a716-446655440000'),
('hex_color', '#FF5733'),
('strong_password', 'WeakPass'),
('strong_password', 'Str0ngP@ssw0rd!'),
]
for pattern_name, value in test_cases:
is_valid = RegexLibrary.validate(pattern_name, value)
status = '✓' if is_valid else '✗'
print(f"{status} {pattern_name:15s}: {value}")
print(f"\n可用模式 ({len(RegexLibrary.list_patterns())}个):")
for i, name in enumerate(RegexLibrary.list_patterns(), 1):
print(f" {i:2d}. {name}")
5.3 最佳实践总结
"""
正则表达式最佳实践清单:
1. 性能优化
✓ 编译频繁使用的正则表达式
✓ 避免灾难性回溯(嵌套量词)
✓ 使用非贪婪匹配 .*? 而非 .*
✓ 使用非捕获组 (?:...) 提高性能
✓ 使用字符类代替 . 当可能时
2. 可读性维护
✓ 使用 re.VERBOSE 添加注释
✓ 给复杂的正则表达式命名
✓ 将复杂的正则拆分为多个简单的
✓ 使用命名分组 (?P<name>...)
3. 正确性保证
✓ 使用原始字符串 r'pattern' 避免转义问题
✓ 注意 Unicode 字符处理
✓ 测试边界情况
✓ 使用 ^ 和 $ 确保完整匹配
4. 安全性考虑
✓ 永远不要信任用户输入的正则表达式
✓ 对用户输入进行转义 re.escape()
✓ 设置超时机制(使用第三方库)
✓ 限制输入长度避免 ReDoS 攻击
5. 调试技巧
✓ 使用在线工具可视化正则表达式
✓ 逐步构建复杂模式
✓ 使用 re.DEBUG 标志查看解析树
✓ 编写单元测试验证正则表达式
"""
# 示例:安全的用户输入处理
def safe_search(pattern, text):
"""安全的正则搜索"""
try:
# 限制输入长度
if len(text) > 10000:
text = text[:10000]
# 编译模式
compiled = re.compile(pattern)
# 执行搜索
return compiled.search(text)
except re.error as e:
print(f"正则表达式错误: {e}")
return None
except Exception as e:
print(f"搜索错误: {e}")
return None
# 示例:使用 re.VERBOSE 提高可读性
EMAIL_PATTERN = re.compile(r"""
^ # 字符串开头
(?P<local> # 本地部分
[a-zA-Z0-9._%+-]+ # 用户名
)
@ # @ 符号
(?P<domain> # 域名部分
[a-zA-Z0-9.-]+ # 域名
\. # 点
[a-zA-Z]{2,} # 顶级域名
)
$ # 字符串结尾
""", re.VERBOSE | re.IGNORECASE)
# 示例:转义用户输入
def find_literal_text(text, search_for):
"""在文本中查找字面量字符串"""
# 转义特殊字符
escaped = re.escape(search_for)
pattern = re.compile(escaped)
return pattern.findall(text)
user_input = "user@example.com"
text = "Contact user@example.com for more info"
matches = find_literal_text(text, user_input)
print(f"查找 '{user_input}': {matches}")
六、总结
正则表达式是处理文本的强大工具,掌握它需要理解三个层次:
- 语法层面:熟悉元字符、量词、字符类等基本元素
- 功能层面:理解分组、断言、替换等高级功能
- 应用层面:知道何时使用正则,何时使用字符串方法
关键要点: - 正则表达式是"写一次,读百次"的代码,可读性至关重要 - 编译频繁使用的模式以提高性能 - 注意灾难性回溯问题 - 使用合适的标志(re.VERBOSE、re.IGNORECASE等) - 测试边界情况和异常输入
何时使用正则表达式: - 复杂的模式匹配和提取 - 数据验证和清洗 - 文本解析和转换 - 日志分析
何时避免使用正则表达式:
- 简单的字符串查找(用 in)
- 简单的开头/结尾检查(用 startswith/endswith)
- 简单的替换(用 replace)
- 解析 HTML/XML(用专门的解析器)
记住:正则表达式是工具箱中的一把利器,但不是唯一的工具。在合适的场景使用合适的工具,才能写出优雅、高效的代码。
本文由 尚先生 原创,转载请注明出处。
评论
0