Python正则

Python基础 2026-04-20 15
预计阅读时间:37 分钟

Python 正则表达式完全指南

一、正则表达式的本质:模式匹配的艺术

正则表达式(Regular Expression)是一种描述字符串模式的语言。它就像一个精密的过滤器,能从海量文本中准确提取你需要的信息。掌握正则表达式,你就拥有了处理文本的超能力。

1.1 为什么需要正则表达式?

# 没有正则表达式的生活:繁琐且易错
def validate_email_manual(email):
    """手动验证邮箱(痛苦的方式)"""
    if '@' not in email:
        return False

    parts = email.split('@')
    if len(parts) != 2:
        return False

    local, domain = parts
    if not local or not domain:
        return False

    if '.' not in domain:
        return False

    # 还有很多规则要检查...
    return True

# 使用正则表达式:简洁且强大
import re

def validate_email_regex(email):
    """使用正则验证邮箱(优雅的方式)"""
    pattern = r'^[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}$'
    return bool(re.match(pattern, email))

# 测试
test_emails = [
    'user@example.com',
    'invalid.email',
    'user@domain',
    'user.name+tag@sub.domain.co.uk'
]

for email in test_emails:
    print(f"{email:30s} -> {'✓' if validate_email_regex(email) else '✗'}")

1.2 Python 的 re 模块基础

import re

# re 模块的主要函数
text = "Python is awesome! Python is powerful!"

# 1. re.search() - 搜索第一个匹配
match = re.search(r'Python', text)
if match:
    print(f"search: 找到 '{match.group()}' 在位置 {match.start()}-{match.end()}")

# 2. re.match() - 从开头匹配
match = re.match(r'Python', text)
print(f"match: {'成功' if match else '失败'}(从开头匹配)")

match = re.match(r'awesome', text)
print(f"match: {'成功' if match else '失败'}(不在开头)")

# 3. re.findall() - 查找所有匹配
matches = re.findall(r'Python', text)
print(f"findall: 找到 {len(matches)} 个匹配: {matches}")

# 4. re.finditer() - 返回迭代器
for match in re.finditer(r'Python', text):
    print(f"finditer: 在位置 {match.start()} 找到 '{match.group()}'")

# 5. re.sub() - 替换
new_text = re.sub(r'Python', 'JavaScript', text)
print(f"sub: {new_text}")

# 6. re.split() - 分割
parts = re.split(r'\s+', text)
print(f"split: {parts}")

二、正则表达式基础语法

2.1 元字符

# 元字符是正则表达式的基石
"""
.    - 匹配任意单个字符(除换行符)
^    - 匹配字符串开头
$    - 匹配字符串结尾
*    - 匹配0次或多次
+    - 匹配1次或多次
?    - 匹配0次或1次
{m}  - 匹配精确m次
{m,n}- 匹配m到n次
[]   - 字符类,匹配其中的任意字符
|    - 或运算符
()   - 分组
\    - 转义字符
"""

# 示例:演示每个元字符
test_strings = [
    "cat",
    "caaat",
    "ct",
    "dog",
    "cBt"
]

print("元字符示例:")
for s in test_strings:
    # . 匹配任意字符
    if re.match(r'c.t', s):
        print(f"  c.t 匹配: {s}")

    # * 匹配0次或多次
    if re.match(r'ca*t', s):
        print(f"  ca*t 匹配: {s}")

    # + 匹配1次或多次
    if re.match(r'ca+t', s):
        print(f"  ca+t 匹配: {s}")

# 字符类 []
print("\n字符类示例:")
text = "The cat sat on the mat"
# 匹配 cat, sat, mat
words = re.findall(r'[csm]at', text)
print(f"  [csm]at 匹配: {words}")

# 范围
print(f"  [0-9] 匹配数字: {re.findall(r'[0-9]', 'a1b2c3')}")
print(f"  [a-z] 匹配小写: {re.findall(r'[a-z]', 'AbC123')}")
print(f"  [A-Z] 匹配大写: {re.findall(r'[A-Z]', 'AbC123')}")
print(f"  [^0-9] 匹配非数字: {re.findall(r'[^0-9]', 'a1b2c3')}")

# 量词
print("\n量词示例:")
print(f"  a{{3}} 精确3个a: {re.findall(r'a{3}', 'aaaaa')}")
print(f"  a{{2,4}} 2-4个a: {re.findall(r'a{2,4}', 'aaaaa')}")
print(f"  a{{2,}} 至少2个a: {re.findall(r'a{2,}', 'aaaaa')}")

2.2 特殊字符类

# 预定义的字符类
"""
\d - 数字 [0-9]
\D - 非数字 [^0-9]
\w - 单词字符 [a-zA-Z0-9_]
\W - 非单词字符 [^a-zA-Z0-9_]
\s - 空白字符 [ \t\n\r\f\v]
\S - 非空白字符 [^ \t\n\r\f\v]
\b - 单词边界
\B - 非单词边界
"""

text = "User123@email.com has 2 cats and 3 dogs!"

print("预定义字符类:")
print(f"  数字: {re.findall(r'\d+', text)}")
print(f"  单词: {re.findall(r'\w+', text)}")
print(f"  非单词字符: {re.findall(r'\W+', text)}")

# 单词边界
print("\n单词边界示例:")
text = "cat category educate"
print(f"  'cat' 单词: {re.findall(r'\bcat\b', text)}")
print(f"  包含'cat': {re.findall(r'cat', text)}")
print(f"  以'cat'开头: {re.findall(r'\bcat', text)}")
print(f"  以'cat'结尾: {re.findall(r'cat\b', text)}")

2.3 分组与捕获

# 分组是正则表达式最强大的特性之一
text = "John Doe, 30 years old, email: john@example.com"
text2 = "Jane Smith, 25 years old, email: jane.smith@company.co.uk"

# 1. 普通分组 ()
pattern = r'(\w+)\s+(\w+),\s+(\d+)\s+years old,\s+email:\s+(\S+@\S+)'

for t in [text, text2]:
    match = re.search(pattern, t)
    if match:
        first_name = match.group(1)
        last_name = match.group(2)
        age = match.group(3)
        email = match.group(4)
        print(f"姓名: {first_name} {last_name}, 年龄: {age}, 邮箱: {email}")
        print(f"  所有分组: {match.groups()}")

# 2. 命名分组 (?P<name>...)
pattern = r'(?P<first>\w+)\s+(?P<last>\w+),\s+(?P<age>\d+)\s+years old,\s+email:\s+(?P<email>\S+@\S+)'

match = re.search(pattern, text)
if match:
    print(f"\n命名分组:")
    print(f"  姓名: {match.group('first')} {match.group('last')}")
    print(f"  年龄: {match.group('age')}")
    print(f"  邮箱: {match.group('email')}")
    print(f"  分组字典: {match.groupdict()}")

# 3. 非捕获分组 (?:...)
# 用于分组但不捕获,提高性能
text = "https://www.example.com/path/to/page"
pattern = r'(?:https?://)?(?:www\.)?([^/]+)'

match = re.search(pattern, text)
if match:
    print(f"\n域名(非捕获分组): {match.group(1)}")

# 4. 反向引用
# 查找重复单词
text = "The the cat sat on on the mat mat"
duplicates = re.findall(r'\b(\w+)\s+\1\b', text, re.IGNORECASE)
print(f"\n重复单词: {duplicates}")

# 查找成对的HTML标签
html = "<div>Content</div> <p>Paragraph</p> <div>Mismatch</p>"
pattern = r'<(\w+)>.*?</\1>'
tags = re.findall(pattern, html)
print(f"成对标签: {tags}")

2.4 断言(零宽断言)

# 断言匹配位置,不消耗字符
"""
(?=...)  - 正向前瞻:后面跟着...
(?!...)  - 负向前瞻:后面不跟着...
(?<=...) - 正向后顾:前面是...
(?<!...) - 负向后顾:前面不是...
"""

text = "apple price: $5.99, orange price: $3.49, banana price: $2.99"

# 1. 正向前瞻:只匹配后面跟着数字的价格
prices_with_dollar = re.findall(r'\$\d+\.\d{2}', text)
print(f"所有价格: {prices_with_dollar}")

# 只提取数字部分(不包含$)
prices_only = re.findall(r'(?<=\$)\d+\.\d{2}', text)
print(f"价格数字(后顾): {prices_only}")

# 2. 负向前瞻:匹配不以特定模式结尾的单词
words = "color colour colur"
# 匹配 colo 后面不是 r 的
pattern = r'colo(?!u)r'
print(f"负向前瞻: {re.findall(pattern, words)}")

# 3. 复杂示例:密码验证
def validate_password(password):
    """
    密码要求:
    - 至少8个字符
    - 包含至少一个大写字母
    - 包含至少一个小写字母
    - 包含至少一个数字
    - 包含至少一个特殊字符
    """
    # 使用正向前瞻同时检查多个条件
    pattern = r'^(?=.*[A-Z])(?=.*[a-z])(?=.*\d)(?=.*[!@#$%^&*])[A-Za-z\d!@#$%^&*]{8,}$'
    return bool(re.match(pattern, password))

test_passwords = [
    "Weak",
    "weakpassword",
    "WEAKPASSWORD",
    "NoNumbers!",
    "StrongP@ss123",
    "Short1!"
]

print("\n密码验证:")
for pwd in test_passwords:
    print(f"  {pwd:15s} -> {'✓' if validate_password(pwd) else '✗'}")

# 4. 更复杂的断言示例
text = "The quick brown fox jumps over the lazy dog"

# 找出所有后面跟着空格和"fox"的单词
words_before_fox = re.findall(r'\b\w+\b(?=\s+fox)', text)
print(f"\n'fox'前面的单词: {words_before_fox}")

# 找出所有前面是"the"的单词(不区分大小写)
words_after_the = re.findall(r'(?<=\b[Tt]he\s+)\w+', text)
print(f"'the'后面的单词: {words_after_the}")

三、re 模块高级功能

3.1 编译正则表达式

# 编译正则表达式可以提高性能(多次使用时)
import timeit

# 未编译
def uncompiled_search(text):
    return re.findall(r'\b\w{4,}\b', text)

# 编译
pattern = re.compile(r'\b\w{4,}\b')
def compiled_search(text):
    return pattern.findall(text)

text = "Python is an awesome programming language" * 1000

# 性能对比
uncompiled_time = timeit.timeit(lambda: uncompiled_search(text), number=100)
compiled_time = timeit.timeit(lambda: compiled_search(text), number=100)

print(f"未编译: {uncompiled_time:.4f}秒")
print(f"已编译: {compiled_time:.4f}秒")
print(f"编译后快 {uncompiled_time/compiled_time:.2f}倍")

# 编译时指定标志
# re.IGNORECASE 或 re.I - 忽略大小写
# re.MULTILINE 或 re.M - 多行模式
# re.DOTALL 或 re.S - . 匹配包括换行符
# re.VERBOSE 或 re.X - 详细模式(可添加注释)

pattern = re.compile(r"""
    ^               # 开头
    (?P<username>   # 用户名分组
        [a-zA-Z0-9._%+-]+
    )
    @               # @ 符号
    (?P<domain>     # 域名分组
        [a-zA-Z0-9.-]+
        \.
        [a-zA-Z]{2,}
    )
    $               # 结尾
""", re.VERBOSE | re.IGNORECASE)

emails = ['user@example.com', 'invalid.email', 'USER@DOMAIN.COM']
for email in emails:
    match = pattern.match(email)
    if match:
        print(f"有效邮箱: {email}")
        print(f"  用户名: {match.group('username')}")
        print(f"  域名: {match.group('domain')}")

3.2 替换的高级用法

# 1. 使用函数进行替换
def format_currency(match):
    """格式化货币"""
    amount = float(match.group(1))
    return f"${amount:,.2f}"

text = "Prices: USD 1234.56, EUR 789.12, GBP 345.67"
# 将货币数字格式化为带千位分隔符
formatted = re.sub(r'(\d+\.\d{2})', format_currency, text)
print(f"格式化货币: {formatted}")

# 2. 处理日期格式
def convert_date_format(match):
    """转换日期格式:MM/DD/YYYY -> YYYY-MM-DD"""
    month, day, year = match.groups()
    return f"{year}-{month.zfill(2)}-{day.zfill(2)}"

text = "Dates: 1/15/2024, 12/25/2023, 6/1/2024"
converted = re.sub(r'(\d{1,2})/(\d{1,2})/(\d{4})', convert_date_format, text)
print(f"日期格式转换: {converted}")

# 3. 使用命名分组进行替换
text = "Name: John Doe, Email: john@example.com"
pattern = r'Name: (?P<name>\w+\s+\w+), Email: (?P<email>\S+@\S+)'

# 使用 \g<name> 引用命名分组
result = re.sub(pattern, r'Contact: \g<name> <\g<email>>', text)
print(f"格式化联系人: {result}")

# 4. 条件替换
def mask_sensitive_data(text):
    """遮蔽敏感数据"""
    # 遮蔽邮箱
    text = re.sub(
        r'(\w{1,3})[\w._%+-]*(\w{1,3}@[\w.-]+\.[a-zA-Z]{2,})',
        r'\1***\2',
        text
    )

    # 遮蔽电话号码(保留后4位)
    text = re.sub(
        r'\b(\d{3})[-.]?(\d{3})[-.]?(\d{4})\b',
        r'***-***-\3',
        text
    )

    # 遮蔽信用卡号(只显示后4位)
    text = re.sub(
        r'\b\d{4}[ -]?\d{4}[ -]?\d{4}[ -]?(\d{4})\b',
        r'****-****-****-\1',
        text
    )

    return text

sensitive_text = """
Contact: john.doe@example.com
Phone: 123-456-7890
Card: 4111-1111-1111-1234
"""

masked = mask_sensitive_data(sensitive_text)
print(f"遮蔽后:\n{masked}")

3.3 贪婪与非贪婪匹配

# 贪婪匹配(默认):尽可能多地匹配
# 非贪婪匹配:尽可能少地匹配(使用 ?)

html = "<div>First div</div> <div>Second div</div>"

# 贪婪匹配
greedy = re.findall(r'<div>.*</div>', html)
print(f"贪婪匹配: {greedy}")
# 输出:['<div>First div</div> <div>Second div</div>'] - 匹配了整个字符串

# 非贪婪匹配
non_greedy = re.findall(r'<div>.*?</div>', html)
print(f"非贪婪匹配: {non_greedy}")
# 输出:['<div>First div</div>', '<div>Second div</div>'] - 分别匹配每个div

# 量词的非贪婪版本
"""
*?   - 0次或多次,尽可能少
+?   - 1次或多次,尽可能少
??   - 0次或1次,尽可能少
{m,n}? - m到n次,尽可能少
"""

# 实际应用:提取HTML标签内容
def extract_tag_content(html, tag):
    """提取指定标签的内容"""
    pattern = f'<{tag}.*?>(.*?)</{tag}>'
    return re.findall(pattern, html, re.DOTALL)

html_content = """
<html>
    <title>My Page</title>
    <body>
        <h1>Welcome</h1>
        <p>This is the first paragraph.</p>
        <p>This is the second paragraph.</p>
    </body>
</html>
"""

titles = extract_tag_content(html_content, 'title')
paragraphs = extract_tag_content(html_content, 'p')

print(f"标题: {titles}")
print(f"段落: {paragraphs}")

四、实战应用

4.1 日志分析

import re
from collections import Counter
from datetime import datetime

# 示例日志格式
sample_log = """
2024-01-15 10:23:45 INFO User alice logged in from 192.168.1.100
2024-01-15 10:24:12 ERROR Database connection failed: timeout
2024-01-15 10:25:01 INFO User bob logged in from 10.0.0.50
2024-01-15 10:25:45 WARNING High memory usage: 85%
2024-01-15 10:26:30 ERROR API request failed: 500 Internal Server Error
2024-01-15 10:27:15 INFO User alice performed action: delete_file
2024-01-15 10:28:00 ERROR Database connection failed: timeout
2024-01-15 10:28:45 INFO User charlie logged in from 172.16.0.20
"""

class LogAnalyzer:
    """日志分析器"""

    # 日志格式正则表达式
    LOG_PATTERN = re.compile(r"""
        ^(\d{4}-\d{2}-\d{2}\s+\d{2}:\d{2}:\d{2})\s+  # 时间戳
        (\w+)\s+                                        # 日志级别
        (.*)$                                           # 消息
    """, re.VERBOSE | re.MULTILINE)

    # IP地址提取模式
    IP_PATTERN = re.compile(r'\b(?:\d{1,3}\.){3}\d{1,3}\b')

    # 用户提取模式
    USER_PATTERN = re.compile(r'User\s+(\w+)')

    def __init__(self, log_text):
        self.log_text = log_text
        self.entries = self._parse_logs()

    def _parse_logs(self):
        """解析日志条目"""
        entries = []
        for match in self.LOG_PATTERN.finditer(self.log_text):
            timestamp_str, level, message = match.groups()
            timestamp = datetime.strptime(timestamp_str, '%Y-%m-%d %H:%M:%S')

            # 提取IP
            ip_match = self.IP_PATTERN.search(message)
            ip = ip_match.group() if ip_match else None

            # 提取用户
            user_match = self.USER_PATTERN.search(message)
            user = user_match.group(1) if user_match else None

            entries.append({
                'timestamp': timestamp,
                'level': level,
                'message': message.strip(),
                'ip': ip,
                'user': user
            })

        return entries

    def get_level_stats(self):
        """统计日志级别"""
        levels = Counter(entry['level'] for entry in self.entries)
        return dict(levels)

    def get_user_activity(self):
        """统计用户活动"""
        users = Counter(entry['user'] for entry in self.entries if entry['user'])
        return dict(users)

    def get_ip_stats(self):
        """统计IP地址"""
        ips = Counter(entry['ip'] for entry in self.entries if entry['ip'])
        return dict(ips)

    def get_errors(self):
        """获取所有错误"""
        return [entry for entry in self.entries if entry['level'] == 'ERROR']

    def find_pattern(self, pattern):
        """搜索特定模式"""
        return [
            entry for entry in self.entries
            if re.search(pattern, entry['message'], re.IGNORECASE)
        ]

# 使用日志分析器
analyzer = LogAnalyzer(sample_log)

print("=== 日志分析结果 ===")
print(f"总日志条目: {len(analyzer.entries)}")
print(f"级别统计: {analyzer.get_level_stats()}")
print(f"用户活动: {analyzer.get_user_activity()}")
print(f"IP统计: {analyzer.get_ip_stats()}")

print("\n错误日志:")
for error in analyzer.get_errors():
    print(f"  [{error['timestamp']}] {error['message']}")

print("\n包含'database'的日志:")
for entry in analyzer.find_pattern(r'database'):
    print(f"  [{entry['level']}] {entry['message']}")

4.2 数据提取与清洗

class DataExtractor:
    """数据提取器"""

    # 常见模式
    PATTERNS = {
        'email': re.compile(r'\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Z|a-z]{2,}\b'),
        'url': re.compile(r'https?://(?:[-\w.]|(?:%[\da-fA-F]{2}))+[^\s]*'),
        'phone': re.compile(r'\b\d{3}[-.]?\d{3}[-.]?\d{4}\b'),
        'date': re.compile(r'\b\d{4}-\d{2}-\d{2}\b|\b\d{2}/\d{2}/\d{4}\b'),
        'ip': re.compile(r'\b(?:\d{1,3}\.){3}\d{1,3}\b'),
        'hashtag': re.compile(r'#\w+'),
        'mention': re.compile(r'@\w+'),
        'price': re.compile(r'\$\d+(?:\.\d{2})?'),
        'time': re.compile(r'\b\d{2}:\d{2}(?::\d{2})?\b'),
    }

    @classmethod
    def extract(cls, text, pattern_type):
        """提取指定类型的数据"""
        if pattern_type not in cls.PATTERNS:
            raise ValueError(f"未知模式: {pattern_type}")

        pattern = cls.PATTERNS[pattern_type]
        return pattern.findall(text)

    @classmethod
    def extract_all(cls, text):
        """提取所有类型的数据"""
        results = {}
        for name, pattern in cls.PATTERNS.items():
            matches = pattern.findall(text)
            if matches:
                results[name] = matches
        return results

    @classmethod
    def clean_html(cls, html):
        """清理HTML标签"""
        # 移除所有HTML标签
        text = re.sub(r'<[^>]+>', '', html)
        # 替换HTML实体
        entities = {
            '&nbsp;': ' ',
            '&lt;': '<',
            '&gt;': '>',
            '&amp;': '&',
            '&quot;': '"',
            '&apos;': "'"
        }
        for entity, char in entities.items():
            text = text.replace(entity, char)
        # 压缩空白字符
        text = re.sub(r'\s+', ' ', text)
        return text.strip()

    @classmethod
    def normalize_phone(cls, phone):
        """标准化电话号码"""
        # 移除所有非数字字符
        digits = re.sub(r'\D', '', phone)

        # 格式化为 (XXX) XXX-XXXX
        if len(digits) == 10:
            return f"({digits[:3]}) {digits[3:6]}-{digits[6:]}"
        elif len(digits) == 11 and digits[0] == '1':
            return f"+1 ({digits[1:4]}) {digits[4:7]}-{digits[7:]}"
        else:
            return phone

# 测试数据提取器
sample_text = """
Contact us at support@example.com or visit https://www.example.com/help
Call us at 123-456-7890 or (987) 654-3210
Follow us on Twitter @example and use #ExampleTag
Prices start at $19.99 and go up to $299.00
Server IP: 192.168.1.1
Meeting at 14:30 on 2024-01-15
"""

print("=== 数据提取示例 ===")
for pattern_name in ['email', 'url', 'phone', 'ip', 'hashtag', 'mention', 'price', 'date', 'time']:
    matches = DataExtractor.extract(sample_text, pattern_name)
    if matches:
        print(f"{pattern_name:10s}: {matches}")

# HTML清理
html_sample = """
<div class="content">
    <h1>Welcome to our &lt;site&gt;</h1>
    <p>This is a paragraph with <strong>bold</strong> text.</p>
    <p>Another &nbsp;&nbsp; paragraph with    multiple     spaces.</p>
</div>
"""

cleaned = DataExtractor.clean_html(html_sample)
print(f"\n清理后的HTML:\n{cleaned}")

# 电话号码标准化
phones = ['123-456-7890', '987.654.3210', '1234567890', '11234567890']
print("\n电话号码标准化:")
for phone in phones:
    print(f"  {phone:15s} -> {DataExtractor.normalize_phone(phone)}")

4.3 代码分析工具

class CodeAnalyzer:
    """Python 代码分析器"""

    # Python 代码模式
    PATTERNS = {
        'import': re.compile(r'^import\s+(\w+)|^from\s+(\w+)\s+import', re.MULTILINE),
        'function': re.compile(r'^def\s+(\w+)\s*\([^)]*\):', re.MULTILINE),
        'class': re.compile(r'^class\s+(\w+)[:(]', re.MULTILINE),
        'variable': re.compile(r'^(\w+)\s*=\s*', re.MULTILINE),
        'decorator': re.compile(r'^@(\w+)', re.MULTILINE),
        'comment': re.compile(r'#.*$', re.MULTILINE),
        'docstring': re.compile(r'""".*?"""|\'\'\'.*?\'\'\'', re.DOTALL),
        'todo': re.compile(r'#\s*TODO:?\s*(.*)$', re.MULTILINE | re.IGNORECASE),
        'fixme': re.compile(r'#\s*FIXME:?\s*(.*)$', re.MULTILINE | re.IGNORECASE),
    }

    def __init__(self, code):
        self.code = code
        self.lines = code.split('\n')

    def count_lines(self):
        """统计代码行数"""
        total = len(self.lines)
        empty = sum(1 for line in self.lines if not line.strip())
        comments = len(self.PATTERNS['comment'].findall(self.code))

        return {
            'total': total,
            'empty': empty,
            'comments': comments,
            'code': total - empty - comments
        }

    def extract_imports(self):
        """提取所有导入"""
        imports = set()
        for match in self.PATTERNS['import'].finditer(self.code):
            module = match.group(1) or match.group(2)
            if module:
                imports.add(module)
        return sorted(imports)

    def extract_functions(self):
        """提取所有函数"""
        functions = []
        for match in self.PATTERNS['function'].finditer(self.code):
            func_name = match.group(1)
            # 尝试提取参数
            line = match.group(0)
            params_match = re.search(r'\((.*?)\)', line)
            params = params_match.group(1) if params_match else ''

            functions.append({
                'name': func_name,
                'params': params,
                'line': self.code[:match.start()].count('\n') + 1
            })
        return functions

    def extract_classes(self):
        """提取所有类"""
        classes = []
        for match in self.PATTERNS['class'].finditer(self.code):
            class_name = match.group(1)

            # 提取父类
            line = match.group(0)
            inheritance_match = re.search(r'class\s+\w+\s*\(([^)]*)\)', line)
            parents = inheritance_match.group(1) if inheritance_match else ''

            classes.append({
                'name': class_name,
                'parents': parents,
                'line': self.code[:match.start()].count('\n') + 1
            })
        return classes

    def extract_todos(self):
        """提取所有TODO注释"""
        todos = []
        for match in self.PATTERNS['todo'].finditer(self.code):
            todos.append({
                'task': match.group(1).strip(),
                'line': self.code[:match.start()].count('\n') + 1
            })
        return todos

    def analyze_complexity(self):
        """简单的复杂度分析"""
        # 计算圈复杂度(简化版)
        complexity = 1  # 基础复杂度
        complexity += len(self.PATTERNS['function'].findall(self.code))
        complexity += len(re.findall(r'\bif\b', self.code))
        complexity += len(re.findall(r'\belif\b', self.code))
        complexity += len(re.findall(r'\bfor\b', self.code))
        complexity += len(re.findall(r'\bwhile\b', self.code))
        complexity += len(re.findall(r'\band\b', self.code))
        complexity += len(re.findall(r'\bor\b', self.code))
        complexity += len(re.findall(r'\bexcept\b', self.code))

        return complexity

# 示例代码
sample_code = """
#!/usr/bin/env python
# -*- coding: utf-8 -*-

import os
import sys
from datetime import datetime
from typing import List, Optional

# TODO: Add error handling
class DataProcessor:
    \"\"\"A class for processing data.\"\"\"

    def __init__(self, data: List):
        self.data = data
        self.processed = False

    def process(self) -> List:
        \"\"\"Process the data.\"\"\"
        # FIXME: This is inefficient for large datasets
        result = []
        for item in self.data:
            if item is not None:
                result.append(self._transform(item))

        self.processed = True
        return result

    def _transform(self, item):
        # TODO: Implement better transformation
        return str(item).upper()

def main():
    # Main function
    processor = DataProcessor([1, 2, 3, None, 5])
    result = processor.process()
    print(f"Result: {result}")

if __name__ == '__main__':
    main()
"""

# 分析代码
analyzer = CodeAnalyzer(sample_code)

print("=== 代码分析报告 ===")
print(f"行数统计: {analyzer.count_lines()}")

print(f"\n导入的模块: {analyzer.extract_imports()}")

print("\n函数列表:")
for func in analyzer.extract_functions():
    print(f"  {func['name']}({func['params']}) - 第{func['line']}行")

print("\n类列表:")
for cls in analyzer.extract_classes():
    parents = f"继承自 {cls['parents']}" if cls['parents'] else "无继承"
    print(f"  {cls['name']} ({parents}) - 第{cls['line']}行")

print("\nTODO列表:")
for todo in analyzer.extract_todos():
    print(f"  第{todo['line']}行: {todo['task']}")

print(f"\n圈复杂度: {analyzer.analyze_complexity()}")

五、性能优化与最佳实践

5.1 正则表达式优化技巧

import timeit

class RegexOptimizer:
    """正则表达式优化示例"""

    @staticmethod
    def compare_performance():
        """比较不同写法的性能"""
        text = "a" * 1000 + "b"

        # 1. 回溯问题:嵌套量词
        patterns = {
            'bad': r'(a+)+b',      # 灾难性回溯
            'good': r'a+b',         # 简单直接
            'better': re.compile(r'a+b')  # 编译后
        }

        print("性能对比:")
        for name, pattern in patterns.items():
            if name == 'better':
                def test(): return pattern.match(text)
            else:
                def test(): return re.match(pattern, text)

            try:
                time = timeit.timeit(test, number=1000)
                print(f"  {name:8s}: {time:.6f}秒")
            except Exception as e:
                print(f"  {name:8s}: 错误 - {e}")

    @staticmethod
    def optimize_pattern(pattern):
        """优化正则表达式模式"""
        tips = []

        # 检查灾难性回溯
        if re.search(r'\([^)]*[+*][^)]*\)[+*]', pattern):
            tips.append("警告:可能存在嵌套量词,可能导致灾难性回溯")

        # 检查是否使用了 .* (贪婪匹配)
        if '.*' in pattern and '.*?' not in pattern:
            tips.append("建议:考虑使用 .*? 非贪婪匹配,或更具体的字符类")

        # 检查字符类优化
        if '[0-9]' in pattern:
            tips.append("建议:使用 \\d 代替 [0-9]")
        if '[a-zA-Z0-9_]' in pattern:
            tips.append("建议:使用 \\w 代替 [a-zA-Z0-9_]")

        # 检查是否应该使用非捕获组
        if '(' in pattern and '(?' not in pattern:
            tips.append("建议:如果不需要捕获,使用 (?:...) 非捕获组提高性能")

        # 检查锚点使用
        if pattern.startswith('.*'):
            tips.append("建议:在开头使用 .* 可能效率低下,考虑使用更具体的模式")

        return tips if tips else ["模式看起来不错!"]

# 运行性能对比
RegexOptimizer.compare_performance()

# 检查模式优化建议
patterns_to_check = [
    r'(a+)+b',
    r'.*keyword.*',
    r'[0-9]{3}-[0-9]{4}',
    r'(https?://)?(www\.)?([a-z]+\.com)'
]

print("\n优化建议:")
for pattern in patterns_to_check:
    print(f"\n模式: {pattern}")
    for tip in RegexOptimizer.optimize_pattern(pattern):
        print(f"  - {tip}")

5.2 常用正则表达式模式库

class RegexLibrary:
    """常用正则表达式库"""

    PATTERNS = {
        # 网络相关
        'email': r'^[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}$',
        'url': r'^https?://(?:www\.)?[-a-zA-Z0-9@:%._+~#=]{1,256}\.[a-zA-Z0-9()]{1,6}\b(?:[-a-zA-Z0-9()@:%_+.~#?&/=]*)$',
        'domain': r'^[a-zA-Z0-9][a-zA-Z0-9-]{0,61}[a-zA-Z0-9]?(?:\.[a-zA-Z]{2,})+$',
        'ipv4': r'^(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?)\.){3}(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?)$',
        'ipv6': r'^(?:[A-F0-9]{1,4}:){7}[A-F0-9]{1,4}$',
        'mac_address': r'^([0-9A-Fa-f]{2}[:-]){5}([0-9A-Fa-f]{2})$',

        # 个人信息
        'phone_us': r'^\(?([0-9]{3})\)?[-. ]?([0-9]{3})[-. ]?([0-9]{4})$',
        'phone_cn': r'^1[3-9]\d{9}$',
        'zipcode_us': r'^\d{5}(?:-\d{4})?$',
        'zipcode_cn': r'^\d{6}$',
        'ssn': r'^\d{3}-\d{2}-\d{4}$',

        # 金融相关
        'credit_card': r'^(?:4[0-9]{12}(?:[0-9]{3})?|5[1-5][0-9]{14}|3[47][0-9]{13}|6(?:011|5[0-9][0-9])[0-9]{12})$',
        'currency': r'^\$?\d{1,3}(?:,?\d{3})*(?:\.\d{2})?$',
        'bitcoin_address': r'^[13][a-km-zA-HJ-NP-Z1-9]{25,34}$',

        # 日期时间
        'date_iso': r'^\d{4}-\d{2}-\d{2}$',
        'date_us': r'^(0[1-9]|1[0-2])/(0[1-9]|[12][0-9]|3[01])/\d{4}$',
        'time_24h': r'^([01]?[0-9]|2[0-3]):[0-5][0-9](?::[0-5][0-9])?$',
        'time_12h': r'^(0?[1-9]|1[0-2]):[0-5][0-9](?::[0-5][0-9])?\s?[APap][Mm]$',
        'datetime_iso': r'^\d{4}-\d{2}-\d{2}[ T]\d{2}:\d{2}:\d{2}(?:\.\d+)?(?:Z|[+-]\d{2}:?\d{2})?$',

        # 编程相关
        'identifier': r'^[a-zA-Z_][a-zA-Z0-9_]*$',
        'semver': r'^\d+\.\d+\.\d+(?:-[0-9A-Za-z-]+(?:\.[0-9A-Za-z-]+)*)?(?:\+[0-9A-Za-z-]+)?$',
        'uuid': r'^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$',

        # 文件路径
        'filename_windows': r'^[^<>:"/\\|?*]+$',
        'filepath_unix': r'^(/[^/ ]*)+/?$',
        'file_extension': r'^.*\.([^.]+)$',

        # 其他
        'hex_color': r'^#?([a-fA-F0-9]{6}|[a-fA-F0-9]{3})$',
        'strong_password': r'^(?=.*[A-Z])(?=.*[a-z])(?=.*\d)(?=.*[!@#$%^&*])[A-Za-z\d!@#$%^&*]{8,}$',
        'username': r'^[a-zA-Z0-9._-]{3,16}$',
        'slug': r'^[a-z0-9]+(?:-[a-z0-9]+)*$',
    }

    @classmethod
    def validate(cls, pattern_name, value):
        """验证值是否符合指定模式"""
        if pattern_name not in cls.PATTERNS:
            raise ValueError(f"未知模式: {pattern_name}")

        pattern = cls.PATTERNS[pattern_name]
        return bool(re.match(pattern, str(value), re.IGNORECASE))

    @classmethod
    def get_pattern(cls, pattern_name):
        """获取模式字符串"""
        return cls.PATTERNS.get(pattern_name)

    @classmethod
    def list_patterns(cls):
        """列出所有可用模式"""
        return sorted(cls.PATTERNS.keys())

# 使用正则库
print("=== 正则表达式库测试 ===\n")

test_cases = [
    ('email', 'user@example.com'),
    ('email', 'invalid-email'),
    ('url', 'https://www.example.com'),
    ('ipv4', '192.168.1.1'),
    ('ipv4', '256.256.256.256'),
    ('phone_cn', '13812345678'),
    ('date_iso', '2024-01-15'),
    ('uuid', '550e8400-e29b-41d4-a716-446655440000'),
    ('hex_color', '#FF5733'),
    ('strong_password', 'WeakPass'),
    ('strong_password', 'Str0ngP@ssw0rd!'),
]

for pattern_name, value in test_cases:
    is_valid = RegexLibrary.validate(pattern_name, value)
    status = '✓' if is_valid else '✗'
    print(f"{status} {pattern_name:15s}: {value}")

print(f"\n可用模式 ({len(RegexLibrary.list_patterns())}个):")
for i, name in enumerate(RegexLibrary.list_patterns(), 1):
    print(f"  {i:2d}. {name}")

5.3 最佳实践总结

"""
正则表达式最佳实践清单:

1. 性能优化
   ✓ 编译频繁使用的正则表达式
   ✓ 避免灾难性回溯(嵌套量词)
   ✓ 使用非贪婪匹配 .*? 而非 .*
   ✓ 使用非捕获组 (?:...) 提高性能
   ✓ 使用字符类代替 . 当可能时

2. 可读性维护
   ✓ 使用 re.VERBOSE 添加注释
   ✓ 给复杂的正则表达式命名
   ✓ 将复杂的正则拆分为多个简单的
   ✓ 使用命名分组 (?P<name>...)

3. 正确性保证
   ✓ 使用原始字符串 r'pattern' 避免转义问题
   ✓ 注意 Unicode 字符处理
   ✓ 测试边界情况
   ✓ 使用 ^ 和 $ 确保完整匹配

4. 安全性考虑
   ✓ 永远不要信任用户输入的正则表达式
   ✓ 对用户输入进行转义 re.escape()
   ✓ 设置超时机制(使用第三方库)
   ✓ 限制输入长度避免 ReDoS 攻击

5. 调试技巧
   ✓ 使用在线工具可视化正则表达式
   ✓ 逐步构建复杂模式
   ✓ 使用 re.DEBUG 标志查看解析树
   ✓ 编写单元测试验证正则表达式
"""

# 示例:安全的用户输入处理
def safe_search(pattern, text):
    """安全的正则搜索"""
    try:
        # 限制输入长度
        if len(text) > 10000:
            text = text[:10000]

        # 编译模式
        compiled = re.compile(pattern)

        # 执行搜索
        return compiled.search(text)
    except re.error as e:
        print(f"正则表达式错误: {e}")
        return None
    except Exception as e:
        print(f"搜索错误: {e}")
        return None

# 示例:使用 re.VERBOSE 提高可读性
EMAIL_PATTERN = re.compile(r"""
    ^                       # 字符串开头
    (?P<local>              # 本地部分
        [a-zA-Z0-9._%+-]+   # 用户名
    )
    @                       # @ 符号
    (?P<domain>             # 域名部分
        [a-zA-Z0-9.-]+      # 域名
        \.                  # 点
        [a-zA-Z]{2,}        # 顶级域名
    )
    $                       # 字符串结尾
""", re.VERBOSE | re.IGNORECASE)

# 示例:转义用户输入
def find_literal_text(text, search_for):
    """在文本中查找字面量字符串"""
    # 转义特殊字符
    escaped = re.escape(search_for)
    pattern = re.compile(escaped)
    return pattern.findall(text)

user_input = "user@example.com"
text = "Contact user@example.com for more info"
matches = find_literal_text(text, user_input)
print(f"查找 '{user_input}': {matches}")

六、总结

正则表达式是处理文本的强大工具,掌握它需要理解三个层次:

  1. 语法层面:熟悉元字符、量词、字符类等基本元素
  2. 功能层面:理解分组、断言、替换等高级功能
  3. 应用层面:知道何时使用正则,何时使用字符串方法

关键要点: - 正则表达式是"写一次,读百次"的代码,可读性至关重要 - 编译频繁使用的模式以提高性能 - 注意灾难性回溯问题 - 使用合适的标志(re.VERBOSE、re.IGNORECASE等) - 测试边界情况和异常输入

何时使用正则表达式: - 复杂的模式匹配和提取 - 数据验证和清洗 - 文本解析和转换 - 日志分析

何时避免使用正则表达式: - 简单的字符串查找(用 in) - 简单的开头/结尾检查(用 startswith/endswith) - 简单的替换(用 replace) - 解析 HTML/XML(用专门的解析器)

记住:正则表达式是工具箱中的一把利器,但不是唯一的工具。在合适的场景使用合适的工具,才能写出优雅、高效的代码。


本文由 尚先生 原创,转载请注明出处。

📖相关推荐

评论

0
暂无评论,来发表第一条评论吧

发表评论

登录 后发表评论