Python利用正则提取字符串中数字的方法汇总

更新时间：2026年02月15日 08:28:26 作者：detayun

本文介绍了10种使用Python正则表达式提取股票代码的方法,从简单直接的findall提取所有数字,到精确匹配特定格式的代码,再到处理多种格式的股票代码和批量处理列表,需要的朋友可以参考下

方法1：最简单直接的方法（推荐）
方法2：使用search提取第一个数字
方法3：精确匹配’cn_'后的数字
方法4：使用findall处理多个数字
方法5：使用split分割提取
方法6：提取并转换为整数
方法7：处理多种格式的股票代码
方法8：使用命名捕获组（更清晰）
方法9：批量处理列表
方法10：完整的股票代码提取工具
方法11：处理您的Redis股票数据场景
快速解决方案
正则表达式说明

方法1：最简单直接的方法（推荐）

import re

text = 'cn_000858'

# 提取所有数字
numbers = re.findall(r'\d+', text)
print(numbers)  # ['000858']

# 如果只需要第一个匹配结果
if numbers:
    result = numbers[0]
    print(result)  # '000858'

方法2：使用search提取第一个数字

import re

text = 'cn_000858'

# 搜索第一个数字序列
match = re.search(r'\d+', text)
if match:
    result = match.group()
    print(result)  # '000858'

方法3：精确匹配’cn_'后的数字

import re

text = 'cn_000858'

# 匹配'cn_'后面的数字
match = re.search(r'cn_(\d+)', text)
if match:
    result = match.group(1)  # group(1)获取第一个捕获组
    print(result)  # '000858'

方法4：使用findall处理多个数字

import re

# 如果字符串中有多个数字
text = 'cn_000858_stock_123'

# 提取所有数字
numbers = re.findall(r'\d+', text)
print(numbers)  # ['000858', '123']

# 提取特定位置的数字
match = re.search(r'cn_(\d+)', text)
if match:
    stock_code = match.group(1)
    print(f"股票代码: {stock_code}")  # '000858'

方法5：使用split分割提取

import re

text = 'cn_000858'

# 使用下划线分割，取最后一部分
parts = text.split('_')
result = parts[-1]
print(result)  # '000858'

# 或者用正则split
result = re.split(r'[^\d]+', text)[-1]
print(result)  # '000858'

方法6：提取并转换为整数

import re

text = 'cn_000858'

# 提取数字并转换为整数（会去掉前导零）
match = re.search(r'\d+', text)
if match:
    number_str = match.group()
    number_int = int(number_str)
    print(f"字符串: {number_str}, 整数: {number_int}")
    # 输出: 字符串: 000858, 整数: 858

# 如果需要保留前导零，保持字符串形式
result = match.group()
print(f"保留前导零: {result}")  # '000858'

方法7：处理多种格式的股票代码

import re

def extract_stock_code(text):
    """
    从各种格式中提取股票代码
    """
    # 匹配多种模式: cn_000858, sh600000, sz000001, 000858等
    patterns = [
        r'cn_(\d+)',           # cn_000858
        r'(?:sh|sz)(\d+)',     # sh600000, sz000001
        r'^(\d{6})$',          # 纯数字6位
        r'_(\d{6})',           # 下划线后6位数字
    ]
    
    for pattern in patterns:
        match = re.search(pattern, text, re.IGNORECASE)
        if match:
            return match.group(1)
    
    return None

# 测试
test_cases = [
    'cn_000858',
    'sh600000',
    'sz000001',
    '000858',
    'stock_cn_000858_data',
]

for text in test_cases:
    code = extract_stock_code(text)
    print(f"{text:20} -> [code]")

方法8：使用命名捕获组（更清晰）

import re

text = 'cn_000858'

# 使用命名捕获组
match = re.search(r'cn_(?P<code>\d+)', text)
if match:
    code = match.group('code')
    print(f"股票代码: [code]")  # '000858'
    
    # 也可以用groupdict()
    print(match.groupdict())  # {'code': '000858'}

方法9：批量处理列表

import re

# 批量处理多个字符串
texts = [
    'cn_000858',
    'cn_000001',
    'sh600000',
    'sz000002',
]

# 使用列表推导式
codes = [re.search(r'\d+', text).group() for text in texts if re.search(r'\d+', text)]
print(codes)  # ['000858', '000001', '600000', '000002']

# 更安全的方式（处理可能没有数字的情况）
codes = []
for text in texts:
    match = re.search(r'\d+', text)
    if match:
        codes.append(match.group())
print(codes)

方法10：完整的股票代码提取工具

import re
from typing import Optional, List

class StockCodeExtractor:
    """股票代码提取器"""
    
    @staticmethod
    def extract(text: str) -> Optional[str]:
        """
        从文本中提取股票代码
        支持格式: cn_000858, sh600000, sz000001, 000858等
        """
        if not text:
            return None
        
        # 模式列表（按优先级排序）
        patterns = [
            (r'cn_(\d{6})', 1),           # cn_000858
            (r'(?:sh|sz|hk)(\d{6})', 1),   # sh600000, sz000001, hk00700
            (r'^(\d{6})$', 1),             # 纯6位数字
            (r'_(\d{6})', 1),              # 下划线后6位数字
            (r'\b(\d{6})\b', 1),           # 单词边界的6位数字
            (r'\d+', 0),                    # 任意数字（最后备选）
        ]
        
        for pattern, group_idx in patterns:
            match = re.search(pattern, text, re.IGNORECASE)
            if match:
                code = match.group(group_idx)
                # 验证是否为6位数字（股票代码通常是6位）
                if len(code) == 6 and code.isdigit():
                    return code
                # 如果不是6位但也没有其他匹配，返回它
                elif group_idx == 0:  # 最后一个模式
                    return code
        
        return None
    
    @staticmethod
    def extract_all(text: str) -> List[str]:
        """
        提取文本中所有可能的股票代码
        """
        # 提取所有6位数字序列
        codes = re.findall(r'\b\d{6}\b', text)
        return codes
    
    @staticmethod
    def validate(code: str) -> bool:
        """
        验证股票代码格式
        """
        return bool(re.match(r'^\d{6}$', code))

# 使用示例
if __name__ == "__main__":
    extractor = StockCodeExtractor()
    
    test_cases = [
        'cn_000858',
        '深南电路 cn_000858',
        'sh600000',
        'sz000001',
        '000858',
        '股票代码: 000858',
        'hk00700',
    ]
    
    print("单个提取:")
    for text in test_cases:
        code = extractor.extract(text)
        is_valid = extractor.validate(code) if code else False
        print(f"{text:25} -> {code:10} (有效: {is_valid})")
    
    print("\n批量提取:")
    text = "关注股票: cn_000858, sh600000, sz000001"
    codes = extractor.extract_all(text)
    print(f"文本: {text}")
    print(f"提取的代码: {codes}")

方法11：处理您的Redis股票数据场景

import re
import json

def extract_stock_code_from_redis(data_str):
    """
    从Redis读取的数据中提取股票代码
    """
    try:
        # 先尝试解析JSON
        data = json.loads(data_str)
        
        # 如果是字典，查找包含代码的字段
        if isinstance(data, dict):
            # 尝试常见的字段名
            code_fields = ['code', 'stock_code', 'symbol', 'ts_code']
            for field in code_fields:
                if field in data:
                    code = data[field]
                    # 提取数字部分
                    match = re.search(r'\d+', str(code))
                    if match:
                        return match.group()
            
            # 如果没有找到，尝试从所有值中提取
            for value in data.values():
                match = re.search(r'\d+', str(value))
                if match and len(match.group()) == 6:
                    return match.group()
        
        # 如果是字符串，直接提取
        elif isinstance(data, str):
            match = re.search(r'\d{6}', data)
            if match:
                return match.group()
    
    except json.JSONDecodeError:
        # 如果不是JSON，直接从字符串提取
        match = re.search(r'\d{6}', data_str)
        if match:
            return match.group()
    
    return None

# 测试
test_data = [
    '{"name": "深南电路", "code": "cn_000858"}',
    '{"ts_code": "000858.SZ", "name": "深南电路"}',
    'cn_000858',
    '000858',
]

for data in test_data:
    code = extract_stock_code_from_redis(data)
    print(f"{data:40} -> 代码: [code]")

快速解决方案

针对您的具体需求，最简单的代码：

import re

# 从 'cn_000858' 提取数字
text = 'cn_000858'
result = re.search(r'\d+', text).group()
print(result)  # '000858'

# 或者一行代码
result = re.findall(r'\d+', 'cn_000858')[0]
print(result)  # '000858'

正则表达式说明

模式	说明	示例
`\d+`	匹配一个或多个数字	‘000858’
`\d{6}`	匹配恰好6个数字	‘000858’
`cn_(\d+)`	匹配’cn_'后的数字	‘000858’
`(?:sh\|sz)(\d+)`	匹配sh或sz后的数字	‘600000’
`\b\d{6}\b`	单词边界的6位数字	‘000858’

选择最适合您场景的方法即可！

以上就是Python使用正则提取字符串中数字的方法汇总的详细内容，更多关于Python正则提取字符串中数字的资料请关注脚本之家其它相关文章！

您可能感兴趣的文章:

Python获取字典变量内存占用的四种方法
本文主要介绍了Python获取字典变量内存占用的四种方法,文中通过示例代码介绍的非常详细,对大家的学习或者工作具有一定的参考学习价值,需要的朋友们下面随着小编来一起学习学习吧
2025-09-09
Python更新所有已安装包的操作
今天小编就为大家分享一篇Python更新所有已安装包的操作，具有很好的参考价值，希望对大家有所帮助。一起跟随小编过来看看吧
2020-02-02
python 中字典嵌套列表的方法
今天小编就为大家分享一篇python 中字典嵌套列表的方法，具有很好的参考价值，希望对大家有所帮助。一起跟随小编过来看看吧
2018-07-07
利用python控制Qt程序的示例详解
这篇文章主要为大家详细介绍了如何利用python实现控制Qt程序，从而进行文本输入，按钮点击等组件控制，感兴趣的小伙伴可以跟随小编一起学习一下
2023-08-08
Python实现的朴素贝叶斯分类器示例
这篇文章主要介绍了Python实现的朴素贝叶斯分类器,结合具体实例形式分析了基于Python实现的朴素贝叶斯分类器相关定义与使用技巧,需要的朋友可以参考下
2018-01-01
python 发送get请求接口详解
这篇文章主要介绍了python 发送get请求接口详解,文中通过示例代码介绍的非常详细，对大家的学习或者工作具有一定的参考学习价值，需要的朋友们下面随着小编来一起学习学习吧
2020-11-11
Python使用ImageIO库实现从图像到视频的高效读写
在现代图像处理与计算机视觉项目中,我们几乎无时无刻不需要与图像文件或视频数据打交道,无论是读取一张照片、保存图像处理结果,还是从视频中提取帧序列,这些操作的第一步都是高效、正确地读写图像数据,所以本文介绍了Python如何使用ImageIO库实现从图像到视频的高效读写
2025-11-11
Tensorflow之梯度裁剪的实现示例
这篇文章主要介绍了Tensorflow之梯度裁剪的实现示例，文中通过示例代码介绍的非常详细，对大家的学习或者工作具有一定的参考学习价值，需要的朋友们下面随着小编来一起学习学习吧
2020-03-03
Python 自动安装 Rising 杀毒软件
平日里经常需要重新安装杀毒软件,我使用的是 Rising 该软件可以将升级后的新版本,压缩成一个安装包,当升级失败造成硬盘中的 Rising
2009-04-04
python实现大转盘抽奖效果
这篇文章主要为大家详细介绍了python实现大转盘抽奖效果，文中示例代码介绍的非常详细，具有一定的参考价值，感兴趣的小伙伴们可以参考一下
2019-01-01