Files
cwExcel/app/finance_parse.py
T
2026-07-18 15:04:19 +08:00

633 lines
26 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# spacy_training/scripts/predict.py
import re
import sys
import cn2an
import os
import os.path
from tqdm import tqdm
from pathlib import Path
# PyInstaller --noconsole 兼容:无控制台时重定向输出到空设备
if sys.stdout is None:
sys.stdout = open(os.devnull, 'w')
if sys.stderr is None:
sys.stderr = open(os.devnull, 'w')
# 单位与连接符定义
MAX_ITEM_OFFSET = 10
UNIT_PATTERN = "个|张|片|块|条|幅|根"
SIZE_UNITS_PATTERN = r'mm|cm|dm|m'
SIZE_JOIN_PATTERN = "xX×"
NUMS_PATTERN = r'\d+(?:\.\d+)' # 修复:原缺少 ? 导致整数不匹配
CNT_PATTERN = rf'{NUMS_PATTERN}?[^克+\s\d-]*'
EXCEPT_TEXT = "解析失败"
INDEX_COL = '序号'
base_colume = ['解析备注', '尺寸备注', '描述', '异常信息']
mapping_size = {
"width_mm": "",
"height_mm": "",
# "original_size": "原始尺寸",
# "total_quantity": "总数量",
"style_count": "款数",
"quantity_per_style": "数量",
"unit": "单位",
"描述": "描述",
"解析备注": "解析备注",
"size_unit": "尺寸单位",
"exception_msg": '异常信息'
}
import spacy
import pandas as pd
# 将 doc 转换为字典
def doc_to_dict(doc):
return {
"text": doc.text,
"entities": [
{
"text": ent.text,
"label": ent.label_,
"start": ent.start_char,
"end": ent.end_char
}
for ent in doc.ents
],
}
def predict_file(df, model_path="./spacy_training/model", key='备注'):
print(model_path)
nlp = spacy.load(model_path)
if "sentencizer" not in nlp.pipe_names:
nlp.add_pipe("sentencizer")
# 统一表头空格
df.columns = df.columns.str.strip()
key = key.strip()
# 流式批处理:用 as_tuples 边读边处理,batch_size=32 兼顾速度与内存
stream = ((str(row[key]).strip(), {k: v for k, v in row.items()}) for _, row in df.iterrows())
for doc, base_row in tqdm(nlp.pipe(stream, as_tuples=True, batch_size=32), total=len(df)):
yield doc.text, base_row, doc_to_dict(doc)
from dataclasses import dataclass
@dataclass
class StructuredOrder:
key: str
width_mm: float
height_mm: float
original_size: str
style_count: int
quantity_per_style: float
unit: str
total_quantity: float
original_qty: str
size_position: list
qty_position: list
exception_msg: str
calc_type: str
size_unit: str
meta: dict
def expand_to_structured(raw_pairs, size_unit=''):
unit_map = {
'': 100, '': 1000, '': 10000,
'百万': 1000000, '千万': 10000000, '亿': 100000000,
}
UNIT_REGEX = re.compile(rf'({UNIT_PATTERN})$')
exception_msg = []
def extract_multiplier(text):
"""
从字符串中提取数值乘数,支持:
- 阿拉伯数字 + 中文单位:3.5万 → 35000
- 纯中文数字 + 单位:三万五千 → 35000
- 纯中文:两千万 → 20000000
"""
# 去除末尾单位(保留前面的部分)
clean = UNIT_REGEX.sub('', text.strip())
# 如果去单位后为空,原字符串可能是纯单位(如“万”),默认系数为1
if not clean:
matched_unit = UNIT_REGEX.search(text)
if matched_unit:
unit = matched_unit.group(1)
return unit_map.get(unit, 1.0)
return 1.0
# 尝试直接转阿拉伯数字
try:
return float(clean)
except ValueError:
pass
# 尝试转中文数字(如“三万五千”虽然不合理,但“三万”可以)
try:
# 注意:cn2an 可以直接处理“三万五千”这种
num = cn2an.cn2an(clean, "smart") # smart 模式支持混合写法
if isinstance(num, (int, float)):
# 检查原字符串是否有单位后缀(比如“三万”中的“万”已被去除,需补回)
matched_unit = UNIT_REGEX.search(text)
if matched_unit:
unit = matched_unit.group(1)
num *= unit_map[unit]
return float(num)
except Exception:
exception_msg.append(f"尺寸解析异常:{text}")
pass
return 1.0
def parse_size(size_str):
join_str = ''.join(re.findall(f"[{SIZE_JOIN_PATTERN}]+", size_str))
if len(join_str) == 0 or re.search('比例|等高|等比|等宽', size_str):
result = f'异常值:{size_str}'
return result, result, size_unit
size_str = size_str.replace(join_str, "x")
match = re.match(rf'({NUMS_PATTERN}?)([a-z]+)?[{SIZE_JOIN_PATTERN}]({NUMS_PATTERN}?)([a-z]+)?', size_str, re.IGNORECASE)
if not match:
return None, None, size_unit
w_val, w_unit, h_val, h_unit = match.groups()
unit = (w_unit or h_unit or size_unit).lower()
conv = {'m': 1000, 'dm': 100, 'cm': 10, 'mm': 1}
if unit not in conv:
print(f'未知单位: {unit} size_str:{size_str} size_unit:{size_unit}')
exception_msg.append(f"尺寸解析异常:{size_str}, 未知单位: {unit}")
return None, None, unit
w_mm = conv[unit] * float(w_val)
h_mm = conv[unit] * float(h_val)
return round(w_mm, 3), round(h_mm, 3), unit
def parse_quantity(qty_str):
qty_str = "".join(qty_str.split())
multi_match = re.match((
rf'[共计]?(.+?)款\w*?([共计]?[\d.一二两三四五六七八九十百千万亿]+)({UNIT_PATTERN})'
), qty_str, re.IGNORECASE)
if multi_match:
style_part, qty_part = multi_match.group(1), multi_match.group(2)
if qty_part[0] in '共计':
qty_part = qty_part[1:]
calc_type = 'total'
else:
calc_type = 'single'
style_count = int(extract_multiplier(style_part))
quantity_per_style = int(extract_multiplier(qty_part))
unit = multi_match.group(3)
return {"style_count": style_count, "quantity_per_style": quantity_per_style, "unit": unit, "calc_type": calc_type}
single_match = re.match(rf'(.+?)({UNIT_PATTERN})', qty_str, re.IGNORECASE)
if single_match:
qty_part = single_match.group(1)
if qty_part[0] in '共计':
qty_part = qty_part[1:]
calc_type = 'total'
else:
calc_type = 'single'
quantity_per_style = int(extract_multiplier(qty_part))
unit = single_match.group(2)
return {"style_count": 1, "quantity_per_style": quantity_per_style, "unit": unit, "calc_type": calc_type}
return {"style_count": 0, "quantity_per_style": 0, "unit": ""}
structured = []
for size_info, qty_info in raw_pairs:
s_start, s_end, s_text, *_ = size_info
q_start, q_end, q_text, *_ = qty_info
if s_text == EXCEPT_TEXT:
width_mm, height_mm = EXCEPT_TEXT, EXCEPT_TEXT
else:
size_parsed = parse_size(s_text)
if size_parsed:
width_mm, height_mm, size_unit = size_parsed
else:
width_mm, height_mm, size_unit = None, None, None
if q_text == EXCEPT_TEXT:
style_count, quantity_per_style, unit = 0, EXCEPT_TEXT, EXCEPT_TEXT
else:
qty_parsed = parse_quantity(q_text)
style_count, quantity_per_style, unit = qty_parsed["style_count"], qty_parsed["quantity_per_style"], qty_parsed["unit"]
structured.append(StructuredOrder(
key=f"{s_start}{s_end}{s_text}",
width_mm=width_mm,
height_mm=height_mm,
original_size=s_text,
style_count=style_count,
quantity_per_style=quantity_per_style,
unit=unit,
total_quantity=style_count * quantity_per_style if style_count and style_count else None,
original_qty=q_text,
size_position=[s_start, s_end],
qty_position=[q_start, q_end],
exception_msg="; ".join(exception_msg),
calc_type=qty_parsed.get('calc_type', None),
size_unit=size_unit,
meta={
"size_text_except": s_text if s_text == EXCEPT_TEXT or (not width_mm and not height_mm) else None,
"qty_text_except": q_text if q_text == EXCEPT_TEXT else None,
}
))
return structured
def decode(doc_dict_list, is_horizontal=False):
ent_group = {
# "单号": ["单号"],
"刮刮膜尺寸": ["刮刮膜尺寸"],
# "产品": ["产品"],
# "材质": ["材质"],
# "工艺": ["工艺"],
# "用户": ["用户信息"],
# "加急": ["加急"],
"数量": ["尺寸", "数量"],
}
columns = list(ent_group.keys())
columns.remove('数量')
columns.extend(base_colume)
group_reflect = {v: k for k, vs in ent_group.items() for v in vs}
result_items = []
exception_list = []
for doc_dict in doc_dict_list:
sentence = doc_dict["text"]
entities = doc_dict["entities"]
is_except = False
# 强关联实体需要进行组合处理
# ======================================== 实体分组 ========================================
ent_group_dict = {k: [] for k in ent_group.keys()}
base_ent_group_dict = {}
for ent in entities:
start, end, label, text = ent["start"], ent["end"], ent["label"], ent["text"]
if label not in group_reflect:
continue
group_name = group_reflect[label]
ent_group_dict[group_name].append((start, end, text, label))
base_ent_group_dict[group_name] = text
result = []
# 解析出尺寸和数量的组合
# ======================================== 解析尺寸和数量 ========================================
group_list = []
size_and_qty_list = [[], []]
# 数量需要和尺寸进行组合处理
base_ent_group_dict.pop("数量", None)
for cnt_item in ent_group_dict["数量"]:
size_info, qty_info = size_and_qty_list
label = cnt_item[-1]
if label == '数量':
qty_info.append(cnt_item)
elif label == '尺寸':
start, end, text, _label = cnt_item
if f'-({text[:2]}' in sentence:
continue
if qty_info:
group_list.append(size_and_qty_list)
size_and_qty_list = [[], []]
size_info, qty_info = size_and_qty_list
size_info.append(cnt_item)
# 存在数量和尺寸
if size_and_qty_list[0] or size_and_qty_list[1]:
group_list.append(size_and_qty_list)
# 如果只有一对尺寸,有可能反过来描述
if len(group_list) == 2:
_1, qty_info = group_list[0]
size_info, _2 = group_list[1]
if not _1 and not _2:
group_list[:] = [[qty_info, size_info]]
# print(f'订单中尺寸和数量组合:{group_list}')
# 解析出尺寸和数量的组合
# ======================================== 解析长、宽、款数、数量 ========================================
size_unit = ''
for size_info, qty_info in group_list:
units = set(''.join(re.findall(SIZE_UNITS_PATTERN, ent[2])) for ent in size_info)
units = [unit for unit in units if unit]
if len(units) == 1:
size_unit = units[0]
size_and_qty_parsed_result = []
for size_info, qty_info in group_list:
size_len, qty_len = len(size_info), len(qty_info)
if size_len == 0:
print(f'解析异常,尺寸为空,数量为{qty_info},原文:{sentence}', entities)
is_except = True
# 起始位置、结束位置、尺寸文本、尺寸标签
qty_info = [(0, 0, EXCEPT_TEXT, '数量')]
if qty_len == 0:
print(f'解析异常,数量为空,尺寸为{size_info},原文:{sentence}', entities)
is_except = True
# 起始位置、结束位置、尺寸文本、尺寸标签
size_info = [(0, 0, EXCEPT_TEXT, '尺寸')]
# continue
# print('size_info, qty_info', size_info, qty_info)
if size_len == qty_len:
decode_desc = "单尺寸、单款式描述的订单" if size_len == 1 else "多尺寸、多款式描述一一匹配的订单"
for size, qty in zip(size_info, qty_info):
expand = expand_to_structured([(size, qty)], size_unit) or []
exceptions = [item.meta for item in expand if item.meta.get('size_text_except') or item.meta.get('qty_text_except')]
if exceptions: print("解析异常", exceptions, '原文', sentence, entities)
for item in expand:
parsed_item = item.__dict__.copy()
parsed_item['描述'] = f"{size[2]}|{qty[2]}"
parsed_item['解析备注'] = decode_desc
parsed_item['size_len'] = size_len
parsed_item['qty_len'] = qty_len
size_and_qty_parsed_result.append(parsed_item)
elif size_len>1 and qty_len>1:
if size_len > qty_len:
total_qty = []
for size, qty in zip(size_info, qty_info):
expand = expand_to_structured([(size, qty)], size_unit) or []
for item in expand:
total_qty.append(item.style_count)
if size_len == sum(total_qty):
decode_desc = "多尺寸、多款式描述:多款式描述和尺寸相等的订单"
item = []
for idx, qty in enumerate(total_qty):
item.extend(qty_info[idx] for _ in range(qty))
print('解析:', size_info, qty_info, item)
# breakpoint()
qty_info = item
qty_len = len(qty_info)
else:
# qty_info = []
decode_desc = f"匹配异常:多尺寸、多款式描述的订单: {size_info} {qty_info}"
else:
# qty_info = []
decode_desc = f"匹配异常:多尺寸、多款式描述的订单: {size_info} {qty_info}"
for size, qty in zip(size_info, qty_info):
expand = expand_to_structured([(size, qty)], size_unit) or []
exceptions = [item.meta for item in expand if item.meta.get('size_text_except') or item.meta.get('qty_text_except')]
if exceptions: print("解析异常", exceptions, '原文', sentence, entities)
for item in expand:
parsed_item = item.__dict__.copy()
parsed_item['style_count'] = 1
parsed_item['描述'] = f"{size[2]}|{qty[2]}"
parsed_item['解析备注'] = decode_desc
parsed_item['size_len'] = size_len
parsed_item['qty_len'] = qty_len
size_and_qty_parsed_result.append(parsed_item)
else:
# 多尺寸-单数量、单尺寸-多数量 告警多尺寸-多数量
iter_item = ((size, qty) for size in size_info for qty in qty_info)
if size_len > 1 and qty_len == 1:
decode_desc = "多尺寸、单款式描述的订单"
elif size_len == 1 and qty_len > 1:
decode_desc = "单尺寸、多款式描述的订单"
else:
decode_desc = f"异常:多尺寸、多款式描述的订单: {size_info} {qty_info}"
for item in iter_item:
size, qty = item
expand = expand_to_structured([(size, qty)], size_unit) or []
exceptions = [item.meta for item in expand if item.meta.get('size_text_except') or item.meta.get('qty_text_except')]
if exceptions: print("解析异常", exceptions, '原文', sentence, entities)
for item in expand:
parsed_item = item.__dict__.copy()
parsed_item['描述'] = f"{size[2]}|{qty[2]}"
parsed_item['解析备注'] = decode_desc
parsed_item['size_len'] = size_len
parsed_item['qty_len'] = qty_len
size_and_qty_parsed_result.append(parsed_item)
result.extend(size_and_qty_parsed_result)
for item in result:
# print('item', item)
new_item = base_ent_group_dict.copy()
for key in mapping_size:
col_name = mapping_size[key]
new_item[col_name] = item[key]
columns.append(col_name)
size_len = item.pop('size_len', None)
qty_len = item.pop('qty_len', None)
calc_type = item.pop('calc_type', None)
if size_len > 1 and calc_type == 'total':
new_item[f'数量'] = f"警告:异常值(总计值:{new_item[f'数量']})"
if size_len == 1 or size_len == qty_len:
pass
# new_item[f'解析备注'] = '单尺寸的订单'
else:
if size_len == new_item['款数']:
new_item[f'款数'] = 1
# new_item[f'解析备注'] = '尺寸数量和款数相同的订单'
elif new_item['款数'] == 1:
new_item[f'款数'] = 1
# new_item[f'解析备注'] = '单款的订单'
else:
new_item[f'款数'] = "异常值:款数和尺寸数不一致"
new_item[f'数量'] = "异常值:款数和尺寸数不一致"
# new_item[f'解析备注'] = '多尺寸并且和款数不同的订单'
result_items.append(new_item)
# print(json.dumps(new_item, indent=4, ensure_ascii=False))
if not result:
if base_ent_group_dict:
base_ent_group_dict['是否异常'] = is_except
base_ent_group_dict[f'解析备注'] = '未解析到实际尺寸的订单1'
result_items.append(base_ent_group_dict.copy())
else:
result_items.append({
"是否异常": is_except,
"解析备注": "未解析到任何实体"
})
return exception_list, columns, result_items
def get_model_path():
"""获取模型路径,兼容开发环境和PyInstaller打包"""
if getattr(sys, '_MEIPASS', None):
# PyInstaller 打包后
return str(Path(sys._MEIPASS) / 'app' / 'resources' / 'models' / '订单尺寸识别' / 'best_model')
# 开发环境
return str(Path(__file__).parent / 'resources' / 'models' / '订单尺寸识别' / 'best_model')
def parse_finance_data(file_path, target_index, is_horizontal, sheet_name="Sheet1"):
df = pd.read_excel(file_path, dtype=str, sheet_name=sheet_name)
total_amount = len(df)
model_path = get_model_path()
print(f"使用模型路径: {model_path}")
items = predict_file(df, model_path, target_index)
print(f"解析完成,共{items}")
# items = predict_file(file_path, './resources/models/lintao/best_model', target_index)
exception_list, results, outputs_columns = [], [], []
origin_item = None
parsed_dict = {}
mapping_size_values = mapping_size.values()
for text, origin_item, item in items:
decode_result = []
base_info = {}
result_index = parsed_dict.get(f"{text}_index")
if result_index:
exceptions, columns = [], []
try:
decode_item = parsed_dict[text][result_index]
for key in origin_item:
if key in mapping_size_values:
continue
decode_item[key] = origin_item[key]
decode_item['尺寸备注'] = '多组尺寸解析'
parsed_dict[f"{text}_index"] += 1
continue
except IndexError:
print(f'解析到的尺寸不足:{text}{result_index} {parsed_dict[text]}')
base_item = parsed_dict[text][0]
for col in base_colume:
base_info[col] = base_item.get(col, '')
base_info[""] = '异常值:没有解析到那么多尺寸'
decode_items = []
parsed_dict[f"{text}_index"] += 1
else:
exceptions, columns, decode_items = decode([item], is_horizontal) or []
# if '(250731202428388557)' in text:
# print(result_index, text, exceptions, columns, decode_items)
# breakpoint()
if len(columns) > len(outputs_columns):
outputs_columns = columns
if len(decode_items) > 1:
for item in decode_items:
item['尺寸备注'] = '多组尺寸解析'
for idx, decode_item in enumerate(decode_items):
base_row = origin_item.copy()
if idx and INDEX_COL in base_row: base_row[INDEX_COL] = ''
# base_row = origin_item.copy() if idx == 0 else {}
base_row.update(decode_item)
is_except = base_row.pop('是否异常', False)
if is_except:
for col in base_colume:
origin_item[col] = base_row.get(col, '')
decode_result.append(origin_item)
exception_list.append(base_row)
else:
decode_result.append(base_row)
if not decode_items:
if base_info:
origin_item.update(base_info)
decode_result.append(origin_item)
else:
if is_horizontal:
new_decode_result = decode_result[0].copy()
for idx, item in enumerate(decode_result[1:], 1):
for key in mapping_size_values:
new_key = f"{key}{idx}"
new_decode_result[new_key] = item[key]
if new_key not in outputs_columns:
outputs_columns.append(new_key)
decode_result = [new_decode_result]
if text not in parsed_dict:
parsed_dict[text] = decode_result
parsed_dict[f"{text}_index"] = 0
parsed_dict[f"{text}_index"] += 1
# if decode_result:
# results.append(decode_result[0])
if len(decode_items) > 1:
for item in decode_result[1:]:
item['尺寸备注'] = '复制新增:多组尺寸解析'
results.extend(decode_result)
# break
print('横向解析', outputs_columns)
if origin_item:
# outputs_columns.
_outputs_columns = list(origin_item.keys()) + outputs_columns
col_set = set()
outputs_columns = []
for col in _outputs_columns:
if col not in col_set:
col_set.add(col)
outputs_columns.append(col)
df = pd.DataFrame(results, columns=outputs_columns)
file_name = os.path.basename(file_path).replace(".xlsx", '')
filename = file_name + '_解析' + ('_横向排列' if is_horizontal else '_纵向排列')
upload_dir = Path(file_path).parent
upload_dir.mkdir(parents=True, exist_ok=True)
save_file_path = upload_dir / f'{filename}_1.xlsx'
# 使用 ExcelWriter 同时写入多个 sheet
with pd.ExcelWriter(save_file_path, engine='openpyxl') as writer:
df.to_excel(writer, sheet_name='正常解析', index=False)
if exception_list:
except_df = pd.DataFrame(exception_list, columns=outputs_columns)
except_df.to_excel(writer, sheet_name='异常解析', index=False)
print(f'解析结果保存在:{save_file_path}')
return save_file_path, total_amount
if __name__ == "__main__":
# print(expand_to_structured([[(0,1,'600x50cm','r'), (0,1,'3款各1张','e')]]))
# exit()
# predict("Apple is opening a new office in Tokyo.")
# predict("Google hired Sarah Connor from Berlin last year.")
file_info = ['d:/会计组/新领图8月.xlsx', '备注']
# file_info = ['d:/会计组/即客2025年8月尺寸整理.xlsx', '系统文件名']
file_info = ['d:/会计组/8月订单明细9.6.xlsx', '文件名']
file_info = ['d:/会计组/ZHX-8月订单明细.xlsx', '文件名']
file_info = ['d:/会计组/国税数据源/智韬2025年8月尺寸整理(1).xlsx', '系统文件名']
file_info = ['d:/会计组/彩印通8月数码9.23.xlsx', 'ERP系统文件名']
file_info = ['d:/会计组/CYT8月明细9.24.xlsx', 'ERP系统文件名']
file_info = ['d:/会计组/9.1-9.26.xlsx', '备注']
file_info = ['d:/会计组/JD.xlsx', '文件名']
file_info = ['d:/会计组/泰州即客2025年9月尺寸整理.xlsx', '系统文件名']
file_info = ['d:/会计组/ZHX需拆明细9月.xlsx', '文件名', False]
file_info = ['d:/会计组/七彩2024年9月账单尺寸整理.xlsx', '系统文件名', True]
file_info = ['d:/会计组/艾印图文2024年9月账单尺寸整理.xlsx', '系统文件名', True]
file_info = ['d:/会计组/9月转印.xlsx', 'erp', True]
file_info = ['d:/会计组/9月CYT.xlsx', '文件名', False]
file_info = ['d:/会计组/智韬2025年9月尺寸整理.xlsx', '系统文件名', True]
file_info = ['d:/会计组/9月UV转印贴.xlsx', '文件名', True]
file_info = ['d:/会计组/彩印通2025年9月(数码).xlsx', '文件名', True]
file_info = ['d:/会计组/9月名片.xlsx', '文件名', True]
file_info = ['d:/会计组/9月不干胶.xlsx', '文件名', True]
file_path, target_index, is_horizontal = file_info
parse_finance_data(file_path, target_index, is_horizontal)