first commit

This commit is contained in:
2026-06-25 17:41:06 +08:00
commit b2f23a933b
370 changed files with 30526 additions and 0 deletions
+34
View File
@@ -0,0 +1,34 @@
# excel_utils.py
import pandas as pd
from typing import Dict, List, Tuple
def get_sheets_and_headers(
file_path: str,
header: int = 0,
skiprows: int = 0,
**kwargs
) -> Dict[str, List[str]]:
"""
读取 Excel 文件,返回每个 sheet 的名称及其表头(列名)。
参数:
file_path (str): Excel 文件路径(支持 .xlsx / .xls
header (int): 表头所在行索引(默认 0,即第一行)
skiprows (int): 跳过的行数(在表头之前)
**kwargs: 透传给 pd.read_excel 的其他参数(如 engine 等)
返回:
Dict[str, List[str]]: {sheet_name: [col1, col2, ...]}
"""
# 使用 nrows=0 只读取表头,不加载数据,性能高
all_sheets = pd.read_excel(
file_path,
sheet_name=None, # 读取所有 sheet
header=header,
skiprows=skiprows,
nrows=0, # ⚡ 关键:只读表头
**kwargs
)
return {name: df.columns.tolist() for name, df in all_sheets.items()}