Initial commit: Auto-Finance 财务报销自动化系统
This commit is contained in:
256
app/extractor.py
Normal file
256
app/extractor.py
Normal file
@@ -0,0 +1,256 @@
|
||||
"""
|
||||
PDF 发票信息提取
|
||||
|
||||
从 PDF 发票文件中提取关键字段,输出为标准化的发票数据列表。
|
||||
|
||||
对外接口:
|
||||
extract_invoices(directory) -> list[dict] 扫描目录下所有 PDF 并提取
|
||||
save_csv(invoices, path) 保存为 CSV
|
||||
save_markdown(invoices, path) 保存为 Markdown 汇总
|
||||
"""
|
||||
|
||||
import csv
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
from . import get_logger
|
||||
|
||||
log = get_logger("extractor")
|
||||
|
||||
CSV_COLUMNS = [
|
||||
"序号", "发票号码", "开票日期", "项目名称", "规格型号",
|
||||
"价税合计", "销售方名称", "人员姓名", "刷卡日期",
|
||||
"公务卡号", "刷卡金额", "备注", "工号",
|
||||
]
|
||||
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# PDF 文件发现与文本提取
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def find_pdf_files(directory: str = ".") -> list[Path]:
|
||||
"""查找目录下所有 PDF 文件(非递归)"""
|
||||
pdf_dir = Path(directory)
|
||||
if not pdf_dir.exists():
|
||||
return []
|
||||
return sorted(pdf_dir.glob("*.pdf"))
|
||||
|
||||
|
||||
def extract_text_from_pdf(filepath: Path) -> str:
|
||||
"""从单个 PDF 中提取全部文本"""
|
||||
try:
|
||||
import pdfplumber
|
||||
except ImportError:
|
||||
raise ImportError("缺少 pdfplumber,请执行: pip install pdfplumber")
|
||||
|
||||
try:
|
||||
parts = []
|
||||
with pdfplumber.open(filepath) as pdf:
|
||||
for page in pdf.pages:
|
||||
text = page.extract_text()
|
||||
if text:
|
||||
parts.append(text)
|
||||
return "\n".join(parts)
|
||||
except Exception as e:
|
||||
log.error(f"无法读取 {filepath.name}: {e}")
|
||||
return ""
|
||||
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# 字段解析
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def _first(regexes: list[str], text: str) -> str | None:
|
||||
"""尝试多个正则,返回第一个匹配组的文本"""
|
||||
for pattern in regexes:
|
||||
m = re.search(pattern, text)
|
||||
if m:
|
||||
return m.group(1).strip()
|
||||
return None
|
||||
|
||||
|
||||
def _parse_line_item(line: str) -> dict | None:
|
||||
"""解析单行明细(*分类*具体名称 格式)"""
|
||||
m = re.match(r"\*([^*]+)\*\s*(.+)", line)
|
||||
if m:
|
||||
return {
|
||||
"项目名称": f"*{m.group(1).strip()}*{m.group(2).strip()}",
|
||||
"规格型号": m.group(2).strip(),
|
||||
}
|
||||
return None
|
||||
|
||||
|
||||
def _extract_line_items(text: str) -> list[dict]:
|
||||
"""从发票文本中提取所有明细行"""
|
||||
items = []
|
||||
skip_keywords = ["项目名称", "合 计", "价税合计", "备注", "开票人"]
|
||||
|
||||
for line in text.split("\n"):
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
if any(kw in line for kw in skip_keywords):
|
||||
continue
|
||||
if "*" in line:
|
||||
item = _parse_line_item(line)
|
||||
if item:
|
||||
items.append(item)
|
||||
|
||||
return items
|
||||
|
||||
|
||||
def _format_date(date_raw: str) -> str:
|
||||
"""将「2026年5月18日」转为「2026/5/18」"""
|
||||
m = re.match(r"(\d{4})年(\d{1,2})月(\d{1,2})日", date_raw)
|
||||
if m:
|
||||
return f"{m.group(1)}/{m.group(2)}/{m.group(3)}"
|
||||
return date_raw
|
||||
|
||||
|
||||
def parse_invoice(text: str) -> dict:
|
||||
"""从发票文本中提取关键字段,返回 dict
|
||||
|
||||
返回字段:
|
||||
发票号码, 开票日期, 销售方名称, 价税合计, _items (明细列表)
|
||||
其他字段(人员姓名等)留空,后续由 OCR 步骤填充
|
||||
"""
|
||||
invoice: dict[str, str] = {}
|
||||
|
||||
invoice["发票号码"] = _first([r"发票号码[::]?\s*(\d+)"], text) or ""
|
||||
|
||||
date_raw = _first([r"开票日期[::]?\s*(\d{4}年\d{1,2}月\d{1,2}日)"], text) or ""
|
||||
invoice["开票日期"] = _format_date(date_raw) if date_raw else ""
|
||||
|
||||
invoice["销售方名称"] = _first(
|
||||
[
|
||||
r"销\s*售?\s*方?\s*名称[::]?\s*(.+?)(?:\n|$)",
|
||||
r"销\s*名称[::]?\s*(.+?)(?:\n|$)",
|
||||
],
|
||||
text,
|
||||
) or ""
|
||||
|
||||
invoice["价税合计"] = _first(
|
||||
[r"价税合计.*?(小写)[¥¥]?\s*(\d+\.?\d*)"], text
|
||||
) or ""
|
||||
|
||||
invoice["_items"] = _extract_line_items(text)
|
||||
|
||||
# 以下字段无法从 PDF 提取,留空由 OCR 步骤填充
|
||||
for key in ("项目名称", "规格型号", "人员姓名", "刷卡日期",
|
||||
"公务卡号", "刷卡金额", "备注", "工号"):
|
||||
if key not in invoice:
|
||||
invoice[key] = ""
|
||||
|
||||
return invoice
|
||||
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# CSV / Markdown 输出
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def save_csv(invoices: list[dict], output_path: str | Path = "invoice_summary.csv"):
|
||||
"""将发票列表保存为 CSV"""
|
||||
csv_path = Path(output_path)
|
||||
|
||||
with open(csv_path, "w", encoding="utf-8", newline="") as f:
|
||||
writer = csv.writer(f)
|
||||
writer.writerow(CSV_COLUMNS)
|
||||
|
||||
for idx, inv in enumerate(invoices, 1):
|
||||
items = inv.get("_items", [])
|
||||
first_item = items[0] if items else {}
|
||||
writer.writerow([
|
||||
idx,
|
||||
inv.get("发票号码", ""),
|
||||
inv.get("开票日期", ""),
|
||||
first_item.get("项目名称", inv.get("项目名称", "")),
|
||||
first_item.get("规格型号", inv.get("规格型号", "")),
|
||||
inv.get("价税合计", ""),
|
||||
inv.get("销售方名称", ""),
|
||||
inv.get("人员姓名", ""),
|
||||
inv.get("刷卡日期", ""),
|
||||
inv.get("公务卡号", ""),
|
||||
inv.get("刷卡金额", ""),
|
||||
inv.get("备注", ""),
|
||||
inv.get("工号", ""),
|
||||
])
|
||||
|
||||
log.info(f"CSV 已保存: {csv_path.name}")
|
||||
|
||||
|
||||
def save_markdown(invoices: list[dict], output_path: str | Path = "invoice_summary.md"):
|
||||
"""将发票列表保存为 Markdown 汇总表"""
|
||||
md_path = Path(output_path)
|
||||
lines = [
|
||||
"# 发票信息汇总表",
|
||||
"",
|
||||
"| 序号 | 发票号码 | 开票日期 | 项目名称 | 规格型号 | 价税合计 | 销售方名称 |",
|
||||
"|------|---------|---------|---------|---------|---------|-----------|",
|
||||
]
|
||||
|
||||
total = 0.0
|
||||
for idx, inv in enumerate(invoices, 1):
|
||||
amount = 0.0
|
||||
try:
|
||||
amount = float(inv.get("价税合计", "0"))
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
total += amount
|
||||
|
||||
items = inv.get("_items", [])
|
||||
first_item = items[0] if items else {}
|
||||
project = first_item.get("项目名称", inv.get("项目名称", "-"))
|
||||
spec = first_item.get("规格型号", inv.get("规格型号", "-"))
|
||||
|
||||
lines.append(
|
||||
f"| {idx} "
|
||||
f"| {inv.get('发票号码', '')} "
|
||||
f"| {inv.get('开票日期', '')} "
|
||||
f"| {project} | {spec} "
|
||||
f"| ¥{amount:,.2f} "
|
||||
f"| {inv.get('销售方名称', '')} |"
|
||||
)
|
||||
|
||||
lines.append("")
|
||||
lines.append(f"**总计: ¥{total:,.2f}**")
|
||||
lines.append("")
|
||||
|
||||
with open(md_path, "w", encoding="utf-8") as f:
|
||||
f.write("\n".join(lines))
|
||||
|
||||
log.info(f"Markdown 已保存: {md_path.name}")
|
||||
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# 主入口
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def extract_invoices(directory: str = ".") -> list[dict]:
|
||||
"""扫描目录下所有 PDF,提取发票信息并返回列表"""
|
||||
target_dir = Path(directory).absolute()
|
||||
|
||||
pdf_files = find_pdf_files(directory)
|
||||
if not pdf_files:
|
||||
log.warning("未找到 PDF 文件")
|
||||
return []
|
||||
|
||||
log.info(f"发现 {len(pdf_files)} 个 PDF 文件")
|
||||
|
||||
all_invoices = []
|
||||
for pdf_path in pdf_files:
|
||||
text = extract_text_from_pdf(pdf_path)
|
||||
if text:
|
||||
invoice = parse_invoice(text)
|
||||
if invoice:
|
||||
all_invoices.append(invoice)
|
||||
else:
|
||||
log.warning(f"未能解析: {pdf_path.name}")
|
||||
else:
|
||||
log.warning(f"未能提取文本: {pdf_path.name}")
|
||||
|
||||
if all_invoices:
|
||||
log.info(f"共处理 {len(all_invoices)} 张发票")
|
||||
else:
|
||||
log.warning("未成功解析任何发票")
|
||||
|
||||
return all_invoices
|
||||
Reference in New Issue
Block a user