Initial commit: Auto-Finance 财务报销自动化系统

This commit is contained in:
wandering
2026-05-25 11:18:12 +08:00
commit f754386695
25 changed files with 3040 additions and 0 deletions

256
app/extractor.py Normal file
View File

@@ -0,0 +1,256 @@
"""
PDF 发票信息提取
从 PDF 发票文件中提取关键字段,输出为标准化的发票数据列表。
对外接口:
extract_invoices(directory) -> list[dict] 扫描目录下所有 PDF 并提取
save_csv(invoices, path) 保存为 CSV
save_markdown(invoices, path) 保存为 Markdown 汇总
"""
import csv
import re
from pathlib import Path
from . import get_logger
log = get_logger("extractor")
CSV_COLUMNS = [
"序号", "发票号码", "开票日期", "项目名称", "规格型号",
"价税合计", "销售方名称", "人员姓名", "刷卡日期",
"公务卡号", "刷卡金额", "备注", "工号",
]
# ------------------------------------------------------------------
# PDF 文件发现与文本提取
# ------------------------------------------------------------------
def find_pdf_files(directory: str = ".") -> list[Path]:
"""查找目录下所有 PDF 文件(非递归)"""
pdf_dir = Path(directory)
if not pdf_dir.exists():
return []
return sorted(pdf_dir.glob("*.pdf"))
def extract_text_from_pdf(filepath: Path) -> str:
"""从单个 PDF 中提取全部文本"""
try:
import pdfplumber
except ImportError:
raise ImportError("缺少 pdfplumber请执行: pip install pdfplumber")
try:
parts = []
with pdfplumber.open(filepath) as pdf:
for page in pdf.pages:
text = page.extract_text()
if text:
parts.append(text)
return "\n".join(parts)
except Exception as e:
log.error(f"无法读取 {filepath.name}: {e}")
return ""
# ------------------------------------------------------------------
# 字段解析
# ------------------------------------------------------------------
def _first(regexes: list[str], text: str) -> str | None:
"""尝试多个正则,返回第一个匹配组的文本"""
for pattern in regexes:
m = re.search(pattern, text)
if m:
return m.group(1).strip()
return None
def _parse_line_item(line: str) -> dict | None:
"""解析单行明细(*分类*具体名称 格式)"""
m = re.match(r"\*([^*]+)\*\s*(.+)", line)
if m:
return {
"项目名称": f"*{m.group(1).strip()}*{m.group(2).strip()}",
"规格型号": m.group(2).strip(),
}
return None
def _extract_line_items(text: str) -> list[dict]:
"""从发票文本中提取所有明细行"""
items = []
skip_keywords = ["项目名称", "合 计", "价税合计", "备注", "开票人"]
for line in text.split("\n"):
line = line.strip()
if not line:
continue
if any(kw in line for kw in skip_keywords):
continue
if "*" in line:
item = _parse_line_item(line)
if item:
items.append(item)
return items
def _format_date(date_raw: str) -> str:
"""将「2026年5月18日」转为「2026/5/18」"""
m = re.match(r"(\d{4})年(\d{1,2})月(\d{1,2})日", date_raw)
if m:
return f"{m.group(1)}/{m.group(2)}/{m.group(3)}"
return date_raw
def parse_invoice(text: str) -> dict:
"""从发票文本中提取关键字段,返回 dict
返回字段:
发票号码, 开票日期, 销售方名称, 价税合计, _items (明细列表)
其他字段(人员姓名等)留空,后续由 OCR 步骤填充
"""
invoice: dict[str, str] = {}
invoice["发票号码"] = _first([r"发票号码[:]?\s*(\d+)"], text) or ""
date_raw = _first([r"开票日期[:]?\s*(\d{4}\d{1,2}月\d{1,2}日)"], text) or ""
invoice["开票日期"] = _format_date(date_raw) if date_raw else ""
invoice["销售方名称"] = _first(
[
r"\s*售?\s*方?\s*名称[:]?\s*(.+?)(?:\n|$)",
r"\s*名称[:]?\s*(.+?)(?:\n|$)",
],
text,
) or ""
invoice["价税合计"] = _first(
[r"价税合计.*?(小写)[¥¥]?\s*(\d+\.?\d*)"], text
) or ""
invoice["_items"] = _extract_line_items(text)
# 以下字段无法从 PDF 提取,留空由 OCR 步骤填充
for key in ("项目名称", "规格型号", "人员姓名", "刷卡日期",
"公务卡号", "刷卡金额", "备注", "工号"):
if key not in invoice:
invoice[key] = ""
return invoice
# ------------------------------------------------------------------
# CSV / Markdown 输出
# ------------------------------------------------------------------
def save_csv(invoices: list[dict], output_path: str | Path = "invoice_summary.csv"):
"""将发票列表保存为 CSV"""
csv_path = Path(output_path)
with open(csv_path, "w", encoding="utf-8", newline="") as f:
writer = csv.writer(f)
writer.writerow(CSV_COLUMNS)
for idx, inv in enumerate(invoices, 1):
items = inv.get("_items", [])
first_item = items[0] if items else {}
writer.writerow([
idx,
inv.get("发票号码", ""),
inv.get("开票日期", ""),
first_item.get("项目名称", inv.get("项目名称", "")),
first_item.get("规格型号", inv.get("规格型号", "")),
inv.get("价税合计", ""),
inv.get("销售方名称", ""),
inv.get("人员姓名", ""),
inv.get("刷卡日期", ""),
inv.get("公务卡号", ""),
inv.get("刷卡金额", ""),
inv.get("备注", ""),
inv.get("工号", ""),
])
log.info(f"CSV 已保存: {csv_path.name}")
def save_markdown(invoices: list[dict], output_path: str | Path = "invoice_summary.md"):
"""将发票列表保存为 Markdown 汇总表"""
md_path = Path(output_path)
lines = [
"# 发票信息汇总表",
"",
"| 序号 | 发票号码 | 开票日期 | 项目名称 | 规格型号 | 价税合计 | 销售方名称 |",
"|------|---------|---------|---------|---------|---------|-----------|",
]
total = 0.0
for idx, inv in enumerate(invoices, 1):
amount = 0.0
try:
amount = float(inv.get("价税合计", "0"))
except (ValueError, TypeError):
pass
total += amount
items = inv.get("_items", [])
first_item = items[0] if items else {}
project = first_item.get("项目名称", inv.get("项目名称", "-"))
spec = first_item.get("规格型号", inv.get("规格型号", "-"))
lines.append(
f"| {idx} "
f"| {inv.get('发票号码', '')} "
f"| {inv.get('开票日期', '')} "
f"| {project} | {spec} "
f"| ¥{amount:,.2f} "
f"| {inv.get('销售方名称', '')} |"
)
lines.append("")
lines.append(f"**总计: ¥{total:,.2f}**")
lines.append("")
with open(md_path, "w", encoding="utf-8") as f:
f.write("\n".join(lines))
log.info(f"Markdown 已保存: {md_path.name}")
# ------------------------------------------------------------------
# 主入口
# ------------------------------------------------------------------
def extract_invoices(directory: str = ".") -> list[dict]:
"""扫描目录下所有 PDF提取发票信息并返回列表"""
target_dir = Path(directory).absolute()
pdf_files = find_pdf_files(directory)
if not pdf_files:
log.warning("未找到 PDF 文件")
return []
log.info(f"发现 {len(pdf_files)} 个 PDF 文件")
all_invoices = []
for pdf_path in pdf_files:
text = extract_text_from_pdf(pdf_path)
if text:
invoice = parse_invoice(text)
if invoice:
all_invoices.append(invoice)
else:
log.warning(f"未能解析: {pdf_path.name}")
else:
log.warning(f"未能提取文本: {pdf_path.name}")
if all_invoices:
log.info(f"共处理 {len(all_invoices)} 张发票")
else:
log.warning("未成功解析任何发票")
return all_invoices