# coding:utf8 import re from typing import Dict, Any from BiddingKG.dl.template_extract.abstract_template import AbstractTemplate, table2list import os import json class OrderInfoTemplate(AbstractTemplate): """ 订单信息提取模板(适配采购订单表格场景) 适配包含采购单位、成交供应商及商品明细的表格提取 """ def __init__(self): super().__init__( template_id=os.path.abspath(__file__), priority=3 ) def check_call_timing(self, preprocessed_data: Dict[str, Any]) -> bool: """检查是否符合模板调用条件""" if "表格列表" not in preprocessed_data: return False tables = preprocessed_data["表格列表"] if len(tables) < 1: return False try: # 获取第一个表格数据 datas = tables[0] if len(datas) < 8: # 最少需要包含订单信息和商品表头行 print("表格行数不足") return False # 验证订单信息区域表头(前6行) order_header_patterns = [ ['订单信息', '(采购单位|采购人单位)'], # 第0行 ['订单信息', '(采购人|采购联系人)'], # 第1行 ['订单信息', '(成交供应商|中标供应商)'], # 第2行 ['订单信息', '(成交金额|中标金额)'], # 第3行 ['订单信息', '运费金额'], # 第4行(可选,允许不匹配) ['订单信息', '(成交时间|中标时间)'] # 第5行 ] # 验证前5行核心订单信息 for i in range(5): if len(datas[i]) < 2: print(f"第{i}行列数不足") return False # 验证第一列是否为"订单信息" if not re.fullmatch(order_header_patterns[i][0], datas[i][0]): print(f"第{i}行第一列不匹配: {datas[i][0]}") return False # 验证第二列是否符合对应模式 if not re.fullmatch(order_header_patterns[i][1], datas[i][1]): print(f"第{i}行第二列不匹配: {datas[i][1]}") return False # 验证商品明细区域表头(第6-7行) if not (re.fullmatch('订单明细', datas[6][0]) and re.fullmatch('商品名称', datas[7][0])): print("商品明细表头不匹配") return False if not (re.fullmatch('品牌', datas[7][1]) and re.fullmatch('数量', datas[7][2]) and re.fullmatch('单价', datas[7][3])): print("商品属性表头不匹配") return False self.header = datas self.data_start = 8 # 商品数据从第8行开始 return True except Exception as e: print(f"表格验证错误: {e}") return False def extract(self, preprocessed_data: Dict[str, Any]) -> Dict[str, Any]: """提取表格中的各类信息""" extract_result = { "招标信息": [], "中标信息": [], "产品信息": [] } # 提取招标信息(采购单位即招标人) purchaser = self.header[0][2] if len(self.header[0]) > 2 else "" contact_person = self.header[1][2] if len(self.header[1]) > 2 else "" if purchaser: extract_result["招标信息"].append({ "招标人": purchaser, "联系人": contact_person }) # 提取中标信息 supplier = self.header[2][2] if len(self.header[2]) > 2 else "" deal_price = self.header[3][2] if len(self.header[3]) > 2 else "" if supplier: # 补充金额单位(如果缺失) if deal_price and not re.search('[万亿美欧日]?元', deal_price): deal_price += "元" extract_result["中标信息"].append({ "中标人": supplier, "中标价": deal_price }) # 提取产品信息 for row in self.header[self.data_start:]: if len(row) < 4: continue # 跳过列数不足的行 product = row[0] brand = row[1] quantity = row[2] unit_price = row[3] # 补充单价单位 if unit_price and not re.search('[万亿美欧日]?元', unit_price): unit_price += "元" extract_result["产品信息"].append({ "产品": product, "品牌": brand, "数量": quantity, "单价": unit_price }) # 过滤空列表和空值字段 filtered_result = {} for key, value in extract_result.items(): if isinstance(value, list): if value: # 过滤列表中每个字典的空值 filtered_list = [] for item in value: filtered_item = {k: v for k, v in item.items() if v} if filtered_item: filtered_list.append(filtered_item) if filtered_list: filtered_result[key] = filtered_list else: if value: filtered_result[key] = value return filtered_result # 使用示例 if __name__ == "__main__": # 读取数据文件 with open("data.json", "r", encoding="utf8") as f: preprocessed_data = json.load(f) # 初始化模板并提取信息 template = OrderInfoTemplate() if template.check_call_timing(preprocessed_data): result = template.extract(preprocessed_data) print("提取结果:", json.dumps(result, ensure_ascii=False, indent=2)) else: print("当前数据不满足模板调用条件")