# coding:utf8 import re from typing import Dict, Any from BiddingKG.dl.template_extract.abstract_template import AbstractTemplate, table2list import os import json class CustomOrderTemplate(AbstractTemplate): """ 定制订单信息提取模板(适配data.json表格结构) 适配包含采购单位、采购人、成交供应商及商品明细的表格提取 """ def __init__(self): super().__init__( template_id=os.path.abspath(__file__), priority=3 ) def check_call_timing(self, preprocessed_data: Dict[str, Any]) -> bool: """检查是否符合模板调用条件""" if "表格列表" not in preprocessed_data: return False tables = preprocessed_data["表格列表"] if len(tables) < 1: return False try: # 获取第一个表格数据 datas = tables[0] if len(datas) < 8: # 最少需要包含订单信息和商品表头行 print("表格行数不足") return False # 验证订单信息区域表头(前5行) order_header_patterns = [ ['', '(采购单位|采购人单位)'], # 第0行 ['', '(采购人|采购联系人)'], # 第1行 ['', '(成交供应商|中标供应商|供应商)'], # 第2行 ['', '(成交金额|中标金额|金额)'], # 第3行 ['', '(成交时间|中标时间|时间)'] # 第4行 ] # 验证前4行核心订单信息(第0-3行) for i in range(4): if len(datas[i]) < 2: print(f"第{i}行列数不足") return False # 验证第一列是否为空(或匹配空字符串) if not re.fullmatch(order_header_patterns[i][0], datas[i][0]): print(f"第{i}行第一列不匹配: {datas[i][0]}") return False # 验证第二列是否符合对应模式 if not re.fullmatch(order_header_patterns[i][1], datas[i][1]): print(f"第{i}行第二列不匹配: {datas[i][1]}") return False # 验证第5行为空行 if not all(not cell.strip() for cell in datas[5]): print("第5行不是空行") return False # 验证商品明细区域表头(第6-7行) if not re.fullmatch('', datas[6][0]): print("第6行第一列不匹配") return False if not (re.fullmatch('品牌/型号', datas[6][1]) and re.fullmatch('数量', datas[6][2]) and re.fullmatch('单价', datas[6][3])): print("商品属性表头不匹配") return False # if not re.fullmatch('(纯水仪耗材|产品名称|商品名称)', datas[7][0]): # print("商品名称表头不匹配") # return False self.header = datas self.data_start = 7 # 商品数据从第7行开始 return True except Exception as e: print(f"表格验证错误: {e}") return False def extract(self, preprocessed_data: Dict[str, Any]) -> Dict[str, Any]: """提取表格中的各类信息""" extract_result = { "招标信息": [], "中标信息": [], "候选人信息": [], "产品信息": [] } # 提取招标信息(采购单位即招标人) purchaser = self.header[0][2] if len(self.header[0]) > 2 else "" contact_person = self.header[1][2] if len(self.header[1]) > 2 else "" if purchaser: # 招标信息必须包含招标人 extract_result["招标信息"].append({ "招标人": purchaser, "联系人": contact_person }) # 提取中标信息 supplier = self.header[2][2] if len(self.header[2]) > 2 else "" deal_price = self.header[3][2] if len(self.header[3]) > 2 else "" if supplier: # 中标信息必须包含中标人 # 补充金额单位 if deal_price and not re.search('[万亿美欧日]?元', deal_price): deal_price += "元" extract_result["中标信息"].append({ "中标人": supplier, "中标价": deal_price }) # 提取产品信息(候选人信息无,不提取) product_row = self.header[self.data_start] if len(product_row) >= 4: product = product_row[0] brand_model = product_row[1] # 分离品牌和型号 brand = "" spec = "" if '/' in brand_model: brand, spec = brand_model.split('/', 1) else: brand = brand_model # 若没有型号,全部视为品牌 quantity = product_row[2] unit_price = product_row[3] # 补充单价单位 if unit_price and not re.search('[万亿美欧日]?元', unit_price): unit_price += "元" extract_result["产品信息"].append({ "产品": product, "品牌": brand, "规格": spec, "数量": quantity, "单价": unit_price }) # 过滤空列表和空值字段 filtered_result = {} for key, value in extract_result.items(): if isinstance(value, list): if value: # 过滤列表中每个字典的空值 filtered_list = [] for item in value: filtered_item = {k: v for k, v in item.items() if v} if filtered_item: filtered_list.append(filtered_item) if filtered_list: filtered_result[key] = filtered_list else: if value: filtered_result[key] = value return filtered_result # 使用示例 if __name__ == "__main__": # 读取数据文件 with open("data.json", "r", encoding="utf8") as f: preprocessed_data = json.load(f) # 初始化模板并提取信息 template = CustomOrderTemplate() if template.check_call_timing(preprocessed_data): result = template.extract(preprocessed_data) print("提取结果:", json.dumps(result, ensure_ascii=False, indent=2)) else: print("当前数据不满足模板调用条件")