# coding:utf8 import re from typing import Dict, Any from BiddingKG.dl.template_extract.abstract_template import AbstractTemplate, table2list import os import json class OceanUniversityTemplate(AbstractTemplate): """ 中国海洋大学采购信息提取模板 适配包含项目信息、采购单位、成交信息及采购清单的表格结构 """ def __init__(self): super().__init__( template_id=os.path.abspath(__file__), priority=3 ) def check_call_timing(self, preprocessed_data: Dict[str, Any]) -> bool: """检查是否符合模板调用条件""" # 站源:中国海洋大学,例子:672542687 if "表格列表" not in preprocessed_data: return False tables = preprocessed_data["表格列表"] if len(tables) < 1: return False try: main_table = tables[0] # 验证表格基本结构(至少包含13行关键信息行) if len(main_table) < 13: print(f"表格行数不足,期望至少13行,实际{len(main_table)}行") return False # 表头行正则匹配(泛化表达) header_patterns = [ # 第0行:项目名称行 (0, 0, '项目信息'), (0, 1, '(项目|工程)名称'), # 第2行:项目编号行 (2, 0, '项目信息'), (2, 1, '项目编号'), # 第3行:采购单位行 (3, 0, '采购单位信息'), (3, 1, '采购单位名称'), # 第4行:采购单位地址行 (4, 0, '采购单位信息'), (4, 1, '采购单位地址'), # 第5行:联系人行 (5, 0, '采购单位信息'), (5, 1, '联系人'), # 第6行:联系方式行 (6, 0, '采购单位信息'), (6, 1, '联系方式'), # 第9行:成交供应商行 (9, 0, '成交信息'), (9, 1, '成交供应商|中标人'), # 第8行:中标价行 (8, 0, '成交信息'), (8, 1, '中标价|成交价'), # 第12行:采购清单表头行 (12, 0, '序号'), (12, 1, '名称'), (12, 2, '规格型号|规格|型号'), (12, 3, '数量') ] # 验证关键表头位置和内容 for row_idx, col_idx, pattern in header_patterns: if row_idx >= len(main_table) or col_idx >= len(main_table[row_idx]): print(f"表格结构异常,行{row_idx}列{col_idx}不存在") return False cell_value = main_table[row_idx][col_idx] if not re.fullmatch(pattern, cell_value, re.IGNORECASE): print(f"表头不匹配:行{row_idx}列{col_idx}期望[{pattern}],实际[{cell_value}]") return False self.main_table = main_table return True except Exception as e: print(f"表格验证错误: {e}") return False def extract(self, preprocessed_data: Dict[str, Any]) -> Dict[str, Any]: """提取表格中的各类信息""" result = { "项目编号": "", "项目名称": "", "招标信息": [], "中标信息": [], "候选人信息": [], # 数据中无候选人信息 "产品信息": [] } main_table = self.main_table # 提取项目基本信息 # 项目名称(第0行第2列) if len(main_table) > 0 and len(main_table[0]) > 2: result["项目名称"] = main_table[0][2].strip() # 项目编号(第2行第2列) if len(main_table) > 2 and len(main_table[2]) > 2: result["项目编号"] = main_table[2][2].strip() # 提取招标信息(采购单位即招标人) tender_info = { "招标人": "", "地址": "", "联系人": "", "电话": "" } # 招标人名称(第3行第2列) if len(main_table) > 3 and len(main_table[3]) > 2: tender_info["招标人"] = main_table[3][2].strip() # 招标人地址(第4行第2列) if len(main_table) > 4 and len(main_table[4]) > 2: tender_info["地址"] = main_table[4][2].strip() # 联系人(第5行第2列) if len(main_table) > 5 and len(main_table[5]) > 2: tender_info["联系人"] = main_table[5][2].strip() # 联系电话(第6行第2列,从"电话:xxx"中提取) if len(main_table) > 6 and len(main_table[6]) > 2: phone_match = re.search(r'电话:(\d+\*{0,4}\d+)', main_table[6][2]) if phone_match: tender_info["电话"] = phone_match.group(1) # 验证招标信息有效性(包含招标人) if tender_info["招标人"]: result["招标信息"].append(tender_info) # 提取中标信息 win_info = { "中标人": "", "中标价": "" } # 中标人(第9行第2列) if len(main_table) > 9 and len(main_table[9]) > 2: win_info["中标人"] = main_table[9][2].strip() # 中标价(第8行第2列,确保有单位) if len(main_table) > 8 and len(main_table[8]) > 2: price = main_table[8][2].strip() if not re.search(r'[万亿]?[元角分]', price): # 尝试从表头推断单位(如果内容无单位) if re.search(r'[万亿]?[元角分]', main_table[8][1]): unit = re.search(r'[万亿]?[元角分]', main_table[8][1]).group() price += unit win_info["中标价"] = price # 验证中标信息有效性(包含中标人) if win_info["中标人"]: result["中标信息"].append(win_info) # 提取产品信息(从第13行开始的采购清单) if len(main_table) > 12: product_start_row = 13 for row in main_table[product_start_row:]: if len(row) < 4: continue # 跳过不完整行 # 解析产品名称和品牌(从名称中提取品牌) product_name = row[1].strip() brand = "" # 尝试从规格型号中提取品牌(格式:品牌/型号) spec = row[2].strip() spec_parts = re.split(r'/', spec, 1) if len(spec_parts) == 2: brand = spec_parts[0].strip() spec = spec_parts[1].strip() product_info = { "产品": product_name, "品牌": brand, "规格": spec, "数量": row[3].strip() } result["产品信息"].append(product_info) # 移除空列表和空值字段 if not result["招标信息"]: del result["招标信息"] if not result["中标信息"]: del result["中标信息"] if not result["候选人信息"]: del result["候选人信息"] if not result["产品信息"]: del result["产品信息"] for key in list(result.keys()): if result[key] == "": del result[key] return result # 使用示例 if __name__ == "__main__": with open("data.json", "r", encoding="utf8") as f: preprocessed_data = json.load(f) template = OceanUniversityTemplate() if template.check_call_timing(preprocessed_data): result = template.extract(preprocessed_data) print("提取结果:", json.dumps(result, ensure_ascii=False, indent=2)) else: print("当前数据不满足模板调用条件")