# coding:utf8 import re from typing import Dict, Any from BiddingKG.dl.template_extract.abstract_template import AbstractTemplate, table2list import os import json class BudgetTemplate(AbstractTemplate): """ 招标信息提取模板(适配多列表格场景) 适配包含"标的名称、标包名称、标包金额、最高限价"列的招标信息表格提取 """ def __init__(self): super().__init__( template_id=os.path.abspath(__file__), priority=5 ) def check_call_timing(self, preprocessed_data: Dict[str, Any]) -> bool: """检查是否符合模板调用条件,验证表头包含标的名称、标包名称、标包金额、最高限价(顺序不固定)""" if "表格列表" not in preprocessed_data: return False tables = preprocessed_data["表格列表"] if len(tables) < 1: return False try: datas = tables[0] # 确保表格有表头行 if len(datas) < 1: return False header = datas[0] if len(datas[0]) < 6: print(f"表头数量异常;实际{len(datas[0])}列,要求最少6列") return False # 定义需要匹配的表头正则(泛化表达) required_headers = [ '序号', # 序号(精确匹配) '标的(名称)?', # 标的名称(精确匹配) '标[包段](名称)?', # 标包名称(精确匹配) '(最高|上限)(投标)?限价(([万亿美欧日]?元(/\w)*))?' # 最高限价(泛化匹配) ] # 记录每个表头的匹配列索引 self.header_indices = {} matched_count = 0 # 遍历表头列,检查是否包含所有必需项(顺序不固定) for col_idx, col_name in enumerate(header): for req_idx, pattern in enumerate(required_headers): if re.fullmatch(pattern, col_name): # 记录匹配到的列索引(使用需求名称作为key) req_key = { 0: "序号", 1: "标的名称", 2: "标包名称", 3: "最高限价" }[req_idx] if req_key not in self.header_indices: self.header_indices[req_key] = col_idx matched_count += 1 break # 一个列只匹配一个需求 # 必须匹配到所有5个必需表头 if matched_count != len(required_headers): print(f"未匹配到所有必需表头,已匹配{matched_count}") return False # 确保表格有数据行 if len(datas) < 2: return False self.header = header self.data = datas[1:] return True except Exception as e: print(f"表格验证错误: {e}") return False def extract(self, preprocessed_data: Dict[str, Any]) -> Dict[str, Any]: """提取表格中的招标信息""" extract_result = {"招标信息": []} for row in self.data: # 跳过长度异常的行 if len(row) != len(self.header): continue # 根据表头索引提取对应数据 subject = row[self.header_indices["标的名称"]] package = row[self.header_indices["标包名称"]] # budget = row[self.header_indices["标包金额"]] limited_price = row[self.header_indices["最高限价"]] # 处理最高限价单位 limit_header = self.header[self.header_indices["最高限价"]] if re.search('[万亿美欧日]?元', limit_header) and not re.search('[万亿美欧日]?元', limited_price): limited_price += re.search('[万亿美欧日]?元', limit_header).group(0) # 整理信息 extract_result["招标信息"].append({ "标的": subject, "标包": package, "限价": limited_price, }) return extract_result # 使用示例 if __name__ == "__main__": # 读取数据文件 with open("data.json", "r", encoding="utf8") as f: preprocessed_data = json.load(f) # 初始化模板并提取信息 template = BudgetTemplate() if template.check_call_timing(preprocessed_data): result = template.extract(preprocessed_data) print("提取结果:", json.dumps(result, ensure_ascii=False, indent=2)) else: print("当前数据不满足模板调用条件")