# coding:utf8 import re from typing import Dict, Any from BiddingKG.dl.template_extract.abstract_template import AbstractTemplate import os import json class BidWinnerTemplate(AbstractTemplate): """ 中标信息提取模板 适配包含"序号、标的名称、标包名称、中标候选人排序、中标候选人名称、投标报价"列的中标信息表格提取 """ def __init__(self): super().__init__( template_id=os.path.abspath(__file__), priority=5 ) def check_call_timing(self, preprocessed_data: Dict[str, Any]) -> bool: """检查是否符合模板调用条件,验证表头包含指定字段(顺序不固定)""" if "表格列表" not in preprocessed_data: return False tables = preprocessed_data["表格列表"] if len(tables) < 1: return False try: datas = tables[0] # 确保表格有表头行 if len(datas) < 1: return False header = datas[0] # 检查表头列数是否足够 if len(datas[0]) < 6: print(f"表头数量异常;实际{len(datas[0])}列,要求最少6列") return False # 定义需要匹配的表头正则(泛化表达) required_headers = [ '序号', # 序号(精确匹配) '标的(名称)?', # 标的名称(泛化匹配) '标[包段](名称)?', # 标包名称(泛化匹配) '(中标|成交)人?候选人排序|推荐排序', # 中标候选人排序(精确匹配) '(中标|成交)候选人名称', # 中标候选人名称(精确匹配) '(投标|中标|成交|响应|最终)下浮率' # 下浮率(泛化匹配) ] # 记录每个表头的匹配列索引 self.header_indices = {} matched_count = 0 # 遍历表头列,检查是否包含所有必需项(顺序不固定) for col_idx, col_name in enumerate(header): for req_idx, pattern in enumerate(required_headers): if re.fullmatch(pattern, col_name): # 记录匹配到的列索引(使用需求名称作为key) req_key = { 0: "序号", 1: "标的名称", 2: "标包名称", 3: "中标候选人排序", 4: "中标候选人名称", 5: "下浮率" }[req_idx] if req_key not in self.header_indices: self.header_indices[req_key] = col_idx matched_count += 1 break # 一个列只匹配一个需求 # 必须匹配到所有必需表头 if matched_count != len(required_headers): print(f"未匹配到所有必需表头,已匹配{matched_count}/{len(required_headers)}") return False # 确保表格有数据行 if len(datas) < 2: print("表格没有数据行") return False self.header = header self.data = datas[1:] return True except Exception as e: print(f"表格验证错误: {e}") return False def extract(self, preprocessed_data: Dict[str, Any]) -> Dict[str, Any]: """提取表格中的中标信息""" extract_result = {"候选人信息": []} for row in self.data: # 跳过长度异常的行 if len(row) != len(self.header): continue # 根据表头索引提取对应数据 serial_number = row[self.header_indices["序号"]] subject = row[self.header_indices["标的名称"]] package = row[self.header_indices["标包名称"]] winner_rank = row[self.header_indices["中标候选人排序"]] winner_name = row[self.header_indices["中标候选人名称"]] bid_price = row[self.header_indices["下浮率"]] # 整理信息 extract_result["候选人信息"].append({ "标的": subject, "标包": package, "排名": winner_rank, "候选人": winner_name, "下浮率": bid_price, }) return extract_result # 使用示例 if __name__ == "__main__": # 读取数据文件 with open("data.json", "r", encoding="utf8") as f: preprocessed_data = json.load(f) # 初始化模板并提取信息 template = BidWinnerTemplate() if template.check_call_timing(preprocessed_data): result = template.extract(preprocessed_data) print("提取结果:", json.dumps(result, ensure_ascii=False, indent=2)) else: print("当前数据不满足模板调用条件")