# coding:utf8 import re from typing import Dict, Any from BiddingKG.dl.template_extract.abstract_template import AbstractTemplate import os import json class CustomCandidateBidTemplate(AbstractTemplate): """ 候选人信息提取模板 适配data.json中行列均为表头的表格(行表头:名次/候选供应商等;列表头:第一名/第二名/第三名) """ def __init__(self): super().__init__( template_id=os.path.abspath(__file__), priority=3 ) def check_call_timing(self, preprocessed_data: Dict[str, Any]) -> bool: """检查是否符合模板调用条件(适配行列双表头结构)""" if "表格列表" not in preprocessed_data: # 例子:727039445 return False tables = preprocessed_data["表格列表"] if len(tables) < 1: return False try: # 目标表格为表格列表第一个元素 self.table = tables[0] # 表格需至少包含行表头(6行)和列表头(至少1列数据列:第一名) if len(self.table) < 6 or len(self.table[0]) < 2: print(f"表格行列数异常:行{len(self.table)}列{len(self.table[0])},预期至少6行2列") return False # 提取行表头(每行第一个单元格)和列表头(第一行所有单元格) row_headers = [row[0].strip() for row in self.table] col_headers = [col.strip() for col in self.table[0]] # 行表头正则匹配规则(泛化表达,fullmatch) row_header_patterns = [ r'名次|排名|排序|序号', # 行1:名次 r'候选(成交|中标|投标)?(人|单位|供应商)(名称)?', # 行2:候选成交供应商 r'(响应报价|投标报?价|报价|参选价)((不含税))?((万?元))?', # 行3:响应报价(元) r'(服务期|工期|交货期|服务时间)(\(日历天\))?', # 行4:服务期(日历天) r'项目负责人|联系人|负责人|联络人', # 行5:项目负责人 r'备注|说明|注释|附注' # 行6:备注 ] # 验证行表头(逐行匹配) for i, (header, pattern) in enumerate(zip(row_headers, row_header_patterns)): if not re.fullmatch(pattern, header): print(f'行表头第{i+1}行未匹配:{pattern} vs {header}') return False # 验证列表头(至少包含"名次"+"第一名/第二名/第三名"等) col_header_pattern = r'名次|第[一二三四五六七八九]+名|第\d+名' for col_header in col_headers: if not re.fullmatch(col_header_pattern, col_header): print(f'列表头列未匹配:{col_header_pattern} vs {col_header}') return False self.row_headers = row_headers self.col_headers = col_headers return True except Exception as e: print(f"表格验证错误: {e}") return False def extract(self, preprocessed_data: Dict[str, Any]) -> Dict[str, Any]: """提取表格中的相关信息(适配行列双表头结构)""" # 初始化返回结构 extract_result = { "项目编号": "", "项目名称": "", "招标信息": [], "中标信息": [], "候选人信息": [], "产品信息": [] } # 行索引映射(行表头对应的数据类型) row_idx_map = { "rank": 0, # 名次行 "candidate": 1, # 候选供应商行 "price": 2, # 响应报价行 "service_time": 3, # 服务期行 "contact": 4 # 项目负责人行 } # 遍历列表头(跳过第一个"名次"列,处理第一名/第二名等列) for col_idx in range(1, len(self.col_headers)): # 提取单列候选人数据(按行索引取对应值) rank = self.table[row_idx_map["rank"]][col_idx].strip() candidate = self.table[row_idx_map["candidate"]][col_idx].strip() price = self.table[row_idx_map["price"]][col_idx].strip() service_time = self.table[row_idx_map["service_time"]][col_idx].strip() contact = self.table[row_idx_map["contact"]][col_idx].strip() # 核心规则:候选人信息必须包含候选人及排名,否则跳过 if not candidate or not rank: continue # 补充金额单位(从行表头提取单位,内容无则补充) price_header = self.row_headers[row_idx_map["price"]] unit_match = re.search(r'(元)|元|万?元', price_header) if unit_match and not re.search(r'(元)|元|万?元', price): unit = unit_match.group(0).replace("(", "").replace(")", "") # 清理括号 # 保留不含税说明并补充单位 if "(不含税)" in price: price = price.replace("(不含税)", "") + f"(不含税){unit}" else: price += unit # 补充服务期单位 if service_time: service_time_pattern = re.search(r'日|天|年|月|周|星期', self.row_headers[row_idx_map["service_time"]]) if service_time_pattern and not re.search(r'日|天|年|月|周|星期', service_time): service_time += service_time_pattern.group(0) # 整理候选人信息 candidate_info = { "标的": "", "标包": "", "包号": "", "候选人": candidate, "投标价": price, "排名": rank, "联系人": contact, "电话": "", "服务时间": service_time, "地址": "" } # 过滤空值字段 candidate_info = {k: v for k, v in candidate_info.items() if v} extract_result["候选人信息"].append(candidate_info) # 最终结果过滤:空值字段/空列表不返回 final_result = {} for key, value in extract_result.items(): if isinstance(value, list): if len(value) > 0: final_result[key] = value else: if value.strip(): final_result[key] = value # 严格遵循规则: # 1. 有排名的候选人信息,不返回中标信息 # 2. 招标信息无招标人/预算,不返回 # 3. 产品信息无产品,不返回 return final_result # 使用示例 if __name__ == "__main__": # 读取data.json数据 with open("data.json", "r", encoding="utf8") as f: preprocessed_data = json.load(f) # 初始化模板并执行提取 template = CustomCandidateBidTemplate() if template.check_call_timing(preprocessed_data): result = template.extract(preprocessed_data) print("提取结果:", json.dumps(result, ensure_ascii=False, indent=2)) else: print("当前数据不满足模板调用条件")