| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169 |
- # coding:utf8
- import re
- from typing import Dict, Any
- from BiddingKG.dl.template_extract.abstract_template import AbstractTemplate
- import os
- import json
- class CustomCandidateBidTemplate(AbstractTemplate):
- """
- 候选人信息提取模板
- 适配data.json中行列均为表头的表格(行表头:名次/候选供应商等;列表头:第一名/第二名/第三名)
- """
- def __init__(self):
- super().__init__(
- template_id=os.path.abspath(__file__),
- priority=3
- )
- def check_call_timing(self, preprocessed_data: Dict[str, Any]) -> bool:
- """检查是否符合模板调用条件(适配行列双表头结构)"""
- if "表格列表" not in preprocessed_data: # 例子:727039445
- return False
- tables = preprocessed_data["表格列表"]
- if len(tables) < 1:
- return False
- try:
- # 目标表格为表格列表第一个元素
- self.table = tables[0]
- # 表格需至少包含行表头(6行)和列表头(至少1列数据列:第一名)
- if len(self.table) < 6 or len(self.table[0]) < 2:
- print(f"表格行列数异常:行{len(self.table)}列{len(self.table[0])},预期至少6行2列")
- return False
- # 提取行表头(每行第一个单元格)和列表头(第一行所有单元格)
- row_headers = [row[0].strip() for row in self.table]
- col_headers = [col.strip() for col in self.table[0]]
- # 行表头正则匹配规则(泛化表达,fullmatch)
- row_header_patterns = [
- r'名次|排名|排序|序号', # 行1:名次
- r'候选(成交|中标|投标)?(人|单位|供应商)(名称)?', # 行2:候选成交供应商
- r'(响应报价|投标报?价|报价|参选价)((不含税))?((万?元))?', # 行3:响应报价(元)
- r'(服务期|工期|交货期|服务时间)(\(日历天\))?', # 行4:服务期(日历天)
- r'项目负责人|联系人|负责人|联络人', # 行5:项目负责人
- r'备注|说明|注释|附注' # 行6:备注
- ]
- # 验证行表头(逐行匹配)
- for i, (header, pattern) in enumerate(zip(row_headers, row_header_patterns)):
- if not re.fullmatch(pattern, header):
- print(f'行表头第{i+1}行未匹配:{pattern} vs {header}')
- return False
- # 验证列表头(至少包含"名次"+"第一名/第二名/第三名"等)
- col_header_pattern = r'名次|第[一二三四五六七八九]+名|第\d+名'
- for col_header in col_headers:
- if not re.fullmatch(col_header_pattern, col_header):
- print(f'列表头列未匹配:{col_header_pattern} vs {col_header}')
- return False
- self.row_headers = row_headers
- self.col_headers = col_headers
- return True
- except Exception as e:
- print(f"表格验证错误: {e}")
- return False
- def extract(self, preprocessed_data: Dict[str, Any]) -> Dict[str, Any]:
- """提取表格中的相关信息(适配行列双表头结构)"""
- # 初始化返回结构
- extract_result = {
- "项目编号": "",
- "项目名称": "",
- "招标信息": [],
- "中标信息": [],
- "候选人信息": [],
- "产品信息": []
- }
- # 行索引映射(行表头对应的数据类型)
- row_idx_map = {
- "rank": 0, # 名次行
- "candidate": 1, # 候选供应商行
- "price": 2, # 响应报价行
- "service_time": 3, # 服务期行
- "contact": 4 # 项目负责人行
- }
- # 遍历列表头(跳过第一个"名次"列,处理第一名/第二名等列)
- for col_idx in range(1, len(self.col_headers)):
- # 提取单列候选人数据(按行索引取对应值)
- rank = self.table[row_idx_map["rank"]][col_idx].strip()
- candidate = self.table[row_idx_map["candidate"]][col_idx].strip()
- price = self.table[row_idx_map["price"]][col_idx].strip()
- service_time = self.table[row_idx_map["service_time"]][col_idx].strip()
- contact = self.table[row_idx_map["contact"]][col_idx].strip()
- # 核心规则:候选人信息必须包含候选人及排名,否则跳过
- if not candidate or not rank:
- continue
- # 补充金额单位(从行表头提取单位,内容无则补充)
- price_header = self.row_headers[row_idx_map["price"]]
- unit_match = re.search(r'(元)|元|万?元', price_header)
- if unit_match and not re.search(r'(元)|元|万?元', price):
- unit = unit_match.group(0).replace("(", "").replace(")", "") # 清理括号
- # 保留不含税说明并补充单位
- if "(不含税)" in price:
- price = price.replace("(不含税)", "") + f"(不含税){unit}"
- else:
- price += unit
- # 补充服务期单位
- if service_time:
- service_time_pattern = re.search(r'日|天|年|月|周|星期', self.row_headers[row_idx_map["service_time"]])
- if service_time_pattern and not re.search(r'日|天|年|月|周|星期', service_time):
- service_time += service_time_pattern.group(0)
- # 整理候选人信息
- candidate_info = {
- "标的": "",
- "标包": "",
- "包号": "",
- "候选人": candidate,
- "投标价": price,
- "排名": rank,
- "联系人": contact,
- "电话": "",
- "服务时间": service_time,
- "地址": ""
- }
- # 过滤空值字段
- candidate_info = {k: v for k, v in candidate_info.items() if v}
- extract_result["候选人信息"].append(candidate_info)
- # 最终结果过滤:空值字段/空列表不返回
- final_result = {}
- for key, value in extract_result.items():
- if isinstance(value, list):
- if len(value) > 0:
- final_result[key] = value
- else:
- if value.strip():
- final_result[key] = value
- # 严格遵循规则:
- # 1. 有排名的候选人信息,不返回中标信息
- # 2. 招标信息无招标人/预算,不返回
- # 3. 产品信息无产品,不返回
- return final_result
- # 使用示例
- if __name__ == "__main__":
- # 读取data.json数据
- with open("data.json", "r", encoding="utf8") as f:
- preprocessed_data = json.load(f)
- # 初始化模板并执行提取
- template = CustomCandidateBidTemplate()
- if template.check_call_timing(preprocessed_data):
- result = template.extract(preprocessed_data)
- print("提取结果:", json.dumps(result, ensure_ascii=False, indent=2))
- else:
- print("当前数据不满足模板调用条件")
|