template1.py 7.3 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169
  1. # coding:utf8
  2. import re
  3. from typing import Dict, Any
  4. from BiddingKG.dl.template_extract.abstract_template import AbstractTemplate
  5. import os
  6. import json
  7. class CustomCandidateBidTemplate(AbstractTemplate):
  8. """
  9. 候选人信息提取模板
  10. 适配data.json中行列均为表头的表格(行表头:名次/候选供应商等;列表头:第一名/第二名/第三名)
  11. """
  12. def __init__(self):
  13. super().__init__(
  14. template_id=os.path.abspath(__file__),
  15. priority=3
  16. )
  17. def check_call_timing(self, preprocessed_data: Dict[str, Any]) -> bool:
  18. """检查是否符合模板调用条件(适配行列双表头结构)"""
  19. if "表格列表" not in preprocessed_data: # 例子:727039445
  20. return False
  21. tables = preprocessed_data["表格列表"]
  22. if len(tables) < 1:
  23. return False
  24. try:
  25. # 目标表格为表格列表第一个元素
  26. self.table = tables[0]
  27. # 表格需至少包含行表头(6行)和列表头(至少1列数据列:第一名)
  28. if len(self.table) < 6 or len(self.table[0]) < 2:
  29. print(f"表格行列数异常:行{len(self.table)}列{len(self.table[0])},预期至少6行2列")
  30. return False
  31. # 提取行表头(每行第一个单元格)和列表头(第一行所有单元格)
  32. row_headers = [row[0].strip() for row in self.table]
  33. col_headers = [col.strip() for col in self.table[0]]
  34. # 行表头正则匹配规则(泛化表达,fullmatch)
  35. row_header_patterns = [
  36. r'名次|排名|排序|序号', # 行1:名次
  37. r'候选(成交|中标|投标)?(人|单位|供应商)(名称)?', # 行2:候选成交供应商
  38. r'(响应报价|投标报?价|报价|参选价)((不含税))?((万?元))?', # 行3:响应报价(元)
  39. r'(服务期|工期|交货期|服务时间)(\(日历天\))?', # 行4:服务期(日历天)
  40. r'项目负责人|联系人|负责人|联络人', # 行5:项目负责人
  41. r'备注|说明|注释|附注' # 行6:备注
  42. ]
  43. # 验证行表头(逐行匹配)
  44. for i, (header, pattern) in enumerate(zip(row_headers, row_header_patterns)):
  45. if not re.fullmatch(pattern, header):
  46. print(f'行表头第{i+1}行未匹配:{pattern} vs {header}')
  47. return False
  48. # 验证列表头(至少包含"名次"+"第一名/第二名/第三名"等)
  49. col_header_pattern = r'名次|第[一二三四五六七八九]+名|第\d+名'
  50. for col_header in col_headers:
  51. if not re.fullmatch(col_header_pattern, col_header):
  52. print(f'列表头列未匹配:{col_header_pattern} vs {col_header}')
  53. return False
  54. self.row_headers = row_headers
  55. self.col_headers = col_headers
  56. return True
  57. except Exception as e:
  58. print(f"表格验证错误: {e}")
  59. return False
  60. def extract(self, preprocessed_data: Dict[str, Any]) -> Dict[str, Any]:
  61. """提取表格中的相关信息(适配行列双表头结构)"""
  62. # 初始化返回结构
  63. extract_result = {
  64. "项目编号": "",
  65. "项目名称": "",
  66. "招标信息": [],
  67. "中标信息": [],
  68. "候选人信息": [],
  69. "产品信息": []
  70. }
  71. # 行索引映射(行表头对应的数据类型)
  72. row_idx_map = {
  73. "rank": 0, # 名次行
  74. "candidate": 1, # 候选供应商行
  75. "price": 2, # 响应报价行
  76. "service_time": 3, # 服务期行
  77. "contact": 4 # 项目负责人行
  78. }
  79. # 遍历列表头(跳过第一个"名次"列,处理第一名/第二名等列)
  80. for col_idx in range(1, len(self.col_headers)):
  81. # 提取单列候选人数据(按行索引取对应值)
  82. rank = self.table[row_idx_map["rank"]][col_idx].strip()
  83. candidate = self.table[row_idx_map["candidate"]][col_idx].strip()
  84. price = self.table[row_idx_map["price"]][col_idx].strip()
  85. service_time = self.table[row_idx_map["service_time"]][col_idx].strip()
  86. contact = self.table[row_idx_map["contact"]][col_idx].strip()
  87. # 核心规则:候选人信息必须包含候选人及排名,否则跳过
  88. if not candidate or not rank:
  89. continue
  90. # 补充金额单位(从行表头提取单位,内容无则补充)
  91. price_header = self.row_headers[row_idx_map["price"]]
  92. unit_match = re.search(r'(元)|元|万?元', price_header)
  93. if unit_match and not re.search(r'(元)|元|万?元', price):
  94. unit = unit_match.group(0).replace("(", "").replace(")", "") # 清理括号
  95. # 保留不含税说明并补充单位
  96. if "(不含税)" in price:
  97. price = price.replace("(不含税)", "") + f"(不含税){unit}"
  98. else:
  99. price += unit
  100. # 补充服务期单位
  101. if service_time:
  102. service_time_pattern = re.search(r'日|天|年|月|周|星期', self.row_headers[row_idx_map["service_time"]])
  103. if service_time_pattern and not re.search(r'日|天|年|月|周|星期', service_time):
  104. service_time += service_time_pattern.group(0)
  105. # 整理候选人信息
  106. candidate_info = {
  107. "标的": "",
  108. "标包": "",
  109. "包号": "",
  110. "候选人": candidate,
  111. "投标价": price,
  112. "排名": rank,
  113. "联系人": contact,
  114. "电话": "",
  115. "服务时间": service_time,
  116. "地址": ""
  117. }
  118. # 过滤空值字段
  119. candidate_info = {k: v for k, v in candidate_info.items() if v}
  120. extract_result["候选人信息"].append(candidate_info)
  121. # 最终结果过滤:空值字段/空列表不返回
  122. final_result = {}
  123. for key, value in extract_result.items():
  124. if isinstance(value, list):
  125. if len(value) > 0:
  126. final_result[key] = value
  127. else:
  128. if value.strip():
  129. final_result[key] = value
  130. # 严格遵循规则:
  131. # 1. 有排名的候选人信息,不返回中标信息
  132. # 2. 招标信息无招标人/预算,不返回
  133. # 3. 产品信息无产品,不返回
  134. return final_result
  135. # 使用示例
  136. if __name__ == "__main__":
  137. # 读取data.json数据
  138. with open("data.json", "r", encoding="utf8") as f:
  139. preprocessed_data = json.load(f)
  140. # 初始化模板并执行提取
  141. template = CustomCandidateBidTemplate()
  142. if template.check_call_timing(preprocessed_data):
  143. result = template.extract(preprocessed_data)
  144. print("提取结果:", json.dumps(result, ensure_ascii=False, indent=2))
  145. else:
  146. print("当前数据不满足模板调用条件")