template1.py 7.9 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219
  1. # coding:utf8
  2. import re
  3. from typing import Dict, Any
  4. from BiddingKG.dl.template_extract.abstract_template import AbstractTemplate, table2list
  5. import os
  6. import json
  7. class OceanUniversityTemplate(AbstractTemplate):
  8. """
  9. 中国海洋大学采购信息提取模板
  10. 适配包含项目信息、采购单位、成交信息及采购清单的表格结构
  11. """
  12. def __init__(self):
  13. super().__init__(
  14. template_id=os.path.abspath(__file__),
  15. priority=3
  16. )
  17. def check_call_timing(self, preprocessed_data: Dict[str, Any]) -> bool:
  18. """检查是否符合模板调用条件""" # 站源:中国海洋大学,例子:672542687
  19. if "表格列表" not in preprocessed_data:
  20. return False
  21. tables = preprocessed_data["表格列表"]
  22. if len(tables) < 1:
  23. return False
  24. try:
  25. main_table = tables[0]
  26. # 验证表格基本结构(至少包含13行关键信息行)
  27. if len(main_table) < 13:
  28. print(f"表格行数不足,期望至少13行,实际{len(main_table)}行")
  29. return False
  30. # 表头行正则匹配(泛化表达)
  31. header_patterns = [
  32. # 第0行:项目名称行
  33. (0, 0, '项目信息'),
  34. (0, 1, '(项目|工程)名称'),
  35. # 第2行:项目编号行
  36. (2, 0, '项目信息'),
  37. (2, 1, '项目编号'),
  38. # 第3行:采购单位行
  39. (3, 0, '采购单位信息'),
  40. (3, 1, '采购单位名称'),
  41. # 第4行:采购单位地址行
  42. (4, 0, '采购单位信息'),
  43. (4, 1, '采购单位地址'),
  44. # 第5行:联系人行
  45. (5, 0, '采购单位信息'),
  46. (5, 1, '联系人'),
  47. # 第6行:联系方式行
  48. (6, 0, '采购单位信息'),
  49. (6, 1, '联系方式'),
  50. # 第9行:成交供应商行
  51. (9, 0, '成交信息'),
  52. (9, 1, '成交供应商|中标人'),
  53. # 第8行:中标价行
  54. (8, 0, '成交信息'),
  55. (8, 1, '中标价|成交价'),
  56. # 第12行:采购清单表头行
  57. (12, 0, '序号'),
  58. (12, 1, '名称'),
  59. (12, 2, '规格型号|规格|型号'),
  60. (12, 3, '数量')
  61. ]
  62. # 验证关键表头位置和内容
  63. for row_idx, col_idx, pattern in header_patterns:
  64. if row_idx >= len(main_table) or col_idx >= len(main_table[row_idx]):
  65. print(f"表格结构异常,行{row_idx}列{col_idx}不存在")
  66. return False
  67. cell_value = main_table[row_idx][col_idx]
  68. if not re.fullmatch(pattern, cell_value, re.IGNORECASE):
  69. print(f"表头不匹配:行{row_idx}列{col_idx}期望[{pattern}],实际[{cell_value}]")
  70. return False
  71. self.main_table = main_table
  72. return True
  73. except Exception as e:
  74. print(f"表格验证错误: {e}")
  75. return False
  76. def extract(self, preprocessed_data: Dict[str, Any]) -> Dict[str, Any]:
  77. """提取表格中的各类信息"""
  78. result = {
  79. "项目编号": "",
  80. "项目名称": "",
  81. "招标信息": [],
  82. "中标信息": [],
  83. "候选人信息": [], # 数据中无候选人信息
  84. "产品信息": []
  85. }
  86. main_table = self.main_table
  87. # 提取项目基本信息
  88. # 项目名称(第0行第2列)
  89. if len(main_table) > 0 and len(main_table[0]) > 2:
  90. result["项目名称"] = main_table[0][2].strip()
  91. # 项目编号(第2行第2列)
  92. if len(main_table) > 2 and len(main_table[2]) > 2:
  93. result["项目编号"] = main_table[2][2].strip()
  94. # 提取招标信息(采购单位即招标人)
  95. tender_info = {
  96. "招标人": "",
  97. "地址": "",
  98. "联系人": "",
  99. "电话": ""
  100. }
  101. # 招标人名称(第3行第2列)
  102. if len(main_table) > 3 and len(main_table[3]) > 2:
  103. tender_info["招标人"] = main_table[3][2].strip()
  104. # 招标人地址(第4行第2列)
  105. if len(main_table) > 4 and len(main_table[4]) > 2:
  106. tender_info["地址"] = main_table[4][2].strip()
  107. # 联系人(第5行第2列)
  108. if len(main_table) > 5 and len(main_table[5]) > 2:
  109. tender_info["联系人"] = main_table[5][2].strip()
  110. # 联系电话(第6行第2列,从"电话:xxx"中提取)
  111. if len(main_table) > 6 and len(main_table[6]) > 2:
  112. phone_match = re.search(r'电话:(\d+\*{0,4}\d+)', main_table[6][2])
  113. if phone_match:
  114. tender_info["电话"] = phone_match.group(1)
  115. # 验证招标信息有效性(包含招标人)
  116. if tender_info["招标人"]:
  117. result["招标信息"].append(tender_info)
  118. # 提取中标信息
  119. win_info = {
  120. "中标人": "",
  121. "中标价": ""
  122. }
  123. # 中标人(第9行第2列)
  124. if len(main_table) > 9 and len(main_table[9]) > 2:
  125. win_info["中标人"] = main_table[9][2].strip()
  126. # 中标价(第8行第2列,确保有单位)
  127. if len(main_table) > 8 and len(main_table[8]) > 2:
  128. price = main_table[8][2].strip()
  129. if not re.search(r'[万亿]?[元角分]', price):
  130. # 尝试从表头推断单位(如果内容无单位)
  131. if re.search(r'[万亿]?[元角分]', main_table[8][1]):
  132. unit = re.search(r'[万亿]?[元角分]', main_table[8][1]).group()
  133. price += unit
  134. win_info["中标价"] = price
  135. # 验证中标信息有效性(包含中标人)
  136. if win_info["中标人"]:
  137. result["中标信息"].append(win_info)
  138. # 提取产品信息(从第13行开始的采购清单)
  139. if len(main_table) > 12:
  140. product_start_row = 13
  141. for row in main_table[product_start_row:]:
  142. if len(row) < 4:
  143. continue # 跳过不完整行
  144. # 解析产品名称和品牌(从名称中提取品牌)
  145. product_name = row[1].strip()
  146. brand = ""
  147. # 尝试从规格型号中提取品牌(格式:品牌/型号)
  148. spec = row[2].strip()
  149. spec_parts = re.split(r'/', spec, 1)
  150. if len(spec_parts) == 2:
  151. brand = spec_parts[0].strip()
  152. spec = spec_parts[1].strip()
  153. product_info = {
  154. "产品": product_name,
  155. "品牌": brand,
  156. "规格": spec,
  157. "数量": row[3].strip()
  158. }
  159. result["产品信息"].append(product_info)
  160. # 移除空列表和空值字段
  161. if not result["招标信息"]:
  162. del result["招标信息"]
  163. if not result["中标信息"]:
  164. del result["中标信息"]
  165. if not result["候选人信息"]:
  166. del result["候选人信息"]
  167. if not result["产品信息"]:
  168. del result["产品信息"]
  169. for key in list(result.keys()):
  170. if result[key] == "":
  171. del result[key]
  172. return result
  173. # 使用示例
  174. if __name__ == "__main__":
  175. with open("data.json", "r", encoding="utf8") as f:
  176. preprocessed_data = json.load(f)
  177. template = OceanUniversityTemplate()
  178. if template.check_call_timing(preprocessed_data):
  179. result = template.extract(preprocessed_data)
  180. print("提取结果:", json.dumps(result, ensure_ascii=False, indent=2))
  181. else:
  182. print("当前数据不满足模板调用条件")