template3.py 7.2 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191
  1. # coding:utf8
  2. import re
  3. from typing import Dict, Any
  4. from BiddingKG.dl.template_extract.abstract_template import AbstractTemplate, table2list
  5. import os
  6. import json
  7. class PurchaseTransactionTemplate(AbstractTemplate):
  8. """
  9. 采购成交信息提取模板(适配物料采购表格场景)
  10. 适配包含中标供应商、物料明细的表格提取
  11. """
  12. def __init__(self):
  13. super().__init__(
  14. template_id=os.path.abspath(__file__),
  15. priority=3
  16. )
  17. self.header = None # 物料采购表格表头
  18. self.data = None # 物料采购表格数据
  19. def check_call_timing(self, preprocessed_data: Dict[str, Any]) -> bool:
  20. """检查是否符合模板调用条件"""
  21. if "表格列表" not in preprocessed_data:
  22. return False
  23. tables = preprocessed_data["表格列表"]
  24. if len(tables) < 1:
  25. return False
  26. try:
  27. # 验证物料采购表格(唯一表格)
  28. purchase_table = tables[0]
  29. if len(purchase_table) < 2: # 至少包含表头+1行数据
  30. return False
  31. # 物料采购表头正则匹配(按实际位置泛化)
  32. purchase_header_patterns = [
  33. r'序号', # 位置0
  34. r'中标供应商|成交供应商|中标单位|成交单位', # 位置1
  35. r'物料编码|物资编码|材料编码|货品编码', # 位置2
  36. r'物料描述|物资描述|材料名称|货品名称|产品名称', # 位置3
  37. r'规格型号|规格|型号|技术参数', # 位置4
  38. r'计量单位|单位|数量单位', # 位置5
  39. r'采购数量|计划数量|需求数量', # 位置6
  40. r'中标数量|成交数量|供货数量', # 位置7
  41. r'成交单价|中标单价|单价|报价', # 位置8
  42. r'到货日期|交货日期|交付日期|到货时间', # 位置9
  43. r'交货地点|到货地点|交付地址|交货地址' # 位置10
  44. ]
  45. # 校验每个表头位置的匹配度
  46. for i, pattern in enumerate(purchase_header_patterns):
  47. if i >= len(purchase_table[0]): # 兼容表头长度不足的情况
  48. return False
  49. if not re.fullmatch(pattern, purchase_table[0][i], re.IGNORECASE):
  50. # print(f'物料采购表头未验证: 位置{i} 预期[{pattern}] 实际[{purchase_table[0][i]}]')
  51. return False
  52. # 赋值表格数据
  53. self.header = purchase_table[0]
  54. self.data = purchase_table[1:]
  55. return True
  56. except Exception as e:
  57. # print(f"表格验证错误: {e}")
  58. return False
  59. def extract(self, preprocessed_data: Dict[str, Any]) -> Dict[str, Any]:
  60. """提取表格中的中标信息、产品信息"""
  61. extract_result = {
  62. "项目编号": "",
  63. "项目名称": "",
  64. "招标信息": [],
  65. "中标信息": [],
  66. "候选人信息": [],
  67. "产品信息": []
  68. }
  69. # 提取中标信息(物料采购表格)
  70. for row in self.data:
  71. if len(row) < 11: # 确保行数据长度匹配表头
  72. continue
  73. # 按表头位置解析字段
  74. seq = row[0]
  75. supplier_name = row[1]
  76. material_code = row[2]
  77. product_name = row[3]
  78. spec = row[4]
  79. unit = row[5]
  80. purchase_quantity = row[6]
  81. bid_quantity = row[7]
  82. unit_price = row[8]
  83. arrival_date = row[9]
  84. delivery_address = row[10]
  85. # 中标信息必须包含中标人
  86. if not supplier_name:
  87. continue
  88. # 补充单价单位(元)
  89. price_unit = "元"
  90. if re.search(r'[万亿美欧日]?元', self.header[8], re.IGNORECASE):
  91. price_unit = re.search(r'[万亿美欧日]?元', self.header[8], re.IGNORECASE).group(0)
  92. if unit_price and not re.search('[万亿美欧日]?元', unit_price):
  93. unit_price += price_unit
  94. # 组装中标信息(过滤空值)
  95. bid_info = {
  96. "标的": product_name,
  97. "中标人": supplier_name,
  98. "中标价": unit_price,
  99. # "地址": delivery_address,
  100. "服务时间": arrival_date # 到货日期作为服务/交付时间
  101. }
  102. # 过滤空值字段
  103. bid_info = {k: v for k, v in bid_info.items() if v}
  104. extract_result["中标信息"].append(bid_info)
  105. # 提取产品信息(物料采购表格)
  106. product_list = []
  107. for row in self.data:
  108. if len(row) < 11:
  109. continue
  110. # 按表头位置解析字段
  111. seq = row[0]
  112. supplier_name = row[1]
  113. material_code = row[2]
  114. product_name = row[3]
  115. spec = row[4]
  116. unit = row[5]
  117. purchase_quantity = row[6]
  118. bid_quantity = row[7]
  119. unit_price = row[8]
  120. arrival_date = row[9]
  121. delivery_address = row[10]
  122. # 补充单价单位
  123. price_unit = "元"
  124. if re.search(r'[万亿美欧日]?元', self.header[8], re.IGNORECASE):
  125. price_unit = re.search(r'[万亿美欧日]?元', self.header[8], re.IGNORECASE).group(0)
  126. if unit_price and not re.search('[万亿美欧日]?元', unit_price):
  127. unit_price += price_unit
  128. # 组装产品信息
  129. product_info = {
  130. "产品": product_name,
  131. "规格": spec,
  132. "数量": bid_quantity, # 中标数量作为实际供货数量
  133. "单位": unit,
  134. "单价": unit_price
  135. }
  136. # 产品信息必须包含产品及至少一个其他要素
  137. other_fields = [product_info["规格"], product_info["数量"],
  138. product_info["单位"], product_info["单价"]]
  139. if product_info["产品"] and any(other_fields):
  140. # 过滤空值字段
  141. product_info = {k: v for k, v in product_info.items() if v}
  142. product_list.append(product_info)
  143. extract_result["产品信息"] = product_list
  144. # 招标信息校验:无招标人/预算则置空
  145. extract_result["招标信息"] = []
  146. # 候选人信息:无候选人及排名则置空
  147. extract_result["候选人信息"] = []
  148. # 整体过滤空值字段(顶层)
  149. extract_result = {k: v for k, v in extract_result.items() if v or v == []}
  150. return extract_result
  151. # 使用示例
  152. if __name__ == "__main__":
  153. # 读取并解析JSON文件
  154. with open("data.json", "r", encoding="utf8") as f:
  155. preprocessed_data = json.load(f)
  156. # 初始化模板并提取信息
  157. template = PurchaseTransactionTemplate()
  158. if template.check_call_timing(preprocessed_data):
  159. result = template.extract(preprocessed_data)
  160. print("提取结果:", json.dumps(result, ensure_ascii=False, indent=4))
  161. else:
  162. print("当前数据不满足模板调用条件")