template2.py 6.6 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177
  1. # coding:utf8
  2. import re
  3. from typing import Dict, Any
  4. from BiddingKG.dl.template_extract.abstract_template import AbstractTemplate, table2list
  5. import os
  6. import json
  7. class CustomOrderTemplate(AbstractTemplate):
  8. """
  9. 定制订单信息提取模板(适配data.json表格结构)
  10. 适配包含采购单位、采购人、成交供应商及商品明细的表格提取
  11. """
  12. def __init__(self):
  13. super().__init__(
  14. template_id=os.path.abspath(__file__),
  15. priority=3
  16. )
  17. def check_call_timing(self, preprocessed_data: Dict[str, Any]) -> bool:
  18. """检查是否符合模板调用条件"""
  19. if "表格列表" not in preprocessed_data:
  20. return False
  21. tables = preprocessed_data["表格列表"]
  22. if len(tables) < 1:
  23. return False
  24. try:
  25. # 获取第一个表格数据
  26. datas = tables[0]
  27. if len(datas) < 8: # 最少需要包含订单信息和商品表头行
  28. print("表格行数不足")
  29. return False
  30. # 验证订单信息区域表头(前5行)
  31. order_header_patterns = [
  32. ['', '(采购单位|采购人单位)'], # 第0行
  33. ['', '(采购人|采购联系人)'], # 第1行
  34. ['', '(成交供应商|中标供应商|供应商)'], # 第2行
  35. ['', '(成交金额|中标金额|金额)'], # 第3行
  36. ['', '(成交时间|中标时间|时间)'] # 第4行
  37. ]
  38. # 验证前4行核心订单信息(第0-3行)
  39. for i in range(4):
  40. if len(datas[i]) < 2:
  41. print(f"第{i}行列数不足")
  42. return False
  43. # 验证第一列是否为空(或匹配空字符串)
  44. if not re.fullmatch(order_header_patterns[i][0], datas[i][0]):
  45. print(f"第{i}行第一列不匹配: {datas[i][0]}")
  46. return False
  47. # 验证第二列是否符合对应模式
  48. if not re.fullmatch(order_header_patterns[i][1], datas[i][1]):
  49. print(f"第{i}行第二列不匹配: {datas[i][1]}")
  50. return False
  51. # 验证第5行为空行
  52. if not all(not cell.strip() for cell in datas[5]):
  53. print("第5行不是空行")
  54. return False
  55. # 验证商品明细区域表头(第6-7行)
  56. if not re.fullmatch('', datas[6][0]):
  57. print("第6行第一列不匹配")
  58. return False
  59. if not (re.fullmatch('品牌/型号', datas[6][1]) and
  60. re.fullmatch('数量', datas[6][2]) and
  61. re.fullmatch('单价', datas[6][3])):
  62. print("商品属性表头不匹配")
  63. return False
  64. # if not re.fullmatch('(纯水仪耗材|产品名称|商品名称)', datas[7][0]):
  65. # print("商品名称表头不匹配")
  66. # return False
  67. self.header = datas
  68. self.data_start = 7 # 商品数据从第7行开始
  69. return True
  70. except Exception as e:
  71. print(f"表格验证错误: {e}")
  72. return False
  73. def extract(self, preprocessed_data: Dict[str, Any]) -> Dict[str, Any]:
  74. """提取表格中的各类信息"""
  75. extract_result = {
  76. "招标信息": [],
  77. "中标信息": [],
  78. "候选人信息": [],
  79. "产品信息": []
  80. }
  81. # 提取招标信息(采购单位即招标人)
  82. purchaser = self.header[0][2] if len(self.header[0]) > 2 else ""
  83. contact_person = self.header[1][2] if len(self.header[1]) > 2 else ""
  84. if purchaser: # 招标信息必须包含招标人
  85. extract_result["招标信息"].append({
  86. "招标人": purchaser,
  87. "联系人": contact_person
  88. })
  89. # 提取中标信息
  90. supplier = self.header[2][2] if len(self.header[2]) > 2 else ""
  91. deal_price = self.header[3][2] if len(self.header[3]) > 2 else ""
  92. if supplier: # 中标信息必须包含中标人
  93. # 补充金额单位
  94. if deal_price and not re.search('[万亿美欧日]?元', deal_price):
  95. deal_price += "元"
  96. extract_result["中标信息"].append({
  97. "中标人": supplier,
  98. "中标价": deal_price
  99. })
  100. # 提取产品信息(候选人信息无,不提取)
  101. product_row = self.header[self.data_start]
  102. if len(product_row) >= 4:
  103. product = product_row[0]
  104. brand_model = product_row[1]
  105. # 分离品牌和型号
  106. brand = ""
  107. spec = ""
  108. if '/' in brand_model:
  109. brand, spec = brand_model.split('/', 1)
  110. else:
  111. brand = brand_model # 若没有型号,全部视为品牌
  112. quantity = product_row[2]
  113. unit_price = product_row[3]
  114. # 补充单价单位
  115. if unit_price and not re.search('[万亿美欧日]?元', unit_price):
  116. unit_price += "元"
  117. extract_result["产品信息"].append({
  118. "产品": product,
  119. "品牌": brand,
  120. "规格": spec,
  121. "数量": quantity,
  122. "单价": unit_price
  123. })
  124. # 过滤空列表和空值字段
  125. filtered_result = {}
  126. for key, value in extract_result.items():
  127. if isinstance(value, list):
  128. if value:
  129. # 过滤列表中每个字典的空值
  130. filtered_list = []
  131. for item in value:
  132. filtered_item = {k: v for k, v in item.items() if v}
  133. if filtered_item:
  134. filtered_list.append(filtered_item)
  135. if filtered_list:
  136. filtered_result[key] = filtered_list
  137. else:
  138. if value:
  139. filtered_result[key] = value
  140. return filtered_result
  141. # 使用示例
  142. if __name__ == "__main__":
  143. # 读取数据文件
  144. with open("data.json", "r", encoding="utf8") as f:
  145. preprocessed_data = json.load(f)
  146. # 初始化模板并提取信息
  147. template = CustomOrderTemplate()
  148. if template.check_call_timing(preprocessed_data):
  149. result = template.extract(preprocessed_data)
  150. print("提取结果:", json.dumps(result, ensure_ascii=False, indent=2))
  151. else:
  152. print("当前数据不满足模板调用条件")