template1.py 5.9 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159
  1. # coding:utf8
  2. import re
  3. from typing import Dict, Any
  4. from BiddingKG.dl.template_extract.abstract_template import AbstractTemplate, table2list
  5. import os
  6. import json
  7. class OrderInfoTemplate(AbstractTemplate):
  8. """
  9. 订单信息提取模板(适配采购订单表格场景)
  10. 适配包含采购单位、成交供应商及商品明细的表格提取
  11. """
  12. def __init__(self):
  13. super().__init__(
  14. template_id=os.path.abspath(__file__),
  15. priority=3
  16. )
  17. def check_call_timing(self, preprocessed_data: Dict[str, Any]) -> bool:
  18. """检查是否符合模板调用条件"""
  19. if "表格列表" not in preprocessed_data:
  20. return False
  21. tables = preprocessed_data["表格列表"]
  22. if len(tables) < 1:
  23. return False
  24. try:
  25. # 获取第一个表格数据
  26. datas = tables[0]
  27. if len(datas) < 8: # 最少需要包含订单信息和商品表头行
  28. print("表格行数不足")
  29. return False
  30. # 验证订单信息区域表头(前6行)
  31. order_header_patterns = [
  32. ['订单信息', '(采购单位|采购人单位)'], # 第0行
  33. ['订单信息', '(采购人|采购联系人)'], # 第1行
  34. ['订单信息', '(成交供应商|中标供应商)'], # 第2行
  35. ['订单信息', '(成交金额|中标金额)'], # 第3行
  36. ['订单信息', '运费金额'], # 第4行(可选,允许不匹配)
  37. ['订单信息', '(成交时间|中标时间)'] # 第5行
  38. ]
  39. # 验证前5行核心订单信息
  40. for i in range(5):
  41. if len(datas[i]) < 2:
  42. print(f"第{i}行列数不足")
  43. return False
  44. # 验证第一列是否为"订单信息"
  45. if not re.fullmatch(order_header_patterns[i][0], datas[i][0]):
  46. print(f"第{i}行第一列不匹配: {datas[i][0]}")
  47. return False
  48. # 验证第二列是否符合对应模式
  49. if not re.fullmatch(order_header_patterns[i][1], datas[i][1]):
  50. print(f"第{i}行第二列不匹配: {datas[i][1]}")
  51. return False
  52. # 验证商品明细区域表头(第6-7行)
  53. if not (re.fullmatch('订单明细', datas[6][0]) and re.fullmatch('商品名称', datas[7][0])):
  54. print("商品明细表头不匹配")
  55. return False
  56. if not (re.fullmatch('品牌', datas[7][1]) and re.fullmatch('数量', datas[7][2]) and re.fullmatch('单价', datas[7][3])):
  57. print("商品属性表头不匹配")
  58. return False
  59. self.header = datas
  60. self.data_start = 8 # 商品数据从第8行开始
  61. return True
  62. except Exception as e:
  63. print(f"表格验证错误: {e}")
  64. return False
  65. def extract(self, preprocessed_data: Dict[str, Any]) -> Dict[str, Any]:
  66. """提取表格中的各类信息"""
  67. extract_result = {
  68. "招标信息": [],
  69. "中标信息": [],
  70. "产品信息": []
  71. }
  72. # 提取招标信息(采购单位即招标人)
  73. purchaser = self.header[0][2] if len(self.header[0]) > 2 else ""
  74. contact_person = self.header[1][2] if len(self.header[1]) > 2 else ""
  75. if purchaser:
  76. extract_result["招标信息"].append({
  77. "招标人": purchaser,
  78. "联系人": contact_person
  79. })
  80. # 提取中标信息
  81. supplier = self.header[2][2] if len(self.header[2]) > 2 else ""
  82. deal_price = self.header[3][2] if len(self.header[3]) > 2 else ""
  83. if supplier:
  84. # 补充金额单位(如果缺失)
  85. if deal_price and not re.search('[万亿美欧日]?元', deal_price):
  86. deal_price += "元"
  87. extract_result["中标信息"].append({
  88. "中标人": supplier,
  89. "中标价": deal_price
  90. })
  91. # 提取产品信息
  92. for row in self.header[self.data_start:]:
  93. if len(row) < 4:
  94. continue # 跳过列数不足的行
  95. product = row[0]
  96. brand = row[1]
  97. quantity = row[2]
  98. unit_price = row[3]
  99. # 补充单价单位
  100. if unit_price and not re.search('[万亿美欧日]?元', unit_price):
  101. unit_price += "元"
  102. extract_result["产品信息"].append({
  103. "产品": product,
  104. "品牌": brand,
  105. "数量": quantity,
  106. "单价": unit_price
  107. })
  108. # 过滤空列表和空值字段
  109. filtered_result = {}
  110. for key, value in extract_result.items():
  111. if isinstance(value, list):
  112. if value:
  113. # 过滤列表中每个字典的空值
  114. filtered_list = []
  115. for item in value:
  116. filtered_item = {k: v for k, v in item.items() if v}
  117. if filtered_item:
  118. filtered_list.append(filtered_item)
  119. if filtered_list:
  120. filtered_result[key] = filtered_list
  121. else:
  122. if value:
  123. filtered_result[key] = value
  124. return filtered_result
  125. # 使用示例
  126. if __name__ == "__main__":
  127. # 读取数据文件
  128. with open("data.json", "r", encoding="utf8") as f:
  129. preprocessed_data = json.load(f)
  130. # 初始化模板并提取信息
  131. template = OrderInfoTemplate()
  132. if template.check_call_timing(preprocessed_data):
  133. result = template.extract(preprocessed_data)
  134. print("提取结果:", json.dumps(result, ensure_ascii=False, indent=2))
  135. else:
  136. print("当前数据不满足模板调用条件")