| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177 |
- # coding:utf8
- import re
- from typing import Dict, Any
- from BiddingKG.dl.template_extract.abstract_template import AbstractTemplate, table2list
- import os
- import json
- class CustomOrderTemplate(AbstractTemplate):
- """
- 定制订单信息提取模板(适配data.json表格结构)
- 适配包含采购单位、采购人、成交供应商及商品明细的表格提取
- """
- def __init__(self):
- super().__init__(
- template_id=os.path.abspath(__file__),
- priority=3
- )
- def check_call_timing(self, preprocessed_data: Dict[str, Any]) -> bool:
- """检查是否符合模板调用条件"""
- if "表格列表" not in preprocessed_data:
- return False
- tables = preprocessed_data["表格列表"]
- if len(tables) < 1:
- return False
- try:
- # 获取第一个表格数据
- datas = tables[0]
- if len(datas) < 8: # 最少需要包含订单信息和商品表头行
- print("表格行数不足")
- return False
- # 验证订单信息区域表头(前5行)
- order_header_patterns = [
- ['', '(采购单位|采购人单位)'], # 第0行
- ['', '(采购人|采购联系人)'], # 第1行
- ['', '(成交供应商|中标供应商|供应商)'], # 第2行
- ['', '(成交金额|中标金额|金额)'], # 第3行
- ['', '(成交时间|中标时间|时间)'] # 第4行
- ]
- # 验证前4行核心订单信息(第0-3行)
- for i in range(4):
- if len(datas[i]) < 2:
- print(f"第{i}行列数不足")
- return False
- # 验证第一列是否为空(或匹配空字符串)
- if not re.fullmatch(order_header_patterns[i][0], datas[i][0]):
- print(f"第{i}行第一列不匹配: {datas[i][0]}")
- return False
- # 验证第二列是否符合对应模式
- if not re.fullmatch(order_header_patterns[i][1], datas[i][1]):
- print(f"第{i}行第二列不匹配: {datas[i][1]}")
- return False
- # 验证第5行为空行
- if not all(not cell.strip() for cell in datas[5]):
- print("第5行不是空行")
- return False
- # 验证商品明细区域表头(第6-7行)
- if not re.fullmatch('', datas[6][0]):
- print("第6行第一列不匹配")
- return False
- if not (re.fullmatch('品牌/型号', datas[6][1]) and
- re.fullmatch('数量', datas[6][2]) and
- re.fullmatch('单价', datas[6][3])):
- print("商品属性表头不匹配")
- return False
- # if not re.fullmatch('(纯水仪耗材|产品名称|商品名称)', datas[7][0]):
- # print("商品名称表头不匹配")
- # return False
- self.header = datas
- self.data_start = 7 # 商品数据从第7行开始
- return True
- except Exception as e:
- print(f"表格验证错误: {e}")
- return False
- def extract(self, preprocessed_data: Dict[str, Any]) -> Dict[str, Any]:
- """提取表格中的各类信息"""
- extract_result = {
- "招标信息": [],
- "中标信息": [],
- "候选人信息": [],
- "产品信息": []
- }
- # 提取招标信息(采购单位即招标人)
- purchaser = self.header[0][2] if len(self.header[0]) > 2 else ""
- contact_person = self.header[1][2] if len(self.header[1]) > 2 else ""
- if purchaser: # 招标信息必须包含招标人
- extract_result["招标信息"].append({
- "招标人": purchaser,
- "联系人": contact_person
- })
- # 提取中标信息
- supplier = self.header[2][2] if len(self.header[2]) > 2 else ""
- deal_price = self.header[3][2] if len(self.header[3]) > 2 else ""
- if supplier: # 中标信息必须包含中标人
- # 补充金额单位
- if deal_price and not re.search('[万亿美欧日]?元', deal_price):
- deal_price += "元"
- extract_result["中标信息"].append({
- "中标人": supplier,
- "中标价": deal_price
- })
- # 提取产品信息(候选人信息无,不提取)
- product_row = self.header[self.data_start]
- if len(product_row) >= 4:
- product = product_row[0]
- brand_model = product_row[1]
- # 分离品牌和型号
- brand = ""
- spec = ""
- if '/' in brand_model:
- brand, spec = brand_model.split('/', 1)
- else:
- brand = brand_model # 若没有型号,全部视为品牌
- quantity = product_row[2]
- unit_price = product_row[3]
- # 补充单价单位
- if unit_price and not re.search('[万亿美欧日]?元', unit_price):
- unit_price += "元"
- extract_result["产品信息"].append({
- "产品": product,
- "品牌": brand,
- "规格": spec,
- "数量": quantity,
- "单价": unit_price
- })
- # 过滤空列表和空值字段
- filtered_result = {}
- for key, value in extract_result.items():
- if isinstance(value, list):
- if value:
- # 过滤列表中每个字典的空值
- filtered_list = []
- for item in value:
- filtered_item = {k: v for k, v in item.items() if v}
- if filtered_item:
- filtered_list.append(filtered_item)
- if filtered_list:
- filtered_result[key] = filtered_list
- else:
- if value:
- filtered_result[key] = value
- return filtered_result
- # 使用示例
- if __name__ == "__main__":
- # 读取数据文件
- with open("data.json", "r", encoding="utf8") as f:
- preprocessed_data = json.load(f)
- # 初始化模板并提取信息
- template = CustomOrderTemplate()
- if template.check_call_timing(preprocessed_data):
- result = template.extract(preprocessed_data)
- print("提取结果:", json.dumps(result, ensure_ascii=False, indent=2))
- else:
- print("当前数据不满足模板调用条件")
|