| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219 |
- # coding:utf8
- import re
- from typing import Dict, Any
- from BiddingKG.dl.template_extract.abstract_template import AbstractTemplate, table2list
- import os
- import json
- class OceanUniversityTemplate(AbstractTemplate):
- """
- 中国海洋大学采购信息提取模板
- 适配包含项目信息、采购单位、成交信息及采购清单的表格结构
- """
- def __init__(self):
- super().__init__(
- template_id=os.path.abspath(__file__),
- priority=3
- )
- def check_call_timing(self, preprocessed_data: Dict[str, Any]) -> bool:
- """检查是否符合模板调用条件""" # 站源:中国海洋大学,例子:672542687
- if "表格列表" not in preprocessed_data:
- return False
- tables = preprocessed_data["表格列表"]
- if len(tables) < 1:
- return False
- try:
- main_table = tables[0]
- # 验证表格基本结构(至少包含13行关键信息行)
- if len(main_table) < 13:
- print(f"表格行数不足,期望至少13行,实际{len(main_table)}行")
- return False
- # 表头行正则匹配(泛化表达)
- header_patterns = [
- # 第0行:项目名称行
- (0, 0, '项目信息'),
- (0, 1, '(项目|工程)名称'),
- # 第2行:项目编号行
- (2, 0, '项目信息'),
- (2, 1, '项目编号'),
- # 第3行:采购单位行
- (3, 0, '采购单位信息'),
- (3, 1, '采购单位名称'),
- # 第4行:采购单位地址行
- (4, 0, '采购单位信息'),
- (4, 1, '采购单位地址'),
- # 第5行:联系人行
- (5, 0, '采购单位信息'),
- (5, 1, '联系人'),
- # 第6行:联系方式行
- (6, 0, '采购单位信息'),
- (6, 1, '联系方式'),
- # 第9行:成交供应商行
- (9, 0, '成交信息'),
- (9, 1, '成交供应商|中标人'),
- # 第8行:中标价行
- (8, 0, '成交信息'),
- (8, 1, '中标价|成交价'),
- # 第12行:采购清单表头行
- (12, 0, '序号'),
- (12, 1, '名称'),
- (12, 2, '规格型号|规格|型号'),
- (12, 3, '数量')
- ]
- # 验证关键表头位置和内容
- for row_idx, col_idx, pattern in header_patterns:
- if row_idx >= len(main_table) or col_idx >= len(main_table[row_idx]):
- print(f"表格结构异常,行{row_idx}列{col_idx}不存在")
- return False
- cell_value = main_table[row_idx][col_idx]
- if not re.fullmatch(pattern, cell_value, re.IGNORECASE):
- print(f"表头不匹配:行{row_idx}列{col_idx}期望[{pattern}],实际[{cell_value}]")
- return False
- self.main_table = main_table
- return True
- except Exception as e:
- print(f"表格验证错误: {e}")
- return False
- def extract(self, preprocessed_data: Dict[str, Any]) -> Dict[str, Any]:
- """提取表格中的各类信息"""
- result = {
- "项目编号": "",
- "项目名称": "",
- "招标信息": [],
- "中标信息": [],
- "候选人信息": [], # 数据中无候选人信息
- "产品信息": []
- }
- main_table = self.main_table
- # 提取项目基本信息
- # 项目名称(第0行第2列)
- if len(main_table) > 0 and len(main_table[0]) > 2:
- result["项目名称"] = main_table[0][2].strip()
- # 项目编号(第2行第2列)
- if len(main_table) > 2 and len(main_table[2]) > 2:
- result["项目编号"] = main_table[2][2].strip()
- # 提取招标信息(采购单位即招标人)
- tender_info = {
- "招标人": "",
- "地址": "",
- "联系人": "",
- "电话": ""
- }
- # 招标人名称(第3行第2列)
- if len(main_table) > 3 and len(main_table[3]) > 2:
- tender_info["招标人"] = main_table[3][2].strip()
- # 招标人地址(第4行第2列)
- if len(main_table) > 4 and len(main_table[4]) > 2:
- tender_info["地址"] = main_table[4][2].strip()
- # 联系人(第5行第2列)
- if len(main_table) > 5 and len(main_table[5]) > 2:
- tender_info["联系人"] = main_table[5][2].strip()
- # 联系电话(第6行第2列,从"电话:xxx"中提取)
- if len(main_table) > 6 and len(main_table[6]) > 2:
- phone_match = re.search(r'电话:(\d+\*{0,4}\d+)', main_table[6][2])
- if phone_match:
- tender_info["电话"] = phone_match.group(1)
- # 验证招标信息有效性(包含招标人)
- if tender_info["招标人"]:
- result["招标信息"].append(tender_info)
- # 提取中标信息
- win_info = {
- "中标人": "",
- "中标价": ""
- }
- # 中标人(第9行第2列)
- if len(main_table) > 9 and len(main_table[9]) > 2:
- win_info["中标人"] = main_table[9][2].strip()
- # 中标价(第8行第2列,确保有单位)
- if len(main_table) > 8 and len(main_table[8]) > 2:
- price = main_table[8][2].strip()
- if not re.search(r'[万亿]?[元角分]', price):
- # 尝试从表头推断单位(如果内容无单位)
- if re.search(r'[万亿]?[元角分]', main_table[8][1]):
- unit = re.search(r'[万亿]?[元角分]', main_table[8][1]).group()
- price += unit
- win_info["中标价"] = price
- # 验证中标信息有效性(包含中标人)
- if win_info["中标人"]:
- result["中标信息"].append(win_info)
- # 提取产品信息(从第13行开始的采购清单)
- if len(main_table) > 12:
- product_start_row = 13
- for row in main_table[product_start_row:]:
- if len(row) < 4:
- continue # 跳过不完整行
- # 解析产品名称和品牌(从名称中提取品牌)
- product_name = row[1].strip()
- brand = ""
- # 尝试从规格型号中提取品牌(格式:品牌/型号)
- spec = row[2].strip()
- spec_parts = re.split(r'/', spec, 1)
- if len(spec_parts) == 2:
- brand = spec_parts[0].strip()
- spec = spec_parts[1].strip()
- product_info = {
- "产品": product_name,
- "品牌": brand,
- "规格": spec,
- "数量": row[3].strip()
- }
- result["产品信息"].append(product_info)
- # 移除空列表和空值字段
- if not result["招标信息"]:
- del result["招标信息"]
- if not result["中标信息"]:
- del result["中标信息"]
- if not result["候选人信息"]:
- del result["候选人信息"]
- if not result["产品信息"]:
- del result["产品信息"]
- for key in list(result.keys()):
- if result[key] == "":
- del result[key]
- return result
- # 使用示例
- if __name__ == "__main__":
- with open("data.json", "r", encoding="utf8") as f:
- preprocessed_data = json.load(f)
- template = OceanUniversityTemplate()
- if template.check_call_timing(preprocessed_data):
- result = template.extract(preprocessed_data)
- print("提取结果:", json.dumps(result, ensure_ascii=False, indent=2))
- else:
- print("当前数据不满足模板调用条件")
|