# -*- coding: utf-8 -*- """表格解析与二维化。 按 ARCHITECTURE.md Phase 4 拆分建议,从 ``interface/Preprocessing.py`` 迁出。 类型:PREPROCESS。 原位置:``interface/Preprocessing.py`` 中以下函数: - ``tableToText`` — HTML 表格转文本,含 fixSpan/getTable 等嵌套辅助函数 - ``table_head_repair_process`` — 表头修复 ``interface/Preprocessing.py`` 仍 re-export 以上全部名称,老 import 不受影响。 """ from __future__ import absolute_import import copy import json import re import time import numpy as np from BiddingKG.dl.common.logging import log from BiddingKG.dl.interface.predictor import getPredictor from BiddingKG.dl.predictors.table_prem import TableTag2List from BiddingKG.dl.model_runtime.embed import formEncoding from BiddingKG.dl.table_head.predict_torch import predict from BiddingKG.dl.preprocess.segmenter import segment __all__ = [ "tableToText", "table_head_repair_process", ] def tableToText(soup, docid=None, return_kv=False): ''' @param: soup:网页html的soup @return:处理完表格信息的网页text ''' def getTrs(tbody): #获取所有的tr trs = [] objs = tbody.find_all(recursive=False) for obj in objs: if obj.name=="tr": trs.append(obj) if obj.name=="tbody": for tr in obj.find_all("tr",recursive=False): trs.append(tr) return trs def fixSpan(tbody): # 处理colspan, rowspan信息补全问题 #trs = tbody.findChildren('tr', recursive=False) trs = getTrs(tbody) ths_len = 0 ths = list() trs_set = set() #修改为先进行列补全再进行行补全,否则可能会出现表格解析混乱 # 遍历每一个tr for indtr, tr in enumerate(trs): ths_tmp = tr.findChildren('th', recursive=False) #不补全含有表格的tr if len(tr.findChildren('table'))>0: continue if len(ths_tmp) > 0: ths_len = ths_len + len(ths_tmp) for th in ths_tmp: ths.append(th) trs_set.add(tr) # 遍历每行中的element tds = tr.findChildren(recursive=False) for indtd, td in enumerate(tds): # 若有colspan 则补全同一行下一个位置 if 'colspan' in td.attrs: if str(re.sub("[^0-9]","",str(td['colspan'])))!="": col = int(re.sub("[^0-9]","",str(td['colspan']))) if col<100 and len(td.get_text())<1000: td['colspan'] = 1 for i in range(1, col, 1): td.insert_after(copy.copy(td)) for indtr, tr in enumerate(trs): ths_tmp = tr.findChildren('th', recursive=False) #不补全含有表格的tr if len(tr.findChildren('table'))>0: continue if len(ths_tmp) > 0: ths_len = ths_len + len(ths_tmp) for th in ths_tmp: ths.append(th) trs_set.add(tr) # 遍历每行中的element tds = tr.findChildren(recursive=False) for indtd, td in enumerate(tds): # 若有rowspan 则补全下一行同样位置 if 'rowspan' in td.attrs: if str(re.sub("[^0-9]","",str(td['rowspan'])))!="": row = int(re.sub("[^0-9]","",str(td['rowspan']))) td['rowspan'] = 1 for i in range(1, row, 1): # 获取下一行的所有td, 在对应的位置插入 if indtr+i= (indtd) and len(tds1)>0: if indtd > 0: tds1[indtd - 1].insert_after(copy.copy(td)) else: tds1[0].insert_before(copy.copy(td)) elif indtd-2>0 and len(tds1) > 0 and len(tds1) == indtd - 1: # 修正某些表格最后一列没补全 tds1[indtd-2].insert_after(copy.copy(td)) def getTable(tbody): #trs = tbody.findChildren('tr', recursive=False) trs = getTrs(tbody) inner_table = [] for tr in trs: tr_line = [] tds = tr.findChildren(['td','th'], recursive=False) if len(tds)==0: if return_kv: tr_line.append([re.sub('\xa0','',tr.get_text()),0]) else: tr_line.append([re.sub('\xa0','',segment(tr,final=False)),0]) # 2021/12/21 修复部分表格没有td 造成数据丢失 for td in tds: if return_kv: tr_line.append([re.sub('\xa0','',td.get_text()),0]) else: tr_line.append([re.sub('\xa0','',segment(td,final=False)),0]) #tr_line.append([td.get_text(),0]) inner_table.append(tr_line) return inner_table #处理表格不对齐的问题 def fixTable(inner_table,fix_value="~~"): maxWidth = 0 for item in inner_table: if len(item)>maxWidth: maxWidth = len(item) if maxWidth > 100: # log('表格列数大于100,表格异常不做处理。') return [] for i in range(len(inner_table)): if len(inner_table[i]) 0 or last_0-len(line)+1 < 0 or last_1 == len(line)-1 or count_1-count_0 >= 3: return True return False def getsimilarity(line, line1): same_count = 0 for item, item1 in zip(line,line1): if item[1] == item1[1]: same_count += 1 return same_count/len(line) def selfrepair(inner_table,index,dye_set,key_set): """ @summary: 计算每个节点受到的挤压度来判断是否需要染色 """ # print("B",inner_table[index]) min_presure = 3 list_dye = [] first = None count = 0 # temp_set = set() temp_set = set(['~~']) # 2023/10/10纠正236239652 受让单位识别不到表头; 受让单位,明细用途:用途名称:陵川县民政局, _index = 0 for item in inner_table[index]: if first is None: first = item[1] if item[0] not in temp_set: count += 1 temp_set.add(item[0]) else: if first == item[1]: if item[0] not in temp_set: temp_set.add(item[0]) count += 1 else: list_dye.append([first,count,_index]) first = item[1] temp_set.add(item[0]) count = 1 _index += 1 list_dye.append([first,count,_index]) if len(list_dye)>1: begin = 0 end = 0 for i in range(len(list_dye)): end = list_dye[i][2] dye_flag = False # 首尾要求压力减一 if i==0: if list_dye[i+1][1]-list_dye[i][1]+1>=min_presure-1: dye_flag = True dye_type = list_dye[i+1][0] elif i==len(list_dye)-1: if list_dye[i-1][1]-list_dye[i][1]+1>=min_presure-1: dye_flag = True dye_type = list_dye[i-1][0] else: if list_dye[i][1]>1: if list_dye[i+1][1]-list_dye[i][1]+1>=min_presure: dye_flag = True dye_type = list_dye[i+1][0] if list_dye[i-1][1]-list_dye[i][1]+1>=min_presure: dye_flag = True dye_type = list_dye[i-1][0] else: if list_dye[i+1][1]+list_dye[i-1][1]-list_dye[i][1]+1>=min_presure: dye_flag = True dye_type = list_dye[i+1][0] if list_dye[i+1][1]+list_dye[i-1][1]-list_dye[i][1]+1>=min_presure: dye_flag = True dye_type = list_dye[i-1][0] if dye_flag: for h in range(begin,end): inner_table[index][h][1] = dye_type dye_set.add((inner_table[index][h][0],dye_type)) key_set.add(inner_table[index][h][0]) begin = end # print("E",inner_table[index]) def otherrepair(inner_table,index,dye_set,key_set): list_provide_repair = [] if index==0 and len(inner_table)>1: list_provide_repair.append(index+1) elif index==len(inner_table)-1: list_provide_repair.append(index-1) else: list_provide_repair.append(index+1) list_provide_repair.append(index-1) for provide_index in list_provide_repair: if not repairNeeded(inner_table[provide_index]): same_prob = getsimilarity(inner_table[index], inner_table[provide_index]) if same_prob>=0.8: for i in range(len(inner_table[provide_index])): if inner_table[index][i][1]!=inner_table[provide_index][i][1]: dye_set.add((inner_table[index][i][0],inner_table[provide_index][i][1])) key_set.add(inner_table[index][i][0]) inner_table[index][i][1] = inner_table[provide_index][i][1] elif same_prob<=0.2: for i in range(len(inner_table[provide_index])): if inner_table[index][i][1]==inner_table[provide_index][i][1]: dye_set.add((inner_table[index][i][0],inner_table[provide_index][i][1])) key_set.add(inner_table[index][i][0]) inner_table[index][i][1] = 0 if inner_table[provide_index][i][1] ==1 else 1 len_dye_set = len(dye_set) height = len(inner_table) for i in range(height): if repairNeeded(inner_table[i]): selfrepair(inner_table, i, dye_set, key_set) #otherrepair(inner_table,i,dye_set,key_set) for h in range(len(inner_table)): for w in range(len(inner_table[0])): if inner_table[h][w][0] in key_set: for item in dye_set: if inner_table[h][w][0] == item[0]: inner_table[h][w][1] = item[1] # 如果两个set长度不相同,则有同一个key被反复染色,将导致无限迭代 if len(dye_set) != len(key_set): for i in range(height): if repairNeeded(inner_table[i]): selfrepair(inner_table,i,dye_set,key_set) #otherrepair(inner_table,i,dye_set,key_set) return if len(dye_set) == len_dye_set: ''' for i in range(height): if repairNeeded(inner_table[i]): otherrepair(inner_table,i,dye_set,key_set) ''' return repairTable(inner_table, dye_set, key_set) def repair_table2(inner_table, show=0, row_no=0): """ @summary: 修复表头识别,将明显错误的进行修正 """ # 循环处理单元格,一次获取需要的 one_head_index_list = [] zero_head_index_list = [] all_head_index_list = [] for i in range(len(inner_table)): head_cnt = 0 for j in range(len(inner_table[i])): # 删除前后逗号 inner_table[i][j][0] = re.sub('^[,,]+', '', inner_table[i][j][0]) inner_table[i][j][0] = re.sub('[,,]+$', '', inner_table[i][j][0]) # 统计表头数 if inner_table[i][j][1] == 1: head_cnt += 1 # 表头数list if head_cnt == 0: zero_head_index_list.append(i) elif head_cnt == 1: one_head_index_list.append(i) elif head_cnt == len(inner_table[i]): all_head_index_list.append(i) # 修复冒号在文本中间的,不能作为表头;(冒号后面需多个字) # 冒号在括号中的除外 # 冒号在最后的,判断后一个格子是否有重复的文字 for i in range(len(inner_table)): for j in range(len(inner_table[i])): _text = inner_table[i][j][0] if len(_text) >= 3 and inner_table[i][j][1] == 1: match = re.search('[::]', _text) if match: start_index, end_index = match.span() if start_index == 0: continue if end_index == len(_text): if len(inner_table[i]) == 2 and j <= len(inner_table[i]) - 2 and (_text in inner_table[i][j+1][0] or inner_table[i][j+1][0] in _text): inner_table[i][j][1] = 0 inner_table[i][j+1][1] = 0 else: continue if re.search('[((]', _text[:start_index]) and re.search('[))]', _text[end_index:]): continue m1 = re.search('[\u4e00-\u9fa50-9a-zA-Z]', _text[:start_index]) m2 = re.search('[\u4e00-\u9fa50-9a-zA-Z]', _text[end_index:]) if m1 and m2 and (len(m2.group()) >= 2 or m2.group() in ['是', '否']): inner_table[i][j][1] = 0 if show: print('inner_table[i]1', inner_table[row_no]) # 修复实际只有几列,但有一列由于重复占了太多行表头识别错误 # for i in range(len(inner_table)): # head_flag_dict = {} # for j in range(len(inner_table[i])): # if inner_table[i][j][0] in head_flag_dict.keys(): # head_flag_dict[inner_table[i][j][0]] += [inner_table[i][j][1]] # else: # head_flag_dict[inner_table[i][j][0]] = [inner_table[i][j][1]] # # if len(head_flag_dict.keys()) == 2: # col_flag = None # col_value = None # for key in head_flag_dict.keys(): # flag_list = head_flag_dict[key] # if len(flag_list) >= 4 and len(set(flag_list)) == 2 and len(set(flag_list[1:])) == 1: # col_flag = flag_list[0] # col_value = key # break # # if col_flag is not None: # for j in range(len(inner_table[i])): # if inner_table[i][j][0] == col_value: # inner_table[i][j][1] = col_flag # 多个重复列的预测值不同,以第一个为准 for i in range(len(inner_table)): col = inner_table[i][0] for j in range(len(inner_table[i])): if inner_table[i][j][0] == col[0]: if inner_table[i][j][1] != col[1]: inner_table[i][j][1] = col[1] else: col = inner_table[i][j] if show: print('inner_table[i]2', inner_table[row_no]) # 修复多个重复的单元格表头不一致 # for i in range(len(inner_table)): # for j in range(len(inner_table[i])-1): # only_chinese1 = ''.join(re.findall('[\u4e00-\u9fa5]+', inner_table[i][j][0])) # only_chinese2 = ''.join(re.findall('[\u4e00-\u9fa5]+', inner_table[i][j+1][0])) # if only_chinese1 == only_chinese2 and inner_table[i][j][1] != inner_table[i][j+1][1]: # inner_table[i][j][1] = 1 # inner_table[i][j+1][1] = 1 # if show: # print('inner_table[i]3', inner_table[row_no]) # # 修复一行几乎都是表头,个别不是;或者一行几乎都是非表头,个别是 # for i in range(len(inner_table)): # head_dict = {} # not_head_dict = {} # for j in range(len(inner_table[i])): # if inner_table[i][j][1] == 1: # if inner_table[i][j][0] not in head_dict: # head_dict[inner_table[i][j][0]] = 1 # else: # if inner_table[i][j][0] not in not_head_dict: # not_head_dict[inner_table[i][j][0]] = 1 # # # 非表头:表头 <= 1:3 # # if len(head_dict.keys()) > 0 and len(not_head_dict.keys()) / len(head_dict.keys()) <= 1/3 and len(head_dict.keys()) >= 3: # # for j in range(len(inner_table[i])): # # if len(re.sub(' ', '', inner_table[i][j][0])) > 0: # # inner_table[i][j][1] = 1 # # # 表头数一个且非表头数大于2且上一行都是表头 # if i > 0 and len(head_dict.keys()) == 1 and len(not_head_dict.keys()) >= 2 and inner_table[i][0][1] == 0: # last_row = inner_table[i-1] # col_list = [] # for j in range(len(last_row)): # if len(re.sub(' ', '', last_row[j][0])) > 0: # if last_row[j][1] == 0: # col_list = [] # break # col_list.append(last_row[j][0]) # if col_list: # col_list = list(set(col_list)) # if len(col_list) > 2: # for j in range(len(inner_table[i])): # if inner_table[i][j][1] == 1: # inner_table[i][j][1] = 0 # 一整个大表格,第一行为表头,下面行中有个别格子被识别为表头 # 候选人后面修复 for index in one_head_index_list: if (index - 1 in zero_head_index_list and index - 2 in zero_head_index_list) \ or (index - 1 in zero_head_index_list and index - 2 in all_head_index_list) \ or (index - 1 in all_head_index_list): for j in range(len(inner_table[index])): inner_table[index][j][1] = 0 zero_head_index_list.append(index) if show: print('inner_table[i]4', inner_table[row_no]) # 修复第一第二第三中标候选人作为表头 first_tenderer = ['第一中标候选人', '第一中标人', '第一中标(成交)人', '第一候选人'] second_tenderer = ['第二中标候选人', '第二中标(成交)候选人', '第二候选人'] third_tenderer = ['第三中标候选人', '第三中标(成交)候选人', '第三候选人'] # n1 next one, n2 next two, l1 last one, l2 last two for i in range(len(inner_table)): row = inner_table[i] n1_row, n2_row = None, None if i+1 < len(inner_table): n1_row = inner_table[i+1] if i+2 < len(inner_table): n2_row = inner_table[i+2] for j in range(len(row)): row_col = row[j] n1_row_col, n2_row_col = None, None row_n1_col, row_n2_col = None, None n1_row_n1_col, n2_row_n1_col, n1_row_n2_col = None, None, None if n1_row: n1_row_col = n1_row[j] if n2_row: n2_row_col = n2_row[j] if j+1 < len(row): row_n1_col = row[j+1] if j+2 < len(row): row_n2_col = row[j+2] if n1_row and j+1 < len(n1_row): n1_row_n1_col = n1_row[j+1] if n2_row and j+1 < len(n2_row): n2_row_n1_col = n2_row[j+1] if n1_row and j+2 < len(n1_row): n1_row_n2_col = n1_row[j+2] # 连续作为行表头 if row_col[0] in first_tenderer and row_n1_col and row_n1_col[1] == 0: if n1_row_col and n1_row_col[0] in second_tenderer and n1_row_n1_col and n1_row_n1_col[1] == 0: inner_table[i][j][1] = 1 inner_table[i+1][j][1] = 1 if n2_row_col and n2_row_col[0] in third_tenderer and n2_row_n1_col and n2_row_n1_col[1] == 0: inner_table[i+2][j][1] = 1 # 连续作为列表头 if row_col[0] in first_tenderer and n1_row_col and n1_row_col[1] == 0: if row_n1_col and row_n1_col[0] in second_tenderer and n1_row_n1_col and n1_row_n1_col[1] == 0: inner_table[i][j][1] = 1 inner_table[i][j+1][1] = 1 if row_n2_col and row_n2_col[0] in third_tenderer and n1_row_n2_col and n1_row_n2_col[1] == 0: inner_table[i][j+2][1] = 1 if show: print('inner_table[i]5', inner_table[row_no]) # 修复表头关键词未作为表头 # 文本匹配关键词,直接作为表头 head_keyword = ['供应商', '总价'] # 末尾匹配关键词且前一列为表头且与前一列文本不同,直接不做表头 head_keyword2 = ['管理中心', '有限公司', '项目采购', ] # 开头匹配关键词,直接不做表头 head_keyword3 = ['详见', '选定', '咨询服务', '标准物资', '电汇', '承兑'] # 文本匹配关键词且前一列为表头,直接作为表头 head_keyword4 = ['综合排名'] # 文本在关键词中,直接不做表头 head_keyword5 = ['殡葬用地'] # n1 next one, n2 next two, l1 last one, l2 last two for i in range(len(inner_table)): row = inner_table[i] for j in range(len(row)): row_col = row[j] row_l1_col = None if j-1 > 0: row_l1_col = row[j-1] match = re.search('[\u4e00-\u9fa50-9a-zA-Z::]+', row_col[0]) if inner_table[i][j][1] == 0 and match and match.group() in head_keyword: inner_table[i][j][1] = 1 for key in head_keyword2: match = re.search(key+'$', row_col[0]) if j > 0 and row_l1_col and row_l1_col[1] == 1 and row_l1_col[0] != row_col[0] and match and row_col[1] == 1: inner_table[i][j][1] = 0 for key in head_keyword3: match = re.search('^'+key, row_col[0]) if match and row_col[1] == 1: inner_table[i][j][1] = 0 for key in head_keyword4: match = re.search(key, row_col[0]) if j > 0 and row_l1_col and row_l1_col[1] == 1 and match and row_col[1] == 0: inner_table[i][j][1] = 1 if row_col[0] in head_keyword5: inner_table[i][j][1] = 0 if show: print('inner_table[i]6', inner_table[row_no]) # 修复姓名被作为表头 # 2023-02-10 取消修复,避免项目名称、编号,单位、单价等作为了非表头 # surname = [ # "赵", "钱", "孙", "李", "周", "吴", "郑", "王", "冯", "陈", "褚", "卫", "蒋", "沈", "韩", "杨", "朱", "秦", "尤", "许", "何", "吕", "施", "张", "孔", "曹", "严", "华", "金", "魏", "陶", "姜", "戚", "谢", "邹", "喻", "柏", "水", "窦", "章", "云", "苏", "潘", "葛", "奚", "范", "彭", "郎", "鲁", "韦", "昌", "马", "苗", "凤", "花", "方", "俞", "任", "袁", "柳", "酆", "鲍", "史", "唐", "费", "廉", "岑", "薛", "雷", "贺", "倪", "汤", "滕", "殷", "罗", "毕", "郝", "邬", "安", "常", "乐", "于", "时", "傅", "皮", "卞", "齐", "康", "伍", "余", "元", "卜", "顾", "孟", "平", "黄", "和", "穆", "萧", "尹", "姚", "邵", "湛", "汪", "祁", "毛", "禹", "狄", "米", "贝", "明", "臧", "计", "伏", "成", "戴", "谈", "宋", "茅", "庞", "熊", "纪", "舒", "屈", "项", "祝", "董", "梁", "杜", "阮", "蓝", "闵", "席", "季", "麻", "强", "贾", "路", "娄", "危", "江", "童", "颜", "郭", "梅", "盛", "林", "刁", "钟", "徐", "邱", "骆", "高", "夏", "蔡", "田", "樊", "胡", "凌", "霍", "虞", "万", "支", "柯", "昝", "管", "卢", "莫", "经", "房", "裘", "缪", "干", "解", "应", "宗", "丁", "宣", "贲", "邓", "郁", "单", "杭", "洪", "包", "诸", "左", "石", "崔", "吉", "钮", "龚", "程", "嵇", "邢", "滑", "裴", "陆", "荣", "翁", "荀", "羊", "於", "惠", "甄", "麴", "家", "封", "芮", "羿", "储", "靳", "汲", "邴", "糜", "松", "井", "段", "富", "巫", "乌", "焦", "巴", "弓", "牧", "隗", "山", "谷", "车", "侯", "宓", "蓬", "全", "郗", "班", "仰", "秋", "仲", "伊", "宫", "宁", "仇", "栾", "暴", "甘", "钭", "厉", "戎", "祖", "武", "符", "刘", "景", "詹", "束", "龙", "叶", "幸", "司", "韶", "郜", "黎", "蓟", "薄", "印", "宿", "白", "怀", "蒲", "邰", "从", "鄂", "索", "咸", "籍", "赖", "卓", "蔺", "屠", "蒙", "池", "乔", "阴", "欎", "胥", "能", "苍", "双", "闻", "莘", "党", "翟", "谭", "贡", "劳", "逄", "姬", "申", "扶", "堵", "冉", "宰", "郦", "雍", "舄", "璩", "桑", "桂", "濮", "牛", "寿", "通", "边", "扈", "燕", "冀", "郏", "浦", "尚", "农", "温", "别", "庄", "晏", "柴", "瞿", "阎", "充", "慕", "连", "茹", "习", "宦", "艾", "鱼", "容", "向", "古", "易", "慎", "戈", "廖", "庾", "终", "暨", "居", "衡", "步", "都", "耿", "满", "弘", "匡", "国", "文", "寇", "广", "禄", "阙", "东", "殴", "殳", "沃", "利", "蔚", "越", "夔", "隆", "师", "巩", "厍", "聂", "晁", "勾", "敖", "融", "冷", "訾", "辛", "阚", "那", "简", "饶", "空", "曾", "毋", "沙", "乜", "养", "鞠", "须", "丰", "巢", "关", "蒯", "相", "查", "後", "荆", "红", "游", "竺", "权", "逯", "盖", "益", "桓", "公", "万俟", "司马", "上官", "欧阳", "夏侯", "诸葛", "闻人", "东方", "赫连", "皇甫", "尉迟", "公羊", "澹台", "公冶", "宗政", "濮阳", "淳于", "单于", "太叔", "申屠", "公孙", "仲孙", "轩辕", "令狐", "钟离", "宇文", "长孙", "慕容", "鲜于", "闾丘", "司徒", "司空", "亓官", "司寇", "仉", "督", "子车", "颛孙", "端木", "巫马", "公西", "漆雕", "乐正", "壤驷", "公良", "拓跋", "夹谷", "宰父", "谷梁", "晋", "楚", "闫", "法", "汝", "鄢", "涂", "钦", "段干", "百里", "东郭", "南门", "呼延", "归", "海", "羊舌", "微生", "岳", "帅", "缑", "亢", "况", "后", "有", "琴", "梁丘", "左丘", "东门", "西门", "商", "牟", "佘", "佴", "伯", "赏", "南宫", "墨", "哈", "谯", "笪", "年", "爱", "阳", "佟", "第五", "言", "福", # ] # for i in range(len(inner_table)): # for j in range(len(inner_table[i])): # if inner_table[i][j][1] == 1 \ # and 2 <= len(inner_table[i][j][0]) <= 4 \ # and (inner_table[i][j][0][0] in surname or inner_table[i][j][0][:2] in surname) \ # and re.search("[^\u4e00-\u9fa5]", inner_table[i][j][0]) is None: # inner_table[i][j][1] = 0 return inner_table def sliceTable(inner_table,fix_value="~~"): #进行分块 height = len(inner_table) width = len(inner_table[0]) head_list = [] head_list.append(0) last_head = None last_is_same_value = False for h in range(height): is_all_key = True#是否是全表头行 is_all_value = True#是否是全属性值 is_same_with_lastHead = True#和上一行的结构是否相同 is_same_value=True#一行的item都一样 #is_same_first_item = True#与上一行的第一项是否相同 same_value = inner_table[h][0][0] for w in range(width): if last_head is not None: if inner_table[h-1][w][0] != fix_value and inner_table[h-1][w][0] != "" and inner_table[h-1][w][1] == 0: is_all_key = False if inner_table[h][w][1]==1: is_all_value = False if inner_table[h][w][1]!= inner_table[h-1][w][1]: is_same_with_lastHead = False if inner_table[h][w][0]!=fix_value and inner_table[h][w][0]!=same_value: is_same_value = False else: if re.search("\d+",same_value) is not None: is_same_value = False if h>0 and inner_table[h][0][0]!=inner_table[h-1][0][0]: is_same_first_item = False last_head = h if last_is_same_value: last_is_same_value = is_same_value continue if is_same_value: # 该块只有表头一行不合法 if h - head_list[-1] > 1: head_list.append(h) last_is_same_value = is_same_value continue if not is_all_key: if not is_same_with_lastHead: # 该块只有表头一行不合法 if h - head_list[-1] > 1 or not is_all_value: # 20260331补充 整行是表头分块 修复 748505279 第二行才是表头分块错误 749002477 第一行最后一格为空非 head_list.append(h) head_list.append(height) return head_list def setHead_initem(inner_table,pat_head,fix_value="~~",prob_min=0.5): set_item = set() height = len(inner_table) width = len(inner_table[0]) empty_set = set() for i in range(height): for j in range(width): item = inner_table[i][j][0] if item.strip()=="": empty_set.add(item) else: set_item.add(item) list_item = list(set_item) if list_item: x = [] for item in list_item: x.append(getPredictor("form").encode(item)) predict_y = getPredictor("form").predict(np.array(x),type="item") _dict = dict() for item,values in zip(list_item,list(predict_y)): _dict[item] = values[1] # print("##",item,values) #print(_dict) for i in range(height): for j in range(width): item = inner_table[i][j][0] if item not in empty_set: inner_table[i][j][1] = 1 if _dict[item]>prob_min else (1 if re.search(pat_head,item) is not None and len(item)<8 else 0) # print("=====") # for item in inner_table: # print(item) # print("======") repairTable(inner_table) head_list = sliceTable(inner_table) return inner_table,head_list def set_head_model(inner_table, show=0): origin_inner_table = copy.deepcopy(inner_table) for i in range(len(inner_table)): for j in range(len(inner_table[i])): # 删掉单格前后符号,以免影响表头预测 col = inner_table[i][j][0] col = re.sub("^[^\u4e00-\u9fa5a-zA-Z0-9]+", "", col) col = re.sub("[^\u4e00-\u9fa5a-zA-Z0-9]+$", "", col) inner_table[i][j] = col # 模型预测表头 # predict_list = predict(inner_table) start_time = time.time() predict_list = predict(inner_table) # print('table head predict cost: ', time.time()-start_time) # 组合结果 for i in range(len(inner_table)): if i == 0 and (inner_table[i] == ['序号', '内容', '说明与要求'] or inner_table[i] == ['序号', '公告事项', '内容']): # 修复 659860507 659817551 这种只识别第一行表头,第二列非表头导致解析错误问题 inner_table[i] = [[it, 0] for it in ['序号', '内容', '说明与要求']] flag = 1 for j in range(1, len(inner_table)): # 判断是否所有行都有3个格 if len(inner_table[j]) != 3: flag = 0 if flag: for j in range(1, len(inner_table)): inner_table[j] = [[t1, t2] for t1,t2 in zip(inner_table[j], [0, 1, 0])] break continue for j in range(len(inner_table[i])): inner_table[i][j] = [origin_inner_table[i][j][0], int(predict_list[i][j])] if origin_inner_table[i][j][0] in ['主要环境影响及预防或者减轻不良环境影响的对策和措施', '建设单位或地方政府作出的相关环保承诺', '公众反馈意见的联系方式', '区县', '项目领域', '成本/收入', '覆盖倍数', '会计所', '律所','建设期', "发行时间" ,"批次" ,"发行额" ,"发行利率" ,"所属债券" ,"专项债作资本金发行额" ,"调整记录","资产面积", "交易底价","交易地点","竞得者", "谈判项目", "资产名称","单元号","交易面积","承租人","租赁单价"] and predict_list[i][j]!=1: inner_table[i][j] = [origin_inner_table[i][j][0], 1] elif predict_list[i][j]!=1 and (re.search('^拟?(中标|中选|成交|承包|(参与)?投标|招标|采购|招租|发包|业主|竞投)(单位|人)(名称|地址|电话)?$' '|^拟?(中标|中选|成交)(供应商|金额|价格|日期)((万?元))?$|^项目(名称|编号)$', origin_inner_table[i][j][0]) or re.match('(?[\d一二三四五六七八九十][\.、)]((招标|采购|\w{2,4})?项目名称|(招标|采购)人|项目概况|估算投资|预计招标时间|招标内容|其他)', origin_inner_table[i][j][0])): inner_table[i][j] = [origin_inner_table[i][j][0], 1] elif origin_inner_table[i][j][0] in ['经评审的最低评标价法'] and predict_list[i][j]==1: inner_table[i][j] = [origin_inner_table[i][j][0], 0] if show: print(json.dumps(inner_table, ensure_ascii=False)) print("="*80) print("table_head before repair") for r in inner_table: print('row', r) print("="*80) # 表头修正 # repairTable(inner_table) inner_table = table_head_repair_process(inner_table, docid) # 组合结果 for i in range(len(inner_table)): for j in range(len(inner_table[i])): inner_table[i][j] = [origin_inner_table[i][j][0], int(inner_table[i][j][1])] if show: print("table_head after repair") for r in inner_table: print('row', r) print("="*80) # 按表头分割表格 head_list = sliceTable(inner_table) return inner_table, head_list def setHead_incontext(inner_table,pat_head,fix_value="~~",prob_min=0.5): data_x,data_position = getPredictor("form").getModel("context").encode(inner_table) predict_y = getPredictor("form").getModel("context").predict(data_x) for _position,_y in zip(data_position,predict_y): _w = _position[0] _h = _position[1] if _y[1]>prob_min: inner_table[_h][_w][1] = 1 else: inner_table[_h][_w][1] = 0 _item = inner_table[_h][_w][0] if re.search(pat_head,_item) is not None and len(_item)<8: inner_table[_h][_w][1] = 1 # print("=====") # for item in inner_table: # print(item) # print("======") height = len(inner_table) width = len(inner_table[0]) for i in range(height): for j in range(width): if re.search("[::]$", inner_table[i][j][0]) and len(inner_table[i][j][0])<8: inner_table[i][j][1] = 1 repairTable(inner_table) head_list = sliceTable(inner_table) # print("inner_table:",inner_table) return inner_table,head_list #设置表头 def setHead_inline(inner_table,prob_min=0.64): pad_row = "@@" pad_col = "##" removePadding(inner_table, pad_row, pad_col) pad_pattern = re.compile(pad_row+"|"+pad_col) height = len(inner_table) width = len(inner_table[0]) head_list = [] head_list.append(0) #行表头 is_head_last = False for i in range(height): is_head = False is_long_value = False #判断是否是全padding值 is_same_value = True same_value = inner_table[i][0][0] for j in range(width): if inner_table[i][j][0]!=same_value and inner_table[i][j][0]!=pad_row: is_same_value = False break #predict is head or not with model temp_item = "" for j in range(width): temp_item += inner_table[i][j][0]+"|" temp_item = re.sub(pad_pattern,"",temp_item) form_prob = getPredictor("form").predict(formEncoding(temp_item,expand=True),type="line") if form_prob is not None: if form_prob[0][1]>prob_min: is_head = True else: is_head = False #print(temp_item,form_prob) if len(inner_table[i][0][0])>40: is_long_value = True if is_head or is_long_value or is_same_value: #不把连续表头分开 if not is_head_last: head_list.append(i) if is_long_value or is_same_value: head_list.append(i+1) if is_head: for j in range(width): inner_table[i][j][1] = 1 is_head_last = is_head head_list.append(height) #列表头 for i in range(len(head_list)-1): head_begin = head_list[i] head_end = head_list[i+1] #最后一列不设置为列表头 for i in range(width-1): is_head = False #predict is head or not with model temp_item = "" for j in range(head_begin,head_end): temp_item += inner_table[j][i][0]+"|" temp_item = re.sub(pad_pattern,"",temp_item) form_prob = getPredictor("form").predict(formEncoding(temp_item,expand=True),type="line") if form_prob is not None: if form_prob[0][1]>prob_min: is_head = True else: is_head = False if is_head: for j in range(head_begin,head_end): inner_table[j][i][1] = 2 addPadding(inner_table, pad_row, pad_col) return inner_table,head_list #设置表头 def setHead_withRule(inner_table,pattern,pat_value,count): height = len(inner_table) width = len(inner_table[0]) head_list = [] head_list.append(0) #行表头 is_head_last = False for i in range(height): set_match = set() is_head = False is_long_value = False is_same_value = True same_value = inner_table[i][0][0] for j in range(width): if inner_table[i][j][0]!=same_value: is_same_value = False break for j in range(width): if re.search(pat_value,inner_table[i][j][0]) is not None: is_head = False break str_find = re.findall(pattern,inner_table[i][j][0]) if len(str_find)>0: set_match.add(inner_table[i][j][0]) if len(set_match)>=count: is_head = True if len(inner_table[i][0][0])>40: is_long_value = True if is_head or is_long_value or is_same_value: if not is_head_last: head_list.append(i) if is_head: for j in range(width): inner_table[i][j][1] = 1 is_head_last = is_head head_list.append(height) #列表头 for i in range(len(head_list)-1): head_begin = head_list[i] head_end = head_list[i+1] #最后一列不设置为列表头 for i in range(width-1): set_match = set() is_head = False for j in range(head_begin,head_end): if re.search(pat_value,inner_table[j][i][0]) is not None: is_head = False break str_find = re.findall(pattern,inner_table[j][i][0]) if len(str_find)>0: set_match.add(inner_table[j][i][0]) if len(set_match)>=count: is_head = True if is_head: for j in range(head_begin,head_end): inner_table[j][i][1] = 2 return inner_table,head_list #取得表格的处理方向 def getDirect(inner_table,begin,end): ''' column_head = set() row_head = set() widths = len(inner_table[0]) for height in range(begin,end): for width in range(widths): if inner_table[height][width][1] ==1: row_head.add(height) if inner_table[height][width][1] ==2: column_head.add(width) company_pattern = re.compile("公司") if 0 in column_head and begin not in row_head: return "column" if 0 in column_head and begin in row_head: for height in range(begin,end): count = 0 count_flag = True for width_index in range(width): if inner_table[height][width_index][1]==0: if re.search(company_pattern,inner_table[height][width_index][0]) is not None: count += 1 else: count_flag = False if count_flag and count>=2: return "column" return "row" ''' count_row_keys = 0 count_column_keys = 0 width = len(inner_table[0]) if begin=2: return "column" # if count_column_keys>count_row_keys: #2022/2/15 此项不够严谨,造成很多错误,故取消 # return "column" return "row" #根据表格处理方向生成句子, def getTableText(inner_table,head_list,key_direct=False): # packPattern = "(标包|[标包][号段名])" packPattern = "(标包|标的|标项|品目|[标包][号段名]|((项目|物资|设备|场次|标段|标的|产品)(名称)))" # 2020/11/23 大网站规则,补充采购类包名 rankPattern = "(排名|排序|名次|序号|评标结果|评审结果|是否中标|推荐(意见|情况)|评标情况|推荐顺序|选取(情况|说明))" # 2020/11/23 大网站规则,添加序号为排序 entityPattern = "((候选|[中投]标|报价)(单位|公司|人|供应商))|供应商名称" moneyPattern = "([中投]标|报价)(金额|价)" height = len(inner_table) width = len(inner_table[0]) text = "" for head_i in range(len(head_list)-1): head_begin = head_list[head_i] head_end = head_list[head_i+1] direct = getDirect(inner_table, head_begin, head_end) #若只有一行,则直接按行读取 if head_end-head_begin==1: text_line = "" for i in range(head_begin,head_end): for w in range(len(inner_table[i])): if inner_table[i][w][1]==1: _punctuation = ":" else: _punctuation = "," #2021/12/15 统一为中文标点,避免 206893924 国际F座1108,1,009,197.49元 if w>0: if inner_table[i][w][0]!= inner_table[i][w-1][0]: text_line += inner_table[i][w][0]+_punctuation else: text_line += inner_table[i][w][0]+_punctuation text_line = text_line+"。" if text_line!="" else text_line text += text_line else: #构建一个共现矩阵 table_occurence = [] for i in range(head_begin,head_end): line_oc = [] for j in range(width): cell = inner_table[i][j] line_oc.append({"text":cell[0],"type":cell[1],"occu_count":0,"left_head":"","top_head":"","left_dis":0,"top_dis":0}) table_occurence.append(line_oc) occu_height = len(table_occurence) occu_width = len(table_occurence[0]) if len(table_occurence)>0 else 0 #为每个属性值寻找表头 for i in range(occu_height): for j in range(occu_width): cell = table_occurence[i][j] #是属性值 if cell["type"]==0 and cell["text"]!="": left_head = "" top_head = "" find_flag = False temp_head = "" for loop_i in range(1,i+1): if not key_direct: key_values = [1,2] else: key_values = [1] if table_occurence[i-loop_i][j]["type"] in key_values: if find_flag: if table_occurence[i-loop_i][j]["text"]!=temp_head: top_head = table_occurence[i-loop_i][j]["text"]+":"+top_head else: top_head = table_occurence[i-loop_i][j]["text"]+":"+top_head find_flag = True temp_head = table_occurence[i-loop_i][j]["text"] table_occurence[i-loop_i][j]["occu_count"] += 1 else: #找到表头后遇到属性值就返回 if find_flag: break cell["top_head"] += top_head find_flag = False temp_head = "" for loop_j in range(1,j+1): if not key_direct: key_values = [1,2] else: key_values = [2] if table_occurence[i][j-loop_j]["type"] in key_values: if find_flag: if table_occurence[i][j-loop_j]["text"]!=temp_head: left_head = table_occurence[i][j-loop_j]["text"]+":"+left_head else: left_head = table_occurence[i][j-loop_j]["text"]+":"+left_head find_flag = True temp_head = table_occurence[i][j-loop_j]["text"] table_occurence[i][j-loop_j]["occu_count"] += 1 else: if find_flag: break cell["left_head"] += left_head if direct=="row": for i in range(occu_height): pack_text = "" rank_text = "" entity_text = "" text_line = "" money_text = "" #在同一句话中重复的可以去掉 text_set = set() head = "" last_text = "" for j in range(width): cell = table_occurence[i][j] if cell["type"]==0 or (cell["type"]==1 and cell["occu_count"]==0): cell = table_occurence[i][j] head = (cell["top_head"]+":") if len(cell["top_head"])>0 else "" if re.search("[单报标限总]价|金额|成交报?价|报价|供应商|候选人|中标人|[利费]率|负责人|工期|服务(期限?|年限|时间|日期|周期)|(履约|履行)期限|合同(期限?|(完成|截止)(日期|时间))", head): head = cell["left_head"] + head else: head += cell["left_head"] if str(head+cell["text"]) in text_set: continue if re.search(packPattern,head) is not None: pack_text += head+cell["text"]+"," elif re.search(rankPattern,head) is not None and re.search('(排名|排序|名次|顺序):?第?[\d一二三]', rank_text)==None: # 2020/11/23 大网站规则发现问题,if 改elif 20240620修复同时有排名及评标情况造成错误 #排名替换为同一种表达 rank_text += head+cell["text"]+"," #print(rank_text) elif re.search(entityPattern,head) is not None: entity_text += head+cell["text"]+"," #print(entity_text) else: if re.search(moneyPattern,head) is not None and entity_text!="": money_text += head+cell["text"]+"," else: text_line += head+cell["text"]+"," text_set.add(str(head+cell["text"])) last_text = cell['text'] tr_text = pack_text+rank_text+entity_text+money_text+text_line text += pack_text+rank_text+entity_text+money_text+text_line # text = text[:-1] + "。" if len(text) > 0 else text if len(text_set-set([' ']))==1 and head == '' and len(last_text)< 25: # 修复367694716分两行表达 text = text if re.search('\w$', text[:-1]) else text[:-1] elif (width == 2 or len(text_set)==1) and head != '' and len(tr_text)<50: # 修复494731937只有两行的,分句不合理 text = text if re.search('\w$', text[:-1]) else text[:-1] else: text = text[:-1] + "。" else: for j in range(occu_width): pack_text = "" rank_text = "" entity_text = "" text_line = "" text_set = set() for i in range(occu_height): cell = table_occurence[i][j] if cell["type"]==0 or (cell["type"]==1 and cell["occu_count"]==0): cell = table_occurence[i][j] head = (cell["left_head"]+"") if len(cell["left_head"])>0 else "" if re.search("[单报标限总]价|金额|成交报?价|报价|供应商|候选人|中标人|[利费]率|负责人|工期|服务(期限?|年限|时间|日期|周期)|(履约|履行)期限|合同(期限?|(完成|截止)(日期|时间))", head): head = cell["top_head"] + head else: head += cell["top_head"] if str(head+cell["text"]) in text_set: continue if re.search(packPattern,head) is not None: pack_text += head+cell["text"]+"," elif re.search(rankPattern,head) is not None: # 2020/11/23 大网站规则发现问题,if 改elif #排名替换为同一种表达 rank_text += head+cell["text"]+"," #print(rank_text) elif re.search(entityPattern,head) is not None and \ re.search('业绩|资格|条件',head)==None and re.search('业绩',cell["text"])==None : #2021/10/19 解决包含业绩的行调到前面问题 entity_text += head+cell["text"]+"," #print(entity_text) else: text_line += head+cell["text"]+"," text_set.add(str(head+cell["text"])) text += pack_text+rank_text+entity_text+text_line text = text[:-1]+"。" if len(text)>0 else text # if direct=="row": # for i in range(head_begin,head_end): # pack_text = "" # rank_text = "" # entity_text = "" # text_line = "" # #在同一句话中重复的可以去掉 # text_set = set() # for j in range(width): # cell = inner_table[i][j] # #是属性值 # if cell[1]==0 and cell[0]!="": # head = "" # # find_flag = False # temp_head = "" # for loop_i in range(0,i+1-head_begin): # if not key_direct: # key_values = [1,2] # else: # key_values = [1] # if inner_table[i-loop_i][j][1] in key_values: # if find_flag: # if inner_table[i-loop_i][j][0]!=temp_head: # head = inner_table[i-loop_i][j][0]+":"+head # else: # head = inner_table[i-loop_i][j][0]+":"+head # find_flag = True # temp_head = inner_table[i-loop_i][j][0] # else: # #找到表头后遇到属性值就返回 # if find_flag: # break # # find_flag = False # temp_head = "" # # # # for loop_j in range(1,j+1): # if not key_direct: # key_values = [1,2] # else: # key_values = [2] # if inner_table[i][j-loop_j][1] in key_values: # if find_flag: # if inner_table[i][j-loop_j][0]!=temp_head: # head = inner_table[i][j-loop_j][0]+":"+head # else: # head = inner_table[i][j-loop_j][0]+":"+head # find_flag = True # temp_head = inner_table[i][j-loop_j][0] # else: # if find_flag: # break # # if str(head+inner_table[i][j][0]) in text_set: # continue # if re.search(packPattern,head) is not None: # pack_text += head+inner_table[i][j][0]+"," # elif re.search(rankPattern,head) is not None: # 2020/11/23 大网站规则发现问题,if 改elif # #排名替换为同一种表达 # rank_text += head+inner_table[i][j][0]+"," # #print(rank_text) # elif re.search(entityPattern,head) is not None: # entity_text += head+inner_table[i][j][0]+"," # #print(entity_text) # else: # text_line += head+inner_table[i][j][0]+"," # text_set.add(str(head+inner_table[i][j][0])) # text += pack_text+rank_text+entity_text+text_line # text = text[:-1]+"。" if len(text)>0 else text # else: # for j in range(width): # # rank_text = "" # entity_text = "" # text_line = "" # text_set = set() # for i in range(head_begin,head_end): # cell = inner_table[i][j] # #是属性值 # if cell[1]==0 and cell[0]!="": # find_flag = False # head = "" # temp_head = "" # # for loop_j in range(1,j+1): # if not key_direct: # key_values = [1,2] # else: # key_values = [2] # if inner_table[i][j-loop_j][1] in key_values: # if find_flag: # if inner_table[i][j-loop_j][0]!=temp_head: # head = inner_table[i][j-loop_j][0]+":"+head # else: # head = inner_table[i][j-loop_j][0]+":"+head # find_flag = True # temp_head = inner_table[i][j-loop_j][0] # else: # if find_flag: # break # find_flag = False # temp_head = "" # for loop_i in range(0,i+1-head_begin): # if not key_direct: # key_values = [1,2] # else: # key_values = [1] # if inner_table[i-loop_i][j][1] in key_values: # if find_flag: # if inner_table[i-loop_i][j][0]!=temp_head: # head = inner_table[i-loop_i][j][0]+":"+head # else: # head = inner_table[i-loop_i][j][0]+":"+head # find_flag = True # temp_head = inner_table[i-loop_i][j][0] # else: # if find_flag: # break # if str(head+inner_table[i][j][0]) in text_set: # continue # if re.search(rankPattern,head) is not None: # rank_text += head+inner_table[i][j][0]+"," # #print(rank_text) # elif re.search(entityPattern,head) is not None: # entity_text += head+inner_table[i][j][0]+"," # #print(entity_text) # else: # text_line += head+inner_table[i][j][0]+"," # text_set.add(str(head+inner_table[i][j][0])) # text += rank_text+entity_text+text_line # text = text[:-1]+"。" if len(text)>0 else text if text.endswith(',。'): text = re.sub(',+。', '。', text) elif text.endswith(','): text = text.rstrip(',') + '。' else: text += '。' # 20260401 表格末尾加句号与其他内容分开 749870793 避免表格与外面混淆 return text def get_table_text_kv(inner_table, head_list, key_direct=False): packPattern = "(标包|标的|标项|品目|[标包][号段名]|((项目|物资|设备|场次|标段|标的|产品)(名称)))" # 2020/11/23 大网站规则,补充采购类包名 rankPattern = "(排名|排序|名次|序号|评标结果|评审结果|是否中标|推荐意见|评标情况|推荐顺序|选取(情况|说明))" # 2020/11/23 大网站规则,添加序号为排序 entityPattern = "((候选|[中投]标|报价)(单位|公司|人|供应商))|供应商名称" moneyPattern = "([中投]标|报价)(金额|价)" width = len(inner_table[0]) text = "" all_table_occurence = [] for head_i in range(len(head_list) - 1): head_begin = head_list[head_i] head_end = head_list[head_i + 1] direct = getDirect(inner_table, head_begin, head_end) # print(inner_table[head_begin:head_end]) # print('direct', direct) # 构建一个共现矩阵 table_occurence = [] for i in range(head_begin, head_end): line_oc = [] for j in range(width): cell = inner_table[i][j] line_oc.append( {"text": cell[0], "type": cell[1], "occu_count": 0, "left_head": "", "top_head": "", "left_dis": 0, "top_dis": 0, "text_row_index": i, "text_col_index": j }) table_occurence.append(line_oc) occu_height = len(table_occurence) occu_width = len(table_occurence[0]) if len(table_occurence) > 0 else 0 # 为每个属性值寻找表头 for i in range(occu_height): for j in range(occu_width): cell = table_occurence[i][j] # 是属性值 if cell["type"] == 0 and cell["text"] != "": left_head = "" top_head = "" find_flag = False temp_head = "" head_row_col_list = [] for loop_i in range(1, i + 1): if not key_direct: key_values = [1, 2] else: key_values = [1] if table_occurence[i - loop_i][j]["type"] in key_values: if find_flag: if table_occurence[i - loop_i][j]["text"] != temp_head: if cell.get("top_head_list"): cell["top_head_list"] += [table_occurence[i - loop_i][j]["text"] + ":"] else: cell["top_head_list"] = [table_occurence[i - loop_i][j]["text"] + ":"] top_head = table_occurence[i - loop_i][j]["text"] + ":" + top_head head_row_col_list.append([i - loop_i, j]) else: if cell.get("top_head_list"): cell["top_head_list"] += [table_occurence[i - loop_i][j]["text"] + ":"] else: cell["top_head_list"] = [table_occurence[i - loop_i][j]["text"] + ":"] top_head = table_occurence[i - loop_i][j]["text"] + ":" + top_head head_row_col_list.append([i - loop_i, j]) find_flag = True temp_head = table_occurence[i - loop_i][j]["text"] table_occurence[i - loop_i][j]["occu_count"] += 1 else: # 找到表头后遇到属性值就返回 if find_flag: break cell["top_head"] += top_head if cell.get("top_head_row_index"): cell["top_head_row_index"] += [x[0] for x in head_row_col_list] else: cell["top_head_row_index"] = [x[0] for x in head_row_col_list] if cell.get("top_head_col_index"): cell["top_head_col_index"] += [x[1] for x in head_row_col_list] else: cell["top_head_col_index"] = [x[1] for x in head_row_col_list] find_flag = False temp_head = "" head_row_col_list = [] for loop_j in range(1, j + 1): if not key_direct: key_values = [1, 2] else: key_values = [2] if table_occurence[i][j - loop_j]["type"] in key_values: if find_flag: if table_occurence[i][j - loop_j]["text"] != temp_head: if cell.get("left_head_list"): cell["left_head_list"] += [table_occurence[i][j - loop_j]["text"] + ":"] else: cell["left_head_list"] = [table_occurence[i][j - loop_j]["text"] + ":"] left_head = table_occurence[i][j - loop_j]["text"] + ":" + left_head head_row_col_list.append([i, j - loop_j]) else: if cell.get("left_head_list"): cell["left_head_list"] += [table_occurence[i][j - loop_j]["text"] + ":"] else: cell["left_head_list"] = [table_occurence[i][j - loop_j]["text"] + ":"] left_head = table_occurence[i][j - loop_j]["text"] + ":" + left_head head_row_col_list.append([i, j - loop_j]) find_flag = True temp_head = table_occurence[i][j - loop_j]["text"] table_occurence[i][j - loop_j]["occu_count"] += 1 else: if find_flag: break cell["left_head"] += left_head if cell.get("left_head_row_index"): cell["left_head_row_index"] += [x[0] for x in head_row_col_list] else: cell["left_head_row_index"] = [x[0] for x in head_row_col_list] if cell.get("left_head_col_index"): cell["left_head_col_index"] += [x[1] for x in head_row_col_list] else: cell["left_head_col_index"] = [x[1] for x in head_row_col_list] # 连接表头和属性值 if direct == "row": for i in range(occu_height): pack_text = "" rank_text = "" entity_text = "" text_line = "" money_text = "" # 在同一句话中重复的可以去掉 text_set = set() head = "" last_text = "" pack_text_location = [] rank_text_location = [] entity_text_location = [] text_line_location = [] money_text_location = [] for j in range(width): cell = table_occurence[i][j] if cell["type"] == 0 or (cell["type"] == 1 and cell["occu_count"] == 0): cell = table_occurence[i][j] head = (cell["top_head"] + ":") if len(cell["top_head"]) > 0 else "" now_top_head = copy.deepcopy(head) now_left_head = copy.deepcopy(cell["left_head"]) if re.search( "[单报标限总]价|金额|成交报?价|报价|供应商|候选人|中标人|[利费]率|负责人|工期|服务(期限?|年限|时间|日期|周期)|(" "履约|履行)期限|合同(期限?|(完成|截止)(日期|时间))", head): head = cell["left_head"] + head left_first = 1 else: head += cell["left_head"] left_first = 0 # print('len(text), len(sub_text), len(head)', cell["text"], len(text), len(sub_text), len(head)) # print('text111', text) # print('pack_text, rank_text, entity_text, money_text, text_line', '1'+pack_text, '2'+rank_text, '3'+entity_text, '4'+money_text, '5'+text_line) # print('head', head) # print('sub_text111', sub_text) if str(head + cell["text"]) in text_set: cell['drop'] = 1 continue if re.search(packPattern, head) is not None: pack_text += head + cell["text"] + "," pack_text_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]] # 2020/11/23 大网站规则发现问题,if 改elif 20240620修复同时有排名及评标情况造成错误 elif re.search(rankPattern, head) is not None and re.search('(排名|排序|名次|顺序):?第?[\d一二三]', rank_text) is None: # 排名替换为同一种表达 rank_text += head + cell["text"] + "," rank_text_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]] elif re.search(entityPattern, head) is not None: entity_text += head + cell["text"] + "," entity_text_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]] else: if re.search(moneyPattern, head) is not None and entity_text != "": money_text += head + cell["text"] + "," money_text_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]] else: text_line += head + cell["text"] + "," text_line_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]] text_set.add(str(head + cell["text"])) last_text = cell['text'] # 计算key value在sentence的index head_location_list = pack_text_location + rank_text_location + entity_text_location + text_line_location + money_text_location current_loc = 0 for ii, jj, head_text, now_left_head, now_top_head, left_first in head_location_list: cell = table_occurence[ii][jj] # 左表头先于右表头 if left_first: cell['left_head_sen_index'] = len(text) + current_loc cell['top_head_sen_index'] = len(text) + current_loc + len(now_left_head) else: cell['left_head_sen_index'] = len(text) + current_loc + len(now_top_head) cell['top_head_sen_index'] = len(text) + current_loc cell['text_sen_index'] = len(text) + current_loc + len(now_left_head + now_top_head) current_loc += len(head_text) tr_text = pack_text + rank_text + entity_text + money_text + text_line text += pack_text + rank_text + entity_text + money_text + text_line # 修复367694716分两行表达 if len(text_set - set([' '])) == 1 and head == '' and len(last_text) < 25: text = text if re.search('\w$', text[:-1]) else text[:-1] # 修复494731937只有两行的,分句不合理 elif (width == 2 or len(text_set) == 1) and head != '' and len(tr_text) < 50: text = text if re.search('\w$', text[:-1]) else text[:-1] else: text = text[:-1] + "。" else: for j in range(occu_width): pack_text = "" rank_text = "" entity_text = "" text_line = "" text_set = set() pack_text_location = [] rank_text_location = [] entity_text_location = [] text_line_location = [] money_text_location = [] for i in range(occu_height): cell = table_occurence[i][j] if cell["type"] == 0 or (cell["type"] == 1 and cell["occu_count"] == 0): cell = table_occurence[i][j] head = (cell["left_head"] + "") if len(cell["left_head"]) > 0 else "" now_top_head = copy.deepcopy(cell["top_head"]) now_left_head = copy.deepcopy(head) if re.search("[单报标限总]价|金额|成交报?价|报价|供应商|候选人|中标人|[利费]率|负责人|工期|服务(期限?|年限|时间|日期|周期)|(履约|履行)期限|合同(期限?|(完成|截止)(日期|时间))", head): head = cell["top_head"] + head left_first = 0 else: head += cell["top_head"] left_first = 1 if str(head + cell["text"]) in text_set: cell['drop'] = 1 continue if re.search(packPattern, head) is not None: pack_text += head + cell["text"] + "," pack_text_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]] # 2020/11/23 大网站规则发现问题,if 改elif elif re.search(rankPattern, head) is not None: # 排名替换为同一种表达 rank_text += head + cell["text"] + "," rank_text_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]] # 2021/10/19 解决包含业绩的行调到前面问题 elif re.search(entityPattern, head) is not None and \ re.search('业绩|资格|条件', head) is None and re.search('业绩', cell["text"]) is None: entity_text += head + cell["text"] + "," entity_text_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]] else: text_line += head + cell["text"] + "," text_line_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]] text_set.add(str(head + cell["text"])) # 计算key value在sentence的index head_location_list = pack_text_location + rank_text_location + entity_text_location + text_line_location + money_text_location current_loc = 0 for ii, jj, head_text, now_left_head, now_top_head, left_first in head_location_list: cell = table_occurence[ii][jj] # 左表头先于右表头 if left_first: cell['left_head_sen_index'] = len(text) + current_loc cell['top_head_sen_index'] = len(text) + current_loc + len(now_left_head) else: cell['left_head_sen_index'] = len(text) + current_loc + len(now_top_head) cell['top_head_sen_index'] = len(text) + current_loc cell['text_sen_index'] = len(text) + current_loc + len(now_left_head + now_top_head) current_loc += len(head_text) text += pack_text + rank_text + entity_text + text_line text = text[:-1] + "。" if len(text) > 0 else text all_table_occurence += table_occurence return text, all_table_occurence def process_dict(text, table): kv_list = [] kv_dict_list = [] # print('text', len(text), text, ), # print('table', table) for r_index, row in enumerate(table): for c_index, col in enumerate(row): # print('col', col) if col['type'] == 1: continue if col.get('drop'): continue if not col.get('left_head_list') and not col.get('top_head_list'): _d = { 'value': col['text'], 'value_row_index': col['text_row_index'], 'value_col_index': col['text_col_index'], 'value_sen_index': col['text_sen_index'], 'sen_value': text[col['text_sen_index']:col['text_sen_index'] + len(col['text'])], } kv_dict_list.append(_d) continue if col.get('text_sen_index') and col.get('text_sen_index') >= len(text): # print('continue1') continue if col.get('left_head_list'): # head, head_row_index, head_col_index 按文本顺序排序 zip_list = list( zip(col.get('left_head_list'), col.get('left_head_row_index'), col.get('left_head_col_index'))) zip_list.sort(key=lambda x: (x[1], x[2])) col['left_head_list'], col['left_head_row_index'], col['left_head_col_index'] = zip(*zip_list) last_head = "" for h_index, head in enumerate(col.get('left_head_list')): _d = { 'key': head, 'value': col['text'], 'key_row_index': col['left_head_row_index'][h_index], 'key_col_index': col['left_head_col_index'][h_index], 'key_sen_index': col['left_head_sen_index'] + len(last_head), 'value_row_index': col['text_row_index'], 'value_col_index': col['text_col_index'], 'value_sen_index': col['text_sen_index'], 'sen_key': text[ col['left_head_sen_index'] + len(last_head):col['left_head_sen_index'] + len( last_head) + len(head)], 'sen_value': text[col['text_sen_index']:col['text_sen_index'] + len(col['text'])], } kv_dict_list.append(_d) last_head += head if col.get('top_head_list'): # head, head_row_index, head_col_index 按文本顺序排序 zip_list = list( zip(col.get('top_head_list'), col.get('top_head_row_index'), col.get('top_head_col_index'))) zip_list.sort(key=lambda x: (x[1], x[2])) col['top_head_list'], col['top_head_row_index'], col['top_head_col_index'] = zip(*zip_list) last_head = "" for h_index, head in enumerate(col.get('top_head_list')): _d = { 'key': head, 'value': col['text'], 'key_row_index': col['top_head_row_index'][h_index], 'key_col_index': col['top_head_col_index'][h_index], 'key_sen_index': col['top_head_sen_index'] + len(last_head), 'value_row_index': col['text_row_index'], 'value_col_index': col['text_col_index'], 'value_sen_index': col['text_sen_index'], 'sen_key': text[col['top_head_sen_index'] + len(last_head):col['top_head_sen_index'] + len( last_head) + len(head)], 'sen_value': text[col['text_sen_index']:col['text_sen_index'] + len(col['text'])], } kv_dict_list.append(_d) last_head += head return kv_list, kv_dict_list def removeFix(inner_table,fix_value="~~"): height = len(inner_table) width = len(inner_table[0]) for h in range(height): for w in range(width): if inner_table[h][w][0]==fix_value: inner_table[h][w][0] = "" def trunTable(tbody,in_attachment): # print(tbody.find('tbody')) # 附件中的表格,排除异常错乱的表格 if in_attachment: if tbody.name=='table': _tbody = tbody.find('tbody') if _tbody is None: _tbody = tbody else: _tbody = tbody _td_len_list = [] for _tr in _tbody.find_all(recursive=False): len_td = len(_tr.find_all(recursive=False)) _td_len_list.append(len_td) if _td_len_list: if len(list(set(_td_len_list))) >= 8 or max(_td_len_list) > 100: string_list = [re.sub("\s+","",i)for i in tbody.strings if i and i!='\n'] tbody.string = ",".join(string_list) table_max_len = 30000 tbody.string = tbody.string[:table_max_len] tbody.name = "turntable" if return_kv: return None, None, None return None # fixSpan(tbody) # inner_table = getTable(tbody) # inner_table = fixTable(inner_table) table2list = TableTag2List() return_html_table = True if return_kv else False if return_html_table: inner_table, html_table = table2list.table2list(tbody, segment, return_html_table,return_kv=return_kv) inner_table = fixTable(inner_table) html_table = fixTable(html_table, "") else: inner_table = table2list.table2list(tbody, segment,return_kv=return_kv) inner_table = fixTable(inner_table) if inner_table == []: string_list = [re.sub("\s+", "", i) for i in tbody.strings if i and i != '\n'] tbody.string = ",".join(string_list) table_max_len = 30000 tbody.string = tbody.string[:table_max_len] # log('异常表格直接取全文') tbody.name = "turntable" if return_kv: return None, None, None return None if len(inner_table)>0 and len(inner_table[0])>0: for tr in inner_table: for td in tr: if isinstance(td, str): tbody.string = segment(tbody,final=False) table_max_len = 30000 tbody.string = tbody.string[:table_max_len] # log('异常表格,不做表格处理,直接取全文') tbody.name = "turntable" if return_kv: return None, None, None return None #inner_table,head_list = setHead_withRule(inner_table,pat_head,pat_value,3) #inner_table,head_list = setHead_inline(inner_table) # inner_table, head_list = setHead_initem(inner_table,pat_head) inner_table, head_list = set_head_model(inner_table) # inner_table,head_list = setHead_incontext(inner_table,pat_head) # print("table_head", inner_table) # print("head_list", head_list) # for begin in range(len(head_list[:-1])): # for item in inner_table[head_list[begin]:head_list[begin+1]]: # print(item) # print("====") removeFix(inner_table) # print("----") # print(head_list) # for item in inner_table: # print(item) # print('inner_table111', inner_table) if return_kv: text1, table = get_table_text_kv(inner_table, head_list) tbody.string = text1 kv_list, kv_dict_list = process_dict(text1, table) # html放入dict for kv_dict in kv_dict_list: html = html_table[kv_dict.get('value_row_index')][kv_dict.get('value_col_index')] kv_dict['value_html'] = html else: tbody.string = getTableText(inner_table,head_list) table_max_len = 30000 tbody.string = tbody.string[:table_max_len] if tbody.string.startswith('。'): # 20251224 修复 602074436 关键词与表格分开 中选单位:名称: tbody.string = ',' + tbody.string[1:] # print(tbody.string) tbody.name = "turntable" if return_kv: return inner_table, kv_dict_list, text1 else: return inner_table if return_kv: return None, None, None return None pat_head = re.compile('^(名称|序号|项目|标项|工程|品目[一二三四1234]|第[一二三四1234](标段|名|候选人|中标)|包段|标包|分包|包号|货物|单位|数量|价格|报价|金额|总价|单价|[招投中]标|候选|编号|得分|评委|评分|名次|排名|排序|科室|方式|工期|时间|产品|开始|结束|联系|日期|面积|姓名|证号|备注|级别|地[点址]|类型|代理|制造|企业资质|质量目标|工期目标|(需求|服务|项目|施工|采购|招租|出租|转让|出让|业主|询价|委托|权属|招标|竞得|抽取|承建)(人|方|单位)(名称)?|(供应商|供货商|服务商)(名称)?)$') #pat_head = re.compile('(名称|序号|项目|工程|品目[一二三四1234]|第[一二三四1234](标段|候选人|中标)|包段|包号|货物|单位|数量|价格|报价|金额|总价|单价|[招投中]标|供应商|候选|编号|得分|评委|评分|名次|排名|排序|科室|方式|工期|时间|产品|开始|结束|联系|日期|面积|姓名|证号|备注|级别|地[点址]|类型|代理)') pat_value = re.compile("(\d{2,}.\d{1}|\d+年\d+月|\d{8,}|\d{3,}-\d{6,}|有限[责任]*公司|^\d+$)") list_innerTable = [] # 2022/2/9 删除干扰标签 for tag in soup.find_all('option'): #例子: 216661412 if 'selected' not in tag.attrs: tag.extract() for ul in soup.find_all('ul'): #例子 156439663 多个不同channel 类别的标题 if ul.find_all('li') == ul.findChildren(recursive=False) and len(set(re.findall( '招标公告|中标结果公示|中标候选人公示|招标答疑|开标评标|合同履?约?公示|资格评审', ul.get_text(), re.S)))>3: ul.extract() # tbodies = soup.find_all('table') # 遍历表格中的每个tbody tbodies = [] in_attachment = False if soup.name=="table": tbodies.append((soup,in_attachment)) for _part in soup.find_all(): if _part.name=='table': tbodies.append((_part,in_attachment)) elif _part.name=='div': if 'class' in _part.attrs and "richTextFetch" in _part['class']: in_attachment = True if return_kv and tbodies: tbodies = tbodies[:1] #逆序处理嵌套表格 # print('len(tbodies)1', len(tbodies)) # for tbody_index in range(1,len(tbodies)+1): tbody_index = 1 while tbody_index < len(tbodies)+1: tbody,_in_attachment = tbodies[len(tbodies)-tbody_index] current_index = len(tbodies)-tbody_index if current_index > 0 and tbodies[current_index - 1][0].find_next_sibling() == tbody and len(tbodies[current_index - 1][0].find_all('tr')) == 1 and len(tbody.find_all('tr'))>=1 and len(tbody.tr.find_all(['th', 'td'])) == len( tbodies[current_index - 1][0].tr.find_all(['th', 'td'])): # 处理相邻表格都只有一行的情况;例:526321576 641881540 for row in tbody.find_all('tr'): # 相邻表格只有一行且列数一样合并表格,标签必须是兄弟节点 避免 665587084 这种不是相邻标签的错误合并 if tbodies[current_index - 1][0].tbody: tbodies[current_index - 1][0].tbody.append(row) else: tbodies[current_index - 1][0].append(row) inner_table = trunTable(tbodies[current_index - 1][0], _in_attachment) if inner_table: list_innerTable.append(inner_table) tbody_index += 2 continue inner_table = trunTable(tbody,_in_attachment) list_innerTable.append(inner_table) tbody_index += 1 # tbodies = soup.find_all('tbody') # 遍历表格中的每个tbody tbodies = [] in_attachment = False if soup.name=="tbody" and soup.get_text().strip() != '': # 20250730 修复 642228216 表格分多个tbody且一三tbody只有tr内容为空导致提取失败 tbodies.append((soup,in_attachment)) for _part in soup.find_all(): if _part.name == 'tbody': tbodies.append((_part, in_attachment)) elif _part.name == 'div': if 'class' in _part.attrs and "richTextFetch" in _part['class']: in_attachment = True if return_kv and tbodies: tbodies = tbodies[:1] #逆序处理嵌套表格 tbody_index = 1 # for tbody_index in range(1,len(tbodies)+1): while tbody_index < len(tbodies) + 1: tbody,_in_attachment = tbodies[len(tbodies)-tbody_index] current_index = len(tbodies) - tbody_index if current_index > 0 and tbodies[current_index - 1][0].find_next_sibling() == tbody and len( tbodies[current_index - 1][0].find_all('tr')) == 1 and len(tbody.find_all('tr'))>=1 and len(tbody.tr.find_all(['th', 'td'])) == len( tbodies[current_index - 1][0].tr.find_all(['th', 'td'])) and len(tbody.tr.find_all(['th', 'td'])) > 2: # 处理相邻表格都只有一行的情况;例:526321576 for row in tbody.find_all('tr'): if tbodies[current_index - 1][0].tbody: tbodies[current_index - 1][0].tbody.append(row) else: tbodies[current_index - 1][0].append(row) inner_table = trunTable(tbodies[current_index - 1][0], _in_attachment) if inner_table: list_innerTable.append(inner_table) tbody_index += 2 continue inner_table = trunTable(tbody,_in_attachment) list_innerTable.append(inner_table) tbody_index += 1 if return_kv: kv_list = [] for x in list_innerTable: if x[1] is not None: kv_list.extend(x[1]) text = "" for x in list_innerTable: if x[2] is not None: text += x[2] return soup, kv_list, text return soup # return list_innerTable def table_head_repair_process(_inner_table, docid=None, show=0, show_row_index=0): def pre_process(inner_table): """ 修复前的预处理 """ # 循环处理单元格,一次获取需要的 for i in range(len(inner_table)): for j in range(len(inner_table[i])): # 删除前后逗号 inner_table[i][j][0] = re.sub('^[,, ]+', '', inner_table[i][j][0]) inner_table[i][j][0] = re.sub('[,, ]+$', '', inner_table[i][j][0]) inner_table[i][j][0] = re.sub('[, ]+', '', inner_table[i][j][0]) return inner_table def repair_by_colon(inner_table): """ 根据冒号修复当前格子的表头值 """ # 修复冒号在文本中间的,不能作为表头;(冒号后面需多个字) # 冒号在括号中的除外 # 冒号在最后的,判断后一个格子是否有重复的文字 for i in range(len(inner_table)): for j in range(len(inner_table[i])): _text = inner_table[i][j][0] if len(_text) >= 3 and inner_table[i][j][1] == 1: match = re.search('[::]', _text) if match: start_index, end_index = match.span() if start_index == 0: continue if end_index == len(_text): if len(inner_table[i]) == 2 and j <= len(inner_table[i]) - 2 and inner_table[i][j+1][0] and (_text in inner_table[i][j+1][0] or inner_table[i][j+1][0] in _text): inner_table[i][j][1] = 0 inner_table[i][j+1][1] = 0 else: continue if re.search('[((]', _text[:start_index]) and re.search('[))]', _text[end_index:]): continue m1 = re.search('[\u4e00-\u9fa50-9a-zA-Z]', _text[:start_index]) m2 = re.search('[\u4e00-\u9fa50-9a-zA-Z]', _text[end_index:]) if m1 and m2 and (len(m2.group()) >= 2 or m2.group() in ['是', '否']): inner_table[i][j][1] = 0 return inner_table def repair_by_duplicate(inner_table): """ 根据列重复修复当前格子的表头值 """ # 统计每个值的表头情况 col_head_dict = {} for i in range(len(inner_table)): for j in range(len(inner_table[i])): col = inner_table[i][j] if col[0] in col_head_dict.keys(): col_head_dict[col[0]] += [col[1]] else: col_head_dict[col[0]] = [col[1]] # 多个重复列的预测值不同,以第一个为准 for i in range(len(inner_table)): col = inner_table[i][0] key = col[0] + '\t' + str(0) dup_dict = {} for j in range(len(inner_table[i])): if inner_table[i][j][0] == col[0]: if key in dup_dict.keys(): dup_dict[key] += [j] else: dup_dict[key] = [j] # if inner_table[i][j][1] != col[1]: # if col != inner_table[i][0]: # inner_table[i][j][1] = col[1] # else: # inner_table[i][0][1] = inner_table[i][j][1] # col = inner_table[i][0] else: col = inner_table[i][j] key = col[0] + '\t' + str(j) dup_dict[key] = [j] # print('dup_dict', dup_dict) # for key in dup_dict.keys(): index_list = dup_dict.get(key) if len(index_list) <= 1: continue # 需要表头不同 table_head_list = [] for index in index_list: table_head_list.append(inner_table[i][index][1]) table_head_list = list(set(table_head_list)) if len(table_head_list) <= 1: continue # 若是职业特殊处理 col = key.split('\t')[0] if re.search('([^人]员|工程师|建造师|经理|安全负责人|技术负责人|合同商务负责人)$', col): table_head_flag = 0 # 看是否包含表头 else: table_head_flag = 0 for index in index_list: if inner_table[i][index][1] == 1: table_head_flag = 1 break table_head = None if index_list[0] > 0 and index_list[-1] == len(inner_table[i]) - 1: table_head = inner_table[i][index_list[0]][1] elif index_list[0] > 0 and index_list[-1] != len(inner_table[i]) - 1: # 查看前面是否有表头-非表头表达 is_head_not_head = 0 for k in range(index_list[0]): if k+1 < index_list[0] and inner_table[i][k][1] == 1 and inner_table[i][k+1][1] == 0: is_head_not_head = 1 break # 查看前后有没有表头 start_has_head = 0 end_has_head = 0 if not is_head_not_head: for k in range(index_list[0], len(inner_table[i])): if inner_table[i][k][0] == inner_table[i][index_list[0]][0]: continue if inner_table[i][k][1] == 1: end_has_head = 1 break for k in range(index_list[0]): if inner_table[i][k][0] == inner_table[i][index_list[0]][0]: continue if inner_table[i][k][1] == 1: start_has_head = 1 break head_list = col_head_dict.get(inner_table[i][index_list[0]][0]) if is_head_not_head: table_head = table_head_flag elif len(head_list) >= 4: if head_list.count(0) > head_list.count(1): table_head = 0 else: table_head = 1 elif not start_has_head and not end_has_head: table_head = 0 else: table_head = table_head_flag elif index_list[0] == 0 and index_list[-1] == len(inner_table[i]) - 1: table_head = table_head_flag elif index_list[0] == 0 and index_list[-1] != len(inner_table[i]) - 1: head_list = col_head_dict.get(inner_table[i][index_list[0]][0]) if len(head_list) >= 4: if head_list.count(0) > head_list.count(1): table_head = 0 else: table_head = 1 else: table_head = table_head_flag if table_head is not None: for index in index_list: inner_table[i][index][1] = table_head return inner_table def repair_by_around(inner_table): """ 根据周围的表头值修复当前格子的表头值 """ one_head_index_list = [] zero_head_index_list = [] all_head_index_list = [] one_not_head_index_list = [] no_dup_index_cnt_dict = {} for i in range(len(inner_table)): head_cnt = 0 head_index = None head_dict = {} for j in range(len(inner_table[i])): # 统计表头数 if inner_table[i][j][1] == 1: head_cnt += 1 head_index = j if inner_table[i][j][0] not in ['~~', '', ' ']: if inner_table[i][j][0] in head_dict.keys(): head_dict[inner_table[i][j][0]] += 1 else: head_dict[inner_table[i][j][0]] = 1 no_dup_index_cnt_dict[i] = len(head_dict.keys()) # 表头数list if head_cnt == 0: zero_head_index_list.append(i) elif head_cnt == 1: # 这个单个表头需满足前面有非表头 find_flag = 0 for k in range(head_index): if inner_table[i][k][1] == 0: find_flag = 1 if find_flag and len(head_dict.keys()) > 2: one_head_index_list.append(i) elif head_cnt == len(inner_table[i]): all_head_index_list.append(i) elif head_cnt == len(inner_table[i]) - 1: one_not_head_index_list.append(i) # 第一行为表头,但有一个不为表头,下面行都非表头,表格行数小于4 if 0 in one_not_head_index_list and 1 in zero_head_index_list and len(inner_table) <= 4: # 不相等的列值大于4 diff_col1 = [] for col in inner_table[0]: if col[0] not in diff_col1 and len(col[0]) >= 1: diff_col1.append(col[0]) diff_col2 = [] for col in inner_table[1]: if col[0] not in diff_col2 and len(col[0]) >= 1: diff_col2.append(col[0]) if len(diff_col1) >= 4 and len(diff_col2) >= 4: for j in range(len(inner_table[0])): inner_table[0][j][1] = 1 one_not_head_index_list.remove(0) all_head_index_list.append(0) # 一行很多列且都为表头,则剩下一个也为表头 for i in range(len(inner_table)): if no_dup_index_cnt_dict.get(i) >= 5 and i in one_not_head_index_list: for j in range(len(inner_table[i])): inner_table[i][j][1] = 1 # 一行很多列且都不为表头,则剩下一个也不为表头,除了第一个 for i in range(len(inner_table)): if no_dup_index_cnt_dict.get(i) >= 5 and i in one_head_index_list \ and inner_table[i][0][1] != 1 and inner_table[i][0][0] != '': for j in range(len(inner_table[i])): inner_table[i][j][1] = 0 # 一整个大表格,第一行为表头,下面行中有个别格子被识别为表头 # 候选人后面修复 for index in one_head_index_list: if (index - 1 in zero_head_index_list and index - 2 in zero_head_index_list) \ or (index - 1 in zero_head_index_list and index - 2 in all_head_index_list) \ or (index - 1 in all_head_index_list): for j in range(len(inner_table[index])): inner_table[index][j][1] = 0 zero_head_index_list.append(index) return inner_table def repair_by_tenderer(inner_table): """ 根据第一第二第三候选人修复当前格子的表头值 """ # 修复第一第二第三中标候选人作为表头 first_tenderer = ['第一中标候选人', '第一中标人', '第一中标(成交)人', '第一候选人'] second_tenderer = ['第二中标候选人', '第二中标(成交)候选人', '第二候选人'] third_tenderer = ['第三中标候选人', '第三中标(成交)候选人', '第三候选人'] # n1 next one, n2 next two, l1 last one, l2 last two for i in range(len(inner_table)): row = inner_table[i] n1_row, n2_row = None, None if i+1 < len(inner_table): n1_row = inner_table[i+1] if i+2 < len(inner_table): n2_row = inner_table[i+2] for j in range(len(row)): row_col = row[j] n1_row_col, n2_row_col = None, None row_n1_col, row_n2_col = None, None n1_row_n1_col, n2_row_n1_col, n1_row_n2_col = None, None, None if n1_row: n1_row_col = n1_row[j] if n2_row: n2_row_col = n2_row[j] if j+1 < len(row): row_n1_col = row[j+1] if j+2 < len(row): row_n2_col = row[j+2] if n1_row and j+1 < len(n1_row): n1_row_n1_col = n1_row[j+1] if n2_row and j+1 < len(n2_row): n2_row_n1_col = n2_row[j+1] if n1_row and j+2 < len(n1_row): n1_row_n2_col = n1_row[j+2] # 连续作为行表头 if row_col[0] in first_tenderer and row_n1_col and row_n1_col[1] == 0: if n1_row_col and n1_row_col[0] in second_tenderer and n1_row_n1_col and n1_row_n1_col[1] == 0: inner_table[i][j][1] = 1 inner_table[i+1][j][1] = 1 if n2_row_col and n2_row_col[0] in third_tenderer and n2_row_n1_col and n2_row_n1_col[1] == 0: inner_table[i+2][j][1] = 1 # 连续作为列表头 if row_col[0] in first_tenderer and n1_row_col and n1_row_col[1] == 0: if row_n1_col and row_n1_col[0] in second_tenderer and n1_row_n1_col and n1_row_n1_col[1] == 0: inner_table[i][j][1] = 1 inner_table[i][j+1][1] = 1 if row_n2_col and row_n2_col[0] in third_tenderer and n1_row_n2_col and n1_row_n2_col[1] == 0: inner_table[i][j+2][1] = 1 return inner_table def repair_by_keywords(inner_table): """ 根据关键词修复当前格子的表头值 """ # 修复表头关键词未作为表头 # 末尾匹配匹配关键词且字数小于7,直接作为表头 head_keyword = ['供应商', '总价', '总价(元)', '总价\(元\)', '品目一', '品目二', '品目三'] # 末尾匹配关键词且前一列为表头且与前一列文本不同,直接不做表头 head_keyword2 = ['管理中心', '有限公司', '项目采购', '确定。', ] # 开头匹配关键词,直接不做表头 head_keyword3 = ['详见', '选定', '咨询服务', '标准物资', '电汇', '承兑', '低档', '高档', '更换配置', '各种数据'] # 文本匹配关键词且前一列为表头,直接作为表头 head_keyword4 = ['综合排名', '工期(交货期)', '检测批', '检测范围', '混凝土设计强检测批的容度等级', '量(个)'] # 文本在关键词中,直接不做表头 head_keyword5 = ['殡葬用地', '电脑包', '电池'] # 文本匹配关键词,直接不作表头 head_keyword6 = ['市场行情', '有限公司', '能提供'] # 末尾匹配关键词,直接不做表头 head_keyword7 = ['基金', '结转', '结余', '税', '结余分配', '协议供货', '房屋', '纳税人', '自然人', '计算所得额'] # 文本匹配关键词且整行都是表头,直接做表头 head_keyword8 = ['备注'] # n1 next one, n2 next two, l1 last one, l2 last two for i in range(len(inner_table)): row = inner_table[i] for j in range(len(row)): row_col = row[j] row_l1_col = None if j-1 >= 0: row_l1_col = row[j-1] for key in head_keyword: match = re.search(key+'$', row_col[0]) if match and len(inner_table[i][j][0]) <= 6: if show: print('match head_keyword') inner_table[i][j][1] = 1 for key in head_keyword2: match = re.search(key+'$', row_col[0]) if j > 0 and row_l1_col and row_l1_col[1] == 1 and row_l1_col[0] != row_col[0] and match and row_col[1] == 1: if show: print('match head_keyword2') inner_table[i][j][1] = 0 for key in head_keyword3: match = re.search('^'+key, row_col[0]) if match and row_col[1] == 1: if show: print('match head_keyword3') inner_table[i][j][1] = 0 for key in head_keyword4: match = re.search(key, row_col[0]) if j > 0 and row_l1_col and row_l1_col[1] == 1 and match and row_col[1] == 0: if show: print('match head_keyword4') inner_table[i][j][1] = 1 if row_col[0] in head_keyword5: if show: print('match head_keyword5') inner_table[i][j][1] = 0 for key in head_keyword6: match = re.search(key, row_col[0]) if match: if show: print('match head_keyword6') inner_table[i][j][1] = 0 for key in head_keyword7: match = re.search(key+'$', row_col[0]) if match and row_col[1] == 1: if show: print('match head_keyword7') inner_table[i][j][1] = 0 if row_col[0] in head_keyword8 and row_col[1] == 0: if show: print('match head_keyword8') all_head_flag = 1 for k in range(len(row)): if row[k][0] in ['', row_col[0]]: continue if row[k][1] == 0: print('row[k]', row[k]) all_head_flag = 0 break # print('all_head_flag', all_head_flag) if all_head_flag: inner_table[i][j][1] = 1 return inner_table def repair_by_length(inner_table): for i in range(len(inner_table)): for j in range(len(inner_table[i])): if len(inner_table[i][j][0]) >= 30: inner_table[i][j][1] = 0 return inner_table def repair_by_summation(inner_table): # 修复合计在中间的特殊情况 if len(inner_table) >= 3 and len(inner_table[1]) == 2 \ and inner_table[1][0][0] == '合计' and inner_table[1][1][0].endswith('%'): inner_table[1][0][1] = 0 inner_table[1][1][1] = 0 return inner_table def repair_by_rank(inner_table): if not inner_table or (inner_table and len(inner_table[0]) < 3): return inner_table for i in range(len(inner_table)): for j in range(len(inner_table[i])-2): if inner_table[i][j][0] in ['第一名'] and inner_table[i][j+1][0] in ['第二名'] and inner_table[i][j+2][0] in ['第三名']: inner_table[i][j][1] = 1 inner_table[i][j+1][1] = 1 inner_table[i][j+2][1] = 1 return inner_table _inner_table = pre_process(_inner_table) compare_inner_table = copy.deepcopy(_inner_table) if show: print('table_head_repair_process1', show_row_index, _inner_table[show_row_index]) _inner_table = repair_by_rank(_inner_table) if _inner_table != compare_inner_table: compare_inner_table = copy.deepcopy(_inner_table) log('table_head repair1.5 ' + str(docid)) if show: print('table_head_repair_process1.5', show_row_index, _inner_table[show_row_index]) _inner_table = repair_by_colon(_inner_table) if _inner_table != compare_inner_table: compare_inner_table = copy.deepcopy(_inner_table) log('table_head repair2 ' + str(docid)) if show: print('table_head_repair_process2', show_row_index, _inner_table[show_row_index]) _inner_table = repair_by_keywords(_inner_table) if _inner_table != compare_inner_table: compare_inner_table = copy.deepcopy(_inner_table) log('table_head repair3 ' + str(docid)) if show: print('table_head_repair_process3', show_row_index, _inner_table[show_row_index]) _inner_table = repair_by_tenderer(_inner_table) if _inner_table != compare_inner_table: compare_inner_table = copy.deepcopy(_inner_table) log('table_head repair4 ' + str(docid)) if show: print('table_head_repair_process4', show_row_index, _inner_table[show_row_index]) _inner_table = repair_by_duplicate(_inner_table) if _inner_table != compare_inner_table: compare_inner_table = copy.deepcopy(_inner_table) log('table_head repair5 ' + str(docid)) if show: print('table_head_repair_process5', show_row_index, _inner_table[show_row_index]) _inner_table = repair_by_around(_inner_table) if _inner_table != compare_inner_table: compare_inner_table = copy.deepcopy(_inner_table) log('table_head repair6 ' + str(docid)) if show: print('table_head_repair_process6', show_row_index, _inner_table[show_row_index]) _inner_table = repair_by_tenderer(_inner_table) if _inner_table != compare_inner_table: compare_inner_table = copy.deepcopy(_inner_table) log('table_head repair7 ' + str(docid)) if show: print('table_head_repair_process7', show_row_index, _inner_table[show_row_index]) _inner_table = repair_by_keywords(_inner_table) if _inner_table != compare_inner_table: compare_inner_table = copy.deepcopy(_inner_table) log('table_head repair8 ' + str(docid)) if show: print('table_head_repair_process8', show_row_index, _inner_table[show_row_index]) _inner_table = repair_by_length(_inner_table) if _inner_table != compare_inner_table: compare_inner_table = copy.deepcopy(_inner_table) print('table_head repair9 ' + str(docid)) if show: print('table_head_repair_process9', show_row_index, _inner_table[show_row_index]) _inner_table = repair_by_summation(_inner_table) if show: print('table_head_repair_process10', show_row_index, _inner_table[show_row_index]) return _inner_table