| 1234567891011121314151617181920212223242526272829303132333435363738394041424344454647484950515253545556575859606162636465666768697071727374757677787980818283848586878889909192939495969798991001011021031041051061071081091101111121131141151161171181191201211221231241251261271281291301311321331341351361371381391401411421431441451461471481491501511521531541551561571581591601611621631641651661671681691701711721731741751761771781791801811821831841851861871881891901911921931941951961971981992002012022032042052062072082092102112122132142152162172182192202212222232242252262272282292302312322332342352362372382392402412422432442452462472482492502512522532542552562572582592602612622632642652662672682692702712722732742752762772782792802812822832842852862872882892902912922932942952962972982993003013023033043053063073083093103113123133143153163173183193203213223233243253263273283293303313323333343353363373383393403413423433443453463473483493503513523533543553563573583593603613623633643653663673683693703713723733743753763773783793803813823833843853863873883893903913923933943953963973983994004014024034044054064074084094104114124134144154164174184194204214224234244254264274284294304314324334344354364374384394404414424434444454464474484494504514524534544554564574584594604614624634644654664674684694704714724734744754764774784794804814824834844854864874884894904914924934944954964974984995005015025035045055065075085095105115125135145155165175185195205215225235245255265275285295305315325335345355365375385395405415425435445455465475485495505515525535545555565575585595605615625635645655665675685695705715725735745755765775785795805815825835845855865875885895905915925935945955965975985996006016026036046056066076086096106116126136146156166176186196206216226236246256266276286296306316326336346356366376386396406416426436446456466476486496506516526536546556566576586596606616626636646656666676686696706716726736746756766776786796806816826836846856866876886896906916926936946956966976986997007017027037047057067077087097107117127137147157167177187197207217227237247257267277287297307317327337347357367377387397407417427437447457467477487497507517527537547557567577587597607617627637647657667677687697707717727737747757767777787797807817827837847857867877887897907917927937947957967977987998008018028038048058068078088098108118128138148158168178188198208218228238248258268278288298308318328338348358368378388398408418428438448458468478488498508518528538548558568578588598608618628638648658668678688698708718728738748758768778788798808818828838848858868878888898908918928938948958968978988999009019029039049059069079089099109119129139149159169179189199209219229239249259269279289299309319329339349359369379389399409419429439449459469479489499509519529539549559569579589599609619629639649659669679689699709719729739749759769779789799809819829839849859869879889899909919929939949959969979989991000100110021003100410051006100710081009101010111012101310141015101610171018101910201021102210231024102510261027102810291030103110321033103410351036103710381039104010411042104310441045104610471048104910501051105210531054105510561057105810591060106110621063106410651066106710681069107010711072107310741075107610771078107910801081108210831084108510861087108810891090109110921093109410951096109710981099110011011102110311041105110611071108110911101111111211131114111511161117111811191120112111221123112411251126112711281129113011311132113311341135113611371138113911401141114211431144114511461147114811491150115111521153115411551156115711581159116011611162116311641165116611671168116911701171117211731174117511761177117811791180118111821183118411851186118711881189119011911192119311941195119611971198119912001201120212031204120512061207120812091210121112121213121412151216121712181219122012211222122312241225122612271228122912301231123212331234123512361237123812391240124112421243124412451246124712481249125012511252125312541255125612571258125912601261126212631264126512661267126812691270127112721273127412751276127712781279128012811282128312841285128612871288128912901291129212931294129512961297129812991300130113021303130413051306130713081309131013111312131313141315131613171318131913201321132213231324132513261327132813291330133113321333133413351336133713381339134013411342134313441345134613471348134913501351135213531354135513561357135813591360136113621363136413651366136713681369137013711372137313741375137613771378137913801381138213831384138513861387138813891390139113921393139413951396139713981399140014011402140314041405140614071408140914101411141214131414141514161417141814191420142114221423142414251426142714281429143014311432143314341435143614371438143914401441144214431444144514461447144814491450145114521453145414551456145714581459146014611462146314641465146614671468146914701471147214731474147514761477147814791480148114821483148414851486148714881489149014911492149314941495149614971498149915001501150215031504150515061507150815091510151115121513151415151516151715181519152015211522152315241525152615271528152915301531153215331534153515361537153815391540154115421543154415451546154715481549155015511552155315541555155615571558155915601561156215631564156515661567156815691570157115721573157415751576157715781579158015811582158315841585158615871588158915901591159215931594159515961597159815991600160116021603160416051606160716081609161016111612161316141615161616171618161916201621162216231624162516261627162816291630163116321633163416351636163716381639164016411642164316441645164616471648164916501651165216531654165516561657165816591660166116621663166416651666166716681669167016711672167316741675167616771678167916801681168216831684168516861687168816891690169116921693169416951696169716981699170017011702170317041705170617071708170917101711171217131714171517161717171817191720172117221723172417251726172717281729173017311732173317341735173617371738173917401741174217431744174517461747174817491750175117521753175417551756175717581759176017611762176317641765176617671768176917701771177217731774177517761777177817791780178117821783178417851786178717881789179017911792179317941795179617971798179918001801180218031804180518061807180818091810181118121813181418151816181718181819182018211822182318241825182618271828182918301831183218331834183518361837183818391840184118421843184418451846184718481849185018511852185318541855185618571858185918601861186218631864186518661867186818691870187118721873187418751876187718781879188018811882188318841885188618871888188918901891189218931894189518961897189818991900190119021903190419051906190719081909191019111912191319141915191619171918191919201921192219231924192519261927192819291930193119321933193419351936193719381939194019411942194319441945194619471948194919501951195219531954195519561957195819591960196119621963196419651966196719681969197019711972197319741975197619771978197919801981198219831984198519861987198819891990199119921993199419951996199719981999200020012002200320042005200620072008200920102011201220132014201520162017201820192020202120222023202420252026202720282029203020312032203320342035203620372038203920402041204220432044204520462047204820492050205120522053205420552056205720582059206020612062206320642065206620672068206920702071207220732074207520762077207820792080208120822083208420852086208720882089209020912092209320942095209620972098209921002101210221032104210521062107210821092110211121122113211421152116211721182119212021212122212321242125212621272128212921302131213221332134213521362137213821392140214121422143214421452146214721482149215021512152215321542155215621572158215921602161216221632164216521662167216821692170217121722173217421752176217721782179218021812182218321842185218621872188218921902191219221932194219521962197219821992200220122022203220422052206220722082209221022112212221322142215221622172218221922202221222222232224222522262227222822292230223122322233223422352236223722382239224022412242224322442245224622472248224922502251225222532254225522562257225822592260226122622263226422652266226722682269227022712272227322742275227622772278227922802281228222832284228522862287228822892290229122922293229422952296229722982299230023012302230323042305230623072308230923102311231223132314231523162317231823192320232123222323232423252326232723282329233023312332233323342335233623372338233923402341234223432344234523462347234823492350235123522353235423552356235723582359236023612362236323642365236623672368236923702371237223732374237523762377237823792380238123822383238423852386238723882389239023912392239323942395239623972398239924002401 |
- # -*- coding: utf-8 -*-
- """表格解析与二维化。
- 按 ARCHITECTURE.md Phase 4 拆分建议,从 ``interface/Preprocessing.py`` 迁出。
- 类型:PREPROCESS。
- 原位置:``interface/Preprocessing.py`` 中以下函数:
- - ``tableToText`` — HTML 表格转文本,含 fixSpan/getTable 等嵌套辅助函数
- - ``table_head_repair_process`` — 表头修复
- ``interface/Preprocessing.py`` 仍 re-export 以上全部名称,老 import 不受影响。
- """
- from __future__ import absolute_import
- import copy
- import json
- import re
- import time
- import numpy as np
- from BiddingKG.dl.common.logging import log
- from BiddingKG.dl.interface.predictor import getPredictor
- from BiddingKG.dl.predictors.table_prem import TableTag2List
- from BiddingKG.dl.model_runtime.embed import formEncoding
- from BiddingKG.dl.table_head.predict_torch import predict
- from BiddingKG.dl.preprocess.segmenter import segment
- __all__ = [
- "tableToText",
- "table_head_repair_process",
- ]
- def tableToText(soup, docid=None, return_kv=False):
- '''
- @param:
- soup:网页html的soup
- @return:处理完表格信息的网页text
- '''
-
- def getTrs(tbody):
- #获取所有的tr
- trs = []
- objs = tbody.find_all(recursive=False)
- for obj in objs:
- if obj.name=="tr":
- trs.append(obj)
- if obj.name=="tbody":
- for tr in obj.find_all("tr",recursive=False):
- trs.append(tr)
- return trs
- def fixSpan(tbody):
- # 处理colspan, rowspan信息补全问题
- #trs = tbody.findChildren('tr', recursive=False)
- trs = getTrs(tbody)
- ths_len = 0
- ths = list()
- trs_set = set()
- #修改为先进行列补全再进行行补全,否则可能会出现表格解析混乱
- # 遍历每一个tr
- for indtr, tr in enumerate(trs):
- ths_tmp = tr.findChildren('th', recursive=False)
- #不补全含有表格的tr
- if len(tr.findChildren('table'))>0:
- continue
- if len(ths_tmp) > 0:
- ths_len = ths_len + len(ths_tmp)
- for th in ths_tmp:
- ths.append(th)
- trs_set.add(tr)
- # 遍历每行中的element
- tds = tr.findChildren(recursive=False)
- for indtd, td in enumerate(tds):
- # 若有colspan 则补全同一行下一个位置
- if 'colspan' in td.attrs:
- if str(re.sub("[^0-9]","",str(td['colspan'])))!="":
- col = int(re.sub("[^0-9]","",str(td['colspan'])))
- if col<100 and len(td.get_text())<1000:
- td['colspan'] = 1
- for i in range(1, col, 1):
- td.insert_after(copy.copy(td))
- for indtr, tr in enumerate(trs):
- ths_tmp = tr.findChildren('th', recursive=False)
- #不补全含有表格的tr
- if len(tr.findChildren('table'))>0:
- continue
- if len(ths_tmp) > 0:
- ths_len = ths_len + len(ths_tmp)
- for th in ths_tmp:
- ths.append(th)
- trs_set.add(tr)
- # 遍历每行中的element
- tds = tr.findChildren(recursive=False)
- for indtd, td in enumerate(tds):
- # 若有rowspan 则补全下一行同样位置
- if 'rowspan' in td.attrs:
- if str(re.sub("[^0-9]","",str(td['rowspan'])))!="":
- row = int(re.sub("[^0-9]","",str(td['rowspan'])))
- td['rowspan'] = 1
- for i in range(1, row, 1):
- # 获取下一行的所有td, 在对应的位置插入
- if indtr+i<len(trs):
- tds1 = trs[indtr + i].findChildren(['td','th'], recursive=False)
- if len(tds1) >= (indtd) and len(tds1)>0:
- if indtd > 0:
- tds1[indtd - 1].insert_after(copy.copy(td))
- else:
- tds1[0].insert_before(copy.copy(td))
- elif indtd-2>0 and len(tds1) > 0 and len(tds1) == indtd - 1: # 修正某些表格最后一列没补全
- tds1[indtd-2].insert_after(copy.copy(td))
- def getTable(tbody):
- #trs = tbody.findChildren('tr', recursive=False)
- trs = getTrs(tbody)
- inner_table = []
- for tr in trs:
- tr_line = []
- tds = tr.findChildren(['td','th'], recursive=False)
- if len(tds)==0:
- if return_kv:
- tr_line.append([re.sub('\xa0','',tr.get_text()),0])
- else:
- tr_line.append([re.sub('\xa0','',segment(tr,final=False)),0]) # 2021/12/21 修复部分表格没有td 造成数据丢失
- for td in tds:
- if return_kv:
- tr_line.append([re.sub('\xa0','',td.get_text()),0])
- else:
- tr_line.append([re.sub('\xa0','',segment(td,final=False)),0])
- #tr_line.append([td.get_text(),0])
- inner_table.append(tr_line)
- return inner_table
-
- #处理表格不对齐的问题
- def fixTable(inner_table,fix_value="~~"):
- maxWidth = 0
- for item in inner_table:
- if len(item)>maxWidth:
- maxWidth = len(item)
- if maxWidth > 100:
- # log('表格列数大于100,表格异常不做处理。')
- return []
- for i in range(len(inner_table)):
- if len(inner_table[i])<maxWidth:
- for j in range(maxWidth-len(inner_table[i])):
- inner_table[i].append([fix_value,0])
- return inner_table
-
- def removePadding(inner_table,pad_row = "@@",pad_col = "##"):
- height = len(inner_table)
- width = len(inner_table[0])
- for i in range(height):
- point = ""
- for j in range(width):
- if inner_table[i][j][0]==point and point!="":
- inner_table[i][j][0] = pad_row
- else:
- if inner_table[i][j][0] not in [pad_row,pad_col]:
- point = inner_table[i][j][0]
- for j in range(width):
- point = ""
- for i in range(height):
- if inner_table[i][j][0]==point and point!="":
- inner_table[i][j][0] = pad_col
- else:
- if inner_table[i][j][0] not in [pad_row,pad_col]:
- point = inner_table[i][j][0]
-
- def addPadding(inner_table,pad_row = "@@",pad_col = "##"):
- height = len(inner_table)
- width = len(inner_table[0])
- for i in range(height):
- for j in range(width):
- if inner_table[i][j][0]==pad_row:
- inner_table[i][j][0] = inner_table[i][j-1][0]
- inner_table[i][j][1] = inner_table[i][j-1][1]
- if inner_table[i][j][0]==pad_col:
- inner_table[i][j][0] = inner_table[i-1][j][0]
- inner_table[i][j][1] = inner_table[i-1][j][1]
- def repairTable(inner_table, dye_set=set(), key_set=set(), fix_value="~~"):
- """
- @summary: 修复表头识别,将明显错误的进行修正
- """
- def repairNeeded(line):
- first_1 = -1
- last_1 = -1
- first_0 = -1
- last_0 = -1
- count_1 = 0
- count_0 = 0
- for i in range(len(line)):
- if line[i][0] == fix_value:
- continue
- if line[i][1]==1:
- if first_1==-1:
- first_1 = i
- last_1 = i
- count_1 += 1
- if line[i][1]==0:
- if first_0 == -1:
- first_0 = i
- last_0 = i
- count_0 += 1
- if first_1 ==-1 or last_0 == -1:
- return False
- # 异常情况:第一个不是表头;最后一个是表头;表头个数远大于属性值个数
- if first_1-0 > 0 or last_0-len(line)+1 < 0 or last_1 == len(line)-1 or count_1-count_0 >= 3:
- return True
- return False
- def getsimilarity(line, line1):
- same_count = 0
- for item, item1 in zip(line,line1):
- if item[1] == item1[1]:
- same_count += 1
- return same_count/len(line)
- def selfrepair(inner_table,index,dye_set,key_set):
- """
- @summary: 计算每个节点受到的挤压度来判断是否需要染色
- """
- # print("B",inner_table[index])
- min_presure = 3
- list_dye = []
- first = None
- count = 0
- # temp_set = set()
- temp_set = set(['~~']) # 2023/10/10纠正236239652 受让单位识别不到表头; 受让单位,明细用途:用途名称:陵川县民政局,
- _index = 0
- for item in inner_table[index]:
- if first is None:
- first = item[1]
- if item[0] not in temp_set:
- count += 1
- temp_set.add(item[0])
- else:
- if first == item[1]:
- if item[0] not in temp_set:
- temp_set.add(item[0])
- count += 1
- else:
- list_dye.append([first,count,_index])
- first = item[1]
- temp_set.add(item[0])
- count = 1
- _index += 1
- list_dye.append([first,count,_index])
- if len(list_dye)>1:
- begin = 0
- end = 0
- for i in range(len(list_dye)):
- end = list_dye[i][2]
- dye_flag = False
- # 首尾要求压力减一
- if i==0:
- if list_dye[i+1][1]-list_dye[i][1]+1>=min_presure-1:
- dye_flag = True
- dye_type = list_dye[i+1][0]
- elif i==len(list_dye)-1:
- if list_dye[i-1][1]-list_dye[i][1]+1>=min_presure-1:
- dye_flag = True
- dye_type = list_dye[i-1][0]
- else:
- if list_dye[i][1]>1:
- if list_dye[i+1][1]-list_dye[i][1]+1>=min_presure:
- dye_flag = True
- dye_type = list_dye[i+1][0]
- if list_dye[i-1][1]-list_dye[i][1]+1>=min_presure:
- dye_flag = True
- dye_type = list_dye[i-1][0]
- else:
- if list_dye[i+1][1]+list_dye[i-1][1]-list_dye[i][1]+1>=min_presure:
- dye_flag = True
- dye_type = list_dye[i+1][0]
- if list_dye[i+1][1]+list_dye[i-1][1]-list_dye[i][1]+1>=min_presure:
- dye_flag = True
- dye_type = list_dye[i-1][0]
- if dye_flag:
- for h in range(begin,end):
- inner_table[index][h][1] = dye_type
- dye_set.add((inner_table[index][h][0],dye_type))
- key_set.add(inner_table[index][h][0])
- begin = end
- # print("E",inner_table[index])
- def otherrepair(inner_table,index,dye_set,key_set):
- list_provide_repair = []
- if index==0 and len(inner_table)>1:
- list_provide_repair.append(index+1)
- elif index==len(inner_table)-1:
- list_provide_repair.append(index-1)
- else:
- list_provide_repair.append(index+1)
- list_provide_repair.append(index-1)
- for provide_index in list_provide_repair:
- if not repairNeeded(inner_table[provide_index]):
- same_prob = getsimilarity(inner_table[index], inner_table[provide_index])
- if same_prob>=0.8:
- for i in range(len(inner_table[provide_index])):
- if inner_table[index][i][1]!=inner_table[provide_index][i][1]:
- dye_set.add((inner_table[index][i][0],inner_table[provide_index][i][1]))
- key_set.add(inner_table[index][i][0])
- inner_table[index][i][1] = inner_table[provide_index][i][1]
- elif same_prob<=0.2:
- for i in range(len(inner_table[provide_index])):
- if inner_table[index][i][1]==inner_table[provide_index][i][1]:
- dye_set.add((inner_table[index][i][0],inner_table[provide_index][i][1]))
- key_set.add(inner_table[index][i][0])
- inner_table[index][i][1] = 0 if inner_table[provide_index][i][1] ==1 else 1
- len_dye_set = len(dye_set)
- height = len(inner_table)
- for i in range(height):
- if repairNeeded(inner_table[i]):
- selfrepair(inner_table, i, dye_set, key_set)
- #otherrepair(inner_table,i,dye_set,key_set)
- for h in range(len(inner_table)):
- for w in range(len(inner_table[0])):
- if inner_table[h][w][0] in key_set:
- for item in dye_set:
- if inner_table[h][w][0] == item[0]:
- inner_table[h][w][1] = item[1]
- # 如果两个set长度不相同,则有同一个key被反复染色,将导致无限迭代
- if len(dye_set) != len(key_set):
- for i in range(height):
- if repairNeeded(inner_table[i]):
- selfrepair(inner_table,i,dye_set,key_set)
- #otherrepair(inner_table,i,dye_set,key_set)
- return
- if len(dye_set) == len_dye_set:
- '''
- for i in range(height):
- if repairNeeded(inner_table[i]):
- otherrepair(inner_table,i,dye_set,key_set)
- '''
- return
- repairTable(inner_table, dye_set, key_set)
- def repair_table2(inner_table, show=0, row_no=0):
- """
- @summary: 修复表头识别,将明显错误的进行修正
- """
- # 循环处理单元格,一次获取需要的
- one_head_index_list = []
- zero_head_index_list = []
- all_head_index_list = []
- for i in range(len(inner_table)):
- head_cnt = 0
- for j in range(len(inner_table[i])):
- # 删除前后逗号
- inner_table[i][j][0] = re.sub('^[,,]+', '', inner_table[i][j][0])
- inner_table[i][j][0] = re.sub('[,,]+$', '', inner_table[i][j][0])
- # 统计表头数
- if inner_table[i][j][1] == 1:
- head_cnt += 1
- # 表头数list
- if head_cnt == 0:
- zero_head_index_list.append(i)
- elif head_cnt == 1:
- one_head_index_list.append(i)
- elif head_cnt == len(inner_table[i]):
- all_head_index_list.append(i)
- # 修复冒号在文本中间的,不能作为表头;(冒号后面需多个字)
- # 冒号在括号中的除外
- # 冒号在最后的,判断后一个格子是否有重复的文字
- for i in range(len(inner_table)):
- for j in range(len(inner_table[i])):
- _text = inner_table[i][j][0]
- if len(_text) >= 3 and inner_table[i][j][1] == 1:
- match = re.search('[::]', _text)
- if match:
- start_index, end_index = match.span()
- if start_index == 0:
- continue
- if end_index == len(_text):
- if len(inner_table[i]) == 2 and j <= len(inner_table[i]) - 2 and (_text in inner_table[i][j+1][0] or inner_table[i][j+1][0] in _text):
- inner_table[i][j][1] = 0
- inner_table[i][j+1][1] = 0
- else:
- continue
- if re.search('[((]', _text[:start_index]) and re.search('[))]', _text[end_index:]):
- continue
- m1 = re.search('[\u4e00-\u9fa50-9a-zA-Z]', _text[:start_index])
- m2 = re.search('[\u4e00-\u9fa50-9a-zA-Z]', _text[end_index:])
- if m1 and m2 and (len(m2.group()) >= 2 or m2.group() in ['是', '否']):
- inner_table[i][j][1] = 0
- if show:
- print('inner_table[i]1', inner_table[row_no])
- # 修复实际只有几列,但有一列由于重复占了太多行表头识别错误
- # for i in range(len(inner_table)):
- # head_flag_dict = {}
- # for j in range(len(inner_table[i])):
- # if inner_table[i][j][0] in head_flag_dict.keys():
- # head_flag_dict[inner_table[i][j][0]] += [inner_table[i][j][1]]
- # else:
- # head_flag_dict[inner_table[i][j][0]] = [inner_table[i][j][1]]
- #
- # if len(head_flag_dict.keys()) == 2:
- # col_flag = None
- # col_value = None
- # for key in head_flag_dict.keys():
- # flag_list = head_flag_dict[key]
- # if len(flag_list) >= 4 and len(set(flag_list)) == 2 and len(set(flag_list[1:])) == 1:
- # col_flag = flag_list[0]
- # col_value = key
- # break
- #
- # if col_flag is not None:
- # for j in range(len(inner_table[i])):
- # if inner_table[i][j][0] == col_value:
- # inner_table[i][j][1] = col_flag
- # 多个重复列的预测值不同,以第一个为准
- for i in range(len(inner_table)):
- col = inner_table[i][0]
- for j in range(len(inner_table[i])):
- if inner_table[i][j][0] == col[0]:
- if inner_table[i][j][1] != col[1]:
- inner_table[i][j][1] = col[1]
- else:
- col = inner_table[i][j]
- if show:
- print('inner_table[i]2', inner_table[row_no])
- # 修复多个重复的单元格表头不一致
- # for i in range(len(inner_table)):
- # for j in range(len(inner_table[i])-1):
- # only_chinese1 = ''.join(re.findall('[\u4e00-\u9fa5]+', inner_table[i][j][0]))
- # only_chinese2 = ''.join(re.findall('[\u4e00-\u9fa5]+', inner_table[i][j+1][0]))
- # if only_chinese1 == only_chinese2 and inner_table[i][j][1] != inner_table[i][j+1][1]:
- # inner_table[i][j][1] = 1
- # inner_table[i][j+1][1] = 1
- # if show:
- # print('inner_table[i]3', inner_table[row_no])
- # # 修复一行几乎都是表头,个别不是;或者一行几乎都是非表头,个别是
- # for i in range(len(inner_table)):
- # head_dict = {}
- # not_head_dict = {}
- # for j in range(len(inner_table[i])):
- # if inner_table[i][j][1] == 1:
- # if inner_table[i][j][0] not in head_dict:
- # head_dict[inner_table[i][j][0]] = 1
- # else:
- # if inner_table[i][j][0] not in not_head_dict:
- # not_head_dict[inner_table[i][j][0]] = 1
- #
- # # 非表头:表头 <= 1:3
- # # if len(head_dict.keys()) > 0 and len(not_head_dict.keys()) / len(head_dict.keys()) <= 1/3 and len(head_dict.keys()) >= 3:
- # # for j in range(len(inner_table[i])):
- # # if len(re.sub(' ', '', inner_table[i][j][0])) > 0:
- # # inner_table[i][j][1] = 1
- #
- # # 表头数一个且非表头数大于2且上一行都是表头
- # if i > 0 and len(head_dict.keys()) == 1 and len(not_head_dict.keys()) >= 2 and inner_table[i][0][1] == 0:
- # last_row = inner_table[i-1]
- # col_list = []
- # for j in range(len(last_row)):
- # if len(re.sub(' ', '', last_row[j][0])) > 0:
- # if last_row[j][1] == 0:
- # col_list = []
- # break
- # col_list.append(last_row[j][0])
- # if col_list:
- # col_list = list(set(col_list))
- # if len(col_list) > 2:
- # for j in range(len(inner_table[i])):
- # if inner_table[i][j][1] == 1:
- # inner_table[i][j][1] = 0
- # 一整个大表格,第一行为表头,下面行中有个别格子被识别为表头
- # 候选人后面修复
- for index in one_head_index_list:
- if (index - 1 in zero_head_index_list and index - 2 in zero_head_index_list) \
- or (index - 1 in zero_head_index_list and index - 2 in all_head_index_list) \
- or (index - 1 in all_head_index_list):
- for j in range(len(inner_table[index])):
- inner_table[index][j][1] = 0
- zero_head_index_list.append(index)
- if show:
- print('inner_table[i]4', inner_table[row_no])
- # 修复第一第二第三中标候选人作为表头
- first_tenderer = ['第一中标候选人', '第一中标人', '第一中标(成交)人', '第一候选人']
- second_tenderer = ['第二中标候选人', '第二中标(成交)候选人', '第二候选人']
- third_tenderer = ['第三中标候选人', '第三中标(成交)候选人', '第三候选人']
- # n1 next one, n2 next two, l1 last one, l2 last two
- for i in range(len(inner_table)):
- row = inner_table[i]
- n1_row, n2_row = None, None
- if i+1 < len(inner_table):
- n1_row = inner_table[i+1]
- if i+2 < len(inner_table):
- n2_row = inner_table[i+2]
- for j in range(len(row)):
- row_col = row[j]
- n1_row_col, n2_row_col = None, None
- row_n1_col, row_n2_col = None, None
- n1_row_n1_col, n2_row_n1_col, n1_row_n2_col = None, None, None
- if n1_row:
- n1_row_col = n1_row[j]
- if n2_row:
- n2_row_col = n2_row[j]
- if j+1 < len(row):
- row_n1_col = row[j+1]
- if j+2 < len(row):
- row_n2_col = row[j+2]
- if n1_row and j+1 < len(n1_row):
- n1_row_n1_col = n1_row[j+1]
- if n2_row and j+1 < len(n2_row):
- n2_row_n1_col = n2_row[j+1]
- if n1_row and j+2 < len(n1_row):
- n1_row_n2_col = n1_row[j+2]
- # 连续作为行表头
- if row_col[0] in first_tenderer and row_n1_col and row_n1_col[1] == 0:
- if n1_row_col and n1_row_col[0] in second_tenderer and n1_row_n1_col and n1_row_n1_col[1] == 0:
- inner_table[i][j][1] = 1
- inner_table[i+1][j][1] = 1
- if n2_row_col and n2_row_col[0] in third_tenderer and n2_row_n1_col and n2_row_n1_col[1] == 0:
- inner_table[i+2][j][1] = 1
- # 连续作为列表头
- if row_col[0] in first_tenderer and n1_row_col and n1_row_col[1] == 0:
- if row_n1_col and row_n1_col[0] in second_tenderer and n1_row_n1_col and n1_row_n1_col[1] == 0:
- inner_table[i][j][1] = 1
- inner_table[i][j+1][1] = 1
- if row_n2_col and row_n2_col[0] in third_tenderer and n1_row_n2_col and n1_row_n2_col[1] == 0:
- inner_table[i][j+2][1] = 1
- if show:
- print('inner_table[i]5', inner_table[row_no])
- # 修复表头关键词未作为表头
- # 文本匹配关键词,直接作为表头
- head_keyword = ['供应商', '总价']
- # 末尾匹配关键词且前一列为表头且与前一列文本不同,直接不做表头
- head_keyword2 = ['管理中心', '有限公司', '项目采购', ]
- # 开头匹配关键词,直接不做表头
- head_keyword3 = ['详见', '选定', '咨询服务', '标准物资', '电汇', '承兑']
- # 文本匹配关键词且前一列为表头,直接作为表头
- head_keyword4 = ['综合排名']
- # 文本在关键词中,直接不做表头
- head_keyword5 = ['殡葬用地']
- # n1 next one, n2 next two, l1 last one, l2 last two
- for i in range(len(inner_table)):
- row = inner_table[i]
- for j in range(len(row)):
- row_col = row[j]
- row_l1_col = None
- if j-1 > 0:
- row_l1_col = row[j-1]
- match = re.search('[\u4e00-\u9fa50-9a-zA-Z::]+', row_col[0])
- if inner_table[i][j][1] == 0 and match and match.group() in head_keyword:
- inner_table[i][j][1] = 1
- for key in head_keyword2:
- match = re.search(key+'$', row_col[0])
- if j > 0 and row_l1_col and row_l1_col[1] == 1 and row_l1_col[0] != row_col[0] and match and row_col[1] == 1:
- inner_table[i][j][1] = 0
- for key in head_keyword3:
- match = re.search('^'+key, row_col[0])
- if match and row_col[1] == 1:
- inner_table[i][j][1] = 0
- for key in head_keyword4:
- match = re.search(key, row_col[0])
- if j > 0 and row_l1_col and row_l1_col[1] == 1 and match and row_col[1] == 0:
- inner_table[i][j][1] = 1
- if row_col[0] in head_keyword5:
- inner_table[i][j][1] = 0
- if show:
- print('inner_table[i]6', inner_table[row_no])
- # 修复姓名被作为表头 # 2023-02-10 取消修复,避免项目名称、编号,单位、单价等作为了非表头
- # surname = [
- # "赵", "钱", "孙", "李", "周", "吴", "郑", "王", "冯", "陈", "褚", "卫", "蒋", "沈", "韩", "杨", "朱", "秦", "尤", "许", "何", "吕", "施", "张", "孔", "曹", "严", "华", "金", "魏", "陶", "姜", "戚", "谢", "邹", "喻", "柏", "水", "窦", "章", "云", "苏", "潘", "葛", "奚", "范", "彭", "郎", "鲁", "韦", "昌", "马", "苗", "凤", "花", "方", "俞", "任", "袁", "柳", "酆", "鲍", "史", "唐", "费", "廉", "岑", "薛", "雷", "贺", "倪", "汤", "滕", "殷", "罗", "毕", "郝", "邬", "安", "常", "乐", "于", "时", "傅", "皮", "卞", "齐", "康", "伍", "余", "元", "卜", "顾", "孟", "平", "黄", "和", "穆", "萧", "尹", "姚", "邵", "湛", "汪", "祁", "毛", "禹", "狄", "米", "贝", "明", "臧", "计", "伏", "成", "戴", "谈", "宋", "茅", "庞", "熊", "纪", "舒", "屈", "项", "祝", "董", "梁", "杜", "阮", "蓝", "闵", "席", "季", "麻", "强", "贾", "路", "娄", "危", "江", "童", "颜", "郭", "梅", "盛", "林", "刁", "钟", "徐", "邱", "骆", "高", "夏", "蔡", "田", "樊", "胡", "凌", "霍", "虞", "万", "支", "柯", "昝", "管", "卢", "莫", "经", "房", "裘", "缪", "干", "解", "应", "宗", "丁", "宣", "贲", "邓", "郁", "单", "杭", "洪", "包", "诸", "左", "石", "崔", "吉", "钮", "龚", "程", "嵇", "邢", "滑", "裴", "陆", "荣", "翁", "荀", "羊", "於", "惠", "甄", "麴", "家", "封", "芮", "羿", "储", "靳", "汲", "邴", "糜", "松", "井", "段", "富", "巫", "乌", "焦", "巴", "弓", "牧", "隗", "山", "谷", "车", "侯", "宓", "蓬", "全", "郗", "班", "仰", "秋", "仲", "伊", "宫", "宁", "仇", "栾", "暴", "甘", "钭", "厉", "戎", "祖", "武", "符", "刘", "景", "詹", "束", "龙", "叶", "幸", "司", "韶", "郜", "黎", "蓟", "薄", "印", "宿", "白", "怀", "蒲", "邰", "从", "鄂", "索", "咸", "籍", "赖", "卓", "蔺", "屠", "蒙", "池", "乔", "阴", "欎", "胥", "能", "苍", "双", "闻", "莘", "党", "翟", "谭", "贡", "劳", "逄", "姬", "申", "扶", "堵", "冉", "宰", "郦", "雍", "舄", "璩", "桑", "桂", "濮", "牛", "寿", "通", "边", "扈", "燕", "冀", "郏", "浦", "尚", "农", "温", "别", "庄", "晏", "柴", "瞿", "阎", "充", "慕", "连", "茹", "习", "宦", "艾", "鱼", "容", "向", "古", "易", "慎", "戈", "廖", "庾", "终", "暨", "居", "衡", "步", "都", "耿", "满", "弘", "匡", "国", "文", "寇", "广", "禄", "阙", "东", "殴", "殳", "沃", "利", "蔚", "越", "夔", "隆", "师", "巩", "厍", "聂", "晁", "勾", "敖", "融", "冷", "訾", "辛", "阚", "那", "简", "饶", "空", "曾", "毋", "沙", "乜", "养", "鞠", "须", "丰", "巢", "关", "蒯", "相", "查", "後", "荆", "红", "游", "竺", "权", "逯", "盖", "益", "桓", "公", "万俟", "司马", "上官", "欧阳", "夏侯", "诸葛", "闻人", "东方", "赫连", "皇甫", "尉迟", "公羊", "澹台", "公冶", "宗政", "濮阳", "淳于", "单于", "太叔", "申屠", "公孙", "仲孙", "轩辕", "令狐", "钟离", "宇文", "长孙", "慕容", "鲜于", "闾丘", "司徒", "司空", "亓官", "司寇", "仉", "督", "子车", "颛孙", "端木", "巫马", "公西", "漆雕", "乐正", "壤驷", "公良", "拓跋", "夹谷", "宰父", "谷梁", "晋", "楚", "闫", "法", "汝", "鄢", "涂", "钦", "段干", "百里", "东郭", "南门", "呼延", "归", "海", "羊舌", "微生", "岳", "帅", "缑", "亢", "况", "后", "有", "琴", "梁丘", "左丘", "东门", "西门", "商", "牟", "佘", "佴", "伯", "赏", "南宫", "墨", "哈", "谯", "笪", "年", "爱", "阳", "佟", "第五", "言", "福",
- # ]
- # for i in range(len(inner_table)):
- # for j in range(len(inner_table[i])):
- # if inner_table[i][j][1] == 1 \
- # and 2 <= len(inner_table[i][j][0]) <= 4 \
- # and (inner_table[i][j][0][0] in surname or inner_table[i][j][0][:2] in surname) \
- # and re.search("[^\u4e00-\u9fa5]", inner_table[i][j][0]) is None:
- # inner_table[i][j][1] = 0
- return inner_table
- def sliceTable(inner_table,fix_value="~~"):
- #进行分块
- height = len(inner_table)
- width = len(inner_table[0])
- head_list = []
- head_list.append(0)
- last_head = None
- last_is_same_value = False
- for h in range(height):
- is_all_key = True#是否是全表头行
- is_all_value = True#是否是全属性值
- is_same_with_lastHead = True#和上一行的结构是否相同
- is_same_value=True#一行的item都一样
- #is_same_first_item = True#与上一行的第一项是否相同
- same_value = inner_table[h][0][0]
- for w in range(width):
- if last_head is not None:
- if inner_table[h-1][w][0] != fix_value and inner_table[h-1][w][0] != "" and inner_table[h-1][w][1] == 0:
- is_all_key = False
- if inner_table[h][w][1]==1:
- is_all_value = False
- if inner_table[h][w][1]!= inner_table[h-1][w][1]:
- is_same_with_lastHead = False
- if inner_table[h][w][0]!=fix_value and inner_table[h][w][0]!=same_value:
- is_same_value = False
- else:
- if re.search("\d+",same_value) is not None:
- is_same_value = False
- if h>0 and inner_table[h][0][0]!=inner_table[h-1][0][0]:
- is_same_first_item = False
- last_head = h
- if last_is_same_value:
- last_is_same_value = is_same_value
- continue
- if is_same_value:
- # 该块只有表头一行不合法
- if h - head_list[-1] > 1:
- head_list.append(h)
- last_is_same_value = is_same_value
- continue
- if not is_all_key:
- if not is_same_with_lastHead:
- # 该块只有表头一行不合法
- if h - head_list[-1] > 1 or not is_all_value: # 20260331补充 整行是表头分块 修复 748505279 第二行才是表头分块错误 749002477 第一行最后一格为空非
- head_list.append(h)
- head_list.append(height)
- return head_list
-
- def setHead_initem(inner_table,pat_head,fix_value="~~",prob_min=0.5):
- set_item = set()
- height = len(inner_table)
- width = len(inner_table[0])
- empty_set = set()
- for i in range(height):
- for j in range(width):
- item = inner_table[i][j][0]
- if item.strip()=="":
- empty_set.add(item)
- else:
- set_item.add(item)
- list_item = list(set_item)
- if list_item:
- x = []
- for item in list_item:
- x.append(getPredictor("form").encode(item))
- predict_y = getPredictor("form").predict(np.array(x),type="item")
- _dict = dict()
- for item,values in zip(list_item,list(predict_y)):
- _dict[item] = values[1]
- # print("##",item,values)
- #print(_dict)
- for i in range(height):
- for j in range(width):
- item = inner_table[i][j][0]
- if item not in empty_set:
- inner_table[i][j][1] = 1 if _dict[item]>prob_min else (1 if re.search(pat_head,item) is not None and len(item)<8 else 0)
- # print("=====")
- # for item in inner_table:
- # print(item)
- # print("======")
- repairTable(inner_table)
- head_list = sliceTable(inner_table)
-
- return inner_table,head_list
- def set_head_model(inner_table, show=0):
- origin_inner_table = copy.deepcopy(inner_table)
- for i in range(len(inner_table)):
- for j in range(len(inner_table[i])):
- # 删掉单格前后符号,以免影响表头预测
- col = inner_table[i][j][0]
- col = re.sub("^[^\u4e00-\u9fa5a-zA-Z0-9]+", "", col)
- col = re.sub("[^\u4e00-\u9fa5a-zA-Z0-9]+$", "", col)
- inner_table[i][j] = col
- # 模型预测表头
- # predict_list = predict(inner_table)
- start_time = time.time()
- predict_list = predict(inner_table)
- # print('table head predict cost: ', time.time()-start_time)
- # 组合结果
- for i in range(len(inner_table)):
- if i == 0 and (inner_table[i] == ['序号', '内容', '说明与要求'] or inner_table[i] == ['序号', '公告事项', '内容']): # 修复 659860507 659817551 这种只识别第一行表头,第二列非表头导致解析错误问题
- inner_table[i] = [[it, 0] for it in ['序号', '内容', '说明与要求']]
- flag = 1
- for j in range(1, len(inner_table)): # 判断是否所有行都有3个格
- if len(inner_table[j]) != 3:
- flag = 0
- if flag:
- for j in range(1, len(inner_table)):
- inner_table[j] = [[t1, t2] for t1,t2 in zip(inner_table[j], [0, 1, 0])]
- break
- continue
- for j in range(len(inner_table[i])):
- inner_table[i][j] = [origin_inner_table[i][j][0], int(predict_list[i][j])]
- if origin_inner_table[i][j][0] in ['主要环境影响及预防或者减轻不良环境影响的对策和措施', '建设单位或地方政府作出的相关环保承诺',
- '公众反馈意见的联系方式', '区县', '项目领域', '成本/收入', '覆盖倍数', '会计所', '律所','建设期',
- "发行时间" ,"批次" ,"发行额" ,"发行利率" ,"所属债券" ,"专项债作资本金发行额" ,"调整记录","资产面积",
- "交易底价","交易地点","竞得者", "谈判项目", "资产名称","单元号","交易面积","承租人","租赁单价"] and predict_list[i][j]!=1:
- inner_table[i][j] = [origin_inner_table[i][j][0], 1]
- elif predict_list[i][j]!=1 and (re.search('^拟?(中标|中选|成交|承包|(参与)?投标|招标|采购|招租|发包|业主|竞投)(单位|人)(名称|地址|电话)?$'
- '|^拟?(中标|中选|成交)(供应商|金额|价格|日期)((万?元))?$|^项目(名称|编号)$', origin_inner_table[i][j][0])
- or re.match('(?[\d一二三四五六七八九十][\.、)]((招标|采购|\w{2,4})?项目名称|(招标|采购)人|项目概况|估算投资|预计招标时间|招标内容|其他)', origin_inner_table[i][j][0])):
- inner_table[i][j] = [origin_inner_table[i][j][0], 1]
- elif origin_inner_table[i][j][0] in ['经评审的最低评标价法'] and predict_list[i][j]==1:
- inner_table[i][j] = [origin_inner_table[i][j][0], 0]
- if show:
- print(json.dumps(inner_table, ensure_ascii=False))
- print("="*80)
- print("table_head before repair")
- for r in inner_table:
- print('row', r)
- print("="*80)
- # 表头修正
- # repairTable(inner_table)
- inner_table = table_head_repair_process(inner_table, docid)
- # 组合结果
- for i in range(len(inner_table)):
- for j in range(len(inner_table[i])):
- inner_table[i][j] = [origin_inner_table[i][j][0], int(inner_table[i][j][1])]
- if show:
- print("table_head after repair")
- for r in inner_table:
- print('row', r)
- print("="*80)
- # 按表头分割表格
- head_list = sliceTable(inner_table)
- return inner_table, head_list
- def setHead_incontext(inner_table,pat_head,fix_value="~~",prob_min=0.5):
- data_x,data_position = getPredictor("form").getModel("context").encode(inner_table)
- predict_y = getPredictor("form").getModel("context").predict(data_x)
- for _position,_y in zip(data_position,predict_y):
- _w = _position[0]
- _h = _position[1]
- if _y[1]>prob_min:
- inner_table[_h][_w][1] = 1
- else:
- inner_table[_h][_w][1] = 0
- _item = inner_table[_h][_w][0]
- if re.search(pat_head,_item) is not None and len(_item)<8:
- inner_table[_h][_w][1] = 1
- # print("=====")
- # for item in inner_table:
- # print(item)
- # print("======")
- height = len(inner_table)
- width = len(inner_table[0])
- for i in range(height):
- for j in range(width):
- if re.search("[::]$", inner_table[i][j][0]) and len(inner_table[i][j][0])<8:
- inner_table[i][j][1] = 1
- repairTable(inner_table)
- head_list = sliceTable(inner_table)
- # print("inner_table:",inner_table)
- return inner_table,head_list
-
- #设置表头
- def setHead_inline(inner_table,prob_min=0.64):
- pad_row = "@@"
- pad_col = "##"
- removePadding(inner_table, pad_row, pad_col)
- pad_pattern = re.compile(pad_row+"|"+pad_col)
- height = len(inner_table)
- width = len(inner_table[0])
- head_list = []
- head_list.append(0)
- #行表头
- is_head_last = False
- for i in range(height):
-
- is_head = False
- is_long_value = False
-
- #判断是否是全padding值
- is_same_value = True
- same_value = inner_table[i][0][0]
- for j in range(width):
- if inner_table[i][j][0]!=same_value and inner_table[i][j][0]!=pad_row:
- is_same_value = False
- break
-
- #predict is head or not with model
- temp_item = ""
- for j in range(width):
- temp_item += inner_table[i][j][0]+"|"
- temp_item = re.sub(pad_pattern,"",temp_item)
- form_prob = getPredictor("form").predict(formEncoding(temp_item,expand=True),type="line")
- if form_prob is not None:
- if form_prob[0][1]>prob_min:
- is_head = True
- else:
- is_head = False
-
- #print(temp_item,form_prob)
- if len(inner_table[i][0][0])>40:
- is_long_value = True
- if is_head or is_long_value or is_same_value:
- #不把连续表头分开
- if not is_head_last:
- head_list.append(i)
- if is_long_value or is_same_value:
- head_list.append(i+1)
- if is_head:
- for j in range(width):
- inner_table[i][j][1] = 1
- is_head_last = is_head
- head_list.append(height)
- #列表头
- for i in range(len(head_list)-1):
- head_begin = head_list[i]
- head_end = head_list[i+1]
- #最后一列不设置为列表头
- for i in range(width-1):
- is_head = False
-
- #predict is head or not with model
- temp_item = ""
- for j in range(head_begin,head_end):
- temp_item += inner_table[j][i][0]+"|"
- temp_item = re.sub(pad_pattern,"",temp_item)
- form_prob = getPredictor("form").predict(formEncoding(temp_item,expand=True),type="line")
- if form_prob is not None:
- if form_prob[0][1]>prob_min:
- is_head = True
- else:
- is_head = False
-
- if is_head:
- for j in range(head_begin,head_end):
- inner_table[j][i][1] = 2
- addPadding(inner_table, pad_row, pad_col)
- return inner_table,head_list
-
- #设置表头
- def setHead_withRule(inner_table,pattern,pat_value,count):
- height = len(inner_table)
- width = len(inner_table[0])
- head_list = []
- head_list.append(0)
- #行表头
- is_head_last = False
- for i in range(height):
- set_match = set()
- is_head = False
- is_long_value = False
- is_same_value = True
- same_value = inner_table[i][0][0]
- for j in range(width):
- if inner_table[i][j][0]!=same_value:
- is_same_value = False
- break
- for j in range(width):
- if re.search(pat_value,inner_table[i][j][0]) is not None:
- is_head = False
- break
- str_find = re.findall(pattern,inner_table[i][j][0])
- if len(str_find)>0:
- set_match.add(inner_table[i][j][0])
- if len(set_match)>=count:
- is_head = True
- if len(inner_table[i][0][0])>40:
- is_long_value = True
- if is_head or is_long_value or is_same_value:
- if not is_head_last:
- head_list.append(i)
- if is_head:
- for j in range(width):
- inner_table[i][j][1] = 1
- is_head_last = is_head
- head_list.append(height)
- #列表头
- for i in range(len(head_list)-1):
- head_begin = head_list[i]
- head_end = head_list[i+1]
- #最后一列不设置为列表头
- for i in range(width-1):
- set_match = set()
- is_head = False
- for j in range(head_begin,head_end):
- if re.search(pat_value,inner_table[j][i][0]) is not None:
- is_head = False
- break
- str_find = re.findall(pattern,inner_table[j][i][0])
- if len(str_find)>0:
- set_match.add(inner_table[j][i][0])
- if len(set_match)>=count:
- is_head = True
- if is_head:
- for j in range(head_begin,head_end):
- inner_table[j][i][1] = 2
- return inner_table,head_list
-
- #取得表格的处理方向
- def getDirect(inner_table,begin,end):
- '''
- column_head = set()
- row_head = set()
- widths = len(inner_table[0])
- for height in range(begin,end):
- for width in range(widths):
- if inner_table[height][width][1] ==1:
- row_head.add(height)
- if inner_table[height][width][1] ==2:
- column_head.add(width)
- company_pattern = re.compile("公司")
- if 0 in column_head and begin not in row_head:
- return "column"
- if 0 in column_head and begin in row_head:
- for height in range(begin,end):
- count = 0
- count_flag = True
- for width_index in range(width):
- if inner_table[height][width_index][1]==0:
- if re.search(company_pattern,inner_table[height][width_index][0]) is not None:
- count += 1
- else:
- count_flag = False
- if count_flag and count>=2:
- return "column"
- return "row"
- '''
- count_row_keys = 0
- count_column_keys = 0
- width = len(inner_table[0])
- if begin<end:
- for w in range(len(inner_table[begin])):
- if inner_table[begin][w][1]!=0:
- count_row_keys += 1
- for h in range(begin,end):
- if inner_table[h][0][1]!=0:
- count_column_keys += 1
-
- company_pattern = re.compile("有限(责任)?公司")
- for height in range(begin,end):
- count_set = set()
- count_flag = True
- for width_index in range(width):
- if inner_table[height][width_index][1]==0:
- if re.search(company_pattern,inner_table[height][width_index][0]) is not None:
- count_set.add(inner_table[height][width_index][0])
- else:
- count_flag = False
- if count_flag and len(count_set)>=2:
- return "column"
- # if count_column_keys>count_row_keys: #2022/2/15 此项不够严谨,造成很多错误,故取消
- # return "column"
- return "row"
-
-
- #根据表格处理方向生成句子,
- def getTableText(inner_table,head_list,key_direct=False):
- # packPattern = "(标包|[标包][号段名])"
- packPattern = "(标包|标的|标项|品目|[标包][号段名]|((项目|物资|设备|场次|标段|标的|产品)(名称)))" # 2020/11/23 大网站规则,补充采购类包名
- rankPattern = "(排名|排序|名次|序号|评标结果|评审结果|是否中标|推荐(意见|情况)|评标情况|推荐顺序|选取(情况|说明))" # 2020/11/23 大网站规则,添加序号为排序
- entityPattern = "((候选|[中投]标|报价)(单位|公司|人|供应商))|供应商名称"
- moneyPattern = "([中投]标|报价)(金额|价)"
- height = len(inner_table)
- width = len(inner_table[0])
- text = ""
- for head_i in range(len(head_list)-1):
-
- head_begin = head_list[head_i]
- head_end = head_list[head_i+1]
-
- direct = getDirect(inner_table, head_begin, head_end)
- #若只有一行,则直接按行读取
- if head_end-head_begin==1:
- text_line = ""
- for i in range(head_begin,head_end):
- for w in range(len(inner_table[i])):
- if inner_table[i][w][1]==1:
- _punctuation = ":"
- else:
- _punctuation = "," #2021/12/15 统一为中文标点,避免 206893924 国际F座1108,1,009,197.49元
- if w>0:
- if inner_table[i][w][0]!= inner_table[i][w-1][0]:
- text_line += inner_table[i][w][0]+_punctuation
- else:
- text_line += inner_table[i][w][0]+_punctuation
- text_line = text_line+"。" if text_line!="" else text_line
- text += text_line
- else:
- #构建一个共现矩阵
- table_occurence = []
- for i in range(head_begin,head_end):
- line_oc = []
- for j in range(width):
- cell = inner_table[i][j]
- line_oc.append({"text":cell[0],"type":cell[1],"occu_count":0,"left_head":"","top_head":"","left_dis":0,"top_dis":0})
- table_occurence.append(line_oc)
- occu_height = len(table_occurence)
- occu_width = len(table_occurence[0]) if len(table_occurence)>0 else 0
- #为每个属性值寻找表头
- for i in range(occu_height):
- for j in range(occu_width):
- cell = table_occurence[i][j]
- #是属性值
- if cell["type"]==0 and cell["text"]!="":
- left_head = ""
- top_head = ""
- find_flag = False
- temp_head = ""
- for loop_i in range(1,i+1):
- if not key_direct:
- key_values = [1,2]
- else:
- key_values = [1]
- if table_occurence[i-loop_i][j]["type"] in key_values:
- if find_flag:
- if table_occurence[i-loop_i][j]["text"]!=temp_head:
- top_head = table_occurence[i-loop_i][j]["text"]+":"+top_head
- else:
- top_head = table_occurence[i-loop_i][j]["text"]+":"+top_head
- find_flag = True
- temp_head = table_occurence[i-loop_i][j]["text"]
- table_occurence[i-loop_i][j]["occu_count"] += 1
- else:
- #找到表头后遇到属性值就返回
- if find_flag:
- break
- cell["top_head"] += top_head
- find_flag = False
- temp_head = ""
- for loop_j in range(1,j+1):
- if not key_direct:
- key_values = [1,2]
- else:
- key_values = [2]
- if table_occurence[i][j-loop_j]["type"] in key_values:
- if find_flag:
- if table_occurence[i][j-loop_j]["text"]!=temp_head:
- left_head = table_occurence[i][j-loop_j]["text"]+":"+left_head
- else:
- left_head = table_occurence[i][j-loop_j]["text"]+":"+left_head
- find_flag = True
- temp_head = table_occurence[i][j-loop_j]["text"]
- table_occurence[i][j-loop_j]["occu_count"] += 1
- else:
- if find_flag:
- break
- cell["left_head"] += left_head
- if direct=="row":
- for i in range(occu_height):
- pack_text = ""
- rank_text = ""
- entity_text = ""
- text_line = ""
- money_text = ""
- #在同一句话中重复的可以去掉
- text_set = set()
- head = ""
- last_text = ""
- for j in range(width):
- cell = table_occurence[i][j]
- if cell["type"]==0 or (cell["type"]==1 and cell["occu_count"]==0):
- cell = table_occurence[i][j]
- head = (cell["top_head"]+":") if len(cell["top_head"])>0 else ""
- if re.search("[单报标限总]价|金额|成交报?价|报价|供应商|候选人|中标人|[利费]率|负责人|工期|服务(期限?|年限|时间|日期|周期)|(履约|履行)期限|合同(期限?|(完成|截止)(日期|时间))", head):
- head = cell["left_head"] + head
- else:
- head += cell["left_head"]
- if str(head+cell["text"]) in text_set:
- continue
- if re.search(packPattern,head) is not None:
- pack_text += head+cell["text"]+","
- elif re.search(rankPattern,head) is not None and re.search('(排名|排序|名次|顺序):?第?[\d一二三]', rank_text)==None: # 2020/11/23 大网站规则发现问题,if 改elif 20240620修复同时有排名及评标情况造成错误
- #排名替换为同一种表达
- rank_text += head+cell["text"]+","
- #print(rank_text)
- elif re.search(entityPattern,head) is not None:
- entity_text += head+cell["text"]+","
- #print(entity_text)
- else:
- if re.search(moneyPattern,head) is not None and entity_text!="":
- money_text += head+cell["text"]+","
- else:
- text_line += head+cell["text"]+","
- text_set.add(str(head+cell["text"]))
- last_text = cell['text']
- tr_text = pack_text+rank_text+entity_text+money_text+text_line
- text += pack_text+rank_text+entity_text+money_text+text_line
- # text = text[:-1] + "。" if len(text) > 0 else text
- if len(text_set-set([' ']))==1 and head == '' and len(last_text)< 25: # 修复367694716分两行表达
- text = text if re.search('\w$', text[:-1]) else text[:-1]
- elif (width == 2 or len(text_set)==1) and head != '' and len(tr_text)<50: # 修复494731937只有两行的,分句不合理
- text = text if re.search('\w$', text[:-1]) else text[:-1]
- else:
- text = text[:-1] + "。"
- else:
- for j in range(occu_width):
- pack_text = ""
- rank_text = ""
- entity_text = ""
- text_line = ""
- text_set = set()
- for i in range(occu_height):
- cell = table_occurence[i][j]
- if cell["type"]==0 or (cell["type"]==1 and cell["occu_count"]==0):
- cell = table_occurence[i][j]
- head = (cell["left_head"]+"") if len(cell["left_head"])>0 else ""
- if re.search("[单报标限总]价|金额|成交报?价|报价|供应商|候选人|中标人|[利费]率|负责人|工期|服务(期限?|年限|时间|日期|周期)|(履约|履行)期限|合同(期限?|(完成|截止)(日期|时间))", head):
- head = cell["top_head"] + head
- else:
- head += cell["top_head"]
- if str(head+cell["text"]) in text_set:
- continue
- if re.search(packPattern,head) is not None:
- pack_text += head+cell["text"]+","
- elif re.search(rankPattern,head) is not None: # 2020/11/23 大网站规则发现问题,if 改elif
- #排名替换为同一种表达
- rank_text += head+cell["text"]+","
- #print(rank_text)
- elif re.search(entityPattern,head) is not None and \
- re.search('业绩|资格|条件',head)==None and re.search('业绩',cell["text"])==None : #2021/10/19 解决包含业绩的行调到前面问题
- entity_text += head+cell["text"]+","
- #print(entity_text)
- else:
- text_line += head+cell["text"]+","
- text_set.add(str(head+cell["text"]))
- text += pack_text+rank_text+entity_text+text_line
- text = text[:-1]+"。" if len(text)>0 else text
- # if direct=="row":
- # for i in range(head_begin,head_end):
- # pack_text = ""
- # rank_text = ""
- # entity_text = ""
- # text_line = ""
- # #在同一句话中重复的可以去掉
- # text_set = set()
- # for j in range(width):
- # cell = inner_table[i][j]
- # #是属性值
- # if cell[1]==0 and cell[0]!="":
- # head = ""
- #
- # find_flag = False
- # temp_head = ""
- # for loop_i in range(0,i+1-head_begin):
- # if not key_direct:
- # key_values = [1,2]
- # else:
- # key_values = [1]
- # if inner_table[i-loop_i][j][1] in key_values:
- # if find_flag:
- # if inner_table[i-loop_i][j][0]!=temp_head:
- # head = inner_table[i-loop_i][j][0]+":"+head
- # else:
- # head = inner_table[i-loop_i][j][0]+":"+head
- # find_flag = True
- # temp_head = inner_table[i-loop_i][j][0]
- # else:
- # #找到表头后遇到属性值就返回
- # if find_flag:
- # break
- #
- # find_flag = False
- # temp_head = ""
- #
- #
- #
- # for loop_j in range(1,j+1):
- # if not key_direct:
- # key_values = [1,2]
- # else:
- # key_values = [2]
- # if inner_table[i][j-loop_j][1] in key_values:
- # if find_flag:
- # if inner_table[i][j-loop_j][0]!=temp_head:
- # head = inner_table[i][j-loop_j][0]+":"+head
- # else:
- # head = inner_table[i][j-loop_j][0]+":"+head
- # find_flag = True
- # temp_head = inner_table[i][j-loop_j][0]
- # else:
- # if find_flag:
- # break
- #
- # if str(head+inner_table[i][j][0]) in text_set:
- # continue
- # if re.search(packPattern,head) is not None:
- # pack_text += head+inner_table[i][j][0]+","
- # elif re.search(rankPattern,head) is not None: # 2020/11/23 大网站规则发现问题,if 改elif
- # #排名替换为同一种表达
- # rank_text += head+inner_table[i][j][0]+","
- # #print(rank_text)
- # elif re.search(entityPattern,head) is not None:
- # entity_text += head+inner_table[i][j][0]+","
- # #print(entity_text)
- # else:
- # text_line += head+inner_table[i][j][0]+","
- # text_set.add(str(head+inner_table[i][j][0]))
- # text += pack_text+rank_text+entity_text+text_line
- # text = text[:-1]+"。" if len(text)>0 else text
- # else:
- # for j in range(width):
- #
- # rank_text = ""
- # entity_text = ""
- # text_line = ""
- # text_set = set()
- # for i in range(head_begin,head_end):
- # cell = inner_table[i][j]
- # #是属性值
- # if cell[1]==0 and cell[0]!="":
- # find_flag = False
- # head = ""
- # temp_head = ""
- #
- # for loop_j in range(1,j+1):
- # if not key_direct:
- # key_values = [1,2]
- # else:
- # key_values = [2]
- # if inner_table[i][j-loop_j][1] in key_values:
- # if find_flag:
- # if inner_table[i][j-loop_j][0]!=temp_head:
- # head = inner_table[i][j-loop_j][0]+":"+head
- # else:
- # head = inner_table[i][j-loop_j][0]+":"+head
- # find_flag = True
- # temp_head = inner_table[i][j-loop_j][0]
- # else:
- # if find_flag:
- # break
- # find_flag = False
- # temp_head = ""
- # for loop_i in range(0,i+1-head_begin):
- # if not key_direct:
- # key_values = [1,2]
- # else:
- # key_values = [1]
- # if inner_table[i-loop_i][j][1] in key_values:
- # if find_flag:
- # if inner_table[i-loop_i][j][0]!=temp_head:
- # head = inner_table[i-loop_i][j][0]+":"+head
- # else:
- # head = inner_table[i-loop_i][j][0]+":"+head
- # find_flag = True
- # temp_head = inner_table[i-loop_i][j][0]
- # else:
- # if find_flag:
- # break
- # if str(head+inner_table[i][j][0]) in text_set:
- # continue
- # if re.search(rankPattern,head) is not None:
- # rank_text += head+inner_table[i][j][0]+","
- # #print(rank_text)
- # elif re.search(entityPattern,head) is not None:
- # entity_text += head+inner_table[i][j][0]+","
- # #print(entity_text)
- # else:
- # text_line += head+inner_table[i][j][0]+","
- # text_set.add(str(head+inner_table[i][j][0]))
- # text += rank_text+entity_text+text_line
- # text = text[:-1]+"。" if len(text)>0 else text
- if text.endswith(',。'):
- text = re.sub(',+。', '。', text)
- elif text.endswith(','):
- text = text.rstrip(',') + '。'
- else:
- text += '。' # 20260401 表格末尾加句号与其他内容分开 749870793 避免表格与外面混淆
- return text
- def get_table_text_kv(inner_table, head_list, key_direct=False):
- packPattern = "(标包|标的|标项|品目|[标包][号段名]|((项目|物资|设备|场次|标段|标的|产品)(名称)))" # 2020/11/23 大网站规则,补充采购类包名
- rankPattern = "(排名|排序|名次|序号|评标结果|评审结果|是否中标|推荐意见|评标情况|推荐顺序|选取(情况|说明))" # 2020/11/23 大网站规则,添加序号为排序
- entityPattern = "((候选|[中投]标|报价)(单位|公司|人|供应商))|供应商名称"
- moneyPattern = "([中投]标|报价)(金额|价)"
- width = len(inner_table[0])
- text = ""
- all_table_occurence = []
- for head_i in range(len(head_list) - 1):
- head_begin = head_list[head_i]
- head_end = head_list[head_i + 1]
- direct = getDirect(inner_table, head_begin, head_end)
- # print(inner_table[head_begin:head_end])
- # print('direct', direct)
- # 构建一个共现矩阵
- table_occurence = []
- for i in range(head_begin, head_end):
- line_oc = []
- for j in range(width):
- cell = inner_table[i][j]
- line_oc.append(
- {"text": cell[0], "type": cell[1], "occu_count": 0, "left_head": "", "top_head": "",
- "left_dis": 0, "top_dis": 0,
- "text_row_index": i, "text_col_index": j
- })
- table_occurence.append(line_oc)
- occu_height = len(table_occurence)
- occu_width = len(table_occurence[0]) if len(table_occurence) > 0 else 0
- # 为每个属性值寻找表头
- for i in range(occu_height):
- for j in range(occu_width):
- cell = table_occurence[i][j]
- # 是属性值
- if cell["type"] == 0 and cell["text"] != "":
- left_head = ""
- top_head = ""
- find_flag = False
- temp_head = ""
- head_row_col_list = []
- for loop_i in range(1, i + 1):
- if not key_direct:
- key_values = [1, 2]
- else:
- key_values = [1]
- if table_occurence[i - loop_i][j]["type"] in key_values:
- if find_flag:
- if table_occurence[i - loop_i][j]["text"] != temp_head:
- if cell.get("top_head_list"):
- cell["top_head_list"] += [table_occurence[i - loop_i][j]["text"] + ":"]
- else:
- cell["top_head_list"] = [table_occurence[i - loop_i][j]["text"] + ":"]
- top_head = table_occurence[i - loop_i][j]["text"] + ":" + top_head
- head_row_col_list.append([i - loop_i, j])
- else:
- if cell.get("top_head_list"):
- cell["top_head_list"] += [table_occurence[i - loop_i][j]["text"] + ":"]
- else:
- cell["top_head_list"] = [table_occurence[i - loop_i][j]["text"] + ":"]
- top_head = table_occurence[i - loop_i][j]["text"] + ":" + top_head
- head_row_col_list.append([i - loop_i, j])
- find_flag = True
- temp_head = table_occurence[i - loop_i][j]["text"]
- table_occurence[i - loop_i][j]["occu_count"] += 1
- else:
- # 找到表头后遇到属性值就返回
- if find_flag:
- break
- cell["top_head"] += top_head
- if cell.get("top_head_row_index"):
- cell["top_head_row_index"] += [x[0] for x in head_row_col_list]
- else:
- cell["top_head_row_index"] = [x[0] for x in head_row_col_list]
- if cell.get("top_head_col_index"):
- cell["top_head_col_index"] += [x[1] for x in head_row_col_list]
- else:
- cell["top_head_col_index"] = [x[1] for x in head_row_col_list]
- find_flag = False
- temp_head = ""
- head_row_col_list = []
- for loop_j in range(1, j + 1):
- if not key_direct:
- key_values = [1, 2]
- else:
- key_values = [2]
- if table_occurence[i][j - loop_j]["type"] in key_values:
- if find_flag:
- if table_occurence[i][j - loop_j]["text"] != temp_head:
- if cell.get("left_head_list"):
- cell["left_head_list"] += [table_occurence[i][j - loop_j]["text"] + ":"]
- else:
- cell["left_head_list"] = [table_occurence[i][j - loop_j]["text"] + ":"]
- left_head = table_occurence[i][j - loop_j]["text"] + ":" + left_head
- head_row_col_list.append([i, j - loop_j])
- else:
- if cell.get("left_head_list"):
- cell["left_head_list"] += [table_occurence[i][j - loop_j]["text"] + ":"]
- else:
- cell["left_head_list"] = [table_occurence[i][j - loop_j]["text"] + ":"]
- left_head = table_occurence[i][j - loop_j]["text"] + ":" + left_head
- head_row_col_list.append([i, j - loop_j])
- find_flag = True
- temp_head = table_occurence[i][j - loop_j]["text"]
- table_occurence[i][j - loop_j]["occu_count"] += 1
- else:
- if find_flag:
- break
- cell["left_head"] += left_head
- if cell.get("left_head_row_index"):
- cell["left_head_row_index"] += [x[0] for x in head_row_col_list]
- else:
- cell["left_head_row_index"] = [x[0] for x in head_row_col_list]
- if cell.get("left_head_col_index"):
- cell["left_head_col_index"] += [x[1] for x in head_row_col_list]
- else:
- cell["left_head_col_index"] = [x[1] for x in head_row_col_list]
- # 连接表头和属性值
- if direct == "row":
- for i in range(occu_height):
- pack_text = ""
- rank_text = ""
- entity_text = ""
- text_line = ""
- money_text = ""
- # 在同一句话中重复的可以去掉
- text_set = set()
- head = ""
- last_text = ""
- pack_text_location = []
- rank_text_location = []
- entity_text_location = []
- text_line_location = []
- money_text_location = []
- for j in range(width):
- cell = table_occurence[i][j]
- if cell["type"] == 0 or (cell["type"] == 1 and cell["occu_count"] == 0):
- cell = table_occurence[i][j]
- head = (cell["top_head"] + ":") if len(cell["top_head"]) > 0 else ""
- now_top_head = copy.deepcopy(head)
- now_left_head = copy.deepcopy(cell["left_head"])
- if re.search(
- "[单报标限总]价|金额|成交报?价|报价|供应商|候选人|中标人|[利费]率|负责人|工期|服务(期限?|年限|时间|日期|周期)|("
- "履约|履行)期限|合同(期限?|(完成|截止)(日期|时间))",
- head):
- head = cell["left_head"] + head
- left_first = 1
- else:
- head += cell["left_head"]
- left_first = 0
- # print('len(text), len(sub_text), len(head)', cell["text"], len(text), len(sub_text), len(head))
- # print('text111', text)
- # print('pack_text, rank_text, entity_text, money_text, text_line', '1'+pack_text, '2'+rank_text, '3'+entity_text, '4'+money_text, '5'+text_line)
- # print('head', head)
- # print('sub_text111', sub_text)
- if str(head + cell["text"]) in text_set:
- cell['drop'] = 1
- continue
- if re.search(packPattern, head) is not None:
- pack_text += head + cell["text"] + ","
- pack_text_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]]
- # 2020/11/23 大网站规则发现问题,if 改elif 20240620修复同时有排名及评标情况造成错误
- elif re.search(rankPattern, head) is not None and re.search('(排名|排序|名次|顺序):?第?[\d一二三]', rank_text) is None:
- # 排名替换为同一种表达
- rank_text += head + cell["text"] + ","
- rank_text_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]]
- elif re.search(entityPattern, head) is not None:
- entity_text += head + cell["text"] + ","
- entity_text_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]]
- else:
- if re.search(moneyPattern, head) is not None and entity_text != "":
- money_text += head + cell["text"] + ","
- money_text_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]]
- else:
- text_line += head + cell["text"] + ","
- text_line_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]]
- text_set.add(str(head + cell["text"]))
- last_text = cell['text']
- # 计算key value在sentence的index
- head_location_list = pack_text_location + rank_text_location + entity_text_location + text_line_location + money_text_location
- current_loc = 0
- for ii, jj, head_text, now_left_head, now_top_head, left_first in head_location_list:
- cell = table_occurence[ii][jj]
- # 左表头先于右表头
- if left_first:
- cell['left_head_sen_index'] = len(text) + current_loc
- cell['top_head_sen_index'] = len(text) + current_loc + len(now_left_head)
- else:
- cell['left_head_sen_index'] = len(text) + current_loc + len(now_top_head)
- cell['top_head_sen_index'] = len(text) + current_loc
- cell['text_sen_index'] = len(text) + current_loc + len(now_left_head + now_top_head)
- current_loc += len(head_text)
- tr_text = pack_text + rank_text + entity_text + money_text + text_line
- text += pack_text + rank_text + entity_text + money_text + text_line
- # 修复367694716分两行表达
- if len(text_set - set([' '])) == 1 and head == '' and len(last_text) < 25:
- text = text if re.search('\w$', text[:-1]) else text[:-1]
- # 修复494731937只有两行的,分句不合理
- elif (width == 2 or len(text_set) == 1) and head != '' and len(tr_text) < 50:
- text = text if re.search('\w$', text[:-1]) else text[:-1]
- else:
- text = text[:-1] + "。"
- else:
- for j in range(occu_width):
- pack_text = ""
- rank_text = ""
- entity_text = ""
- text_line = ""
- text_set = set()
- pack_text_location = []
- rank_text_location = []
- entity_text_location = []
- text_line_location = []
- money_text_location = []
- for i in range(occu_height):
- cell = table_occurence[i][j]
- if cell["type"] == 0 or (cell["type"] == 1 and cell["occu_count"] == 0):
- cell = table_occurence[i][j]
- head = (cell["left_head"] + "") if len(cell["left_head"]) > 0 else ""
- now_top_head = copy.deepcopy(cell["top_head"])
- now_left_head = copy.deepcopy(head)
- if re.search("[单报标限总]价|金额|成交报?价|报价|供应商|候选人|中标人|[利费]率|负责人|工期|服务(期限?|年限|时间|日期|周期)|(履约|履行)期限|合同(期限?|(完成|截止)(日期|时间))", head):
- head = cell["top_head"] + head
- left_first = 0
- else:
- head += cell["top_head"]
- left_first = 1
- if str(head + cell["text"]) in text_set:
- cell['drop'] = 1
- continue
- if re.search(packPattern, head) is not None:
- pack_text += head + cell["text"] + ","
- pack_text_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]]
- # 2020/11/23 大网站规则发现问题,if 改elif
- elif re.search(rankPattern, head) is not None:
- # 排名替换为同一种表达
- rank_text += head + cell["text"] + ","
- rank_text_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]]
- # 2021/10/19 解决包含业绩的行调到前面问题
- elif re.search(entityPattern, head) is not None and \
- re.search('业绩|资格|条件', head) is None and re.search('业绩', cell["text"]) is None:
- entity_text += head + cell["text"] + ","
- entity_text_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]]
- else:
- text_line += head + cell["text"] + ","
- text_line_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]]
- text_set.add(str(head + cell["text"]))
- # 计算key value在sentence的index
- head_location_list = pack_text_location + rank_text_location + entity_text_location + text_line_location + money_text_location
- current_loc = 0
- for ii, jj, head_text, now_left_head, now_top_head, left_first in head_location_list:
- cell = table_occurence[ii][jj]
- # 左表头先于右表头
- if left_first:
- cell['left_head_sen_index'] = len(text) + current_loc
- cell['top_head_sen_index'] = len(text) + current_loc + len(now_left_head)
- else:
- cell['left_head_sen_index'] = len(text) + current_loc + len(now_top_head)
- cell['top_head_sen_index'] = len(text) + current_loc
- cell['text_sen_index'] = len(text) + current_loc + len(now_left_head + now_top_head)
- current_loc += len(head_text)
- text += pack_text + rank_text + entity_text + text_line
- text = text[:-1] + "。" if len(text) > 0 else text
- all_table_occurence += table_occurence
- return text, all_table_occurence
- def process_dict(text, table):
- kv_list = []
- kv_dict_list = []
- # print('text', len(text), text, ),
- # print('table', table)
- for r_index, row in enumerate(table):
- for c_index, col in enumerate(row):
- # print('col', col)
- if col['type'] == 1:
- continue
- if col.get('drop'):
- continue
- if not col.get('left_head_list') and not col.get('top_head_list'):
- _d = {
- 'value': col['text'],
- 'value_row_index': col['text_row_index'],
- 'value_col_index': col['text_col_index'],
- 'value_sen_index': col['text_sen_index'],
- 'sen_value': text[col['text_sen_index']:col['text_sen_index'] + len(col['text'])],
- }
- kv_dict_list.append(_d)
- continue
- if col.get('text_sen_index') and col.get('text_sen_index') >= len(text):
- # print('continue1')
- continue
- if col.get('left_head_list'):
- # head, head_row_index, head_col_index 按文本顺序排序
- zip_list = list(
- zip(col.get('left_head_list'), col.get('left_head_row_index'), col.get('left_head_col_index')))
- zip_list.sort(key=lambda x: (x[1], x[2]))
- col['left_head_list'], col['left_head_row_index'], col['left_head_col_index'] = zip(*zip_list)
- last_head = ""
- for h_index, head in enumerate(col.get('left_head_list')):
- _d = {
- 'key': head,
- 'value': col['text'],
- 'key_row_index': col['left_head_row_index'][h_index],
- 'key_col_index': col['left_head_col_index'][h_index],
- 'key_sen_index': col['left_head_sen_index'] + len(last_head),
- 'value_row_index': col['text_row_index'],
- 'value_col_index': col['text_col_index'],
- 'value_sen_index': col['text_sen_index'],
- 'sen_key': text[
- col['left_head_sen_index'] + len(last_head):col['left_head_sen_index'] + len(
- last_head) + len(head)],
- 'sen_value': text[col['text_sen_index']:col['text_sen_index'] + len(col['text'])],
- }
- kv_dict_list.append(_d)
- last_head += head
- if col.get('top_head_list'):
- # head, head_row_index, head_col_index 按文本顺序排序
- zip_list = list(
- zip(col.get('top_head_list'), col.get('top_head_row_index'), col.get('top_head_col_index')))
- zip_list.sort(key=lambda x: (x[1], x[2]))
- col['top_head_list'], col['top_head_row_index'], col['top_head_col_index'] = zip(*zip_list)
- last_head = ""
- for h_index, head in enumerate(col.get('top_head_list')):
- _d = {
- 'key': head,
- 'value': col['text'],
- 'key_row_index': col['top_head_row_index'][h_index],
- 'key_col_index': col['top_head_col_index'][h_index],
- 'key_sen_index': col['top_head_sen_index'] + len(last_head),
- 'value_row_index': col['text_row_index'],
- 'value_col_index': col['text_col_index'],
- 'value_sen_index': col['text_sen_index'],
- 'sen_key': text[col['top_head_sen_index'] + len(last_head):col['top_head_sen_index'] + len(
- last_head) + len(head)],
- 'sen_value': text[col['text_sen_index']:col['text_sen_index'] + len(col['text'])],
- }
- kv_dict_list.append(_d)
- last_head += head
- return kv_list, kv_dict_list
- def removeFix(inner_table,fix_value="~~"):
- height = len(inner_table)
- width = len(inner_table[0])
- for h in range(height):
- for w in range(width):
- if inner_table[h][w][0]==fix_value:
- inner_table[h][w][0] = ""
-
- def trunTable(tbody,in_attachment):
- # print(tbody.find('tbody'))
- # 附件中的表格,排除异常错乱的表格
- if in_attachment:
- if tbody.name=='table':
- _tbody = tbody.find('tbody')
- if _tbody is None:
- _tbody = tbody
- else:
- _tbody = tbody
- _td_len_list = []
- for _tr in _tbody.find_all(recursive=False):
- len_td = len(_tr.find_all(recursive=False))
- _td_len_list.append(len_td)
- if _td_len_list:
- if len(list(set(_td_len_list))) >= 8 or max(_td_len_list) > 100:
- string_list = [re.sub("\s+","",i)for i in tbody.strings if i and i!='\n']
- tbody.string = ",".join(string_list)
- table_max_len = 30000
- tbody.string = tbody.string[:table_max_len]
- tbody.name = "turntable"
- if return_kv:
- return None, None, None
- return None
- # fixSpan(tbody)
- # inner_table = getTable(tbody)
- # inner_table = fixTable(inner_table)
- table2list = TableTag2List()
- return_html_table = True if return_kv else False
- if return_html_table:
- inner_table, html_table = table2list.table2list(tbody, segment, return_html_table,return_kv=return_kv)
- inner_table = fixTable(inner_table)
- html_table = fixTable(html_table, "")
- else:
- inner_table = table2list.table2list(tbody, segment,return_kv=return_kv)
- inner_table = fixTable(inner_table)
- if inner_table == []:
- string_list = [re.sub("\s+", "", i) for i in tbody.strings if i and i != '\n']
- tbody.string = ",".join(string_list)
- table_max_len = 30000
- tbody.string = tbody.string[:table_max_len]
- # log('异常表格直接取全文')
- tbody.name = "turntable"
- if return_kv:
- return None, None, None
- return None
- if len(inner_table)>0 and len(inner_table[0])>0:
- for tr in inner_table:
- for td in tr:
- if isinstance(td, str):
- tbody.string = segment(tbody,final=False)
- table_max_len = 30000
- tbody.string = tbody.string[:table_max_len]
- # log('异常表格,不做表格处理,直接取全文')
- tbody.name = "turntable"
- if return_kv:
- return None, None, None
- return None
- #inner_table,head_list = setHead_withRule(inner_table,pat_head,pat_value,3)
- #inner_table,head_list = setHead_inline(inner_table)
- # inner_table, head_list = setHead_initem(inner_table,pat_head)
- inner_table, head_list = set_head_model(inner_table)
- # inner_table,head_list = setHead_incontext(inner_table,pat_head)
- # print("table_head", inner_table)
- # print("head_list", head_list)
- # for begin in range(len(head_list[:-1])):
- # for item in inner_table[head_list[begin]:head_list[begin+1]]:
- # print(item)
- # print("====")
- removeFix(inner_table)
-
- # print("----")
- # print(head_list)
- # for item in inner_table:
- # print(item)
- # print('inner_table111', inner_table)
- if return_kv:
- text1, table = get_table_text_kv(inner_table, head_list)
- tbody.string = text1
- kv_list, kv_dict_list = process_dict(text1, table)
- # html放入dict
- for kv_dict in kv_dict_list:
- html = html_table[kv_dict.get('value_row_index')][kv_dict.get('value_col_index')]
- kv_dict['value_html'] = html
- else:
- tbody.string = getTableText(inner_table,head_list)
- table_max_len = 30000
- tbody.string = tbody.string[:table_max_len]
- if tbody.string.startswith('。'): # 20251224 修复 602074436 关键词与表格分开 中选单位:名称:
- tbody.string = ',' + tbody.string[1:]
- # print(tbody.string)
- tbody.name = "turntable"
- if return_kv:
- return inner_table, kv_dict_list, text1
- else:
- return inner_table
- if return_kv:
- return None, None, None
- return None
-
- pat_head = re.compile('^(名称|序号|项目|标项|工程|品目[一二三四1234]|第[一二三四1234](标段|名|候选人|中标)|包段|标包|分包|包号|货物|单位|数量|价格|报价|金额|总价|单价|[招投中]标|候选|编号|得分|评委|评分|名次|排名|排序|科室|方式|工期|时间|产品|开始|结束|联系|日期|面积|姓名|证号|备注|级别|地[点址]|类型|代理|制造|企业资质|质量目标|工期目标|(需求|服务|项目|施工|采购|招租|出租|转让|出让|业主|询价|委托|权属|招标|竞得|抽取|承建)(人|方|单位)(名称)?|(供应商|供货商|服务商)(名称)?)$')
- #pat_head = re.compile('(名称|序号|项目|工程|品目[一二三四1234]|第[一二三四1234](标段|候选人|中标)|包段|包号|货物|单位|数量|价格|报价|金额|总价|单价|[招投中]标|供应商|候选|编号|得分|评委|评分|名次|排名|排序|科室|方式|工期|时间|产品|开始|结束|联系|日期|面积|姓名|证号|备注|级别|地[点址]|类型|代理)')
- pat_value = re.compile("(\d{2,}.\d{1}|\d+年\d+月|\d{8,}|\d{3,}-\d{6,}|有限[责任]*公司|^\d+$)")
- list_innerTable = []
- # 2022/2/9 删除干扰标签
- for tag in soup.find_all('option'): #例子: 216661412
- if 'selected' not in tag.attrs:
- tag.extract()
- for ul in soup.find_all('ul'): #例子 156439663 多个不同channel 类别的标题
- if ul.find_all('li') == ul.findChildren(recursive=False) and len(set(re.findall(
- '招标公告|中标结果公示|中标候选人公示|招标答疑|开标评标|合同履?约?公示|资格评审',
- ul.get_text(), re.S)))>3:
- ul.extract()
- # tbodies = soup.find_all('table')
- # 遍历表格中的每个tbody
- tbodies = []
- in_attachment = False
- if soup.name=="table":
- tbodies.append((soup,in_attachment))
- for _part in soup.find_all():
- if _part.name=='table':
- tbodies.append((_part,in_attachment))
- elif _part.name=='div':
- if 'class' in _part.attrs and "richTextFetch" in _part['class']:
- in_attachment = True
- if return_kv and tbodies:
- tbodies = tbodies[:1]
- #逆序处理嵌套表格
- # print('len(tbodies)1', len(tbodies))
- # for tbody_index in range(1,len(tbodies)+1):
- tbody_index = 1
- while tbody_index < len(tbodies)+1:
- tbody,_in_attachment = tbodies[len(tbodies)-tbody_index]
- current_index = len(tbodies)-tbody_index
- if current_index > 0 and tbodies[current_index - 1][0].find_next_sibling() == tbody and len(tbodies[current_index - 1][0].find_all('tr')) == 1 and len(tbody.find_all('tr'))>=1 and len(tbody.tr.find_all(['th', 'td'])) == len(
- tbodies[current_index - 1][0].tr.find_all(['th', 'td'])): # 处理相邻表格都只有一行的情况;例:526321576 641881540
- for row in tbody.find_all('tr'): # 相邻表格只有一行且列数一样合并表格,标签必须是兄弟节点 避免 665587084 这种不是相邻标签的错误合并
- if tbodies[current_index - 1][0].tbody:
- tbodies[current_index - 1][0].tbody.append(row)
- else:
- tbodies[current_index - 1][0].append(row)
- inner_table = trunTable(tbodies[current_index - 1][0], _in_attachment)
- if inner_table:
- list_innerTable.append(inner_table)
- tbody_index += 2
- continue
- inner_table = trunTable(tbody,_in_attachment)
- list_innerTable.append(inner_table)
- tbody_index += 1
- # tbodies = soup.find_all('tbody')
- # 遍历表格中的每个tbody
- tbodies = []
- in_attachment = False
- if soup.name=="tbody" and soup.get_text().strip() != '': # 20250730 修复 642228216 表格分多个tbody且一三tbody只有tr内容为空导致提取失败
- tbodies.append((soup,in_attachment))
- for _part in soup.find_all():
- if _part.name == 'tbody':
- tbodies.append((_part, in_attachment))
- elif _part.name == 'div':
- if 'class' in _part.attrs and "richTextFetch" in _part['class']:
- in_attachment = True
- if return_kv and tbodies:
- tbodies = tbodies[:1]
- #逆序处理嵌套表格
- tbody_index = 1
- # for tbody_index in range(1,len(tbodies)+1):
- while tbody_index < len(tbodies) + 1:
- tbody,_in_attachment = tbodies[len(tbodies)-tbody_index]
- current_index = len(tbodies) - tbody_index
- if current_index > 0 and tbodies[current_index - 1][0].find_next_sibling() == tbody and len(
- tbodies[current_index - 1][0].find_all('tr')) == 1 and len(tbody.find_all('tr'))>=1 and len(tbody.tr.find_all(['th', 'td'])) == len(
- tbodies[current_index - 1][0].tr.find_all(['th', 'td'])) and len(tbody.tr.find_all(['th', 'td'])) > 2: # 处理相邻表格都只有一行的情况;例:526321576
- for row in tbody.find_all('tr'):
- if tbodies[current_index - 1][0].tbody:
- tbodies[current_index - 1][0].tbody.append(row)
- else:
- tbodies[current_index - 1][0].append(row)
- inner_table = trunTable(tbodies[current_index - 1][0], _in_attachment)
- if inner_table:
- list_innerTable.append(inner_table)
- tbody_index += 2
- continue
- inner_table = trunTable(tbody,_in_attachment)
- list_innerTable.append(inner_table)
- tbody_index += 1
- if return_kv:
- kv_list = []
- for x in list_innerTable:
- if x[1] is not None:
- kv_list.extend(x[1])
- text = ""
- for x in list_innerTable:
- if x[2] is not None:
- text += x[2]
- return soup, kv_list, text
- return soup
- # return list_innerTable
- def table_head_repair_process(_inner_table, docid=None, show=0, show_row_index=0):
- def pre_process(inner_table):
- """
- 修复前的预处理
- """
- # 循环处理单元格,一次获取需要的
- for i in range(len(inner_table)):
- for j in range(len(inner_table[i])):
- # 删除前后逗号
- inner_table[i][j][0] = re.sub('^[,, ]+', '', inner_table[i][j][0])
- inner_table[i][j][0] = re.sub('[,, ]+$', '', inner_table[i][j][0])
- inner_table[i][j][0] = re.sub('[, ]+', '', inner_table[i][j][0])
- return inner_table
- def repair_by_colon(inner_table):
- """
- 根据冒号修复当前格子的表头值
- """
- # 修复冒号在文本中间的,不能作为表头;(冒号后面需多个字)
- # 冒号在括号中的除外
- # 冒号在最后的,判断后一个格子是否有重复的文字
- for i in range(len(inner_table)):
- for j in range(len(inner_table[i])):
- _text = inner_table[i][j][0]
- if len(_text) >= 3 and inner_table[i][j][1] == 1:
- match = re.search('[::]', _text)
- if match:
- start_index, end_index = match.span()
- if start_index == 0:
- continue
- if end_index == len(_text):
- if len(inner_table[i]) == 2 and j <= len(inner_table[i]) - 2 and inner_table[i][j+1][0] and (_text in inner_table[i][j+1][0] or inner_table[i][j+1][0] in _text):
- inner_table[i][j][1] = 0
- inner_table[i][j+1][1] = 0
- else:
- continue
- if re.search('[((]', _text[:start_index]) and re.search('[))]', _text[end_index:]):
- continue
- m1 = re.search('[\u4e00-\u9fa50-9a-zA-Z]', _text[:start_index])
- m2 = re.search('[\u4e00-\u9fa50-9a-zA-Z]', _text[end_index:])
- if m1 and m2 and (len(m2.group()) >= 2 or m2.group() in ['是', '否']):
- inner_table[i][j][1] = 0
- return inner_table
- def repair_by_duplicate(inner_table):
- """
- 根据列重复修复当前格子的表头值
- """
- # 统计每个值的表头情况
- col_head_dict = {}
- for i in range(len(inner_table)):
- for j in range(len(inner_table[i])):
- col = inner_table[i][j]
- if col[0] in col_head_dict.keys():
- col_head_dict[col[0]] += [col[1]]
- else:
- col_head_dict[col[0]] = [col[1]]
- # 多个重复列的预测值不同,以第一个为准
- for i in range(len(inner_table)):
- col = inner_table[i][0]
- key = col[0] + '\t' + str(0)
- dup_dict = {}
- for j in range(len(inner_table[i])):
- if inner_table[i][j][0] == col[0]:
- if key in dup_dict.keys():
- dup_dict[key] += [j]
- else:
- dup_dict[key] = [j]
- # if inner_table[i][j][1] != col[1]:
- # if col != inner_table[i][0]:
- # inner_table[i][j][1] = col[1]
- # else:
- # inner_table[i][0][1] = inner_table[i][j][1]
- # col = inner_table[i][0]
- else:
- col = inner_table[i][j]
- key = col[0] + '\t' + str(j)
- dup_dict[key] = [j]
- # print('dup_dict', dup_dict)
- #
- for key in dup_dict.keys():
- index_list = dup_dict.get(key)
- if len(index_list) <= 1:
- continue
- # 需要表头不同
- table_head_list = []
- for index in index_list:
- table_head_list.append(inner_table[i][index][1])
- table_head_list = list(set(table_head_list))
- if len(table_head_list) <= 1:
- continue
- # 若是职业特殊处理
- col = key.split('\t')[0]
- if re.search('([^人]员|工程师|建造师|经理|安全负责人|技术负责人|合同商务负责人)$', col):
- table_head_flag = 0
- # 看是否包含表头
- else:
- table_head_flag = 0
- for index in index_list:
- if inner_table[i][index][1] == 1:
- table_head_flag = 1
- break
- table_head = None
- if index_list[0] > 0 and index_list[-1] == len(inner_table[i]) - 1:
- table_head = inner_table[i][index_list[0]][1]
- elif index_list[0] > 0 and index_list[-1] != len(inner_table[i]) - 1:
- # 查看前面是否有表头-非表头表达
- is_head_not_head = 0
- for k in range(index_list[0]):
- if k+1 < index_list[0] and inner_table[i][k][1] == 1 and inner_table[i][k+1][1] == 0:
- is_head_not_head = 1
- break
- # 查看前后有没有表头
- start_has_head = 0
- end_has_head = 0
- if not is_head_not_head:
- for k in range(index_list[0], len(inner_table[i])):
- if inner_table[i][k][0] == inner_table[i][index_list[0]][0]:
- continue
- if inner_table[i][k][1] == 1:
- end_has_head = 1
- break
- for k in range(index_list[0]):
- if inner_table[i][k][0] == inner_table[i][index_list[0]][0]:
- continue
- if inner_table[i][k][1] == 1:
- start_has_head = 1
- break
- head_list = col_head_dict.get(inner_table[i][index_list[0]][0])
- if is_head_not_head:
- table_head = table_head_flag
- elif len(head_list) >= 4:
- if head_list.count(0) > head_list.count(1):
- table_head = 0
- else:
- table_head = 1
- elif not start_has_head and not end_has_head:
- table_head = 0
- else:
- table_head = table_head_flag
- elif index_list[0] == 0 and index_list[-1] == len(inner_table[i]) - 1:
- table_head = table_head_flag
- elif index_list[0] == 0 and index_list[-1] != len(inner_table[i]) - 1:
- head_list = col_head_dict.get(inner_table[i][index_list[0]][0])
- if len(head_list) >= 4:
- if head_list.count(0) > head_list.count(1):
- table_head = 0
- else:
- table_head = 1
- else:
- table_head = table_head_flag
- if table_head is not None:
- for index in index_list:
- inner_table[i][index][1] = table_head
- return inner_table
- def repair_by_around(inner_table):
- """
- 根据周围的表头值修复当前格子的表头值
- """
- one_head_index_list = []
- zero_head_index_list = []
- all_head_index_list = []
- one_not_head_index_list = []
- no_dup_index_cnt_dict = {}
- for i in range(len(inner_table)):
- head_cnt = 0
- head_index = None
- head_dict = {}
- for j in range(len(inner_table[i])):
- # 统计表头数
- if inner_table[i][j][1] == 1:
- head_cnt += 1
- head_index = j
- if inner_table[i][j][0] not in ['~~', '', ' ']:
- if inner_table[i][j][0] in head_dict.keys():
- head_dict[inner_table[i][j][0]] += 1
- else:
- head_dict[inner_table[i][j][0]] = 1
- no_dup_index_cnt_dict[i] = len(head_dict.keys())
- # 表头数list
- if head_cnt == 0:
- zero_head_index_list.append(i)
- elif head_cnt == 1:
- # 这个单个表头需满足前面有非表头
- find_flag = 0
- for k in range(head_index):
- if inner_table[i][k][1] == 0:
- find_flag = 1
- if find_flag and len(head_dict.keys()) > 2:
- one_head_index_list.append(i)
- elif head_cnt == len(inner_table[i]):
- all_head_index_list.append(i)
- elif head_cnt == len(inner_table[i]) - 1:
- one_not_head_index_list.append(i)
- # 第一行为表头,但有一个不为表头,下面行都非表头,表格行数小于4
- if 0 in one_not_head_index_list and 1 in zero_head_index_list and len(inner_table) <= 4:
- # 不相等的列值大于4
- diff_col1 = []
- for col in inner_table[0]:
- if col[0] not in diff_col1 and len(col[0]) >= 1:
- diff_col1.append(col[0])
- diff_col2 = []
- for col in inner_table[1]:
- if col[0] not in diff_col2 and len(col[0]) >= 1:
- diff_col2.append(col[0])
- if len(diff_col1) >= 4 and len(diff_col2) >= 4:
- for j in range(len(inner_table[0])):
- inner_table[0][j][1] = 1
- one_not_head_index_list.remove(0)
- all_head_index_list.append(0)
- # 一行很多列且都为表头,则剩下一个也为表头
- for i in range(len(inner_table)):
- if no_dup_index_cnt_dict.get(i) >= 5 and i in one_not_head_index_list:
- for j in range(len(inner_table[i])):
- inner_table[i][j][1] = 1
- # 一行很多列且都不为表头,则剩下一个也不为表头,除了第一个
- for i in range(len(inner_table)):
- if no_dup_index_cnt_dict.get(i) >= 5 and i in one_head_index_list \
- and inner_table[i][0][1] != 1 and inner_table[i][0][0] != '':
- for j in range(len(inner_table[i])):
- inner_table[i][j][1] = 0
- # 一整个大表格,第一行为表头,下面行中有个别格子被识别为表头
- # 候选人后面修复
- for index in one_head_index_list:
- if (index - 1 in zero_head_index_list and index - 2 in zero_head_index_list) \
- or (index - 1 in zero_head_index_list and index - 2 in all_head_index_list) \
- or (index - 1 in all_head_index_list):
- for j in range(len(inner_table[index])):
- inner_table[index][j][1] = 0
- zero_head_index_list.append(index)
- return inner_table
- def repair_by_tenderer(inner_table):
- """
- 根据第一第二第三候选人修复当前格子的表头值
- """
- # 修复第一第二第三中标候选人作为表头
- first_tenderer = ['第一中标候选人', '第一中标人', '第一中标(成交)人', '第一候选人']
- second_tenderer = ['第二中标候选人', '第二中标(成交)候选人', '第二候选人']
- third_tenderer = ['第三中标候选人', '第三中标(成交)候选人', '第三候选人']
- # n1 next one, n2 next two, l1 last one, l2 last two
- for i in range(len(inner_table)):
- row = inner_table[i]
- n1_row, n2_row = None, None
- if i+1 < len(inner_table):
- n1_row = inner_table[i+1]
- if i+2 < len(inner_table):
- n2_row = inner_table[i+2]
- for j in range(len(row)):
- row_col = row[j]
- n1_row_col, n2_row_col = None, None
- row_n1_col, row_n2_col = None, None
- n1_row_n1_col, n2_row_n1_col, n1_row_n2_col = None, None, None
- if n1_row:
- n1_row_col = n1_row[j]
- if n2_row:
- n2_row_col = n2_row[j]
- if j+1 < len(row):
- row_n1_col = row[j+1]
- if j+2 < len(row):
- row_n2_col = row[j+2]
- if n1_row and j+1 < len(n1_row):
- n1_row_n1_col = n1_row[j+1]
- if n2_row and j+1 < len(n2_row):
- n2_row_n1_col = n2_row[j+1]
- if n1_row and j+2 < len(n1_row):
- n1_row_n2_col = n1_row[j+2]
- # 连续作为行表头
- if row_col[0] in first_tenderer and row_n1_col and row_n1_col[1] == 0:
- if n1_row_col and n1_row_col[0] in second_tenderer and n1_row_n1_col and n1_row_n1_col[1] == 0:
- inner_table[i][j][1] = 1
- inner_table[i+1][j][1] = 1
- if n2_row_col and n2_row_col[0] in third_tenderer and n2_row_n1_col and n2_row_n1_col[1] == 0:
- inner_table[i+2][j][1] = 1
- # 连续作为列表头
- if row_col[0] in first_tenderer and n1_row_col and n1_row_col[1] == 0:
- if row_n1_col and row_n1_col[0] in second_tenderer and n1_row_n1_col and n1_row_n1_col[1] == 0:
- inner_table[i][j][1] = 1
- inner_table[i][j+1][1] = 1
- if row_n2_col and row_n2_col[0] in third_tenderer and n1_row_n2_col and n1_row_n2_col[1] == 0:
- inner_table[i][j+2][1] = 1
- return inner_table
- def repair_by_keywords(inner_table):
- """
- 根据关键词修复当前格子的表头值
- """
- # 修复表头关键词未作为表头
- # 末尾匹配匹配关键词且字数小于7,直接作为表头
- head_keyword = ['供应商', '总价', '总价(元)', '总价\(元\)', '品目一', '品目二', '品目三']
- # 末尾匹配关键词且前一列为表头且与前一列文本不同,直接不做表头
- head_keyword2 = ['管理中心', '有限公司', '项目采购', '确定。', ]
- # 开头匹配关键词,直接不做表头
- head_keyword3 = ['详见', '选定', '咨询服务', '标准物资', '电汇', '承兑', '低档', '高档',
- '更换配置', '各种数据']
- # 文本匹配关键词且前一列为表头,直接作为表头
- head_keyword4 = ['综合排名', '工期(交货期)', '检测批', '检测范围', '混凝土设计强检测批的容度等级',
- '量(个)']
- # 文本在关键词中,直接不做表头
- head_keyword5 = ['殡葬用地', '电脑包', '电池']
- # 文本匹配关键词,直接不作表头
- head_keyword6 = ['市场行情', '有限公司', '能提供']
- # 末尾匹配关键词,直接不做表头
- head_keyword7 = ['基金', '结转', '结余', '税', '结余分配', '协议供货', '房屋',
- '纳税人', '自然人', '计算所得额']
- # 文本匹配关键词且整行都是表头,直接做表头
- head_keyword8 = ['备注']
- # n1 next one, n2 next two, l1 last one, l2 last two
- for i in range(len(inner_table)):
- row = inner_table[i]
- for j in range(len(row)):
- row_col = row[j]
- row_l1_col = None
- if j-1 >= 0:
- row_l1_col = row[j-1]
- for key in head_keyword:
- match = re.search(key+'$', row_col[0])
- if match and len(inner_table[i][j][0]) <= 6:
- if show:
- print('match head_keyword')
- inner_table[i][j][1] = 1
- for key in head_keyword2:
- match = re.search(key+'$', row_col[0])
- if j > 0 and row_l1_col and row_l1_col[1] == 1 and row_l1_col[0] != row_col[0] and match and row_col[1] == 1:
- if show:
- print('match head_keyword2')
- inner_table[i][j][1] = 0
- for key in head_keyword3:
- match = re.search('^'+key, row_col[0])
- if match and row_col[1] == 1:
- if show:
- print('match head_keyword3')
- inner_table[i][j][1] = 0
- for key in head_keyword4:
- match = re.search(key, row_col[0])
- if j > 0 and row_l1_col and row_l1_col[1] == 1 and match and row_col[1] == 0:
- if show:
- print('match head_keyword4')
- inner_table[i][j][1] = 1
- if row_col[0] in head_keyword5:
- if show:
- print('match head_keyword5')
- inner_table[i][j][1] = 0
- for key in head_keyword6:
- match = re.search(key, row_col[0])
- if match:
- if show:
- print('match head_keyword6')
- inner_table[i][j][1] = 0
- for key in head_keyword7:
- match = re.search(key+'$', row_col[0])
- if match and row_col[1] == 1:
- if show:
- print('match head_keyword7')
- inner_table[i][j][1] = 0
- if row_col[0] in head_keyword8 and row_col[1] == 0:
- if show:
- print('match head_keyword8')
- all_head_flag = 1
- for k in range(len(row)):
- if row[k][0] in ['', row_col[0]]:
- continue
- if row[k][1] == 0:
- print('row[k]', row[k])
- all_head_flag = 0
- break
- # print('all_head_flag', all_head_flag)
- if all_head_flag:
- inner_table[i][j][1] = 1
- return inner_table
- def repair_by_length(inner_table):
- for i in range(len(inner_table)):
- for j in range(len(inner_table[i])):
- if len(inner_table[i][j][0]) >= 30:
- inner_table[i][j][1] = 0
- return inner_table
- def repair_by_summation(inner_table):
- # 修复合计在中间的特殊情况
- if len(inner_table) >= 3 and len(inner_table[1]) == 2 \
- and inner_table[1][0][0] == '合计' and inner_table[1][1][0].endswith('%'):
- inner_table[1][0][1] = 0
- inner_table[1][1][1] = 0
- return inner_table
- def repair_by_rank(inner_table):
- if not inner_table or (inner_table and len(inner_table[0]) < 3):
- return inner_table
- for i in range(len(inner_table)):
- for j in range(len(inner_table[i])-2):
- if inner_table[i][j][0] in ['第一名'] and inner_table[i][j+1][0] in ['第二名'] and inner_table[i][j+2][0] in ['第三名']:
- inner_table[i][j][1] = 1
- inner_table[i][j+1][1] = 1
- inner_table[i][j+2][1] = 1
- return inner_table
- _inner_table = pre_process(_inner_table)
- compare_inner_table = copy.deepcopy(_inner_table)
- if show:
- print('table_head_repair_process1', show_row_index, _inner_table[show_row_index])
- _inner_table = repair_by_rank(_inner_table)
- if _inner_table != compare_inner_table:
- compare_inner_table = copy.deepcopy(_inner_table)
- log('table_head repair1.5 ' + str(docid))
- if show:
- print('table_head_repair_process1.5', show_row_index, _inner_table[show_row_index])
- _inner_table = repair_by_colon(_inner_table)
- if _inner_table != compare_inner_table:
- compare_inner_table = copy.deepcopy(_inner_table)
- log('table_head repair2 ' + str(docid))
- if show:
- print('table_head_repair_process2', show_row_index, _inner_table[show_row_index])
- _inner_table = repair_by_keywords(_inner_table)
- if _inner_table != compare_inner_table:
- compare_inner_table = copy.deepcopy(_inner_table)
- log('table_head repair3 ' + str(docid))
- if show:
- print('table_head_repair_process3', show_row_index, _inner_table[show_row_index])
- _inner_table = repair_by_tenderer(_inner_table)
- if _inner_table != compare_inner_table:
- compare_inner_table = copy.deepcopy(_inner_table)
- log('table_head repair4 ' + str(docid))
- if show:
- print('table_head_repair_process4', show_row_index, _inner_table[show_row_index])
- _inner_table = repair_by_duplicate(_inner_table)
- if _inner_table != compare_inner_table:
- compare_inner_table = copy.deepcopy(_inner_table)
- log('table_head repair5 ' + str(docid))
- if show:
- print('table_head_repair_process5', show_row_index, _inner_table[show_row_index])
- _inner_table = repair_by_around(_inner_table)
- if _inner_table != compare_inner_table:
- compare_inner_table = copy.deepcopy(_inner_table)
- log('table_head repair6 ' + str(docid))
- if show:
- print('table_head_repair_process6', show_row_index, _inner_table[show_row_index])
- _inner_table = repair_by_tenderer(_inner_table)
- if _inner_table != compare_inner_table:
- compare_inner_table = copy.deepcopy(_inner_table)
- log('table_head repair7 ' + str(docid))
- if show:
- print('table_head_repair_process7', show_row_index, _inner_table[show_row_index])
- _inner_table = repair_by_keywords(_inner_table)
- if _inner_table != compare_inner_table:
- compare_inner_table = copy.deepcopy(_inner_table)
- log('table_head repair8 ' + str(docid))
- if show:
- print('table_head_repair_process8', show_row_index, _inner_table[show_row_index])
- _inner_table = repair_by_length(_inner_table)
- if _inner_table != compare_inner_table:
- compare_inner_table = copy.deepcopy(_inner_table)
- print('table_head repair9 ' + str(docid))
- if show:
- print('table_head_repair_process9', show_row_index, _inner_table[show_row_index])
- _inner_table = repair_by_summation(_inner_table)
- if show:
- print('table_head_repair_process10', show_row_index, _inner_table[show_row_index])
- return _inner_table
|