table_parser.py 122 KB

1234567891011121314151617181920212223242526272829303132333435363738394041424344454647484950515253545556575859606162636465666768697071727374757677787980818283848586878889909192939495969798991001011021031041051061071081091101111121131141151161171181191201211221231241251261271281291301311321331341351361371381391401411421431441451461471481491501511521531541551561571581591601611621631641651661671681691701711721731741751761771781791801811821831841851861871881891901911921931941951961971981992002012022032042052062072082092102112122132142152162172182192202212222232242252262272282292302312322332342352362372382392402412422432442452462472482492502512522532542552562572582592602612622632642652662672682692702712722732742752762772782792802812822832842852862872882892902912922932942952962972982993003013023033043053063073083093103113123133143153163173183193203213223233243253263273283293303313323333343353363373383393403413423433443453463473483493503513523533543553563573583593603613623633643653663673683693703713723733743753763773783793803813823833843853863873883893903913923933943953963973983994004014024034044054064074084094104114124134144154164174184194204214224234244254264274284294304314324334344354364374384394404414424434444454464474484494504514524534544554564574584594604614624634644654664674684694704714724734744754764774784794804814824834844854864874884894904914924934944954964974984995005015025035045055065075085095105115125135145155165175185195205215225235245255265275285295305315325335345355365375385395405415425435445455465475485495505515525535545555565575585595605615625635645655665675685695705715725735745755765775785795805815825835845855865875885895905915925935945955965975985996006016026036046056066076086096106116126136146156166176186196206216226236246256266276286296306316326336346356366376386396406416426436446456466476486496506516526536546556566576586596606616626636646656666676686696706716726736746756766776786796806816826836846856866876886896906916926936946956966976986997007017027037047057067077087097107117127137147157167177187197207217227237247257267277287297307317327337347357367377387397407417427437447457467477487497507517527537547557567577587597607617627637647657667677687697707717727737747757767777787797807817827837847857867877887897907917927937947957967977987998008018028038048058068078088098108118128138148158168178188198208218228238248258268278288298308318328338348358368378388398408418428438448458468478488498508518528538548558568578588598608618628638648658668678688698708718728738748758768778788798808818828838848858868878888898908918928938948958968978988999009019029039049059069079089099109119129139149159169179189199209219229239249259269279289299309319329339349359369379389399409419429439449459469479489499509519529539549559569579589599609619629639649659669679689699709719729739749759769779789799809819829839849859869879889899909919929939949959969979989991000100110021003100410051006100710081009101010111012101310141015101610171018101910201021102210231024102510261027102810291030103110321033103410351036103710381039104010411042104310441045104610471048104910501051105210531054105510561057105810591060106110621063106410651066106710681069107010711072107310741075107610771078107910801081108210831084108510861087108810891090109110921093109410951096109710981099110011011102110311041105110611071108110911101111111211131114111511161117111811191120112111221123112411251126112711281129113011311132113311341135113611371138113911401141114211431144114511461147114811491150115111521153115411551156115711581159116011611162116311641165116611671168116911701171117211731174117511761177117811791180118111821183118411851186118711881189119011911192119311941195119611971198119912001201120212031204120512061207120812091210121112121213121412151216121712181219122012211222122312241225122612271228122912301231123212331234123512361237123812391240124112421243124412451246124712481249125012511252125312541255125612571258125912601261126212631264126512661267126812691270127112721273127412751276127712781279128012811282128312841285128612871288128912901291129212931294129512961297129812991300130113021303130413051306130713081309131013111312131313141315131613171318131913201321132213231324132513261327132813291330133113321333133413351336133713381339134013411342134313441345134613471348134913501351135213531354135513561357135813591360136113621363136413651366136713681369137013711372137313741375137613771378137913801381138213831384138513861387138813891390139113921393139413951396139713981399140014011402140314041405140614071408140914101411141214131414141514161417141814191420142114221423142414251426142714281429143014311432143314341435143614371438143914401441144214431444144514461447144814491450145114521453145414551456145714581459146014611462146314641465146614671468146914701471147214731474147514761477147814791480148114821483148414851486148714881489149014911492149314941495149614971498149915001501150215031504150515061507150815091510151115121513151415151516151715181519152015211522152315241525152615271528152915301531153215331534153515361537153815391540154115421543154415451546154715481549155015511552155315541555155615571558155915601561156215631564156515661567156815691570157115721573157415751576157715781579158015811582158315841585158615871588158915901591159215931594159515961597159815991600160116021603160416051606160716081609161016111612161316141615161616171618161916201621162216231624162516261627162816291630163116321633163416351636163716381639164016411642164316441645164616471648164916501651165216531654165516561657165816591660166116621663166416651666166716681669167016711672167316741675167616771678167916801681168216831684168516861687168816891690169116921693169416951696169716981699170017011702170317041705170617071708170917101711171217131714171517161717171817191720172117221723172417251726172717281729173017311732173317341735173617371738173917401741174217431744174517461747174817491750175117521753175417551756175717581759176017611762176317641765176617671768176917701771177217731774177517761777177817791780178117821783178417851786178717881789179017911792179317941795179617971798179918001801180218031804180518061807180818091810181118121813181418151816181718181819182018211822182318241825182618271828182918301831183218331834183518361837183818391840184118421843184418451846184718481849185018511852185318541855185618571858185918601861186218631864186518661867186818691870187118721873187418751876187718781879188018811882188318841885188618871888188918901891189218931894189518961897189818991900190119021903190419051906190719081909191019111912191319141915191619171918191919201921192219231924192519261927192819291930193119321933193419351936193719381939194019411942194319441945194619471948194919501951195219531954195519561957195819591960196119621963196419651966196719681969197019711972197319741975197619771978197919801981198219831984198519861987198819891990199119921993199419951996199719981999200020012002200320042005200620072008200920102011201220132014201520162017201820192020202120222023202420252026202720282029203020312032203320342035203620372038203920402041204220432044204520462047204820492050205120522053205420552056205720582059206020612062206320642065206620672068206920702071207220732074207520762077207820792080208120822083208420852086208720882089209020912092209320942095209620972098209921002101210221032104210521062107210821092110211121122113211421152116211721182119212021212122212321242125212621272128212921302131213221332134213521362137213821392140214121422143214421452146214721482149215021512152215321542155215621572158215921602161216221632164216521662167216821692170217121722173217421752176217721782179218021812182218321842185218621872188218921902191219221932194219521962197219821992200220122022203220422052206220722082209221022112212221322142215221622172218221922202221222222232224222522262227222822292230223122322233223422352236223722382239224022412242224322442245224622472248224922502251225222532254225522562257225822592260226122622263226422652266226722682269227022712272227322742275227622772278227922802281228222832284228522862287228822892290229122922293229422952296229722982299230023012302230323042305230623072308230923102311231223132314231523162317231823192320232123222323232423252326232723282329233023312332233323342335233623372338233923402341234223432344234523462347234823492350235123522353235423552356235723582359236023612362236323642365236623672368236923702371237223732374237523762377237823792380238123822383238423852386238723882389239023912392239323942395239623972398239924002401
  1. # -*- coding: utf-8 -*-
  2. """表格解析与二维化。
  3. 按 ARCHITECTURE.md Phase 4 拆分建议,从 ``interface/Preprocessing.py`` 迁出。
  4. 类型:PREPROCESS。
  5. 原位置:``interface/Preprocessing.py`` 中以下函数:
  6. - ``tableToText`` — HTML 表格转文本,含 fixSpan/getTable 等嵌套辅助函数
  7. - ``table_head_repair_process`` — 表头修复
  8. ``interface/Preprocessing.py`` 仍 re-export 以上全部名称,老 import 不受影响。
  9. """
  10. from __future__ import absolute_import
  11. import copy
  12. import json
  13. import re
  14. import time
  15. import numpy as np
  16. from BiddingKG.dl.common.logging import log
  17. from BiddingKG.dl.interface.predictor import getPredictor
  18. from BiddingKG.dl.predictors.table_prem import TableTag2List
  19. from BiddingKG.dl.model_runtime.embed import formEncoding
  20. from BiddingKG.dl.table_head.predict_torch import predict
  21. from BiddingKG.dl.preprocess.segmenter import segment
  22. __all__ = [
  23. "tableToText",
  24. "table_head_repair_process",
  25. ]
  26. def tableToText(soup, docid=None, return_kv=False):
  27. '''
  28. @param:
  29. soup:网页html的soup
  30. @return:处理完表格信息的网页text
  31. '''
  32. def getTrs(tbody):
  33. #获取所有的tr
  34. trs = []
  35. objs = tbody.find_all(recursive=False)
  36. for obj in objs:
  37. if obj.name=="tr":
  38. trs.append(obj)
  39. if obj.name=="tbody":
  40. for tr in obj.find_all("tr",recursive=False):
  41. trs.append(tr)
  42. return trs
  43. def fixSpan(tbody):
  44. # 处理colspan, rowspan信息补全问题
  45. #trs = tbody.findChildren('tr', recursive=False)
  46. trs = getTrs(tbody)
  47. ths_len = 0
  48. ths = list()
  49. trs_set = set()
  50. #修改为先进行列补全再进行行补全,否则可能会出现表格解析混乱
  51. # 遍历每一个tr
  52. for indtr, tr in enumerate(trs):
  53. ths_tmp = tr.findChildren('th', recursive=False)
  54. #不补全含有表格的tr
  55. if len(tr.findChildren('table'))>0:
  56. continue
  57. if len(ths_tmp) > 0:
  58. ths_len = ths_len + len(ths_tmp)
  59. for th in ths_tmp:
  60. ths.append(th)
  61. trs_set.add(tr)
  62. # 遍历每行中的element
  63. tds = tr.findChildren(recursive=False)
  64. for indtd, td in enumerate(tds):
  65. # 若有colspan 则补全同一行下一个位置
  66. if 'colspan' in td.attrs:
  67. if str(re.sub("[^0-9]","",str(td['colspan'])))!="":
  68. col = int(re.sub("[^0-9]","",str(td['colspan'])))
  69. if col<100 and len(td.get_text())<1000:
  70. td['colspan'] = 1
  71. for i in range(1, col, 1):
  72. td.insert_after(copy.copy(td))
  73. for indtr, tr in enumerate(trs):
  74. ths_tmp = tr.findChildren('th', recursive=False)
  75. #不补全含有表格的tr
  76. if len(tr.findChildren('table'))>0:
  77. continue
  78. if len(ths_tmp) > 0:
  79. ths_len = ths_len + len(ths_tmp)
  80. for th in ths_tmp:
  81. ths.append(th)
  82. trs_set.add(tr)
  83. # 遍历每行中的element
  84. tds = tr.findChildren(recursive=False)
  85. for indtd, td in enumerate(tds):
  86. # 若有rowspan 则补全下一行同样位置
  87. if 'rowspan' in td.attrs:
  88. if str(re.sub("[^0-9]","",str(td['rowspan'])))!="":
  89. row = int(re.sub("[^0-9]","",str(td['rowspan'])))
  90. td['rowspan'] = 1
  91. for i in range(1, row, 1):
  92. # 获取下一行的所有td, 在对应的位置插入
  93. if indtr+i<len(trs):
  94. tds1 = trs[indtr + i].findChildren(['td','th'], recursive=False)
  95. if len(tds1) >= (indtd) and len(tds1)>0:
  96. if indtd > 0:
  97. tds1[indtd - 1].insert_after(copy.copy(td))
  98. else:
  99. tds1[0].insert_before(copy.copy(td))
  100. elif indtd-2>0 and len(tds1) > 0 and len(tds1) == indtd - 1: # 修正某些表格最后一列没补全
  101. tds1[indtd-2].insert_after(copy.copy(td))
  102. def getTable(tbody):
  103. #trs = tbody.findChildren('tr', recursive=False)
  104. trs = getTrs(tbody)
  105. inner_table = []
  106. for tr in trs:
  107. tr_line = []
  108. tds = tr.findChildren(['td','th'], recursive=False)
  109. if len(tds)==0:
  110. if return_kv:
  111. tr_line.append([re.sub('\xa0','',tr.get_text()),0])
  112. else:
  113. tr_line.append([re.sub('\xa0','',segment(tr,final=False)),0]) # 2021/12/21 修复部分表格没有td 造成数据丢失
  114. for td in tds:
  115. if return_kv:
  116. tr_line.append([re.sub('\xa0','',td.get_text()),0])
  117. else:
  118. tr_line.append([re.sub('\xa0','',segment(td,final=False)),0])
  119. #tr_line.append([td.get_text(),0])
  120. inner_table.append(tr_line)
  121. return inner_table
  122. #处理表格不对齐的问题
  123. def fixTable(inner_table,fix_value="~~"):
  124. maxWidth = 0
  125. for item in inner_table:
  126. if len(item)>maxWidth:
  127. maxWidth = len(item)
  128. if maxWidth > 100:
  129. # log('表格列数大于100,表格异常不做处理。')
  130. return []
  131. for i in range(len(inner_table)):
  132. if len(inner_table[i])<maxWidth:
  133. for j in range(maxWidth-len(inner_table[i])):
  134. inner_table[i].append([fix_value,0])
  135. return inner_table
  136. def removePadding(inner_table,pad_row = "@@",pad_col = "##"):
  137. height = len(inner_table)
  138. width = len(inner_table[0])
  139. for i in range(height):
  140. point = ""
  141. for j in range(width):
  142. if inner_table[i][j][0]==point and point!="":
  143. inner_table[i][j][0] = pad_row
  144. else:
  145. if inner_table[i][j][0] not in [pad_row,pad_col]:
  146. point = inner_table[i][j][0]
  147. for j in range(width):
  148. point = ""
  149. for i in range(height):
  150. if inner_table[i][j][0]==point and point!="":
  151. inner_table[i][j][0] = pad_col
  152. else:
  153. if inner_table[i][j][0] not in [pad_row,pad_col]:
  154. point = inner_table[i][j][0]
  155. def addPadding(inner_table,pad_row = "@@",pad_col = "##"):
  156. height = len(inner_table)
  157. width = len(inner_table[0])
  158. for i in range(height):
  159. for j in range(width):
  160. if inner_table[i][j][0]==pad_row:
  161. inner_table[i][j][0] = inner_table[i][j-1][0]
  162. inner_table[i][j][1] = inner_table[i][j-1][1]
  163. if inner_table[i][j][0]==pad_col:
  164. inner_table[i][j][0] = inner_table[i-1][j][0]
  165. inner_table[i][j][1] = inner_table[i-1][j][1]
  166. def repairTable(inner_table, dye_set=set(), key_set=set(), fix_value="~~"):
  167. """
  168. @summary: 修复表头识别,将明显错误的进行修正
  169. """
  170. def repairNeeded(line):
  171. first_1 = -1
  172. last_1 = -1
  173. first_0 = -1
  174. last_0 = -1
  175. count_1 = 0
  176. count_0 = 0
  177. for i in range(len(line)):
  178. if line[i][0] == fix_value:
  179. continue
  180. if line[i][1]==1:
  181. if first_1==-1:
  182. first_1 = i
  183. last_1 = i
  184. count_1 += 1
  185. if line[i][1]==0:
  186. if first_0 == -1:
  187. first_0 = i
  188. last_0 = i
  189. count_0 += 1
  190. if first_1 ==-1 or last_0 == -1:
  191. return False
  192. # 异常情况:第一个不是表头;最后一个是表头;表头个数远大于属性值个数
  193. if first_1-0 > 0 or last_0-len(line)+1 < 0 or last_1 == len(line)-1 or count_1-count_0 >= 3:
  194. return True
  195. return False
  196. def getsimilarity(line, line1):
  197. same_count = 0
  198. for item, item1 in zip(line,line1):
  199. if item[1] == item1[1]:
  200. same_count += 1
  201. return same_count/len(line)
  202. def selfrepair(inner_table,index,dye_set,key_set):
  203. """
  204. @summary: 计算每个节点受到的挤压度来判断是否需要染色
  205. """
  206. # print("B",inner_table[index])
  207. min_presure = 3
  208. list_dye = []
  209. first = None
  210. count = 0
  211. # temp_set = set()
  212. temp_set = set(['~~']) # 2023/10/10纠正236239652 受让单位识别不到表头; 受让单位,明细用途:用途名称:陵川县民政局,
  213. _index = 0
  214. for item in inner_table[index]:
  215. if first is None:
  216. first = item[1]
  217. if item[0] not in temp_set:
  218. count += 1
  219. temp_set.add(item[0])
  220. else:
  221. if first == item[1]:
  222. if item[0] not in temp_set:
  223. temp_set.add(item[0])
  224. count += 1
  225. else:
  226. list_dye.append([first,count,_index])
  227. first = item[1]
  228. temp_set.add(item[0])
  229. count = 1
  230. _index += 1
  231. list_dye.append([first,count,_index])
  232. if len(list_dye)>1:
  233. begin = 0
  234. end = 0
  235. for i in range(len(list_dye)):
  236. end = list_dye[i][2]
  237. dye_flag = False
  238. # 首尾要求压力减一
  239. if i==0:
  240. if list_dye[i+1][1]-list_dye[i][1]+1>=min_presure-1:
  241. dye_flag = True
  242. dye_type = list_dye[i+1][0]
  243. elif i==len(list_dye)-1:
  244. if list_dye[i-1][1]-list_dye[i][1]+1>=min_presure-1:
  245. dye_flag = True
  246. dye_type = list_dye[i-1][0]
  247. else:
  248. if list_dye[i][1]>1:
  249. if list_dye[i+1][1]-list_dye[i][1]+1>=min_presure:
  250. dye_flag = True
  251. dye_type = list_dye[i+1][0]
  252. if list_dye[i-1][1]-list_dye[i][1]+1>=min_presure:
  253. dye_flag = True
  254. dye_type = list_dye[i-1][0]
  255. else:
  256. if list_dye[i+1][1]+list_dye[i-1][1]-list_dye[i][1]+1>=min_presure:
  257. dye_flag = True
  258. dye_type = list_dye[i+1][0]
  259. if list_dye[i+1][1]+list_dye[i-1][1]-list_dye[i][1]+1>=min_presure:
  260. dye_flag = True
  261. dye_type = list_dye[i-1][0]
  262. if dye_flag:
  263. for h in range(begin,end):
  264. inner_table[index][h][1] = dye_type
  265. dye_set.add((inner_table[index][h][0],dye_type))
  266. key_set.add(inner_table[index][h][0])
  267. begin = end
  268. # print("E",inner_table[index])
  269. def otherrepair(inner_table,index,dye_set,key_set):
  270. list_provide_repair = []
  271. if index==0 and len(inner_table)>1:
  272. list_provide_repair.append(index+1)
  273. elif index==len(inner_table)-1:
  274. list_provide_repair.append(index-1)
  275. else:
  276. list_provide_repair.append(index+1)
  277. list_provide_repair.append(index-1)
  278. for provide_index in list_provide_repair:
  279. if not repairNeeded(inner_table[provide_index]):
  280. same_prob = getsimilarity(inner_table[index], inner_table[provide_index])
  281. if same_prob>=0.8:
  282. for i in range(len(inner_table[provide_index])):
  283. if inner_table[index][i][1]!=inner_table[provide_index][i][1]:
  284. dye_set.add((inner_table[index][i][0],inner_table[provide_index][i][1]))
  285. key_set.add(inner_table[index][i][0])
  286. inner_table[index][i][1] = inner_table[provide_index][i][1]
  287. elif same_prob<=0.2:
  288. for i in range(len(inner_table[provide_index])):
  289. if inner_table[index][i][1]==inner_table[provide_index][i][1]:
  290. dye_set.add((inner_table[index][i][0],inner_table[provide_index][i][1]))
  291. key_set.add(inner_table[index][i][0])
  292. inner_table[index][i][1] = 0 if inner_table[provide_index][i][1] ==1 else 1
  293. len_dye_set = len(dye_set)
  294. height = len(inner_table)
  295. for i in range(height):
  296. if repairNeeded(inner_table[i]):
  297. selfrepair(inner_table, i, dye_set, key_set)
  298. #otherrepair(inner_table,i,dye_set,key_set)
  299. for h in range(len(inner_table)):
  300. for w in range(len(inner_table[0])):
  301. if inner_table[h][w][0] in key_set:
  302. for item in dye_set:
  303. if inner_table[h][w][0] == item[0]:
  304. inner_table[h][w][1] = item[1]
  305. # 如果两个set长度不相同,则有同一个key被反复染色,将导致无限迭代
  306. if len(dye_set) != len(key_set):
  307. for i in range(height):
  308. if repairNeeded(inner_table[i]):
  309. selfrepair(inner_table,i,dye_set,key_set)
  310. #otherrepair(inner_table,i,dye_set,key_set)
  311. return
  312. if len(dye_set) == len_dye_set:
  313. '''
  314. for i in range(height):
  315. if repairNeeded(inner_table[i]):
  316. otherrepair(inner_table,i,dye_set,key_set)
  317. '''
  318. return
  319. repairTable(inner_table, dye_set, key_set)
  320. def repair_table2(inner_table, show=0, row_no=0):
  321. """
  322. @summary: 修复表头识别,将明显错误的进行修正
  323. """
  324. # 循环处理单元格,一次获取需要的
  325. one_head_index_list = []
  326. zero_head_index_list = []
  327. all_head_index_list = []
  328. for i in range(len(inner_table)):
  329. head_cnt = 0
  330. for j in range(len(inner_table[i])):
  331. # 删除前后逗号
  332. inner_table[i][j][0] = re.sub('^[,,]+', '', inner_table[i][j][0])
  333. inner_table[i][j][0] = re.sub('[,,]+$', '', inner_table[i][j][0])
  334. # 统计表头数
  335. if inner_table[i][j][1] == 1:
  336. head_cnt += 1
  337. # 表头数list
  338. if head_cnt == 0:
  339. zero_head_index_list.append(i)
  340. elif head_cnt == 1:
  341. one_head_index_list.append(i)
  342. elif head_cnt == len(inner_table[i]):
  343. all_head_index_list.append(i)
  344. # 修复冒号在文本中间的,不能作为表头;(冒号后面需多个字)
  345. # 冒号在括号中的除外
  346. # 冒号在最后的,判断后一个格子是否有重复的文字
  347. for i in range(len(inner_table)):
  348. for j in range(len(inner_table[i])):
  349. _text = inner_table[i][j][0]
  350. if len(_text) >= 3 and inner_table[i][j][1] == 1:
  351. match = re.search('[::]', _text)
  352. if match:
  353. start_index, end_index = match.span()
  354. if start_index == 0:
  355. continue
  356. if end_index == len(_text):
  357. if len(inner_table[i]) == 2 and j <= len(inner_table[i]) - 2 and (_text in inner_table[i][j+1][0] or inner_table[i][j+1][0] in _text):
  358. inner_table[i][j][1] = 0
  359. inner_table[i][j+1][1] = 0
  360. else:
  361. continue
  362. if re.search('[((]', _text[:start_index]) and re.search('[))]', _text[end_index:]):
  363. continue
  364. m1 = re.search('[\u4e00-\u9fa50-9a-zA-Z]', _text[:start_index])
  365. m2 = re.search('[\u4e00-\u9fa50-9a-zA-Z]', _text[end_index:])
  366. if m1 and m2 and (len(m2.group()) >= 2 or m2.group() in ['是', '否']):
  367. inner_table[i][j][1] = 0
  368. if show:
  369. print('inner_table[i]1', inner_table[row_no])
  370. # 修复实际只有几列,但有一列由于重复占了太多行表头识别错误
  371. # for i in range(len(inner_table)):
  372. # head_flag_dict = {}
  373. # for j in range(len(inner_table[i])):
  374. # if inner_table[i][j][0] in head_flag_dict.keys():
  375. # head_flag_dict[inner_table[i][j][0]] += [inner_table[i][j][1]]
  376. # else:
  377. # head_flag_dict[inner_table[i][j][0]] = [inner_table[i][j][1]]
  378. #
  379. # if len(head_flag_dict.keys()) == 2:
  380. # col_flag = None
  381. # col_value = None
  382. # for key in head_flag_dict.keys():
  383. # flag_list = head_flag_dict[key]
  384. # if len(flag_list) >= 4 and len(set(flag_list)) == 2 and len(set(flag_list[1:])) == 1:
  385. # col_flag = flag_list[0]
  386. # col_value = key
  387. # break
  388. #
  389. # if col_flag is not None:
  390. # for j in range(len(inner_table[i])):
  391. # if inner_table[i][j][0] == col_value:
  392. # inner_table[i][j][1] = col_flag
  393. # 多个重复列的预测值不同,以第一个为准
  394. for i in range(len(inner_table)):
  395. col = inner_table[i][0]
  396. for j in range(len(inner_table[i])):
  397. if inner_table[i][j][0] == col[0]:
  398. if inner_table[i][j][1] != col[1]:
  399. inner_table[i][j][1] = col[1]
  400. else:
  401. col = inner_table[i][j]
  402. if show:
  403. print('inner_table[i]2', inner_table[row_no])
  404. # 修复多个重复的单元格表头不一致
  405. # for i in range(len(inner_table)):
  406. # for j in range(len(inner_table[i])-1):
  407. # only_chinese1 = ''.join(re.findall('[\u4e00-\u9fa5]+', inner_table[i][j][0]))
  408. # only_chinese2 = ''.join(re.findall('[\u4e00-\u9fa5]+', inner_table[i][j+1][0]))
  409. # if only_chinese1 == only_chinese2 and inner_table[i][j][1] != inner_table[i][j+1][1]:
  410. # inner_table[i][j][1] = 1
  411. # inner_table[i][j+1][1] = 1
  412. # if show:
  413. # print('inner_table[i]3', inner_table[row_no])
  414. # # 修复一行几乎都是表头,个别不是;或者一行几乎都是非表头,个别是
  415. # for i in range(len(inner_table)):
  416. # head_dict = {}
  417. # not_head_dict = {}
  418. # for j in range(len(inner_table[i])):
  419. # if inner_table[i][j][1] == 1:
  420. # if inner_table[i][j][0] not in head_dict:
  421. # head_dict[inner_table[i][j][0]] = 1
  422. # else:
  423. # if inner_table[i][j][0] not in not_head_dict:
  424. # not_head_dict[inner_table[i][j][0]] = 1
  425. #
  426. # # 非表头:表头 <= 1:3
  427. # # if len(head_dict.keys()) > 0 and len(not_head_dict.keys()) / len(head_dict.keys()) <= 1/3 and len(head_dict.keys()) >= 3:
  428. # # for j in range(len(inner_table[i])):
  429. # # if len(re.sub(' ', '', inner_table[i][j][0])) > 0:
  430. # # inner_table[i][j][1] = 1
  431. #
  432. # # 表头数一个且非表头数大于2且上一行都是表头
  433. # if i > 0 and len(head_dict.keys()) == 1 and len(not_head_dict.keys()) >= 2 and inner_table[i][0][1] == 0:
  434. # last_row = inner_table[i-1]
  435. # col_list = []
  436. # for j in range(len(last_row)):
  437. # if len(re.sub(' ', '', last_row[j][0])) > 0:
  438. # if last_row[j][1] == 0:
  439. # col_list = []
  440. # break
  441. # col_list.append(last_row[j][0])
  442. # if col_list:
  443. # col_list = list(set(col_list))
  444. # if len(col_list) > 2:
  445. # for j in range(len(inner_table[i])):
  446. # if inner_table[i][j][1] == 1:
  447. # inner_table[i][j][1] = 0
  448. # 一整个大表格,第一行为表头,下面行中有个别格子被识别为表头
  449. # 候选人后面修复
  450. for index in one_head_index_list:
  451. if (index - 1 in zero_head_index_list and index - 2 in zero_head_index_list) \
  452. or (index - 1 in zero_head_index_list and index - 2 in all_head_index_list) \
  453. or (index - 1 in all_head_index_list):
  454. for j in range(len(inner_table[index])):
  455. inner_table[index][j][1] = 0
  456. zero_head_index_list.append(index)
  457. if show:
  458. print('inner_table[i]4', inner_table[row_no])
  459. # 修复第一第二第三中标候选人作为表头
  460. first_tenderer = ['第一中标候选人', '第一中标人', '第一中标(成交)人', '第一候选人']
  461. second_tenderer = ['第二中标候选人', '第二中标(成交)候选人', '第二候选人']
  462. third_tenderer = ['第三中标候选人', '第三中标(成交)候选人', '第三候选人']
  463. # n1 next one, n2 next two, l1 last one, l2 last two
  464. for i in range(len(inner_table)):
  465. row = inner_table[i]
  466. n1_row, n2_row = None, None
  467. if i+1 < len(inner_table):
  468. n1_row = inner_table[i+1]
  469. if i+2 < len(inner_table):
  470. n2_row = inner_table[i+2]
  471. for j in range(len(row)):
  472. row_col = row[j]
  473. n1_row_col, n2_row_col = None, None
  474. row_n1_col, row_n2_col = None, None
  475. n1_row_n1_col, n2_row_n1_col, n1_row_n2_col = None, None, None
  476. if n1_row:
  477. n1_row_col = n1_row[j]
  478. if n2_row:
  479. n2_row_col = n2_row[j]
  480. if j+1 < len(row):
  481. row_n1_col = row[j+1]
  482. if j+2 < len(row):
  483. row_n2_col = row[j+2]
  484. if n1_row and j+1 < len(n1_row):
  485. n1_row_n1_col = n1_row[j+1]
  486. if n2_row and j+1 < len(n2_row):
  487. n2_row_n1_col = n2_row[j+1]
  488. if n1_row and j+2 < len(n1_row):
  489. n1_row_n2_col = n1_row[j+2]
  490. # 连续作为行表头
  491. if row_col[0] in first_tenderer and row_n1_col and row_n1_col[1] == 0:
  492. if n1_row_col and n1_row_col[0] in second_tenderer and n1_row_n1_col and n1_row_n1_col[1] == 0:
  493. inner_table[i][j][1] = 1
  494. inner_table[i+1][j][1] = 1
  495. if n2_row_col and n2_row_col[0] in third_tenderer and n2_row_n1_col and n2_row_n1_col[1] == 0:
  496. inner_table[i+2][j][1] = 1
  497. # 连续作为列表头
  498. if row_col[0] in first_tenderer and n1_row_col and n1_row_col[1] == 0:
  499. if row_n1_col and row_n1_col[0] in second_tenderer and n1_row_n1_col and n1_row_n1_col[1] == 0:
  500. inner_table[i][j][1] = 1
  501. inner_table[i][j+1][1] = 1
  502. if row_n2_col and row_n2_col[0] in third_tenderer and n1_row_n2_col and n1_row_n2_col[1] == 0:
  503. inner_table[i][j+2][1] = 1
  504. if show:
  505. print('inner_table[i]5', inner_table[row_no])
  506. # 修复表头关键词未作为表头
  507. # 文本匹配关键词,直接作为表头
  508. head_keyword = ['供应商', '总价']
  509. # 末尾匹配关键词且前一列为表头且与前一列文本不同,直接不做表头
  510. head_keyword2 = ['管理中心', '有限公司', '项目采购', ]
  511. # 开头匹配关键词,直接不做表头
  512. head_keyword3 = ['详见', '选定', '咨询服务', '标准物资', '电汇', '承兑']
  513. # 文本匹配关键词且前一列为表头,直接作为表头
  514. head_keyword4 = ['综合排名']
  515. # 文本在关键词中,直接不做表头
  516. head_keyword5 = ['殡葬用地']
  517. # n1 next one, n2 next two, l1 last one, l2 last two
  518. for i in range(len(inner_table)):
  519. row = inner_table[i]
  520. for j in range(len(row)):
  521. row_col = row[j]
  522. row_l1_col = None
  523. if j-1 > 0:
  524. row_l1_col = row[j-1]
  525. match = re.search('[\u4e00-\u9fa50-9a-zA-Z::]+', row_col[0])
  526. if inner_table[i][j][1] == 0 and match and match.group() in head_keyword:
  527. inner_table[i][j][1] = 1
  528. for key in head_keyword2:
  529. match = re.search(key+'$', row_col[0])
  530. if j > 0 and row_l1_col and row_l1_col[1] == 1 and row_l1_col[0] != row_col[0] and match and row_col[1] == 1:
  531. inner_table[i][j][1] = 0
  532. for key in head_keyword3:
  533. match = re.search('^'+key, row_col[0])
  534. if match and row_col[1] == 1:
  535. inner_table[i][j][1] = 0
  536. for key in head_keyword4:
  537. match = re.search(key, row_col[0])
  538. if j > 0 and row_l1_col and row_l1_col[1] == 1 and match and row_col[1] == 0:
  539. inner_table[i][j][1] = 1
  540. if row_col[0] in head_keyword5:
  541. inner_table[i][j][1] = 0
  542. if show:
  543. print('inner_table[i]6', inner_table[row_no])
  544. # 修复姓名被作为表头 # 2023-02-10 取消修复,避免项目名称、编号,单位、单价等作为了非表头
  545. # surname = [
  546. # "赵", "钱", "孙", "李", "周", "吴", "郑", "王", "冯", "陈", "褚", "卫", "蒋", "沈", "韩", "杨", "朱", "秦", "尤", "许", "何", "吕", "施", "张", "孔", "曹", "严", "华", "金", "魏", "陶", "姜", "戚", "谢", "邹", "喻", "柏", "水", "窦", "章", "云", "苏", "潘", "葛", "奚", "范", "彭", "郎", "鲁", "韦", "昌", "马", "苗", "凤", "花", "方", "俞", "任", "袁", "柳", "酆", "鲍", "史", "唐", "费", "廉", "岑", "薛", "雷", "贺", "倪", "汤", "滕", "殷", "罗", "毕", "郝", "邬", "安", "常", "乐", "于", "时", "傅", "皮", "卞", "齐", "康", "伍", "余", "元", "卜", "顾", "孟", "平", "黄", "和", "穆", "萧", "尹", "姚", "邵", "湛", "汪", "祁", "毛", "禹", "狄", "米", "贝", "明", "臧", "计", "伏", "成", "戴", "谈", "宋", "茅", "庞", "熊", "纪", "舒", "屈", "项", "祝", "董", "梁", "杜", "阮", "蓝", "闵", "席", "季", "麻", "强", "贾", "路", "娄", "危", "江", "童", "颜", "郭", "梅", "盛", "林", "刁", "钟", "徐", "邱", "骆", "高", "夏", "蔡", "田", "樊", "胡", "凌", "霍", "虞", "万", "支", "柯", "昝", "管", "卢", "莫", "经", "房", "裘", "缪", "干", "解", "应", "宗", "丁", "宣", "贲", "邓", "郁", "单", "杭", "洪", "包", "诸", "左", "石", "崔", "吉", "钮", "龚", "程", "嵇", "邢", "滑", "裴", "陆", "荣", "翁", "荀", "羊", "於", "惠", "甄", "麴", "家", "封", "芮", "羿", "储", "靳", "汲", "邴", "糜", "松", "井", "段", "富", "巫", "乌", "焦", "巴", "弓", "牧", "隗", "山", "谷", "车", "侯", "宓", "蓬", "全", "郗", "班", "仰", "秋", "仲", "伊", "宫", "宁", "仇", "栾", "暴", "甘", "钭", "厉", "戎", "祖", "武", "符", "刘", "景", "詹", "束", "龙", "叶", "幸", "司", "韶", "郜", "黎", "蓟", "薄", "印", "宿", "白", "怀", "蒲", "邰", "从", "鄂", "索", "咸", "籍", "赖", "卓", "蔺", "屠", "蒙", "池", "乔", "阴", "欎", "胥", "能", "苍", "双", "闻", "莘", "党", "翟", "谭", "贡", "劳", "逄", "姬", "申", "扶", "堵", "冉", "宰", "郦", "雍", "舄", "璩", "桑", "桂", "濮", "牛", "寿", "通", "边", "扈", "燕", "冀", "郏", "浦", "尚", "农", "温", "别", "庄", "晏", "柴", "瞿", "阎", "充", "慕", "连", "茹", "习", "宦", "艾", "鱼", "容", "向", "古", "易", "慎", "戈", "廖", "庾", "终", "暨", "居", "衡", "步", "都", "耿", "满", "弘", "匡", "国", "文", "寇", "广", "禄", "阙", "东", "殴", "殳", "沃", "利", "蔚", "越", "夔", "隆", "师", "巩", "厍", "聂", "晁", "勾", "敖", "融", "冷", "訾", "辛", "阚", "那", "简", "饶", "空", "曾", "毋", "沙", "乜", "养", "鞠", "须", "丰", "巢", "关", "蒯", "相", "查", "後", "荆", "红", "游", "竺", "权", "逯", "盖", "益", "桓", "公", "万俟", "司马", "上官", "欧阳", "夏侯", "诸葛", "闻人", "东方", "赫连", "皇甫", "尉迟", "公羊", "澹台", "公冶", "宗政", "濮阳", "淳于", "单于", "太叔", "申屠", "公孙", "仲孙", "轩辕", "令狐", "钟离", "宇文", "长孙", "慕容", "鲜于", "闾丘", "司徒", "司空", "亓官", "司寇", "仉", "督", "子车", "颛孙", "端木", "巫马", "公西", "漆雕", "乐正", "壤驷", "公良", "拓跋", "夹谷", "宰父", "谷梁", "晋", "楚", "闫", "法", "汝", "鄢", "涂", "钦", "段干", "百里", "东郭", "南门", "呼延", "归", "海", "羊舌", "微生", "岳", "帅", "缑", "亢", "况", "后", "有", "琴", "梁丘", "左丘", "东门", "西门", "商", "牟", "佘", "佴", "伯", "赏", "南宫", "墨", "哈", "谯", "笪", "年", "爱", "阳", "佟", "第五", "言", "福",
  547. # ]
  548. # for i in range(len(inner_table)):
  549. # for j in range(len(inner_table[i])):
  550. # if inner_table[i][j][1] == 1 \
  551. # and 2 <= len(inner_table[i][j][0]) <= 4 \
  552. # and (inner_table[i][j][0][0] in surname or inner_table[i][j][0][:2] in surname) \
  553. # and re.search("[^\u4e00-\u9fa5]", inner_table[i][j][0]) is None:
  554. # inner_table[i][j][1] = 0
  555. return inner_table
  556. def sliceTable(inner_table,fix_value="~~"):
  557. #进行分块
  558. height = len(inner_table)
  559. width = len(inner_table[0])
  560. head_list = []
  561. head_list.append(0)
  562. last_head = None
  563. last_is_same_value = False
  564. for h in range(height):
  565. is_all_key = True#是否是全表头行
  566. is_all_value = True#是否是全属性值
  567. is_same_with_lastHead = True#和上一行的结构是否相同
  568. is_same_value=True#一行的item都一样
  569. #is_same_first_item = True#与上一行的第一项是否相同
  570. same_value = inner_table[h][0][0]
  571. for w in range(width):
  572. if last_head is not None:
  573. if inner_table[h-1][w][0] != fix_value and inner_table[h-1][w][0] != "" and inner_table[h-1][w][1] == 0:
  574. is_all_key = False
  575. if inner_table[h][w][1]==1:
  576. is_all_value = False
  577. if inner_table[h][w][1]!= inner_table[h-1][w][1]:
  578. is_same_with_lastHead = False
  579. if inner_table[h][w][0]!=fix_value and inner_table[h][w][0]!=same_value:
  580. is_same_value = False
  581. else:
  582. if re.search("\d+",same_value) is not None:
  583. is_same_value = False
  584. if h>0 and inner_table[h][0][0]!=inner_table[h-1][0][0]:
  585. is_same_first_item = False
  586. last_head = h
  587. if last_is_same_value:
  588. last_is_same_value = is_same_value
  589. continue
  590. if is_same_value:
  591. # 该块只有表头一行不合法
  592. if h - head_list[-1] > 1:
  593. head_list.append(h)
  594. last_is_same_value = is_same_value
  595. continue
  596. if not is_all_key:
  597. if not is_same_with_lastHead:
  598. # 该块只有表头一行不合法
  599. if h - head_list[-1] > 1 or not is_all_value: # 20260331补充 整行是表头分块 修复 748505279 第二行才是表头分块错误 749002477 第一行最后一格为空非
  600. head_list.append(h)
  601. head_list.append(height)
  602. return head_list
  603. def setHead_initem(inner_table,pat_head,fix_value="~~",prob_min=0.5):
  604. set_item = set()
  605. height = len(inner_table)
  606. width = len(inner_table[0])
  607. empty_set = set()
  608. for i in range(height):
  609. for j in range(width):
  610. item = inner_table[i][j][0]
  611. if item.strip()=="":
  612. empty_set.add(item)
  613. else:
  614. set_item.add(item)
  615. list_item = list(set_item)
  616. if list_item:
  617. x = []
  618. for item in list_item:
  619. x.append(getPredictor("form").encode(item))
  620. predict_y = getPredictor("form").predict(np.array(x),type="item")
  621. _dict = dict()
  622. for item,values in zip(list_item,list(predict_y)):
  623. _dict[item] = values[1]
  624. # print("##",item,values)
  625. #print(_dict)
  626. for i in range(height):
  627. for j in range(width):
  628. item = inner_table[i][j][0]
  629. if item not in empty_set:
  630. inner_table[i][j][1] = 1 if _dict[item]>prob_min else (1 if re.search(pat_head,item) is not None and len(item)<8 else 0)
  631. # print("=====")
  632. # for item in inner_table:
  633. # print(item)
  634. # print("======")
  635. repairTable(inner_table)
  636. head_list = sliceTable(inner_table)
  637. return inner_table,head_list
  638. def set_head_model(inner_table, show=0):
  639. origin_inner_table = copy.deepcopy(inner_table)
  640. for i in range(len(inner_table)):
  641. for j in range(len(inner_table[i])):
  642. # 删掉单格前后符号,以免影响表头预测
  643. col = inner_table[i][j][0]
  644. col = re.sub("^[^\u4e00-\u9fa5a-zA-Z0-9]+", "", col)
  645. col = re.sub("[^\u4e00-\u9fa5a-zA-Z0-9]+$", "", col)
  646. inner_table[i][j] = col
  647. # 模型预测表头
  648. # predict_list = predict(inner_table)
  649. start_time = time.time()
  650. predict_list = predict(inner_table)
  651. # print('table head predict cost: ', time.time()-start_time)
  652. # 组合结果
  653. for i in range(len(inner_table)):
  654. if i == 0 and (inner_table[i] == ['序号', '内容', '说明与要求'] or inner_table[i] == ['序号', '公告事项', '内容']): # 修复 659860507 659817551 这种只识别第一行表头,第二列非表头导致解析错误问题
  655. inner_table[i] = [[it, 0] for it in ['序号', '内容', '说明与要求']]
  656. flag = 1
  657. for j in range(1, len(inner_table)): # 判断是否所有行都有3个格
  658. if len(inner_table[j]) != 3:
  659. flag = 0
  660. if flag:
  661. for j in range(1, len(inner_table)):
  662. inner_table[j] = [[t1, t2] for t1,t2 in zip(inner_table[j], [0, 1, 0])]
  663. break
  664. continue
  665. for j in range(len(inner_table[i])):
  666. inner_table[i][j] = [origin_inner_table[i][j][0], int(predict_list[i][j])]
  667. if origin_inner_table[i][j][0] in ['主要环境影响及预防或者减轻不良环境影响的对策和措施', '建设单位或地方政府作出的相关环保承诺',
  668. '公众反馈意见的联系方式', '区县', '项目领域', '成本/收入', '覆盖倍数', '会计所', '律所','建设期',
  669. "发行时间" ,"批次" ,"发行额" ,"发行利率" ,"所属债券" ,"专项债作资本金发行额" ,"调整记录","资产面积",
  670. "交易底价","交易地点","竞得者", "谈判项目", "资产名称","单元号","交易面积","承租人","租赁单价"] and predict_list[i][j]!=1:
  671. inner_table[i][j] = [origin_inner_table[i][j][0], 1]
  672. elif predict_list[i][j]!=1 and (re.search('^拟?(中标|中选|成交|承包|(参与)?投标|招标|采购|招租|发包|业主|竞投)(单位|人)(名称|地址|电话)?$'
  673. '|^拟?(中标|中选|成交)(供应商|金额|价格|日期)((万?元))?$|^项目(名称|编号)$', origin_inner_table[i][j][0])
  674. or re.match('(?[\d一二三四五六七八九十][\.、)]((招标|采购|\w{2,4})?项目名称|(招标|采购)人|项目概况|估算投资|预计招标时间|招标内容|其他)', origin_inner_table[i][j][0])):
  675. inner_table[i][j] = [origin_inner_table[i][j][0], 1]
  676. elif origin_inner_table[i][j][0] in ['经评审的最低评标价法'] and predict_list[i][j]==1:
  677. inner_table[i][j] = [origin_inner_table[i][j][0], 0]
  678. if show:
  679. print(json.dumps(inner_table, ensure_ascii=False))
  680. print("="*80)
  681. print("table_head before repair")
  682. for r in inner_table:
  683. print('row', r)
  684. print("="*80)
  685. # 表头修正
  686. # repairTable(inner_table)
  687. inner_table = table_head_repair_process(inner_table, docid)
  688. # 组合结果
  689. for i in range(len(inner_table)):
  690. for j in range(len(inner_table[i])):
  691. inner_table[i][j] = [origin_inner_table[i][j][0], int(inner_table[i][j][1])]
  692. if show:
  693. print("table_head after repair")
  694. for r in inner_table:
  695. print('row', r)
  696. print("="*80)
  697. # 按表头分割表格
  698. head_list = sliceTable(inner_table)
  699. return inner_table, head_list
  700. def setHead_incontext(inner_table,pat_head,fix_value="~~",prob_min=0.5):
  701. data_x,data_position = getPredictor("form").getModel("context").encode(inner_table)
  702. predict_y = getPredictor("form").getModel("context").predict(data_x)
  703. for _position,_y in zip(data_position,predict_y):
  704. _w = _position[0]
  705. _h = _position[1]
  706. if _y[1]>prob_min:
  707. inner_table[_h][_w][1] = 1
  708. else:
  709. inner_table[_h][_w][1] = 0
  710. _item = inner_table[_h][_w][0]
  711. if re.search(pat_head,_item) is not None and len(_item)<8:
  712. inner_table[_h][_w][1] = 1
  713. # print("=====")
  714. # for item in inner_table:
  715. # print(item)
  716. # print("======")
  717. height = len(inner_table)
  718. width = len(inner_table[0])
  719. for i in range(height):
  720. for j in range(width):
  721. if re.search("[::]$", inner_table[i][j][0]) and len(inner_table[i][j][0])<8:
  722. inner_table[i][j][1] = 1
  723. repairTable(inner_table)
  724. head_list = sliceTable(inner_table)
  725. # print("inner_table:",inner_table)
  726. return inner_table,head_list
  727. #设置表头
  728. def setHead_inline(inner_table,prob_min=0.64):
  729. pad_row = "@@"
  730. pad_col = "##"
  731. removePadding(inner_table, pad_row, pad_col)
  732. pad_pattern = re.compile(pad_row+"|"+pad_col)
  733. height = len(inner_table)
  734. width = len(inner_table[0])
  735. head_list = []
  736. head_list.append(0)
  737. #行表头
  738. is_head_last = False
  739. for i in range(height):
  740. is_head = False
  741. is_long_value = False
  742. #判断是否是全padding值
  743. is_same_value = True
  744. same_value = inner_table[i][0][0]
  745. for j in range(width):
  746. if inner_table[i][j][0]!=same_value and inner_table[i][j][0]!=pad_row:
  747. is_same_value = False
  748. break
  749. #predict is head or not with model
  750. temp_item = ""
  751. for j in range(width):
  752. temp_item += inner_table[i][j][0]+"|"
  753. temp_item = re.sub(pad_pattern,"",temp_item)
  754. form_prob = getPredictor("form").predict(formEncoding(temp_item,expand=True),type="line")
  755. if form_prob is not None:
  756. if form_prob[0][1]>prob_min:
  757. is_head = True
  758. else:
  759. is_head = False
  760. #print(temp_item,form_prob)
  761. if len(inner_table[i][0][0])>40:
  762. is_long_value = True
  763. if is_head or is_long_value or is_same_value:
  764. #不把连续表头分开
  765. if not is_head_last:
  766. head_list.append(i)
  767. if is_long_value or is_same_value:
  768. head_list.append(i+1)
  769. if is_head:
  770. for j in range(width):
  771. inner_table[i][j][1] = 1
  772. is_head_last = is_head
  773. head_list.append(height)
  774. #列表头
  775. for i in range(len(head_list)-1):
  776. head_begin = head_list[i]
  777. head_end = head_list[i+1]
  778. #最后一列不设置为列表头
  779. for i in range(width-1):
  780. is_head = False
  781. #predict is head or not with model
  782. temp_item = ""
  783. for j in range(head_begin,head_end):
  784. temp_item += inner_table[j][i][0]+"|"
  785. temp_item = re.sub(pad_pattern,"",temp_item)
  786. form_prob = getPredictor("form").predict(formEncoding(temp_item,expand=True),type="line")
  787. if form_prob is not None:
  788. if form_prob[0][1]>prob_min:
  789. is_head = True
  790. else:
  791. is_head = False
  792. if is_head:
  793. for j in range(head_begin,head_end):
  794. inner_table[j][i][1] = 2
  795. addPadding(inner_table, pad_row, pad_col)
  796. return inner_table,head_list
  797. #设置表头
  798. def setHead_withRule(inner_table,pattern,pat_value,count):
  799. height = len(inner_table)
  800. width = len(inner_table[0])
  801. head_list = []
  802. head_list.append(0)
  803. #行表头
  804. is_head_last = False
  805. for i in range(height):
  806. set_match = set()
  807. is_head = False
  808. is_long_value = False
  809. is_same_value = True
  810. same_value = inner_table[i][0][0]
  811. for j in range(width):
  812. if inner_table[i][j][0]!=same_value:
  813. is_same_value = False
  814. break
  815. for j in range(width):
  816. if re.search(pat_value,inner_table[i][j][0]) is not None:
  817. is_head = False
  818. break
  819. str_find = re.findall(pattern,inner_table[i][j][0])
  820. if len(str_find)>0:
  821. set_match.add(inner_table[i][j][0])
  822. if len(set_match)>=count:
  823. is_head = True
  824. if len(inner_table[i][0][0])>40:
  825. is_long_value = True
  826. if is_head or is_long_value or is_same_value:
  827. if not is_head_last:
  828. head_list.append(i)
  829. if is_head:
  830. for j in range(width):
  831. inner_table[i][j][1] = 1
  832. is_head_last = is_head
  833. head_list.append(height)
  834. #列表头
  835. for i in range(len(head_list)-1):
  836. head_begin = head_list[i]
  837. head_end = head_list[i+1]
  838. #最后一列不设置为列表头
  839. for i in range(width-1):
  840. set_match = set()
  841. is_head = False
  842. for j in range(head_begin,head_end):
  843. if re.search(pat_value,inner_table[j][i][0]) is not None:
  844. is_head = False
  845. break
  846. str_find = re.findall(pattern,inner_table[j][i][0])
  847. if len(str_find)>0:
  848. set_match.add(inner_table[j][i][0])
  849. if len(set_match)>=count:
  850. is_head = True
  851. if is_head:
  852. for j in range(head_begin,head_end):
  853. inner_table[j][i][1] = 2
  854. return inner_table,head_list
  855. #取得表格的处理方向
  856. def getDirect(inner_table,begin,end):
  857. '''
  858. column_head = set()
  859. row_head = set()
  860. widths = len(inner_table[0])
  861. for height in range(begin,end):
  862. for width in range(widths):
  863. if inner_table[height][width][1] ==1:
  864. row_head.add(height)
  865. if inner_table[height][width][1] ==2:
  866. column_head.add(width)
  867. company_pattern = re.compile("公司")
  868. if 0 in column_head and begin not in row_head:
  869. return "column"
  870. if 0 in column_head and begin in row_head:
  871. for height in range(begin,end):
  872. count = 0
  873. count_flag = True
  874. for width_index in range(width):
  875. if inner_table[height][width_index][1]==0:
  876. if re.search(company_pattern,inner_table[height][width_index][0]) is not None:
  877. count += 1
  878. else:
  879. count_flag = False
  880. if count_flag and count>=2:
  881. return "column"
  882. return "row"
  883. '''
  884. count_row_keys = 0
  885. count_column_keys = 0
  886. width = len(inner_table[0])
  887. if begin<end:
  888. for w in range(len(inner_table[begin])):
  889. if inner_table[begin][w][1]!=0:
  890. count_row_keys += 1
  891. for h in range(begin,end):
  892. if inner_table[h][0][1]!=0:
  893. count_column_keys += 1
  894. company_pattern = re.compile("有限(责任)?公司")
  895. for height in range(begin,end):
  896. count_set = set()
  897. count_flag = True
  898. for width_index in range(width):
  899. if inner_table[height][width_index][1]==0:
  900. if re.search(company_pattern,inner_table[height][width_index][0]) is not None:
  901. count_set.add(inner_table[height][width_index][0])
  902. else:
  903. count_flag = False
  904. if count_flag and len(count_set)>=2:
  905. return "column"
  906. # if count_column_keys>count_row_keys: #2022/2/15 此项不够严谨,造成很多错误,故取消
  907. # return "column"
  908. return "row"
  909. #根据表格处理方向生成句子,
  910. def getTableText(inner_table,head_list,key_direct=False):
  911. # packPattern = "(标包|[标包][号段名])"
  912. packPattern = "(标包|标的|标项|品目|[标包][号段名]|((项目|物资|设备|场次|标段|标的|产品)(名称)))" # 2020/11/23 大网站规则,补充采购类包名
  913. rankPattern = "(排名|排序|名次|序号|评标结果|评审结果|是否中标|推荐(意见|情况)|评标情况|推荐顺序|选取(情况|说明))" # 2020/11/23 大网站规则,添加序号为排序
  914. entityPattern = "((候选|[中投]标|报价)(单位|公司|人|供应商))|供应商名称"
  915. moneyPattern = "([中投]标|报价)(金额|价)"
  916. height = len(inner_table)
  917. width = len(inner_table[0])
  918. text = ""
  919. for head_i in range(len(head_list)-1):
  920. head_begin = head_list[head_i]
  921. head_end = head_list[head_i+1]
  922. direct = getDirect(inner_table, head_begin, head_end)
  923. #若只有一行,则直接按行读取
  924. if head_end-head_begin==1:
  925. text_line = ""
  926. for i in range(head_begin,head_end):
  927. for w in range(len(inner_table[i])):
  928. if inner_table[i][w][1]==1:
  929. _punctuation = ":"
  930. else:
  931. _punctuation = "," #2021/12/15 统一为中文标点,避免 206893924 国际F座1108,1,009,197.49元
  932. if w>0:
  933. if inner_table[i][w][0]!= inner_table[i][w-1][0]:
  934. text_line += inner_table[i][w][0]+_punctuation
  935. else:
  936. text_line += inner_table[i][w][0]+_punctuation
  937. text_line = text_line+"。" if text_line!="" else text_line
  938. text += text_line
  939. else:
  940. #构建一个共现矩阵
  941. table_occurence = []
  942. for i in range(head_begin,head_end):
  943. line_oc = []
  944. for j in range(width):
  945. cell = inner_table[i][j]
  946. line_oc.append({"text":cell[0],"type":cell[1],"occu_count":0,"left_head":"","top_head":"","left_dis":0,"top_dis":0})
  947. table_occurence.append(line_oc)
  948. occu_height = len(table_occurence)
  949. occu_width = len(table_occurence[0]) if len(table_occurence)>0 else 0
  950. #为每个属性值寻找表头
  951. for i in range(occu_height):
  952. for j in range(occu_width):
  953. cell = table_occurence[i][j]
  954. #是属性值
  955. if cell["type"]==0 and cell["text"]!="":
  956. left_head = ""
  957. top_head = ""
  958. find_flag = False
  959. temp_head = ""
  960. for loop_i in range(1,i+1):
  961. if not key_direct:
  962. key_values = [1,2]
  963. else:
  964. key_values = [1]
  965. if table_occurence[i-loop_i][j]["type"] in key_values:
  966. if find_flag:
  967. if table_occurence[i-loop_i][j]["text"]!=temp_head:
  968. top_head = table_occurence[i-loop_i][j]["text"]+":"+top_head
  969. else:
  970. top_head = table_occurence[i-loop_i][j]["text"]+":"+top_head
  971. find_flag = True
  972. temp_head = table_occurence[i-loop_i][j]["text"]
  973. table_occurence[i-loop_i][j]["occu_count"] += 1
  974. else:
  975. #找到表头后遇到属性值就返回
  976. if find_flag:
  977. break
  978. cell["top_head"] += top_head
  979. find_flag = False
  980. temp_head = ""
  981. for loop_j in range(1,j+1):
  982. if not key_direct:
  983. key_values = [1,2]
  984. else:
  985. key_values = [2]
  986. if table_occurence[i][j-loop_j]["type"] in key_values:
  987. if find_flag:
  988. if table_occurence[i][j-loop_j]["text"]!=temp_head:
  989. left_head = table_occurence[i][j-loop_j]["text"]+":"+left_head
  990. else:
  991. left_head = table_occurence[i][j-loop_j]["text"]+":"+left_head
  992. find_flag = True
  993. temp_head = table_occurence[i][j-loop_j]["text"]
  994. table_occurence[i][j-loop_j]["occu_count"] += 1
  995. else:
  996. if find_flag:
  997. break
  998. cell["left_head"] += left_head
  999. if direct=="row":
  1000. for i in range(occu_height):
  1001. pack_text = ""
  1002. rank_text = ""
  1003. entity_text = ""
  1004. text_line = ""
  1005. money_text = ""
  1006. #在同一句话中重复的可以去掉
  1007. text_set = set()
  1008. head = ""
  1009. last_text = ""
  1010. for j in range(width):
  1011. cell = table_occurence[i][j]
  1012. if cell["type"]==0 or (cell["type"]==1 and cell["occu_count"]==0):
  1013. cell = table_occurence[i][j]
  1014. head = (cell["top_head"]+":") if len(cell["top_head"])>0 else ""
  1015. if re.search("[单报标限总]价|金额|成交报?价|报价|供应商|候选人|中标人|[利费]率|负责人|工期|服务(期限?|年限|时间|日期|周期)|(履约|履行)期限|合同(期限?|(完成|截止)(日期|时间))", head):
  1016. head = cell["left_head"] + head
  1017. else:
  1018. head += cell["left_head"]
  1019. if str(head+cell["text"]) in text_set:
  1020. continue
  1021. if re.search(packPattern,head) is not None:
  1022. pack_text += head+cell["text"]+","
  1023. elif re.search(rankPattern,head) is not None and re.search('(排名|排序|名次|顺序):?第?[\d一二三]', rank_text)==None: # 2020/11/23 大网站规则发现问题,if 改elif 20240620修复同时有排名及评标情况造成错误
  1024. #排名替换为同一种表达
  1025. rank_text += head+cell["text"]+","
  1026. #print(rank_text)
  1027. elif re.search(entityPattern,head) is not None:
  1028. entity_text += head+cell["text"]+","
  1029. #print(entity_text)
  1030. else:
  1031. if re.search(moneyPattern,head) is not None and entity_text!="":
  1032. money_text += head+cell["text"]+","
  1033. else:
  1034. text_line += head+cell["text"]+","
  1035. text_set.add(str(head+cell["text"]))
  1036. last_text = cell['text']
  1037. tr_text = pack_text+rank_text+entity_text+money_text+text_line
  1038. text += pack_text+rank_text+entity_text+money_text+text_line
  1039. # text = text[:-1] + "。" if len(text) > 0 else text
  1040. if len(text_set-set([' ']))==1 and head == '' and len(last_text)< 25: # 修复367694716分两行表达
  1041. text = text if re.search('\w$', text[:-1]) else text[:-1]
  1042. elif (width == 2 or len(text_set)==1) and head != '' and len(tr_text)<50: # 修复494731937只有两行的,分句不合理
  1043. text = text if re.search('\w$', text[:-1]) else text[:-1]
  1044. else:
  1045. text = text[:-1] + "。"
  1046. else:
  1047. for j in range(occu_width):
  1048. pack_text = ""
  1049. rank_text = ""
  1050. entity_text = ""
  1051. text_line = ""
  1052. text_set = set()
  1053. for i in range(occu_height):
  1054. cell = table_occurence[i][j]
  1055. if cell["type"]==0 or (cell["type"]==1 and cell["occu_count"]==0):
  1056. cell = table_occurence[i][j]
  1057. head = (cell["left_head"]+"") if len(cell["left_head"])>0 else ""
  1058. if re.search("[单报标限总]价|金额|成交报?价|报价|供应商|候选人|中标人|[利费]率|负责人|工期|服务(期限?|年限|时间|日期|周期)|(履约|履行)期限|合同(期限?|(完成|截止)(日期|时间))", head):
  1059. head = cell["top_head"] + head
  1060. else:
  1061. head += cell["top_head"]
  1062. if str(head+cell["text"]) in text_set:
  1063. continue
  1064. if re.search(packPattern,head) is not None:
  1065. pack_text += head+cell["text"]+","
  1066. elif re.search(rankPattern,head) is not None: # 2020/11/23 大网站规则发现问题,if 改elif
  1067. #排名替换为同一种表达
  1068. rank_text += head+cell["text"]+","
  1069. #print(rank_text)
  1070. elif re.search(entityPattern,head) is not None and \
  1071. re.search('业绩|资格|条件',head)==None and re.search('业绩',cell["text"])==None : #2021/10/19 解决包含业绩的行调到前面问题
  1072. entity_text += head+cell["text"]+","
  1073. #print(entity_text)
  1074. else:
  1075. text_line += head+cell["text"]+","
  1076. text_set.add(str(head+cell["text"]))
  1077. text += pack_text+rank_text+entity_text+text_line
  1078. text = text[:-1]+"。" if len(text)>0 else text
  1079. # if direct=="row":
  1080. # for i in range(head_begin,head_end):
  1081. # pack_text = ""
  1082. # rank_text = ""
  1083. # entity_text = ""
  1084. # text_line = ""
  1085. # #在同一句话中重复的可以去掉
  1086. # text_set = set()
  1087. # for j in range(width):
  1088. # cell = inner_table[i][j]
  1089. # #是属性值
  1090. # if cell[1]==0 and cell[0]!="":
  1091. # head = ""
  1092. #
  1093. # find_flag = False
  1094. # temp_head = ""
  1095. # for loop_i in range(0,i+1-head_begin):
  1096. # if not key_direct:
  1097. # key_values = [1,2]
  1098. # else:
  1099. # key_values = [1]
  1100. # if inner_table[i-loop_i][j][1] in key_values:
  1101. # if find_flag:
  1102. # if inner_table[i-loop_i][j][0]!=temp_head:
  1103. # head = inner_table[i-loop_i][j][0]+":"+head
  1104. # else:
  1105. # head = inner_table[i-loop_i][j][0]+":"+head
  1106. # find_flag = True
  1107. # temp_head = inner_table[i-loop_i][j][0]
  1108. # else:
  1109. # #找到表头后遇到属性值就返回
  1110. # if find_flag:
  1111. # break
  1112. #
  1113. # find_flag = False
  1114. # temp_head = ""
  1115. #
  1116. #
  1117. #
  1118. # for loop_j in range(1,j+1):
  1119. # if not key_direct:
  1120. # key_values = [1,2]
  1121. # else:
  1122. # key_values = [2]
  1123. # if inner_table[i][j-loop_j][1] in key_values:
  1124. # if find_flag:
  1125. # if inner_table[i][j-loop_j][0]!=temp_head:
  1126. # head = inner_table[i][j-loop_j][0]+":"+head
  1127. # else:
  1128. # head = inner_table[i][j-loop_j][0]+":"+head
  1129. # find_flag = True
  1130. # temp_head = inner_table[i][j-loop_j][0]
  1131. # else:
  1132. # if find_flag:
  1133. # break
  1134. #
  1135. # if str(head+inner_table[i][j][0]) in text_set:
  1136. # continue
  1137. # if re.search(packPattern,head) is not None:
  1138. # pack_text += head+inner_table[i][j][0]+","
  1139. # elif re.search(rankPattern,head) is not None: # 2020/11/23 大网站规则发现问题,if 改elif
  1140. # #排名替换为同一种表达
  1141. # rank_text += head+inner_table[i][j][0]+","
  1142. # #print(rank_text)
  1143. # elif re.search(entityPattern,head) is not None:
  1144. # entity_text += head+inner_table[i][j][0]+","
  1145. # #print(entity_text)
  1146. # else:
  1147. # text_line += head+inner_table[i][j][0]+","
  1148. # text_set.add(str(head+inner_table[i][j][0]))
  1149. # text += pack_text+rank_text+entity_text+text_line
  1150. # text = text[:-1]+"。" if len(text)>0 else text
  1151. # else:
  1152. # for j in range(width):
  1153. #
  1154. # rank_text = ""
  1155. # entity_text = ""
  1156. # text_line = ""
  1157. # text_set = set()
  1158. # for i in range(head_begin,head_end):
  1159. # cell = inner_table[i][j]
  1160. # #是属性值
  1161. # if cell[1]==0 and cell[0]!="":
  1162. # find_flag = False
  1163. # head = ""
  1164. # temp_head = ""
  1165. #
  1166. # for loop_j in range(1,j+1):
  1167. # if not key_direct:
  1168. # key_values = [1,2]
  1169. # else:
  1170. # key_values = [2]
  1171. # if inner_table[i][j-loop_j][1] in key_values:
  1172. # if find_flag:
  1173. # if inner_table[i][j-loop_j][0]!=temp_head:
  1174. # head = inner_table[i][j-loop_j][0]+":"+head
  1175. # else:
  1176. # head = inner_table[i][j-loop_j][0]+":"+head
  1177. # find_flag = True
  1178. # temp_head = inner_table[i][j-loop_j][0]
  1179. # else:
  1180. # if find_flag:
  1181. # break
  1182. # find_flag = False
  1183. # temp_head = ""
  1184. # for loop_i in range(0,i+1-head_begin):
  1185. # if not key_direct:
  1186. # key_values = [1,2]
  1187. # else:
  1188. # key_values = [1]
  1189. # if inner_table[i-loop_i][j][1] in key_values:
  1190. # if find_flag:
  1191. # if inner_table[i-loop_i][j][0]!=temp_head:
  1192. # head = inner_table[i-loop_i][j][0]+":"+head
  1193. # else:
  1194. # head = inner_table[i-loop_i][j][0]+":"+head
  1195. # find_flag = True
  1196. # temp_head = inner_table[i-loop_i][j][0]
  1197. # else:
  1198. # if find_flag:
  1199. # break
  1200. # if str(head+inner_table[i][j][0]) in text_set:
  1201. # continue
  1202. # if re.search(rankPattern,head) is not None:
  1203. # rank_text += head+inner_table[i][j][0]+","
  1204. # #print(rank_text)
  1205. # elif re.search(entityPattern,head) is not None:
  1206. # entity_text += head+inner_table[i][j][0]+","
  1207. # #print(entity_text)
  1208. # else:
  1209. # text_line += head+inner_table[i][j][0]+","
  1210. # text_set.add(str(head+inner_table[i][j][0]))
  1211. # text += rank_text+entity_text+text_line
  1212. # text = text[:-1]+"。" if len(text)>0 else text
  1213. if text.endswith(',。'):
  1214. text = re.sub(',+。', '。', text)
  1215. elif text.endswith(','):
  1216. text = text.rstrip(',') + '。'
  1217. else:
  1218. text += '。' # 20260401 表格末尾加句号与其他内容分开 749870793 避免表格与外面混淆
  1219. return text
  1220. def get_table_text_kv(inner_table, head_list, key_direct=False):
  1221. packPattern = "(标包|标的|标项|品目|[标包][号段名]|((项目|物资|设备|场次|标段|标的|产品)(名称)))" # 2020/11/23 大网站规则,补充采购类包名
  1222. rankPattern = "(排名|排序|名次|序号|评标结果|评审结果|是否中标|推荐意见|评标情况|推荐顺序|选取(情况|说明))" # 2020/11/23 大网站规则,添加序号为排序
  1223. entityPattern = "((候选|[中投]标|报价)(单位|公司|人|供应商))|供应商名称"
  1224. moneyPattern = "([中投]标|报价)(金额|价)"
  1225. width = len(inner_table[0])
  1226. text = ""
  1227. all_table_occurence = []
  1228. for head_i in range(len(head_list) - 1):
  1229. head_begin = head_list[head_i]
  1230. head_end = head_list[head_i + 1]
  1231. direct = getDirect(inner_table, head_begin, head_end)
  1232. # print(inner_table[head_begin:head_end])
  1233. # print('direct', direct)
  1234. # 构建一个共现矩阵
  1235. table_occurence = []
  1236. for i in range(head_begin, head_end):
  1237. line_oc = []
  1238. for j in range(width):
  1239. cell = inner_table[i][j]
  1240. line_oc.append(
  1241. {"text": cell[0], "type": cell[1], "occu_count": 0, "left_head": "", "top_head": "",
  1242. "left_dis": 0, "top_dis": 0,
  1243. "text_row_index": i, "text_col_index": j
  1244. })
  1245. table_occurence.append(line_oc)
  1246. occu_height = len(table_occurence)
  1247. occu_width = len(table_occurence[0]) if len(table_occurence) > 0 else 0
  1248. # 为每个属性值寻找表头
  1249. for i in range(occu_height):
  1250. for j in range(occu_width):
  1251. cell = table_occurence[i][j]
  1252. # 是属性值
  1253. if cell["type"] == 0 and cell["text"] != "":
  1254. left_head = ""
  1255. top_head = ""
  1256. find_flag = False
  1257. temp_head = ""
  1258. head_row_col_list = []
  1259. for loop_i in range(1, i + 1):
  1260. if not key_direct:
  1261. key_values = [1, 2]
  1262. else:
  1263. key_values = [1]
  1264. if table_occurence[i - loop_i][j]["type"] in key_values:
  1265. if find_flag:
  1266. if table_occurence[i - loop_i][j]["text"] != temp_head:
  1267. if cell.get("top_head_list"):
  1268. cell["top_head_list"] += [table_occurence[i - loop_i][j]["text"] + ":"]
  1269. else:
  1270. cell["top_head_list"] = [table_occurence[i - loop_i][j]["text"] + ":"]
  1271. top_head = table_occurence[i - loop_i][j]["text"] + ":" + top_head
  1272. head_row_col_list.append([i - loop_i, j])
  1273. else:
  1274. if cell.get("top_head_list"):
  1275. cell["top_head_list"] += [table_occurence[i - loop_i][j]["text"] + ":"]
  1276. else:
  1277. cell["top_head_list"] = [table_occurence[i - loop_i][j]["text"] + ":"]
  1278. top_head = table_occurence[i - loop_i][j]["text"] + ":" + top_head
  1279. head_row_col_list.append([i - loop_i, j])
  1280. find_flag = True
  1281. temp_head = table_occurence[i - loop_i][j]["text"]
  1282. table_occurence[i - loop_i][j]["occu_count"] += 1
  1283. else:
  1284. # 找到表头后遇到属性值就返回
  1285. if find_flag:
  1286. break
  1287. cell["top_head"] += top_head
  1288. if cell.get("top_head_row_index"):
  1289. cell["top_head_row_index"] += [x[0] for x in head_row_col_list]
  1290. else:
  1291. cell["top_head_row_index"] = [x[0] for x in head_row_col_list]
  1292. if cell.get("top_head_col_index"):
  1293. cell["top_head_col_index"] += [x[1] for x in head_row_col_list]
  1294. else:
  1295. cell["top_head_col_index"] = [x[1] for x in head_row_col_list]
  1296. find_flag = False
  1297. temp_head = ""
  1298. head_row_col_list = []
  1299. for loop_j in range(1, j + 1):
  1300. if not key_direct:
  1301. key_values = [1, 2]
  1302. else:
  1303. key_values = [2]
  1304. if table_occurence[i][j - loop_j]["type"] in key_values:
  1305. if find_flag:
  1306. if table_occurence[i][j - loop_j]["text"] != temp_head:
  1307. if cell.get("left_head_list"):
  1308. cell["left_head_list"] += [table_occurence[i][j - loop_j]["text"] + ":"]
  1309. else:
  1310. cell["left_head_list"] = [table_occurence[i][j - loop_j]["text"] + ":"]
  1311. left_head = table_occurence[i][j - loop_j]["text"] + ":" + left_head
  1312. head_row_col_list.append([i, j - loop_j])
  1313. else:
  1314. if cell.get("left_head_list"):
  1315. cell["left_head_list"] += [table_occurence[i][j - loop_j]["text"] + ":"]
  1316. else:
  1317. cell["left_head_list"] = [table_occurence[i][j - loop_j]["text"] + ":"]
  1318. left_head = table_occurence[i][j - loop_j]["text"] + ":" + left_head
  1319. head_row_col_list.append([i, j - loop_j])
  1320. find_flag = True
  1321. temp_head = table_occurence[i][j - loop_j]["text"]
  1322. table_occurence[i][j - loop_j]["occu_count"] += 1
  1323. else:
  1324. if find_flag:
  1325. break
  1326. cell["left_head"] += left_head
  1327. if cell.get("left_head_row_index"):
  1328. cell["left_head_row_index"] += [x[0] for x in head_row_col_list]
  1329. else:
  1330. cell["left_head_row_index"] = [x[0] for x in head_row_col_list]
  1331. if cell.get("left_head_col_index"):
  1332. cell["left_head_col_index"] += [x[1] for x in head_row_col_list]
  1333. else:
  1334. cell["left_head_col_index"] = [x[1] for x in head_row_col_list]
  1335. # 连接表头和属性值
  1336. if direct == "row":
  1337. for i in range(occu_height):
  1338. pack_text = ""
  1339. rank_text = ""
  1340. entity_text = ""
  1341. text_line = ""
  1342. money_text = ""
  1343. # 在同一句话中重复的可以去掉
  1344. text_set = set()
  1345. head = ""
  1346. last_text = ""
  1347. pack_text_location = []
  1348. rank_text_location = []
  1349. entity_text_location = []
  1350. text_line_location = []
  1351. money_text_location = []
  1352. for j in range(width):
  1353. cell = table_occurence[i][j]
  1354. if cell["type"] == 0 or (cell["type"] == 1 and cell["occu_count"] == 0):
  1355. cell = table_occurence[i][j]
  1356. head = (cell["top_head"] + ":") if len(cell["top_head"]) > 0 else ""
  1357. now_top_head = copy.deepcopy(head)
  1358. now_left_head = copy.deepcopy(cell["left_head"])
  1359. if re.search(
  1360. "[单报标限总]价|金额|成交报?价|报价|供应商|候选人|中标人|[利费]率|负责人|工期|服务(期限?|年限|时间|日期|周期)|("
  1361. "履约|履行)期限|合同(期限?|(完成|截止)(日期|时间))",
  1362. head):
  1363. head = cell["left_head"] + head
  1364. left_first = 1
  1365. else:
  1366. head += cell["left_head"]
  1367. left_first = 0
  1368. # print('len(text), len(sub_text), len(head)', cell["text"], len(text), len(sub_text), len(head))
  1369. # print('text111', text)
  1370. # print('pack_text, rank_text, entity_text, money_text, text_line', '1'+pack_text, '2'+rank_text, '3'+entity_text, '4'+money_text, '5'+text_line)
  1371. # print('head', head)
  1372. # print('sub_text111', sub_text)
  1373. if str(head + cell["text"]) in text_set:
  1374. cell['drop'] = 1
  1375. continue
  1376. if re.search(packPattern, head) is not None:
  1377. pack_text += head + cell["text"] + ","
  1378. pack_text_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]]
  1379. # 2020/11/23 大网站规则发现问题,if 改elif 20240620修复同时有排名及评标情况造成错误
  1380. elif re.search(rankPattern, head) is not None and re.search('(排名|排序|名次|顺序):?第?[\d一二三]', rank_text) is None:
  1381. # 排名替换为同一种表达
  1382. rank_text += head + cell["text"] + ","
  1383. rank_text_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]]
  1384. elif re.search(entityPattern, head) is not None:
  1385. entity_text += head + cell["text"] + ","
  1386. entity_text_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]]
  1387. else:
  1388. if re.search(moneyPattern, head) is not None and entity_text != "":
  1389. money_text += head + cell["text"] + ","
  1390. money_text_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]]
  1391. else:
  1392. text_line += head + cell["text"] + ","
  1393. text_line_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]]
  1394. text_set.add(str(head + cell["text"]))
  1395. last_text = cell['text']
  1396. # 计算key value在sentence的index
  1397. head_location_list = pack_text_location + rank_text_location + entity_text_location + text_line_location + money_text_location
  1398. current_loc = 0
  1399. for ii, jj, head_text, now_left_head, now_top_head, left_first in head_location_list:
  1400. cell = table_occurence[ii][jj]
  1401. # 左表头先于右表头
  1402. if left_first:
  1403. cell['left_head_sen_index'] = len(text) + current_loc
  1404. cell['top_head_sen_index'] = len(text) + current_loc + len(now_left_head)
  1405. else:
  1406. cell['left_head_sen_index'] = len(text) + current_loc + len(now_top_head)
  1407. cell['top_head_sen_index'] = len(text) + current_loc
  1408. cell['text_sen_index'] = len(text) + current_loc + len(now_left_head + now_top_head)
  1409. current_loc += len(head_text)
  1410. tr_text = pack_text + rank_text + entity_text + money_text + text_line
  1411. text += pack_text + rank_text + entity_text + money_text + text_line
  1412. # 修复367694716分两行表达
  1413. if len(text_set - set([' '])) == 1 and head == '' and len(last_text) < 25:
  1414. text = text if re.search('\w$', text[:-1]) else text[:-1]
  1415. # 修复494731937只有两行的,分句不合理
  1416. elif (width == 2 or len(text_set) == 1) and head != '' and len(tr_text) < 50:
  1417. text = text if re.search('\w$', text[:-1]) else text[:-1]
  1418. else:
  1419. text = text[:-1] + "。"
  1420. else:
  1421. for j in range(occu_width):
  1422. pack_text = ""
  1423. rank_text = ""
  1424. entity_text = ""
  1425. text_line = ""
  1426. text_set = set()
  1427. pack_text_location = []
  1428. rank_text_location = []
  1429. entity_text_location = []
  1430. text_line_location = []
  1431. money_text_location = []
  1432. for i in range(occu_height):
  1433. cell = table_occurence[i][j]
  1434. if cell["type"] == 0 or (cell["type"] == 1 and cell["occu_count"] == 0):
  1435. cell = table_occurence[i][j]
  1436. head = (cell["left_head"] + "") if len(cell["left_head"]) > 0 else ""
  1437. now_top_head = copy.deepcopy(cell["top_head"])
  1438. now_left_head = copy.deepcopy(head)
  1439. if re.search("[单报标限总]价|金额|成交报?价|报价|供应商|候选人|中标人|[利费]率|负责人|工期|服务(期限?|年限|时间|日期|周期)|(履约|履行)期限|合同(期限?|(完成|截止)(日期|时间))", head):
  1440. head = cell["top_head"] + head
  1441. left_first = 0
  1442. else:
  1443. head += cell["top_head"]
  1444. left_first = 1
  1445. if str(head + cell["text"]) in text_set:
  1446. cell['drop'] = 1
  1447. continue
  1448. if re.search(packPattern, head) is not None:
  1449. pack_text += head + cell["text"] + ","
  1450. pack_text_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]]
  1451. # 2020/11/23 大网站规则发现问题,if 改elif
  1452. elif re.search(rankPattern, head) is not None:
  1453. # 排名替换为同一种表达
  1454. rank_text += head + cell["text"] + ","
  1455. rank_text_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]]
  1456. # 2021/10/19 解决包含业绩的行调到前面问题
  1457. elif re.search(entityPattern, head) is not None and \
  1458. re.search('业绩|资格|条件', head) is None and re.search('业绩', cell["text"]) is None:
  1459. entity_text += head + cell["text"] + ","
  1460. entity_text_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]]
  1461. else:
  1462. text_line += head + cell["text"] + ","
  1463. text_line_location += [[i, j, head + cell["text"] + ",", now_left_head, now_top_head, left_first]]
  1464. text_set.add(str(head + cell["text"]))
  1465. # 计算key value在sentence的index
  1466. head_location_list = pack_text_location + rank_text_location + entity_text_location + text_line_location + money_text_location
  1467. current_loc = 0
  1468. for ii, jj, head_text, now_left_head, now_top_head, left_first in head_location_list:
  1469. cell = table_occurence[ii][jj]
  1470. # 左表头先于右表头
  1471. if left_first:
  1472. cell['left_head_sen_index'] = len(text) + current_loc
  1473. cell['top_head_sen_index'] = len(text) + current_loc + len(now_left_head)
  1474. else:
  1475. cell['left_head_sen_index'] = len(text) + current_loc + len(now_top_head)
  1476. cell['top_head_sen_index'] = len(text) + current_loc
  1477. cell['text_sen_index'] = len(text) + current_loc + len(now_left_head + now_top_head)
  1478. current_loc += len(head_text)
  1479. text += pack_text + rank_text + entity_text + text_line
  1480. text = text[:-1] + "。" if len(text) > 0 else text
  1481. all_table_occurence += table_occurence
  1482. return text, all_table_occurence
  1483. def process_dict(text, table):
  1484. kv_list = []
  1485. kv_dict_list = []
  1486. # print('text', len(text), text, ),
  1487. # print('table', table)
  1488. for r_index, row in enumerate(table):
  1489. for c_index, col in enumerate(row):
  1490. # print('col', col)
  1491. if col['type'] == 1:
  1492. continue
  1493. if col.get('drop'):
  1494. continue
  1495. if not col.get('left_head_list') and not col.get('top_head_list'):
  1496. _d = {
  1497. 'value': col['text'],
  1498. 'value_row_index': col['text_row_index'],
  1499. 'value_col_index': col['text_col_index'],
  1500. 'value_sen_index': col['text_sen_index'],
  1501. 'sen_value': text[col['text_sen_index']:col['text_sen_index'] + len(col['text'])],
  1502. }
  1503. kv_dict_list.append(_d)
  1504. continue
  1505. if col.get('text_sen_index') and col.get('text_sen_index') >= len(text):
  1506. # print('continue1')
  1507. continue
  1508. if col.get('left_head_list'):
  1509. # head, head_row_index, head_col_index 按文本顺序排序
  1510. zip_list = list(
  1511. zip(col.get('left_head_list'), col.get('left_head_row_index'), col.get('left_head_col_index')))
  1512. zip_list.sort(key=lambda x: (x[1], x[2]))
  1513. col['left_head_list'], col['left_head_row_index'], col['left_head_col_index'] = zip(*zip_list)
  1514. last_head = ""
  1515. for h_index, head in enumerate(col.get('left_head_list')):
  1516. _d = {
  1517. 'key': head,
  1518. 'value': col['text'],
  1519. 'key_row_index': col['left_head_row_index'][h_index],
  1520. 'key_col_index': col['left_head_col_index'][h_index],
  1521. 'key_sen_index': col['left_head_sen_index'] + len(last_head),
  1522. 'value_row_index': col['text_row_index'],
  1523. 'value_col_index': col['text_col_index'],
  1524. 'value_sen_index': col['text_sen_index'],
  1525. 'sen_key': text[
  1526. col['left_head_sen_index'] + len(last_head):col['left_head_sen_index'] + len(
  1527. last_head) + len(head)],
  1528. 'sen_value': text[col['text_sen_index']:col['text_sen_index'] + len(col['text'])],
  1529. }
  1530. kv_dict_list.append(_d)
  1531. last_head += head
  1532. if col.get('top_head_list'):
  1533. # head, head_row_index, head_col_index 按文本顺序排序
  1534. zip_list = list(
  1535. zip(col.get('top_head_list'), col.get('top_head_row_index'), col.get('top_head_col_index')))
  1536. zip_list.sort(key=lambda x: (x[1], x[2]))
  1537. col['top_head_list'], col['top_head_row_index'], col['top_head_col_index'] = zip(*zip_list)
  1538. last_head = ""
  1539. for h_index, head in enumerate(col.get('top_head_list')):
  1540. _d = {
  1541. 'key': head,
  1542. 'value': col['text'],
  1543. 'key_row_index': col['top_head_row_index'][h_index],
  1544. 'key_col_index': col['top_head_col_index'][h_index],
  1545. 'key_sen_index': col['top_head_sen_index'] + len(last_head),
  1546. 'value_row_index': col['text_row_index'],
  1547. 'value_col_index': col['text_col_index'],
  1548. 'value_sen_index': col['text_sen_index'],
  1549. 'sen_key': text[col['top_head_sen_index'] + len(last_head):col['top_head_sen_index'] + len(
  1550. last_head) + len(head)],
  1551. 'sen_value': text[col['text_sen_index']:col['text_sen_index'] + len(col['text'])],
  1552. }
  1553. kv_dict_list.append(_d)
  1554. last_head += head
  1555. return kv_list, kv_dict_list
  1556. def removeFix(inner_table,fix_value="~~"):
  1557. height = len(inner_table)
  1558. width = len(inner_table[0])
  1559. for h in range(height):
  1560. for w in range(width):
  1561. if inner_table[h][w][0]==fix_value:
  1562. inner_table[h][w][0] = ""
  1563. def trunTable(tbody,in_attachment):
  1564. # print(tbody.find('tbody'))
  1565. # 附件中的表格,排除异常错乱的表格
  1566. if in_attachment:
  1567. if tbody.name=='table':
  1568. _tbody = tbody.find('tbody')
  1569. if _tbody is None:
  1570. _tbody = tbody
  1571. else:
  1572. _tbody = tbody
  1573. _td_len_list = []
  1574. for _tr in _tbody.find_all(recursive=False):
  1575. len_td = len(_tr.find_all(recursive=False))
  1576. _td_len_list.append(len_td)
  1577. if _td_len_list:
  1578. if len(list(set(_td_len_list))) >= 8 or max(_td_len_list) > 100:
  1579. string_list = [re.sub("\s+","",i)for i in tbody.strings if i and i!='\n']
  1580. tbody.string = ",".join(string_list)
  1581. table_max_len = 30000
  1582. tbody.string = tbody.string[:table_max_len]
  1583. tbody.name = "turntable"
  1584. if return_kv:
  1585. return None, None, None
  1586. return None
  1587. # fixSpan(tbody)
  1588. # inner_table = getTable(tbody)
  1589. # inner_table = fixTable(inner_table)
  1590. table2list = TableTag2List()
  1591. return_html_table = True if return_kv else False
  1592. if return_html_table:
  1593. inner_table, html_table = table2list.table2list(tbody, segment, return_html_table,return_kv=return_kv)
  1594. inner_table = fixTable(inner_table)
  1595. html_table = fixTable(html_table, "")
  1596. else:
  1597. inner_table = table2list.table2list(tbody, segment,return_kv=return_kv)
  1598. inner_table = fixTable(inner_table)
  1599. if inner_table == []:
  1600. string_list = [re.sub("\s+", "", i) for i in tbody.strings if i and i != '\n']
  1601. tbody.string = ",".join(string_list)
  1602. table_max_len = 30000
  1603. tbody.string = tbody.string[:table_max_len]
  1604. # log('异常表格直接取全文')
  1605. tbody.name = "turntable"
  1606. if return_kv:
  1607. return None, None, None
  1608. return None
  1609. if len(inner_table)>0 and len(inner_table[0])>0:
  1610. for tr in inner_table:
  1611. for td in tr:
  1612. if isinstance(td, str):
  1613. tbody.string = segment(tbody,final=False)
  1614. table_max_len = 30000
  1615. tbody.string = tbody.string[:table_max_len]
  1616. # log('异常表格,不做表格处理,直接取全文')
  1617. tbody.name = "turntable"
  1618. if return_kv:
  1619. return None, None, None
  1620. return None
  1621. #inner_table,head_list = setHead_withRule(inner_table,pat_head,pat_value,3)
  1622. #inner_table,head_list = setHead_inline(inner_table)
  1623. # inner_table, head_list = setHead_initem(inner_table,pat_head)
  1624. inner_table, head_list = set_head_model(inner_table)
  1625. # inner_table,head_list = setHead_incontext(inner_table,pat_head)
  1626. # print("table_head", inner_table)
  1627. # print("head_list", head_list)
  1628. # for begin in range(len(head_list[:-1])):
  1629. # for item in inner_table[head_list[begin]:head_list[begin+1]]:
  1630. # print(item)
  1631. # print("====")
  1632. removeFix(inner_table)
  1633. # print("----")
  1634. # print(head_list)
  1635. # for item in inner_table:
  1636. # print(item)
  1637. # print('inner_table111', inner_table)
  1638. if return_kv:
  1639. text1, table = get_table_text_kv(inner_table, head_list)
  1640. tbody.string = text1
  1641. kv_list, kv_dict_list = process_dict(text1, table)
  1642. # html放入dict
  1643. for kv_dict in kv_dict_list:
  1644. html = html_table[kv_dict.get('value_row_index')][kv_dict.get('value_col_index')]
  1645. kv_dict['value_html'] = html
  1646. else:
  1647. tbody.string = getTableText(inner_table,head_list)
  1648. table_max_len = 30000
  1649. tbody.string = tbody.string[:table_max_len]
  1650. if tbody.string.startswith('。'): # 20251224 修复 602074436 关键词与表格分开 中选单位:名称:
  1651. tbody.string = ',' + tbody.string[1:]
  1652. # print(tbody.string)
  1653. tbody.name = "turntable"
  1654. if return_kv:
  1655. return inner_table, kv_dict_list, text1
  1656. else:
  1657. return inner_table
  1658. if return_kv:
  1659. return None, None, None
  1660. return None
  1661. pat_head = re.compile('^(名称|序号|项目|标项|工程|品目[一二三四1234]|第[一二三四1234](标段|名|候选人|中标)|包段|标包|分包|包号|货物|单位|数量|价格|报价|金额|总价|单价|[招投中]标|候选|编号|得分|评委|评分|名次|排名|排序|科室|方式|工期|时间|产品|开始|结束|联系|日期|面积|姓名|证号|备注|级别|地[点址]|类型|代理|制造|企业资质|质量目标|工期目标|(需求|服务|项目|施工|采购|招租|出租|转让|出让|业主|询价|委托|权属|招标|竞得|抽取|承建)(人|方|单位)(名称)?|(供应商|供货商|服务商)(名称)?)$')
  1662. #pat_head = re.compile('(名称|序号|项目|工程|品目[一二三四1234]|第[一二三四1234](标段|候选人|中标)|包段|包号|货物|单位|数量|价格|报价|金额|总价|单价|[招投中]标|供应商|候选|编号|得分|评委|评分|名次|排名|排序|科室|方式|工期|时间|产品|开始|结束|联系|日期|面积|姓名|证号|备注|级别|地[点址]|类型|代理)')
  1663. pat_value = re.compile("(\d{2,}.\d{1}|\d+年\d+月|\d{8,}|\d{3,}-\d{6,}|有限[责任]*公司|^\d+$)")
  1664. list_innerTable = []
  1665. # 2022/2/9 删除干扰标签
  1666. for tag in soup.find_all('option'): #例子: 216661412
  1667. if 'selected' not in tag.attrs:
  1668. tag.extract()
  1669. for ul in soup.find_all('ul'): #例子 156439663 多个不同channel 类别的标题
  1670. if ul.find_all('li') == ul.findChildren(recursive=False) and len(set(re.findall(
  1671. '招标公告|中标结果公示|中标候选人公示|招标答疑|开标评标|合同履?约?公示|资格评审',
  1672. ul.get_text(), re.S)))>3:
  1673. ul.extract()
  1674. # tbodies = soup.find_all('table')
  1675. # 遍历表格中的每个tbody
  1676. tbodies = []
  1677. in_attachment = False
  1678. if soup.name=="table":
  1679. tbodies.append((soup,in_attachment))
  1680. for _part in soup.find_all():
  1681. if _part.name=='table':
  1682. tbodies.append((_part,in_attachment))
  1683. elif _part.name=='div':
  1684. if 'class' in _part.attrs and "richTextFetch" in _part['class']:
  1685. in_attachment = True
  1686. if return_kv and tbodies:
  1687. tbodies = tbodies[:1]
  1688. #逆序处理嵌套表格
  1689. # print('len(tbodies)1', len(tbodies))
  1690. # for tbody_index in range(1,len(tbodies)+1):
  1691. tbody_index = 1
  1692. while tbody_index < len(tbodies)+1:
  1693. tbody,_in_attachment = tbodies[len(tbodies)-tbody_index]
  1694. current_index = len(tbodies)-tbody_index
  1695. if current_index > 0 and tbodies[current_index - 1][0].find_next_sibling() == tbody and len(tbodies[current_index - 1][0].find_all('tr')) == 1 and len(tbody.find_all('tr'))>=1 and len(tbody.tr.find_all(['th', 'td'])) == len(
  1696. tbodies[current_index - 1][0].tr.find_all(['th', 'td'])): # 处理相邻表格都只有一行的情况;例:526321576 641881540
  1697. for row in tbody.find_all('tr'): # 相邻表格只有一行且列数一样合并表格,标签必须是兄弟节点 避免 665587084 这种不是相邻标签的错误合并
  1698. if tbodies[current_index - 1][0].tbody:
  1699. tbodies[current_index - 1][0].tbody.append(row)
  1700. else:
  1701. tbodies[current_index - 1][0].append(row)
  1702. inner_table = trunTable(tbodies[current_index - 1][0], _in_attachment)
  1703. if inner_table:
  1704. list_innerTable.append(inner_table)
  1705. tbody_index += 2
  1706. continue
  1707. inner_table = trunTable(tbody,_in_attachment)
  1708. list_innerTable.append(inner_table)
  1709. tbody_index += 1
  1710. # tbodies = soup.find_all('tbody')
  1711. # 遍历表格中的每个tbody
  1712. tbodies = []
  1713. in_attachment = False
  1714. if soup.name=="tbody" and soup.get_text().strip() != '': # 20250730 修复 642228216 表格分多个tbody且一三tbody只有tr内容为空导致提取失败
  1715. tbodies.append((soup,in_attachment))
  1716. for _part in soup.find_all():
  1717. if _part.name == 'tbody':
  1718. tbodies.append((_part, in_attachment))
  1719. elif _part.name == 'div':
  1720. if 'class' in _part.attrs and "richTextFetch" in _part['class']:
  1721. in_attachment = True
  1722. if return_kv and tbodies:
  1723. tbodies = tbodies[:1]
  1724. #逆序处理嵌套表格
  1725. tbody_index = 1
  1726. # for tbody_index in range(1,len(tbodies)+1):
  1727. while tbody_index < len(tbodies) + 1:
  1728. tbody,_in_attachment = tbodies[len(tbodies)-tbody_index]
  1729. current_index = len(tbodies) - tbody_index
  1730. if current_index > 0 and tbodies[current_index - 1][0].find_next_sibling() == tbody and len(
  1731. tbodies[current_index - 1][0].find_all('tr')) == 1 and len(tbody.find_all('tr'))>=1 and len(tbody.tr.find_all(['th', 'td'])) == len(
  1732. tbodies[current_index - 1][0].tr.find_all(['th', 'td'])) and len(tbody.tr.find_all(['th', 'td'])) > 2: # 处理相邻表格都只有一行的情况;例:526321576
  1733. for row in tbody.find_all('tr'):
  1734. if tbodies[current_index - 1][0].tbody:
  1735. tbodies[current_index - 1][0].tbody.append(row)
  1736. else:
  1737. tbodies[current_index - 1][0].append(row)
  1738. inner_table = trunTable(tbodies[current_index - 1][0], _in_attachment)
  1739. if inner_table:
  1740. list_innerTable.append(inner_table)
  1741. tbody_index += 2
  1742. continue
  1743. inner_table = trunTable(tbody,_in_attachment)
  1744. list_innerTable.append(inner_table)
  1745. tbody_index += 1
  1746. if return_kv:
  1747. kv_list = []
  1748. for x in list_innerTable:
  1749. if x[1] is not None:
  1750. kv_list.extend(x[1])
  1751. text = ""
  1752. for x in list_innerTable:
  1753. if x[2] is not None:
  1754. text += x[2]
  1755. return soup, kv_list, text
  1756. return soup
  1757. # return list_innerTable
  1758. def table_head_repair_process(_inner_table, docid=None, show=0, show_row_index=0):
  1759. def pre_process(inner_table):
  1760. """
  1761. 修复前的预处理
  1762. """
  1763. # 循环处理单元格,一次获取需要的
  1764. for i in range(len(inner_table)):
  1765. for j in range(len(inner_table[i])):
  1766. # 删除前后逗号
  1767. inner_table[i][j][0] = re.sub('^[,, ]+', '', inner_table[i][j][0])
  1768. inner_table[i][j][0] = re.sub('[,, ]+$', '', inner_table[i][j][0])
  1769. inner_table[i][j][0] = re.sub('[, ]+', '', inner_table[i][j][0])
  1770. return inner_table
  1771. def repair_by_colon(inner_table):
  1772. """
  1773. 根据冒号修复当前格子的表头值
  1774. """
  1775. # 修复冒号在文本中间的,不能作为表头;(冒号后面需多个字)
  1776. # 冒号在括号中的除外
  1777. # 冒号在最后的,判断后一个格子是否有重复的文字
  1778. for i in range(len(inner_table)):
  1779. for j in range(len(inner_table[i])):
  1780. _text = inner_table[i][j][0]
  1781. if len(_text) >= 3 and inner_table[i][j][1] == 1:
  1782. match = re.search('[::]', _text)
  1783. if match:
  1784. start_index, end_index = match.span()
  1785. if start_index == 0:
  1786. continue
  1787. if end_index == len(_text):
  1788. if len(inner_table[i]) == 2 and j <= len(inner_table[i]) - 2 and inner_table[i][j+1][0] and (_text in inner_table[i][j+1][0] or inner_table[i][j+1][0] in _text):
  1789. inner_table[i][j][1] = 0
  1790. inner_table[i][j+1][1] = 0
  1791. else:
  1792. continue
  1793. if re.search('[((]', _text[:start_index]) and re.search('[))]', _text[end_index:]):
  1794. continue
  1795. m1 = re.search('[\u4e00-\u9fa50-9a-zA-Z]', _text[:start_index])
  1796. m2 = re.search('[\u4e00-\u9fa50-9a-zA-Z]', _text[end_index:])
  1797. if m1 and m2 and (len(m2.group()) >= 2 or m2.group() in ['是', '否']):
  1798. inner_table[i][j][1] = 0
  1799. return inner_table
  1800. def repair_by_duplicate(inner_table):
  1801. """
  1802. 根据列重复修复当前格子的表头值
  1803. """
  1804. # 统计每个值的表头情况
  1805. col_head_dict = {}
  1806. for i in range(len(inner_table)):
  1807. for j in range(len(inner_table[i])):
  1808. col = inner_table[i][j]
  1809. if col[0] in col_head_dict.keys():
  1810. col_head_dict[col[0]] += [col[1]]
  1811. else:
  1812. col_head_dict[col[0]] = [col[1]]
  1813. # 多个重复列的预测值不同,以第一个为准
  1814. for i in range(len(inner_table)):
  1815. col = inner_table[i][0]
  1816. key = col[0] + '\t' + str(0)
  1817. dup_dict = {}
  1818. for j in range(len(inner_table[i])):
  1819. if inner_table[i][j][0] == col[0]:
  1820. if key in dup_dict.keys():
  1821. dup_dict[key] += [j]
  1822. else:
  1823. dup_dict[key] = [j]
  1824. # if inner_table[i][j][1] != col[1]:
  1825. # if col != inner_table[i][0]:
  1826. # inner_table[i][j][1] = col[1]
  1827. # else:
  1828. # inner_table[i][0][1] = inner_table[i][j][1]
  1829. # col = inner_table[i][0]
  1830. else:
  1831. col = inner_table[i][j]
  1832. key = col[0] + '\t' + str(j)
  1833. dup_dict[key] = [j]
  1834. # print('dup_dict', dup_dict)
  1835. #
  1836. for key in dup_dict.keys():
  1837. index_list = dup_dict.get(key)
  1838. if len(index_list) <= 1:
  1839. continue
  1840. # 需要表头不同
  1841. table_head_list = []
  1842. for index in index_list:
  1843. table_head_list.append(inner_table[i][index][1])
  1844. table_head_list = list(set(table_head_list))
  1845. if len(table_head_list) <= 1:
  1846. continue
  1847. # 若是职业特殊处理
  1848. col = key.split('\t')[0]
  1849. if re.search('([^人]员|工程师|建造师|经理|安全负责人|技术负责人|合同商务负责人)$', col):
  1850. table_head_flag = 0
  1851. # 看是否包含表头
  1852. else:
  1853. table_head_flag = 0
  1854. for index in index_list:
  1855. if inner_table[i][index][1] == 1:
  1856. table_head_flag = 1
  1857. break
  1858. table_head = None
  1859. if index_list[0] > 0 and index_list[-1] == len(inner_table[i]) - 1:
  1860. table_head = inner_table[i][index_list[0]][1]
  1861. elif index_list[0] > 0 and index_list[-1] != len(inner_table[i]) - 1:
  1862. # 查看前面是否有表头-非表头表达
  1863. is_head_not_head = 0
  1864. for k in range(index_list[0]):
  1865. if k+1 < index_list[0] and inner_table[i][k][1] == 1 and inner_table[i][k+1][1] == 0:
  1866. is_head_not_head = 1
  1867. break
  1868. # 查看前后有没有表头
  1869. start_has_head = 0
  1870. end_has_head = 0
  1871. if not is_head_not_head:
  1872. for k in range(index_list[0], len(inner_table[i])):
  1873. if inner_table[i][k][0] == inner_table[i][index_list[0]][0]:
  1874. continue
  1875. if inner_table[i][k][1] == 1:
  1876. end_has_head = 1
  1877. break
  1878. for k in range(index_list[0]):
  1879. if inner_table[i][k][0] == inner_table[i][index_list[0]][0]:
  1880. continue
  1881. if inner_table[i][k][1] == 1:
  1882. start_has_head = 1
  1883. break
  1884. head_list = col_head_dict.get(inner_table[i][index_list[0]][0])
  1885. if is_head_not_head:
  1886. table_head = table_head_flag
  1887. elif len(head_list) >= 4:
  1888. if head_list.count(0) > head_list.count(1):
  1889. table_head = 0
  1890. else:
  1891. table_head = 1
  1892. elif not start_has_head and not end_has_head:
  1893. table_head = 0
  1894. else:
  1895. table_head = table_head_flag
  1896. elif index_list[0] == 0 and index_list[-1] == len(inner_table[i]) - 1:
  1897. table_head = table_head_flag
  1898. elif index_list[0] == 0 and index_list[-1] != len(inner_table[i]) - 1:
  1899. head_list = col_head_dict.get(inner_table[i][index_list[0]][0])
  1900. if len(head_list) >= 4:
  1901. if head_list.count(0) > head_list.count(1):
  1902. table_head = 0
  1903. else:
  1904. table_head = 1
  1905. else:
  1906. table_head = table_head_flag
  1907. if table_head is not None:
  1908. for index in index_list:
  1909. inner_table[i][index][1] = table_head
  1910. return inner_table
  1911. def repair_by_around(inner_table):
  1912. """
  1913. 根据周围的表头值修复当前格子的表头值
  1914. """
  1915. one_head_index_list = []
  1916. zero_head_index_list = []
  1917. all_head_index_list = []
  1918. one_not_head_index_list = []
  1919. no_dup_index_cnt_dict = {}
  1920. for i in range(len(inner_table)):
  1921. head_cnt = 0
  1922. head_index = None
  1923. head_dict = {}
  1924. for j in range(len(inner_table[i])):
  1925. # 统计表头数
  1926. if inner_table[i][j][1] == 1:
  1927. head_cnt += 1
  1928. head_index = j
  1929. if inner_table[i][j][0] not in ['~~', '', ' ']:
  1930. if inner_table[i][j][0] in head_dict.keys():
  1931. head_dict[inner_table[i][j][0]] += 1
  1932. else:
  1933. head_dict[inner_table[i][j][0]] = 1
  1934. no_dup_index_cnt_dict[i] = len(head_dict.keys())
  1935. # 表头数list
  1936. if head_cnt == 0:
  1937. zero_head_index_list.append(i)
  1938. elif head_cnt == 1:
  1939. # 这个单个表头需满足前面有非表头
  1940. find_flag = 0
  1941. for k in range(head_index):
  1942. if inner_table[i][k][1] == 0:
  1943. find_flag = 1
  1944. if find_flag and len(head_dict.keys()) > 2:
  1945. one_head_index_list.append(i)
  1946. elif head_cnt == len(inner_table[i]):
  1947. all_head_index_list.append(i)
  1948. elif head_cnt == len(inner_table[i]) - 1:
  1949. one_not_head_index_list.append(i)
  1950. # 第一行为表头,但有一个不为表头,下面行都非表头,表格行数小于4
  1951. if 0 in one_not_head_index_list and 1 in zero_head_index_list and len(inner_table) <= 4:
  1952. # 不相等的列值大于4
  1953. diff_col1 = []
  1954. for col in inner_table[0]:
  1955. if col[0] not in diff_col1 and len(col[0]) >= 1:
  1956. diff_col1.append(col[0])
  1957. diff_col2 = []
  1958. for col in inner_table[1]:
  1959. if col[0] not in diff_col2 and len(col[0]) >= 1:
  1960. diff_col2.append(col[0])
  1961. if len(diff_col1) >= 4 and len(diff_col2) >= 4:
  1962. for j in range(len(inner_table[0])):
  1963. inner_table[0][j][1] = 1
  1964. one_not_head_index_list.remove(0)
  1965. all_head_index_list.append(0)
  1966. # 一行很多列且都为表头,则剩下一个也为表头
  1967. for i in range(len(inner_table)):
  1968. if no_dup_index_cnt_dict.get(i) >= 5 and i in one_not_head_index_list:
  1969. for j in range(len(inner_table[i])):
  1970. inner_table[i][j][1] = 1
  1971. # 一行很多列且都不为表头,则剩下一个也不为表头,除了第一个
  1972. for i in range(len(inner_table)):
  1973. if no_dup_index_cnt_dict.get(i) >= 5 and i in one_head_index_list \
  1974. and inner_table[i][0][1] != 1 and inner_table[i][0][0] != '':
  1975. for j in range(len(inner_table[i])):
  1976. inner_table[i][j][1] = 0
  1977. # 一整个大表格,第一行为表头,下面行中有个别格子被识别为表头
  1978. # 候选人后面修复
  1979. for index in one_head_index_list:
  1980. if (index - 1 in zero_head_index_list and index - 2 in zero_head_index_list) \
  1981. or (index - 1 in zero_head_index_list and index - 2 in all_head_index_list) \
  1982. or (index - 1 in all_head_index_list):
  1983. for j in range(len(inner_table[index])):
  1984. inner_table[index][j][1] = 0
  1985. zero_head_index_list.append(index)
  1986. return inner_table
  1987. def repair_by_tenderer(inner_table):
  1988. """
  1989. 根据第一第二第三候选人修复当前格子的表头值
  1990. """
  1991. # 修复第一第二第三中标候选人作为表头
  1992. first_tenderer = ['第一中标候选人', '第一中标人', '第一中标(成交)人', '第一候选人']
  1993. second_tenderer = ['第二中标候选人', '第二中标(成交)候选人', '第二候选人']
  1994. third_tenderer = ['第三中标候选人', '第三中标(成交)候选人', '第三候选人']
  1995. # n1 next one, n2 next two, l1 last one, l2 last two
  1996. for i in range(len(inner_table)):
  1997. row = inner_table[i]
  1998. n1_row, n2_row = None, None
  1999. if i+1 < len(inner_table):
  2000. n1_row = inner_table[i+1]
  2001. if i+2 < len(inner_table):
  2002. n2_row = inner_table[i+2]
  2003. for j in range(len(row)):
  2004. row_col = row[j]
  2005. n1_row_col, n2_row_col = None, None
  2006. row_n1_col, row_n2_col = None, None
  2007. n1_row_n1_col, n2_row_n1_col, n1_row_n2_col = None, None, None
  2008. if n1_row:
  2009. n1_row_col = n1_row[j]
  2010. if n2_row:
  2011. n2_row_col = n2_row[j]
  2012. if j+1 < len(row):
  2013. row_n1_col = row[j+1]
  2014. if j+2 < len(row):
  2015. row_n2_col = row[j+2]
  2016. if n1_row and j+1 < len(n1_row):
  2017. n1_row_n1_col = n1_row[j+1]
  2018. if n2_row and j+1 < len(n2_row):
  2019. n2_row_n1_col = n2_row[j+1]
  2020. if n1_row and j+2 < len(n1_row):
  2021. n1_row_n2_col = n1_row[j+2]
  2022. # 连续作为行表头
  2023. if row_col[0] in first_tenderer and row_n1_col and row_n1_col[1] == 0:
  2024. if n1_row_col and n1_row_col[0] in second_tenderer and n1_row_n1_col and n1_row_n1_col[1] == 0:
  2025. inner_table[i][j][1] = 1
  2026. inner_table[i+1][j][1] = 1
  2027. if n2_row_col and n2_row_col[0] in third_tenderer and n2_row_n1_col and n2_row_n1_col[1] == 0:
  2028. inner_table[i+2][j][1] = 1
  2029. # 连续作为列表头
  2030. if row_col[0] in first_tenderer and n1_row_col and n1_row_col[1] == 0:
  2031. if row_n1_col and row_n1_col[0] in second_tenderer and n1_row_n1_col and n1_row_n1_col[1] == 0:
  2032. inner_table[i][j][1] = 1
  2033. inner_table[i][j+1][1] = 1
  2034. if row_n2_col and row_n2_col[0] in third_tenderer and n1_row_n2_col and n1_row_n2_col[1] == 0:
  2035. inner_table[i][j+2][1] = 1
  2036. return inner_table
  2037. def repair_by_keywords(inner_table):
  2038. """
  2039. 根据关键词修复当前格子的表头值
  2040. """
  2041. # 修复表头关键词未作为表头
  2042. # 末尾匹配匹配关键词且字数小于7,直接作为表头
  2043. head_keyword = ['供应商', '总价', '总价(元)', '总价\(元\)', '品目一', '品目二', '品目三']
  2044. # 末尾匹配关键词且前一列为表头且与前一列文本不同,直接不做表头
  2045. head_keyword2 = ['管理中心', '有限公司', '项目采购', '确定。', ]
  2046. # 开头匹配关键词,直接不做表头
  2047. head_keyword3 = ['详见', '选定', '咨询服务', '标准物资', '电汇', '承兑', '低档', '高档',
  2048. '更换配置', '各种数据']
  2049. # 文本匹配关键词且前一列为表头,直接作为表头
  2050. head_keyword4 = ['综合排名', '工期(交货期)', '检测批', '检测范围', '混凝土设计强检测批的容度等级',
  2051. '量(个)']
  2052. # 文本在关键词中,直接不做表头
  2053. head_keyword5 = ['殡葬用地', '电脑包', '电池']
  2054. # 文本匹配关键词,直接不作表头
  2055. head_keyword6 = ['市场行情', '有限公司', '能提供']
  2056. # 末尾匹配关键词,直接不做表头
  2057. head_keyword7 = ['基金', '结转', '结余', '税', '结余分配', '协议供货', '房屋',
  2058. '纳税人', '自然人', '计算所得额']
  2059. # 文本匹配关键词且整行都是表头,直接做表头
  2060. head_keyword8 = ['备注']
  2061. # n1 next one, n2 next two, l1 last one, l2 last two
  2062. for i in range(len(inner_table)):
  2063. row = inner_table[i]
  2064. for j in range(len(row)):
  2065. row_col = row[j]
  2066. row_l1_col = None
  2067. if j-1 >= 0:
  2068. row_l1_col = row[j-1]
  2069. for key in head_keyword:
  2070. match = re.search(key+'$', row_col[0])
  2071. if match and len(inner_table[i][j][0]) <= 6:
  2072. if show:
  2073. print('match head_keyword')
  2074. inner_table[i][j][1] = 1
  2075. for key in head_keyword2:
  2076. match = re.search(key+'$', row_col[0])
  2077. if j > 0 and row_l1_col and row_l1_col[1] == 1 and row_l1_col[0] != row_col[0] and match and row_col[1] == 1:
  2078. if show:
  2079. print('match head_keyword2')
  2080. inner_table[i][j][1] = 0
  2081. for key in head_keyword3:
  2082. match = re.search('^'+key, row_col[0])
  2083. if match and row_col[1] == 1:
  2084. if show:
  2085. print('match head_keyword3')
  2086. inner_table[i][j][1] = 0
  2087. for key in head_keyword4:
  2088. match = re.search(key, row_col[0])
  2089. if j > 0 and row_l1_col and row_l1_col[1] == 1 and match and row_col[1] == 0:
  2090. if show:
  2091. print('match head_keyword4')
  2092. inner_table[i][j][1] = 1
  2093. if row_col[0] in head_keyword5:
  2094. if show:
  2095. print('match head_keyword5')
  2096. inner_table[i][j][1] = 0
  2097. for key in head_keyword6:
  2098. match = re.search(key, row_col[0])
  2099. if match:
  2100. if show:
  2101. print('match head_keyword6')
  2102. inner_table[i][j][1] = 0
  2103. for key in head_keyword7:
  2104. match = re.search(key+'$', row_col[0])
  2105. if match and row_col[1] == 1:
  2106. if show:
  2107. print('match head_keyword7')
  2108. inner_table[i][j][1] = 0
  2109. if row_col[0] in head_keyword8 and row_col[1] == 0:
  2110. if show:
  2111. print('match head_keyword8')
  2112. all_head_flag = 1
  2113. for k in range(len(row)):
  2114. if row[k][0] in ['', row_col[0]]:
  2115. continue
  2116. if row[k][1] == 0:
  2117. print('row[k]', row[k])
  2118. all_head_flag = 0
  2119. break
  2120. # print('all_head_flag', all_head_flag)
  2121. if all_head_flag:
  2122. inner_table[i][j][1] = 1
  2123. return inner_table
  2124. def repair_by_length(inner_table):
  2125. for i in range(len(inner_table)):
  2126. for j in range(len(inner_table[i])):
  2127. if len(inner_table[i][j][0]) >= 30:
  2128. inner_table[i][j][1] = 0
  2129. return inner_table
  2130. def repair_by_summation(inner_table):
  2131. # 修复合计在中间的特殊情况
  2132. if len(inner_table) >= 3 and len(inner_table[1]) == 2 \
  2133. and inner_table[1][0][0] == '合计' and inner_table[1][1][0].endswith('%'):
  2134. inner_table[1][0][1] = 0
  2135. inner_table[1][1][1] = 0
  2136. return inner_table
  2137. def repair_by_rank(inner_table):
  2138. if not inner_table or (inner_table and len(inner_table[0]) < 3):
  2139. return inner_table
  2140. for i in range(len(inner_table)):
  2141. for j in range(len(inner_table[i])-2):
  2142. if inner_table[i][j][0] in ['第一名'] and inner_table[i][j+1][0] in ['第二名'] and inner_table[i][j+2][0] in ['第三名']:
  2143. inner_table[i][j][1] = 1
  2144. inner_table[i][j+1][1] = 1
  2145. inner_table[i][j+2][1] = 1
  2146. return inner_table
  2147. _inner_table = pre_process(_inner_table)
  2148. compare_inner_table = copy.deepcopy(_inner_table)
  2149. if show:
  2150. print('table_head_repair_process1', show_row_index, _inner_table[show_row_index])
  2151. _inner_table = repair_by_rank(_inner_table)
  2152. if _inner_table != compare_inner_table:
  2153. compare_inner_table = copy.deepcopy(_inner_table)
  2154. log('table_head repair1.5 ' + str(docid))
  2155. if show:
  2156. print('table_head_repair_process1.5', show_row_index, _inner_table[show_row_index])
  2157. _inner_table = repair_by_colon(_inner_table)
  2158. if _inner_table != compare_inner_table:
  2159. compare_inner_table = copy.deepcopy(_inner_table)
  2160. log('table_head repair2 ' + str(docid))
  2161. if show:
  2162. print('table_head_repair_process2', show_row_index, _inner_table[show_row_index])
  2163. _inner_table = repair_by_keywords(_inner_table)
  2164. if _inner_table != compare_inner_table:
  2165. compare_inner_table = copy.deepcopy(_inner_table)
  2166. log('table_head repair3 ' + str(docid))
  2167. if show:
  2168. print('table_head_repair_process3', show_row_index, _inner_table[show_row_index])
  2169. _inner_table = repair_by_tenderer(_inner_table)
  2170. if _inner_table != compare_inner_table:
  2171. compare_inner_table = copy.deepcopy(_inner_table)
  2172. log('table_head repair4 ' + str(docid))
  2173. if show:
  2174. print('table_head_repair_process4', show_row_index, _inner_table[show_row_index])
  2175. _inner_table = repair_by_duplicate(_inner_table)
  2176. if _inner_table != compare_inner_table:
  2177. compare_inner_table = copy.deepcopy(_inner_table)
  2178. log('table_head repair5 ' + str(docid))
  2179. if show:
  2180. print('table_head_repair_process5', show_row_index, _inner_table[show_row_index])
  2181. _inner_table = repair_by_around(_inner_table)
  2182. if _inner_table != compare_inner_table:
  2183. compare_inner_table = copy.deepcopy(_inner_table)
  2184. log('table_head repair6 ' + str(docid))
  2185. if show:
  2186. print('table_head_repair_process6', show_row_index, _inner_table[show_row_index])
  2187. _inner_table = repair_by_tenderer(_inner_table)
  2188. if _inner_table != compare_inner_table:
  2189. compare_inner_table = copy.deepcopy(_inner_table)
  2190. log('table_head repair7 ' + str(docid))
  2191. if show:
  2192. print('table_head_repair_process7', show_row_index, _inner_table[show_row_index])
  2193. _inner_table = repair_by_keywords(_inner_table)
  2194. if _inner_table != compare_inner_table:
  2195. compare_inner_table = copy.deepcopy(_inner_table)
  2196. log('table_head repair8 ' + str(docid))
  2197. if show:
  2198. print('table_head_repair_process8', show_row_index, _inner_table[show_row_index])
  2199. _inner_table = repair_by_length(_inner_table)
  2200. if _inner_table != compare_inner_table:
  2201. compare_inner_table = copy.deepcopy(_inner_table)
  2202. print('table_head repair9 ' + str(docid))
  2203. if show:
  2204. print('table_head_repair_process9', show_row_index, _inner_table[show_row_index])
  2205. _inner_table = repair_by_summation(_inner_table)
  2206. if show:
  2207. print('table_head_repair_process10', show_row_index, _inner_table[show_row_index])
  2208. return _inner_table