html_cleaner.py 21 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409
  1. # -*- coding: utf-8 -*-
  2. """HTML 清洗、异常字符处理、附件前处理。
  3. 按 ARCHITECTURE.md Phase 4 拆分建议,从 ``interface/Preprocessing.py`` 迁出。
  4. 类型:PREPROCESS。
  5. 原位置:``interface/Preprocessing.py`` 中以下函数:
  6. - ``special_treatment`` — 特殊数据源 HTML 预处理
  7. - ``article_limit`` — 正文/附件字数限制
  8. - ``attachment_filelink`` — 附件文件链接处理
  9. - ``del_achievement`` — 删除业绩内容
  10. - ``split_header`` — 空格分割多表头处理
  11. ``interface/Preprocessing.py`` 仍 re-export 以上全部名称,老 import 不受影响。
  12. """
  13. from __future__ import absolute_import
  14. import re
  15. from BiddingKG.dl.common.logging import log
  16. __all__ = [
  17. "special_treatment",
  18. "article_limit",
  19. "attachment_filelink",
  20. "del_achievement",
  21. "split_header",
  22. ]
  23. def special_treatment(sourceContent, web_source_no):
  24. try:
  25. if web_source_no == 'DX000202-1':
  26. ser = re.search('中标供应商及中标金额:【(([\w()]{5,20}-[\d,.]+,)+)】', sourceContent)
  27. if ser:
  28. new = ""
  29. l = ser.group(1).split(',')
  30. for i in range(len(l)):
  31. it = l[i]
  32. if '-' in it:
  33. role, money = it.split('-')
  34. new += '标段%d, 中标供应商: ' % (i + 1) + role + ',中标金额:' + money + '。'
  35. sourceContent = sourceContent.replace(ser.group(0), new, 1)
  36. elif web_source_no == '00753-14':
  37. body = sourceContent.find("body")
  38. body_child = body.find_all(recursive=False)
  39. pcontent = body
  40. if 'id' in body_child[0].attrs:
  41. if len(body_child) <= 2 and body_child[0]['id'] == 'pcontent':
  42. pcontent = body_child[0]
  43. # pcontent = sourceContent.find("div", id="pcontent")
  44. pcontent = pcontent.find_all(recursive=False)[0]
  45. first_table = None
  46. for idx in range(len(pcontent.find_all(recursive=False))):
  47. t_part = pcontent.find_all(recursive=False)[idx]
  48. if t_part.name != "table":
  49. break
  50. if idx == 0:
  51. first_table = t_part
  52. else:
  53. for _tr in t_part.find("tbody").find_all(recursive=False):
  54. first_table.find("tbody").append(_tr)
  55. t_part.clear()
  56. elif web_source_no == 'DX008357-11':
  57. body = sourceContent.find("body")
  58. body_child = body.find_all(recursive=False)
  59. pcontent = body
  60. if 'id' in body_child[0].attrs:
  61. if len(body_child) <= 2 and body_child[0]['id'] == 'pcontent':
  62. pcontent = body_child[0]
  63. # pcontent = sourceContent.find("div", id="pcontent")
  64. pcontent = pcontent.find_all(recursive=False)[0]
  65. error_table = []
  66. is_error_table = False
  67. for part in pcontent.find_all(recursive=False):
  68. if is_error_table:
  69. if part.name == "table":
  70. error_table.append(part)
  71. else:
  72. break
  73. if part.name == "div" and part.get_text(strip=True) == "中标候选单位:":
  74. is_error_table = True
  75. first_table = None
  76. for idx in range(len(error_table)):
  77. t_part = error_table[idx]
  78. # if t_part.name != "table":
  79. # break
  80. if idx == 0:
  81. for _tr in t_part.find("tbody").find_all(recursive=False):
  82. if _tr.get_text(strip=True) == "":
  83. _tr.decompose()
  84. first_table = t_part
  85. else:
  86. for _tr in t_part.find("tbody").find_all(recursive=False):
  87. if _tr.get_text(strip=True) != "":
  88. first_table.find("tbody").append(_tr)
  89. t_part.clear()
  90. elif web_source_no == '18021-2':
  91. body = sourceContent.find("body")
  92. body_child = body.find_all(recursive=False)
  93. pcontent = body
  94. if 'id' in body_child[0].attrs:
  95. if len(body_child) <= 2 and body_child[0]['id'] == 'pcontent':
  96. pcontent = body_child[0]
  97. # pcontent = sourceContent.find("div", id="pcontent")
  98. td = pcontent.find_all("td")
  99. for _td in td:
  100. if str(_td.string).strip() == "报价金额":
  101. _td.string = "单价"
  102. elif web_source_no == '13740-2':
  103. # “xxx成为成交供应商”
  104. re_match = re.search("[^,。]+成为[^,。]*成交供应商", sourceContent)
  105. if re_match:
  106. sourceContent = sourceContent.replace(re_match.group(), "成交人:" + re_match.group())
  107. elif web_source_no == '03786-10':
  108. ser1 = re.search('中标价:([\d,.]+)', sourceContent)
  109. ser2 = re.search('合同金额[((]万元[))]:([\d,.]+)', sourceContent)
  110. if ser1 and ser2:
  111. m1 = ser1.group(1).replace(',', '')
  112. m2 = ser2.group(1).replace(',', '')
  113. if float(m1) < 100000 and (m1.split('.')[0] == m2.split('.')[0] or m2 == '0'):
  114. new = '中标价(万元):' + m1
  115. sourceContent = sourceContent.replace(ser1.group(0), new, 1)
  116. elif web_source_no=='00076-4':
  117. ser = re.search('主要标的数量:([0-9一]+)\w{,3},主要标的单价:([\d,.]+)元?,合同金额:(.00),', sourceContent)
  118. if ser:
  119. num = ser.group(1).replace('一', '1')
  120. try:
  121. num = 1 if num == '0' else num
  122. unit_price = ser.group(2).replace(',', '')
  123. total_price = str(int(num) * float(unit_price))
  124. new = '合同金额:' + total_price
  125. sourceContent = sourceContent.replace('合同金额:.00', new, 1)
  126. except Exception as e:
  127. log('preprocessing.py special_treatment exception')
  128. elif web_source_no=='DX000105-2':
  129. if re.search("成交公示", sourceContent) and re.search(',投标人:', sourceContent) and re.search(',成交人:', sourceContent)==None:
  130. sourceContent = sourceContent.replace(',投标人:', ',成交人:')
  131. elif web_source_no in ['03795-1', '03795-2']:
  132. if re.search('中标单位如下', sourceContent) and re.search(',投标人:', sourceContent) and re.search(',中标人:', sourceContent)==None:
  133. sourceContent = sourceContent.replace(',投标人:', ',中标人:')
  134. elif web_source_no in ['04080-3', '04080-4']:
  135. ser = re.search('合同金额:([0-9,]+.[0-9]{3,})(.{,4})', sourceContent)
  136. if ser and '万' not in ser.group(2):
  137. sourceContent = sourceContent.replace('合同金额:', '合同金额(万元):')
  138. elif web_source_no=='03761-3':
  139. ser = re.search('中标价,([0-9]+)[.0-9]*%', sourceContent)
  140. if ser and int(ser.group(1))>100:
  141. sourceContent = sourceContent.replace(ser.group(0), ser.group(0)[:-1]+'元')
  142. elif web_source_no=='00695-7':
  143. ser = re.search('支付金额:', sourceContent)
  144. if ser:
  145. sourceContent = sourceContent.replace('支付金额:', '合同金额:')
  146. elif web_source_no=='00811-8':
  147. if re.search('是否中标:是', sourceContent) and re.search('排名:\d,', sourceContent):
  148. sourceContent = re.sub('排名:\d,', '候选', sourceContent)
  149. elif web_source_no=='DX000726-6':
  150. sourceContent = re.sub('卖方[::\s]+宝山钢铁股份有限公司', '招标单位:宝山钢铁股份有限公司', sourceContent)
  151. elif web_source_no=='DX008791-1':
  152. sourceContent = re.sub('收货单位:', '最终用户:', sourceContent)
  153. elif web_source_no=='DX011971':
  154. sourceContent = re.sub('公司主体:', '业主单位:', sourceContent)
  155. return sourceContent
  156. except Exception as e:
  157. log('特殊数据源: %s 预处理特别修改抛出异常: %s'%(web_source_no, e))
  158. return sourceContent
  159. def article_limit(soup,limit_words=30000):
  160. sub_space = re.compile("\s+")
  161. def soup_limit(_soup,_count,_recursion_depth,max_count=30000,max_gap=500,max_recursion_depth=900):
  162. """
  163. :param _soup: soup
  164. :param _count: 当前字数
  165. :param max_count: 字数最大限制
  166. :param max_gap: 超过限制后的最大误差
  167. :return:
  168. """
  169. _gap = _count - max_count
  170. _is_skip = False
  171. next_soup = None
  172. # 跳过层级结构为1的标签,向下取值
  173. # while len(_soup.find_all(recursive=False)) == 1 and \
  174. # _soup.get_text(strip=True) == _soup.find_all(recursive=False)[0].get_text(strip=True):
  175. # _soup = _soup.find_all(recursive=False)[0]
  176. # _recursion_depth += 1
  177. while len(_soup.find_all(recursive=False)) == 1:
  178. _recursion_depth += 1
  179. if _soup.get_text(strip=True) == _soup.find_all(recursive=False)[0].get_text(strip=True):
  180. if _recursion_depth > max_recursion_depth:
  181. _soup.string = str(_soup.get_text())[:max_count - _count]
  182. next_soup = None
  183. return _count, _recursion_depth, _gap, next_soup
  184. else:
  185. _soup = _soup.find_all(recursive=False)[0]
  186. else:
  187. _count += len(_soup.get_text(strip=True)) - len(_soup.find_all(recursive=False)[0].get_text(strip=True))
  188. if _count >= max_count or _recursion_depth > max_recursion_depth:
  189. _is_skip = True
  190. next_soup = None
  191. _count -= len(_soup.get_text(strip=True)) - len(_soup.find_all(recursive=False)[0].get_text(strip=True))
  192. _soup.string = str(_soup.get_text())[:max_count - _count]
  193. return _count, _recursion_depth, _gap, next_soup
  194. else:
  195. _soup = _soup.find_all(recursive=False)[0]
  196. # 无结构的纯文本直接取值
  197. if len(_soup.find_all(recursive=False)) == 0:
  198. _soup.string = str(_soup.get_text())[:max_count-_count]
  199. _count += len(re.sub(sub_space, "", _soup.string))
  200. _gap = _count - max_count
  201. next_soup = None
  202. else:
  203. _recursion_depth += 1
  204. for _soup_part in _soup.find_all(recursive=False):
  205. if not _is_skip:
  206. _count += len(re.sub(sub_space, "", _soup_part.get_text()))
  207. if _count >= max_count:
  208. _gap = _count - max_count
  209. if _gap <= max_gap:
  210. _is_skip = True
  211. else:
  212. _is_skip = True
  213. if _recursion_depth <= max_recursion_depth:
  214. next_soup = _soup_part
  215. _count -= len(re.sub(sub_space, "", _soup_part.get_text()))
  216. else: # 超出最大递归层级时,直接切片取值
  217. next_soup = None
  218. _count -= len(re.sub(sub_space, "", _soup_part.get_text()))
  219. _soup_part.string = str(_soup_part.get_text())[:max_count - _count]
  220. continue
  221. else:
  222. _soup_part.decompose()
  223. return _count,_recursion_depth,_gap,next_soup
  224. text_count = 0
  225. max_recursion_depth = 900 # 最大递归
  226. recursion_depth = 0
  227. have_attachment = False
  228. attachment_part = None
  229. for child in soup.find_all(recursive=True):
  230. if child.name == 'div' and 'class' in child.attrs:
  231. if "richTextFetch" in child['class']:
  232. child.insert_before("##attachment##。") # 句号分开,避免项目名称等提取
  233. attachment_part = child
  234. have_attachment = True
  235. break
  236. if not have_attachment:
  237. # 无附件,通过get_text()方法与limit_words大小判断是否要限制字数
  238. if len(re.sub(sub_space, "", soup.get_text())) > limit_words:
  239. text_count,recursion_depth,gap,n_soup = soup_limit(soup,text_count,recursion_depth,max_count=limit_words,max_gap=1000,max_recursion_depth=max_recursion_depth)
  240. while n_soup:
  241. text_count,recursion_depth, gap, n_soup = soup_limit(n_soup, text_count,recursion_depth, max_count=limit_words, max_gap=1000,max_recursion_depth=max_recursion_depth)
  242. else:
  243. # 有附件
  244. _text = re.sub(sub_space, "", soup.get_text())
  245. _text_split = _text.split("##attachment##")
  246. # 正文部分
  247. if len(_text_split[0])>limit_words:
  248. main_soup = attachment_part.parent
  249. main_text = main_soup.find_all(recursive=False)[0]
  250. text_count,recursion_depth, gap, n_soup = soup_limit(main_text, text_count,recursion_depth, max_count=limit_words, max_gap=1000,max_recursion_depth=max_recursion_depth)
  251. while n_soup:
  252. text_count,recursion_depth, gap, n_soup = soup_limit(n_soup, text_count,recursion_depth, max_count=limit_words, max_gap=1000,max_recursion_depth=max_recursion_depth)
  253. # 附件部分
  254. if len(_text_split[1])>limit_words:
  255. # attachment_html纯文本,无子结构
  256. if len(attachment_part.find_all(recursive=False))==0:
  257. attachment_part.string = str(attachment_part.get_text())[:limit_words]
  258. else:
  259. attachment_text_nums = 0
  260. attachment_skip = False
  261. for part in attachment_part.find_all(recursive=False):
  262. if not attachment_skip:
  263. if part.name == 'div' and 'filemd5' in part.attrs:
  264. if len(part.find_all(recursive=False)) == 0: #无结构的纯文本直接取值
  265. if not attachment_skip:
  266. last_attachment_text_nums = attachment_text_nums
  267. attachment_text_nums = attachment_text_nums + len(re.sub(sub_space, "", part.get_text()))
  268. if attachment_text_nums >=limit_words:
  269. part.string = str(part.get_text())[:limit_words - last_attachment_text_nums]
  270. attachment_skip = True
  271. else:
  272. part.decompose()
  273. else:
  274. for p_part in part.find_all(recursive=False):
  275. last_attachment_text_nums = attachment_text_nums
  276. attachment_text_nums = attachment_text_nums + len(re.sub(sub_space, "", p_part.get_text()))
  277. if not attachment_skip:
  278. if attachment_text_nums >= limit_words:
  279. p_part.string = str(p_part.get_text())[:limit_words - last_attachment_text_nums]
  280. attachment_skip = True
  281. else:
  282. p_part.decompose()
  283. else:
  284. last_attachment_text_nums = attachment_text_nums
  285. attachment_text_nums = attachment_text_nums + len(re.sub(sub_space, "", part.get_text()))
  286. if attachment_text_nums>=limit_words and not attachment_skip:
  287. part.string = str(part.get_text())[:limit_words-last_attachment_text_nums]
  288. attachment_skip = True
  289. else:
  290. part.decompose()
  291. return soup
  292. def attachment_filelink(soup):
  293. have_attachment = False
  294. attachment_part = None
  295. for child in soup.find_all(recursive=True):
  296. if child.name == 'div' and 'class' in child.attrs:
  297. if "richTextFetch" in child['class']:
  298. attachment_part = child
  299. have_attachment = True
  300. break
  301. if not have_attachment:
  302. return soup
  303. else:
  304. # 附件类型:图片、表格
  305. attachment_type = re.compile("\.(?:png|jpg|jpeg|tif|bmp|xlsx|xls)$")
  306. attachment_dict = dict()
  307. for _attachment in attachment_part.find_all(recursive=False):
  308. if _attachment.name == 'div' and 'filemd5' in _attachment.attrs:
  309. # print('filemd5',_attachment['filemd5'])
  310. attachment_dict[_attachment['filemd5']] = _attachment
  311. # print(attachment_dict)
  312. for child in soup.find_all(recursive=True):
  313. if child.name == 'div' and 'class' in child.attrs:
  314. if "richTextFetch" in child['class']:
  315. break
  316. if "filelink" in child.attrs and child['filelink'] in attachment_dict:
  317. if re.search(attachment_type,str(child.string).strip()) or \
  318. ('original' in child.attrs and re.search(attachment_type,str(child['original']).strip())) or \
  319. ('href' in child.attrs and re.search(attachment_type,str(child['href']).strip())):
  320. # 附件插入正文标识
  321. child.insert_before("。##attachment_begin##")
  322. child.insert_after("。##attachment_end##")
  323. child.replace_with(attachment_dict[child['filelink']])
  324. # print('格式化输出',soup.prettify())
  325. return soup
  326. def del_achievement(text):
  327. if re.search('中标|成交|入围|结果|评标|开标|候选人', text[:500]) == None or re.search('业绩', text) == None:
  328. return text
  329. p0 = '[,。;]((\d{1,2})|\d{1,2}、)[\w、]{,8}:|((\d{1,2})|\d{1,2}、)|。' # 例子 264392818
  330. p1 = '业绩[:,](\d、[-\w()、]{6,30}(工程|项目|勘察|设计|施工|监理|总承包|采购|更新)[\w()]{,10}[,;])+' # 例子 257717618
  331. p2 = '(类似业绩情况:|业绩:)(\w{,20}:)?(((\d)|\d、)项目名称:[-\w(),;、\d\s:]{5,100}[;。])+' # 例子 264345826
  332. p3 = '(投标|类似|(类似)?项目|合格|有效|企业|工程)?业绩(名称|信息|\d)?:(项目名称:)?[-\w()、]{6,50}(项目|工程|勘察|设计|施工|监理|总承包|采购|更新)'
  333. l = []
  334. tmp = []
  335. for it in re.finditer(p0, text):
  336. if it.group(0)[-3:] in ['业绩:', '荣誉:']:
  337. if tmp != []:
  338. del_text = text[tmp[0]:it.start()]
  339. l.append(del_text)
  340. tmp = []
  341. tmp.append(it.start())
  342. elif tmp != []:
  343. del_text = text[tmp[0]:it.start()]
  344. l.append(del_text)
  345. tmp = []
  346. if tmp != []:
  347. del_text = text[tmp[0]:]
  348. l.append(del_text)
  349. for del_text in l:
  350. text = text.replace(del_text, '')
  351. # print('删除业绩信息:', del_text)
  352. for rs in re.finditer(p1, text):
  353. # print('删除业绩信息:', rs.group(0))
  354. text = text.replace(rs.group(0), '')
  355. for rs in re.finditer(p2, text):
  356. # print('删除业绩信息:', rs.group(0))
  357. text = text.replace(rs.group(0), '')
  358. for rs in re.finditer(p3, text):
  359. # print('删除业绩信息:', rs.group(0))
  360. text = text.replace(rs.group(0), '')
  361. return text
  362. def split_header(soup):
  363. '''
  364. 处理 空格分割多个表头的情况 : 主要标的名称 规格型号(或服务要求) 主要标的数量 主要标的单价 合同金额(万元)
  365. :param soup: bs4 soup 对象
  366. :return:
  367. '''
  368. header = []
  369. attrs = []
  370. flag = 0
  371. tag = None
  372. for p in soup.find_all('p'):
  373. text = p.get_text()
  374. if re.search('主要标的数量\s+主要标的单价((万?元))?\s+合同金额', text):
  375. header = re.split('\s{3,}', text) if re.search('\s{3,}', text) else re.split('\s+', text)
  376. flag = 1
  377. tag = p
  378. tag.string = ''
  379. continue
  380. if flag:
  381. attrs = re.split('\s{3,}', text) if re.search('\s{3,}', text) else re.split('\s+', text)
  382. if header and len(header) == len(attrs) and tag:
  383. s = ""
  384. for head, attr in zip(header, attrs):
  385. s += head + ':' + attr + ','
  386. # tag.string = s
  387. # p.extract()
  388. p.string = s
  389. else:
  390. break