# -*- coding: utf-8 -*-
"""HTML 清洗、异常字符处理、附件前处理。
按 ARCHITECTURE.md Phase 4 拆分建议,从 ``interface/Preprocessing.py`` 迁出。
类型:PREPROCESS。
原位置:``interface/Preprocessing.py`` 中以下函数:
- ``special_treatment`` — 特殊数据源 HTML 预处理
- ``article_limit`` — 正文/附件字数限制
- ``attachment_filelink`` — 附件文件链接处理
- ``del_achievement`` — 删除业绩内容
- ``split_header`` — 空格分割多表头处理
``interface/Preprocessing.py`` 仍 re-export 以上全部名称,老 import 不受影响。
"""
from __future__ import absolute_import
import re
from BiddingKG.dl.common.logging import log
__all__ = [
"special_treatment",
"article_limit",
"attachment_filelink",
"del_achievement",
"split_header",
]
def special_treatment(sourceContent, web_source_no):
try:
if web_source_no == 'DX000202-1':
ser = re.search('中标供应商及中标金额:【(([\w()]{5,20}-[\d,.]+,)+)】', sourceContent)
if ser:
new = ""
l = ser.group(1).split(',')
for i in range(len(l)):
it = l[i]
if '-' in it:
role, money = it.split('-')
new += '标段%d, 中标供应商: ' % (i + 1) + role + ',中标金额:' + money + '。'
sourceContent = sourceContent.replace(ser.group(0), new, 1)
elif web_source_no == '00753-14':
body = sourceContent.find("body")
body_child = body.find_all(recursive=False)
pcontent = body
if 'id' in body_child[0].attrs:
if len(body_child) <= 2 and body_child[0]['id'] == 'pcontent':
pcontent = body_child[0]
# pcontent = sourceContent.find("div", id="pcontent")
pcontent = pcontent.find_all(recursive=False)[0]
first_table = None
for idx in range(len(pcontent.find_all(recursive=False))):
t_part = pcontent.find_all(recursive=False)[idx]
if t_part.name != "table":
break
if idx == 0:
first_table = t_part
else:
for _tr in t_part.find("tbody").find_all(recursive=False):
first_table.find("tbody").append(_tr)
t_part.clear()
elif web_source_no == 'DX008357-11':
body = sourceContent.find("body")
body_child = body.find_all(recursive=False)
pcontent = body
if 'id' in body_child[0].attrs:
if len(body_child) <= 2 and body_child[0]['id'] == 'pcontent':
pcontent = body_child[0]
# pcontent = sourceContent.find("div", id="pcontent")
pcontent = pcontent.find_all(recursive=False)[0]
error_table = []
is_error_table = False
for part in pcontent.find_all(recursive=False):
if is_error_table:
if part.name == "table":
error_table.append(part)
else:
break
if part.name == "div" and part.get_text(strip=True) == "中标候选单位:":
is_error_table = True
first_table = None
for idx in range(len(error_table)):
t_part = error_table[idx]
# if t_part.name != "table":
# break
if idx == 0:
for _tr in t_part.find("tbody").find_all(recursive=False):
if _tr.get_text(strip=True) == "":
_tr.decompose()
first_table = t_part
else:
for _tr in t_part.find("tbody").find_all(recursive=False):
if _tr.get_text(strip=True) != "":
first_table.find("tbody").append(_tr)
t_part.clear()
elif web_source_no == '18021-2':
body = sourceContent.find("body")
body_child = body.find_all(recursive=False)
pcontent = body
if 'id' in body_child[0].attrs:
if len(body_child) <= 2 and body_child[0]['id'] == 'pcontent':
pcontent = body_child[0]
# pcontent = sourceContent.find("div", id="pcontent")
td = pcontent.find_all("td")
for _td in td:
if str(_td.string).strip() == "报价金额":
_td.string = "单价"
elif web_source_no == '13740-2':
# “xxx成为成交供应商”
re_match = re.search("[^,。]+成为[^,。]*成交供应商", sourceContent)
if re_match:
sourceContent = sourceContent.replace(re_match.group(), "成交人:" + re_match.group())
elif web_source_no == '03786-10':
ser1 = re.search('中标价:([\d,.]+)', sourceContent)
ser2 = re.search('合同金额[((]万元[))]:([\d,.]+)', sourceContent)
if ser1 and ser2:
m1 = ser1.group(1).replace(',', '')
m2 = ser2.group(1).replace(',', '')
if float(m1) < 100000 and (m1.split('.')[0] == m2.split('.')[0] or m2 == '0'):
new = '中标价(万元):' + m1
sourceContent = sourceContent.replace(ser1.group(0), new, 1)
elif web_source_no=='00076-4':
ser = re.search('主要标的数量:([0-9一]+)\w{,3},主要标的单价:([\d,.]+)元?,合同金额:(.00),', sourceContent)
if ser:
num = ser.group(1).replace('一', '1')
try:
num = 1 if num == '0' else num
unit_price = ser.group(2).replace(',', '')
total_price = str(int(num) * float(unit_price))
new = '合同金额:' + total_price
sourceContent = sourceContent.replace('合同金额:.00', new, 1)
except Exception as e:
log('preprocessing.py special_treatment exception')
elif web_source_no=='DX000105-2':
if re.search("成交公示", sourceContent) and re.search(',投标人:', sourceContent) and re.search(',成交人:', sourceContent)==None:
sourceContent = sourceContent.replace(',投标人:', ',成交人:')
elif web_source_no in ['03795-1', '03795-2']:
if re.search('中标单位如下', sourceContent) and re.search(',投标人:', sourceContent) and re.search(',中标人:', sourceContent)==None:
sourceContent = sourceContent.replace(',投标人:', ',中标人:')
elif web_source_no in ['04080-3', '04080-4']:
ser = re.search('合同金额:([0-9,]+.[0-9]{3,})(.{,4})', sourceContent)
if ser and '万' not in ser.group(2):
sourceContent = sourceContent.replace('合同金额:', '合同金额(万元):')
elif web_source_no=='03761-3':
ser = re.search('中标价,([0-9]+)[.0-9]*%', sourceContent)
if ser and int(ser.group(1))>100:
sourceContent = sourceContent.replace(ser.group(0), ser.group(0)[:-1]+'元')
elif web_source_no=='00695-7':
ser = re.search('支付金额:', sourceContent)
if ser:
sourceContent = sourceContent.replace('支付金额:', '合同金额:')
elif web_source_no=='00811-8':
if re.search('是否中标:是', sourceContent) and re.search('排名:\d,', sourceContent):
sourceContent = re.sub('排名:\d,', '候选', sourceContent)
elif web_source_no=='DX000726-6':
sourceContent = re.sub('卖方[::\s]+宝山钢铁股份有限公司', '招标单位:宝山钢铁股份有限公司', sourceContent)
elif web_source_no=='DX008791-1':
sourceContent = re.sub('收货单位:', '最终用户:', sourceContent)
elif web_source_no=='DX011971':
sourceContent = re.sub('公司主体:', '业主单位:', sourceContent)
return sourceContent
except Exception as e:
log('特殊数据源: %s 预处理特别修改抛出异常: %s'%(web_source_no, e))
return sourceContent
def article_limit(soup,limit_words=30000):
sub_space = re.compile("\s+")
def soup_limit(_soup,_count,_recursion_depth,max_count=30000,max_gap=500,max_recursion_depth=900):
"""
:param _soup: soup
:param _count: 当前字数
:param max_count: 字数最大限制
:param max_gap: 超过限制后的最大误差
:return:
"""
_gap = _count - max_count
_is_skip = False
next_soup = None
# 跳过层级结构为1的标签,向下取值
# while len(_soup.find_all(recursive=False)) == 1 and \
# _soup.get_text(strip=True) == _soup.find_all(recursive=False)[0].get_text(strip=True):
# _soup = _soup.find_all(recursive=False)[0]
# _recursion_depth += 1
while len(_soup.find_all(recursive=False)) == 1:
_recursion_depth += 1
if _soup.get_text(strip=True) == _soup.find_all(recursive=False)[0].get_text(strip=True):
if _recursion_depth > max_recursion_depth:
_soup.string = str(_soup.get_text())[:max_count - _count]
next_soup = None
return _count, _recursion_depth, _gap, next_soup
else:
_soup = _soup.find_all(recursive=False)[0]
else:
_count += len(_soup.get_text(strip=True)) - len(_soup.find_all(recursive=False)[0].get_text(strip=True))
if _count >= max_count or _recursion_depth > max_recursion_depth:
_is_skip = True
next_soup = None
_count -= len(_soup.get_text(strip=True)) - len(_soup.find_all(recursive=False)[0].get_text(strip=True))
_soup.string = str(_soup.get_text())[:max_count - _count]
return _count, _recursion_depth, _gap, next_soup
else:
_soup = _soup.find_all(recursive=False)[0]
# 无结构的纯文本直接取值
if len(_soup.find_all(recursive=False)) == 0:
_soup.string = str(_soup.get_text())[:max_count-_count]
_count += len(re.sub(sub_space, "", _soup.string))
_gap = _count - max_count
next_soup = None
else:
_recursion_depth += 1
for _soup_part in _soup.find_all(recursive=False):
if not _is_skip:
_count += len(re.sub(sub_space, "", _soup_part.get_text()))
if _count >= max_count:
_gap = _count - max_count
if _gap <= max_gap:
_is_skip = True
else:
_is_skip = True
if _recursion_depth <= max_recursion_depth:
next_soup = _soup_part
_count -= len(re.sub(sub_space, "", _soup_part.get_text()))
else: # 超出最大递归层级时,直接切片取值
next_soup = None
_count -= len(re.sub(sub_space, "", _soup_part.get_text()))
_soup_part.string = str(_soup_part.get_text())[:max_count - _count]
continue
else:
_soup_part.decompose()
return _count,_recursion_depth,_gap,next_soup
text_count = 0
max_recursion_depth = 900 # 最大递归
recursion_depth = 0
have_attachment = False
attachment_part = None
for child in soup.find_all(recursive=True):
if child.name == 'div' and 'class' in child.attrs:
if "richTextFetch" in child['class']:
child.insert_before("##attachment##。") # 句号分开,避免项目名称等提取
attachment_part = child
have_attachment = True
break
if not have_attachment:
# 无附件,通过get_text()方法与limit_words大小判断是否要限制字数
if len(re.sub(sub_space, "", soup.get_text())) > limit_words:
text_count,recursion_depth,gap,n_soup = soup_limit(soup,text_count,recursion_depth,max_count=limit_words,max_gap=1000,max_recursion_depth=max_recursion_depth)
while n_soup:
text_count,recursion_depth, gap, n_soup = soup_limit(n_soup, text_count,recursion_depth, max_count=limit_words, max_gap=1000,max_recursion_depth=max_recursion_depth)
else:
# 有附件
_text = re.sub(sub_space, "", soup.get_text())
_text_split = _text.split("##attachment##")
# 正文部分
if len(_text_split[0])>limit_words:
main_soup = attachment_part.parent
main_text = main_soup.find_all(recursive=False)[0]
text_count,recursion_depth, gap, n_soup = soup_limit(main_text, text_count,recursion_depth, max_count=limit_words, max_gap=1000,max_recursion_depth=max_recursion_depth)
while n_soup:
text_count,recursion_depth, gap, n_soup = soup_limit(n_soup, text_count,recursion_depth, max_count=limit_words, max_gap=1000,max_recursion_depth=max_recursion_depth)
# 附件部分
if len(_text_split[1])>limit_words:
# attachment_html纯文本,无子结构
if len(attachment_part.find_all(recursive=False))==0:
attachment_part.string = str(attachment_part.get_text())[:limit_words]
else:
attachment_text_nums = 0
attachment_skip = False
for part in attachment_part.find_all(recursive=False):
if not attachment_skip:
if part.name == 'div' and 'filemd5' in part.attrs:
if len(part.find_all(recursive=False)) == 0: #无结构的纯文本直接取值
if not attachment_skip:
last_attachment_text_nums = attachment_text_nums
attachment_text_nums = attachment_text_nums + len(re.sub(sub_space, "", part.get_text()))
if attachment_text_nums >=limit_words:
part.string = str(part.get_text())[:limit_words - last_attachment_text_nums]
attachment_skip = True
else:
part.decompose()
else:
for p_part in part.find_all(recursive=False):
last_attachment_text_nums = attachment_text_nums
attachment_text_nums = attachment_text_nums + len(re.sub(sub_space, "", p_part.get_text()))
if not attachment_skip:
if attachment_text_nums >= limit_words:
p_part.string = str(p_part.get_text())[:limit_words - last_attachment_text_nums]
attachment_skip = True
else:
p_part.decompose()
else:
last_attachment_text_nums = attachment_text_nums
attachment_text_nums = attachment_text_nums + len(re.sub(sub_space, "", part.get_text()))
if attachment_text_nums>=limit_words and not attachment_skip:
part.string = str(part.get_text())[:limit_words-last_attachment_text_nums]
attachment_skip = True
else:
part.decompose()
return soup
def attachment_filelink(soup):
have_attachment = False
attachment_part = None
for child in soup.find_all(recursive=True):
if child.name == 'div' and 'class' in child.attrs:
if "richTextFetch" in child['class']:
attachment_part = child
have_attachment = True
break
if not have_attachment:
return soup
else:
# 附件类型:图片、表格
attachment_type = re.compile("\.(?:png|jpg|jpeg|tif|bmp|xlsx|xls)$")
attachment_dict = dict()
for _attachment in attachment_part.find_all(recursive=False):
if _attachment.name == 'div' and 'filemd5' in _attachment.attrs:
# print('filemd5',_attachment['filemd5'])
attachment_dict[_attachment['filemd5']] = _attachment
# print(attachment_dict)
for child in soup.find_all(recursive=True):
if child.name == 'div' and 'class' in child.attrs:
if "richTextFetch" in child['class']:
break
if "filelink" in child.attrs and child['filelink'] in attachment_dict:
if re.search(attachment_type,str(child.string).strip()) or \
('original' in child.attrs and re.search(attachment_type,str(child['original']).strip())) or \
('href' in child.attrs and re.search(attachment_type,str(child['href']).strip())):
# 附件插入正文标识
child.insert_before("。##attachment_begin##")
child.insert_after("。##attachment_end##")
child.replace_with(attachment_dict[child['filelink']])
# print('格式化输出',soup.prettify())
return soup
def del_achievement(text):
if re.search('中标|成交|入围|结果|评标|开标|候选人', text[:500]) == None or re.search('业绩', text) == None:
return text
p0 = '[,。;]((\d{1,2})|\d{1,2}、)[\w、]{,8}:|((\d{1,2})|\d{1,2}、)|。' # 例子 264392818
p1 = '业绩[:,](\d、[-\w()、]{6,30}(工程|项目|勘察|设计|施工|监理|总承包|采购|更新)[\w()]{,10}[,;])+' # 例子 257717618
p2 = '(类似业绩情况:|业绩:)(\w{,20}:)?(((\d)|\d、)项目名称:[-\w(),;、\d\s:]{5,100}[;。])+' # 例子 264345826
p3 = '(投标|类似|(类似)?项目|合格|有效|企业|工程)?业绩(名称|信息|\d)?:(项目名称:)?[-\w()、]{6,50}(项目|工程|勘察|设计|施工|监理|总承包|采购|更新)'
l = []
tmp = []
for it in re.finditer(p0, text):
if it.group(0)[-3:] in ['业绩:', '荣誉:']:
if tmp != []:
del_text = text[tmp[0]:it.start()]
l.append(del_text)
tmp = []
tmp.append(it.start())
elif tmp != []:
del_text = text[tmp[0]:it.start()]
l.append(del_text)
tmp = []
if tmp != []:
del_text = text[tmp[0]:]
l.append(del_text)
for del_text in l:
text = text.replace(del_text, '')
# print('删除业绩信息:', del_text)
for rs in re.finditer(p1, text):
# print('删除业绩信息:', rs.group(0))
text = text.replace(rs.group(0), '')
for rs in re.finditer(p2, text):
# print('删除业绩信息:', rs.group(0))
text = text.replace(rs.group(0), '')
for rs in re.finditer(p3, text):
# print('删除业绩信息:', rs.group(0))
text = text.replace(rs.group(0), '')
return text
def split_header(soup):
'''
处理 空格分割多个表头的情况 : 主要标的名称 规格型号(或服务要求) 主要标的数量 主要标的单价 合同金额(万元)
:param soup: bs4 soup 对象
:return:
'''
header = []
attrs = []
flag = 0
tag = None
for p in soup.find_all('p'):
text = p.get_text()
if re.search('主要标的数量\s+主要标的单价((万?元))?\s+合同金额', text):
header = re.split('\s{3,}', text) if re.search('\s{3,}', text) else re.split('\s+', text)
flag = 1
tag = p
tag.string = ''
continue
if flag:
attrs = re.split('\s{3,}', text) if re.search('\s{3,}', text) else re.split('\s+', text)
if header and len(header) == len(attrs) and tag:
s = ""
for head, attr in zip(header, attrs):
s += head + ':' + attr + ','
# tag.string = s
# p.extract()
p.string = s
else:
break