| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409 |
- # -*- coding: utf-8 -*-
- """HTML 清洗、异常字符处理、附件前处理。
- 按 ARCHITECTURE.md Phase 4 拆分建议,从 ``interface/Preprocessing.py`` 迁出。
- 类型:PREPROCESS。
- 原位置:``interface/Preprocessing.py`` 中以下函数:
- - ``special_treatment`` — 特殊数据源 HTML 预处理
- - ``article_limit`` — 正文/附件字数限制
- - ``attachment_filelink`` — 附件文件链接处理
- - ``del_achievement`` — 删除业绩内容
- - ``split_header`` — 空格分割多表头处理
- ``interface/Preprocessing.py`` 仍 re-export 以上全部名称,老 import 不受影响。
- """
- from __future__ import absolute_import
- import re
- from BiddingKG.dl.common.logging import log
- __all__ = [
- "special_treatment",
- "article_limit",
- "attachment_filelink",
- "del_achievement",
- "split_header",
- ]
- def special_treatment(sourceContent, web_source_no):
- try:
- if web_source_no == 'DX000202-1':
- ser = re.search('中标供应商及中标金额:【(([\w()]{5,20}-[\d,.]+,)+)】', sourceContent)
- if ser:
- new = ""
- l = ser.group(1).split(',')
- for i in range(len(l)):
- it = l[i]
- if '-' in it:
- role, money = it.split('-')
- new += '标段%d, 中标供应商: ' % (i + 1) + role + ',中标金额:' + money + '。'
- sourceContent = sourceContent.replace(ser.group(0), new, 1)
- elif web_source_no == '00753-14':
- body = sourceContent.find("body")
- body_child = body.find_all(recursive=False)
- pcontent = body
- if 'id' in body_child[0].attrs:
- if len(body_child) <= 2 and body_child[0]['id'] == 'pcontent':
- pcontent = body_child[0]
- # pcontent = sourceContent.find("div", id="pcontent")
- pcontent = pcontent.find_all(recursive=False)[0]
- first_table = None
- for idx in range(len(pcontent.find_all(recursive=False))):
- t_part = pcontent.find_all(recursive=False)[idx]
- if t_part.name != "table":
- break
- if idx == 0:
- first_table = t_part
- else:
- for _tr in t_part.find("tbody").find_all(recursive=False):
- first_table.find("tbody").append(_tr)
- t_part.clear()
- elif web_source_no == 'DX008357-11':
- body = sourceContent.find("body")
- body_child = body.find_all(recursive=False)
- pcontent = body
- if 'id' in body_child[0].attrs:
- if len(body_child) <= 2 and body_child[0]['id'] == 'pcontent':
- pcontent = body_child[0]
- # pcontent = sourceContent.find("div", id="pcontent")
- pcontent = pcontent.find_all(recursive=False)[0]
- error_table = []
- is_error_table = False
- for part in pcontent.find_all(recursive=False):
- if is_error_table:
- if part.name == "table":
- error_table.append(part)
- else:
- break
- if part.name == "div" and part.get_text(strip=True) == "中标候选单位:":
- is_error_table = True
- first_table = None
- for idx in range(len(error_table)):
- t_part = error_table[idx]
- # if t_part.name != "table":
- # break
- if idx == 0:
- for _tr in t_part.find("tbody").find_all(recursive=False):
- if _tr.get_text(strip=True) == "":
- _tr.decompose()
- first_table = t_part
- else:
- for _tr in t_part.find("tbody").find_all(recursive=False):
- if _tr.get_text(strip=True) != "":
- first_table.find("tbody").append(_tr)
- t_part.clear()
- elif web_source_no == '18021-2':
- body = sourceContent.find("body")
- body_child = body.find_all(recursive=False)
- pcontent = body
- if 'id' in body_child[0].attrs:
- if len(body_child) <= 2 and body_child[0]['id'] == 'pcontent':
- pcontent = body_child[0]
- # pcontent = sourceContent.find("div", id="pcontent")
- td = pcontent.find_all("td")
- for _td in td:
- if str(_td.string).strip() == "报价金额":
- _td.string = "单价"
- elif web_source_no == '13740-2':
- # “xxx成为成交供应商”
- re_match = re.search("[^,。]+成为[^,。]*成交供应商", sourceContent)
- if re_match:
- sourceContent = sourceContent.replace(re_match.group(), "成交人:" + re_match.group())
- elif web_source_no == '03786-10':
- ser1 = re.search('中标价:([\d,.]+)', sourceContent)
- ser2 = re.search('合同金额[((]万元[))]:([\d,.]+)', sourceContent)
- if ser1 and ser2:
- m1 = ser1.group(1).replace(',', '')
- m2 = ser2.group(1).replace(',', '')
- if float(m1) < 100000 and (m1.split('.')[0] == m2.split('.')[0] or m2 == '0'):
- new = '中标价(万元):' + m1
- sourceContent = sourceContent.replace(ser1.group(0), new, 1)
- elif web_source_no=='00076-4':
- ser = re.search('主要标的数量:([0-9一]+)\w{,3},主要标的单价:([\d,.]+)元?,合同金额:(.00),', sourceContent)
- if ser:
- num = ser.group(1).replace('一', '1')
- try:
- num = 1 if num == '0' else num
- unit_price = ser.group(2).replace(',', '')
- total_price = str(int(num) * float(unit_price))
- new = '合同金额:' + total_price
- sourceContent = sourceContent.replace('合同金额:.00', new, 1)
- except Exception as e:
- log('preprocessing.py special_treatment exception')
- elif web_source_no=='DX000105-2':
- if re.search("成交公示", sourceContent) and re.search(',投标人:', sourceContent) and re.search(',成交人:', sourceContent)==None:
- sourceContent = sourceContent.replace(',投标人:', ',成交人:')
- elif web_source_no in ['03795-1', '03795-2']:
- if re.search('中标单位如下', sourceContent) and re.search(',投标人:', sourceContent) and re.search(',中标人:', sourceContent)==None:
- sourceContent = sourceContent.replace(',投标人:', ',中标人:')
- elif web_source_no in ['04080-3', '04080-4']:
- ser = re.search('合同金额:([0-9,]+.[0-9]{3,})(.{,4})', sourceContent)
- if ser and '万' not in ser.group(2):
- sourceContent = sourceContent.replace('合同金额:', '合同金额(万元):')
- elif web_source_no=='03761-3':
- ser = re.search('中标价,([0-9]+)[.0-9]*%', sourceContent)
- if ser and int(ser.group(1))>100:
- sourceContent = sourceContent.replace(ser.group(0), ser.group(0)[:-1]+'元')
- elif web_source_no=='00695-7':
- ser = re.search('支付金额:', sourceContent)
- if ser:
- sourceContent = sourceContent.replace('支付金额:', '合同金额:')
- elif web_source_no=='00811-8':
- if re.search('是否中标:是', sourceContent) and re.search('排名:\d,', sourceContent):
- sourceContent = re.sub('排名:\d,', '候选', sourceContent)
- elif web_source_no=='DX000726-6':
- sourceContent = re.sub('卖方[::\s]+宝山钢铁股份有限公司', '招标单位:宝山钢铁股份有限公司', sourceContent)
- elif web_source_no=='DX008791-1':
- sourceContent = re.sub('收货单位:', '最终用户:', sourceContent)
- elif web_source_no=='DX011971':
- sourceContent = re.sub('公司主体:', '业主单位:', sourceContent)
- return sourceContent
- except Exception as e:
- log('特殊数据源: %s 预处理特别修改抛出异常: %s'%(web_source_no, e))
- return sourceContent
- def article_limit(soup,limit_words=30000):
- sub_space = re.compile("\s+")
- def soup_limit(_soup,_count,_recursion_depth,max_count=30000,max_gap=500,max_recursion_depth=900):
- """
- :param _soup: soup
- :param _count: 当前字数
- :param max_count: 字数最大限制
- :param max_gap: 超过限制后的最大误差
- :return:
- """
- _gap = _count - max_count
- _is_skip = False
- next_soup = None
- # 跳过层级结构为1的标签,向下取值
- # while len(_soup.find_all(recursive=False)) == 1 and \
- # _soup.get_text(strip=True) == _soup.find_all(recursive=False)[0].get_text(strip=True):
- # _soup = _soup.find_all(recursive=False)[0]
- # _recursion_depth += 1
- while len(_soup.find_all(recursive=False)) == 1:
- _recursion_depth += 1
- if _soup.get_text(strip=True) == _soup.find_all(recursive=False)[0].get_text(strip=True):
- if _recursion_depth > max_recursion_depth:
- _soup.string = str(_soup.get_text())[:max_count - _count]
- next_soup = None
- return _count, _recursion_depth, _gap, next_soup
- else:
- _soup = _soup.find_all(recursive=False)[0]
- else:
- _count += len(_soup.get_text(strip=True)) - len(_soup.find_all(recursive=False)[0].get_text(strip=True))
- if _count >= max_count or _recursion_depth > max_recursion_depth:
- _is_skip = True
- next_soup = None
- _count -= len(_soup.get_text(strip=True)) - len(_soup.find_all(recursive=False)[0].get_text(strip=True))
- _soup.string = str(_soup.get_text())[:max_count - _count]
- return _count, _recursion_depth, _gap, next_soup
- else:
- _soup = _soup.find_all(recursive=False)[0]
- # 无结构的纯文本直接取值
- if len(_soup.find_all(recursive=False)) == 0:
- _soup.string = str(_soup.get_text())[:max_count-_count]
- _count += len(re.sub(sub_space, "", _soup.string))
- _gap = _count - max_count
- next_soup = None
- else:
- _recursion_depth += 1
- for _soup_part in _soup.find_all(recursive=False):
- if not _is_skip:
- _count += len(re.sub(sub_space, "", _soup_part.get_text()))
- if _count >= max_count:
- _gap = _count - max_count
- if _gap <= max_gap:
- _is_skip = True
- else:
- _is_skip = True
- if _recursion_depth <= max_recursion_depth:
- next_soup = _soup_part
- _count -= len(re.sub(sub_space, "", _soup_part.get_text()))
- else: # 超出最大递归层级时,直接切片取值
- next_soup = None
- _count -= len(re.sub(sub_space, "", _soup_part.get_text()))
- _soup_part.string = str(_soup_part.get_text())[:max_count - _count]
- continue
- else:
- _soup_part.decompose()
- return _count,_recursion_depth,_gap,next_soup
- text_count = 0
- max_recursion_depth = 900 # 最大递归
- recursion_depth = 0
- have_attachment = False
- attachment_part = None
- for child in soup.find_all(recursive=True):
- if child.name == 'div' and 'class' in child.attrs:
- if "richTextFetch" in child['class']:
- child.insert_before("##attachment##。") # 句号分开,避免项目名称等提取
- attachment_part = child
- have_attachment = True
- break
- if not have_attachment:
- # 无附件,通过get_text()方法与limit_words大小判断是否要限制字数
- if len(re.sub(sub_space, "", soup.get_text())) > limit_words:
- text_count,recursion_depth,gap,n_soup = soup_limit(soup,text_count,recursion_depth,max_count=limit_words,max_gap=1000,max_recursion_depth=max_recursion_depth)
- while n_soup:
- text_count,recursion_depth, gap, n_soup = soup_limit(n_soup, text_count,recursion_depth, max_count=limit_words, max_gap=1000,max_recursion_depth=max_recursion_depth)
- else:
- # 有附件
- _text = re.sub(sub_space, "", soup.get_text())
- _text_split = _text.split("##attachment##")
- # 正文部分
- if len(_text_split[0])>limit_words:
- main_soup = attachment_part.parent
- main_text = main_soup.find_all(recursive=False)[0]
- text_count,recursion_depth, gap, n_soup = soup_limit(main_text, text_count,recursion_depth, max_count=limit_words, max_gap=1000,max_recursion_depth=max_recursion_depth)
- while n_soup:
- text_count,recursion_depth, gap, n_soup = soup_limit(n_soup, text_count,recursion_depth, max_count=limit_words, max_gap=1000,max_recursion_depth=max_recursion_depth)
- # 附件部分
- if len(_text_split[1])>limit_words:
- # attachment_html纯文本,无子结构
- if len(attachment_part.find_all(recursive=False))==0:
- attachment_part.string = str(attachment_part.get_text())[:limit_words]
- else:
- attachment_text_nums = 0
- attachment_skip = False
- for part in attachment_part.find_all(recursive=False):
- if not attachment_skip:
- if part.name == 'div' and 'filemd5' in part.attrs:
- if len(part.find_all(recursive=False)) == 0: #无结构的纯文本直接取值
- if not attachment_skip:
- last_attachment_text_nums = attachment_text_nums
- attachment_text_nums = attachment_text_nums + len(re.sub(sub_space, "", part.get_text()))
- if attachment_text_nums >=limit_words:
- part.string = str(part.get_text())[:limit_words - last_attachment_text_nums]
- attachment_skip = True
- else:
- part.decompose()
- else:
- for p_part in part.find_all(recursive=False):
- last_attachment_text_nums = attachment_text_nums
- attachment_text_nums = attachment_text_nums + len(re.sub(sub_space, "", p_part.get_text()))
- if not attachment_skip:
- if attachment_text_nums >= limit_words:
- p_part.string = str(p_part.get_text())[:limit_words - last_attachment_text_nums]
- attachment_skip = True
- else:
- p_part.decompose()
- else:
- last_attachment_text_nums = attachment_text_nums
- attachment_text_nums = attachment_text_nums + len(re.sub(sub_space, "", part.get_text()))
- if attachment_text_nums>=limit_words and not attachment_skip:
- part.string = str(part.get_text())[:limit_words-last_attachment_text_nums]
- attachment_skip = True
- else:
- part.decompose()
- return soup
- def attachment_filelink(soup):
- have_attachment = False
- attachment_part = None
- for child in soup.find_all(recursive=True):
- if child.name == 'div' and 'class' in child.attrs:
- if "richTextFetch" in child['class']:
- attachment_part = child
- have_attachment = True
- break
- if not have_attachment:
- return soup
- else:
- # 附件类型:图片、表格
- attachment_type = re.compile("\.(?:png|jpg|jpeg|tif|bmp|xlsx|xls)$")
- attachment_dict = dict()
- for _attachment in attachment_part.find_all(recursive=False):
- if _attachment.name == 'div' and 'filemd5' in _attachment.attrs:
- # print('filemd5',_attachment['filemd5'])
- attachment_dict[_attachment['filemd5']] = _attachment
- # print(attachment_dict)
- for child in soup.find_all(recursive=True):
- if child.name == 'div' and 'class' in child.attrs:
- if "richTextFetch" in child['class']:
- break
- if "filelink" in child.attrs and child['filelink'] in attachment_dict:
- if re.search(attachment_type,str(child.string).strip()) or \
- ('original' in child.attrs and re.search(attachment_type,str(child['original']).strip())) or \
- ('href' in child.attrs and re.search(attachment_type,str(child['href']).strip())):
- # 附件插入正文标识
- child.insert_before("。##attachment_begin##")
- child.insert_after("。##attachment_end##")
- child.replace_with(attachment_dict[child['filelink']])
- # print('格式化输出',soup.prettify())
- return soup
- def del_achievement(text):
- if re.search('中标|成交|入围|结果|评标|开标|候选人', text[:500]) == None or re.search('业绩', text) == None:
- return text
- p0 = '[,。;]((\d{1,2})|\d{1,2}、)[\w、]{,8}:|((\d{1,2})|\d{1,2}、)|。' # 例子 264392818
- p1 = '业绩[:,](\d、[-\w()、]{6,30}(工程|项目|勘察|设计|施工|监理|总承包|采购|更新)[\w()]{,10}[,;])+' # 例子 257717618
- p2 = '(类似业绩情况:|业绩:)(\w{,20}:)?(((\d)|\d、)项目名称:[-\w(),;、\d\s:]{5,100}[;。])+' # 例子 264345826
- p3 = '(投标|类似|(类似)?项目|合格|有效|企业|工程)?业绩(名称|信息|\d)?:(项目名称:)?[-\w()、]{6,50}(项目|工程|勘察|设计|施工|监理|总承包|采购|更新)'
- l = []
- tmp = []
- for it in re.finditer(p0, text):
- if it.group(0)[-3:] in ['业绩:', '荣誉:']:
- if tmp != []:
- del_text = text[tmp[0]:it.start()]
- l.append(del_text)
- tmp = []
- tmp.append(it.start())
- elif tmp != []:
- del_text = text[tmp[0]:it.start()]
- l.append(del_text)
- tmp = []
- if tmp != []:
- del_text = text[tmp[0]:]
- l.append(del_text)
- for del_text in l:
- text = text.replace(del_text, '')
- # print('删除业绩信息:', del_text)
- for rs in re.finditer(p1, text):
- # print('删除业绩信息:', rs.group(0))
- text = text.replace(rs.group(0), '')
- for rs in re.finditer(p2, text):
- # print('删除业绩信息:', rs.group(0))
- text = text.replace(rs.group(0), '')
- for rs in re.finditer(p3, text):
- # print('删除业绩信息:', rs.group(0))
- text = text.replace(rs.group(0), '')
- return text
- def split_header(soup):
- '''
- 处理 空格分割多个表头的情况 : 主要标的名称 规格型号(或服务要求) 主要标的数量 主要标的单价 合同金额(万元)
- :param soup: bs4 soup 对象
- :return:
- '''
- header = []
- attrs = []
- flag = 0
- tag = None
- for p in soup.find_all('p'):
- text = p.get_text()
- if re.search('主要标的数量\s+主要标的单价((万?元))?\s+合同金额', text):
- header = re.split('\s{3,}', text) if re.search('\s{3,}', text) else re.split('\s+', text)
- flag = 1
- tag = p
- tag.string = ''
- continue
- if flag:
- attrs = re.split('\s{3,}', text) if re.search('\s{3,}', text) else re.split('\s+', text)
- if header and len(header) == len(attrs) and tag:
- s = ""
- for head, attr in zip(header, attrs):
- s += head + ':' + attr + ','
- # tag.string = s
- # p.extract()
- p.string = s
- else:
- break
|