# -*- coding: utf-8 -*- """HTML 清洗、异常字符处理、附件前处理。 按 ARCHITECTURE.md Phase 4 拆分建议,从 ``interface/Preprocessing.py`` 迁出。 类型:PREPROCESS。 原位置:``interface/Preprocessing.py`` 中以下函数: - ``special_treatment`` — 特殊数据源 HTML 预处理 - ``article_limit`` — 正文/附件字数限制 - ``attachment_filelink`` — 附件文件链接处理 - ``del_achievement`` — 删除业绩内容 - ``split_header`` — 空格分割多表头处理 ``interface/Preprocessing.py`` 仍 re-export 以上全部名称,老 import 不受影响。 """ from __future__ import absolute_import import re from BiddingKG.dl.common.logging import log __all__ = [ "special_treatment", "article_limit", "attachment_filelink", "del_achievement", "split_header", ] def special_treatment(sourceContent, web_source_no): try: if web_source_no == 'DX000202-1': ser = re.search('中标供应商及中标金额:【(([\w()]{5,20}-[\d,.]+,)+)】', sourceContent) if ser: new = "" l = ser.group(1).split(',') for i in range(len(l)): it = l[i] if '-' in it: role, money = it.split('-') new += '标段%d, 中标供应商: ' % (i + 1) + role + ',中标金额:' + money + '。' sourceContent = sourceContent.replace(ser.group(0), new, 1) elif web_source_no == '00753-14': body = sourceContent.find("body") body_child = body.find_all(recursive=False) pcontent = body if 'id' in body_child[0].attrs: if len(body_child) <= 2 and body_child[0]['id'] == 'pcontent': pcontent = body_child[0] # pcontent = sourceContent.find("div", id="pcontent") pcontent = pcontent.find_all(recursive=False)[0] first_table = None for idx in range(len(pcontent.find_all(recursive=False))): t_part = pcontent.find_all(recursive=False)[idx] if t_part.name != "table": break if idx == 0: first_table = t_part else: for _tr in t_part.find("tbody").find_all(recursive=False): first_table.find("tbody").append(_tr) t_part.clear() elif web_source_no == 'DX008357-11': body = sourceContent.find("body") body_child = body.find_all(recursive=False) pcontent = body if 'id' in body_child[0].attrs: if len(body_child) <= 2 and body_child[0]['id'] == 'pcontent': pcontent = body_child[0] # pcontent = sourceContent.find("div", id="pcontent") pcontent = pcontent.find_all(recursive=False)[0] error_table = [] is_error_table = False for part in pcontent.find_all(recursive=False): if is_error_table: if part.name == "table": error_table.append(part) else: break if part.name == "div" and part.get_text(strip=True) == "中标候选单位:": is_error_table = True first_table = None for idx in range(len(error_table)): t_part = error_table[idx] # if t_part.name != "table": # break if idx == 0: for _tr in t_part.find("tbody").find_all(recursive=False): if _tr.get_text(strip=True) == "": _tr.decompose() first_table = t_part else: for _tr in t_part.find("tbody").find_all(recursive=False): if _tr.get_text(strip=True) != "": first_table.find("tbody").append(_tr) t_part.clear() elif web_source_no == '18021-2': body = sourceContent.find("body") body_child = body.find_all(recursive=False) pcontent = body if 'id' in body_child[0].attrs: if len(body_child) <= 2 and body_child[0]['id'] == 'pcontent': pcontent = body_child[0] # pcontent = sourceContent.find("div", id="pcontent") td = pcontent.find_all("td") for _td in td: if str(_td.string).strip() == "报价金额": _td.string = "单价" elif web_source_no == '13740-2': # “xxx成为成交供应商” re_match = re.search("[^,。]+成为[^,。]*成交供应商", sourceContent) if re_match: sourceContent = sourceContent.replace(re_match.group(), "成交人:" + re_match.group()) elif web_source_no == '03786-10': ser1 = re.search('中标价:([\d,.]+)', sourceContent) ser2 = re.search('合同金额[((]万元[))]:([\d,.]+)', sourceContent) if ser1 and ser2: m1 = ser1.group(1).replace(',', '') m2 = ser2.group(1).replace(',', '') if float(m1) < 100000 and (m1.split('.')[0] == m2.split('.')[0] or m2 == '0'): new = '中标价(万元):' + m1 sourceContent = sourceContent.replace(ser1.group(0), new, 1) elif web_source_no=='00076-4': ser = re.search('主要标的数量:([0-9一]+)\w{,3},主要标的单价:([\d,.]+)元?,合同金额:(.00),', sourceContent) if ser: num = ser.group(1).replace('一', '1') try: num = 1 if num == '0' else num unit_price = ser.group(2).replace(',', '') total_price = str(int(num) * float(unit_price)) new = '合同金额:' + total_price sourceContent = sourceContent.replace('合同金额:.00', new, 1) except Exception as e: log('preprocessing.py special_treatment exception') elif web_source_no=='DX000105-2': if re.search("成交公示", sourceContent) and re.search(',投标人:', sourceContent) and re.search(',成交人:', sourceContent)==None: sourceContent = sourceContent.replace(',投标人:', ',成交人:') elif web_source_no in ['03795-1', '03795-2']: if re.search('中标单位如下', sourceContent) and re.search(',投标人:', sourceContent) and re.search(',中标人:', sourceContent)==None: sourceContent = sourceContent.replace(',投标人:', ',中标人:') elif web_source_no in ['04080-3', '04080-4']: ser = re.search('合同金额:([0-9,]+.[0-9]{3,})(.{,4})', sourceContent) if ser and '万' not in ser.group(2): sourceContent = sourceContent.replace('合同金额:', '合同金额(万元):') elif web_source_no=='03761-3': ser = re.search('中标价,([0-9]+)[.0-9]*%', sourceContent) if ser and int(ser.group(1))>100: sourceContent = sourceContent.replace(ser.group(0), ser.group(0)[:-1]+'元') elif web_source_no=='00695-7': ser = re.search('支付金额:', sourceContent) if ser: sourceContent = sourceContent.replace('支付金额:', '合同金额:') elif web_source_no=='00811-8': if re.search('是否中标:是', sourceContent) and re.search('排名:\d,', sourceContent): sourceContent = re.sub('排名:\d,', '候选', sourceContent) elif web_source_no=='DX000726-6': sourceContent = re.sub('卖方[::\s]+宝山钢铁股份有限公司', '招标单位:宝山钢铁股份有限公司', sourceContent) elif web_source_no=='DX008791-1': sourceContent = re.sub('收货单位:', '最终用户:', sourceContent) elif web_source_no=='DX011971': sourceContent = re.sub('公司主体:', '业主单位:', sourceContent) return sourceContent except Exception as e: log('特殊数据源: %s 预处理特别修改抛出异常: %s'%(web_source_no, e)) return sourceContent def article_limit(soup,limit_words=30000): sub_space = re.compile("\s+") def soup_limit(_soup,_count,_recursion_depth,max_count=30000,max_gap=500,max_recursion_depth=900): """ :param _soup: soup :param _count: 当前字数 :param max_count: 字数最大限制 :param max_gap: 超过限制后的最大误差 :return: """ _gap = _count - max_count _is_skip = False next_soup = None # 跳过层级结构为1的标签,向下取值 # while len(_soup.find_all(recursive=False)) == 1 and \ # _soup.get_text(strip=True) == _soup.find_all(recursive=False)[0].get_text(strip=True): # _soup = _soup.find_all(recursive=False)[0] # _recursion_depth += 1 while len(_soup.find_all(recursive=False)) == 1: _recursion_depth += 1 if _soup.get_text(strip=True) == _soup.find_all(recursive=False)[0].get_text(strip=True): if _recursion_depth > max_recursion_depth: _soup.string = str(_soup.get_text())[:max_count - _count] next_soup = None return _count, _recursion_depth, _gap, next_soup else: _soup = _soup.find_all(recursive=False)[0] else: _count += len(_soup.get_text(strip=True)) - len(_soup.find_all(recursive=False)[0].get_text(strip=True)) if _count >= max_count or _recursion_depth > max_recursion_depth: _is_skip = True next_soup = None _count -= len(_soup.get_text(strip=True)) - len(_soup.find_all(recursive=False)[0].get_text(strip=True)) _soup.string = str(_soup.get_text())[:max_count - _count] return _count, _recursion_depth, _gap, next_soup else: _soup = _soup.find_all(recursive=False)[0] # 无结构的纯文本直接取值 if len(_soup.find_all(recursive=False)) == 0: _soup.string = str(_soup.get_text())[:max_count-_count] _count += len(re.sub(sub_space, "", _soup.string)) _gap = _count - max_count next_soup = None else: _recursion_depth += 1 for _soup_part in _soup.find_all(recursive=False): if not _is_skip: _count += len(re.sub(sub_space, "", _soup_part.get_text())) if _count >= max_count: _gap = _count - max_count if _gap <= max_gap: _is_skip = True else: _is_skip = True if _recursion_depth <= max_recursion_depth: next_soup = _soup_part _count -= len(re.sub(sub_space, "", _soup_part.get_text())) else: # 超出最大递归层级时,直接切片取值 next_soup = None _count -= len(re.sub(sub_space, "", _soup_part.get_text())) _soup_part.string = str(_soup_part.get_text())[:max_count - _count] continue else: _soup_part.decompose() return _count,_recursion_depth,_gap,next_soup text_count = 0 max_recursion_depth = 900 # 最大递归 recursion_depth = 0 have_attachment = False attachment_part = None for child in soup.find_all(recursive=True): if child.name == 'div' and 'class' in child.attrs: if "richTextFetch" in child['class']: child.insert_before("##attachment##。") # 句号分开,避免项目名称等提取 attachment_part = child have_attachment = True break if not have_attachment: # 无附件,通过get_text()方法与limit_words大小判断是否要限制字数 if len(re.sub(sub_space, "", soup.get_text())) > limit_words: text_count,recursion_depth,gap,n_soup = soup_limit(soup,text_count,recursion_depth,max_count=limit_words,max_gap=1000,max_recursion_depth=max_recursion_depth) while n_soup: text_count,recursion_depth, gap, n_soup = soup_limit(n_soup, text_count,recursion_depth, max_count=limit_words, max_gap=1000,max_recursion_depth=max_recursion_depth) else: # 有附件 _text = re.sub(sub_space, "", soup.get_text()) _text_split = _text.split("##attachment##") # 正文部分 if len(_text_split[0])>limit_words: main_soup = attachment_part.parent main_text = main_soup.find_all(recursive=False)[0] text_count,recursion_depth, gap, n_soup = soup_limit(main_text, text_count,recursion_depth, max_count=limit_words, max_gap=1000,max_recursion_depth=max_recursion_depth) while n_soup: text_count,recursion_depth, gap, n_soup = soup_limit(n_soup, text_count,recursion_depth, max_count=limit_words, max_gap=1000,max_recursion_depth=max_recursion_depth) # 附件部分 if len(_text_split[1])>limit_words: # attachment_html纯文本,无子结构 if len(attachment_part.find_all(recursive=False))==0: attachment_part.string = str(attachment_part.get_text())[:limit_words] else: attachment_text_nums = 0 attachment_skip = False for part in attachment_part.find_all(recursive=False): if not attachment_skip: if part.name == 'div' and 'filemd5' in part.attrs: if len(part.find_all(recursive=False)) == 0: #无结构的纯文本直接取值 if not attachment_skip: last_attachment_text_nums = attachment_text_nums attachment_text_nums = attachment_text_nums + len(re.sub(sub_space, "", part.get_text())) if attachment_text_nums >=limit_words: part.string = str(part.get_text())[:limit_words - last_attachment_text_nums] attachment_skip = True else: part.decompose() else: for p_part in part.find_all(recursive=False): last_attachment_text_nums = attachment_text_nums attachment_text_nums = attachment_text_nums + len(re.sub(sub_space, "", p_part.get_text())) if not attachment_skip: if attachment_text_nums >= limit_words: p_part.string = str(p_part.get_text())[:limit_words - last_attachment_text_nums] attachment_skip = True else: p_part.decompose() else: last_attachment_text_nums = attachment_text_nums attachment_text_nums = attachment_text_nums + len(re.sub(sub_space, "", part.get_text())) if attachment_text_nums>=limit_words and not attachment_skip: part.string = str(part.get_text())[:limit_words-last_attachment_text_nums] attachment_skip = True else: part.decompose() return soup def attachment_filelink(soup): have_attachment = False attachment_part = None for child in soup.find_all(recursive=True): if child.name == 'div' and 'class' in child.attrs: if "richTextFetch" in child['class']: attachment_part = child have_attachment = True break if not have_attachment: return soup else: # 附件类型:图片、表格 attachment_type = re.compile("\.(?:png|jpg|jpeg|tif|bmp|xlsx|xls)$") attachment_dict = dict() for _attachment in attachment_part.find_all(recursive=False): if _attachment.name == 'div' and 'filemd5' in _attachment.attrs: # print('filemd5',_attachment['filemd5']) attachment_dict[_attachment['filemd5']] = _attachment # print(attachment_dict) for child in soup.find_all(recursive=True): if child.name == 'div' and 'class' in child.attrs: if "richTextFetch" in child['class']: break if "filelink" in child.attrs and child['filelink'] in attachment_dict: if re.search(attachment_type,str(child.string).strip()) or \ ('original' in child.attrs and re.search(attachment_type,str(child['original']).strip())) or \ ('href' in child.attrs and re.search(attachment_type,str(child['href']).strip())): # 附件插入正文标识 child.insert_before("。##attachment_begin##") child.insert_after("。##attachment_end##") child.replace_with(attachment_dict[child['filelink']]) # print('格式化输出',soup.prettify()) return soup def del_achievement(text): if re.search('中标|成交|入围|结果|评标|开标|候选人', text[:500]) == None or re.search('业绩', text) == None: return text p0 = '[,。;]((\d{1,2})|\d{1,2}、)[\w、]{,8}:|((\d{1,2})|\d{1,2}、)|。' # 例子 264392818 p1 = '业绩[:,](\d、[-\w()、]{6,30}(工程|项目|勘察|设计|施工|监理|总承包|采购|更新)[\w()]{,10}[,;])+' # 例子 257717618 p2 = '(类似业绩情况:|业绩:)(\w{,20}:)?(((\d)|\d、)项目名称:[-\w(),;、\d\s:]{5,100}[;。])+' # 例子 264345826 p3 = '(投标|类似|(类似)?项目|合格|有效|企业|工程)?业绩(名称|信息|\d)?:(项目名称:)?[-\w()、]{6,50}(项目|工程|勘察|设计|施工|监理|总承包|采购|更新)' l = [] tmp = [] for it in re.finditer(p0, text): if it.group(0)[-3:] in ['业绩:', '荣誉:']: if tmp != []: del_text = text[tmp[0]:it.start()] l.append(del_text) tmp = [] tmp.append(it.start()) elif tmp != []: del_text = text[tmp[0]:it.start()] l.append(del_text) tmp = [] if tmp != []: del_text = text[tmp[0]:] l.append(del_text) for del_text in l: text = text.replace(del_text, '') # print('删除业绩信息:', del_text) for rs in re.finditer(p1, text): # print('删除业绩信息:', rs.group(0)) text = text.replace(rs.group(0), '') for rs in re.finditer(p2, text): # print('删除业绩信息:', rs.group(0)) text = text.replace(rs.group(0), '') for rs in re.finditer(p3, text): # print('删除业绩信息:', rs.group(0)) text = text.replace(rs.group(0), '') return text def split_header(soup): ''' 处理 空格分割多个表头的情况 : 主要标的名称 规格型号(或服务要求) 主要标的数量 主要标的单价 合同金额(万元) :param soup: bs4 soup 对象 :return: ''' header = [] attrs = [] flag = 0 tag = None for p in soup.find_all('p'): text = p.get_text() if re.search('主要标的数量\s+主要标的单价((万?元))?\s+合同金额', text): header = re.split('\s{3,}', text) if re.search('\s{3,}', text) else re.split('\s+', text) flag = 1 tag = p tag.string = '' continue if flag: attrs = re.split('\s{3,}', text) if re.search('\s{3,}', text) else re.split('\s+', text) if header and len(header) == len(attrs) and tag: s = "" for head, attr in zip(header, attrs): s += head + ':' + attr + ',' # tag.string = s # p.extract() p.string = s else: break