# -*- coding: utf-8 -*- """大纲提取与分句。 按 ARCHITECTURE.md Phase 4 拆分建议,合并 ``interface/outline_extractor.py`` 与 ``interface/Preprocessing.py`` 中的大纲相关函数。 类型:PREPROCESS。 来源: - ``interface/Preprocessing.py``: ``get_preprocessed_outline``, ``change2num``, ``re_num``, ``num_dict`` - ``interface/outline_extractor.py``: ``Sentence2``, ``extract_sentence_list``, patterns, ``extract_parameters``, ``extract_addr`` ``interface/Preprocessing.py`` 和 ``interface/outline_extractor.py`` 仍 re-export 以上全部名称。 """ from __future__ import absolute_import import re from BiddingKG.dl.interface.htmlparser import ParseDocument, get_childs __all__ = [ "re_num", "num_dict", "change2num", "get_preprocessed_outline", "Sentence2", "extract_sentence_list", "requirement_pattern", "winter_pattern", "aptitude_pattern", "addr_bidopen_pattern", "addr_bidsend_pattern", "pinmu_name_pattern", "policy_pattern", "not_policy_pattern", "correction_pattern", "extract_parameters", "extract_addr", ] re_num = re.compile("[二三四五六七八九]十[一二三四五六七八九]?|十[一二三四五六七八九]|[一二三四五六七八九十]") num_dict = { "一": 1, "二": 2, "三": 3, "四": 4, "五": 5, "六": 6, "七": 7, "八": 8, "九": 9, "十": 10} # 一百以内的中文大写转换为数字 def change2num(text): result_num = -1 # text = text[:6] match = re_num.search(text) if match: _num = match.group() if num_dict.get(_num): return num_dict.get(_num) else: tenths = 1 the_unit = 0 num_split = _num.split("十") if num_dict.get(num_split[0]): tenths = num_dict.get(num_split[0]) if num_dict.get(num_split[1]): the_unit = num_dict.get(num_split[1]) result_num = tenths * 10 + the_unit elif re.search("\d{1,2}",text): _num = re.search("\d{1,2}",text).group() result_num = int(_num) return result_num #大纲分段处理 def get_preprocessed_outline(soup): pattern_0 = re.compile("^(?:[二三四五六七八九]十[一二三四五六七八九]?|十[一二三四五六七八九]|[一二三四五六七八九十])[、.\.]") pattern_1 = re.compile("^[\((]?(?:[二三四五六七八九]十[一二三四五六七八九]?|十[一二三四五六七八九]|[一二三四五六七八九十])[\))]") pattern_2 = re.compile("^\d{1,2}[、.\.](?=[^\d]{1,2}|$)") pattern_3 = re.compile("^[\((]?\d{1,2}[\))]") pattern_list = [pattern_0, pattern_1, pattern_2, pattern_3] body = soup.find("body") if body == None: return soup # 修复 无body的报错 例子:264419050 body_child = body.find_all(recursive=False) deal_part = body # print(body_child[0]['id']) if 'id' in body_child[0].attrs: if len(body_child) <= 2 and body_child[0]['id'] == 'pcontent': deal_part = body_child[0] if len(deal_part.find_all(recursive=False))>2: deal_part = deal_part.parent skip_tag = ['turntable', 'tbody', 'th', 'tr', 'td', 'table','thead','tfoot'] for part in deal_part.find_all(recursive=False): # 查找解析文本的主干部分 is_main_text = False through_text_num = 0 while (not is_main_text and part.find_all(recursive=False)): while len(part.find_all(recursive=False)) == 1 and part.get_text(strip=True) == \ part.find_all(recursive=False)[0].get_text(strip=True): part = part.find_all(recursive=False)[0] max_len = len(part.get_text(strip=True)) is_main_text = True for t_part in part.find_all(recursive=False): if t_part.name not in skip_tag and t_part.get_text(strip=True)!="": through_text_num += 1 if t_part.get_text(strip=True)!="" and len(t_part.get_text(strip=True))/max_len>=0.65: if t_part.name not in skip_tag: is_main_text = False part = t_part break else: while len(t_part.find_all(recursive=False)) == 1 and t_part.get_text(strip=True) == \ t_part.find_all(recursive=False)[0].get_text(strip=True): t_part = t_part.find_all(recursive=False)[0] if through_text_num>2: is_table = True for _t_part in t_part.find_all(recursive=False): if _t_part.name not in skip_tag: is_table = False break if not is_table: is_main_text = False part = t_part break else: is_main_text = False part = t_part break is_find = False for _pattern in pattern_list: last_index = 0 handle_list = [] for _part in part.find_all(recursive=False): if _part.name not in skip_tag and _part.get_text(strip=True) != "": # print('text:', _part.get_text(strip=True)) re_match = re.search(_pattern, _part.get_text(strip=True)) if re_match: outline_index = change2num(re_match.group()) if last_index < outline_index: # _part.insert_before("##split##") handle_list.append(_part) last_index = outline_index if len(handle_list)>1: is_find = True for _part in handle_list: _part.insert_before("##split##") if is_find: break # print(soup) return soup class Sentence2(): def __init__(self,text,sentence_index,wordOffset_begin,wordOffset_end): self.name = 'sentence2' self.text = text self.sentence_index = sentence_index self.wordOffset_begin = wordOffset_begin self.wordOffset_end = wordOffset_end def get_text(self): return self.text def extract_sentence_list(sentence_list): new_sentence2_list = [] new_sentence2_list_attach = [] for sentence in sentence_list: sentence_index = sentence.sentence_index sentence_text = sentence.sentence_text begin_index = 0 end_index = 0 for it in re.finditer('([^一二三四五六七八九十,。][一二三四五六七八九十]{1,3}|[^\d\.、,。a-zA-Z]\d{1,2}(\.\d{1,2}){,2})、', sentence_text): # 例:289699210 1、招标内容:滑触线及配件2、招标品牌:3、参标供应商经营形式要求:厂家4、参标供应商资质要求:5、 temp = it.group(0) sentence_text = sentence_text.replace(temp, temp[0] + ',' + temp[1:]) for item in re.finditer('[,。:;;!!?]+', sentence_text): # 20240725去掉英文问号,避免网址被分隔 end_index = item.end() # if end_index!=len(sentence_text): # # if end_index-begin_index<6 and item.group(0) in [',', ';', ';'] and re.match('[一二三四五六七八九十\d.]+、', sentence_text[begin_index:end_index])==None: # 20240725 注销,避免标题提取错误 # # continue if end_index != len(sentence_text) and re.match('[一二三四五六七八九十\d.]{1,2}[、,.]+$', sentence_text[begin_index:end_index]): # 避免表格序号和内容在不同表格情况 例:293178161 continue new_sentence_text = sentence_text[begin_index:end_index] sentence2 = Sentence2(new_sentence_text,sentence_index,begin_index,end_index) if sentence.in_attachment: new_sentence2_list_attach.append(sentence2) else: new_sentence2_list.append(sentence2) begin_index = end_index if end_index!=len(sentence_text): end_index = len(sentence_text) new_sentence_text = sentence_text[begin_index:end_index] sentence2 = Sentence2(new_sentence_text, sentence_index, begin_index, end_index) if sentence.in_attachment: new_sentence2_list_attach.append(sentence2) else: new_sentence2_list.append(sentence2) return new_sentence2_list, new_sentence2_list_attach requirement_pattern = "(采购需求|需求分析|项目说明|(采购|合同|招标|询比?价|项目|服务|工程|标的|需求|建设)(的?(主要|简要|基本|具体|名称及))?" \ "(内容|概况|概述|范围|信息|规模|简介|介绍|说明|摘要|情况)([及与和]((其它|\w{,2})[要需]求|发包范围|数量))?" \ "|招标项目技术要求|服务要求|服务需求|项目目标|需求内容如下|建设规模)为?([::,]|$)" winter_pattern = "((乙方|竞得|受让|买受|签约|供货|供应|承做|承包|承建|承销|承保|承接|承制|承担|承修|承租(?:(包))?|入围|入选|竞买|中标|中选|中价|中签|成交|候选)[\u4e00-\u9fa5]{0,5}" \ "(公示)?(信息|概况|情况|名称|联系人|联系方式|负责人)|中标公示单位|合同主体)为?([::,、]|$)" aptitude_pattern = "资质(资格)要求|资格(资质)要求|单位要求|资质及业绩要求|((资格|资质|准入)[的及]?(要求|条件|标准|限定|门槛)|竞买资格及要求|供应商报价须知)|按以下要求参与竞买|((报名|应征|竞买|投标|竞投|受让|报价|竞价|竞包|竞租|承租|申请|参与|参选|遴选)的?(人|方|单位|企业|客户|机构)?|供应商|受让方)((必?须|需|应[该当]?)(具备|满足|符合|提供)+以?下?)?的?(一般|基本|主要)?(条件|要求|资格(能力)?|资质)+|乙方应当符合下列要求|参与比选条件|合格的投标人|询价要求" addr_bidopen_pattern = "([开评]标|开启|评选|比选|磋商|遴选|寻源|采购|招标|竞价|议价|委托|询比?价|比价|谈判|邀标|邀请|洽谈|约谈|选取|抽取|抽选|递交\w{,4}文件)[))]?(时间[与及和、])?(地址|地点)([与及和、]时间)?([::,]|$)|开启([::,]|$)" addr_bidsend_pattern = "((\w{,4}文件)?(提交|递交)(\w{,4}文件)?|投标)(截止时间[与及和、])?地[点址]([与及和、]截止时间)?([::,]|$)" pinmu_name_pattern = "采购品目(名称)?([::,]|$)" policy_pattern = "《.+?(通知|办法|条例|规定|规程|规范|须知|规则|标准|细则|意见|协议|条件|要求|手册|法典|方案|指南|指引|法)》" not_policy_pattern = "(表|函|书|证|\d页|公告|合同|文件|清单)》$|采购合同|响应方须知|响应文件格式|营业执照|开标一览|采购需求" correction_pattern = "(更正|更改|修正|修改|变更|延期)(信息|内容|事项|详情)" def extract_parameters(parse_document): ''' 通过大纲、预处理后文本正则获取需要字段 :param parse_document: ParseDocument() 方法返回结果 :return: ''' list_data = parse_document.tree requirement_text = '' # 采购内容 aptitude_text = '' # 资质要求 addr_bidopen_text = '' # 开标地址 addr_bidsend_text = '' # 投标地址 requirement_scope = [] # 采购内容始末位置 winter_scope = [] # 中标信息始末位置 pinmu_name = '' # 品目名称 list_policy = [] # 政策法规 correction_content = "" # 更正内容 out_lines = [] _find_count = 0 _data_i = -1 while _data_i= 4 else _text # out_lines.append((outline_text, _data['sentence_index'], _data['wordOffset_begin'])) childs = get_childs([_data]) if len(childs) > 0: scope = ((childs[0]['sentence_index'], childs[0]['wordOffset_begin']), (childs[-1]['sentence_index'], childs[-1]['wordOffset_end'])) else: scope = ((_data['sentence_index'], _data['wordOffset_begin']), (_data['sentence_index'], _data['wordOffset_end'])) out_lines.append((outline_text, _data['sentence_title'], _data['title_index'], _data['next_index'], scope)) if re.search(requirement_pattern,_text[:30]) is not None and re.search('符合采购需求,', _text[:30])==None: b = (_data['sentence_index'], _data['wordOffset_begin']) childs = get_childs([_data]) for c in childs: # requirement_text += c["text"]+"\n" requirement_text += c["text"] e = (c['sentence_index'], c["wordOffset_end"]) if len(childs)>0 else (_data['sentence_index'], _data['wordOffset_end']) requirement_scope.append(b) requirement_scope.append(e) _data_i += len(childs) _data_i -= 1 _data_i = -1 # 中标信息 while _data_i0 else (_data['sentence_index'], _data['wordOffset_end']) winter_scope.append(b) winter_scope.append(e) _data_i += len(childs) _data_i -= 1 _data_i = -1 # 更正内容 while _data_i < len(list_data) - 1: _data_i += 1 _data = list_data[_data_i] _type = _data["type"] _text = _data["text"].strip() if _type == "sentence": if _data["sentence_title"] is not None: if re.search(correction_pattern, _text[:20]) is not None: childs = get_childs([_data]) correction_text = "" for c in childs: correction_text += c["text"].strip() # print('correction_text',correction_text) correction_content += correction_text _data_i += len(childs) _data_i -= 1 _data_i = -1 while _data_i120 and re.search(aptitude_pattern,cell_text) is not None: aptitude_text += cell_text+"\n" _data_i = -1 while _data_i < len(list_data) - 1: _data_i += 1 _data = list_data[_data_i] _type = _data["type"] _text = _data["text"].strip() # print(_data.keys()) if _type == "sentence": if _data["sentence_title"] is not None: if re.search(addr_bidopen_pattern, _text[:20]) is not None: childs = get_childs([_data], max_depth=1) for c in childs: addr_bidopen_text += c["text"] _data_i += len(childs) _data_i -= 1 elif re.search(addr_bidsend_pattern, _text[:20]): childs = get_childs([_data], max_depth=1) for c in childs: addr_bidsend_text += c["text"] _data_i += len(childs) _data_i -= 1 elif re.search(pinmu_name_pattern, _text): childs = get_childs([_data], max_depth=1) for c in childs: pinmu_name += c["text"] _data_i += len(childs) _data_i -= 1 _data_i = -1 while _data_i([\w()()【】]{2,25}([省市县区州旗]|采购网|平台|公司)[\w()()【】-]{,60}))[,。]', addr_bidopen_text) addr_bidopen_text = ser.group('addr') if ser else '' ser = re.search('地[址点][:为](?P([\w()()【】]{2,25}([省市县区州旗]|采购网|平台|公司)[\w()()【】-]{,60}))[,。]', addr_bidsend_text) addr_bidsend_text = ser.group('addr') if ser else '' if re.search('开启', addr_bidopen_text) and re.search('时间:\d{2,4}年\d{1,2}月\d{1,2}日', addr_bidopen_text) and len(addr_bidopen_text)<40: # 优化类似 364991684只有时间没地址情况 addr_bidopen_text = "" ser = re.search(pinmu_name_pattern, pinmu_name) if ser: pinmu_name = pinmu_name[ser.end():] if re.search('[^\w]$', pinmu_name): pinmu_name = pinmu_name[:-1] if len(out_lines) < 3: # 小于三个的大纲去掉 out_lines = [] # else: # text_, title_type, title_index, next_index, scope = out_lines[-1] # if scope[0][0] < scope[1][0]:# 最后一个大纲范围取当句,避免错误 # out_lines[-1] = (text_, title_type, title_index, next_index,(scope[0], (scope[0][0]+1, 0))) return requirement_text, aptitude_text, addr_bidopen_text, addr_bidsend_text, out_lines, requirement_scope, pinmu_name, list_policy, winter_scope,correction_content def extract_addr(content): ''' 通过正则提取地址 :param content: 公告预处理后文本 :return: ''' addr_bidopen_text = '' ser = re.search('([开评]标|开启|评选|比选|磋商|遴选|寻源|采购|招标|竞价|议价|委托|询比?价|比价|谈判|邀标|邀请|洽谈|约谈|选取|抽取|抽选|递交\w{,4}文件))?(会议)?地[点址]([((]网址[))])?[:为][^,;。]{2,100}[,;。]', content) if ser: addr_bidopen_text = ser.group(0) return addr_bidopen_text