# -*- coding: utf-8 -*- """分句与文本清洗。 按 ARCHITECTURE.md Phase 4 拆分建议,从 ``interface/Preprocessing.py`` 迁出。 类型:PREPROCESS。 原位置:``interface/Preprocessing.py`` 中以下函数: - ``segment`` — HTML 文本清洗与标点规范化 - ``get_preprocessed_sentences`` — 分句与大纲解析 ``interface/Preprocessing.py`` 仍 re-export 以上全部名称,老 import 不受影响。 """ from __future__ import absolute_import import re import time from BiddingKG.dl.common.nerUtils import getTokens from BiddingKG.dl.interface.Entitys import Sentences, Outline __all__ = [ "segment", "get_preprocessed_sentences", ] def segment(soup,final=True): # print("==") # print(soup) # print("====") #segList = ["tr","div","h1", "h2", "h3", "h4", "h5", "h6", "header"] subspaceList = ["td","span","p"] # 20250723去掉 a 标签,角色会被补充a标签 if soup.name in subspaceList: #判断有值叶子节点数 _count = 0 for child in soup.find_all(recursive=True): if child.get_text().strip()!="" and len(child.find_all())==0: _count += 1 if _count<=1: text = soup.get_text() # 2020/11/24 大网站规则添加 if 'title' in soup.attrs: if '...' in soup.get_text() and soup.get_text().strip()[:-3] in soup.attrs['title']: text = soup.attrs['title'] _list = [] for x in re.split("\s+",text): if x.strip()!="": _list.append(len(x)) if len(_list)>0: _minLength = min(_list) if _minLength>2: _substr = "," else: _substr = "" else: _substr = "" text = text.replace("\r\n",",").replace("\n",",") text = re.sub("\s+",_substr,text) # text = re.sub("\s+","##space##",text) return text segList = ["title"] commaList = ["div","br","td","p","li","h1","h2","h3","h4","h5","h6"] #commaList = [] spaceList = ["span"] tbodies = soup.find_all('tbody') if len(tbodies) == 0: tbodies = soup.find_all('table') # 递归遍历所有节点,插入符号 for child in soup.find_all(recursive=True): # print(child.name,child.get_text()) if child.name in segList: child.insert_after("。") if child.name in commaList: if child.name in ["p", "div"] and re.search('(排名:[\d]+|排名第[一二三四五\d])$', child.get_text().strip()): child.insert_after(";") else: child.insert_after(",") # if child.name != "td" and re.match('[((][一二三四五六七八九十]+[))]|[一二三四五六七八九十]+\s*、', child.get_text().strip()): # 大纲前面用句号分割 20240930 注销,修复529501491关键词 三、中标供应商:(一)单位名称被分句 # child.insert_before("。") # if child.name == 'div' and 'class' in child.attrs: # # 添加附件"attachment"标识 # if "richTextFetch" in child['class']: # child.insert_before("##attachment##") # print(child.parent) # if child.name in subspaceList: # child.insert_before("#subs"+str(child.name)+"#") # child.insert_after("#sube"+str(child.name)+"#") # if child.name in spaceList: # child.insert_after(" ") # if child.name in subspaceList and len(child.get_text()) > 5 and re.search('\w$', child.get_text()): # 20250718 补充逗号避免642261504 分隔不了 联系邮箱:2034758276@qq.com中治交通建设集团有限公司北京分公司2025年6月25日, 20250801 注释掉 653089421 等多篇公告公司被span拆分 # child.insert_after(",") text = str(soup.get_text()) #替换英文冒号为中文冒号 text = re.sub("(?<=[\u4e00-\u9fa5]):|:(?=[\u4e00-\u9fa5])",":",text) #替换为中文逗号 text = re.sub("(?<=[\u4e00-\u9fa5]),|,(?=[\u4e00-\u9fa5])",",",text) #替换为中文分号 text = re.sub("(?<=[\u4e00-\u9fa5]);|;(?=[\u4e00-\u9fa5])",";",text) # 感叹号替换为中文句号 text = re.sub("(?<=[\u4e00-\u9fa5])[!!]|[!!](?=[\u4e00-\u9fa5])","。",text) #替换格式未识别的问号为" " ,update:2021/7/20 text = re.sub("[?\?]{2,}|\n"," ",text) #替换"""为"“",否则导入deepdive出错 # text = text.replace('"',"“").replace("\r","").replace("\n",",") text = text.replace('"',"“").replace("\r","").replace("\n","").replace("\\n","") #2022/1/4修复 非分段\n 替换为逗号造成 公司拆分 span \n南航\n上海\n分公司 text = re.sub("( )+"," ",text) # 空白符替换 # print('==1',text) # text = re.sub("\s{4,}",",",text) # 解决公告中的" "空格替换问题 if re.search("\s{4,}",text): _text = "" for _sent in re.split("。+",text): for _sent2 in re.split(',+',_sent): for _sent3 in re.split(":+",_sent2): pre_t = '' for _t in re.split("\s{4,}",_sent3): if len(_t)<3 or len(pre_t)<3 or re.search('[^\w\s]$', pre_t): # 20240726 前文小于3字或以符合结尾的不加 避免乱加逗号 例:2) 申请人的资格要求 _text += _t else: _text += ","+_t pre_t = _t _text += ":" _text = _text[:-1] _text += "," _text = _text[:-1] _text += "。" _text = _text[:-1] text = _text # print('==2',text) #替换标点 #替换连续的标点 if final: text = re.sub("##space##"," ",text) punc_pattern = "(?P[。,;::,\s]+)" list_punc = re.findall(punc_pattern,text) list_punc.sort(key=lambda x:len(x),reverse=True) for punc_del in list_punc: if len(punc_del)>1: if len(punc_del.strip())>0: if ":" in punc_del.strip(): if "。" in punc_del.strip(): text = re.sub(punc_del, ":。", text) else: text = re.sub(punc_del,":",text) else: text = re.sub(punc_del,punc_del.strip()[0],text) #2021/12/09 修正由于某些标签后插入符号把原来符号替换 else: text = re.sub(punc_del," ",text) # 多个空字符替换为一个空格(防止时间类连接),后面还有对空格处理 #将连续的中文句号替换为一个 # text_split = text.split("。") # text_split = [x for x in text_split if len(x)>0] # text = "。".join(text_split) text = re.sub('。+', '。', text).lstrip('。') # 20240703 修复上面的方法造成文末句号丢失问题。 # #删除标签中的所有空格 # for subs in subspaceList: # patten = "#subs"+str(subs)+"#(.*?)#sube"+str(subs)+"#" # while(True): # oneMatch = re.search(re.compile(patten),text) # if oneMatch is not None: # _match = oneMatch.group(1) # text = text.replace("#subs"+str(subs)+"#"+_match+"#sube"+str(subs)+"#",_match) # else: # break # text过大报错 LOOP_LEN = 10000 LOOP_BEGIN = 0 _text = "" if len(text)<10000000: while(LOOP_BEGIN0 and _sen not in sentences_set: # 去重导致内容丢失 if len(_sen)>0 and (len(sentences)>0 and _sen != sentences[-1] or len(sentences)==0): # 2024/07/25 改为顺序去重 # 标识在附件里的句子 if re.search("##attachment##",_sen): attachment_begin_index = len(sentences) # _sen = re.sub("##attachment##","",_sen) sentences.append(_sen) sentences_set.add(_sen) _begin = _iter.span()[1] _sen = article_processed[_begin:] if re.search("##attachment##", _sen): # _sen = re.sub("##attachment##", "", _sen) attachment_begin_index = len(sentences) # if len(_sen)>0 and _sen not in sentences_set: if len(_sen)>0 and (len(sentences)>0 and _sen != sentences[-1] or len(sentences)==0): # 2024/07/25 改为顺序去重 sentences.append(_sen) sentences_set.add(_sen) # 解析outline大纲分段 outline_list = [] if re.search("##split##",article.content): temp_sentences = [] last_sentence_index = (-1,-1) outline_index = 0 for sentence_index in range(len(sentences)): sentence_text = sentences[sentence_index] for _ in re.findall("##split##", sentence_text): _match = re.search("##split##", sentence_text) if last_sentence_index[0] > -1: sentence_begin_index,wordOffset_begin = last_sentence_index sentence_end_index = sentence_index wordOffset_end = _match.start() if sentence_begin_index=attachment_begin_index: outline_list.append(Outline(doc_id,outline_index,'',sentence_begin_index,attachment_begin_index-1,wordOffset_begin,len(sentences[attachment_begin_index-1]))) else: outline_list.append(Outline(doc_id,outline_index,'',sentence_begin_index,sentence_end_index,wordOffset_begin,wordOffset_end)) outline_index += 1 sentence_text = re.sub("##split##,?", "", sentence_text,count=1) last_sentence_index = (sentence_index,_match.start()) temp_sentences.append(sentence_text) if attachment_begin_index>-1 and last_sentence_index[0]= attachment_begin_index and attachment_begin_index!=-1: in_attachment = True tokens = tokens_all[sentence_index] #pos_tag = pos_all[sentence_index] pos_tag = "" ner_entitys = "" list_sentences_temp.append(Sentences(doc_id=doc_id,sentence_index=sentence_index,sentence_text=sentence_text,tokens=tokens,pos_tags=pos_tag,ner_tags=ner_entitys,in_attachment=in_attachment)) if len(list_sentences_temp)==0: list_sentences_temp.append(Sentences(doc_id=doc_id,sentence_index=0,sentence_text="sentence_text",tokens=[],pos_tags=[],ner_tags="")) list_sentences.append(list_sentences_temp) list_outlines.append(outline_list) article.content = re.sub("##attachment_begin##|##attachment_end##", "", article.content) return list_sentences,list_outlines