| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348 |
- # -*- coding: utf-8 -*-
- """分句与文本清洗。
- 按 ARCHITECTURE.md Phase 4 拆分建议,从 ``interface/Preprocessing.py`` 迁出。
- 类型:PREPROCESS。
- 原位置:``interface/Preprocessing.py`` 中以下函数:
- - ``segment`` — HTML 文本清洗与标点规范化
- - ``get_preprocessed_sentences`` — 分句与大纲解析
- ``interface/Preprocessing.py`` 仍 re-export 以上全部名称,老 import 不受影响。
- """
- from __future__ import absolute_import
- import re
- import time
- from BiddingKG.dl.common.nerUtils import getTokens
- from BiddingKG.dl.interface.Entitys import Sentences, Outline
- __all__ = [
- "segment",
- "get_preprocessed_sentences",
- ]
- def segment(soup,final=True):
- # print("==")
- # print(soup)
- # print("====")
- #segList = ["tr","div","h1", "h2", "h3", "h4", "h5", "h6", "header"]
- subspaceList = ["td","span","p"] # 20250723去掉 a 标签,角色会被补充a标签
- if soup.name in subspaceList:
- #判断有值叶子节点数
- _count = 0
- for child in soup.find_all(recursive=True):
- if child.get_text().strip()!="" and len(child.find_all())==0:
- _count += 1
- if _count<=1:
- text = soup.get_text()
- # 2020/11/24 大网站规则添加
- if 'title' in soup.attrs:
- if '...' in soup.get_text() and soup.get_text().strip()[:-3] in soup.attrs['title']:
- text = soup.attrs['title']
- _list = []
- for x in re.split("\s+",text):
- if x.strip()!="":
- _list.append(len(x))
- if len(_list)>0:
- _minLength = min(_list)
- if _minLength>2:
- _substr = ","
- else:
- _substr = ""
- else:
- _substr = ""
- text = text.replace("\r\n",",").replace("\n",",")
- text = re.sub("\s+",_substr,text)
- # text = re.sub("\s+","##space##",text)
- return text
- segList = ["title"]
- commaList = ["div","br","td","p","li","h1","h2","h3","h4","h5","h6"]
- #commaList = []
- spaceList = ["span"]
- tbodies = soup.find_all('tbody')
- if len(tbodies) == 0:
- tbodies = soup.find_all('table')
- # 递归遍历所有节点,插入符号
- for child in soup.find_all(recursive=True):
- # print(child.name,child.get_text())
- if child.name in segList:
- child.insert_after("。")
- if child.name in commaList:
- if child.name in ["p", "div"] and re.search('(排名:[\d]+|排名第[一二三四五\d])$', child.get_text().strip()):
- child.insert_after(";")
- else:
- child.insert_after(",")
- # if child.name != "td" and re.match('[((][一二三四五六七八九十]+[))]|[一二三四五六七八九十]+\s*、', child.get_text().strip()): # 大纲前面用句号分割 20240930 注销,修复529501491关键词 三、中标供应商:(一)单位名称被分句
- # child.insert_before("。")
- # if child.name == 'div' and 'class' in child.attrs:
- # # 添加附件"attachment"标识
- # if "richTextFetch" in child['class']:
- # child.insert_before("##attachment##")
- # print(child.parent)
- # if child.name in subspaceList:
- # child.insert_before("#subs"+str(child.name)+"#")
- # child.insert_after("#sube"+str(child.name)+"#")
- # if child.name in spaceList:
- # child.insert_after(" ")
- # if child.name in subspaceList and len(child.get_text()) > 5 and re.search('\w$', child.get_text()): # 20250718 补充逗号避免642261504 分隔不了 联系邮箱:2034758276@qq.com中治交通建设集团有限公司北京分公司2025年6月25日, 20250801 注释掉 653089421 等多篇公告公司被span拆分
- # child.insert_after(",")
- text = str(soup.get_text())
- #替换英文冒号为中文冒号
- text = re.sub("(?<=[\u4e00-\u9fa5]):|:(?=[\u4e00-\u9fa5])",":",text)
- #替换为中文逗号
- text = re.sub("(?<=[\u4e00-\u9fa5]),|,(?=[\u4e00-\u9fa5])",",",text)
- #替换为中文分号
- text = re.sub("(?<=[\u4e00-\u9fa5]);|;(?=[\u4e00-\u9fa5])",";",text)
- # 感叹号替换为中文句号
- text = re.sub("(?<=[\u4e00-\u9fa5])[!!]|[!!](?=[\u4e00-\u9fa5])","。",text)
- #替换格式未识别的问号为" " ,update:2021/7/20
- text = re.sub("[?\?]{2,}|\n"," ",text)
- #替换"""为"“",否则导入deepdive出错
- # text = text.replace('"',"“").replace("\r","").replace("\n",",")
- text = text.replace('"',"“").replace("\r","").replace("\n","").replace("\\n","") #2022/1/4修复 非分段\n 替换为逗号造成 公司拆分 span \n南航\n上海\n分公司
- text = re.sub("( )+"," ",text) # 空白符替换
- # print('==1',text)
- # text = re.sub("\s{4,}",",",text)
- # 解决公告中的" "空格替换问题
- if re.search("\s{4,}",text):
- _text = ""
- for _sent in re.split("。+",text):
- for _sent2 in re.split(',+',_sent):
- for _sent3 in re.split(":+",_sent2):
- pre_t = ''
- for _t in re.split("\s{4,}",_sent3):
- if len(_t)<3 or len(pre_t)<3 or re.search('[^\w\s]$', pre_t): # 20240726 前文小于3字或以符合结尾的不加 避免乱加逗号 例:2) 申请人的资格要求
- _text += _t
- else:
- _text += ","+_t
- pre_t = _t
- _text += ":"
- _text = _text[:-1]
- _text += ","
- _text = _text[:-1]
- _text += "。"
- _text = _text[:-1]
- text = _text
- # print('==2',text)
- #替换标点
- #替换连续的标点
- if final:
- text = re.sub("##space##"," ",text)
- punc_pattern = "(?P<del>[。,;::,\s]+)"
- list_punc = re.findall(punc_pattern,text)
- list_punc.sort(key=lambda x:len(x),reverse=True)
- for punc_del in list_punc:
- if len(punc_del)>1:
- if len(punc_del.strip())>0:
- if ":" in punc_del.strip():
- if "。" in punc_del.strip():
- text = re.sub(punc_del, ":。", text)
- else:
- text = re.sub(punc_del,":",text)
- else:
- text = re.sub(punc_del,punc_del.strip()[0],text) #2021/12/09 修正由于某些标签后插入符号把原来符号替换
- else:
- text = re.sub(punc_del," ",text) # 多个空字符替换为一个空格(防止时间类连接),后面还有对空格处理
- #将连续的中文句号替换为一个
- # text_split = text.split("。")
- # text_split = [x for x in text_split if len(x)>0]
- # text = "。".join(text_split)
- text = re.sub('。+', '。', text).lstrip('。') # 20240703 修复上面的方法造成文末句号丢失问题。
- # #删除标签中的所有空格
- # for subs in subspaceList:
- # patten = "#subs"+str(subs)+"#(.*?)#sube"+str(subs)+"#"
- # while(True):
- # oneMatch = re.search(re.compile(patten),text)
- # if oneMatch is not None:
- # _match = oneMatch.group(1)
- # text = text.replace("#subs"+str(subs)+"#"+_match+"#sube"+str(subs)+"#",_match)
- # else:
- # break
- # text过大报错
- LOOP_LEN = 10000
- LOOP_BEGIN = 0
- _text = ""
- if len(text)<10000000:
- while(LOOP_BEGIN<len(text)):
- _text += re.sub(")",")",re.sub("(","(",re.sub("\s(?!\d{1,2}[::]\d{2}|\d{1,2}[点时])","",text[LOOP_BEGIN:LOOP_BEGIN+LOOP_LEN])))
- LOOP_BEGIN += LOOP_LEN
- text = _text
- # 附件标识前修改为句号,避免正文和附件内容混合在一起
- text = re.sub("[^。](?=##attachment##)","。",text)
- text = re.sub("[^。](?=##attachment_begin##)","。",text)
- text = re.sub("[^。](?=##attachment_end##)","。",text)
- text = re.sub("##attachment_begin##。","##attachment_begin##",text)
- text = re.sub("##attachment_end##。","##attachment_end##",text)
- return text
- def get_preprocessed_sentences(list_articles,useselffool=True,cost_time=dict()):
- '''
- :param list_articles: 经过预处理的article text
- :return: list_sentences
- '''
- list_sentences = []
- list_outlines = []
- for article in list_articles:
- list_sentences_temp = []
- list_entitys_temp = []
- doc_id = article.id
- _send_doc_id = article.doc_id
- _title = article.title
- #表格处理
- key_preprocess = "tableToText"
- start_time = time.time()
- article_processed = article.content
- if len(_title)<100 and _title not in article_processed: # 把标题放到正文
- article_processed = _title + ',' + article_processed # 2023/01/06 标题正文加逗号分割,预防标题后面是产品,正文开头是公司实体,实体识别把产品和公司作为整个角色实体
- attachment_begin_index = -1
- if key_preprocess not in cost_time:
- cost_time[key_preprocess] = 0
- cost_time[key_preprocess] += time.time()-start_time
- #nlp处理
- outline_list = [] # 20240906 修复下面条件不成立时,后面 list_outlines.append(outline_list) 名称未定义报错
- if article_processed is not None and len(article_processed)!=0:
- split_patten = "。"
- sentences = []
- _begin = 0
- sentences_set = set()
- for _iter in re.finditer(split_patten,article_processed):
- _sen = article_processed[_begin:_iter.span()[1]]
- # if len(_sen)>0 and _sen not in sentences_set: # 去重导致内容丢失
- if len(_sen)>0 and (len(sentences)>0 and _sen != sentences[-1] or len(sentences)==0): # 2024/07/25 改为顺序去重
- # 标识在附件里的句子
- if re.search("##attachment##",_sen):
- attachment_begin_index = len(sentences)
- # _sen = re.sub("##attachment##","",_sen)
- sentences.append(_sen)
- sentences_set.add(_sen)
- _begin = _iter.span()[1]
- _sen = article_processed[_begin:]
- if re.search("##attachment##", _sen):
- # _sen = re.sub("##attachment##", "", _sen)
- attachment_begin_index = len(sentences)
- # if len(_sen)>0 and _sen not in sentences_set:
- if len(_sen)>0 and (len(sentences)>0 and _sen != sentences[-1] or len(sentences)==0): # 2024/07/25 改为顺序去重
- sentences.append(_sen)
- sentences_set.add(_sen)
- # 解析outline大纲分段
- outline_list = []
- if re.search("##split##",article.content):
- temp_sentences = []
- last_sentence_index = (-1,-1)
- outline_index = 0
- for sentence_index in range(len(sentences)):
- sentence_text = sentences[sentence_index]
- for _ in re.findall("##split##", sentence_text):
- _match = re.search("##split##", sentence_text)
- if last_sentence_index[0] > -1:
- sentence_begin_index,wordOffset_begin = last_sentence_index
- sentence_end_index = sentence_index
- wordOffset_end = _match.start()
- if sentence_begin_index<attachment_begin_index and sentence_end_index>=attachment_begin_index:
- outline_list.append(Outline(doc_id,outline_index,'',sentence_begin_index,attachment_begin_index-1,wordOffset_begin,len(sentences[attachment_begin_index-1])))
- else:
- outline_list.append(Outline(doc_id,outline_index,'',sentence_begin_index,sentence_end_index,wordOffset_begin,wordOffset_end))
- outline_index += 1
- sentence_text = re.sub("##split##,?", "", sentence_text,count=1)
- last_sentence_index = (sentence_index,_match.start())
- temp_sentences.append(sentence_text)
- if attachment_begin_index>-1 and last_sentence_index[0]<attachment_begin_index:
- outline_list.append(Outline(doc_id,outline_index,'',last_sentence_index[0],attachment_begin_index-1,last_sentence_index[1],len(temp_sentences[attachment_begin_index-1])))
- else:
- outline_list.append(Outline(doc_id,outline_index,'',last_sentence_index[0],len(sentences)-1,last_sentence_index[1],len(temp_sentences[-1])))
- sentences = temp_sentences
- #解析outline的outline_text内容
- for _outline in outline_list:
- if _outline.sentence_begin_index==_outline.sentence_end_index:
- _text = sentences[_outline.sentence_begin_index][_outline.wordOffset_begin:_outline.wordOffset_end]
- else:
- _text = ""
- for idx in range(_outline.sentence_begin_index,_outline.sentence_end_index+1):
- if idx==_outline.sentence_begin_index:
- _text += sentences[idx][_outline.wordOffset_begin:]
- elif idx==_outline.sentence_end_index:
- _text += sentences[idx][:_outline.wordOffset_end]
- else:
- _text += sentences[idx]
- _outline.outline_text = _text
- _outline_summary = re.split("[::,]",_text,1)[0]
- if len(_outline_summary)<30:
- _outline.outline_summary = _outline_summary
- # print(_outline.outline_index,_outline.outline_text)
- article.content = "".join(sentences)
- # sentences.append(article_processed[_begin:])
- article.content = re.sub('[,。\s]+。', '。', article.content) # 处理连续标点
- lemmas = []
- doc_offsets = []
- dep_types = []
- dep_tokens = []
- time1 = time.time()
- '''
- tokens_all = fool.cut(sentences)
- #pos_all = fool.LEXICAL_ANALYSER.pos(tokens_all)
- #ner_tag_all = fool.LEXICAL_ANALYSER.ner_labels(sentences,tokens_all)
- ner_entitys_all = fool.ner(sentences)
- '''
- #限流执行
- key_nerToken = "nerToken"
- start_time = time.time()
- # tokens_all = getTokens(sentences,useselffool=useselffool)
- sentences = [_sen for _sen in sentences if _sen]
- tokens_all = getTokens([re.sub("##attachment_begin##|##attachment_end##","",_sen) for _sen in sentences],useselffool=useselffool)
- if key_nerToken not in cost_time:
- cost_time[key_nerToken] = 0
- cost_time[key_nerToken] += round(time.time()-start_time,2)
- in_attachment = False
- for sentence_index in range(len(sentences)):
- sentence_text = sentences[sentence_index]
- if re.search("##attachment_begin##",sentence_text):
- in_attachment = True
- sentence_text = re.sub("##attachment_begin##","",sentence_text)
- if re.search("##attachment_end##",sentence_text):
- in_attachment = False
- sentence_text = re.sub("##attachment_end##", "", sentence_text)
- if sentence_index >= attachment_begin_index and attachment_begin_index!=-1:
- in_attachment = True
- tokens = tokens_all[sentence_index]
- #pos_tag = pos_all[sentence_index]
- pos_tag = ""
- ner_entitys = ""
- list_sentences_temp.append(Sentences(doc_id=doc_id,sentence_index=sentence_index,sentence_text=sentence_text,tokens=tokens,pos_tags=pos_tag,ner_tags=ner_entitys,in_attachment=in_attachment))
- if len(list_sentences_temp)==0:
- list_sentences_temp.append(Sentences(doc_id=doc_id,sentence_index=0,sentence_text="sentence_text",tokens=[],pos_tags=[],ner_tags=""))
- list_sentences.append(list_sentences_temp)
- list_outlines.append(outline_list)
- article.content = re.sub("##attachment_begin##|##attachment_end##", "", article.content)
- return list_sentences,list_outlines
|