segmenter.py 16 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348
  1. # -*- coding: utf-8 -*-
  2. """分句与文本清洗。
  3. 按 ARCHITECTURE.md Phase 4 拆分建议,从 ``interface/Preprocessing.py`` 迁出。
  4. 类型:PREPROCESS。
  5. 原位置:``interface/Preprocessing.py`` 中以下函数:
  6. - ``segment`` — HTML 文本清洗与标点规范化
  7. - ``get_preprocessed_sentences`` — 分句与大纲解析
  8. ``interface/Preprocessing.py`` 仍 re-export 以上全部名称,老 import 不受影响。
  9. """
  10. from __future__ import absolute_import
  11. import re
  12. import time
  13. from BiddingKG.dl.common.nerUtils import getTokens
  14. from BiddingKG.dl.interface.Entitys import Sentences, Outline
  15. __all__ = [
  16. "segment",
  17. "get_preprocessed_sentences",
  18. ]
  19. def segment(soup,final=True):
  20. # print("==")
  21. # print(soup)
  22. # print("====")
  23. #segList = ["tr","div","h1", "h2", "h3", "h4", "h5", "h6", "header"]
  24. subspaceList = ["td","span","p"] # 20250723去掉 a 标签,角色会被补充a标签
  25. if soup.name in subspaceList:
  26. #判断有值叶子节点数
  27. _count = 0
  28. for child in soup.find_all(recursive=True):
  29. if child.get_text().strip()!="" and len(child.find_all())==0:
  30. _count += 1
  31. if _count<=1:
  32. text = soup.get_text()
  33. # 2020/11/24 大网站规则添加
  34. if 'title' in soup.attrs:
  35. if '...' in soup.get_text() and soup.get_text().strip()[:-3] in soup.attrs['title']:
  36. text = soup.attrs['title']
  37. _list = []
  38. for x in re.split("\s+",text):
  39. if x.strip()!="":
  40. _list.append(len(x))
  41. if len(_list)>0:
  42. _minLength = min(_list)
  43. if _minLength>2:
  44. _substr = ","
  45. else:
  46. _substr = ""
  47. else:
  48. _substr = ""
  49. text = text.replace("\r\n",",").replace("\n",",")
  50. text = re.sub("\s+",_substr,text)
  51. # text = re.sub("\s+","##space##",text)
  52. return text
  53. segList = ["title"]
  54. commaList = ["div","br","td","p","li","h1","h2","h3","h4","h5","h6"]
  55. #commaList = []
  56. spaceList = ["span"]
  57. tbodies = soup.find_all('tbody')
  58. if len(tbodies) == 0:
  59. tbodies = soup.find_all('table')
  60. # 递归遍历所有节点,插入符号
  61. for child in soup.find_all(recursive=True):
  62. # print(child.name,child.get_text())
  63. if child.name in segList:
  64. child.insert_after("。")
  65. if child.name in commaList:
  66. if child.name in ["p", "div"] and re.search('(排名:[\d]+|排名第[一二三四五\d])$', child.get_text().strip()):
  67. child.insert_after(";")
  68. else:
  69. child.insert_after(",")
  70. # if child.name != "td" and re.match('[((][一二三四五六七八九十]+[))]|[一二三四五六七八九十]+\s*、', child.get_text().strip()): # 大纲前面用句号分割 20240930 注销,修复529501491关键词 三、中标供应商:(一)单位名称被分句
  71. # child.insert_before("。")
  72. # if child.name == 'div' and 'class' in child.attrs:
  73. # # 添加附件"attachment"标识
  74. # if "richTextFetch" in child['class']:
  75. # child.insert_before("##attachment##")
  76. # print(child.parent)
  77. # if child.name in subspaceList:
  78. # child.insert_before("#subs"+str(child.name)+"#")
  79. # child.insert_after("#sube"+str(child.name)+"#")
  80. # if child.name in spaceList:
  81. # child.insert_after(" ")
  82. # if child.name in subspaceList and len(child.get_text()) > 5 and re.search('\w$', child.get_text()): # 20250718 补充逗号避免642261504 分隔不了 联系邮箱:2034758276@qq.com中治交通建设集团有限公司北京分公司2025年6月25日, 20250801 注释掉 653089421 等多篇公告公司被span拆分
  83. # child.insert_after(",")
  84. text = str(soup.get_text())
  85. #替换英文冒号为中文冒号
  86. text = re.sub("(?<=[\u4e00-\u9fa5]):|:(?=[\u4e00-\u9fa5])",":",text)
  87. #替换为中文逗号
  88. text = re.sub("(?<=[\u4e00-\u9fa5]),|,(?=[\u4e00-\u9fa5])",",",text)
  89. #替换为中文分号
  90. text = re.sub("(?<=[\u4e00-\u9fa5]);|;(?=[\u4e00-\u9fa5])",";",text)
  91. # 感叹号替换为中文句号
  92. text = re.sub("(?<=[\u4e00-\u9fa5])[!!]|[!!](?=[\u4e00-\u9fa5])","。",text)
  93. #替换格式未识别的问号为" " ,update:2021/7/20
  94. text = re.sub("[?\?]{2,}|\n"," ",text)
  95. #替换"""为"“",否则导入deepdive出错
  96. # text = text.replace('"',"“").replace("\r","").replace("\n",",")
  97. text = text.replace('"',"“").replace("\r","").replace("\n","").replace("\\n","") #2022/1/4修复 非分段\n 替换为逗号造成 公司拆分 span \n南航\n上海\n分公司
  98. text = re.sub("(&nbsp)+"," ",text) # 空白符替换
  99. # print('==1',text)
  100. # text = re.sub("\s{4,}",",",text)
  101. # 解决公告中的" "空格替换问题
  102. if re.search("\s{4,}",text):
  103. _text = ""
  104. for _sent in re.split("。+",text):
  105. for _sent2 in re.split(',+',_sent):
  106. for _sent3 in re.split(":+",_sent2):
  107. pre_t = ''
  108. for _t in re.split("\s{4,}",_sent3):
  109. if len(_t)<3 or len(pre_t)<3 or re.search('[^\w\s]$', pre_t): # 20240726 前文小于3字或以符合结尾的不加 避免乱加逗号 例:2) 申请人的资格要求
  110. _text += _t
  111. else:
  112. _text += ","+_t
  113. pre_t = _t
  114. _text += ":"
  115. _text = _text[:-1]
  116. _text += ","
  117. _text = _text[:-1]
  118. _text += "。"
  119. _text = _text[:-1]
  120. text = _text
  121. # print('==2',text)
  122. #替换标点
  123. #替换连续的标点
  124. if final:
  125. text = re.sub("##space##"," ",text)
  126. punc_pattern = "(?P<del>[。,;::,\s]+)"
  127. list_punc = re.findall(punc_pattern,text)
  128. list_punc.sort(key=lambda x:len(x),reverse=True)
  129. for punc_del in list_punc:
  130. if len(punc_del)>1:
  131. if len(punc_del.strip())>0:
  132. if ":" in punc_del.strip():
  133. if "。" in punc_del.strip():
  134. text = re.sub(punc_del, ":。", text)
  135. else:
  136. text = re.sub(punc_del,":",text)
  137. else:
  138. text = re.sub(punc_del,punc_del.strip()[0],text) #2021/12/09 修正由于某些标签后插入符号把原来符号替换
  139. else:
  140. text = re.sub(punc_del," ",text) # 多个空字符替换为一个空格(防止时间类连接),后面还有对空格处理
  141. #将连续的中文句号替换为一个
  142. # text_split = text.split("。")
  143. # text_split = [x for x in text_split if len(x)>0]
  144. # text = "。".join(text_split)
  145. text = re.sub('。+', '。', text).lstrip('。') # 20240703 修复上面的方法造成文末句号丢失问题。
  146. # #删除标签中的所有空格
  147. # for subs in subspaceList:
  148. # patten = "#subs"+str(subs)+"#(.*?)#sube"+str(subs)+"#"
  149. # while(True):
  150. # oneMatch = re.search(re.compile(patten),text)
  151. # if oneMatch is not None:
  152. # _match = oneMatch.group(1)
  153. # text = text.replace("#subs"+str(subs)+"#"+_match+"#sube"+str(subs)+"#",_match)
  154. # else:
  155. # break
  156. # text过大报错
  157. LOOP_LEN = 10000
  158. LOOP_BEGIN = 0
  159. _text = ""
  160. if len(text)<10000000:
  161. while(LOOP_BEGIN<len(text)):
  162. _text += re.sub(")",")",re.sub("(","(",re.sub("\s(?!\d{1,2}[::]\d{2}|\d{1,2}[点时])","",text[LOOP_BEGIN:LOOP_BEGIN+LOOP_LEN])))
  163. LOOP_BEGIN += LOOP_LEN
  164. text = _text
  165. # 附件标识前修改为句号,避免正文和附件内容混合在一起
  166. text = re.sub("[^。](?=##attachment##)","。",text)
  167. text = re.sub("[^。](?=##attachment_begin##)","。",text)
  168. text = re.sub("[^。](?=##attachment_end##)","。",text)
  169. text = re.sub("##attachment_begin##。","##attachment_begin##",text)
  170. text = re.sub("##attachment_end##。","##attachment_end##",text)
  171. return text
  172. def get_preprocessed_sentences(list_articles,useselffool=True,cost_time=dict()):
  173. '''
  174. :param list_articles: 经过预处理的article text
  175. :return: list_sentences
  176. '''
  177. list_sentences = []
  178. list_outlines = []
  179. for article in list_articles:
  180. list_sentences_temp = []
  181. list_entitys_temp = []
  182. doc_id = article.id
  183. _send_doc_id = article.doc_id
  184. _title = article.title
  185. #表格处理
  186. key_preprocess = "tableToText"
  187. start_time = time.time()
  188. article_processed = article.content
  189. if len(_title)<100 and _title not in article_processed: # 把标题放到正文
  190. article_processed = _title + ',' + article_processed # 2023/01/06 标题正文加逗号分割,预防标题后面是产品,正文开头是公司实体,实体识别把产品和公司作为整个角色实体
  191. attachment_begin_index = -1
  192. if key_preprocess not in cost_time:
  193. cost_time[key_preprocess] = 0
  194. cost_time[key_preprocess] += time.time()-start_time
  195. #nlp处理
  196. outline_list = [] # 20240906 修复下面条件不成立时,后面 list_outlines.append(outline_list) 名称未定义报错
  197. if article_processed is not None and len(article_processed)!=0:
  198. split_patten = "。"
  199. sentences = []
  200. _begin = 0
  201. sentences_set = set()
  202. for _iter in re.finditer(split_patten,article_processed):
  203. _sen = article_processed[_begin:_iter.span()[1]]
  204. # if len(_sen)>0 and _sen not in sentences_set: # 去重导致内容丢失
  205. if len(_sen)>0 and (len(sentences)>0 and _sen != sentences[-1] or len(sentences)==0): # 2024/07/25 改为顺序去重
  206. # 标识在附件里的句子
  207. if re.search("##attachment##",_sen):
  208. attachment_begin_index = len(sentences)
  209. # _sen = re.sub("##attachment##","",_sen)
  210. sentences.append(_sen)
  211. sentences_set.add(_sen)
  212. _begin = _iter.span()[1]
  213. _sen = article_processed[_begin:]
  214. if re.search("##attachment##", _sen):
  215. # _sen = re.sub("##attachment##", "", _sen)
  216. attachment_begin_index = len(sentences)
  217. # if len(_sen)>0 and _sen not in sentences_set:
  218. if len(_sen)>0 and (len(sentences)>0 and _sen != sentences[-1] or len(sentences)==0): # 2024/07/25 改为顺序去重
  219. sentences.append(_sen)
  220. sentences_set.add(_sen)
  221. # 解析outline大纲分段
  222. outline_list = []
  223. if re.search("##split##",article.content):
  224. temp_sentences = []
  225. last_sentence_index = (-1,-1)
  226. outline_index = 0
  227. for sentence_index in range(len(sentences)):
  228. sentence_text = sentences[sentence_index]
  229. for _ in re.findall("##split##", sentence_text):
  230. _match = re.search("##split##", sentence_text)
  231. if last_sentence_index[0] > -1:
  232. sentence_begin_index,wordOffset_begin = last_sentence_index
  233. sentence_end_index = sentence_index
  234. wordOffset_end = _match.start()
  235. if sentence_begin_index<attachment_begin_index and sentence_end_index>=attachment_begin_index:
  236. outline_list.append(Outline(doc_id,outline_index,'',sentence_begin_index,attachment_begin_index-1,wordOffset_begin,len(sentences[attachment_begin_index-1])))
  237. else:
  238. outline_list.append(Outline(doc_id,outline_index,'',sentence_begin_index,sentence_end_index,wordOffset_begin,wordOffset_end))
  239. outline_index += 1
  240. sentence_text = re.sub("##split##,?", "", sentence_text,count=1)
  241. last_sentence_index = (sentence_index,_match.start())
  242. temp_sentences.append(sentence_text)
  243. if attachment_begin_index>-1 and last_sentence_index[0]<attachment_begin_index:
  244. outline_list.append(Outline(doc_id,outline_index,'',last_sentence_index[0],attachment_begin_index-1,last_sentence_index[1],len(temp_sentences[attachment_begin_index-1])))
  245. else:
  246. outline_list.append(Outline(doc_id,outline_index,'',last_sentence_index[0],len(sentences)-1,last_sentence_index[1],len(temp_sentences[-1])))
  247. sentences = temp_sentences
  248. #解析outline的outline_text内容
  249. for _outline in outline_list:
  250. if _outline.sentence_begin_index==_outline.sentence_end_index:
  251. _text = sentences[_outline.sentence_begin_index][_outline.wordOffset_begin:_outline.wordOffset_end]
  252. else:
  253. _text = ""
  254. for idx in range(_outline.sentence_begin_index,_outline.sentence_end_index+1):
  255. if idx==_outline.sentence_begin_index:
  256. _text += sentences[idx][_outline.wordOffset_begin:]
  257. elif idx==_outline.sentence_end_index:
  258. _text += sentences[idx][:_outline.wordOffset_end]
  259. else:
  260. _text += sentences[idx]
  261. _outline.outline_text = _text
  262. _outline_summary = re.split("[::,]",_text,1)[0]
  263. if len(_outline_summary)<30:
  264. _outline.outline_summary = _outline_summary
  265. # print(_outline.outline_index,_outline.outline_text)
  266. article.content = "".join(sentences)
  267. # sentences.append(article_processed[_begin:])
  268. article.content = re.sub('[,。\s]+。', '。', article.content) # 处理连续标点
  269. lemmas = []
  270. doc_offsets = []
  271. dep_types = []
  272. dep_tokens = []
  273. time1 = time.time()
  274. '''
  275. tokens_all = fool.cut(sentences)
  276. #pos_all = fool.LEXICAL_ANALYSER.pos(tokens_all)
  277. #ner_tag_all = fool.LEXICAL_ANALYSER.ner_labels(sentences,tokens_all)
  278. ner_entitys_all = fool.ner(sentences)
  279. '''
  280. #限流执行
  281. key_nerToken = "nerToken"
  282. start_time = time.time()
  283. # tokens_all = getTokens(sentences,useselffool=useselffool)
  284. sentences = [_sen for _sen in sentences if _sen]
  285. tokens_all = getTokens([re.sub("##attachment_begin##|##attachment_end##","",_sen) for _sen in sentences],useselffool=useselffool)
  286. if key_nerToken not in cost_time:
  287. cost_time[key_nerToken] = 0
  288. cost_time[key_nerToken] += round(time.time()-start_time,2)
  289. in_attachment = False
  290. for sentence_index in range(len(sentences)):
  291. sentence_text = sentences[sentence_index]
  292. if re.search("##attachment_begin##",sentence_text):
  293. in_attachment = True
  294. sentence_text = re.sub("##attachment_begin##","",sentence_text)
  295. if re.search("##attachment_end##",sentence_text):
  296. in_attachment = False
  297. sentence_text = re.sub("##attachment_end##", "", sentence_text)
  298. if sentence_index >= attachment_begin_index and attachment_begin_index!=-1:
  299. in_attachment = True
  300. tokens = tokens_all[sentence_index]
  301. #pos_tag = pos_all[sentence_index]
  302. pos_tag = ""
  303. ner_entitys = ""
  304. list_sentences_temp.append(Sentences(doc_id=doc_id,sentence_index=sentence_index,sentence_text=sentence_text,tokens=tokens,pos_tags=pos_tag,ner_tags=ner_entitys,in_attachment=in_attachment))
  305. if len(list_sentences_temp)==0:
  306. list_sentences_temp.append(Sentences(doc_id=doc_id,sentence_index=0,sentence_text="sentence_text",tokens=[],pos_tags=[],ner_tags=""))
  307. list_sentences.append(list_sentences_temp)
  308. list_outlines.append(outline_list)
  309. article.content = re.sub("##attachment_begin##|##attachment_end##", "", article.content)
  310. return list_sentences,list_outlines