time_attrs.py 65 KB

1234567891011121314151617181920212223242526272829303132333435363738394041424344454647484950515253545556575859606162636465666768697071727374757677787980818283848586878889909192939495969798991001011021031041051061071081091101111121131141151161171181191201211221231241251261271281291301311321331341351361371381391401411421431441451461471481491501511521531541551561571581591601611621631641651661671681691701711721731741751761771781791801811821831841851861871881891901911921931941951961971981992002012022032042052062072082092102112122132142152162172182192202212222232242252262272282292302312322332342352362372382392402412422432442452462472482492502512522532542552562572582592602612622632642652662672682692702712722732742752762772782792802812822832842852862872882892902912922932942952962972982993003013023033043053063073083093103113123133143153163173183193203213223233243253263273283293303313323333343353363373383393403413423433443453463473483493503513523533543553563573583593603613623633643653663673683693703713723733743753763773783793803813823833843853863873883893903913923933943953963973983994004014024034044054064074084094104114124134144154164174184194204214224234244254264274284294304314324334344354364374384394404414424434444454464474484494504514524534544554564574584594604614624634644654664674684694704714724734744754764774784794804814824834844854864874884894904914924934944954964974984995005015025035045055065075085095105115125135145155165175185195205215225235245255265275285295305315325335345355365375385395405415425435445455465475485495505515525535545555565575585595605615625635645655665675685695705715725735745755765775785795805815825835845855865875885895905915925935945955965975985996006016026036046056066076086096106116126136146156166176186196206216226236246256266276286296306316326336346356366376386396406416426436446456466476486496506516526536546556566576586596606616626636646656666676686696706716726736746756766776786796806816826836846856866876886896906916926936946956966976986997007017027037047057067077087097107117127137147157167177187197207217227237247257267277287297307317327337347357367377387397407417427437447457467477487497507517527537547557567577587597607617627637647657667677687697707717727737747757767777787797807817827837847857867877887897907917927937947957967977987998008018028038048058068078088098108118128138148158168178188198208218228238248258268278288298308318328338348358368378388398408418428438448458468478488498508518528538548558568578588598608618628638648658668678688698708718728738748758768778788798808818828838848858868878888898908918928938948958968978988999009019029039049059069079089099109119129139149159169179189199209219229239249259269279289299309319329339349359369379389399409419429439449459469479489499509519529539549559569579589599609619629639649659669679689699709719729739749759769779789799809819829839849859869879889899909919929939949959969979989991000100110021003100410051006100710081009101010111012
  1. # -*- coding: utf-8 -*-
  2. """Phase 6 migration from interface/getAttributes.py.
  3. 本模块承载从 ``BiddingKG.dl.interface.getAttributes`` 原样迁出的时间相关属性
  4. 装配函数,作为 Phase 6 装配层(assembly)重组的一部分。原位置与行号:
  5. - ``turnBidWay`` — 采购方式归一(原 3315-3333 行)
  6. - ``my_time_format_pattern`` — 时间正则常量(原 3370 行)
  7. - ``time_sub_pattern`` — 时间分隔符正则常量(原 3371 行)
  8. - ``my_timeFormat`` — 时间文本格式化为 ``YYYY-MM-DD`` 列表(原 3373-3457 行)
  9. - ``timeAdd`` — 日期加减天数/分钟(原 3459-3463 行)
  10. - ``getTimeAttributes`` — 装配时间类属性(原 3465-4210 行)
  11. - ``getOtherAttributes`` — 装配其它属性(资金来源/服务期等)(原 4413-4529 行)
  12. 注:``extract_serviceTime``、``get_days_between``、``turnMoneySource`` 已先一步
  13. 下沉到 ``BiddingKG.dl.common.attr_utils``,本模块按依赖方向约束改为从
  14. ``common.attr_utils`` 显式导入,避免反向依赖 ``interface.getAttributes``。
  15. """
  16. from __future__ import absolute_import
  17. import re
  18. import time
  19. from datetime import datetime
  20. from decimal import Decimal
  21. from BiddingKG.dl.common.attr_utils import extract_serviceTime, get_days_between, turnMoneySource
  22. from BiddingKG.dl.ratio.re_ratio import getUnifyNum
  23. from BiddingKG.dl.common.Utils import isValidDate, getDigitsDic
  24. __all__ = ["turnBidWay", "my_timeFormat", "timeAdd", "getTimeAttributes", "getOtherAttributes",
  25. "my_time_format_pattern", "time_sub_pattern"]
  26. def turnBidWay(bidway):
  27. if bidway in ("邀请招标","采购方式:邀请"):
  28. return "邀请招标"
  29. elif bidway in ("询价","询单","询比","采购方式:询价"):
  30. return "询价"
  31. elif bidway in ("竞谈","竞争性谈判","公开竞谈"):
  32. return "竞争性谈判"
  33. elif bidway in ("竞争性磋商","磋商"):
  34. return "竞争性磋商"
  35. elif bidway in ("竞价","竞标","电子竞价","以电子竞价","电子书面竞投"):
  36. return "竞价"
  37. elif bidway in ("公开招标","网上电子投标","网上招标","采购方式:公开","招标为其他"):
  38. return "公开招标"
  39. elif bidway in ("单一来源"):
  40. return "单一来源"
  41. elif bidway in ("比选"):
  42. return "比选"
  43. else:
  44. return "其他"
  45. my_time_format_pattern = re.compile("((?:(?P<year>20\d{2}|\d{2}|二[零〇0][零〇一二三四五六七八九0]{2})\s*[-/年.]\s*)?(?P<month>\d{1,2}|[一二三四五六七八九十]{1,3})\s*[-/月.]\s*(?P<day>\d{1,2}|[一二三四五六七八九十]{1,3}))(?:.?(?:[0-1][0-9]|2[0-4]|[0-9]):[0-5][0-9](?::[0-5][0-9])?)?")
  46. time_sub_pattern = re.compile("[\[\]【】()(){}{}]+")
  47. from BiddingKG.dl.ratio.re_ratio import getUnifyNum
  48. def my_timeFormat(_time,page_time):
  49. # if page_time:
  50. # current_year = time.strftime("%Y",time.localtime(int(datetime.strptime(page_time, '%Y-%m-%d').timestamp())))
  51. # else:
  52. # current_year = time.strftime("%Y",time.localtime())
  53. _time = re.sub(time_sub_pattern,"",_time)
  54. all_match = re.finditer(my_time_format_pattern,_time)
  55. time_list = []
  56. idx = 0
  57. global_year = ""
  58. for _match in all_match:
  59. if len(_match.group())>0:
  60. idx += 1
  61. legal = True
  62. year = ""
  63. month = ""
  64. day = ""
  65. for k,v in _match.groupdict().items():
  66. if k=="year":
  67. year = v
  68. if k=="month":
  69. month = v
  70. if k=="day":
  71. day = v
  72. if year!="":
  73. if year==None: # 例:5月18日
  74. if idx==2 and global_year: # 例:2025年5月14日-5月18日,第二个时间没年份
  75. year = global_year
  76. else:
  77. legal = False
  78. else:
  79. if re.search("^\d+$", year):
  80. if len(year) == 2:
  81. year = "20" + year
  82. # if int(year) - int(current_year) > 5 or int(year) - int(current_year) < -1:
  83. # legal = False
  84. # else:
  85. # if int(year) - int(current_year)>10 or int(year) - int(current_year) < -1:
  86. # legal = False
  87. else:
  88. _year = ""
  89. for word in year:
  90. if word == '0':
  91. _year += word
  92. else:
  93. _year += str(getDigitsDic(word))
  94. year = _year
  95. else:
  96. legal = False
  97. if month!="":
  98. if re.search("^\d+$", month):
  99. if int(month) > 12:
  100. legal = False
  101. else:
  102. month = int(getUnifyNum(month))
  103. if month >= 1 and month <= 12:
  104. month = str(month)
  105. else:
  106. legal = False
  107. else:
  108. legal = False
  109. if day!="":
  110. if re.search("^\d+$", day):
  111. if int(day) > 31:
  112. legal = False
  113. else:
  114. day = int(getUnifyNum(day))
  115. if day >= 1 and day <= 31:
  116. day = str(day)
  117. else:
  118. legal = False
  119. else:
  120. legal = False
  121. if legal and not isValidDate(int(year),int(month),int(day)):
  122. legal = False
  123. if legal:
  124. # 数字字符格式化
  125. year = str(int(year))
  126. month = str(int(month))
  127. day = str(int(day))
  128. time_list.append("%s-%s-%s"%(year,month.rjust(2,"0"),day.rjust(2,"0")))
  129. # if idx==1 and not global_year:
  130. if not global_year:
  131. global_year = year
  132. return time_list,global_year
  133. def timeAdd(_time,days,format="%Y-%m-%d",minutes=0):
  134. a = time.mktime(time.strptime(_time,format))+86400*days+60*minutes
  135. _time1 = time.strftime(format,time.localtime(a))
  136. return _time1
  137. def getTimeAttributes(list_entity,list_sentence,page_time):
  138. # from BiddingKG.dl.interface.htmlparser import get_childs
  139. # document_tree = parse_document.tree
  140. # new_document_tree = []
  141. # _data_i = -1
  142. # while _data_i < len(document_tree) - 1:
  143. # _data_i += 1
  144. # _data = document_tree[_data_i]
  145. # _type = _data["type"]
  146. # if _type == "sentence":
  147. # if _data["sentence_title"] is not None:
  148. # new_document_tree.append(_data)
  149. # document_tree = new_document_tree
  150. time_entitys = [i for i in list_entity if i.entity_type=='time']
  151. time_entitys = sorted(time_entitys,key=lambda x:(x.sentence_index, x.begin_index))
  152. list_sentence = sorted(list_sentence,key=lambda x:x.sentence_index)
  153. dict_time = {
  154. "time_release": [], # 1 发布时间
  155. "time_bidopen": [], # 2 开标时间
  156. "time_bidclose": [], # 3 截标时间
  157. 'time_bidstart': [], # 12 投标(开始)时间、响应文件接收(开始)时间
  158. 'time_publicityStart': [], # 4 公示开始时间(公示时间、公示期)
  159. 'time_publicityEnd': [], # 5 公示截止时间
  160. 'time_getFileStart': [], # 6 文件获取开始时间(文件获取时间)
  161. 'time_getFileEnd': [], # 7 文件获取截止时间
  162. 'time_registrationStart': [], # 8 报名开始时间(报名时间)
  163. 'time_registrationEnd': [], # 9 报名截止时间
  164. 'time_earnestMoneyStart': [], #10 保证金递交开始时间(保证金递交时间)
  165. 'time_earnestMoneyEnd': [] , # 11 保证金递交截止时间
  166. 'time_commencement':[] , #13 开工日期
  167. 'time_completion': [], # 14 竣工日期
  168. 'time_listingStart': [], # 15 挂牌开始日期(挂牌时间)
  169. 'time_listingEnd': [], # 16 挂牌结束日期、挂牌截止日期
  170. 'time_signContract': [], # 17 合同签订时间
  171. 'time_contractStart': [], # 18 合同开始时间
  172. 'time_contractEnd': [] # 19 合同结束时间
  173. }
  174. dict_time2label = {
  175. "time_release": 1, # 1 发布时间
  176. "time_bidopen": 2, # 2 开标时间
  177. "time_bidclose": 3, # 3 截标时间
  178. 'time_bidstart': 12, # 12 投标(开始)时间、响应文件接收(开始)时间
  179. 'time_publicityStart': 4, # 4 公示开始时间(公示时间、公示期)
  180. 'time_publicityEnd': 5, # 5 公示截止时间
  181. 'time_getFileStart': 6, # 6 文件获取开始时间(文件获取时间)
  182. 'time_getFileEnd': 7, # 7 文件获取截止时间
  183. 'time_registrationStart': 8, # 8 报名开始时间(报名时间)
  184. 'time_registrationEnd': 9, # 9 报名截止时间
  185. 'time_earnestMoneyStart': 10, # 10 保证金递交开始时间(保证金递交时间)
  186. 'time_earnestMoneyEnd': 11, # 11 保证金递交截止时间
  187. 'time_commencement': 13, # 13 开工日期
  188. 'time_completion': 14, # 14 竣工日期
  189. 'time_listingStart': 15, # 15 挂牌开始日期(挂牌时间)
  190. 'time_listingEnd': 16, # 16 挂牌结束日期、挂牌截止日期
  191. 'time_signContract': 17, # 17 合同签订时间
  192. 'time_contractStart': 18, # 18 合同开始时间
  193. 'time_contractEnd': 19 # 19 合同结束时间
  194. }
  195. last_sentence_index = 0
  196. last_time_type = ""
  197. last_time_index = {
  198. 'time_bidstart':"time_bidclose",
  199. 'time_publicityStart':"time_publicityEnd",
  200. 'time_getFileStart':"time_getFileEnd",
  201. 'time_registrationStart':"time_registrationEnd",
  202. 'time_earnestMoneyStart':"time_earnestMoneyEnd",
  203. 'time_commencement':"time_completion",
  204. 'time_listingStart':"time_listingEnd",
  205. 'time_contractStart':"time_contractEnd"
  206. }
  207. # time_entitys = [[_entity,my_timeFormat(_entity.entity_text,page_time)] for _entity in time_entitys]
  208. new_time_entitys = []
  209. year_list = []
  210. for _entity in time_entitys:
  211. _time_list,_year = my_timeFormat(_entity.entity_text,page_time)
  212. _in_attachment = _entity.in_attachment
  213. if _time_list:
  214. new_time_entitys.append([_entity,_time_list,_year])
  215. year_list.append([_year,_in_attachment])
  216. get_all_time = False if False in [i[1] for i in year_list] else True
  217. if page_time:
  218. current_year = time.strftime("%Y",time.localtime(int(datetime.strptime(page_time, '%Y-%m-%d').timestamp())))
  219. year_list.append([current_year,False])
  220. else:
  221. current_year = time.strftime("%Y",time.localtime())
  222. if get_all_time:
  223. year_list = [i[0] for i in year_list]
  224. else:
  225. year_list = [i[0] for i in year_list if not i[1]]
  226. year_list = [(y,year_list.count(y)) for y in year_list if y[:2]=='20']
  227. year_list.sort(key=lambda x:x[1],reverse=True)
  228. most_year = year_list[0][0] if year_list else ""
  229. if most_year:
  230. time_entitys = [item for item in new_time_entitys if int(item[2])-int(most_year)<=10 and int(item[2])-int(most_year)>=-1]
  231. else:
  232. time_entitys = new_time_entitys
  233. # print(time_entitys)
  234. for entity_idx in range(len(time_entitys)):
  235. entity = time_entitys[entity_idx][0]
  236. extract_time = time_entitys[entity_idx][1]
  237. sentence_text = list_sentence[entity.sentence_index].sentence_text
  238. previous_entity = time_entitys[entity_idx-1][0] if entity_idx!=0 else None
  239. previous_extract_time = time_entitys[entity_idx-1][1] if entity_idx!=0 else None
  240. next_entity = time_entitys[entity_idx+1][0] if entity_idx!=len(time_entitys)-1 else None
  241. next_extract_time = time_entitys[entity_idx+1][1] if entity_idx!=len(time_entitys)-1 else None
  242. # 实体有效上下文
  243. entity_context_begin = previous_entity.wordOffset_end if previous_entity and previous_entity.sentence_index==entity.sentence_index else 0
  244. entity_context_end = next_entity.wordOffset_begin if next_entity and next_entity.sentence_index==entity.sentence_index else len(sentence_text)
  245. if entity.sentence_index!=last_sentence_index:
  246. # sentence_index 不同句子重置last_time_type
  247. last_time_type = ""
  248. entity_left = sentence_text[max(entity_context_begin, entity.wordOffset_begin - 2):entity.wordOffset_begin]
  249. entity_left2 = sentence_text[max(entity_context_begin, entity.wordOffset_begin - 10):entity.wordOffset_begin]
  250. entity_left3 = sentence_text[max(entity_context_begin, entity.wordOffset_begin - 30):entity.wordOffset_begin]
  251. entity_right = sentence_text[entity.wordOffset_end:min(entity.wordOffset_end + 3,entity_context_end)]
  252. entity_right2 = sentence_text[entity.wordOffset_end:entity_context_end]
  253. entity_right2 = re.sub(r"http[s]?://(?:[a-zA-Z]|[0-9]|[$-_@.&+]|[!*\\(\\),]|(?:%[0-9a-fA-F][0-9a-fA-F]))+",'',entity_right2)[:60] # 去除网址
  254. # print(entity.entity_text,entity_right2)
  255. label_prob = entity.values[entity.label]
  256. entity_text = entity.entity_text
  257. in_attachment = entity.in_attachment
  258. if extract_time:
  259. definite_time_list = []
  260. t = re.compile("(北京时间)?(?P<day>下午|上午|早上)?(?P<hour>(?:[0-1][0-9]|2[0-4]|[0-9]))[::时点](?P<half_hour>半)?(?P<minute>(?:[0-5][0-9]|[0-9]))?[::分]?(?P<second>[0-5][0-9])?秒?")
  261. _entity_text = re.sub(" (?=[^\d])|(?<=[^\d]) ","",entity_text)
  262. _entity_text_len = len(_entity_text)
  263. _entity_text = _entity_text + sentence_text[entity.wordOffset_end:entity.wordOffset_end+20]
  264. t_in_word_num = len(re.findall(t,_entity_text))
  265. # t_out_of_word = re.search("^[^\d]{,2}"+t.pattern,re.sub(" (?=[^\d])|(?<=[^\d]) ","",sentence_text[entity.wordOffset_end:]))
  266. begin_index = 0
  267. definite_time_idx_list = []
  268. for _num in range(t_in_word_num):
  269. if begin_index> _entity_text_len + 8:
  270. break
  271. t_in_word = re.search(t, _entity_text[begin_index:])
  272. # print(_entity_text[begin_index:])
  273. if t_in_word:
  274. if _num==0 and t_in_word.start() > _entity_text_len + 8:
  275. break
  276. begin_index += t_in_word.end()
  277. # print('t_in_word',entity_text,t_in_word.groupdict())
  278. day = t_in_word.groupdict().get('day',"")
  279. hour = t_in_word.groupdict().get('hour',"")
  280. half_hour = t_in_word.groupdict().get('half_hour',"")
  281. minute = t_in_word.groupdict().get('minute',"")
  282. second = t_in_word.groupdict().get('second',"")
  283. if not minute:
  284. second = ""
  285. if hour:
  286. if day=='下午' and int(hour)<12:
  287. hour = str(int(hour)+12)
  288. if int(hour)>=24:
  289. continue
  290. else:
  291. hour = "00"
  292. if not minute:
  293. if half_hour:
  294. minute = "30"
  295. else:
  296. minute = "00"
  297. if int(minute)>=60:
  298. continue
  299. if not second:
  300. second = "00"
  301. if int(second)>=60:
  302. continue
  303. definite_time = "%s:%s:%s"%(hour.rjust(2,"0"),minute.rjust(2,"0"),second.rjust(2,"0"))
  304. # print(definite_time)
  305. definite_time_list.append(definite_time)
  306. definite_time_idx_list.append([begin_index-len(t_in_word.group()),begin_index])
  307. if len(extract_time)==1 and len(definite_time_list)>=2: # 实体只包含一个时间,"2024-12-09 09:00~16:00" 考虑单个时间对应两个详细时间段的识别
  308. # 前两个详细时间的间隔
  309. distance = definite_time_idx_list[1][0] - definite_time_idx_list[0][1]
  310. if distance<=8 and int(definite_time_list[1][:2])>=int(definite_time_list[0][:2]): # 判断详细时间都‘小时’顺序从小到大
  311. new_extract_time = []
  312. for d_time in definite_time_list[:2]:
  313. if d_time == "24:00:00": # 修正不规范时间表述
  314. d_time = "23:59:59"
  315. new_extract_time.append(extract_time[0] + " " + d_time)
  316. extract_time = new_extract_time
  317. else:
  318. if definite_time_list[0] == "24:00:00": # 修正不规范时间表述
  319. definite_time_list[0] = "23:59:59"
  320. if definite_time_list[0] != "00:00:00":
  321. extract_time[0] = extract_time[0] + " " + definite_time_list[0]
  322. else:
  323. min_len = min(len(extract_time),len(definite_time_list))
  324. for i in range(min_len):
  325. if definite_time_list[i] == "24:00:00": # 修正不规范时间表述
  326. definite_time_list[i] = "23:59:59"
  327. if definite_time_list[i] != "00:00:00":
  328. extract_time[i] = extract_time[i] + " " + definite_time_list[i]
  329. if extract_time:
  330. # 时间变更prob优化
  331. if re.search("原",entity_left2):
  332. last_index = 0
  333. for item in re.finditer("原",entity_left2):
  334. last_index = item.start() + 1
  335. label_prob = label_prob - 0.2 * last_index / len(entity_left2)
  336. # print('prob优化',label_prob,extract_time)
  337. elif (re.search("改正|更正|修正|修改|更改|变更|延期|调整|变成",entity_left3[-20:]) or re.search("现",entity_left3[-10:])) and not re.search("更正日期",entity_left3[-8:]):
  338. last_time_label = dict_time2label.get(last_time_type,None)
  339. if last_time_label and entity.label==0:
  340. entity.label = last_time_label
  341. label_prob = 1.5
  342. elif last_time_label and entity.label==last_time_label:# 前后两个相同类型的时间为变更关系
  343. label_prob = 2
  344. # 优化多个并列的时间,如:开标时间和截标时间,截标时间和报名结束时间
  345. if entity.label in [2,3,9]:
  346. if entity.label==2 and re.search("截标|投标.{,2}截止|([递提]交|接收)(?:文件)?.{,2}截止|报价.{,2}截止|响应.{,2}截止|文件.{,2}([递提]交|接收)",entity_left3):
  347. dict_time['time_bidclose'].append((extract_time[0], label_prob-0.1, in_attachment))
  348. if entity.label==3 and re.search("开标|(评审|比选).{,2}(?:开始)?(时间|日期)|选取.{,2}(时间|日期)",entity_left3):
  349. dict_time['time_bidopen'].append((extract_time[0], label_prob-0.1, in_attachment))
  350. if entity.label==3 and re.search("报名",entity_left3):
  351. dict_time['time_registrationEnd'].append((extract_time[0], 0.5, in_attachment))
  352. if entity.label==3 and re.search("获取",entity_left3[-20:]):
  353. dict_time['time_getFileEnd'].append((extract_time[0], 0.45, in_attachment))
  354. if entity.label==9 and re.search("截标|投标.{,2}截止|([递提]交|接收)(?:文件)?.{,2}截止|报价.{,2}截止|响应.{,2}截止|文件.{,2}([递提]交|接收)",entity_left3):
  355. dict_time['time_bidclose'].append((extract_time[0], label_prob-0.1, in_attachment))
  356. if entity.label in [11, 3]:
  357. if entity.label==11 and re.search("文件.{,2}([递提]交|接收)|截标|投标.{,2}截止|([递提]交|接收)(?:文件)?.{,2}截止|报价.{,2}截止|响应.{,2}截止",entity_left3):
  358. dict_time['time_bidclose'].append((extract_time[0], 0.5, in_attachment))
  359. if entity.label==3 and re.search("保证金.{,2}(接受|收取)|(接受|收取).{,2}保证金",entity_left3):
  360. dict_time['time_earnestMoneyEnd'].append((extract_time[0], 0.5, in_attachment))
  361. if entity.label in [6, 7]:
  362. if re.search("文件.{,2}([递提]交|接收)|截标|投标.{,2}截止|([递提]交|接收)(?:文件)?.{,2}截止|报价.{,2}截止|响应.{,2}截止",entity_left3):
  363. dict_time['time_bidclose'].append((extract_time[0], 0.5, in_attachment))
  364. if entity.label==0:
  365. if re.search("文件.{,2}([递提]交|接收)|截标|投标.{,2}截止|([递提]交|接收)(?:文件)?.{,2}截止|报价.{,2}截止|响应.{,2}截止",entity_left3):
  366. if len(extract_time)>=2:
  367. dict_time['time_bidstart'].append((extract_time[0], 0.45, in_attachment))
  368. dict_time['time_bidclose'].append((extract_time[1], 0.45, in_attachment))
  369. else:
  370. dict_time['time_bidclose'].append((extract_time[0], 0.45, in_attachment))
  371. if entity.label==6:
  372. # "文件获取时间"和"报名时间"并列
  373. if re.search("报名",entity_left3):
  374. if len(extract_time)==1:
  375. dict_time['time_registrationStart'].append((extract_time[0], 0.51, in_attachment))
  376. else:
  377. dict_time['time_registrationStart'].append((extract_time[0], 0.51, in_attachment))
  378. dict_time['time_registrationEnd'].append((extract_time[1], 0.51, in_attachment))
  379. # 获取文件/报名/报价 时间补充(上下文表达过长无法通过模型识别)
  380. # if entity.label == 0:
  381. # if re.search("(获取|领取|售卖|出售|购买|下载).{,4}(招标|投标|采购)?(文件|标书)|(文件|标书).{,4}(获取|售卖|出售|发售|购买)", entity_left3):
  382. # if len(extract_time)==2:
  383. # dict_time['time_getFileStart'].append((extract_time[0], 0.51, in_attachment))
  384. # dict_time['time_getFileEnd'].append((extract_time[1], 0.51, in_attachment))
  385. # else:
  386. # if next_entity and next_entity.sentence_index==entity.sentence_index:
  387. # mid_text = sentence_text[entity.wordOffset_end:next_entity.wordOffset_begin]
  388. # if len(mid_text)<=10 and re.search("至|到|[-—]|[~~]",mid_text) and len(next_extract_time)==1:
  389. # dict_time['time_getFileStart'].append((extract_time[0], 0.51, in_attachment))
  390. # dict_time['time_getFileEnd'].append((next_extract_time[0], 0.51, in_attachment))
  391. # if not dict_time['time_getFileEnd']:
  392. # if re.search("前|止|截止", entity_right) or re.search("前",entity_text[-2:]):
  393. # dict_time['time_getFileEnd'].append((extract_time[0], 0.51, in_attachment))
  394. # elif re.search("起|开?始", entity_right) or re.search("起",entity_text[-2:]):
  395. # dict_time['time_getFileStart'].append((extract_time[0], 0.51, in_attachment))
  396. # if re.search("(进行|在线|线下|线上|网上).{,2}报名|报名.{,2}(开始)?(时间|日期)", entity_left3):
  397. # if len(extract_time)==2:
  398. # dict_time['time_registrationStart'].append((extract_time[0], 0.51, in_attachment))
  399. # dict_time['time_registrationEnd'].append((extract_time[1], 0.51, in_attachment))
  400. # else:
  401. # if next_entity and next_entity.sentence_index==entity.sentence_index:
  402. # mid_text = sentence_text[entity.wordOffset_end:next_entity.wordOffset_begin]
  403. # if len(mid_text)<=10 and re.search("至|到|[-—]|[~~]",mid_text) and len(next_extract_time)==1:
  404. # dict_time['time_registrationStart'].append((extract_time[0], 0.51, in_attachment))
  405. # dict_time['time_registrationEnd'].append((next_extract_time[0], 0.51, in_attachment))
  406. # if not dict_time['time_registrationEnd']:
  407. # if re.search("前|止|截止", entity_right) or re.search("前",entity_text[-2:]):
  408. # dict_time['time_registrationEnd'].append((extract_time[0], 0.51, in_attachment))
  409. # elif re.search("起|开?始", entity_right) or re.search("起",entity_text[-2:]):
  410. # dict_time['time_registrationStart'].append((extract_time[0], 0.51, in_attachment))
  411. #
  412. # if re.search("(获取|售卖|出售|购买).{,4}(招标|投标|采购)?(文件|标书)|(文件|标书).{,4}(获取|售卖|出售|发售|购买)", entity_right2):
  413. # if len(extract_time)==2:
  414. # dict_time['time_getFileStart'].append((extract_time[0], 0.51, in_attachment))
  415. # dict_time['time_getFileEnd'].append((extract_time[1], 0.51, in_attachment))
  416. # else:
  417. # if previous_entity and previous_entity.sentence_index==entity.sentence_index:
  418. # mid_text = sentence_text[previous_entity.wordOffset_end:entity.wordOffset_begin]
  419. # if len(mid_text)<=10 and re.search("至|到|[-—]|[~~]",mid_text) and len(previous_extract_time)==1:
  420. # dict_time['time_getFileStart'].append((previous_extract_time[0], 0.51, in_attachment))
  421. # dict_time['time_getFileEnd'].append((extract_time[0], 0.51, in_attachment))
  422. # if not dict_time['time_getFileEnd']:
  423. # if re.search("前|止|截止", entity_right) or re.search("前",entity_text[-2:]):
  424. # dict_time['time_getFileEnd'].append((extract_time[0], 0.51, in_attachment))
  425. # elif re.search("起|开?始", entity_right) or re.search("起",entity_text[-2:]):
  426. # dict_time['time_getFileStart'].append((extract_time[0], 0.51, in_attachment))
  427. # if re.search("(进行|在线|线下).{,2}报名", entity_right2):
  428. # if len(extract_time) == 2:
  429. # dict_time['time_registrationStart'].append((extract_time[0], 0.51, in_attachment))
  430. # dict_time['time_registrationEnd'].append((extract_time[1], 0.51, in_attachment))
  431. # else:
  432. # if previous_entity and previous_entity.sentence_index==entity.sentence_index:
  433. # mid_text = sentence_text[previous_entity.wordOffset_end:entity.wordOffset_begin]
  434. # if len(mid_text)<=10 and re.search("至|到|[-—]|[~~]",mid_text) and len(previous_extract_time)==1:
  435. # dict_time['time_registrationStart'].append((previous_extract_time[0], 0.51, in_attachment))
  436. # dict_time['time_registrationEnd'].append((extract_time[0], 0.51, in_attachment))
  437. # if not dict_time['time_registrationEnd']:
  438. # if re.search("前|止|截止", entity_right) or re.search("前",entity_text[-2:]):
  439. # dict_time['time_registrationEnd'].append((extract_time[0], 0.51, in_attachment))
  440. # elif re.search("起|开?始", entity_right) or re.search("起",entity_text[-2:]):
  441. # dict_time['time_registrationStart'].append((extract_time[0], 0.51, in_attachment))
  442. # if re.search("(进行|开始).{,4}(报价|投标|竞价)", entity_right2):
  443. # if len(extract_time) == 2:
  444. # dict_time['time_bidstart'].append((extract_time[0], 0.51, in_attachment))
  445. # # dict_time['time_bidclose'].append((extract_time[1], 0.51, in_attachment))
  446. # 补充公告末尾处的发布时间
  447. if entity.label==0:
  448. if entity.is_tail:
  449. entity.label = 1
  450. entity.values[1] = 0.5
  451. dict_time['time_release'].append((extract_time[0], 0.5, in_attachment))
  452. # 2022/12/12 新增挂牌时间正则
  453. if re.search("挂牌.{,4}(?:时间|日期)",entity_left2):
  454. if re.search("挂牌.{,4}(?:时间|日期)",entity_left2).end()>len(entity_left2)/2:
  455. if len(extract_time) == 1:
  456. if re.search("挂牌.?(开始|起始).?(?:时间|日期)",entity_left2):
  457. dict_time['time_listingStart'].append((extract_time[0], 0.5, in_attachment))
  458. last_time_type = 'time_listingStart'
  459. elif re.search("挂牌.?(截[止至]|结束).?(?:时间|日期)",entity_left2):
  460. dict_time['time_listingEnd'].append((extract_time[0], 0.5, in_attachment))
  461. last_time_type = 'time_listingEnd'
  462. elif re.search("挂牌.?(?:时间|日期)",entity_left2):
  463. if re.search("前|止|截止",entity_right) or re.search("至|止|到",entity_left) or re.search("前",entity_text[-2:]):
  464. dict_time['time_listingEnd'].append((extract_time[0], 0.5, in_attachment))
  465. last_time_type = 'time_listingEnd'
  466. else:
  467. dict_time['time_listingStart'].append((extract_time[0], 0.5, in_attachment))
  468. last_time_type = 'time_listingStart'
  469. else:
  470. dict_time['time_listingStart'].append((extract_time[0], 0.5, in_attachment))
  471. dict_time['time_listingEnd'].append((extract_time[1], 0.5, in_attachment))
  472. last_time_type = ''
  473. last_sentence_index = entity.sentence_index
  474. continue
  475. # 2023/9/13 新增合同相关时间
  476. if re.search("合同|服务|履[约行]", entity_left3[-15:]):
  477. if len(extract_time) == 1:
  478. if re.search("(合同.{,2}签[订定署].{,2}|签[订定署].{,2}合同.{,2})(?:时间|日期)|合同签[订定署].{,1}$", entity_left2):
  479. dict_time['time_signContract'].append((extract_time[0], 0.5, in_attachment))
  480. last_time_type = 'time_signContract'
  481. last_sentence_index = entity.sentence_index
  482. continue
  483. elif re.search("(合同|服务|履约|(合同|服务)履行).{,4}(?:起始|开始)(?:时间|日期)", entity_left3[-15:]):
  484. dict_time['time_contractStart'].append((extract_time[0], 0.55, in_attachment))
  485. last_time_type = 'time_contractStart'
  486. last_sentence_index = entity.sentence_index
  487. continue
  488. elif re.search("(合同|服务|履约).{,2}(?:完成|截止|结束)(?:时间|日期|时限)", entity_left2):
  489. dict_time['time_contractEnd'].append((extract_time[0], 0.55, in_attachment))
  490. last_time_type = 'time_contractEnd'
  491. last_sentence_index = entity.sentence_index
  492. continue
  493. elif re.search("(?:合同|服务|履约|(合同|服务)履行)(?:期限?|有效期)|(?:服务|履约|(合同|服务)履行)(?:时间|日期|周期)|服务[时年]限|合同周期", entity_left2):
  494. if re.search("到|至|截[至止]",entity_left) or re.search("前|止|截止",entity_right) or re.search("前",entity_text[-2:]):
  495. dict_time['time_contractEnd'].append((extract_time[0], 0.5, in_attachment))
  496. last_time_type = 'time_contractEnd'
  497. else:
  498. dict_time['time_contractStart'].append((extract_time[0], 0.5, in_attachment))
  499. last_time_type = 'time_contractStart'
  500. last_sentence_index = entity.sentence_index
  501. continue
  502. else:
  503. if re.search("(?:合同|服务|履约|(合同|服务)履行)(?:期限?|有效期)|(?:服务|履约|(合同|服务)履行)(?:时间|日期|周期)|服务[时年]限|合同周期", entity_left2):
  504. # 排除开始和借宿时间一样的错误模板,例:“履约期限:2023年02月15日至2023年02月15日”
  505. if extract_time[0]!=extract_time[1]:
  506. dict_time['time_contractStart'].append((extract_time[0], 0.6, in_attachment))
  507. dict_time['time_contractEnd'].append((extract_time[1], 0.6, in_attachment))
  508. last_time_type = ''
  509. last_sentence_index = entity.sentence_index
  510. continue
  511. # 服务期限表达补充
  512. if entity.label==0:
  513. re_service = '合同期限|工期/交货期/服务期|工期\(交货期\)|合格工期|服务期限|工期' \
  514. '|工期要求|项目周期|工期\(交货期\)|计划工期\(服务期限\)|服务时限|履行期限|服务周期|供货期限' \
  515. '|合格工期|计划工期\(服务期\)|服务期|服务,期|交货\(完工\)(时间|日期)|交付\(服务、完工\)(时间|日期)' \
  516. '|交货(时间|日期)|工期承诺|(服务|合同|施工|实施|工程|设计)的?(年限|期限|周期|期:)' \
  517. '|服务期限为|计划工期|工期要求|服务期限|服务期' \
  518. '|投标工期|设计工期|合格服务周期|总工期|服务(时间|日期)(范围)?|流转期限|维护期限|服务时限|交货期' \
  519. '|完成(时间|日期)|服务期限|中标工期|项目周期|期限要求|供货期|合同履行日期|计划的?周期' \
  520. '|履约期限|合同约定完成时限|合同完成日期|承诺完成日期' \
  521. '|合同起始日起|合同履约期|履约截止日期|承包期限|合同完成日期' \
  522. '|服务期间|服务履行期|委托(管理)?期限|履约期限、地点等简要信息'
  523. if len(extract_time)==2:
  524. if re.search(re_service,entity_left2) or re.search("履约期限、地点等简要信息",entity_left3[-20:]):
  525. dict_time['time_contractStart'].append((extract_time[0], 0.5, in_attachment))
  526. dict_time['time_contractEnd'].append((extract_time[1], 0.5, in_attachment))
  527. last_time_type = ''
  528. # 报价/投标时间补充(规则补充)
  529. if entity.label == 0:
  530. if re.search("[报竞]价.{,2}(开始|起始).{,2}(时间|日期)",entity_left2):
  531. entity.label = 12
  532. label_prob = 0.8
  533. elif re.search("[报竞]价.{,2}起止.{,2}(时间|日期)",entity_left2):
  534. entity.label = 12
  535. label_prob = 0.6
  536. elif re.search("响应.{,2}文件([递提]交|接收).{,2}(时间|日期)[::]|([递提]交|接收).{,2}响应.{,2}文件.{,2}(时间|日期)[::]",entity_left2):
  537. entity.label = 3
  538. label_prob = 0.501
  539. elif re.search("响应.{,2}文件([递提]交|接收).{,2}(时间|日期)|([递提]交|接收).{,2}响应.{,2}文件.{,2}(时间|日期)",entity_left2) and not re.search("截[止至]",entity_left2):
  540. entity.label = 12
  541. label_prob = 0.51
  542. elif re.search("[报竞]价.{,2}截[止至].{,2}(时间|日期)",entity_left2):
  543. entity.label = 3
  544. label_prob = 0.8
  545. elif re.search("(竞价|报价).?(时间|日期)",entity_left2):
  546. entity.label = 12
  547. label_prob = 0.51
  548. elif re.search("(竞价|报价).?(时间|日期)",entity_left3) and re.search("参与|报价|有意",entity_left2):
  549. entity.label = 12
  550. label_prob = 0.501
  551. # 文档结构补充
  552. # if entity.label == 0:
  553. # re_registration = re.compile("报名|(文件|标书)[\u4e00-\u9fa5、]{,4}(获取|出售|售卖|购买|下载)|"
  554. # "(获取|出售|售卖|购买|下载)[\u4e00-\u9fa5、]{,4}(文件|标书)")
  555. # _data_i = -1
  556. # while _data_i < len(document_tree) - 1:
  557. # _data_i += 1
  558. # _data = document_tree[_data_i]
  559. # _type = _data["type"]
  560. # _text = _data["text"].strip()
  561. # childs = get_childs([_data])
  562. # last_child = childs[-1]
  563. # if entity.sentence_index>=_data.sentence_index and entity.wordOffset_begin>=_data.wordOffset_begin and
  564. # ():
  565. # if re.search(re_registration, re.split("[::;;,]", _text)[0][:20]) is not None:
  566. #
  567. # content_text = ""
  568. # for c in childs:
  569. # content_text += c["text"] + ""
  570. # print('concat_text', content_text)
  571. if re.search("[,;](完成|截止|结束)(时间|日期)", entity_left2[-8:]) and entity.label==0:
  572. if entity.sentence_index == last_sentence_index:
  573. time_type = last_time_index.get(last_time_type)
  574. if time_type:
  575. dict_time[time_type].append((extract_time[0], 0.5 + label_prob / 10,in_attachment))
  576. last_time_type = ""
  577. last_sentence_index = entity.sentence_index
  578. continue
  579. if re.search("至|到|[日\d][-—]$|[~~]", entity_left):
  580. if entity.sentence_index == last_sentence_index:
  581. time_type = last_time_index.get(last_time_type)
  582. if time_type:
  583. dict_time[time_type].append((extract_time[0], 0.5 + label_prob / 10,in_attachment))
  584. last_time_type = ""
  585. last_sentence_index = entity.sentence_index
  586. continue
  587. if entity.label!=0:
  588. if entity.label==1 and label_prob>0.5:
  589. dict_time['time_release'].append((extract_time[0],label_prob,in_attachment))
  590. last_time_type = 'time_release'
  591. elif entity.label==2 and label_prob>0.5:
  592. dict_time['time_bidopen'].append((extract_time[0],label_prob,in_attachment))
  593. last_time_type = 'time_bidopen'
  594. elif entity.label==3 and label_prob>0.5:
  595. if re.search("(资格)?预审文件[\u4e00-\u9fa5]{,4}截止",entity_left3[-15:]):
  596. # 排除"资格预审文件提交截止时间"
  597. last_time_type = ''
  598. continue
  599. if len(extract_time)==1:
  600. dict_time['time_bidclose'].append((extract_time[0],label_prob,in_attachment))
  601. last_time_type = 'time_bidclose'
  602. elif len(extract_time)==2:
  603. dict_time['time_bidstart'].append((extract_time[0], 0.6, in_attachment))
  604. dict_time['time_bidclose'].append((extract_time[1], label_prob, in_attachment))
  605. last_time_type = 'time_bidclose'
  606. elif entity.label==12 and label_prob>0.5:
  607. if len(extract_time)==1:
  608. if re.search("前|止|截止",entity_right) or re.search("至|止|到",entity_left) or re.search("前",entity_text[-2:]):
  609. dict_time['time_bidclose'].append((extract_time[0], label_prob,in_attachment))
  610. last_time_type = 'time_bidclose'
  611. else:
  612. dict_time['time_bidstart'].append((extract_time[0], label_prob,in_attachment))
  613. last_time_type = 'time_bidstart'
  614. else:
  615. dict_time['time_bidstart'].append((extract_time[0],label_prob,in_attachment))
  616. dict_time['time_bidclose'].append((extract_time[1],label_prob,in_attachment))
  617. last_time_type = ''
  618. elif entity.label==4 and label_prob>0.5:
  619. if len(extract_time)==1:
  620. if re.search("前|止|截止",entity_right) or re.search("至|止|到",entity_left) or re.search("前",entity_text[-2:]):
  621. dict_time['time_publicityEnd'].append((extract_time[0], label_prob,in_attachment))
  622. last_time_type = 'time_publicityEnd'
  623. else:
  624. dict_time['time_publicityStart'].append((extract_time[0], label_prob,in_attachment))
  625. last_time_type = 'time_publicityStart'
  626. else:
  627. dict_time['time_publicityStart'].append((extract_time[0],label_prob,in_attachment))
  628. dict_time['time_publicityEnd'].append((extract_time[1],label_prob,in_attachment))
  629. last_time_type = ''
  630. elif entity.label==5 and label_prob>0.5:
  631. if len(extract_time)==1:
  632. dict_time['time_publicityEnd'].append((extract_time[0], label_prob,in_attachment))
  633. last_time_type = 'time_publicityEnd'
  634. else:
  635. dict_time['time_publicityStart'].append((extract_time[0],label_prob,in_attachment))
  636. dict_time['time_publicityEnd'].append((extract_time[1],label_prob,in_attachment))
  637. last_time_type = ''
  638. elif entity.label==6 and label_prob>0.5:
  639. if len(extract_time)==1:
  640. if (re.search("前|截?止",entity_right) and re.search("前|截?止(?!时间|日期)",entity_right2[:len(entity_right)+3])) or re.search("至|止|到",entity_left) or re.search("前",entity_text[-2:]):
  641. dict_time['time_getFileEnd'].append((extract_time[0], label_prob,in_attachment))
  642. last_time_type = 'time_getFileEnd'
  643. else:
  644. dict_time['time_getFileStart'].append((extract_time[0], label_prob,in_attachment))
  645. last_time_type = 'time_getFileStart'
  646. else:
  647. dict_time['time_getFileStart'].append((extract_time[0],label_prob,in_attachment))
  648. dict_time['time_getFileEnd'].append((extract_time[1],label_prob,in_attachment))
  649. last_time_type = ''
  650. elif entity.label==7 and label_prob>0.5:
  651. if len(extract_time)==1:
  652. dict_time['time_getFileEnd'].append((extract_time[0], label_prob,in_attachment))
  653. last_time_type = 'time_getFileEnd'
  654. else:
  655. dict_time['time_getFileStart'].append((extract_time[0],label_prob,in_attachment))
  656. dict_time['time_getFileEnd'].append((extract_time[1],label_prob,in_attachment))
  657. last_time_type = ''
  658. elif entity.label==8 and label_prob>0.5:
  659. if len(extract_time)==1:
  660. if re.search("前|止|截止",entity_right) or re.search("至|止|到",entity_left) or re.search("前",entity_text[-2:]):
  661. dict_time['time_registrationEnd'].append((extract_time[0], label_prob,in_attachment))
  662. last_time_type = 'time_registrationEnd'
  663. else:
  664. dict_time['time_registrationStart'].append((extract_time[0], label_prob,in_attachment))
  665. last_time_type = 'time_registrationStart'
  666. else:
  667. dict_time['time_registrationStart'].append((extract_time[0],label_prob,in_attachment))
  668. dict_time['time_registrationEnd'].append((extract_time[1],label_prob,in_attachment))
  669. last_time_type = ''
  670. elif entity.label==9 and label_prob>0.5:
  671. if len(extract_time)==1:
  672. dict_time['time_registrationEnd'].append((extract_time[0], label_prob,in_attachment))
  673. last_time_type = 'time_registrationEnd'
  674. else:
  675. dict_time['time_registrationStart'].append((extract_time[0],label_prob,in_attachment))
  676. dict_time['time_registrationEnd'].append((extract_time[1],label_prob,in_attachment))
  677. last_time_type = ''
  678. elif entity.label==10 and label_prob>0.5:
  679. if len(extract_time)==1:
  680. if re.search("前|止|截止",entity_right) or re.search("至|止|到",entity_left) or re.search("前",entity_text[-2:]):
  681. dict_time['time_earnestMoneyEnd'].append((extract_time[0], label_prob,in_attachment))
  682. last_time_type = 'time_earnestMoneyEnd'
  683. else:
  684. dict_time['time_earnestMoneyStart'].append((extract_time[0], label_prob,in_attachment))
  685. last_time_type = 'time_earnestMoneyStart'
  686. else:
  687. dict_time['time_earnestMoneyStart'].append((extract_time[0],label_prob,in_attachment))
  688. dict_time['time_earnestMoneyEnd'].append((extract_time[1],label_prob,in_attachment))
  689. last_time_type = ''
  690. elif entity.label==11 and label_prob>0.5:
  691. if len(extract_time)==1:
  692. dict_time['time_earnestMoneyEnd'].append((extract_time[0], label_prob,in_attachment))
  693. last_time_type = 'time_earnestMoneyEnd'
  694. else:
  695. dict_time['time_earnestMoneyStart'].append((extract_time[0],label_prob,in_attachment))
  696. dict_time['time_earnestMoneyEnd'].append((extract_time[1],label_prob,in_attachment))
  697. last_time_type = ''
  698. elif entity.label==13 and label_prob>0.5:
  699. if len(extract_time)==1:
  700. if re.search("前|止|截止",entity_right) or re.search("至|止|到",entity_left) or re.search("前",entity_text[-2:]):
  701. dict_time['time_completion'].append((extract_time[0], label_prob,in_attachment))
  702. last_time_type = 'time_completion'
  703. else:
  704. dict_time['time_commencement'].append((extract_time[0], label_prob,in_attachment))
  705. last_time_type = 'time_commencement'
  706. else:
  707. dict_time['time_commencement'].append((extract_time[0],label_prob,in_attachment))
  708. dict_time['time_completion'].append((extract_time[1],label_prob,in_attachment))
  709. last_time_type = ''
  710. elif entity.label==14 and label_prob>0.5:
  711. if len(extract_time)==1:
  712. dict_time['time_completion'].append((extract_time[0], label_prob,in_attachment))
  713. last_time_type = 'time_completion'
  714. else:
  715. dict_time['time_commencement'].append((extract_time[0],label_prob,in_attachment))
  716. dict_time['time_completion'].append((extract_time[1],label_prob,in_attachment))
  717. last_time_type = ''
  718. else:
  719. last_time_type = ""
  720. else:
  721. last_time_type = ""
  722. else:
  723. last_time_type = ""
  724. last_sentence_index = entity.sentence_index
  725. # 通过文档分析树形结构补充部分时间实体
  726. def add_time_by_parseDocument(dict_time,parse_document):
  727. from BiddingKG.dl.interface.htmlparser import get_childs
  728. document_tree = parse_document.tree
  729. # if not dict_time['time_getFileStart'] or not dict_time['time_getFileEnd']:
  730. # time_pattern = re.compile("")
  731. concat_text_list = []
  732. if not dict_time['time_registrationStart'] or not dict_time['time_registrationEnd']:
  733. re_registration = re.compile("报名|(文件|标书)[\u4e00-\u9fa5、]{,4}(获取|出售|售卖|购买|下载)|"
  734. "(获取|出售|售卖|购买|下载)[\u4e00-\u9fa5、]{,4}(文件|标书)")
  735. _data_i = -1
  736. while _data_i < len(document_tree) - 1:
  737. _data_i += 1
  738. _data = document_tree[_data_i]
  739. _type = _data["type"]
  740. _text = _data["text"].strip()
  741. # print(_data.keys())
  742. if _type == "sentence":
  743. print('_text:',_text,_data["sentence_title"])
  744. if _data["sentence_title"] is not None:
  745. print("aptitude_pattern", _text)
  746. print(_data['sentence_index'],_data['wordOffset_begin'],_data['wordOffset_end'])
  747. if re.search(re_registration, re.split("[::;;。]",_text)[0][:15]) is not None:
  748. childs = get_childs([_data])
  749. concat_text = ""
  750. for c in childs:
  751. concat_text += c["text"] + ""
  752. print('concat_text',concat_text)
  753. concat_text_list.append(concat_text)
  754. _data_i += len(childs)-1
  755. # if _type == "table":
  756. # list_table = _data["list_table"]
  757. # parent_title = _data["parent_title"]
  758. # if list_table is not None:
  759. # for line in list_table[:2]:
  760. # for cell_i in range(len(line)):
  761. # cell = line[cell_i]
  762. # cell_text = cell[0]
  763. # if len(cell_text) > 120 and re.search(re_registration, cell_text) is not None:
  764. # concat_text += cell_text + "\n"
  765. print('_text',concat_text_list)
  766. for text in concat_text_list:
  767. time_list = re.finditer(my_time_format_pattern,text)
  768. time_list = [(i,my_timeFormat(i.group(),page_time)) for i in time_list]
  769. for time_idx in range(len(time_list)):
  770. _time = time_list[time_idx][0]
  771. extract_time = time_list[time_idx][1]
  772. entity_left = text[:_time.start()]
  773. entity_left = re.split("[。;;!??]",entity_left)[-1]
  774. # entity_left2 = sentence_text[
  775. # max(entity_context_begin, entity.wordOffset_begin - 10):entity.wordOffset_begin]
  776. # entity_left3 = sentence_text[
  777. # max(entity_context_begin, entity.wordOffset_begin - 30):entity.wordOffset_begin]
  778. entity_right = text[_time.end():]
  779. entity_right = re.split("[。;;!??]",entity_right)[0]
  780. # entity_right2 = sentence_text[entity.wordOffset_end:entity_context_end]
  781. entity_right2 = re.sub(r"(http[s]?://)?(?:[a-zA-Z]|[0-9]|[$-_@.&+]|[!*\\(\\),]|(?:%[0-9a-fA-F][0-9a-fA-F])){6,}",
  782. '', entity_right)[:60] # 去除网址
  783. print('entity_right2',entity_right2)
  784. if re.search("(进行|在线|线下).{,2}报名", entity_right2):
  785. print('报名text',entity_right2)
  786. if len(extract_time) == 2:
  787. dict_time['time_registrationStart'].append((extract_time[0], 0.51, in_attachment))
  788. dict_time['time_registrationEnd'].append((extract_time[1], 0.51, in_attachment))
  789. else:
  790. if previous_entity and previous_entity.sentence_index==entity.sentence_index:
  791. mid_text = sentence_text[previous_entity.wordOffset_end:entity.wordOffset_begin]
  792. if len(mid_text)<=10 and re.search("至|到|[-—]|[~~]",mid_text) and len(previous_extract_time)==1:
  793. dict_time['time_registrationStart'].append((previous_extract_time[0], 0.51, in_attachment))
  794. dict_time['time_registrationEnd'].append((extract_time[0], 0.51, in_attachment))
  795. if not dict_time['time_registrationEnd']:
  796. if re.search("前|止|截止", entity_right) or re.search("前",entity_text[-2:]):
  797. dict_time['time_registrationEnd'].append((extract_time[0], 0.51, in_attachment))
  798. elif re.search("起|开?始", entity_right) or re.search("起",entity_text[-2:]):
  799. dict_time['time_registrationStart'].append((extract_time[0], 0.51, in_attachment))
  800. return dict_time
  801. # dict_time = add_time_by_parseDocument(dict_time,parse_document)
  802. # print(dict_time)
  803. result_dict = dict((key,"") for key in dict_time.keys())
  804. for time_type,value in dict_time.items():
  805. list_time = dict_time[time_type]
  806. if list_time:
  807. for in_attachment in [False,True]:
  808. _list_time = [_time for _time in list_time if _time[2]==in_attachment]
  809. if time_type in ['time_getFileEnd','time_registrationEnd','time_bidclose','time_bidopen']:
  810. # 过滤与page_time差距过大的时间(需加上公告类别判断[52,101,114])
  811. # print(time_type,_list_time,timeAdd(page_time,-7))
  812. tmp_list_time = []
  813. # for add_day in [ 1, 0, -7, -15, -31]:
  814. for add_day in [ 1, 0, -7, -15, -31, -60, -90]:
  815. tmp_list_time = [_time for _time in list_time if _time[0][:10] >= timeAdd(page_time,add_day) and _time[0][:10] < timeAdd(page_time,180)]
  816. if tmp_list_time:
  817. break
  818. _list_time = tmp_list_time
  819. else:
  820. if time_type not in ['time_contractEnd','time_completion']:
  821. _list_time = [_time for _time in list_time if _time[0][:10] >= timeAdd(page_time, -180) and _time[0][:10] < timeAdd(page_time,180)]
  822. else:
  823. _list_time = [_time for _time in list_time if _time[0][:10] >= timeAdd(page_time, -180)]
  824. if _list_time:
  825. _list_time.sort(key=lambda x:(x[1],len(x[0])),reverse=True) # sort_key: label_prob,时间文本长度(优先有具体时分秒的)
  826. if in_attachment==True and len(result_dict[time_type])>0:
  827. break
  828. result_dict[time_type] = _list_time[0][0]
  829. # print('time result_dict',result_dict)
  830. # result_dict 纠错
  831. if not result_dict['time_bidclose']:
  832. if result_dict['time_bidstart']: # 无截标时间,投标开始和开标时间一样
  833. if result_dict['time_bidstart'][:10] in result_dict['time_bidopen']:
  834. result_dict['time_bidstart'] = ""
  835. result_dict['time_bidclose'] = result_dict['time_bidopen']
  836. if not result_dict['time_bidclose']:
  837. if result_dict['time_getFileEnd']: # 无截标时间,获取文件截止时间和开标时间一样
  838. if result_dict['time_getFileEnd'][:10] in result_dict['time_bidopen']:
  839. result_dict['time_bidclose'] = result_dict['time_bidopen']
  840. else:
  841. if result_dict['time_bidopen']: # 截标时间 和 开标时间 时分秒互补
  842. if len(result_dict['time_bidclose'])<len(result_dict['time_bidopen']) and result_dict['time_bidclose'] in result_dict['time_bidopen']:
  843. result_dict['time_bidclose'] = result_dict['time_bidopen']
  844. elif len(result_dict['time_bidclose'])>len(result_dict['time_bidopen']) and result_dict['time_bidopen'] in result_dict['time_bidclose']:
  845. result_dict['time_bidopen'] = result_dict['time_bidclose']
  846. # 截标时间与开标时间差距过大时,互补修复
  847. if result_dict['time_bidopen'][:10]>=page_time:
  848. if result_dict['time_bidclose'][:10] < timeAdd(result_dict['time_bidopen'][:10], -7) and len(dict_time['time_bidclose'])>0:
  849. result_dict['time_bidclose'] = result_dict['time_bidopen'][:10]
  850. if result_dict['time_bidclose'][:10]>=page_time:
  851. if result_dict['time_bidopen'][:10] < timeAdd(result_dict['time_bidclose'][:10], -7) and len(dict_time['time_bidopen'])>0:
  852. result_dict['time_bidopen'] = result_dict['time_bidclose'][:10]
  853. # 开始结束对应时间纠错
  854. for begin_time_type,end_time_type in last_time_index.items():
  855. if result_dict[end_time_type] and result_dict[end_time_type][:10] < result_dict[begin_time_type][:10]:
  856. # print('error time',end_time_type,result_dict[end_time_type])
  857. result_dict[end_time_type] = ""
  858. return result_dict
  859. def getOtherAttributes(list_entity,page_time,prem,channel_dic):
  860. dict_other = {"moneysource":"",
  861. "person_review":[],
  862. "serviceTime":"",
  863. "product":[],
  864. "total_tendereeMoney":0,
  865. "total_tendereeMoneyUnit":''}
  866. list_serviceTime = []
  867. last_moneysource_prob = 0
  868. for entity in list_entity:
  869. if entity.entity_type == 'bidway':
  870. dict_other["bidway"] = turnBidWay(entity.entity_text)
  871. elif entity.entity_type=='moneysource':
  872. if dict_other["moneysource"] and entity.in_attachment:
  873. continue
  874. if not dict_other["moneysource"]:
  875. dict_other["moneysource"] = entity.entity_text
  876. last_moneysource_prob = entity.prob
  877. elif entity.prob>last_moneysource_prob:
  878. dict_other["moneysource"] = entity.entity_text
  879. last_moneysource_prob = entity.prob
  880. elif entity.entity_type=='serviceTime':
  881. # print(entity.entity_text)
  882. # if list_serviceTime and entity.in_attachment:
  883. # continue
  884. if re.search("[^之]日|天|年|月|周|星期", entity.entity_text) or re.search("\d{4}[-./]\d{1,2}", entity.entity_text):
  885. list_serviceTime.append(entity)
  886. elif entity.entity_type=="person" and entity.label ==4 and entity.entity_text not in dict_other["person_review"]: # 20240624评审专家去重
  887. dict_other["person_review"].append(entity.entity_text)
  888. elif entity.entity_type=='product' and entity.entity_text not in dict_other["product"]: #顺序去重保留
  889. dict_other["product"].append(entity.entity_text)
  890. elif entity.entity_type=='money' and entity.notes=='总投资' and float(entity.entity_text)>5000000 and float(dict_other["total_tendereeMoney"])<float(entity.entity_text): # 20250702 限制总投资为500万的以上
  891. dict_other["total_tendereeMoney"] = str(Decimal(entity.entity_text))
  892. dict_other["total_tendereeMoneyUnit"] = entity.money_unit
  893. time_contractEnd = prem[0].get("time_contractEnd","")[:10]
  894. time_contractStart = prem[0].get("time_contractStart","")[:10]
  895. serviceTime_dict = {"service_start":"", "service_end":"", "service_days": 0}
  896. if time_contractEnd:
  897. serviceTime_dict['service_end'] = time_contractEnd
  898. if time_contractStart:
  899. if get_days_between(time_contractStart,time_contractEnd)>0:
  900. serviceTime_dict['service_start'] = time_contractStart
  901. # print([i.entity_text for i in list_serviceTime])
  902. if list_serviceTime and not serviceTime_dict['service_end']:
  903. list_serviceTime_inAtt = [serviceTime for serviceTime in list_serviceTime if serviceTime.in_attachment==1]
  904. list_serviceTime = [serviceTime for serviceTime in list_serviceTime if serviceTime.in_attachment==0]
  905. error_serviceTime = []
  906. for list_time in [list_serviceTime,list_serviceTime_inAtt]:
  907. if not serviceTime_dict['service_end'] and not serviceTime_dict['service_days']:
  908. list_time.sort(key=lambda x: (x.prob,-x.sentence_index,-x.begin_index), reverse=True)
  909. for _serviceTime in list_time:
  910. # 优先取具体时间(20XX年x月x日-20XX年x月x日)
  911. if re.search("20\d{2}[年/.\-]\d{1,2}[月/.\-]\d{1,2}日?[^。\d半一二三四五六七八九十壹两叁贰肆伍陆柒捌玖拾;;]{,4}20\d{2}[年/.\-]\d{1,2}[月/.\-]\d{1,2}日?",_serviceTime.entity_text):
  912. _extract_time,_ = my_timeFormat(_serviceTime.entity_text,page_time)
  913. if _extract_time and len(_extract_time)==2:
  914. # 排除开始和结束时间一样的错误模板,例:“履约期限:2023年02月15日至2023年02月15日”
  915. if _extract_time[0]!=_extract_time[1]:
  916. # dict_other["serviceTime"] = _serviceTime.entity_text
  917. # extract_time = extract_serviceTime(_serviceTime.entity_text)
  918. # if extract_time['service_end']:
  919. serviceTime_dict['service_start'] = _extract_time[0]
  920. serviceTime_dict['service_end'] = _extract_time[1]
  921. break
  922. else:
  923. error_serviceTime.append(_serviceTime.entity_text)
  924. if not serviceTime_dict['service_end']:
  925. for _serviceTime in list_time:
  926. # 优先取具体时间(20XX年x月-20XX年x月)
  927. if re.search("20\d{2}[年/.\-]\d{1,2}月?[^。\d半一二三四五六七八九十壹两叁贰肆伍陆柒捌玖拾;;]{,3}20\d{2}[年/.\-]\d{1,2}月?", _serviceTime.entity_text):
  928. # dict_other["serviceTime"] = _serviceTime.entity_text
  929. extract_time = extract_serviceTime(_serviceTime.entity_text,page_time)
  930. if extract_time['service_end']:
  931. serviceTime_dict = extract_time
  932. break
  933. if not serviceTime_dict['service_end']:
  934. for _serviceTime in list_time:
  935. # 优先取具体时间(20XX年x月x日)
  936. if re.search("20\d{2}[年/.\-]\d{1,2}[月/.\-]\d{1,2}日?",_serviceTime.entity_text):
  937. if _serviceTime.entity_text not in error_serviceTime:
  938. # dict_other["serviceTime"] = _serviceTime.entity_text
  939. extract_time = extract_serviceTime(_serviceTime.entity_text,page_time)
  940. if extract_time['service_end']:
  941. serviceTime_dict = extract_time
  942. break
  943. if not serviceTime_dict['service_end'] and not serviceTime_dict['service_days']:
  944. for _serviceTime in list_time:
  945. if _serviceTime.entity_text not in error_serviceTime:
  946. # dict_other["serviceTime"] = _serviceTime.entity_text
  947. extract_time = extract_serviceTime(_serviceTime.entity_text,page_time)
  948. # service_days > 3
  949. if extract_time['service_end'] or extract_time['service_days']>3:
  950. serviceTime_dict = extract_time
  951. break
  952. # 若上一步仍无结果,取消service_days > 3 的条件
  953. if not serviceTime_dict['service_end'] and not serviceTime_dict['service_days']:
  954. for _serviceTime in list_time:
  955. if _serviceTime.entity_text not in error_serviceTime:
  956. # dict_other["serviceTime"] = _serviceTime.entity_text
  957. extract_time = extract_serviceTime(_serviceTime.entity_text,page_time)
  958. if extract_time['service_end'] or extract_time['service_days']:
  959. serviceTime_dict = extract_time
  960. break
  961. if serviceTime_dict['service_start'] and serviceTime_dict['service_end']:
  962. service_days = get_days_between(serviceTime_dict['service_start'],serviceTime_dict['service_end'])
  963. serviceTime_dict['service_days'] = service_days
  964. dict_other["serviceTime"] = serviceTime_dict
  965. if not time_contractEnd and channel_dic['docchannel']['docchannel']=='合同公告': # 用serviceTime补充合同开始结束时间,公告类型为合同公告
  966. if serviceTime_dict['service_start'] and serviceTime_dict['service_end']:
  967. prem[0]["time_contractStart"] = serviceTime_dict['service_start']
  968. prem[0]["time_contractEnd"] = serviceTime_dict['service_end']
  969. if dict_other['moneysource']:
  970. dict_other['moneysource'] = turnMoneySource(dict_other['moneysource'])
  971. # dict_other["product"] = list(set(dict_other["product"])) # 已在添加时 顺序去重保留
  972. return dict_other