re_bidway.py 27 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626
  1. # DEPRECATED(Phase1): 本文件中的 PostgreSQL 硬编码连接(host=192.168.*,
  2. # password=postgres 等)将在后续 training/ Phase 迁移到 BiddingKG.dl.infra.db。
  3. # 迁移完成前可临时使用:from BiddingKG.dl.infra.db import get_connection;
  4. # conn = get_connection("<dbname>")
  5. # 详见 ARCHITECTURE.md 第 11 章 Phase 1 与 REFACTOR_LOG.md。
  6. import ast
  7. import pandas as pd
  8. import re
  9. # from BiddingKG.dl.interface import Entitys
  10. # def re_bidway_old(text):
  11. # df = pd.read_csv("C:\\Users\\admin\\Desktop\\bidway_text.csv")
  12. #
  13. # reg = re.compile(u'(采购方式|竞价方式|招标方式|询价类型|交易方式|寻源策略|招标形式|询价方式'
  14. # u'|发包方式|发包类型|开展方式|招标类型)(.*)'
  15. # u'(公开招标|竞争性磋商|竞争性谈判|公开采购|单一来源'
  16. # u'|电子书面竞投|邀请招标|定向公开|询价采购|抽签摇号'
  17. # u'|网上电子投标|比质比价|询单|询比采购|比选|单一来源采购'
  18. # u'|网上招标|其他'
  19. # u'|竞谈竞价|网上直购|公开竞谈'
  20. # u'|库内邀请|库内公开发包)')
  21. #
  22. # # reg = re.compile(u'(公开招标|竞争性磋商|竞争性谈判采购|公开采购|单一来源|网络竞价'
  23. # # u'|竞争性谈判|公开询价|邀请招标|公开招募|公开询比价|电子书面竞投'
  24. # # u'|网上电子投标|比质比价|定向询单|国内比选|电子竞价'
  25. # # u'|公开招租|公开竞标方式|网上招标|公开招标|国内竞争性谈判'
  26. # # u'|国内竞争性磋商|公开竞谈|定向询价|网上询价|网上竞价|公开比选|磋商采购|网上直购'
  27. # # u'|库内邀请|询价采购|询比采购|分散采购|单一来源采购)')
  28. #
  29. # reg2 = re.compile(u'(采用|以|)'
  30. # u'(公开招标|竞争性磋商|竞争性谈判|公开采购|单一来源'
  31. # u'|竞争性谈判|询价|电子书面竞投|电子竞价'
  32. # u'|网上电子投标|比质比价|询单|询比采购|比选|单一来源采购'
  33. # u'|网上招标|分散采购'
  34. # u'|竞谈竞价|网上直购|公开竞谈'
  35. # u'|库内邀请)'
  36. # u'(采购方式|方式)')
  37. #
  38. # reg1 = re.compile(
  39. # # u'(公开招标|竞争性磋商|竞争性谈判采购|公开采购|单一来源采购|网络竞价|公开招商方式'
  40. # # u'|竞争性谈判|公开询价|询价采购|邀请招标|公开招募|公开询比|电子书面竞投'
  41. # # u'|网上电子投标|比质比价|定向询单|询比采购|国内比选|单一来源|公开选取|库内公开发包'
  42. # # u'|公开招租|公开竞标方式|网上招标|公开招标|竞争性谈判|公开招投标'
  43. # # u'|国内竞争性磋商|公开竞谈|定向询价|网上询价|网上竞价|公开比选|磋商采购|网上直购'
  44. # # u'|国际公开竞争性招标)'
  45. # u'(公开招标|竞争性磋商|竞争性谈判|公开采购|单一来源'
  46. # u'|竞争性谈判|询价|电子书面竞投'
  47. # u'|网上电子投标|比质比价|询单|询比采购|比选|单一来源采购'
  48. # u'|网上招标|分散采购'
  49. # u'|竞谈竞价|网上直购|公开竞谈'
  50. # u'|库内邀请)'
  51. # )
  52. #
  53. # # 都切为4个字符
  54. # # reg1_not = re.compile(u'(及单一来源|询价小组成员|除单一来源|竞争性谈判邀请函|询价记录)')
  55. # reg1_not = re.compile(u'(及单一来|价小组成|除单一来|性谈判邀|询价记录)')
  56. #
  57. # reg3 = re.compile(u'(采购方式:邀请|采购方式:公开|采购方式:询价|分散采购|公开招标|竞价|磋商|询比|竞标|邀请招标|公开招募|公开招租)')
  58. #
  59. #
  60. # reg_standard = re.compile(u'(公开招标|竞争性磋商|竞争性谈判|单一来源'
  61. # u'|竞争性谈判|询价|邀请招标|公开招募|询比|电子书面竞投'
  62. # u'|网上电子投标|比质比价|询单|比选'
  63. # u'|公开招租|网上招标|分散采购'
  64. # u'|网上直购|公开竞谈|采购方式:邀请|采购方式:公开|采购方式:询价)'
  65. # )
  66. #
  67. # text_list = df["text"].to_list()
  68. # # text_list = []
  69. # # text_list.append(text)
  70. # text_index_list = []
  71. # output_list = []
  72. # for index in range(len(text_list)):
  73. # # 全文下标
  74. # text_index = [0, 0]
  75. #
  76. # input_str = text_list[index]
  77. #
  78. # # 把一些混淆的词先替换掉
  79. # input_str = re.sub(reg1_not, "####", input_str)
  80. #
  81. # match = reg.search(input_str)
  82. # output_str = None
  83. # # 根据正则表达式匹配
  84. # if match:
  85. # # 更新全文下标
  86. # text_index[0] = match.start()
  87. # text_index[1] = match.end()
  88. #
  89. # # 判断长度,截断
  90. # if len(match.group()) >= 15:
  91. # ss = re.split(",|\.|,|。|;|;", match.group())
  92. # # 判断所需的字符串在哪一段
  93. # for i in range(len(ss)):
  94. # if re.search(reg1, ss[i]):
  95. # output_str = ss[i]
  96. #
  97. # # 更新全文下标
  98. # front_len, back_len = calculateLen(ss, i)
  99. # text_index[0] = text_index[0] + front_len + i
  100. # text_index[1] = text_index[1] - back_len + len(ss) -1 - i
  101. #
  102. # break
  103. # else:
  104. # output_str = match.group()
  105. #
  106. # else:
  107. # match2 = re.search(reg2, input_str)
  108. # if match2:
  109. # # 更新全文下标
  110. # text_index[0] = match2.start()
  111. # text_index[1] = match2.end()
  112. #
  113. # output_str = match2.group()
  114. #
  115. # else:
  116. # match1 = re.search(reg1, input_str)
  117. # if match1:
  118. # # 更新全文下标
  119. # text_index[0] = match1.start()
  120. # text_index[1] = match1.end()
  121. # output_str = match1.group()
  122. #
  123. # # 再判断一次长度
  124. # if output_str is not None:
  125. # if len(output_str) >= 15:
  126. # match2 = re.search(reg2, input_str)
  127. # if match2:
  128. # # 更新全文下标
  129. # text_index[0] = match2.start()
  130. # text_index[1] = match2.end()
  131. #
  132. # output_str = match2.group()
  133. # if len(output_str) >= 15:
  134. # match1 = re.search(reg1, input_str)
  135. # if match1:
  136. # # 更新全文下标
  137. # text_index[0] = match1.start()
  138. # text_index[1] = match1.end()
  139. #
  140. # output_str = match1.group()
  141. #
  142. # # 最后输出还为空,匹配一些易混淆的词
  143. # if output_str is None:
  144. # match3 = re.search(reg3, input_str)
  145. # if match3:
  146. # # 更新全文下标
  147. # text_index[0] = match3.start()
  148. # text_index[1] = match3.end()
  149. #
  150. # output_str = match3.group()
  151. #
  152. # # 处理前缀等无用词
  153. # if output_str is not None:
  154. # match5 = re.search("分散采购|采购方式:邀请", output_str)
  155. # if not match5:
  156. # # 公开采购转为公开招标
  157. # output_str = re.sub("公开采购", "公开招标", output_str)
  158. #
  159. # # 去掉第一个字符冒号
  160. # ss = re.split(":|:", output_str)
  161. # output_str = ss[-1]
  162. # # 更新全文下标
  163. # front_len, back_len = calculateLen(ss, len(ss) - 1)
  164. # text_index[0] = text_index[0] + front_len + len(ss) - 1
  165. #
  166. # # 去掉采购、方式、采用
  167. # match6 = re.search("(采用|出售|直接(|现就本次|招标为)", output_str)
  168. # match7 = re.search("(采购|方式|进行)", output_str)
  169. # output_str = re.sub("(采购|方式|采用|出售|进行|直接(|现就本次|招标为)", "", output_str)
  170. # # 更新全文下标
  171. # if match6:
  172. # text_index[0] += match6.end() - match6.start()
  173. # if match7:
  174. # text_index[1] -= match7.end() - match7.start()
  175. #
  176. # # 使用标准标签过滤
  177. # match4 = re.search(reg_standard, output_str)
  178. # if match4:
  179. # output_str = match4.group()
  180. # # 更新全文下标
  181. # text_index[0] += match4.start()
  182. # text_index[1] = text_index[0] + match4.end() - match4.start()
  183. #
  184. # output_list.append(output_str)
  185. # # text_index_list.append(str(text_index))
  186. # text_index_list.append(text_index)
  187. #
  188. # # df["re"] = pd.DataFrame(output_list)
  189. # # df["text_index"] = pd.DataFrame(text_index_list)
  190. #
  191. # # index_to_word = []
  192. # # for index, row in df.iterrows():
  193. # # i_list = ast.literal_eval(row["text_index"])
  194. # # word = row["text"][i_list[0]:i_list[1]]
  195. # # if len(word) >= 20:
  196. # # word = ""
  197. # # index_to_word.append(word)
  198. #
  199. #
  200. # # df["index2word"] = pd.DataFrame(index_to_word)
  201. # # df.to_csv("C:\\Users\\admin\\Desktop\\bidway_text2.csv")
  202. #
  203. # return output_list[0], text_index_list[0]
  204. normal_bidway = "公开招标|邀请招标|竞争性谈判|竞争性磋商|单一来源|框架协议|询价"
  205. bidway = '单一来源' \
  206. '|国内竞争性磋商|竞争性磋商|竞争性谈判|网络竞价|网上竞价|公开竞谈|公开竞价|电子竞价|竞价|竞标|竞谈竞价|电子书面竞投' \
  207. '|公开比选|比质比价|比选' \
  208. '|公开招标|公开招租|公开招募|公开选取|公开招投标' \
  209. '|网上直购|网上招标|网上电子投标|网上挂牌' \
  210. '|邀请招标' \
  211. '|网上询价|公开询价|非定向询价|定向询价|询比价|询单|询价|询比' \
  212. '|库内邀请|库内公开发包|内部邀标' \
  213. '|定点采购议价|定点采购' \
  214. '|竞争性评审|框架协议'
  215. not_bidway = '及单一来源|询价小组成员|除单一来源|竞争性谈判邀请函|询价记录|自由竞价' \
  216. '|限时竞价|咨询单位|询价单'
  217. not_bidway_preffix = "本次|拟|参加|无效|标的|联合体|参与|否决|除|可以选择|包括|涉及|非"
  218. not_bidway_suffix = "文件|报名|邀请|项目|失败|数量|编号|后|时间|类型|名称|和|成交" \
  219. "|标题|开始|结束|产品|报价|供应商|部门|监督|需求|范围|入围|内容|人" \
  220. "|条件|公司|保证金|完毕|事件|成功|活动|地点|标|会|须知|范围" \
  221. "|响应|报价|采购公示|的原因|采购供应商|价|采购人员|失败|小组"
  222. bidway_preffix = '采购方式|竞价方式|招标方式|询价类型|交易方式|寻源策略|招标形式|询价方式' \
  223. '|发包方式|发包类型|开展方式|招标类型|选取方式|招租方式'
  224. bidway_special = '采购方式:公开|采购方式:邀请|采购方式:询价' \
  225. '|招标方式:.公开|采购方式:.公开' \
  226. '|分散采购' \
  227. ''
  228. def re_not_bidway(_str):
  229. match = re.findall(not_bidway, _str)
  230. if match:
  231. for word in match:
  232. instead = "#" * len(word)
  233. _str = re.sub(word, instead, _str)
  234. reg_not1 = "(" + bidway + ")" + "(" + not_bidway_suffix + ")"
  235. match = re.findall(reg_not1, _str)
  236. if match:
  237. for word in match:
  238. word_add = ""
  239. for w in word:
  240. word_add += w
  241. instead = "#" * len(word_add)
  242. _str = re.sub(word_add, instead, _str)
  243. reg_not2 = "(" + not_bidway_preffix + ")" + "(" + bidway + ")"
  244. match = re.findall(reg_not2, _str)
  245. if match:
  246. for word in match:
  247. word_add = ""
  248. for w in word:
  249. word_add += w
  250. instead = "#" * len(word_add)
  251. _str = re.sub(word_add, instead, _str)
  252. return _str
  253. def re_standard_bidway(_str):
  254. reg_standard = "(?P<preffix>" + bidway_preffix + ")" \
  255. + "(?P<char>.{1,2})" \
  256. + "(?P<value>" + bidway + ")"
  257. match = re.finditer(reg_standard, _str)
  258. bidway_list = []
  259. if match:
  260. for m in match:
  261. keyword = m.group('value')
  262. keyword_index = list(m.span('value'))
  263. behind_str = _str[m.start(): m.end()+30]
  264. if len(re.findall(normal_bidway, behind_str))>1:
  265. keyword = ''
  266. for it in re.finditer('(?P<sign>.{1,2})(?P<bidway>'+normal_bidway+')+', behind_str): # 招标方式后面多个选择处理
  267. if '□' != it.group('sign')[-1]:
  268. keyword = it.group('bidway')
  269. keyword_index = [m.start()+it.start('bidway'), m.start()+it.end('bidway')]
  270. break
  271. # m_dict = m.groupdict()
  272. # m_span = m.span()
  273. # keyword = ""
  274. # keyword_index = [m_span[0], m_span[1]]
  275. # for key in m_dict.keys():
  276. # if key == "value":
  277. # keyword = m_dict.get(key)
  278. # else:
  279. # keyword_index[0] += len(m_dict.get(key))
  280. bidway_list.append([keyword, keyword_index])
  281. return bidway_list
  282. def re_normal_bidway(_str):
  283. ser = re.search("("+normal_bidway+")(转为?|变更为|更改为)"+"(?P<bidway>(" + normal_bidway + "))", _str) # 如果方式变更取变更后的
  284. if ser:
  285. return [[ser.group('bidway'), list(ser.span('bidway'))]]
  286. reg_all = "(?P<value>" + normal_bidway + ")"
  287. match = re.finditer(reg_all, _str)
  288. bidway_list = []
  289. bidway_set = set()
  290. if match:
  291. for m in match:
  292. keyword = m.group()
  293. if keyword == '公开招标' and m.start()>0 and _str[m.start()-1]=='非':
  294. continue
  295. keyword_index = list(m.span())
  296. bidway_set.add(keyword)
  297. bidway_list.append([keyword, keyword_index])
  298. if len(bidway_list) == 0: # 如果找不到标准方式,匹配简称方式
  299. ser = re.search('(?P<bidway>(磋商|谈判))(公告|成交|结果)', _str)
  300. if ser:
  301. return [[ser.group('bidway'), list(ser.span('bidway'))]]
  302. if len(bidway_set) > 1: # 匹配到多种招标方式返回空
  303. return []
  304. return bidway_list
  305. def re_all_bidway(_str):
  306. reg_all = "(?P<value>" + normal_bidway + ")" # 优先匹配规范的招标方式
  307. match = re.finditer(reg_all, _str)
  308. bidway_list = []
  309. if match:
  310. for m in match:
  311. keyword = m.group()
  312. keyword_index = list(m.span())
  313. bidway_list.append([keyword, keyword_index])
  314. return bidway_list
  315. reg_all = "(?P<value>" + bidway + ")"
  316. match = re.finditer(reg_all, _str)
  317. bidway_list = []
  318. if match:
  319. for m in match:
  320. keyword = m.group()
  321. keyword_index = list(m.span())
  322. bidway_list.append([keyword, keyword_index])
  323. return bidway_list
  324. def re_special_bidway(_str):
  325. reg_special = "(?P<value>" + bidway_special + ")"
  326. match = re.finditer(reg_special, _str)
  327. bidway_list = []
  328. if match:
  329. for m in match:
  330. keyword = m.group()
  331. keyword_index = list(m.span())
  332. bidway_list.append([keyword, keyword_index])
  333. return bidway_list
  334. def get_one_word(bidway_list):
  335. # 若有多个,去重,输出较长的
  336. word = None
  337. text_index = [0, 0]
  338. if len(bidway_list) > 1:
  339. word_dict = {}
  340. for bw in bidway_list:
  341. if bw[0] in word_dict.keys():
  342. if bw[1][0] < word_dict.get(bw[0])[0]:
  343. word_dict[bw[0]] = bw[1]
  344. else:
  345. word_dict[bw[0]] = bw[1]
  346. word_list = []
  347. for key in word_dict.keys():
  348. word_list.append([key, word_dict.get(key)[0]])
  349. if len(word_list) > 1:
  350. word_list.sort(key=lambda x: (-int(x[1]), len(x[0])))
  351. word = word_list[-1][0]
  352. text_index = word_dict.get(word)
  353. elif word_list:
  354. word = word_list[0][0]
  355. text_index = word_dict.get(word)
  356. else:
  357. text_index = [0, 0]
  358. elif len(bidway_list) == 1:
  359. word = bidway_list[0][0]
  360. text_index = bidway_list[0][1]
  361. return word, text_index
  362. def re_bidway(text, title):
  363. # 优先匹配标题标准招标方式
  364. if len(title)<100:
  365. bidway_list = re_normal_bidway(title)
  366. if bidway_list:
  367. word, text_index = get_one_word(bidway_list)
  368. return word, text_index
  369. # 替换易混淆词
  370. text_clean = re_not_bidway(text)
  371. title_clean = re_not_bidway(title)
  372. # 查找符合标准形式的
  373. bidway_list = re_standard_bidway(text_clean)
  374. if bidway_list:
  375. word = bidway_list[0][0]
  376. text_index = bidway_list[0][1]
  377. return word, text_index
  378. # 无符合标准形式的,查找title里的所有形式
  379. bidway_list = re_all_bidway(title_clean)
  380. if bidway_list:
  381. word, text_index = get_one_word(bidway_list)
  382. return word, text_index
  383. # 无符合标准形式的,查找所有形式
  384. bidway_list = re_all_bidway(text_clean)
  385. if bidway_list:
  386. word, text_index = get_one_word(bidway_list)
  387. return word, text_index
  388. # 还无结果,查找特殊形式
  389. bidway_list = re_special_bidway(text_clean)
  390. if bidway_list:
  391. word = bidway_list[0][0]
  392. text_index = bidway_list[0][1]
  393. return word, text_index
  394. # 查无结果
  395. return None, [0, 0]
  396. def extract_bidway(text, title):
  397. list_bidway = []
  398. word, text_index_list = re_bidway(text, title)
  399. if word is not None:
  400. if text_index_list[1]-text_index_list[0] != len(word) \
  401. or text_index_list[1]-text_index_list[0] >= 10:
  402. return []
  403. d = {"body": word, "begin_index": text_index_list[0], "end_index": text_index_list[1]}
  404. list_bidway.append(d)
  405. # print(d.get("body"), d.get("begin_index"), d.get("end_index"))
  406. return list_bidway
  407. bidway_dict = {'询价': '询价', '竞争性谈判': '竞争性谈判',
  408. '公开比选': '其他', '国内竞争性磋商': '竞争性磋商',
  409. '招标方式:t公开': '公开招标', '竞价': '竞价',
  410. '竞标': '竞价', '电子竞价': '竞价',
  411. '电子书面竞投': '竞价', '单一来源': '单一来源',
  412. '网上竞价': '竞价', '公开招标': '公开招标',
  413. '询比': '询价', '定点采购': '其他',
  414. '招标方式:■公开': '公开招标', '交易其他,付款其他': '其他',
  415. '竞争性评审': '竞争性磋商', '公开招租': '其他', '\\N': '',
  416. '比选': '其他', '比质比价': '其他', '分散采购': '其他',
  417. '内部邀标': '邀请招标', '邀请招标': '邀请招标',
  418. '网上招标': '公开招标', '非定向询价': '询价',
  419. '网络竞价': '竞价', '公开询价': '询价',
  420. '定点采购议价': '其他', '询单': '询价',
  421. '网上挂牌': '其他', '网上直购': '其他',
  422. '定向询价': '询价', '采购方式:公开': '公开招标',
  423. '磋商': '竞争性磋商', '公开招投标': '公开招标',
  424. '招标方式:√公开': '公开招标', '公开选取': '公开招标',
  425. '网上电子投标': '公开招标', '公开竞谈': '竞争性谈判',
  426. '竞争性磋商': '竞争性磋商', '采购方式:邀请': '邀请招标',
  427. '公开竞价': '竞价', '其他': '其他', '公开招募': '其他',
  428. '网上询价': '询价', '框架协议': '框架协议', '谈判':'竞争性谈判'}
  429. # bidway名称统一规范
  430. def bidway_integrate(bidway):
  431. integrate_name = bidway_dict.get(bidway,"其他")
  432. return integrate_name
  433. def bidway_normalize(key):
  434. if re.search('公开招标|公开发包', key):
  435. return '公开招标'
  436. elif re.search('单一来源', key):
  437. return '单一来源'
  438. elif re.search('磋商', key):
  439. return '竞争性磋商'
  440. elif re.search('谈判', key):
  441. return '竞争性谈判'
  442. elif re.search('竞谈|竞价|竞投|竞标', key):
  443. return '竞价'
  444. elif re.search('询价|询比|比价|询单', key):
  445. return '询价'
  446. elif re.search('邀请|邀标', key):
  447. return '邀请招标'
  448. else:
  449. return bidway_dict.get(key, '其他')
  450. def test_csv():
  451. df = pd.read_csv("C:\\Users\\Administrator\\Desktop\\bidway_text.csv")
  452. predict_list = []
  453. for index, row in df.iterrows():
  454. word, text_index = re_bidway(row["text"], "")
  455. if word:
  456. predict = [word, text_index]
  457. else:
  458. predict = []
  459. print("predict", predict)
  460. predict_list.append(str(predict))
  461. predict_df = pd.DataFrame(predict_list)
  462. df = pd.concat([df, predict_df], axis=1)
  463. df.to_csv("C:\\Users\\Administrator\\Desktop\\bidway_result.csv")
  464. print("finish write!")
  465. def test_str():
  466. s = '政府采购项目招标方式:公开招标,联系人:黎明。代理机构地址:广州市天河区'
  467. s = '''
  468. ,关于人防工程技术咨询服务项目【重新招标】单一来源谈判的通知,各投标人:深圳市国际招标有限公司受中共
  469. 深圳市委军民融合发展委员会办公室委托,就人防工程技术咨询服务项目【重新招标】(项目编号:0658-2171
  470. 1A60965),进行公开招标,因投标单位不足三家,公开招标失败,现经采购单位同意,采用单一来源谈判方式
  471. 确定中标供应商,邀请中国建筑标准设计研究院有限公司前来谈判,一、项目编号:0658-21711A60965,二
  472. 、项目名称:人防工程技术咨询服务项目【重新招标】,三、凡被邀请参加谈判的供应商必须按照原招标文件第
  473. 六章要求制作谈判文件正本一本,副本二本,按规定的时间密封递交并参加谈判,四、谈判内容:投标价格、项
  474. 目实施方案、售后服务方案和其它相关事项,五、地点及时间:1、因疫情影响本项目谈判响应文件采用邮寄方
  475. 式接收文件,2、文件接收截止时间:2021年11月5日14:30(北京时间),3、谈判响应文件邮寄地址:深圳
  476. 市罗湖区嘉宾路2018号深华商业大厦裙楼6层600A。收件人:郑工,电话:18806665013,3、谈判地点:
  477. 线上谈判,六、谈判的相关规则按原招标文件的相应规定执行;有关谈判事宜详见招标文件第六章《公开招标失
  478. 败后后续采购程序和投标须知》,1、采购人信息,名称:中共深圳市委军民融合发展委员会办公室,地址:深
  479. 圳市福田区新洲路5008号,联系方式:刘先生,电话:0755-88100332,2、采购代理机构信息,名称:深
  480. 圳市国际招标有限公司,地址:罗湖总部:深圳市罗湖区嘉宾路2018号深华商业大厦裙楼6层,深圳湾总部:深
  481. 圳市南山区沙河西路与白石路交汇处深圳湾科技生态园9栋B4座6楼,联系方式:0755-22918634,监督举报
  482. 电话:0755-22965602、0755-86660475,特此通知,深圳市国际招标有限公司,2021年11月1日,更多
  483. 咨询报价请点击:http://zbcloud.net/bidbulletin/69495.htm,
  484. '''
  485. print(extract_bidway(s, title=""))
  486. def test_html():
  487. # html_path = "C:/Users/Administrator/Desktop/3.html"
  488. html_path = 'd:/html/2.html'
  489. with open(html_path, "r", encoding='utf-8') as f:
  490. s = f.read()
  491. print(extract_bidway(s, title=""))
  492. def get_valuate():
  493. import psycopg2
  494. conn = psycopg2.connect(host='192.168.2.103', port='5432', user='postgres', password='postgres', dbname='iepy')
  495. cursor = conn.cursor()
  496. sql = "select c1.docid, c1.doctitle, c1.extract_json, c2.text from corpus_otherinput c1 left join corpus_iedocument c2 on c1.docid=c2.human_identifier where c1.new_extract notnull;" # where docid='110635873'
  497. # sql = "select c1.docid, c1.doctitle from corpus_otherinput c1;"
  498. # sql = "select text from corpus_iedocument limit 50000;"
  499. cursor.execute(sql)
  500. datas = []
  501. olds = []
  502. news = []
  503. label_old = []
  504. label_new = []
  505. labels = []
  506. for row in cursor.fetchall():
  507. docid = row[0]
  508. doctitle = row[1]
  509. ex = row[2]
  510. text = row[3]
  511. ser = re.search('"bidway": "(\w{,6})"', ex)
  512. # print('ser:', ser)
  513. old = ser.group(1) if ser else ""
  514. pred = extract_bidway(text, title=doctitle)
  515. # list_bidway = extract_bidway(text, title=doctitle)
  516. # print('list_bidway', list_bidway)
  517. # if list_bidway:
  518. # bidway = list_bidway[0].get("body")
  519. # # bidway名称统一规范
  520. # bidway = bidway_integrate(bidway)
  521. # else:
  522. # bidway = ""
  523. # print('bidway: ', bidway)
  524. pred = pred[0]['body'] if len(pred) > 0 else ""
  525. new = bidway_dict.get(pred, "其他") if pred!="" else ""
  526. sql2 = "select value from brat_bratannotation where document_id='{0}' and value like '%bidway%' limit 4;".format(docid)
  527. cursor.execute(sql2)
  528. lb_new = docid + "_"
  529. lb_old = docid + "_"
  530. tmp_l = []
  531. for row in cursor.fetchall():
  532. lb = row[0].split()[-1]
  533. lb = bidway_dict.get(lb, "其他") # 新准确率:0.9642, 召回率: 0.9642, F1: 0.8965
  534. # lb = bidway_normalize(lb) # 旧准确率:0.9287, 召回率: 0.9287, F1: 0.8011 新准确率:0.9692, 召回率: 0.9692, F1: 0.9105
  535. tmp_l.append(lb)
  536. if lb == new:
  537. lb_new = docid + "_" + lb
  538. if lb == old:
  539. lb_old = docid + "_" + lb
  540. olds.append(docid + "_" + old)
  541. news.append(docid + "_" + new)
  542. label_new.append(lb_new)
  543. label_old.append(lb_old)
  544. labels.append(';'.join(tmp_l))
  545. datas.append((docid, docid + "_" + old, lb_old, docid + "_" + new, lb_new, ';'.join(tmp_l)))
  546. eq_old = len(set(olds)&set(label_old))
  547. eq_new = len(set(news)&set(label_new))
  548. acc_old = eq_old/len(set(olds))
  549. recall_old = eq_old/len(set(label_old))
  550. f1_old = acc_old*recall_old/2*(acc_old+recall_old)
  551. acc_new = eq_new/len(set(news))
  552. recall_new = eq_new/len(set(label_new))
  553. f1_new = acc_new*recall_new/2*(acc_new+recall_new)
  554. print('旧准确率:%.4f, 召回率: %.4f, F1: %.4f'%(acc_old, recall_old, f1_old))
  555. print('新准确率:%.4f, 召回率: %.4f, F1: %.4f'%(acc_new, recall_new, f1_new))
  556. df = pd.DataFrame(datas, columns=['docid', 'pred_old', 'label_old', 'pred_new', 'label_new', 'labels'])
  557. df['old_pos'] = df.apply(lambda x:1 if x['pred_old']==x['label_old'] else 0, axis=1)
  558. df['new_pos'] = df.apply(lambda x:1 if x['pred_new']==x['label_new'] else 0, axis=1)
  559. df.to_csv('E:/其他数据/招标方式预测结果.csv', index=False)
  560. if __name__ == "__main__":
  561. # extract_bidway(s)
  562. # test_csv()
  563. test_str()
  564. # test_html()
  565. pass