label.py 41 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753754755756757758759760761762763764765766767768769770771772773774775776777778779780781782783784785786787788789790791792793794795796797798799800801802803804805806807808809810811812813814815816817818819820821822823824825826827828829830831832833834835836837838839840841842843844845846847848849850851852853854855856
  1. # -*- coding: utf-8 -*-
  2. """标签类预测器。
  3. 按 ARCHITECTURE.md Phase 5 拆分建议,从 ``interface/predictor.py`` 迁出以下
  4. 标签相关类:
  5. - ``ProjectLabel`` — 项目标签提取(原 predictor.py 第 5585-5891 行)
  6. - ``IndustryLabel`` — 行业标签分类(原 5895-5942 行)
  7. - ``PropertyLabel`` — 产权分类二级标签(原 6041-6201 行)
  8. - ``BiddingScore`` — 评分表投标人得分/排名提取(原 9835-10039 行)
  9. 原 ``from common.Utils import *`` / ``from interface.modelFactory import *``
  10. 已替换为显式 import;``os.path.dirname(__file__)`` 路径引用替换为
  11. ``predictors._common.INTERFACE_DIR``。
  12. ``interface/predictor.py`` 仍保留原定义,老 import 不受影响。
  13. """
  14. from __future__ import absolute_import
  15. import re
  16. import pandas as pd
  17. from bs4 import BeautifulSoup
  18. from BiddingKG.dl.interface.classification_process import classify_text, process_rules
  19. from BiddingKG.dl.predictors._common import INTERFACE_DIR, get_role
  20. from BiddingKG.dl.predictors.table_prem import TableTag2List, get_header_line
  21. __all__ = [
  22. "ProjectLabel",
  23. "IndustryLabel",
  24. "PropertyLabel",
  25. "BiddingScore",
  26. ]
  27. class ProjectLabel():
  28. def __init__(self, ):
  29. self.keyword_list = self.get_label_keywords()
  30. self.kongjing_keyword_list = self.get_kongjing_keywords()
  31. self.ICT_smart_compute_keyword_list = self.get_ICT_smart_compute_keywords()
  32. def get_label_keywords(self):
  33. import csv
  34. path = INTERFACE_DIR+'/project_label_keywords.csv'
  35. with open(path, 'r',encoding='utf-8') as f:
  36. reader = csv.reader(f)
  37. key_word_list = []
  38. for r in reader:
  39. if r[0] == '类型':
  40. continue
  41. type = r[0]
  42. key_wrod = r[1]
  43. key_paichuci = str(r[2])
  44. key_paichuci = key_paichuci if key_paichuci and key_paichuci != 'nan' else ""
  45. type_paichuci = str(r[3])
  46. type_paichuci = type_paichuci if type_paichuci and type_paichuci != 'nan' else ""
  47. key_word_list.append((type, key_wrod, key_paichuci, type_paichuci))
  48. return key_word_list
  49. def get_kongjing_keywords(self):
  50. import csv
  51. path = INTERFACE_DIR+'/kongjing_label_keywords.csv'
  52. with open(path, 'r',encoding='utf-8') as f:
  53. reader = csv.reader(f)
  54. key_word_list = []
  55. for r in reader:
  56. if r[0] == '关键词':
  57. continue
  58. key_wrod = r[0]
  59. key_wrod2 = str(r[1])
  60. key_wrod2 = key_wrod2 if key_wrod2 and key_wrod2 != 'nan' else ""
  61. search_type = r[2]
  62. info_type_list = str(r[3])
  63. info_type_list = info_type_list if info_type_list and info_type_list != 'nan' else ""
  64. key_word_list.append((key_wrod, key_wrod2, search_type, info_type_list))
  65. return key_word_list
  66. def get_ICT_smart_compute_keywords(self):
  67. import csv
  68. path = INTERFACE_DIR+'/ICT智算_label_keywords.csv'
  69. with open(path, 'r',encoding='utf-8') as f:
  70. reader = csv.reader(f)
  71. key_word_list = []
  72. for r in reader:
  73. if r[0] == '关键词':
  74. continue
  75. key_wrod = r[0]
  76. key_wrod2 = str(r[1])
  77. key_wrod2 = key_wrod2 if key_wrod2 and key_wrod2 != 'nan' else ""
  78. search_type = r[2]
  79. info_type_list = str(r[3])
  80. info_type_list = info_type_list if info_type_list and info_type_list != 'nan' else ""
  81. key_word_list.append((key_wrod, key_wrod2, search_type, info_type_list))
  82. return key_word_list
  83. def extract_core_text(self,all_text,tenderee="",agency=""):
  84. # 剔除 招标单位、代理机构名称
  85. if tenderee:
  86. all_text = all_text.replace(tenderee, " ")
  87. if agency:
  88. all_text = all_text.replace(agency, " ")
  89. # 定义需要匹配的关键词列表
  90. keywords = [
  91. '项目名称', '工程名称', '采购名称', '标段名称', '项目的名称', '设备名称', '申购主题',
  92. '申购单主题', '标的', '商品名称', '二级目录', '招标内容', '项目内容', '商品清单',
  93. '标的名称', '采购内容', '集成要求', '概况介绍', '品目分类', '招标范围', '采购范围',
  94. '项目采购分', '采购合同', '招标合同','产品名称','服务内容','采购品目名称','货物名称',
  95. '采购需求概况','项目概况','招标范围','采购条目名称','物资名称','物料名称','建设规模',
  96. '建设内容','采购项目概况','采购包名称','物料描述','商品信息','服务品目','标项名称',
  97. '规格描述','采购标的','服务名称','采购单名称','明细信息','申购主题','需求详情'
  98. ]
  99. # 创建正则表达式模式,匹配任意一个关键词
  100. pattern = r'(' + '|'.join(re.escape(kw) for kw in keywords) + r')'
  101. # 查找所有匹配位置
  102. matches = list(re.finditer(pattern, all_text))
  103. if not matches:
  104. return "" # 没有找到关键词
  105. all_match_text = []
  106. for _match in matches:
  107. start_pos = _match.end() # 关键词结束位置
  108. # 提取关键词之后的内容
  109. after_text = all_text[start_pos:]
  110. # 提取最多45个汉字
  111. chinese_chars = []
  112. count = 0
  113. for char in after_text:
  114. # 判断是否为汉字 (Unicode范围)
  115. if '\u4e00' <= char <= '\u9fff':
  116. count += 1
  117. if count > 45:
  118. break
  119. chinese_chars.append(char)
  120. chinese_chars = ''.join(chinese_chars).strip()
  121. if chinese_chars and chinese_chars not in all_match_text:
  122. all_match_text.append(chinese_chars)
  123. # 将字符列表组合成字符串
  124. main_content_text = ','.join(all_match_text).strip()
  125. return main_content_text
  126. def predict(self, doctitle,product,project_name,prem,all_text):
  127. doctitle = doctitle if doctitle else ""
  128. product = product if product else ""
  129. product = ",".join(set(product.split(','))) # 产品词去重
  130. project_name = project_name if project_name else ""
  131. tenderee = ""
  132. agency = ""
  133. sub_project_names = [] # 标段名称
  134. try:
  135. for k,v in prem[0]['prem'].items():
  136. # sub_project_names.append(k)
  137. sub_project_names.append(v.get("name",""))
  138. for link in v['roleList']:
  139. if link['role_name'] == 'tenderee' and tenderee == "":
  140. tenderee = link['role_text']
  141. if link['role_name'] == 'agency' and agency == "":
  142. agency = link['role_text']
  143. except Exception as e:
  144. # print('解析prem 获取招标人、代理人出错')
  145. pass
  146. sub_project_names = ";".join(sub_project_names)
  147. main_content_text = self.extract_core_text(all_text,tenderee,agency)
  148. # 核心字段:标题+产品词+项目名称+标段名称
  149. # main_text = ",".join([doctitle, product, project_name, sub_project_names])
  150. # 核心字段:标题+项目名称+产品词+正文定位词后45个字
  151. main_text = ",".join([doctitle, project_name, product, main_content_text])
  152. # 剔除 招标单位、代理机构名称
  153. if tenderee:
  154. doctitle = doctitle.replace(tenderee, " ")
  155. main_text = main_text.replace(tenderee, " ")
  156. if agency:
  157. doctitle = doctitle.replace(agency, " ")
  158. main_text = main_text.replace(agency, " ")
  159. doctitle_dict = dict()
  160. main_text_dict = dict()
  161. for item in self.keyword_list:
  162. _type = item[0]
  163. key_wrod = item[1]
  164. # 关键词排除词
  165. key_paichuci = item[2]
  166. key_paichuci_s = "|".join([re.escape(word) for word in key_paichuci.strip('、').split('、')])
  167. # 类型排除词
  168. type_paichuci = item[3]
  169. if type_paichuci:
  170. paichuci_split = type_paichuci.strip('、').split('、')
  171. if re.search("|".join([re.escape(word) for word in paichuci_split]), main_text):
  172. continue
  173. if doctitle:
  174. if key_wrod in doctitle:
  175. if not key_paichuci_s or (key_paichuci_s and not re.search(key_paichuci_s, doctitle)):
  176. key_wrod_count1 = doctitle.count(key_wrod)
  177. if _type not in doctitle_dict:
  178. # doctitle_dict[_type] = {'关键词': [], '排除词': type_paichuci}
  179. doctitle_dict[_type] = []
  180. doctitle_dict[_type].append((key_wrod, key_wrod_count1))
  181. if main_text:
  182. if key_wrod in main_text:
  183. if not key_paichuci_s or (key_paichuci_s and not re.search(key_paichuci_s, main_text)):
  184. key_wrod_count2 = main_text.count(key_wrod)
  185. if _type not in main_text_dict:
  186. # main_text_dict[_type] = {'关键词': [], '排除词': type_paichuci}
  187. main_text_dict[_type] = []
  188. main_text_dict[_type].append((key_wrod, key_wrod_count2))
  189. # 排序 doctitle
  190. for k, v in doctitle_dict.items():
  191. doctitle_dict[k].sort(key=lambda x: x[1], reverse=True)
  192. # 按匹配次数保留前10个标签
  193. if len(doctitle_dict) > 10:
  194. doctitle_labels = [(k, sum(w[1] for w in doctitle_dict[k])) for k in doctitle_dict]
  195. doctitle_labels.sort(key=lambda x: x[1], reverse=True)
  196. for item in doctitle_labels[10:]:
  197. doctitle_dict.pop(item[0])
  198. # main_text
  199. pop_list = []
  200. for k, v in main_text_dict.items():
  201. if sum([j[1] for j in main_text_dict[k]]) == 1:
  202. # 关键词匹配次数等于1的标签
  203. pop_list.append(k)
  204. main_text_dict[k].sort(key=lambda x: x[1], reverse=True)
  205. # 核心字段标签,若存在同一个标签的关键词匹配次数大于1,则只保留关键词匹配次数大于1的标签,关键词匹配次数等于1的标签不要
  206. if len(pop_list) < len(main_text_dict):
  207. for k in pop_list:
  208. main_text_dict.pop(k)
  209. # 按匹配次数保留前10个标签
  210. if len(main_text_dict) > 10:
  211. main_text_labels = [(k, sum(w[1] for w in main_text_dict[k])) for k in main_text_dict]
  212. main_text_labels.sort(key=lambda x: x[1], reverse=True)
  213. for item in main_text_labels[10:]:
  214. main_text_dict.pop(item[0])
  215. return {"标题":doctitle_dict,"核心字段":main_text_dict},main_content_text
  216. def predict_other(self,project_label,industry,doctitle,project_name,product,list_articles,main_content_text):
  217. # doctextcon 取正文内容
  218. doctextcon = list_articles[0].content.split('##attachment##')[0]
  219. info_type = industry.get('industry',{}).get("class_name","")
  220. doctitle = doctitle if doctitle else ""
  221. product = product if product else ""
  222. product = ",".join(set(product.split(','))) # 产品词去重
  223. project_name = project_name if project_name else ""
  224. main_content_text = main_content_text if main_content_text else ""
  225. # 空净通
  226. get_kongjing_label = False
  227. keywords_list = []
  228. for item in self.kongjing_keyword_list:
  229. key_wrod = item[0]
  230. key_wrod2 = item[1]
  231. search_type = item[2]
  232. info_type_list = item[3]
  233. info_type_list = info_type_list.strip('|').split("|") if info_type_list else []
  234. search_text = ""
  235. if search_type=='正文':
  236. search_text = ",".join([doctextcon,doctitle,project_name,product])
  237. elif search_type=='产品':
  238. search_text = ",".join([doctitle,project_name,product])
  239. if search_type=='行业':
  240. # ’行业’类型直接用info_type匹配关键词
  241. if info_type==key_wrod:
  242. # 匹配关键词记录
  243. keywords_list.append(key_wrod)
  244. get_kongjing_label = True
  245. break
  246. else:
  247. if key_wrod in search_text:
  248. if key_wrod2 and key_wrod2 not in search_text:
  249. continue
  250. if info_type_list and info_type not in info_type_list:
  251. continue
  252. # 匹配关键词记录
  253. if key_wrod2:
  254. keywords_list.append(key_wrod+'+'+key_wrod2)
  255. else:
  256. keywords_list.append(key_wrod)
  257. get_kongjing_label = True
  258. break
  259. if get_kongjing_label:
  260. project_label["核心字段"]["空净通"] = [[word,1] for word in keywords_list][:10]
  261. # ICT智算
  262. get_ICT_label = False
  263. keywords_list = []
  264. for item in self.ICT_smart_compute_keyword_list:
  265. key_wrod = item[0]
  266. key_wrod2 = item[1]
  267. search_type = item[2]
  268. info_type_list = item[3]
  269. info_type_list = info_type_list.strip('|').split("|") if info_type_list else []
  270. search_text = ""
  271. if search_type=='正文':
  272. search_text = ",".join([doctextcon,doctitle,project_name,product])
  273. elif search_type=='产品':
  274. search_text = ",".join([doctitle,project_name,product])
  275. elif search_type=='core_text':
  276. # core_text规则:标题+项目名称+产品词+正文定位词后45个字
  277. search_text = doctitle + ',' + project_name + "," + product + "," + main_content_text
  278. if search_type=='行业':
  279. # ’行业’类型直接用info_type匹配关键词
  280. if info_type==key_wrod:
  281. # 匹配关键词记录
  282. keywords_list.append(key_wrod)
  283. get_ICT_label = True
  284. break
  285. else:
  286. if key_wrod in search_text:
  287. if key_wrod2 and key_wrod2 not in search_text:
  288. continue
  289. if info_type_list and info_type not in info_type_list:
  290. continue
  291. # 匹配关键词记录
  292. if key_wrod2:
  293. keywords_list.append(key_wrod+'+'+key_wrod2)
  294. else:
  295. keywords_list.append(key_wrod)
  296. get_ICT_label = True
  297. break
  298. if info_type in ['计算机设备','监控设备','通信设备','信息系统集成和物联网技术服务','运行维护服务','信息处理和存储支持服务','互联网安全服务','互联网接入及相关服务','电信']:
  299. # 新增规则info_type符合范围直接判定
  300. get_ICT_label = True
  301. if get_ICT_label:
  302. project_label["核心字段"]["ICT智算"] = [[word,1] for word in keywords_list][:10]
  303. return project_label
  304. # 行业标签
  305. class IndustryLabel():
  306. def __init__(self):
  307. # self.keyword_list = self.get_label_keywords()
  308. pass
  309. def predict(self,doctitle,article,product,prem):
  310. doctitle = doctitle if doctitle else ""
  311. product = product if product else ""
  312. product = ",".join(set(product.split(','))) # 产品词去重
  313. all_text = article.content
  314. all_text = re.sub('\s+', ' ', all_text)
  315. tenderee = ""
  316. agency = ""
  317. try:
  318. for k,v in prem[0]['prem'].items():
  319. for link in v['roleList']:
  320. if link['role_name'] == 'tenderee' and tenderee == "":
  321. tenderee = link['role_text']
  322. if link['role_name'] == 'agency' and agency == "":
  323. agency = link['role_text']
  324. except Exception as e:
  325. # print('解析prem 获取招标人、代理人出错')
  326. pass
  327. # 剔除 招标单位、代理机构名称
  328. if tenderee:
  329. doctitle = doctitle.replace(tenderee, " ")
  330. all_text = all_text.replace(tenderee, " ")
  331. if agency:
  332. doctitle = doctitle.replace(agency, " ")
  333. all_text = all_text.replace(agency, " ")
  334. # category_1, category_2, category_3, matched_keywords, rule_id = product_classify_process(doctitle, all_text, product)
  335. category_1, category_2, category_3, matched_keywords, rule_id = classify_text(doctitle, all_text, product)
  336. # print(category_1, category_2, category_3, matched_keywords, rule_id)
  337. tenderee_label = process_rules(tenderee)
  338. new_tenderee_label = []
  339. for k in tenderee_label.keys():
  340. if '-' in k:
  341. new_tenderee_label.append({"first_level":k.split('-')[0],"second_level":k.split('-')[1]})
  342. # new_tenderee_label.append({"first_level":k.split('-')[0],"second_level":k.split('-')[1],"code":k.split('-')[2]})
  343. # print(new_tenderee_label)
  344. if category_2=="标题排除":
  345. category_1 = "其他"
  346. category_2 = ""
  347. rule_id = ""
  348. return {"first_level":category_1,"second_level":category_2},new_tenderee_label
  349. # def get_label_keywords(self):
  350. # import csv
  351. # path = os.path.dirname(__file__)+'/industry_label_keywords.csv'
  352. # with open(path, 'r',encoding='utf-8') as f:
  353. # reader = csv.reader(f)
  354. # key_word_list = []
  355. # for r in reader:
  356. # if r[0] == '一级标签':
  357. # continue
  358. # first_level = r[0]
  359. # second_level = str(r[1])
  360. # second_level = second_level.strip() if second_level and second_level != 'nan' else ""
  361. # key_word = str(r[2]).strip()
  362. # all_paichuci = str(r[3])
  363. # all_paichuci = all_paichuci.strip() if all_paichuci and all_paichuci != 'nan' else ""
  364. # title_paichuci = str(r[4])
  365. # title_paichuci = title_paichuci.strip() if title_paichuci and title_paichuci != 'nan' else ""
  366. # product_paichuci = str(r[5])
  367. # product_paichuci = product_paichuci.strip() if product_paichuci and product_paichuci != 'nan' else ""
  368. # key_word_list.append((first_level, second_level, key_word, all_paichuci,title_paichuci,product_paichuci))
  369. # return key_word_list
  370. #
  371. # def predict(self, doctitle,article,product,prem):
  372. #
  373. # doctitle = doctitle if doctitle else ""
  374. # product = product if product else ""
  375. # product = ",".join(set(product.split(','))) # 产品词去重
  376. # all_text = article.content
  377. # tenderee = ""
  378. # agency = ""
  379. # try:
  380. # for k,v in prem[0]['prem'].items():
  381. # for link in v['roleList']:
  382. # if link['role_name'] == 'tenderee' and tenderee == "":
  383. # tenderee = link['role_text']
  384. # if link['role_name'] == 'agency' and agency == "":
  385. # agency = link['role_text']
  386. # except Exception as e:
  387. # # print('解析prem 获取招标人、代理人出错')
  388. # pass
  389. # # 剔除 招标单位、代理机构名称
  390. # if tenderee:
  391. # doctitle = doctitle.replace(tenderee, " ")
  392. # all_text = all_text.replace(tenderee, " ")
  393. # if agency:
  394. # doctitle = doctitle.replace(agency, " ")
  395. # all_text = all_text.replace(agency, " ")
  396. #
  397. # label_list = []
  398. # for item in self.keyword_list:
  399. # first_level = item[0]
  400. # second_level = item[1]
  401. # key_word = item[2]
  402. # key_word = key_word.strip('、').split('、')
  403. # # 全文排除词
  404. # all_paichuci = item[3]
  405. # all_paichuci = "|".join([re.escape(word) for word in all_paichuci.strip('、').split('、')])
  406. # # 标题排除词
  407. # title_paichuci = item[4]
  408. # title_paichuci = "|".join([re.escape(word) for word in title_paichuci.strip('、').split('、')])
  409. # # 产品排除词
  410. # product_paichuci = item[5]
  411. # product_paichuci = "|".join([re.escape(word) for word in product_paichuci.strip('、').split('、')])
  412. #
  413. #
  414. # if doctitle and title_paichuci:
  415. # if re.search(title_paichuci,doctitle):
  416. # continue
  417. # if product and product_paichuci:
  418. # if re.search(product_paichuci,product):
  419. # continue
  420. # if all_text:
  421. # if all_paichuci:
  422. # if re.search(all_paichuci,all_text):
  423. # continue
  424. # get_label = False
  425. # for _keyword in key_word:
  426. # if '+' not in _keyword:
  427. # if _keyword in all_text:
  428. # get_label = True
  429. # break
  430. # else:
  431. # get_keyword = True
  432. # for _word in _keyword.split("+"):
  433. # if _word not in all_text:
  434. # get_keyword = False
  435. # break
  436. # if get_keyword:
  437. # get_label = True
  438. # break
  439. # if get_label:
  440. # label_list.append({"first_level":first_level,"second_level":second_level})
  441. #
  442. # return label_list
  443. # 产权分类二级标签
  444. class PropertyLabel():
  445. '''
  446. 产权分类二级标签
  447. 全部类别:
  448. 股权, 债权, 知识产权, 矿权, 房产, 土地, 交通运输工具, 闲置物资、设备、材料, 其他
  449. '''
  450. def __init__(self, ):
  451. car = "比亚迪|奇瑞|奥迪|宝马|菲尼迪|雷克萨斯|三菱|铃木|马自达|奔驰|劳斯莱斯|北京现代|" \
  452. "宾利|兰博基尼|布加迪|保时捷|斯柯达|雪佛兰|别克|凯迪拉克|庞蒂亚克|克尔维特|福特|林肯|克莱斯勒|道奇|JEEP品牌"
  453. self.keywords_dict = {
  454. "房产": "房产|住宅|公寓|商铺|车位|写字楼|办公楼|别墅|综合楼|在建工程|厂房|车库|车房|房转让|房屋|商品房|商业用房|"
  455. "宅基地|[\u4e00-\u9fa5]{,2}用房|店面|商业房|门[面市]房|仓库|铺位|地下室|\d号?(房|室|门市|门面|商?铺|单元|户)|不动产|"
  456. "自建房|铺面|商务楼|商住楼|阁楼|(杂物|储物|储藏)(房|间|室)|套房|[\da-zA-Z](栋|棟|幢|层|座|号?楼|单元)\d{1,4}(号|房|室|商?铺|户)|"
  457. "[\da-zA-Z](栋|棟|幢|层|座|号?楼|单元)\d{2,}|门面+转让|楼+变卖|房地产",
  458. "交通运输工具": "车辆|轿车|汽车(?!用品|库|位|衡)|公车|客车|货车|面包车|SUV|新能源车|二手车|车辆|商用车|机动车|观光车|巴车|"
  459. "船舶|四驱" + "|" + car,
  460. "股权": "\d.?股|股权(?!交易中心)|\d%(比例)?.?股|\d万.?股|\d.?元/股|增资(?!源)|扩股|股(转让|出售)|百分之[一二三四五六七八九十]{1,3}股",
  461. "债权": "债权|债权转让|债权人|债务人|原债权人|新债权人|金融资产",
  462. "土地": "住宅用地|商业用地|工业用地|国有[\u4e00-\u9fa5]{,3}[土用]地|集体土地|划拨|流转|地块编号|"
  463. "土地使用权证|土地经营权|土地证|土地[发承]包|[\u4e00-\u9fa5]{,2}用地|土地\d{1,3}(亩|公?顷)|\d{1,3}(亩|公?顷)(使用|经营)权|"
  464. "承包土地|(地块|土地)承包|水面经营权|[鱼水]塘|鱼池|(水面|旱田)[\u4e00-\u9fa5]{,3}[发承]包|水面资源|(水面|水田)[\u4e00-\u9fa5]{,3}权|"
  465. "四荒|林地|林场|林木所有权|采伐权|水利设施所有权|水利设施使用权|海域|滩涂|林业产权|旱田|水田|机动田|机动地|耕地|荒地|农田|"
  466. "苗圃地|塘口",
  467. "矿权": "矿权|矿业权|采矿许可|探矿权|采矿权|开采权|矿产资源处置|矿[\u4e00-\u9fa5]{1,3}开[发采]",
  468. "知识产权": "知识产权(?!局)|商标|专利|著作权|版权|商业秘密|科研成果",
  469. "闲置物资、设备、材料": "(废旧|报废|废|闲置|二手|淘汰)(物资|资产|机械|设备|仪器|汽车|车|钢铁|钢材|钢|金属|塑料|材料|导管|漆|渣|有色|品|[\u4e00-\u9fa5]{,2}车|偶头)|"
  470. "(金属|机械|设备|仪器|汽车|钢铁|钢材|钢|塑料|有色|)废料|废液|废旧|报废|边角料|残次品|(热轧|冷轧|酸洗|镀铝|热镀|镀锌|镀镁)|"
  471. "机[器械]设备|医疗设备|生产设备|办公设备|仪器|仪表|设备出租|设备租赁|拖拉机|收割机|插秧机|挖机|车床|挖掘机|电机|"
  472. "戒指|弃渣|电解质块|茶杯|装置|花瓶|女表|手表|男表|硫磺|物资|书画|茶叶|油茶|红茶|[茗名]茶|白酒|红酒|酒水|酒品|名酒|毛石|[石金木铁矿铜锌铝钢]料|"
  473. "零部件",
  474. "经营权": "经营权",
  475. "租赁": "房+租|市场+续约|资产+出租|租赁|续租|招租|出租|租金|房租"
  476. }
  477. self.cqjy_keywords = self.get_cqjy_keywords()
  478. self.score_idx = ["股权", "债权", "知识产权", "矿权", "房产", "土地", "交通运输工具", "闲置物资、设备、材料"]
  479. def get_cqjy_keywords(self):
  480. import csv
  481. path = INTERFACE_DIR+'/property_label_products.csv'
  482. with open(path, 'r',encoding='utf-8') as f:
  483. reader = csv.reader(f)
  484. key_word_list = []
  485. for r in reader:
  486. if r[0] == 'product':
  487. continue
  488. key_wrod = r[0]
  489. _type = r[1]
  490. key_word_list.append((_type, key_wrod))
  491. return key_word_list
  492. def get_type(self, text):
  493. keyword_list = []
  494. for key, value in self.keywords_dict.items():
  495. keyword = "|".join([i for i in value.split("|") if '+' not in i])
  496. keyword2 = [i for i in value.split("|") if '+' in i]
  497. if re.search(keyword, text):
  498. re1 = [i for i in re.finditer(keyword, text)][-1]
  499. keyword_list.append((key, re1.start()))
  500. else:
  501. # 组合词 查询
  502. for k in keyword2:
  503. k1, k2 = k.split('+')
  504. if re.search(k1, text) and re.search(k2, text):
  505. keyword_list.append((key, re.search(k2, text).start()))
  506. break
  507. return keyword_list
  508. def get_type2(self, text, cqjy_type_list):
  509. have_type = [i[0] for i in cqjy_type_list]
  510. for item in self.cqjy_keywords:
  511. _type = item[0]
  512. key_wrod = item[1]
  513. if _type not in have_type:
  514. if '+' in key_wrod:
  515. k1, k2 = key_wrod.split('+')
  516. if re.search(k1, text) and re.search(k2, text):
  517. cqjy_type_list.append((_type, re.search(k2, text).start()))
  518. have_type.append(_type)
  519. else:
  520. if key_wrod in text:
  521. cqjy_type_list.append((_type, text.index(key_wrod)))
  522. have_type.append(_type)
  523. return cqjy_type_list
  524. def predict(self, doctitle,product,project_name,prem,channel_dic):
  525. docchannel = channel_dic['docchannel']['doctype']
  526. # print('docchannel',docchannel)
  527. if docchannel not in ['土地矿产', '拍卖出让', '产权交易']:
  528. return ""
  529. doctitle = doctitle if doctitle else ""
  530. product = product if product else ""
  531. product = ",".join(set(product.split(','))) # 产品词去重
  532. project_name = project_name if project_name else ""
  533. tenderee = ""
  534. agency = ""
  535. try:
  536. for k,v in prem[0]['prem'].items():
  537. for link in v['roleList']:
  538. if link['role_name'] == 'tenderee' and tenderee == "":
  539. tenderee = link['role_text']
  540. if link['role_name'] == 'agency' and agency == "":
  541. agency = link['role_text']
  542. except Exception as e:
  543. # print('解析prem 获取招标人、代理人出错')
  544. pass
  545. cqjy_type = []
  546. idx = 0
  547. for text in [doctitle, project_name, product]:
  548. if tenderee:
  549. text = text.replace(tenderee, "")
  550. if agency:
  551. text = text.replace(agency, "")
  552. cqjy_type = self.get_type(text)
  553. if not cqjy_type:
  554. cqjy_type = self.get_type2(text, cqjy_type)
  555. idx += 1
  556. if idx == 2: # project_name
  557. if len(re.split("[,、]", text)) > 9:
  558. cqjy_type = []
  559. if idx == 3: # product
  560. if len(text.split(",")) > 15:
  561. cqjy_type = []
  562. if cqjy_type:
  563. break
  564. cqjy_type2 = [i[0] for i in cqjy_type]
  565. if cqjy_type:
  566. # 类别优先级调整
  567. if "租赁" in cqjy_type2:
  568. cqjy_type2 = ['租赁']
  569. elif "经营权" in cqjy_type2:
  570. cqjy_type2 = ['经营权']
  571. elif "股权" in cqjy_type2 or "债权" in cqjy_type2 or "知识产权" in cqjy_type2:
  572. cqjy_type.sort(key=lambda x: self.score_idx.index(x[0]))
  573. cqjy_type = cqjy_type[0]
  574. cqjy_type2 = [cqjy_type[0]]
  575. elif len(cqjy_type2) == 2 and "房产" in cqjy_type2 and "土地" in cqjy_type2:
  576. cqjy_type2 = ['房产']
  577. else:
  578. # 权重排序,取第一位
  579. if idx in [1, 2]: # doctitle, project_name
  580. cqjy_type.sort(key=lambda x: x[1], reverse=True)
  581. cqjy_type = cqjy_type[0]
  582. cqjy_type2 = [cqjy_type[0]]
  583. else:
  584. cqjy_type.sort(key=lambda x: self.score_idx.index(x[0]))
  585. cqjy_type = cqjy_type[0]
  586. cqjy_type2 = [cqjy_type[0]]
  587. cqjy_type2 = ",".join(cqjy_type2)
  588. if not cqjy_type2:
  589. cqjy_type2 = '其他'
  590. return cqjy_type2
  591. class BiddingScore():
  592. def __init__(self):
  593. self.head_rule_dic = {
  594. "tenderer": "((候选|入围|入选|投标|应答|响应)(供应商库)?的?(人|人?单位|机构|供应商|供货商|服务商|投标人|(中标)?公司|(中标)?企业|银行)|(通过)?名单|中标候选人)(名称|名单|全称|\d)?$|^供应商(名称|信息)?$|投标个人/单位", #补充 368295593 投标个人/单位 提取
  595. "score_price": "(价格|报价|单价|总价|经济)(部分|\w{,2})?([得评]分|评审)",
  596. "score_technical": "技术(部分|\w{,2})?标?([得评]分|评审)",
  597. "score_commercial": "商务(部分|\w{,2})?标?([得评]分|评审)",
  598. "score_integrity": "诚信(部分|\w{,2})?([得评]分|评审)",
  599. "score_comprehensive": "(综合(标|评估)?|总|最终)得?分$",
  600. "ranking": "(得分)?排名",
  601. "qualification_review": "资格性审查|是否通过资格",
  602. "compliance_review": "符合性审查|是否通过符合"
  603. }
  604. self.tb = TableTag2List()
  605. def get_table_info(self, df, nlp_enterprise):
  606. def get_header_index(datas):
  607. '''
  608. 根据表格表头判断结果0/1 得到哪些行和列是表头
  609. :param datas: 表格内容表头判断结果数据[[1,1,1,1],[0,0,0,0]]
  610. :return: 表头所在的行和列序号
  611. '''
  612. header_row = []
  613. header_col = []
  614. df_h = pd.DataFrame(datas) # 表头判断数据 , columns=columns
  615. for i in df_h.index:
  616. line = df_h.loc[i].values
  617. if sum(line) == len(line):
  618. header_row.append((i, sum(line) / len(line)))
  619. elif sum(line) / len(line) > 0.8:
  620. header_row.append((i, sum(line) / len(line)))
  621. elif len(line) > 3 and len(re.findall('11', ''.join([str(it) for it in line]))) > len(
  622. re.findall('10', ''.join([str(it) for it in line]))):
  623. header_row.append((i, sum(line) / len(line)))
  624. for i in df_h.columns:
  625. col = df_h[i].values
  626. if sum(col) == len(col):
  627. header_col.append((i, sum(col) / len(col)))
  628. elif sum(col) / len(col) > 0.8:
  629. header_col.append((i, sum(col) / len(col)))
  630. elif len(col) > 3 and len(re.findall('11', ''.join([str(it) for it in line]))) > len(
  631. re.findall('10', ''.join([str(it) for it in line]))):
  632. header_col.append((i, sum(col) / len(col)))
  633. return header_row, header_col
  634. def get_header(l, head_rule_dic):
  635. header_dic = {}
  636. for i in range(len(l)):
  637. text = l[i]
  638. num = 0
  639. tmp_dic = {}
  640. for k, v in head_rule_dic.items():
  641. # print('k : ', k)
  642. if re.search(v, text):
  643. tmp_dic[k] = i
  644. num += 1
  645. # if num > 1:
  646. # if tmp_dic.keys() == set(['qualification_review', 'compliance_review']):
  647. # for k, v in tmp_dic.items():
  648. # if k not in header_dic:
  649. # header_dic[k] = v
  650. # elif tmp_dic:
  651. for k, v in tmp_dic.items():
  652. if k not in header_dic:
  653. header_dic[k] = v
  654. return header_dic
  655. def get_score(text):
  656. text = text.strip()
  657. if re.search('^\d{1,2}(\.\d{2})$', text):
  658. return text
  659. elif re.search('^\d{1,2}(\.\d{2})?[\d,,;\.]*$', text):
  660. return text
  661. return ''
  662. result_l = []
  663. datas = []
  664. for i in df.index:
  665. line = get_header_line(df.loc[i].values)
  666. datas.append(line)
  667. header_row, header_col = get_header_index(datas)
  668. if len(header_col) == 1 and header_col[0][0] > 1: # 列表头不可能在第1列后面开始
  669. header_col = []
  670. if len(header_row) >= 1 and len(header_col) == 0: # 有行表头无列表头
  671. i = 0
  672. while i < len(header_row):
  673. idx, ratio = header_row[i]
  674. if idx + 1 >= len(df):
  675. break
  676. header_dic = get_header(df.loc[idx].values, self.head_rule_dic)
  677. i += 1
  678. range_from = idx + 1
  679. range_to = len(df)
  680. if i < len(header_row):
  681. next_header = i
  682. for j in range(i, len(header_row)):
  683. idx2, ratio2 = header_row[j]
  684. if idx2 - idx == 1:
  685. header_dic2 = get_header(df.loc[idx2].values, self.head_rule_dic)
  686. if set(df.loc[idx].values) & set(df.loc[idx2].values) != set():
  687. header_dic.update(header_dic2)
  688. else:
  689. header_dic = header_dic2
  690. range_from = idx2 + 1
  691. range_to = len(df)
  692. next_header = j + 1
  693. idx = idx2
  694. else:
  695. range_from = idx + 1
  696. range_to = idx2
  697. next_header = j
  698. break
  699. i = next_header
  700. if len(header_dic) >= 2 and 'tenderer' in header_dic:
  701. for index in range(range_from, range_to):
  702. tmp_dic = {}
  703. for k, v in header_dic.items():
  704. if k.startswith('score'):
  705. content = get_score(df.loc[index, v])
  706. elif k == 'tenderer':
  707. content = get_role(df.loc[index, v], nlp_enterprise)
  708. elif k == 'ranking':
  709. content = df.loc[index, v] if re.search('^第?[\d一二三四五六七八九十]+名?$',df.loc[index, v]) else ''
  710. else:
  711. content = df.loc[index, v]
  712. if content != '':
  713. tmp_dic[k] = content
  714. if len(tmp_dic) > 1 and 'tenderer' in tmp_dic and tmp_dic not in result_l:
  715. result_l.append(tmp_dic)
  716. elif len(header_row) == 0 and len(header_col) >= 1:
  717. i = 0
  718. while i < len(header_col):
  719. idx, ratio = header_col[i]
  720. if idx + 1 >= len(df.columns):
  721. break
  722. header_dic = get_header(df[idx].values, self.head_rule_dic)
  723. i += 1
  724. range_from = idx + 1
  725. range_to = len(df.columns)
  726. if i < len(header_col):
  727. next_header = i
  728. for j in range(i, len(header_col)):
  729. idx2, ratio2 = header_col[j]
  730. if idx2 - idx == 1:
  731. header_dic2 = get_header(df[idx2].values, self.head_rule_dic)
  732. if set(df[idx].values) & set(df[idx2].values) != set():
  733. header_dic.update(header_dic2)
  734. else:
  735. header_dic = header_dic2
  736. range_from = idx2 + 1
  737. range_to = len(df.columns)
  738. next_header = j + 1
  739. idx = idx2
  740. else:
  741. range_from = idx + 1
  742. range_to = idx2
  743. next_header = j
  744. break
  745. i = next_header
  746. if len(header_dic.keys()&set(['tenderer','score_technical', 'score_commercial', 'score_price', 'score_comprehensive'])) >= 2 and 'tenderer' in header_dic:
  747. for index in range(range_from, range_to):
  748. tmp_dic = {}
  749. for k, v in header_dic.items():
  750. if k.startswith('score'):
  751. content = get_score(df.loc[v, index])
  752. elif k == 'tenderer':
  753. content = get_role(df.loc[v, index], nlp_enterprise)
  754. elif k == 'ranking':
  755. content = df.loc[v, index] if re.search('^第?[\d一二三四五六七八九十]+名?$', df.loc[v, index]) else ''
  756. else:
  757. content = df.loc[v, index]
  758. if content != '':
  759. tmp_dic[k] = content
  760. if len(tmp_dic) > 2 and 'tenderer' in tmp_dic and tmp_dic not in result_l:
  761. result_l.append(tmp_dic)
  762. elif len(header_row) == 1 and len(header_col) == 1:
  763. pass
  764. return result_l
  765. def predict(self, html, nlp_enterprise=[]):
  766. html = re.sub("<html>|</html>|<body>|</body>", "", html)
  767. html = re.sub("##attachment##", "", html)
  768. soup = BeautifulSoup(html, 'lxml')
  769. richText = soup.find(name='div', attrs={'class': 'richTextFetch'})
  770. self.nlp_enterprise = nlp_enterprise
  771. if richText:
  772. richText = richText.extract() # 过滤掉附件
  773. tables = soup.find_all('table')
  774. if len(tables) == 0 and richText:
  775. tables = richText.find_all('table')
  776. tables.reverse()
  777. rs_dic = {}
  778. for table in tables:
  779. trs = self.tb.table2list(table)
  780. if len(trs)>1 and len(trs[0])>1 and len(set([len(tr) for tr in trs])) == 1:
  781. df = pd.DataFrame(trs)
  782. rs_l = self.get_table_info(df, nlp_enterprise)
  783. for d in rs_l:
  784. if d['tenderer'] not in rs_dic:
  785. rs_dic[d['tenderer']] = d
  786. elif len(d) > len(rs_dic[d['tenderer']]):
  787. rs_dic[d['tenderer']] = d
  788. table.extract()
  789. return list(rs_dic.values())