general_data.py 7.0 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154
  1. # DEPRECATED(Phase1): 本文件中的 PostgreSQL 硬编码连接(host=192.168.*,
  2. # password=postgres 等)将在后续 training/ Phase 迁移到 BiddingKG.dl.infra.db。
  3. # 迁移完成前可临时使用:from BiddingKG.dl.infra.db import get_connection;
  4. # conn = get_connection("<dbname>")
  5. # 详见 ARCHITECTURE.md 第 11 章 Phase 1 与 REFACTOR_LOG.md。
  6. import pandas as pd
  7. import psycopg2
  8. import pickle
  9. import re
  10. def get_data():
  11. '''
  12. @summary: 取出待标注的数据到excel中
  13. '''
  14. conn = psycopg2.connect(dbname='BidiPro', user='postgres',password='postgres',host='192.168.2.101')
  15. cursor = conn.cursor()
  16. #sql = '''SELECT e.doc_id,e.entity_id,e.sentence_index,e.entity_text,e.entity_type,e.begin_index,e.end_index,s.tokens from entity_mention e,sentences s
  17. #WHERE s.doc_id=e.doc_id AND s.sentence_index=e.sentence_index AND entity_type in ('person') ORDER BY doc_id,sentence_index,begin_index LIMIT 20000
  18. #'''
  19. sql = '''SELECT e.doc_id,e.entity_id,e.sentence_index,e.entity_text,e.entity_type,e.begin_index,e.end_index,s.tokens from entity_mention e,sentences s
  20. WHERE s.doc_id=e.doc_id AND s.sentence_index=e.sentence_index AND entity_type in ('person')
  21. and e.doc_id in (select id from articles_processed order by id desc limit 4000) ORDER BY doc_id,sentence_index,begin_index limit 20000
  22. '''
  23. cursor.execute(sql)
  24. rows = cursor.fetchmany(5000)
  25. new_df = pd.DataFrame()
  26. i = 0
  27. while(rows):
  28. df = pd.DataFrame(rows, columns=['doc_id','entity_id','sentence_index','entity_text','entity_type','begin_index','end_index','tokens'])
  29. i += 1
  30. new_df = pd.concat([new_df, df],ignore_index=True)
  31. #df.to_excel('data/person_'+str(i)+'.xls', encoding='utf-8',index=False)
  32. rows = cursor.fetchmany(5000)
  33. with open('data/person.pk', 'wb') as f:
  34. pickle.dump(new_df, f)
  35. #new_df.to_excel('data/person_total.xls', encoding='utf-8', index=False)
  36. #print(rows)
  37. cursor.close()
  38. conn.close()
  39. def label_data():
  40. '''
  41. @summary: 先通过规则预标注
  42. '''
  43. file2 = 'data/person_label.xls'
  44. with open('data/person.pk', 'rb') as f:
  45. data = pickle.load(f)
  46. # 分类:未知0 招标1 代理2 中标3 监督4 施工员5 联系人6
  47. zhaobiao = re.compile('采购中心|采购单位|采购人|采购经办人|招标单位|建设单位|招标人|项目单位|比选人|发包人|项目业主')
  48. daili = re.compile('代理机构|招标代理|采购代理|采购机构|招标代理机构|招标代理人')
  49. # zhongbiao = re.compile('供应商|法人代表|法定代表|中标人|中标单位|中标候选人|第[一|二|三|1|2|3]名|中标项目|项目负责人|项目经理')
  50. jiandu = re.compile('评标|评审|审批|审查|评委|监标|专家|小组|成员|名单|监督|监管|监察|监审|主管|受理|处室|反映|异议|质疑|(\d{2}\.\d{2})[^\d]')
  51. shigong = re.compile('甲方代表|管理人员|管理机构人员|施工员|安全员|质检员|质量员|材料员|预算员|建造师|造价员|监理员|监理人员|项目总监')
  52. lianxi = re.compile('经办人|联系人|联系方式|联系电话|法人代表|法定代表|中标供应商|中标人|中标单位|中标候选人|第[一|二|三|1|2|3]名|中标项目|项目负责人|项目经理')
  53. pattern_pos = re.compile('联系方式|联系人|项目负责人|项目经理|法人|法定代表|级别及证书|采购人|第一|第二|第三|第1|第2|第3')
  54. #pattern = re.compile('采购代理|采购机构|采购人|代理机构|项目负责人|联系人|技术负责人|第一|第二|第三|中标人|中标供应商|中标机构|中标候选人|招标|代理|资质|法人代表')
  55. pattern_neg = re.compile('监管|监察|监督|主管|受理|处室|异议|反映|评委|评审|评标|监标|委员会|磋商|专家|小组|人员类别|管理人员|人员配备|成员|名单')
  56. count = 0
  57. span = 10
  58. tokens = data['tokens']
  59. ben = data['begin_index']
  60. end = data['end_index']
  61. ent_id = data['entity_id']
  62. ent = data['entity_text']
  63. ent_type = data['entity_type']
  64. sen_index = data['sentence_index']
  65. pre_ent = []
  66. cur_ent = []
  67. label = [] # 标签列表
  68. ent_idl = []
  69. shiti = []
  70. s_list = []
  71. b_list = []
  72. for i in range(len(tokens)):
  73. if ent_type[i] == 'person':
  74. begin1 = ben[i] - span if ben[i] > span else 0
  75. end1 = end[i] + span if end[i] + span < len(tokens[i]) else len(tokens[i])
  76. pre_ent.append(tokens[i][begin1:ben[i]])
  77. cur_ent.append(tokens[i][end[i]:end1])
  78. ent_idl.append(ent_id[i])
  79. shiti.append(ent[i])
  80. s_list.append(sen_index[i])
  81. b_list.append(ben[i])
  82. str_tok = ''.join(tokens[i][begin1:ben[i]])
  83. str_tok = re.sub(',|\s','',str_tok)
  84. cur_tok = ''.join(tokens[i][begin1:ben[i]])
  85. cur_tok = re.sub(',|\s','',cur_tok)
  86. if re.findall(jiandu, str_tok):
  87. flag = 0
  88. elif re.findall(zhaobiao, str_tok):
  89. flag = 1
  90. elif re.findall(daili, str_tok):
  91. flag = 2
  92. # elif re.findall(zhongbiao, str_tok):
  93. # flag = 3
  94. elif re.findall(shigong, str_tok):
  95. flag = 0
  96. elif re.findall(lianxi, str_tok):
  97. flag = 3
  98. else:
  99. flag = 0
  100. count += 1
  101. label.append(flag)
  102. else:
  103. pass
  104. new_data = {'pre_ent':pre_ent, 'label':label,'cur_ent':cur_ent, 'entity_id':ent_idl, 'shiti':shiti, 'sentence_index':s_list, 'begin_index':b_list}
  105. data_label = pd.DataFrame(new_data)
  106. data_label.to_excel(file2, encoding='utf-8', index=False, columns=['entity_id','sentence_index','begin_index','pre_ent','label','cur_ent','shiti'])
  107. with open('data/person_label.pk', 'wb') as f:
  108. pickle.dump(data_label, f)
  109. def post_data():
  110. '''
  111. @summary: 将标注好的数据推送到数据库
  112. '''
  113. conn = psycopg2.connect(dbname='BidiPro', user='postgres',password='postgres',host='192.168.2.101')
  114. cursor = conn.cursor()
  115. table = 'person_label'
  116. cursor.execute(" select to_regclass('"+table+"') is null ")
  117. notExists = cursor.fetchall()[0][0]
  118. if notExists:
  119. cursor.execute(" create table "+table+" (entity_id text,label int)")
  120. else:
  121. cursor.execute(" delete from "+table)
  122. df3 = pd.read_excel('data/person_label.xls', header=0)
  123. df3.head(3)
  124. entity_id = df3['entity_id']
  125. label = df3['label']
  126. for i in range(len(entity_id)):
  127. sql = " insert into "+table+"(entity_id,label) values('"+str(df3['entity_id'][i])+"',"+str(int(label[i]))+")"
  128. #print(sql)
  129. cursor.execute(sql)
  130. conn.commit()
  131. cursor.close()
  132. conn.close()
  133. if __name__ == '__main__':
  134. #get_data()
  135. label_data()
  136. post_data()