# -*- coding: utf-8 -*- '''``common/Utils.py`` — Phase 3 拆分后的兼容层。 按 ARCHITECTURE.md Phase 3 拆分建议,原 ``common/Utils.py`` 中的函数和全局 变量已按职责迁移到以下 7 个目标模块: | 目标文件 | 迁出内容 | 类型 | |---|---|---| | ``common/json_encoder.py`` | ``MyEncoder`` | RULE FRAMEWORK | | ``model_runtime/viterbi.py`` | ``viterbi_decode`` | CORE | | ``model_runtime/vocab.py`` | ``vocab_word``、char/id 映射等 | CORE | | ``model_runtime/embed.py`` | w2v 加载、``embedding`` 等 | CORE | | ``services/external_inference/paieas.py`` | PAI-EAS 客户端 | INFRA | | ``common/context_utils.py`` | 上下文窗口、金额标准化等纯函数 | CORE | | ``common/logging.py`` | ``log`` / ``debug`` / ``logger`` | INFRA | 本文件通过 re-export 保持全部旧 import 路径可用: ``from BiddingKG.dl.common.Utils import viterbi_decode`` 仍然有效 ``from BiddingKG.dl.common.Utils import log`` 仍然有效 ``from BiddingKG.dl.common.Utils import *`` 仍然有效 未迁移的函数(时间处理、keras metrics、HTML 工具、pickle save/load、 encodeInput 等)仍保留在本文件中,后续 Phase 再拆。 ''' from __future__ import absolute_import import os import re import time import pickle import sys import traceback import numpy as np from keras import backend as K from lxml import etree from datetime import date import calendar # ============================================================ # Re-exports from model_runtime # ============================================================ from BiddingKG.dl.model_runtime.viterbi import viterbi_decode from BiddingKG.dl.model_runtime.embed import ( model_w2v, lock_model_w2v, model_word, lock_model_word, model_word_file, Lazy_load, getLazyLoad, getw2vfilepath, getFileFromSysPath, getModel_w2v, getModel_word, embedding, embedding_word, embedding_word_forward, formEncoding, ) from BiddingKG.dl.model_runtime.vocab import ( vocab_word, vocab_words, file_vocab_word, file_vocab_words, fool_char_to_id, getIndexOfWord, getIndexOfWords, getIndexOfWord_fool, getVocabAndMatrix, changeIndexFromWordToWords, ) # ============================================================ # Re-exports from services # ============================================================ from BiddingKG.dl.services.external_inference.paieas import ( USE_PAI_EAS, API_URL, USE_API, tf_predict_pb2, selffool_authorization, selffool_url, selffool_seg_authorization, selffool_seg_url, codename_authorization, codename_url, form_item_authorization, form_item_url, person_authorization, person_url, role_authorization, role_url, money_authorization, money_url, codeclasses_authorization, codeclasses_url, limitRun, get_values, vpc_requests, ) # ============================================================ # Re-exports from common submodules # ============================================================ from BiddingKG.dl.common.json_encoder import MyEncoder from BiddingKG.dl.common.logging import log, debug, logger from BiddingKG.dl.common.context_utils import ( # 上下文窗口 spanWindow, get_context, findAllIndex, find_index, # 金额标准化 getUnifyMoney, getDigitsDic, getMultipleFactor, partMoney, uniform_num, uniform_package_name, money_process, get_money_entity, # 其他纯函数 combine, fitDataByRule, clean_company, cut_repeat_name, is_all_winner, is_deposit_project, find_package, # 模块级正则 package_number_pattern, filter_package_pattern, ) # ============================================================ # 以下为未迁移函数,仍保留在本文件中 # ============================================================ def getCurrent_date(format="%Y-%m-%d %H:%M:%S"): _time = time.strftime(format, time.localtime()) return _time def encodeInput(data, word_len, word_flag=True, userFool=False): result = [] out_index = 0 for item in data: if out_index in [0]: list_word = item[-word_len:] else: list_word = item[:word_len] temp = [] if word_flag: for word in list_word: if userFool: temp.append(getIndexOfWord_fool(word)) else: temp.append(getIndexOfWord(word)) list_append = [] temp_len = len(temp) while temp_len < word_len: if userFool: list_append.append(0) else: list_append.append(getIndexOfWord("")) temp_len += 1 if out_index in [0]: temp = list_append + temp else: temp = temp + list_append else: for words in list_word: temp.append(getIndexOfWords(words)) list_append = [] temp_len = len(temp) while temp_len < word_len: list_append.append(getIndexOfWords("")) temp_len += 1 if out_index in [0, 1]: temp = list_append + temp else: temp = temp + list_append result.append(temp) out_index += 1 return result def encodeInput_form(input, MAX_LEN=30): x = np.zeros([MAX_LEN]) for i in range(len(input)): if i >= MAX_LEN: break x[i] = getIndexOfWord(input[i]) return x def save(object_to_save, path): ''' 保存对象 @Arugs: object_to_save: 需要保存的对象 @Return: 保存的路径 ''' with open(path, 'wb') as f: pickle.dump(object_to_save, f) def load(path): ''' 读取对象 @Arugs: path: 读取的路径 @Return: 读取的对象 ''' with open(path, 'rb') as f: object1 = pickle.load(f) return object1 # 时间合法性判断 def isValidDate(year, month, day): try: date(year, month, day) except: return False else: return True time_format_pattern = re.compile("((?P20\d{2}|\d{2}|二[零〇0][零〇一二三四五六七八九0]{2})\s*[-/年.]\s*(?P\d{1,2}|[一二三四五六七八九十]{1,3})\s*[-/月.]?\s*(?P\d{1,2}|[一二三四五六七八九十]{1,3})?)") from BiddingKG.dl.ratio.re_ratio import getUnifyNum def get_maxday(year, month): # calendar.monthrange(year, month) 返回一个元组,其中第一个元素是那个月第一天的星期几(0-6代表周一到周日), # 第二个元素是那个月的天数。 _, last_day = calendar.monthrange(year, month) return last_day def timeFormat(_time, default_first_day=True): ''' 日期格式化:年-月-日 :param _time: :param default_first_day: True取当月第一天,否则取最后一天 :return: ''' current_year = time.strftime("%Y", time.localtime()) all_match = re.finditer(time_format_pattern, _time) for _match in all_match: if len(_match.group()) > 0: legal = True year = "" month = "" day = "" for k, v in _match.groupdict().items(): if k == "year": year = v if k == "month": month = v if k == "day": day = v if year != "": if re.search("^\d+$", year): if len(year) == 2: year = "20" + year if int(year) - int(current_year) > 10: legal = False else: _year = "" for word in year: if word == '0': _year += word else: _year += str(getDigitsDic(word)) year = _year else: legal = False if month != "": if re.search("^\d+$", month): if int(month) > 12: legal = False else: month = int(getUnifyNum(month)) if month >= 1 and month <= 12: month = str(month) else: legal = False else: legal = False if day == None: day = "01" if (default_first_day or legal == False) else str(get_maxday(int(year), int(month))) if day != "": if re.search("^\d+$", day): if int(day) > 31: legal = False else: day = int(getUnifyNum(day)) if day >= 1 and day <= 31: day = str(day) else: legal = False else: legal = False # print(year,month,day) if not isValidDate(int(year), int(month), int(day)): legal = False if legal: return "%s-%s-%s" % (year, month.rjust(2, "0"), day.rjust(2, "0")) return "" def del_tabel_achievement(soup): if re.search('中标|成交|入围|结果|评标|开标|候选人', soup.text[:800]) == None or re.search('业绩|类似项目', soup.text) == None: return None p1 = '(中标|成交)(单位|候选人)的?(企业|项目|项目负责人|\w{,5})?业绩|类似(项目)?业绩|\w{,10}业绩$|业绩(公示|情况|荣誉)|近年完成的项目类似项目情况表|类似项目' '''删除前面标签 命中业绩规则;当前标签为表格且公布业绩相关信息的去除''' for tag in soup.find_all('table'): pre_text = "" if tag.findPreviousSibling() != None: pre_text = tag.findPreviousSibling().text.strip() if pre_text == "" and tag.findPreviousSibling().findPreviousSibling() != None: # 修复表格前一标签没内容,再前一个才有内容情况 pre_text = tag.findPreviousSibling().findPreviousSibling().text.strip() tr_text = tag.find('tr').text.strip() if tag.find('tr') != None else "" if len(pre_text) < 20 and re.search('^[((]?[\d一二三四五六七八九十]+[)、)]近年完成的(项目)?类似项目情况表|类似项目历史成交信息|类似项目的?采购预算', pre_text): # 删除 650469910 评标公告 历史业绩 547305313 del_tag = tag.extract() # print('删除表格业绩内容', del_tag.text) # print(re.search(p1, pre_text),pre_text, len(pre_text), re.findall('序号|中标候选人名称|项目名称|工程名称|合同金额|建设单位|业主', tr_text)) elif re.search(p1, pre_text) and len(pre_text) < 20 and tag.find('tr') != None and len(tr_text) < 100: _count = 0 for td in tag.find('tr').find_all('td'): td_text = td.text.strip() if len(td_text) > 25: break if len(td_text) < 25 and re.search('中标候选人|第[一二三四五1-5]候选人|(项目|业绩|工程)名称|\w{,10}业绩$|合同(金额|价格)|建设单位|采购单位|业主|甲方|发包人', td_text): _count += 1 if _count >= 2: pre_tag = tag.findPreviousSibling().extract() del_tag = tag.extract() # print('删除表格业绩内容', pre_tag.text + del_tag.text) break elif re.search('业绩名称', tr_text) and re.search('建设单位|采购单位|业主', tr_text) and len(tr_text) < 100: del_tag = tag.extract() # print('删除表格业绩内容', del_tag.text) elif re.search('^项目管理机构主要人员$', tr_text): # 修复598985057 去除表格业绩 del_tag = tag.extract() # print('删除表格业绩内容', del_tag.text) del_trs = [] '''删除表格某些行公布的业绩信息''' for tag in soup.find_all('table'): text = tag.text if re.search('业绩', text) == None: continue # for tr in tag.find_all('tr'): trs = tag.find_all('tr') i = 0 while i < len(trs): tr = trs[i] if len(tr.find_all('td')) == 2 and tr.td != None and tr.td.findNextSibling() != None: td1_text = tr.td.text td2_text = tr.td.findNextSibling().text if re.search('业绩', td1_text) != None and len(td1_text) < 10 and len(re.findall('(\d、|(\d))?[-\w()、]+(工程|项目|勘察|设计|施工|监理|总承包|采购|更新)', td2_text)) >= 2: # del_tag = tr.extract() # print('删除表格业绩内容', del_tag.text) del_trs.append(tr) elif tr.td != None and re.search('^业绩|业绩$', tr.td.text.strip()) and len(tr.td.text.strip()) < 25: rows = tr.td.attrs.get('rowspan', '') cols = tr.td.attrs.get('colspan', '') if rows.isdigit() and int(rows) > 2: for j in range(int(rows)): if i + j < len(trs): del_trs.append(trs[i + j]) i += j elif cols.isdigit() and int(cols) > 3 and len(tr.find_all('td')) == 1 and i + 2 < len(trs): next_tr_cols = 0 td_num = 0 for td in trs[i + 1].find_all('td'): td_num += 1 if td.attrs.get('colspan', '').isdigit(): next_tr_cols += int(td.attrs.get('colspan', '')) if next_tr_cols == int(cols): del_trs.append(tr) for j in range(1, len(trs) - i): if len(trs[i + j].find_all('td')) == 1: break elif len(trs[i + j].find_all('td')) >= td_num - 1: del_trs.append(trs[i + j]) else: break i += j i += 1 for tr in del_trs: del_tag = tr.extract() # print('删除表格业绩内容', del_tag.text) def recall(y_true, y_pred): ''' 计算召回率 @Argus: y_true: 正确的标签 y_pred: 模型预测的标签 @Return 召回率 ''' c1 = K.sum(K.round(K.clip(y_true * y_pred, 0, 1))) c3 = K.sum(K.round(K.clip(y_true, 0, 1))) if c3 == 0: return 0 recall = c1 / c3 return recall def f1_score(y_true, y_pred): ''' 计算F1 @Argus: y_true: 正确的标签 y_pred: 模型预测的标签 @Return F1值 ''' c1 = K.sum(K.round(K.clip(y_true * y_pred, 0, 1))) c2 = K.sum(K.round(K.clip(y_pred, 0, 1))) c3 = K.sum(K.round(K.clip(y_true, 0, 1))) precision = c1 / c2 if c3 == 0: recall = 0 else: recall = c1 / c3 f1_score = 2 * (precision * recall) / (precision + recall) return f1_score def precision(y_true, y_pred): ''' 计算精确率 @Argus: y_true: 正确的标签 y_pred: 模型预测的标签 @Return 精确率 ''' c1 = K.sum(K.round(K.clip(y_true * y_pred, 0, 1))) c2 = K.sum(K.round(K.clip(y_pred, 0, 1))) precision = c1 / c2 return precision def is_adjacent_or_have_adjacent_parents(element1, element2): """ 判断两个标签是否相邻,或它们的父级标签是否相邻 """ # 获取两个元素的父元素 parent1 = element1.getparent() parent2 = element2.getparent() # 如果两个元素有相同的父元素,检查它们是否相邻 if parent1 is parent2: siblings = parent1.xpath('./*') index1 = siblings.index(element1) index2 = siblings.index(element2) # 检查索引差值是否为1(相邻) return abs(index1 - index2) == 1 elif len(parent1.xpath('./*')) == len(parent2.xpath('./*')) == 1: # 如果父元素不同,检查父元素是否相邻(有相同的父元素) grandparent1 = parent1.getparent() if parent1 is not None else None grandparent2 = parent2.getparent() if parent2 is not None else None if grandparent1 is grandparent2 and grandparent1 is not None: # 检查父元素是否相邻 # siblings = list(grandparent1) siblings = grandparent1.xpath('./*') index1 = siblings.index(parent1) index2 = siblings.index(parent2) # 检查索引差值是否为1(相邻) return abs(index1 - index2) == 1 # 如果以上都不满足,返回False return False def merge_single_row_tables(html): tree = etree.HTML(html) if tree is None: return '
' + html + '
' # 步骤1:用XPath定位所有只有一行的表格 # XPath逻辑:table下直接子节点tr的数量等于1 single_row_tables = tree.xpath('//table[count(.//tr) = 1]') # 步骤2:筛选相邻的单行列表格并合并 i = 0 shold_merge_tables = [] while i < len(single_row_tables) - 1: current_table = single_row_tables[i] next_table = single_row_tables[i + 1] if len(current_table.xpath('.//tr')) == len( next_table.xpath('.//tr')) == 1 and is_adjacent_or_have_adjacent_parents(current_table, next_table): shold_merge_tables.append(current_table) shold_merge_tables.append(next_table) j = i + 1 while j < len(single_row_tables) - 1: current_table = single_row_tables[j] next_table = single_row_tables[j + 1] if len(current_table.xpath('.//tr')) == len( next_table.xpath('.//tr')) == 1 and is_adjacent_or_have_adjacent_parents(current_table, next_table): shold_merge_tables.append(next_table) j = j + 1 else: break first_table = shold_merge_tables[0] n = 0 for next_table in shold_merge_tables[1:]: n += 1 first_table.extend(next_table.xpath('./*')) parent = next_table.getparent() parent.remove(next_table) shold_merge_tables = [] i = j else: i += 1 merged_html = etree.tostring(tree, encoding='unicode', pretty_print=True) return merged_html if __name__ == "__main__": # print(fool_char_to_id[">"]) print(getUnifyMoney('伍仟贰佰零壹拾伍万零捌佰壹拾元陆角伍分')) # model = getModel_w2v() # vocab,matrix = getVocabAndMatrix(model, Embedding_size=128) # save([vocab,matrix],"vocabMatrix_words.pk")