| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547 |
- # -*- coding: utf-8 -*-
- '''``common/Utils.py`` — Phase 3 拆分后的兼容层。
- 按 ARCHITECTURE.md Phase 3 拆分建议,原 ``common/Utils.py`` 中的函数和全局
- 变量已按职责迁移到以下 7 个目标模块:
- | 目标文件 | 迁出内容 | 类型 |
- |---|---|---|
- | ``common/json_encoder.py`` | ``MyEncoder`` | RULE FRAMEWORK |
- | ``model_runtime/viterbi.py`` | ``viterbi_decode`` | CORE |
- | ``model_runtime/vocab.py`` | ``vocab_word``、char/id 映射等 | CORE |
- | ``model_runtime/embed.py`` | w2v 加载、``embedding`` 等 | CORE |
- | ``services/external_inference/paieas.py`` | PAI-EAS 客户端 | INFRA |
- | ``common/context_utils.py`` | 上下文窗口、金额标准化等纯函数 | CORE |
- | ``common/logging.py`` | ``log`` / ``debug`` / ``logger`` | INFRA |
- 本文件通过 re-export 保持全部旧 import 路径可用:
- ``from BiddingKG.dl.common.Utils import viterbi_decode`` 仍然有效
- ``from BiddingKG.dl.common.Utils import log`` 仍然有效
- ``from BiddingKG.dl.common.Utils import *`` 仍然有效
- 未迁移的函数(时间处理、keras metrics、HTML 工具、pickle save/load、
- encodeInput 等)仍保留在本文件中,后续 Phase 再拆。
- '''
- from __future__ import absolute_import
- import os
- import re
- import time
- import pickle
- import sys
- import traceback
- import numpy as np
- from keras import backend as K
- from lxml import etree
- from datetime import date
- import calendar
- # ============================================================
- # Re-exports from model_runtime
- # ============================================================
- from BiddingKG.dl.model_runtime.viterbi import viterbi_decode
- from BiddingKG.dl.model_runtime.embed import (
- model_w2v,
- lock_model_w2v,
- model_word,
- lock_model_word,
- model_word_file,
- Lazy_load,
- getLazyLoad,
- getw2vfilepath,
- getFileFromSysPath,
- getModel_w2v,
- getModel_word,
- embedding,
- embedding_word,
- embedding_word_forward,
- formEncoding,
- )
- from BiddingKG.dl.model_runtime.vocab import (
- vocab_word,
- vocab_words,
- file_vocab_word,
- file_vocab_words,
- fool_char_to_id,
- getIndexOfWord,
- getIndexOfWords,
- getIndexOfWord_fool,
- getVocabAndMatrix,
- changeIndexFromWordToWords,
- )
- # ============================================================
- # Re-exports from services
- # ============================================================
- from BiddingKG.dl.services.external_inference.paieas import (
- USE_PAI_EAS,
- API_URL,
- USE_API,
- tf_predict_pb2,
- selffool_authorization,
- selffool_url,
- selffool_seg_authorization,
- selffool_seg_url,
- codename_authorization,
- codename_url,
- form_item_authorization,
- form_item_url,
- person_authorization,
- person_url,
- role_authorization,
- role_url,
- money_authorization,
- money_url,
- codeclasses_authorization,
- codeclasses_url,
- limitRun,
- get_values,
- vpc_requests,
- )
- # ============================================================
- # Re-exports from common submodules
- # ============================================================
- from BiddingKG.dl.common.json_encoder import MyEncoder
- from BiddingKG.dl.common.logging import log, debug, logger
- from BiddingKG.dl.common.context_utils import (
- # 上下文窗口
- spanWindow,
- get_context,
- findAllIndex,
- find_index,
- # 金额标准化
- getUnifyMoney,
- getDigitsDic,
- getMultipleFactor,
- partMoney,
- uniform_num,
- uniform_package_name,
- money_process,
- get_money_entity,
- # 其他纯函数
- combine,
- fitDataByRule,
- clean_company,
- cut_repeat_name,
- is_all_winner,
- is_deposit_project,
- find_package,
- # 模块级正则
- package_number_pattern,
- filter_package_pattern,
- )
- # ============================================================
- # 以下为未迁移函数,仍保留在本文件中
- # ============================================================
- def getCurrent_date(format="%Y-%m-%d %H:%M:%S"):
- _time = time.strftime(format, time.localtime())
- return _time
- def encodeInput(data, word_len, word_flag=True, userFool=False):
- result = []
- out_index = 0
- for item in data:
- if out_index in [0]:
- list_word = item[-word_len:]
- else:
- list_word = item[:word_len]
- temp = []
- if word_flag:
- for word in list_word:
- if userFool:
- temp.append(getIndexOfWord_fool(word))
- else:
- temp.append(getIndexOfWord(word))
- list_append = []
- temp_len = len(temp)
- while temp_len < word_len:
- if userFool:
- list_append.append(0)
- else:
- list_append.append(getIndexOfWord("<pad>"))
- temp_len += 1
- if out_index in [0]:
- temp = list_append + temp
- else:
- temp = temp + list_append
- else:
- for words in list_word:
- temp.append(getIndexOfWords(words))
- list_append = []
- temp_len = len(temp)
- while temp_len < word_len:
- list_append.append(getIndexOfWords("<pad>"))
- temp_len += 1
- if out_index in [0, 1]:
- temp = list_append + temp
- else:
- temp = temp + list_append
- result.append(temp)
- out_index += 1
- return result
- def encodeInput_form(input, MAX_LEN=30):
- x = np.zeros([MAX_LEN])
- for i in range(len(input)):
- if i >= MAX_LEN:
- break
- x[i] = getIndexOfWord(input[i])
- return x
- def save(object_to_save, path):
- '''
- 保存对象
- @Arugs:
- object_to_save: 需要保存的对象
- @Return:
- 保存的路径
- '''
- with open(path, 'wb') as f:
- pickle.dump(object_to_save, f)
- def load(path):
- '''
- 读取对象
- @Arugs:
- path: 读取的路径
- @Return:
- 读取的对象
- '''
- with open(path, 'rb') as f:
- object1 = pickle.load(f)
- return object1
- # 时间合法性判断
- def isValidDate(year, month, day):
- try:
- date(year, month, day)
- except:
- return False
- else:
- return True
- time_format_pattern = re.compile("((?P<year>20\d{2}|\d{2}|二[零〇0][零〇一二三四五六七八九0]{2})\s*[-/年.]\s*(?P<month>\d{1,2}|[一二三四五六七八九十]{1,3})\s*[-/月.]?\s*(?P<day>\d{1,2}|[一二三四五六七八九十]{1,3})?)")
- from BiddingKG.dl.ratio.re_ratio import getUnifyNum
- def get_maxday(year, month):
- # calendar.monthrange(year, month) 返回一个元组,其中第一个元素是那个月第一天的星期几(0-6代表周一到周日),
- # 第二个元素是那个月的天数。
- _, last_day = calendar.monthrange(year, month)
- return last_day
- def timeFormat(_time, default_first_day=True):
- '''
- 日期格式化:年-月-日
- :param _time:
- :param default_first_day: True取当月第一天,否则取最后一天
- :return:
- '''
- current_year = time.strftime("%Y", time.localtime())
- all_match = re.finditer(time_format_pattern, _time)
- for _match in all_match:
- if len(_match.group()) > 0:
- legal = True
- year = ""
- month = ""
- day = ""
- for k, v in _match.groupdict().items():
- if k == "year":
- year = v
- if k == "month":
- month = v
- if k == "day":
- day = v
- if year != "":
- if re.search("^\d+$", year):
- if len(year) == 2:
- year = "20" + year
- if int(year) - int(current_year) > 10:
- legal = False
- else:
- _year = ""
- for word in year:
- if word == '0':
- _year += word
- else:
- _year += str(getDigitsDic(word))
- year = _year
- else:
- legal = False
- if month != "":
- if re.search("^\d+$", month):
- if int(month) > 12:
- legal = False
- else:
- month = int(getUnifyNum(month))
- if month >= 1 and month <= 12:
- month = str(month)
- else:
- legal = False
- else:
- legal = False
- if day == None:
- day = "01" if (default_first_day or legal == False) else str(get_maxday(int(year), int(month)))
- if day != "":
- if re.search("^\d+$", day):
- if int(day) > 31:
- legal = False
- else:
- day = int(getUnifyNum(day))
- if day >= 1 and day <= 31:
- day = str(day)
- else:
- legal = False
- else:
- legal = False
- # print(year,month,day)
- if not isValidDate(int(year), int(month), int(day)):
- legal = False
- if legal:
- return "%s-%s-%s" % (year, month.rjust(2, "0"), day.rjust(2, "0"))
- return ""
- def del_tabel_achievement(soup):
- if re.search('中标|成交|入围|结果|评标|开标|候选人', soup.text[:800]) == None or re.search('业绩|类似项目', soup.text) == None:
- return None
- p1 = '(中标|成交)(单位|候选人)的?(企业|项目|项目负责人|\w{,5})?业绩|类似(项目)?业绩|\w{,10}业绩$|业绩(公示|情况|荣誉)|近年完成的项目类似项目情况表|类似项目'
- '''删除前面标签 命中业绩规则;当前标签为表格且公布业绩相关信息的去除'''
- for tag in soup.find_all('table'):
- pre_text = ""
- if tag.findPreviousSibling() != None:
- pre_text = tag.findPreviousSibling().text.strip()
- if pre_text == "" and tag.findPreviousSibling().findPreviousSibling() != None: # 修复表格前一标签没内容,再前一个才有内容情况
- pre_text = tag.findPreviousSibling().findPreviousSibling().text.strip()
- tr_text = tag.find('tr').text.strip() if tag.find('tr') != None else ""
- if len(pre_text) < 20 and re.search('^[((]?[\d一二三四五六七八九十]+[)、)]近年完成的(项目)?类似项目情况表|类似项目历史成交信息|类似项目的?采购预算', pre_text): # 删除 650469910 评标公告 历史业绩 547305313
- del_tag = tag.extract()
- # print('删除表格业绩内容', del_tag.text)
- # print(re.search(p1, pre_text),pre_text, len(pre_text), re.findall('序号|中标候选人名称|项目名称|工程名称|合同金额|建设单位|业主', tr_text))
- elif re.search(p1, pre_text) and len(pre_text) < 20 and tag.find('tr') != None and len(tr_text) < 100:
- _count = 0
- for td in tag.find('tr').find_all('td'):
- td_text = td.text.strip()
- if len(td_text) > 25:
- break
- if len(td_text) < 25 and re.search('中标候选人|第[一二三四五1-5]候选人|(项目|业绩|工程)名称|\w{,10}业绩$|合同(金额|价格)|建设单位|采购单位|业主|甲方|发包人', td_text):
- _count += 1
- if _count >= 2:
- pre_tag = tag.findPreviousSibling().extract()
- del_tag = tag.extract()
- # print('删除表格业绩内容', pre_tag.text + del_tag.text)
- break
- elif re.search('业绩名称', tr_text) and re.search('建设单位|采购单位|业主', tr_text) and len(tr_text) < 100:
- del_tag = tag.extract()
- # print('删除表格业绩内容', del_tag.text)
- elif re.search('^项目管理机构主要人员$', tr_text): # 修复598985057 去除表格业绩
- del_tag = tag.extract()
- # print('删除表格业绩内容', del_tag.text)
- del_trs = []
- '''删除表格某些行公布的业绩信息'''
- for tag in soup.find_all('table'):
- text = tag.text
- if re.search('业绩', text) == None:
- continue
- # for tr in tag.find_all('tr'):
- trs = tag.find_all('tr')
- i = 0
- while i < len(trs):
- tr = trs[i]
- if len(tr.find_all('td')) == 2 and tr.td != None and tr.td.findNextSibling() != None:
- td1_text = tr.td.text
- td2_text = tr.td.findNextSibling().text
- if re.search('业绩', td1_text) != None and len(td1_text) < 10 and len(re.findall('(\d、|(\d))?[-\w()、]+(工程|项目|勘察|设计|施工|监理|总承包|采购|更新)', td2_text)) >= 2:
- # del_tag = tr.extract()
- # print('删除表格业绩内容', del_tag.text)
- del_trs.append(tr)
- elif tr.td != None and re.search('^业绩|业绩$', tr.td.text.strip()) and len(tr.td.text.strip()) < 25:
- rows = tr.td.attrs.get('rowspan', '')
- cols = tr.td.attrs.get('colspan', '')
- if rows.isdigit() and int(rows) > 2:
- for j in range(int(rows)):
- if i + j < len(trs):
- del_trs.append(trs[i + j])
- i += j
- elif cols.isdigit() and int(cols) > 3 and len(tr.find_all('td')) == 1 and i + 2 < len(trs):
- next_tr_cols = 0
- td_num = 0
- for td in trs[i + 1].find_all('td'):
- td_num += 1
- if td.attrs.get('colspan', '').isdigit():
- next_tr_cols += int(td.attrs.get('colspan', ''))
- if next_tr_cols == int(cols):
- del_trs.append(tr)
- for j in range(1, len(trs) - i):
- if len(trs[i + j].find_all('td')) == 1:
- break
- elif len(trs[i + j].find_all('td')) >= td_num - 1:
- del_trs.append(trs[i + j])
- else:
- break
- i += j
- i += 1
- for tr in del_trs:
- del_tag = tr.extract()
- # print('删除表格业绩内容', del_tag.text)
- def recall(y_true, y_pred):
- '''
- 计算召回率
- @Argus:
- y_true: 正确的标签
- y_pred: 模型预测的标签
- @Return
- 召回率
- '''
- c1 = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))
- c3 = K.sum(K.round(K.clip(y_true, 0, 1)))
- if c3 == 0:
- return 0
- recall = c1 / c3
- return recall
- def f1_score(y_true, y_pred):
- '''
- 计算F1
- @Argus:
- y_true: 正确的标签
- y_pred: 模型预测的标签
- @Return
- F1值
- '''
- c1 = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))
- c2 = K.sum(K.round(K.clip(y_pred, 0, 1)))
- c3 = K.sum(K.round(K.clip(y_true, 0, 1)))
- precision = c1 / c2
- if c3 == 0:
- recall = 0
- else:
- recall = c1 / c3
- f1_score = 2 * (precision * recall) / (precision + recall)
- return f1_score
- def precision(y_true, y_pred):
- '''
- 计算精确率
- @Argus:
- y_true: 正确的标签
- y_pred: 模型预测的标签
- @Return
- 精确率
- '''
- c1 = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))
- c2 = K.sum(K.round(K.clip(y_pred, 0, 1)))
- precision = c1 / c2
- return precision
- def is_adjacent_or_have_adjacent_parents(element1, element2):
- """
- 判断两个标签是否相邻,或它们的父级标签是否相邻
- """
- # 获取两个元素的父元素
- parent1 = element1.getparent()
- parent2 = element2.getparent()
- # 如果两个元素有相同的父元素,检查它们是否相邻
- if parent1 is parent2:
- siblings = parent1.xpath('./*')
- index1 = siblings.index(element1)
- index2 = siblings.index(element2)
- # 检查索引差值是否为1(相邻)
- return abs(index1 - index2) == 1
- elif len(parent1.xpath('./*')) == len(parent2.xpath('./*')) == 1:
- # 如果父元素不同,检查父元素是否相邻(有相同的父元素)
- grandparent1 = parent1.getparent() if parent1 is not None else None
- grandparent2 = parent2.getparent() if parent2 is not None else None
- if grandparent1 is grandparent2 and grandparent1 is not None:
- # 检查父元素是否相邻
- # siblings = list(grandparent1)
- siblings = grandparent1.xpath('./*')
- index1 = siblings.index(parent1)
- index2 = siblings.index(parent2)
- # 检查索引差值是否为1(相邻)
- return abs(index1 - index2) == 1
- # 如果以上都不满足,返回False
- return False
- def merge_single_row_tables(html):
- tree = etree.HTML(html)
- if tree is None:
- return '<div>' + html + '</div>'
- # 步骤1:用XPath定位所有只有一行的表格
- # XPath逻辑:table下直接子节点tr的数量等于1
- single_row_tables = tree.xpath('//table[count(.//tr) = 1]')
- # 步骤2:筛选相邻的单行列表格并合并
- i = 0
- shold_merge_tables = []
- while i < len(single_row_tables) - 1:
- current_table = single_row_tables[i]
- next_table = single_row_tables[i + 1]
- if len(current_table.xpath('.//tr')) == len(
- next_table.xpath('.//tr')) == 1 and is_adjacent_or_have_adjacent_parents(current_table, next_table):
- shold_merge_tables.append(current_table)
- shold_merge_tables.append(next_table)
- j = i + 1
- while j < len(single_row_tables) - 1:
- current_table = single_row_tables[j]
- next_table = single_row_tables[j + 1]
- if len(current_table.xpath('.//tr')) == len(
- next_table.xpath('.//tr')) == 1 and is_adjacent_or_have_adjacent_parents(current_table,
- next_table):
- shold_merge_tables.append(next_table)
- j = j + 1
- else:
- break
- first_table = shold_merge_tables[0]
- n = 0
- for next_table in shold_merge_tables[1:]:
- n += 1
- first_table.extend(next_table.xpath('./*'))
- parent = next_table.getparent()
- parent.remove(next_table)
- shold_merge_tables = []
- i = j
- else:
- i += 1
- merged_html = etree.tostring(tree, encoding='unicode', pretty_print=True)
- return merged_html
- if __name__ == "__main__":
- # print(fool_char_to_id[">"])
- print(getUnifyMoney('伍仟贰佰零壹拾伍万零捌佰壹拾元陆角伍分'))
- # model = getModel_w2v()
- # vocab,matrix = getVocabAndMatrix(model, Embedding_size=128)
- # save([vocab,matrix],"vocabMatrix_words.pk")
|