# -*- coding: utf-8 -*- """Training / model-saving helpers. 按 ARCHITECTURE.md Phase 5 拆分建议,从 ``interface/predictor.py`` 迁出的 训练 / 模型保存相关函数。这些函数构建或加载 Keras / TensorFlow 模型,并将其 导出为 TensorFlow ``SavedModel`` 产物以供部署调用(codename / role / money / person / form / codesplit / timesplit)。 原位置:``interface/predictor.py`` 末尾(约第 10155-10448 行)。 """ from __future__ import absolute_import import os import re import sys import pickle import h5py import numpy as np import tensorflow as tf from keras import models, layers # from keras_contrib.layers.crf import CRF from BiddingKG.dl.common.Utils import load, precision, recall, f1_score from BiddingKG.dl.complaint.punishNo_tf import BiLSTM_CRF_tfmodel from BiddingKG.dl.predictors.prem import PREMPredict, EPCPredict from BiddingKG.dl.predictors.form import FormPredictor from BiddingKG.dl.predictors._common import INTERFACE_DIR __all__ = [ "getSavedModel", "getBiLSTMCRFModel", "h5_to_graph", "initialize_uninitialized", "save_codename_model", "save_role_model", "save_money_model", "save_person_model", "save_form_model", "save_codesplit_model", "save_timesplit_model", ] def getSavedModel(): #predictor = FormPredictor() graph = tf.Graph() with graph.as_default(): model = tf.keras.models.load_model("../form/model/model_form.model_item.hdf5",custom_objects={"precision":precision,"recall":recall,"f1_score":f1_score}) #print(tf.graph_util.remove_training_nodes(model)) tf.saved_model.simple_save( tf.keras.backend.get_session(), "./h5_savedmodel/", inputs={"image": model.input}, outputs={"scores": model.output} ) def getBiLSTMCRFModel(MAX_LEN,vocab,EMBED_DIM,BiRNN_UNITS,chunk_tags,weights): ''' model = models.Sequential() model.add(layers.Embedding(len(vocab), EMBED_DIM, mask_zero=True)) # Random embedding model.add(layers.Bidirectional(layers.LSTM(BiRNN_UNITS // 2, return_sequences=True))) crf = CRF(len(chunk_tags), sparse_target=True) model.add(crf) model.summary() model.compile('adam', loss=crf.loss_function, metrics=[crf.accuracy]) return model ''' input = layers.Input(shape=(None,),dtype="int32") if weights is not None: embedding = layers.embeddings.Embedding(len(vocab),EMBED_DIM,mask_zero=True,weights=[weights],trainable=True)(input) else: embedding = layers.embeddings.Embedding(len(vocab),EMBED_DIM,mask_zero=True)(input) bilstm = layers.Bidirectional(layers.LSTM(BiRNN_UNITS//2,return_sequences=True))(embedding) bilstm_dense = layers.TimeDistributed(layers.Dense(len(chunk_tags)))(bilstm) crf = CRF(len(chunk_tags),sparse_target=True) crf_out = crf(bilstm_dense) model = models.Model(input=[input],output = [crf_out]) model.summary() model.compile(optimizer = 'adam', loss = crf.loss_function, metrics = [crf.accuracy]) return model def h5_to_graph(sess,graph,h5file): f = h5py.File(h5file,'r') #打开h5文件 def getValue(v): _value = f["model_weights"] list_names = str(v.name).split("/") for _index in range(len(list_names)): print(v.name) if _index==1: _value = _value[list_names[0]] _value = _value[list_names[_index]] return _value def _load_attributes_from_hdf5_group(group, name): """Loads attributes of the specified name from the HDF5 group. This method deals with an inherent problem of HDF5 file which is not able to store data larger than HDF5_OBJECT_HEADER_LIMIT bytes. # Arguments group: A pointer to a HDF5 group. name: A name of the attributes to load. # Returns data: Attributes data. """ if name in group.attrs: data = [n.decode('utf8') for n in group.attrs[name]] else: data = [] chunk_id = 0 while ('%s%d' % (name, chunk_id)) in group.attrs: data.extend([n.decode('utf8') for n in group.attrs['%s%d' % (name, chunk_id)]]) chunk_id += 1 return data def readGroup(gr,parent_name,data): for subkey in gr: print(subkey) if parent_name!=subkey: if parent_name=="": _name = subkey else: _name = parent_name+"/"+subkey else: _name = parent_name if str(type(gr[subkey]))=="": readGroup(gr[subkey],_name,data) else: data.append([_name,gr[subkey].value]) print(_name,gr[subkey].shape) layer_names = _load_attributes_from_hdf5_group(f["model_weights"], 'layer_names') list_name_value = [] readGroup(f["model_weights"], "", list_name_value) ''' for k, name in enumerate(layer_names): g = f["model_weights"][name] weight_names = _load_attributes_from_hdf5_group(g, 'weight_names') #weight_values = [np.asarray(g[weight_name]) for weight_name in weight_names] for weight_name in weight_names: list_name_value.append([weight_name,np.asarray(g[weight_name])]) ''' for name_value in list_name_value: name = name_value[0] ''' if re.search("dense",name) is not None: name = name[:7]+"_1"+name[7:] ''' value = name_value[1] print(name,graph.get_tensor_by_name(name),np.shape(value)) sess.run(tf.assign(graph.get_tensor_by_name(name),value)) def initialize_uninitialized(sess): global_vars = tf.global_variables() is_not_initialized = sess.run([tf.is_variable_initialized(var) for var in global_vars]) not_initialized_vars = [v for (v, f) in zip(global_vars, is_not_initialized) if not f] adam_vars = [] for _vars in not_initialized_vars: if re.search("Adam",_vars.name) is not None: adam_vars.append(_vars) print([str(i.name) for i in adam_vars]) # only for testing if len(adam_vars): sess.run(tf.variables_initializer(adam_vars)) def save_codename_model(): # filepath = "../projectCode/models/model_project_"+str(60)+"_"+str(200)+".hdf5" filepath = "../../dl_dev/projectCode/models_tf/59-L0.471516189943-F0.8802154826344823-P0.8789179683459191-R0.8815168335321886/model.ckpt" vocabpath = "../projectCode/models/vocab.pk" classlabelspath = "../projectCode/models/classlabels.pk" # vocab = load(vocabpath) # class_labels = load(classlabelspath) w2v_matrix = load('codename_w2v_matrix.pk') graph = tf.get_default_graph() with graph.as_default() as g: '''''' # model = getBiLSTMCRFModel(None, vocab, 60, 200, class_labels,weights=None) #model = models.load_model(filepath,custom_objects={'precision':precision,'recall':recall,'f1_score':f1_score,"CRF":CRF,"loss":CRF.loss_function}) sess = tf.Session(graph=g) # sess = tf.keras.backend.get_session() char_input, logits, target, keepprob, length, crf_loss, trans, train_op = BiLSTM_CRF_tfmodel(sess, w2v_matrix) #with sess.as_default(): sess.run(tf.global_variables_initializer()) # print(sess.run("time_distributed_1/kernel:0")) # model.load_weights(filepath) saver = tf.train.Saver() saver.restore(sess, filepath) # print("logits",sess.run(logits)) # print("#",sess.run("time_distributed_1/kernel:0")) # x = load("codename_x.pk") #y = model.predict(x) # y = sess.run(model.output,feed_dict={model.input:x}) # for item in np.argmax(y,-1): # print(item) tf.saved_model.simple_save( sess, "./codename_savedmodel_tf/", inputs={"inputs": char_input, "inputs_length":length, 'keepprob':keepprob}, outputs={"logits": logits, "trans":trans} ) def save_role_model(): ''' @summary: 保存model为savedModel,部署到PAI平台上调用 ''' model_role = PREMPredict().model_role with model_role.graph.as_default(): model = model_role.getModel() sess = tf.Session(graph=model_role.graph) print(type(model.input)) sess.run(tf.global_variables_initializer()) h5_to_graph(sess, model_role.graph, model_role.model_role_file) model = model_role.getModel() tf.saved_model.simple_save(sess, "./role_savedmodel/", inputs={"input0":model.input[0], "input1":model.input[1], "input2":model.input[2]}, outputs={"outputs":model.output} ) def save_money_model(): model_file = os.path.join(INTERFACE_DIR, "../money/models/model_money_word.h5") graph = tf.Graph() with graph.as_default(): sess = tf.Session(graph=graph) with sess.as_default(): # model = model_money.getModel() # model.summary() # sess.run(tf.global_variables_initializer()) # h5_to_graph(sess, model_money.graph, model_money.model_money_file) model = models.load_model(model_file,custom_objects={'precision':precision,'recall':recall,'f1_score':f1_score}) model.summary() print(model.weights) tf.saved_model.simple_save(sess, "./money_savedmodel2/", inputs = {"input0":model.input[0], "input1":model.input[1], "input2":model.input[2]}, outputs = {"outputs":model.output} ) def save_person_model(): model_person = EPCPredict().model_person with model_person.graph.as_default(): x = load("person_x.pk") _data = np.transpose(np.array(x),(1,0,2,3)) model = model_person.getModel() sess = tf.Session(graph=model_person.graph) with sess.as_default(): sess.run(tf.global_variables_initializer()) model_person.load_weights() #h5_to_graph(sess, model_person.graph, model_person.model_person_file) predict_y = sess.run(model.output,feed_dict={model.input[0]:_data[0],model.input[1]:_data[1]}) #predict_y = model.predict([_data[0],_data[1]]) print(np.argmax(predict_y,-1)) tf.saved_model.simple_save(sess, "./person_savedmodel/", inputs={"input0":model.input[0], "input1":model.input[1]}, outputs = {"outputs":model.output}) def save_form_model(): model_form = FormPredictor() with model_form.graph.as_default(): model = model_form.getModel("item") sess = tf.Session(graph=model_form.graph) sess.run(tf.global_variables_initializer()) h5_to_graph(sess, model_form.graph, model_form.model_file_item) tf.saved_model.simple_save(sess, "./form_savedmodel/", inputs={"inputs":model.input}, outputs = {"outputs":model.output}) def save_codesplit_model(): filepath_code = "../../dl_dev/projectCode/models/model_code.hdf5" graph = tf.Graph() with graph.as_default(): model_code = models.load_model(filepath_code, custom_objects={'precision':precision,'recall':recall,'f1_score':f1_score}) sess = tf.Session() sess.run(tf.global_variables_initializer()) h5_to_graph(sess, graph, filepath_code) tf.saved_model.simple_save(sess, "./codesplit_savedmodel/", inputs={"input0":model_code.input[0], "input1":model_code.input[1], "input2":model_code.input[2]}, outputs={"outputs":model_code.output}) def save_timesplit_model(): filepath = '../time/model_label_time_classify.model.hdf5' with tf.Graph().as_default() as graph: time_model = models.load_model(filepath, custom_objects={'precision': precision, 'recall': recall, 'f1_score': f1_score}) with tf.Session() as sess: sess.run(tf.global_variables_initializer()) h5_to_graph(sess, graph, filepath) tf.saved_model.simple_save(sess, "./timesplit_model/", inputs={"input0":time_model.input[0], "input1":time_model.input[1]}, outputs={"outputs":time_model.output})