| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129 |
- # -*- coding: utf-8 -*-
- """字/词 vocab 与 char/id 映射。
- 按 ARCHITECTURE.md Phase 3 拆分建议,从 ``common/Utils.py`` 迁出。
- 类型:CORE(模型推理基础,人工主导)。
- 原位置:``common/Utils.py`` 中以下函数和全局变量:
- - ``vocab_word`` / ``vocab_words``
- - ``file_vocab_word`` / ``file_vocab_words``
- - ``fool_char_to_id``(模块加载时从 ``fool_char_to_id.pk`` 读取)
- - ``getIndexOfWord`` / ``getIndexOfWords`` / ``getIndexOfWord_fool``
- - ``getVocabAndMatrix``
- - ``changeIndexFromWordToWords``
- 依赖关系:
- - ``getIndexOfWord`` / ``getIndexOfWords`` 调用 ``getModel_word`` / ``getModel_w2v``
- (来自 ``model_runtime.embed``)和 ``save`` / ``load``
- (仍留在 ``common/Utils.py``,本文件用延迟 import 避免循环依赖)。
- - ``fool_char_to_id`` 在模块加载时用 ``pickle.load`` 直接读取,不依赖 ``save`` / ``load``。
- ``common/Utils.py`` 仍 re-export 以上全部名称,老 import 不受影响。
- 按 ARCHITECTURE.md §4.3 依赖方向约束:
- model_runtime -> domain, infra only
- """
- from __future__ import absolute_import
- import os
- import pickle
- import numpy as np
- from BiddingKG.dl.model_runtime import embed as _embed
- __all__ = [
- "vocab_word",
- "vocab_words",
- "file_vocab_word",
- "file_vocab_words",
- "fool_char_to_id",
- "getIndexOfWord",
- "getIndexOfWords",
- "getIndexOfWord_fool",
- "getVocabAndMatrix",
- "changeIndexFromWordToWords",
- ]
- vocab_word = None
- vocab_words = None
- file_vocab_word = "vocab_word.pk"
- file_vocab_words = "vocab_words.pk"
- fool_char_to_id = pickle.load(
- open(os.path.dirname(os.path.abspath(__file__)) + "/../common/fool_char_to_id.pk", 'rb')
- )
- def getVocabAndMatrix(model, Embedding_size=60):
- '''
- @summary:获取子向量的词典和子向量矩阵
- '''
- vocab = ["<pad>"] + model.index2word
- embedding_matrix = np.zeros((len(vocab), Embedding_size))
- for i in range(1, len(vocab)):
- embedding_matrix[i] = model[vocab[i]]
- return vocab, embedding_matrix
- def getIndexOfWord(word):
- global vocab_word, file_vocab_word
- if vocab_word is None:
- if os.path.exists(file_vocab_word):
- # 延迟 import save/load 避免 common/Utils.py <-> model_runtime/vocab.py 循环依赖
- from BiddingKG.dl.common.Utils import load, save
- vocab = load(file_vocab_word)
- vocab_word = dict((w, i) for i, w in enumerate(np.array(vocab)))
- else:
- model = _embed.getModel_word()
- vocab, _ = getVocabAndMatrix(model, Embedding_size=60)
- vocab_word = dict((w, i) for i, w in enumerate(np.array(vocab)))
- from BiddingKG.dl.common.Utils import save
- save(vocab, file_vocab_word)
- if word in vocab_word.keys():
- return vocab_word[word]
- else:
- return vocab_word['<pad>']
- def changeIndexFromWordToWords(tokens, word_index):
- '''
- @summary:转换某个字的字偏移为词偏移
- '''
- before_index = 0
- after_index = 0
- for i in range(len(tokens)):
- after_index = after_index + len(tokens[i])
- if before_index <= word_index and after_index > word_index:
- return i
- before_index = after_index
- return i + 1
- def getIndexOfWords(words):
- global vocab_words, file_vocab_words
- if vocab_words is None:
- if os.path.exists(file_vocab_words):
- from BiddingKG.dl.common.Utils import load, save
- vocab = load(file_vocab_words)
- vocab_words = dict((w, i) for i, w in enumerate(np.array(vocab)))
- else:
- model = _embed.getModel_w2v()
- vocab, _ = getVocabAndMatrix(model, Embedding_size=128)
- vocab_words = dict((w, i) for i, w in enumerate(np.array(vocab)))
- from BiddingKG.dl.common.Utils import save
- save(vocab, file_vocab_words)
- if words in vocab_words.keys():
- return vocab_words[words]
- else:
- return vocab_words["<pad>"]
- def getIndexOfWord_fool(word):
- if word in fool_char_to_id.keys():
- return fool_char_to_id[word]
- else:
- return fool_char_to_id["[UNK]"]
|