# -*- coding: utf-8 -*-
"""
Created on Mon May 7 19:43:04 2018
@author: Administrator
"""
from __future__ import absolute_import
from __future__ import division
from __future__ import print_function
import collections
import tensorflow as tf
import os
import numpy as np
def _read_words(filename):
with tf.gfile.GFile(filename, "r") as f:
return f.read().replace("\n", "").split()
def _build_vocab(filename):
data = _read_words(filename)
counter = collections.Counter(data)
count_pairs = sorted(counter.items(), key=lambda x: (-x[1], x[0]))
words, _ = list(zip(*count_pairs))
word_to_id = dict(zip(words, range(len(words))))
return word_to_id
2021-04-22 21:03:53
2KB
reader
1