python3环境下汉字转拼音

来源:互联网 发布:各种旋转矩阵公式 编辑:程序博客网 时间:2024/05/16 12:58

参考:https://github.com/cleverdeng/pinyin.py

上述代码适合python2.7的代码,但是对于python3.5的环境却不是很友好,所以将代码改成python3.5环境下可以运行的代码,代码如下:


# -*- coding:utf-8 -*-"""    Author:cleverdeng    E-mail:clverdeng@gmail.com"""__version__ = '0.9'__all__ = ["PinYin"]import os.pathclass PinYin(object):    def __init__(self, dict_file='word.data'):        self.word_dict = {}        self.dict_file = dict_file    def load_word(self):        if not os.path.exists(self.dict_file):            raise IOError("NotFoundFile")        with open(self.dict_file) as f_obj:            for f_line in f_obj.readlines():                try:                    line = f_line.split('    ')                    self.word_dict[line[0]] = line[1]                except:                    line = f_line.split('   ')                    self.word_dict[line[0]] = line[1]    def hanzi2pinyin(self, string=""):        result = []        for char in string:            key = '%X' % ord(char)            result.append(self.word_dict.get(key, char).split()[0][:-1].lower())        return result    def hanzi2pinyin_split(self, string="", split=""):        result = self.hanzi2pinyin(string=string)        if split == "":            return result        else:            return split.join(result)if __name__ == "__main__":    test = PinYin()    test.load_word()    string = "钓鱼岛是中国的"    print( "in: %s" % string)    print( "out: %s" % str(test.hanzi2pinyin(string=string)))    print( "out: %s" % test.hanzi2pinyin_split(string=string, split="-"))


原创粉丝点击