标签:一个 charset ret content return TE cto lang data
在python处理文本的过程中,经常会有文本字符集转换的情况,import chardet
def convert_encoding(data,new_coding=‘UTF-8‘):
# 任意字符集转换
encoding = chardet.detect(data)[‘encoding‘]
if new_coding.upper() != encoding.upper():
data = data.decode(encoding,data).encode(new_coding)
return data
import icu
def convert_encoding2(data,new_coding=‘UTF-8‘):
encoding = icu.CharsetDetector(data).detect().getName()
# encoding = chardet.detect(content)[‘encoding‘]
if new_coding.upper() != encoding.upper():
# data = data.decode(encoding,data).encode(new_coding)
data = unicode(data,coding).encode(new_coding)
return data
import cchardet
def convert_encoding3(data,new_coding=‘UTF-8‘):
encoding = cchardet.detect(data)[‘encoding‘]
if new_coding.upper() != encoding.upper():
data = data.decode(encoding,data).encode(new_coding)
return data
此处使用方法一
#转换成utf-8
convert_encoding(data,‘utf-8‘)
#转抱成GBK
convert_encoding(data,‘gbk‘)
#转抱成GB2312
convert_encoding(data,‘gbk‘)
标签:一个 charset ret content return TE cto lang data
原文地址:http://blog.51cto.com/yangrong/2130811