
生成 ocr key 字符集 alphabet
import pickle as pkl
#----------- 生成 ocr key 字符集 alphabet
alphabet_set = set()
# 数据集label
infofiles_label = ['/home/jlb/下载/rec_data_lesson_demo/train.txt', '/home/jlb/下载/rec_data_lesson_demo/val.txt']
# ppocr中文key
infofiles = ['/home/jlb/Project/OCR/ocr/recognize/ppocr_keys_v1.txt']
for infofile in infofiles:
f = open(infofile)
content = f.readlines()
f.close()
for line in content:
line = line.replace("\n", "")
alphabet_set.add(line)
for infofile in infofiles_label:
f = open(infofile)
content = f.readlines()
f.close()
for line in content:
if len(line.strip())>0:
if len(line.strip().split('\t'))!=2:
print(line)
else:
fname,label = line.strip().split('\t')
for ch in label:
alphabet_set.add(ch)
alphabet_list = sorted(list(alphabet_set))
pkl.dump(alphabet_list, open('alphabet.pkl','wb'))
读取字符集 alphabet
#----------- 读取字符集 alphabet
alphabet_list = pkl.load(open('/home/jlb/Project/OCR/ocr/recognize/alphabet.pkl','rb'))
alphabet = [ord(ch) for ch in alphabet_list]
alphabet_v2 = alphabet
print("----------------------- alphabet_v2:", len(alphabet_list), alphabet_list)
这段代码主要实现了两个功能:一是从指定的数据集和PPOCR中文key文件中生成OCR字符集alphabet,并保存为pkl文件;二是读取这个pkl文件,将字符集转换为对应的ASCII码列表。整个过程涉及到文本处理和序列化操作。
4273

被折叠的 条评论
为什么被折叠?



