生成 ocr key 字符集 alphabet
import pickle as pkl #----------- 生成 ocr key 字符集 alphabet alphabet_set = set() # 数据集label infofiles_label = ['/home/jlb/下载/rec_data_lesson_demo/train.txt', '/home/jlb/下载/rec_data_lesson_demo/val.txt'] # ppocr中文key infofiles = ['/home/jlb/Project/OCR/ocr/recognize/ppocr_keys_v1.txt'] for infofile in infofiles: f = open(infofile) content = f.readlines() f.close() for line in content: line = line.replace("\n", "") alphabet_set.add(line) for infofile in infofiles_label: f = open(infofile) content = f.readlines() f.close() for line in content: if len(line.strip())>0: if len(line.strip().split('\t'))!=2: print(line) else: fname,label = line.strip().split('\t') for ch in label: alphabet_set.add(ch) alphabet_list = sorted(list(alphabet_set)) pkl.dump(alphabet_list, open('alphabet.pkl','wb'))
读取字符集 alphabet
#----------- 读取字符集 alphabet
alphabet_list = pkl.load(open('/home/jlb/Project/OCR/ocr/recognize/alphabet.pkl','rb'))
alphabet = [ord(ch) for ch in alphabet_list]
alphabet_v2 = alphabet
print("----------------------- alphabet_v2:", len(alphabet_list), alphabet_list)