cwenzi/neuroflow-cpp
1
1import json
2
3path = r'D:\neuroflow-C++\configs\tokenizer_128k.json'
4d = json.load(open(path, 'r', encoding='utf-8'))
5vocab = d['vocab']
6
7# Rebuild vocab with sorted IDs (0, 1, 2, ..., 127999)
8# This ensures JSON serialization puts them in order
9sorted_vocab = {}
10id_to_key = {}
11for k, v in vocab.items():
12 id_to_key[v] = k
13
14for i in range(128000):
15 if i in id_to_key:
16 sorted_vocab[id_to_key[i]] = i
17 else:
18 sorted_vocab['<pad_extra_' + str(i) + '>'] = i
19
20d['vocab'] = sorted_vocab
21d['vocab_size'] = 128000
22
23# Verify
24assert len(sorted_vocab) == 128000
25all_ids = set(sorted_vocab.values())
26assert all_ids == set(range(128000))
27
28with open(path, 'w', encoding='utf-8') as f:
29 json.dump(d, f, ensure_ascii=False, indent=2)
30
31print('Re-saved tokenizer_128k.json')
32print('Total tokens:', len(sorted_vocab))
33print('ID coverage:', min(sorted_vocab.values()), '-', max(sorted_vocab.values()))