ContextReq commited on
Commit
9710741
·
verified ·
1 Parent(s): ecb8961

uploaded work done on vocab

Browse files
Files changed (3) hide show
  1. vocab.bin +3 -0
  2. vocab.txt +109 -0
  3. vocab_to_bin.py +53 -0
vocab.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:04eaea28ee517ea93c73ae159633ec91ab3e5dc55aca69742e606ec155685bcc
3
+ size 264
vocab.txt ADDED
@@ -0,0 +1,109 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ <unk>
2
+ <bos>
3
+ <eos>
4
+ <pad>
5
+ <msk>
6
+ <spc>
7
+ <tab>
8
+ <nwl>
9
+ A
10
+ B
11
+ C
12
+ D
13
+ E
14
+ F
15
+ G
16
+ H
17
+ I
18
+ J
19
+ K
20
+ L
21
+ M
22
+ N
23
+ O
24
+ P
25
+ Q
26
+ R
27
+ S
28
+ T
29
+ U
30
+ V
31
+ W
32
+ X
33
+ Y
34
+ Z
35
+ a
36
+ b
37
+ c
38
+ d
39
+ e
40
+ f
41
+ g
42
+ h
43
+ i
44
+ j
45
+ k
46
+ l
47
+ m
48
+ n
49
+ o
50
+ p
51
+ q
52
+ r
53
+ s
54
+ t
55
+ u
56
+ v
57
+ w
58
+ x
59
+ y
60
+ z
61
+ 0
62
+ 1
63
+ 2
64
+ 3
65
+ 4
66
+ 5
67
+ 6
68
+ 7
69
+ 8
70
+ 9
71
+ !
72
+ "
73
+ #
74
+ $
75
+ %
76
+ &
77
+ '
78
+ (
79
+ )
80
+ *
81
+ +
82
+ ,
83
+ -
84
+ .
85
+ /
86
+ :
87
+ ;
88
+ <
89
+ =
90
+ >
91
+ ?
92
+ @
93
+ [
94
+ \
95
+ ]
96
+ ^
97
+ _
98
+ `
99
+ {
100
+ |
101
+ }
102
+ ~
103
+ —
104
+ –
105
+ ‘
106
+ ’
107
+ “
108
+ ”
109
+ …
vocab_to_bin.py ADDED
@@ -0,0 +1,53 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """vocab.txt -> vocab.bin (strict, minimal)
2
+
3
+ Format: u8 length + token bytes, one entry per line, id = entry order.
4
+ Refuses to run if the file looks wrong instead of silently corrupting.
5
+ """
6
+ import sys
7
+ from pathlib import Path
8
+
9
+ HERE = Path(__file__).resolve().parent
10
+ SRC = HERE / "vocab.txt"
11
+ DST = HERE / "vocab.bin"
12
+ SPECIALS = ["<unk>", "<bos>", "<eos>", "<pad>", "<msk>", "<spc>", "<tab>", "<nwl>"]
13
+ EXPECTED_COUNT = 109 # 8 specials + 94 ascii + 7 smart punct
14
+
15
+
16
+ def main():
17
+ data = SRC.read_bytes()
18
+ if b"\r" in data:
19
+ sys.exit("refusing: CR present (CRLF line endings?) — fix the file first")
20
+ lines = data.split(b"\n")
21
+ if lines and lines[-1] == b"":
22
+ lines.pop()
23
+ if len(lines) != EXPECTED_COUNT:
24
+ sys.exit(f"refusing: {len(lines)} tokens, expected {EXPECTED_COUNT}")
25
+ for i, tok in enumerate(lines):
26
+ if not tok:
27
+ sys.exit(f"refusing: empty token at line {i + 1}")
28
+ if len(tok) > 255:
29
+ sys.exit(f"refusing: token {i + 1} longer than 255 bytes")
30
+ for i, want in enumerate(SPECIALS):
31
+ if lines[i] != want.encode():
32
+ sys.exit(f"refusing: specials block wrong at id {i}: {lines[i]!r} != {want!r}")
33
+ out = bytearray()
34
+ for tok in lines:
35
+ out.append(len(tok))
36
+ out.extend(tok)
37
+ DST.write_bytes(bytes(out))
38
+ print(f"wrote {DST.name}: {len(out)} bytes, {len(lines)} tokens "
39
+ f"({sum(len(t) for t in lines)} token bytes + {len(lines)} length bytes)")
40
+
41
+ # round-trip check
42
+ back = bytearray()
43
+ p = 0
44
+ while p < len(out):
45
+ n = out[p]
46
+ back.extend(out[p + 1:p + 1 + n])
47
+ p += 1 + n
48
+ assert back == b"\n".join(lines) + (b"\n" if lines else b""), "round-trip failed"
49
+ print("round-trip: OK")
50
+
51
+
52
+ if __name__ == "__main__":
53
+ main()