Skip to content

Commit 5fa04e2

Browse files
fix: swap encode/decode to read correct vocab direction
1 parent 83e9f6d commit 5fa04e2

1 file changed

Lines changed: 78 additions & 34 deletions

File tree

‎src/Models/FallbackModel.php‎

Lines changed: 78 additions & 34 deletions
Original file line numberDiff line numberDiff line change
@@ -6,104 +6,148 @@
66

77
use Codewithkyrian\Tokenizers\Contracts\ModelInterface;
88

9+
/**
10+
* A minimal vocabulary-mapping model with no subword algorithm.
11+
*
12+
* Used for tokenizers where tokens map 1:1 to characters or bytes
13+
* without BPE, WordPiece, or Unigram segmentation — typically CTC
14+
* models such as Wav2Vec2.
15+
*
16+
* `tokenize()` is an identity transform; `encode()` and `decode()`
17+
* are simple dictionary lookups.
18+
*/
919
class FallbackModel implements ModelInterface
1020
{
1121
/**
22+
* Maps token strings to integer IDs.
23+
*
24+
* @var array<string, int>
25+
*/
26+
protected array $tokenToId = [];
27+
28+
/**
29+
* Maps integer IDs back to token strings.
30+
*
1231
* @var array<int, string>
1332
*/
14-
protected array $vocab = [];
33+
protected array $idToToken = [];
1534

1635
/**
17-
* @var array<string, int>
36+
* The unknown-token string, used as a fallback when a token
37+
* or ID cannot be found in the vocabulary.
1838
*/
19-
protected array $vocabReversed = [];
2039
protected ?string $unkToken;
2140

2241
/**
23-
* @param array<int, string> $vocab the vocabulary
24-
* @param null|string $unkToken the unknown token
42+
* @param array<string, int> $vocab Token → ID mapping (e.g. `['e' => 5, …]`)
43+
* @param null|string $unkToken Fallback string for unknown tokens / IDs
2544
*/
26-
public function __construct(
27-
array $vocab = [],
28-
?string $unkToken = null
29-
) {
45+
public function __construct(array $vocab = [], ?string $unkToken = null)
46+
{
47+
$this->tokenToId = $vocab;
48+
$this->idToToken = array_flip($this->tokenToId);
3049
$this->unkToken = $unkToken;
31-
32-
// Populate vocab
33-
foreach ($vocab as $token => $id) {
34-
$this->vocab[$token] = $id;
35-
$this->vocabReversed[$id] = $token;
36-
}
3750
}
3851

3952
/**
40-
* @param string[] $messages the messages to tokenize
53+
* Identity transform — returns tokens unchanged.
54+
*
55+
* @param string[] $messages Input token strings
4156
*
42-
* @return string[]
57+
* @return string[] Same token strings, unchanged
4358
*/
4459
public function tokenize(array $messages): array
4560
{
4661
return $messages;
4762
}
4863

4964
/**
50-
* @param string[] $tokens the tokens to encode
65+
* Convert token strings to their integer IDs.
5166
*
52-
* @return int[]
67+
* Unknown tokens resolve to the unk-token's ID if available,
68+
* otherwise to 0.
69+
*
70+
* @param string[] $tokens Token strings to encode
71+
*
72+
* @return int[] Integer token IDs
5373
*/
5474
public function encode(array $tokens): array
5575
{
56-
return array_map(function ($token) {
57-
return $this->vocabReversed[$token] ?? $this->vocabReversed[$this->unkToken] ?? 0;
58-
}, $tokens);
76+
return array_map(
77+
fn (string $token): int => $this->tokenToId[$token]
78+
?? $this->tokenToId[$this->unkToken]
79+
?? 0,
80+
$tokens,
81+
);
5982
}
6083

6184
/**
62-
* @param int[] $ids the IDs to decode
85+
* Convert integer IDs back to their token strings.
86+
*
87+
* Unknown IDs resolve to the unk-token string if available,
88+
* otherwise to an empty string.
6389
*
64-
* @return string[]
90+
* @param int[] $ids Integer token IDs to decode
91+
*
92+
* @return string[] Token strings
6593
*/
6694
public function decode(array $ids): array
6795
{
68-
return array_map(fn ($id) => $this->vocab[$id] ?? $this->unkToken ?? '', $ids);
96+
return array_map(
97+
fn (int $id): string => $this->idToToken[$id]
98+
?? $this->unkToken
99+
?? '',
100+
$ids,
101+
);
69102
}
70103

71104
/**
105+
* Return the full token → ID vocabulary.
106+
*
72107
* @return array<int, string>
73108
*/
74109
public function getVocab(): array
75110
{
76-
return $this->vocab;
111+
return $this->idToToken;
77112
}
78113

114+
/**
115+
* Return the number of tokens in the vocabulary.
116+
*/
79117
public function getVocabSize(): int
80118
{
81-
return \count($this->vocab);
119+
return \count($this->idToToken);
82120
}
83121

84122
/**
85-
* @param string $token the token to add
86-
* @param int $id the ID of the token
123+
* Add a token or override an existing one.
124+
*
125+
* @param string $token The token string
126+
* @param int $id The integer ID to assign
87127
*/
88128
public function addToken(string $token, int $id): void
89129
{
90-
$this->vocab[$id] = $token;
91-
$this->vocabReversed[$token] = $id;
130+
$this->tokenToId[$token] = $id;
131+
$this->idToToken[$id] = $token;
92132
}
93133

134+
/**
135+
* Return configuration, a single config key, or a default value.
136+
*
137+
* @return ($key is null ? array<string, mixed> : mixed)
138+
*/
94139
public function getConfig(?string $key = null, mixed $default = null): mixed
95140
{
96141
if (null !== $key) {
97142
return match ($key) {
98-
'vocab' => $this->vocab,
143+
'vocab' => $this->tokenToId,
99144
'unk_token' => $this->unkToken,
100145
default => $default,
101146
};
102147
}
103148

104-
// 2. Full Config Reconstruction
105149
return [
106-
'vocab' => $this->vocab,
150+
'vocab' => $this->tokenToId,
107151
'unk_token' => $this->unkToken,
108152
];
109153
}

0 commit comments

Comments
 (0)