Skip to content

Commit e7bbac6

Browse files
committed
fix tests
1 parent 59268d4 commit e7bbac6

File tree

1 file changed

+4
-4
lines changed

1 file changed

+4
-4
lines changed

tests/torchtune/modules/tokenizers/test_tiktoken.py

Lines changed: 4 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -38,7 +38,6 @@ def texts(self):
3838
@pytest.fixture
3939
def token_ids(self):
4040
return [
41-
0,
4241
73,
4342
503,
4443
654,
@@ -64,17 +63,18 @@ def token_ids(self):
6463
511,
6564
115,
6665
46,
67-
-1,
6866
]
6967

7068
def test_encode(self, tokenizer, texts, token_ids):
71-
assert tokenizer.encode(texts[0]) == token_ids
69+
assert tokenizer.encode(texts[0], add_bos=True, add_eos=True) == [
70+
0
71+
] + token_ids + [-1]
7272

7373
def test_decode(self, tokenizer, texts, token_ids):
7474
assert tokenizer.decode(token_ids) == texts[0]
7575

7676
def test_encode_and_decode(self, tokenizer, texts):
77-
token_ids = tokenizer.encode(texts[0])
77+
token_ids = tokenizer.encode(texts[0], add_bos=False, add_eos=False)
7878
decoded_text = tokenizer.decode(token_ids)
7979
assert texts[0] == decoded_text
8080

0 commit comments

Comments
 (0)