nltk--nltk
3454a55636
cffconvert / validate (push) Has been cancelled
ci-workflow / pre-commit (push) Has been cancelled
ci-workflow / Minimal NLTK Download Test (macos-latest) (push) Has been cancelled
ci-workflow / Minimal NLTK Download Test (ubuntu-latest) (push) Has been cancelled
ci-workflow / Minimal NLTK Download Test (windows-latest) (push) Has been cancelled
ci-workflow / Python 3.10 on macos-latest (push) Has been cancelled
ci-workflow / Python 3.11 on macos-latest (push) Has been cancelled
ci-workflow / Python 3.12 on macos-latest (push) Has been cancelled
ci-workflow / Python 3.13 on macos-latest (push) Has been cancelled
ci-workflow / Python 3.14 on macos-latest (push) Has been cancelled
ci-workflow / Python 3.14t on macos-latest (push) Has been cancelled
ci-workflow / Python 3.10 on ubuntu-latest (push) Has been cancelled
ci-workflow / Python 3.11 on ubuntu-latest (push) Has been cancelled
ci-workflow / Python 3.12 on ubuntu-latest (push) Has been cancelled
ci-workflow / Python 3.13 on ubuntu-latest (push) Has been cancelled
ci-workflow / Python 3.14 on ubuntu-latest (push) Has been cancelled
ci-workflow / Python 3.14t on ubuntu-latest (push) Has been cancelled
ci-workflow / Python 3.10 on windows-latest (push) Has been cancelled
ci-workflow / Python 3.11 on windows-latest (push) Has been cancelled
ci-workflow / Python 3.12 on windows-latest (push) Has been cancelled
ci-workflow / Python 3.13 on windows-latest (push) Has been cancelled
ci-workflow / Python 3.14 on windows-latest (push) Has been cancelled
31 行
969 B
Python
31 行
969 B
Python
# Natural Language Toolkit: Language Model Unit Tests
|
|
#
|
|
# Copyright (C) 2001-2026 NLTK Project
|
|
# Author: Ilia Kurenkov <ilia.kurenkov@gmail.com>
|
|
# URL: <https://www.nltk.org/>
|
|
# For license information, see LICENSE.TXT
|
|
import unittest
|
|
|
|
from nltk.lm.preprocessing import padded_everygram_pipeline
|
|
|
|
|
|
class TestPreprocessing(unittest.TestCase):
|
|
def test_padded_everygram_pipeline(self):
|
|
expected_train = [
|
|
[
|
|
("<s>",),
|
|
("<s>", "a"),
|
|
("a",),
|
|
("a", "b"),
|
|
("b",),
|
|
("b", "c"),
|
|
("c",),
|
|
("c", "</s>"),
|
|
("</s>",),
|
|
]
|
|
]
|
|
expected_vocab = ["<s>", "a", "b", "c", "</s>"]
|
|
train_data, vocab_data = padded_everygram_pipeline(2, [["a", "b", "c"]])
|
|
self.assertEqual([list(sent) for sent in train_data], expected_train)
|
|
self.assertEqual(list(vocab_data), expected_vocab)
|