diff --git a/.github/workflows/python.yml b/.github/workflows/python.yml index 42554d5d18..a86d24a580 100644 --- a/.github/workflows/python.yml +++ b/.github/workflows/python.yml @@ -129,7 +129,7 @@ jobs: working-directory: ./bindings/python run: | source .env/bin/activate - ty check py_src tests + ty check py_src --exclude py_src/tokenizers/implementations - name: Run tests working-directory: ./bindings/python diff --git a/bindings/python/Cargo.toml b/bindings/python/Cargo.toml index e53badecc2..26ea783247 100644 --- a/bindings/python/Cargo.toml +++ b/bindings/python/Cargo.toml @@ -14,21 +14,24 @@ serde = { version = "1.0", features = ["rc", "derive"] } serde_json = "1.0" libc = "0.2" env_logger = "0.11" -pyo3 = { version = "0.26", features = ["abi3", "abi3-py39", "py-clone"] } -pyo3-async-runtimes = { version = "0.26", features = ["tokio-runtime"] } +pyo3 = { version = "0.27.2",default-features = false, features = ["abi3", "abi3-py39", "py-clone", "experimental-inspect"] } +pyo3-async-runtimes = { version = "0.27.0", features = ["tokio-runtime"] } tokio = { version = "1.47.1", features = ["rt", "rt-multi-thread", "macros", "signal"] } once_cell = "1.19.0" -numpy = "0.26" +numpy = "0.27" ndarray = "0.16" itertools = "0.14" ahash = { version = "0.8.11", features = ["serde"] } +pyo3-ffi = "0.27.2" [dependencies.tokenizers] path = "../../tokenizers" [dev-dependencies] tempfile = "3.10" -pyo3 = { version = "0.26", features = ["auto-initialize"] } +pyo3 = { version = "0.27.2", features = ["auto-initialize"] } + [features] -default = ["pyo3/extension-module"] +default = ["ext-module"] +ext-module = ["pyo3/extension-module"] diff --git a/bindings/python/Makefile b/bindings/python/Makefile index 701b782a52..29ee4db50d 100644 --- a/bindings/python/Makefile +++ b/bindings/python/Makefile @@ -7,17 +7,18 @@ check_dirs := examples py_src/tokenizers tests # Format source code automatically style: + cargo run --manifest-path tools/stub-gen/Cargo.toml python stub.py ruff check $(check_dirs) --fix ruff format $(check_dirs) - ty check py_src tests + ty check py_src --exclude py_src/tokenizers/implementations # Check the source code is formatted correctly check-style: python stub.py --check ruff check $(check_dirs) ruff format --check $(check_dirs) - ty check py_src tests + ty check py_src --exclude py_src/tokenizers/implementations TESTS_RESOURCES = $(DATA_DIR)/small.txt $(DATA_DIR)/roberta.json diff --git a/bindings/python/README.md b/bindings/python/README.md index b04dbdbe6d..c22cac52ba 100644 --- a/bindings/python/README.md +++ b/bindings/python/README.md @@ -169,6 +169,49 @@ tokenizer = Tokenizer.from_file("byte-level-bpe.tokenizer.json") encoded = tokenizer.encode("I can feel the magic, can you?") ``` -### Typing support and `stub.py` +### Typing support and stub generation -The compiled PyO3 extension does not expose type annotations, so editors and type checkers would otherwise see most objects as `Any`. The `stub.py` helper walks the loaded extension modules, renders `.pyi` stub files (plus minimal forwarding `__init__.py` shims), and formats them so that tools like mypy/pyright can understand the public API. Run `python stub.py` whenever you change the Python-visible surface to keep the generated stubs in sync. +The compiled PyO3 extension does not expose type annotations, so editors and type checkers would otherwise see most objects as `Any`. To provide full typing support, we use a two-step stub generation process: + +1. **Rust introspection** (`tools/stub-gen/`): Uses `pyo3-introspection` to analyze the compiled extension and generate `.pyi` stub files +2. **Python enrichment** (`stub.py`): Adds docstrings from the runtime module and generates forwarding `__init__.py` shims + +#### Running stub generation + +The easiest way to regenerate stubs is via `make style`: + +```bash +cd bindings/python +make style +``` + +This will: +1. Build the extension with `maturin develop --release` +2. Run introspection to generate `.pyi` files +3. Enrich stubs with docstrings via `stub.py` +4. Format with `ruff` + +#### Running manually + +To run the stub generator directly: + +```bash +cd bindings/python +cargo run --manifest-path tools/stub-gen/Cargo.toml +python stub.py +``` + +The stub generator automatically: +- Builds the extension using maturin +- Copies the built `.so` to the project root for introspection +- Detects and sets `PYTHONHOME` for embedded Python (handles uv/venv environments) +- Generates stubs to `py_src/tokenizers/` + +#### Troubleshooting + +If you encounter Python initialization errors, you can manually set `PYTHONHOME`: + +```bash +export PYTHONHOME=$(python3 -c 'import sys; print(sys.base_prefix)') +cargo run --manifest-path tools/stub-gen/Cargo.toml +``` diff --git a/bindings/python/py_src/tokenizers/__init__.py b/bindings/python/py_src/tokenizers/__init__.py index d689252a22..efd574298f 100644 --- a/bindings/python/py_src/tokenizers/__init__.py +++ b/bindings/python/py_src/tokenizers/__init__.py @@ -75,7 +75,7 @@ class SplitDelimiterBehavior(Enum): CONTIGUOUS = "contiguous" -from .tokenizers import ( # type: ignore[import] +from .tokenizers import ( AddedToken, Encoding, NormalizedString, diff --git a/bindings/python/py_src/tokenizers/__init__.pyi b/bindings/python/py_src/tokenizers/__init__.pyi index 44f19b8a44..ea9bfde147 100644 --- a/bindings/python/py_src/tokenizers/__init__.pyi +++ b/bindings/python/py_src/tokenizers/__init__.pyi @@ -1,165 +1,95 @@ -# Generated content DO NOT EDIT -class AddedToken: - """ - Represents a token that can be be added to a :class:`~tokenizers.Tokenizer`. - It can have special options that defines the way it should behave. - - Args: - content (:obj:`str`): The content of the token - - single_word (:obj:`bool`, defaults to :obj:`False`): - Defines whether this token should only match single words. If :obj:`True`, this - token will never match inside of a word. For example the token ``ing`` would match - on ``tokenizing`` if this option is :obj:`False`, but not if it is :obj:`True`. - The notion of "`inside of a word`" is defined by the word boundaries pattern in - regular expressions (ie. the token should start and end with word boundaries). - - lstrip (:obj:`bool`, defaults to :obj:`False`): - Defines whether this token should strip all potential whitespaces on its left side. - If :obj:`True`, this token will greedily match any whitespace on its left. For - example if we try to match the token ``[MASK]`` with ``lstrip=True``, in the text - ``"I saw a [MASK]"``, we would match on ``" [MASK]"``. (Note the space on the left). - - rstrip (:obj:`bool`, defaults to :obj:`False`): - Defines whether this token should strip all potential whitespaces on its right - side. If :obj:`True`, this token will greedily match any whitespace on its right. - It works just like :obj:`lstrip` but on the right. - - normalized (:obj:`bool`, defaults to :obj:`True` with :meth:`~tokenizers.Tokenizer.add_tokens` and :obj:`False` with :meth:`~tokenizers.Tokenizer.add_special_tokens`): - Defines whether this token should match against the normalized version of the input - text. For example, with the added token ``"yesterday"``, and a normalizer in charge of - lowercasing the text, the token could be extract from the input ``"I saw a lion - Yesterday"``. - special (:obj:`bool`, defaults to :obj:`False` with :meth:`~tokenizers.Tokenizer.add_tokens` and :obj:`False` with :meth:`~tokenizers.Tokenizer.add_special_tokens`): - Defines whether this token should be skipped when decoding. - - """ - def __init__(self, content=None, single_word=False, lstrip=False, rstrip=False, normalized=True, special=False): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass +import _typeshed +import tokenizers +import tokenizers.decoders +import tokenizers.models +import tokenizers.normalizers +import tokenizers.pre_tokenizers +import tokenizers.processors +import tokenizers.trainers +import typing + +__version__: typing.Final[str] +class AddedToken: + def __eq__(self, /, other: tokenizers.AddedToken) -> bool: + """Return self==value.""" + ... + def __ge__(self, /, other: tokenizers.AddedToken) -> bool: + """Return self>=value.""" + ... + def __getstate__(self, /) -> typing.Any: ... + def __gt__(self, /, other: tokenizers.AddedToken) -> bool: + """Return self>value.""" + ... + def __hash__(self, /) -> int: + """Return hash(self).""" + ... + def __le__(self, /, other: tokenizers.AddedToken) -> bool: + """Return self<=value.""" + ... + def __lt__(self, /, other: tokenizers.AddedToken) -> bool: + """Return self bool: + """Return self!=value.""" + ... + def __new__(cls, /, content: str | None = None, **kwargs) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + def __repr__(self, /) -> str: + """Return repr(self).""" + ... + def __setstate__(self, /, state: typing.Any) -> typing.Any: ... + def __str__(self, /) -> str: + """Return str(self).""" + ... @property - def content(self): - """ - Get the content of this :obj:`AddedToken` - """ - pass - + def content(self, /) -> str: + """Get the content of this :obj:`AddedToken`""" + ... @content.setter - def content(self, value): - """ - Get the content of this :obj:`AddedToken` - """ - pass - + def content(self, /, content: str) -> None: + """Get the content of this :obj:`AddedToken`""" + ... @property - def lstrip(self): - """ - Get the value of the :obj:`lstrip` option - """ - pass - - @lstrip.setter - def lstrip(self, value): - """ - Get the value of the :obj:`lstrip` option - """ - pass - + def lstrip(self, /) -> bool: + """Get the value of the :obj:`lstrip` option""" + ... @property - def normalized(self): - """ - Get the value of the :obj:`normalized` option - """ - pass - - @normalized.setter - def normalized(self, value): - """ - Get the value of the :obj:`normalized` option - """ - pass - + def normalized(self, /) -> bool: + """Get the value of the :obj:`normalized` option""" + ... @property - def rstrip(self): - """ - Get the value of the :obj:`rstrip` option - """ - pass - - @rstrip.setter - def rstrip(self, value): - """ - Get the value of the :obj:`rstrip` option - """ - pass - + def rstrip(self, /) -> bool: + """Get the value of the :obj:`rstrip` option""" + ... @property - def single_word(self): - """ - Get the value of the :obj:`single_word` option - """ - pass - - @single_word.setter - def single_word(self, value): - """ - Get the value of the :obj:`single_word` option - """ - pass - + def single_word(self, /) -> bool: + """Get the value of the :obj:`single_word` option""" + ... @property - def special(self): - """ - Get the value of the :obj:`special` option - """ - pass - + def special(self, /) -> bool: + """Get the value of the :obj:`special` option""" + ... @special.setter - def special(self, value): - """ - Get the value of the :obj:`special` option - """ - pass + def special(self, /, special: bool) -> None: + """Get the value of the :obj:`special` option""" + ... class Encoding: - """ - The :class:`~tokenizers.Encoding` represents the output of a :class:`~tokenizers.Tokenizer`. - """ - def __init__(self): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - + def __getstate__(self, /) -> typing.Any: ... + def __len__(self, /) -> int: + """Return len(self).""" + ... + def __new__(cls, /) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + def __repr__(self, /) -> str: + """Return repr(self).""" + ... + def __setstate__(self, /, state: typing.Any) -> typing.Any: ... @property - def attention_mask(self): - """ - The attention mask - - This indicates to the LM which tokens should be attended to, and which should not. - This is especially important when batching sequences, where we need to applying - padding. - - Returns: - :obj:`List[int]`: The attention mask - """ - pass - - @attention_mask.setter - def attention_mask(self, value): + def attention_mask(self, /) -> typing.Any: """ The attention mask @@ -170,9 +100,8 @@ class Encoding: Returns: :obj:`List[int]`: The attention mask """ - pass - - def char_to_token(self, char_pos, sequence_index=0): + ... + def char_to_token(self, /, char_pos: int, sequence_index: int = 0) -> typing.Any: """ Get the token that contains the char at the given position in the input sequence. @@ -185,9 +114,8 @@ class Encoding: Returns: :obj:`int`: The index of the token that contains this char in the encoded sequence """ - pass - - def char_to_word(self, char_pos, sequence_index=0): + ... + def char_to_word(self, /, char_pos: int, sequence_index: int = 0) -> typing.Any: """ Get the word that contains the char at the given position in the input sequence. @@ -200,23 +128,9 @@ class Encoding: Returns: :obj:`int`: The index of the word that contains this char in the input sequence """ - pass - + ... @property - def ids(self): - """ - The generated IDs - - The IDs are the main input to a Language Model. They are the token indices, - the numerical representations that a LM understands. - - Returns: - :obj:`List[int]`: The list of IDs - """ - pass - - @ids.setter - def ids(self, value): + def ids(self, /) -> typing.Any: """ The generated IDs @@ -226,10 +140,9 @@ class Encoding: Returns: :obj:`List[int]`: The list of IDs """ - pass - + ... @staticmethod - def merge(encodings, growing_offsets=True): + def merge(encodings: typing.Any, growing_offsets: bool = True) -> Encoding: """ Merge the list of encodings into one final :class:`~tokenizers.Encoding` @@ -243,30 +156,18 @@ class Encoding: Returns: :class:`~tokenizers.Encoding`: The resulting Encoding """ - pass - + ... @property - def n_sequences(self): + def n_sequences(self, /) -> int: """ The number of sequences represented Returns: :obj:`int`: The number of sequences in this :class:`~tokenizers.Encoding` """ - pass - - @n_sequences.setter - def n_sequences(self, value): - """ - The number of sequences represented - - Returns: - :obj:`int`: The number of sequences in this :class:`~tokenizers.Encoding` - """ - pass - + ... @property - def offsets(self): + def offsets(self, /) -> typing.Any: """ The offsets associated to each token @@ -276,23 +177,9 @@ class Encoding: Returns: A :obj:`List` of :obj:`Tuple[int, int]`: The list of offsets """ - pass - - @offsets.setter - def offsets(self, value): - """ - The offsets associated to each token - - These offsets let's you slice the input string, and thus retrieve the original - part that led to producing the corresponding token. - - Returns: - A :obj:`List` of :obj:`Tuple[int, int]`: The list of offsets - """ - pass - + ... @property - def overflowing(self): + def overflowing(self, /) -> typing.Any: """ A :obj:`List` of overflowing :class:`~tokenizers.Encoding` @@ -304,24 +191,8 @@ class Encoding: variations to cover all the possible combinations, while respecting the provided maximum length. """ - pass - - @overflowing.setter - def overflowing(self, value): - """ - A :obj:`List` of overflowing :class:`~tokenizers.Encoding` - - When using truncation, the :class:`~tokenizers.Tokenizer` takes care of splitting - the output into as many pieces as required to match the specified maximum length. - This field lets you retrieve all the subsequent pieces. - - When you use pairs of sequences, the overflowing pieces will contain enough - variations to cover all the possible combinations, while respecting the provided - maximum length. - """ - pass - - def pad(self, length, direction="right", pad_id=0, pad_type_id=0, pad_token="[PAD]"): + ... + def pad(self, /, length: int, **kwargs) -> None: """ Pad the :class:`~tokenizers.Encoding` at the given length @@ -341,24 +212,9 @@ class Encoding: pad_token (:obj:`str`, defaults to `[PAD]`): The pad token to use """ - pass - + ... @property - def sequence_ids(self): - """ - The generated sequence indices. - - They represent the index of the input sequence associated to each token. - The sequence id can be None if the token is not related to any input sequence, - like for example with special tokens. - - Returns: - A :obj:`List` of :obj:`Optional[int]`: A list of optional sequence index. - """ - pass - - @sequence_ids.setter - def sequence_ids(self, value): + def sequence_ids(self, /) -> typing.Any: """ The generated sequence indices. @@ -369,19 +225,17 @@ class Encoding: Returns: A :obj:`List` of :obj:`Optional[int]`: A list of optional sequence index. """ - pass - - def set_sequence_id(self, sequence_id): + ... + def set_sequence_id(self, /, sequence_id: int) -> None: """ Set the given sequence index Set the given sequence index for the whole range of tokens contained in this :class:`~tokenizers.Encoding`. """ - pass - + ... @property - def special_tokens_mask(self): + def special_tokens_mask(self, /) -> typing.Any: """ The special token mask @@ -390,21 +244,8 @@ class Encoding: Returns: :obj:`List[int]`: The special tokens mask """ - pass - - @special_tokens_mask.setter - def special_tokens_mask(self, value): - """ - The special token mask - - This indicates which tokens are special tokens, and which are not. - - Returns: - :obj:`List[int]`: The special tokens mask - """ - pass - - def token_to_chars(self, token_index): + ... + def token_to_chars(self, /, token_index: int) -> typing.Any: """ Get the offsets of the token at the given index. @@ -419,9 +260,8 @@ class Encoding: Returns: :obj:`Tuple[int, int]`: The token offsets :obj:`(first, last + 1)` """ - pass - - def token_to_sequence(self, token_index): + ... + def token_to_sequence(self, /, token_index: int) -> typing.Any: """ Get the index of the sequence represented by the given token. @@ -435,9 +275,8 @@ class Encoding: Returns: :obj:`int`: The sequence id of the given token """ - pass - - def token_to_word(self, token_index): + ... + def token_to_word(self, /, token_index: int) -> typing.Any: """ Get the index of the word that contains the token in one of the input sequences. @@ -452,10 +291,9 @@ class Encoding: Returns: :obj:`int`: The index of the word in the relevant input sequence. """ - pass - + ... @property - def tokens(self): + def tokens(self, /) -> typing.Any: """ The generated tokens @@ -464,21 +302,8 @@ class Encoding: Returns: :obj:`List[str]`: The list of tokens """ - pass - - @tokens.setter - def tokens(self, value): - """ - The generated tokens - - They are the string representation of the IDs. - - Returns: - :obj:`List[str]`: The list of tokens - """ - pass - - def truncate(self, max_length, stride=0, direction="right"): + ... + def truncate(self, /, max_length: int, stride: int = 0, direction: str = "right") -> None: """ Truncate the :class:`~tokenizers.Encoding` at the given length @@ -495,10 +320,9 @@ class Encoding: direction (:obj:`str`, defaults to :obj:`right`): Truncate direction """ - pass - + ... @property - def type_ids(self): + def type_ids(self, /) -> typing.Any: """ The generated type IDs @@ -508,23 +332,9 @@ class Encoding: Returns: :obj:`List[int]`: The list of type ids """ - pass - - @type_ids.setter - def type_ids(self, value): - """ - The generated type IDs - - Generally used for tasks like sequence classification or question answering, - these tokens let the LM know which input sequence corresponds to each tokens. - - Returns: - :obj:`List[int]`: The list of type ids - """ - pass - + ... @property - def word_ids(self): + def word_ids(self, /) -> typing.Any: """ The generated word indices. @@ -539,27 +349,8 @@ class Encoding: Returns: A :obj:`List` of :obj:`Optional[int]`: A list of optional word index. """ - pass - - @word_ids.setter - def word_ids(self, value): - """ - The generated word indices. - - They represent the index of the word associated to each token. - When the input is pre-tokenized, they correspond to the ID of the given input label, - otherwise they correspond to the words indices as defined by the - :class:`~tokenizers.pre_tokenizers.PreTokenizer` that was used. - - For special tokens and such (any token that was generated from something that was - not part of the input), the output is :obj:`None` - - Returns: - A :obj:`List` of :obj:`Optional[int]`: A list of optional word index. - """ - pass - - def word_to_chars(self, word_index, sequence_index=0): + ... + def word_to_chars(self, /, word_index: int, sequence_index: int = 0) -> typing.Any: """ Get the offsets of the word at the given index in one of the input sequences. @@ -572,9 +363,8 @@ class Encoding: Returns: :obj:`Tuple[int, int]`: The range of characters (span) :obj:`(first, last + 1)` """ - pass - - def word_to_tokens(self, word_index, sequence_index=0): + ... + def word_to_tokens(self, /, word_index: int, sequence_index: int = 0) -> typing.Any: """ Get the encoded tokens corresponding to the word at the given index in one of the input sequences. @@ -588,10 +378,9 @@ class Encoding: Returns: :obj:`Tuple[int, int]`: The range of tokens: :obj:`(first, last + 1)` """ - pass - + ... @property - def words(self): + def words(self, /) -> typing.Any: """ The generated word indices. @@ -610,157 +399,69 @@ class Encoding: Returns: A :obj:`List` of :obj:`Optional[int]`: A list of optional word index. """ - pass - - @words.setter - def words(self, value): - """ - The generated word indices. - - .. warning:: - This is deprecated and will be removed in a future version. - Please use :obj:`~tokenizers.Encoding.word_ids` instead. - - They represent the index of the word associated to each token. - When the input is pre-tokenized, they correspond to the ID of the given input label, - otherwise they correspond to the words indices as defined by the - :class:`~tokenizers.pre_tokenizers.PreTokenizer` that was used. - - For special tokens and such (any token that was generated from something that was - not part of the input), the output is :obj:`None` - - Returns: - A :obj:`List` of :obj:`Optional[int]`: A list of optional word index. - """ - pass + ... class NormalizedString: - """ - NormalizedString - - A NormalizedString takes care of modifying an "original" string, to obtain a "normalized" one. - While making all the requested modifications, it keeps track of the alignment information - between the two versions of the string. - - Args: - sequence: str: - The string sequence used to initialize this NormalizedString - """ - def __init__(self, sequence): - pass - - def __getitem__(self, key): - """ - Return self[key]. - """ - pass - - def __getstate__(self, /): - """ - Helper for pickle. - """ - pass - - def append(self, s): - """ - Append the given sequence to the string - """ - pass - - def clear(self): - """ - Clears the string - """ - pass - - def filter(self, func): - """ - Filter each character of the string using the given func - """ - pass - - def for_each(self, func): - """ - Calls the given function for each character of the string - """ - pass - - def lowercase(self): - """ - Lowercase the string - """ - pass - - def lstrip(self): - """ - Strip the left of the string - """ - pass - - def map(self, func): + def __getitem__(self, /, range: int | tuple[int, int] | typing.Any) -> typing.Any: + """Return self[key].""" + ... + def __new__(cls, /, sequence: str) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + def __repr__(self, /) -> str: + """Return repr(self).""" + ... + def __str__(self, /) -> str: + """Return str(self).""" + ... + def append(self, /, s: str) -> None: + """Append the given sequence to the string""" + ... + def clear(self, /) -> None: + """Clears the string""" + ... + def filter(self, /, func: typing.Any) -> typing.Any: + """Filter each character of the string using the given func""" + ... + def for_each(self, /, func: typing.Any) -> typing.Any: + """Calls the given function for each character of the string""" + ... + def lowercase(self, /) -> None: + """Lowercase the string""" + ... + def lstrip(self, /) -> None: + """Strip the left of the string""" + ... + def map(self, /, func: typing.Any) -> typing.Any: """ Calls the given function for each character of the string Replaces each character of the string using the returned value. Each returned value **must** be a str of length 1 (ie a character). """ - pass - - def nfc(self): - """ - Runs the NFC normalization - """ - pass - - def nfd(self): - """ - Runs the NFD normalization - """ - pass - - def nfkc(self): - """ - Runs the NFKC normalization - """ - pass - - def nfkd(self): - """ - Runs the NFKD normalization - """ - pass - + ... + def nfc(self, /) -> None: + """Runs the NFC normalization""" + ... + def nfd(self, /) -> None: + """Runs the NFD normalization""" + ... + def nfkc(self, /) -> None: + """Runs the NFKC normalization""" + ... + def nfkd(self, /) -> None: + """Runs the NFKD normalization""" + ... @property - def normalized(self): - """ - The normalized part of the string - """ - pass - - @normalized.setter - def normalized(self, value): - """ - The normalized part of the string - """ - pass - + def normalized(self, /) -> str: + """The normalized part of the string""" + ... @property - def original(self): - """ """ - pass - - @original.setter - def original(self, value): - """ """ - pass - - def prepend(self, s): - """ - Prepend the given sequence to the string - """ - pass - - def replace(self, pattern, content): + def original(self, /) -> str: ... + def prepend(self, /, s: str) -> None: + """Prepend the given sequence to the string""" + ... + def replace(self, /, pattern: str | tokenizers.Regex, content: str) -> typing.Any: """ Replace the content of the given pattern with the provided content @@ -771,21 +472,14 @@ class NormalizedString: content: str: The content to be used as replacement """ - pass - - def rstrip(self): - """ - Strip the right of the string - """ - pass - - def slice(self, range): - """ - Slice the string using the given range - """ - pass - - def split(self, pattern, behavior): + ... + def rstrip(self, /) -> None: + """Strip the right of the string""" + ... + def slice(self, /, range: int | tuple[int, int] | typing.Any) -> typing.Any: + """Slice the string using the given range""" + ... + def split(self, /, pattern: str | tokenizers.Regex, behavior: typing.Any) -> typing.Any: """ Split the NormalizedString using the given pattern and the specified behavior @@ -801,48 +495,19 @@ class NormalizedString: Returns: A list of NormalizedString, representing each split """ - pass - - def strip(self): - """ - Strip both ends of the string - """ - pass - - def uppercase(self): - """ - Uppercase the string - """ - pass + ... + def strip(self, /) -> None: + """Strip both ends of the string""" + ... + def uppercase(self, /) -> None: + """Uppercase the string""" + ... class PreTokenizedString: - """ - PreTokenizedString - - Wrapper over a string, that provides a way to normalize, pre-tokenize, tokenize the - underlying string, while keeping track of the alignment information (offsets). - - The PreTokenizedString manages what we call `splits`. Each split represents a substring - which is a subpart of the original string, with the relevant offsets and tokens. - - When calling one of the methods used to modify the PreTokenizedString (namely one of - `split`, `normalize` or `tokenize), only the `splits` that don't have any associated - tokens will get modified. - - Args: - sequence: str: - The string sequence used to initialize this PreTokenizedString - """ - def __init__(self, sequence): - pass - - def __getstate__(self, /): - """ - Helper for pickle. - """ - pass - - def get_splits(self, offset_referential="original", offset_type="char"): + def __new__(cls, /, s: str) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + def get_splits(self, /, offset_referential: typing.Any = ..., offset_type: typing.Any = ...) -> typing.Any: """ Get the splits currently managed by the PreTokenizedString @@ -861,9 +526,8 @@ class PreTokenizedString: Returns A list of splits """ - pass - - def normalize(self, func): + ... + def normalize(self, /, func: typing.Any) -> typing.Any: """ Normalize each split of the `PreTokenizedString` using the given `func` @@ -873,9 +537,8 @@ class PreTokenizedString: does not need to return anything, just calling the methods on the provided NormalizedString allow its modification. """ - pass - - def split(self, func): + ... + def split(self, /, func: typing.Any) -> typing.Any: """ Split the PreTokenizedString using the given `func` @@ -888,9 +551,8 @@ class PreTokenizedString: In order for the offsets to be tracked accurately, any returned `NormalizedString` should come from calling either `.split` or `.slice` on the received one. """ - pass - - def to_encoding(self, type_id=0, word_idx=None): + ... + def to_encoding(self, /, type_id: int = 0, word_idx: int | None = None) -> Encoding: """ Return an Encoding generated from this PreTokenizedString @@ -906,9 +568,8 @@ class PreTokenizedString: Returns: An Encoding """ - pass - - def tokenize(self, func): + ... + def tokenize(self, /, func: typing.Any) -> typing.Any: """ Tokenize each split of the `PreTokenizedString` using the given `func` @@ -917,91 +578,39 @@ class PreTokenizedString: The function used to tokenize each underlying split. This function must return a list of Token generated from the input str. """ - pass + ... class Regex: - """ - Instantiate a new Regex with the given pattern - """ - def __init__(self, pattern): - pass - - def __getstate__(self, /): - """ - Helper for pickle. - """ - pass + def __new__(cls, /, s: str) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... class Token: - def __init__(self, id, value, offsets): - pass - - def __getstate__(self, /): - """ - Helper for pickle. - """ - pass - - def as_tuple(self): - """ """ - pass - + def __new__(cls, /, id: int, value: str, offsets: typing.Any) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + def as_tuple(self, /) -> typing.Any: ... @property - def id(self): - """ """ - pass - - @id.setter - def id(self, value): - """ """ - pass - + def id(self, /) -> int: ... @property - def offsets(self): - """ """ - pass - - @offsets.setter - def offsets(self, value): - """ """ - pass - + def offsets(self, /) -> typing.Any: ... @property - def value(self): - """ """ - pass - - @value.setter - def value(self, value): - """ """ - pass + def value(self, /) -> str: ... class Tokenizer: - """ - A :obj:`Tokenizer` works as a pipeline. It processes some raw text as input - and outputs an :class:`~tokenizers.Encoding`. - - Args: - model (:class:`~tokenizers.models.Model`): - The core algorithm that this :obj:`Tokenizer` should be using. - - """ - def __init__(self, model): - pass - - def __getnewargs__(self): - """ """ - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - def add_special_tokens(self, tokens): + def __getnewargs__(self, /) -> typing.Any: ... + def __getstate__(self, /) -> typing.Any: ... + def __new__(cls, /, model: tokenizers.models.Model) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + def __repr__(self, /) -> str: + """Return repr(self).""" + ... + def __setstate__(self, /, state: typing.Any) -> typing.Any: ... + def __str__(self, /) -> str: + """Return str(self).""" + ... + def add_special_tokens(self, /, tokens: typing.Any) -> int: """ Add the given special tokens to the Tokenizer. @@ -1020,9 +629,8 @@ class Tokenizer: Returns: :obj:`int`: The number of tokens that were created in the vocabulary """ - pass - - def add_tokens(self, tokens): + ... + def add_tokens(self, /, tokens: typing.Any) -> int: """ Add the given tokens to the vocabulary @@ -1037,9 +645,8 @@ class Tokenizer: Returns: :obj:`int`: The number of tokens that were created in the vocabulary """ - pass - - def async_decode_batch(self, sequences, skip_special_tokens=True): + ... + def async_decode_batch(self, /, sequences: typing.Any, skip_special_tokens: bool = True) -> typing.Any: """ Decode a batch of ids back to their corresponding string @@ -1053,9 +660,15 @@ class Tokenizer: Returns: :obj:`List[str]`: A list of decoded strings """ - pass - - def async_encode(self, sequence, pair=None, is_pretokenized=False, add_special_tokens=True): + ... + def async_encode( + self, + /, + sequence: typing.Any, + pair: typing.Any | None = None, + is_pretokenized: bool = False, + add_special_tokens: bool = True, + ) -> typing.Any: """ Asynchronously encode the given input with character offsets. @@ -1085,11 +698,11 @@ class Tokenizer: Returns: :class:`~tokenizers.Encoding`: The encoded result - """ - pass - - def async_encode_batch(self, input, is_pretokenized=False, add_special_tokens=True): + ... + def async_encode_batch( + self, /, input: typing.Any, is_pretokenized: bool = False, add_special_tokens: bool = True + ) -> typing.Any: """ Asynchronously encode the given batch of inputs with character offsets. @@ -1122,11 +735,11 @@ class Tokenizer: Returns: A :obj:`List` of :class:`~tokenizers.Encoding`: The encoded batch - """ - pass - - def async_encode_batch_fast(self, input, is_pretokenized=False, add_special_tokens=True): + ... + def async_encode_batch_fast( + self, /, input: typing.Any, is_pretokenized: bool = False, add_special_tokens: bool = True + ) -> typing.Any: """ Asynchronously encode the given batch of inputs without tracking character offsets. @@ -1159,11 +772,9 @@ class Tokenizer: Returns: A :obj:`List` of :class:`~tokenizers.Encoding`: The encoded batch - """ - pass - - def decode(self, ids, skip_special_tokens=True): + ... + def decode(self, /, ids: typing.Any, skip_special_tokens: bool = True) -> str: """ Decode the given list of ids back to a string @@ -1179,9 +790,8 @@ class Tokenizer: Returns: :obj:`str`: The decoded string """ - pass - - def decode_batch(self, sequences, skip_special_tokens=True): + ... + def decode_batch(self, /, sequences: typing.Any, skip_special_tokens: bool = True) -> list[str]: """ Decode a batch of ids back to their corresponding string @@ -1195,25 +805,16 @@ class Tokenizer: Returns: :obj:`List[str]`: A list of decoded strings """ - pass - + ... @property - def decoder(self): - """ - The `optional` :class:`~tokenizers.decoders.Decoder` in use by the Tokenizer - """ - pass - + def decoder(self, /) -> typing.Any: + """The `optional` :class:`~tokenizers.decoders.Decoder` in use by the Tokenizer""" + ... @decoder.setter - def decoder(self, value): - """ - The `optional` :class:`~tokenizers.decoders.Decoder` in use by the Tokenizer - """ - pass - - def enable_padding( - self, direction="right", pad_id=0, pad_type_id=0, pad_token="[PAD]", length=None, pad_to_multiple_of=None - ): + def decoder(self, /, decoder: tokenizers.decoders.Decoder | None) -> None: + """The `optional` :class:`~tokenizers.decoders.Decoder` in use by the Tokenizer""" + ... + def enable_padding(self, /, **kwargs) -> None: """ Enable the padding @@ -1239,9 +840,8 @@ class Tokenizer: If specified, the length at which to pad. If not specified we pad using the size of the longest sequence in a batch. """ - pass - - def enable_truncation(self, max_length, stride=0, strategy="longest_first", direction="right"): + ... + def enable_truncation(self, /, max_length: int, **kwargs) -> None: """ Enable truncation @@ -1260,9 +860,15 @@ class Tokenizer: direction (:obj:`str`, defaults to :obj:`right`): Truncate direction """ - pass - - def encode(self, sequence, pair=None, is_pretokenized=False, add_special_tokens=True): + ... + def encode( + self, + /, + sequence: typing.Any, + pair: typing.Any | None = None, + is_pretokenized: bool = False, + add_special_tokens: bool = True, + ) -> Encoding: """ Encode the given sequence and pair. This method can process raw text sequences as well as already pre-tokenized sequences. @@ -1297,11 +903,11 @@ class Tokenizer: Returns: :class:`~tokenizers.Encoding`: The encoded result - """ - pass - - def encode_batch(self, input, is_pretokenized=False, add_special_tokens=True): + ... + def encode_batch( + self, /, input: typing.Any, is_pretokenized: bool = False, add_special_tokens: bool = True + ) -> list[Encoding]: """ Encode the given batch of inputs. This method accept both raw text sequences as well as already pre-tokenized sequences. The reason we use `PySequence` is @@ -1335,11 +941,11 @@ class Tokenizer: Returns: A :obj:`List` of :class:`~tokenizers.Encoding`: The encoded batch - """ - pass - - def encode_batch_fast(self, input, is_pretokenized=False, add_special_tokens=True): + ... + def encode_batch_fast( + self, /, input: typing.Any, is_pretokenized: bool = False, add_special_tokens: bool = True + ) -> list[Encoding]: """ Encode the given batch of inputs. This method is faster than `encode_batch` because it doesn't keep track of offsets, they will be all zeros. @@ -1371,12 +977,10 @@ class Tokenizer: Returns: A :obj:`List` of :class:`~tokenizers.Encoding`: The encoded batch - """ - pass - + ... @property - def encode_special_tokens(self): + def encode_special_tokens(self, /) -> bool: """ Modifies the tokenizer in order to use or not the special tokens during encoding. @@ -1384,12 +988,10 @@ class Tokenizer: Args: value (:obj:`bool`): Whether to use the special tokens or not - """ - pass - + ... @encode_special_tokens.setter - def encode_special_tokens(self, value): + def encode_special_tokens(self, /, value: bool) -> None: """ Modifies the tokenizer in order to use or not the special tokens during encoding. @@ -1397,12 +999,10 @@ class Tokenizer: Args: value (:obj:`bool`): Whether to use the special tokens or not - """ - pass - + ... @staticmethod - def from_buffer(buffer): + def from_buffer(buffer: typing.Any) -> Tokenizer: """ Instantiate a new :class:`~tokenizers.Tokenizer` from the given buffer. @@ -1413,10 +1013,9 @@ class Tokenizer: Returns: :class:`~tokenizers.Tokenizer`: The new tokenizer """ - pass - + ... @staticmethod - def from_file(path): + def from_file(path: str) -> Tokenizer: """ Instantiate a new :class:`~tokenizers.Tokenizer` from the file at the given path. @@ -1428,10 +1027,9 @@ class Tokenizer: Returns: :class:`~tokenizers.Tokenizer`: The new tokenizer """ - pass - + ... @staticmethod - def from_pretrained(identifier, revision="main", token=None): + def from_pretrained(identifier: str, revision: str = ..., token: str | None = None) -> Tokenizer: """ Instantiate a new :class:`~tokenizers.Tokenizer` from an existing file on the Hugging Face Hub. @@ -1449,10 +1047,9 @@ class Tokenizer: Returns: :class:`~tokenizers.Tokenizer`: The new tokenizer """ - pass - + ... @staticmethod - def from_str(json): + def from_str(json: str) -> Tokenizer: """ Instantiate a new :class:`~tokenizers.Tokenizer` from the given JSON string. @@ -1464,18 +1061,16 @@ class Tokenizer: Returns: :class:`~tokenizers.Tokenizer`: The new tokenizer """ - pass - - def get_added_tokens_decoder(self): + ... + def get_added_tokens_decoder(self, /) -> dict[int, AddedToken]: """ Get the underlying vocabulary Returns: :obj:`Dict[int, AddedToken]`: The vocabulary """ - pass - - def get_vocab(self, with_added_tokens=True): + ... + def get_vocab(self, /, with_added_tokens: bool = True) -> dict[str, int]: """ Get the underlying vocabulary @@ -1486,9 +1081,8 @@ class Tokenizer: Returns: :obj:`Dict[str, int]`: The vocabulary """ - pass - - def get_vocab_size(self, with_added_tokens=True): + ... + def get_vocab_size(self, /, with_added_tokens: bool = True) -> int: """ Get the size of the underlying vocabulary @@ -1499,9 +1093,8 @@ class Tokenizer: Returns: :obj:`int`: The size of the vocabulary """ - pass - - def id_to_token(self, id): + ... + def id_to_token(self, /, id: int) -> str | None: """ Convert the given id to its corresponding token if it exists @@ -1512,58 +1105,38 @@ class Tokenizer: Returns: :obj:`Optional[str]`: An optional token, :obj:`None` if out of vocabulary """ - pass - + ... @property - def model(self): - """ - The :class:`~tokenizers.models.Model` in use by the Tokenizer - """ - pass - + def model(self, /) -> typing.Any: + """The :class:`~tokenizers.models.Model` in use by the Tokenizer""" + ... @model.setter - def model(self, value): - """ - The :class:`~tokenizers.models.Model` in use by the Tokenizer - """ - pass - - def no_padding(self): - """ - Disable padding - """ - pass - - def no_truncation(self): - """ - Disable truncation - """ - pass - + def model(self, /, model: tokenizers.models.Model) -> None: + """The :class:`~tokenizers.models.Model` in use by the Tokenizer""" + ... + def no_padding(self, /) -> None: + """Disable padding""" + ... + def no_truncation(self, /) -> None: + """Disable truncation""" + ... @property - def normalizer(self): - """ - The `optional` :class:`~tokenizers.normalizers.Normalizer` in use by the Tokenizer - """ - pass - + def normalizer(self, /) -> typing.Any: + """The `optional` :class:`~tokenizers.normalizers.Normalizer` in use by the Tokenizer""" + ... @normalizer.setter - def normalizer(self, value): - """ - The `optional` :class:`~tokenizers.normalizers.Normalizer` in use by the Tokenizer - """ - pass - - def num_special_tokens_to_add(self, is_pair): + def normalizer(self, /, normalizer: tokenizers.normalizers.Normalizer | None) -> None: + """The `optional` :class:`~tokenizers.normalizers.Normalizer` in use by the Tokenizer""" + ... + def num_special_tokens_to_add(self, /, is_pair: bool) -> int: """ Return the number of special tokens that would be added for single/pair sentences. :param is_pair: Boolean indicating if the input would be a single sentence or a pair :return: """ - pass - + ... @property - def padding(self): + def padding(self, /) -> typing.Any: """ Get the current padding parameters @@ -1573,22 +1146,14 @@ class Tokenizer: (:obj:`dict`, `optional`): A dict with the current padding parameters if padding is enabled """ - pass - - @padding.setter - def padding(self, value): - """ - Get the current padding parameters - - `Cannot be set, use` :meth:`~tokenizers.Tokenizer.enable_padding` `instead` - - Returns: - (:obj:`dict`, `optional`): - A dict with the current padding parameters if padding is enabled - """ - pass - - def post_process(self, encoding, pair=None, add_special_tokens=True): + ... + def post_process( + self, + /, + encoding: tokenizers.Encoding, + pair: tokenizers.Encoding | None = None, + add_special_tokens: bool = True, + ) -> tokenizers.Encoding: """ Apply all the post-processing steps to the given encodings. @@ -1613,37 +1178,24 @@ class Tokenizer: Returns: :class:`~tokenizers.Encoding`: The final post-processed encoding """ - pass - + ... @property - def post_processor(self): - """ - The `optional` :class:`~tokenizers.processors.PostProcessor` in use by the Tokenizer - """ - pass - + def post_processor(self, /) -> typing.Any: + """The `optional` :class:`~tokenizers.processors.PostProcessor` in use by the Tokenizer""" + ... @post_processor.setter - def post_processor(self, value): - """ - The `optional` :class:`~tokenizers.processors.PostProcessor` in use by the Tokenizer - """ - pass - + def post_processor(self, /, processor: tokenizers.processors.PostProcessor | None) -> None: + """The `optional` :class:`~tokenizers.processors.PostProcessor` in use by the Tokenizer""" + ... @property - def pre_tokenizer(self): - """ - The `optional` :class:`~tokenizers.pre_tokenizers.PreTokenizer` in use by the Tokenizer - """ - pass - + def pre_tokenizer(self, /) -> typing.Any: + """The `optional` :class:`~tokenizers.pre_tokenizers.PreTokenizer` in use by the Tokenizer""" + ... @pre_tokenizer.setter - def pre_tokenizer(self, value): - """ - The `optional` :class:`~tokenizers.pre_tokenizers.PreTokenizer` in use by the Tokenizer - """ - pass - - def save(self, path, pretty=True): + def pre_tokenizer(self, /, pretok: tokenizers.pre_tokenizers.PreTokenizer | None) -> None: + """The `optional` :class:`~tokenizers.pre_tokenizers.PreTokenizer` in use by the Tokenizer""" + ... + def save(self, /, path: str, pretty: bool = True) -> None: """ Save the :class:`~tokenizers.Tokenizer` to the file at the given path. @@ -1654,9 +1206,8 @@ class Tokenizer: pretty (:obj:`bool`, defaults to :obj:`True`): Whether the JSON file should be pretty formatted. """ - pass - - def to_str(self, pretty=False): + ... + def to_str(self, /, pretty: bool = False) -> str: """ Gets a serialized string representing this :class:`~tokenizers.Tokenizer`. @@ -1667,9 +1218,8 @@ class Tokenizer: Returns: :obj:`str`: A string representing the serialized Tokenizer """ - pass - - def token_to_id(self, token): + ... + def token_to_id(self, /, token: str) -> int | None: """ Convert the given token to its corresponding id if it exists @@ -1680,9 +1230,8 @@ class Tokenizer: Returns: :obj:`Optional[int]`: An optional id, :obj:`None` if out of vocabulary """ - pass - - def train(self, files, trainer=None): + ... + def train(self, /, files: typing.Any, trainer: tokenizers.trainers.Trainer | None = None) -> typing.Any: """ Train the Tokenizer using the given files. @@ -1697,9 +1246,10 @@ class Tokenizer: trainer (:obj:`~tokenizers.trainers.Trainer`, `optional`): An optional trainer that should be used to train our Model """ - pass - - def train_from_iterator(self, iterator, trainer=None, length=None): + ... + def train_from_iterator( + self, /, iterator: typing.Any, trainer: tokenizers.trainers.Trainer | None = None, length: int | None = None + ) -> typing.Any: """ Train the Tokenizer using the provided iterator. @@ -1721,10 +1271,9 @@ class Tokenizer: The total number of sequences in the iterator. This is used to provide meaningful progress tracking """ - pass - + ... @property - def truncation(self): + def truncation(self, /) -> typing.Any: """ Get the currently set truncation parameters @@ -1734,67 +1283,6 @@ class Tokenizer: (:obj:`dict`, `optional`): A dict with the current truncation parameters if truncation is enabled """ - pass - - @truncation.setter - def truncation(self, value): - """ - Get the currently set truncation parameters + ... - `Cannot set, use` :meth:`~tokenizers.Tokenizer.enable_truncation` `instead` - - Returns: - (:obj:`dict`, `optional`): - A dict with the current truncation parameters if truncation is enabled - """ - pass - -from enum import Enum -from typing import List, Tuple, Union, Any - -Offsets = Tuple[int, int] -TextInputSequence = str -PreTokenizedInputSequence = Union[List[str], Tuple[str, ...]] -TextEncodeInput = Union[ - TextInputSequence, - Tuple[TextInputSequence, TextInputSequence], - List[TextInputSequence], -] -PreTokenizedEncodeInput = Union[ - PreTokenizedInputSequence, - Tuple[PreTokenizedInputSequence, PreTokenizedInputSequence], - List[PreTokenizedInputSequence], -] -InputSequence = Union[TextInputSequence, PreTokenizedInputSequence] -EncodeInput = Union[TextEncodeInput, PreTokenizedEncodeInput] - -class OffsetReferential(Enum): - ORIGINAL = "original" - NORMALIZED = "normalized" - -class OffsetType(Enum): - BYTE = "byte" - CHAR = "char" - -class SplitDelimiterBehavior(Enum): - REMOVED = "removed" - ISOLATED = "isolated" - MERGED_WITH_PREVIOUS = "merged_with_previous" - MERGED_WITH_NEXT = "merged_with_next" - CONTIGUOUS = "contiguous" - -from .implementations import ( - BertWordPieceTokenizer, - ByteLevelBPETokenizer, - CharBPETokenizer, - SentencePieceBPETokenizer, - SentencePieceUnigramTokenizer, -) - -def __getattr__(name: str) -> Any: ... - -BertWordPieceTokenizer: Any -ByteLevelBPETokenizer: Any -CharBPETokenizer: Any -SentencePieceBPETokenizer: Any -SentencePieceUnigramTokenizer: Any +def __getattr__(name: str) -> _typeshed.Incomplete: ... diff --git a/bindings/python/py_src/tokenizers/decoders.pyi b/bindings/python/py_src/tokenizers/decoders.pyi new file mode 100644 index 0000000000..8c6377256e --- /dev/null +++ b/bindings/python/py_src/tokenizers/decoders.pyi @@ -0,0 +1,147 @@ +import tokenizers +import tokenizers.decoders +import typing + +class BPEDecoder: + def __new__(cls, /, suffix: str = ...) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def suffix(self, /) -> str: ... + @suffix.setter + def suffix(self, /, suffix: str) -> None: ... + +class ByteFallback: + def __new__(cls, /) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + +class ByteLevel: + def __new__(cls, /, **_kwargs) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + +class CTC: + def __new__(cls, /, pad_token: str = ..., word_delimiter_token: str = ..., cleanup: bool = True) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def cleanup(self, /) -> bool: ... + @cleanup.setter + def cleanup(self, /, cleanup: bool) -> None: ... + @property + def pad_token(self, /) -> str: ... + @pad_token.setter + def pad_token(self, /, pad_token: str) -> None: ... + @property + def word_delimiter_token(self, /) -> str: ... + @word_delimiter_token.setter + def word_delimiter_token(self, /, word_delimiter_token: str) -> None: ... + +class DecodeStream: + def __new__(cls, /, ids: typing.Any | None = None, skip_special_tokens: bool | None = False) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + def step(self, /, tokenizer: tokenizers.Tokenizer, id: typing.Any) -> typing.Any: + """ + Streaming decode step + + Args: + tokenizer (:class:`~tokenizers.Tokenizer`): + The tokenizer to use for decoding + id (:obj:`int` or `List[int]`): + The next token id or list of token ids to add to the stream + + + Returns: + :obj:`Optional[str]`: The next decoded string chunk, or None if not enough + tokens have been provided yet. + """ + ... + +class Decoder: + def __getstate__(self, /) -> typing.Any: ... + def __repr__(self, /) -> str: + """Return repr(self).""" + ... + def __setstate__(self, /, state: typing.Any) -> typing.Any: ... + def __str__(self, /) -> str: + """Return str(self).""" + ... + @staticmethod + def custom(decoder: typing.Any) -> tokenizers.decoders.Decoder: ... + def decode(self, /, tokens: typing.Any) -> str: + """ + Decode the given list of tokens to a final string + + Args: + tokens (:obj:`List[str]`): + The list of tokens to decode + + Returns: + :obj:`str`: The decoded string + """ + ... + +class Fuse: + def __new__(cls, /) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + +class Metaspace: + def __new__(cls, /, replacement: str = "▁", prepend_scheme: str = ..., split: bool = True) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def prepend_scheme(self, /) -> str: ... + @prepend_scheme.setter + def prepend_scheme(self, /, prepend_scheme: str) -> typing.Any: ... + @property + def replacement(self, /) -> str: ... + @replacement.setter + def replacement(self, /, replacement: str) -> None: ... + @property + def split(self, /) -> bool: ... + @split.setter + def split(self, /, split: bool) -> None: ... + +class Replace: + def __new__(cls, /, pattern: str | tokenizers.Regex, content: str) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + +class Sequence: + def __getnewargs__(self, /) -> typing.Any: ... + def __new__(cls, /, decoders_py: typing.Any) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + +class Strip: + def __new__(cls, /, content: str = " ", left: int = 0, right: int = 0) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def content(self, /) -> str: ... + @content.setter + def content(self, /, content: str) -> None: ... + @property + def start(self, /) -> int: ... + @start.setter + def start(self, /, start: int) -> None: ... + @property + def stop(self, /) -> int: ... + @stop.setter + def stop(self, /, stop: int) -> None: ... + +class WordPiece: + def __new__(cls, /, prefix: str = ..., cleanup: bool = True) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def cleanup(self, /) -> bool: ... + @cleanup.setter + def cleanup(self, /, cleanup: bool) -> None: ... + @property + def prefix(self, /) -> str: ... + @prefix.setter + def prefix(self, /, prefix: str) -> None: ... diff --git a/bindings/python/py_src/tokenizers/decoders/__init__.pyi b/bindings/python/py_src/tokenizers/decoders/__init__.pyi deleted file mode 100644 index 29fb501c99..0000000000 --- a/bindings/python/py_src/tokenizers/decoders/__init__.pyi +++ /dev/null @@ -1,569 +0,0 @@ -# Generated content DO NOT EDIT -class DecodeStream: - """ - Class needed for streaming decode - - """ - def __init__(self, ids=None, skip_special_tokens=False): - pass - - def __getstate__(self, /): - """ - Helper for pickle. - """ - pass - - def step(self, tokenizer, id): - """ - Streaming decode step - - Args: - tokenizer (:class:`~tokenizers.Tokenizer`): - The tokenizer to use for decoding - id (:obj:`int` or `List[int]`): - The next token id or list of token ids to add to the stream - - - Returns: - :obj:`Optional[str]`: The next decoded string chunk, or None if not enough - tokens have been provided yet. - """ - pass - -class Decoder: - """ - Base class for all decoders - - This class is not supposed to be instantiated directly. Instead, any implementation of - a Decoder will return an instance of this class when instantiated. - """ - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(decoder): - """ """ - pass - - def decode(self, tokens): - """ - Decode the given list of tokens to a final string - - Args: - tokens (:obj:`List[str]`): - The list of tokens to decode - - Returns: - :obj:`str`: The decoded string - """ - pass - -class BPEDecoder(Decoder): - """ - BPEDecoder Decoder - - Args: - suffix (:obj:`str`, `optional`, defaults to :obj:``): - The suffix that was used to characterize an end-of-word. This suffix will - be replaced by whitespaces during the decoding - """ - def __init__(self, suffix=""): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(decoder): - """ """ - pass - - def decode(self, tokens): - """ - Decode the given list of tokens to a final string - - Args: - tokens (:obj:`List[str]`): - The list of tokens to decode - - Returns: - :obj:`str`: The decoded string - """ - pass - - @property - def suffix(self): - """ """ - pass - - @suffix.setter - def suffix(self, value): - """ """ - pass - -class ByteFallback(Decoder): - """ - ByteFallback Decoder - ByteFallback is a simple trick which converts tokens looking like `<0x61>` - to pure bytes, and attempts to make them into a string. If the tokens - cannot be decoded you will get � instead for each inconvertible byte token - - """ - def __init__(self): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(decoder): - """ """ - pass - - def decode(self, tokens): - """ - Decode the given list of tokens to a final string - - Args: - tokens (:obj:`List[str]`): - The list of tokens to decode - - Returns: - :obj:`str`: The decoded string - """ - pass - -class ByteLevel(Decoder): - """ - ByteLevel Decoder - - This decoder is to be used in tandem with the :class:`~tokenizers.pre_tokenizers.ByteLevel` - :class:`~tokenizers.pre_tokenizers.PreTokenizer`. - """ - def __init__(self): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(decoder): - """ """ - pass - - def decode(self, tokens): - """ - Decode the given list of tokens to a final string - - Args: - tokens (:obj:`List[str]`): - The list of tokens to decode - - Returns: - :obj:`str`: The decoded string - """ - pass - -class CTC(Decoder): - """ - CTC Decoder - - Args: - pad_token (:obj:`str`, `optional`, defaults to :obj:``): - The pad token used by CTC to delimit a new token. - word_delimiter_token (:obj:`str`, `optional`, defaults to :obj:`|`): - The word delimiter token. It will be replaced by a - cleanup (:obj:`bool`, `optional`, defaults to :obj:`True`): - Whether to cleanup some tokenization artifacts. - Mainly spaces before punctuation, and some abbreviated english forms. - """ - def __init__(self, pad_token="", word_delimiter_token="|", cleanup=True): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @property - def cleanup(self): - """ """ - pass - - @cleanup.setter - def cleanup(self, value): - """ """ - pass - - @staticmethod - def custom(decoder): - """ """ - pass - - def decode(self, tokens): - """ - Decode the given list of tokens to a final string - - Args: - tokens (:obj:`List[str]`): - The list of tokens to decode - - Returns: - :obj:`str`: The decoded string - """ - pass - - @property - def pad_token(self): - """ """ - pass - - @pad_token.setter - def pad_token(self, value): - """ """ - pass - - @property - def word_delimiter_token(self): - """ """ - pass - - @word_delimiter_token.setter - def word_delimiter_token(self, value): - """ """ - pass - -class Fuse(Decoder): - """ - Fuse Decoder - Fuse simply fuses every token into a single string. - This is the last step of decoding, this decoder exists only if - there is need to add other decoders *after* the fusion - """ - def __init__(self): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(decoder): - """ """ - pass - - def decode(self, tokens): - """ - Decode the given list of tokens to a final string - - Args: - tokens (:obj:`List[str]`): - The list of tokens to decode - - Returns: - :obj:`str`: The decoded string - """ - pass - -class Metaspace(Decoder): - """ - Metaspace Decoder - - Args: - replacement (:obj:`str`, `optional`, defaults to :obj:`▁`): - The replacement character. Must be exactly one character. By default we - use the `▁` (U+2581) meta symbol (Same as in SentencePiece). - - prepend_scheme (:obj:`str`, `optional`, defaults to :obj:`"always"`): - Whether to add a space to the first word if there isn't already one. This - lets us treat `hello` exactly like `say hello`. - Choices: "always", "never", "first". First means the space is only added on the first - token (relevant when special tokens are used or other pre_tokenizer are used). - """ - def __init__(self, replacement="▁", prepend_scheme="always", split=True): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(decoder): - """ """ - pass - - def decode(self, tokens): - """ - Decode the given list of tokens to a final string - - Args: - tokens (:obj:`List[str]`): - The list of tokens to decode - - Returns: - :obj:`str`: The decoded string - """ - pass - - @property - def prepend_scheme(self): - """ """ - pass - - @prepend_scheme.setter - def prepend_scheme(self, value): - """ """ - pass - - @property - def replacement(self): - """ """ - pass - - @replacement.setter - def replacement(self, value): - """ """ - pass - - @property - def split(self): - """ """ - pass - - @split.setter - def split(self, value): - """ """ - pass - -class Replace(Decoder): - """ - Replace Decoder - - This decoder is to be used in tandem with the :class:`~tokenizers.pre_tokenizers.Replace` - :class:`~tokenizers.pre_tokenizers.PreTokenizer`. - """ - def __init__(self, pattern, content): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(decoder): - """ """ - pass - - def decode(self, tokens): - """ - Decode the given list of tokens to a final string - - Args: - tokens (:obj:`List[str]`): - The list of tokens to decode - - Returns: - :obj:`str`: The decoded string - """ - pass - -class Sequence(Decoder): - """ - Sequence Decoder - - Args: - decoders (:obj:`List[Decoder]`) - The decoders that need to be chained - """ - def __init__(self, decoders): - pass - - def __getnewargs__(self): - """ """ - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(decoder): - """ """ - pass - - def decode(self, tokens): - """ - Decode the given list of tokens to a final string - - Args: - tokens (:obj:`List[str]`): - The list of tokens to decode - - Returns: - :obj:`str`: The decoded string - """ - pass - -class Strip(Decoder): - """ - Strip normalizer - Strips n left characters of each token, or n right characters of each token - """ - def __init__(self, content=" ", left=0, right=0): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @property - def content(self): - """ """ - pass - - @content.setter - def content(self, value): - """ """ - pass - - @staticmethod - def custom(decoder): - """ """ - pass - - def decode(self, tokens): - """ - Decode the given list of tokens to a final string - - Args: - tokens (:obj:`List[str]`): - The list of tokens to decode - - Returns: - :obj:`str`: The decoded string - """ - pass - - @property - def start(self): - """ """ - pass - - @start.setter - def start(self, value): - """ """ - pass - - @property - def stop(self): - """ """ - pass - - @stop.setter - def stop(self, value): - """ """ - pass - -class WordPiece(Decoder): - """ - WordPiece Decoder - - Args: - prefix (:obj:`str`, `optional`, defaults to :obj:`##`): - The prefix to use for subwords that are not a beginning-of-word - - cleanup (:obj:`bool`, `optional`, defaults to :obj:`True`): - Whether to cleanup some tokenization artifacts. Mainly spaces before punctuation, - and some abbreviated english forms. - """ - def __init__(self, prefix="##", cleanup=True): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @property - def cleanup(self): - """ """ - pass - - @cleanup.setter - def cleanup(self, value): - """ """ - pass - - @staticmethod - def custom(decoder): - """ """ - pass - - def decode(self, tokens): - """ - Decode the given list of tokens to a final string - - Args: - tokens (:obj:`List[str]`): - The list of tokens to decode - - Returns: - :obj:`str`: The decoded string - """ - pass - - @property - def prefix(self): - """ """ - pass - - @prefix.setter - def prefix(self, value): - """ """ - pass diff --git a/bindings/python/py_src/tokenizers/models.pyi b/bindings/python/py_src/tokenizers/models.pyi new file mode 100644 index 0000000000..21285853be --- /dev/null +++ b/bindings/python/py_src/tokenizers/models.pyi @@ -0,0 +1,285 @@ +import typing + +class BPE: + def __new__( + cls, /, vocab: typing.Any | str | None = None, merges: typing.Any | str | None = None, **kwargs + ) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + def _clear_cache(self, /) -> None: + """Clears the internal cache""" + ... + def _resize_cache(self, /, capacity: int) -> None: + """Resize the internal cache""" + ... + @property + def byte_fallback(self, /) -> bool: ... + @byte_fallback.setter + def byte_fallback(self, /, byte_fallback: bool) -> None: ... + @property + def continuing_subword_prefix(self, /) -> typing.Any: ... + @continuing_subword_prefix.setter + def continuing_subword_prefix(self, /, continuing_subword_prefix: str | None) -> None: ... + @property + def dropout(self, /) -> typing.Any: ... + @dropout.setter + def dropout(self, /, dropout: float | None) -> None: ... + @property + def end_of_word_suffix(self, /) -> typing.Any: ... + @end_of_word_suffix.setter + def end_of_word_suffix(self, /, end_of_word_suffix: str | None) -> None: ... + @classmethod + def from_file(cls, /, vocab: str, merges: str, **kwargs) -> BPE: + """ + Instantiate a BPE model from the given files. + + This method is roughly equivalent to doing:: + + vocab, merges = BPE.read_file(vocab_filename, merges_filename) + bpe = BPE(vocab, merges) + + If you don't need to keep the :obj:`vocab, merges` values lying around, + this method is more optimized than manually calling + :meth:`~tokenizers.models.BPE.read_file` to initialize a :class:`~tokenizers.models.BPE` + + Args: + vocab (:obj:`str`): + The path to a :obj:`vocab.json` file + + merges (:obj:`str`): + The path to a :obj:`merges.txt` file + + Returns: + :class:`~tokenizers.models.BPE`: An instance of BPE loaded from these files + """ + ... + @property + def fuse_unk(self, /) -> bool: ... + @fuse_unk.setter + def fuse_unk(self, /, fuse_unk: bool) -> None: ... + @property + def ignore_merges(self, /) -> bool: ... + @ignore_merges.setter + def ignore_merges(self, /, ignore_merges: bool) -> None: ... + @staticmethod + def read_file(vocab: str, merges: str) -> typing.Any: + """ + Read a :obj:`vocab.json` and a :obj:`merges.txt` files + + This method provides a way to read and parse the content of these files, + returning the relevant data structures. If you want to instantiate some BPE models + from memory, this method gives you the expected input from the standard files. + + Args: + vocab (:obj:`str`): + The path to a :obj:`vocab.json` file + + merges (:obj:`str`): + The path to a :obj:`merges.txt` file + + Returns: + A :obj:`Tuple` with the vocab and the merges: + The vocabulary and merges loaded into memory + """ + ... + @property + def unk_token(self, /) -> typing.Any: ... + @unk_token.setter + def unk_token(self, /, unk_token: str | None) -> None: ... + +class Model: + def __getstate__(self, /) -> typing.Any: ... + def __new__(cls, /) -> Model: + """Create and return a new object. See help(type) for accurate signature.""" + ... + def __repr__(self, /) -> str: + """Return repr(self).""" + ... + def __setstate__(self, /, state: typing.Any) -> typing.Any: ... + def __str__(self, /) -> str: + """Return str(self).""" + ... + def get_trainer(self, /) -> typing.Any: + """ + Get the associated :class:`~tokenizers.trainers.Trainer` + + Retrieve the :class:`~tokenizers.trainers.Trainer` associated to this + :class:`~tokenizers.models.Model`. + + Returns: + :class:`~tokenizers.trainers.Trainer`: The Trainer used to train this model + """ + ... + def id_to_token(self, /, id: int) -> typing.Any: + """ + Get the token associated to an ID + + Args: + id (:obj:`int`): + An ID to convert to a token + + Returns: + :obj:`str`: The token associated to the ID + """ + ... + def save(self, /, folder: str, prefix: str | None = None, name: str | None = None) -> list[str]: + """ + Save the current model + + Save the current model in the given folder, using the given prefix for the various + files that will get created. + Any file with the same name that already exists in this folder will be overwritten. + + Args: + folder (:obj:`str`): + The path to the target folder in which to save the various files + + prefix (:obj:`str`, `optional`): + An optional prefix, used to prefix each file name + + Returns: + :obj:`List[str]`: The list of saved files + """ + ... + def token_to_id(self, /, token: str) -> typing.Any: + """ + Get the ID associated to a token + + Args: + token (:obj:`str`): + A token to convert to an ID + + Returns: + :obj:`int`: The ID associated to the token + """ + ... + def tokenize(self, /, sequence: str) -> typing.Any: + """ + Tokenize a sequence + + Args: + sequence (:obj:`str`): + A sequence to tokenize + + Returns: + A :obj:`List` of :class:`~tokenizers.Token`: The generated tokens + """ + ... + +class Unigram: + def __new__( + cls, /, vocab: typing.Any | None = None, unk_id: int | None = None, byte_fallback: bool | None = None + ) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + def _clear_cache(self, /) -> None: + """Clears the internal cache""" + ... + def _resize_cache(self, /, capacity: int) -> None: + """Resize the internal cache""" + ... + +class WordLevel: + def __new__(cls, /, vocab: typing.Any | str | None = None, unk_token: str | None = None) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @classmethod + def from_file(cls, /, vocab: str, unk_token: str | None = None) -> WordLevel: + """ + Instantiate a WordLevel model from the given file + + This method is roughly equivalent to doing:: + + vocab = WordLevel.read_file(vocab_filename) + wordlevel = WordLevel(vocab) + + If you don't need to keep the :obj:`vocab` values lying around, this method is + more optimized than manually calling :meth:`~tokenizers.models.WordLevel.read_file` to + initialize a :class:`~tokenizers.models.WordLevel` + + Args: + vocab (:obj:`str`): + The path to a :obj:`vocab.json` file + + Returns: + :class:`~tokenizers.models.WordLevel`: An instance of WordLevel loaded from file + """ + ... + @staticmethod + def read_file(vocab: str) -> typing.Any: + """ + Read a :obj:`vocab.json` + + This method provides a way to read and parse the content of a vocabulary file, + returning the relevant data structures. If you want to instantiate some WordLevel models + from memory, this method gives you the expected input from the standard files. + + Args: + vocab (:obj:`str`): + The path to a :obj:`vocab.json` file + + Returns: + :obj:`Dict[str, int]`: The vocabulary as a :obj:`dict` + """ + ... + @property + def unk_token(self, /) -> str: ... + @unk_token.setter + def unk_token(self, /, unk_token: str) -> None: ... + +class WordPiece: + def __new__(cls, /, vocab: typing.Any | str | None = None, **kwargs) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def continuing_subword_prefix(self, /) -> str: ... + @continuing_subword_prefix.setter + def continuing_subword_prefix(self, /, continuing_subword_prefix: str) -> None: ... + @classmethod + def from_file(cls, /, vocab: str, **kwargs) -> WordPiece: + """ + Instantiate a WordPiece model from the given file + + This method is roughly equivalent to doing:: + + vocab = WordPiece.read_file(vocab_filename) + wordpiece = WordPiece(vocab) + + If you don't need to keep the :obj:`vocab` values lying around, this method is + more optimized than manually calling :meth:`~tokenizers.models.WordPiece.read_file` to + initialize a :class:`~tokenizers.models.WordPiece` + + Args: + vocab (:obj:`str`): + The path to a :obj:`vocab.txt` file + + Returns: + :class:`~tokenizers.models.WordPiece`: An instance of WordPiece loaded from file + """ + ... + @property + def max_input_chars_per_word(self, /) -> int: ... + @max_input_chars_per_word.setter + def max_input_chars_per_word(self, /, max: int) -> None: ... + @staticmethod + def read_file(vocab: str) -> typing.Any: + """ + Read a :obj:`vocab.txt` file + + This method provides a way to read and parse the content of a standard `vocab.txt` + file as used by the WordPiece Model, returning the relevant data structures. If you + want to instantiate some WordPiece models from memory, this method gives you the + expected input from the standard files. + + Args: + vocab (:obj:`str`): + The path to a :obj:`vocab.txt` file + + Returns: + :obj:`Dict[str, int]`: The vocabulary as a :obj:`dict` + """ + ... + @property + def unk_token(self, /) -> str: ... + @unk_token.setter + def unk_token(self, /, unk_token: str) -> None: ... diff --git a/bindings/python/py_src/tokenizers/models/__init__.py b/bindings/python/py_src/tokenizers/models/__init__.py index 68ac211aa8..5adfc8e25c 100644 --- a/bindings/python/py_src/tokenizers/models/__init__.py +++ b/bindings/python/py_src/tokenizers/models/__init__.py @@ -1,8 +1,9 @@ # Generated content DO NOT EDIT + from .. import models -Model = models.Model BPE = models.BPE +Model = models.Model Unigram = models.Unigram WordLevel = models.WordLevel WordPiece = models.WordPiece diff --git a/bindings/python/py_src/tokenizers/models/__init__.pyi b/bindings/python/py_src/tokenizers/models/__init__.pyi deleted file mode 100644 index 2548697410..0000000000 --- a/bindings/python/py_src/tokenizers/models/__init__.pyi +++ /dev/null @@ -1,744 +0,0 @@ -# Generated content DO NOT EDIT -class Model: - """ - Base class for all models - - The model represents the actual tokenization algorithm. This is the part that - will contain and manage the learned vocabulary. - - This class cannot be constructed directly. Please use one of the concrete models. - """ - def __init__(self): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - def get_trainer(self): - """ - Get the associated :class:`~tokenizers.trainers.Trainer` - - Retrieve the :class:`~tokenizers.trainers.Trainer` associated to this - :class:`~tokenizers.models.Model`. - - Returns: - :class:`~tokenizers.trainers.Trainer`: The Trainer used to train this model - """ - pass - - def id_to_token(self, id): - """ - Get the token associated to an ID - - Args: - id (:obj:`int`): - An ID to convert to a token - - Returns: - :obj:`str`: The token associated to the ID - """ - pass - - def save(self, folder, prefix): - """ - Save the current model - - Save the current model in the given folder, using the given prefix for the various - files that will get created. - Any file with the same name that already exists in this folder will be overwritten. - - Args: - folder (:obj:`str`): - The path to the target folder in which to save the various files - - prefix (:obj:`str`, `optional`): - An optional prefix, used to prefix each file name - - Returns: - :obj:`List[str]`: The list of saved files - """ - pass - - def token_to_id(self, tokens): - """ - Get the ID associated to a token - - Args: - token (:obj:`str`): - A token to convert to an ID - - Returns: - :obj:`int`: The ID associated to the token - """ - pass - - def tokenize(self, sequence): - """ - Tokenize a sequence - - Args: - sequence (:obj:`str`): - A sequence to tokenize - - Returns: - A :obj:`List` of :class:`~tokenizers.Token`: The generated tokens - """ - pass - -class BPE(Model): - """ - An implementation of the BPE (Byte-Pair Encoding) algorithm - - Args: - vocab (:obj:`Dict[str, int]`, `optional`): - A dictionary of string keys and their ids :obj:`{"am": 0,...}` - - merges (:obj:`List[Tuple[str, str]]`, `optional`): - A list of pairs of tokens (:obj:`Tuple[str, str]`) :obj:`[("a", "b"),...]` - - cache_capacity (:obj:`int`, `optional`): - The number of words that the BPE cache can contain. The cache allows - to speed-up the process by keeping the result of the merge operations - for a number of words. - - dropout (:obj:`float`, `optional`): - A float between 0 and 1 that represents the BPE dropout to use. - - unk_token (:obj:`str`, `optional`): - The unknown token to be used by the model. - - continuing_subword_prefix (:obj:`str`, `optional`): - The prefix to attach to subword units that don't represent a beginning of word. - - end_of_word_suffix (:obj:`str`, `optional`): - The suffix to attach to subword units that represent an end of word. - - fuse_unk (:obj:`bool`, `optional`): - Whether to fuse any subsequent unknown tokens into a single one - - byte_fallback (:obj:`bool`, `optional`): - Whether to use spm byte-fallback trick (defaults to False) - - ignore_merges (:obj:`bool`, `optional`): - Whether or not to match tokens with the vocab before using merges. - """ - def __init__( - self, - vocab=None, - merges=None, - cache_capacity=None, - dropout=None, - unk_token=None, - continuing_subword_prefix=None, - end_of_word_suffix=None, - fuse_unk=None, - byte_fallback=False, - ignore_merges=False, - ): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @property - def byte_fallback(self): - """ """ - pass - - @byte_fallback.setter - def byte_fallback(self, value): - """ """ - pass - - @property - def continuing_subword_prefix(self): - """ """ - pass - - @continuing_subword_prefix.setter - def continuing_subword_prefix(self, value): - """ """ - pass - - @property - def dropout(self): - """ """ - pass - - @dropout.setter - def dropout(self, value): - """ """ - pass - - @property - def end_of_word_suffix(self): - """ """ - pass - - @end_of_word_suffix.setter - def end_of_word_suffix(self, value): - """ """ - pass - - @staticmethod - def from_file(vocab, merges, **kwargs): - """ - Instantiate a BPE model from the given files. - - This method is roughly equivalent to doing:: - - vocab, merges = BPE.read_file(vocab_filename, merges_filename) - bpe = BPE(vocab, merges) - - If you don't need to keep the :obj:`vocab, merges` values lying around, - this method is more optimized than manually calling - :meth:`~tokenizers.models.BPE.read_file` to initialize a :class:`~tokenizers.models.BPE` - - Args: - vocab (:obj:`str`): - The path to a :obj:`vocab.json` file - - merges (:obj:`str`): - The path to a :obj:`merges.txt` file - - Returns: - :class:`~tokenizers.models.BPE`: An instance of BPE loaded from these files - """ - pass - - @property - def fuse_unk(self): - """ """ - pass - - @fuse_unk.setter - def fuse_unk(self, value): - """ """ - pass - - def get_trainer(self): - """ - Get the associated :class:`~tokenizers.trainers.Trainer` - - Retrieve the :class:`~tokenizers.trainers.Trainer` associated to this - :class:`~tokenizers.models.Model`. - - Returns: - :class:`~tokenizers.trainers.Trainer`: The Trainer used to train this model - """ - pass - - def id_to_token(self, id): - """ - Get the token associated to an ID - - Args: - id (:obj:`int`): - An ID to convert to a token - - Returns: - :obj:`str`: The token associated to the ID - """ - pass - - @property - def ignore_merges(self): - """ """ - pass - - @ignore_merges.setter - def ignore_merges(self, value): - """ """ - pass - - @staticmethod - def read_file(vocab, merges): - """ - Read a :obj:`vocab.json` and a :obj:`merges.txt` files - - This method provides a way to read and parse the content of these files, - returning the relevant data structures. If you want to instantiate some BPE models - from memory, this method gives you the expected input from the standard files. - - Args: - vocab (:obj:`str`): - The path to a :obj:`vocab.json` file - - merges (:obj:`str`): - The path to a :obj:`merges.txt` file - - Returns: - A :obj:`Tuple` with the vocab and the merges: - The vocabulary and merges loaded into memory - """ - pass - - def save(self, folder, prefix): - """ - Save the current model - - Save the current model in the given folder, using the given prefix for the various - files that will get created. - Any file with the same name that already exists in this folder will be overwritten. - - Args: - folder (:obj:`str`): - The path to the target folder in which to save the various files - - prefix (:obj:`str`, `optional`): - An optional prefix, used to prefix each file name - - Returns: - :obj:`List[str]`: The list of saved files - """ - pass - - def token_to_id(self, tokens): - """ - Get the ID associated to a token - - Args: - token (:obj:`str`): - A token to convert to an ID - - Returns: - :obj:`int`: The ID associated to the token - """ - pass - - def tokenize(self, sequence): - """ - Tokenize a sequence - - Args: - sequence (:obj:`str`): - A sequence to tokenize - - Returns: - A :obj:`List` of :class:`~tokenizers.Token`: The generated tokens - """ - pass - - @property - def unk_token(self): - """ """ - pass - - @unk_token.setter - def unk_token(self, value): - """ """ - pass - -class Unigram(Model): - """ - An implementation of the Unigram algorithm - - Args: - vocab (:obj:`List[Tuple[str, float]]`, `optional`, `optional`): - A list of vocabulary items and their relative score [("am", -0.2442),...] - """ - def __init__(self, vocab=None, unk_id=None, byte_fallback=None): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - def get_trainer(self): - """ - Get the associated :class:`~tokenizers.trainers.Trainer` - - Retrieve the :class:`~tokenizers.trainers.Trainer` associated to this - :class:`~tokenizers.models.Model`. - - Returns: - :class:`~tokenizers.trainers.Trainer`: The Trainer used to train this model - """ - pass - - def id_to_token(self, id): - """ - Get the token associated to an ID - - Args: - id (:obj:`int`): - An ID to convert to a token - - Returns: - :obj:`str`: The token associated to the ID - """ - pass - - def save(self, folder, prefix): - """ - Save the current model - - Save the current model in the given folder, using the given prefix for the various - files that will get created. - Any file with the same name that already exists in this folder will be overwritten. - - Args: - folder (:obj:`str`): - The path to the target folder in which to save the various files - - prefix (:obj:`str`, `optional`): - An optional prefix, used to prefix each file name - - Returns: - :obj:`List[str]`: The list of saved files - """ - pass - - def token_to_id(self, tokens): - """ - Get the ID associated to a token - - Args: - token (:obj:`str`): - A token to convert to an ID - - Returns: - :obj:`int`: The ID associated to the token - """ - pass - - def tokenize(self, sequence): - """ - Tokenize a sequence - - Args: - sequence (:obj:`str`): - A sequence to tokenize - - Returns: - A :obj:`List` of :class:`~tokenizers.Token`: The generated tokens - """ - pass - -class WordLevel(Model): - """ - An implementation of the WordLevel algorithm - - Most simple tokenizer model based on mapping tokens to their corresponding id. - - Args: - vocab (:obj:`str`, `optional`): - A dictionary of string keys and their ids :obj:`{"am": 0,...}` - - unk_token (:obj:`str`, `optional`): - The unknown token to be used by the model. - """ - def __init__(self, vocab=None, unk_token=None): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def from_file(vocab, unk_token=None): - """ - Instantiate a WordLevel model from the given file - - This method is roughly equivalent to doing:: - - vocab = WordLevel.read_file(vocab_filename) - wordlevel = WordLevel(vocab) - - If you don't need to keep the :obj:`vocab` values lying around, this method is - more optimized than manually calling :meth:`~tokenizers.models.WordLevel.read_file` to - initialize a :class:`~tokenizers.models.WordLevel` - - Args: - vocab (:obj:`str`): - The path to a :obj:`vocab.json` file - - Returns: - :class:`~tokenizers.models.WordLevel`: An instance of WordLevel loaded from file - """ - pass - - def get_trainer(self): - """ - Get the associated :class:`~tokenizers.trainers.Trainer` - - Retrieve the :class:`~tokenizers.trainers.Trainer` associated to this - :class:`~tokenizers.models.Model`. - - Returns: - :class:`~tokenizers.trainers.Trainer`: The Trainer used to train this model - """ - pass - - def id_to_token(self, id): - """ - Get the token associated to an ID - - Args: - id (:obj:`int`): - An ID to convert to a token - - Returns: - :obj:`str`: The token associated to the ID - """ - pass - - @staticmethod - def read_file(vocab): - """ - Read a :obj:`vocab.json` - - This method provides a way to read and parse the content of a vocabulary file, - returning the relevant data structures. If you want to instantiate some WordLevel models - from memory, this method gives you the expected input from the standard files. - - Args: - vocab (:obj:`str`): - The path to a :obj:`vocab.json` file - - Returns: - :obj:`Dict[str, int]`: The vocabulary as a :obj:`dict` - """ - pass - - def save(self, folder, prefix): - """ - Save the current model - - Save the current model in the given folder, using the given prefix for the various - files that will get created. - Any file with the same name that already exists in this folder will be overwritten. - - Args: - folder (:obj:`str`): - The path to the target folder in which to save the various files - - prefix (:obj:`str`, `optional`): - An optional prefix, used to prefix each file name - - Returns: - :obj:`List[str]`: The list of saved files - """ - pass - - def token_to_id(self, tokens): - """ - Get the ID associated to a token - - Args: - token (:obj:`str`): - A token to convert to an ID - - Returns: - :obj:`int`: The ID associated to the token - """ - pass - - def tokenize(self, sequence): - """ - Tokenize a sequence - - Args: - sequence (:obj:`str`): - A sequence to tokenize - - Returns: - A :obj:`List` of :class:`~tokenizers.Token`: The generated tokens - """ - pass - - @property - def unk_token(self): - """ """ - pass - - @unk_token.setter - def unk_token(self, value): - """ """ - pass - -class WordPiece(Model): - """ - An implementation of the WordPiece algorithm - - Args: - vocab (:obj:`Dict[str, int]`, `optional`): - A dictionary of string keys and their ids :obj:`{"am": 0,...}` - - unk_token (:obj:`str`, `optional`): - The unknown token to be used by the model. - - max_input_chars_per_word (:obj:`int`, `optional`): - The maximum number of characters to authorize in a single word. - """ - def __init__(self, vocab=None, unk_token="[UNK]", max_input_chars_per_word=100, continuing_subword_prefix="##"): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @property - def continuing_subword_prefix(self): - """ """ - pass - - @continuing_subword_prefix.setter - def continuing_subword_prefix(self, value): - """ """ - pass - - @staticmethod - def from_file(vocab, **kwargs): - """ - Instantiate a WordPiece model from the given file - - This method is roughly equivalent to doing:: - - vocab = WordPiece.read_file(vocab_filename) - wordpiece = WordPiece(vocab) - - If you don't need to keep the :obj:`vocab` values lying around, this method is - more optimized than manually calling :meth:`~tokenizers.models.WordPiece.read_file` to - initialize a :class:`~tokenizers.models.WordPiece` - - Args: - vocab (:obj:`str`): - The path to a :obj:`vocab.txt` file - - Returns: - :class:`~tokenizers.models.WordPiece`: An instance of WordPiece loaded from file - """ - pass - - def get_trainer(self): - """ - Get the associated :class:`~tokenizers.trainers.Trainer` - - Retrieve the :class:`~tokenizers.trainers.Trainer` associated to this - :class:`~tokenizers.models.Model`. - - Returns: - :class:`~tokenizers.trainers.Trainer`: The Trainer used to train this model - """ - pass - - def id_to_token(self, id): - """ - Get the token associated to an ID - - Args: - id (:obj:`int`): - An ID to convert to a token - - Returns: - :obj:`str`: The token associated to the ID - """ - pass - - @property - def max_input_chars_per_word(self): - """ """ - pass - - @max_input_chars_per_word.setter - def max_input_chars_per_word(self, value): - """ """ - pass - - @staticmethod - def read_file(vocab): - """ - Read a :obj:`vocab.txt` file - - This method provides a way to read and parse the content of a standard `vocab.txt` - file as used by the WordPiece Model, returning the relevant data structures. If you - want to instantiate some WordPiece models from memory, this method gives you the - expected input from the standard files. - - Args: - vocab (:obj:`str`): - The path to a :obj:`vocab.txt` file - - Returns: - :obj:`Dict[str, int]`: The vocabulary as a :obj:`dict` - """ - pass - - def save(self, folder, prefix): - """ - Save the current model - - Save the current model in the given folder, using the given prefix for the various - files that will get created. - Any file with the same name that already exists in this folder will be overwritten. - - Args: - folder (:obj:`str`): - The path to the target folder in which to save the various files - - prefix (:obj:`str`, `optional`): - An optional prefix, used to prefix each file name - - Returns: - :obj:`List[str]`: The list of saved files - """ - pass - - def token_to_id(self, tokens): - """ - Get the ID associated to a token - - Args: - token (:obj:`str`): - A token to convert to an ID - - Returns: - :obj:`int`: The ID associated to the token - """ - pass - - def tokenize(self, sequence): - """ - Tokenize a sequence - - Args: - sequence (:obj:`str`): - A sequence to tokenize - - Returns: - A :obj:`List` of :class:`~tokenizers.Token`: The generated tokens - """ - pass - - @property - def unk_token(self): - """ """ - pass - - @unk_token.setter - def unk_token(self, value): - """ """ - pass diff --git a/bindings/python/py_src/tokenizers/normalizers.pyi b/bindings/python/py_src/tokenizers/normalizers.pyi new file mode 100644 index 0000000000..a25ab319c9 --- /dev/null +++ b/bindings/python/py_src/tokenizers/normalizers.pyi @@ -0,0 +1,170 @@ +import tokenizers +import tokenizers.normalizers +import typing + +class BertNormalizer: + def __new__( + cls, + /, + clean_text: bool = True, + handle_chinese_chars: bool = True, + strip_accents: bool | None = None, + lowercase: bool = True, + ) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def clean_text(self, /) -> bool: ... + @clean_text.setter + def clean_text(self, /, clean_text: bool) -> None: ... + @property + def handle_chinese_chars(self, /) -> bool: ... + @handle_chinese_chars.setter + def handle_chinese_chars(self, /, handle_chinese_chars: bool) -> None: ... + @property + def lowercase(self, /) -> bool: ... + @lowercase.setter + def lowercase(self, /, lowercase: bool) -> None: ... + @property + def strip_accents(self, /) -> typing.Any: ... + @strip_accents.setter + def strip_accents(self, /, strip_accents: bool | None) -> None: ... + +class ByteLevel: + def __new__(cls, /) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + +class Lowercase: + def __new__(cls, /) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + +class NFC: + def __new__(cls, /) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + +class NFD: + def __new__(cls, /) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + +class NFKC: + def __new__(cls, /) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + +class NFKD: + def __new__(cls, /) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + +class Nmt: + def __new__(cls, /) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + +class Normalizer: + def __getstate__(self, /) -> typing.Any: ... + def __repr__(self, /) -> str: + """Return repr(self).""" + ... + def __setstate__(self, /, state: typing.Any) -> typing.Any: ... + def __str__(self, /) -> str: + """Return str(self).""" + ... + @staticmethod + def custom(obj: typing.Any) -> tokenizers.normalizers.Normalizer: ... + def normalize(self, /, normalized: tokenizers.NormalizedString | tokenizers.NormalizedStringRefMut) -> typing.Any: + """ + Normalize a :class:`~tokenizers.NormalizedString` in-place + + This method allows to modify a :class:`~tokenizers.NormalizedString` to + keep track of the alignment information. If you just want to see the result + of the normalization on a raw string, you can use + :meth:`~tokenizers.normalizers.Normalizer.normalize_str` + + Args: + normalized (:class:`~tokenizers.NormalizedString`): + The normalized string on which to apply this + :class:`~tokenizers.normalizers.Normalizer` + """ + ... + def normalize_str(self, /, sequence: str) -> str: + """ + Normalize the given string + + This method provides a way to visualize the effect of a + :class:`~tokenizers.normalizers.Normalizer` but it does not keep track of the alignment + information. If you need to get/convert offsets, you can use + :meth:`~tokenizers.normalizers.Normalizer.normalize` + + Args: + sequence (:obj:`str`): + A string to normalize + + Returns: + :obj:`str`: A string after normalization + """ + ... + +class Precompiled: + def __new__(cls, /, precompiled_charsmap: typing.Any) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + +class Prepend: + def __new__(cls, /, prepend: str = ...) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def prepend(self, /) -> str: ... + @prepend.setter + def prepend(self, /, prepend: str) -> None: ... + +class Replace: + def __new__(cls, /, pattern: str | tokenizers.Regex, content: str) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def content(self, /) -> str: ... + @content.setter + def content(self, /, content: str) -> None: ... + @property + def pattern(self, /) -> typing.Any: ... + @pattern.setter + def pattern(self, /, _pattern: str | tokenizers.Regex) -> typing.Any: ... + +class Sequence: + def __getitem__(self, /, index: int) -> typing.Any: + """Return self[key].""" + ... + def __getnewargs__(self, /) -> typing.Any: ... + def __len__(self, /) -> int: + """Return len(self).""" + ... + def __new__(cls, /, normalizers: typing.Any) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + def __setitem__(self, /, index: int, value: typing.Any) -> typing.Any: + """Set self[key] to value.""" + ... + +class Strip: + def __new__(cls, /, left: bool = True, right: bool = True) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def left(self, /) -> bool: ... + @left.setter + def left(self, /, left: bool) -> None: ... + @property + def right(self, /) -> bool: ... + @right.setter + def right(self, /, right: bool) -> None: ... + +class StripAccents: + def __new__(cls, /) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... diff --git a/bindings/python/py_src/tokenizers/normalizers/__init__.pyi b/bindings/python/py_src/tokenizers/normalizers/__init__.pyi deleted file mode 100644 index 8d920e0ed7..0000000000 --- a/bindings/python/py_src/tokenizers/normalizers/__init__.pyi +++ /dev/null @@ -1,946 +0,0 @@ -# Generated content DO NOT EDIT -class Normalizer: - """ - Base class for all normalizers - - This class is not supposed to be instantiated directly. Instead, any implementation of a - Normalizer will return an instance of this class when instantiated. - """ - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(normalizer): - """ """ - pass - - def normalize(self, normalized): - """ - Normalize a :class:`~tokenizers.NormalizedString` in-place - - This method allows to modify a :class:`~tokenizers.NormalizedString` to - keep track of the alignment information. If you just want to see the result - of the normalization on a raw string, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize_str` - - Args: - normalized (:class:`~tokenizers.NormalizedString`): - The normalized string on which to apply this - :class:`~tokenizers.normalizers.Normalizer` - """ - pass - - def normalize_str(self, sequence): - """ - Normalize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.normalizers.Normalizer` but it does not keep track of the alignment - information. If you need to get/convert offsets, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize` - - Args: - sequence (:obj:`str`): - A string to normalize - - Returns: - :obj:`str`: A string after normalization - """ - pass - -class BertNormalizer(Normalizer): - """ - BertNormalizer - - Takes care of normalizing raw text before giving it to a Bert model. - This includes cleaning the text, handling accents, chinese chars and lowercasing - - Args: - clean_text (:obj:`bool`, `optional`, defaults to :obj:`True`): - Whether to clean the text, by removing any control characters - and replacing all whitespaces by the classic one. - - handle_chinese_chars (:obj:`bool`, `optional`, defaults to :obj:`True`): - Whether to handle chinese chars by putting spaces around them. - - strip_accents (:obj:`bool`, `optional`): - Whether to strip all accents. If this option is not specified (ie == None), - then it will be determined by the value for `lowercase` (as in the original Bert). - - lowercase (:obj:`bool`, `optional`, defaults to :obj:`True`): - Whether to lowercase. - """ - def __init__(self, clean_text=True, handle_chinese_chars=True, strip_accents=None, lowercase=True): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @property - def clean_text(self): - """ """ - pass - - @clean_text.setter - def clean_text(self, value): - """ """ - pass - - @staticmethod - def custom(normalizer): - """ """ - pass - - @property - def handle_chinese_chars(self): - """ """ - pass - - @handle_chinese_chars.setter - def handle_chinese_chars(self, value): - """ """ - pass - - @property - def lowercase(self): - """ """ - pass - - @lowercase.setter - def lowercase(self, value): - """ """ - pass - - def normalize(self, normalized): - """ - Normalize a :class:`~tokenizers.NormalizedString` in-place - - This method allows to modify a :class:`~tokenizers.NormalizedString` to - keep track of the alignment information. If you just want to see the result - of the normalization on a raw string, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize_str` - - Args: - normalized (:class:`~tokenizers.NormalizedString`): - The normalized string on which to apply this - :class:`~tokenizers.normalizers.Normalizer` - """ - pass - - def normalize_str(self, sequence): - """ - Normalize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.normalizers.Normalizer` but it does not keep track of the alignment - information. If you need to get/convert offsets, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize` - - Args: - sequence (:obj:`str`): - A string to normalize - - Returns: - :obj:`str`: A string after normalization - """ - pass - - @property - def strip_accents(self): - """ """ - pass - - @strip_accents.setter - def strip_accents(self, value): - """ """ - pass - -class ByteLevel(Normalizer): - """ - Bytelevel Normalizer - """ - def __init__(self): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(normalizer): - """ """ - pass - - def normalize(self, normalized): - """ - Normalize a :class:`~tokenizers.NormalizedString` in-place - - This method allows to modify a :class:`~tokenizers.NormalizedString` to - keep track of the alignment information. If you just want to see the result - of the normalization on a raw string, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize_str` - - Args: - normalized (:class:`~tokenizers.NormalizedString`): - The normalized string on which to apply this - :class:`~tokenizers.normalizers.Normalizer` - """ - pass - - def normalize_str(self, sequence): - """ - Normalize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.normalizers.Normalizer` but it does not keep track of the alignment - information. If you need to get/convert offsets, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize` - - Args: - sequence (:obj:`str`): - A string to normalize - - Returns: - :obj:`str`: A string after normalization - """ - pass - -class Lowercase(Normalizer): - """ - Lowercase Normalizer - """ - def __init__(self): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(normalizer): - """ """ - pass - - def normalize(self, normalized): - """ - Normalize a :class:`~tokenizers.NormalizedString` in-place - - This method allows to modify a :class:`~tokenizers.NormalizedString` to - keep track of the alignment information. If you just want to see the result - of the normalization on a raw string, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize_str` - - Args: - normalized (:class:`~tokenizers.NormalizedString`): - The normalized string on which to apply this - :class:`~tokenizers.normalizers.Normalizer` - """ - pass - - def normalize_str(self, sequence): - """ - Normalize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.normalizers.Normalizer` but it does not keep track of the alignment - information. If you need to get/convert offsets, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize` - - Args: - sequence (:obj:`str`): - A string to normalize - - Returns: - :obj:`str`: A string after normalization - """ - pass - -class NFC(Normalizer): - """ - NFC Unicode Normalizer - """ - def __init__(self): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(normalizer): - """ """ - pass - - def normalize(self, normalized): - """ - Normalize a :class:`~tokenizers.NormalizedString` in-place - - This method allows to modify a :class:`~tokenizers.NormalizedString` to - keep track of the alignment information. If you just want to see the result - of the normalization on a raw string, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize_str` - - Args: - normalized (:class:`~tokenizers.NormalizedString`): - The normalized string on which to apply this - :class:`~tokenizers.normalizers.Normalizer` - """ - pass - - def normalize_str(self, sequence): - """ - Normalize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.normalizers.Normalizer` but it does not keep track of the alignment - information. If you need to get/convert offsets, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize` - - Args: - sequence (:obj:`str`): - A string to normalize - - Returns: - :obj:`str`: A string after normalization - """ - pass - -class NFD(Normalizer): - """ - NFD Unicode Normalizer - """ - def __init__(self): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(normalizer): - """ """ - pass - - def normalize(self, normalized): - """ - Normalize a :class:`~tokenizers.NormalizedString` in-place - - This method allows to modify a :class:`~tokenizers.NormalizedString` to - keep track of the alignment information. If you just want to see the result - of the normalization on a raw string, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize_str` - - Args: - normalized (:class:`~tokenizers.NormalizedString`): - The normalized string on which to apply this - :class:`~tokenizers.normalizers.Normalizer` - """ - pass - - def normalize_str(self, sequence): - """ - Normalize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.normalizers.Normalizer` but it does not keep track of the alignment - information. If you need to get/convert offsets, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize` - - Args: - sequence (:obj:`str`): - A string to normalize - - Returns: - :obj:`str`: A string after normalization - """ - pass - -class NFKC(Normalizer): - """ - NFKC Unicode Normalizer - """ - def __init__(self): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(normalizer): - """ """ - pass - - def normalize(self, normalized): - """ - Normalize a :class:`~tokenizers.NormalizedString` in-place - - This method allows to modify a :class:`~tokenizers.NormalizedString` to - keep track of the alignment information. If you just want to see the result - of the normalization on a raw string, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize_str` - - Args: - normalized (:class:`~tokenizers.NormalizedString`): - The normalized string on which to apply this - :class:`~tokenizers.normalizers.Normalizer` - """ - pass - - def normalize_str(self, sequence): - """ - Normalize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.normalizers.Normalizer` but it does not keep track of the alignment - information. If you need to get/convert offsets, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize` - - Args: - sequence (:obj:`str`): - A string to normalize - - Returns: - :obj:`str`: A string after normalization - """ - pass - -class NFKD(Normalizer): - """ - NFKD Unicode Normalizer - """ - def __init__(self): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(normalizer): - """ """ - pass - - def normalize(self, normalized): - """ - Normalize a :class:`~tokenizers.NormalizedString` in-place - - This method allows to modify a :class:`~tokenizers.NormalizedString` to - keep track of the alignment information. If you just want to see the result - of the normalization on a raw string, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize_str` - - Args: - normalized (:class:`~tokenizers.NormalizedString`): - The normalized string on which to apply this - :class:`~tokenizers.normalizers.Normalizer` - """ - pass - - def normalize_str(self, sequence): - """ - Normalize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.normalizers.Normalizer` but it does not keep track of the alignment - information. If you need to get/convert offsets, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize` - - Args: - sequence (:obj:`str`): - A string to normalize - - Returns: - :obj:`str`: A string after normalization - """ - pass - -class Nmt(Normalizer): - """ - Nmt normalizer - """ - def __init__(self): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(normalizer): - """ """ - pass - - def normalize(self, normalized): - """ - Normalize a :class:`~tokenizers.NormalizedString` in-place - - This method allows to modify a :class:`~tokenizers.NormalizedString` to - keep track of the alignment information. If you just want to see the result - of the normalization on a raw string, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize_str` - - Args: - normalized (:class:`~tokenizers.NormalizedString`): - The normalized string on which to apply this - :class:`~tokenizers.normalizers.Normalizer` - """ - pass - - def normalize_str(self, sequence): - """ - Normalize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.normalizers.Normalizer` but it does not keep track of the alignment - information. If you need to get/convert offsets, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize` - - Args: - sequence (:obj:`str`): - A string to normalize - - Returns: - :obj:`str`: A string after normalization - """ - pass - -class Precompiled(Normalizer): - """ - Precompiled normalizer - Don't use manually it is used for compatibility for SentencePiece. - """ - def __init__(self, precompiled_charsmap): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(normalizer): - """ """ - pass - - def normalize(self, normalized): - """ - Normalize a :class:`~tokenizers.NormalizedString` in-place - - This method allows to modify a :class:`~tokenizers.NormalizedString` to - keep track of the alignment information. If you just want to see the result - of the normalization on a raw string, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize_str` - - Args: - normalized (:class:`~tokenizers.NormalizedString`): - The normalized string on which to apply this - :class:`~tokenizers.normalizers.Normalizer` - """ - pass - - def normalize_str(self, sequence): - """ - Normalize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.normalizers.Normalizer` but it does not keep track of the alignment - information. If you need to get/convert offsets, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize` - - Args: - sequence (:obj:`str`): - A string to normalize - - Returns: - :obj:`str`: A string after normalization - """ - pass - -class Prepend(Normalizer): - """ - Prepend normalizer - """ - def __init__(self, prepend): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(normalizer): - """ """ - pass - - def normalize(self, normalized): - """ - Normalize a :class:`~tokenizers.NormalizedString` in-place - - This method allows to modify a :class:`~tokenizers.NormalizedString` to - keep track of the alignment information. If you just want to see the result - of the normalization on a raw string, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize_str` - - Args: - normalized (:class:`~tokenizers.NormalizedString`): - The normalized string on which to apply this - :class:`~tokenizers.normalizers.Normalizer` - """ - pass - - def normalize_str(self, sequence): - """ - Normalize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.normalizers.Normalizer` but it does not keep track of the alignment - information. If you need to get/convert offsets, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize` - - Args: - sequence (:obj:`str`): - A string to normalize - - Returns: - :obj:`str`: A string after normalization - """ - pass - - @property - def prepend(self): - """ """ - pass - - @prepend.setter - def prepend(self, value): - """ """ - pass - -class Replace(Normalizer): - """ - Replace normalizer - """ - def __init__(self, pattern, content): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @property - def content(self): - """ """ - pass - - @content.setter - def content(self, value): - """ """ - pass - - @staticmethod - def custom(normalizer): - """ """ - pass - - def normalize(self, normalized): - """ - Normalize a :class:`~tokenizers.NormalizedString` in-place - - This method allows to modify a :class:`~tokenizers.NormalizedString` to - keep track of the alignment information. If you just want to see the result - of the normalization on a raw string, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize_str` - - Args: - normalized (:class:`~tokenizers.NormalizedString`): - The normalized string on which to apply this - :class:`~tokenizers.normalizers.Normalizer` - """ - pass - - def normalize_str(self, sequence): - """ - Normalize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.normalizers.Normalizer` but it does not keep track of the alignment - information. If you need to get/convert offsets, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize` - - Args: - sequence (:obj:`str`): - A string to normalize - - Returns: - :obj:`str`: A string after normalization - """ - pass - - @property - def pattern(self): - """ """ - pass - - @pattern.setter - def pattern(self, value): - """ """ - pass - -class Sequence(Normalizer): - """ - Allows concatenating multiple other Normalizer as a Sequence. - All the normalizers run in sequence in the given order - - Args: - normalizers (:obj:`List[Normalizer]`): - A list of Normalizer to be run as a sequence - """ - def __init__(self, normalizers): - pass - - def __getitem__(self, key): - """ - Return self[key]. - """ - pass - - def __getnewargs__(self): - """ """ - pass - - def __getstate__(self): - """ """ - pass - - def __setitem__(self, key, value): - """ - Set self[key] to value. - """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(normalizer): - """ """ - pass - - def normalize(self, normalized): - """ - Normalize a :class:`~tokenizers.NormalizedString` in-place - - This method allows to modify a :class:`~tokenizers.NormalizedString` to - keep track of the alignment information. If you just want to see the result - of the normalization on a raw string, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize_str` - - Args: - normalized (:class:`~tokenizers.NormalizedString`): - The normalized string on which to apply this - :class:`~tokenizers.normalizers.Normalizer` - """ - pass - - def normalize_str(self, sequence): - """ - Normalize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.normalizers.Normalizer` but it does not keep track of the alignment - information. If you need to get/convert offsets, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize` - - Args: - sequence (:obj:`str`): - A string to normalize - - Returns: - :obj:`str`: A string after normalization - """ - pass - -class Strip(Normalizer): - """ - Strip normalizer - """ - def __init__(self, left=True, right=True): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(normalizer): - """ """ - pass - - @property - def left(self): - """ """ - pass - - @left.setter - def left(self, value): - """ """ - pass - - def normalize(self, normalized): - """ - Normalize a :class:`~tokenizers.NormalizedString` in-place - - This method allows to modify a :class:`~tokenizers.NormalizedString` to - keep track of the alignment information. If you just want to see the result - of the normalization on a raw string, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize_str` - - Args: - normalized (:class:`~tokenizers.NormalizedString`): - The normalized string on which to apply this - :class:`~tokenizers.normalizers.Normalizer` - """ - pass - - def normalize_str(self, sequence): - """ - Normalize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.normalizers.Normalizer` but it does not keep track of the alignment - information. If you need to get/convert offsets, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize` - - Args: - sequence (:obj:`str`): - A string to normalize - - Returns: - :obj:`str`: A string after normalization - """ - pass - - @property - def right(self): - """ """ - pass - - @right.setter - def right(self, value): - """ """ - pass - -class StripAccents(Normalizer): - """ - StripAccents normalizer - """ - def __init__(self): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(normalizer): - """ """ - pass - - def normalize(self, normalized): - """ - Normalize a :class:`~tokenizers.NormalizedString` in-place - - This method allows to modify a :class:`~tokenizers.NormalizedString` to - keep track of the alignment information. If you just want to see the result - of the normalization on a raw string, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize_str` - - Args: - normalized (:class:`~tokenizers.NormalizedString`): - The normalized string on which to apply this - :class:`~tokenizers.normalizers.Normalizer` - """ - pass - - def normalize_str(self, sequence): - """ - Normalize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.normalizers.Normalizer` but it does not keep track of the alignment - information. If you need to get/convert offsets, you can use - :meth:`~tokenizers.normalizers.Normalizer.normalize` - - Args: - sequence (:obj:`str`): - A string to normalize - - Returns: - :obj:`str`: A string after normalization - """ - pass - -from typing import Dict - -NORMALIZERS: Dict[str, Normalizer] - -def unicode_normalizer_from_str(normalizer: str) -> Normalizer: ... diff --git a/bindings/python/py_src/tokenizers/pre_tokenizers.pyi b/bindings/python/py_src/tokenizers/pre_tokenizers.pyi new file mode 100644 index 0000000000..911cce3b96 --- /dev/null +++ b/bindings/python/py_src/tokenizers/pre_tokenizers.pyi @@ -0,0 +1,186 @@ +import tokenizers +import tokenizers.pre_tokenizers +import typing + +class BertPreTokenizer: + def __new__(cls, /) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + +class ByteLevel: + def __new__( + cls, /, add_prefix_space: bool = True, trim_offsets: bool = True, use_regex: bool = True, **_kwargs + ) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def add_prefix_space(self, /) -> bool: ... + @add_prefix_space.setter + def add_prefix_space(self, /, add_prefix_space: bool) -> None: ... + @staticmethod + def alphabet() -> typing.Any: + """ + Returns the alphabet used by this PreTokenizer. + + Since the ByteLevel works as its name suggests, at the byte level, it + encodes each byte value to a unique visible character. This means that there is a + total of 256 different characters composing this alphabet. + + Returns: + :obj:`List[str]`: A list of characters that compose the alphabet + """ + ... + @property + def trim_offsets(self, /) -> bool: ... + @trim_offsets.setter + def trim_offsets(self, /, trim_offsets: bool) -> None: ... + @property + def use_regex(self, /) -> bool: ... + @use_regex.setter + def use_regex(self, /, use_regex: bool) -> None: ... + +class CharDelimiterSplit: + def __getnewargs__(self, /) -> typing.Any: ... + def __new__(cls, /, delimiter: str) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def delimiter(self, /) -> str: ... + @delimiter.setter + def delimiter(self, /, delimiter: str) -> None: ... + +class Digits: + def __new__(cls, /, individual_digits: bool = False) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def individual_digits(self, /) -> bool: ... + @individual_digits.setter + def individual_digits(self, /, individual_digits: bool) -> None: ... + +class FixedLength: + def __new__(cls, /, length: int = 5) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def length(self, /) -> int: ... + @length.setter + def length(self, /, length: int) -> None: ... + +class Metaspace: + def __new__(cls, /, replacement: str = "▁", prepend_scheme: str = ..., split: bool = True) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def prepend_scheme(self, /) -> str: ... + @prepend_scheme.setter + def prepend_scheme(self, /, prepend_scheme: str) -> typing.Any: ... + @property + def replacement(self, /) -> str: ... + @replacement.setter + def replacement(self, /, replacement: str) -> None: ... + @property + def split(self, /) -> bool: ... + @split.setter + def split(self, /, split: bool) -> None: ... + +class PreTokenizer: + def __getstate__(self, /) -> typing.Any: ... + def __repr__(self, /) -> str: + """Return repr(self).""" + ... + def __setstate__(self, /, state: typing.Any) -> typing.Any: ... + def __str__(self, /) -> str: + """Return str(self).""" + ... + @staticmethod + def custom(pretok: typing.Any) -> tokenizers.pre_tokenizers.PreTokenizer: ... + def pre_tokenize(self, /, pretok: tokenizers.PreTokenizedString) -> typing.Any: + """ + Pre-tokenize a :class:`~tokenizers.PyPreTokenizedString` in-place + + This method allows to modify a :class:`~tokenizers.PreTokenizedString` to + keep track of the pre-tokenization, and leverage the capabilities of the + :class:`~tokenizers.PreTokenizedString`. If you just want to see the result of + the pre-tokenization of a raw string, you can use + :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize_str` + + Args: + pretok (:class:`~tokenizers.PreTokenizedString): + The pre-tokenized string on which to apply this + :class:`~tokenizers.pre_tokenizers.PreTokenizer` + """ + ... + def pre_tokenize_str(self, /, s: str) -> typing.Any: + """ + Pre tokenize the given string + + This method provides a way to visualize the effect of a + :class:`~tokenizers.pre_tokenizers.PreTokenizer` but it does not keep track of the + alignment, nor does it provide all the capabilities of the + :class:`~tokenizers.PreTokenizedString`. If you need some of these, you can use + :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize` + + Args: + sequence (:obj:`str`): + A string to pre-tokeize + + Returns: + :obj:`List[Tuple[str, Offsets]]`: + A list of tuple with the pre-tokenized parts and their offsets + """ + ... + +class Punctuation: + def __new__(cls, /, behavior: typing.Any = ...) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def behavior(self, /) -> str: ... + @behavior.setter + def behavior(self, /, behavior: str) -> typing.Any: ... + +class Sequence: + def __getitem__(self, /, index: int) -> typing.Any: + """Return self[key].""" + ... + def __getnewargs__(self, /) -> typing.Any: ... + def __new__(cls, /, pre_tokenizers: typing.Any) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + def __setitem__(self, /, index: int, value: typing.Any) -> typing.Any: + """Set self[key] to value.""" + ... + +class Split: + def __getnewargs__(self, /) -> typing.Any: ... + def __new__(cls, /, pattern: str | tokenizers.Regex, behavior: typing.Any, invert: bool = False) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def behavior(self, /) -> str: ... + @behavior.setter + def behavior(self, /, behavior: str) -> typing.Any: ... + @property + def invert(self, /) -> bool: ... + @invert.setter + def invert(self, /, invert: bool) -> None: ... + @property + def pattern(self, /) -> typing.Any: ... + @pattern.setter + def pattern(self, /, _pattern: str | tokenizers.Regex) -> typing.Any: ... + +class UnicodeScripts: + def __new__(cls, /) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + +class Whitespace: + def __new__(cls, /) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + +class WhitespaceSplit: + def __new__(cls, /) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... diff --git a/bindings/python/py_src/tokenizers/pre_tokenizers/__init__.py b/bindings/python/py_src/tokenizers/pre_tokenizers/__init__.py index db8ddc2080..54bf038c0d 100644 --- a/bindings/python/py_src/tokenizers/pre_tokenizers/__init__.py +++ b/bindings/python/py_src/tokenizers/pre_tokenizers/__init__.py @@ -1,13 +1,14 @@ # Generated content DO NOT EDIT + from .. import pre_tokenizers -PreTokenizer = pre_tokenizers.PreTokenizer BertPreTokenizer = pre_tokenizers.BertPreTokenizer ByteLevel = pre_tokenizers.ByteLevel CharDelimiterSplit = pre_tokenizers.CharDelimiterSplit Digits = pre_tokenizers.Digits FixedLength = pre_tokenizers.FixedLength Metaspace = pre_tokenizers.Metaspace +PreTokenizer = pre_tokenizers.PreTokenizer Punctuation = pre_tokenizers.Punctuation Sequence = pre_tokenizers.Sequence Split = pre_tokenizers.Split diff --git a/bindings/python/py_src/tokenizers/pre_tokenizers/__init__.pyi b/bindings/python/py_src/tokenizers/pre_tokenizers/__init__.pyi deleted file mode 100644 index 1e58d5d040..0000000000 --- a/bindings/python/py_src/tokenizers/pre_tokenizers/__init__.pyi +++ /dev/null @@ -1,1015 +0,0 @@ -# Generated content DO NOT EDIT -class PreTokenizer: - """ - Base class for all pre-tokenizers - - This class is not supposed to be instantiated directly. Instead, any implementation of a - PreTokenizer will return an instance of this class when instantiated. - """ - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(pretok): - """ """ - pass - - def pre_tokenize(self, pretok): - """ - Pre-tokenize a :class:`~tokenizers.PyPreTokenizedString` in-place - - This method allows to modify a :class:`~tokenizers.PreTokenizedString` to - keep track of the pre-tokenization, and leverage the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you just want to see the result of - the pre-tokenization of a raw string, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize_str` - - Args: - pretok (:class:`~tokenizers.PreTokenizedString): - The pre-tokenized string on which to apply this - :class:`~tokenizers.pre_tokenizers.PreTokenizer` - """ - pass - - def pre_tokenize_str(self, sequence): - """ - Pre tokenize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.pre_tokenizers.PreTokenizer` but it does not keep track of the - alignment, nor does it provide all the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you need some of these, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize` - - Args: - sequence (:obj:`str`): - A string to pre-tokeize - - Returns: - :obj:`List[Tuple[str, Offsets]]`: - A list of tuple with the pre-tokenized parts and their offsets - """ - pass - -class BertPreTokenizer(PreTokenizer): - """ - BertPreTokenizer - - This pre-tokenizer splits tokens on spaces, and also on punctuation. - Each occurrence of a punctuation character will be treated separately. - """ - def __init__(self): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(pretok): - """ """ - pass - - def pre_tokenize(self, pretok): - """ - Pre-tokenize a :class:`~tokenizers.PyPreTokenizedString` in-place - - This method allows to modify a :class:`~tokenizers.PreTokenizedString` to - keep track of the pre-tokenization, and leverage the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you just want to see the result of - the pre-tokenization of a raw string, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize_str` - - Args: - pretok (:class:`~tokenizers.PreTokenizedString): - The pre-tokenized string on which to apply this - :class:`~tokenizers.pre_tokenizers.PreTokenizer` - """ - pass - - def pre_tokenize_str(self, sequence): - """ - Pre tokenize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.pre_tokenizers.PreTokenizer` but it does not keep track of the - alignment, nor does it provide all the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you need some of these, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize` - - Args: - sequence (:obj:`str`): - A string to pre-tokeize - - Returns: - :obj:`List[Tuple[str, Offsets]]`: - A list of tuple with the pre-tokenized parts and their offsets - """ - pass - -class ByteLevel(PreTokenizer): - """ - ByteLevel PreTokenizer - - This pre-tokenizer takes care of replacing all bytes of the given string - with a corresponding representation, as well as splitting into words. - - Args: - add_prefix_space (:obj:`bool`, `optional`, defaults to :obj:`True`): - Whether to add a space to the first word if there isn't already one. This - lets us treat `hello` exactly like `say hello`. - use_regex (:obj:`bool`, `optional`, defaults to :obj:`True`): - Set this to :obj:`False` to prevent this `pre_tokenizer` from using - the GPT2 specific regexp for spliting on whitespace. - """ - def __init__(self, add_prefix_space=True, trim_offsets=True, use_regex=True): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @property - def add_prefix_space(self): - """ """ - pass - - @add_prefix_space.setter - def add_prefix_space(self, value): - """ """ - pass - - @staticmethod - def alphabet(): - """ - Returns the alphabet used by this PreTokenizer. - - Since the ByteLevel works as its name suggests, at the byte level, it - encodes each byte value to a unique visible character. This means that there is a - total of 256 different characters composing this alphabet. - - Returns: - :obj:`List[str]`: A list of characters that compose the alphabet - """ - pass - - @staticmethod - def custom(pretok): - """ """ - pass - - def pre_tokenize(self, pretok): - """ - Pre-tokenize a :class:`~tokenizers.PyPreTokenizedString` in-place - - This method allows to modify a :class:`~tokenizers.PreTokenizedString` to - keep track of the pre-tokenization, and leverage the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you just want to see the result of - the pre-tokenization of a raw string, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize_str` - - Args: - pretok (:class:`~tokenizers.PreTokenizedString): - The pre-tokenized string on which to apply this - :class:`~tokenizers.pre_tokenizers.PreTokenizer` - """ - pass - - def pre_tokenize_str(self, sequence): - """ - Pre tokenize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.pre_tokenizers.PreTokenizer` but it does not keep track of the - alignment, nor does it provide all the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you need some of these, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize` - - Args: - sequence (:obj:`str`): - A string to pre-tokeize - - Returns: - :obj:`List[Tuple[str, Offsets]]`: - A list of tuple with the pre-tokenized parts and their offsets - """ - pass - - @property - def trim_offsets(self): - """ """ - pass - - @trim_offsets.setter - def trim_offsets(self, value): - """ """ - pass - - @property - def use_regex(self): - """ """ - pass - - @use_regex.setter - def use_regex(self, value): - """ """ - pass - -class CharDelimiterSplit(PreTokenizer): - """ - This pre-tokenizer simply splits on the provided char. Works like `.split(delimiter)` - - Args: - delimiter: str: - The delimiter char that will be used to split input - """ - def __init__(self, delimiter): - pass - - def __getnewargs__(self): - """ """ - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(pretok): - """ """ - pass - - @property - def delimiter(self): - """ """ - pass - - @delimiter.setter - def delimiter(self, value): - """ """ - pass - - def pre_tokenize(self, pretok): - """ - Pre-tokenize a :class:`~tokenizers.PyPreTokenizedString` in-place - - This method allows to modify a :class:`~tokenizers.PreTokenizedString` to - keep track of the pre-tokenization, and leverage the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you just want to see the result of - the pre-tokenization of a raw string, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize_str` - - Args: - pretok (:class:`~tokenizers.PreTokenizedString): - The pre-tokenized string on which to apply this - :class:`~tokenizers.pre_tokenizers.PreTokenizer` - """ - pass - - def pre_tokenize_str(self, sequence): - """ - Pre tokenize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.pre_tokenizers.PreTokenizer` but it does not keep track of the - alignment, nor does it provide all the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you need some of these, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize` - - Args: - sequence (:obj:`str`): - A string to pre-tokeize - - Returns: - :obj:`List[Tuple[str, Offsets]]`: - A list of tuple with the pre-tokenized parts and their offsets - """ - pass - -class Digits(PreTokenizer): - """ - This pre-tokenizer simply splits using the digits in separate tokens - - Args: - individual_digits (:obj:`bool`, `optional`, defaults to :obj:`False`): - If set to True, digits will each be separated as follows:: - - "Call 123 please" -> "Call ", "1", "2", "3", " please" - - If set to False, digits will grouped as follows:: - - "Call 123 please" -> "Call ", "123", " please" - """ - def __init__(self, individual_digits=False): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(pretok): - """ """ - pass - - @property - def individual_digits(self): - """ """ - pass - - @individual_digits.setter - def individual_digits(self, value): - """ """ - pass - - def pre_tokenize(self, pretok): - """ - Pre-tokenize a :class:`~tokenizers.PyPreTokenizedString` in-place - - This method allows to modify a :class:`~tokenizers.PreTokenizedString` to - keep track of the pre-tokenization, and leverage the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you just want to see the result of - the pre-tokenization of a raw string, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize_str` - - Args: - pretok (:class:`~tokenizers.PreTokenizedString): - The pre-tokenized string on which to apply this - :class:`~tokenizers.pre_tokenizers.PreTokenizer` - """ - pass - - def pre_tokenize_str(self, sequence): - """ - Pre tokenize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.pre_tokenizers.PreTokenizer` but it does not keep track of the - alignment, nor does it provide all the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you need some of these, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize` - - Args: - sequence (:obj:`str`): - A string to pre-tokeize - - Returns: - :obj:`List[Tuple[str, Offsets]]`: - A list of tuple with the pre-tokenized parts and their offsets - """ - pass - -class FixedLength(PreTokenizer): - """ - This pre-tokenizer splits the text into fixed length chunks as used - [here](https://www.biorxiv.org/content/10.1101/2023.01.11.523679v1.full) - - Args: - length (:obj:`int`, `optional`, defaults to :obj:`5`): - The length of the chunks to split the text into. - - Strings are split on the character level rather than the byte level to avoid - splitting unicode characters consisting of multiple bytes. - """ - def __init__(self, length=5): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(pretok): - """ """ - pass - - @property - def length(self): - """ """ - pass - - @length.setter - def length(self, value): - """ """ - pass - - def pre_tokenize(self, pretok): - """ - Pre-tokenize a :class:`~tokenizers.PyPreTokenizedString` in-place - - This method allows to modify a :class:`~tokenizers.PreTokenizedString` to - keep track of the pre-tokenization, and leverage the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you just want to see the result of - the pre-tokenization of a raw string, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize_str` - - Args: - pretok (:class:`~tokenizers.PreTokenizedString): - The pre-tokenized string on which to apply this - :class:`~tokenizers.pre_tokenizers.PreTokenizer` - """ - pass - - def pre_tokenize_str(self, sequence): - """ - Pre tokenize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.pre_tokenizers.PreTokenizer` but it does not keep track of the - alignment, nor does it provide all the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you need some of these, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize` - - Args: - sequence (:obj:`str`): - A string to pre-tokeize - - Returns: - :obj:`List[Tuple[str, Offsets]]`: - A list of tuple with the pre-tokenized parts and their offsets - """ - pass - -class Metaspace(PreTokenizer): - """ - Metaspace pre-tokenizer - - This pre-tokenizer replaces any whitespace by the provided replacement character. - It then tries to split on these spaces. - - Args: - replacement (:obj:`str`, `optional`, defaults to :obj:`▁`): - The replacement character. Must be exactly one character. By default we - use the `▁` (U+2581) meta symbol (Same as in SentencePiece). - - prepend_scheme (:obj:`str`, `optional`, defaults to :obj:`"always"`): - Whether to add a space to the first word if there isn't already one. This - lets us treat `hello` exactly like `say hello`. - Choices: "always", "never", "first". First means the space is only added on the first - token (relevant when special tokens are used or other pre_tokenizer are used). - - """ - def __init__(self, replacement="_", prepend_scheme="always", split=True): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(pretok): - """ """ - pass - - def pre_tokenize(self, pretok): - """ - Pre-tokenize a :class:`~tokenizers.PyPreTokenizedString` in-place - - This method allows to modify a :class:`~tokenizers.PreTokenizedString` to - keep track of the pre-tokenization, and leverage the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you just want to see the result of - the pre-tokenization of a raw string, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize_str` - - Args: - pretok (:class:`~tokenizers.PreTokenizedString): - The pre-tokenized string on which to apply this - :class:`~tokenizers.pre_tokenizers.PreTokenizer` - """ - pass - - def pre_tokenize_str(self, sequence): - """ - Pre tokenize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.pre_tokenizers.PreTokenizer` but it does not keep track of the - alignment, nor does it provide all the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you need some of these, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize` - - Args: - sequence (:obj:`str`): - A string to pre-tokeize - - Returns: - :obj:`List[Tuple[str, Offsets]]`: - A list of tuple with the pre-tokenized parts and their offsets - """ - pass - - @property - def prepend_scheme(self): - """ """ - pass - - @prepend_scheme.setter - def prepend_scheme(self, value): - """ """ - pass - - @property - def replacement(self): - """ """ - pass - - @replacement.setter - def replacement(self, value): - """ """ - pass - - @property - def split(self): - """ """ - pass - - @split.setter - def split(self, value): - """ """ - pass - -class Punctuation(PreTokenizer): - """ - This pre-tokenizer simply splits on punctuation as individual characters. - - Args: - behavior (:class:`~tokenizers.SplitDelimiterBehavior`): - The behavior to use when splitting. - Choices: "removed", "isolated" (default), "merged_with_previous", "merged_with_next", - "contiguous" - """ - def __init__(self, behavior="isolated"): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @property - def behavior(self): - """ """ - pass - - @behavior.setter - def behavior(self, value): - """ """ - pass - - @staticmethod - def custom(pretok): - """ """ - pass - - def pre_tokenize(self, pretok): - """ - Pre-tokenize a :class:`~tokenizers.PyPreTokenizedString` in-place - - This method allows to modify a :class:`~tokenizers.PreTokenizedString` to - keep track of the pre-tokenization, and leverage the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you just want to see the result of - the pre-tokenization of a raw string, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize_str` - - Args: - pretok (:class:`~tokenizers.PreTokenizedString): - The pre-tokenized string on which to apply this - :class:`~tokenizers.pre_tokenizers.PreTokenizer` - """ - pass - - def pre_tokenize_str(self, sequence): - """ - Pre tokenize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.pre_tokenizers.PreTokenizer` but it does not keep track of the - alignment, nor does it provide all the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you need some of these, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize` - - Args: - sequence (:obj:`str`): - A string to pre-tokeize - - Returns: - :obj:`List[Tuple[str, Offsets]]`: - A list of tuple with the pre-tokenized parts and their offsets - """ - pass - -class Sequence(PreTokenizer): - """ - This pre-tokenizer composes other pre_tokenizers and applies them in sequence - """ - def __init__(self, pretokenizers): - pass - - def __getitem__(self, key): - """ - Return self[key]. - """ - pass - - def __getnewargs__(self): - """ """ - pass - - def __getstate__(self): - """ """ - pass - - def __setitem__(self, key, value): - """ - Set self[key] to value. - """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(pretok): - """ """ - pass - - def pre_tokenize(self, pretok): - """ - Pre-tokenize a :class:`~tokenizers.PyPreTokenizedString` in-place - - This method allows to modify a :class:`~tokenizers.PreTokenizedString` to - keep track of the pre-tokenization, and leverage the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you just want to see the result of - the pre-tokenization of a raw string, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize_str` - - Args: - pretok (:class:`~tokenizers.PreTokenizedString): - The pre-tokenized string on which to apply this - :class:`~tokenizers.pre_tokenizers.PreTokenizer` - """ - pass - - def pre_tokenize_str(self, sequence): - """ - Pre tokenize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.pre_tokenizers.PreTokenizer` but it does not keep track of the - alignment, nor does it provide all the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you need some of these, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize` - - Args: - sequence (:obj:`str`): - A string to pre-tokeize - - Returns: - :obj:`List[Tuple[str, Offsets]]`: - A list of tuple with the pre-tokenized parts and their offsets - """ - pass - -class Split(PreTokenizer): - """ - Split PreTokenizer - - This versatile pre-tokenizer splits using the provided pattern and - according to the provided behavior. The pattern can be inverted by - making use of the invert flag. - - Args: - pattern (:obj:`str` or :class:`~tokenizers.Regex`): - A pattern used to split the string. Usually a string or a regex built with `tokenizers.Regex`. - If you want to use a regex pattern, it has to be wrapped around a `tokenizers.Regex`, - otherwise we consider is as a string pattern. For example `pattern="|"` - means you want to split on `|` (imagine a csv file for example), while - `pattern=tokenizers.Regex("1|2")` means you split on either '1' or '2'. - behavior (:class:`~tokenizers.SplitDelimiterBehavior`): - The behavior to use when splitting. - Choices: "removed", "isolated", "merged_with_previous", "merged_with_next", - "contiguous" - - invert (:obj:`bool`, `optional`, defaults to :obj:`False`): - Whether to invert the pattern. - """ - def __init__(self, pattern, behavior, invert=False): - pass - - def __getnewargs__(self): - """ """ - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @property - def behavior(self): - """ """ - pass - - @behavior.setter - def behavior(self, value): - """ """ - pass - - @staticmethod - def custom(pretok): - """ """ - pass - - @property - def invert(self): - """ """ - pass - - @invert.setter - def invert(self, value): - """ """ - pass - - @property - def pattern(self): - """ """ - pass - - @pattern.setter - def pattern(self, value): - """ """ - pass - - def pre_tokenize(self, pretok): - """ - Pre-tokenize a :class:`~tokenizers.PyPreTokenizedString` in-place - - This method allows to modify a :class:`~tokenizers.PreTokenizedString` to - keep track of the pre-tokenization, and leverage the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you just want to see the result of - the pre-tokenization of a raw string, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize_str` - - Args: - pretok (:class:`~tokenizers.PreTokenizedString): - The pre-tokenized string on which to apply this - :class:`~tokenizers.pre_tokenizers.PreTokenizer` - """ - pass - - def pre_tokenize_str(self, sequence): - """ - Pre tokenize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.pre_tokenizers.PreTokenizer` but it does not keep track of the - alignment, nor does it provide all the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you need some of these, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize` - - Args: - sequence (:obj:`str`): - A string to pre-tokeize - - Returns: - :obj:`List[Tuple[str, Offsets]]`: - A list of tuple with the pre-tokenized parts and their offsets - """ - pass - -class UnicodeScripts(PreTokenizer): - """ - This pre-tokenizer splits on characters that belong to different language family - It roughly follows https://github.com/google/sentencepiece/blob/master/data/Scripts.txt - Actually Hiragana and Katakana are fused with Han, and 0x30FC is Han too. - This mimicks SentencePiece Unigram implementation. - """ - def __init__(self): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(pretok): - """ """ - pass - - def pre_tokenize(self, pretok): - """ - Pre-tokenize a :class:`~tokenizers.PyPreTokenizedString` in-place - - This method allows to modify a :class:`~tokenizers.PreTokenizedString` to - keep track of the pre-tokenization, and leverage the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you just want to see the result of - the pre-tokenization of a raw string, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize_str` - - Args: - pretok (:class:`~tokenizers.PreTokenizedString): - The pre-tokenized string on which to apply this - :class:`~tokenizers.pre_tokenizers.PreTokenizer` - """ - pass - - def pre_tokenize_str(self, sequence): - """ - Pre tokenize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.pre_tokenizers.PreTokenizer` but it does not keep track of the - alignment, nor does it provide all the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you need some of these, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize` - - Args: - sequence (:obj:`str`): - A string to pre-tokeize - - Returns: - :obj:`List[Tuple[str, Offsets]]`: - A list of tuple with the pre-tokenized parts and their offsets - """ - pass - -class Whitespace(PreTokenizer): - """ - This pre-tokenizer splits on word boundaries according to the `\w+|[^\w\s]+` - regex pattern. It splits on word characters or characters that aren't words or - whitespaces (punctuation such as hyphens, apostrophes, commas, etc.). - - Example: - Use the `Whitespace` function as shown below:: - - ```python - from tokenizers.pre_tokenizers import Whitespace - - pre_tokenizer = Whitespace() - text = "Hello, world! Let's try the Whitespace pre-tokenizer." - pre_tokenizer.pre_tokenize_str(text) - [('Hello', (0, 5)), - (',', (5, 6)), - ('world', (7, 12)), - ('!', (12, 13)), - ('Let', (14, 17)), - ("'", (17, 18)), - ('s', (18, 19)), - ('try', (20, 23)), - ('the', (24, 27)), - ('Whitespace', (28, 38)), - ('pre', (39, 42)), - ('-', (42, 43)), - ('tokenizer', (43, 52)), - ('.', (52, 53))] - ``` - """ - def __init__(self): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(pretok): - """ """ - pass - - def pre_tokenize(self, pretok): - """ - Pre-tokenize a :class:`~tokenizers.PyPreTokenizedString` in-place - - This method allows to modify a :class:`~tokenizers.PreTokenizedString` to - keep track of the pre-tokenization, and leverage the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you just want to see the result of - the pre-tokenization of a raw string, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize_str` - - Args: - pretok (:class:`~tokenizers.PreTokenizedString): - The pre-tokenized string on which to apply this - :class:`~tokenizers.pre_tokenizers.PreTokenizer` - """ - pass - - def pre_tokenize_str(self, sequence): - """ - Pre tokenize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.pre_tokenizers.PreTokenizer` but it does not keep track of the - alignment, nor does it provide all the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you need some of these, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize` - - Args: - sequence (:obj:`str`): - A string to pre-tokeize - - Returns: - :obj:`List[Tuple[str, Offsets]]`: - A list of tuple with the pre-tokenized parts and their offsets - """ - pass - -class WhitespaceSplit(PreTokenizer): - """ - This pre-tokenizer simply splits on the whitespace. Works like `.split()` - """ - def __init__(self): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @staticmethod - def custom(pretok): - """ """ - pass - - def pre_tokenize(self, pretok): - """ - Pre-tokenize a :class:`~tokenizers.PyPreTokenizedString` in-place - - This method allows to modify a :class:`~tokenizers.PreTokenizedString` to - keep track of the pre-tokenization, and leverage the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you just want to see the result of - the pre-tokenization of a raw string, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize_str` - - Args: - pretok (:class:`~tokenizers.PreTokenizedString): - The pre-tokenized string on which to apply this - :class:`~tokenizers.pre_tokenizers.PreTokenizer` - """ - pass - - def pre_tokenize_str(self, sequence): - """ - Pre tokenize the given string - - This method provides a way to visualize the effect of a - :class:`~tokenizers.pre_tokenizers.PreTokenizer` but it does not keep track of the - alignment, nor does it provide all the capabilities of the - :class:`~tokenizers.PreTokenizedString`. If you need some of these, you can use - :meth:`~tokenizers.pre_tokenizers.PreTokenizer.pre_tokenize` - - Args: - sequence (:obj:`str`): - A string to pre-tokeize - - Returns: - :obj:`List[Tuple[str, Offsets]]`: - A list of tuple with the pre-tokenized parts and their offsets - """ - pass diff --git a/bindings/python/py_src/tokenizers/processors.pyi b/bindings/python/py_src/tokenizers/processors.pyi new file mode 100644 index 0000000000..258c2dc091 --- /dev/null +++ b/bindings/python/py_src/tokenizers/processors.pyi @@ -0,0 +1,137 @@ +import tokenizers +import typing + +class BertProcessing: + def __getnewargs__(self, /) -> typing.Any: ... + def __new__(cls, /, sep: typing.Any, cls_token: typing.Any) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def cls(self, /) -> typing.Any: ... + @cls.setter + def cls(self, /, cls: typing.Any) -> typing.Any: ... + @property + def sep(self, /) -> typing.Any: ... + @sep.setter + def sep(self, /, sep: typing.Any) -> typing.Any: ... + +class ByteLevel: + def __new__( + cls, + /, + add_prefix_space: bool | None = None, + trim_offsets: bool | None = None, + use_regex: bool | None = None, + **_kwargs, + ) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def add_prefix_space(self, /) -> bool: ... + @add_prefix_space.setter + def add_prefix_space(self, /, add_prefix_space: bool) -> None: ... + @property + def trim_offsets(self, /) -> bool: ... + @trim_offsets.setter + def trim_offsets(self, /, trim_offsets: bool) -> None: ... + @property + def use_regex(self, /) -> bool: ... + @use_regex.setter + def use_regex(self, /, use_regex: bool) -> None: ... + +class PostProcessor: + def __getstate__(self, /) -> typing.Any: ... + def __repr__(self, /) -> str: + """Return repr(self).""" + ... + def __setstate__(self, /, state: typing.Any) -> typing.Any: ... + def __str__(self, /) -> str: + """Return str(self).""" + ... + def num_special_tokens_to_add(self, /, is_pair: bool) -> int: + """ + Return the number of special tokens that would be added for single/pair sentences. + + Args: + is_pair (:obj:`bool`): + Whether the input would be a pair of sequences + + Returns: + :obj:`int`: The number of tokens to add + """ + ... + def process( + self, + /, + encoding: tokenizers.Encoding, + pair: tokenizers.Encoding | None = None, + add_special_tokens: bool = True, + ) -> tokenizers.Encoding: + """ + Post-process the given encodings, generating the final one + + Args: + encoding (:class:`~tokenizers.Encoding`): + The encoding for the first sequence + + pair (:class:`~tokenizers.Encoding`, `optional`): + The encoding for the pair sequence + + add_special_tokens (:obj:`bool`): + Whether to add the special tokens + + Return: + :class:`~tokenizers.Encoding`: The final encoding + """ + ... + +class RobertaProcessing: + def __getnewargs__(self, /) -> typing.Any: ... + def __new__( + cls, /, sep: typing.Any, cls_token: typing.Any, trim_offsets: bool = True, add_prefix_space: bool = True + ) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def add_prefix_space(self, /) -> bool: ... + @add_prefix_space.setter + def add_prefix_space(self, /, add_prefix_space: bool) -> None: ... + @property + def cls(self, /) -> typing.Any: ... + @cls.setter + def cls(self, /, cls: typing.Any) -> typing.Any: ... + @property + def sep(self, /) -> typing.Any: ... + @sep.setter + def sep(self, /, sep: typing.Any) -> typing.Any: ... + @property + def trim_offsets(self, /) -> bool: ... + @trim_offsets.setter + def trim_offsets(self, /, trim_offsets: bool) -> None: ... + +class Sequence: + def __getitem__(self, /, index: int) -> typing.Any: + """Return self[key].""" + ... + def __getnewargs__(self, /) -> typing.Any: ... + def __new__(cls, /, processors_py: typing.Any) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + def __setitem__(self, /, index: int, value: typing.Any) -> typing.Any: + """Set self[key] to value.""" + ... + +class TemplateProcessing: + def __new__( + cls, + /, + single: typing.Any | None = None, + pair: typing.Any | None = None, + special_tokens: typing.Any | None = None, + ) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def single(self, /) -> str: ... + @single.setter + def single(self, /, single: typing.Any) -> typing.Any: ... diff --git a/bindings/python/py_src/tokenizers/processors/__init__.py b/bindings/python/py_src/tokenizers/processors/__init__.py index 06d124037b..9bc48012a5 100644 --- a/bindings/python/py_src/tokenizers/processors/__init__.py +++ b/bindings/python/py_src/tokenizers/processors/__init__.py @@ -1,9 +1,10 @@ # Generated content DO NOT EDIT + from .. import processors -PostProcessor = processors.PostProcessor BertProcessing = processors.BertProcessing ByteLevel = processors.ByteLevel +PostProcessor = processors.PostProcessor RobertaProcessing = processors.RobertaProcessing Sequence = processors.Sequence TemplateProcessing = processors.TemplateProcessing diff --git a/bindings/python/py_src/tokenizers/processors/__init__.pyi b/bindings/python/py_src/tokenizers/processors/__init__.pyi deleted file mode 100644 index 0d49520c63..0000000000 --- a/bindings/python/py_src/tokenizers/processors/__init__.pyi +++ /dev/null @@ -1,519 +0,0 @@ -# Generated content DO NOT EDIT -class PostProcessor: - """ - Base class for all post-processors - - This class is not supposed to be instantiated directly. Instead, any implementation of - a PostProcessor will return an instance of this class when instantiated. - """ - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - def num_special_tokens_to_add(self, is_pair): - """ - Return the number of special tokens that would be added for single/pair sentences. - - Args: - is_pair (:obj:`bool`): - Whether the input would be a pair of sequences - - Returns: - :obj:`int`: The number of tokens to add - """ - pass - - def process(self, encoding, pair=None, add_special_tokens=True): - """ - Post-process the given encodings, generating the final one - - Args: - encoding (:class:`~tokenizers.Encoding`): - The encoding for the first sequence - - pair (:class:`~tokenizers.Encoding`, `optional`): - The encoding for the pair sequence - - add_special_tokens (:obj:`bool`): - Whether to add the special tokens - - Return: - :class:`~tokenizers.Encoding`: The final encoding - """ - pass - -class BertProcessing(PostProcessor): - """ - This post-processor takes care of adding the special tokens needed by - a Bert model: - - - a SEP token - - a CLS token - - Args: - sep (:obj:`Tuple[str, int]`): - A tuple with the string representation of the SEP token, and its id - - cls (:obj:`Tuple[str, int]`): - A tuple with the string representation of the CLS token, and its id - """ - def __init__(self, sep, cls): - pass - - def __getnewargs__(self): - """ """ - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @property - def cls(self): - """ """ - pass - - @cls.setter - def cls(self, value): - """ """ - pass - - def num_special_tokens_to_add(self, is_pair): - """ - Return the number of special tokens that would be added for single/pair sentences. - - Args: - is_pair (:obj:`bool`): - Whether the input would be a pair of sequences - - Returns: - :obj:`int`: The number of tokens to add - """ - pass - - def process(self, encoding, pair=None, add_special_tokens=True): - """ - Post-process the given encodings, generating the final one - - Args: - encoding (:class:`~tokenizers.Encoding`): - The encoding for the first sequence - - pair (:class:`~tokenizers.Encoding`, `optional`): - The encoding for the pair sequence - - add_special_tokens (:obj:`bool`): - Whether to add the special tokens - - Return: - :class:`~tokenizers.Encoding`: The final encoding - """ - pass - - @property - def sep(self): - """ """ - pass - - @sep.setter - def sep(self, value): - """ """ - pass - -class ByteLevel(PostProcessor): - """ - This post-processor takes care of trimming the offsets. - - By default, the ByteLevel BPE might include whitespaces in the produced tokens. If you don't - want the offsets to include these whitespaces, then this PostProcessor must be used. - - Args: - trim_offsets (:obj:`bool`): - Whether to trim the whitespaces from the produced offsets. - - add_prefix_space (:obj:`bool`, `optional`, defaults to :obj:`True`): - If :obj:`True`, keeps the first token's offset as is. If :obj:`False`, increments - the start of the first token's offset by 1. Only has an effect if :obj:`trim_offsets` - is set to :obj:`True`. - """ - def __init__(self, add_prefix_space=None, trim_offsets=None, use_regex=None): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @property - def add_prefix_space(self): - """ """ - pass - - @add_prefix_space.setter - def add_prefix_space(self, value): - """ """ - pass - - def num_special_tokens_to_add(self, is_pair): - """ - Return the number of special tokens that would be added for single/pair sentences. - - Args: - is_pair (:obj:`bool`): - Whether the input would be a pair of sequences - - Returns: - :obj:`int`: The number of tokens to add - """ - pass - - def process(self, encoding, pair=None, add_special_tokens=True): - """ - Post-process the given encodings, generating the final one - - Args: - encoding (:class:`~tokenizers.Encoding`): - The encoding for the first sequence - - pair (:class:`~tokenizers.Encoding`, `optional`): - The encoding for the pair sequence - - add_special_tokens (:obj:`bool`): - Whether to add the special tokens - - Return: - :class:`~tokenizers.Encoding`: The final encoding - """ - pass - - @property - def trim_offsets(self): - """ """ - pass - - @trim_offsets.setter - def trim_offsets(self, value): - """ """ - pass - - @property - def use_regex(self): - """ """ - pass - - @use_regex.setter - def use_regex(self, value): - """ """ - pass - -class RobertaProcessing(PostProcessor): - """ - This post-processor takes care of adding the special tokens needed by - a Roberta model: - - - a SEP token - - a CLS token - - It also takes care of trimming the offsets. - By default, the ByteLevel BPE might include whitespaces in the produced tokens. If you don't - want the offsets to include these whitespaces, then this PostProcessor should be initialized - with :obj:`trim_offsets=True` - - Args: - sep (:obj:`Tuple[str, int]`): - A tuple with the string representation of the SEP token, and its id - - cls (:obj:`Tuple[str, int]`): - A tuple with the string representation of the CLS token, and its id - - trim_offsets (:obj:`bool`, `optional`, defaults to :obj:`True`): - Whether to trim the whitespaces from the produced offsets. - - add_prefix_space (:obj:`bool`, `optional`, defaults to :obj:`True`): - Whether the add_prefix_space option was enabled during pre-tokenization. This - is relevant because it defines the way the offsets are trimmed out. - """ - def __init__(self, sep, cls, trim_offsets=True, add_prefix_space=True): - pass - - def __getnewargs__(self): - """ """ - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - @property - def add_prefix_space(self): - """ """ - pass - - @add_prefix_space.setter - def add_prefix_space(self, value): - """ """ - pass - - @property - def cls(self): - """ """ - pass - - @cls.setter - def cls(self, value): - """ """ - pass - - def num_special_tokens_to_add(self, is_pair): - """ - Return the number of special tokens that would be added for single/pair sentences. - - Args: - is_pair (:obj:`bool`): - Whether the input would be a pair of sequences - - Returns: - :obj:`int`: The number of tokens to add - """ - pass - - def process(self, encoding, pair=None, add_special_tokens=True): - """ - Post-process the given encodings, generating the final one - - Args: - encoding (:class:`~tokenizers.Encoding`): - The encoding for the first sequence - - pair (:class:`~tokenizers.Encoding`, `optional`): - The encoding for the pair sequence - - add_special_tokens (:obj:`bool`): - Whether to add the special tokens - - Return: - :class:`~tokenizers.Encoding`: The final encoding - """ - pass - - @property - def sep(self): - """ """ - pass - - @sep.setter - def sep(self, value): - """ """ - pass - - @property - def trim_offsets(self): - """ """ - pass - - @trim_offsets.setter - def trim_offsets(self, value): - """ """ - pass - -class Sequence(PostProcessor): - """ - Sequence Processor - - Args: - processors (:obj:`List[PostProcessor]`) - The processors that need to be chained - """ - def __init__(self, processors): - pass - - def __getitem__(self, key): - """ - Return self[key]. - """ - pass - - def __getnewargs__(self): - """ """ - pass - - def __getstate__(self): - """ """ - pass - - def __setitem__(self, key, value): - """ - Set self[key] to value. - """ - pass - - def __setstate__(self, state): - """ """ - pass - - def num_special_tokens_to_add(self, is_pair): - """ - Return the number of special tokens that would be added for single/pair sentences. - - Args: - is_pair (:obj:`bool`): - Whether the input would be a pair of sequences - - Returns: - :obj:`int`: The number of tokens to add - """ - pass - - def process(self, encoding, pair=None, add_special_tokens=True): - """ - Post-process the given encodings, generating the final one - - Args: - encoding (:class:`~tokenizers.Encoding`): - The encoding for the first sequence - - pair (:class:`~tokenizers.Encoding`, `optional`): - The encoding for the pair sequence - - add_special_tokens (:obj:`bool`): - Whether to add the special tokens - - Return: - :class:`~tokenizers.Encoding`: The final encoding - """ - pass - -class TemplateProcessing(PostProcessor): - """ - Provides a way to specify templates in order to add the special tokens to each - input sequence as relevant. - - Let's take :obj:`BERT` tokenizer as an example. It uses two special tokens, used to - delimitate each sequence. :obj:`[CLS]` is always used at the beginning of the first - sequence, and :obj:`[SEP]` is added at the end of both the first, and the pair - sequences. The final result looks like this: - - - Single sequence: :obj:`[CLS] Hello there [SEP]` - - Pair sequences: :obj:`[CLS] My name is Anthony [SEP] What is my name? [SEP]` - - With the type ids as following:: - - [CLS] ... [SEP] ... [SEP] - 0 0 0 1 1 - - You can achieve such behavior using a TemplateProcessing:: - - TemplateProcessing( - single="[CLS] $0 [SEP]", - pair="[CLS] $A [SEP] $B:1 [SEP]:1", - special_tokens=[("[CLS]", 1), ("[SEP]", 0)], - ) - - In this example, each input sequence is identified using a ``$`` construct. This identifier - lets us specify each input sequence, and the type_id to use. When nothing is specified, - it uses the default values. Here are the different ways to specify it: - - - Specifying the sequence, with default ``type_id == 0``: ``$A`` or ``$B`` - - Specifying the `type_id` with default ``sequence == A``: ``$0``, ``$1``, ``$2``, ... - - Specifying both: ``$A:0``, ``$B:1``, ... - - The same construct is used for special tokens: ``(:)?``. - - **Warning**: You must ensure that you are giving the correct tokens/ids as these - will be added to the Encoding without any further check. If the given ids correspond - to something totally different in a `Tokenizer` using this `PostProcessor`, it - might lead to unexpected results. - - Args: - single (:obj:`Template`): - The template used for single sequences - - pair (:obj:`Template`): - The template used when both sequences are specified - - special_tokens (:obj:`Tokens`): - The list of special tokens used in each sequences - - Types: - - Template (:obj:`str` or :obj:`List`): - - If a :obj:`str` is provided, the whitespace is used as delimiter between tokens - - If a :obj:`List[str]` is provided, a list of tokens - - Tokens (:obj:`List[Union[Tuple[int, str], Tuple[str, int], dict]]`): - - A :obj:`Tuple` with both a token and its associated ID, in any order - - A :obj:`dict` with the following keys: - - "id": :obj:`str` => The special token id, as specified in the Template - - "ids": :obj:`List[int]` => The associated IDs - - "tokens": :obj:`List[str]` => The associated tokens - - The given dict expects the provided :obj:`ids` and :obj:`tokens` lists to have - the same length. - """ - def __init__(self, single=None, pair=None, special_tokens=None): - pass - - def __getstate__(self): - """ """ - pass - - def __setstate__(self, state): - """ """ - pass - - def num_special_tokens_to_add(self, is_pair): - """ - Return the number of special tokens that would be added for single/pair sentences. - - Args: - is_pair (:obj:`bool`): - Whether the input would be a pair of sequences - - Returns: - :obj:`int`: The number of tokens to add - """ - pass - - def process(self, encoding, pair=None, add_special_tokens=True): - """ - Post-process the given encodings, generating the final one - - Args: - encoding (:class:`~tokenizers.Encoding`): - The encoding for the first sequence - - pair (:class:`~tokenizers.Encoding`, `optional`): - The encoding for the pair sequence - - add_special_tokens (:obj:`bool`): - Whether to add the special tokens - - Return: - :class:`~tokenizers.Encoding`: The final encoding - """ - pass - - @property - def single(self): - """ """ - pass - - @single.setter - def single(self, value): - """ """ - pass diff --git a/bindings/python/py_src/tokenizers/py.typed b/bindings/python/py_src/tokenizers/py.typed new file mode 100644 index 0000000000..e69de29bb2 diff --git a/bindings/python/py_src/tokenizers/trainers.pyi b/bindings/python/py_src/tokenizers/trainers.pyi new file mode 100644 index 0000000000..687dcd710b --- /dev/null +++ b/bindings/python/py_src/tokenizers/trainers.pyi @@ -0,0 +1,131 @@ +import typing + +class BpeTrainer: + def __new__(cls, /, **kwargs) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def continuing_subword_prefix(self, /) -> typing.Any: ... + @continuing_subword_prefix.setter + def continuing_subword_prefix(self, /, prefix: str | None) -> None: ... + @property + def end_of_word_suffix(self, /) -> typing.Any: ... + @end_of_word_suffix.setter + def end_of_word_suffix(self, /, suffix: str | None) -> None: ... + @property + def initial_alphabet(self, /) -> typing.Any: ... + @initial_alphabet.setter + def initial_alphabet(self, /, alphabet: typing.Any) -> None: ... + @property + def limit_alphabet(self, /) -> typing.Any: ... + @limit_alphabet.setter + def limit_alphabet(self, /, limit: int | None) -> None: ... + @property + def max_token_length(self, /) -> typing.Any: ... + @max_token_length.setter + def max_token_length(self, /, limit: int | None) -> None: ... + @property + def min_frequency(self, /) -> int: ... + @min_frequency.setter + def min_frequency(self, /, freq: int) -> None: ... + @property + def show_progress(self, /) -> bool: ... + @show_progress.setter + def show_progress(self, /, show_progress: bool) -> None: ... + @property + def special_tokens(self, /) -> typing.Any: ... + @special_tokens.setter + def special_tokens(self, /, special_tokens: typing.Any) -> typing.Any: ... + @property + def vocab_size(self, /) -> int: ... + @vocab_size.setter + def vocab_size(self, /, vocab_size: int) -> None: ... + +class Trainer: + def __getstate__(self, /) -> typing.Any: ... + def __repr__(self, /) -> str: + """Return repr(self).""" + ... + def __setstate__(self, /, state: typing.Any) -> typing.Any: ... + def __str__(self, /) -> str: + """Return str(self).""" + ... + +class UnigramTrainer: + def __new__(cls, /, **kwargs) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def initial_alphabet(self, /) -> typing.Any: ... + @initial_alphabet.setter + def initial_alphabet(self, /, alphabet: typing.Any) -> None: ... + @property + def show_progress(self, /) -> bool: ... + @show_progress.setter + def show_progress(self, /, show_progress: bool) -> None: ... + @property + def special_tokens(self, /) -> typing.Any: ... + @special_tokens.setter + def special_tokens(self, /, special_tokens: typing.Any) -> typing.Any: ... + @property + def vocab_size(self, /) -> int: ... + @vocab_size.setter + def vocab_size(self, /, vocab_size: int) -> None: ... + +class WordLevelTrainer: + def __new__(cls, /, **kwargs) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def min_frequency(self, /) -> int: ... + @min_frequency.setter + def min_frequency(self, /, freq: int) -> None: ... + @property + def show_progress(self, /) -> bool: ... + @show_progress.setter + def show_progress(self, /, show_progress: bool) -> None: ... + @property + def special_tokens(self, /) -> typing.Any: ... + @special_tokens.setter + def special_tokens(self, /, special_tokens: typing.Any) -> typing.Any: ... + @property + def vocab_size(self, /) -> int: ... + @vocab_size.setter + def vocab_size(self, /, vocab_size: int) -> None: ... + +class WordPieceTrainer: + def __new__(cls, /, **kwargs) -> None: + """Create and return a new object. See help(type) for accurate signature.""" + ... + @property + def continuing_subword_prefix(self, /) -> typing.Any: ... + @continuing_subword_prefix.setter + def continuing_subword_prefix(self, /, prefix: str | None) -> None: ... + @property + def end_of_word_suffix(self, /) -> typing.Any: ... + @end_of_word_suffix.setter + def end_of_word_suffix(self, /, suffix: str | None) -> None: ... + @property + def initial_alphabet(self, /) -> typing.Any: ... + @initial_alphabet.setter + def initial_alphabet(self, /, alphabet: typing.Any) -> None: ... + @property + def limit_alphabet(self, /) -> typing.Any: ... + @limit_alphabet.setter + def limit_alphabet(self, /, limit: int | None) -> None: ... + @property + def min_frequency(self, /) -> int: ... + @min_frequency.setter + def min_frequency(self, /, freq: int) -> None: ... + @property + def show_progress(self, /) -> bool: ... + @show_progress.setter + def show_progress(self, /, show_progress: bool) -> None: ... + @property + def special_tokens(self, /) -> typing.Any: ... + @special_tokens.setter + def special_tokens(self, /, special_tokens: typing.Any) -> typing.Any: ... + @property + def vocab_size(self, /) -> int: ... + @vocab_size.setter + def vocab_size(self, /, vocab_size: int) -> None: ... diff --git a/bindings/python/py_src/tokenizers/trainers/__init__.py b/bindings/python/py_src/tokenizers/trainers/__init__.py index 22f94c50b7..959dd42eaa 100644 --- a/bindings/python/py_src/tokenizers/trainers/__init__.py +++ b/bindings/python/py_src/tokenizers/trainers/__init__.py @@ -1,8 +1,9 @@ # Generated content DO NOT EDIT + from .. import trainers -Trainer = trainers.Trainer BpeTrainer = trainers.BpeTrainer +Trainer = trainers.Trainer UnigramTrainer = trainers.UnigramTrainer WordLevelTrainer = trainers.WordLevelTrainer WordPieceTrainer = trainers.WordPieceTrainer diff --git a/bindings/python/pyproject.toml b/bindings/python/pyproject.toml index b10806929e..83bebe756e 100644 --- a/bindings/python/pyproject.toml +++ b/bindings/python/pyproject.toml @@ -64,3 +64,6 @@ lint.ignore = [ # Fixtures unused import "F811", ] + +[tool.ty.rules] +invalid-method-override = "ignore" \ No newline at end of file diff --git a/bindings/python/src/decoders.rs b/bindings/python/src/decoders.rs index 2ac1e49ec8..b88b2b9e67 100644 --- a/bindings/python/src/decoders.rs +++ b/bindings/python/src/decoders.rs @@ -98,7 +98,7 @@ impl PyDecoder { })?; Ok(()) } - Err(e) => Err(e), + Err(e) => Err(e.into()), } } @@ -581,20 +581,31 @@ impl Decoder for PyDecoderWrapper { /// Decoders Module #[pymodule] -pub fn decoders(m: &Bound<'_, PyModule>) -> PyResult<()> { - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - Ok(()) +pub mod decoders { + #[pymodule_export] + pub use super::PyBPEDecoder; + #[pymodule_export] + pub use super::PyByteFallbackDec; + #[pymodule_export] + pub use super::PyByteLevelDec; + #[pymodule_export] + pub use super::PyCTCDecoder; + #[pymodule_export] + pub use super::PyDecodeStream; + #[pymodule_export] + pub use super::PyDecoder; + #[pymodule_export] + pub use super::PyFuseDec; + #[pymodule_export] + pub use super::PyMetaspaceDec; + #[pymodule_export] + pub use super::PyReplaceDec; + #[pymodule_export] + pub use super::PySequenceDecoder; + #[pymodule_export] + pub use super::PyStrip; + #[pymodule_export] + pub use super::PyWordPieceDec; } /// Class needed for streaming decode @@ -630,8 +641,10 @@ enum StreamInput { Ids(Vec), } -impl FromPyObject<'_> for StreamInput { - fn extract_bound(obj: &Bound<'_, PyAny>) -> PyResult { +impl<'a, 'py> FromPyObject<'a, 'py> for StreamInput { + type Error = PyErr; + + fn extract(obj: Borrowed<'a, 'py, PyAny>) -> Result { if let Ok(id) = obj.extract::() { Ok(StreamInput::Id(id)) } else if let Ok(ids) = obj.extract::>() { diff --git a/bindings/python/src/encoding.rs b/bindings/python/src/encoding.rs index 7261bbbbc1..c4657924d3 100644 --- a/bindings/python/src/encoding.rs +++ b/bindings/python/src/encoding.rs @@ -49,7 +49,7 @@ impl PyEncoding { })?; Ok(()) } - Err(e) => Err(e), + Err(e) => Err(e.into()), } } @@ -77,7 +77,7 @@ impl PyEncoding { /// Returns: /// :class:`~tokenizers.Encoding`: The resulting Encoding #[staticmethod] - #[pyo3(signature = (encodings, growing_offsets = true))] + #[pyo3(signature = (encodings, growing_offsets = true) -> "Encoding")] #[pyo3(text_signature = "(encodings, growing_offsets=True)")] fn merge(encodings: Vec>, growing_offsets: bool) -> PyEncoding { tk::tokenizer::Encoding::merge( @@ -385,7 +385,7 @@ impl PyEncoding { /// /// pad_token (:obj:`str`, defaults to `[PAD]`): /// The pad token to use - #[pyo3(signature = (length, **kwargs))] + #[pyo3(signature = (length, **kwargs) -> "None")] #[pyo3( text_signature = "(self, length, direction='right', pad_id=0, pad_type_id=0, pad_token='[PAD]')" )] @@ -437,7 +437,7 @@ impl PyEncoding { /// /// direction (:obj:`str`, defaults to :obj:`right`): /// Truncate direction - #[pyo3(signature = (max_length, stride = 0, direction = "right"))] + #[pyo3(signature = (max_length, stride = 0, direction = "right") -> "None")] #[pyo3(text_signature = "(self, max_length, stride=0, direction='right')")] fn truncate(&mut self, max_length: usize, stride: usize, direction: &str) -> PyResult<()> { let tdir = match direction { diff --git a/bindings/python/src/lib.rs b/bindings/python/src/lib.rs index 1aad0755cf..086112f79d 100644 --- a/bindings/python/src/lib.rs +++ b/bindings/python/src/lib.rs @@ -31,8 +31,6 @@ mod trainers; mod utils; use pyo3::prelude::*; -use pyo3::wrap_pymodule; - pub const VERSION: &str = env!("CARGO_PKG_VERSION"); // For users using multiprocessing in python, it is quite easy to fork the process running @@ -50,31 +48,54 @@ extern "C" fn child_after_fork() { /// Tokenizers Module #[pymodule] -pub fn tokenizers(m: &Bound<'_, PyModule>) -> PyResult<()> { - let _ = env_logger::try_init_from_env("TOKENIZERS_LOG"); +pub mod tokenizers { + use super::*; + + #[pymodule_export] + pub use super::encoding::PyEncoding; + #[pymodule_export] + pub use super::token::PyToken; + #[pymodule_export] + pub use super::tokenizer::PyAddedToken; + #[pymodule_export] + pub use super::tokenizer::PyTokenizer; + #[pymodule_export] + pub use super::utils::PyNormalizedString; + #[pymodule_export] + pub use super::utils::PyPreTokenizedString; + #[pymodule_export] + pub use super::utils::PyRegex; + + #[pymodule_export] + pub use super::decoders::decoders; + #[pymodule_export] + pub use super::models::models; + #[pymodule_export] + pub use super::normalizers::normalizers; + #[pymodule_export] + pub use super::pre_tokenizers::pre_tokenizers; + #[pymodule_export] + pub use super::processors::processors; + #[pymodule_export] + pub use super::trainers::trainers; + + #[allow(non_upper_case_globals)] + #[pymodule_export] + pub const __version__: &str = env!("CARGO_PKG_VERSION"); - // Register the fork callback - #[cfg(target_family = "unix")] - unsafe { - if !REGISTERED_FORK_CALLBACK { - libc::pthread_atfork(None, None, Some(child_after_fork)); - REGISTERED_FORK_CALLBACK = true; + #[pymodule_init] + fn init(_m: &Bound<'_, PyModule>) -> PyResult<()> { + let _ = env_logger::try_init_from_env("TOKENIZERS_LOG"); + + // Register the fork callback + #[cfg(target_family = "unix")] + unsafe { + if !REGISTERED_FORK_CALLBACK { + libc::pthread_atfork(None, None, Some(child_after_fork)); + REGISTERED_FORK_CALLBACK = true; + } } - } - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_wrapped(wrap_pymodule!(models::models))?; - m.add_wrapped(wrap_pymodule!(pre_tokenizers::pre_tokenizers))?; - m.add_wrapped(wrap_pymodule!(decoders::decoders))?; - m.add_wrapped(wrap_pymodule!(processors::processors))?; - m.add_wrapped(wrap_pymodule!(normalizers::normalizers))?; - m.add_wrapped(wrap_pymodule!(trainers::trainers))?; - m.add("__version__", env!("CARGO_PKG_VERSION"))?; - Ok(()) + Ok(()) + } } diff --git a/bindings/python/src/models.rs b/bindings/python/src/models.rs index d093fa1424..068e952507 100644 --- a/bindings/python/src/models.rs +++ b/bindings/python/src/models.rs @@ -90,7 +90,7 @@ where #[pymethods] impl PyModel { #[new] - #[pyo3(signature = (), text_signature = "(self)")] + #[pyo3(signature = () -> "Model", text_signature = "(self)")] fn __new__() -> Self { // Instantiate a default empty model. This doesn't really make sense, but we need // to be able to instantiate an empty model for pickle capabilities. @@ -116,7 +116,7 @@ impl PyModel { })?; Ok(()) } - Err(e) => Err(e), + Err(e) => Err(e.into()), } } @@ -178,7 +178,7 @@ impl PyModel { /// /// Returns: /// :obj:`List[str]`: The list of saved files - #[pyo3(signature = (folder, prefix=None, name=None), text_signature = "(self, folder, prefix)")] + #[pyo3(signature = (folder, prefix=None, name=None) -> "list[str]", text_signature = "(self, folder, prefix)")] fn save<'a>( &self, py: Python<'_>, @@ -506,7 +506,7 @@ impl PyBPE { /// Returns: /// :class:`~tokenizers.models.BPE`: An instance of BPE loaded from these files #[classmethod] - #[pyo3(signature = (vocab, merges, **kwargs))] + #[pyo3(signature = (vocab, merges, **kwargs) -> "BPE")] #[pyo3(text_signature = "(vocab, merges, **kwargs)")] fn from_file( _cls: &Bound<'_, PyType>, @@ -531,7 +531,7 @@ impl PyBPE { } /// Clears the internal cache - #[pyo3(signature = ())] + #[pyo3(signature = () -> "None")] #[pyo3(text_signature = "(self)")] fn _clear_cache(self_: PyRef) -> PyResult<()> { let super_ = self_.as_ref(); @@ -543,7 +543,7 @@ impl PyBPE { } /// Resize the internal cache - #[pyo3(signature = (capacity))] + #[pyo3(signature = (capacity) -> "None")] #[pyo3(text_signature = "(self, capacity)")] fn _resize_cache(self_: PyRef, capacity: usize) -> PyResult<()> { let super_ = self_.as_ref(); @@ -705,7 +705,7 @@ impl PyWordPiece { /// Returns: /// :class:`~tokenizers.models.WordPiece`: An instance of WordPiece loaded from file #[classmethod] - #[pyo3(signature = (vocab, **kwargs))] + #[pyo3(signature = (vocab, **kwargs) -> "WordPiece")] #[pyo3(text_signature = "(vocab, **kwargs)")] fn from_file( _cls: &Bound<'_, PyType>, @@ -825,7 +825,7 @@ impl PyWordLevel { /// Returns: /// :class:`~tokenizers.models.WordLevel`: An instance of WordLevel loaded from file #[classmethod] - #[pyo3(signature = (vocab, unk_token = None))] + #[pyo3(signature = (vocab, unk_token = None)-> "WordLevel")] #[pyo3(text_signature = "(vocab, unk_token=None)")] fn from_file( _cls: &Bound<'_, PyType>, @@ -879,7 +879,7 @@ impl PyUnigram { } /// Clears the internal cache - #[pyo3(signature = ())] + #[pyo3(signature = () -> "None")] #[pyo3(text_signature = "(self)")] fn _clear_cache(self_: PyRef) -> PyResult<()> { let super_ = self_.as_ref(); @@ -891,7 +891,7 @@ impl PyUnigram { } /// Resize the internal cache - #[pyo3(signature = (capacity))] + #[pyo3(signature = (capacity) -> "None")] #[pyo3(text_signature = "(self, capacity)")] fn _resize_cache(self_: PyRef, capacity: usize) -> PyResult<()> { let super_ = self_.as_ref(); @@ -905,13 +905,17 @@ impl PyUnigram { /// Models Module #[pymodule] -pub fn models(m: &Bound<'_, PyModule>) -> PyResult<()> { - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - Ok(()) +pub mod models { + #[pymodule_export] + pub use super::PyBPE; + #[pymodule_export] + pub use super::PyModel; + #[pymodule_export] + pub use super::PyUnigram; + #[pymodule_export] + pub use super::PyWordLevel; + #[pymodule_export] + pub use super::PyWordPiece; } #[cfg(test)] diff --git a/bindings/python/src/normalizers.rs b/bindings/python/src/normalizers.rs index 3c255a5e62..2891682c06 100644 --- a/bindings/python/src/normalizers.rs +++ b/bindings/python/src/normalizers.rs @@ -135,7 +135,7 @@ impl PyNormalizer { })?; Ok(()) } - Err(e) => Err(e), + Err(e) => Err(e.into()), } } @@ -765,23 +765,37 @@ impl Normalizer for PyNormalizerWrapper { /// Normalizers Module #[pymodule] -pub fn normalizers(m: &Bound<'_, PyModule>) -> PyResult<()> { - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - Ok(()) +pub mod normalizers { + #[pymodule_export] + pub use super::PyBertNormalizer; + #[pymodule_export] + pub use super::PyByteLevel; + #[pymodule_export] + pub use super::PyLowercase; + #[pymodule_export] + pub use super::PyNFC; + #[pymodule_export] + pub use super::PyNFD; + #[pymodule_export] + pub use super::PyNFKC; + #[pymodule_export] + pub use super::PyNFKD; + #[pymodule_export] + pub use super::PyNmt; + #[pymodule_export] + pub use super::PyNormalizer; + #[pymodule_export] + pub use super::PyPrecompiled; + #[pymodule_export] + pub use super::PyPrepend; + #[pymodule_export] + pub use super::PyReplace; + #[pymodule_export] + pub use super::PySequence; + #[pymodule_export] + pub use super::PyStrip; + #[pymodule_export] + pub use super::PyStripAccents; } #[cfg(test)] diff --git a/bindings/python/src/pre_tokenizers.rs b/bindings/python/src/pre_tokenizers.rs index d71ec7b332..2b124d9c08 100644 --- a/bindings/python/src/pre_tokenizers.rs +++ b/bindings/python/src/pre_tokenizers.rs @@ -142,7 +142,7 @@ impl PyPreTokenizer { self.pretok = unpickled; Ok(()) } - Err(e) => Err(e), + Err(e) => Err(e.into()), } } @@ -954,21 +954,33 @@ impl PreTokenizer for PyPreTokenizerWrapper { /// PreTokenizers Module #[pymodule] -pub fn pre_tokenizers(m: &Bound<'_, PyModule>) -> PyResult<()> { - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - Ok(()) +pub mod pre_tokenizers { + #[pymodule_export] + pub use super::PyBertPreTokenizer; + #[pymodule_export] + pub use super::PyByteLevel; + #[pymodule_export] + pub use super::PyCharDelimiterSplit; + #[pymodule_export] + pub use super::PyDigits; + #[pymodule_export] + pub use super::PyFixedLength; + #[pymodule_export] + pub use super::PyMetaspace; + #[pymodule_export] + pub use super::PyPreTokenizer; + #[pymodule_export] + pub use super::PyPunctuation; + #[pymodule_export] + pub use super::PySequence; + #[pymodule_export] + pub use super::PySplit; + #[pymodule_export] + pub use super::PyUnicodeScripts; + #[pymodule_export] + pub use super::PyWhitespace; + #[pymodule_export] + pub use super::PyWhitespaceSplit; } #[cfg(test)] diff --git a/bindings/python/src/processors.rs b/bindings/python/src/processors.rs index 3973bd7952..75b5dee84c 100644 --- a/bindings/python/src/processors.rs +++ b/bindings/python/src/processors.rs @@ -117,7 +117,7 @@ impl PyPostProcessor { })?; Ok(()) } - Err(e) => Err(e), + Err(e) => Err(e.into()), } } @@ -148,7 +148,9 @@ impl PyPostProcessor { /// /// Return: /// :class:`~tokenizers.Encoding`: The final encoding - #[pyo3(signature = (encoding, pair = None, add_special_tokens = true))] + #[pyo3( + signature = (encoding, pair = None, add_special_tokens = true) -> "tokenizers.Encoding" + )] #[pyo3(text_signature = "(self, encoding, pair=None, add_special_tokens=True)")] fn process( &self, @@ -319,9 +321,12 @@ pub struct PyBertProcessing {} #[pymethods] impl PyBertProcessing { #[new] - #[pyo3(text_signature = "(self, sep, cls)")] - fn new(sep: (String, u32), cls: (String, u32)) -> (Self, PyPostProcessor) { - (PyBertProcessing {}, BertProcessing::new(sep, cls).into()) + #[pyo3(text_signature = "(self, sep, cls_token: str| int)")] + fn new(sep: (String, u32), cls_token: (String, u32)) -> (Self, PyPostProcessor) { + ( + PyBertProcessing {}, + BertProcessing::new(sep, cls_token).into(), + ) } fn __getnewargs__<'p>(&self, py: Python<'p>) -> PyResult> { @@ -392,14 +397,17 @@ pub struct PyRobertaProcessing {} #[pymethods] impl PyRobertaProcessing { #[new] - #[pyo3(signature = (sep, cls, trim_offsets = true, add_prefix_space = true), text_signature = "(self, sep, cls, trim_offsets=True, add_prefix_space=True)")] + #[pyo3( + signature = (sep, cls_token, trim_offsets = true, add_prefix_space = true), + text_signature = "(self, sep, cls_token, trim_offsets=True, add_prefix_space=True)" + )] fn new( sep: (String, u32), - cls: (String, u32), + cls_token: (String, u32), trim_offsets: bool, add_prefix_space: bool, ) -> (Self, PyPostProcessor) { - let proc = RobertaProcessing::new(sep, cls) + let proc = RobertaProcessing::new(sep, cls_token) .trim_offsets(trim_offsets) .add_prefix_space(add_prefix_space); (PyRobertaProcessing {}, proc.into()) @@ -549,13 +557,15 @@ impl From for SpecialToken { } } -impl FromPyObject<'_> for PySpecialToken { - fn extract_bound(ob: &Bound<'_, PyAny>) -> PyResult { +impl<'a, 'py> FromPyObject<'a, 'py> for PySpecialToken { + type Error = PyErr; + + fn extract(ob: Borrowed<'a, 'py, PyAny>) -> Result { if let Ok(v) = ob.extract::<(String, u32)>() { Ok(Self(v.into())) } else if let Ok(v) = ob.extract::<(u32, String)>() { Ok(Self(v.into())) - } else if let Ok(d) = ob.downcast::() { + } else if let Ok(d) = ob.cast::() { let id = d .get_item("id")? .ok_or_else(|| exceptions::PyValueError::new_err("`id` must be specified"))? @@ -589,8 +599,10 @@ impl From for Template { } } -impl FromPyObject<'_> for PyTemplate { - fn extract_bound(ob: &Bound<'_, PyAny>) -> PyResult { +impl<'a, 'py> FromPyObject<'a, 'py> for PyTemplate { + type Error = PyErr; + + fn extract(ob: Borrowed<'a, 'py, PyAny>) -> Result { if let Ok(s) = ob.extract::() { Ok(Self( s.try_into().map_err(exceptions::PyValueError::new_err)?, @@ -807,14 +819,19 @@ impl PySequence { /// Processors Module #[pymodule] -pub fn processors(m: &Bound<'_, PyModule>) -> PyResult<()> { - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - Ok(()) +pub mod processors { + #[pymodule_export] + pub use super::PyBertProcessing; + #[pymodule_export] + pub use super::PyByteLevel; + #[pymodule_export] + pub use super::PyPostProcessor; + #[pymodule_export] + pub use super::PyRobertaProcessing; + #[pymodule_export] + pub use super::PySequence; + #[pymodule_export] + pub use super::PyTemplateProcessing; } #[cfg(test)] diff --git a/bindings/python/src/tokenizer.rs b/bindings/python/src/tokenizer.rs index 0cd06985ca..ddbfeaa30c 100644 --- a/bindings/python/src/tokenizer.rs +++ b/bindings/python/src/tokenizer.rs @@ -160,7 +160,7 @@ impl PyAddedToken { } fn __setstate__(&mut self, py: Python, state: Py) -> PyResult<()> { - match state.downcast_bound::(py) { + match state.cast_bound::(py) { Ok(state) => { for (key, value) in state { let key: String = key.extract()?; @@ -266,8 +266,10 @@ impl PyAddedToken { } struct TextInputSequence<'s>(tk::InputSequence<'s>); -impl<'s> FromPyObject<'s> for TextInputSequence<'s> { - fn extract_bound(ob: &Bound<'s, PyAny>) -> PyResult { +impl<'a, 'py> FromPyObject<'a, 'py> for TextInputSequence<'py> { + type Error = PyErr; + + fn extract(ob: Borrowed<'a, 'py, PyAny>) -> Result { let err = exceptions::PyTypeError::new_err("TextInputSequence must be str"); if let Ok(s) = ob.extract::() { Ok(Self(s.into())) @@ -283,8 +285,10 @@ impl<'s> From> for tk::InputSequence<'s> { } struct PyArrayUnicode(Vec); -impl FromPyObject<'_> for PyArrayUnicode { - fn extract_bound(ob: &Bound<'_, PyAny>) -> PyResult { +impl<'a, 'py> FromPyObject<'a, 'py> for PyArrayUnicode { + type Error = PyErr; + + fn extract(ob: Borrowed<'a, 'py, PyAny>) -> Result { // SAFETY Making sure the pointer is a valid numpy array requires calling numpy C code if unsafe { npyffi::PyArray_Check(ob.py(), ob.as_ptr()) } == 0 { return Err(exceptions::PyTypeError::new_err("Expected an np.array")); @@ -352,15 +356,17 @@ impl From for tk::InputSequence<'_> { struct PyArrayStr(Vec); -impl FromPyObject<'_> for PyArrayStr { - fn extract_bound(ob: &Bound<'_, PyAny>) -> PyResult { - let array = ob.downcast::>>()?; +impl<'a, 'py> FromPyObject<'a, 'py> for PyArrayStr { + type Error = PyErr; + + fn extract(ob: Borrowed<'a, 'py, PyAny>) -> Result { + let array = ob.cast::>>()?; let seq = array .readonly() .as_array() .iter() .map(|obj| { - let s = obj.downcast_bound::(ob.py())?; + let s = obj.cast_bound::(ob.py())?; Ok(s.to_string_lossy().into_owned()) }) .collect::>>()?; @@ -375,20 +381,22 @@ impl From for tk::InputSequence<'_> { } struct PreTokenizedInputSequence<'s>(tk::InputSequence<'s>); -impl<'s> FromPyObject<'s> for PreTokenizedInputSequence<'s> { - fn extract_bound(ob: &Bound<'s, PyAny>) -> PyResult { +impl<'a, 'py> FromPyObject<'a, 'py> for PreTokenizedInputSequence<'py> { + type Error = PyErr; + + fn extract(ob: Borrowed<'a, 'py, PyAny>) -> Result { if let Ok(seq) = ob.extract::() { return Ok(Self(seq.into())); } if let Ok(seq) = ob.extract::() { return Ok(Self(seq.into())); } - if let Ok(s) = ob.downcast::() { + if let Ok(s) = ob.cast::() { if let Ok(seq) = s.extract::>() { return Ok(Self(seq.into())); } } - if let Ok(s) = ob.downcast::() { + if let Ok(s) = ob.cast::() { if let Ok(seq) = s.extract::>() { return Ok(Self(seq.into())); } @@ -405,18 +413,21 @@ impl<'s> From> for tk::InputSequence<'s> { } struct TextEncodeInput<'s>(tk::EncodeInput<'s>); -impl<'s> FromPyObject<'s> for TextEncodeInput<'s> { - fn extract_bound(ob: &Bound<'s, PyAny>) -> PyResult { +impl<'a, 'py> FromPyObject<'a, 'py> for TextEncodeInput<'py> { + type Error = PyErr; + + fn extract(ob: Borrowed<'a, 'py, PyAny>) -> Result { if let Ok(i) = ob.extract::() { return Ok(Self(i.into())); } if let Ok((i1, i2)) = ob.extract::<(TextInputSequence, TextInputSequence)>() { return Ok(Self((i1, i2).into())); } - if let Ok(arr) = ob.extract::>>() { + if let Ok(arr) = ob.extract::>>() { if arr.len() == 2 { - let first = arr[0].extract::()?; - let second = arr[1].extract::()?; + let py = ob.py(); + let first = arr[0].bind(py).extract::()?; + let second = arr[1].bind(py).extract::()?; return Ok(Self((first, second).into())); } } @@ -431,8 +442,10 @@ impl<'s> From> for tk::tokenizer::EncodeInput<'s> { } } struct PreTokenizedEncodeInput<'s>(tk::EncodeInput<'s>); -impl<'s> FromPyObject<'s> for PreTokenizedEncodeInput<'s> { - fn extract_bound(ob: &Bound<'s, PyAny>) -> PyResult { +impl<'a, 'py> FromPyObject<'a, 'py> for PreTokenizedEncodeInput<'py> { + type Error = PyErr; + + fn extract(ob: Borrowed<'a, 'py, PyAny>) -> Result { if let Ok(i) = ob.extract::() { return Ok(Self(i.into())); } @@ -440,10 +453,11 @@ impl<'s> FromPyObject<'s> for PreTokenizedEncodeInput<'s> { { return Ok(Self((i1, i2).into())); } - if let Ok(arr) = ob.extract::>>() { + if let Ok(arr) = ob.extract::>>() { if arr.len() == 2 { - let first = arr[0].extract::()?; - let second = arr[1].extract::()?; + let py = ob.py(); + let first = arr[0].bind(py).extract::()?; + let second = arr[1].bind(py).extract::()?; return Ok(Self((first, second).into())); } } @@ -492,10 +506,10 @@ impl PyTokenizer { if let Ok(seq) = ob.extract::() { return Ok(seq.0); } - if let Ok(list) = ob.downcast::() { + if let Ok(list) = ob.cast::() { return list.extract::>(); } - if let Ok(tup) = ob.downcast::() { + if let Ok(tup) = ob.cast::() { return tup.extract::>(); } Err(exceptions::PyTypeError::new_err( @@ -513,7 +527,7 @@ impl PyTokenizer { for it in items { if is_pretokenized { // Pair? - if let Ok(tup) = it.downcast::() { + if let Ok(tup) = it.cast::() { if tup.len() == 2 { let a = Self::extract_pretok_seq(&tup.get_item(0)?)?; let b = Self::extract_pretok_seq(&tup.get_item(1)?)?; @@ -521,7 +535,7 @@ impl PyTokenizer { continue; } } - if let Ok(lst) = it.downcast::() { + if let Ok(lst) = it.cast::() { if lst.len() == 2 { let a = Self::extract_pretok_seq(&lst.get_item(0)?)?; let b = Self::extract_pretok_seq(&lst.get_item(1)?)?; @@ -534,7 +548,7 @@ impl PyTokenizer { out.push(tk::EncodeInput::Single(a.into())); } else { // Raw text: pair? - if let Ok(tup) = it.downcast::() { + if let Ok(tup) = it.cast::() { if tup.len() == 2 { let a: String = tup.get_item(0)?.extract()?; let b: String = tup.get_item(1)?.extract()?; @@ -542,10 +556,10 @@ impl PyTokenizer { continue; } } - if let Ok(lst) = it.downcast::() { + if let Ok(lst) = it.cast::() { if lst.len() == 2 - && lst.get_item(0)?.downcast::().is_ok() - && lst.get_item(1)?.downcast::().is_ok() + && lst.get_item(0)?.cast::().is_ok() + && lst.get_item(1)?.cast::().is_ok() { let a: String = lst.get_item(0)?.extract()?; let b: String = lst.get_item(1)?.extract()?; @@ -617,7 +631,7 @@ impl PyTokenizer { })?; Ok(()) } - Err(e) => Err(e), + Err(e) => Err(e.into()), } } @@ -639,6 +653,7 @@ impl PyTokenizer { /// Returns: /// :class:`~tokenizers.Tokenizer`: The new tokenizer #[staticmethod] + #[pyo3(signature = (json) -> "Tokenizer")] #[pyo3(text_signature = "(json)")] fn from_str(json: &str) -> PyResult { let tokenizer: PyResult<_> = ToPyResult(json.parse()).into(); @@ -655,6 +670,7 @@ impl PyTokenizer { /// Returns: /// :class:`~tokenizers.Tokenizer`: The new tokenizer #[staticmethod] + #[pyo3(signature = (path) -> "Tokenizer")] #[pyo3(text_signature = "(path)")] fn from_file(path: &str) -> PyResult { let tokenizer: PyResult<_> = ToPyResult(Tokenizer::from_file(path)).into(); @@ -670,6 +686,7 @@ impl PyTokenizer { /// Returns: /// :class:`~tokenizers.Tokenizer`: The new tokenizer #[staticmethod] + #[pyo3(signature = (buffer) -> "Tokenizer")] #[pyo3(text_signature = "(buffer)")] fn from_buffer(buffer: &Bound<'_, PyBytes>) -> PyResult { let tokenizer = serde_json::from_slice(buffer.as_bytes()).map_err(|e| { @@ -696,7 +713,7 @@ impl PyTokenizer { /// Returns: /// :class:`~tokenizers.Tokenizer`: The new tokenizer #[staticmethod] - #[pyo3(signature = (identifier, revision = String::from("main"), token = None))] + #[pyo3(signature = (identifier, revision = String::from("main"), token = None) -> "Tokenizer")] #[pyo3(text_signature = "(identifier, revision=\"main\", token=None)")] fn from_pretrained( identifier: &str, @@ -731,7 +748,7 @@ impl PyTokenizer { /// /// Returns: /// :obj:`str`: A string representing the serialized Tokenizer - #[pyo3(signature = (pretty = false))] + #[pyo3(signature = (pretty = false) -> "str")] #[pyo3(text_signature = "(self, pretty=False)")] fn to_str(&self, pretty: bool) -> PyResult { ToPyResult(self.tokenizer.to_string(pretty)).into() @@ -745,7 +762,7 @@ impl PyTokenizer { /// /// pretty (:obj:`bool`, defaults to :obj:`True`): /// Whether the JSON file should be pretty formatted. - #[pyo3(signature = (path, pretty = true))] + #[pyo3(signature = (path, pretty = true) -> "None")] #[pyo3(text_signature = "(self, path, pretty=True)")] fn save(&self, path: &str, pretty: bool) -> PyResult<()> { ToPyResult(self.tokenizer.save(path, pretty)).into() @@ -779,7 +796,7 @@ impl PyTokenizer { /// /// Returns: /// :obj:`Dict[str, int]`: The vocabulary - #[pyo3(signature = (with_added_tokens = true))] + #[pyo3(signature = (with_added_tokens = true) -> "dict[str, int]")] #[pyo3(text_signature = "(self, with_added_tokens=True)")] fn get_vocab(&self, with_added_tokens: bool) -> HashMap { self.tokenizer.get_vocab(with_added_tokens) @@ -789,7 +806,7 @@ impl PyTokenizer { /// /// Returns: /// :obj:`Dict[int, AddedToken]`: The vocabulary - #[pyo3(signature = ())] + #[pyo3(signature = () -> "dict[int, AddedToken]")] #[pyo3(text_signature = "(self)")] fn get_added_tokens_decoder(&self) -> BTreeMap { let mut sorted_map = BTreeMap::new(); @@ -809,7 +826,7 @@ impl PyTokenizer { /// /// Returns: /// :obj:`int`: The size of the vocabulary - #[pyo3(signature = (with_added_tokens = true))] + #[pyo3(signature = (with_added_tokens = true) -> "int")] #[pyo3(text_signature = "(self, with_added_tokens=True)")] fn get_vocab_size(&self, with_added_tokens: bool) -> usize { self.tokenizer.get_vocab_size(with_added_tokens) @@ -831,7 +848,7 @@ impl PyTokenizer { /// /// direction (:obj:`str`, defaults to :obj:`right`): /// Truncate direction - #[pyo3(signature = (max_length, **kwargs))] + #[pyo3(signature = (max_length, **kwargs) -> "None")] #[pyo3( text_signature = "(self, max_length, stride=0, strategy='longest_first', direction='right')" )] @@ -938,7 +955,7 @@ impl PyTokenizer { /// length (:obj:`int`, `optional`): /// If specified, the length at which to pad. If not specified we pad using the size of /// the longest sequence in a batch. - #[pyo3(signature = (**kwargs))] + #[pyo3(signature = (**kwargs) -> "None")] #[pyo3( text_signature = "(self, direction='right', pad_id=0, pad_type_id=0, pad_token='[PAD]', length=None, pad_to_multiple_of=None)" )] @@ -1066,7 +1083,7 @@ impl PyTokenizer { /// Returns: /// :class:`~tokenizers.Encoding`: The encoded result /// - #[pyo3(signature = (sequence, pair = None, is_pretokenized = false, add_special_tokens = true))] + #[pyo3(signature = (sequence, pair = None, is_pretokenized = false, add_special_tokens = true) -> "Encoding")] #[pyo3( text_signature = "(self, sequence, pair=None, is_pretokenized=False, add_special_tokens=True)" )] @@ -1196,7 +1213,7 @@ impl PyTokenizer { /// Returns: /// A :obj:`List` of :class:`~tokenizers.Encoding`: The encoded batch /// - #[pyo3(signature = (input, is_pretokenized = false, add_special_tokens = true))] + #[pyo3(signature = (input, is_pretokenized = false, add_special_tokens = true) -> "list[Encoding]")] #[pyo3(text_signature = "(self, input, is_pretokenized=False, add_special_tokens=True)")] fn encode_batch( &self, @@ -1315,7 +1332,7 @@ impl PyTokenizer { /// Returns: /// A :obj:`List` of :class:`~tokenizers.Encoding`: The encoded batch /// - #[pyo3(signature = (input, is_pretokenized = false, add_special_tokens = true))] + #[pyo3(signature = (input, is_pretokenized = false, add_special_tokens = true) -> "list[Encoding]")] #[pyo3(text_signature = "(self, input, is_pretokenized=False, add_special_tokens=True)")] fn encode_batch_fast( &self, @@ -1421,7 +1438,7 @@ impl PyTokenizer { /// /// Returns: /// :obj:`str`: The decoded string - #[pyo3(signature = (ids, skip_special_tokens = true))] + #[pyo3(signature = (ids, skip_special_tokens = true) -> "str")] #[pyo3(text_signature = "(self, ids, skip_special_tokens=True)")] fn decode(&self, ids: Vec, skip_special_tokens: bool) -> PyResult { ToPyResult(self.tokenizer.decode(&ids, skip_special_tokens)).into() @@ -1438,7 +1455,7 @@ impl PyTokenizer { /// /// Returns: /// :obj:`List[str]`: A list of decoded strings - #[pyo3(signature = (sequences, skip_special_tokens = true))] + #[pyo3(signature = (sequences, skip_special_tokens = true) -> "list[str]")] #[pyo3(text_signature = "(self, sequences, skip_special_tokens=True)")] fn decode_batch( &self, @@ -1495,7 +1512,7 @@ impl PyTokenizer { /// /// Returns: /// :obj:`Optional[int]`: An optional id, :obj:`None` if out of vocabulary - #[pyo3(text_signature = "(self, token)")] + #[pyo3(signature = (token) -> "int | None", text_signature = "(self, token)")] fn token_to_id(&self, token: &str) -> Option { self.tokenizer.token_to_id(token) } @@ -1508,7 +1525,7 @@ impl PyTokenizer { /// /// Returns: /// :obj:`Optional[str]`: An optional token, :obj:`None` if out of vocabulary - #[pyo3(text_signature = "(self, id)")] + #[pyo3(signature = (id) -> "str | None", text_signature = "(self, id)")] fn id_to_token(&self, id: u32) -> Option { self.tokenizer.id_to_token(id) } @@ -1667,7 +1684,7 @@ impl PyTokenizer { // Each element of the iterator can either be: // - An iterator, to allow batching // - A string - if let Ok(s) = element.downcast::() { + if let Ok(s) = element.cast::() { itertools::Either::Right(std::iter::once(s.to_cow().map(|s| s.into_owned()))) } else { match element.try_iter() { diff --git a/bindings/python/src/trainers.rs b/bindings/python/src/trainers.rs index 53415fff02..8e90406726 100644 --- a/bindings/python/src/trainers.rs +++ b/bindings/python/src/trainers.rs @@ -64,7 +64,7 @@ impl PyTrainer { self.trainer = unpickled; Ok(()) } - Err(e) => Err(e), + Err(e) => Err(e.into()), } } @@ -321,7 +321,7 @@ impl PyBpeTrainer { "show_progress" => builder = builder.show_progress(val.extract()?), "special_tokens" => { builder = builder.special_tokens( - val.downcast::()? + val.cast::()? .into_iter() .map(|token| { if let Ok(content) = token.extract::() { @@ -528,7 +528,7 @@ impl PyWordPieceTrainer { "show_progress" => builder = builder.show_progress(val.extract()?), "special_tokens" => { builder = builder.special_tokens( - val.downcast::()? + val.cast::()? .into_iter() .map(|token| { if let Ok(content) = token.extract::() { @@ -678,7 +678,7 @@ impl PyWordLevelTrainer { } "special_tokens" => { builder.special_tokens( - val.downcast::()? + val.cast::()? .into_iter() .map(|token| { if let Ok(content) = token.extract::() { @@ -851,7 +851,7 @@ impl PyUnigramTrainer { ) } "special_tokens" => builder.special_tokens( - val.downcast::()? + val.cast::()? .into_iter() .map(|token| { if let Ok(content) = token.extract::() { @@ -887,13 +887,17 @@ impl PyUnigramTrainer { /// Trainers Module #[pymodule] -pub fn trainers(m: &Bound<'_, PyModule>) -> PyResult<()> { - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - m.add_class::()?; - Ok(()) +pub mod trainers { + #[pymodule_export] + pub use super::PyBpeTrainer; + #[pymodule_export] + pub use super::PyTrainer; + #[pymodule_export] + pub use super::PyUnigramTrainer; + #[pymodule_export] + pub use super::PyWordLevelTrainer; + #[pymodule_export] + pub use super::PyWordPieceTrainer; } #[cfg(test)] diff --git a/bindings/python/src/utils/normalization.rs b/bindings/python/src/utils/normalization.rs index 5efdd1d137..dc3f49dfe6 100644 --- a/bindings/python/src/utils/normalization.rs +++ b/bindings/python/src/utils/normalization.rs @@ -90,8 +90,10 @@ impl PyRange<'_> { #[derive(Clone)] pub struct PySplitDelimiterBehavior(pub SplitDelimiterBehavior); -impl FromPyObject<'_> for PySplitDelimiterBehavior { - fn extract_bound(obj: &Bound<'_, PyAny>) -> PyResult { +impl<'a, 'py> FromPyObject<'a, 'py> for PySplitDelimiterBehavior { + type Error = PyErr; + + fn extract(obj: Borrowed<'a, 'py, PyAny>) -> Result { let s = obj.extract::()?; Ok(Self(match s.as_ref() { diff --git a/bindings/python/src/utils/pretokenization.rs b/bindings/python/src/utils/pretokenization.rs index 6a14888d3b..e533cbd4ef 100644 --- a/bindings/python/src/utils/pretokenization.rs +++ b/bindings/python/src/utils/pretokenization.rs @@ -56,7 +56,8 @@ fn tokenize(pretok: &mut PreTokenizedString, func: &Bound<'_, PyAny>) -> PyResul ToPyResult(pretok.tokenize(|normalized| { let output = func.call((normalized.get(),), None)?; Ok(output - .extract::>()? + .extract::>() + .map_err(PyErr::from)? .into_iter() .map(|obj| Ok(Token::from(obj.extract::()?))) .collect::>>()?) @@ -68,8 +69,10 @@ fn tokenize(pretok: &mut PreTokenizedString, func: &Bound<'_, PyAny>) -> PyResul /// This is an enum #[derive(Clone)] pub struct PyOffsetReferential(OffsetReferential); -impl FromPyObject<'_> for PyOffsetReferential { - fn extract_bound(obj: &Bound<'_, PyAny>) -> PyResult { +impl<'a, 'py> FromPyObject<'a, 'py> for PyOffsetReferential { + type Error = PyErr; + + fn extract(obj: Borrowed<'a, 'py, PyAny>) -> Result { let s = obj.extract::()?; Ok(Self(match s.as_ref() { @@ -84,8 +87,10 @@ impl FromPyObject<'_> for PyOffsetReferential { #[derive(Clone)] pub struct PyOffsetType(OffsetType); -impl FromPyObject<'_> for PyOffsetType { - fn extract_bound(obj: &Bound<'_, PyAny>) -> PyResult { +impl<'a, 'py> FromPyObject<'a, 'py> for PyOffsetType { + type Error = PyErr; + + fn extract(obj: Borrowed<'a, 'py, PyAny>) -> Result { let s = obj.extract::()?; Ok(Self(match s.as_ref() { @@ -223,7 +228,7 @@ impl PyPreTokenizedString { /// /// Returns: /// An Encoding - #[pyo3(signature = (type_id = 0, word_idx = None))] + #[pyo3(signature = (type_id = 0, word_idx = None) -> "Encoding")] #[pyo3(text_signature = "(self, type_id=0, word_idx=None)")] fn to_encoding(&self, type_id: u32, word_idx: Option) -> PyResult { to_encoding(&self.pretok, type_id, word_idx) diff --git a/bindings/python/stub.py b/bindings/python/stub.py index ec6ae86564..ade619d899 100644 --- a/bindings/python/stub.py +++ b/bindings/python/stub.py @@ -1,173 +1,38 @@ import argparse +import ast import inspect import os +import subprocess +import sys from pathlib import Path -INDENT = " " * 4 GENERATED_COMMENT = "# Generated content DO NOT EDIT\n" -OVERRIDES = { - ("tokenizers", "AddedToken", "__init__"): "(self, content=None, single_word=False, lstrip=False, rstrip=False, normalized=True, special=False)", - ("tokenizers.decoders", "Strip", "__init__"): "(self, content=' ', left=0, right=0)", - ("tokenizers.processors", "TemplateProcessing", "__init__"): "(self, single=None, pair=None, special_tokens=None)", -} - - -def do_indent(text: str, indent: str): - return text.replace("\n", f"\n{indent}") - - -def function(obj, indent, text_signature=None, owner=None): - name = obj.__name__ - - # 1) Figure out a usable text_signature - if text_signature is None: - text_signature = getattr(obj, "__text_signature__", None) - if owner is not None: - key = (getattr(owner, "__module__", ""), owner.__name__, name) - if key in OVERRIDES: - text_signature = OVERRIDES[key] - if text_signature is None: - text_signature = "()" - else: - text_signature = text_signature.replace("$self", "self").replace(" /,", "") - - if name in ("__getitem__", "__setitem__"): - # Always expose magic indexing methods, even if they lack a __text_signature__ - # (PyO3 magic methods often do). - if name == "__getitem__": - text_signature = "(self, key)" - else: - text_signature = "(self, key, value)" - - # 2) Safely handle missing docstrings - doc = obj.__doc__ or "" - - string = "" - string += f"{indent}def {name}{text_signature}:\n" - indent += INDENT - string += f'{indent}"""\n' - if doc: - string += f"{indent}{do_indent(doc, indent)}\n" - string += f'{indent}"""\n' - string += f"{indent}pass\n" - string += "\n\n" - return string - - -def member_sort(member): - if inspect.isclass(member): - value = 10 + len(inspect.getmro(member)) - else: - value = 1 - return value - - -def fn_predicate(obj): - always = {"__getitem__", "__setitem__", "__getstate__", "__setstate__", "__getnewargs__"} - if inspect.ismethoddescriptor(obj) or inspect.isbuiltin(obj): - name = obj.__name__ - # Always expose magic indexing methods, even if they start with "_" - # or lack a __text_signature__ (PyO3 magic methods often do). - if name in always: - return True - return obj.__text_signature__ and not obj.__name__.startswith("_") - - if inspect.isgetsetdescriptor(obj): - return not obj.__name__.startswith("_") - return False - -def get_module_members(module): - members = [ +def public_members(module): + return [ member for name, member in inspect.getmembers(module) if not name.startswith("_") and not inspect.ismodule(member) ] - members.sort(key=member_sort) - return members - - -def pyi_file(obj, indent="", owner=None): - string = "" - if inspect.ismodule(obj): - string += GENERATED_COMMENT - members = get_module_members(obj) - for member in members: - string += pyi_file(member, indent) - - elif inspect.isclass(obj): - indent += INDENT - mro = inspect.getmro(obj) - if len(mro) > 2: - inherit = f"({mro[1].__name__})" - else: - inherit = "" - string += f"class {obj.__name__}{inherit}:\n" - - body = "" - if obj.__doc__: - body += f'{indent}"""\n{indent}{do_indent(obj.__doc__, indent)}\n{indent}"""\n' - - fns = inspect.getmembers(obj, fn_predicate) - - # Init - if obj.__text_signature__: - init_sig = OVERRIDES.get((obj.__module__, obj.__name__, "__init__"), obj.__text_signature__) - init_sig = init_sig.replace("$self", "self").replace(" /,", "") - body += f"{indent}def __init__{init_sig}:\n" - body += f"{indent + INDENT}pass\n" - body += "\n" - - for name, fn in fns: - body += pyi_file(fn, indent=indent, owner=obj) - - if not body: - body += f"{indent}pass\n" - - string += body - string += "\n\n" - - elif inspect.isbuiltin(obj): - string += f"{indent}@staticmethod\n" - string += function(obj, indent, owner=owner) - - elif inspect.ismethoddescriptor(obj): - string += function(obj, indent, owner=owner) - - elif inspect.isgetsetdescriptor(obj): - string += f"{indent}@property\n" - string += function(obj, indent, text_signature="(self)", owner=owner) - # Expose setter in stubs for properties that are writable in Python. - # If a descriptor is actually read-only at runtime, type checkers may still allow - # assignment but the runtime will raise, which is acceptable for stubs. - string += f"{indent}@{obj.__name__}.setter\n" - string += function(obj, indent, text_signature="(self, value)", owner=owner) - else: - raise Exception(f"Object {obj} is not supported") - return string - -def py_file(module, origin): - members = get_module_members(module) - string = GENERATED_COMMENT - string += f"from .. import {origin}\n" - string += "\n" +def forwarder(module, origin): + members = public_members(module) + lines = [GENERATED_COMMENT, f"from .. import {origin}", ""] for member in members: - name = member.__name__ - string += f"{name} = {origin}.{name}\n" - return string - - -import subprocess -from typing import List, Optional, Tuple + if getattr(member, "__module__", "") == "typing": + continue + member_name = getattr(member, "__name__", None) + if member_name: + lines.append(f"{member_name} = {origin}.{member_name}") + lines.append("") + return "\n".join(lines) -def do_ruff(code, is_pyi: bool): - command = ["ruff", "format", "--config", "pyproject.toml"] - command.extend(["--stdin-filename", "test.pyi" if is_pyi else "test.py", "-"]) +def do_ruff(code): + command = ["ruff", "format", "--config", "pyproject.toml", "--stdin-filename", "__init__.py", "-"] process = subprocess.Popen(command, stdout=subprocess.PIPE, stderr=subprocess.PIPE, stdin=subprocess.PIPE) stdout, stderr = process.communicate(input=code.encode("utf-8")) if stderr: @@ -177,101 +42,145 @@ def do_ruff(code, is_pyi: bool): return stdout.decode("utf-8") -def write(module, directory, origin, check=False): - submodules = [(name, member) for name, member in inspect.getmembers(module) if inspect.ismodule(member)] - - filename = os.path.join(directory, "__init__.pyi") - pyi_content = pyi_file(module) - - # Inject extra hints for hand-written Python modules layered on top of the extension. - if origin == "tokenizers": - extra = """ -from enum import Enum -from typing import List, Tuple, Union, Any - -Offsets = Tuple[int, int] -TextInputSequence = str -PreTokenizedInputSequence = Union[List[str], Tuple[str, ...]] -TextEncodeInput = Union[ - TextInputSequence, - Tuple[TextInputSequence, TextInputSequence], - List[TextInputSequence], -] -PreTokenizedEncodeInput = Union[ - PreTokenizedInputSequence, - Tuple[PreTokenizedInputSequence, PreTokenizedInputSequence], - List[PreTokenizedInputSequence], -] -InputSequence = Union[TextInputSequence, PreTokenizedInputSequence] -EncodeInput = Union[TextEncodeInput, PreTokenizedEncodeInput] - - -class OffsetReferential(Enum): - ORIGINAL = "original" - NORMALIZED = "normalized" - - -class OffsetType(Enum): - BYTE = "byte" - CHAR = "char" - - -class SplitDelimiterBehavior(Enum): - REMOVED = "removed" - ISOLATED = "isolated" - MERGED_WITH_PREVIOUS = "merged_with_previous" - MERGED_WITH_NEXT = "merged_with_next" - CONTIGUOUS = "contiguous" - -from .implementations import ( - BertWordPieceTokenizer, - ByteLevelBPETokenizer, - CharBPETokenizer, - SentencePieceBPETokenizer, - SentencePieceUnigramTokenizer, -) - -def __getattr__(name: str) -> Any: ... -BertWordPieceTokenizer: Any -ByteLevelBPETokenizer: Any -CharBPETokenizer: Any -SentencePieceBPETokenizer: Any -SentencePieceUnigramTokenizer: Any -""" - pyi_content += extra - - if origin == "normalizers": - pyi_content += """ -from typing import Dict - -NORMALIZERS: Dict[str, Normalizer] - -def unicode_normalizer_from_str(normalizer: str) -> Normalizer: ... -""" +def format_docstring(docstring: str, indent: int = 4) -> str: + """Format a docstring for insertion into a .pyi file.""" + if not docstring: + return "" + + indent_str = " " * indent + lines = docstring.strip().split("\n") + + if len(lines) == 1: + # Single line docstring + return f'{indent_str}"""{lines[0]}"""\n' + + # Multi-line docstring + result = [f'{indent_str}"""'] + result.extend(indent_str + line.rstrip() for line in lines) + result.append(f'{indent_str}"""') + return "\n".join(result) + "\n" + +def get_module(module_name: str): + """Get module by name, handling tokenizers submodules.""" + import tokenizers + if module_name == "tokenizers": + return tokenizers + return getattr(tokenizers, module_name.split(".")[-1], None) + + +def add_docstring_to_stub(line: str, docstring: str, indent: int) -> str: + """Convert 'def foo(): ...' or multi-line ending with '...' to include docstring.""" + if line.rstrip().endswith('...'): + base = line.rstrip()[:-3] # Remove ... + if not base.rstrip().endswith(':'): + base = base.rstrip() + ':' + inner = ' ' * (indent + 4) + return f"{base}\n{inner}\"\"\"{docstring}\"\"\"\n{inner}..." + return line + + +def add_docstrings_to_pyi(pyi_file: Path, module_name: str): + """Add docstrings from the actual module to a .pyi file.""" + module = get_module(module_name) + if module is None: + print(f"Could not find module {module_name}") + return + + content = pyi_file.read_text() try: - pyi_content = do_ruff(pyi_content, is_pyi=True) - except Exception as e: - print(f"Ruff error: {e}") + tree = ast.parse(content) + except SyntaxError as e: + print(f"Could not parse {pyi_file}: {e}") + return + + lines = content.splitlines(keepends=True) + + # Collect insertions: (start_line, end_line, docstring, indent) + # start_line and end_line are 0-indexed + insertions = [] + + for node in ast.walk(tree): + if isinstance(node, ast.ClassDef): + obj = getattr(module, node.name, None) + if obj and getattr(obj, "__doc__", None): + indent = len(lines[node.lineno - 1]) - len(lines[node.lineno - 1].lstrip()) + insertions.append((node.lineno - 1, node.lineno - 1, obj.__doc__.strip(), indent)) + + # Process methods and properties within the class + for item in node.body: + if isinstance(item, (ast.FunctionDef, ast.AsyncFunctionDef)): + class_obj = getattr(module, node.name, None) + if not class_obj: + continue + # For properties, get descriptor; for methods, get method + method_obj = getattr(class_obj, item.name, None) + if method_obj and getattr(method_obj, "__doc__", None): + start = item.lineno - 1 + end = (item.end_lineno - 1) if item.end_lineno else start + indent = len(lines[start]) - len(lines[start].lstrip()) + insertions.append((start, end, method_obj.__doc__.strip(), indent)) + + elif isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)): + # Only process top-level functions (not nested in classes) + # Check if this is a top-level node + if node in tree.body: + obj = getattr(module, node.name, None) + if obj and getattr(obj, "__doc__", None): + start = node.lineno - 1 + end = (node.end_lineno - 1) if node.end_lineno else start + indent = len(lines[start]) - len(lines[start].lstrip()) + insertions.append((start, end, obj.__doc__.strip(), indent)) + + # Sort by start line descending to apply from bottom to top + insertions.sort(key=lambda x: x[0], reverse=True) + + # Apply insertions + for start, end, docstring, indent in insertions: + # Get all lines for this definition + def_lines = [lines[i].rstrip('\n') for i in range(start, end + 1)] + combined = '\n'.join(def_lines) + + # Check if it ends with ... (stub pattern) + if combined.rstrip().endswith('...'): + inner_indent = ' ' * (indent + 4) + formatted_doc = format_docstring(docstring, indent=indent + 4).rstrip('\n') + + if len(def_lines) == 1: + # Single line: def foo(): ... + base = def_lines[0].rstrip()[:-3].rstrip() + if not base.endswith(':'): + base += ':' + new_content = f"{base}\n{formatted_doc}\n{inner_indent}...\n" + else: + # Multi-line signature + last_line = def_lines[-1].rstrip()[:-3].rstrip() # Remove ... + new_lines = def_lines[:-1] + [last_line] + new_content = '\n'.join(new_lines) + '\n' + formatted_doc + '\n' + inner_indent + '...\n' + + # Replace lines[start:end+1] with new_content + lines[start:end + 1] = [new_content] + + # Write back + pyi_file.write_text(''.join(lines)) + print(f"Added docstrings to {pyi_file}") + +def write(module, directory, origin, check=False): + submodules = [ + (name, member) + for name, member in inspect.getmembers(module) + if inspect.ismodule(member) + ] os.makedirs(directory, exist_ok=True) - if check: - with open(filename, "r") as f: - data = f.read() - assert data == pyi_content, f"The content of {filename} seems outdated, please run `python stub.py`" - else: - with open(filename, "w") as f: - f.write(pyi_content) filename = os.path.join(directory, "__init__.py") - py_content = py_file(module, origin) + py_content = forwarder(module, origin) try: - py_content = do_ruff(py_content, is_pyi=False) + py_content = do_ruff(py_content) except Exception as e: print(f"Ruff error: {e}") - os.makedirs(directory, exist_ok=True) - is_auto = False if not os.path.exists(filename): is_auto = True @@ -290,8 +199,42 @@ def unicode_normalizer_from_str(normalizer: str) -> Normalizer: ... with open(filename, "w") as f: f.write(py_content) + print(f"Wrote stub for module: {origin}, submodules: {[name for name, _ in submodules]}") for name, submodule in submodules: - write(submodule, os.path.join(directory, name), f"{name}", check=check) + try: + write(submodule, os.path.join(directory, name), f"{name}", check=check) + except Exception as e: + print(f"Something went wrong with {name}, {submodule}: {e}") + +def process_all_pyi_files(base_dir: str): + """Process all .pyi files in the directory tree to add docstrings.""" + base_path = Path(base_dir) + + # Map relative paths to module names + for pyi_file in base_path.rglob("*.pyi"): + # Skip __init__.pyi as it typically just has imports + if pyi_file.name == "__init__.pyi" and pyi_file.parent.name == "tokenizers": + # Process the main tokenizers module + add_docstrings_to_pyi(pyi_file, "tokenizers") + elif pyi_file.name == "__init__.pyi": + # Skip other __init__.pyi files + continue + elif pyi_file.name == "tokenizers.pyi": + # Skip the tokenizers.pyi file as it's just imports + continue + else: + # Convert file path to module name + # e.g., py_src/tokenizers/decoders.pyi -> tokenizers.decoders + rel_path = pyi_file.relative_to(base_path) + # rel_path will be something like "decoders.pyi" or "subdir/file.pyi" + # We want to convert to "tokenizers.decoders" or "tokenizers.subdir.file" + parts = list(rel_path.parts[:-1]) + [rel_path.stem] + if parts: + module_name = "tokenizers." + ".".join(parts) + else: + module_name = "tokenizers" + + add_docstrings_to_pyi(pyi_file, module_name) if __name__ == "__main__": @@ -301,5 +244,9 @@ def unicode_normalizer_from_str(normalizer: str) -> Normalizer: ... args = parser.parse_args() import tokenizers - # `tokenizers.tokenizers` is the extension module; attribute access is dynamic. - write(tokenizers.tokenizers, "py_src/tokenizers/", "tokenizers", check=args.check) # type: ignore[attr-defined] + write(tokenizers, "py_src/tokenizers/", "tokenizers", check=args.check) # type: ignore[attr-defined] + + # Process all .pyi files to add docstrings + if not args.check: + print("\nAdding docstrings to .pyi files...") + process_all_pyi_files("py_src/tokenizers") diff --git a/bindings/python/tests/documentation/test_tutorial_train_from_iterators.py b/bindings/python/tests/documentation/test_tutorial_train_from_iterators.py index 87ba55a2fd..f2a4183384 100644 --- a/bindings/python/tests/documentation/test_tutorial_train_from_iterators.py +++ b/bindings/python/tests/documentation/test_tutorial_train_from_iterators.py @@ -2,7 +2,7 @@ import gzip import os -import datasets +import datasets # type: ignore[import-not-found] import pytest from ..utils import data_dir, train_files @@ -32,7 +32,7 @@ def get_tokenizer_trainer(): @staticmethod def load_dummy_dataset(): # START load_dataset - import datasets + import datasets # type: ignore[import-not-found] dataset = datasets.load_dataset("wikitext", "wikitext-103-raw-v1", split="train+test+validation") # END load_dataset @@ -73,13 +73,13 @@ def test_datasets(self): def batch_iterator(batch_size=1000): # Only keep the text column to avoid decoding the rest of the columns unnecessarily tok_dataset = dataset.select_columns("text") - for batch in tok_dataset.iter(batch_size): # type: ignore[attr-defined] + for batch in tok_dataset.iter(batch_size): yield batch["text"] # END def_batch_iterator # START train_datasets - tokenizer.train_from_iterator(batch_iterator(), trainer=trainer, length=len(dataset)) # type: ignore[arg-type] + tokenizer.train_from_iterator(batch_iterator(), trainer=trainer, length=len(dataset)) # END train_datasets def test_gzip(self, setup_gzip_files): diff --git a/bindings/python/tools/stub-gen/Cargo.toml b/bindings/python/tools/stub-gen/Cargo.toml new file mode 100644 index 0000000000..c8761e153b --- /dev/null +++ b/bindings/python/tools/stub-gen/Cargo.toml @@ -0,0 +1,14 @@ +[package] +name = "stub-gen" +version = "0.1.0" +edition = "2021" +description = "Stub generation tool for tokenizers Python bindings" + +[[bin]] +name = "stub-gen" +path = "src/main.rs" + +[dependencies] +env_logger = "0.11" +pyo3 = { version = "0.27.2", default-features = false, features = ["auto-initialize"] } +pyo3-introspection = "0.27.2" diff --git a/bindings/python/tools/stub-gen/src/main.rs b/bindings/python/tools/stub-gen/src/main.rs new file mode 100644 index 0000000000..f47fbd8449 --- /dev/null +++ b/bindings/python/tools/stub-gen/src/main.rs @@ -0,0 +1,194 @@ +use pyo3::prelude::*; +use pyo3::types::PyList; +use std::ffi::OsString; +use std::path::{Path, PathBuf}; +use std::process::Command; + +fn main() -> Result<(), Box> { + env_logger::try_init().ok(); + + let manifest_dir = find_manifest_dir()?; + let cdylib = manifest_dir.join("tokenizers.abi3.so"); + let out_dir = manifest_dir.join("py_src/tokenizers"); + + build_extension(&manifest_dir)?; + refresh_cdylib(&manifest_dir, &cdylib)?; + setup_python_env()?; + generate_stubs(&cdylib, &out_dir)?; + Ok(()) +} + +/// Set up PYTHONHOME environment variable if not already set. +/// This is needed for PyO3 embedded Python to find the standard library, +/// especially when using virtual environments created by uv. +fn setup_python_env() -> Result<(), Box> { + if std::env::var_os("PYTHONHOME").is_some() { + return Ok(()); + } + + // Query Python for its base_prefix (the actual Python installation, not venv) + let output = Command::new("python3") + .args(["-c", "import sys; print(sys.base_prefix, end='')"]) + .output()?; + + if !output.status.success() { + return Err("Failed to query Python base_prefix".into()); + } + + let base_prefix = String::from_utf8(output.stdout)?; + if !base_prefix.is_empty() { + println!("Setting PYTHONHOME={}", base_prefix); + std::env::set_var("PYTHONHOME", &base_prefix); + } + + Ok(()) +} + +fn find_manifest_dir() -> Result> { + // Look for the bindings/python directory relative to current working directory + // or from the tool's location + let cwd = std::env::current_dir()?; + + // Check if we're already in bindings/python + if cwd.join("pyproject.toml").exists() && cwd.join("py_src").exists() { + return Ok(cwd); + } + + // Check if bindings/python exists relative to cwd + let bindings_python = cwd.join("bindings/python"); + if bindings_python.join("pyproject.toml").exists() { + return Ok(bindings_python); + } + + // Try to find it from the executable location + if let Ok(exe) = std::env::current_exe() { + // Go up from tools/stub-gen/target/... to bindings/python + let mut path = exe.as_path(); + for _ in 0..10 { + if let Some(parent) = path.parent() { + if parent.join("pyproject.toml").exists() && parent.join("py_src").exists() { + return Ok(parent.to_path_buf()); + } + path = parent; + } + } + } + + Err("Could not find bindings/python directory. Run from the tokenizers root or bindings/python directory.".into()) +} + +fn generate_stubs(cdylib: &Path, out_dir: &Path) -> Result<(), Box> { + if !cdylib.is_file() { + return Err(format!("Failed to locate cdylib at {}", cdylib.display()).into()); + } + + println!("Initializing python"); + Python::initialize(); + let cdylib = cdylib.to_path_buf(); + let out_dir = out_dir.to_path_buf(); + + Python::attach(|py| -> PyResult<()> { + println!("Gathering Python environment information..."); + let sys = py.import("sys")?; + println!("sys.version = {}", sys.getattr("version")?); + println!("sys.executable = {}", sys.getattr("executable")?); + println!("sys.prefix = {}", sys.getattr("prefix")?); + println!("sys.base_prefix = {}", sys.getattr("base_prefix")?); + + let so_dir = cdylib + .parent() + .unwrap_or_else(|| Path::new(".")) + .to_path_buf(); + + let bindings = sys.getattr("path")?; + let sys_path = bindings.cast::()?; + sys_path.insert(0, so_dir.to_str().unwrap())?; + + let old = std::env::var_os("PYTHONPATH"); + let mut new = OsString::new(); + new.push(&so_dir); + if let Some(old) = old { + new.push(":"); + new.push(old); + } + std::env::set_var("PYTHONPATH", &new); + println!("New PYTHONPATH={:?}", std::env::var_os("PYTHONPATH")); + let sysconfig = PyModule::import(py, "sysconfig")?; + let python_version = sysconfig.call_method0("get_python_version")?; + println!("Using python version: {}", python_version); + let python_lib = sysconfig.call_method("get_config_var", ("LIBDEST",), None)?; + println!("Using python lib: {}", python_lib); + let python_site_packages = sysconfig.call_method("get_path", ("purelib",), None)?; + println!("Using python site-packages: {}", python_site_packages); + py.run( + c"import tokenizers; import sys; print('import ok:', tokenizers.__file__); print('sys.path[0]=', sys.path[0])", + None, + None, + ) + .unwrap_or_else(|e| panic!("Failed to import tokenizers: {:?}", e)); + + println!("Generating stub files"); + assert!( + cdylib.is_file(), + "Failed to locate cdylib at {}", + cdylib.display() + ); + println!("Found cdylib at {}", cdylib.display()); + + let main_module_name = "tokenizers"; + let python_module = pyo3_introspection::introspect_cdylib(&cdylib, main_module_name) + .unwrap_or_else(|_| panic!("Failed introspection of {}", main_module_name)); + let type_stubs = pyo3_introspection::module_stub_files(&python_module); + + for (rel_path, contents) in type_stubs { + let out_path = out_dir.join(&rel_path); + if let Some(parent) = out_path.parent() { + std::fs::create_dir_all(parent) + .unwrap_or_else(|_| panic!("Failed introspection of {}", main_module_name)) + } + std::fs::write(&out_path, contents).expect("Failed to write stubs file"); + println!("Generated stub: {}", out_path.display()); + } + + Ok(()) + })?; + + Ok(()) +} + +fn build_extension(manifest_dir: &Path) -> Result<(), Box> { + println!("Building and installing extension (release)..."); + let status = Command::new("maturin") + .current_dir(manifest_dir) + .args(["develop", "--release"]) + .status()?; + + if !status.success() { + return Err("`maturin develop` failed".into()); + } + + Ok(()) +} + +fn refresh_cdylib(manifest_dir: &Path, cdylib: &Path) -> Result<(), Box> { + let built_cdylib = manifest_dir.join(format!( + "target/release/{}tokenizers.{}", + std::env::consts::DLL_PREFIX, + std::env::consts::DLL_EXTENSION + )); + + if !built_cdylib.is_file() { + return Err(format!( + "Could not find built cdylib at {}.", + built_cdylib.display() + ) + .into()); + } + + println!( + "Refreshing cdylib used for introspection: {}", + cdylib.display() + ); + std::fs::copy(&built_cdylib, cdylib)?; + Ok(()) +}