a
    dS                     @   s  U d Z ddlZddlZddlZddlmZ ddlmZmZm	Z	m
Z
mZ ddlmZ ddlmZmZ ddlmZ dd	lmZ dd
lmZmZmZmZmZ ddlmZ ddlmZ ddlm Z m!Z!m"Z"m#Z#m$Z$ e rddl%m&Z& ndZ&e'e(Z)ere Z*ee+e
e	e+ e	e+ f f e,d< 	nede r,dnde r:dndffdde rPdndffdde rfdnde rtdndffddde rdndffde rdnddffd d!d"e rd#nde rd$ndffd%d&e rd'ndffd(d)d*d+de rdndffd,d-e rd.ndffd/de r&d0ndffd1d2e r<d3ndffd4d5e rRd6nde r`d7ndffd8d9de rxdndffd:d2e rd3ndffd;d<e rd=ndffd>d<e rd=ndffd?d@e rdAndffdBdCe rdDndffdEe rdFnde rdGndffdHdIdJd2e r"d3ndffdKdLe r8dMndffdNe rLdOnde rZdPndffdQdRe rpdSndffdTdUe rdVndffdWdXe rdYndffdZde rdndffd[e rd\nddffd]d^d_d`e rdandffdbdcdde rdendffdfde rdndffdge r$dhnddffdid-e r<d.ndffdjd-e rRd.ndffdkd-e rhd.ndffdlde r~dmndffdndod-e rd.ndffdpdqd<e rd=ndffdrdse rdtndffdudvd2e rd3ndffdwdxdye rdzndffd{d|e r
d}ndffd~de r dndffdde r6dndffdde rLdndffdde rbdndffde rvdnde rdndffdde rdndffde rdnde rdndffddde rdndffde rdnddffde rdnddffde rdnde r"dndffde r6dnde rDdndffdd2e rZd3ndffdde rpdndffdde rdnddffdde rdndffdde rdndffde rdnde rdndffdde rdndffdde rdndffde rdnde r$dndffde r8dnde rFdndffde rZdnde rhdndffdd<e r~d=ndffdde rdndffdd-e rd.ndffdd<e rd=ndffde rd&nde rd'ndffde rd&nde rd'ndffdddde rdndffde r2dnddffddde rLdndffddde rddndffde rxdnde rdndffde rdnde rdndffdde rdndffdd2e rd3ndffdd2e rd3ndffddde 	rdndffdde 	rdmndffde 	r,dnddffdde 	rDdnddffddde 	r^dndffde 	rrdnde 	rdndffde 	rdnde 	rdndffddddde 	rdndffdde 	rdndffddddde 	rdndffdd<e 
rd=ndffde 
rdnde 
r(dndffdde 
r>dnddffde 
rTdnde 
rbdndffde 
rvdnde 
rdndffde 
rdnde 
rdndffde 
rdnde 
rdndffde 
rdnde 
rdndffgZ*ee e*Z-dd  e . D Z/e+dddZ0dee+ej1f e	ee+ej1f  e2e2e	ee+e+f  e	ee2e+f  e	e+ e2e+d	ddZ3G d	d
 d
Z4dS (  z Auto Tokenizer class.    N)OrderedDict)TYPE_CHECKINGDictOptionalTupleUnion   )PretrainedConfig)get_class_from_dynamic_moduleresolve_trust_remote_code)PreTrainedTokenizer)TOKENIZER_CONFIG_FILE)cached_fileextract_commit_hashis_sentencepiece_availableis_tokenizers_availablelogging   )EncoderDecoderConfig   )_LazyAutoMapping)CONFIG_MAPPING_NAMES
AutoConfigconfig_class_to_model_typemodel_type_to_module_name!replace_list_option_in_docstrings)PreTrainedTokenizerFastTOKENIZER_MAPPING_NAMESZalbertZAlbertTokenizerZAlbertTokenizerFastZalignZBertTokenizerZBertTokenizerFast)Zbart)ZBartTokenizerZBartTokenizerFastZbarthezZBarthezTokenizerZBarthezTokenizerFast)Zbartpho)ZBartphoTokenizerNZbertzbert-generationZBertGenerationTokenizer)zbert-japanese)ZBertJapaneseTokenizerN)Zbertweet)ZBertweetTokenizerNZbig_birdZBigBirdTokenizerZBigBirdTokenizerFastZbigbird_pegasusZPegasusTokenizerZPegasusTokenizerFast)Zbiogpt)ZBioGptTokenizerN)Z
blenderbot)ZBlenderbotTokenizerZBlenderbotTokenizerFast)zblenderbot-small)ZBlenderbotSmallTokenizerNZblipzblip-2ZGPT2TokenizerZGPT2TokenizerFastZbloomZBloomTokenizerFastZbridgetowerZRobertaTokenizerZRobertaTokenizerFast)Zbyt5)ZByT5TokenizerNZ	camembertZCamembertTokenizerZCamembertTokenizerFast)Zcanine)ZCanineTokenizerNZchinese_clipZclapZclipZCLIPTokenizerZCLIPTokenizerFastZclipsegZcodegenZCodeGenTokenizerZCodeGenTokenizerFastZconvbertZConvBertTokenizerZConvBertTokenizerFastZcpmZCpmTokenizerZCpmTokenizerFast)Zcpmant)ZCpmAntTokenizerN)Zctrl)ZCTRLTokenizerNzdata2vec-textZdebertaZDebertaTokenizerZDebertaTokenizerFastz
deberta-v2ZDebertaV2TokenizerZDebertaV2TokenizerFastZ
distilbertZDistilBertTokenizerZDistilBertTokenizerFastZdprZDPRQuestionEncoderTokenizerZDPRQuestionEncoderTokenizerFastZelectraZElectraTokenizerZElectraTokenizerFastZernieZernie_mZErnieMTokenizer)Zesm)ZEsmTokenizerN)Zflaubert)ZFlaubertTokenizerNZfnetZFNetTokenizerZFNetTokenizerFast)Zfsmt)ZFSMTTokenizerNZfunnelZFunnelTokenizerZFunnelTokenizerFastgitzgpt-sw3ZGPTSw3TokenizerZgpt2Zgpt_bigcodeZgpt_neoZgpt_neoxZGPTNeoXTokenizerFast)Zgpt_neox_japanese)ZGPTNeoXJapaneseTokenizerNZgptj)zgptsan-japanese)ZGPTSanJapaneseTokenizerNZgroupvitZherbertZHerbertTokenizerZHerbertTokenizerFast)ZhubertZWav2Vec2CTCTokenizerNZibert)Zjukebox)ZJukeboxTokenizerNZlayoutlmZLayoutLMTokenizerZLayoutLMTokenizerFastZ
layoutlmv2ZLayoutLMv2TokenizerZLayoutLMv2TokenizerFastZ
layoutlmv3ZLayoutLMv3TokenizerZLayoutLMv3TokenizerFastZ	layoutxlmZLayoutXLMTokenizerZLayoutXLMTokenizerFastZledZLEDTokenizerZLEDTokenizerFastZliltZllamaZLlamaTokenizerZLlamaTokenizerFastZ
longformerZLongformerTokenizerZLongformerTokenizerFastZlongt5ZT5TokenizerZT5TokenizerFast)Zluke)ZLukeTokenizerNZlxmertZLxmertTokenizerZLxmertTokenizerFastZm2m_100ZM2M100TokenizerZmarianZMarianTokenizerZmbartZMBartTokenizerZMBartTokenizerFastZmbart50ZMBart50TokenizerZMBart50TokenizerFastZmegazmegatron-bert)zmgp-str)ZMgpstrTokenizerNZmlukeZMLukeTokenizerZ
mobilebertZMobileBertTokenizerZMobileBertTokenizerFastZmpnetZMPNetTokenizerZMPNetTokenizerFastZmt5ZMT5TokenizerZMT5TokenizerFastZmvpZMvpTokenizerZMvpTokenizerFastZnezhaZnllbZNllbTokenizerZNllbTokenizerFastznllb-moeZnystromformerZ	oneformerz
openai-gptZOpenAIGPTTokenizerZOpenAIGPTTokenizerFastoptZowlvitZpegasusZ	pegasus_x)Z	perceiver)ZPerceiverTokenizerN)Zphobert)ZPhobertTokenizerNZ
pix2structZplbartZPLBartTokenizer)Z
prophetnet)ZProphetNetTokenizerNZqdqbert)Zrag)ZRagTokenizerNrealmZRealmTokenizerZRealmTokenizerFastZreformerZReformerTokenizerZReformerTokenizerFastZrembertZRemBertTokenizerZRemBertTokenizerFastZ	retribertZRetriBertTokenizerZRetriBertTokenizerFastZrobertazroberta-prelayernorm)Zroc_bert)ZRoCBertTokenizerNZroformerZRoFormerTokenizerZRoFormerTokenizerFastZrwkvZspeech_to_textZSpeech2TextTokenizer)Zspeech_to_text_2)ZSpeech2Text2TokenizerNZspeecht5ZSpeechT5Tokenizer)Zsplinter)ZSplinterTokenizerZSplinterTokenizerFastZsqueezebertZSqueezeBertTokenizerZSqueezeBertTokenizerFastZswitch_transformersZt5)Ztapas)ZTapasTokenizerN)Ztapex)ZTapexTokenizerN)z
transfo-xl)ZTransfoXLTokenizerNZviltZvisual_bert)Zwav2vec2r   )zwav2vec2-conformerr   )Zwav2vec2_phoneme)ZWav2Vec2PhonemeCTCTokenizerNZwhisperZWhisperTokenizerZWhisperTokenizerFastZxclipZxglmZXGLMTokenizerZXGLMTokenizerFast)Zxlm)ZXLMTokenizerNzxlm-prophetnetZXLMProphetNetTokenizerzxlm-robertaZXLMRobertaTokenizerZXLMRobertaTokenizerFastzxlm-roberta-xlZxlnetZXLNetTokenizerZXLNetTokenizerFastZxmodZyosoc                 C   s   i | ]\}}||qS  r"   ).0kvr"   r"   s/var/www/html/stable-diffusion-webui/venv/lib/python3.9/site-packages/transformers/models/auto/tokenization_auto.py
<dictcomp>~      r'   )
class_namec              	   C   s   | dkrt S t D ]R\}}| |v rt|}td| d}zt|| W   S  tyd   Y qY q0 qtj	 D ].\}}|D ] }t|dd | kr~|    S q~qrtd}t
|| rt|| S d S )Nr   .ztransformers.models__name__Ztransformers)r   r   itemsr   	importlibimport_modulegetattrAttributeErrorTOKENIZER_MAPPING_extra_contenthasattr)r)   module_nameZ
tokenizersmoduleconfig	tokenizerZmain_moduler"   r"   r&   tokenizer_class_from_name  s$    


r8   F )	pretrained_model_name_or_path	cache_dirforce_downloadresume_downloadproxiesuse_auth_tokenrevisionlocal_files_only	subfolderc	                 K   s   |	 dd}
t| t||||||||dd|
d}|du rDtd i S t||
}
t|dd}t|}W d   n1 sz0    Y  |
|d< |S )a  
    Loads the tokenizer configuration from a pretrained model tokenizer configuration.

    Args:
        pretrained_model_name_or_path (`str` or `os.PathLike`):
            This can be either:

            - a string, the *model id* of a pretrained model configuration hosted inside a model repo on
              huggingface.co. Valid model ids can be located at the root-level, like `bert-base-uncased`, or namespaced
              under a user or organization name, like `dbmdz/bert-base-german-cased`.
            - a path to a *directory* containing a configuration file saved using the
              [`~PreTrainedTokenizer.save_pretrained`] method, e.g., `./my_model_directory/`.

        cache_dir (`str` or `os.PathLike`, *optional*):
            Path to a directory in which a downloaded pretrained model configuration should be cached if the standard
            cache should not be used.
        force_download (`bool`, *optional*, defaults to `False`):
            Whether or not to force to (re-)download the configuration files and override the cached versions if they
            exist.
        resume_download (`bool`, *optional*, defaults to `False`):
            Whether or not to delete incompletely received file. Attempts to resume the download if such a file exists.
        proxies (`Dict[str, str]`, *optional*):
            A dictionary of proxy servers to use by protocol or endpoint, e.g., `{'http': 'foo.bar:3128',
            'http://hostname': 'foo.bar:4012'}.` The proxies are used on each request.
        use_auth_token (`str` or *bool*, *optional*):
            The token to use as HTTP bearer authorization for remote files. If `True`, will use the token generated
            when running `huggingface-cli login` (stored in `~/.huggingface`).
        revision (`str`, *optional*, defaults to `"main"`):
            The specific model version to use. It can be a branch name, a tag name, or a commit id, since we use a
            git-based system for storing models and other artifacts on huggingface.co, so `revision` can be any
            identifier allowed by git.
        local_files_only (`bool`, *optional*, defaults to `False`):
            If `True`, will only try to load the tokenizer configuration from local files.
        subfolder (`str`, *optional*, defaults to `""`):
            In case the tokenizer config is located inside a subfolder of the model repo on huggingface.co, you can
            specify the folder name here.

    <Tip>

    Passing `use_auth_token=True` is required when you want to use a private model.

    </Tip>

    Returns:
        `Dict`: The configuration of the tokenizer.

    Examples:

    ```python
    # Download configuration from huggingface.co and cache.
    tokenizer_config = get_tokenizer_config("bert-base-uncased")
    # This model does not have a tokenizer config so the result will be an empty dict.
    tokenizer_config = get_tokenizer_config("xlm-roberta-base")

    # Save a pretrained tokenizer locally and you can reload its config
    from transformers import AutoTokenizer

    tokenizer = AutoTokenizer.from_pretrained("bert-base-cased")
    tokenizer.save_pretrained("tokenizer-test")
    tokenizer_config = get_tokenizer_config("tokenizer-test")
    ```_commit_hashNF)r;   r<   r=   r>   r?   r@   rA   rB   Z%_raise_exceptions_for_missing_entriesZ'_raise_exceptions_for_connection_errorsrC   z\Could not locate the tokenizer configuration file, will try to use the model config instead.zutf-8)encoding)	getr   r   loggerinfor   openjsonload)r:   r;   r<   r=   r>   r?   r@   rA   rB   kwargsZcommit_hashZresolved_config_filereaderresultr"   r"   r&   get_tokenizer_config  s0    I

(rN   c                   @   s6   e Zd ZdZdd Zeeedd Zd	ddZ	dS )
AutoTokenizera  
    This is a generic tokenizer class that will be instantiated as one of the tokenizer classes of the library when
    created with the [`AutoTokenizer.from_pretrained`] class method.

    This class cannot be instantiated directly using `__init__()` (throws an error).
    c                 C   s   t dd S )Nz}AutoTokenizer is designed to be instantiated using the `AutoTokenizer.from_pretrained(pretrained_model_name_or_path)` method.)EnvironmentError)selfr"   r"   r&   __init__	  s    zAutoTokenizer.__init__c              	   O   s  | dd}d|d< | dd}| dd}| dd}|durd}t|d}	|	du rtd| d	d
dd t D  d|	\}
}|r|durt|}n
td |du rt|
}|du rtd|
 d|j	|g|R i |S t
|fi |}d|v r|d |d< |d}d}d|v rVt|d ttfrF|d }n|d dd}|du rt|tstj	|fd|i|}|j}t|drd|jv r|jd }|du}|dupt|tv }t||||}|r>|r>|r |d dur |d }n|d }t||fi |}| dd}|j	|g|R i |S |durd}|rp|dsp| d}t|}|du r|}t|}|du rtd| d|j	|g|R i |S t|tr t|jt|jurtd|jj d|jj d |j}tt|j}|durtt| \}}|rV|s>|du rV|j	|g|R i |S |durx|j	|g|R i |S tdtd|j dd
d d t D  ddS )!a9  
        Instantiate one of the tokenizer classes of the library from a pretrained model vocabulary.

        The tokenizer class to instantiate is selected based on the `model_type` property of the config object (either
        passed as an argument or loaded from `pretrained_model_name_or_path` if possible), or when it's missing, by
        falling back to using pattern matching on `pretrained_model_name_or_path`:

        List options

        Params:
            pretrained_model_name_or_path (`str` or `os.PathLike`):
                Can be either:

                    - A string, the *model id* of a predefined tokenizer hosted inside a model repo on huggingface.co.
                      Valid model ids can be located at the root-level, like `bert-base-uncased`, or namespaced under a
                      user or organization name, like `dbmdz/bert-base-german-cased`.
                    - A path to a *directory* containing vocabulary files required by the tokenizer, for instance saved
                      using the [`~PreTrainedTokenizer.save_pretrained`] method, e.g., `./my_model_directory/`.
                    - A path or url to a single saved vocabulary file if and only if the tokenizer only requires a
                      single vocabulary file (like Bert or XLNet), e.g.: `./my_model_directory/vocab.txt`. (Not
                      applicable to all derived classes)
            inputs (additional positional arguments, *optional*):
                Will be passed along to the Tokenizer `__init__()` method.
            config ([`PretrainedConfig`], *optional*)
                The configuration object used to dertermine the tokenizer class to instantiate.
            cache_dir (`str` or `os.PathLike`, *optional*):
                Path to a directory in which a downloaded pretrained model configuration should be cached if the
                standard cache should not be used.
            force_download (`bool`, *optional*, defaults to `False`):
                Whether or not to force the (re-)download the model weights and configuration files and override the
                cached versions if they exist.
            resume_download (`bool`, *optional*, defaults to `False`):
                Whether or not to delete incompletely received files. Will attempt to resume the download if such a
                file exists.
            proxies (`Dict[str, str]`, *optional*):
                A dictionary of proxy servers to use by protocol or endpoint, e.g., `{'http': 'foo.bar:3128',
                'http://hostname': 'foo.bar:4012'}`. The proxies are used on each request.
            revision (`str`, *optional*, defaults to `"main"`):
                The specific model version to use. It can be a branch name, a tag name, or a commit id, since we use a
                git-based system for storing models and other artifacts on huggingface.co, so `revision` can be any
                identifier allowed by git.
            subfolder (`str`, *optional*):
                In case the relevant files are located inside a subfolder of the model repo on huggingface.co (e.g. for
                facebook/rag-token-base), specify it here.
            use_fast (`bool`, *optional*, defaults to `True`):
                Use a [fast Rust-based tokenizer](https://huggingface.co/docs/tokenizers/index) if it is supported for
                a given model. If a fast tokenizer is not available for a given model, a normal Python-based tokenizer
                is returned instead.
            tokenizer_type (`str`, *optional*):
                Tokenizer type to be loaded.
            trust_remote_code (`bool`, *optional*, defaults to `False`):
                Whether or not to allow for custom models defined on the Hub in their own modeling files. This option
                should only be set to `True` for repositories you trust and in which you have read the code, as it will
                execute code present on the Hub on your local machine.
            kwargs (additional keyword arguments, *optional*):
                Will be passed to the Tokenizer `__init__()` method. Can be used to set special tokens like
                `bos_token`, `eos_token`, `unk_token`, `sep_token`, `pad_token`, `cls_token`, `mask_token`,
                `additional_special_tokens`. See parameters in the `__init__()` for more details.

        Examples:

        ```python
        >>> from transformers import AutoTokenizer

        >>> # Download vocabulary from huggingface.co and cache.
        >>> tokenizer = AutoTokenizer.from_pretrained("bert-base-uncased")

        >>> # Download vocabulary from huggingface.co (user-uploaded) and cache.
        >>> tokenizer = AutoTokenizer.from_pretrained("dbmdz/bert-base-german-cased")

        >>> # If vocabulary files are in a directory (e.g. tokenizer was saved using *save_pretrained('./test/saved_model/')*)
        >>> # tokenizer = AutoTokenizer.from_pretrained("./test/bert_saved_model/")

        >>> # Download vocabulary from huggingface.co and define model-specific arguments
        >>> tokenizer = AutoTokenizer.from_pretrained("roberta-base", add_prefix_space=True)
        ```r6   NTZ
_from_autouse_fasttokenizer_typetrust_remote_codezPassed `tokenizer_type` z3 does not exist. `tokenizer_type` should be one of z, c                 s   s   | ]
}|V  qd S Nr"   r#   cr"   r"   r&   	<genexpr>m  r(   z0AutoTokenizer.from_pretrained.<locals>.<genexpr>r*   zt`use_fast` is set to `True` but the tokenizer class does not have a fast version.  Falling back to the slow version.zTokenizer class z is not currently imported.rC   tokenizer_classauto_maprO   r   r   Zcode_revisionZFastz- does not exist or is not currently imported.z The encoder model config class: z3 is different from the decoder model config class: z. It is not recommended to use the `AutoTokenizer.from_pretrained()` method in this case. Please use the encoder and decoder specific tokenizer classes.zzThis tokenizer cannot be instantiated. Please make sure you have `sentencepiece` installed in order to use this tokenizer.z!Unrecognized configuration class z8 to build an AutoTokenizer.
Model type should be one of c                 s   s   | ]}|j V  qd S rV   )r+   rW   r"   r"   r&   rY     r(   )popr   rE   
ValueErrorjoinkeysr8   rF   warningfrom_pretrainedrN   
isinstancetuplelistr	   r   rZ   r3   r[   typer1   r   r
   endswithr   decoderencoder	__class__r   r+   )clsr:   inputsrK   r6   rS   rT   rU   rZ   Ztokenizer_class_tupleZtokenizer_class_nameZtokenizer_fast_class_nameZtokenizer_configZconfig_tokenizer_classZtokenizer_auto_mapZhas_remote_codeZhas_local_codeZ	class_ref_Ztokenizer_class_candidateZ
model_typeZtokenizer_class_pyZtokenizer_class_fastr"   r"   r&   ra     s    O















zAutoTokenizer.from_pretrainedNc                 C   s   |du r|du rt d|dur2t|tr2t d|durLt|trLt d|dur|durt|tr|j|krt d|j d| d| tjv rt|  \}}|du r|}|du r|}t| ||f dS )a  
        Register a new tokenizer in this mapping.


        Args:
            config_class ([`PretrainedConfig`]):
                The configuration corresponding to the model to register.
            slow_tokenizer_class ([`PretrainedTokenizer`], *optional*):
                The slow tokenizer to register.
            slow_tokenizer_class ([`PretrainedTokenizerFast`], *optional*):
                The fast tokenizer to register.
        NzKYou need to pass either a `slow_tokenizer_class` or a `fast_tokenizer_classz:You passed a fast tokenizer in the `slow_tokenizer_class`.z:You passed a slow tokenizer in the `fast_tokenizer_class`.zThe fast tokenizer class you are passing has a `slow_tokenizer_class` attribute that is not consistent with the slow tokenizer class you passed (fast tokenizer has z and you passed z!. Fix one of those so they match!)r]   
issubclassr   r   slow_tokenizer_classr1   r2   register)Zconfig_classrn   Zfast_tokenizer_classZexisting_slowZexisting_fastr"   r"   r&   ro     s8    
zAutoTokenizer.register)NN)
r+   
__module____qualname____doc__rR   classmethodr   r   ra   ro   r"   r"   r"   r&   rO     s    DrO   )NFFNNNFr9   )5rr   r-   rI   oscollectionsr   typingr   r   r   r   r   Zconfiguration_utilsr	   Zdynamic_module_utilsr
   r   Ztokenization_utilsr   Ztokenization_utils_baser   utilsr   r   r   r   r   Zencoder_decoderr   Zauto_factoryr   Zconfiguration_autor   r   r   r   r   Ztokenization_utils_fastr   Z
get_loggerr+   rF   r   str__annotations__r1   r,   ZCONFIG_TO_TYPEr8   PathLikeboolrN   rO   r"   r"   r"   r&   <module>   s`  	
*    J
       d