a
    d<                     @   s~   d Z ddlZddlmZ ddlmZ eeZdddd	Z	G d
d deZ
G dd deZG dd deZG dd deZdS )z SAM model configuration    N   )PretrainedConfig)loggingzEhttps://huggingface.co/facebook/sam-vit-huge/resolve/main/config.jsonzFhttps://huggingface.co/facebook/sam-vit-large/resolve/main/config.jsonzEhttps://huggingface.co/facebook/sam-vit-base/resolve/main/config.json)zfacebook/sam-vit-hugezfacebook/sam-vit-largezfacebook/sam-vit-basec                       s"   e Zd ZdZd
 fdd		Z  ZS )SamPromptEncoderConfiga  
    This is the configuration class to store the configuration of a [`SamPromptEncoder`]. The [`SamPromptEncoder`]
    module is used to encode the input 2D points and bounding boxes. Instantiating a configuration defaults will yield
    a similar configuration to that of the SAM-vit-h
    [facebook/sam-vit-huge](https://huggingface.co/facebook/sam-vit-huge) architecture.

    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
    documentation from [`PretrainedConfig`] for more information.

    Args:
        hidden_size (`int`, *optional*, defaults to 256):
            Dimensionality of the hidden states.
        image_size (`int`, *optional*, defaults to 1024):
            The expected output resolution of the image.
        patch_size (`int`, *optional*, defaults to 16):
            The size (resolution) of each patch.
        mask_input_channels (`int`, *optional*, defaults to 16):
            The number of channels to be fed to the `MaskDecoder` module.
        num_point_embeddings (`int`, *optional*, defaults to 4):
            The number of point embeddings to be used.
        hidden_act (`str`, *optional*, defaults to `"gelu"`):
            The non-linear activation function in the encoder and pooler.
                geluư>c           	         sJ   t  jf i | || _|| _|| _|| | _|| _|| _|| _|| _	d S N)
super__init__hidden_size
image_size
patch_sizeZimage_embedding_sizemask_input_channelsnum_point_embeddings
hidden_actlayer_norm_eps)	selfr   r   r   r   r   r   r   kwargs	__class__ r/var/www/html/stable-diffusion-webui/venv/lib/python3.9/site-packages/transformers/models/sam/configuration_sam.pyr   9   s    
zSamPromptEncoderConfig.__init__)r   r   r   r   r	   r
   r   __name__
__module____qualname____doc__r   __classcell__r   r   r   r   r       s          r   c                
       s"   e Zd ZdZd fd	d
	Z  ZS )SamMaskDecoderConfiga  
    This is the configuration class to store the configuration of a [`SamMaskDecoder`]. It is used to instantiate a SAM
    mask decoder to the specified arguments, defining the model architecture. Instantiating a configuration defaults
    will yield a similar configuration to that of the SAM-vit-h
    [facebook/sam-vit-huge](https://huggingface.co/facebook/sam-vit-huge) architecture.

    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
    documentation from [`PretrainedConfig`] for more information.

    Args:
        hidden_size (`int`, *optional*, defaults to 256):
            Dimensionality of the hidden states.
        hidden_act (`str`, *optional*, defaults to `"relu"`):
            The non-linear activation function used inside the `SamMaskDecoder` module.
        mlp_dim (`int`, *optional*, defaults to 2048):
            Dimensionality of the "intermediate" (i.e., feed-forward) layer in the Transformer encoder.
        num_hidden_layers (`int`, *optional*, defaults to 2):
            Number of hidden layers in the Transformer encoder.
        num_attention_heads (`int`, *optional*, defaults to 8):
            Number of attention heads for each attention layer in the Transformer encoder.
        attention_downsample_rate (`int`, *optional*, defaults to 2):
            The downsampling rate of the attention layer.
        num_multimask_outputs (`int`, *optional*, defaults to 3):
            The number of outputs from the `SamMaskDecoder` module. In the Segment Anything paper, this is set to 3.
        iou_head_depth (`int`, *optional*, defaults to 3):
            The number of layers in the IoU head module.
        iou_head_hidden_dim (`int`, *optional*, defaults to 256):
            The dimensionality of the hidden states in the IoU head module.
        layer_norm_eps (`float`, *optional*, defaults to 1e-6):
            The epsilon used by the layer normalization layers.

    r   relu         r   r   c                    sR   t  jf i | || _|| _|| _|| _|| _|| _|| _|| _	|	| _
|
| _d S r   )r   r   r   r   mlp_dimnum_hidden_layersnum_attention_headsattention_downsample_ratenum_multimask_outputsiou_head_depthiou_head_hidden_dimr   )r   r   r   r'   r(   r)   r*   r+   r,   r-   r   r   r   r   r   r   q   s    zSamMaskDecoderConfig.__init__)
r   r#   r$   r%   r&   r%   r   r   r   r   r   r   r   r   r   r"   O   s   #          r"   c                       sT   e Zd ZdZddddddddd	d
ddddddddddg dddf fdd	Z  ZS )SamVisionConfiga-  
    This is the configuration class to store the configuration of a [`SamVisionModel`]. It is used to instantiate a SAM
    vision encoder according to the specified arguments, defining the model architecture. Instantiating a configuration
    defaults will yield a similar configuration to that of the SAM ViT-h
    [facebook/sam-vit-huge](https://huggingface.co/facebook/sam-vit-huge) architecture.

    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
    documentation from [`PretrainedConfig`] for more information.

    Args:
        hidden_size (`int`, *optional*, defaults to 768):
            Dimensionality of the encoder layers and the pooler layer.
        intermediate_size (`int`, *optional*, defaults to 6144):
            Dimensionality of the "intermediate" (i.e., feed-forward) layer in the Transformer encoder.
        projection_dim (`int`, *optional*, defaults to 512):
            Dimensionality of the projection layer in the Transformer encoder.
        output_channels (`int`, *optional*, defaults to 256):
            Dimensionality of the output channels in the Patch Encoder.
        num_hidden_layers (`int`, *optional*, defaults to 12):
            Number of hidden layers in the Transformer encoder.
        num_attention_heads (`int`, *optional*, defaults to 12):
            Number of attention heads for each attention layer in the Transformer encoder.
        num_channels (`int`, *optional*, defaults to 3):
            Number of channels in the input image.
        image_size (`int`, *optional*, defaults to 1024):
            Expected resolution. Target size of the resized input image.
        patch_size (`int`, *optional*, defaults to 16):
            Size of the patches to be extracted from the input image.
        hidden_act (`str`, *optional*, defaults to `"gelu"`):
            The non-linear activation function (function or string)
        layer_norm_eps (`float`, *optional*, defaults to 1e-6):
            The epsilon used by the layer normalization layers.
        dropout (`float`, *optional*, defaults to 0.0):
            The dropout probability.
        attention_dropout (`float`, *optional*, defaults to 0.0):
            The dropout ratio for the attention probabilities.
        initializer_range (`float`, *optional*, defaults to 1e-10):
            The standard deviation of the truncated_normal_initializer for initializing all weight matrices.
        initializer_factor (`float`, *optional*, defaults to 1.0):
            A factor for multiplying the initializer range.
        qkv_bias (`bool`, *optional*, defaults to `True`):
            Whether to add a bias to query, key, value projections.
        mlp_ratio (`float`, *optional*, defaults to 4.0):
            Ratio of mlp hidden dim to embedding dim.
        use_abs_pos (`bool`, *optional*, defaults to True):
            Whether to use absolute position embedding.
        use_rel_pos (`bool`, *optional*, defaults to True):
            Whether to use relative position embedding.
        window_size (`int`, *optional*, defaults to 14):
            Window size for relative position.
        global_attn_indexes (`List[int]`, *optional*, defaults to `[2, 5, 8, 11]`):
            The indexes of the global attention layers.
        num_pos_feats (`int`, *optional*, defaults to 128):
            The dimensionality of the position embedding.
        mlp_dim (`int`, *optional*, defaults to None):
            The dimensionality of the MLP layer in the Transformer encoder. If `None`, defaults to `mlp_ratio *
            hidden_size`.
    i   i   i   r      r   r   r   r
   r   g        g|=g      ?Tg      @   )r%      r&         Nc                    s   t  jf i | || _|| _|| _|| _|| _|| _|| _|| _	|	| _
|
| _|| _|| _|| _|| _|| _|| _|| _|| _|| _|| _|| _|| _|d u rt|| n|| _d S r   )r   r   r   intermediate_sizeprojection_dimoutput_channelsr(   r)   num_channelsr   r   r   r   dropoutattention_dropoutinitializer_rangeinitializer_factorqkv_bias	mlp_ratiouse_abs_posuse_rel_poswindow_sizeglobal_attn_indexesnum_pos_featsintr'   )r   r   r4   r5   r6   r(   r)   r7   r   r   r   r   r8   r9   r:   r;   r<   r=   r>   r?   r@   rA   rB   r'   r   r   r   r   r      s0    zSamVisionConfig.__init__r   r   r   r   r   r.      s2   =r.   c                       s2   e Zd ZdZdZdZd
 fdd	Zdd	 Z  ZS )	SamConfiga  
    [`SamConfig`] is the configuration class to store the configuration of a [`SamModel`]. It is used to instantiate a
    SAM model according to the specified arguments, defining the vision model, prompt-encoder model and mask decoder
    configs. Instantiating a configuration with the defaults will yield a similar configuration to that of the
    SAM-ViT-H [facebook/sam-vit-huge](https://huggingface.co/facebook/sam-vit-huge) architecture.

    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
    documentation from [`PretrainedConfig`] for more information.

    Args:
        vision_config (Union[`dict`, `SamVisionConfig`], *optional*):
            Dictionary of configuration options used to initialize [`SamVisionConfig`].
        prompt_encoder_config (Union[`dict`, `SamPromptEncoderConfig`], *optional*):
            Dictionary of configuration options used to initialize [`SamPromptEncoderConfig`].
        mask_decoder_config (Union[`dict`, `SamMaskDecoderConfig`], *optional*):
            Dictionary of configuration options used to initialize [`SamMaskDecoderConfig`].

        kwargs (*optional*):
            Dictionary of keyword arguments.

    Example:

    ```python
    >>> from transformers import (
    ...     SamVisionConfig,
    ...     SamPromptEncoderConfig,
    ...     SamMaskDecoderConfig,
    ...     SamModel,
    ... )

    >>> # Initializing a SamConfig with `"facebook/sam-vit-huge"` style configuration
    >>> configuration = SamConfig()

    >>> # Initializing a SamModel (with random weights) from the `"facebook/sam-vit-huge"` style configuration
    >>> model = SamModel(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config

    >>> # We can also initialize a SamConfig from a SamVisionConfig, SamPromptEncoderConfig, and SamMaskDecoderConfig

    >>> # Initializing SAM vision, SAM Q-Former and language model configurations
    >>> vision_config = SamVisionConfig()
    >>> prompt_encoder_config = SamPromptEncoderConfig()
    >>> mask_decoder_config = SamMaskDecoderConfig()

    >>> config = SamConfig(vision_config, prompt_encoder_config, mask_decoder_config)
    ```ZsamTN{Gz?c                    s   t  jf i | |d ur|ni }|d ur.|ni }|d ur>|ni }t|trT| }t|trf| }t|trx| }tf i || _tf i || _tf i || _	|| _
d S r   )r   r   
isinstancer.   to_dictr   r"   vision_configprompt_encoder_configmask_decoder_configr:   )r   rH   rI   rJ   r:   r   r   r   r   r   3  s    


zSamConfig.__init__c                 C   sF   t | j}| j |d< | j |d< | j |d< | jj|d< |S )z
        Serializes this instance to a Python dictionary. Override the default [`~PretrainedConfig.to_dict`].

        Returns:
            `Dict[str, any]`: Dictionary of all the attributes that make up this configuration instance,
        rH   rI   rJ   
model_type)	copydeepcopy__dict__rH   rG   rI   rJ   r   rK   )r   outputr   r   r   rG   L  s    zSamConfig.to_dict)NNNrE   )	r   r   r   r    rK   Zis_compositionr   rG   r!   r   r   r   r   rD      s   1    rD   )r    rL   Zconfiguration_utilsr   utilsr   Z
get_loggerr   loggerZ!SAM_PRETRAINED_CONFIG_ARCHIVE_MAPr   r"   r.   rD   r   r   r   r   <module>   s   
/=r