
    ^j                     j    d dl mZmZ d dlmZ d dlmZ  ed      e G d de                    ZdgZy)	   )PreTrainedConfigstrict)RopeParameters)auto_docstringzEuroBERT/EuroBERT-210m)
checkpointc                       e Zd ZU dZdZdgZddddddddZdgdgfd	d
gd	gfd	gd	gfdZdZe	e
d<   dZe	e
d<   dZe	e
d<   dZe	e
d<   dZe	e
d<   dZe	dz  e
d<   dZee
d<   dZe	e
d<   dZee
d<   dZee
d<   dZee
d <   d!Ze	dz  e
d"<   d#Ze	dz  e
d$<   d!Ze	ee	   z  dz  e
d%<   d&Ze	e
d'<   d(Zee
d)<   dZee z  dz  e
d*<   d(Z!ee
d+<   d,Z"e	ez  e
d-<   d(Z#ee
d.<   dZ$e	dz  e
d/<   d0Z%e	e
d1<   d2Z&ee
d3<    fd4Z'd5 Z( xZ)S )6EuroBertConfiga  
    mask_token_id (`int`, *optional*, defaults to 128002):
        Mask token id.
    classifier_pooling (`str`, *optional*, defaults to `"late"`):
        The pooling strategy to use for the classifier. Can be one of ['bos', 'mean', 'late'].

    ```python
    >>> from transformers import EuroBertModel, EuroBertConfig

    >>> # Initializing a EuroBert eurobert-base style configuration
    >>> configuration = EuroBertConfig()

    >>> # Initializing a model from the eurobert-base style configuration
    >>> model = EuroBertModel(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```eurobertpast_key_valuescolwiserowwise)zlayers.*.self_attn.q_projzlayers.*.self_attn.k_projzlayers.*.self_attn.v_projzlayers.*.self_attn.o_projzlayers.*.mlp.gate_projzlayers.*.mlp.up_projzlayers.*.mlp.down_proj	input_idsinputs_embedshidden_statesattention_mask)embed_tokenslayersnormi  
vocab_sizei   hidden_sizei   intermediate_size   num_hidden_layersnum_attention_headsNnum_key_value_headssilu
hidden_acti    max_position_embeddingsg{Gz?initializer_rangegh㈵>rms_norm_epsT	use_cachei pad_token_idi  bos_token_ideos_token_id   pretraining_tpFtie_word_embeddingsrope_parametersattention_biasg        attention_dropoutmlp_biashead_dimi mask_token_idlateclassifier_poolingc                     | j                   | j                  | _         | j                  | j                  | j                  z  | _        | j                   | j                  | _         t	        |   di | y )N )r   r   r,   r   super__post_init__)selfkwargs	__class__s     ~/var/www/ramen.bs-engineer-server.com/venv/lib/python3.12/site-packages/transformers/models/eurobert/configuration_eurobert.pyr3   zEuroBertConfig.__post_init__\   si    ##+'+'?'?D$==  ,,0H0HHDM##+'+'?'?D$''    c                     | j                   | j                  z  dk7  r&t        d| j                    d| j                   d      y)zOPart of `@strict`-powered validation. Validates the architecture of the config.    zThe hidden size (z6) is not a multiple of the number of attention heads (z).N)r   r   
ValueError)r4   s    r7   validate_architecturez$EuroBertConfig.validate_architecturef   sS    d666!;#D$4$4#5 622327  <r8   )*__name__
__module____qualname____doc__
model_typekeys_to_ignore_at_inferencebase_model_tp_planbase_model_pp_planr   int__annotations__r   r   r   r   r   r   strr   r   floatr    r!   boolr"   r#   r$   listr&   r'   r(   r   dictr)   r*   r+   r,   r-   r/   r3   r<   __classcell__)r6   s   @r7   r	   r	      s   & J#4"5 &/%.%.%."+ )"+ &(9:#%568IJ!"_$56 JK!s!s!!&*t*J#'S'#u#L%It%L#*%%L#*%+1L#S	/D(1NC %%48O^d*T18 ND %(sU{(HdHcDjM3$$(r8   r	   N)	configuration_utilsr   r   modeling_rope_utilsr   utilsr   r	   __all__r1   r8   r7   <module>rQ      sH   . < 1 # 34N% N  5Nb 
r8   