
    ^j                         d Z ddlZddlZddlmZ ddlmZmZmZ ddlm	Z
 ddlmZ ddlmZmZmZ dd	lmZ dd
lmZ ddlmZ ddlmZmZmZmZmZmZmZmZ ddl m!Z! ddl"m#Z# ddl$m%Z%m&Z& ddl'm(Z(  e&jR                  e*      Z+ G d dejX                        Z- G d dejX                        Z. G d dejX                        Z/ G d dejX                        Z0 G d dejX                        Z1 G d dejX                        Z2 G d dejX                        Z3 G d  d!e      Z4 G d" d#ejX                        Z5 G d$ d%ejX                        Z6 G d& d'ejX                        Z7 G d( d)ejX                        Z8e% G d* d+e!             Z9 e%d,-       G d. d/e9             Z:e% G d0 d1e9             Z; e%d2-       G d3 d4e9e             Z< e%d5-       G d6 d7e9             Z=e% G d8 d9e9             Z>e% G d: d;e9             Z?e% G d< d=e9             Z@g d>ZAy)?zPyTorch RemBERT model.    N)nn)BCEWithLogitsLossCrossEntropyLossMSELoss   )initialization)ACT2FN)CacheDynamicCacheEncoderDecoderCache)GenerationMixin)create_bidirectional_mask)GradientCheckpointingLayer))BaseModelOutputWithPastAndCrossAttentions,BaseModelOutputWithPoolingAndCrossAttentions!CausalLMOutputWithCrossAttentionsMaskedLMOutputMultipleChoiceModelOutputQuestionAnsweringModelOutputSequenceClassifierOutputTokenClassifierOutput)PreTrainedModel)apply_chunking_to_forward)auto_docstringlogging   )RemBertConfigc                        e Zd ZdZ fdZ	 	 	 	 	 ddej                  dz  dej                  dz  dej                  dz  dej                  dz  ded	ej                  fd
Z
 xZS )RemBertEmbeddingszGConstruct the embeddings from word, position and token_type embeddings.c                 |   t         |           t        j                  |j                  |j
                  |j                        | _        t        j                  |j                  |j
                        | _	        t        j                  |j                  |j
                        | _        t        j                  |j
                  |j                        | _        t        j                  |j                        | _        | j#                  dt%        j&                  |j                        j)                  d      d       y )N)padding_idxepsposition_idsr   F)
persistent)super__init__r   	Embedding
vocab_sizeinput_embedding_sizepad_token_idword_embeddingsmax_position_embeddingsposition_embeddingstype_vocab_sizetoken_type_embeddings	LayerNormlayer_norm_epsDropouthidden_dropout_probdropoutregister_buffertorcharangeexpandselfconfig	__class__s     w/var/www/ramen.bs-engineer-server.com/venv/lib/python3.12/site-packages/transformers/models/rembert/modeling_rembert.pyr)   zRemBertEmbeddings.__init__2   s    !||v::H[H[ 
 $&<<0N0NPVPkPk#l %'\\&2H2H&JeJe%f"f&A&AvG\G\]zz&"<"<= 	ELL)G)GHOOPWXej 	 	
    N	input_idstoken_type_idsr$   inputs_embedspast_key_values_lengthreturnc                    ||j                         }n|j                         d d }|d   }|| j                  d d |||z   f   }|:t        j                  |t        j                  | j                  j
                        }|| j                  |      }| j                  |      }||z   }	| j                  |      }
|	|
z  }	| j                  |	      }	| j                  |	      }	|	S )Nr&   r   dtypedevice)sizer$   r9   zeroslongrJ   r.   r2   r0   r3   r7   )r=   rB   rC   r$   rD   rE   input_shape
seq_lengthr2   
embeddingsr0   s              r@   forwardzRemBertEmbeddings.forwardB   s      #..*K',,.s3K ^
,,Q0FVlIl0l-lmL!"[[EJJtO`O`OgOghN  00;M $ : :> J"%::
"66|D))
^^J/
\\*-
rA   )NNNNr   )__name__
__module____qualname____doc__r)   r9   
LongTensorFloatTensorintTensorrQ   __classcell__r?   s   @r@   r   r   /   s    Q
$ .2260426&'##d* ((4/ &&-	
 ((4/ !$ 
rA   r   c                   V     e Zd Z fdZdej
                  dej
                  fdZ xZS )RemBertPoolerc                     t         |           t        j                  |j                  |j                        | _        t        j                         | _        y N)r(   r)   r   Linearhidden_sizedenseTanh
activationr<   s     r@   r)   zRemBertPooler.__init__e   s9    YYv1163E3EF
'')rA   hidden_statesrF   c                 \    |d d df   }| j                  |      }| j                  |      }|S )Nr   )rb   rd   )r=   re   first_token_tensorpooled_outputs       r@   rQ   zRemBertPooler.forwardj   s6     +1a40

#566rA   rR   rS   rT   r)   r9   rY   rQ   rZ   r[   s   @r@   r]   r]   d   s#    $
U\\ ell rA   r]   c                        e Zd Zd
 fd	Z	 	 	 	 ddej
                  dej                  dz  dej                  dz  dedz  dede	fd	Z
 xZS )RemBertSelfAttentionNc                    t         |           |j                  |j                  z  dk7  r2t	        |d      s&t        d|j                   d|j                   d      |j                  | _        t        |j                  |j                  z        | _        | j                  | j                  z  | _        t        j                  |j                  | j                        | _        t        j                  |j                  | j                        | _        t        j                  |j                  | j                        | _        t        j                  |j                        | _        |j"                  | _        || _        y )Nr   embedding_sizezThe hidden size (z6) is not a multiple of the number of attention heads ())r(   r)   ra   num_attention_headshasattr
ValueErrorrX   attention_head_sizeall_head_sizer   r`   querykeyvaluer5   attention_probs_dropout_probr7   
is_decoder	layer_idxr=   r>   ry   r?   s      r@   r)   zRemBertSelfAttention.__init__t   s0    : ::a?PVXhHi#F$6$6#7 8 445Q8 
 $*#=#= #&v'9'9F<V<V'V#W !558P8PPYYv1143E3EF
99V//1C1CDYYv1143E3EF
zz&"E"EF ++"rA   re   attention_maskencoder_hidden_statespast_key_valuesoutput_attentionsrF   c                 v   |j                   d d }g |d| j                  }| j                  |      j                  |      j	                  dd      }	d}
|d u}|St        |t              rA|j                  j                  | j                        }
|r|j                  }n|j                  }n|}|r|n|}|rK|I|
rGj                  | j                     j                  }|j                  | j                     j                  }ng |j                   d d d| j                  }| j                  |      j                  |      j	                  dd      }| j!                  |      j                  |      j	                  dd      }|Kj#                  ||| j                        \  }}|r)t        |t              rd|j                  | j                  <   t%        j&                  |	|j	                  dd            }|t)        j*                  | j                        z  }|||z   }t,        j.                  j1                  |d      }| j3                  |      }t%        j&                  ||      }|j5                  dddd	      j7                         }|j9                         d d | j:                  fz   } |j                  | }||fS )
Nr&   r      FTdimr   r   )shaperr   rt   view	transpose
isinstancer   
is_updatedgetry   cross_attention_cacheself_attention_cachelayerskeysvaluesru   rv   updater9   matmulmathsqrtr   
functionalsoftmaxr7   permute
contiguousrK   rs   )r=   re   r{   r|   r}   r~   kwargsrN   hidden_shapequery_layerr   is_cross_attentioncurr_past_key_valuescurrent_states	key_layervalue_layerkv_shapeattention_scoresattention_probscontext_layernew_context_layer_shapes                        r@   rQ   zRemBertSelfAttention.forward   s    $))#2.CCbC$*B*BCjj/44\BLLQPQR
2$>&/+>?,77;;DNNK
%+:+P+P(+:+O+O('6$2D.-/"=*,33DNNCHHI.55dnnELLKQ--cr2QBQ8P8PQH055h?II!QOI**^499(CMMaQRSK*)=)D)DYP[]a]k]k)l&	;%*_FY*ZAEO..t~~> !<<Y5H5HR5PQ+dii8P8P.QQ%/.@ --//0@b/I ,,7_kB%--aAq9DDF"/"4"4"6s";t?Q?Q>S"S***,CDo--rA   r_   NNNFrR   rS   rT   r)   r9   rY   rW   r
   booltuplerQ   rZ   r[   s   @r@   rk   rk   s   sz    #0 48:>(,"'@.||@. ))D0@.  %0047	@.
 @.  @. 
@.rA   rk   c                   n     e Zd Z fdZdej
                  dej
                  dej
                  fdZ xZS )RemBertSelfOutputc                 (   t         |           t        j                  |j                  |j                        | _        t        j                  |j                  |j                        | _        t        j                  |j                        | _
        y Nr"   )r(   r)   r   r`   ra   rb   r3   r4   r5   r6   r7   r<   s     r@   r)   zRemBertSelfOutput.__init__   s`    YYv1163E3EF
f&8&8f>S>STzz&"<"<=rA   re   input_tensorrF   c                 r    | j                  |      }| j                  |      }| j                  ||z         }|S r_   rb   r7   r3   r=   re   r   s      r@   rQ   zRemBertSelfOutput.forward   7    

=1]3}|'CDrA   ri   r[   s   @r@   r   r      1    >U\\  RWR^R^ rA   r   c                        e Zd Zd
 fd	Z	 	 	 	 ddej
                  dej                  dz  dej                  dz  dedz  dedz  de	ej
                     fd	Z
 xZS )RemBertAttentionNc                 f    t         |           t        ||      | _        t	        |      | _        y )Nry   )r(   r)   rk   r=   r   outputrz   s      r@   r)   zRemBertAttention.__init__   s(    (9E	'/rA   re   r{   r|   r}   r~   rF   c                 n    | j                  |||||      }| j                  |d   |      }|f|dd  z   }	|	S )Nr{   r|   r}   r~   r   r   )r=   r   )
r=   re   r{   r|   r}   r~   r   self_outputsattention_outputoutputss
             r@   rQ   zRemBertAttention.forward   sV     yy)"7+/ ! 
  ;;|AF#%QR(88rA   r_   r   r   r[   s   @r@   r   r      s    0 48:>(,).|| ))D0  %0047	
   $; 
u||	rA   r   c                   V     e Zd Z fdZdej
                  dej
                  fdZ xZS )RemBertIntermediatec                    t         |           t        j                  |j                  |j
                        | _        t        |j                  t              rt        |j                     | _        y |j                  | _        y r_   )r(   r)   r   r`   ra   intermediate_sizerb   r   
hidden_actstrr	   intermediate_act_fnr<   s     r@   r)   zRemBertIntermediate.__init__   s]    YYv1163K3KL
f''-'-f.?.?'@D$'-'8'8D$rA   re   rF   c                 J    | j                  |      }| j                  |      }|S r_   )rb   r   r=   re   s     r@   rQ   zRemBertIntermediate.forward   s&    

=100?rA   ri   r[   s   @r@   r   r      s#    9U\\ ell rA   r   c                   n     e Zd Z fdZdej
                  dej
                  dej
                  fdZ xZS )RemBertOutputc                 (   t         |           t        j                  |j                  |j
                        | _        t        j                  |j
                  |j                        | _        t        j                  |j                        | _        y r   )r(   r)   r   r`   r   ra   rb   r3   r4   r5   r6   r7   r<   s     r@   r)   zRemBertOutput.__init__  s`    YYv779K9KL
f&8&8f>S>STzz&"<"<=rA   re   r   rF   c                 r    | j                  |      }| j                  |      }| j                  ||z         }|S r_   r   r   s      r@   rQ   zRemBertOutput.forward  r   rA   ri   r[   s   @r@   r   r     r   rA   r   c                        e Zd Zd fd	Z	 	 	 	 	 ddej
                  dej                  dz  dej                  dz  dej                  dz  dedz  dedz  d	e	ej
                     fd
Z
d Z xZS )RemBertLayerNc                 h   t         |           |j                  | _        d| _        t	        ||      | _        |j                  | _        |j                  | _        | j                  r,| j                  st        |  d      t	        ||      | _	        t        |      | _        t        |      | _        y )Nr   z> should be used as a decoder model if cross attention is addedr   )r(   r)   chunk_size_feed_forwardseq_len_dimr   	attentionrx   add_cross_attentionrq   crossattentionr   intermediater   r   rz   s      r@   r)   zRemBertLayer.__init__  s    '-'E'E$)&)< ++#)#=#= ##?? D6)g!hii"26Y"OD/7#F+rA   re   r{   r|   encoder_attention_maskr}   r~   rF   c                 @   | j                  ||||      }|d   }	|dd  }
| j                  r@|>t        | d      st        d|  d      | j	                  |	||||      }|d   }	|
|dd  z   }
t        | j                  | j                  | j                  |	      }|f|
z   }
|
S )N)r{   r~   r}   r   r   r   z'If `encoder_hidden_states` are passed, z` has to be instantiated with cross-attention layers by setting `config.add_cross_attention=True`r   )	r   rx   rp   rq   r   r   feed_forward_chunkr   r   )r=   re   r{   r|   r   r}   r~   r   self_attention_outputsr   r   cross_attention_outputslayer_outputs                r@   rQ   zRemBertLayer.forward%  s     "&)/+	 "0 "
 2!4(,??4@4!12 =dV DD D 
 '+&9&9 5&; /"3 ': '#  7q9 7 ;;G0##T%A%A4CSCSUe
  /G+rA   c                 L    | j                  |      }| j                  ||      }|S r_   )r   r   )r=   r   intermediate_outputr   s       r@   r   zRemBertLayer.feed_forward_chunkQ  s,    "//0@A{{#68HIrA   r_   )NNNNF)rR   rS   rT   r)   r9   rY   rW   r
   r   r   rQ   r   rZ   r[   s   @r@   r   r     s    ,$ 48:>;?(,).)||) ))D0)  %0047	)
 !& 1 1D 8) )  $;) 
u||	)XrA   r   c                        e Zd Z fdZ	 	 	 	 	 	 	 	 ddej
                  dej                  dz  dej                  dz  dej                  dz  dedz  dedz  d	ed
edede	e
z  fdZ xZS )RemBertEncoderc           	      2   t         |           || _        t        j                  |j
                  |j                        | _        t        j                  t        |j                        D cg c]  }t        ||       c}      | _        d| _        y c c}w )Nr   F)r(   r)   r>   r   r`   r,   ra   embedding_hidden_mapping_in
ModuleListrangenum_hidden_layersr   layergradient_checkpointing)r=   r>   ir?   s      r@   r)   zRemBertEncoder.__init__X  sq    +-99V5P5PRXRdRd+e(]]uU[UmUmOn#o!L1$E#op
&+# $ps   ,BNre   r{   r|   r   r}   	use_cacher~   output_hidden_statesreturn_dictrF   c
           	      n   | j                   r%| j                  r|rt        j                  d       d}|r6|4t	        t        | j                        t        | j                              }| j                  |      }|rdnd }|rdnd }|r| j                  j                  rdnd }t        | j                        D ]K  \  }}|r||fz   } |||||||      }|d   }|s#||d   fz   }| j                  j                  sC||d   fz   }M |r||fz   }|	st        d |||||fD              S t        |||||	      S )
NzZ`use_cache=True` is incompatible with gradient checkpointing. Setting `use_cache=False`...F)r>    r   r   r   c              3   $   K   | ]  }|| 
 y wr_   r   ).0vs     r@   	<genexpr>z)RemBertEncoder.forward.<locals>.<genexpr>  s      
 = 
s   )last_hidden_stater}   re   
attentionscross_attentions)r   trainingloggerwarning_oncer   r   r>   r   r   	enumerater   r   r   )r=   re   r{   r|   r   r}   r   r~   r   r   r   all_hidden_statesall_self_attentionsall_cross_attentionsr   layer_modulelayer_outputss                    r@   rQ   zRemBertEncoder.forward`  s    &&4==##p "	01,dkk2RT`hlhshsTtuO88G"6BD$5b4%64;;;Z;Zr`d(4 	VOA|#$58H$H!(%&!M *!,M &9]1=M<O&O#;;22+?=QRCSBU+U(#	V&   1]4D D 
 "#%'(
 
 
 9+++*1
 	
rA   )NNNNNFFT)rR   rS   rT   r)   r9   rY   rW   r
   r   r   r   rQ   rZ   r[   s   @r@   r   r   W  s    , 48:>;?(,!%"'%* D
||D
 ))D0D
  %0047	D

 !& 1 1D 8D
 D
 $;D
  D
 #D
 D
 
:	:D
rA   r   c                   V     e Zd Z fdZdej
                  dej
                  fdZ xZS )RemBertPredictionHeadTransformc                 h   t         |           t        j                  |j                  |j                        | _        t        |j                  t              rt        |j                     | _
        n|j                  | _
        t        j                  |j                  |j                        | _        y r   )r(   r)   r   r`   ra   rb   r   r   r   r	   transform_act_fnr3   r4   r<   s     r@   r)   z'RemBertPredictionHeadTransform.__init__  s{    YYv1163E3EF
f''-$*6+<+<$=D!$*$5$5D!f&8&8f>S>STrA   re   rF   c                 l    | j                  |      }| j                  |      }| j                  |      }|S r_   )rb   r   r3   r   s     r@   rQ   z&RemBertPredictionHeadTransform.forward  s4    

=1--m<}5rA   ri   r[   s   @r@   r   r     s$    UU\\ ell rA   r   c                   V     e Zd Z fdZdej
                  dej
                  fdZ xZS )RemBertLMPredictionHeadc                 n   t         |           t        j                  |j                  |j
                        | _        t        j                  |j
                  |j                        | _        t        |j                     | _        t        j                  |j
                  |j                        | _        y r   )r(   r)   r   r`   ra   output_embedding_sizerb   r+   decoderr	   r   rd   r3   r4   r<   s     r@   r)   z RemBertLMPredictionHead.__init__  sz    YYv1163O3OP
yy!=!=v?P?PQ !2!23f&B&BH]H]^rA   re   rF   c                     | j                  |      }| j                  |      }| j                  |      }| j                  |      }|S r_   )rb   rd   r3   r  r   s     r@   rQ   zRemBertLMPredictionHead.forward  s@    

=16}5]3rA   ri   r[   s   @r@   r   r     s$    _U\\ ell rA   r   c                   V     e Zd Z fdZdej
                  dej
                  fdZ xZS )RemBertOnlyMLMHeadc                 B    t         |           t        |      | _        y r_   )r(   r)   r   predictionsr<   s     r@   r)   zRemBertOnlyMLMHead.__init__  s    26:rA   sequence_outputrF   c                 (    | j                  |      }|S r_   )r  )r=   r  prediction_scoress      r@   rQ   zRemBertOnlyMLMHead.forward  s     ,,_=  rA   ri   r[   s   @r@   r  r    s#    ;!u|| ! !rA   r  c                   2     e Zd ZU eed<   dZdZ fdZ xZS )RemBertPreTrainedModelr>   rembertTc                     t         |   |       t        |t              rZt	        j
                  |j                  t        j                  |j                  j                  d         j                  d             y y )Nr&   r%   )r(   _init_weightsr   r   initcopy_r$   r9   r:   r   r;   )r=   moduler?   s     r@   r  z$RemBertPreTrainedModel._init_weights  s[    f%f/0JJv**ELL9L9L9R9RSU9V,W,^,^_f,gh 1rA   )	rR   rS   rT   r   __annotations__base_model_prefixsupports_gradient_checkpointingr  rZ   r[   s   @r@   r  r    s!    !&*#i irA   r  a
  
    The model can behave as an encoder (with only self-attention) as well as a decoder, in which case a layer of
    cross-attention is added between the self-attention layers, following the architecture described in [Attention is
    all you need](https://huggingface.co/papers/1706.03762) by Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit,
    Llion Jones, Aidan N. Gomez, Lukasz Kaiser and Illia Polosukhin.

    To behave as an decoder the model needs to be initialized with the `is_decoder` argument of the configuration set
    to `True`. To be used in a Seq2Seq model, the model needs to initialized with both `is_decoder` argument and
    `add_cross_attention` set to `True`; an `encoder_hidden_states` is then expected as an input to the forward pass.
    )custom_introc                   f    e Zd Zd fd	Zd Zd Ze	 	 	 	 	 	 	 	 	 	 	 	 ddej                  dz  dej                  dz  dej                  dz  dej                  dz  d	ej                  dz  d
ej                  dz  dej                  dz  de
dz  dedz  dedz  dedz  dedz  deez  fd       Z xZS )RemBertModelc                     t         |   |       || _        t        |      | _        t        |      | _        |rt        |      nd| _        | j                          y)zv
        add_pooling_layer (bool, *optional*, defaults to `True`):
            Whether to add a pooling layer
        N)
r(   r)   r>   r   rP   r   encoderr]   pooler	post_init)r=   r>   add_pooling_layerr?   s      r@   r)   zRemBertModel.__init__  sM    
 	 +F3%f-/@mF+d 	rA   c                 .    | j                   j                  S r_   rP   r.   r=   s    r@   get_input_embeddingsz!RemBertModel.get_input_embeddings  s    ...rA   c                 &    || j                   _        y r_   r  )r=   rv   s     r@   set_input_embeddingsz!RemBertModel.set_input_embeddings   s    */'rA   NrB   r{   rC   r$   rD   r|   r   r}   r   r~   r   r   rF   c                 8   |
|
n| j                   j                  }
||n| j                   j                  }||n| j                   j                  }| j                   j                  r|	|	n| j                   j
                  }	nd}	||t        d      |#| j                  ||       |j                         }n!||j                         d d }nt        d      |\  }}||j                  n|j                  }|dn|j                         }|t        j                  |||z   f|      }|&t        j                  |t        j                  |      }| j                  |||||      }t!        | j                   ||	      }|t!        | j                   |||
      }| j#                  ||||||	|
||	      }|d   }| j$                  | j%                  |      nd }|s
||f|dd  z   S t'        |||j(                  |j*                  |j,                  |j.                        S )NFzDYou cannot specify both input_ids and inputs_embeds at the same timer&   z5You have to specify either input_ids or inputs_embedsr   )rJ   rH   )rB   r$   rC   rD   rE   )r>   rD   r{   )r>   rD   r{   r|   )r{   r|   r   r}   r   r~   r   r   r   )r   pooler_outputr}   re   r   r   )r>   r~   r   r   rx   r   rq   %warn_if_padding_and_no_attention_maskrK   rJ   get_seq_lengthr9   onesrL   rM   rP   r   r  r  r   r}   re   r   r   )r=   rB   r{   rC   r$   rD   r|   r   r}   r   r~   r   r   r   rN   
batch_sizerO   rJ   rE   embedding_outputencoder_outputsr  rh   s                          r@   rQ   zRemBertModel.forward  sV   " 2C1N-TXT_T_TqTq$8$D $++JjJj 	 &1%<k$++BYBY;;!!%.%:	@U@UII ]%>cdd"66y.Q#..*K&',,.s3KTUU!,
J%.%:!!@T@T&5&=?CaCaCc!"ZZ*jCY6Y)ZdjkN!"[[EJJvVN??%)'#9 + 
 3;;*)
 "-%>{{.5&;	&" ,,)"7#9+/!5# ' 

 *!,8<8OO4UY#]3oab6III;-'+;;)77&11,==
 	
rA   )T)NNNNNNNNNNNN)rR   rS   rT   r)   r!  r#  r   r9   rV   rW   r
   r   r   r   rQ   rZ   r[   s   @r@   r  r    sB    /0  .226260426:>;?(,!%)-,0#']
##d*]
 ((4/]
 ((4/	]

 &&-]
 ((4/]
  %0047]
 !& 1 1D 8]
 ]
 $;]
  $;]
 #Tk]
 D[]
 
=	=]
 ]
rA   r  c                   l    e Zd Z fdZd Zd Ze	 	 	 	 	 	 	 	 	 	 	 ddej                  dz  dej                  dz  dej                  dz  dej                  dz  d	ej                  dz  d
ej                  dz  dej                  dz  dej                  dz  de
dz  de
dz  de
dz  deez  fd       Z xZS )RemBertForMaskedLMc                     t         |   |       |j                  rt        j	                  d       t        |d      | _        t        |      | _        | j                          y )NznIf you want to use `RemBertForMaskedLM` make sure `config.is_decoder=False` for bi-directional self-attention.Fr  
r(   r)   rx   r   warningr  r  r  clsr  r<   s     r@   r)   zRemBertForMaskedLM.__init__f  sR     NN1
 $FeD%f- 	rA   c                 B    | j                   j                  j                  S r_   r2  r  r  r   s    r@   get_output_embeddingsz(RemBertForMaskedLM.get_output_embeddingsu      xx##+++rA   c                 :    || j                   j                  _        y r_   r4  r=   new_embeddingss     r@   set_output_embeddingsz(RemBertForMaskedLM.set_output_embeddingsx      '5$rA   NrB   r{   rC   r$   rD   r|   r   labelsr~   r   r   rF   c                    ||n| j                   j                  }| j                  ||||||||	|
|
      }|d   }| j                  |      }d}|Ft	               } ||j                  d| j                   j                        |j                  d            }|s|f|dd z   }||f|z   S |S t        |||j                  |j                        S )a  
        labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Labels for computing the masked language modeling loss. Indices should be in `[-100, 0, ...,
            config.vocab_size]` (see `input_ids` docstring) Tokens with indices set to `-100` are ignored (masked), the
            loss is only computed for the tokens with labels in `[0, ..., config.vocab_size]`.
        N)	r{   rC   r$   rD   r|   r   r~   r   r   r   r&   r   losslogitsre   r   )
r>   r   r  r2  r   r   r+   r   re   r   )r=   rB   r{   rC   r$   rD   r|   r   r<  r~   r   r   r   r   r  r
  masked_lm_lossloss_fctr   s                      r@   rQ   zRemBertForMaskedLM.forward{  s    , &1%<k$++BYBY,,))%'"7#9/!5#  
 "!* HH_5')H%&7&<&<RAWAW&XZ`ZeZefhZijN')GABK7F3A3M^%.YSYY$!//))	
 	
rA   )NNNNNNNNNNN)rR   rS   rT   r)   r5  r:  r   r9   rV   rW   r   r   r   rQ   rZ   r[   s   @r@   r-  r-  d  s(   ,6  .226260426:>;?*.)-,0#'5
##d*5
 ((4/5
 ((4/	5

 &&-5
 ((4/5
  %00475
 !& 1 1D 85
   4'5
  $;5
 #Tk5
 D[5
 
	5
 5
rA   r-  zS
    RemBERT Model with a `language modeling` head on top for CLM fine-tuning.
    c            !           e Zd Z fdZd Zd Ze	 	 	 	 	 	 	 	 	 	 	 	 	 	 ddej                  dz  dej                  dz  dej                  dz  dej                  dz  d	ej                  dz  d
ej                  dz  dej                  dz  de
dz  dej                  dz  dedz  dedz  dedz  dedz  deej                  z  deez  fd       Z xZS )RemBertForCausalLMc                     t         |   |       |j                  st        j	                  d       t        |d      | _        t        |      | _        | j                          y )NzOIf you want to use `RemBertForCausalLM` as a standalone, add `is_decoder=True.`Fr/  r0  r<   s     r@   r)   zRemBertForCausalLM.__init__  sL       NNlm#FeD%f- 	rA   c                 B    | j                   j                  j                  S r_   r4  r   s    r@   r5  z(RemBertForCausalLM.get_output_embeddings  r6  rA   c                 :    || j                   j                  _        y r_   r4  r8  s     r@   r:  z(RemBertForCausalLM.set_output_embeddings  r;  rA   NrB   r{   rC   r$   rD   r|   r   r}   r<  r   r~   r   r   logits_to_keeprF   c                    ||n| j                   j                  }| j                  |||||||||
|||      }|d   }t        |t              rt        | d      n|}| j                  |dd|ddf         }d}|	* | j                  d||	| j                   j                  d|}|s|f|dd z   }||f|z   S |S t        |||j                  |j                  |j                  |j                        S )a  
        labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Labels for computing the left-to-right language modeling loss (next word prediction). Indices should be in
            `[-100, 0, ..., config.vocab_size]` (see `input_ids` docstring) Tokens with indices set to `-100` are
            ignored (masked), the loss is only computed for the tokens with labels n `[0, ..., config.vocab_size]`.

        Example:

        ```python
        >>> from transformers import AutoTokenizer, RemBertForCausalLM, RemBertConfig
        >>> import torch

        >>> tokenizer = AutoTokenizer.from_pretrained("google/rembert")
        >>> config = RemBertConfig.from_pretrained("google/rembert")
        >>> config.is_decoder = True
        >>> model = RemBertForCausalLM.from_pretrained("google/rembert", config=config)

        >>> inputs = tokenizer("Hello, my dog is cute", return_tensors="pt")
        >>> outputs = model(**inputs)

        >>> prediction_logits = outputs.logits
        ```N)r{   rC   r$   rD   r|   r   r}   r   r~   r   r   r   )r@  r<  r+   r   )r?  r@  r}   re   r   r   r   )r>   r   r  r   rX   slicer2  loss_functionr+   r   r}   re   r   r   )r=   rB   r{   rC   r$   rD   r|   r   r}   r<  r   r~   r   r   rH  r   r   re   slice_indicesr@  r?  r   s                         r@   rQ   zRemBertForCausalLM.forward  s)   R &1%<k$++BYBY,,))%'"7#9+/!5#  
  
8B>SV8W~ot4]k-=!(;<=%4%%pVFt{{OeOepiopDY,F)-)9TGf$EvE0#33!//))$55
 	
rA   )NNNNNNNNNNNNNr   )rR   rS   rT   r)   r5  r:  r   r9   rV   rW   r
   r   rX   rY   r   r   rQ   rZ   r[   s   @r@   rD  rD    sr   
,6  .226260426:>;?(,*.!%)-,0#'-.M
##d*M
 ((4/M
 ((4/	M

 &&-M
 ((4/M
  %0047M
 !& 1 1D 8M
 M
   4'M
 $;M
  $;M
 #TkM
 D[M
 ell*M
" 
2	2#M
 M
rA   rD  z
    RemBERT Model transformer with a sequence classification/regression head on top (a linear layer on top of the
    pooled output) e.g. for GLUE tasks.
    c                        e Zd Z fdZe	 	 	 	 	 	 	 	 	 ddej                  dz  dej                  dz  dej                  dz  dej                  dz  dej                  dz  dej                  dz  d	edz  d
edz  dedz  de	e
z  fd       Z xZS ) RemBertForSequenceClassificationc                 ,   t         |   |       |j                  | _        t        |      | _        t        j                  |j                        | _        t        j                  |j                  |j                        | _        | j                          y r_   r(   r)   
num_labelsr  r  r   r5   classifier_dropout_probr7   r`   ra   
classifierr  r<   s     r@   r)   z)RemBertForSequenceClassification.__init__$  si      ++#F+zz&"@"@A))F$6$68I8IJ 	rA   NrB   r{   rC   r$   rD   r<  r~   r   r   rF   c
           
      >   |	|	n| j                   j                  }	| j                  ||||||||	      }|d   }| j                  |      }| j	                  |      }d}|| j                   j
                  | j                  dk(  rd| j                   _        nl| j                  dkD  rL|j                  t        j                  k(  s|j                  t        j                  k(  rd| j                   _        nd| j                   _        | j                   j
                  dk(  rIt               }| j                  dk(  r& ||j                         |j                               }n |||      }n| j                   j
                  dk(  r=t               } ||j                  d| j                        |j                  d            }n,| j                   j
                  dk(  rt               } |||      }|	s|f|dd z   }||f|z   S |S t!        |||j"                  |j$                  	      S )
a  
        labels (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
            Labels for computing the sequence classification/regression loss. Indices should be in `[0, ...,
            config.num_labels - 1]`. If `config.num_labels == 1` a regression loss is computed (Mean-Square loss), If
            `config.num_labels > 1` a classification loss is computed (Cross-Entropy).
        Nr{   rC   r$   rD   r~   r   r   r   
regressionsingle_label_classificationmulti_label_classificationr&   r   r>  )r>   r   r  r7   rS  problem_typerQ  rI   r9   rM   rX   r   squeezer   r   r   r   re   r   )r=   rB   r{   rC   r$   rD   r<  r~   r   r   r   r   rh   r@  r?  rB  r   s                    r@   rQ   z(RemBertForSequenceClassification.forward.  s   ( &1%<k$++BYBY,,))%'/!5#  	
  
]3/{{''/??a'/;DKK,__q(fllejj.HFLL\a\e\eLe/LDKK,/KDKK,{{''<7"9??a'#FNN$4fnn6FGD#FF3D))-JJ+-B @&++b/R))-II,./Y,F)-)9TGf$EvE'!//))	
 	
rA   	NNNNNNNNN)rR   rS   rT   r)   r   r9   rW   rV   r   r   r   rQ   rZ   r[   s   @r@   rN  rN    s      /337261526*.)-,0#'D
$$t+D
 ))D0D
 ((4/	D

 ''$.D
 ((4/D
   4'D
  $;D
 #TkD
 D[D
 
)	)D
 D
rA   rN  c                        e Zd Z fdZe	 	 	 	 	 	 	 	 	 ddej                  dz  dej                  dz  dej                  dz  dej                  dz  dej                  dz  dej                  dz  d	edz  d
edz  dedz  de	e
z  fd       Z xZS )RemBertForMultipleChoicec                     t         |   |       t        |      | _        t	        j
                  |j                        | _        t	        j                  |j                  d      | _
        | j                          y )Nr   )r(   r)   r  r  r   r5   rR  r7   r`   ra   rS  r  r<   s     r@   r)   z!RemBertForMultipleChoice.__init__x  sV     #F+zz&"@"@A))F$6$6: 	rA   NrB   r{   rC   r$   rD   r<  r~   r   r   rF   c
           
      J   |	|	n| j                   j                  }	||j                  d   n|j                  d   }|!|j                  d|j	                  d            nd}|!|j                  d|j	                  d            nd}|!|j                  d|j	                  d            nd}|!|j                  d|j	                  d            nd}|1|j                  d|j	                  d      |j	                  d            nd}| j                  ||||||||	      }|d   }| j                  |      }| j                  |      }|j                  d|      }d}|t               } |||      }|	s|f|dd z   }||f|z   S |S t        |||j                  |j                        S )a[  
        input_ids (`torch.LongTensor` of shape `(batch_size, num_choices, sequence_length)`):
            Indices of input sequence tokens in the vocabulary.

            Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
            [`PreTrainedTokenizer.__call__`] for details.

            [What are input IDs?](../glossary#input-ids)
        token_type_ids (`torch.LongTensor` of shape `(batch_size, num_choices, sequence_length)`, *optional*):
            Segment token indices to indicate first and second portions of the inputs. Indices are selected in `[0,
            1]`:

            - 0 corresponds to a *sentence A* token,
            - 1 corresponds to a *sentence B* token.

            [What are token type IDs?](../glossary#token-type-ids)
        position_ids (`torch.LongTensor` of shape `(batch_size, num_choices, sequence_length)`, *optional*):
            Indices of positions of each input sequence tokens in the position embeddings. Selected in the range `[0,
            config.max_position_embeddings - 1]`.

            [What are position IDs?](../glossary#position-ids)
        inputs_embeds (`torch.FloatTensor` of shape `(batch_size, num_choices, sequence_length, hidden_size)`, *optional*):
            Optionally, instead of passing `input_ids` you can choose to directly pass an embedded representation. This
            is useful if you want more control over how to convert *input_ids* indices into associated vectors than the
            model's internal embedding lookup matrix.
        labels (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
            Labels for computing the multiple choice classification loss. Indices should be in `[0, ...,
            num_choices-1]` where `num_choices` is the size of the second dimension of the input tensors. (See
            `input_ids` above)
        Nr   r&   r   rU  r   r>  )r>   r   r   r   rK   r  r7   rS  r   r   re   r   )r=   rB   r{   rC   r$   rD   r<  r~   r   r   r   num_choicesr   rh   r@  reshaped_logitsr?  rB  r   s                      r@   rQ   z RemBertForMultipleChoice.forward  s   X &1%<k$++BYBY,5,Aiooa(}GZGZ[\G]>G>SINN2y~~b'9:Y]	M[Mg,,R1D1DR1HImqM[Mg,,R1D1DR1HImqGSG_|((\->->r-BCei ( r=#5#5b#9=;M;Mb;QR 	 ,,))%'/!5#  	
  
]3/ ++b+6')HOV4D%''!"+5F)-)9TGf$EvE("!//))	
 	
rA   r[  )rR   rS   rT   r)   r   r9   rW   rV   r   r   r   rQ   rZ   r[   s   @r@   r]  r]  v  s      /337261526*.)-,0#'W
$$t+W
 ))D0W
 ((4/	W

 ''$.W
 ((4/W
   4'W
  $;W
 #TkW
 D[W
 
*	*W
 W
rA   r]  c                        e Zd Z fdZe	 	 	 	 	 	 	 	 	 ddej                  dz  dej                  dz  dej                  dz  dej                  dz  dej                  dz  dej                  dz  d	edz  d
edz  dedz  de	e
z  fd       Z xZS )RemBertForTokenClassificationc                 0   t         |   |       |j                  | _        t        |d      | _        t        j                  |j                        | _        t        j                  |j                  |j                        | _        | j                          y NFr/  rP  r<   s     r@   r)   z&RemBertForTokenClassification.__init__  sk      ++#FeDzz&"@"@A))F$6$68I8IJ 	rA   NrB   r{   rC   r$   rD   r<  r~   r   r   rF   c
           
         |	|	n| j                   j                  }	| j                  ||||||||	      }|d   }| j                  |      }| j	                  |      }d}|<t               } ||j                  d| j                        |j                  d            }|	s|f|dd z   }||f|z   S |S t        |||j                  |j                        S )z
        labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
            Labels for computing the token classification loss. Indices should be in `[0, ..., config.num_labels - 1]`.
        NrU  r   r&   r   r>  )r>   r   r  r7   rS  r   r   rQ  r   re   r   )r=   rB   r{   rC   r$   rD   r<  r~   r   r   r   r   r  r@  r?  rB  r   s                    r@   rQ   z%RemBertForTokenClassification.forward  s    $ &1%<k$++BYBY,,))%'/!5#  	
 "!*,,71')HFKKDOO<fkk"oNDY,F)-)9TGf$EvE$!//))	
 	
rA   r[  )rR   rS   rT   r)   r   r9   rW   rV   r   r   r   rQ   rZ   r[   s   @r@   rc  rc    s    	  /337261526*.)-,0#'1
$$t+1
 ))D01
 ((4/	1

 ''$.1
 ((4/1
   4'1
  $;1
 #Tk1
 D[1
 
&	&1
 1
rA   rc  c                   @    e Zd Z fdZe	 	 	 	 	 	 	 	 	 	 ddej                  dz  dej                  dz  dej                  dz  dej                  dz  dej                  dz  dej                  dz  d	ej                  dz  d
edz  dedz  dedz  de	e
z  fd       Z xZS )RemBertForQuestionAnsweringc                     t         |   |       |j                  | _        t        |d      | _        t        j                  |j                  |j                        | _        | j                          y re  )
r(   r)   rQ  r  r  r   r`   ra   
qa_outputsr  r<   s     r@   r)   z$RemBertForQuestionAnswering.__init__!  sU      ++#FeD))F$6$68I8IJ 	rA   NrB   r{   rC   r$   rD   start_positionsend_positionsr~   r   r   rF   c           
         |
|
n| j                   j                  }
| j                  |||||||	|
      }|d   }| j                  |      }|j	                  dd      \  }}|j                  d      }|j                  d      }d }||t        |j                               dkD  r|j                  d      }t        |j                               dkD  r|j                  d      }|j                  d      }|j                  d|       |j                  d|       t        |      } |||      } |||      }||z   dz  }|
s||f|dd  z   }||f|z   S |S t        ||||j                  |j                        S )	NrU  r   r   r&   r   )ignore_indexr   )r?  start_logits
end_logitsre   r   )r>   r   r  rj  splitrZ  lenrK   clamp_r   r   re   r   )r=   rB   r{   rC   r$   rD   rk  rl  r~   r   r   r   r   r  r@  ro  rp  
total_lossignored_indexrB  
start_lossend_lossr   s                          r@   rQ   z#RemBertForQuestionAnswering.forward,  s    &1%<k$++BYBY,,))%'/!5#  	
 "!*1#)<<r<#: j#++B/''+

&=+D?'')*Q."1"9"9""==%%'(1, - 5 5b 9(--a0M""1m4  M2']CH!,@J
M:H$x/14J"J/'!"+=F/9/EZMF*Q6Q+%!!//))
 	
rA   )
NNNNNNNNNN)rR   rS   rT   r)   r   r9   rW   rV   r   r   r   rQ   rZ   r[   s   @r@   rh  rh    s   	  /3372615263715)-,0#'=
$$t+=
 ))D0=
 ((4/	=

 ''$.=
 ((4/=
 ))D0=
 ''$.=
  $;=
 #Tk=
 D[=
 
-	-=
 =
rA   rh  )	rD  r-  r]  rh  rN  rc  r   r  r  )BrU   r   r9   r   torch.nnr   r   r    r   r  activationsr	   cache_utilsr
   r   r   
generationr   masking_utilsr   modeling_layersr   modeling_outputsr   r   r   r   r   r   r   r   modeling_utilsr   pytorch_utilsr   utilsr   r   configuration_rembertr   
get_loggerrR   r   Moduler   r]   rk   r   r   r   r   r   r   r   r   r  r  r  r-  rD  rN  r]  rc  rh  __all__r   rA   r@   <module>r     sF       A A & ! C C ) 6 9	 	 	 . 6 , 0 
		H	%1		 1jBII V.299 V.t		 ryy 8"))  BII ?- ?DM
RYY M
bRYY "bii "! ! i_ i i 	u
) u
u
p L
/ L
 L
^ 
a
/ a

a
H P
'= P
P
f c
5 c
 c
L >
$: >
 >
B J
"8 J
 J
Z
rA   