Ë
    T^(h®†  ã                   óœ  — d Z ddlmZmZmZmZmZ ddlZddlZddlm	Z	 ddl
mZmZ ddlmZ ddlmZmZmZ dd	lmZmZ dd
lmZmZ ddlmZmZmZmZmZ ddlm Z   ejB                  e"«      Z#dZ$dZ% G d„ de	jL                  «      Z' G d„ de	jL                  «      Z(	 d7de	jL                  dejR                  dejR                  dejR                  deejR                     de*de*fd„Z+ G d„ de	jL                  «      Z, G d„ de	jL                  «      Z- G d „ d!e	jL                  «      Z. G d"„ d#e	jL                  «      Z/ G d$„ d%e	jL                  «      Z0 G d&„ d'e	jL                  «      Z1 G d(„ d)e	jL                  «      Z2 G d*„ d+e	jL                  «      Z3 G d,„ d-e«      Z4d.Z5d/Z6 ed0e5«       G d1„ d2e4«      «       Z7 ed3e5«       G d4„ d5e4«      «       Z8g d6¢Z9y)8zPyTorch ViViT model.é    )ÚCallableÚOptionalÚSetÚTupleÚUnionN)Únn)ÚCrossEntropyLossÚMSELossé   )ÚACT2FN)ÚBaseModelOutputÚBaseModelOutputWithPoolingÚImageClassifierOutput)ÚALL_ATTENTION_FUNCTIONSÚPreTrainedModel)Ú find_pruneable_heads_and_indicesÚprune_linear_layer)Úadd_start_docstringsÚ%add_start_docstrings_to_model_forwardÚloggingÚreplace_return_docstringsÚ	torch_inté   )ÚVivitConfigzgoogle/vivit-b-16x2-kinetics400r   c                   ó0   ‡ — e Zd ZdZˆ fd„Zddefd„Zˆ xZS )ÚVivitTubeletEmbeddingsa’  
    Construct Vivit Tubelet embeddings.

    This module turns a batch of videos of shape (batch_size, num_frames, num_channels, height, width) into a tensor of
    shape (batch_size, seq_len, hidden_size) to be consumed by a Transformer encoder.

    The seq_len (the number of patches) equals (number of frames // tubelet_size[0]) * (height // tubelet_size[1]) *
    (width // tubelet_size[2]).
    c                 óì  •— t         ‰| �  «        |j                  | _        |j                  | _        |j                  | _        | j                  | j
                  d   z  | j                  | j
                  d   z  z  | j                  | j
                  d   z  z  | _        |j                  | _        t        j                  |j                  |j                  |j                  |j                  ¬«      | _        y )Né   r   r   )Úkernel_sizeÚstride)ÚsuperÚ__init__Ú
num_framesÚ
image_sizeÚtubelet_sizeÚ
patch_sizeÚnum_patchesÚhidden_sizeÚ	embed_dimr   ÚConv3dÚnum_channelsÚ
projection©ÚselfÚconfigÚ	__class__s     €úf/var/www/skyplay_api_hub/venv/lib/python3.12/site-packages/transformers/models/vivit/modeling_vivit.pyr"   zVivitTubeletEmbeddings.__init__7   sÇ   ø€ Ü‰ÑÔØ ×+Ñ+ˆŒØ ×+Ñ+ˆŒØ ×-Ñ-ˆŒà�_‰_ §¡°Ñ 2Ñ2Ø�‰ $§/¡/°!Ñ"4Ñ4ñ6à�‰ $§/¡/°!Ñ"4Ñ4ñ6ð 	Ôð
  ×+Ñ+ˆŒäŸ)™)Ø×Ñ ×!3Ñ!3À×ATÑATÐ]c×]pÑ]pô
ˆ�ó    Úinterpolate_pos_encodingc                 ó\  — |j                   \  }}}}}|sP|| j                  k7  s|| j                  k7  r2t        d|› d|› d| j                  d   › d| j                  d   › d�	«      ‚|j                  ddddd	«      }| j	                  |«      }|j                  d«      j                  dd«      }|S )
NzImage image size (Ú*z) doesn't match model (r   r   z).r   r   é   )Úshaper$   Ú
ValueErrorÚpermuter,   ÚflattenÚ	transpose)	r.   Úpixel_valuesr3   Ú
batch_sizer#   r+   ÚheightÚwidthÚxs	            r1   ÚforwardzVivitTubeletEmbeddings.forwardG   sÄ   € Ø>J×>PÑ>PÑ;ˆ
�J ¨f°eÙ'¨V°t·±Ò-FÈ%ÐSW×SbÑSbÒJbÜØ$ V H¨A¨e¨WÐ4KÈDÏOÉOÐ\]ÑL^ÐK_Ð_`Ðae×apÑapÐqrÑasÐ`tÐtvÐwóð ð
 $×+Ñ+¨A¨q°!°Q¸Ó:ˆà�O‰O˜LÓ)ˆð �I‰I�a‹L×"Ñ" 1 aÓ(ˆØˆr2   ©F)Ú__name__Ú
__module__Ú__qualname__Ú__doc__r"   ÚboolrA   Ú__classcell__©r0   s   @r1   r   r   ,   s   ø„ ñô
ñ ¸d÷ r2   r   c                   óp   ‡ — e Zd ZdZˆ fd„Zdej                  dededej                  fd„Zd
de	fd	„Z
ˆ xZS )ÚVivitEmbeddingszˆ
    Vivit Embeddings.

    Creates embeddings from a video using VivitTubeletEmbeddings, adds CLS token and positional embeddings.
    c                 óÒ  •— t         ‰| �  «        t        j                  t	        j
                  dd|j                  «      «      | _        t        |«      | _	        t        j                  t	        j
                  d| j                  j                  dz   |j                  «      «      | _        t        j                  |j                  «      | _        |j                  dd  | _        || _        y )Nr   )r!   r"   r   Ú	ParameterÚtorchÚzerosr(   Ú	cls_tokenr   Úpatch_embeddingsr'   Úposition_embeddingsÚDropoutÚhidden_dropout_probÚdropoutr%   r&   r/   r-   s     €r1   r"   zVivitEmbeddings.__init___   s©   ø€ Ü‰ÑÔäŸ™¤e§k¡k°!°Q¸×8JÑ8JÓ&KÓLˆŒÜ 6°vÓ >ˆÔä#%§<¡<Ü�K‰K˜˜4×0Ñ0×<Ñ<¸qÑ@À&×BTÑBTÓUó$
ˆÔ ô —z‘z &×"<Ñ"<Ó=ˆŒØ ×-Ñ-¨a¨bÐ1ˆŒØˆ�r2   Ú
embeddingsr>   r?   Úreturnc                 ó²  — |j                   d   dz
  }| j                  j                   d   dz
  }t        j                  j	                  «       s||k(  r||k(  r| j                  S | j                  dd…dd…f   }| j                  dd…dd…f   }|j                   d   }|| j
                  d   z  }	|| j
                  d   z  }
t        |dz  «      }|j                  d|||«      }|j                  dddd«      }t        j                  j                  ||	|
fdd	¬
«      }|j                  dddd«      j                  dd|«      }t        j                  ||fd¬«      S )a   
        This method allows to interpolate the pre-trained position encodings, to be able to use the model on higher resolution
        images. This method is also adapted to support torch.jit tracing.

        Adapted from:
        - https://github.com/facebookresearch/dino/blob/de9ee3df6cf39fac952ab558447af1fa1365362a/vision_transformer.py#L174-L194, and
        - https://github.com/facebookresearch/dinov2/blob/e1277af2ba9496fbadf7aec6eba56e8d882d1e35/dinov2/models/vision_transformer.py#L179-L211
        r   Néÿÿÿÿr   g      à?r   r   ÚbicubicF)ÚsizeÚmodeÚalign_corners©Údim)r7   rR   rN   ÚjitÚ
is_tracingr&   r   Úreshaper9   r   Ú
functionalÚinterpolateÚviewÚcat)r.   rV   r>   r?   r'   Únum_positionsÚclass_pos_embedÚpatch_pos_embedr_   Ú
new_heightÚ	new_widthÚsqrt_num_positionss               r1   r3   z(VivitEmbeddings.interpolate_pos_encodingm   sj  € ð !×&Ñ& qÑ)¨AÑ-ˆØ×0Ñ0×6Ñ6°qÑ9¸AÑ=ˆô �y‰y×#Ñ#Ô%¨+¸Ò*FÈ6ÐUZÊ?Ø×+Ñ+Ð+à×2Ñ2²1°b°q°b°5Ñ9ˆØ×2Ñ2²1°a±b°5Ñ9ˆà×Ñ˜rÑ"ˆà˜tŸ™¨qÑ1Ñ1ˆ
Ø˜TŸ_™_¨QÑ/Ñ/ˆ	ä& }°cÑ'9Ó:ÐØ)×1Ñ1°!Ð5GÐI[Ð]`ÓaˆØ)×1Ñ1°!°Q¸¸1Ó=ˆäŸ-™-×3Ñ3ØØ˜iÐ(ØØð	 4ó 
ˆð *×1Ñ1°!°Q¸¸1Ó=×BÑBÀ1ÀbÈ#ÓNˆä�y‰y˜/¨?Ð;ÀÔCÐCr2   r3   c                 ó0  — |j                   \  }}}}}| j                  ||¬«      }| j                  j                  |ddg«      }	t	        j
                  |	|fd¬«      }|r|| j                  |||«      z   }n|| j                  z   }| j                  |«      }|S )N©r3   r   r^   )	r7   rQ   rP   ÚtilerN   rf   r3   rR   rU   )
r.   r<   r3   r=   r#   r+   r>   r?   rV   Ú
cls_tokenss
             r1   rA   zVivitEmbeddings.forward•   s¢   € Ø>J×>PÑ>PÑ;ˆ
�J ¨f°eØ×*Ñ*¨<ÐRjÐ*Ókˆ
à—^‘^×(Ñ(¨*°a¸Ð);Ó<ˆ
Ü—Y‘Y 
¨JÐ7¸QÔ?ˆ
ñ $Ø# d×&CÑ&CÀJÐPVÐX]Ó&^Ñ^‰Jà# d×&>Ñ&>Ñ>ˆJà—\‘\ *Ó-ˆ
àÐr2   rB   )rC   rD   rE   rF   r"   rN   ÚTensorÚintr3   rG   rA   rH   rI   s   @r1   rK   rK   X   sL   ø„ ñôð&D°5·<±<ð &DÈð &DÐUXð &DÐ]b×]iÑ]ió &DñP¸d÷ r2   rK   ÚmoduleÚqueryÚkeyÚvalueÚattention_maskÚscalingrU   c                 óÀ  — t        j                  ||j                  dd«      «      |z  }t        j                  j                  |dt         j                  ¬«      j                  |j                  «      }t        j                  j                  ||| j                  ¬«      }|�||z  }t        j                  ||«      }	|	j                  dd«      j                  «       }	|	|fS )NrY   éþÿÿÿ)r_   Údtype)ÚpÚtrainingr   r   )rN   Úmatmulr;   r   rc   ÚsoftmaxÚfloat32Útor{   rU   r}   Ú
contiguous)
rs   rt   ru   rv   rw   rx   rU   ÚkwargsÚattn_weightsÚattn_outputs
             r1   Úeager_attention_forwardr†   ¨   sÀ   € ô —<‘<  s§}¡}°R¸Ó'<Ó=ÀÑG€Lô —=‘=×(Ñ(¨¸2ÄUÇ]Á]Ð(ÓS×VÑVÐW\×WbÑWbÓc€Lô —=‘=×(Ñ(¨¸È6Ï?É?Ð(Ó[€Lð Ð!Ø# nÑ4ˆä—,‘,˜|¨UÓ3€KØ×'Ñ'¨¨1Ó-×8Ñ8Ó:€Kà˜Ð$Ð$r2   c            
       óè   ‡ — e Zd Zdeddfˆ fd„Zdej                  dej                  fd„Z	 d
deej                     de	de
eej                  ej                  f   eej                     f   fd	„Zˆ xZS )ÚVivitSelfAttentionr/   rW   Nc                 ó2  •— t         ‰| �  «        |j                  |j                  z  dk7  r2t	        |d«      s&t        d|j                  › d|j                  › d�«      ‚|| _        |j                  | _        t        |j                  |j                  z  «      | _        | j                  | j                  z  | _	        |j                  | _        | j                  dz  | _        d| _        t        j                  |j                  | j                  |j                   ¬«      | _        t        j                  |j                  | j                  |j                   ¬«      | _        t        j                  |j                  | j                  |j                   ¬«      | _        y )	Nr   Úembedding_sizezThe hidden size z4 is not a multiple of the number of attention heads ú.g      à¿F)Úbias)r!   r"   r(   Únum_attention_headsÚhasattrr8   r/   rr   Úattention_head_sizeÚall_head_sizeÚattention_probs_dropout_probÚdropout_probrx   Ú	is_causalr   ÚLinearÚqkv_biasrt   ru   rv   r-   s     €r1   r"   zVivitSelfAttention.__init__È   sF  ø€ Ü‰ÑÔØ×Ñ × :Ñ :Ñ:¸aÒ?ÌÐPVÐXhÔHiÜØ" 6×#5Ñ#5Ð"6ð 7Ø×3Ñ3Ð4°Að7óð ð
 ˆŒØ#)×#=Ñ#=ˆÔ Ü#& v×'9Ñ'9¸F×<VÑ<VÑ'VÓ#WˆÔ Ø!×5Ñ5¸×8PÑ8PÑPˆÔØ"×?Ñ?ˆÔØ×/Ñ/°Ñ5ˆŒØˆŒä—Y‘Y˜v×1Ñ1°4×3EÑ3EÈFÏOÉOÔ\ˆŒ
Ü—9‘9˜V×/Ñ/°×1CÑ1CÈ&Ï/É/ÔZˆŒÜ—Y‘Y˜v×1Ñ1°4×3EÑ3EÈFÏOÉOÔ\ˆ�
r2   r@   c                 ó¤   — |j                  «       d d | j                  | j                  fz   }|j                  |«      }|j	                  dddd«      S )NrY   r   r   r   r   )r[   r�   r�   re   r9   )r.   r@   Únew_x_shapes      r1   Útranspose_for_scoresz'VivitSelfAttention.transpose_for_scoresÜ   sL   € Ø—f‘f“h˜s �m t×'?Ñ'?À×AYÑAYÐ&ZÑZˆØ�F‰F�;ÓˆØ�y‰y˜˜A˜q !Ó$Ð$r2   Ú	head_maskÚoutput_attentionsc           
      ó˜  — | j                  | j                  |«      «      }| j                  | j                  |«      «      }| j                  | j                  |«      «      }t        }| j
                  j                  dk7  rN| j
                  j                  dk(  r|rt        j                  d«       nt        | j
                  j                     } || ||||| j                  | j                  | j                  sdn| j                  ¬«      \  }}	|j                  «       d d | j                  fz   }
|j!                  |
«      }|r||	f}|S |f}|S )NÚeagerÚsdpazã`torch.nn.functional.scaled_dot_product_attention` does not support `output_attentions=True`. Falling back to eager attention. This warning can be removed using the argument `attn_implementation="eager"` when loading the model.ç        )r“   rx   rU   rz   )r˜   ru   rv   rt   r†   r/   Ú_attn_implementationÚloggerÚwarning_oncer   r“   rx   r}   r’   r[   r�   rb   )r.   Úhidden_statesr™   rš   Ú	key_layerÚvalue_layerÚquery_layerÚattention_interfaceÚcontext_layerÚattention_probsÚnew_context_layer_shapeÚoutputss               r1   rA   zVivitSelfAttention.forwardá   s=  € ð ×-Ñ-¨d¯h©h°}Ó.EÓFˆ	Ø×/Ñ/°·
±
¸=Ó0IÓJˆØ×/Ñ/°·
±
¸=Ó0IÓJˆä(?ÐØ�;‰;×+Ñ+¨wÒ6Ø�{‰{×/Ñ/°6Ò9Ñ>OÜ×#Ñ#ðLõô
 '>¸d¿k¹k×>^Ñ>^Ñ&_Ð#á)<ØØØØØØ—n‘nØ—L‘LØ#Ÿ}š}‘C°$×2CÑ2Cô	*
Ñ&ˆ�ð #0×"4Ñ"4Ó"6°s¸Ð";¸t×?QÑ?QÐ>SÑ"SÐØ%×-Ñ-Ð.EÓFˆá6G�= /Ð2ˆàˆð O\ÐM]ˆàˆr2   ©NF)rC   rD   rE   r   r"   rN   rq   r˜   r   rG   r   r   rA   rH   rI   s   @r1   rˆ   rˆ   Ç   s†   ø„ ð]˜{ð ]¨tõ ]ð(% e§l¡lð %°u·|±|ó %ð bgñ!Ø(0°·±Ñ(>ð!ØZ^ð!à	ˆu�U—\‘\ 5§<¡<Ð/Ñ0°%¸¿¹Ñ2EÐEÑ	F÷!r2   rˆ   c                   ó|   ‡ — e Zd ZdZdeddfˆ fd„Zdej                  dej                  dej                  fd„Zˆ xZ	S )	ÚVivitSelfOutputz¢
    The residual connection is defined in VivitLayer instead of here (as is the case with other models), due to the
    layernorm applied before each block.
    r/   rW   Nc                 óÈ   •— t         ‰| �  «        t        j                  |j                  |j                  «      | _        t        j                  |j                  «      | _        y ©N)	r!   r"   r   r”   r(   ÚdenserS   rT   rU   r-   s     €r1   r"   zVivitSelfOutput.__init__  sB   ø€ Ü‰ÑÔÜ—Y‘Y˜v×1Ñ1°6×3EÑ3EÓFˆŒ
Ü—z‘z &×"<Ñ"<Ó=ˆ�r2   r¢   Úinput_tensorc                 óJ   — | j                  |«      }| j                  |«      }|S r¯   ©r°   rU   ©r.   r¢   r±   s      r1   rA   zVivitSelfOutput.forward  s$   € ØŸ
™
 =Ó1ˆØŸ™ ]Ó3ˆàÐr2   )
rC   rD   rE   rF   r   r"   rN   rq   rA   rH   rI   s   @r1   r­   r­     sD   ø„ ñð
>˜{ð >¨tõ >ð
 U§\¡\ð ÀÇÁð ÐRW×R^ÑR^÷ r2   r­   c                   óà   ‡ — e Zd Zdeddfˆ fd„Zdee   ddfd„Z	 	 ddej                  de
ej                     d	edeeej                  ej                  f   eej                     f   fd
„Zˆ xZS )ÚVivitAttentionr/   rW   Nc                 ó€   •— t         ‰| �  «        t        |«      | _        t	        |«      | _        t        «       | _        y r¯   )r!   r"   rˆ   Ú	attentionr­   ÚoutputÚsetÚpruned_headsr-   s     €r1   r"   zVivitAttention.__init__  s0   ø€ Ü‰ÑÔÜ+¨FÓ3ˆŒÜ% fÓ-ˆŒÜ›EˆÕr2   Úheadsc                 ó>  — t        |«      dk(  ry t        || j                  j                  | j                  j                  | j
                  «      \  }}t        | j                  j                  |«      | j                  _        t        | j                  j                  |«      | j                  _        t        | j                  j                  |«      | j                  _	        t        | j                  j                  |d¬«      | j                  _        | j                  j                  t        |«      z
  | j                  _        | j                  j                  | j                  j                  z  | j                  _        | j
                  j                  |«      | _        y )Nr   r   r^   )Úlenr   r¸   r�   r�   r»   r   rt   ru   rv   r¹   r°   r�   Úunion)r.   r¼   Úindexs      r1   Úprune_headszVivitAttention.prune_heads   s  € Üˆu‹:˜Š?ØÜ7Ø�4—>‘>×5Ñ5°t·~±~×7YÑ7YÐ[_×[lÑ[ló
‰ˆˆuô
  2°$·.±.×2FÑ2FÈÓNˆ�‰ÔÜ/°·±×0BÑ0BÀEÓJˆ�‰ÔÜ1°$·.±.×2FÑ2FÈÓNˆ�‰ÔÜ.¨t¯{©{×/@Ñ/@À%ÈQÔOˆ�‰Ôð .2¯^©^×-OÑ-OÔRUÐV[ÓR\Ñ-\ˆ�‰Ô*Ø'+§~¡~×'IÑ'IÈDÏNÉN×LnÑLnÑ'nˆ�‰Ô$Ø ×-Ñ-×3Ñ3°EÓ:ˆÕr2   r¢   r™   rš   c                 óh   — | j                  |||«      }| j                  |d   |«      }|f|dd  z   }|S )Nr   r   )r¸   r¹   )r.   r¢   r™   rš   Úself_outputsÚattention_outputrª   s          r1   rA   zVivitAttention.forward2  sE   € ð —~‘~ m°YÐ@QÓRˆàŸ;™; |°A¡¸ÓFÐà#Ð%¨°Q°RÐ(8Ñ8ˆØˆr2   r«   )rC   rD   rE   r   r"   r   rr   rÁ   rN   rq   r   rG   r   r   rA   rH   rI   s   @r1   r¶   r¶     s’   ø„ ð"˜{ð "¨tõ "ð;  S¡ð ;¨dó ;ð* -1Ø"'ñ	à—|‘|ðð ˜EŸL™LÑ)ðð  ð	ð
 
ˆu�U—\‘\ 5§<¡<Ð/Ñ0°%¸¿¹Ñ2EÐEÑ	F÷r2   r¶   c                   ó$   ‡ — e Zd Zˆ fd„Zd„ Zˆ xZS )ÚVivitIntermediatec                 óP  •— t         ‰| �  «        t        j                  |j                  |j
                  «      | _        t        j                  |j                  «      | _	        t        |j                  t        «      rt        |j                     | _        y |j                  | _        y r¯   )r!   r"   r   r”   r(   Úintermediate_sizer°   rS   rT   rU   Ú
isinstanceÚ
hidden_actÚstrr   Úintermediate_act_fnr-   s     €r1   r"   zVivitIntermediate.__init__A  ss   ø€ Ü‰ÑÔÜ—Y‘Y˜v×1Ñ1°6×3KÑ3KÓLˆŒ
Ü—z‘z &×"<Ñ"<Ó=ˆŒÜ�f×'Ñ'¬Ô-Ü'-¨f×.?Ñ.?Ñ'@ˆDÕ$à'-×'8Ñ'8ˆDÕ$r2   c                 ól   — | j                  |«      }| j                  |«      }| j                  |«      }|S r¯   )r°   rÌ   rU   )r.   r¢   s     r1   rA   zVivitIntermediate.forwardJ  s4   € ØŸ
™
 =Ó1ˆØ×0Ñ0°Ó?ˆØŸ™ ]Ó3ˆàÐr2   ©rC   rD   rE   r"   rA   rH   rI   s   @r1   rÆ   rÆ   @  s   ø„ ô9ör2   rÆ   c                   ó$   ‡ — e Zd Zˆ fd„Zd„ Zˆ xZS )ÚVivitOutputc                 óÈ   •— t         ‰| �  «        t        j                  |j                  |j
                  «      | _        t        j                  |j                  «      | _	        y r¯   )
r!   r"   r   r”   rÈ   r(   r°   rS   rT   rU   r-   s     €r1   r"   zVivitOutput.__init__S  sB   ø€ Ü‰ÑÔÜ—Y‘Y˜v×7Ñ7¸×9KÑ9KÓLˆŒ
Ü—z‘z &×"<Ñ"<Ó=ˆ�r2   c                 óT   — | j                  |«      }| j                  |«      }||z   }|S r¯   r³   r´   s      r1   rA   zVivitOutput.forwardX  s.   € ØŸ
™
 =Ó1ˆàŸ™ ]Ó3ˆà%¨Ñ4ˆàÐr2   rÎ   rI   s   @r1   rÐ   rÐ   R  s   ø„ ô>ö
r2   rÐ   c                   ó*   ‡ — e Zd ZdZˆ fd„Zdd„Zˆ xZS )Ú
VivitLayerzNThis corresponds to the EncoderBlock class in the scenic/vivit implementation.c                 ór  •— t         ‰| �  «        |j                  | _        d| _        t	        |«      | _        t        |«      | _        t        |«      | _	        t        j                  |j                  |j                  ¬«      | _        t        j                  |j                  |j                  ¬«      | _        y )Nr   ©Úeps)r!   r"   Úchunk_size_feed_forwardÚseq_len_dimr¶   r¸   rÆ   ÚintermediaterÐ   r¹   r   Ú	LayerNormr(   Úlayer_norm_epsÚlayernorm_beforeÚlayernorm_afterr-   s     €r1   r"   zVivitLayer.__init__e  s‡   ø€ Ü‰ÑÔØ'-×'EÑ'EˆÔ$ØˆÔÜ'¨Ó/ˆŒÜ-¨fÓ5ˆÔÜ! &Ó)ˆŒÜ "§¡¨V×-?Ñ-?ÀV×EZÑEZÔ [ˆÔÜ!Ÿ|™|¨F×,>Ñ,>ÀF×DYÑDYÔZˆÕr2   c                 óÞ   — | j                  | j                  |«      ||¬«      }|d   }|dd  }||z   }| j                  |«      }| j                  |«      }| j	                  ||«      }|f|z   }|S )N)rš   r   r   )r¸   rÝ   rÞ   rÚ   r¹   )r.   r¢   r™   rš   Úself_attention_outputsrÄ   rª   Úlayer_outputs           r1   rA   zVivitLayer.forwardo  s”   € Ø!%§¡à×!Ñ! -Ó0ØØ/ð	 "0ó "
Ðð 2°!Ñ4Ðà(¨¨Ð,ˆð )¨=Ñ8ˆð ×+Ñ+¨MÓ:ˆØ×(Ñ(¨Ó6ˆð —{‘{ <°Ó?ˆà�/ GÑ+ˆàˆr2   r«   )rC   rD   rE   rF   r"   rA   rH   rI   s   @r1   rÔ   rÔ   b  s   ø„ ÙXô[÷r2   rÔ   c                   ó.   ‡ — e Zd Zˆ fd„Z	 	 	 	 dd„Zˆ xZS )ÚVivitEncoderc                 óÐ   •— t         ‰| �  «        || _        t        j                  t        |j                  «      D �cg c]  }t        |«      ‘Œ c}«      | _        d| _	        y c c}w r«   )
r!   r"   r/   r   Ú
ModuleListÚrangeÚnum_hidden_layersrÔ   ÚlayerÚgradient_checkpointing)r.   r/   Ú_r0   s      €r1   r"   zVivitEncoder.__init__Š  sN   ø€ Ü‰ÑÔØˆŒÜ—]‘]ÄÀf×F^ÑF^Ó@_Ö#`¸1¤J¨vÕ$6Ò#`ÓaˆŒ
Ø&+ˆÕ#ùò $as   ½A#c                 ót  — |rdnd }|rdnd }t        | j                  «      D ]h  \  }}	|r||fz   }|�||   nd }
| j                  r+| j                  r| j	                  |	j
                  ||
|«      }n
 |	||
|«      }|d   }|sŒ`||d   fz   }Œj |r||fz   }|st        d„ |||fD «       «      S t        |||¬«      S )N© r   r   c              3   ó&   K  — | ]	  }|€Œ|–— Œ y ­wr¯   rì   )Ú.0Úvs     r1   ú	<genexpr>z'VivitEncoder.forward.<locals>.<genexpr>´  s   è ø€ Òm˜qÐ_`Ñ_lœÑmùs   ‚Š)Úlast_hidden_stater¢   Ú
attentions)Ú	enumeraterè   ré   r}   Ú_gradient_checkpointing_funcÚ__call__Útupler   )r.   r¢   r™   rš   Úoutput_hidden_statesÚreturn_dictÚall_hidden_statesÚall_self_attentionsÚiÚlayer_moduleÚlayer_head_maskÚlayer_outputss               r1   rA   zVivitEncoder.forward�  sÿ   € ñ #7™B¸DÐÙ$5™b¸4Ðä(¨¯©Ó4ò 	P‰OˆAˆ|Ù#Ø$5¸Ð8HÑ$HÐ!à.7Ð.C˜i¨šlÈˆOà×*Ò*¨t¯}ª}Ø $× AÑ AØ ×)Ñ)Ø!Ø#Ø%ó	!‘ñ !-¨]¸OÐM^Ó _�à)¨!Ñ,ˆMâ Ø&9¸]È1Ñ=MÐ<OÑ&OÑ#ð'	Pñ*  Ø 1°]Ð4DÑ DÐáÜÑm ]Ð4EÐGZÐ$[ÔmÓmÐmÜØ+Ø+Ø*ô
ð 	
r2   )NFFTrÎ   rI   s   @r1   rã   rã   ‰  s   ø„ ô,ð ØØ"Ø÷)
r2   rã   c                   ó$   ‡ — e Zd Zˆ fd„Zd„ Zˆ xZS )ÚVivitPoolerc                 ó²   •— t         ‰| �  «        t        j                  |j                  |j                  «      | _        t        j                  «       | _        y r¯   )r!   r"   r   r”   r(   r°   ÚTanhÚ
activationr-   s     €r1   r"   zVivitPooler.__init__½  s9   ø€ Ü‰ÑÔÜ—Y‘Y˜v×1Ñ1°6×3EÑ3EÓFˆŒ
ÜŸ'™'›)ˆ�r2   c                 ó\   — |d d …df   }| j                  |«      }| j                  |«      }|S )Nr   )r°   r  )r.   r¢   Úfirst_token_tensorÚpooled_outputs       r1   rA   zVivitPooler.forwardÂ  s6   € ð +ª1¨a¨4Ñ0ÐØŸ
™
Ð#5Ó6ˆØŸ™¨Ó6ˆØÐr2   rÎ   rI   s   @r1   r   r   ¼  s   ø„ ô$ö
r2   r   c                   ó2   — e Zd ZdZeZdZdZdZg Z	dZ
dZd„ Zy)ÚVivitPreTrainedModelz†
    An abstract class to handle weights initialization and a simple interface for downloading and loading pretrained
    models.
    Úvivitr<   Tc                 óÔ  — t        |t        j                  t        j                  f«      rm|j                  j
                  j                  d| j                  j                  ¬«       |j                  �%|j                  j
                  j                  «        yyt        |t        j                  «      rz|j                  j
                  j                  d| j                  j                  ¬«       |j                  �2|j                  j
                  |j                     j                  «        yyt        |t        j                  «      rJ|j                  j
                  j                  «        |j                  j
                  j                  d«       yt        |t        «      rI|j                   j
                  j                  «        |j"                  j
                  j                  «        yy)zInitialize the weightsrž   )ÚmeanÚstdNg      ð?)rÉ   r   r”   r*   ÚweightÚdataÚnormal_r/   Úinitializer_rangerŒ   Úzero_Ú	EmbeddingÚpadding_idxrÛ   Úfill_rK   rP   rR   )r.   rs   s     r1   Ú_init_weightsz"VivitPreTrainedModel._init_weightsÙ  sI  € ä�fœrŸy™y¬"¯)©)Ð4Ô5ð �M‰M×Ñ×&Ñ&¨C°T·[±[×5RÑ5RÐ&ÔSØ�{‰{Ð&Ø—‘× Ñ ×&Ñ&Õ(ð 'ä˜¤§¡Ô-Ø�M‰M×Ñ×&Ñ&¨C°T·[±[×5RÑ5RÐ&ÔSØ×!Ñ!Ð-Ø—‘×"Ñ" 6×#5Ñ#5Ñ6×<Ñ<Õ>ð .ä˜¤§¡Ô-Ø�K‰K×Ñ×"Ñ"Ô$Ø�M‰M×Ñ×$Ñ$ SÕ)Ü˜¤Ô0Ø×Ñ×!Ñ!×'Ñ'Ô)Ø×&Ñ&×+Ñ+×1Ñ1Õ3ð 1r2   N)rC   rD   rE   rF   r   Úconfig_classÚbase_model_prefixÚmain_input_nameÚsupports_gradient_checkpointingÚ_no_split_modulesÚ_supports_sdpaÚ_supports_flash_attn_2r  rì   r2   r1   r  r  Ë  s5   „ ñð
 €LØÐØ$€OØ&*Ð#ØÐØ€NØ!Ðó4r2   r  aG  
    This model is a PyTorch [torch.nn.Module](https://pytorch.org/docs/stable/nn.html#torch.nn.Module) subclass. Use it
    as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to general usage and
    behavior.

    Parameters:
        config ([`VivitConfig`]): Model configuration class with all the parameters of the model.
            Initializing with a config file does not load the weights associated with the model, only the
            configuration. Check out the [`~PreTrainedModel.from_pretrained`] method to load the model weights.
aã  
    Args:
        pixel_values (`torch.FloatTensor` of shape `(batch_size, num_frames, num_channels, height, width)`):
            Pixel values. Pixel values can be obtained using [`VivitImageProcessor`]. See
            [`VivitImageProcessor.preprocess`] for details.

        head_mask (`torch.FloatTensor` of shape `(num_heads,)` or `(num_layers, num_heads)`, *optional*):
            Mask to nullify selected heads of the self-attention modules. Mask values selected in `[0, 1]`:

            - 1 indicates the head is **not masked**,
            - 0 indicates the head is **masked**.

        output_attentions (`bool`, *optional*):
            Whether or not to return the attentions tensors of all attention layers. See `attentions` under returned
            tensors for more detail.
        output_hidden_states (`bool`, *optional*):
            Whether or not to return the hidden states of all layers. See `hidden_states` under returned tensors for
            more detail.
        interpolate_pos_encoding (`bool`, *optional*, `False`):
            Whether to interpolate the pre-trained position encodings.
        return_dict (`bool`, *optional*):
            Whether or not to return a [`~utils.ModelOutput`] instead of a plain tuple.
z_The bare ViViT Transformer model outputting raw hidden-states without any specific head on top.c                   óø   ‡ — e Zd Zdˆ fd„	Zd„ Zd„ Z ee«       ee	e
¬«      	 	 	 	 	 	 ddeej                     deej                     dee   dee   d	ed
ee   deeej                     e	f   fd„«       «       Zˆ xZS )Ú
VivitModelc                 ó  •— t         ‰| �  |«       || _        t        |«      | _        t        |«      | _        t        j                  |j                  |j                  ¬«      | _        |rt        |«      nd | _        | j                  «        y )NrÖ   )r!   r"   r/   rK   rV   rã   Úencoderr   rÛ   r(   rÜ   Ú	layernormr   ÚpoolerÚ	post_init)r.   r/   Úadd_pooling_layerr0   s      €r1   r"   zVivitModel.__init__  si   ø€ Ü‰Ñ˜Ô ØˆŒä)¨&Ó1ˆŒÜ# FÓ+ˆŒäŸ™ f×&8Ñ&8¸f×>SÑ>SÔTˆŒÙ->”k &Ô)ÀDˆŒð 	�‰Õr2   c                 ó.   — | j                   j                  S r¯   )rV   rQ   )r.   s    r1   Úget_input_embeddingszVivitModel.get_input_embeddings#  s   € Ø�‰×/Ñ/Ð/r2   c                 ó˜   — |j                  «       D ]7  \  }}| j                  j                  |   j                  j	                  |«       Œ9 y)z¡
        Prunes heads of the model.

        Args:
            heads_to_prune:
                dict of {layer_num: list of heads to prune in this layer}
        N)Úitemsr   rè   r¸   rÁ   )r.   Úheads_to_prunerè   r¼   s       r1   Ú_prune_headszVivitModel._prune_heads&  sE   € ð +×0Ñ0Ó2ò 	C‰LˆE�5Ø�L‰L×Ñ˜uÑ%×/Ñ/×;Ñ;¸EÕBñ	Cr2   ©Úoutput_typer  r<   r™   rš   r÷   r3   rø   rW   c                 ó  — |�|n| j                   j                  }|�|n| j                   j                  }|�|n| j                   j                  }|€t	        d«      ‚| j                  || j                   j                  «      }| j                  ||¬«      }| j                  |||||¬«      }|d   }	| j                  |	«      }	| j                  �| j                  |	«      nd}
|s
|	|
f|dd z   S t        |	|
|j                  |j                  ¬«      S )a  
        Returns:

        Examples:

        ```python
        >>> import av
        >>> import numpy as np

        >>> from transformers import VivitImageProcessor, VivitModel
        >>> from huggingface_hub import hf_hub_download

        >>> np.random.seed(0)


        >>> def read_video_pyav(container, indices):
        ...     '''
        ...     Decode the video with PyAV decoder.
        ...     Args:
        ...         container (`av.container.input.InputContainer`): PyAV container.
        ...         indices (`List[int]`): List of frame indices to decode.
        ...     Returns:
        ...         result (np.ndarray): np array of decoded frames of shape (num_frames, height, width, 3).
        ...     '''
        ...     frames = []
        ...     container.seek(0)
        ...     start_index = indices[0]
        ...     end_index = indices[-1]
        ...     for i, frame in enumerate(container.decode(video=0)):
        ...         if i > end_index:
        ...             break
        ...         if i >= start_index and i in indices:
        ...             frames.append(frame)
        ...     return np.stack([x.to_ndarray(format="rgb24") for x in frames])


        >>> def sample_frame_indices(clip_len, frame_sample_rate, seg_len):
        ...     '''
        ...     Sample a given number of frame indices from the video.
        ...     Args:
        ...         clip_len (`int`): Total number of frames to sample.
        ...         frame_sample_rate (`int`): Sample every n-th frame.
        ...         seg_len (`int`): Maximum allowed index of sample's last frame.
        ...     Returns:
        ...         indices (`List[int]`): List of sampled frame indices
        ...     '''
        ...     converted_len = int(clip_len * frame_sample_rate)
        ...     end_idx = np.random.randint(converted_len, seg_len)
        ...     start_idx = end_idx - converted_len
        ...     indices = np.linspace(start_idx, end_idx, num=clip_len)
        ...     indices = np.clip(indices, start_idx, end_idx - 1).astype(np.int64)
        ...     return indices


        >>> # video clip consists of 300 frames (10 seconds at 30 FPS)
        >>> file_path = hf_hub_download(
        ...     repo_id="nielsr/video-demo", filename="eating_spaghetti.mp4", repo_type="dataset"
        ... )
        >>> container = av.open(file_path)

        >>> # sample 32 frames
        >>> indices = sample_frame_indices(clip_len=32, frame_sample_rate=1, seg_len=container.streams.video[0].frames)
        >>> video = read_video_pyav(container=container, indices=indices)

        >>> image_processor = VivitImageProcessor.from_pretrained("google/vivit-b-16x2-kinetics400")
        >>> model = VivitModel.from_pretrained("google/vivit-b-16x2-kinetics400")

        >>> # prepare video for the model
        >>> inputs = image_processor(list(video), return_tensors="pt")

        >>> # forward pass
        >>> outputs = model(**inputs)
        >>> last_hidden_states = outputs.last_hidden_state
        >>> list(last_hidden_states.shape)
        [1, 3137, 768]
        ```Nz You have to specify pixel_valuesrn   )r™   rš   r÷   rø   r   r   )rñ   Úpooler_outputr¢   rò   )r/   rš   r÷   Úuse_return_dictr8   Úget_head_maskrç   rV   r   r!  r"  r   r¢   rò   )r.   r<   r™   rš   r÷   r3   rø   Úembedding_outputÚencoder_outputsÚsequence_outputr  s              r1   rA   zVivitModel.forward1  s*  € ðn 2CÐ1NÑ-ÐTX×T_ÑT_×TqÑTqÐà$8Ð$DÑ È$Ï+É+×JjÑJjð 	ð &1Ð%<‘kÀ$Ç+Á+×B]ÑB]ˆàÐÜÐ?Ó@Ð@à×&Ñ& y°$·+±+×2OÑ2OÓPˆ	àŸ?™?¨<ÐRj˜?ÓkÐàŸ,™,ØØØ/Ø!5Ø#ð 'ó 
ˆð *¨!Ñ,ˆØŸ.™.¨Ó9ˆØ8<¿¹Ð8O˜Ÿ™ OÔ4ÐUYˆáØ# ]Ð3°oÀaÀbÐ6IÑIÐIä)Ø-Ø'Ø)×7Ñ7Ø&×1Ñ1ô	
ð 	
r2   )T)NNNNFN)rC   rD   rE   r"   r&  r*  r   ÚVIVIT_INPUTS_DOCSTRINGr   r   Ú_CONFIG_FOR_DOCr   rN   ÚFloatTensorrG   r   r   rA   rH   rI   s   @r1   r  r    sÙ   ø„ õ
ò0ò	Cñ +Ð+AÓBÙÐ+EÐTcÔdð 59Ø15Ø,0Ø/3Ø).Ø&*ñu
à˜u×0Ñ0Ñ1ðu
ð ˜E×-Ñ-Ñ.ðu
ð $ D™>ð	u
ð
 ' t™nðu
ð #'ðu
ð ˜d‘^ðu
ð 
ˆu�U×&Ñ&Ñ'Ð)CÐCÑ	Dòu
ó eó Côu
r2   r  aá  
    ViViT Transformer model with a video classification head on top (a linear layer on top of the final hidden state of the
[CLS] token) e.g. for Kinetics-400.

    <Tip>

        Note that it's possible to fine-tune ViT on higher resolution images than the ones it has been trained on, by
        setting `interpolate_pos_encoding` to `True` in the forward of the model. This will interpolate the pre-trained
        position embeddings to the higher resolution.

    </Tip>
    c                   ó
  ‡ — e Zd Zˆ fd„Z ee«       eee¬«      	 	 	 	 	 	 	 dde	e
j                     de	e
j                     de	e
j                     de	e   de	e   ded	e	e   d
eee
j                     ef   fd„«       «       Zˆ xZS )ÚVivitForVideoClassificationc                 ó.  •— t         ‰| �  |«       |j                  | _        t        |d¬«      | _        |j                  dkD  r*t        j                  |j                  |j                  «      nt        j                  «       | _	        | j                  «        y )NF)r$  r   )r!   r"   Ú
num_labelsr  r	  r   r”   r(   ÚIdentityÚ
classifierr#  r-   s     €r1   r"   z$VivitForVideoClassification.__init__»  ss   ø€ Ü‰Ñ˜Ô à ×+Ñ+ˆŒÜ ¸%Ô@ˆŒ
ð OU×N_ÑN_ÐbcÒNcœ"Ÿ)™) F×$6Ñ$6¸×8IÑ8IÔJÔik×itÑitÓivˆŒð 	�‰Õr2   r+  r<   r™   Úlabelsrš   r÷   r3   rø   rW   c                 ó  — |�|n| j                   j                  }| j                  ||||||¬«      }|d   }	| j                  |	dd…ddd…f   «      }
d}|�}| j                  dk(  r2t        «       } ||
j                  d«      |j                  d«      «      }n<t        «       } ||
j                  d| j                  «      |j                  d«      «      }|s|
f|dd z   }|�|f|z   S |S t        ||
|j                  |j                  ¬«      S )a(  
        labels (`torch.LongTensor` of shape `(batch_size,)`, *optional*):
            Labels for computing the image classification/regression loss. Indices should be in `[0, ...,
            config.num_labels - 1]`. If `config.num_labels == 1` a regression loss is computed (Mean-Square loss), If
            `config.num_labels > 1` a classification loss is computed (Cross-Entropy).

        Returns:

        Examples:

        ```python
        >>> import av
        >>> import numpy as np
        >>> import torch

        >>> from transformers import VivitImageProcessor, VivitForVideoClassification
        >>> from huggingface_hub import hf_hub_download

        >>> np.random.seed(0)


        >>> def read_video_pyav(container, indices):
        ...     '''
        ...     Decode the video with PyAV decoder.
        ...     Args:
        ...         container (`av.container.input.InputContainer`): PyAV container.
        ...         indices (`List[int]`): List of frame indices to decode.
        ...     Returns:
        ...         result (np.ndarray): np array of decoded frames of shape (num_frames, height, width, 3).
        ...     '''
        ...     frames = []
        ...     container.seek(0)
        ...     start_index = indices[0]
        ...     end_index = indices[-1]
        ...     for i, frame in enumerate(container.decode(video=0)):
        ...         if i > end_index:
        ...             break
        ...         if i >= start_index and i in indices:
        ...             frames.append(frame)
        ...     return np.stack([x.to_ndarray(format="rgb24") for x in frames])


        >>> def sample_frame_indices(clip_len, frame_sample_rate, seg_len):
        ...     '''
        ...     Sample a given number of frame indices from the video.
        ...     Args:
        ...         clip_len (`int`): Total number of frames to sample.
        ...         frame_sample_rate (`int`): Sample every n-th frame.
        ...         seg_len (`int`): Maximum allowed index of sample's last frame.
        ...     Returns:
        ...         indices (`List[int]`): List of sampled frame indices
        ...     '''
        ...     converted_len = int(clip_len * frame_sample_rate)
        ...     end_idx = np.random.randint(converted_len, seg_len)
        ...     start_idx = end_idx - converted_len
        ...     indices = np.linspace(start_idx, end_idx, num=clip_len)
        ...     indices = np.clip(indices, start_idx, end_idx - 1).astype(np.int64)
        ...     return indices


        >>> # video clip consists of 300 frames (10 seconds at 30 FPS)
        >>> file_path = hf_hub_download(
        ...     repo_id="nielsr/video-demo", filename="eating_spaghetti.mp4", repo_type="dataset"
        ... )
        >>> container = av.open(file_path)

        >>> # sample 32 frames
        >>> indices = sample_frame_indices(clip_len=32, frame_sample_rate=4, seg_len=container.streams.video[0].frames)
        >>> video = read_video_pyav(container=container, indices=indices)

        >>> image_processor = VivitImageProcessor.from_pretrained("google/vivit-b-16x2-kinetics400")
        >>> model = VivitForVideoClassification.from_pretrained("google/vivit-b-16x2-kinetics400")

        >>> inputs = image_processor(list(video), return_tensors="pt")

        >>> with torch.no_grad():
        ...     outputs = model(**inputs)
        ...     logits = outputs.logits

        >>> # model predicts one of the 400 Kinetics-400 classes
        >>> predicted_label = logits.argmax(-1).item()
        >>> print(model.config.id2label[predicted_label])
        LABEL_116
        ```N)r™   rš   r÷   r3   rø   r   r   rY   r   )ÚlossÚlogitsr¢   rò   )r/   r/  r	  r<  r:  r
   re   r	   r   r¢   rò   )r.   r<   r™   r=  rš   r÷   r3   rø   rª   r3  r@  r?  Úloss_fctr¹   s                 r1   rA   z#VivitForVideoClassification.forwardÇ  s  € ð@ &1Ð%<‘kÀ$Ç+Á+×B]ÑB]ˆà—*‘*ØØØ/Ø!5Ø%=Ø#ð ó 
ˆð " !™*ˆà—‘ ²°A²q°Ñ!9Ó:ˆàˆØÐØ�‰ !Ò#ä"›9�Ù §¡¨B£°·±¸R³ÓA‘ä+Ó-�Ù §¡¨B°·±Ó @À&Ç+Á+ÈbÃ/ÓR�áØ�Y ¨¨ Ñ,ˆFØ)-Ð)9�T�G˜fÑ$ÐE¸vÐEä$ØØØ!×/Ñ/Ø×)Ñ)ô	
ð 	
r2   )NNNNNFN)rC   rD   rE   r"   r   r4  r   r   r5  r   rN   r6  Ú
LongTensorrG   r   r   rA   rH   rI   s   @r1   r8  r8  «  sæ   ø„ ô 
ñ +Ð+AÓBÙÐ+@ÈÔ_ð 59Ø15Ø-1Ø,0Ø/3Ø).Ø&*ñ@
à˜u×0Ñ0Ñ1ð@
ð ˜E×-Ñ-Ñ.ð@
ð ˜×)Ñ)Ñ*ð	@
ð
 $ D™>ð@
ð ' t™nð@
ð #'ð@
ð ˜d‘^ð@
ð 
ˆu�U×&Ñ&Ñ'Ð)>Ð>Ñ	?ò@
ó `ó Cô@
r2   r8  )r  r  r8  )rž   ):rF   Útypingr   r   r   r   r   rN   Útorch.utils.checkpointr   Útorch.nnr	   r
   Úactivationsr   Úmodeling_outputsr   r   r   Úmodeling_utilsr   r   Úpytorch_utilsr   r   Úutilsr   r   r   r   r   Úconfiguration_vivitr   Ú
get_loggerrC   r    Ú_CHECKPOINT_FOR_DOCr5  ÚModuler   rK   rq   Úfloatr†   rˆ   r­   r¶   rÆ   rÐ   rÔ   rã   r   r  ÚVIVIT_START_DOCSTRINGr4  r  r8  Ú__all__rì   r2   r1   ú<module>rR     så  ðñ ç 8Õ 8ã Û Ý ß .å !ß bÑ bß Fß Q÷õ õ -ð 
ˆ×	Ñ	˜HÓ	%€à7Ð Ø€ô)˜RŸY™Yô )ôXL�b—i‘iô Lðn ñ%Ø�I‰Ið%à�<‰<ð%ð 
�‰ð%ð �<‰<ð	%ð
 ˜UŸ\™\Ñ*ð%ð ð%ð ó%ô>;˜Ÿ™ô ;ô~�b—i‘iô ô&$�R—Y‘Yô $ôN˜Ÿ	™	ô ô$�"—)‘)ô ô $�—‘ô $ôN0
�2—9‘9ô 0
ôf�"—)‘)ô ô4˜?ô 4ðD	Ð ðÐ ñ2 ØeØóôS
Ð%ó S
ó	ðS
ñl ðð óôO
Ð"6ó O
óðO
òd P�r2   