Ë
    l^(h€  ã                  ó‚   — d dl mZ d dlZd dlmZmZ d dlmZmZ d dl	Z	 ej                  e«      Ze G d„ d«      «       Zy)é    )ÚannotationsN)Ú	dataclassÚfield)ÚAnyÚCallablec                  ój   — e Zd ZU dZded<    ed„ ¬«      Zded<    eedd¬	«      Zd
ed<   dd„Z	dd„Z
y)ÚSentenceTransformerDataCollatoraŽ  Collator for a SentenceTransformers model.
    This encodes the text columns to {column}_input_ids and {column}_attention_mask columns.
    This works with the two text dataset that is used as the example in the training overview:
    https://www.sbert.net/docs/sentence_transformer/training_overview.html

    It is important that the columns are in the expected order. For example, if your dataset has columns
    "answer", "question" in that order, then the MultipleNegativesRankingLoss will consider
    "answer" as the anchor and "question" as the positive, and it will (unexpectedly) optimize for
    "given the answer, what is the question?".
    r   Útokenize_fnc                 ó
   — ddgS )NÚlabelÚscore© r   ó    úa/var/www/skyplay_api_hub/venv/lib/python3.12/site-packages/sentence_transformers/data_collator.pyú<lambda>z(SentenceTransformerDataCollator.<lambda>   s   € ÀGÈWÐCU€ r   )Údefault_factoryú	list[str]Úvalid_label_columnsF)r   ÚinitÚreprzset[tuple[str]]Ú_warned_columnsc                óà  — t        |d   j                  «       «      }i }d|v r|j                  d«       |d   d   |d<   t        |«      | j                  vr| j                  |«       | j                  D ]B  }||v sŒt        j                  |D �cg c]  }||   ‘Œ	 c}«      |d<   |j                  |«        n |D ]¢  }|j                  d«      rK|d t        d«        |v r:t        j                  |D �cg c]  }||   ‘Œ	 c}t        j                  ¬«      ||<   Œ_| j                  |D �cg c]  }||   ‘Œ	 c}«      }|j                  «       D ]  \  }}	|	||› d|› �<   Œ Œ¤ |S c c}w c c}w c c}w )Nr   Údataset_namer   Ú_prompt_length)ÚdtypeÚ_)ÚlistÚkeysÚremoveÚtupler   Úmaybe_warn_about_column_orderr   ÚtorchÚtensorÚendswithÚlenÚintr
   Úitems)
ÚselfÚfeaturesÚcolumn_namesÚbatchÚlabel_columnÚrowÚcolumn_nameÚ	tokenizedÚkeyÚvalues
             r   Ú__call__z(SentenceTransformerDataCollator.__call__   sŒ  € Ü˜H Q™K×,Ñ,Ó.Ó/ˆð ˆà˜\Ñ)Ø×Ñ Ô/Ø$,¨Q¡K°Ñ$?ˆE�.Ñ!ä�Ó d×&:Ñ&:Ñ:Ø×.Ñ.¨|Ô<ð !×4Ñ4ò 	ˆLØ˜|Ò+Ü!&§¡ÈHÖ.UÀS¨s°<Ó/@Ò.UÓ!V��g‘Ø×#Ñ# LÔ1Ùð		ð (ò 	6ˆKà×#Ñ#Ð$4Ô5¸+ÐF^ÌÐM]ÓI^ÐH^Ð:_ÐcoÑ:oÜ%*§\¡\ÈxÖ2XÈ°3°{Ó3CÒ2XÔ`e×`iÑ`iÔ%j��kÑ"Øà×(Ñ(ÀhÖ)O¸s¨#¨kÓ*:Ò)OÓPˆIØ'Ÿo™oÓ/ò 6‘
��UØ05�˜˜ Q s eÐ,Ò-ñ6ð	6ð ˆùò /Vùò 3Yùò *Ps   ÂE!
Ã2E&
Ä(E+
c                ót  — dddddddddddœ
}|j                  «       D ]t  \  }}||v sŒ|j                  |«      |k7  sŒ |dv rg d¢}n|dv rddg}n|d	v rd
dg}n|dv rg d¢}t        j                  d|›d|j                  |«      › d|› d› d�	«        n | j                  j                  t        |«      «       y)zBWarn the user if the columns are likely not in the expected order.r   é   é   )
ÚanchorÚpositiveÚnegativeÚquestionÚanswerÚqueryÚresponseÚ
hypothesisÚ
entailmentÚcontradiction)r6   r7   r8   )r9   r:   r9   r:   )r;   r<   r;   r<   )r=   r>   r?   zColumn z is at index z?, whereas a column with this name is usually expected at index a?  . Note that the column order can be important for some losses, e.g. MultipleNegativesRankingLoss will always consider the first column as the anchor and the second as the positive, regardless of the dataset column names. Consider renaming the columns to match the expected order, e.g.:
dataset = dataset.select_columns(ú)N)r'   ÚindexÚloggerÚwarningr   Úaddr    )r(   r*   Úcolumn_name_to_expected_idxr.   Úexpected_idxÚproposed_fix_columnss         r   r!   z=SentenceTransformerDataCollator.maybe_warn_about_column_order=   s  € ð ØØØØØØØØØñ'
Ð#ð *E×)JÑ)JÓ)Lò 	Ñ%ˆK˜Ø˜lÒ*¨|×/AÑ/AÀ+Ó/NÐR^Ó/^ØÐ"DÑDÚ+MÑ(Ø Ð$:Ñ:Ø,6¸Ð+AÑ(Ø Ð$9Ñ9Ø,3°ZÐ+@Ñ(Ø Ð$QÑQÚ+XÐ(ä—‘Ø˜k˜_¨M¸,×:LÑ:LÈ[Ó:YÐ9Zð [LØLXÈ>ð Z8ð 9MÐ7MÈQðPôñ ð)	ð, 	×Ñ× Ñ ¤ |Ó!4Õ5r   N)r)   zlist[dict[str, Any]]Úreturnzdict[str, torch.Tensor])r*   r   rH   ÚNone)Ú__name__Ú
__module__Ú__qualname__Ú__doc__Ú__annotations__r   r   Úsetr   r2   r!   r   r   r   r	   r	      s?   … ñ	ð ÓÙ%*Ñ;UÔ%VÐ˜ÓVÙ',¸SÀuÐSXÔ'Y€O�_ÓYóô@%6r   r	   )Ú
__future__r   ÚloggingÚdataclassesr   r   Útypingr   r   r"   Ú	getLoggerrJ   rB   r	   r   r   r   ú<module>rU      sB   ðÝ "ã ß (ß  ã à	ˆ×	Ñ	˜8Ó	$€ð ÷U6ð U6ó ñU6r   