Ë
    TêñiØC  ã            
      óŽ  — d dl mZ d dlZd dlZd dlZd dlmZ d dlmZ d dl	Z
d dlZd dlmZ d dlmZ ddlmZ dd	lmZ  ej&                  e«      Zerd d
lmZ ddddddedddf
	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 dd„Zddddef	 	 	 	 	 	 	 	 	 	 	 	 	 dd„Zdd„Zdddef	 	 	 	 	 	 	 	 	 	 	 	 	 dd„Z	 	 	 	 d	 	 	 	 	 	 	 	 	 	 	 dd„Zy)é    )ÚannotationsN)ÚCallable)ÚTYPE_CHECKING)ÚTensor)Útqdmé   )Úcos_sim)Únormalize_embeddings)ÚSentenceTransformerFé    iˆ  i † i ¡ éd   c           	     óT   — | j                  |||d|	|
|¬«      }t        ||||||¬«      S )a@	  
    Given a list of sentences / texts, this function performs paraphrase mining. It compares all sentences against all
    other sentences and returns a list with the pairs that have the highest cosine similarity score.

    Args:
        model (SentenceTransformer): SentenceTransformer model for embedding computation
        sentences (List[str]): A list of strings (texts or sentences)
        show_progress_bar (bool, optional): Plotting of a progress bar. Defaults to False.
        batch_size (int, optional): Number of texts that are encoded simultaneously by the model. Defaults to 32.
        query_chunk_size (int, optional): Search for most similar pairs for #query_chunk_size at the same time. Decrease, to lower memory footprint (increases run-time). Defaults to 5000.
        corpus_chunk_size (int, optional): Compare a sentence simultaneously against #corpus_chunk_size other sentences. Decrease, to lower memory footprint (increases run-time). Defaults to 100000.
        max_pairs (int, optional): Maximal number of text pairs returned. Defaults to 500000.
        top_k (int, optional): For each sentence, we retrieve up to top_k other sentences. Defaults to 100.
        score_function (Callable[[Tensor, Tensor], Tensor], optional): Function for computing scores. By default, cosine similarity. Defaults to cos_sim.
        truncate_dim (int, optional): The dimension to truncate sentence embeddings to. If None, uses the model's ones. Defaults to None.
        prompt_name (Optional[str], optional): The name of a predefined prompt to use when encoding the sentence.
            It must match a key in the model `prompts` dictionary, which can be set during model initialization
            or loaded from the model configuration.

            Ignored if `prompt` is provided. Defaults to None.

        prompt (Optional[str], optional): A raw prompt string to prepend directly to the input sentence during encoding.

            For instance, `prompt="query: "` transforms the sentence "What is the capital of France?" into:
            "query: What is the capital of France?". Use this to override the prompt logic entirely and supply your own prefix.
            This takes precedence over `prompt_name`. Defaults to None.

    Returns:
        List[List[Union[float, int]]]: Returns a list of triplets with the format [score, id1, id2]
    T)Úshow_progress_barÚ
batch_sizeÚconvert_to_tensorÚtruncate_dimÚprompt_nameÚprompt)Úquery_chunk_sizeÚcorpus_chunk_sizeÚ	max_pairsÚtop_kÚscore_function)ÚencodeÚparaphrase_mining_embeddings)ÚmodelÚ	sentencesr   r   r   r   r   r   r   r   r   r   Ú
embeddingss                úf/var/www/pod-logistic/pod-ai/venv/lib/python3.12/site-packages/sentence_transformers/util/retrieval.pyÚparaphrase_miningr       sN   € ð\ —‘ØØ+ØØØ!ØØð ó €Jô (ØØ)Ø+ØØØ%ôð ó    c                óê  — |dz  }t        j                  «       }d}d}t        dt        | «      |«      D �])  }	t        dt        | «      |«      D �]  }
 || |
|
|z    | |	|	|z    «      }t	        j
                  |t        |t        |d   «      «      ddd¬«      \  }}|j                  «       j                  «       }|j                  «       j                  «       }t        t        |«      «      D ]n  }t        ||   «      D ][  \  }}|
|z   }|	|z   }||k7  sŒ||   |   |kD  sŒ"|j                  ||   |   ||f«       |dz  }||k\  sŒG|j                  «       }|d   }Œ] Œp �Œ �Œ, t        «       }g }|j                  «       sg|j                  «       \  }}}t        ||g«      \  }}||k7  r-||f|vr'|j                  ||f«       |j!                  |||g«       |j                  «       sŒgt        |d„ d¬«      }|S )	aì  
    Given a list of sentences / texts, this function performs paraphrase mining. It compares all sentences against all
    other sentences and returns a list with the pairs that have the highest cosine similarity score.

    Args:
        embeddings (Tensor): A tensor with the embeddings
        query_chunk_size (int): Search for most similar pairs for #query_chunk_size at the same time. Decrease, to lower memory footprint (increases run-time).
        corpus_chunk_size (int): Compare a sentence simultaneously against #corpus_chunk_size other sentences. Decrease, to lower memory footprint (increases run-time).
        max_pairs (int): Maximal number of text pairs returned.
        top_k (int): For each sentence, we retrieve up to top_k other sentences
        score_function (Callable[[Tensor, Tensor], Tensor]): Function for computing scores. By default, cosine similarity.

    Returns:
        List[List[Union[float, int]]]: Returns a list of triplets with the format [score, id1, id2]
    r   éÿÿÿÿr   TF©ÚdimÚlargestÚsortedc                ó   — | d   S )Nr   © ©Úxs    r   ú<lambda>z.paraphrase_mining_embeddings.<locals>.<lambda>ž   s
   € °!°A±$€ r!   ©ÚkeyÚreverse)ÚqueueÚPriorityQueueÚrangeÚlenÚtorchÚtopkÚminÚcpuÚtolistÚ	enumerateÚputÚgetÚsetÚemptyr'   ÚaddÚappend)r   r   r   r   r   r   ÚpairsÚ	min_scoreÚ	num_addedÚcorpus_start_idxÚquery_start_idxÚscoresÚscores_top_k_valuesÚscores_top_k_idxÚ	query_itrÚ	top_k_idxÚ
corpus_itrÚiÚjÚentryÚadded_pairsÚ
pairs_listÚscoreÚsorted_iÚsorted_js                            r   r   r   Y   s/  € ð0 
ˆQ�J€Eô ×ÑÓ!€EØ€IØ€Iä! !¤S¨£_Ð6GÓHó 1ÐÜ$ Q¬¨J«Ð9IÓJó 	1ˆOÙ#Ø˜?¨_Ð?OÑ-OÐPØÐ+Ð.>ÐARÑ.RÐSóˆFô
 5:·J±JØœ˜E¤3 v¨a¡y£>Ó2¸À4ÐPUô5Ñ1ÐÐ!1ð #6×"9Ñ"9Ó";×"BÑ"BÓ"DÐØ/×3Ñ3Ó5×<Ñ<Ó>Ðä"¤3 v£;Ó/ò 1�	Ü-6Ð7GÈ	Ñ7RÓ-Sò 
1Ñ)�I˜zØ'¨)Ñ3�AØ(¨:Ñ5�Aà˜A“vÐ"5°iÑ"@ÀÑ"KÈiÓ"WØŸ	™	Ð#6°yÑ#AÀ)Ñ#LÈaÐQRÐ"SÔTØ! Q™˜	à$¨	Ó1Ø$)§I¡I£K˜EØ(-¨a©™Iñ
1ò1ò	1ð1ô6 “%€KØ€JØ�k‰kŒmØ—i‘i“k‰ˆˆq�!Ü# Q¨ F›^Ñˆ�(à�xÒ X¨xÐ$8ÀÑ$KØ�O‰O˜X xÐ0Ô1Ø×Ñ˜u h°Ð9Ô:ð �k‰k�mô ˜
©ÀÔE€JØÐr!   c                 ó   — t        | i |¤ŽS )z8This function is deprecated. Use semantic_search instead)Úsemantic_search)ÚargsÚkwargss     r   Úinformation_retrievalrW   ¢   s   € ä˜DÐ+ FÑ+Ð+r!   é
   c                óF  — t        | t        j                  t        j                  f«      rt	        j
                  | «      } n%t        | t        «      rt	        j                  | «      } t        | j                  «      dk(  r| j                  d«      } t        |t        j                  t        j                  f«      rt	        j
                  |«      }n%t        |t        «      rt	        j                  |«      }|j                  | j                  k7  r| j                  |j                  «      } t        t        | «      «      D �cg c]  }g ‘Œ }}t        dt        | «      |«      D �]Ù  }t        ||z   t        | «      «      }	| j                  r5t	        j                   ||	| j                  ¬«      }
| j#                  d|
«      }n| ||	 }t        dt        |«      |«      D �]^  }t        ||z   t        |«      «      }|j                  r5t	        j                   |||j                  ¬«      }
|j#                  d|
«      }n||| } |||«      }t	        j$                  |t        |t        |d   «      «      ddd¬«      \  }}|j'                  «       j)                  «       }|j'                  «       j)                  «       }t        t        |«      «      D ]n  }t+        ||   ||   «      D ]W  \  }}||z   }||z   }t        ||   «      |k  rt-        j.                  ||   ||f«       Œ=t-        j0                  ||   ||f«       ŒY Œp �Œa �ŒÜ t        t        |«      «      D ]I  }t        t        ||   «      «      D ]  }||   |   \  }}||dœ||   |<   Œ t3        ||   d„ d¬	«      ||<   ŒK |S c c}w )
a3  
    This function performs by default a cosine similarity search between a list of query embeddings  and a list of corpus embeddings.
    It can be used for Information Retrieval / Semantic Search for corpora up to about 1 Million entries.

    Args:
        query_embeddings (:class:`~torch.Tensor`): A 2 dimensional tensor with the query embeddings. Can be a sparse tensor.
        corpus_embeddings (:class:`~torch.Tensor`): A 2 dimensional tensor with the corpus embeddings. Can be a sparse tensor.
        query_chunk_size (int, optional): Process 100 queries simultaneously. Increasing that value increases the speed, but requires more memory. Defaults to 100.
        corpus_chunk_size (int, optional): Scans the corpus 100k entries at a time. Increasing that value increases the speed, but requires more memory. Defaults to 500000.
        top_k (int, optional): Retrieve top k matching entries. Defaults to 10.
        score_function (Callable[[:class:`~torch.Tensor`, :class:`~torch.Tensor`], :class:`~torch.Tensor`], optional): Function for computing scores. By default, cosine similarity.

    Returns:
        List[List[Dict[str, Union[int, float]]]]: A list with one entry for each query. Each entry is a list of dictionaries with the keys 'corpus_id' and 'score', sorted by decreasing cosine similarity scores.
    r   r   ©ÚdeviceTFr$   )Ú	corpus_idrP   c                ó   — | d   S )NrP   r)   r*   s    r   r,   z!semantic_search.<locals>.<lambda>ý   s   € Ð\]Ð^eÑ\f€ r!   r-   )Ú
isinstanceÚnpÚndarrayÚgenericr4   Ú
from_numpyÚlistÚstackr3   ÚshapeÚ	unsqueezer[   Útor2   r6   Ú	is_sparseÚarangeÚindex_selectr5   r7   r8   ÚzipÚheapqÚheappushÚheappushpopr'   )Úquery_embeddingsÚcorpus_embeddingsr   r   r   r   Ú_Úqueries_result_listrD   Úquery_end_idxÚindicesÚquery_chunkrC   Úcorpus_end_idxÚcorpus_chunkÚ
cos_scoresÚcos_scores_top_k_valuesÚcos_scores_top_k_idxrH   Úsub_corpus_idrP   r\   Úquery_idÚdoc_itrs                           r   rT   rT   §   s¨  € ô0 Ð"¤R§Z¡Z´·±Ð$<Ô=Ü ×+Ñ+Ð,<Ó=ÑÜ	Ð$¤dÔ	+Ü Ÿ;™;Ð'7Ó8Ðä
Ð×!Ñ!Ó" aÒ'Ø+×5Ñ5°aÓ8ÐäÐ#¤b§j¡j´"·*±*Ð%=Ô>Ü!×,Ñ,Ð->Ó?ÑÜ	Ð%¤tÔ	,Ü!ŸK™KÐ(9Ó:Ðð ×ÑÐ#3×#:Ñ#:Ò:Ø+×.Ñ.Ð/@×/GÑ/GÓHÐä',¬SÐ1AÓ-BÓ'CÖD !š2ÐDÐÐDä  ¤CÐ(8Ó$9Ð;KÓLó $]ˆÜ˜OÐ.>Ñ>ÄÐDTÓ@UÓVˆØ×%Ò%Ü—l‘l ?°MÐJZ×JaÑJaÔbˆGØ*×7Ñ7¸¸7ÓC‰Kà*¨?¸=ÐIˆKô !& a¬Ð->Ó)?ÐARÓ Só 	]ÐÜ Ð!1Ð4EÑ!EÄsÐK\ÓG]Ó^ˆNØ ×*Ò*ÜŸ,™,Ð'7¸ÐPa×PhÑPhÔi�Ø0×=Ñ=¸aÀÓI‘à0Ð1AÀ.ÐQ�ñ (¨°\ÓBˆJô =B¿J¹JØœC ¤s¨:°a©=Ó'9Ó:ÀÈ4ÐX]ô=Ñ9Ð#Ð%9ð '>×&AÑ&AÓ&C×&JÑ&JÓ&LÐ#Ø#7×#;Ñ#;Ó#=×#DÑ#DÓ#FÐ ä"¤3 z£?Ó3ò 	]�	Ü,/Ð0DÀYÑ0OÐQhÐirÑQsÓ,tò ]Ñ(�M 5Ø 0°=Ñ @�IØ.°Ñ:�HÜÐ.¨xÑ8Ó9¸EÒAÜŸ™Ø/°Ñ9¸EÀ9Ð;Mõô ×)Ñ)Ð*=¸hÑ*GÈ%ÐQZÐI[Õ\ñ]ò	]ò%	]ð$]ôN œ#Ð1Ó2Ó3ò vˆÜœSÐ!4°XÑ!>Ó?Ó@ò 	^ˆGØ2°8Ñ<¸WÑEÑˆE�9ØCLÐW\Ñ5]Ð Ñ)¨'Ò2ð	^ô )/Ð/BÀ8Ñ/LÑRfÐptÔ(uÐ˜HÒ%ð	vð Ðùò_ Es   Ä>	Nc                óÊ  — t        | t        j                  «      st        j                  | «      } t        j                  || j                  ¬«      }t        | «      } g }t        |t        | «      «      }t        t        d|z  d«      t        | «      «      }t        t        dt        | «      |«      d| ¬«      D �]š  }| |||z    | j                  z  }| j                  j                  dv r“||k\  }	|	j                  d«      }
|
|k\  }|j                  «       sŒ]|
|   }
||   }|
j                  «       }|j                  |d	¬
«      \  }}t!        |
|«      D ]'  \  }}|j#                  |d| j%                  «       «       Œ) ŒÄ|j                  |d	¬
«      \  }}t        t        |«      «      D ]ª  }||   d   |k\  sŒ||   j                  |d	¬
«      \  }}|d   |kD  rV|t        | «      k  rHt        d|z  t        | «      «      }||   j                  |d	¬
«      \  }}|d   |kD  r|t        | «      k  rŒH|j#                  |||k\     j%                  «       «       Œ¬ �Œ� t'        |d„ d	¬«      }g }t)        «       }t+        |«      D ]U  \  }}g }|D ]  }||vsŒ|j#                  |«       Œ t        |«      |k\  sŒ4|j#                  |«       |j-                  |«       ŒW t'        |d„ d	¬«      }|S )a¼  
    Function for Fast Community Detection.

    Finds in the embeddings all communities, i.e. embeddings that are close (closer than threshold).
    Returns only communities that are larger than min_community_size. The communities are returned
    in decreasing order. The first element in each list is the central point in the community.

    Args:
        embeddings (torch.Tensor or numpy.ndarray): The input embeddings.
        threshold (float): The threshold for determining if two embeddings are close. Defaults to 0.75.
        min_community_size (int): The minimum size of a community to be considered. Defaults to 10.
        batch_size (int): The batch size for computing cosine similarity scores. Defaults to 1024.
        show_progress_bar (bool): Whether to show a progress bar during computation. Defaults to False.

    Returns:
        List[List[int]]: A list of communities, where each community is represented as a list of indices.
    rZ   é   é2   r   zFinding clusters)ÚdescÚdisable)ÚcudaÚnpur   T)Úkr&   Nr#   c                ó   — t        | «      S ©N©r3   r*   s    r   r,   z%community_detection.<locals>.<lambda>S  s
   € ÌÈAË€ r!   r-   c                ó   — t        | «      S r‡   rˆ   r*   s    r   r,   z%community_detection.<locals>.<lambda>c  s
   € Ä#ÀaÃ&€ r!   )r^   r4   r   Útensorr[   r
   r6   r3   Úmaxr   r2   ÚTÚtypeÚsumÚanyr5   rk   r?   r8   r'   r<   r9   Úupdate)r   Ú	thresholdÚmin_community_sizer   r   Úextracted_communitiesÚsort_max_sizeÚ	start_idxrx   Úthreshold_maskÚrow_wise_countÚlarge_enough_maskr…   rq   Útop_k_indicesÚcountrt   Útop_k_valuesrK   Útop_val_largeÚtop_idx_largeÚunique_communitiesÚextracted_idsÚ
cluster_idÚ	communityÚnon_overlapped_communityÚidxs                              r   Úcommunity_detectionr¤     s.  € ô0 �j¤%§,¡,Ô/Ü—\‘\ *Ó-ˆ
ä—‘˜Y¨z×/@Ñ/@ÔA€IÜ% jÓ1€JàÐô Ð/´°Z³ÓAÐÜœ˜AÐ 2Ñ2°BÓ7¼¸Z»ÓI€MäÜˆa”�Z“ *Ó-Ð4FÐTeÐPeôó *eˆ	ð   	¨I¸
Ñ,BÐCÀjÇlÁlÑRˆ
ð ×Ñ×!Ñ! _Ñ4à'¨9Ñ4ˆNØ+×/Ñ/°Ó2ˆNð !/Ð2DÑ DÐØ$×(Ñ(Ô*Øà+Ð,=Ñ>ˆNØ#Ð$5Ñ6ˆJð ×"Ñ"Ó$ˆAØ)Ÿ™°¸D˜ÓAÑˆAˆ}ô #& n°mÓ"Dò G‘��wØ%×,Ñ,¨W°V°e¨_×-CÑ-CÓ-EÕFñGð )Ÿo™oÐ0BÈD˜oÓQ‰OˆL˜!ô œ3˜|Ó,Ó-ò 
e�Ø ‘? 2Ñ&¨)Ó3à3=¸a±=×3EÑ3EÈÐ_cÐ3EÓ3dÑ0�M =ð (¨Ñ+¨iÒ7¸MÌCÐPZËOÒ<[Ü(+¨A°Ñ,=¼sÀ:»Ó(O˜Ø7AÀ!±}×7IÑ7IÈMÐcgÐ7IÓ7hÑ4˜ }ð (¨Ñ+¨iÒ7¸MÌCÐPZËOÓ<[ð *×0Ñ0°¸}ÐPYÑ?YÑ1Z×1aÑ1aÓ1cÕdò
eðA*eôZ #Ð#8Ñ>NÐX\Ô]Ðð ÐÜ“E€Mä!*Ð+@Ó!Aò ;Ñˆ
�IØ#%Ð Øò 	5ˆCØ˜-Ò'Ø(×/Ñ/°Õ4ð	5ô Ð'Ó(Ð,>Ó>Ø×%Ñ%Ð&>Ô?Ø× Ñ Ð!9Õ:ð;ô  Ð 2Ñ8HÐRVÔWÐàÐr!   )r   r   r   z	list[str]r   Úboolr   Úintr   r¦   r   r¦   r   r¦   r   r¦   r   ú"Callable[[Tensor, Tensor], Tensor]r   z
int | Noner   ú
str | Noner   r¨   Úreturnúlist[list[float | int]])r   r   r   r¦   r   r¦   r   r¦   r   r¦   r   r§   r©   rª   )r©   ú"list[list[dict[str, int | float]]])ro   r   rp   r   r   r¦   r   r¦   r   r¦   r   r§   r©   r«   )g      è?rX   i   F)r   ztorch.Tensor | np.ndarrayr‘   Úfloatr’   r¦   r   r¦   r   r¥   r©   zlist[list[int]])Ú
__future__r   rl   Úloggingr0   Úcollections.abcr   Útypingr   Únumpyr_   r4   r   Útqdm.autonotebookr   Ú
similarityr	   rŠ   r
   Ú	getLoggerÚ__name__ÚloggerÚ0sentence_transformers.sentence_transformer.modelr   r    r   rW   rT   r¤   r)   r!   r   ú<module>r¸      sû  ðÝ "ã Û Û Ý $Ý  ã Û Ý Ý "å Ý (à	ˆ×	Ñ	˜8Ó	$€áÝTð $ØØ Ø#ØØØ9@Ø#Ø"Øð?Øð?àð?ð ð?ð ð	?ð
 ð?ð ð?ð ð?ð ð?ð 7ð?ð ð?ð ð?ð ð?ð ó?ðH !Ø#ØØØ9@ðFØðFàðFð ðFð ð	Fð
 ðFð 7ðFð óFóR,ð  Ø#ØØ9@ðXØðXàðXð ðXð ð	Xð
 ðXð 7ðXð (óXðz Ø ØØ#ðcØ)ðcàðcð ðcð ð	cð
 ðcð ôcr!   