§
    ™ŠtjÃD  ã            
      ó  — d dl mZ d dlZd dlZd dlZd dlmZ d dlmZ d dl	Z
d dlZd dlmZ d dlmZ ddlmZ dd	lmZ  ej        e¦  «        Zerd d
lmZ ddddddedddf
d8d&„Zddddefd9d)„Zd:d+„Zddd,efd;d/„Z	 	 	 	 d<d=d7„ZdS )>é    )ÚannotationsN)ÚCallable)ÚTYPE_CHECKING)ÚTensor)Útqdmé   )Úcos_sim)Únormalize_embeddings)ÚSentenceTransformerFé    iˆ  i † i ¡ éd   Úmodelr   Ú	sentencesú	list[str]Úshow_progress_barÚboolÚ
batch_sizeÚintÚquery_chunk_sizeÚcorpus_chunk_sizeÚ	max_pairsÚtop_kÚscore_functionú"Callable[[Tensor, Tensor], Tensor]Útruncate_dimú
int | NoneÚprompt_nameú
str | NoneÚpromptÚreturnúlist[list[float | int]]c           	     ód   — |                       |||d|	|
|¬¦  «        }t          ||||||¬¦  «        S )a@	  
    Given a list of sentences / texts, this function performs paraphrase mining. It compares all sentences against all
    other sentences and returns a list with the pairs that have the highest cosine similarity score.

    Args:
        model (SentenceTransformer): SentenceTransformer model for embedding computation
        sentences (List[str]): A list of strings (texts or sentences)
        show_progress_bar (bool, optional): Plotting of a progress bar. Defaults to False.
        batch_size (int, optional): Number of texts that are encoded simultaneously by the model. Defaults to 32.
        query_chunk_size (int, optional): Search for most similar pairs for #query_chunk_size at the same time. Decrease, to lower memory footprint (increases run-time). Defaults to 5000.
        corpus_chunk_size (int, optional): Compare a sentence simultaneously against #corpus_chunk_size other sentences. Decrease, to lower memory footprint (increases run-time). Defaults to 100000.
        max_pairs (int, optional): Maximal number of text pairs returned. Defaults to 500000.
        top_k (int, optional): For each sentence, we retrieve up to top_k other sentences. Defaults to 100.
        score_function (Callable[[Tensor, Tensor], Tensor], optional): Function for computing scores. By default, cosine similarity. Defaults to cos_sim.
        truncate_dim (int, optional): The dimension to truncate sentence embeddings to. If None, uses the model's ones. Defaults to None.
        prompt_name (Optional[str], optional): The name of a predefined prompt to use when encoding the sentence.
            It must match a key in the model `prompts` dictionary, which can be set during model initialization
            or loaded from the model configuration.

            Ignored if `prompt` is provided. Defaults to None.

        prompt (Optional[str], optional): A raw prompt string to prepend directly to the input sentence during encoding.

            For instance, `prompt="query: "` transforms the sentence "What is the capital of France?" into:
            "query: What is the capital of France?". Use this to override the prompt logic entirely and supply your own prefix.
            This takes precedence over `prompt_name`. Defaults to None.

    Returns:
        List[List[Union[float, int]]]: Returns a list of triplets with the format [score, id1, id2]
    T)r   r   Úconvert_to_tensorr   r   r   )r   r   r   r   r   )ÚencodeÚparaphrase_mining_embeddings)r   r   r   r   r   r   r   r   r   r   r   r   Ú
embeddingss                úb/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/sentence_transformers/util/retrieval.pyÚparaphrase_miningr(      s]   € ð\ —’ØØ+ØØØ!ØØð ñ ô €Jõ (ØØ)Ø+ØØØ%ðñ ô ð ó    r&   r   c                óÒ  — |dz  }t          j        ¦   «         }d}d}t          dt          | ¦  «        |¦  «        D �]x}	t          dt          | ¦  «        |¦  «        D �]U}
 || |
|
|z   …         | |	|	|z   …         ¦  «        }t	          j        |t          |t          |d         ¦  «        ¦  «        ddd¬¦  «        \  }}|                     ¦   «                              ¦   «         }|                     ¦   «                              ¦   «         }t          t          |¦  «        ¦  «        D ]Š}t          ||         ¦  «        D ]r\  }}|
|z   }|	|z   }||k    r]||         |         |k    rK| 
                    ||         |         ||f¦  «         |dz  }||k    r|                     ¦   «         }|d         }ŒsŒ‹�ŒW�Œzt          ¦   «         }g }|                     ¦   «         s{|                     ¦   «         \  }}}t          ||g¦  «        \  }}||k    r5||f|vr/|                     ||f¦  «         |                     |||g¦  «         |                     ¦   «         ¯{t          |d„ d¬¦  «        }|S )	aì  
    Given a list of sentences / texts, this function performs paraphrase mining. It compares all sentences against all
    other sentences and returns a list with the pairs that have the highest cosine similarity score.

    Args:
        embeddings (Tensor): A tensor with the embeddings
        query_chunk_size (int): Search for most similar pairs for #query_chunk_size at the same time. Decrease, to lower memory footprint (increases run-time).
        corpus_chunk_size (int): Compare a sentence simultaneously against #corpus_chunk_size other sentences. Decrease, to lower memory footprint (increases run-time).
        max_pairs (int): Maximal number of text pairs returned.
        top_k (int): For each sentence, we retrieve up to top_k other sentences
        score_function (Callable[[Tensor, Tensor], Tensor]): Function for computing scores. By default, cosine similarity.

    Returns:
        List[List[Union[float, int]]]: Returns a list of triplets with the format [score, id1, id2]
    r   éÿÿÿÿr   TF©ÚdimÚlargestÚsortedc                ó   — | d         S )Nr   © ©Úxs    r'   ú<lambda>z.paraphrase_mining_embeddings.<locals>.<lambda>ž   s
   € °!°A´$€ r)   ©ÚkeyÚreverse)ÚqueueÚPriorityQueueÚrangeÚlenÚtorchÚtopkÚminÚcpuÚtolistÚ	enumerateÚputÚgetÚsetÚemptyr/   ÚaddÚappend)r&   r   r   r   r   r   ÚpairsÚ	min_scoreÚ	num_addedÚcorpus_start_idxÚquery_start_idxÚscoresÚscores_top_k_valuesÚscores_top_k_idxÚ	query_itrÚ	top_k_idxÚ
corpus_itrÚiÚjÚentryÚadded_pairsÚ
pairs_listÚscoreÚsorted_iÚsorted_js                            r'   r%   r%   Y   sµ  € ð0 
ˆQ�J€Eõ ÔÑ!Ô!€EØ€IØ€Iå! !¥S¨¡_¤_Ð6GÑHÔHð 1ñ 1ÐÝ$ Q­¨J©¬Ð9IÑJÔJð 	1ñ 	1ˆOØ#�^Ø˜?¨_Ð?OÑ-OÐOÔPØÐ+Ð.>ÐARÑ.RÐRÔSñô ˆFõ
 5:´JØ�˜E¥3 v¨a¤y¡>¤>Ñ2Ô2¸À4ÐPUð5ñ 5ô 5Ñ1ÐÐ!1ð #6×"9Ò"9Ñ";Ô";×"BÒ"BÑ"DÔ"DÐØ/×3Ò3Ñ5Ô5×<Ò<Ñ>Ô>Ðå"¥3 v¡;¤;Ñ/Ô/ð 1ð 1�	Ý-6Ð7GÈ	Ô7RÑ-SÔ-Sð 
1ð 
1Ñ)�I˜zØ'¨)Ñ3�AØ(¨:Ñ5�Aà˜A’v�vÐ"5°iÔ"@ÀÔ"KÈiÒ"WÐ"WØŸ	š	Ð#6°yÔ#AÀ)Ô#LÈaÐQRÐ"SÑTÔTÐTØ! Q™˜	à$¨	Ò1Ð1Ø$)§I¢I¡K¤K˜EØ(-¨a¬˜Iøð
1ñ1ñ	1õ4 ‘%”%€KØ€JØ�kŠk‰mŒmð ;Ø—i’i‘k”k‰ˆˆq�!Ý# Q¨ F™^œ^Ñˆ�(à�xÒÐ X¨xÐ$8ÀÐ$KÐ$KØ�OŠO˜X xÐ0Ñ1Ô1Ð1Ø×Ò˜u h°Ð9Ñ:Ô:Ð:ð �kŠk‰mŒmð ;õ ˜
¨¨ÀÐEÑEÔE€JØÐr)   ú"list[list[dict[str, int | float]]]c                 ó   — t          | i |¤ŽS )z8This function is deprecated. Use semantic_search instead)Úsemantic_search)ÚargsÚkwargss     r'   Úinformation_retrievalr`   ¢   s   € å˜DÐ+ FÐ+Ð+Ð+r)   é
   Úquery_embeddingsÚcorpus_embeddingsc                ó  — t          | t          j        t          j        f¦  «        rt	          j        | ¦  «        } n)t          | t          ¦  «        rt	          j        | ¦  «        } t          | j	        ¦  «        dk    r|  
                    d¦  «        } t          |t          j        t          j        f¦  «        rt	          j        |¦  «        }n)t          |t          ¦  «        rt	          j        |¦  «        }|j        | j        k    r|                      |j        ¦  «        } d„ t          t          | ¦  «        ¦  «        D ¦   «         }t          dt          | ¦  «        |¦  «        D �]"}t          ||z   t          | ¦  «        ¦  «        }| j        r3t	          j        ||| j        ¬¦  «        }	|                      d|	¦  «        }
n
| ||…         }
t          dt          |¦  «        |¦  «        D �]›}t          ||z   t          |¦  «        ¦  «        }|j        r3t	          j        |||j        ¬¦  «        }	|                     d|	¦  «        }n
|||…         } ||
|¦  «        }t	          j        |t          |t          |d         ¦  «        ¦  «        ddd¬¦  «        \  }}|                     ¦   «                              ¦   «         }|                     ¦   «                              ¦   «         }t          t          |¦  «        ¦  «        D ]‚}t+          ||         ||         ¦  «        D ]c\  }}||z   }||z   }t          ||         ¦  «        |k     rt-          j        ||         ||f¦  «         ŒFt-          j        ||         ||f¦  «         ŒdŒƒ�Œ��Œ$t          t          |¦  «        ¦  «        D ]b}t          t          ||         ¦  «        ¦  «        D ]!}||         |         \  }}||dœ||         |<   Œ"t3          ||         d	„ d¬
¦  «        ||<   Œc|S )a3  
    This function performs by default a cosine similarity search between a list of query embeddings  and a list of corpus embeddings.
    It can be used for Information Retrieval / Semantic Search for corpora up to about 1 Million entries.

    Args:
        query_embeddings (:class:`~torch.Tensor`): A 2 dimensional tensor with the query embeddings. Can be a sparse tensor.
        corpus_embeddings (:class:`~torch.Tensor`): A 2 dimensional tensor with the corpus embeddings. Can be a sparse tensor.
        query_chunk_size (int, optional): Process 100 queries simultaneously. Increasing that value increases the speed, but requires more memory. Defaults to 100.
        corpus_chunk_size (int, optional): Scans the corpus 100k entries at a time. Increasing that value increases the speed, but requires more memory. Defaults to 500000.
        top_k (int, optional): Retrieve top k matching entries. Defaults to 10.
        score_function (Callable[[:class:`~torch.Tensor`, :class:`~torch.Tensor`], :class:`~torch.Tensor`], optional): Function for computing scores. By default, cosine similarity.

    Returns:
        List[List[Dict[str, Union[int, float]]]]: A list with one entry for each query. Each entry is a list of dictionaries with the keys 'corpus_id' and 'score', sorted by decreasing cosine similarity scores.
    r   r   c                ó   — g | ]}g ‘ŒS r1   r1   )Ú.0Ú_s     r'   ú
<listcomp>z#semantic_search.<locals>.<listcomp>Ð   s   € ÐDÐDÐD !˜2ÐDÐDÐDr)   ©ÚdeviceTFr,   )Ú	corpus_idrX   c                ó   — | d         S )NrX   r1   r2   s    r'   r4   z!semantic_search.<locals>.<lambda>ý   s   € Ð\]Ð^eÔ\f€ r)   r5   )Ú
isinstanceÚnpÚndarrayÚgenericr<   Ú
from_numpyÚlistÚstackr;   ÚshapeÚ	unsqueezerj   Útor:   r>   Ú	is_sparseÚarangeÚindex_selectr=   r?   r@   ÚzipÚheapqÚheappushÚheappushpopr/   )rb   rc   r   r   r   r   Úqueries_result_listrL   Úquery_end_idxÚindicesÚquery_chunkrK   Úcorpus_end_idxÚcorpus_chunkÚ
cos_scoresÚcos_scores_top_k_valuesÚcos_scores_top_k_idxrP   Úsub_corpus_idrX   rk   Úquery_idÚdoc_itrs                          r'   r]   r]   §   se  € õ0 Ð"¥R¤Zµ´Ð$<Ñ=Ô=ð 9Ý Ô+Ð,<Ñ=Ô=ÐÐÝ	Ð$¥dÑ	+Ô	+ð 9Ý œ;Ð'7Ñ8Ô8Ðå
ÐÔ!Ñ"Ô" aÒ'Ð'Ø+×5Ò5°aÑ8Ô8ÐåÐ#¥b¤jµ"´*Ð%=Ñ>Ô>ð ;Ý!Ô,Ð->Ñ?Ô?ÐÐÝ	Ð%¥tÑ	,Ô	,ð ;Ý!œKÐ(9Ñ:Ô:Ðð ÔÐ#3Ô#:Ò:Ð:Ø+×.Ò.Ð/@Ô/GÑHÔHÐàDÐD¥u­SÐ1AÑ-BÔ-BÑ'CÔ'CÐDÑDÔDÐå  ¥CÐ(8Ñ$9Ô$9Ð;KÑLÔLð $]ñ $]ˆÝ˜OÐ.>Ñ>ÅÐDTÑ@UÔ@UÑVÔVˆØÔ%ð 	JÝ”l ?°MÐJZÔJaÐbÑbÔbˆGØ*×7Ò7¸¸7ÑCÔCˆKˆKà*¨?¸=Ð+HÔIˆKõ !& a­Ð->Ñ)?Ô)?ÐARÑ SÔ Sð 	]ñ 	]ÐÝ Ð!1Ð4EÑ!EÅsÐK\ÑG]ÔG]Ñ^Ô^ˆNØ Ô*ð RÝœ,Ð'7¸ÐPaÔPhÐiÑiÔi�Ø0×=Ò=¸aÀÑIÔI��à0Ð1AÀ.Ð1PÔQ�ð (˜¨°\ÑBÔBˆJõ =B¼JØ�C ¥s¨:°a¬=Ñ'9Ô'9Ñ:Ô:ÀÈ4ÐX]ð=ñ =ô =Ñ9Ð#Ð%9ð '>×&AÒ&AÑ&CÔ&C×&JÒ&JÑ&LÔ&LÐ#Ø#7×#;Ò#;Ñ#=Ô#=×#DÒ#DÑ#FÔ#FÐ å"¥3 z¡?¤?Ñ3Ô3ð 	]ð 	]�	Ý,/Ð0DÀYÔ0OÐQhÐirÔQsÑ,tÔ,tð ]ð ]Ñ(�M 5Ø 0°=Ñ @�IØ.°Ñ:�HÝÐ.¨xÔ8Ñ9Ô9¸EÒAÐAÝœØ/°Ô9¸EÀ9Ð;Mñô ð ð õ Ô)Ð*=¸hÔ*GÈ%ÐQZÐI[Ñ\Ô\Ð\Ð\ð]ñ	]ñ%	]õ< �#Ð1Ñ2Ô2Ñ3Ô3ð vð vˆÝ�SÐ!4°XÔ!>Ñ?Ô?Ñ@Ô@ð 	^ð 	^ˆGØ2°8Ô<¸WÔEÑˆE�9ØCLÐW\Ð5]Ð5]Ð Ô)¨'Ñ2Ð2Ý(.Ð/BÀ8Ô/LÐRfÐRfÐptÐ(uÑ(uÔ(uÐ˜HÑ%Ð%àÐr)   ç      è?é   útorch.Tensor | np.ndarrayÚ	thresholdÚfloatÚmin_community_sizeúlist[list[int]]c                óÆ  — t          | t          j        ¦  «        st          j        | ¦  «        } t          j        || j        ¬¦  «        }t          | ¦  «        } g }t          |t          | ¦  «        ¦  «        }t          t          d|z  d¦  «        t          | ¦  «        ¦  «        }t          t          dt          | ¦  «        |¦  «        d| ¬¦  «        D �]o}| |||z   …         | j        z  }| j        j        dv rþ||k    }	|	                     d¦  «        }
|
|k    }|                     ¦   «         sŒ\|
|         }
||         }|
                     ¦   «         }|                     |d	¬
¦  «        \  }}|                     ¦   «         }t#          |
                     ¦   «         |¦  «        D ]Q\  }}|                     |d|…                              ¦   «                              t,          j        ¦  «        ¦  «         ŒR�Œ$|                     |d	¬
¦  «        \  }}t          t          |¦  «        ¦  «        D �]}||         d         |k    rþ||                              |d	¬
¦  «        \  }}|d         |k    rr|t          | ¦  «        k     r_t          d|z  t          | ¦  «        ¦  «        }||                              |d	¬
¦  «        \  }}|d         |k    r|t          | ¦  «        k     °_|                     |||k                                  ¦   «                              ¦   «                              t,          j        ¦  «        ¦  «         �Œ�Œqt1          |d„ d	¬¦  «        }g }t-          j        t          | ¦  «        t4          ¬¦  «        }|D ]>}|||                   }t          |¦  «        |k    r|                     |¦  «         d	||<   Œ?d„ t1          |d„ d	¬¦  «        D ¦   «         }|S )a¼  
    Function for Fast Community Detection.

    Finds in the embeddings all communities, i.e. embeddings that are close (closer than threshold).
    Returns only communities that are larger than min_community_size. The communities are returned
    in decreasing order. The first element in each list is the central point in the community.

    Args:
        embeddings (torch.Tensor or numpy.ndarray): The input embeddings.
        threshold (float): The threshold for determining if two embeddings are close. Defaults to 0.75.
        min_community_size (int): The minimum size of a community to be considered. Defaults to 10.
        batch_size (int): The batch size for computing cosine similarity scores. Defaults to 1024.
        show_progress_bar (bool): Whether to show a progress bar during computation. Defaults to False.

    Returns:
        List[List[int]]: A list of communities, where each community is represented as a list of indices.
    ri   é   é2   r   zFinding clusters)ÚdescÚdisable)ÚcudaÚnpur   T)Úkr.   Nr+   c                ó    — t          | ¦  «        S ©N©r;   r2   s    r'   r4   z%community_detection.<locals>.<lambda>V  s   € ÍÈAÉÌ€ r)   r5   )Údtypec                ó6   — g | ]}|                      ¦   «         ‘ŒS r1   )r@   )rf   Ú	communitys     r'   rh   z'community_detection.<locals>.<listcomp>c  s1   € ð ð ð Ø(ˆ	×ÒÑÔðð ð r)   c                ó    — t          | ¦  «        S rš   r›   r2   s    r'   r4   z%community_detection.<locals>.<lambda>d  s   € ÕUXÐYZÑU[ÔU[€ r)   )rm   r<   r   Útensorrj   r
   r>   r;   Úmaxr   r:   ÚTÚtypeÚsumÚanyr=   r?   rz   r@   rG   ÚnumpyÚastypern   Úuint32r/   Úzerosr   )r&   r�   r�   r   r   Úextracted_communitiesÚsort_max_sizeÚ	start_idxr„   Úthreshold_maskÚrow_wise_countÚlarge_enough_maskr˜   rg   Útop_k_indicesÚcountr€   Útop_k_valuesrS   Útop_val_largeÚtop_idx_largeÚunique_communitiesÚusedrž   Únon_overlapped_communitys                            r'   Úcommunity_detectionr¸     s  € õ0 �j¥%¤,Ñ/Ô/ð .Ý”\ *Ñ-Ô-ˆ
å”˜Y¨zÔ/@ÐAÑAÔA€IÝ% jÑ1Ô1€JàÐõ Ð/µ°Z±´ÑAÔAÐÝ�˜AÐ 2Ñ2°BÑ7Ô7½¸Z¹¼ÑIÔI€MåÝˆa•�Z‘” *Ñ-Ô-Ð4FÐTeÐPeðñ ô ð -ñ -ˆ	ð   	¨I¸
Ñ,BÐ BÔCÀjÄlÑRˆ
ð ÔÔ! _Ð4Ð4à'¨9Ò4ˆNØ+×/Ò/°Ñ2Ô2ˆNð !/Ð2DÒ DÐØ$×(Ò(Ñ*Ô*ð Øà+Ð,=Ô>ˆNØ#Ð$5Ô6ˆJð ×"Ò"Ñ$Ô$ˆAØ)Ÿš°¸D˜ÑAÔAÑˆAˆ}ð *×-Ò-Ñ/Ô/ˆMÝ"% n×&;Ò&;Ñ&=Ô&=¸}Ñ"MÔ"Mð Xð X‘��wØ%×,Ò,¨W°V°e°V¬_×-BÒ-BÑ-DÔ-D×-KÒ-KÍBÌIÑ-VÔ-VÑWÔWÐWÐWñXð )ŸošoÐ0BÈD˜oÑQÔQ‰OˆL˜!õ �3˜|Ñ,Ô,Ñ-Ô-ð ñ �Ø ”? 2Ô&¨)Ò3Ð3à3=¸a´=×3EÒ3EÈÐ_cÐ3EÑ3dÔ3dÑ0�M =ð (¨Ô+¨yÒ8Ð8¸]ÍSÐQ[É_Ì_Ò=\Ð=\Ý(+¨A°Ñ,=½sÀ:¹¼Ñ(OÔ(O˜Ø7AÀ!´}×7IÒ7IÈMÐcgÐ7IÑ7hÔ7hÑ4˜ }ð (¨Ô+¨yÒ8Ð8¸]ÍSÐQ[É_Ì_Ò=\Ð=\ð *×0Ò0Ø% m°yÒ&@ÔA×EÒEÑGÔG×MÒMÑOÔO×VÒVÕWYÔW`ÑaÔañô ð ùñõ #Ð#8Ð>NÐ>NÐX\Ð]Ñ]Ô]Ðð ÐÝŒ8•C˜
‘O”O­4Ð0Ñ0Ô0€Dà*ð 2ð 2ˆ	à#,¨d°9¬oÐ-=Ô#>Ð ÝÐ'Ñ(Ô(Ð,>Ò>Ð>Ø×%Ò%Ð&>Ñ?Ô?Ð?Ø-1ˆDÐ)Ñ*øðð Ý,2Ð3EÐK[ÐK[ÐeiÐ,jÑ,jÔ,jðñ ô Ðð Ðr)   )r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r    r!   )r&   r   r   r   r   r   r   r   r   r   r   r   r    r!   )r    r[   )rb   r   rc   r   r   r   r   r   r   r   r   r   r    r[   )rŠ   ra   r‹   F)r&   rŒ   r�   rŽ   r�   r   r   r   r   r   r    r�   )Ú
__future__r   r{   Úloggingr8   Úcollections.abcr   Útypingr   r¦   rn   r<   r   Útqdm.autonotebookr   Ú
similarityr	   r    r
   Ú	getLoggerÚ__name__ÚloggerÚ0sentence_transformers.sentence_transformer.modelr   r(   r%   r`   r]   r¸   r1   r)   r'   ú<module>rÃ      sÃ  ðØ "Ð "Ð "Ð "Ð "Ð "à €€€Ø €€€Ø €€€Ø $Ð $Ð $Ð $Ð $Ð $Ø  Ð  Ð  Ð  Ð  Ð  à Ð Ð Ð Ø €€€Ø Ð Ð Ð Ð Ð Ø "Ð "Ð "Ð "Ð "Ð "à Ð Ð Ð Ð Ð Ø (Ð (Ð (Ð (Ð (Ð (à	ˆÔ	˜8Ñ	$Ô	$€àð UØTÐTÐTÐTÐTÐTð $ØØ Ø#ØØØ9@Ø#Ø"Øð?ð ?ð ?ð ?ð ?ðH !Ø#ØØØ9@ðFð Fð Fð Fð FðR,ð ,ð ,ð ,ð  Ø#ØØ9@ðXð Xð Xð Xð Xðz Ø ØØ#ðeð eð eð eð eð eð er)   