§
    �Štja+  ã                   ó„   — d dl Z d dlZd dlmZmZmZmZmZmZ d dl	m
Z
 dddœd„Zdd„Zdd„Zdd	„Zdddœd
„Zdddœd„ZdS )é    N)Ú_flatten_dense_tensorsÚ_get_device_indexÚ_handle_complexÚ_reorder_tensors_asÚ_take_tensorsÚ_unflatten_dense_tensors)Únccl)Úoutc                óø   — t          | ¦  «        } |du |du z  st          d|› d|› �¦  «        ‚|�,d„ |D ¦   «         }t          j                             | |¦  «        S t          j                             | |¦  «        S )aï  Broadcasts a tensor to specified GPU devices.

    Args:
        tensor (Tensor): tensor to broadcast. Can be on CPU or GPU.
        devices (Iterable[torch.device, str or int], optional): an iterable of
          GPU devices, among which to broadcast.
        out (Sequence[Tensor], optional, keyword-only): the GPU tensors to
          store output results.

    .. note::
        Exactly one of :attr:`devices` and :attr:`out` must be specified.

    Returns:
        - If :attr:`devices` is specified,
            a tuple containing copies of :attr:`tensor`, placed on
            :attr:`devices`.
        - If :attr:`out` is specified,
            a tuple containing :attr:`out` tensors, each containing a copy of
            :attr:`tensor`.
    NzFExactly one of 'devices' and 'out' must be specified, but got devices=z	 and out=c                 ó,   — g | ]}t          |¦  «        ‘ŒS © ©r   ©Ú.0Úds     úT/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/torch/nn/parallel/comm.pyú
<listcomp>zbroadcast.<locals>.<listcomp>+   ó!   € Ð9Ð9Ð9¨AÕ$ QÑ'Ô'Ð9Ð9Ð9ó    )r   ÚRuntimeErrorÚtorchÚ_CÚ
_broadcastÚ_broadcast_out)ÚtensorÚdevicesr
   s      r   Ú	broadcastr      sš   € õ* ˜VÑ$Ô$€FØ˜ˆ_ ¨ Ñ-ð 
ÝØlÐU\ÐlÐlÐgjÐlÐlñ
ô 
ð 	
ð ÐØ9Ð9°Ð9Ñ9Ô9ˆÝŒx×"Ò" 6¨7Ñ3Ô3Ð3õ Œx×&Ò& v¨sÑ3Ô3Ð3r   é    c                 ót   — d„ |D ¦   «         }d„ | D ¦   «         } t           j                             | ||¦  «        S )a.  Broadcast a sequence of tensors to the specified GPUs.

    Small tensors are first coalesced into a buffer to reduce the number of synchronizations.

    Args:
        tensors (sequence): tensors to broadcast. Must be on the same device,
          either CPU or GPU.
        devices (Iterable[torch.device, str or int]): an iterable of GPU
          devices, among which to broadcast.
        buffer_size (int): maximum size of the buffer used for coalescing

    Returns:
        A tuple containing copies of :attr:`tensor`, placed on :attr:`devices`.
    c                 ó,   — g | ]}t          |¦  «        ‘ŒS r   r   r   s     r   r   z'broadcast_coalesced.<locals>.<listcomp>A   s!   € Ð5Ð5Ð5¨Õ  Ñ#Ô#Ð5Ð5Ð5r   c                 ó,   — g | ]}t          |¦  «        ‘ŒS r   ©r   ©r   Úts     r   r   z'broadcast_coalesced.<locals>.<listcomp>B   ó    € Ð3Ð3Ð3 a�˜qÑ!Ô!Ð3Ð3Ð3r   )r   r   Ú_broadcast_coalesced)Útensorsr   Úbuffer_sizes      r   Úbroadcast_coalescedr)   2   sD   € ð 6Ð5¨WÐ5Ñ5Ô5€GØ3Ð3¨7Ð3Ñ3Ô3€GÝŒ8×(Ò(¨°'¸;ÑGÔGÐGr   c           	      ó$  ‡— t          |d¬¦  «        }| d                              ¦   «         }dŠt          | ¦  «        D ]Â\  }}|j        j        dk    rt          d|› d�¦  «        ‚|                     ¦   «         |k    r|Š|                     ¦   «         |k    rhd                     d	„ |                     ¦   «         D ¦   «         ¦  «        }d                     d
„ |D ¦   «         ¦  «        }t          d|› d|› d|› �¦  «        ‚ŒÃ‰€t          d¦  «        ‚t          | ¦  «        dk    r| d         S t          j        | ¦  «        r2t          j        | ‰         ¦  «        }t          j        | |‰¬¦  «         n�t          j        | ‰         j        j        |¦  «        }ˆfd„t          | ¦  «        D ¦   «         }	| ‰         |	d                              |d¬¦  «        z   }|	dd…         D ],}
|                     |
                     |d¬¦  «        ¦  «         Œ-|S )aë  Sum tensors from multiple GPUs.

    All inputs should have matching shapes, dtype, and layout. The output tensor
    will be of the same shape, dtype, and layout.

    Args:
        inputs (Iterable[Tensor]): an iterable of tensors to add.
        destination (int, optional): a device on which the output will be
            placed (default: current device).

    Returns:
        A tensor containing an elementwise sum of all inputs, placed on the
        :attr:`destination` device.
    T)Úoptionalr   NÚcpuz7reduce_add expects all inputs to be on GPUs, but input z
 is on CPUÚxc              3   ó4   K  — | ]}t          |¦  «        V — Œd S ©N©Ústr©r   r-   s     r   ú	<genexpr>zreduce_add.<locals>.<genexpr>`   s(   è è € Ð6Ð6 a�3˜q™6œ6Ð6Ð6Ð6Ð6Ð6Ð6r   c              3   ó4   K  — | ]}t          |¦  «        V — Œd S r/   r0   r2   s     r   r3   zreduce_add.<locals>.<genexpr>a   s(   è è € Ð;Ð;¨1¥ A¡¤Ð;Ð;Ð;Ð;Ð;Ð;r   zinput z has invalid size: got z, but expected zLreduce_add expects destination to be on the same GPU with one of the tensorsé   )ÚoutputÚrootc                 ó&   •— g | ]\  }}|‰k    ¯|‘ŒS r   r   )r   Úir$   Ú
root_indexs      €r   r   zreduce_add.<locals>.<listcomp>r   s"   ø€ ÐFÐFÐF™˜˜A°a¸:²o°o�1°o°o°or   )ÚdeviceÚnon_blocking)r   ÚsizeÚ	enumerater;   ÚtypeÚAssertionErrorÚ
get_deviceÚjoinÚ
ValueErrorr   Úlenr	   Úis_availabler   Ú
empty_likeÚreduceÚtoÚadd_)ÚinputsÚdestinationÚ
input_sizer9   ÚinpÚgotÚexpectedÚresultÚdestination_deviceÚnonrootÚotherr:   s              @r   Ú
reduce_addrT   F   sP  ø€ õ $ K¸$Ð?Ñ?Ô?€KØ˜”—’Ñ!Ô!€JØ€JÝ˜FÑ#Ô#ð ð ‰ˆˆ3ØŒ:Œ?˜eÒ#Ð#Ý ØWÈ!ÐWÐWÐWñô ð ð �>Š>ÑÔ˜{Ò*Ð*ØˆJØ�8Š8‰:Œ:˜Ò#Ð#Ø—(’(Ð6Ð6¨3¯8ª8©:¬:Ð6Ñ6Ô6Ñ6Ô6ˆCØ—x’xÐ;Ð;°
Ð;Ñ;Ô;Ñ;Ô;ˆHÝØQ˜ÐQÐQ°3ÐQÐQÀxÐQÐQñô ð ð $ð ÐÝØZñ
ô 
ð 	
õ ˆ6�{„{�aÒÐØ�aŒyÐåÔ˜Ñ Ô ð PÝÔ! &¨Ô"4Ñ5Ô5ˆÝŒ�F 6°
Ð;Ñ;Ô;Ð;Ð;å"œ\¨&°Ô*<Ô*CÔ*HÈ+ÑVÔVÐØFÐFÐFÐF¥¨6Ñ!2Ô!2ÐFÑFÔFˆà˜
Ô# g¨a¤j§m¢mØ%°Dð '4ñ '
ô '
ñ 
ˆð ˜Q˜R˜R”[ð 	Pð 	PˆEØ�KŠK˜ŸšÐ(:È˜ÑNÔNÑOÔOÐOÐOØ€Mr   c                 óÚ  ‡— d„ | D ¦   «         }g }g }t          | ddiŽD ]Å}t          d„ |D ¦   «         ¦  «        rAt          ||¦  «        }|                     |¦  «         |                     |d         ¦  «         Œ\t          ||d¬¦  «        D ]5\  }}	|                     |	j        r|	                     ¦   «         n|	¦  «         Œ6|                     |d         d         ¦  «         ŒÆˆfd„|D ¦   «         }
t          |
ddiŽD ]Q}d	„ |D ¦   «         }t          ||¦  «        }t          ||d         ¦  «        D ]}	|                     |	j        ¦  «         ŒŒRt          t          ||¦  «        ¦  «        S )
a\  Sum tensors from multiple GPUs.

    Small tensors are first coalesced into a buffer to reduce the number
    of synchronizations.

    Args:
        inputs (Iterable[Iterable[Tensor]]): iterable of iterables that
            contain tensors from a single device.
        destination (int, optional): a device on which the output will be
            placed (default: current device).
        buffer_size (int): maximum size of the buffer used for coalescing

    Returns:
        A tuple of tensors containing an elementwise sum of each group of
        inputs, placed on the ``destination`` device.
    c                 ó   — g | ]}g ‘ŒS r   r   )r   Ú_s     r   r   z(reduce_add_coalesced.<locals>.<listcomp>�   s   € Ð 4Ð 4Ð 4¨ Ð 4Ð 4Ð 4r   ÚstrictTc              3   ó$   K  — | ]}|j         V — Œd S r/   )Ú	is_sparser#   s     r   r3   z'reduce_add_coalesced.<locals>.<genexpr>”   s$   è è € Ð3Ð3˜qˆqŒ{Ð3Ð3Ð3Ð3Ð3Ð3r   r   )rX   éÿÿÿÿc                 ó0   •— g | ]}t          |‰¦  «        ‘ŒS r   )r   )r   r'   r(   s     €r   r   z(reduce_add_coalesced.<locals>.<listcomp>œ   s#   ø€ ÐMÐMÐM°G�M˜' ;Ñ/Ô/ÐMÐMÐMr   c                 ó,   — g | ]}t          |¦  «        ‘ŒS r   )r   )r   Úchunks     r   r   z(reduce_add_coalesced.<locals>.<listcomp>Ÿ   s.   € ð 
ð 
ð 
Ø.3Õ" 5Ñ)Ô)ð
ð 
ð 
r   )
ÚzipÚallrT   ÚappendrZ   Úto_denser   ÚdataÚtupler   )rJ   rK   r(   Údense_tensorsr6   Ú	ref_orderÚtensor_at_gpusrP   Úcollr$   ÚitrsÚchunksÚflat_tensorsÚflat_results     `           r   Úreduce_add_coalescedrm   |   sÁ  ø€ ð& !5Ð 4¨VÐ 4Ñ 4Ô 4€MØ€FØ€Iå˜vÐ3¨dÐ3Ð3ð 3ð 3ˆÝÐ3Ð3 NÐ3Ñ3Ô3Ñ3Ô3ð 	3Ý °Ñ<Ô<ˆFØ�MŠM˜&Ñ!Ô!Ð!Ø×Ò˜^¨AÔ.Ñ/Ô/Ð/Ð/å˜}¨nÀTÐJÑJÔJð @ð @‘��aØ—’¨A¬KÐ>˜AŸJšJ™LœL˜L¸QÑ?Ô?Ð?Ð?Ø×Ò˜]¨1Ô-¨bÔ1Ñ2Ô2Ð2Ð2ØMÐMÐMÐM¸}ÐMÑMÔM€Då�tÐ) DÐ)Ð)ð 	"ð 	"ˆð
ð 
Ø7=ð
ñ 
ô 
ˆõ ! ¨{Ñ;Ô;ˆÝ)¨+°v¸a´yÑAÔAð 	"ð 	"ˆAð �MŠM˜!œ&Ñ!Ô!Ð!Ð!ð		"õ
 Õ$ V¨YÑ7Ô7Ñ8Ô8Ð8r   c          	      óJ  — t          | ¦  «        } |€<d„ |D ¦   «         }t          t          j                             | ||||¦  «        ¦  «        S |�t          d|› �¦  «        ‚|�t          d|› �¦  «        ‚t          t          j                             | |||¦  «        ¦  «        S )a<  Scatters tensor across multiple GPUs.

    Args:
        tensor (Tensor): tensor to scatter. Can be on CPU or GPU.
        devices (Iterable[torch.device, str or int], optional): an iterable of
          GPU devices, among which to scatter.
        chunk_sizes (Iterable[int], optional): sizes of chunks to be placed on
          each device. It should match :attr:`devices` in length and sums to
          ``tensor.size(dim)``. If not specified, :attr:`tensor` will be divided
          into equal chunks.
        dim (int, optional): A dimension along which to chunk :attr:`tensor`.
          Default: ``0``.
        streams (Iterable[torch.cuda.Stream], optional): an iterable of Streams, among
          which to execute the scatter. If not specified, the default stream will
          be utilized.
        out (Sequence[Tensor], optional, keyword-only): the GPU tensors to
          store output results. Sizes of these tensors must match that of
          :attr:`tensor`, except for :attr:`dim`, where the total size must
          sum to ``tensor.size(dim)``.

    .. note::
        Exactly one of :attr:`devices` and :attr:`out` must be specified. When
        :attr:`out` is specified, :attr:`chunk_sizes` must not be specified and
        will be inferred from sizes of :attr:`out`.

    Returns:
        - If :attr:`devices` is specified,
            a tuple containing chunks of :attr:`tensor`, placed on
            :attr:`devices`.
        - If :attr:`out` is specified,
            a tuple containing :attr:`out` tensors, each containing a chunk of
            :attr:`tensor`.
    Nc                 ó,   — g | ]}t          |¦  «        ‘ŒS r   r   r   s     r   r   zscatter.<locals>.<listcomp>Ð   r   r   zI'devices' must not be specified when 'out' is specified, but got devices=zQ'chunk_sizes' must not be specified when 'out' is specified, but got chunk_sizes=)r   rd   r   r   Ú_scatterr   Ú_scatter_out)r   r   Úchunk_sizesÚdimÚstreamsr
   s         r   Úscatterru   «   sÀ   € õD ˜VÑ$Ô$€FØ
€{à9Ð9°Ð9Ñ9Ô9ˆÝ•U”X×&Ò& v¨w¸ÀSÈ'ÑRÔRÑSÔSÐSàÐÝØeÐ\cÐeÐeñô ð ð Ð"ÝØqÐdoÐqÐqñô ð õ •U”X×*Ò*¨6°3¸¸WÑEÔEÑFÔFÐFr   c                ó2  — d„ | D ¦   «         } |€U|dk    rt          j        dt          d¬¦  «         t          |dd¬¦  «        }t          j                             | ||¦  «        S |�t          d	|› �¦  «        ‚t          j                             | ||¦  «        S )
a²  Gathers tensors from multiple GPU devices.

    Args:
        tensors (Iterable[Tensor]): an iterable of tensors to gather.
          Tensor sizes in all dimensions other than :attr:`dim` have to match.
        dim (int, optional): a dimension along which the tensors will be
          concatenated. Default: ``0``.
        destination (torch.device, str, or int, optional): the output device.
          Can be CPU or CUDA. Default: the current CUDA device.
        out (Tensor, optional, keyword-only): the tensor to store gather result.
          Its sizes must match those of :attr:`tensors`, except for :attr:`dim`,
          where the size must equal ``sum(tensor.size(dim) for tensor in tensors)``.
          Can be on CPU or CUDA.

    .. note::
        :attr:`destination` must not be specified when :attr:`out` is specified.

    Returns:
        - If :attr:`destination` is specified,
            a tensor located on :attr:`destination` device, that is a result of
            concatenating :attr:`tensors` along :attr:`dim`.
        - If :attr:`out` is specified,
            the :attr:`out` tensor, now containing results of concatenating
            :attr:`tensors` along :attr:`dim`.
    c                 ó,   — g | ]}t          |¦  «        ‘ŒS r   r"   r#   s     r   r   zgather.<locals>.<listcomp>ø   r%   r   Nr[   zjUsing -1 to represent CPU tensor is deprecated. Please use a device object or string instead, e.g., "cpu".é   )Ú
stacklevelT)Ú	allow_cpur+   zQ'destination' must not be specified when 'out' is specified, but got destination=)	ÚwarningsÚwarnÚFutureWarningr   r   r   Ú_gatherr   Ú_gather_out)r'   rs   rK   r
   s       r   Úgatherr€   Þ   s¼   € ð4 4Ð3¨7Ð3Ñ3Ô3€GØ
€{Ø˜"ÒÐÝŒMð@åØð	ñ ô ð õ (¨¸tÈdÐSÑSÔSˆÝŒx×Ò ¨¨kÑ:Ô:Ð:àÐ"ÝØqÐdoÐqÐqñô ð õ Œx×#Ò# G¨S°#Ñ6Ô6Ð6r   r/   )r   )Nr   )NNr   N)r   N)r{   r   Útorch._utilsr   r   r   r   r   r   Ú
torch.cudar	   r   r)   rT   rm   ru   r€   r   r   r   ú<module>rƒ      s0  ðà €€€à €€€ðð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð Ð Ð Ð Ð Ð ð4¨4ð 4ð 4ð 4ð 4ð 4ðDHð Hð Hð Hð(3ð 3ð 3ð 3ðl,9ð ,9ð ,9ð ,9ð^0GÐPTð 0Gð 0Gð 0Gð 0Gð 0Gðf*7°Dð *7ð *7ð *7ð *7ð *7ð *7ð *7r   