Ë
    3êñia+  ã                   ó‚   — d dl Z d dlZd dlmZmZmZmZmZmZ d dl	m
Z
 dddœd„Zdd„Zdd„Zdd„Zdddœd	„Zdddœd
„Zy)é    N)Ú_flatten_dense_tensorsÚ_get_device_indexÚ_handle_complexÚ_reorder_tensors_asÚ_take_tensorsÚ_unflatten_dense_tensors)Únccl)Úoutc                ó
  — t        | «      } |du |du z  st        d|› d|› �«      ‚|�8|D �cg c]  }t        |«      ‘Œ }}t        j                  j                  | |«      S t        j                  j                  | |«      S c c}w )aï  Broadcasts a tensor to specified GPU devices.

    Args:
        tensor (Tensor): tensor to broadcast. Can be on CPU or GPU.
        devices (Iterable[torch.device, str or int], optional): an iterable of
          GPU devices, among which to broadcast.
        out (Sequence[Tensor], optional, keyword-only): the GPU tensors to
          store output results.

    .. note::
        Exactly one of :attr:`devices` and :attr:`out` must be specified.

    Returns:
        - If :attr:`devices` is specified,
            a tuple containing copies of :attr:`tensor`, placed on
            :attr:`devices`.
        - If :attr:`out` is specified,
            a tuple containing :attr:`out` tensors, each containing a copy of
            :attr:`tensor`.
    NzFExactly one of 'devices' and 'out' must be specified, but got devices=z	 and out=)r   ÚRuntimeErrorr   ÚtorchÚ_CÚ
_broadcastÚ_broadcast_out)ÚtensorÚdevicesr
   Úds       úX/var/www/pod-logistic/pod-ai/venv/lib/python3.12/site-packages/torch/nn/parallel/comm.pyÚ	broadcastr      s–   € ô* ˜VÓ$€FØ˜ˆ_ ¨ Ò-ÜØTÐU\ÐT]Ð]fÐgjÐfkÐló
ð 	
ð ÐØ18Ö9¨AÔ$ QÕ'Ð9ˆÐ9Ü�x‰x×"Ñ" 6¨7Ó3Ð3ô �x‰x×&Ñ& v¨sÓ3Ð3ùò	 :s   ¬B c                 ó¸   — |D �cg c]  }t        |«      ‘Œ }}| D �cg c]  }t        |«      ‘Œ } }t        j                  j	                  | ||«      S c c}w c c}w )a.  Broadcast a sequence of tensors to the specified GPUs.

    Small tensors are first coalesced into a buffer to reduce the number of synchronizations.

    Args:
        tensors (sequence): tensors to broadcast. Must be on the same device,
          either CPU or GPU.
        devices (Iterable[torch.device, str or int]): an iterable of GPU
          devices, among which to broadcast.
        buffer_size (int): maximum size of the buffer used for coalescing

    Returns:
        A tuple containing copies of :attr:`tensor`, placed on :attr:`devices`.
    )r   r   r   r   Ú_broadcast_coalesced)Útensorsr   Úbuffer_sizer   Úts        r   Úbroadcast_coalescedr   2   sV   € ð .5Ö5¨Ô  Õ#Ð5€GÐ5Ø+2Ö3 aŒ˜qÕ!Ð3€GÐ3Ü�8‰8×(Ñ(¨°'¸;ÓGÐGùò 6ùÚ3s
   …A�Ac           	      ó¾  — t        |d¬«      }| d   j                  «       }d}t        | «      D ]§  \  }}|j                  j                  dk(  rt        d|› d�«      ‚|j                  «       |k(  r|}|j                  «       |k7  sŒWdj                  d	„ |j                  «       D «       «      }dj                  d
„ |D «       «      }t        d|› d|› d|› �«      ‚ |€t        d«      ‚t        | «      dk(  r| d   S t        j                  | «      r2t        j                  | |   «      }t        j                  | ||¬«       |S t        j                  | |   j                  j                  |«      }	t        | «      D ��
cg c]  \  }}
||k7  sŒ|
‘Œ }}}
| |   |d   j!                  |	d¬«      z   }|dd D ]$  }|j#                  |j!                  |	d¬«      «       Œ& |S c c}
}w )aë  Sum tensors from multiple GPUs.

    All inputs should have matching shapes, dtype, and layout. The output tensor
    will be of the same shape, dtype, and layout.

    Args:
        inputs (Iterable[Tensor]): an iterable of tensors to add.
        destination (int, optional): a device on which the output will be
            placed (default: current device).

    Returns:
        A tensor containing an elementwise sum of all inputs, placed on the
        :attr:`destination` device.
    T)Úoptionalr   NÚcpuz7reduce_add expects all inputs to be on GPUs, but input z
 is on CPUÚxc              3   ó2   K  — | ]  }t        |«      –— Œ y ­w©N©Ústr©Ú.0r   s     r   ú	<genexpr>zreduce_add.<locals>.<genexpr>`   s   è ø€ Ò6 aœ3˜qŸ6Ñ6ùó   ‚c              3   ó2   K  — | ]  }t        |«      –— Œ y ­wr!   r"   r$   s     r   r&   zreduce_add.<locals>.<genexpr>a   s   è ø€ Ò;¨1¤ A§Ñ;ùr'   zinput z has invalid size: got z, but expected zLreduce_add expects destination to be on the same GPU with one of the tensorsé   )ÚoutputÚroot)ÚdeviceÚnon_blocking)r   ÚsizeÚ	enumerater,   ÚtypeÚAssertionErrorÚ
get_deviceÚjoinÚ
ValueErrorr   Úlenr	   Úis_availabler   Ú
empty_likeÚreduceÚtoÚadd_)ÚinputsÚdestinationÚ
input_sizeÚ
root_indexÚiÚinpÚgotÚexpectedÚresultÚdestination_devicer   ÚnonrootÚothers                r   Ú
reduce_addrG   F   sò  € ô $ K¸$Ô?€KØ˜‘—‘Ó!€JØ€JÜ˜FÓ#ò ‰ˆˆ3Ø�:‰:�?‰?˜eÒ#Ü ØIÈ!ÈÈJÐWóð ð �>‰>Ó˜{Ò*ØˆJØ�8‰8‹:˜Ó#Ø—(‘(Ñ6¨3¯8©8«:Ô6Ó6ˆCØ—x‘xÑ;°
Ô;Ó;ˆHÜØ˜˜Ð2°3°%°ÀxÀjÐQóð ðð ÐÜØZó
ð 	
ô ˆ6ƒ{�aÒØ�a‰yÐä×Ñ˜Ô Ü×!Ñ! &¨Ñ"4Ó5ˆÜ�‰�F 6°
Õ;ð €Mô #Ÿ\™\¨&°Ñ*<×*CÑ*C×*HÑ*HÈ+ÓVÐÜ!*¨6Ó!2×F™˜˜A°a¸:³o’1ÐFˆÑFà˜
Ñ# g¨a¡j§m¡mØ%°Dð '4ó '
ñ 
ˆð ˜Q˜R�[ò 	PˆEØ�K‰K˜Ÿ™Ð(:È˜ÓNÕOð	Pà€Mùó Gs   Å:GÆGc                 óÄ  — | D �cg c]  }g ‘Œ }}g }g }t        | ddiŽD ]   }t        d„ |D «       «      r2t        ||«      }|j                  |«       |j                  |d   «       ŒGt        ||d¬«      D ]2  \  }	}
|	j                  |
j                  r|
j                  «       n|
«       Œ4 |j                  |d   d   «       Œ¢ |D �cg c]  }t        ||«      ‘Œ }}t        |ddiŽD ]U  }|D �cg c]  }t        |«      ‘Œ }}t        ||«      }t        ||d   «      D ]  }
|j                  |
j                  «       Œ ŒW t        t        ||«      «      S c c}w c c}w c c}w )a\  Sum tensors from multiple GPUs.

    Small tensors are first coalesced into a buffer to reduce the number
    of synchronizations.

    Args:
        inputs (Iterable[Iterable[Tensor]]): iterable of iterables that
            contain tensors from a single device.
        destination (int, optional): a device on which the output will be
            placed (default: current device).
        buffer_size (int): maximum size of the buffer used for coalescing

    Returns:
        A tuple of tensors containing an elementwise sum of each group of
        inputs, placed on the ``destination`` device.
    ÚstrictTc              3   ó4   K  — | ]  }|j                   –— Œ y ­wr!   )Ú	is_sparse)r%   r   s     r   r&   z'reduce_add_coalesced.<locals>.<genexpr>”   s   è ø€ Ò3˜qˆq�{�{Ñ3ùs   ‚r   )rI   éÿÿÿÿ)ÚzipÚallrG   ÚappendrK   Úto_denser   r   r   ÚdataÚtupler   )r;   r<   r   Ú_Údense_tensorsr*   Ú	ref_orderÚtensor_at_gpusrC   Úcollr   r   ÚitrsÚchunksÚchunkÚflat_tensorsÚflat_results                    r   Úreduce_add_coalescedr]   |   sy  € ð& .4Ö 4¨¢Ð 4€MÐ 4Ø€FØ€Iä˜vÐ3¨dÑ3ò 3ˆÜÑ3 NÔ3Ô3Ü °Ó<ˆFØ�M‰M˜&Ô!Ø×Ñ˜^¨AÑ.Õ/ä˜}¨nÀTÔJò @‘��aØ—‘¨A¯KªK˜AŸJ™JœL¸QÕ?ð@à×Ñ˜]¨1Ñ-¨bÑ1Õ2ð3ð @MÖM°GŒM˜' ;Õ/ÐM€DÐMä�tÐ) DÑ)ò 	"ˆà7=ö
Ø.3Ô" 5Õ)ð
ˆð 
ô ! ¨{Ó;ˆÜ)¨+°v¸a±yÓAò 	"ˆAð �M‰M˜!Ÿ&™&Õ!ñ		"ð	"ô Ô$ V¨YÓ7Ó8Ð8ùò3 !5ùò Nùò
s   …	EÃEÃ-Ec          	      óD  — t        | «      } |€D|D �cg c]  }t        |«      ‘Œ }}t        t        j                  j                  | ||||«      «      S |�t        d|› �«      ‚|�t        d|› �«      ‚t        t        j                  j                  | |||«      «      S c c}w )a<  Scatters tensor across multiple GPUs.

    Args:
        tensor (Tensor): tensor to scatter. Can be on CPU or GPU.
        devices (Iterable[torch.device, str or int], optional): an iterable of
          GPU devices, among which to scatter.
        chunk_sizes (Iterable[int], optional): sizes of chunks to be placed on
          each device. It should match :attr:`devices` in length and sums to
          ``tensor.size(dim)``. If not specified, :attr:`tensor` will be divided
          into equal chunks.
        dim (int, optional): A dimension along which to chunk :attr:`tensor`.
          Default: ``0``.
        streams (Iterable[torch.cuda.Stream], optional): an iterable of Streams, among
          which to execute the scatter. If not specified, the default stream will
          be utilized.
        out (Sequence[Tensor], optional, keyword-only): the GPU tensors to
          store output results. Sizes of these tensors must match that of
          :attr:`tensor`, except for :attr:`dim`, where the total size must
          sum to ``tensor.size(dim)``.

    .. note::
        Exactly one of :attr:`devices` and :attr:`out` must be specified. When
        :attr:`out` is specified, :attr:`chunk_sizes` must not be specified and
        will be inferred from sizes of :attr:`out`.

    Returns:
        - If :attr:`devices` is specified,
            a tuple containing chunks of :attr:`tensor`, placed on
            :attr:`devices`.
        - If :attr:`out` is specified,
            a tuple containing :attr:`out` tensors, each containing a chunk of
            :attr:`tensor`.
    zI'devices' must not be specified when 'out' is specified, but got devices=zQ'chunk_sizes' must not be specified when 'out' is specified, but got chunk_sizes=)r   r   rR   r   r   Ú_scatterr   Ú_scatter_out)r   r   Úchunk_sizesÚdimÚstreamsr
   r   s          r   Úscatterrd   «   s¶   € ôD ˜VÓ$€FØ
€{à18Ö9¨AÔ$ QÕ'Ð9ˆÐ9Ü”U—X‘X×&Ñ& v¨w¸ÀSÈ'ÓRÓSÐSàÐÜØ[Ð\cÐ[dÐeóð ð Ð"ÜØcÐdoÐcpÐqóð ô ”U—X‘X×*Ñ*¨6°3¸¸WÓEÓFÐFùò :s   ’Bc                óB  — | D �cg c]  }t        |«      ‘Œ } }|€P|dk(  rt        j                  dt        d¬«       t	        |dd¬«      }t
        j                  j                  | ||«      S |�t        d|› �«      ‚t
        j                  j                  | ||«      S c c}w )a²  Gathers tensors from multiple GPU devices.

    Args:
        tensors (Iterable[Tensor]): an iterable of tensors to gather.
          Tensor sizes in all dimensions other than :attr:`dim` have to match.
        dim (int, optional): a dimension along which the tensors will be
          concatenated. Default: ``0``.
        destination (torch.device, str, or int, optional): the output device.
          Can be CPU or CUDA. Default: the current CUDA device.
        out (Tensor, optional, keyword-only): the tensor to store gather result.
          Its sizes must match those of :attr:`tensors`, except for :attr:`dim`,
          where the size must equal ``sum(tensor.size(dim) for tensor in tensors)``.
          Can be on CPU or CUDA.

    .. note::
        :attr:`destination` must not be specified when :attr:`out` is specified.

    Returns:
        - If :attr:`destination` is specified,
            a tensor located on :attr:`destination` device, that is a result of
            concatenating :attr:`tensors` along :attr:`dim`.
        - If :attr:`out` is specified,
            the :attr:`out` tensor, now containing results of concatenating
            :attr:`tensors` along :attr:`dim`.
    rL   zjUsing -1 to represent CPU tensor is deprecated. Please use a device object or string instead, e.g., "cpu".é   )Ú
stacklevelT)Ú	allow_cpur   zQ'destination' must not be specified when 'out' is specified, but got destination=)
r   ÚwarningsÚwarnÚFutureWarningr   r   r   Ú_gatherr   Ú_gather_out)r   rb   r<   r
   r   s        r   Úgatherrn   Þ   s©   € ð4 ,3Ö3 aŒ˜qÕ!Ð3€GÐ3Ø
€{Ø˜"ÒÜ�M‰Mð@äØõ	ô (¨¸tÈdÔSˆÜ�x‰x×Ñ ¨¨kÓ:Ð:àÐ"ÜØcÐdoÐcpÐqóð ô �x‰x×#Ñ# G¨S°#Ó6Ð6ùò! 4s   …Br!   )é    )Nro   )NNr   N)r   N)ri   r   Útorch._utilsr   r   r   r   r   r   Ú
torch.cudar	   r   r   rG   r]   rd   rn   © ó    r   ú<module>rt      sU   ðã ã ÷÷ õ ð4¨4ô 4óDHó(3ól,9ð^0GÐPTô 0Gðf*7°Dõ *7rs   