ó
    !Eñia+  ã                   ó”   • S SK r S SKrS SKJrJrJrJrJrJr  S SK	J
r
  SSS.S jjrSS jrSS jrSS jrSSS.S	 jjrSSS.S
 jjrg)é    N)Ú_flatten_dense_tensorsÚ_get_device_indexÚ_handle_complexÚ_reorder_tensors_asÚ_take_tensorsÚ_unflatten_dense_tensors)Únccl)Úoutc                ó  • [        U 5      n USL USL -  (       d  [        SU SU 35      eUb:  U Vs/ s H  n[        U5      PM     nn[        R                  R                  X5      $ [        R                  R                  X5      $ s  snf )a¯  Broadcasts a tensor to specified GPU devices.

Args:
    tensor (Tensor): tensor to broadcast. Can be on CPU or GPU.
    devices (Iterable[torch.device, str or int], optional): an iterable of
      GPU devices, among which to broadcast.
    out (Sequence[Tensor], optional, keyword-only): the GPU tensors to
      store output results.

.. note::
    Exactly one of :attr:`devices` and :attr:`out` must be specified.

Returns:
    - If :attr:`devices` is specified,
        a tuple containing copies of :attr:`tensor`, placed on
        :attr:`devices`.
    - If :attr:`out` is specified,
        a tuple containing :attr:`out` tensors, each containing a copy of
        :attr:`tensor`.
NzFExactly one of 'devices' and 'out' must be specified, but got devices=z	 and out=)r   ÚRuntimeErrorr   ÚtorchÚ_CÚ
_broadcastÚ_broadcast_out)ÚtensorÚdevicesr
   Úds       ÚS/home/mande/repo/quber/.venv/lib/python3.13/site-packages/torch/nn/parallel/comm.pyÚ	broadcastr      s“   € ô* ˜VÓ$€FØ˜ˆ_ ¨ ×-ÜØTÐU\ÐT]Ð]fÐgjÐfkÐló
ð 	
ð ÑÙ18Ó9²¨AÔ$ QÖ'±ˆÐ9Ü�x‰x×"Ñ" 6Ó3Ð3ô �x‰x×&Ñ& vÓ3Ð3ùò	 :s   ²Bc                 óÂ   • U Vs/ s H  n[        U5      PM     nnU  Vs/ s H  n[        U5      PM     n n[        R                  R	                  XU5      $ s  snf s  snf )a  Broadcast a sequence of tensors to the specified GPUs.

Small tensors are first coalesced into a buffer to reduce the number of synchronizations.

Args:
    tensors (sequence): tensors to broadcast. Must be on the same device,
      either CPU or GPU.
    devices (Iterable[torch.device, str or int]): an iterable of GPU
      devices, among which to broadcast.
    buffer_size (int): maximum size of the buffer used for coalescing

Returns:
    A tuple containing copies of :attr:`tensor`, placed on :attr:`devices`.
)r   r   r   r   Ú_broadcast_coalesced)Útensorsr   Úbuffer_sizer   Úts        r   Úbroadcast_coalescedr   2   sV   € ñ .5Ó5ªW¨Ô  Ö#©W€GÐ5Ù+2Ó3ª7 aŒ˜qÖ!©7€GÐ3Ü�8‰8×(Ñ(¨¸;ÓGÐGùò 6ùÚ3s
   …A Ac           	      óÆ  • [        USS9nU S   R                  5       nSn[        U 5       Hª  u  pEUR                  R                  S:X  a  [        SU S35      eUR                  5       U:X  a  UnUR                  5       U:w  d  MZ  SR                  S	 UR                  5        5       5      nSR                  S
 U 5       5      n[        SU SU SU 35      e   Uc  [        S5      e[        U 5      S:X  a  U S   $ [        R                  " U 5      (       a/  [        R                  " X   5      n[        R                  " XUS9  U$ [        R                  " X   R                  R                  U5      n	[        U 5       VV
s/ s H  u  pJXC:w  d  M  U
PM     nnn
X   US   R!                  U	SS9-   nUSS  H"  nUR#                  UR!                  U	SS95        M$     U$ s  sn
nf )aÃ  Sum tensors from multiple GPUs.

All inputs should have matching shapes, dtype, and layout. The output tensor
will be of the same shape, dtype, and layout.

Args:
    inputs (Iterable[Tensor]): an iterable of tensors to add.
    destination (int, optional): a device on which the output will be
        placed (default: current device).

Returns:
    A tensor containing an elementwise sum of all inputs, placed on the
    :attr:`destination` device.
T)Úoptionalr   NÚcpuz7reduce_add expects all inputs to be on GPUs, but input z
 is on CPUÚxc              3   ó8   #   • U  H  n[        U5      v •  M     g 7f©N©Ústr©Ú.0r   s     r   Ú	<genexpr>Úreduce_add.<locals>.<genexpr>`   s   é € Ð6ª: aœ3˜qŸ6˜6ª:ùó   ‚c              3   ó8   #   • U  H  n[        U5      v •  M     g 7fr!   r"   r$   s     r   r&   r'   a   s   é € Ð;²
¨1¤ A§ ²
ùr(   zinput z has invalid size: got z, but expected zLreduce_add expects destination to be on the same GPU with one of the tensorsé   )ÚoutputÚroot)ÚdeviceÚnon_blocking)r   ÚsizeÚ	enumerater-   ÚtypeÚAssertionErrorÚ
get_deviceÚjoinÚ
ValueErrorr   Úlenr	   Úis_availabler   Ú
empty_likeÚreduceÚtoÚadd_)ÚinputsÚdestinationÚ
input_sizeÚ
root_indexÚiÚinpÚgotÚexpectedÚresultÚdestination_devicer   ÚnonrootÚothers                r   Ú
reduce_addrH   F   sÞ  € ô $ K¸$Ñ?€KØ˜‘—‘Ó!€JØ€JÜ˜FÖ#‰ˆØ�:‰:�?‰?˜eÓ#Ü ØIÈ!ÈÈJÐWóð ð �>‰>Ó˜{Ó*ØˆJØ�8‰8‹:˜Õ#Ø—(‘(Ñ6¨3¯8©8¬:Ó6Ó6ˆCØ—x‘xÑ;±
Ó;Ó;ˆHÜØ˜˜Ð2°3°%°ÀxÀjÐQóð ñ $ð ÑÜØZó
ð 	
ô ˆ6ƒ{�aÓØ�a‰yÐä×Ò˜× Ñ Ü×!Ò! &Ñ"4Ó5ˆÜ�Š�F°
Ò;ð €Mô #Ÿ\š\¨&Ñ*<×*CÑ*C×*HÑ*HÈ+ÓVÐÜ!*¨6Ô!2ÔFÒ!2™˜°a±o—1Ñ!2ˆÑFàÑ# g¨a¡j§m¡mØ%°Dð '4ð '
ñ 
ˆð ˜Q˜R“[ˆEØ�K‰K˜Ÿ™Ð(:È˜ÐNÖOñ !à€Mùó Gs   Æ GÆGc                 óê  • U  Vs/ s H  n/ PM     nn/ n/ n[        U SS06 H¨  n[        S U 5       5      (       a2  [        Xq5      nUR                  U5        UR                  US   5        ML  [        XGSS9 H7  u  pšU	R                  U
R                  (       a  U
R                  5       OU
5        M9     UR                  US   S   5        Mª     U Vs/ s H  n[        X²5      PM     nn[        USS06 HZ  nU Vs/ s H  n[        U5      PM     nn[        Xñ5      n[        UUS   5       H  n
UR                  U
R                  5        M      M\     [        [        XV5      5      $ s  snf s  snf s  snf )a,  Sum tensors from multiple GPUs.

Small tensors are first coalesced into a buffer to reduce the number
of synchronizations.

Args:
    inputs (Iterable[Iterable[Tensor]]): iterable of iterables that
        contain tensors from a single device.
    destination (int, optional): a device on which the output will be
        placed (default: current device).
    buffer_size (int): maximum size of the buffer used for coalescing

Returns:
    A tuple of tensors containing an elementwise sum of each group of
    inputs, placed on the ``destination`` device.
ÚstrictTc              3   ó8   #   • U  H  oR                   v •  M     g 7fr!   )Ú	is_sparse)r%   r   s     r   r&   Ú'reduce_add_coalesced.<locals>.<genexpr>”   s   é € Ð3¢N˜q�{Ž{¢Nùr(   r   )rJ   éÿÿÿÿ)ÚzipÚallrH   ÚappendrL   Úto_denser   r   r   ÚdataÚtupler   )r<   r=   r   Ú_Údense_tensorsr+   Ú	ref_orderÚtensor_at_gpusrD   Úcollr   r   ÚitrsÚchunksÚchunkÚflat_tensorsÚflat_results                    r   Úreduce_add_coalescedr_   |   s`  € ñ& .4Ó 4ªV¨£©V€MÐ 4Ø€FØ€Iä˜vÐ3¨dÔ3ˆÜÑ3¡NÓ3×3Ñ3Ü Ó<ˆFØ�M‰M˜&Ô!Ø×Ñ˜^¨AÑ.Ö/ä˜}ÀTÔJ‘�Ø—‘¨A¯K¯K˜AŸJ™JœL¸QÖ?ñ Kà×Ñ˜]¨1Ñ-¨bÑ1Ö2ñ 4ñ @MÓMº}°GŒM˜'Ö/¹}€DÐMä�tÐ) DÔ)ˆá7=ó
Ú7=¨eÔ" 5Ö)±vð 	ð 
ô ! Ó;ˆÜ)¨+°v¸a±yÖAˆAð �M‰M˜!Ÿ&™&Ö!ó	 Bñ *ô Ô$ VÓ7Ó8Ð8ùò3 !5ùò Nùò
s   …E&ÃE+Ã;E0c          	      óH  • [        U 5      n UcE  U Vs/ s H  n[        U5      PM     nn[        [        R                  R                  XX#U5      5      $ Ub  [        SU 35      eUb  [        SU 35      e[        [        R                  R                  XX45      5      $ s  snf )aÈ  Scatters tensor across multiple GPUs.

Args:
    tensor (Tensor): tensor to scatter. Can be on CPU or GPU.
    devices (Iterable[torch.device, str or int], optional): an iterable of
      GPU devices, among which to scatter.
    chunk_sizes (Iterable[int], optional): sizes of chunks to be placed on
      each device. It should match :attr:`devices` in length and sums to
      ``tensor.size(dim)``. If not specified, :attr:`tensor` will be divided
      into equal chunks.
    dim (int, optional): A dimension along which to chunk :attr:`tensor`.
      Default: ``0``.
    streams (Iterable[torch.cuda.Stream], optional): an iterable of Streams, among
      which to execute the scatter. If not specified, the default stream will
      be utilized.
    out (Sequence[Tensor], optional, keyword-only): the GPU tensors to
      store output results. Sizes of these tensors must match that of
      :attr:`tensor`, except for :attr:`dim`, where the total size must
      sum to ``tensor.size(dim)``.

.. note::
    Exactly one of :attr:`devices` and :attr:`out` must be specified. When
    :attr:`out` is specified, :attr:`chunk_sizes` must not be specified and
    will be inferred from sizes of :attr:`out`.

Returns:
    - If :attr:`devices` is specified,
        a tuple containing chunks of :attr:`tensor`, placed on
        :attr:`devices`.
    - If :attr:`out` is specified,
        a tuple containing :attr:`out` tensors, each containing a chunk of
        :attr:`tensor`.
zI'devices' must not be specified when 'out' is specified, but got devices=zQ'chunk_sizes' must not be specified when 'out' is specified, but got chunk_sizes=)r   r   rT   r   r   Ú_scatterr   Ú_scatter_out)r   r   Úchunk_sizesÚdimÚstreamsr
   r   s          r   Úscatterrf   «   s¯   € ôD ˜VÓ$€FØ
�{á18Ó9²¨AÔ$ QÖ'±ˆÐ9Ü”U—X‘X×&Ñ& v¸È'ÓRÓSÐSàÑÜØ[Ð\cÐ[dÐeóð ð Ñ"ÜØcÐdoÐcpÐqóð ô ”U—X‘X×*Ñ*¨6¸ÓEÓFÐFùò :s   “Bc                ó@  • U  Vs/ s H  n[        U5      PM     n nUcK  US:X  a  [        R                  " S[        SS9  [	        USSS9n[
        R                  R                  XU5      $ Ub  [        SU 35      e[
        R                  R                  XU5      $ s  snf )a^  Gathers tensors from multiple GPU devices.

Args:
    tensors (Iterable[Tensor]): an iterable of tensors to gather.
      Tensor sizes in all dimensions other than :attr:`dim` have to match.
    dim (int, optional): a dimension along which the tensors will be
      concatenated. Default: ``0``.
    destination (torch.device, str, or int, optional): the output device.
      Can be CPU or CUDA. Default: the current CUDA device.
    out (Tensor, optional, keyword-only): the tensor to store gather result.
      Its sizes must match those of :attr:`tensors`, except for :attr:`dim`,
      where the size must equal ``sum(tensor.size(dim) for tensor in tensors)``.
      Can be on CPU or CUDA.

.. note::
    :attr:`destination` must not be specified when :attr:`out` is specified.

Returns:
    - If :attr:`destination` is specified,
        a tensor located on :attr:`destination` device, that is a result of
        concatenating :attr:`tensors` along :attr:`dim`.
    - If :attr:`out` is specified,
        the :attr:`out` tensor, now containing results of concatenating
        :attr:`tensors` along :attr:`dim`.
rN   zjUsing -1 to represent CPU tensor is deprecated. Please use a device object or string instead, e.g., "cpu".é   )Ú
stacklevelT)Ú	allow_cpur   zQ'destination' must not be specified when 'out' is specified, but got destination=)
r   ÚwarningsÚwarnÚFutureWarningr   r   r   Ú_gatherr   Ú_gather_out)r   rd   r=   r
   r   s        r   Úgatherrp   Þ   s¦   € ñ4 ,3Ó3ª7 aŒ˜qÖ!©7€GÐ3Ø
�{Ø˜"ÓÜ�MŠMð@äØò	ô (¨¸tÈdÑSˆÜ�x‰x×Ñ ¨kÓ:Ð:àÑ"ÜØcÐdoÐcpÐqóð ô �x‰x×#Ñ# G°#Ó6Ð6ùò! 4s   …Br!   )é    )Nrq   )NNr   N)r   N)rk   r   Útorch._utilsr   r   r   r   r   r   Ú
torch.cudar	   r   r   rH   r_   rf   rp   © ó    r   Ú<module>rv      sU   ðã ã ÷÷ õ ð4¨4ö 4ôDHô(3ôl,9ð^0GÐPTö 0Gðf*7°D÷ *7ru   