Ë
    èÿæi�  ã                   óÔ  — d dl Zd dlZd dlZd dlZd dlZd dlmZmZm	Z	m
Z
mZmZ d dlmZmZ d dlmZ d dlmZ d dlmZmZ d dlmZmZ d dlmZ d d	lmZ d d
lmZm Z  d dl!m"Z" d dl#m$Z$ d dl%m&Z& d dl'm(Z(m)Z) d dl*m
Z+ d dl,m-Z- d dl,m.Z. d dl/m0Z0 g d¢Z1 G d„ dejd                  «      Z3 G d„ de4«      Z5 G d„ d«      Z6 G d„ de«      Z7 G d„ de«      Z8 G d„ deejd                  «      Z9y) é    N)ÚconfigÚ	serializeÚsigutilsÚtypesÚtypingÚutils)ÚCacheÚ	CacheImpl)Úglobal_compiler_lock)Ú
Dispatcher)ÚNumbaPerformanceWarningÚNumbaValueError)ÚPurposeÚtypeof)Úget_current_device)Úwrap_arg)Úcompile_cudaÚCUDACompiler)Údriver)Úget_context)Úcuda_target)Úmissing_launch_config_msgÚnormalize_kernel_dimensions)r   ©Úcuda)Ú_dispatcher)Úwarn)ÚhsinÚhcosÚhlogÚhlog10Úhlog2ÚhexpÚhexp10Úhexp2ÚhsqrtÚhrsqrtÚhfloorÚhceilÚhrcpÚhrintÚhtruncÚhdivc                   ó   ‡ — e Zd ZdZe	 	 	 dˆ fd„	«       Zed„ «       Zed„ «       Zd„ Z	ed„ «       Z
ed„ «       Zeˆ fd„«       Zd	„ Zd
„ Zed„ «       Zed„ «       Zed„ «       Zed„ «       Zed„ «       Zd„ Zd„ Zd„ Zd„ Zdd„Zdd„Zdd„Zd„ Zˆ xZS )Ú_Kernelz„
    CUDA Kernel specialized for a given set of argument types. When called, this
    object launches the kernel on the device.
    c                 ó  •— |rt        d«      ‚t        ‰| �	  «        d| _        d | _        || _        || _        || _        || _        |xs g | _	        ||
rdnddœ}t        «       j                  }t        | j
                  t        j                  | j                  | j                  |||||¬«	      }|j                  }| j
                  j                   }|j"                  }|j$                  }|j'                  |j(                  |j*                  ||||||	«      \  }}|sg }d|j-                  «       v | _        | j.                  rd|_        t2        D �cg c]  }d	|› �|j-                  «       v r|‘Œ }}|rqt4        j6                  j9                  t4        j6                  j;                  t<        «      «      }t4        j6                  j?                  |d
«      }|jA                  |«       |D ]  }|jC                  |«       Œ |jD                  | _#        |jH                  | _$        |jJ                  | _&        || _'        |jP                  | _(        || _        |j*                  | _        |jR                  | _)        g | _*        g | _+        g | _,        y c c}w )Nz,Cannot compile a device function as a kernelFé   r   )ÚfastmathÚopt©ÚdebugÚlineinfoÚinliner2   Únvvm_optionsÚccÚcudaCGGetIntrinsicHandleTÚ__numba_wrapper_zcpp_function_wrappers.cu)-ÚRuntimeErrorÚsuperÚ__init__Ú
objectmodeÚentry_pointÚpy_funcÚargtypesr5   r6   Ú
extensionsr   Úcompute_capabilityr   r   ÚvoidÚtarget_contextÚ__code__Úco_filenameÚco_firstlinenoÚprepare_cuda_kernelÚlibraryÚfndescÚget_asm_strÚcooperativeÚneeds_cudadevrtÚcuda_fp16_math_funcsÚosÚpathÚdirnameÚabspathÚ__file__ÚjoinÚappendÚadd_linking_fileÚnameÚ
entry_nameÚ	signatureÚtype_annotationÚ_type_annotationÚ_codelibraryÚcall_helperÚenvironmentÚ_referenced_environmentsÚliftedÚreload_init)ÚselfrA   rB   Úlinkr5   r6   r7   r2   rC   Úmax_registersr3   Údevicer8   r9   ÚcresÚtgt_ctxÚcodeÚfilenameÚlinenumÚlibÚkernelÚfnÚresÚbasedirÚfunctions_cu_pathÚfilepathÚ	__class__s                             €új/Volumes/fast/ai/experiments/voice-extract-mac/.venv/lib/python3.12/site-packages/numba/cuda/dispatcher.pyr>   z_Kernel.__init__.   sU  ø€ ñ
 ÜÐMÓNÐNä‰ÑÔð  ˆŒð  ˆÔàˆŒØ ˆŒØˆŒ
Ø ˆŒØ$Ò*¨ˆŒð !Ù‘1 ñ
ˆô
  Ó!×4Ñ4ˆÜ˜DŸL™L¬%¯*©*°d·m±mØ"&§*¡*Ø%-Ø#)Ø%-Ø)5Ø!ô#ˆð ×%Ñ%ˆØ�|‰|×$Ñ$ˆØ×#Ñ#ˆØ×%Ñ%ˆØ×1Ñ1°$·,±,ÀÇÁØ27¸À<Ø2:¸GØ2?óA‰ˆˆVñ
 ØˆDð 6¸¿¹Ó9JÐJˆÔà×ÒØ"&ˆCÔå0ó BÑ0�bØ% b TÐ*¨c¯o©oÓ.?Ñ?ò Ð0ˆð Bñ ä—g‘g—o‘o¤b§g¡g§o¡o´hÓ&?Ó@ˆGÜ "§¡§¡¨WØ-Gó!IÐà�K‰KÐ)Ô*ãˆHØ× Ñ  Õ*ð ð !Ÿ+™+ˆŒØŸ™ˆŒØ $× 4Ñ 4ˆÔØˆÔØ×+Ñ+ˆÔð &ˆÔØ—k‘kˆŒØ×+Ñ+ˆÔØ(*ˆÔ%ØˆŒØˆÕùò;Bs   ÅJc                 ó   — | j                   S ©N)r^   ©rd   s    ru   rK   z_Kernel.library‹   s   € à× Ñ Ð ó    c                 ó   — | j                   S rw   )r]   rx   s    ru   r\   z_Kernel.type_annotation�   s   € à×$Ñ$Ð$ry   c                 ó   — | j                   S rw   )ra   rx   s    ru   Ú_find_referenced_environmentsz%_Kernel._find_referenced_environments“   s   € Ø×,Ñ,Ð,ry   c                 ó6   — | j                   j                  «       S rw   )rF   Úcodegenrx   s    ru   r~   z_Kernel.codegen–   s   € à×"Ñ"×*Ñ*Ó,Ð,ry   c                 ó@   — t        | j                  j                  «      S rw   )Útupler[   Úargsrx   s    ru   Úargument_typesz_Kernel.argument_typesš   s   € ä�T—^‘^×(Ñ(Ó)Ð)ry   c	                 óÒ   •— | j                  | «      }	t        | |	�  «        d|	_        ||	_        ||	_        ||	_        d|	_        ||	_        ||	_	        ||	_
        ||	_        ||	_        |	S )ú&
        Rebuild an instance.
        N)Ú__new__r=   r>   r@   rN   rZ   r[   r]   r^   r5   r6   r_   rC   )ÚclsrN   rY   r[   Úcodelibraryr5   r6   r_   rC   Úinstancert   s             €ru   Ú_rebuildz_Kernel._rebuildž   st   ø€ ð —;‘;˜sÓ#ˆäˆc�8Ñ%Ô'à#ˆÔØ*ˆÔØ"ˆÔØ&ˆÔØ$(ˆÔ!Ø +ˆÔØˆŒØ$ˆÔØ*ˆÔØ(ˆÔØˆry   c           
      óÈ   — t        | j                  | j                  | j                  | j                  | j
                  | j                  | j                  | j                  ¬«      S )a  
        Reduce the instance for serialization.
        Compiled definitions are serialized in PTX form.
        Type annotation are discarded.
        Thread, block and shared memory configuration are serialized.
        Stream information is discarded.
        )rN   rY   r[   r‡   r5   r6   r_   rC   )	ÚdictrN   rZ   r[   r^   r5   r6   r_   rC   rx   s    ru   Ú_reduce_statesz_Kernel._reduce_states´   sL   € ô  × 0Ñ 0°t·±Ø"Ÿn™n¸$×:KÑ:KØŸ*™*¨t¯}©}Ø $× 0Ñ 0¸T¿_¹_ôNð 	Nry   c                 ó8   — | j                   j                  «        y)z7
        Force binding to current CUDA context
        N)r^   Ú
get_cufuncrx   s    ru   Úbindz_Kernel.bindÁ   s   € ð 	×Ñ×$Ñ$Õ&ry   c                 ó^   — | j                   j                  «       j                  j                  S )zN
        The number of registers used by each thread for this kernel.
        )r^   rŽ   ÚattrsÚregsrx   s    ru   Úregs_per_threadz_Kernel.regs_per_threadÇ   s%   € ð
 × Ñ ×+Ñ+Ó-×3Ñ3×8Ñ8Ð8ry   c                 ó^   — | j                   j                  «       j                  j                  S )zD
        The amount of constant memory used by this kernel.
        )r^   rŽ   r‘   Úconstrx   s    ru   Úconst_mem_sizez_Kernel.const_mem_sizeÎ   ó%   € ð
 × Ñ ×+Ñ+Ó-×3Ñ3×9Ñ9Ð9ry   c                 ó^   — | j                   j                  «       j                  j                  S )zM
        The amount of shared memory used per block for this kernel.
        )r^   rŽ   r‘   Úsharedrx   s    ru   Úshared_mem_per_blockz_Kernel.shared_mem_per_blockÕ   s%   € ð
 × Ñ ×+Ñ+Ó-×3Ñ3×:Ñ:Ð:ry   c                 ó^   — | j                   j                  «       j                  j                  S )z:
        The maximum allowable threads per block.
        )r^   rŽ   r‘   Ú
maxthreadsrx   s    ru   Úmax_threads_per_blockz_Kernel.max_threads_per_blockÜ   s%   € ð
 × Ñ ×+Ñ+Ó-×3Ñ3×>Ñ>Ð>ry   c                 ó^   — | j                   j                  «       j                  j                  S )zM
        The amount of local memory used per thread for this kernel.
        )r^   rŽ   r‘   Úlocalrx   s    ru   Úlocal_mem_per_threadz_Kernel.local_mem_per_threadã   r—   ry   c                 ó6   — | j                   j                  «       S )z6
        Returns the LLVM IR for this kernel.
        )r^   Úget_llvm_strrx   s    ru   Úinspect_llvmz_Kernel.inspect_llvmê   s   € ð × Ñ ×-Ñ-Ó/Ð/ry   c                 ó:   — | j                   j                  |¬«      S )z7
        Returns the PTX code for this kernel.
        )r9   )r^   rM   )rd   r9   s     ru   Úinspect_asmz_Kernel.inspect_asmð   s   € ð × Ñ ×,Ñ,°Ð,Ó3Ð3ry   c                 ó6   — | j                   j                  «       S )zv
        Returns the CFG of the SASS for this kernel.

        Requires nvdisasm to be available on the PATH.
        )r^   Úget_sass_cfgrx   s    ru   Úinspect_sass_cfgz_Kernel.inspect_sass_cfgö   s   € ð × Ñ ×-Ñ-Ó/Ð/ry   c                 ó6   — | j                   j                  «       S )zp
        Returns the SASS code for this kernel.

        Requires nvdisasm to be available on the PATH.
        )r^   Úget_sassrx   s    ru   Úinspect_sassz_Kernel.inspect_sassþ   s   € ð × Ñ ×)Ñ)Ó+Ð+ry   c                 ó  — | j                   €t        d«      ‚|€t        j                  }t	        | j
                  ›d| j                  ›�|¬«       t	        d|¬«       t	        | j                   |¬«       t	        d|¬«       y)úÚ
        Produce a dump of the Python source of this function annotated with the
        corresponding Numba IR and type information. The dump is written to
        *file*, or *sys.stdout* if *file* is *None*.
        Nz Type annotation is not availableÚ ©ÚfilezP--------------------------------------------------------------------------------zP================================================================================)r]   Ú
ValueErrorÚsysÚstdoutÚprintrZ   r‚   )rd   r°   s     ru   Úinspect_typesz_Kernel.inspect_types  sg   € ð × Ñ Ð(ÜÐ?Ó@Ð@àˆ<Ü—:‘:ˆDä˜Ÿ›¨$×*=Ò*=Ð>ÀTÕJÜˆh˜TÕ"Üˆd×#Ñ#¨$Õ/Üˆh˜TÖ"ry   c                 óô   — t        «       }| j                  j                  «       }t        |t        «      rt        j                  d„ |«      }|j                  |||«      }|j                  j                  }||z  S )aÕ  
        Calculates the maximum number of blocks that can be launched for this
        kernel in a cooperative grid in the current context, for the given block
        and dynamic shared memory sizes.

        :param blockdim: Block dimensions, either as a scalar for a 1D block, or
                         a tuple for 2D or 3D blocks.
        :param dynsmemsize: Dynamic shared memory size in bytes.
        :return: The maximum number of blocks in the grid.
        c                 ó   — | |z  S rw   © )ÚxÚys     ru   Ú<lambda>z5_Kernel.max_cooperative_grid_blocks.<locals>.<lambda>&  s   € °Q¸²Ury   )
r   r^   rŽ   Ú
isinstancer€   Ú	functoolsÚreduceÚ$get_active_blocks_per_multiprocessorrg   ÚMULTIPROCESSOR_COUNT)rd   ÚblockdimÚdynsmemsizeÚctxÚcufuncÚactive_per_smÚsm_counts          ru   Úmax_cooperative_grid_blocksz#_Kernel.max_cooperative_grid_blocks  sq   € ô ‹mˆØ×"Ñ"×-Ñ-Ó/ˆä�h¤Ô&Ü ×'Ñ'Ñ(:¸HÓEˆHØ×@Ñ@ÀØAIØALóNˆð —:‘:×2Ñ2ˆØ˜xÑ'Ð'ry   c                 óà  ‡— | j                   j                  «       Š| j                  r|‰j                  dz   }‰j                  j                  |«      \  }}|t        j                  t        j                  «      k(  sJ ‚t        j                  «       }	|j                  d|¬«       g }
g }t        | j                  |«      D ]  \  }}| j                  ||||
|«       Œ t        j                  r t        j                  j!                  d«      }nd }|xr |j"                  xs |}t        j$                  ‰j"                  g|¢|¢|‘|‘|‘­d| j&                  iŽ | j                  rõt        j(                  t        j*                  	«      «       |	j,                  dk7  r¼ˆfd„}dD �cg c]  } |d|z   «      ‘Œ }}dD �cg c]  } |d|z   «      ‘Œ }}|	j,                  }| j.                  j1                  |«      \  }}}|€d	}n1|\  }}}t2        j4                  j7                  |«      }d
|›d|›d|›d�}|›d|›d|›�}|r|›d|d   ›�f|dd  z   }n|f} ||Ž ‚|
D ]	  } |«        Œ y c c}w c c}w )NÚ__errcode__r   )ÚstreamrN   c                 óô   •— ‰j                   j                  ‰j                  ›d| ›d�«      \  }}t        j                  «       }t        j                  t        j                  |«      ||«       |j                  S )NÚ__)	ÚmoduleÚget_global_symbolrY   ÚctypesÚc_intr   Údevice_to_hostÚ	addressofÚvalue)rY   ÚmemÚszÚvalrÄ   s       €ru   Úload_symbolz#_Kernel.launch.<locals>.load_symbolS  s`   ø€ Ø$Ÿm™m×=Ñ=Ø?E¿{»{Ú?Cð?Eó F‘G�C˜ô !Ÿ,™,›.�CÜ×)Ñ)¬&×*:Ñ*:¸3Ó*?ÀÀbÔIØŸ9™9Ð$ry   ÚzyxÚtidÚctaidÚ zIn function z, file z, line z, ztid=z ctaid=z: é   )r^   rŽ   r5   rY   rÍ   rÎ   rÏ   ÚsizeofrÐ   ÚmemsetÚzipr‚   Ú_prepare_argsr   ÚUSE_NV_BINDINGÚbindingÚCUstreamÚhandleÚlaunch_kernelrN   rÑ   rÒ   rÓ   r_   Úget_exceptionrQ   rR   rT   )rd   r�   ÚgriddimrÁ   rÊ   Ú	sharedmemÚexcnameÚexcmemÚexcszÚexcvalÚretrÚ
kernelargsÚtÚvÚzero_streamÚstream_handler×   ÚirÙ   rÚ   rj   ÚexcclsÚexc_argsÚlocÚlocinfoÚsymrs   ÚlinenoÚprefixÚwbrÄ   s                                 @ru   Úlaunchz_Kernel.launch-  sq  ø€ à×"Ñ"×-Ñ-Ó/ˆà�:Š:Ø—k‘k MÑ1ˆGØ"ŸM™M×;Ñ;¸GÓD‰MˆF�EØœFŸM™M¬&¯,©,Ó7Ò7Ð7Ð7Ü—\‘\“^ˆFØ�M‰M˜! FˆMÔ+ð ˆàˆ
Ü˜×+Ñ+¨TÖ2‰DˆAˆqØ×Ñ˜q ! V¨T°:Õ>ð 3ô × Ò Ü Ÿ.™.×1Ñ1°!Ó4‰KàˆKàÒ0 6§=¡=Ò?°Kˆô 	×Ñ˜VŸ]™]ð 	;Ø%ð	;à&ð	;ð 'ð	;ð +ð		;ð
 (ò	;ð *.×)9Ñ)9ò	;ð �:Š:Ü×!Ñ!¤&×"2Ñ"2°6Ó":¸FÀEÔJØ�|‰|˜qÒ ô%ñ 8=Ó=±u°!‘{ 5¨1¡9Õ-°u�Ð=Ù;@ÓA¹5°a™ W¨q¡[Õ1¸5�ÐAØ—|‘|�Ø(,×(8Ñ(8×(FÑ(FÀtÓ(LÑ%�˜ #à�;Ø ‘Gà,/Ñ)�C˜ 6Ü!Ÿw™wŸ™¨xÓ8‘HÚFIÚFNÚFLðO�Gò 18º¹eÐD�ÙÚ,2°H¸Q²KÐ @ÐBØ   ˜ñ %‘Hð  &˜w�HÙ˜hÐ'Ð'ó ˆBÙ�Dñ ùò/ >ùÚAs   Æ$I&Æ<I+c                 óÄ  — t        | j                  «      D ]  }|j                  ||||¬«      \  }}Œ t        |t        j
                  «      �ršt        |«      j                  ||«      }t        j                  }t        j                  d«      }	t        j                  d«      }
 ||j                  «      } ||j                  j                  «      }t        j                  |«      }t        j                   rt#        |«      }t        j                  |«      }|j%                  |	«       |j%                  |
«       |j%                  |«       |j%                  |«       |j%                  |«       t'        |j(                  «      D ]&  }|j%                   ||j*                  |   «      «       Œ( t'        |j(                  «      D ]&  }|j%                   ||j,                  |   «      «       Œ( yt        |t        j.                  «      r+ t1        t        d|z  «      |«      }|j%                  |«       y|t        j2                  k(  rWt        j4                  t7        j2                  |«      j9                  t6        j:                  «      «      }|j%                  |«       y|t        j<                  k(  r't        j>                  |«      }|j%                  |«       y|t        j@                  k(  r't        jB                  |«      }|j%                  |«       y|t        jD                  k(  r0t        jF                  t#        |«      «      }|j%                  |«       y|t        jH                  k(  r]|j%                  t        jB                  |jJ                  «      «       |j%                  t        jB                  |jL                  «      «       y|t        jN                  k(  r]|j%                  t        j>                  |jJ                  «      «       |j%                  t        j>                  |jL                  «      «       yt        |t        jP                  t        jR                  f«      rB|j%                  t        jT                  |j9                  t6        jV                  «      «      «       yt        |t        jX                  «      rgt        |«      j                  ||«      }|jZ                  }t        j                   rt        j                  t#        |«      «      }|j%                  |«       yt        |t        j\                  «      rCt_        |«      t_        |«      k(  sJ ‚ta        ||«      D ]  \  }}| jc                  |||||«       Œ yt        |t        jd                  «      r+	 | jc                  |j                  |jf                  |||«       yti        ||«      ‚# th        $ r ti        ||«      ‚w xY w)zF
        Convert arguments to ctypes and append to kernelargs
        )rÊ   rí   r   zc_%sN)5ÚreversedrC   Úprepare_argsr¼   r   ÚArrayr   Ú	to_devicerÏ   Ú	c_ssize_tÚc_void_pÚsizeÚdtypeÚitemsizer   Údevice_pointerrá   ÚintrW   ÚrangeÚndimÚshapeÚstridesÚIntegerÚgetattrÚfloat16Úc_uint16ÚnpÚviewÚuint16Úfloat64Úc_doubleÚfloat32Úc_floatÚbooleanÚc_uint8Ú	complex64ÚrealÚimagÚ
complex128Ú
NPDatetimeÚNPTimedeltaÚc_int64Úint64ÚRecordÚdevice_ctypes_pointerÚ	BaseTupleÚlenrß   rà   Ú
EnumMemberrÓ   ÚNotImplementedError)rd   ÚtyrÖ   rÊ   rí   rî   Ú	extensionÚdevaryÚc_intpÚmeminfoÚparentÚnitemsr  ÚptrÚdataÚaxÚcvalÚdevrecrï   rð   s                       ru   rà   z_Kernel._prepare_argsu  s?  € ô " $§/¡/Ö2ˆIØ×,Ñ,ØØØØð	 -ó ‰GˆB‘ð 3ô �bœ%Ÿ+™+Õ&Ü˜c“]×,Ñ,¨T°6Ó:ˆFä×%Ñ%ˆFä—o‘o aÓ(ˆGÜ—_‘_ QÓ'ˆFÙ˜FŸK™KÓ(ˆFÙ˜fŸl™l×3Ñ3Ó4ˆHä×'Ñ'¨Ó/ˆCä×$Ò$Ü˜#“h�ä—?‘? 3Ó'ˆDà×Ñ˜gÔ&Ø×Ñ˜fÔ%Ø×Ñ˜fÔ%Ø×Ñ˜hÔ'Ø×Ñ˜dÔ#Ü˜FŸK™KÖ(�Ø×!Ñ!¡&¨¯©°bÑ)9Ó":Õ;ð )ä˜FŸK™KÖ(�Ø×!Ñ!¡&¨¯©¸Ñ);Ó"<Õ=ñ )ô ˜œEŸM™MÔ*Ø/”7œ6 6¨B¡;Ó/°Ó4ˆDØ×Ñ˜dÕ#à”5—=‘=Ò Ü—?‘?¤2§:¡:¨c£?×#7Ñ#7¼¿	¹	Ó#BÓCˆDØ×Ñ˜dÕ#à”5—=‘=Ò Ü—?‘? 3Ó'ˆDØ×Ñ˜dÕ#à”5—=‘=Ò Ü—>‘> #Ó&ˆDØ×Ñ˜dÕ#à”5—=‘=Ò Ü—>‘>¤# c£(Ó+ˆDØ×Ñ˜dÕ#à”5—?‘?Ò"Ø×ÑœfŸn™n¨S¯X©XÓ6Ô7Ø×ÑœfŸn™n¨S¯X©XÓ6Õ7à”5×#Ñ#Ò#Ø×ÑœfŸo™o¨c¯h©hÓ7Ô8Ø×ÑœfŸo™o¨c¯h©hÓ7Õ8ä˜œU×-Ñ-¬u×/@Ñ/@ÐAÔBØ×ÑœfŸn™n¨S¯X©X´b·h±hÓ-?Ó@ÕAä˜œEŸL™LÔ)Ü˜c“]×,Ñ,¨T°6Ó:ˆFØ×.Ñ.ˆCÜ×$Ò$Ü—o‘o¤c¨#£hÓ/�Ø×Ñ˜cÕ"ä˜œEŸO™OÔ,Ü�r“7œc #›hÒ&Ð&Ð&Ü˜B ž‘��1Ø×"Ñ" 1 a¨°°zÕBñ %ô ˜œE×,Ñ,Ô-ð3Ø×"Ñ"Ø—H‘H˜cŸi™i¨°°zõô & b¨#Ó.Ð.øô	 'ò 3Ü)¨"¨cÓ2Ð2ð3ús   Ö)W	 ×	W)	NFFFFNNTFrw   )r   ©r   r   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   r>   ÚpropertyrK   r\   r|   r~   r‚   Úclassmethodr‰   rŒ   r�   r“   r–   rš   r�   r    r£   r¥   r¨   r«   rµ   rÇ   rü   rà   Ú__classcell__©rt   s   @ru   r/   r/   (   s+  ø„ ñð
 Ø;@ØJNØ6;ôZó ðZðx ñ!ó ð!ð ñ%ó ð%ò-ð ñ-ó ð-ð ñ*ó ð*ð óó ðò*Nò'ð ñ9ó ð9ð ñ:ó ð:ð ñ;ó ð;ð ñ?ó ð?ð ñ:ó ð:ò0ò4ò0ò,ó#ó"(ó,FöP\/ry   r/   c                   ó   — e Zd Zd„ Zd„ Zd„ Zy)ÚForAllc                 óp   — |dk  rt        d|z  «      ‚|| _        || _        || _        || _        || _        y )Nr   z0Can't create ForAll with negative task count: %s)r±   Ú
dispatcherÚntasksÚthread_per_blockrÊ   rè   )rd   r@  rA  ÚtpbrÊ   rè   s         ru   r>   zForAll.__init__Õ  sE   € Ø�AŠ:ÜÐOØ%ñ&ó 'ð 'à$ˆŒØˆŒØ #ˆÔØˆŒØ"ˆ�ry   c                 ó&  — | j                   dk(  ry | j                  j                  r| j                  }n | j                  j                  |Ž }| j	                  |«      }| j                   |z   dz
  |z  } |||| j
                  | j                  f   |Ž S )Nr   rÜ   )rA  r@  ÚspecializedÚ
specializeÚ_compute_thread_per_blockrÊ   rè   )rd   r�   rE  rÁ   rç   s        ru   Ú__call__zForAll.__call__ß  s”   € Ø�;‰;˜!ÒØà�?‰?×&Ò&ØŸ/™/‰Kà4˜$Ÿ/™/×4Ñ4°dÐ;ˆKØ×1Ñ1°+Ó>ˆØ—;‘; Ñ)¨AÑ-°(Ñ:ˆð+ˆ{˜7 H¨d¯k©kØŸ>™>ð*ñ +Ø,0ð2ð 	2ry   c                 ó$  — | j                   }|dk7  r|S t        «       }t        t        |j                  j                  «       «      «      }t        |j                  j                  «       d| j                  d¬«      } |j                  di |¤Ž\  }}|S )Nr   i   )ÚfuncÚb2d_funcÚmemsizeÚblocksizelimitr¸   )rB  r   ÚnextÚiterÚ	overloadsÚvaluesr‹   r^   rŽ   rè   Úget_max_potential_block_size)rd   r@  rC  rÃ   rn   ÚkwargsÚ_s          ru   rG  z ForAll._compute_thread_per_blockí  s‹   € Ø×#Ñ#ˆà�!Š8ØˆJô “-ˆCô œ$˜z×3Ñ3×:Ñ:Ó<Ó=Ó>ˆFÜØ×(Ñ(×3Ñ3Ó5ØØŸ™Ø#ô	ˆFð 6�S×5Ñ5Ñ?¸Ñ?‰FˆAˆsØˆJry   N)r5  r6  r7  r>   rH  rG  r¸   ry   ru   r>  r>  Ô  s   „ ò#ò2óry   r>  c                   ó   — e Zd Zd„ Zd„ Zy)Ú_LaunchConfigurationc                 óÒ   — || _         || _        || _        || _        || _        t
        j                  r4d}|d   |d   z  |d   z  }||k  rd|› d�}t        t        |«      «       y y y )Né€   r   rÜ   é   z
Grid size zB will likely result in GPU under-utilization due to low occupancy.)	r@  rç   rÁ   rÊ   rè   r   ÚCUDA_LOW_OCCUPANCY_WARNINGSr   r   )	rd   r@  rç   rÁ   rÊ   rè   Úmin_grid_sizeÚ	grid_sizeÚmsgs	            ru   r>   z_LaunchConfiguration.__init__  s…   € Ø$ˆŒØˆŒØ ˆŒØˆŒØ"ˆŒä×-Ò-ð  ˆMØ ™
 W¨Q¡ZÑ/°'¸!±*Ñ<ˆIØ˜=Ò(Ø# I ;ð /Að A�äÔ,¨SÓ1Õ2ð )ð .ry   c                 ó�   — | j                   j                  || j                  | j                  | j                  | j
                  «      S rw   )r@  Úcallrç   rÁ   rÊ   rè   ©rd   r�   s     ru   rH  z_LaunchConfiguration.__call__  s6   € Ø�‰×#Ñ# D¨$¯,©,¸¿¹Ø$(§K¡K°·±óAð 	Ary   N)r5  r6  r7  r>   rH  r¸   ry   ru   rV  rV    s   „ ò3ó.Ary   rV  c                   ó   — e Zd Zd„ Zd„ Zd„ Zy)ÚCUDACacheImplc                 ó"   — |j                  «       S rw   )rŒ   )rd   rn   s     ru   r¾   zCUDACacheImpl.reduce   s   € Ø×$Ñ$Ó&Ð&ry   c                 ó,   — t        j                  di |¤ŽS )Nr¸   )r/   r‰   )rd   rF   Úpayloads      ru   ÚrebuildzCUDACacheImpl.rebuild#  s   € Ü×ÑÑ* 'Ñ*Ð*ry   c                  ó   — y)NTr¸   )rd   rh   s     ru   Úcheck_cachablezCUDACacheImpl.check_cachable&  s   € ð ry   N)r5  r6  r7  r¾   rf  rh  r¸   ry   ru   rb  rb    s   „ ò'ò+óry   rb  c                   ó&   ‡ — e Zd ZdZeZˆ fd„Zˆ xZS )Ú	CUDACachezS
    Implements a cache that saves and loads CUDA kernels and compile results.
    c                 ól   •— ddl m}  |d«      5  t        ‰| �  ||«      cd d d «       S # 1 sw Y   y xY w)Nr   )Útarget_overrider   )Únumba.core.target_extensionrl  r=   Úload_overload)rd   ÚsigrF   rl  rt   s       €ru   rn  zCUDACache.load_overload7  s,   ø€ õ 	@Ù˜VÕ$Ü‘7Ñ(¨¨nÓ=÷ %×$Ò$ús   �*ª3)r5  r6  r7  r8  rb  Ú_impl_classrn  r;  r<  s   @ru   rj  rj  1  s   ø„ ñð  €K÷>ð >ry   rj  c                   óD  ‡ — e Zd ZdZdZeZefˆ fd„	Ze	d„ «       Z
d„ Z ej                  d¬«      d"d„«       Zd	„ Zd#d
„Ze	d„ «       Zd„ Zd„ Zd„ Zd„ Zd„ Ze	d„ «       Zd$d„Zd$d„Zd$d„Zd$d„Zd$d„Zd„ Zd$d„Zd„ Zd„ Z d$d„Z!d$d„Z"d$d„Z#d$d„Z$d$d„Z%e&d „ «       Z'd!„ Z(ˆ xZ)S )%ÚCUDADispatchera–  
    CUDA Dispatcher object. When configured and called, the dispatcher will
    specialize itself for the given arguments (if no suitable specialized
    version already exists) & compute capability, and launch on the device
    associated with the current context.

    Dispatcher objects are not to be constructed by the user, but instead are
    created using the :func:`numba.cuda.jit` decorator.
    Fc                 óF   •— t         ‰| �  |||¬«       d| _        i | _        y )N)ÚtargetoptionsÚpipeline_classF)r=   r>   Ú_specializedÚspecializations)rd   rA   rt  ru  rt   s       €ru   r>   zCUDADispatcher.__init__R  s0   ø€ Ü‰Ñ˜°Ø(6ð 	ô 	8ð "ˆÔð  "ˆÕry   c                 ó,   — t        j                  | «      S rw   )Ú
cuda_typesrr  rx   s    ru   Ú_numba_type_zCUDADispatcher._numba_type_b  s   € ä×(Ñ(¨Ó.Ð.ry   c                 ó8   — t        | j                  «      | _        y rw   )rj  rA   Ú_cacherx   s    ru   Úenable_cachingzCUDADispatcher.enable_cachingf  s   € Ü §¡Ó-ˆ�ry   rX  )Úmaxsizec                 ó>   — t        ||«      \  }}t        | ||||«      S rw   )r   rV  )rd   rç   rÁ   rÊ   rè   s        ru   Ú	configurezCUDADispatcher.configurei  s&   € ä7¸ÀÓJÑˆ�Ü# D¨'°8¸VÀYÓOÐOry   c                 óP   — t        |«      dvrt        d«      ‚ | j                  |Ž S )N)rY  r1   é   z.must specify at least the griddim and blockdim)r%  r±   r€  r`  s     ru   Ú__getitem__zCUDADispatcher.__getitem__n  s+   € Üˆt‹9˜IÑ%ÜÐMÓNÐNØˆt�~‰~˜tÐ$Ð$ry   c                 ó"   — t        | ||||¬«      S )a3  Returns a 1D-configured dispatcher for a given number of tasks.

        This assumes that:

        - the kernel maps the Global Thread ID ``cuda.grid(1)`` to tasks on a
          1-1 basis.
        - the kernel checks that the Global Thread ID is upper-bounded by
          ``ntasks``, and does nothing if it is not.

        :param ntasks: The number of tasks.
        :param tpb: The size of a block. An appropriate value is chosen if this
                    parameter is not supplied.
        :param stream: The stream on which the configured dispatcher will be
                       launched.
        :param sharedmem: The number of bytes of dynamic shared memory required
                          by the kernel.
        :return: A configured dispatcher, ready to launch on a set of
                 arguments.)rC  rÊ   rè   )r>  )rd   rA  rC  rÊ   rè   s        ru   ÚforallzCUDADispatcher.foralls  s   € ô( �d˜F¨°FÀiÔPÐPry   c                 ó8   — | j                   j                  d«      S )aS  
        A list of objects that must have a `prepare_args` function. When a
        specialized kernel is called, each argument will be passed through
        to the `prepare_args` (from the last object in this list to the
        first). The arguments to `prepare_args` are:

        - `ty` the numba type of the argument
        - `val` the argument value itself
        - `stream` the CUDA stream used for the current call to the kernel
        - `retr` a list of zero-arg functions that you may want to append
          post-call cleanup work to.

        The `prepare_args` function must return a tuple `(ty, val)`, which
        will be passed in turn to the next right-most `extension`. After all
        the extensions have been called, the resulting `(ty, val)` will be
        passed into Numba's default argument marshalling logic.
        rC   )rt  Úgetrx   s    ru   rC   zCUDADispatcher.extensions‰  s   € ð& ×!Ñ!×%Ñ% lÓ3Ð3ry   c                 ó    — t        t        «      ‚rw   )r±   r   )rd   r�   rS  s      ru   rH  zCUDADispatcher.__call__ž  s   € äÔ2Ó3Ð3ry   c                 óà   — | j                   r-t        t        | j                  j	                  «       «      «      }n t        j                  j                  | g|¢­Ž }|j                  |||||«       y)zJ
        Compile if necessary and invoke this kernel with *args*.
        N)	rE  rN  rO  rP  rQ  r   r   Ú
_cuda_callrü   )rd   r�   rç   rÁ   rÊ   rè   rn   s          ru   r_  zCUDADispatcher.call¢  sX   € ð ×ÒÜœ$˜tŸ~™~×4Ñ4Ó6Ó7Ó8‰Fä ×+Ñ+×6Ñ6°tÐC¸dÒCˆFà�‰�d˜G X¨v°yÕAry   c                 ó„   — |rJ ‚|D �cg c]  }| j                  |«      ‘Œ }}| j                  t        |«      «      S c c}w rw   )Útypeof_pyvalÚcompiler€   )rd   r�   ÚkwsÚarB   s        ru   Ú_compile_for_argsz CUDADispatcher._compile_for_args­  s@   € áˆˆwÙ26Ó7±$¨Q�D×%Ñ% aÕ(°$ˆÐ7Ø�|‰|œE (›OÓ,Ð,ùò 8s   ‰=c                 óì   — 	 t        |t        j                  «      S # t        t        f$ rH t        j                  |«      r1t        t        j                  |d¬«      t        j                  «      cY S ‚ w xY w)NF)Úsync)r   r   Úargumentr   r±   r   Úis_cuda_arrayÚas_cuda_array)rd   rÖ   s     ru   rŒ  zCUDADispatcher.typeof_pyval³  sh   € ð		Ü˜#œw×/Ñ/Ó0Ð0øÜ¤Ð,ò 	Ü×!Ñ! #Ô&ô œd×0Ñ0°¸5ÔAÜ%×.Ñ.ó0ò 0ð ð	ús   ‚ œAA3Á1A3c                 ó€  ‡ — ‰ j                   rt        d«      ‚t        «       j                  }t	        ˆ fd„|D «       «      }‰ j
                  j                  ||f«      }|r|S ‰ j                  }t        ‰ j                  |¬«      }|j                  |«       |j                  «        d|_        |‰ j
                  ||f<   |S )zd
        Create a new instance of this dispatcher specialized for the given
        *args*.
        zDispatcher already specializedc              3   ó@   •K  — | ]  }‰j                  |«      –— Œ y ­wrw   )rŒ  )Ú.0r�  rd   s     €ru   Ú	<genexpr>z,CUDADispatcher.specialize.<locals>.<genexpr>Ê  s   øè ø€ Ð<±t°!˜×*Ñ*¨1×-±tùs   ƒ)rt  T)rE  r<   r   rD   r€   rw  r‡  rt  rr  rA   r�  Údisable_compilerv  )rd   r�   r9   rB   Úspecializationrt  s   `     ru   rF  zCUDADispatcher.specializeÁ  s¶   ø€ ð
 ×ÒÜÐ?Ó@Ð@äÓ!×4Ñ4ˆÜÓ<±tÓ<Ó<ˆà×-Ñ-×1Ñ1°2°x°.ÓAˆÙØ!Ð!à×*Ñ*ˆÜ'¨¯©Ø6CôEˆà×Ñ˜xÔ(Ø×&Ñ&Ô(Ø&*ˆÔ#Ø-;ˆ×Ñ˜R ˜\Ñ*ØÐry   c                 ó   — | j                   S )z>
        True if the Dispatcher has been specialized.
        )rv  rx   s    ru   rE  zCUDADispatcher.specializedÙ  s   € ð
 × Ñ Ð ry   c                 óL  — |�#| j                   |j                     j                  S | j                  r6t	        t        | j                   j                  «       «      «      j                  S | j                   j                  «       D ��ci c]  \  }}||j                  “Œ c}}S c c}}w )aÑ  
        Returns the number of registers used by each thread in this kernel for
        the device in the current context.

        :param signature: The signature of the compiled kernel to get register
                          usage for. This may be omitted for a specialized
                          kernel.
        :return: The number of registers used by the compiled variant of the
                 kernel for the given signature and current device.
        )rP  r�   r“   rE  rN  rO  rQ  Úitems©rd   r[   ro  Úoverloads       ru   Úget_regs_per_threadz"CUDADispatcher.get_regs_per_threadà  s–   € ð Ð Ø—>‘> )§.¡.Ñ1×AÑAÐAØ×ÒÜœ˜TŸ^™^×2Ñ2Ó4Ó5Ó6×FÑFÐFð *.¯©×)=Ñ)=Ô)?ôAÙ)?™˜˜Xð ˜×1Ñ1Ñ1Ø)?òAð Aùó Aó   ÂB c                 óL  — |�#| j                   |j                     j                  S | j                  r6t	        t        | j                   j                  «       «      «      j                  S | j                   j                  «       D ��ci c]  \  }}||j                  “Œ c}}S c c}}w )aù  
        Returns the size in bytes of constant memory used by this kernel for
        the device in the current context.

        :param signature: The signature of the compiled kernel to get constant
                          memory usage for. This may be omitted for a
                          specialized kernel.
        :return: The size in bytes of constant memory allocated by the
                 compiled variant of the kernel for the given signature and
                 current device.
        )rP  r�   r–   rE  rN  rO  rQ  rž  rŸ  s       ru   Úget_const_mem_sizez!CUDADispatcher.get_const_mem_sizeó  s–   € ð Ð Ø—>‘> )§.¡.Ñ1×@Ñ@Ð@Ø×ÒÜœ˜TŸ^™^×2Ñ2Ó4Ó5Ó6×EÑEÐEð *.¯©×)=Ñ)=Ô)?ôAÙ)?™˜˜Xð ˜×0Ñ0Ñ0Ø)?òAð Aùó Ar¢  c                 óL  — |�#| j                   |j                     j                  S | j                  r6t	        t        | j                   j                  «       «      «      j                  S | j                   j                  «       D ��ci c]  \  }}||j                  “Œ c}}S c c}}w )aÆ  
        Returns the size in bytes of statically allocated shared memory
        for this kernel.

        :param signature: The signature of the compiled kernel to get shared
                          memory usage for. This may be omitted for a
                          specialized kernel.
        :return: The amount of shared memory allocated by the compiled variant
                 of the kernel for the given signature and current device.
        )rP  r�   rš   rE  rN  rO  rQ  rž  rŸ  s       ru   Úget_shared_mem_per_blockz'CUDADispatcher.get_shared_mem_per_block  ó–   € ð Ð Ø—>‘> )§.¡.Ñ1×FÑFÐFØ×ÒÜœ˜TŸ^™^×2Ñ2Ó4Ó5Ó6×KÑKÐKð *.¯©×)=Ñ)=Ô)?ôAÙ)?™˜˜Xð ˜×6Ñ6Ñ6Ø)?òAð Aùó Ar¢  c                 óL  — |�#| j                   |j                     j                  S | j                  r6t	        t        | j                   j                  «       «      «      j                  S | j                   j                  «       D ��ci c]  \  }}||j                  “Œ c}}S c c}}w )a(  
        Returns the maximum allowable number of threads per block
        for this kernel. Exceeding this threshold will result in
        the kernel failing to launch.

        :param signature: The signature of the compiled kernel to get the max
                          threads per block for. This may be omitted for a
                          specialized kernel.
        :return: The maximum allowable threads per block for the compiled
                 variant of the kernel for the given signature and current
                 device.
        )rP  r�   r�   rE  rN  rO  rQ  rž  rŸ  s       ru   Úget_max_threads_per_blockz(CUDADispatcher.get_max_threads_per_block  s–   € ð Ð Ø—>‘> )§.¡.Ñ1×GÑGÐGØ×ÒÜœ˜TŸ^™^×2Ñ2Ó4Ó5Ó6×LÑLÐLð *.¯©×)=Ñ)=Ô)?ôAÙ)?™˜˜Xð ˜×7Ñ7Ñ7Ø)?òAð Aùó Ar¢  c                 óL  — |�#| j                   |j                     j                  S | j                  r6t	        t        | j                   j                  «       «      «      j                  S | j                   j                  «       D ��ci c]  \  }}||j                  “Œ c}}S c c}}w )a¹  
        Returns the size in bytes of local memory per thread
        for this kernel.

        :param signature: The signature of the compiled kernel to get local
                          memory usage for. This may be omitted for a
                          specialized kernel.
        :return: The amount of local memory allocated by the compiled variant
                 of the kernel for the given signature and current device.
        )rP  r�   r    rE  rN  rO  rQ  rž  rŸ  s       ru   Úget_local_mem_per_threadz'CUDADispatcher.get_local_mem_per_thread/  r§  r¢  c                 ó*  — | j                   r| j                  t        |«      «       | j                  j                  }dj                  |«      }t        j                  ||| j                  ¬«      }t        j                  | j                  «      }||||fS )zØ
        Get a typing.ConcreteTemplate for this dispatcher and the given
        *args* and *kws* types.  This allows resolution of the return type.

        A (template, pysig, args, kws) tuple is returned.
        zCallTemplate({0}))ÚkeyÚ
signatures)Ú_can_compileÚcompile_devicer€   rA   r5  Úformatr   Úmake_concrete_templateÚnopython_signaturesr   Úpysignature)rd   r�   rŽ  Ú	func_namerY   Úcall_templateÚpysigs          ru   Úget_call_templatez CUDADispatcher.get_call_templateB  s�   € ð ×ÒØ×Ñ¤ d£Ô,ð —L‘L×)Ñ)ˆ	Ø"×)Ñ)¨)Ó4ˆä×5Ñ5Ø�i¨D×,DÑ,DôFˆä×!Ñ! $§,¡,Ó/ˆà˜e T¨3Ð.Ð.ry   c                 ó   — || j                   v�r"| j                  5  | j                  j                  d«      }| j                  j                  d«      }| j                  j                  d«      }| j                  j                  d«      }| j                  j                  d«      rdnd|dœ}t	        «       j
                  }t        | j                  ||||||||¬	«	      }	|	| j                   |<   |	j                  j                  |	j                  |	j                  |	j                  g«       d
d
d
«       |	S | j                   |   }	|	S # 1 sw Y   	S xY w)zÎCompile the device function for the given argument types.

        Each signature is compiled once by caching the compiled function inside
        this object.

        Returns the `CompileResult`.
        r5   r6   r7   r2   r3   r1   r   )r3   r2   r4   N)rP  Ú_compiling_counterrt  r‡  r   rD   r   rA   rF   Úinsert_user_functionr@   rL   rK   )
rd   r�   Úreturn_typer5   r6   r7   r2   r8   r9   rh   s
             ru   r°  zCUDADispatcher.compile_device]  s6  € ð �t—~‘~Ò%Ø×(Ó(à×*Ñ*×.Ñ.¨wÓ7�Ø×-Ñ-×1Ñ1°*Ó=�Ø×+Ñ+×/Ñ/°Ó9�Ø×-Ñ-×1Ñ1°*Ó=�ð !%× 2Ñ 2× 6Ñ 6°uÔ =™1À1Ø (ñ �ô
 (Ó)×<Ñ<�Ü# D§L¡L°+¸tØ*/Ø-5Ø+1Ø-5Ø1=Ø')ô+�ð (,�—‘˜tÑ$à×#Ñ#×8Ñ8¸×9IÑ9IØ9=¿¹Ø:>¿,¹,¸ôI÷- )ð8 ˆð —>‘> $Ñ'ˆDàˆ÷9 )ð8 ˆús   œDEÅEc                 ó†   — |D �cg c]  }|j                   ‘Œ }}| j                  ||d¬«       || j                  |<   y c c}w )NTr   )Ú_codeÚ_insertrP  )rd   rn   rB   r�  Úc_sigs        ru   Úadd_overloadzCUDADispatcher.add_overload„  s?   € Ù"*Ó+¡(˜Q�—“ (ˆÐ+Ø�‰�U˜F¨ˆÔ.Ø#)ˆ�‰�xÒ ùò ,s   …>c                 ó¬  — t        j                  |«      \  }}|�|t        j                  k(  sJ ‚| j                  r,t        t        | j                  j                  «       «      «      S | j                  j                  |«      }|�|S | j                  j                  || j                  «      }|�| j                  |xx   dz  cc<   n{| j                  |xx   dz  cc<   | j                  st!        d«      ‚t#        | j$                  |fi | j&                  ¤Ž}|j)                  «        | j                  j+                  ||«       | j-                  ||«       |S )z
        Compile and bind to the current context a version of this kernel
        specialized for the given signature.
        rÜ   zCompilation disabled)r   Únormalize_signaturer   ÚnonerE  rN  rO  rP  rQ  r‡  r|  rn  Ú	targetctxÚ_cache_hitsÚ_cache_missesr¯  r<   r/   rA   rt  r�   Úsave_overloadrÁ  )rd   ro  rB   r¼  rn   s        ru   r�  zCUDADispatcher.compile‰  s)  € ô
 !)× <Ñ <¸SÓ AÑˆ�+ØÐ" k´U·Z±ZÒ&?Ð?Ð?ð ×ÒÜœ˜TŸ^™^×2Ñ2Ó4Ó5Ó6Ð6à—^‘^×'Ñ'¨Ó1ˆFØÐ!Ø�ð —‘×*Ñ*¨3°·±Ó?ˆàÐØ×Ñ˜SÓ! QÑ&Ô!ð ×Ñ˜sÓ# qÑ(Ó#Ø×$Ò$Ü"Ð#9Ó:Ð:ä˜TŸ\™\¨8ÑJ°t×7IÑ7IÑJˆFà�K‰KŒMØ�K‰K×%Ñ% c¨6Ô2à×Ñ˜& (Ô+àˆry   c                 óè  — | j                   j                  d«      }|�F|r'| j                  |   j                  j	                  «       S | j                  |   j                  «       S |rF| j                  j                  «       D ��ci c]   \  }}||j                  j	                  «       “Œ" c}}S | j                  j                  «       D ��ci c]  \  }}||j                  «       “Œ c}}S c c}}w c c}}w )zó
        Return the LLVM IR for this kernel.

        :param signature: A tuple of argument types.
        :return: The LLVM IR for the given signature, or a dict of LLVM IR
                 for all previously-encountered signatures.

        rg   )rt  r‡  rP  rK   r¢   r£   rž  )rd   r[   rg   ro  r   s        ru   r£   zCUDADispatcher.inspect_llvm­  sô   € ð ×#Ñ#×'Ñ'¨Ó1ˆØÐ ÙØ—~‘~ iÑ0×8Ñ8×EÑEÓGÐGà—~‘~ iÑ0×=Ñ=Ó?Ð?áà-1¯^©^×-AÑ-AÔ-CôEÙ-C™M˜C ð ˜X×-Ñ-×:Ñ:Ó<Ñ<Ø-CòEð Eð .2¯^©^×-AÑ-AÔ-CôEÙ-C™M˜C ð ˜X×2Ñ2Ó4Ñ4Ø-CòEð EùóEùóEs   Â%C(Ã	C.c                 ó  — t        «       j                  }| j                  j                  d«      }|�H|r(| j                  |   j
                  j                  |«      S | j                  |   j                  |«      S |rG| j                  j                  «       D ��ci c]!  \  }}||j
                  j                  |«      “Œ# c}}S | j                  j                  «       D ��ci c]  \  }}||j                  |«      “Œ c}}S c c}}w c c}}w )a+  
        Return this kernel's PTX assembly code for for the device in the
        current context.

        :param signature: A tuple of argument types.
        :return: The PTX code for the given signature, or a dict of PTX codes
                 for all previously-encountered signatures.
        rg   )	r   rD   rt  r‡  rP  rK   rM   r¥   rž  )rd   r[   r9   rg   ro  r   s         ru   r¥   zCUDADispatcher.inspect_asmÄ  s
  € ô  Ó!×4Ñ4ˆØ×#Ñ#×'Ñ'¨Ó1ˆØÐ ÙØ—~‘~ iÑ0×8Ñ8×DÑDÀRÓHÐHà—~‘~ iÑ0×<Ñ<¸RÓ@Ð@áà-1¯^©^×-AÑ-AÔ-CôEÙ-C™M˜C ð ˜X×-Ñ-×9Ñ9¸"Ó=Ñ=Ø-CòEð Eð .2¯^©^×-AÑ-AÔ-CôEÙ-C™M˜C ð ˜X×1Ñ1°"Ó5Ñ5Ø-CòEð EùóEùóEs   Â&D Ã Dc                 ó  — | j                   j                  d«      rt        d«      ‚|�| j                  |   j	                  «       S | j                  j                  «       D ��ci c]  \  }}||j	                  «       “Œ c}}S c c}}w )aƒ  
        Return this kernel's CFG for the device in the current context.

        :param signature: A tuple of argument types.
        :return: The CFG for the given signature, or a dict of CFGs
                 for all previously-encountered signatures.

        The CFG for the device in the current context is returned.

        Requires nvdisasm to be available on the PATH.
        rg   z'Cannot get the CFG of a device function)rt  r‡  r<   rP  r¨   rž  ©rd   r[   ro  Údefns       ru   r¨   zCUDADispatcher.inspect_sass_cfgÜ  sˆ   € ð ×Ñ×!Ñ! (Ô+ÜÐHÓIÐIàÐ Ø—>‘> )Ñ,×=Ñ=Ó?Ð?ð &*§^¡^×%9Ñ%9Ô%;ô=Ù%;™	˜˜Tð ˜×.Ñ.Ó0Ñ0Ø%;ò=ð =ùó =ó   Á#Bc                 ó  — | j                   j                  d«      rt        d«      ‚|�| j                  |   j	                  «       S | j                  j                  «       D ��ci c]  \  }}||j	                  «       “Œ c}}S c c}}w )a§  
        Return this kernel's SASS assembly code for for the device in the
        current context.

        :param signature: A tuple of argument types.
        :return: The SASS code for the given signature, or a dict of SASS codes
                 for all previously-encountered signatures.

        SASS for the device in the current context is returned.

        Requires nvdisasm to be available on the PATH.
        rg   z(Cannot inspect SASS of a device function)rt  r‡  r<   rP  r«   rž  rÌ  s       ru   r«   zCUDADispatcher.inspect_sassñ  sˆ   € ð ×Ñ×!Ñ! (Ô+ÜÐIÓJÐJàÐ Ø—>‘> )Ñ,×9Ñ9Ó;Ð;ð &*§^¡^×%9Ñ%9Ô%;ô=Ù%;™	˜˜Tð ˜×*Ñ*Ó,Ñ,Ø%;ò=ð =ùó =rÎ  c                 ó�   — |€t         j                  }| j                  j                  «       D ]  \  }}|j	                  |¬«       Œ y)r­   Nr¯   )r²   r³   rP  rž  rµ   )rd   r°   rT  rÍ  s       ru   rµ   zCUDADispatcher.inspect_types  s>   € ð ˆ<Ü—:‘:ˆDà—~‘~×+Ñ+Ö-‰GˆAˆtØ×Ñ DÐÕ)ñ .ry   c                 ó   —  | ||«      }|S )r„   r¸   )r†   rA   rt  rˆ   s       ru   r‰   zCUDADispatcher._rebuild  s   € ñ
 �w Ó.ˆØˆry   c                 óD   — t        | j                  | j                  ¬«      S )zd
        Reduce the instance for serialization.
        Compiled definitions are discarded.
        )rA   rt  )r‹   rA   rt  rx   s    ru   rŒ   zCUDADispatcher._reduce_states  s    € ô
 ˜DŸL™LØ"&×"4Ñ"4ô6ð 	6ry   r4  )r   r   r   rw   )*r5  r6  r7  r8  Ú
_fold_argsr   Útargetdescrr   r>   r9  rz  r}  r½   Ú	lru_cacher€  rƒ  r…  rC   rH  r_  r�  rŒ  rF  rE  r¡  r¤  r¦  r©  r«  r¸  r°  rÁ  r�  r£   r¥   r¨   r«   rµ   r:  r‰   rŒ   r;  r<  s   @ru   rr  rr  @  s  ø„ ñð €Jà€Kà>Jõ "ð  ñ/ó ð/ò.ð €Y×Ñ Ô%òPó &ðPò%ó
Qð, ñ4ó ð4ò(4ò	Bò-òòð0 ñ!ó ð!óAó&Aó(Aó&Aó*Aò&/ó6%òN*ò
"óHEó.Eó0=ó*=ó,
*ð ñó ðö6ry   rr  ):Únumpyr  rQ   r²   rÏ   r½   Ú
numba.corer   r   r   r   r   r   Únumba.core.cachingr	   r
   Únumba.core.compiler_lockr   Únumba.core.dispatcherr   Únumba.core.errorsr   r   Únumba.core.typing.typeofr   r   Únumba.cuda.apir   Únumba.cuda.argsr   Únumba.cuda.compilerr   r   Únumba.cuda.cudadrvr   Únumba.cuda.cudadrv.devicesr   Únumba.cuda.descriptorr   Únumba.cuda.errorsr   r   Ú
numba.cudary  Únumbar   r   Úwarningsr   rP   ÚReduceMixinr/   Úobjectr>  rV  rb  rj  rr  r¸   ry   ru   Ú<module>ré     s°   ðÛ Û 	Û 
Û Û ç H× Hß /Ý 9Ý ,ß Fß 4å -Ý $ß :Ý %Ý 2Ý -÷<å *å Ý å ò*Ð ôi/ˆi×#Ñ#ô i/ôX+ˆVô +÷\Añ Aô:�Iô ô$>�ô >ôa6�Z ×!6Ñ!6õ a6ry   