o
    8ήcw                     @   s  d dl Zd dlZd dlZd dlZd dlZd dlmZmZm	Z	m
Z
mZmZ d dlmZmZ d dlmZ d dlmZ d dlmZ d dlmZmZ d dlmZ d d	lmZ d d
lmZmZ d dl m!Z! d dl"m#Z# d dl$m%Z% d dl&m'Z' d dl(m)Z)m*Z* d dl+m
Z, d dl-m.Z. d dl-m/Z/ d dl0m1Z1 G dd dej2Z3G dd de4Z5G dd dZ6G dd deZ7G dd deZ8G dd deej2Z9dS )     N)config	serializesigutilstypestypingutils)Cache	CacheImpl)global_compiler_lock)
Dispatcher)NumbaPerformanceWarning)Purposetypeofget_current_device)wrap_arg)compile_cudaCUDACompiler)driver)get_context)get_cudalib)cuda_target)missing_launch_config_msgnormalize_kernel_dimensions)r   cuda)_dispatcher)warnc                       s   e Zd ZdZe			d* fdd	Zedd Zed	d
 Zdd Z	edd Z
edd Ze fddZdd Zdd Zedd Zedd Zdd Zdd Zdd  Zd+d!d"Zd,d$d%Zd-d&d'Zd(d) Z  ZS )._Kernelz
    CUDA Kernel specialized for a given set of argument types. When called, this
    object launches the kernel on the device.
    NFTc              
      sL  |rt dt   d| _d | _|| _|| _|| _|| _|p g | _	| j| j||
r+dndd}t
| jtj| j| j| j|||d}|j}| jj}|j}|j}||j|j|||||	\}}|s`g }d| v | _| jrs|tdd	d
 |D ]}|| qu|j| _|j| _|j| _|| _|j| _|| _|j| _|j | _ g | _!g | _"g | _#d S )Nz,Cannot compile a device function as a kernelF   r   )debuglineinfofastmathopt)r    r!   inliner"   nvvm_optionscudaCGGetIntrinsicHandle	cudadevrtT)static)$RuntimeErrorsuper__init__
objectmodeentry_pointpy_funcargtypesr    r!   
extensionsr   r   voidtarget_context__code__co_filenameco_firstlinenoprepare_cuda_kernellibraryfndescget_asm_strcooperativeappendr   add_linking_filename
entry_name	signaturetype_annotation_type_annotation_codelibrarycall_helperenvironment_referenced_environmentsliftedreload_init)selfr.   r/   linkr    r!   r$   r"   r0   max_registersr#   devicer%   crestgt_ctxcodefilenamelinenumlibkernelfilepath	__class__ </tmp/pip-target-vg8gfxp4/lib/python/numba/cuda/dispatcher.pyr+   %   sb   



z_Kernel.__init__c                 C      | j S N)rB   rH   rV   rV   rW   r7   x      z_Kernel.libraryc                 C   rX   rY   )rA   rZ   rV   rV   rW   r@   |   r[   z_Kernel.type_annotationc                 C   rX   rY   )rE   rZ   rV   rV   rW   _find_referenced_environments   s   z%_Kernel._find_referenced_environmentsc                 C   
   | j  S rY   )r2   codegenrZ   rV   rV   rW   r^         
z_Kernel.codegenc                 C   s   t | jjS rY   )tupler?   argsrZ   rV   rV   rW   argument_types   s   z_Kernel.argument_typesc	           
         sX   |  | }	t| |	  d|	_||	_||	_||	_d|	_||	_||	_	||	_
||	_||	_|	S )&
        Rebuild an instance.
        N)__new__r*   r+   r-   r:   r>   r?   rA   rB   r    r!   rC   r0   )
clsr:   r=   r?   codelibraryr    r!   rC   r0   instancerT   rV   rW   _rebuild   s   
z_Kernel._rebuildc              
   C   s(   t | j| j| j| j| j| j| j| jdS )a  
        Reduce the instance for serialization.
        Compiled definitions are serialized in PTX form.
        Type annotation are discarded.
        Thread, block and shared memory configuration are serialized.
        Stream information is discarded.
        )r:   r=   r?   rf   r    r!   rC   r0   )	dictr:   r>   r?   rB   r    r!   rC   r0   rZ   rV   rV   rW   _reduce_states   s
   
z_Kernel._reduce_statesc                 C   s   | j   dS )z7
        Force binding to current CUDA context
        N)rB   
get_cufuncrZ   rV   rV   rW   bind      z_Kernel.bindc                 C   s   t  S )z,
        Get current active context
        r   rZ   rV   rV   rW   rK         z_Kernel.devicec                 C   s   | j  jjS )zN
        The number of registers used by each thread for this kernel.
        )rB   rk   attrsregsrZ   rV   rV   rW   regs_per_thread   s   z_Kernel.regs_per_threadc                 C   r]   )z6
        Returns the LLVM IR for this kernel.
        )rB   get_llvm_strrZ   rV   rV   rW   inspect_llvm   s   
z_Kernel.inspect_llvmc                 C   s   | j j|dS )z7
        Returns the PTX code for this kernel.
        cc)rB   r9   )rH   ru   rV   rV   rW   inspect_asm   rm   z_Kernel.inspect_asmc                 C   r]   )zp
        Returns the SASS code for this kernel.

        Requires nvdisasm to be available on the PATH.
        )rB   get_sassrZ   rV   rV   rW   inspect_sass   s   
z_Kernel.inspect_sassc                 C   sb   | j du r	td|du rtj}td| j| jf |d td|d t| j |d td|d dS )
        Produce a dump of the Python source of this function annotated with the
        corresponding Numba IR and type information. The dump is written to
        *file*, or *sys.stdout* if *file* is *None*.
        Nz Type annotation is not availablez%s %sfilezP--------------------------------------------------------------------------------zP================================================================================)rA   
ValueErrorsysstdoutprintr>   rb   )rH   r{   rV   rV   rW   inspect_types   s   
z_Kernel.inspect_typesr   c                 C   sH   t  }| j }t|trtdd |}||||}|jj	}|| S )a  
        Calculates the maximum number of blocks that can be launched for this
        kernel in a cooperative grid in the current context, for the given block
        and dynamic shared memory sizes.

        :param blockdim: Block dimensions, either as a scalar for a 1D block, or
                         a tuple for 2D or 3D blocks.
        :param dynsmemsize: Dynamic shared memory size in bytes.
        :return: The maximum number of blocks in the grid.
        c                 S   s   | | S rY   rV   )xyrV   rV   rW   <lambda>   s    z5_Kernel.max_cooperative_grid_blocks.<locals>.<lambda>)
r   rB   rk   
isinstancer`   	functoolsreduce$get_active_blocks_per_multiprocessorrK   MULTIPROCESSOR_COUNT)rH   blockdimdynsmemsizectxcufuncactive_per_smsm_countrV   rV   rW   max_cooperative_grid_blocks   s   

z#_Kernel.max_cooperative_grid_blocksc                    s  | j   | jr* jd } j|\}}|ttjksJ t }	|j	d|d g }
g }t
| j|D ]\}}| ||||
| q4tjrLtjd}nd }|rS|jpT|}tj jg|||||R d| ji | jrtt|	|| |	jdkr݇ fddfddd	D }fd
dd	D }|	j}| j|\}}}|d u rd}n|\}}}tj|}d|||f }d|||f }|rd||d f f|dd   }|| |f}|| |
D ]}|  qd S )N__errcode__r   )streamr:   c                    s<    j d j| f \}}t }tt||| |jS )Nz%s__%s__)	moduleget_global_symbolr=   ctypesc_intr   device_to_host	addressofvalue)r=   memszval)r   rV   rW   load_symbol#  s   
z#_Kernel.launch.<locals>.load_symbolc                       g | ]} d | qS )tidrV   .0ir   rV   rW   
<listcomp>+      z"_Kernel.launch.<locals>.<listcomp>zyxc                    r   )ctaidrV   r   r   rV   rW   r   ,  r    z"In function %r, file %s, line %s, z%stid=%s ctaid=%sz%s: %s   )rB   rk   r    r=   r   r   r   sizeofr   memsetziprb   _prepare_argsr   USE_NV_BINDINGbindingCUstreamhandlelaunch_kernelr:   r   r   r   rC   get_exceptionospathabspath)rH   ra   griddimr   r   	sharedmemexcnameexcmemexcszexcvalretr
kernelargstvzero_streamstream_handler   r   rN   excclsexc_argsloclocinfosymrS   linenoprefixwbrV   )r   r   rW   launch   sn   





z_Kernel.launchc                 C   sX  t | jD ]}|j||||d\}}qt|tjrt|||}tj	}t
d}	t
d}
||j}||jj}t|}tjrEt|}t
|}||	 ||
 || || || t|jD ]}|||j|  qht|jD ]}|||j|  qzdS t|tjrttd| |}|| dS |tjkrtt|tj}|| dS |tjkrt|}|| dS |tj krt!|}|| dS |tj"krt#t|}|| dS |tj$kr|t!|j% |t!|j& dS |tj'kr |t|j% |t|j& dS t|tj(tj)fr8|t*|tj+ dS t|tj,r\t|||}|j-}tjrUt
t|}|| dS t|tj.rt/|t/|ksnJ t0||D ]\}}| 1||||| qsdS t|tj2rz| 1|j|j3||| W dS  t4y   t4||w t4||)zF
        Convert arguments to ctypes and append to kernelargs
        )r   r   r   zc_%sN)5reversedr0   prepare_argsr   r   Arrayr   	to_devicer   	c_ssize_tc_void_psizedtypeitemsizer   device_pointerr   intr;   rangendimshapestridesIntegergetattrfloat16c_uint16npviewuint16float64c_doublefloat32c_floatbooleanc_uint8	complex64realimag
complex128
NPDatetimeNPTimedeltac_int64int64Recorddevice_ctypes_pointer	BaseTuplelenr   r   
EnumMemberr   NotImplementedError)rH   tyr   r   r   r   	extensiondevaryc_intpmeminfoparentnitemsr   ptrdataaxcvaldevrecr   r   rV   rV   rW   r   E  s   


















z_Kernel._prepare_args)	NFFFFNNTFrY   )r   r   r   )__name__
__module____qualname____doc__r
   r+   propertyr7   r@   r\   r^   rb   classmethodrh   rj   rl   rK   rq   rs   rv   rx   r   r   r   r   __classcell__rV   rV   rT   rW   r      s>    R








Hr   c                   @   $   e Zd Zdd Zdd Zdd ZdS )ForAllc                 C   s6   |dk r
t d| || _|| _|| _|| _|| _d S )Nr   z0Can't create ForAll with negative task count: %s)r|   
dispatcherntasksthread_per_blockr   r   )rH   r  r  tpbr   r   rV   rV   rW   r+     s   
zForAll.__init__c                 G   s^   | j dkrd S | jjr| j}n| jj| }| |}| j | d | }|||| j| jf | S )Nr   r   )r  r  specialized
specialize_compute_thread_per_blockr   r   )rH   ra   r  r   r   rV   rV   rW   __call__  s   


zForAll.__call__c                 C   sZ   | j }|dkr	|S t }tt|j }t|j d| j	dd}|j
di |\}}|S )Nr   i   )funcb2d_funcmemsizeblocksizelimitrV   )r  r   nextiter	overloadsvaluesri   rB   rk   r   get_max_potential_block_size)rH   r  r  r   rR   kwargs_rV   rV   rW   r    s   z ForAll._compute_thread_per_blockN)r  r  r  r+   r  r  rV   rV   rV   rW   r
    s    
r
  c                   @   s   e Zd Zdd Zdd ZdS )_LaunchConfigurationc           	      C   sl   || _ || _|| _|| _|| _tjr2d}|d |d  |d  }||k r4d| d}tt| d S d S d S )N   r   r      z
Grid size zB will likely result in GPU under-utilization due to low occupancy.)	r  r   r   r   r   r   CUDA_LOW_OCCUPANCY_WARNINGSr   r   )	rH   r  r   r   r   r   min_grid_size	grid_sizemsgrV   rV   rW   r+     s   	z_LaunchConfiguration.__init__c                 G   s   | j || j| j| j| jS rY   )r  callr   r   r   r   rH   ra   rV   rV   rW   r    s   z_LaunchConfiguration.__call__N)r  r  r  r+   r  rV   rV   rV   rW   r    s    r  c                   @   r	  )CUDACacheImplc                 C   s   |  S rY   )rj   )rH   rR   rV   rV   rW   r     s   zCUDACacheImpl.reducec                 C   s   t jdi |S )NrV   )r   rh   )rH   r2   payloadrV   rV   rW   rebuild     zCUDACacheImpl.rebuildc                 C   s   dS )NTrV   )rH   rL   rV   rV   rW   check_cachable  s   zCUDACacheImpl.check_cachableN)r  r  r  r   r)  r+  rV   rV   rV   rW   r'    s    r'  c                   @   s   e Zd ZdZeZdS )	CUDACachezS
    Implements a cache that saves and loads CUDA kernels and compile results.
    N)r  r  r  r  r'  _impl_classrV   rV   rV   rW   r,    s    r,  c                       s  e Zd ZdZdZeZef fdd	Ze	dd Z
dd Zejd	d
d9ddZdd Zd:ddZe	dd Zdd Zdd Zdd Zdd Zdd Ze	dd Zd;d!d"Zd#d$ Zd%d& Zd'd( Zd)d* Zd;d+d,Zd;d-d.Zd;d/d0Zd;d1d2Z d3d4 Z!e"d5d6 Z#d7d8 Z$  Z%S )<CUDADispatchera  
    CUDA Dispatcher object. When configured and called, the dispatcher will
    specialize itself for the given arguments (if no suitable specialized
    version already exists) & compute capability, and launch on the device
    associated with the current context.

    Dispatcher objects are not to be constructed by the user, but instead are
    created using the :func:`numba.cuda.jit` decorator.
    Fc                    s"   t  j|||d d| _i | _d S )N)targetoptionspipeline_classF)r*   r+   _specializedspecializations)rH   r.   r/  r0  rT   rV   rW   r+     s
   
	
zCUDADispatcher.__init__c                 C   s
   t | S rY   )
cuda_typesr.  rZ   rV   rV   rW   _numba_type_*  r_   zCUDADispatcher._numba_type_c                 C   s   t | j| _d S rY   )r,  r.   _cacherZ   rV   rV   rW   enable_caching.  r*  zCUDADispatcher.enable_cachingr  )maxsizer   c                 C   s   t ||\}}t| ||||S rY   )r   r  )rH   r   r   r   r   rV   rV   rW   	configure1  s   zCUDADispatcher.configurec                 C   s   t |dvr
td| j| S )N)r   r      z.must specify at least the griddim and blockdim)r   r|   r8  r&  rV   rV   rW   __getitem__6  s   
zCUDADispatcher.__getitem__c                 C   s   t | ||||dS )a3  Returns a 1D-configured dispatcher for a given number of tasks.

        This assumes that:

        - the kernel maps the Global Thread ID ``cuda.grid(1)`` to tasks on a
          1-1 basis.
        - the kernel checks that the Global Thread ID is upper-bounded by
          ``ntasks``, and does nothing if it is not.

        :param ntasks: The number of tasks.
        :param tpb: The size of a block. An appropriate value is chosen if this
                    parameter is not supplied.
        :param stream: The stream on which the configured dispatcher will be
                       launched.
        :param sharedmem: The number of bytes of dynamic shared memory required
                          by the kernel.
        :return: A configured dispatcher, ready to launch on a set of
                 arguments.)r  r   r   )r
  )rH   r  r  r   r   rV   rV   rW   forall;  s   zCUDADispatcher.forallc                 C   s   | j dS )aS  
        A list of objects that must have a `prepare_args` function. When a
        specialized kernel is called, each argument will be passed through
        to the `prepare_args` (from the last object in this list to the
        first). The arguments to `prepare_args` are:

        - `ty` the numba type of the argument
        - `val` the argument value itself
        - `stream` the CUDA stream used for the current call to the kernel
        - `retr` a list of zero-arg functions that you may want to append
          post-call cleanup work to.

        The `prepare_args` function must return a tuple `(ty, val)`, which
        will be passed in turn to the next right-most `extension`. After all
        the extensions have been called, the resulting `(ty, val)` will be
        passed into Numba's default argument marshalling logic.
        r0   )r/  getrZ   rV   rV   rW   r0   Q  s   zCUDADispatcher.extensionsc                 O   s   t trY   )r|   r   )rH   ra   r  rV   rV   rW   r  f  s   zCUDADispatcher.__call__c                 C   sD   | j rtt| j }n
tjj| g|R  }|||||| dS )zJ
        Compile if necessary and invoke this kernel with *args*.
        N)	r  r  r  r  r  r   r   
_cuda_callr   )rH   ra   r   r   r   r   rR   rV   rV   rW   r%  j  s   zCUDADispatcher.callc                    s(   |rJ  fdd|D }  t|S )Nc                    s   g | ]}  |qS rV   )typeof_pyvalr   arZ   rV   rW   r   x  s    z4CUDADispatcher._compile_for_args.<locals>.<listcomp>)compiler`   )rH   ra   kwsr/   rV   rZ   rW   _compile_for_argsu  s   z CUDADispatcher._compile_for_argsc                 C   sD   zt |tjW S  ty!   t|r t tj|ddtj Y S  w )NF)sync)r   r   argumentr|   r   is_cuda_arrayas_cuda_array)rH   r   rV   rV   rW   r>  {  s   
zCUDADispatcher.typeof_pyvalc                    s   t  j}t fdd|D } jrtd j||f}|r"|S  j}t j	|d}|
| |  d|_| j||f< |S )zd
        Create a new instance of this dispatcher specialized for the given
        *args*.
        c                    s   g | ]} j |qS rV   )	typingctxresolve_argument_typer?  rZ   rV   rW   r     r   z-CUDADispatcher.specialize.<locals>.<listcomp>zDispatcher already specialized)r/  T)r   compute_capabilityr`   r  r)   r2  r<  r/  r.  r.   rA  disable_compiler1  )rH   ra   ru   r/   specializationr/  rV   rZ   rW   r    s$   
zCUDADispatcher.specializec                 C   rX   )z>
        True if the Dispatcher has been specialized.
        )r1  rZ   rV   rV   rW   r    rn   zCUDADispatcher.specializedNc                 C   sD   |dur| j |j jS | jrtt| j  jS dd | j  D S )a  
        Returns the number of registers used by each thread in this kernel for
        the device in the current context.

        :param signature: The signature of the compiled kernel to get register
                          usage for. This may be omitted for a specialized
                          kernel.
        :return: The number of registers used by the compiled variant of the
                 kernel for the given signature and current device.
        Nc                 S   s   i | ]\}}||j qS rV   )rq   r   sigoverloadrV   rV   rW   
<dictcomp>  s    z6CUDADispatcher.get_regs_per_thread.<locals>.<dictcomp>)r  ra   rq   r  r  r  r  itemsrH   r?   rV   rV   rW   get_regs_per_thread  s   z"CUDADispatcher.get_regs_per_threadc                 C   sP   | j r
| t| | jj}d|}tj||| jd}t	
| j}||||fS )z
        Get a typing.ConcreteTemplate for this dispatcher and the given
        *args* and *kws* types.  This allows resolution of the return type.

        A (template, pysig, args, kws) tuple is returned.
        zCallTemplate({0}))key
signatures)_can_compilecompile_devicer`   r.   r  formatr   make_concrete_templatenopython_signaturesr   pysignature)rH   ra   rB  	func_namer=   call_templatepysigrV   rV   rW   get_call_template  s   
z CUDADispatcher.get_call_templatec              
   C   s   || j vrX| jF | jd}| jd}| jd}|| jdr$dnd|d}t| jd|||||d	}|| j |< |j|j|j	|j
g W d   |S 1 sQw   Y  |S | j | }|S )
zCompile the device function for the given argument types.

        Each signature is compiled once by caching the compiled function inside
        this object.

        Returns the `CompileResult`.
        r    r$   r"   r#   r   r   )r    r#   r"   N)r    r$   r"   r%   )r  _compiling_counterr/  r<  r   r.   r2   insert_user_functionr-   r8   r7   )rH   ra   r    r$   r"   r%   rL   rV   rV   rW   rW    s4   





zCUDADispatcher.compile_devicec                 C   s,   dd |D }| j ||dd || j|< d S )Nc                 S   s   g | ]}|j qS rV   )_coder?  rV   rV   rW   r     s    z/CUDADispatcher.add_overload.<locals>.<listcomp>Tr   )_insertr  )rH   rR   r/   c_sigrV   rV   rW   add_overload  s   zCUDADispatcher.add_overloadc                 C   s   t |\}}|du s|tjksJ | jrtt| j S | j	|}|dur*|S | j
|| j}|dur@| j|  d7  < n&| j|  d7  < | jsPtdt| j|fi | j}|  | j
|| | || |S )z
        Compile and bind to the current context a version of this kernel
        specialized for the given signature.
        Nr   zCompilation disabled)r   normalize_signaturer   noner  r  r  r  r  r<  r5  load_overload	targetctx_cache_hits_cache_missesrV  r)   r   r.   r/  rl   save_overloadre  )rH   rN  r/   return_typerR   rV   rV   rW   rA    s$   zCUDADispatcher.compilec                 C   sb   | j d}|dur|r| j| j S | j|  S |r'dd | j D S dd | j D S )z
        Return the LLVM IR for this kernel.

        :param signature: A tuple of argument types.
        :return: The LLVM IR for the given signature, or a dict of LLVM IR
                 for all previously-encountered signatures.

        rK   Nc                 S   s   i | ]
\}}||j  qS rV   )r7   rr   rM  rV   rV   rW   rP  4      z/CUDADispatcher.inspect_llvm.<locals>.<dictcomp>c                 S      i | ]	\}}||  qS rV   )rs   rM  rV   rV   rW   rP  7      )r/  r<  r  r7   rr   rs   rQ  rH   r?   rK   rV   rV   rW   rs   #  s   	zCUDADispatcher.inspect_llvmc                    sv   t  j | jd}|dur!|r| j| j S | j|  S |r/ fdd| j D S  fdd| j D S )a+  
        Return this kernel's PTX assembly code for for the device in the
        current context.

        :param signature: A tuple of argument types.
        :return: The PTX code for the given signature, or a dict of PTX codes
                 for all previously-encountered signatures.
        rK   Nc                    s   i | ]\}}||j  qS rV   )r7   r9   rM  rt   rV   rW   rP  L  s    z.CUDADispatcher.inspect_asm.<locals>.<dictcomp>c                    s   i | ]
\}}||  qS rV   )rv   rM  rt   rV   rW   rP  O  rn  )	r   rJ  r/  r<  r  r7   r9   rv   rQ  rq  rV   rt   rW   rv   :  s   	

zCUDADispatcher.inspect_asmc                 C   s>   | j dr
td|dur| j|  S dd | j D S )a  
        Return this kernel's SASS assembly code for for the device in the
        current context.

        :param signature: A tuple of argument types.
        :return: The SASS code for the given signature, or a dict of SASS codes
                 for all previously-encountered signatures.

        SASS for the device in the current context is returned.

        Requires nvdisasm to be available on the PATH.
        rK   z(Cannot inspect SASS of a device functionNc                 S   ro  rV   )rx   )r   rN  defnrV   rV   rW   rP  e  rp  z/CUDADispatcher.inspect_sass.<locals>.<dictcomp>)r/  r<  r)   r  rx   rQ  rR  rV   rV   rW   rx   R  s   zCUDADispatcher.inspect_sassc                 C   s2   |du rt j}| j D ]
\}}|j|d qdS )ry   Nrz   )r}   r~   r  rQ  r   )rH   r{   r  rr  rV   rV   rW   r   h  s
   zCUDADispatcher.inspect_typesc                 C   s   | j  D ]}|  qd S rY   )r  r  rl   )rH   rr  rV   rV   rW   rl   t  s   
zCUDADispatcher.bindc                 C   s   | ||}|S )rc   rV   )re   r.   r/  rg   rV   rV   rW   rh   x  s   
zCUDADispatcher._rebuildc                 C   s   t | j| jdS )zd
        Reduce the instance for serialization.
        Compiled definitions are discarded.
        )r.   r/  )ri   r.   r/  rZ   rV   rV   rW   rj     s   zCUDADispatcher._reduce_statesr  )r   r   r   rY   )&r  r  r  r  
_fold_argsr   targetdescrr   r+   r  r4  r6  r   	lru_cacher8  r:  r;  r0   r  r%  rC  r>  r  r  rS  r_  rW  re  rA  rs   rv   rx   r   rl   r  rh   rj   r  rV   rV   rT   rW   r.    sD    





$
$



r.  ):numpyr   r   r}   r   r   
numba.corer   r   r   r   r   r   numba.core.cachingr   r	   numba.core.compiler_lockr
   numba.core.dispatcherr   numba.core.errorsr   numba.core.typing.typeofr   r   numba.cuda.apir   numba.cuda.argsr   numba.cuda.compilerr   r   numba.cuda.cudadrvr   numba.cuda.cudadrv.devicesr   numba.cuda.cudadrv.libsr   numba.cuda.descriptorr   numba.cuda.errorsr   r   
numba.cudar3  numbar   r   warningsr   ReduceMixinr   objectr
  r  r'  r,  r.  rV   rV   rV   rW   <module>   s@        .