U
    hâËdy  ã                   @   sÀ  d Z ddlZddlZddlZddlZddlmZ ddlZddl	Z	ddl	m
Z
 ddlmZ ddlmZ ddlmZmZ ddlmZ dd	lmZ dd
lmZ ddlmZ ddlmZ ddlmZ zeedƒdƒZW n e k
rê   dd„ ZY nX dd„ Z!dd„ Z"dd„ Z#G dd„ de
j$ƒZ%G dd„ de%ƒZ&edd„ ƒZ'G dd„ de%ƒZ(G dd„ de)ƒZ*G d d!„ d!e%ej+ƒZ,G d"d#„ d#e%ej+ƒZ-d5d$d%„Z.d6d&d'„Z/d(d)„ Z0d*d+„ Z1d,Z2d-d.„ Z3d7d1d2„Z4d3d4„ Z5dS )8zÑ
A CUDA ND Array is recognized by checking the __cuda_memory__ attribute
on the object.  If it exists and evaluate to True, it must define shape,
strides, dtype and size attributes similar to a NumPy ndarray.
é    N)Úc_void_p)Ú_devicearray)Údevices)Údriver)ÚtypesÚconfig)Úto_fixed_tuple)Ú
dummyarray)Únumpy_support)Úprepare_shape_strides_dtype)ÚNumbaPerformanceWarning)ÚwarnÚ	lru_cachec                 C   s   | S ©N© )Úfuncr   r   úW/home/sam/Atlas/atlas_env/lib/python3.8/site-packages/numba/cuda/cudadrv/devicearray.pyr      s    c                 C   s   t | ddƒS )z$Check if an object is a CUDA ndarrayÚ__cuda_ndarray__F)Úgetattr©Úobjr   r   r   Úis_cuda_ndarray#   s    r   c                    sB   t ˆ ƒ ‡ fdd„}|dtƒ |dtƒ |dtjƒ |dtƒ dS )z,Verify the CUDA ndarray interface for an objc                    s6   t ˆ | ƒst| ƒ‚ttˆ | ƒ|ƒs2td| |f ƒ‚d S )Nz%s must be of type %s)ÚhasattrÚAttributeErrorÚ
isinstancer   )ÚattrÚtypr   r   r   Úrequires_attr,   s    
z4verify_cuda_ndarray_interface.<locals>.requires_attrÚshapeÚstridesÚdtypeÚsizeN)Úrequire_cuda_ndarrayÚtupleÚnpr    Úint)r   r   r   r   r   Úverify_cuda_ndarray_interface(   s    

r&   c                 C   s   t | ƒstdƒ‚dS )z9Raises ValueError is is_cuda_ndarray(obj) evaluates Falsezrequire an cuda ndarray objectN)r   Ú
ValueErrorr   r   r   r   r"   8   s    r"   c                   @   sÆ   e Zd ZdZdZdZd%dd„Zedd„ ƒZd&d	d
„Z	edd„ ƒZ
d'dd„Zdd„ Zedd„ ƒZedd„ ƒZejd(dd„ƒZejd)dd„ƒZd*dd„Zdd„ Zdd„ Zd+dd „Zd!d"„ Zed#d$„ ƒZdS ),ÚDeviceNDArrayBasez$A on GPU NDArray representation
    Tr   Nc                 C   s"  t |tƒr|f}t |tƒr |f}t |¡}t|ƒ| _t|ƒ| jkrJtdƒ‚tj 	d|||j
¡| _t|ƒ| _t|ƒ| _|| _tt tj| jd¡ƒ| _| jdkrÜ|dkrÎt | j| j| jj
¡| _t ¡  | j¡}nt |¡| _n6tjrðtj d¡}ntdƒ}tjt ¡ |dd�}d| _|| _ || _!dS )a5  
        Args
        ----

        shape
            array shape.
        strides
            array strides.
        dtype
            data type as np.dtype coercible object.
        stream
            cuda stream.
        gpu_data
            user provided device memory for the ndarray data buffer
        zstrides not match ndimr   é   N)ÚcontextZpointerr!   )"r   r%   r$   r    ÚlenÚndimr'   r	   ÚArrayZ	from_descÚitemsizeÚ_dummyr#   r   r   Ú	functoolsÚreduceÚoperatorÚmulr!   Ú_driverZmemory_size_from_infoÚ
alloc_sizer   Úget_contextZmemallocZdevice_memory_sizeÚUSE_NV_BINDINGÚbindingÚCUdeviceptrr   ZMemoryPointerÚgpu_dataÚstream)Úselfr   r   r    r;   r:   Únullr   r   r   Ú__init__D   sD    



ÿ


  ÿ
 ÿzDeviceNDArrayBase.__init__c                 C   s‚   t jr"| jd k	rt| jƒ}q<d}n| jjd k	r8| jj}nd}t| jƒt| ƒrPd nt| jƒ|df| j	j
| jdkrxt| jƒnd ddœS )Nr   Fé   )r   r   ÚdataZtypestrr;   Úversion)r4   r7   Údevice_ctypes_pointerr%   Úvaluer#   r   Úis_contiguousr   r    Ústrr;   )r<   Zptrr   r   r   Ú__cuda_array_interface__w   s    

úz*DeviceNDArrayBase.__cuda_array_interface__c                 C   s   t   | ¡}||_|S )zBind a CUDA stream to this object so that all subsequent operation
        on this array defaults to the given stream.
        )Úcopyr;   )r<   r;   Úcloner   r   r   Úbind�   s    
zDeviceNDArrayBase.bindc                 C   s   |   ¡ S r   ©Ú	transpose©r<   r   r   r   ÚT•   s    zDeviceNDArrayBase.Tc                 C   s|   |rt |ƒt t| jƒƒkr| S | jdkr6d}t|ƒ‚nB|d k	rdt|ƒtt| jƒƒkrdtd|f ƒ‚nddlm} || ƒS d S )Né   z2transposing a non-2D DeviceNDArray isn't supportedzinvalid axes list %rr   rJ   )r#   Úranger,   ÚNotImplementedErrorÚsetr'   Znumba.cuda.kernels.transposerK   )r<   ZaxesÚmsgrK   r   r   r   rK   ™   s    

zDeviceNDArrayBase.transposec                 C   s   |s
| j S |S r   ©r;   )r<   r;   r   r   r   Ú_default_stream¥   s    z!DeviceNDArrayBase._default_streamc                 C   sR   d| j k}| jd r|sd}n| jd r2|s2d}nd}t | j¡}t || j|¡S )ún
        Magic attribute expected by Numba to get the numba type that
        represents this object.
        r   ÚC_CONTIGUOUSÚCÚF_CONTIGUOUSÚFÚA)r   Úflagsr
   Ú
from_dtyper    r   r-   r,   )r<   Ú	broadcastZlayoutr    r   r   r   Ú_numba_type_¨   s    
zDeviceNDArrayBase._numba_type_c                 C   s2   | j dkr&tjrtj d¡S tdƒS n| j jS dS )z:Returns the ctypes pointer to the GPU data buffer
        Nr   )r:   r4   r7   r8   r9   r   rB   rL   r   r   r   rB   Ç   s
    

z'DeviceNDArrayBase.device_ctypes_pointerc                 C   s®   |j dkrdS t| ƒ |  |¡}t| ƒt|ƒ }}t |¡rdt|ƒ t||ƒ tj| || j|d� nFt	j
||jd rxdndd|jd  d	�}t||ƒ tj| || j|d� dS )
zŸCopy `ary` to `self`.

        If `ary` is a CUDA memory, perform a device-to-device transfer.
        Otherwise, perform a a host-to-device transfer.
        r   NrS   rV   rW   rY   TZ	WRITEABLE)ÚorderÚsubokrG   )r!   Úsentry_contiguousrT   Ú
array_corer4   Úis_device_memoryÚcheck_array_compatibilityÚdevice_to_devicer5   r$   Úarrayr[   Zhost_to_device)r<   Úaryr;   Z	self_coreZary_corer   r   r   Úcopy_to_deviceÓ   s&    




ü
ÿz DeviceNDArrayBase.copy_to_devicec                 C   sÐ   t dd„ | jD ƒƒr(d}t| | j¡ƒ‚| jdks:tdƒ‚|  |¡}|dkr`tj| jtj	d�}nt
| |ƒ |}| jdkrŒtj|| | j|d� |dkrÌ| jdkr´tj| j| j|d	�}ntj| j| j| j|d
�}|S )a^  Copy ``self`` to ``ary`` or create a new Numpy ndarray
        if ``ary`` is ``None``.

        If a CUDA ``stream`` is given, then the transfer will be made
        asynchronously as part as the given stream.  Otherwise, the transfer is
        synchronous: the function returns after the copy is finished.

        Always returns the host array.

        Example::

            import numpy as np
            from numba import cuda

            arr = np.arange(1000)
            d_arr = cuda.to_device(arr)

            my_kernel[100, 100](d_arr)

            result_array = d_arr.copy_to_host()
        c                 s   s   | ]}|d k V  qdS )r   Nr   )Ú.0Úsr   r   r   Ú	<genexpr>	  s     z1DeviceNDArrayBase.copy_to_host.<locals>.<genexpr>z2D->H copy not implemented for negative strides: {}r   zNegative memory sizeN©r   r    rS   )r   r    Úbuffer)r   r    r   rm   )Úanyr   rP   Úformatr5   ÚAssertionErrorrT   r$   ÚemptyÚbyterd   r4   Údevice_to_hostr!   Úndarrayr   r    )r<   rg   r;   rR   Úhostaryr   r   r   Úcopy_to_hostò   s.    


ÿ
ÿ ÿzDeviceNDArrayBase.copy_to_hostc                 c   s¼   |   |¡}| jdkrtdƒ‚| jd | jjkr6tdƒ‚tt t	| j
ƒ| ¡ƒ}| j}| jj}t|ƒD ]R}|| }t|| | j
ƒ}|| f}	| j || || ¡}
t|	|| j||
d�V  qddS )zžSplit the array into equal partition of the `section` size.
        If the array cannot be equally divided, the last section will be
        smaller.
        r)   zonly support 1d arrayr   zonly support unit stride©r    r;   r:   N)rT   r,   r'   r   r    r.   r%   ÚmathÚceilÚfloatr!   rO   Úminr:   ÚviewÚDeviceNDArray)r<   Úsectionr;   Znsectr   r.   ÚiÚbeginÚendr   r:   r   r   r   Úsplit!  s     


ÿzDeviceNDArrayBase.splitc                 C   s   | j S )zEReturns a device memory object that is used as the argument.
        )r:   rL   r   r   r   Úas_cuda_arg6  s    zDeviceNDArrayBase.as_cuda_argc                 C   s0   t  ¡  | j¡}t| j| j| jd�}t||d�S )zÌ
        Returns a *IpcArrayHandle* object that is safe to serialize and transfer
        to another process to share the local allocation.

        Note: this feature is only available on Linux.
        )r   r   r    )Ú
ipc_handleÚ
array_desc)	r   r6   Úget_ipc_handler:   Údictr   r   r    ÚIpcArrayHandle)r<   ZipchÚdescr   r   r   r†   ;  s    z DeviceNDArrayBase.get_ipc_handlec                 C   s2   | j j|d�\}}t|j|j| j|  |¡| jd�S )a(  
        Remove axes of size one from the array shape.

        Parameters
        ----------
        axis : None or int or tuple of ints, optional
            Subset of dimensions to remove. A `ValueError` is raised if an axis
            with size greater than one is selected. If `None`, all axes with
            size one are removed.
        stream : cuda stream or 0, optional
            Default stream for the returned view of the array.

        Returns
        -------
        DeviceNDArray
            Squeezed view into the array.

        )Úaxis©r   r   r    r;   r:   )r/   Úsqueezer}   r   r   r    rT   r:   )r<   rŠ   r;   Z	new_dummyÚ_r   r   r   rŒ   F  s    ûzDeviceNDArrayBase.squeezec                 C   sŒ   t  |¡}t| jƒ}t| jƒ}| jj|jkrv|  ¡ s<tdƒ‚t|d | jj |jƒ\|d< }|dkrltdƒ‚|j|d< t	|||| j
| jd�S )zeReturns a new object by reinterpretting the dtype without making a
        copy of the data.
        zHTo change to a dtype of a different size, the array must be C-contiguouséÿÿÿÿr   zuWhen changing to a larger dtype, its size must be a divisor of the total size in bytes of the last axis of the array.r‹   )r$   r    Úlistr   r   r.   Úis_c_contiguousr'   Údivmodr}   r;   r:   )r<   r    r   r   Úremr   r   r   r|   b  s0    


ÿþÿ
ûzDeviceNDArrayBase.viewc                 C   s   | j j| j S r   )r    r.   r!   rL   r   r   r   Únbytes‡  s    zDeviceNDArrayBase.nbytes)r   N)r   )N)r   )Nr   )r   )Nr   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__Z__cuda_memory__r   r>   ÚpropertyrF   rI   rM   rK   rT   r^   rB   r   Úrequire_contextrh   rv   r‚   rƒ   r†   rŒ   r|   r“   r   r   r   r   r(   >   s4   
3





.

%r(   c                       sŠ   e Zd ZdZd‡ fdd„	Zedd„ ƒZedd	„ ƒZej	d
d„ ƒZ
ej	ddd„ƒZddd„Zej	dd„ ƒZej	ddd„ƒZddd„Z‡  ZS )ÚDeviceRecordz
    An on-GPU record type
    r   Nc                    s$   d}d}t t| ƒ |||||¡ d S ©Nr   )Úsuperrš   r>   )r<   r    r;   r:   r   r   ©Ú	__class__r   r   r>   “  s
    ÿzDeviceRecord.__init__c                 C   s   t | jjƒS ©zÿ
        For `numpy.ndarray` compatibility. Ideally this would return a
        `np.core.multiarray.flagsobj`, but that needs to be constructed
        with an existing `numpy.ndarray` (as the C- and F- contiguous flags
        aren't writeable).
        ©r‡   r/   r[   rL   r   r   r   r[   ™  s    zDeviceRecord.flagsc                 C   s   t  | j¡S )rU   )r
   r\   r    rL   r   r   r   r^   £  s    zDeviceRecord._numba_type_c                 C   s
   |   |¡S r   ©Ú_do_getitem©r<   Úitemr   r   r   Ú__getitem__«  s    zDeviceRecord.__getitem__c                 C   s   |   ||¡S ©z0Do `__getitem__(item)` with CUDA stream
        r¡   ©r<   r¤   r;   r   r   r   Úgetitem¯  s    zDeviceRecord.getitemc           
      C   s¤   |   |¡}| jj| \}}| j |¡}|jdkrr|jd k	rHt|||d�S tj	d|d�}t
j|||j|d� |d S t|jd |jd dƒ\}}}	t|||	||d�S d S )	Nr   rw   r)   ©r    ©ÚdstÚsrcr!   r;   r   rW   ©r   r   r    r:   r;   )rT   r    Úfieldsr:   r|   r   Únamesrš   r$   rq   r4   rs   r.   r   Zsubdtyper}   )
r<   r¤   r;   r   ÚoffsetÚnewdataru   r   r   r    r   r   r   r¢   µ  s2    


ÿþ þÿ þzDeviceRecord._do_getitemc                 C   s   |   ||¡S r   ©Ú_do_setitem©r<   ÚkeyrC   r   r   r   Ú__setitem__Í  s    zDeviceRecord.__setitem__c                 C   s   | j |||d�S ©z6Do `__setitem__(key, value)` with CUDA stream
        rS   r²   ©r<   rµ   rC   r;   r   r   r   ÚsetitemÑ  s    zDeviceRecord.setitemc                 C   sŽ   |   |¡}| }|r$t ¡ }| ¡ }| jj| \}}| j |¡}t| ƒ|||d�}	t	|	j |¡|d�\}
}t
 |	|
|
jj|¡ |rŠ| ¡  d S )Nrw   rS   )rT   r   r6   Úget_default_streamr    r®   r:   r|   ÚtypeÚauto_devicer4   re   r.   Úsynchronize)r<   rµ   rC   r;   ÚsynchronousÚctxr   r°   r±   ÚlhsÚrhsr�   r   r   r   r³   ×  s    
zDeviceRecord._do_setitem)r   N)r   )r   )r   )r   )r”   r•   r–   r—   r>   r˜   r[   r^   r   r™   r¥   r¨   r¢   r¶   r¹   r³   Ú__classcell__r   r   r�   r   rš   �  s    
	



rš   c                    s>   ddl m‰  ˆdkr&ˆ jdd„ ƒ}|S ˆ j‡ ‡fdd„ƒ}|S )zÙ
    A separate method so we don't need to compile code every assignment (!).

    :param ndim: We need to have static array sizes for cuda.local.array, so
        bake in the number of dimensions into the kernel
    r   )Úcudac                 S   s   |d | d< d S r›   r   )rÀ   rÁ   r   r   r   Úkernel  s    z_assign_kernel.<locals>.kernelc                    sÐ   ˆ   d¡}d}t| jƒD ]}|| j| 9 }q||kr8d S ˆ jjdˆftjd�}tˆd ddƒD ]L}|| j|  |d|f< || j|  |j| dk |d|f< || j|  }q^|t|d ˆƒ | t|d ˆƒ< d S )Nr)   rN   rl   rŽ   r   )	ÚgridrO   r,   r   Úlocalrf   r   Úint64r   )rÀ   rÁ   ÚlocationÚ
n_elementsr   Úidx©rÃ   r,   r   r   rÄ     s    
þ$)ÚnumbarÃ   Zjit)r,   rÄ   r   rË   r   Ú_assign_kernelö  s    
rÍ   c                   @   s    e Zd ZdZdd„ Zedd„ ƒZdd„ Zdd	d
„Zdd„ Z	dd„ Z
d dd„Zejdd„ ƒZejd!dd„ƒZd"dd„Zejdd„ ƒZejd#dd„ƒZd$dd„ZdS )%r}   z
    An on-GPU array type
    c                 C   s   | j jS )zA
        Return true if the array is Fortran-contiguous.
        )r/   Zis_f_contigrL   r   r   r   Úis_f_contiguous&  s    zDeviceNDArray.is_f_contiguousc                 C   s   t | jjƒS rŸ   r    rL   r   r   r   r[   ,  s    zDeviceNDArray.flagsc                 C   s   | j jS )z;
        Return true if the array is C-contiguous.
        )r/   Zis_c_contigrL   r   r   r   r�   6  s    zDeviceNDArray.is_c_contiguousNc                 C   s"   |r|   ¡  |¡S |   ¡  ¡ S dS )zE
        :return: an `numpy.ndarray`, so copies to the host.
        N)rv   Ú	__array__)r<   r    r   r   r   rÏ   <  s    zDeviceNDArray.__array__c                 C   s
   | j d S )Nr   )r   rL   r   r   r   Ú__len__E  s    zDeviceNDArray.__len__c                 O   s”   t |ƒdkr&t|d ttfƒr&|d }t| ƒ}|| jkrP|| j| j| j| jd�S | j	j
||Ž\}}|| j	jgkrˆ||j|j| j| jd�S tdƒ‚dS )z¶
        Reshape the array without changing its contents, similarly to
        :meth:`numpy.ndarray.reshape`. Example::

            d_arr = d_arr.reshape(20, 50, order='F')
        r)   r   )r   r   r    r:   úoperation requires copyingN)r+   r   r#   r�   r»   r   r   r    r:   r/   ÚreshapeÚextentrP   )r<   ZnewshapeÚkwsÚclsÚnewarrÚextentsr   r   r   rÒ   H  s    

 ÿ
 ÿzDeviceNDArray.reshaperW   r   c                 C   sX   |   |¡}t| ƒ}| jj|d�\}}|| jjgkrL||j|j| j| j|d�S t	dƒ‚dS )z¹
        Flattens a contiguous array without changing its contents, similar to
        :meth:`numpy.ndarray.ravel`. If the array is not contiguous, raises an
        exception.
        )r_   r­   rÑ   N)
rT   r»   r/   ÚravelrÓ   r   r   r    r:   rP   )r<   r_   r;   rÕ   rÖ   r×   r   r   r   rØ   `  s    

 þzDeviceNDArray.ravelc                 C   s
   |   |¡S r   r¡   r£   r   r   r   r¥   r  s    zDeviceNDArray.__getitem__c                 C   s   |   ||¡S r¦   r¡   r§   r   r   r   r¨   v  s    zDeviceNDArray.getitemc                 C   sÚ   |   |¡}| j |¡}t| ¡ ƒ}t| ƒ}t|ƒdkr°| jj|d Ž }|j	s–| j
jd k	rht| j
||d�S tjd| j
d�}tj||| jj|d� |d S ||j|j| j
||d�S n&| jj|jŽ }||j|j| j
||d�S d S )Nr)   r   rw   r©   rª   r­   )rT   r/   r¥   r�   Ziter_contiguous_extentr»   r+   r:   r|   Zis_arrayr    r¯   rš   r$   rq   r4   rs   r.   r   r   rÓ   )r<   r¤   r;   Úarrr×   rÕ   r±   ru   r   r   r   r¢   |  s8    
ÿþ
  ÿ
  ÿzDeviceNDArray._do_getitemc                 C   s   |   ||¡S r   r²   r´   r   r   r   r¶   ™  s    zDeviceNDArray.__setitem__c                 C   s   | j |||d�S r·   r²   r¸   r   r   r   r¹   �  s    zDeviceNDArray.setitemc                 C   s\  |   |¡}| }|r$t ¡ }| ¡ }| j |¡}| jj|jŽ }t	|t
jƒrTd}d}	n|j}|j}	t| ƒ||	| j||d�}
t||dd�\}}|j|
jkrªtd|j|
jf ƒ‚tj|
jtjd�}|j||
j|j d …< |j|Ž }tt|
j|jƒƒD ].\}\}}|dkrî||krîtd|||f ƒ‚qît tj|
jd¡}t|
jƒj||d	�|
|ƒ |�rX| ¡  d S )
Nr   r­   T)r;   Úuser_explicitz$Can't assign %s-D array to %s-D selfr©   r)   zCCan't copy sequence with size %d to array axis %d with dimension %drS   ) rT   r   r6   rº   r/   r¥   r:   r|   rÓ   r   r	   ZElementr   r   r»   r    r¼   r,   r'   r$   ZonesrÇ   rÒ   Ú	enumerateÚzipr0   r1   r2   r3   rÍ   Úforallr½   )r<   rµ   rC   r;   r¾   r¿   rÙ   r±   r   r   rÀ   rÁ   r�   Z	rhs_shaper   ÚlÚrrÉ   r   r   r   r³   £  sJ    
û	þ
ÿzDeviceNDArray._do_setitem)N)rW   r   )r   )r   )r   )r   )r”   r•   r–   r—   rÎ   r˜   r[   r�   rÏ   rÐ   rÒ   rØ   r   r™   r¥   r¨   r¢   r¶   r¹   r³   r   r   r   r   r}   "  s&   
	
	



r}   c                   @   s8   e Zd ZdZdd„ Zdd„ Zdd„ Zdd	„ Zd
d„ ZdS )rˆ   a"  
    An IPC array handle that can be serialized and transfer to another process
    in the same machine for share a GPU allocation.

    On the destination process, use the *.open()* method to creates a new
    *DeviceNDArray* object that shares the allocation from the original process.
    To release the resources, call the *.close()* method.  After that, the
    destination can no longer use the shared array object.  (Note: the
    underlying weakref to the resource is now dead.)

    This object implements the context-manager interface that calls the
    *.open()* and *.close()* method automatically::

        with the_ipc_array_handle as ipc_array:
            # use ipc_array here as a normal gpu array object
            some_code(ipc_array)
        # ipc_array is dead at this point
    c                 C   s   || _ || _d S r   )Ú_array_descÚ_ipc_handle)r<   r„   r…   r   r   r   r>   î  s    zIpcArrayHandle.__init__c                 C   s$   | j  t ¡ ¡}tf d|i| j—ŽS )z˜
        Returns a new *DeviceNDArray* that shares the allocation from the
        original process.  Must not be used on the original process.
        r:   )rá   Úopenr   r6   r}   rà   )r<   Zdptrr   r   r   râ   ò  s    zIpcArrayHandle.openc                 C   s   | j  ¡  dS )z5
        Closes the IPC handle to the array.
        N)rá   ÚcloserL   r   r   r   rã   ú  s    zIpcArrayHandle.closec                 C   s   |   ¡ S r   )râ   rL   r   r   r   Ú	__enter__   s    zIpcArrayHandle.__enter__c                 C   s   |   ¡  d S r   )rã   )r<   r»   rC   Ú	tracebackr   r   r   Ú__exit__  s    zIpcArrayHandle.__exit__N)	r”   r•   r–   r—   r>   râ   rã   rä   ræ   r   r   r   r   rˆ   Û  s   rˆ   c                   @   s   e Zd ZdZddd„ZdS )ÚMappedNDArrayz4
    A host array that uses CUDA mapped memory.
    r   c                 C   s   || _ || _d S r   ©r:   r;   ©r<   r:   r;   r   r   r   Údevice_setup  s    zMappedNDArray.device_setupN)r   ©r”   r•   r–   r—   rê   r   r   r   r   rç     s   rç   c                   @   s   e Zd ZdZddd„ZdS )ÚManagedNDArrayz5
    A host array that uses CUDA managed memory.
    r   c                 C   s   || _ || _d S r   rè   ré   r   r   r   rê     s    zManagedNDArray.device_setupN)r   rë   r   r   r   r   rì     s   rì   c                 C   s   t | j| j| j||d�S )z/Create a DeviceNDArray object that is like ary.©r;   r:   )r}   r   r   r    )rg   r;   r:   r   r   r   Úfrom_array_like  s    ÿrî   c                 C   s   t | j||d�S )z.Create a DeviceRecord object that is like rec.rí   )rš   r    )Zrecr;   r:   r   r   r   Úfrom_record_like!  s    rï   c                 C   sF   | j r| js| S g }| j D ]}| |dkr.dntdƒ¡ q| t|ƒ S )aG  
    Extract the repeated core of a broadcast array.

    Broadcast arrays are by definition non-contiguous due to repeated
    dimensions, i.e., dimensions with stride 0. In order to ascertain memory
    contiguity and copy the underlying data from such arrays, we must create
    a view without the repeated dimensions.

    r   N)r   r!   ÚappendÚslicer#   )rg   Z
core_indexÚstrider   r   r   rb   &  s    

rb   c                 C   sR   | j j}tt| jƒt| jƒƒD ].\}}|dkr|dkr||krD dS ||9 }qdS )zÓ
    Returns True iff `ary` is C-style contiguous while ignoring
    broadcasted and 1-sized dimensions.
    As opposed to array_core(), it does not call require_context(),
    which can be quite expensive.
    r)   r   FT)r    r.   rÜ   Úreversedr   r   )rg   r!   r   rò   r   r   r   rD   8  s    
rD   z™Array contains non-contiguous buffer and cannot be transferred as a single memory region. Please ensure contiguous buffer with numpy .ascontiguousarray()c                 C   s(   t | ƒ}|jd s$|jd s$ttƒ‚d S )NrV   rX   )rb   r[   r'   Úerrmsg_contiguous_buffer)rg   Úcorer   r   r   ra   N  s    ra   TFc                 C   s¸   t  | ¡r| dfS t| dƒr,tj | ¡dfS t| tjƒrFt	| |d�}n$tj
| ddd�} t| ƒ t| |d�}|r¬tjrž|sžt| tƒsžt| tjƒržd}tt|ƒƒ |j| |d� |dfS dS )zº
    Create a DeviceRecord or DeviceArray like obj and optionally copy data from
    host to device. If obj already represents device memory, it is returned and
    no copy is made.
    FrF   rS   T)rG   r`   zGHost array used in CUDA kernel will incur copy overhead to/from device.N)r4   rc   r   rÌ   rÃ   Zas_cuda_arrayr   r$   Úvoidrï   rf   ra   rî   r   ZCUDA_WARN_ON_IMPLICIT_COPYr}   rt   r   r   rh   )r   r;   rG   rÚ   ZdevobjrR   r   r   r   r¼   T  s2    

ýÿþ
ýr¼   c                 C   s|   |   ¡ |  ¡  }}| j|jkr2td| j|jf ƒ‚|j|jkrRtd| j|jf ƒ‚| jrx|j|jkrxtd| j|jf ƒ‚d S )Nzincompatible dtype: %s vs. %szincompatible shape: %s vs. %szincompatible strides: %s vs. %s)rŒ   r    Ú	TypeErrorr   r'   r!   r   )Zary1Zary2Zary1sqZary2sqr   r   r   rd   {  s    
ÿ
ÿ
ÿrd   )r   N)r   N)r   TF)6r—   rx   r0   r2   rG   Úctypesr   Únumpyr$   rÌ   r   Znumba.cuda.cudadrvr   r   r4   Z
numba.corer   r   Znumba.np.unsafe.ndarrayr   Z
numba.miscr	   Znumba.npr
   Znumba.cuda.api_utilr   Znumba.core.errorsr   Úwarningsr   r   r   r   r   r&   r"   ZDeviceArrayr(   rš   rÍ   r}   Úobjectrˆ   rt   rç   rì   rî   rï   rb   rD   rô   ra   r¼   rd   r   r   r   r   Ú<module>   sV     Sg
+ :,




'