o
    ôT·j‹D  ã                   @  st  d dl mZ d dlmZmZ d dlZd dlmZ d dl	m
Z
 d dlmZ d dlmZ d dlmZ d dlZd d	lmZmZ d d
lmZ d dlmZmZ d dlmZmZmZmZ d dlm Z m!Z!m"Z" erhd dlm#Z# ej$ej%ej&ej'ej(ej)ej)dœZ*ej&ej+dfej)ej,e
fej$ej-dfej%ej-dfej'ej-dfej.ej,dfej(ej/d fiZ0ej-dej+dej,diZ1G dd„ deƒZ2dS )é    )Úannotations)ÚTYPE_CHECKINGÚAnyN)Úinfer_dtype)ÚiNaT)ÚNoBufferPresent)Úcache_readonly)ÚBaseMaskedDtype)Ú
ArrowDtypeÚDatetimeTZDtype)Úis_string_dtype)ÚPandasBufferÚPandasBufferPyarrow)ÚColumnÚColumnBuffersÚColumnNullTypeÚ	DtypeKind)ÚArrowCTypesÚ
EndiannessÚdtype_to_arrow_c_fmt)ÚBuffer)ÚiÚuÚfÚbÚUÚMÚméÿÿÿÿzThis column is non-nullablezThis column uses NaN as nullz!This column uses a sentinel valuec                   @  s¾   e Zd ZdZd1d2d	d
„Zd3dd„Zed3dd„ƒZed4dd„ƒZ	d4dd„Z
edd„ ƒZedd„ ƒZed3dd„ƒZed5dd„ƒZd3dd„Zd6d7d#d$„Zd8d&d'„Zd9d)d*„Zd:d,d-„Zd;d/d0„Zd S )<ÚPandasColumnaö  
    A column object, with only the methods and properties required by the
    interchange protocol defined.
    A column can contain one or more chunks. Each chunk can contain up to three
    buffers - a data buffer, a mask buffer (depending on null representation),
    and an offsets buffer (if variable-size binary; e.g., variable-length
    strings).
    Note: this Column object can only be produced by ``__dataframe__``, so
          doesn't need its own version or ``__column__`` protocol.
    TÚcolumnú	pd.SeriesÚ
allow_copyÚboolÚreturnÚNonec                 C  sN   t |tjƒrtd|j› d�ƒ‚t |tjƒstdt|ƒ› d�ƒ‚|| _|| _	dS )zu
        Note: doesn't deal with extension arrays yet, just assume a regular
        Series/ndarray for now.
        z·Expected a Series, got a DataFrame. This likely happened because you called __dataframe__ on a DataFrame which, after converting column names to string, resulted in duplicated names: zD. Please rename these columns before using the interchange protocol.zColumns of type ú not handled yetN)
Ú
isinstanceÚpdÚ	DataFrameÚ	TypeErrorÚcolumnsÚSeriesÚNotImplementedErrorÚtypeÚ_colÚ_allow_copy)Úselfr    r"   © r2   úa/home/dinkstrade/pdmp-scanner/venv/lib/python3.10/site-packages/pandas/core/interchange/column.pyÚ__init__T   s   ýÿ
zPandasColumn.__init__Úintc                 C  s   | j jS )z2
        Size of the column, in elements.
        )r/   Úsize©r1   r2   r2   r3   r6   h   s   zPandasColumn.sizec                 C  ó   dS )z7
        Offset of first element. Always zero.
        r   r2   r7   r2   r2   r3   Úoffsetn   s   zPandasColumn.offsetútuple[DtypeKind, int, str, str]c                 C  s~   | j j}t|tjƒr!| j jj}|  |j¡\}}}}tj	||t
jfS t|ƒr:t| j ƒdv r6tjdt|ƒt
jfS tdƒ‚|  |¡S )N)ÚstringÚemptyé   z.Non-string object dtypes are not supported yet)r/   Údtyper'   r(   ÚCategoricalDtypeÚvaluesÚcodesÚ_dtype_from_pandasdtyper   ÚCATEGORICALr   ÚNATIVEr   r   ÚSTRINGr   r-   )r1   r>   rA   Ú_ÚbitwidthÚc_arrow_dtype_f_strr2   r2   r3   r>   v   s.   

ûüü
zPandasColumn.dtypec                 C  s–   t  |jd¡}|du rtd|› d�ƒ‚t|tƒr|jj}nt|tƒr'|j	j}nt|t
ƒr1|jj}n|j}|dkr@||jtj|fS ||jd t|ƒ|fS )z/
        See `self.dtype` for details.
        Nú
Data type z& not supported by interchange protocolzbool[pyarrow]r=   )Ú	_NP_KINDSÚgetÚkindÚ
ValueErrorr'   r
   Únumpy_dtypeÚ	byteorderr   Úbaser	   Úitemsizer   ÚBOOLr   )r1   r>   rL   rO   r2   r2   r3   rB   ”   s"   





üz$PandasColumn._dtype_from_pandasdtypec                 C  s:   | j d tjkstdƒ‚| jjjdtt 	| jjj
¡ƒdœS )a:  
        If the dtype is categorical, there are two options:
        - There are only values in the data buffer.
        - There is a separate non-categorical Column encoding for categorical values.

        Raises TypeError if the dtype is not categorical

        Content of returned dict:
            - "is_ordered" : bool, whether the ordering of dictionary indices is
                             semantically meaningful.
            - "is_dictionary" : bool, whether a dictionary-style mapping of
                                categorical values to other objects exists
            - "categories" : Column representing the (implicit) mapping of indices to
                             category values (e.g. an array of cat1, cat2, ...).
                             None if not a dictionary-style categorical.
        r   zCdescribe_categorical only works on a column with categorical dtype!T)Ú
is_orderedÚis_dictionaryÚ
categories)r>   r   rC   r*   r/   ÚcatÚorderedr   r(   r,   rU   r7   r2   r2   r3   Údescribe_categoricalµ   s   ÿýz!PandasColumn.describe_categoricalc                 C  sž   t | jjtƒrtj}d}||fS t | jjtƒr/| jjjj	d  
¡ d d u r*tjd fS tjdfS | jd }zt| \}}W ||fS  tyN   td|› d�ƒ‚w )Né   r   rI   z not yet supported)r'   r/   r>   r	   r   ÚUSE_BYTEMASKr
   ÚarrayÚ	_pa_arrayÚchunksÚbuffersÚNON_NULLABLEÚUSE_BITMASKÚ_NULL_DESCRIPTIONÚKeyErrorr-   )r1   Úcolumn_null_dtypeÚ
null_valuerL   ÚnullÚvaluer2   r2   r3   Údescribe_nullÒ   s   


ýÿzPandasColumn.describe_nullc                 C  s   | j  ¡  ¡  ¡ S )zB
        Number of null elements. Should always be known.
        )r/   ÚisnaÚsumÚitemr7   r2   r2   r3   Ú
null_countæ   s   zPandasColumn.null_countúdict[str, pd.Index]c                 C  s   d| j jiS )z8
        Store specific metadata of the column.
        zpandas.index)r/   Úindexr7   r2   r2   r3   Úmetadataí   s   zPandasColumn.metadatac                 C  r8   )zE
        Return the number of chunks the column consists of.
        rY   r2   r7   r2   r2   r3   Ú
num_chunksô   s   zPandasColumn.num_chunksNÚn_chunksú
int | Nonec                 c  sv   � |r6|dkr6t | jƒ}|| }|| dkr|d7 }td|| |ƒD ]}t| jj||| … | jƒV  q"dS | V  dS )zy
        Return an iterator yielding the chunks.
        See `DataFrame.get_chunks` for details on ``n_chunks``.
        rY   r   N)Úlenr/   Úranger   Úilocr0   )r1   rp   r6   ÚstepÚstartr2   r2   r3   Ú
get_chunksú   s   €
ÿÿ
zPandasColumn.get_chunksr   c                 C  s\   |   ¡ dddœ}z|  ¡ |d< W n	 ty   Y nw z	|  ¡ |d< W |S  ty-   Y |S w )a`  
        Return a dictionary containing the underlying buffers.
        The returned dictionary has the following contents:
            - "data": a two-element tuple whose first element is a buffer
                      containing the data and whose second element is the data
                      buffer's associated dtype.
            - "validity": a two-element tuple whose first element is a buffer
                          containing mask values indicating missing data and
                          whose second element is the mask value buffer's
                          associated dtype. None if the null representation is
                          not a bit or byte mask.
            - "offsets": a two-element tuple whose first element is a buffer
                         containing the offset values for variable-size binary
                         data (e.g., variable-length strings) and whose second
                         element is the offsets buffer's associated dtype. None
                         if the data buffer does not have an associated offsets
                         buffer.
        N)ÚdataÚvalidityÚoffsetsry   rz   )Ú_get_data_bufferÚ_get_validity_bufferr   Ú_get_offsets_buffer)r1   r^   r2   r2   r3   Úget_buffers  s    ýÿýýzPandasColumn.get_buffersú.tuple[Buffer, tuple[DtypeKind, int, str, str]]c           	      C  sˆ  | j d tjtjtjtjtjfv ri| j }| j d tjkr/t| j d ƒdkr/| jj	 
d¡ ¡ }n/| jj}t| jj tƒr>|j}n t| jj tƒr[|jjd }t| ¡ d t|ƒd�}||fS |j}t|| jd�}||fS | j d tjkr‡| jjj}t|| jd�}|  |j ¡}||fS | j d tjkrº| j ¡ }tƒ }|D ]}t|tƒr©| |j dd	�¡ q™tt!j"|d
d�ƒ}| j }||fS t#d| jj › d�ƒ‚)zZ
        Return the buffer containing the data and the buffer's associated dtype.
        r   é   é   NrY   ©Úlength)r"   úutf-8©ÚencodingÚuint8)r>   rI   r&   )$r>   r   ÚINTÚUINTÚFLOATrR   ÚDATETIMErr   r/   ÚdtÚ
tz_convertÚto_numpyr[   r'   r	   Ú_datar
   r\   r]   r   r^   Ú_ndarrayr   r0   rC   r@   Ú_codesrB   rE   Ú	bytearrayÚstrÚextendÚencodeÚnpÚ
frombufferr-   )	r1   r>   Únp_arrÚarrÚbufferrA   Úbufr   Úobjr2   r2   r3   r{   0  sN   û	"
þç
ë

€þzPandasColumn._get_data_bufferútuple[Buffer, Any] | Nonec                 C  s`  | j \}}t| jjtƒr7| jjjjd }tj	dt
j	tjf}| ¡ d du r'dS t| ¡ d t|ƒd�}||fS t| jjtƒrT| jjj}t|ƒ}tj	dt
j	tjf}||fS | jd tjkr˜| j ¡ }|dk}| }tjt|ƒftjd�}t|ƒD ]\}	}
t|
tƒr‚|n|||	< qwt|ƒ}tj	dt
j	tjf}||fS zt| › d�}W t|ƒ‚ ty¯   tdƒ‚w )	zÒ
        Return the buffer containing the mask values indicating missing data and
        the buffer's associated dtype.
        Raises NoBufferPresent if null representation is not a bit or byte mask.
        r   rY   Nr‚   r=   ©Úshaper>   z! so does not have a separate maskzSee self.describe_null)rg   r'   r/   r>   r
   r[   r\   r]   r   rR   r   r   rD   r^   r   rr   r	   Ú_maskr   rE   rŽ   r–   ÚzerosÚbool_Ú	enumerater“   Ú_NO_VALIDITY_BUFFERrb   r-   r   )r1   re   Úinvalidr™   r>   rš   Úmaskr›   Úvalidr   rœ   Úmsgr2   r2   r3   r|   n  s@   

þ

üþz!PandasColumn._get_validity_bufferútuple[PandasBuffer, Any]c           	      C  s¢   | j d tjkrM| j ¡ }d}tjt|ƒd ftjd�}t	|ƒD ]\}}t
|tƒr5|jdd�}|t|ƒ7 }|||d < q t|ƒ}tjdtjtjf}||fS tdƒ‚)a  
        Return the buffer containing the offset values for variable-size binary
        data (e.g., variable-length strings) and the buffer's associated dtype.
        Raises NoBufferPresent if the data buffer does not have an associated
        offsets buffer.
        r   rY   rž   r„   r…   é@   zJThis column has a fixed-length dtype so it does not have an offsets buffer)r>   r   rE   r/   rŽ   r–   r¡   rr   Úint64r£   r'   r“   r•   r   rˆ   r   ÚINT64r   rD   r   )	r1   r@   Úptrrz   r   Úvr   rš   r>   r2   r2   r3   r}   §  s&   

üûÿz PandasColumn._get_offsets_buffer)T)r    r!   r"   r#   r$   r%   )r$   r5   )r$   r:   )r$   rl   )N)rp   rq   )r$   r   )r$   r   )r$   r�   )r$   r©   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__r4   r6   Úpropertyr9   r   r>   rB   rX   rg   rk   rn   ro   rw   r~   r{   r|   r}   r2   r2   r2   r3   r   H   s.    

!




%
>9r   )3Ú
__future__r   Útypingr   r   Únumpyr–   Úpandas._libs.libr   Úpandas._libs.tslibsr   Úpandas.errorsr   Úpandas.util._decoratorsr   Úpandas.core.dtypes.dtypesr	   Úpandasr(   r
   r   Úpandas.api.typesr   Úpandas.core.interchange.bufferr   r   Ú*pandas.core.interchange.dataframe_protocolr   r   r   r   Úpandas.core.interchange.utilsr   r   r   r   rˆ   r‰   rŠ   rR   rE   r‹   rJ   ÚUSE_NANÚUSE_SENTINELr_   rC   rZ   ra   r¤   r   r2   r2   r2   r3   Ú<module>   sJ    ùöý