
    IZj<                      d dl mZ d dlZd dlZd dlZd dlZd dlmZmZ d dl	m
Z
 d dlmZmZ d dlmZ d dlmZmZmZmZmZmZmZmZmZmZmZ d dlmZ d d	lmZmZ d
dl m!Z! d dl"m#Z# d dl$m%Z% d
dl&m'Z'm(Z(m)Z)m*Z*m+Z,m-Z. d dl/Z0d dl1Z/d dl2m3Z4 d dl5m6Z7 d dl8Z9d
dl:m;Z;m<Z<m=Z= d
dl>m?Z?m@Z@ d
dlAmBZBmCZCmDZDmEZEmFZFmGZGmHZHmIZImJZJmKZKmLZL d
dlMmNZN d
dlOmPZPmQZQ d
dlRmSZSmTZTmUZUmVZVmWZWmXZXmYZYmZZZm[Z[m\Z\m]Z]m^Z^m_Z_ d
dl`maZambZbmcZcmdZdmeZemfZf d
dlAmgZg ed         ZhdZidZjdd Zkdd#Zldd$dd'ZmerFd
d(lnmoZo d
d)lpmqZrmsZsmtZtmuZumvZvmwZwmxZxmyZymzZzm{Z{m|Z|m}Z}m~Z~ d
d*lAmZ d dl+Z+d dlZd
d+lmZmZmZmZmZmZmZmZ 	 ddd/Zdd2Z	 	 	 	 dd5d6dd@Z	 dddCZddGZddIZ	 	 	 dddKZddLZdM Z	 dddNddOZddRZddSZdT Z G dU dVe          Zq G dW dXeq          Z	 	 	 	 dddZZdd]Zdd_ZddeZ	 	 	 	 dddlZddpZddrZddtZddvZddxZddzZdd|Zdd}Zdd~Z G d d          Ze
 G d d                      Ze
 G d d                      Ze
 G d d                      Ze
 G d d                      Z G d d          Z G d d          ZdS )    )annotationsN)ABCabstractmethod)	dataclass)datetime	timedelta)cached_property)TYPE_CHECKINGAnyCallableDictIterableListLiteralOptionalTupleUnionoverload)urlparse)_register_optional_convertersto_scannable   )__version__)peek_reader)LOOP)_check_for_hugging_face_check_for_lance_check_for_pandaslancepandaspolars)DATAVECVECTOR_COLUMN_NAME)EmbeddingFunctionConfigEmbeddingFunctionRegistry)BTreeIvfFlatIvfPqIvfSqBitmapIvfRq	LabelListHnswPqHnswSqHnswFlatFTS)LanceMergeInsertBuilder)
LanceModelmodel_to_dict)AsyncFTSQueryAsyncHybridQuery
AsyncQueryAsyncTakeQueryAsyncVectorQueryFullTextQueryLanceEmptyQueryBuilderLanceFtsQueryBuilderLanceHybridQueryBuilderLanceQueryBuilderLanceVectorQueryBuilderLanceTakeQueryBuilderQuery)add_notefs_from_uriget_uri_schemeinfer_vector_column_namejoin_urivalue_to_sql)lang_mapping)lazybytesdescriptions)jiebalindera)zunknown base tokenizerzInvalid directory path:zFailed to load JiebazFailed to load tokenizer configz&Failed to initialize default tokenizer	exceptionBaseExceptionnotestrreturnNonec                    t          | dd          pd}| j        r-t          | j        d         t                    r| j        d         nd}||vr||vrt	          | |           d S d S d S )N	__notes__ r    )getattrargs
isinstancerQ   rB   )rN   rP   existing_notesmessages       Y/Users/jameslopez/projects/MentorCore/.venv/lib/python3.11/site-packages/lancedb/table.py_add_unique_noter^   f   s    YR88>BN >	():C@@		q 
 >!!d'&9&9D!!!!! "!&9&9    base_tokenizerboolc                D     t           fdt          D                       S )Nc              3  T   K   | ]"}|k    p                     | d           V  #dS )/N)
startswith).0prefixr`   s     r]   	<genexpr>z-_is_model_backed_tokenizer.<locals>.<genexpr>r   sU         	& KN$=$=lll$K$K     r_   )any _MODEL_BACKED_TOKENIZER_PREFIXES)r`   s   `r]   _is_model_backed_tokenizerrk   q   s;        6     r_   )languagerl   Optional[str]c               .   t          |           |?dv r;d                    t          j                              }t	          | d|            d S t          |          sd S t          fdt          D                       sd S t	          | d           d S )Nz"not support the requested languagez, zSupported languages: c              3      K   | ]}|v V  	d S NrV   )rf   markerr\   s     r]   rh   z,_maybe_add_fts_error_note.<locals>.<genexpr>   s(      NNVv NNNNNNr_   aK  Model-backed tokenizers such as 'jieba/default' and 'lindera/ipadic' require tokenizer models in Lance's language model home. Set LANCE_LANGUAGE_MODEL_HOME to override the default platform data directory under 'lance/language_models'. Expected layouts include '<model-home>/jieba/default/...' and '<model-home>/lindera/ipadic/...'.)rQ   joinrH   valuesr^   rk   ri   _MODEL_BACKED_TOKENIZER_ERRORS)rN   r`   rl   supported_langsr\   s       @r]   _maybe_add_fts_error_noterv   x   s     )nnG D O O))L$7$9$9::$MO$M$MNNN%n55 NNNN/MNNNNN 	-    r_   LanceDBConnection)TableOptimizeStatsCleanupStatsCompactionStatsTagAddColumnsResult	AddResultAlterColumnsResultDeleteResultDropColumnsResultLsmWriteSpecMergeResultUpdateResult)IndexConfig)	QueryTypeOnBadVectorsTypeAddMode
CreateModeVectorIndexTypeScalarIndexTypeBaseTokenizerTypeDistanceTypeschemaOptional[pa.Schema]pa.RecordBatchReaderc                `   ddl m} t          |           rt          | |j                  rC| j        j        }t          j        	                    || j
                                                  S t          | |j        j                  rt          | d           }d|j        vr9|                    t          j        dt          j                                        }t          j        	                    |t'          |                     S t          | t(                    rt+          d          t          | t,                    rt+          d          t          | t.                    rA| sD|t+          d          t          j                            | |                                          S t          | d         t(                    r^| d         j                                        }d | D             } t          j                            | |                                          S t          | d         t          j                  r1t          j        	                    |                                           S t          j                            |                                           S t=          |           rt          | t>          j                   rt          j        !                    | d	
          }|j"        j#        |j"        j#        ni }d |$                                D             }|%                    |                                          S t          | t          j                  r|                                 S t          | t          j                  r&t          j        	                    | j"        | g          S tM          |           r@t          | tN          j(                  r&| )                                                                S t          | t          j*        j                  r&| )                                                                S t          | t          j*        j+                  r|                                 S t          | t          j                  r| S tY          |           j-        .                    d          r6| j        j/        dk    r&| 0                                                                S tY          |           j-        .                    d          rH| j        j/        dk    r8| 1                                0                                                                S t          | td                    rtg          |           S ti          dtY          |            d          )Nr   )datasetssplitz6Cannot add a single LanceModel to a table. Use a list.z6Cannot add a single dictionary to a table. Use a list.z4Cannot create table from empty list without a schemar   c                ,    g | ]}t          |          S rV   )r4   )rf   ds     r]   
<listcomp>z(_into_pyarrow_reader.<locals>.<listcomp>   s     333M!$$333r_   F)preserve_indexc                &    i | ]\  }}|d k    ||S )s   pandasrV   rf   kvs      r]   
<dictcomp>z(_into_pyarrow_reader.<locals>.<dictcomp>   s#    @@@Ai1r_   r!   	DataFrame	LazyFramezUnknown data type z. Supported types: list of dicts, pandas DataFrame, polars DataFrame, pyarrow Table/RecordBatch, or Pydantic models. See https://docs.lancedb.com/tables/ for examples.)5lancedb.dependenciesr   r   rZ   Datasetfeaturesarrow_schemapaRecordBatchReaderfrom_batchesdata
to_batchesdataset_dictDatasetDict_schema_from_hfnamesappendfieldstring_to_batches_with_splitr3   
ValueErrordictlistry   from_pylist	to_reader	__class__to_arrow_schemaRecordBatchr   pdr   from_pandasr   metadataitemsreplace_schema_metadatar   r   LanceDatasetscannerdatasetScannertype
__module__re   __name__to_arrowcollectr   _iterator_to_reader	TypeError)r   r   r   tablemetas        r]   _into_pyarrow_readerr      s    .-----t$$ 
dH,-- 		]/F'44VTY=Q=Q=S=STTTh3?@@ 	$T400Ffl**rx'E'EFF'44.t44   $
## SQRRR$ SQRRR$ 4
 	I~ !WXXX8''V'<<FFHHH d1gz** 	:!W&6688F33d333D8''V'<<FFHHHQ00 	:8((..88:::8''--77999	4	 	  $
Zbl%C%C $
$$T%$@@(-(=(Iu|$$r@@@@@,,T22<<>>>	D"(	#	# 
~~	D".	)	) 
#00tfEEE	$		 
JtU5G$H$H 
||~~'')))	D"*,	-	- 
||~~'')))	D"*,	-	- 
~~	D".	/	/ 
T

((22
N#{22}}((***T

((22
N#{22||~~&&((22444	D(	#	# 
"4(((Ad A A A
 
 	
r_   r   r   c                     t          t                               j         fd}t          j                             |                      S )Nc               3  D  K   E d {V  D ]} t          |                                           }|j        k    rI	 |                              }n2# t          j        j        $ r t          d d| j                   w xY w|                                E d {V  d S )NzfInput iterator yielded a batch with schema that does not match the schema of other batches.
Expected:
z
Got:
)	r   read_allr   castr   libArrowInvalidr   r   )batchr   r   firstr   s     r]   genz _iterator_to_reader.<locals>.gen  s       	* 	*E2599BBDDE|v%%!JJv..EEv*   $E&,E E6;lE E   ''))))))))))	* 	*s   A/B)r   nextr   r   r   r   )r   r   r   r   s   ` @@r]   r   r      se     !d,,E\F* * * * * * * ,,VSSUU;;;r_   error        F)allow_subschema'DATA'target_schemar   Optional[dict]on_bad_vectorsr   
fill_valuefloatr   c               .   t          | |          }t          |||          }t          |||||          }|t          |          \  }}|r(|                    t          |j        |                    }t          |           t          |||          }|S )a  
    Handle input data, applying all standard transformations.

    This includes:

     * Converting the data to a PyArrow Table
     * Adding vector columns defined in the metadata
     * Adding embedding metadata into the schema
     * Casting the table to the target schema
     * Handling bad vectors

    Parameters
    ----------
    target_schema : Optional[pa.Schema], default None
        The schema to cast the table to. This is typically the schema of the table
        if it already exists. Otherwise it might be a user-requested schema.
    allow_subschema : bool, default False
        If True, the input table is allowed to omit columns from the target schema.
        The target schema will be filtered to only include columns that are present
        in the input table before casting.
    metadata : Optional[dict], default None
        The embedding metadata to add to the schema.
    on_bad_vectors : Literal["error", "drop", "fill", "null"], default "error"
        What to do if any of the vectors are not the same size or contains NaNs.
    fill_value : float, default 0.0
        The value to use when filling vectors. Only used if on_bad_vectors="fill".
        All entries in the vector will be set to this value.
    r   )r   r   r   r   )	r   _append_vector_columns_handle_bad_vectors_infer_target_schemawith_metadata_merge_metadatar   _validate_schema_cast_to_target_schema)r   r   r   r   r   r   readers          r]   _sanitize_datar     s    R "$66F#FMHMMMF !%#  F  4V < <v 
%33M2H==
 
 ]####FM?KKFMr_   r   	pa.Schemac                     j                             |d          r S t          t          t	           j                             t          t	          |                              }t          j         ||j                  |s/t                    t          |          k    rt          d          |rt                    t          |          k    rdt          t          t	           j                             t          t	                                        }t          j         ||j                   fd}t
          j
                             |                      S )NTcheck_metadatar   z>Input table has different number of columns than target schemac               3     K   D ]w} t           j                            | g                                                                        }|r.t           j                            |d         j                  V  xd S )Nr   r   )r   ry   r   r   r   r   from_arrayscolumns)r   cast_batchesr   reordered_schemas     r]   r   z#_cast_to_target_schema.<locals>.genu  s       	 	E %%ug..334DEEPPRR   n00 O+4D 1     	 	r_   )r   equals_align_field_typesr   iterr   r   lenr   _infer_subschemar   r   )r   r   r   fieldsr   r   s   `    @r]   r   r   \  s`    }M$?? T&-%8%8 9 94]@S@S;T;TUUFy-2HIII 
s#344M8J8JJJL
 
 	
  N3/00C4F4FFF!fm$$%%tD1A,B,B'C'C
 
 9Vm6LMMM	 	 	 	 	 	 ,,-=ssuuEEEr_   r   List[pa.Field]target_fieldsc           	     ,   g }| D ]t          fd|D             d          }|t          dj         d          t          j                            |j                  ret          j                            j                  r8t          j        t          j        j	        |j        j	                            }n|j        }nt          j        
                    |j                  r]t          j                  r@t          j        t          j        j        g|j        j        g          d                   }n|j        }nt          j                            |j                  r[t          j                  r?t          j        t          j        j        g|j        j        g          d                   }n|j        }nt          j                            |j                  rft          j                  rJt          j        t          j        j        g|j        j        g          d         |j        j                  }n|j        }n|j        }|                    t          j        j        |j        |j                             |S )zD
    Apply the data types from the target_fields to the fields.
    c              3  <   K   | ]}|j         j         k    |V  d S rp   name)rf   fr   s     r]   rh   z%_align_field_types.<locals>.<genexpr>  s1      NN15:9M9MQ9M9M9M9MNNr_   NzField 'z' not found in target schemar   )r   r   r   r   types	is_structr   structr   r   is_list_is_list_likelist_value_fieldis_large_list
large_listis_fixed_size_list	list_sizer   r   nullabler   )r   r   
new_fieldstarget_fieldnew_typer   s        @r]   r   r     s    J 1
 1
NNNNNNNPTUUOuzOOOPPP8l/00 *	)x!!%*-- -9&
)$)0   (,Xl/00  	)UZ(( -8&/0%*67    (,X##L$566 	)UZ(( -=&/0%*67    (,X(():;; 	)UZ(( 	-8&/0%*67   !%/  (,#(HHUZ5><;PQQ	
 	
 	
 	
 r_   reference_fieldsc                   g }d |D             }| D ]}|                     |j                  }|"t          d                    |                    t          j                            |j                  rWt	          j        t          |j        j
        |j        j
                            }t	          j        |j        ||j                  }n|}|                    |           |S )z
    Transform the list of fields so the types match the reference_fields.

    The order of the fields is preserved.

    ``schema`` may have fewer fields than `reference_fields`, but it may not have
    more fields.

    c                    i | ]
}|j         |S rV   r   rf   r   s     r]   r   z$_infer_subschema.<locals>.<dictcomp>  s    222Aafa222r_   NzUnexpected field in schema: {})getr   r   formatr   r  r  r   r  r   r   r   r  r   )r   r  r   lookupr   	referencer  	new_fields           r]   r   r     s     F22!1222F ! !JJuz**	=DDUKKLLL8in-- 	"y J%N)  H 
" II "Ii    Mr_   Union[pa.Schema, LanceModel]c                   t          j        |          r)t          |t                    r|                                }| '|	||j        }t          | ||||          } | j        }n"| t          j	        
                    g |          } |(| t          d          t          | d          r| j        }|rt          |j        |          }|                    |          }t          | t          j	                  r|                     |          } n:t          | t          j                  r t          j                            ||           } | |fS )N)r   r   r   z&Either data or schema must be providedr   )inspectisclass
issubclassr3   r   r   r   r   r   ry   r   r   hasattrr   r   rZ   r   r   r   )r   r   r   r   r   s        r]   sanitize_create_tabler    s^    v 5:fj#A#A 5 #2244 2H)!
 
 
 8''F33D~<EFFFT8$$ 	![F C"6?H==%%h//dBH%% 	C//99DDb233 	C'44VTBBD<r_   c                    |                                  D ]2}||j        j        }||j        j        k    rd}t          |          3|S )z
    Extract pyarrow schema from HuggingFace DatasetDict
    and validate that they're all the same schema between
    splits
    NzCAll datasets in a HuggingFace DatasetDict must have the same schema)rs   r   r   r   )r   r   r   msgs       r]   r   r     s[     ;;== ! !>%2FFw'444WCC..  5 Mr_   c           
   #    K   |                                  D ]\  }}|j                                        D ]}t          j                            |g          }d|j        vrC|                    dt          j        |g|j	        z  t          j
                                        }|                                D ]}|V  dS )zm
    Return a generator of RecordBatches from a HuggingFace DatasetDict
    with an extra `split` column
    r   N)r   r   r   r   ry   r   column_namesappend_columnarraynum_rowsr   )r   keyr   r   r   bs         r]   r   r   $  s      
 

  W\,,.. 	 	EH))5'22Ee000++RXseen&<bikkJJ  %%''  	 r_   r   c                  	 t          |          }nt          j        |          }t          j                                        |          		s S t           j                  }	                                D ]\  }}| j        j        vr	                    |          }nTt          j        t          j                    |j                                                  }t          j	        ||d          }|                    |           t          j        | j        j                  	 fd}t          j                             |                      S )z{
    Use the embedding function to automatically embed the source columns and add the
    vector columns to the table.
    NT)r   r  r   c               3    K   D ]Y}                                  D ]<\  }}|j        }|| j        v}|s>t          j        t          j        | |                                                             r|                    | |j                           }|rV| 	                    
                    |          t          j        |
                    |          j                            } |                     | j                            |          
                    |          t          j        |
                    |          j                            } >| V  [d S )Nr   )r   functionr#  pcallis_nullas_py$compute_source_embeddings_with_retrysource_columnr$  r   r   r%  r   
set_columnindex)	r   vector_columnconffuncno_vector_columncol_data	functionsr   r   s	         r]   r   z#_append_vector_columns.<locals>.genR  sa      	 	E'0'8'8  #t}#08J#J # rvbj}9M.N.N'O'O'U'U'W'W #HHd01   H ( 
 % 3 3"LL77HXFLL4O4O4TUUU! !
 !& 0 0!.44]CC"LL77HXFLL4O4O4TUUU! !
 KKKK'	 	r_   )r   r   r&   get_instanceparse_functionsr   r   r   r   r   r   r  float32r,  ndimsr   r   r   )
r   r   r   r   r5  r6  r   dtyper   r:  s
   ``       @r]   r   r   4  sV    ~"8,,"6?H==)688HHRRI &-  F(00 ! !t 333!]33t}/B/B/D/DEEUTJJJMM%   Yv(>???F      , ,,VSSUU;;;r_   base
table_namec                    t          | |          }t          |          }|j        dk    r|                    d          }|                    d                                          S )z
    Get a table path that can be used in PyArrow FS.

    Removes any weird schemes (such as "s3+ddb") and drops any query params.
    zs3+ddbs3)schemeNquery)
_table_urir   rD  _replacegeturl)r@  rA  uriparseds       r]   _table_pathrL  k  s^     T:
&
&Cc]]F}  --???&&--///r_   c                (    t          | | d          S )Nz.lance)rF   )r@  rA  s     r]   rG  rG  {  s    DZ///000r_   c                J    | du rddl m}  |d          dfS | du s| dS | dfS )	zNormalize a ``progress`` parameter for :meth:`Table.add`.

    Returns ``(progress_obj, owns)`` where *owns* is True when we created a
    tqdm bar that the caller must close.
    Tr   )tqdmz rows)unitFN)NF)	tqdm.autorO  )progressrO  s     r]   _normalize_progressrS    sX     4""""""t!!!4''5H,{U?r_   c                     e Zd ZdZeedd                        Zeedd                        Zeedd                        Zeedd
                        Z	ddZ
eedd                        Zeddd            ZdddZedd            ZddZddZddded ddfd!dd"d#dd$d%dd dd&
dd9Zdd;Z ed%<          fddAZeddC            Zed dDdddEddH            ZddIdJdIddIdKdLdMd d d d dNdNdIdddOddcZe	 	 	 	 dddr            ZddvZe	 	 	 	 	 ddd            ZedIddd            ZddZedIddd            Zeddddd            Zeddd            Z edd            Z!edd            Z"edd            Z#edd            Z$e	 	 ddddd            Z%e	 ddIddd            Z&ed             Z'eddIdIddd            Z(edd            Z)edd            Z*edd            Z+edd            Z,edd            Z-edd            Z.ed             Z/eddd            Z0edd            Z1e2dd            Z3ddĄZ4dń Z5eddƄ            Z6edǄ             Z7dS )ry   a&  
    A Table is a collection of Records in a LanceDB Database.

    Examples
    --------

    Create using [DBConnection.create_table][lancedb.DBConnection.create_table]
    (more examples in that method's documentation).

    >>> import lancedb
    >>> db = lancedb.connect("./.lancedb")
    >>> table = db.create_table("my_table", data=[{"vector": [1.1, 1.2], "b": 2}])
    >>> table.head()
    pyarrow.Table
    vector: fixed_size_list<item: float>[2]
      child 0, item: float
    b: int64
    ----
    vector: [[[1.1,1.2]]]
    b: [[2]]

    Can append new data with [Table.add()][lancedb.table.Table.add].

    >>> table.add([{"vector": [0.5, 1.3], "b": 4}])
    AddResult(version=2)

    Can query the table with [Table.search][lancedb.table.Table.search].

    >>> table.search([0.4, 0.4]).select(["b", "vector"]).to_pandas()
       b      vector  _distance
    0  4  [0.5, 1.3]       0.82
    1  2  [1.1, 1.2]       1.13

    Search queries are much faster when an index is created. See
    [Table.create_index][lancedb.table.Table.create_index].
    rR   rQ   c                    t           )zThe name of this TableNotImplementedErrorselfs    r]   r   z
Table.name  
     "!r_   intc                    t           )zThe version of this TablerV  rX  s    r]   versionzTable.version  rZ  r_   r   c                    t           )lThe [Arrow Schema](https://arrow.apache.org/docs/python/api/datatypes.html#)
        of this Table

        rV  rX  s    r]   r   zTable.schema  
     "!r_   Tagsc                    t           )aW  Tag management for the table.

        Similar to Git, tags are a way to add metadata to a specific version of the
        table.

        .. warning::

            Tagged versions are exempted from the :py:meth:`cleanup_old_versions()`
            process.

            To remove a version that has been tagged, you must first
            :py:meth:`~Tags.delete` the associated tag.

        Examples
        --------

        .. code-block:: python

            table = db.open_table("my_table")
            table.tags.create("v2-prod-20250203", 10)

            tags = table.tags.list()

        rV  rX  s    r]   tagsz
Table.tags  s
    6 "!r_   c                ,    |                      d          S )z The number of rows in this TableN)
count_rowsrX  s    r]   __len__zTable.__len__  s    t$$$r_   "Dict[str, EmbeddingFunctionConfig]c                    dS )z^
        Get a mapping from vector column name to it's configured embedding function.
        NrV   rX  s    r]   embedding_functionszTable.embedding_functions        r_   Nfilterrm   c                    t           )
        Count the number of rows in the table.

        Parameters
        ----------
        filter: str, optional
            A SQL where clause to filter the rows to count.
        rV  rY  rk  s     r]   re  zTable.count_rows  s
     "!r_   rI   	blob_modeBlobMode'pandas.DataFrame'c                @     |                                  j        di |S )a~  Return the table as a pandas DataFrame.

        Parameters
        ----------
        blob_mode: str, default "lazy"
            Controls how blob columns are returned for backends that support
            Lance blob-aware pandas conversion.
        **kwargs
            Forwarded to PyArrow / Lance pandas conversion.

        Returns
        -------
        pd.DataFrame
        rV   )r   	to_pandasrY  ro  kwargss      r]   rs  zTable.to_pandas  s%     )t}}(226222r_   pa.Tablec                    t           )_Return the table as a pyarrow Table.

        Returns
        -------
        pa.Table
        rV  rX  s    r]   r   zTable.to_arrow  s
     "!r_   lance.LanceDatasetc                    t           )znReturn the table as a lance.LanceDataset.

        Returns
        -------
        lance.LanceDataset
        rV  rY  ru  s     r]   to_lancezTable.to_lance  r`  r_   'pl.DataFrame'c                    t           )zjReturn the table as a polars.DataFrame.

        Returns
        -------
        polars.DataFrame
        rV  r{  s     r]   	to_polarszTable.to_polars!  r`  r_   l2   `   TIVF_PQ   2      ,  )

index_typewait_timeoutnum_bitsmax_iterationssample_ratemef_constructionr   traintarget_partition_sizevector_column_namereplacera   acceleratorindex_cache_sizeOptional[int]r  r   r  Optional[timedelta]r  r  r  r  r  r   r  r  c       
            t           )a  Create an index on the table.

        Parameters
        ----------
        metric: str, default "l2"
            The distance metric to use when creating the index.
            Valid values are "l2", "cosine", "dot", or "hamming".
            l2 is euclidean distance.
            Hamming is available only for binary vectors.
        num_partitions: int, default 256
            The number of IVF partitions to use when creating the index.
            Default is 256.
        num_sub_vectors: int, default 96
            The number of PQ sub-vectors to use when creating the index.
            Default is 96.
        vector_column_name: str, default "vector"
            The vector column name to create the index.
        replace: bool, default True
            - If True, replace the existing index if it exists.

            - If False, raise an error if duplicate index exists.
        accelerator: str, default None
            If set, use the given accelerator to create the index.
            Only support "cuda" for now.
        index_cache_size : int, optional
            The size of the index cache in number of entries. Default value is 256.
        num_bits: int
            The number of bits to encode sub-vectors. Only used with the IVF_PQ index.
            Only 4 and 8 are supported.
        wait_timeout: timedelta, optional
            The timeout to wait if indexing is asynchronous.
        name: str, optional
            The name of the index. If not provided, a default name will be generated.
        train: bool, default True
            Whether to train the index with existing data. Vector indices always train
            with existing data.
        rV  )rY  metricnum_partitionsnum_sub_vectorsr  r  r  r  r  r  r  r  r  r  r  r   r  r  s                     r]   create_indexzTable.create_index*      t "!r_   rS   c                    t           )a  
        Drop an index from the table.

        Parameters
        ----------
        name: str
            The name of the index to drop.

        Notes
        -----
        This does not delete the index from disk, it just removes it from the table.
        To delete the index, run [optimize][lancedb.table.Table.optimize]
        after dropping the index.

        Use [list_indices][lancedb.table.Table.list_indices] to find the names of
        the indices.
        rV  rY  r   s     r]   
drop_indexzTable.drop_indexf  s
    $ "!r_   secondsindex_namesIterable[str]timeoutr   c                    t           )  
        Wait for indexing to complete for the given index names.
        This will poll the table until all the indices are fully indexed,
        or raise a timeout exception if the timeout is reached.

        Parameters
        ----------
        index_names: str
            The name of the indices to poll
        timeout: timedelta
            Timeout to wait for asynchronous indexing. The default is 5 minutes.
        rV  rY  r  r  s      r]   wait_for_indexzTable.wait_for_indexz  s
     "!r_   TableStatisticsc                    t           )9
        Retrieve table and fragment statistics.
        rV  rX  s    r]   statszTable.stats  s
    
 "!r_   BTREE)r  r  r  r   columnr   c                   t           )a	  Create a scalar index on a column.

        Parameters
        ----------
        column : str
            The column to be indexed.  Must be a boolean, integer, float,
            or string column.
        replace : bool, default True
            Replace the existing index if it exists.
        index_type: Literal["BTREE", "BITMAP", "LABEL_LIST"], default "BTREE"
            The type of index to create.
        wait_timeout: timedelta, optional
            The timeout to wait if indexing is asynchronous.
        name: str, optional
            The name of the index. If not provided, a default name will be generated.
        Examples
        --------

        Scalar indices, like vector indices, can be used to speed up scans.  A scalar
        index can speed up scans that contain filter expressions on the indexed column.
        For example, the following scan will be faster if the column ``my_col`` has
        a scalar index:

        >>> import lancedb # doctest: +SKIP
        >>> db = lancedb.connect("/data/lance") # doctest: +SKIP
        >>> img_table = db.open_table("images") # doctest: +SKIP
        >>> my_df = img_table.search().where("my_col = 7", # doctest: +SKIP
        ...                                  prefilter=True).to_pandas()

        Scalar indices can also speed up scans containing a vector search and a
        prefilter:

        >>> import lancedb # doctest: +SKIP
        >>> db = lancedb.connect("/data/lance") # doctest: +SKIP
        >>> img_table = db.open_table("images") # doctest: +SKIP
        >>> img_table.search([1, 2, 3, 4], vector_column_name="vector") # doctest: +SKIP
        ...     .where("my_col != 7", prefilter=True)
        ...     .to_pandas()

        Scalar indices can only speed up scans for basic filters using
        equality, comparison, range (e.g. ``my_col BETWEEN 0 AND 100``), and set
        membership (e.g. `my_col IN (0, 1, 2)`)

        Scalar indices can be used if the filter contains multiple indexed columns and
        the filter criteria are AND'd or OR'd together
        (e.g. ``my_col < 0 AND other_col> 100``)

        Scalar indices may be used if the filter contains non-indexed columns but,
        depending on the structure of the filter, they may not be usable.  For example,
        if the column ``not_indexed`` does not have a scalar index then the filter
        ``my_col = 0 OR not_indexed = 1`` will not be able to use any scalar index on
        ``my_col``.
        rV  )rY  r  r  r  r  r   s         r]   create_scalar_indexzTable.create_scalar_index  s    ~ "!r_   F   @simpleEnglish(      )ordering_field_namesr  writer_heap_sizeuse_tantivytokenizer_namewith_positionr`   rl   max_token_length
lower_casestemremove_stop_wordsascii_foldingngram_min_lengthngram_max_lengthprefix_onlyr  r   field_namesUnion[str, List[str]]r  Optional[Union[str, List[str]]]r  r  r  r  r`   r   rl   r  r  r  r  r  r  r  r  c                   t           )u  Create a full-text search index on the table.

        Warning - this API is highly experimental and is highly likely to change
        in the future.

        Parameters
        ----------
        field_names: str or list of str
            The name of the field to index. Native FTS indexes can only be
            created on a single field at a time. To search over multiple text
            fields, create a separate FTS index for each field.
        replace: bool, default False
            If True, replace the existing index if it exists. Note that this is
            not yet an atomic operation; the index will be temporarily
            unavailable while the new index is being created.
        writer_heap_size: int, default 1GB
            Deprecated legacy Tantivy parameter. Any value other than the
            default raises an error.
        ordering_field_names:
            Deprecated legacy Tantivy parameter. Setting this raises an error.
        tokenizer_name: str, default "default"
            A compatibility alias for native tokenizer configs. Can be "raw",
            "default" or the 2 letter language code followed by "_stem". So
            for english it would be "en_stem". For new native FTS indexes, use
            ``base_tokenizer`` directly; ``tokenizer_name`` is a legacy
            compatibility alias and does not expose model-backed tokenizer names
            such as ``jieba/default`` or ``lindera/ipadic``.
        use_tantivy: bool, default False
            Deprecated legacy Tantivy parameter. Setting this to True raises an
            error.
        with_position: bool, default False
            If False, do not store the positions of the terms in the text.
            This can reduce the size of the index and improve indexing speed.
            But it will raise an exception for phrase queries.
        base_tokenizer : str, default "simple"
            The base tokenizer to use for tokenization. Options are:
            - "simple": Splits text by whitespace and punctuation.
            - "whitespace": Split text by whitespace, but not punctuation.
            - "raw": No tokenization. The entire text is treated as a single token.
            - "ngram": N-Gram tokenizer.
            - "jieba/*": Jieba tokenizer loaded from Lance's language model home.
            - "lindera/*": Lindera tokenizer loaded from Lance's language model home.
        language : str, default "English"
            The language to use for stemming and stop-word removal. This is not
            the primary way to enable CJK tokenization.
        max_token_length : int, default 40
            The maximum token length to index. Tokens longer than this length will be
            ignored.
        lower_case : bool, default True
            Whether to convert the token to lower case. This makes queries
            case-insensitive.
        stem : bool, default True
            Whether to stem the token. Stemming reduces words to their root form.
            For example, in English "running" and "runs" would both be reduced to "run".
        remove_stop_words : bool, default True
            Whether to remove stop words. Stop words are common words that are often
            removed from text before indexing. For example, in English "the" and "and".
        ascii_folding : bool, default True
            Whether to fold ASCII characters. This converts accented characters to
            their ASCII equivalent. For example, "café" would be converted to "cafe".
        ngram_min_length: int, default 3
            The minimum length of an n-gram.
        ngram_max_length: int, default 3
            The maximum length of an n-gram.
        prefix_only: bool, default False
            Whether to only index the prefix of the token for ngram tokenizer.
        wait_timeout: timedelta, optional
            The timeout to wait if indexing is asynchronous.
        name: str, optional
            The name of the index. If not provided, a default name will be generated.

        Notes
        -----
        Model-backed tokenizers such as ``jieba/default`` and ``lindera/ipadic``
        require tokenizer models in Lance's language model home. Set
        ``LANCE_LANGUAGE_MODEL_HOME`` to override the default platform data
        directory under ``lance/language_models``.
        rV  )rY  r  r  r  r  r  r  r  r`   rl   r  r  r  r  r  r  r  r  r  r   s                       r]   create_fts_indexzTable.create_fts_index  s    L "!r_   r   r   r   r   r"   moder   r   r   r   r   rR  $Optional[Union[bool, Callable, Any]]r   c                    t           )aj  Add more data to the [Table](Table).

        Parameters
        ----------
        data: DATA
            The data to insert into the table. Acceptable types are:

            - list-of-dict

            - pandas.DataFrame

            - pyarrow.Table or pyarrow.RecordBatch
        mode: str
            The mode to use when writing the data. Valid values are
            "append" and "overwrite".
        on_bad_vectors: str, default "error"
            What to do if any of the vectors are not the same size or contains NaNs.
            One of "error", "drop", "fill".
        fill_value: float, default 0.
            The value to use when filling vectors. Only used if on_bad_vectors="fill".
        progress: bool, callable, or tqdm-like, optional
            Progress reporting during the add operation. Can be:

            - ``True`` to automatically create and display a tqdm progress
              bar (requires ``tqdm`` to be installed)::

                table.add(data, progress=True)

            - A **callable** that receives a dict with keys ``output_rows``,
              ``output_bytes``, ``total_rows``, ``elapsed_seconds``,
              ``active_tasks``, ``total_tasks``, and ``done``::

                def on_progress(p):
                    print(f"{p['output_rows']}/{p['total_rows']} rows, "
                          f"{p['active_tasks']}/{p['total_tasks']} workers")
                table.add(data, progress=on_progress)

            - A **tqdm-compatible** progress bar whose ``total`` and
              ``update()`` will be called automatically. The postfix shows
              write throughput (MB/s) and active worker count::

                with tqdm() as pbar:
                    table.add(data, progress=pbar)

        Returns
        -------
        AddResult
            An object containing the new version number of the table after adding data.
        rV  )rY  r   r  r   r   rR  s         r]   addz	Table.add;  r  r_   onUnion[str, Iterable[str]]r2   c                    t          |t                    r|gnt          t          |                    }t	          | |          S a	  
        Returns a [`LanceMergeInsertBuilder`][lancedb.merge.LanceMergeInsertBuilder]
        that can be used to create a "merge insert" operation

        This operation can add rows, update rows, and remove rows all in a single
        transaction. It is a very generic tool that can be used to create
        behaviors like "insert if not exists", "update or insert (i.e. upsert)",
        or even replace a portion of existing data with new data (e.g. replace
        all data where month="january")

        The merge insert operation works by combining new data from a
        **source table** with existing data in a **target table** by using a
        join.  There are three categories of records.

        "Matched" records are records that exist in both the source table and
        the target table. "Not matched" records exist only in the source table
        (e.g. these are new data) "Not matched by source" records exist only
        in the target table (this is old data)

        The builder returned by this method can be used to customize what
        should happen for each category of data.

        Please note that the data may appear to be reordered as part of this
        operation.  This is because updated rows will be deleted from the
        dataset and then reinserted at the end with the new values.

        Parameters
        ----------

        on: Union[str, Iterable[str]]
            A column (or columns) to join on.  This is how records from the
            source table and target table are matched.  Typically this is some
            kind of key or id column.

        Examples
        --------
        >>> import lancedb
        >>> data = pa.table({"a": [2, 1, 3], "b": ["a", "b", "c"]})
        >>> db = lancedb.connect("./.lancedb")
        >>> table = db.create_table("my_table", data)
        >>> new_data = pa.table({"a": [2, 3, 4], "b": ["x", "y", "z"]})
        >>> # Perform a "upsert" operation
        >>> res = table.merge_insert("a")     \
        ...      .when_matched_update_all()     \
        ...      .when_not_matched_insert_all() \
        ...      .execute(new_data)
        >>> res
        MergeResult(version=2, num_updated_rows=2, num_inserted_rows=1, num_deleted_rows=0, num_attempts=1)
        >>> # The order of new rows is non-deterministic since we use
        >>> # a hash-join as part of this operation and so we sort here
        >>> table.to_arrow().sort_by("a").to_pandas()
           a  b
        0  1  b
        1  2  x
        2  3  y
        3  4  z
        rZ   rQ   r   r   r2   rY  r  s     r]   merge_insertzTable.merge_insertw  ;    t  C((<bTTd488nn&tR000r_   autorF  BOptional[Union[VEC, str, 'PIL.Image.Image', Tuple, FullTextQuery]]
query_typer   ordering_field_namefts_columnsr>   c                    t           )a~  Create a search query to find the nearest neighbors
        of the given query vector. We currently support [vector search][search]
        and [full-text search][experimental-full-text-search].

        All query options are defined in
        [LanceQueryBuilder][lancedb.query.LanceQueryBuilder].

        Examples
        --------
        >>> import lancedb
        >>> db = lancedb.connect("./.lancedb")
        >>> data = [
        ...    {"original_width": 100, "caption": "bar", "vector": [0.1, 2.3, 4.5]},
        ...    {"original_width": 2000, "caption": "foo",  "vector": [0.5, 3.4, 1.3]},
        ...    {"original_width": 3000, "caption": "test", "vector": [0.3, 6.2, 2.6]}
        ... ]
        >>> table = db.create_table("my_table", data)
        >>> query = [0.4, 1.4, 2.4]
        >>> (table.search(query)
        ...     .where("original_width > 1000", prefilter=True)
        ...     .select(["caption", "original_width", "vector"])
        ...     .limit(2)
        ...     .to_pandas())
          caption  original_width           vector  _distance
        0     foo            2000  [0.5, 3.4, 1.3]   5.220000
        1    test            3000  [0.3, 6.2, 2.6]  23.089996

        Parameters
        ----------
        query: list/np.ndarray/str/PIL.Image.Image, default None
            The targetted vector to search for.

            - *default None*.
            Acceptable types are: list, np.ndarray, PIL.Image.Image

            - If None then the select/where/limit clauses are applied to filter
            the table
        vector_column_name: str, optional
            The name of the vector column to search.

            The vector column needs to be a pyarrow fixed size list type

            - If not specified then the vector column is inferred from
            the table schema

            - If the table has multiple vector columns then the *vector_column_name*
            needs to be specified. Otherwise, an error is raised.
        query_type: str
            *default "auto"*.
            Acceptable types are: "vector", "fts", "hybrid", or "auto"

            - If "auto" then the query type is inferred from the query;

                - If `query` is a list/np.ndarray then the query type is
                "vector";

                - If `query` is a PIL.Image.Image then either do vector search,
                or raise an error if no corresponding embedding function is found.

            - If `query` is a string, then the query type is "vector" if the
            table has embedding functions else the query type is "fts"

        Returns
        -------
        LanceQueryBuilder
            A query builder object representing the query.
            Once executed, the query returns

            - selected columns

            - the vector

            - and also the "_distance" column which is the distance between the query
            vector and the returned vector.
        rV  rY  rF  r  r  r  r  s         r]   searchzTable.search  s    l "!r_   )with_row_idoffsets	list[int]r  r@   c                   dS )a  
        Take a list of offsets from the table.

        Offsets are 0-indexed and relative to the current version of the table.  Offsets
        are not stable.  A row with an offset of N may have a different offset in a
        different version of the table (e.g. if an earlier row is deleted).

        Offsets are mostly useful for sampling as the set of all valid offsets is easily
        known in advance to be [0, len(table)).

        No guarantees are made regarding the order in which results are returned.  If
        you desire an output order that matches the order of the given offsets, you will
        need to add the row offset column to the output and align it yourself.

        Parameters
        ----------
        offsets: list[int]
            The offsets to take.

        Returns
        -------
        pa.RecordBatch
            A record batch containing the rows at the given offsets.
        NrV   )rY  r  r  s      r]   take_offsetszTable.take_offsets  rj  r_   pa.RecordBatchc                   t                    }t          t          |                    }t          |fd          }dg|z  }t          |          D ]}||||         <   | j        j        }|                    d           |                                                   |          	                                
                    d                              |                                                              dg          }|S )a  
        Take a list of offsets from the table and return as a record batch.

        This method uses the `take_offsets` method to take the rows.  However, it
        aligns the offsets to the passed in offsets.  This means the return type
        is a record batch (and so users should take care not to pass in too many
        offsets)

        Note: this method is primarily intended to fulfill the Dataset contract
        for pytorch.

        Parameters
        ----------
        offsets: list[int]
            The offsets to take.

        Returns
        -------
        pa.RecordBatch
            A record batch containing the rows at the given offsets.
        c                    |          S rp   rV   )idxr  s    r]   <lambda>z$Table.__getitems__.<locals>.<lambda>I  s    gcl r_   )r'  r   
_rowoffset)r   r   rangesortedr   r   r   r  selectr   sort_bytakecombine_chunksdrop_columns)	rY  r  num_offsetsindicespermutationpermutation_invir   tbls	    `       r]   __getitems__zTable.__getitems__*  s    : 'llu[))**W*B*B*B*BCCC#+{## 	0 	0A./OKN+++#|$$$g&&VG__XZZW\""T/""^\<.)) 	 
r_   row_idsc                   dS )ap  
        Take a list of row ids from the table.

        Row ids are not stable and are relative to the current version of the table.
        They can change due to compaction and updates.

        No guarantees are made regarding the order in which results are returned.  If
        you desire an output order that matches the order of the given ids, you will
        need to add the row id column to the output and align it yourself.

        Unlike offsets, row ids are not 0-indexed and no assumptions should be made
        about the possible range of row ids.  In order to use this method you must
        first obtain the row ids by scanning or searching the table.

        Even so, row ids are more stable than offsets and can be useful in some
        situations.

        There is an ongoing effort to make row ids stable which is tracked at
        https://github.com/lancedb/lancedb/issues/1120

        Parameters
        ----------
        row_ids: list[int]
            The row ids to take.

        Returns
        -------
        AsyncTakeQuery
            A query object that can be executed to get the rows.
        NrV   )rY  r  r  s      r]   take_row_idszTable.take_row_ids\  rj  r_   
batch_sizer  rA   r  r   c                   d S rp   rV   )rY  rF  r  r  s       r]   _execute_queryzTable._execute_query  s	      #sr_   verboseOptional[bool]c                    d S rp   rV   rY  rF  r  s      r]   _explain_planzTable._explain_plan  s    SVSVr_   c                    d S rp   rV   rY  rF  s     r]   _analyze_planzTable._analyze_plan  s    25#r_   c                    d S rp   rV   r  s     r]   _output_schemazTable._output_schema  s    9<r_   mergenew_datar   c                    d S rp   rV   rY  r  r  r   r   s        r]   	_do_mergezTable._do_merge  s	     cr_   wherer   c                    t           )aj  Delete rows from the table.

        This can be used to delete a single row, many rows, all rows, or
        sometimes no rows (if your predicate matches nothing).

        Parameters
        ----------
        where: str
            The SQL where clause to use when deleting rows.

            - For example, 'x = 2' or 'x IN (1, 2, 3)'.

            The filter must not be empty, or it will error.

        Returns
        -------
        DeleteResult
            An object containing the new version number of the table after deletion.

        Examples
        --------
        >>> import lancedb
        >>> data = [
        ...    {"x": 1, "vector": [1.0, 2]},
        ...    {"x": 2, "vector": [3.0, 4]},
        ...    {"x": 3, "vector": [5.0, 6]}
        ... ]
        >>> db = lancedb.connect("./.lancedb")
        >>> table = db.create_table("my_table", data)
        >>> table.to_pandas()
           x      vector
        0  1  [1.0, 2.0]
        1  2  [3.0, 4.0]
        2  3  [5.0, 6.0]
        >>> table.delete("x = 2")
        DeleteResult(num_deleted_rows=1, version=2)
        >>> table.to_pandas()
           x      vector
        0  1  [1.0, 2.0]
        1  3  [5.0, 6.0]

        If you have a list of values to delete, you can combine them into a
        stringified list and use the `IN` operator:

        >>> to_remove = [1, 5]
        >>> to_remove = ", ".join([str(v) for v in to_remove])
        >>> to_remove
        '1, 5'
        >>> table.delete(f"x IN ({to_remove})")
        DeleteResult(num_deleted_rows=1, version=3)
        >>> table.to_pandas()
           x      vector
        0  3  [5.0, 6.0]
        rV  rY  r  s     r]   deletezTable.delete  s    p "!r_   
values_sqlrs   r   r  Optional[Dict[str, str]]r   c                   t           )a  
        This can be used to update zero to all rows depending on how many
        rows match the where clause. If no where clause is provided, then
        all rows will be updated.

        Either `values` or `values_sql` must be provided. You cannot provide
        both.

        Parameters
        ----------
        where: str, optional
            The SQL where clause to use when updating rows. For example, 'x = 2'
            or 'x IN (1, 2, 3)'. The filter must not be empty, or it will error.
        values: dict, optional
            The values to update. The keys are the column names and the values
            are the values to set.
        values_sql: dict, optional
            The values to update, expressed as SQL expression strings. These can
            reference existing columns. For example, {"x": "x + 1"} will increment
            the x column by 1.

        Returns
        -------
        UpdateResult
            - rows_updated: The number of rows that were updated
            - version: The new version number of the table after the update

        Examples
        --------
        >>> import lancedb
        >>> import pandas as pd
        >>> data = pd.DataFrame({"x": [1, 2, 3], "vector": [[1.0, 2], [3, 4], [5, 6]]})
        >>> db = lancedb.connect("./.lancedb")
        >>> table = db.create_table("my_table", data)
        >>> table.to_pandas()
           x      vector
        0  1  [1.0, 2.0]
        1  2  [3.0, 4.0]
        2  3  [5.0, 6.0]
        >>> table.update(where="x = 2", values={"vector": [10.0, 10]})
        UpdateResult(rows_updated=1, version=2)
        >>> table.to_pandas()
           x        vector
        0  1    [1.0, 2.0]
        1  3    [5.0, 6.0]
        2  2  [10.0, 10.0]
        >>> table.update(values_sql={"x": "x + 1"})
        UpdateResult(rows_updated=3, version=3)
        >>> table.to_pandas()
           x        vector
        0  2    [1.0, 2.0]
        1  4    [5.0, 6.0]
        2  3  [10.0, 10.0]
        rV  rY  r  rs   r  s       r]   updatezTable.update  s    | "!r_   delete_unverified
older_thanr  'CleanupStats'c                   dS )aT  
        Clean up old versions of the table, freeing disk space.

        Parameters
        ----------
        older_than: timedelta, default None
            The minimum age of the version to delete. If None, then this defaults
            to two weeks.
        delete_unverified: bool, default False
            Because they may be part of an in-progress transaction, files newer
            than 7 days old are not deleted by default. If you are sure that
            there are no in-progress transactions, then you can set this to True
            to delete all files older than `older_than`.

        Returns
        -------
        CleanupStats
            The stats of the cleanup operation, including how many bytes were
            freed.

        See Also
        --------
        [Table.optimize][lancedb.table.Table.optimize]: A more comprehensive
            optimization operation that includes cleanup as well as other operations.

        Notes
        -----
        This function is not available in LanceDb Cloud (since LanceDB
        Cloud manages cleanup for you automatically)
        NrV   rY  r  r  s      r]   cleanup_old_versionszTable.cleanup_old_versions  rj  r_   c                    dS )a  
        Run the compaction process on the table.
        This can be run after making several small appends to optimize the table
        for faster reads.

        Arguments are passed onto Lance's
        [compact_files][lance.dataset.DatasetOptimizer.compact_files].
        For most cases, the default should be fine.

        See Also
        --------
        [Table.optimize][lancedb.table.Table.optimize]: A more comprehensive
            optimization operation that includes cleanup as well as other operations.

        Notes
        -----
        This function is not available in LanceDB Cloud (since LanceDB
        Cloud manages compaction for you automatically)
        NrV   )rY  rY   ru  s      r]   compact_fileszTable.compact_files:  rj  r_   cleanup_older_thanr  retrainr  r   c                   dS )  
        Optimize the on-disk data and indices for better performance.

        Modeled after ``VACUUM`` in PostgreSQL.

        Optimization covers three operations:

         * Compaction: Merges small files into larger ones
         * Prune: Removes old versions of the dataset
         * Index: Optimizes the indices, adding new data to existing indices

        Parameters
        ----------
        cleanup_older_than: timedelta, optional default 7 days
            All files belonging to versions older than this will be removed.  Set
            to 0 days to remove all versions except the latest.  The latest version
            is never removed.
        delete_unverified: bool, default False
            Files leftover from a failed transaction may appear to be part of an
            in-progress operation (e.g. appending new data) and these files will not
            be deleted unless they are at least 7 days old. If delete_unverified is True
            then these files will be deleted regardless of their age.

            .. warning::

                This should only be set to True if you can guarantee that no other
                process is currently working on this dataset. Otherwise the dataset
                could be put into a corrupted state.

        retrain: bool, default False
            This parameter is no longer used and is deprecated.

        The frequency an application should call optimize is based on the frequency of
        data modifications.  If data is frequently added, deleted, or updated then
        optimize should be run frequently.  A good rule of thumb is to run optimize if
        you have added or modified 100,000 or more records or run more than 20 data
        modification operations.
        NrV   rY  r  r  r   s       r]   optimizezTable.optimizeP  rj  r_   Iterable[IndexConfig]c                    dS )z}
        List all indices that have been created with
        [Table.create_index][lancedb.table.Table.create_index]
        NrV   rX  s    r]   list_indiceszTable.list_indices  rj  r_   
index_nameOptional[IndexStatistics]c                    dS )G  
        Retrieve statistics about an index

        Parameters
        ----------
        index_name: str
            The name of the index to retrieve statistics for

        Returns
        -------
        IndexStatistics or None
            The statistics about the index. Returns None if the index does not exist.
        NrV   rY  r(  s     r]   index_statszTable.index_stats  rj  r_   
transforms6Dict[str, str] | pa.Field | List[pa.Field] | pa.Schemac                    dS )a  
        Add new columns with defined values.

        Parameters
        ----------
        transforms: Dict[str, str], pa.Field, List[pa.Field], pa.Schema
            A map of column name to a SQL expression to use to calculate the
            value of the new column. These expressions will be evaluated for
            each row in the table, and can reference existing columns.
            Alternatively, a pyarrow Field or Schema can be provided to add
            new columns with the specified data types. The new columns will
            be initialized with null values.

        Returns
        -------
        AddColumnsResult
            version: the new version number of the table after adding columns.
        NrV   rY  r.  s     r]   add_columnszTable.add_columns  rj  r_   alterationsIterable[Dict[str, str]]c                    dS )a  
        Alter column names and nullability.

        Parameters
        ----------
        alterations : Iterable[Dict[str, Any]]
            A sequence of dictionaries, each with the following keys:
            - "path": str
                The column path to alter. For a top-level column, this is the name.
                For a nested column, this is the dot-separated path, e.g. "a.b.c".
            - "rename": str, optional
                The new name of the column. If not specified, the column name is
                not changed.
            - "data_type": pyarrow.DataType, optional
               The new data type of the column. Existing values will be casted
               to this type. If not specified, the column data type is not changed.
            - "nullable": bool, optional
                Whether the column should be nullable. If not specified, the column
                nullability is not changed. Only non-nullable columns can be changed
                to nullable. Currently, you cannot change a nullable column to
                non-nullable.

        Returns
        -------
        AlterColumnsResult
            version: the new version number of the table after the alteration.
        NrV   rY  r3  s     r]   alter_columnszTable.alter_columns  rj  r_   r   r   c                    dS )a-  
        Drop columns from the table.

        Parameters
        ----------
        columns : Iterable[str]
            The names of the columns to drop.

        Returns
        -------
        DropColumnsResult
            version: the new version number of the table dropping the columns.
        NrV   rY  r   s     r]   r  zTable.drop_columns  rj  r_   r]  Union[int, str]c                    dS )h  
        Checks out a specific version of the Table

        Any read operation on the table will now access the data at the checked out
        version. As a consequence, calling this method will disable any read consistency
        interval that was previously set.

        This is a read-only operation that turns the table into a sort of "view"
        or "detached head".  Other table instances will not be affected.  To make the
        change permanent you can use the `[Self::restore]` method.

        Any operation that modifies the table will fail while the table is in a checked
        out state.

        Parameters
        ----------
        version: int | str,
            The version to check out. A version number (`int`) or a tag
            (`str`) can be provided.

        To return the table to a normal state use `[Self::checkout_latest]`
        NrV   rY  r]  s     r]   checkoutzTable.checkout  rj  r_   c                    dS z
        Ensures the table is pointing at the latest version

        This can be used to manually update a table when the read_consistency_interval
        is None
        It can also be used to undo a `[Self::checkout]` operation
        NrV   rX  s    r]   checkout_latestzTable.checkout_latest  rj  r_   Optional[Union[int, str]]c                    dS )a  Restore a version of the table. This is an in-place operation.

        This creates a new version where the data is equivalent to the
        specified previous version. Data is not copied (as of python-v0.2.1).

        Parameters
        ----------
        version : int or str, default None
            The version number or version tag to restore.
            If unspecified then restores the currently checked out version.
            If the currently checked out version is the
            latest version then this is a no-op.
        NrV   r=  s     r]   restorezTable.restore  rj  r_   List[Dict[str, Any]]c                    dS )List all versions of the tableNrV   rX  s    r]   list_versionszTable.list_versions  rj  r_   c                @    t          | j        j        | j                  S rp   )rG  _connrJ  r   rX  s    r]   _dataset_urizTable._dataset_uri  s    $*.$)444r_   "Tuple[str, pa_fs.FileSystem, bool]c                   ddl m} t          | |          st          | j                  dk    rdS t          | j        dd          }t          |          \  }}|                    |          j        t          j
        j        k    }|||fS )Nr   )RemoteTablefile)rW   NF_indicesfts)remote.tablerN  rZ   rD   rK  rF   rC   get_file_infor   pa_fsFileTypeNotFound)rY  rN  pathfsindex_existss        r]   _get_fts_index_pathzTable._get_fts_index_path  s    ------dK(( 	%N4;L,M,MQW,W,W$$):u==t$$D''--2en6MMb,''r_   c                `    |                                  \  }}}|rt          d| d          d S )Nz%Legacy Tantivy FTS index detected at zo. Tantivy-based FTS has been removed. Delete the legacy index and recreate it with table.create_fts_index(...).)rZ  r   )rY  rW  _existss       r]   _ensure_no_legacy_fts_indexz!Table._ensure_no_legacy_fts_index   sT    2244a 	// / /  	 	r_   c                    dS 
        Check if the table is using the new v2 manifest paths.

        Returns
        -------
        bool
            True if the table is using the new v2 manifest paths, False otherwise.
        NrV   rX  s    r]   uses_v2_manifest_pathszTable.uses_v2_manifest_paths*  rj  r_   c                    dS )ap  
        Migrate the manifest paths to the new format.

        This will update the manifest to use the new v2 format for paths.

        This function is idempotent, and can be run multiple times without
        changing the state of the object store.

        !!! danger

            This should not be run while other concurrent operations are happening.
            And it should also run until completion before resuming other operations.

        You can use
        [Table.uses_v2_manifest_paths][lancedb.table.Table.uses_v2_manifest_paths]
        to check if the table is already using the new path style.
        NrV   rX  s    r]   migrate_v2_manifest_pathszTable.migrate_v2_manifest_paths5  rj  r_   rR   rQ   rR   r[  rR   r   rR   ra  rR   rg  rp   rk  rm   rR   r[  rI   )ro  rp  rR   rq  rR   rv  rR   ry  )rR   r}  )r  rQ   r  ra   r  rm   r  r  r  r   r  r  r  r[  r  r[  r  r[  r  r[  r  r[  r   rm   r  ra   r  r  r   rQ   rR   rS   r  r  r  r   rR   rS   rR   r  )
r  rQ   r  ra   r  r   r  r  r   rm   )&r  r  r  r  r  ra   r  r  r  ra   r  rm   r  ra   r`   r   rl   rQ   r  r  r  ra   r  ra   r  ra   r  ra   r  r[  r  r[  r  ra   r  r  r   rm   r   r   r   Nr   r"   r  r   r   r   r   r   rR  r  rR   r   r  r  rR   r2   NNr  NNrF  r  r  rm   r  r   r  rm   r  r  rR   r>   )r  r  r  ra   rR   r@   )r  r  rR   r  )r  r  r  ra   rR   r@   rF  rA   r  r  r  r  rR   r   FrF  rA   r  r  rR   rQ   rF  rA   rR   rQ   rF  rA   rR   r   
r  r2   r  r"   r   r   r   r   rR   r   r  rQ   rR   r   NNr  rm   rs   r   r  r  rR   r   r  r  r  ra   rR   r  r  r  r  ra   r   ra   rR   r%  r(  rQ   rR   r)  )r.  r/  )r3  r4  r   r  rR   r   r]  r:  r]  rB  rR   rE  )rR   rL  rR   ra   )8r   r   __qualname____doc__propertyr   r   r]  r   rc  rf  ri  re  rs  r   r|  r  r$   r  r  r   r  r  r  r  r  r  r  r  r  r  r  r   r  r  r
  r  r  r  r  r$  r'  r-  r2  r7  r  r>  rA  rD  rH  r	   rK  rZ  r^  rb  rd  rV   r_   r]   ry   ry     s       # #J " " " ^ X" " " " ^ X" " " " ^ X" " " " ^ X"6% % % %    ^ X
 	" 	" 	" 	" ^	"3 3 3 3 3" " " " ^"" " " "" " " " "4%)*.:" '/,0 ""/3':" :" :" :" :" :"x" " " "* @IyQT?U?U?U" " " " "" " " " ^" 
 &-,0">" >" >" >" >" ^>"H AE*<!(,#,4!*,"&" ! !!,0"-f" f" f" f" f" f"P  !+29=9" 9" 9" 9" ^9"v<1 <1 <1 <1| 
 ,0 &-17;U" U" U" U" ^U"n 9>     ^80 0 0 0d 9>          ^ D 
 %)'+# # # # # ^# VVVV ^V555 ^5<<< ^<   ^ 7" 7" 7" ^7"r   $!%="
 04=" =" =" =" =" ^="~  +/# #(	# # # # # ^#J   ^*  37"', , , , , ^,\    ^    ^    ^,    ^:    ^    ^0   ^     ^ - - - ^- 5 5 5 _5( ( ( (      ^   ^  r_   ry   c                     e Zd ZdZdddddddddddZedd            Zedd            Zedd            Ze	dd             Z
e	dddddddd!dd"            Zedd#            Zdd%Zedd'            Zdd)Zedd+            Zdd/Zdd1Zedd3            Zd d6Zd7 Zd!d"d9Zd!d#d;Zdd<Zdd=Zd$d%d@Zd&d'dEZd%dFZd!d(dHZdIddedJdddKdLdMdNdOdPfddJddQd)daZ d*dcZ!d*ddZ"d!d+dfZ# e$dPg          fd,dlZ%d-dnZ&eddo            Z'd.dpZ(d.dqZ)dJdrddsd/dvZ*ddwdxdwddwdydzd{dJdJdJdJd|d|dwdd}d0dZ+e,d1d            Z-	 	 	 	 d2d3dZ.	 	 d4d5dZ/ed6d            Z0e1	 	 	 	 	 d7d8d            Z2e1	 	 	 	 	 d9d:d            Z2e1	 	 	 	 	 d;d<d            Z2e1	 	 	 	 	 d=d>d            Z2	 	 	 	 	 d=d?dÄZ2e		 	 	 	 	 	 	 d@ddddddddŜdAdф            Z3dBdԄZ4	 	 d4dd՜dCdڄZ5dddۜdDdZ6dEdFdZ7dGdZ8dHdZ9dIdZ:edJd            Z; e<j=        de>d          	 d!dwddKd            Z? e<j=        de>d          dLd            Z@ddwdwddMdZAdNdZBdOdZCdPdZDdQdZEdRdZFdSd	ZGdTdZHdUdZIdVdZJd ZKdWdZLdS (X  
LanceTablea  
    A table in a LanceDB database.

    This can be opened in two modes: standard and time-travel.

    Standard mode is the default. In this mode, the table is mutable and tracks
    the latest version of the table. The level of read consistency is controlled
    by the `read_consistency_interval` parameter on the connection.

    Time-travel mode is activated by specifying a version number. In this mode,
    the table is immutable and fixed to a specific version. This is useful for
    querying historical versions of the table.
    N)namespace_pathstorage_optionsr  locationnamespace_clientmanaged_versioningpushdown_operations_async
connection'LanceDBConnection'r   rQ   r  Optional[List[str]]r  r  r  r  r  rm   r  Optional[Any]r  r  r  Optional[set]r  
AsyncTablec                   |g }|| _         || _        || _        || _        |	pt	                      | _        |
	|
| _        d S t          j        |j         	                    |||||||                    | _        d S )N)r  r  r  r  r  r  )
rJ  _namespace_path	_location_namespace_clientset_pushdown_operations_tabler   run
open_table)rY  r  r   r  r  r  r  r  r  r  r  s              r]   __init__zLanceTable.__init__Y  s     !N
-!!1$7$@355! DKKK( ++#1$3%5%%5'9 ,  
 
DKKKr_   rR   c                    | j         j        S rp   )r  r   rX  s    r]   r   zLanceTable.name}  s    {r_   	List[str]c                    | j         S )z'Return the namespace path of the table.)r  rX  s    r]   	namespacezLanceTable.namespace  s     ##r_   c                d    | j         r#d                    | j         | j        gz             S | j        S )z9Return the full identifier of the table (namespace$name).$)r  rr   r   rX  s    r]   idzLanceTable.id  s6      	@88D0DI;>???yr_   r  LanceDBTablec                    ddl m} t          |          } |j        |                                          } | ||j        |          S )Nr   rw   )r  )dbrx   r  
from_innerdatabaser   )clsr  rx   	async_tblconns        r]   r  zLanceTable.from_inner  s^    ))))))sOO	+ +CLLNN;;sN
 
 
 	
r_   r  r  r  r  r  r  r  c                   |g } | |||||||||		  	        }
	 |
j          n8# t          $ r+}dt          |          v rt          d| d          |d }~ww xY w|
S )Nr  z
Not found:zTable z does not exist)r]  r   rQ   FileNotFoundError)r  r  r   r  r  r  r  r  r  r  r  es               r]   openzLanceTable.open  s     !Nc)+--1 3

 

 

	KKK 	 	 	s1vv%%'(F(F(F(FGGGG	
 
s   " 
A&AAc                \    | j         | j         S t          | j        j        | j                  S rp   )r  rL  rJ  rJ  r   rX  s    r]   _dataset_pathzLanceTable._dataset_path  s*    
 >%>!4:>49555r_   ry  c                   	 ddl }n# t          $ r t          d          w xY w| j        6| j        | j        gz   } |j        d| j        | j        j        | j        |d|S  |j        | j	        f| j        | j        j        d|S )z+Return the LanceDataset backing this table.r   N^The lance library is required to use this function. Please install with `pip install pylance`.)r]  r  r  table_idr]  r  rV   )
r   ImportErrorr  r  r   r   r]  rJ  r  r  )rY  ru  r   r  s       r]   r|  zLanceTable.to_lance  s    	LLLL 	 	 	=  	 !-+tyk9H 5=  $
 :!%!7!	 
    u}
L J6
 
 	
 
 	
s    !r   c                X    t          j        | j                                                  S )zwReturn the schema of the table.

        Returns
        -------
        pa.Schema
            A PyArrow schema object.)r   r  r  r   rX  s    r]   r   zLanceTable.schema  s"     x**,,---r_   rE  c                X    t          j        | j                                                  S )rG  )r   r  r  rH  rX  s    r]   rH  zLanceTable.list_versions  s     x1133444r_   r[  c                X    t          j        | j                                                  S )z$Get the current version of the table)r   r  r  r]  rX  s    r]   r]  zLanceTable.version  s"     x++--...r_   r  r  r@   c                P    t          | j                            |                    S rp   )r@   r  r  rY  r  s     r]   r  zLanceTable.take_offsets       $T[%=%=g%F%FGGGr_   r  c                P    t          | j                            |                    S rp   )r@   r  r  rY  r  s     r]   r  zLanceTable.take_row_ids  r  r_   ra  c                *    t          | j                  S )a0  Tag management for the table.

        Similar to Git, tags are a way to add metadata to a specific version of the
        table.

        .. warning::

            Tagged versions are exempted from the :py:meth:`cleanup_old_versions()`
            process.

            To remove a version that has been tagged, you must first
            :py:meth:`~Tags.delete` the associated tag.

        Returns
        -------
        Tags
            The tag manager for managing tags for the table.

        Examples
        --------
        >>> import lancedb
        >>> db = lancedb.connect("./.lancedb")
        >>> table = db.create_table("my_table",
        ...    [{"vector": [1.1, 0.9], "type": "vector"}])
        >>> table.tags.create("v1", table.version)
        >>> table.add([{"vector": [0.5, 0.2], "type": "vector"}])
        AddResult(version=2)
        >>> tags = table.tags.list()
        >>> print(tags["v1"]["version"])
        1
        >>> table.checkout("v1")
        >>> table.to_pandas()
               vector    type
        0  [1.1, 0.9]  vector
        )ra  r  rX  s    r]   rc  zLanceTable.tags  s    J DK   r_   r]  r:  c                ^    t          j        | j                            |                     dS )a  Checkout a version of the table. This is an in-place operation.

        This allows viewing previous versions of the table. If you wish to
        keep writing to the dataset starting from an old version, then use
        the `restore` function.

        Calling this method will set the table into time-travel mode. If you
        wish to return to standard mode, call `checkout_latest`.

        Parameters
        ----------
        version: int | str,
            The version to check out. A version number (`int`) or a tag
            (`str`) can be provided.

        Examples
        --------
        >>> import lancedb
        >>> db = lancedb.connect("./.lancedb")
        >>> table = db.create_table("my_table",
        ...    [{"vector": [1.1, 0.9], "type": "vector"}])
        >>> table.version
        1
        >>> table.to_pandas()
               vector    type
        0  [1.1, 0.9]  vector
        >>> table.add([{"vector": [0.5, 0.2], "type": "vector"}])
        AddResult(version=2)
        >>> table.version
        2
        >>> table.checkout(1)
        >>> table.to_pandas()
               vector    type
        0  [1.1, 0.9]  vector
        N)r   r  r  r>  r=  s     r]   r>  zLanceTable.checkout#  s+    H 	%%g../////r_   c                \    t          j        | j                                                   dS )zCheckout the latest version of the table. This is an in-place operation.

        The table will be set back into standard mode, and will track the latest
        version of the table.
        N)r   r  r  rA  rX  s    r]   rA  zLanceTable.checkout_latestI  s(     	,,../////r_   rB  c                    |,t          j        | j                            |                     t          j        | j                                                   dS )a  Restore a version of the table. This is an in-place operation.

        This creates a new version where the data is equivalent to the
        specified previous version. Data is not copied (as of python-v0.2.1).

        Parameters
        ----------
        version : int or str, default None
            The version number or version tag to restore.
            If unspecified then restores the currently checked out version.
            If the currently checked out version is the
            latest version then this is a no-op.

        Examples
        --------
        >>> import lancedb
        >>> db = lancedb.connect("./.lancedb")
        >>> table = db.create_table("my_table", [
        ...     {"vector": [1.1, 0.9], "type": "vector"}])
        >>> table.version
        1
        >>> table.to_pandas()
               vector    type
        0  [1.1, 0.9]  vector
        >>> table.add([{"vector": [0.5, 0.2], "type": "vector"}])
        AddResult(version=2)
        >>> table.version
        2
        >>> table.tags.create("v2", 2)
        >>> table.restore(1)
        >>> table.to_pandas()
               vector    type
        0  [1.1, 0.9]  vector
        >>> len(table.list_versions())
        3
        >>> table.restore("v2")
        >>> table.to_pandas()
               vector    type
        0  [1.1, 0.9]  vector
        1  [0.5, 0.2]  vector
        >>> len(table.list_versions())
        4
        N)r   r  r  r>  rD  r=  s     r]   rD  zLanceTable.restoreQ  sO    X HT[))'22333$$&&'''''r_   rk  c                Z    t          j        | j                            |                    S rp   )r   r  r  re  rn  s     r]   re  zLanceTable.count_rows  s"    x..v66777r_   c                    | j         j         d| j        }| j        j        "|d                    | j        j                  z  }|d| j        dz  }|S )Nz(name=z , read_consistency_interval={!r}z, _conn=))r   r   r   rJ  read_consistency_intervalr  )rY  vals     r]   __repr__zLanceTable.__repr__  sk    (==	==:/;5<<
4  C 	)$*))))
r_   c                *    |                                  S rp   )r  rX  s    r]   __str__zLanceTable.__str__  s    }}r_      rv  c                Z    t          j        | j                            |                    S )z%Return the first n rows of the table.)r   r  r  headrY  ns     r]   r  zLanceTable.head  s"    x((++,,,r_   rI   ro  rp  'pd.DataFrame'c                    |dk    r>| j         t          | j                  dk    r |                                 j        di |S  |                                 j        dd|i|S )a1  Return the table as a pandas DataFrame.

        Parameters
        ----------
        blob_mode: str, default "lazy"
            Controls how Lance blob columns are returned.
        **kwargs
            Forwarded to Lance pandas conversion.

        Returns
        -------
        pd.DataFrame
        rI   Nmemoryro  rV   )r  rD   r  r   rs  r|  rt  s      r]   rs  zLanceTable.to_pandas  su     ".d011X==,4==??,66v666(t}}(GG9GGGGr_   c                X    t          j        | j                                                  S )zVReturn the table as a pyarrow Table.

        Returns
        -------
        pa.Table)r   r  r  r   rX  s    r]   r   zLanceTable.to_arrow  s"     x,,..///r_   'pl.LazyFrame'c                R    ddl m}  ||           }t          j        |d|          S )aF  Return the table as a polars LazyFrame.

        Parameters
        ----------
        batch_size: int, optional
            Passed to polars. This is the maximum row count for
            scanned pyarrow record batches

        Note
        ----
        1. This requires polars to be installed separately
        2. Currently we've disabled push-down of the filters from polars
           because polars pushdown into pyarrow uses pyarrow compute
           expressions rather than SQl strings (which LanceDB supports)

        Returns
        -------
        pl.LazyFrame
        r   )PyarrowDatasetAdapterF)allow_pyarrow_filterr  )lancedb.integrations.pyarrowr  plscan_pyarrow_dataset)rY  r  r  r   s       r]   r  zLanceTable.to_polars  sH    ( 	GFFFFF''--&%J
 
 
 	
r_   r  Tr  r  r  r  r  r  )r   r  r  r  r   r  r  ra   r  r  r  `Literal['IVF_FLAT', 'IVF_SQ', 'IVF_PQ', 'IVF_RQ', 'IVF_HNSW_SQ', 'IVF_HNSW_PQ', 'IVF_HNSW_FLAT']r  r  r  r  r  r  c                  |I|                                                      ||	||||||||||           |                                  dS |	dk    rt          |||
||          }n|	dk    rt	          |||
||          }n|	dk    rt          |||||
||          }n|	dk    rt          ||||
||	          }nk|	d
k    rt          |||||
||||	  	        }nL|	dk    rt          |||
||||          }n/|	dk    rt          |||
||||          }nt          d|	           t          j        | j                            |||||                    S )zCreate an index on the table.N)r  r  r  r  r  r  r  r  r  r  r  r  IVF_FLAT)distance_typer  r  r  r  IVF_SQr  )r  r  r  r  r  r  r  IVF_RQ)r  r  r  r  r  r  IVF_HNSW_PQ)	r  r  r  r  r  r  r  r  r  IVF_HNSW_SQ)r  r  r  r  r  r  r  IVF_HNSW_FLATUnknown index type )r  configr   r  )r|  r  rA  r(   r*   r)   r,   r.   r/   r0   r   r   r  r  )rY  r  r  r  r  r  r  r  r  r  r  r  r  r  r   r  r  r  s                     r]   r  zLanceTable.create_index  s?   : "MMOO(()%- /'!1! /&; )      """F:%%$--'&;  FF 8##$--'&;  FF 8##$- /!-'&;  FF 8##$-!-'&;  FF =(($- /!-' /&;
 
 
FF =(($--' /&;  FF ?**$--' /&;  FF ?:??@@@xK$$" %  
 
 	
r_   rS   c                Z    t          j        | j                            |                    S )z
        Drops an index from the table

        Parameters
        ----------
        name: str
            The name of the index to drop
        )r   r  r  r  r  s     r]   r  zLanceTable.drop_indexL	  s$     x..t44555r_   c                Z    t          j        | j                            |                    S )  
        Prewarm an index in the table.

        This is a hint to the database that the index will be accessed in the
        future and should be loaded into memory if possible.  This can reduce
        cold-start latency for subsequent queries.

        This call initiates prewarming and returns once the request is accepted.
        It is idempotent and safe to call from multiple clients concurrently.

        It is generally wasteful to call this if the index does not fit into the
        available cache.  Not all index types support prewarming; unsupported
        indices will silently ignore the request.

        Parameters
        ----------
        name: str
            The name of the index to prewarm
        )r   r  r  prewarm_indexr  s     r]   r  zLanceTable.prewarm_indexW	  s$    ( x11$77888r_   r   c                Z    t          j        | j                            |                    S )n  
        Prewarm data for the table.

        This is a hint to the database that the given columns will be accessed
        in the future and the database should prefetch the data if possible.
        Currently only supported on remote tables.

        This call initiates prewarming and returns once the request is accepted.
        It is idempotent and safe to call from multiple clients concurrently.

        This operation has a large upfront cost but can speed up future queries
        that need to fetch the given columns.  Large columns such as embeddings
        or binary data may not be practical to prewarm.  This feature is intended
        for workloads that issue many queries against the same columns.

        Parameters
        ----------
        columns: list of str, optional
            The columns to prewarm. If None, all columns are prewarmed.
        )r   r  r  prewarm_datar9  s     r]   r  zLanceTable.prewarm_datam	  s$    * x0099:::r_   r  r  r  r  r   c                \    t          j        | j                            ||                    S rp   )r   r  r  r  r  s      r]   r  zLanceTable.wait_for_index	  s&     x22;HHIIIr_   r  c                X    t          j        | j                                                  S rp   )r   r  r  r  rX  s    r]   r  zLanceTable.stats	  s     x))++,,,r_   c                X    t          j        | j                                                  S rp   )r   r  r  rJ  rX  s    r]   rJ  zLanceTable.uri	  s    x))***r_   c                X    t          j        | j                                                  S )  Get the initial storage options that were passed in when opening this table.

        For dynamically refreshed options (e.g., credential vending), use
        :meth:`latest_storage_options`.

        Warning: This is an internal API and the return value is subject to change.

        Returns
        -------
        Optional[Dict[str, str]]
            The storage options, or None if no storage options were configured.
        )r   r  r  initial_storage_optionsrX  s    r]   r  z"LanceTable.initial_storage_options	  s"     x;;==>>>r_   c                X    t          j        | j                                                  S )
  Get the latest storage options, refreshing from provider if configured.

        This method is useful for credential vending scenarios where storage options
        may be refreshed dynamically. If no dynamic provider is configured, this
        returns the initial static options.

        Warning: This is an internal API and the return value is subject to change.

        Returns
        -------
        Optional[Dict[str, str]]
            The storage options, or None if no storage options were configured.
        )r   r  r  latest_storage_optionsrX  s    r]   r  z!LanceTable.latest_storage_options	  s"     x::<<===r_   r  )r  r  r   r  r   c                  |dk    rt                      }n<|dk    rt                      }n'|dk    rt                      }nt          d|           t	          j        | j                            ||||                    S )Nr  BITMAP
LABEL_LISTr  r  r  r   )r'   r+   r-   r   r   r  r  r  )rY  r  r  r  r   r  s         r]   r  zLanceTable.create_scalar_index	  s       WWFF8##XXFF<''[[FF?:??@@@xK$$VWVRV$WW
 
 	
r_   Fr  r  r  r  r  )r  r  r  r  r  r  r`   rl   r  r  r  r  r  r  r  r  r   r  r  r  r  r  r  r  r  r`   r   rl   r  r  r  r  r  r  r  r  c                  |                                   |rt          d          |t          d          |dk    rt          d          t          |t                    st          d          |||	||
|||||||d}n|                     |          }t          d	i |}	 t          j        | j        	                    ||||                     d S # t          t          f$ r#}t          ||j        |j                   |d }~ww xY w)
Nz^Tantivy-based FTS has been removed. Remove use_tantivy and recreate the index with native FTS.zXordering_field_names was only supported by the removed Tantivy-based FTS implementation.r  zTwriter_heap_size was only supported by the removed Tantivy-based FTS implementation.zNative FTS indexes can only be created on a single field at a time. To search over multiple text fields, create a separate FTS index for each field.)r`   rl   r  r  r  r  r  r  r  r  r  r  r`   rl   rV   )r^  r   rZ   rQ   infer_tokenizer_configsr1   r   r  r  r  RuntimeErrorrv   r`   rl   )rY  r  r  r  r  r  r  r  r`   rl   r  r  r  r  r  r  r  r  r   tokenizer_configsr  r  s                         r]   r  zLanceTable.create_fts_index	  s   . 	((*** 	M    +4   1114   +s++ 	5   !"0$!.$4(%6!.$4$4*! ! !% < <^ L L 
 

 
	H((#!	 )       L) 	 	 	%%4   
 G	s   #0C D	&DD	r   c                x   | dk    rddddddddddd
S | d	k    rd	dd dddddddd
S | d
k    rd
dd dddddddd
S t          |           dk    rt          d|            | d d         }| dd          dk    rt          d|            |t          vrt          d|           dt          |         ddddddddd
S )Ndefaultr  r  r  TFr  )
r`   rl   r  r  r  r  r  r  r  r  raw
whitespace   zInvalid tokenizer name    _stemzInvalid language code )r   r   rH   )r  langs     r]   r  z"LanceTable.infer_tokenizer_configs
  sg   Y&&"*%$&"%*!&$%$%$   u$$"'%$(#%*!&$%$%$   |++".%$(#%*!&$%$%$   ~!##G~GGHHHbqb!"##'))G~GGHHH|##<d<<===&$T* "!&" ! ! 
 
 	
r_   r   r   r   r   r"   r  r   r   r   r   r   rR  r  r   c           	         t          |          \  }}	 t          j        | j                            |||||                    |r|                                 S S # |r|                                 w w xY w)a  Add data to the table.
        If vector columns are missing and the table
        has embedding functions, then the vector columns
        are automatically computed and added.

        Parameters
        ----------
        data: list-of-dict, pd.DataFrame
            The data to insert into the table.
        mode: str
            The mode to use when writing the data. Valid values are
            "append" and "overwrite".
        on_bad_vectors: str, default "error"
            What to do if any of the vectors are not the same size or contains NaNs.
            One of "error", "drop", "fill", "null".
        fill_value: float, default 0.
            The value to use when filling vectors. Only used if on_bad_vectors="fill".
        progress: bool, callable, or tqdm-like, optional
            A callback or tqdm-compatible progress bar. See
            :meth:`Table.add` for details.

        Returns
        -------
        int
            The number of vectors in the table.
        r  r   r   rR  )rS  r   r  r  r  close)rY  r   r  r   r   rR  ownss          r]   r  zLanceTable.addW
  s    D -X66$	!8#1)%       !    !t !    !s   0A A5other_tableUnion[LanceTable, DATA]left_onright_onr   &Optional[Union[pa.Schema, LanceModel]]c                    t          |t                    r|                                }nt          ||          }|                                                     ||||           |                                  dS )a  Merge another table into this table.

        Performs a left join, where the dataset is the left side and other_table
        is the right side. Rows existing in the dataset but not on the left will
        be filled with null values, unless Lance doesn't support null values for
        some types, in which case an error will be raised. The only overlapping
        column allowed is the join column. If other overlapping columns exist,
        an error will be raised.

        Parameters
        ----------
        other_table: LanceTable or Reader-like
            The data to be merged. Acceptable types are:
            - Pandas DataFrame, Pyarrow Table, Dataset, Scanner,
            Iterator[RecordBatch], or RecordBatchReader
            - LanceTable
        left_on: str
            The name of the column in the dataset to join on.
        right_on: str or None
            The name of the column in other_table to join on. If None, defaults to
            left_on.
        schema: pa.Schema or LanceModel, optional
            The schema of the other_table.
            If not provided, the schema is inferred from the data.

        Examples
        --------
        >>> import lancedb
        >>> import pyarrow as pa
        >>> df = pa.table({'x': [1, 2, 3], 'y': ['a', 'b', 'c']})
        >>> db = lancedb.connect("./.lancedb")
        >>> table = db.create_table("dataset", df)
        >>> table.to_pandas()
           x  y
        0  1  a
        1  2  b
        2  3  c
        >>> new_df = pa.table({'x': [1, 2, 3], 'z': ['d', 'e', 'f']})
        >>> table.merge(new_df, 'x')
        >>> table.to_pandas()
           x  y  z
        0  1  a  d
        1  2  b  e
        2  3  c  f
        )r  r  r   N)rZ   r  r|  r   r  rA  )rY  r  r  r  r   s        r]   r  zLanceTable.merge
  s    h k:.. 	%..00KK( K 	8F 	 	
 	
 	
 	r_   rg  c                b    t          j                                        | j        j                  S )   
        Get the embedding functions for the table

        Returns
        -------
        funcs: Dict[str, EmbeddingFunctionConfig]
            A mapping of the vector column to the embedding function
            or empty dict if not configured.
        )r&   r;  r<  r   r   rX  s    r]   ri  zLanceTable.embedding_functions
  s-     )577GGK 
 
 	
r_   vectorrF  3Optional[Union[VEC, str, 'PIL.Image.Image', Tuple]]r  Literal['vector']r  r  r?   c                    d S rp   rV   r  s         r]   r  zLanceTable.search
  s	     #&#r_   rQ  Literal['fts']r<   c                    d S rp   rV   r  s         r]   r  zLanceTable.search
  s	      #sr_   hybridr  Literal['hybrid']r=   c                    d S rp   rV   r  s         r]   r  zLanceTable.search
  s	     #&#r_   r  r   r;   c                    d S rp   rV   r  s         r]   r  zLanceTable.search
  s	     "%r_   r>   c                    t          |t                    rd}t          | j        |||          }t	          j        | |||||pg           S )a  Create a search query to find the nearest neighbors
        of the given query vector. We currently support [vector search][search]
        and [full-text search][search].

        Examples
        --------
        >>> import lancedb
        >>> db = lancedb.connect("./.lancedb")
        >>> data = [
        ...    {"original_width": 100, "caption": "bar", "vector": [0.1, 2.3, 4.5]},
        ...    {"original_width": 2000, "caption": "foo",  "vector": [0.5, 3.4, 1.3]},
        ...    {"original_width": 3000, "caption": "test", "vector": [0.3, 6.2, 2.6]}
        ... ]
        >>> table = db.create_table("my_table", data)
        >>> query = [0.4, 1.4, 2.4]
        >>> (table.search(query)
        ...     .where("original_width > 1000", prefilter=True)
        ...     .select(["caption", "original_width", "vector"])
        ...     .limit(2)
        ...     .to_pandas())
          caption  original_width           vector  _distance
        0     foo            2000  [0.5, 3.4, 1.3]   5.220000
        1    test            3000  [0.3, 6.2, 2.6]  23.089996

        Parameters
        ----------
        query: list/np.ndarray/str/PIL.Image.Image, default None
            The targetted vector to search for.

            - *default None*.
            Acceptable types are: list, np.ndarray, PIL.Image.Image

            - If None then the select/[where][sql]/limit clauses are applied
            to filter the table
        vector_column_name: str, optional
            The name of the vector column to search.

            The vector column needs to be a pyarrow fixed size list type
            *default "vector"*

            - If not specified then the vector column is inferred from
            the table schema

            - If the table has multiple vector columns then the *vector_column_name*
            needs to be specified. Otherwise, an error is raised.
        query_type: str, default "auto"
            "vector", "fts", or "auto"
            If "auto" then the query type is inferred from the query;
            If `query` is a list/np.ndarray then the query type is "vector";
            If `query` is a PIL.Image.Image then either do vector search
            or raise an error if no corresponding embedding function is found.
            If the `query` is a string, then the query type is "vector" if the
            table has embedding functions, else the query type is "fts"
        fts_columns: str or list of str, default None
            The column(s) to search in for full-text search.
            If None then the search is performed on all indexed columns.
            For now, only one column can be searched at a time.

        Returns
        -------
        LanceQueryBuilder
            A query builder object representing the query.
            Once executed, the query returns selected columns, the vector,
            and also the "_distance" column which is the distance between the query
            vector and the returned vector.
        rQ  r   r  rF  r  )r  r  r  )rZ   r:   rE   r   r>   creater  s         r]   r  zLanceTable.search  sr    X e]++ 	J5;!1	
 
 
 !'1 3#)r
 
 
 	
r_   r(  )r  r  data_storage_versionenable_v2_manifest_pathsr  r  r  r  rx   Optional[DATA]r   r   exist_okri  'Optional[List[EmbeddingFunctionConfig]]Optional[Dict[str, str | bool]]r)  r*  c
                  |
g }
|                      |           }||_        |
|_        ||_        ||_        |pt                      |_        |$t          j        ddt                     |i }||d<   |$t          j        ddt                     |i }||d<   t          j        |j        j                            ||||||||	|
|||                    |_        |S )	a.	  
        Create a new table.

        Examples
        --------
        >>> import lancedb
        >>> data = [
        ...    {"x": 1, "vector": [1.0, 2]},
        ...    {"x": 2, "vector": [3.0, 4]},
        ...    {"x": 3, "vector": [5.0, 6]}
        ... ]
        >>> db = lancedb.connect("./.lancedb")
        >>> table = db.create_table("my_table", data)
        >>> table.to_pandas()
           x      vector
        0  1  [1.0, 2.0]
        1  2  [3.0, 4.0]
        2  3  [5.0, 6.0]

        Parameters
        ----------
        db: LanceDB
            The LanceDB instance to create the table in.
        name: str
            The name of the table to create.
        data: list-of-dict, dict, pd.DataFrame, default None
            The data to insert into the table.
            At least one of `data` or `schema` must be provided.
        schema: pa.Schema or LanceModel, optional
            The schema of the table. If not provided,
            the schema is inferred from the data.
            At least one of `data` or `schema` must be provided.
        mode: str, default "create"
            The mode to use when writing the data. Valid values are
            "create", "overwrite", and "append".
        exist_ok: bool, default False
            If the table already exists then raise an error if False,
            otherwise just open the table, it will not add the provided
            data but will validate against any schema that's specified.
        on_bad_vectors: str, default "error"
            What to do if any of the vectors are not the same size or contains NaNs.
            One of "error", "drop", "fill", "null".
        fill_value: float, default 0.
            The value to use when filling vectors. Only used if on_bad_vectors="fill".
        embedding_functions: list of EmbeddingFunctionModel, default None
            The embedding functions to use when creating the table.
        data_storage_version: optional, str, default "stable"
            Deprecated.  Set `storage_options` when connecting to the database and set
            `new_table_data_storage_version` in the options.
        enable_v2_manifest_paths: optional, bool, default False
            Deprecated.  Set `storage_options` when connecting to the database and set
            `new_table_enable_v2_manifest_paths` in the options.
        NzEsetting data_storage_version directly on create_table is deprecated. zUse database_options instead.new_table_data_storage_versionz=setting enable_v2_manifest_paths directly on create_table is z)deprecated. Use database_options instead."new_table_enable_v2_manifest_paths)
r   r  r,  r   r   ri  r  r  r  r  )__new__rJ  r  r  r  r  r  warningswarnDeprecationWarningr   r  create_tabler  )r  r  r   r   r   r  r,  r   r   ri  r  r  r)  r*  r  r  r  rY  s                     r]   r(  zLanceTable.create_  s&   T !N{{3
-!!1$7$@355!+MW/"  
 &"$@TO<=#/MO;"  
 &"$( @A hJ))!-%$7- /!!1 *  
 
  r_   r  r   c                Z    t          j        | j                            |                    S rp   )r   r  r  r  r  s     r]   r  zLanceTable.delete  s"    x**511222r_   r  rs   r   r  r   c               `    t          j        | j                            |||                    S )a7  
        This can be used to update zero to all rows depending on how many
        rows match the where clause.

        Parameters
        ----------
        where: str, optional
            The SQL where clause to use when updating rows. For example, 'x = 2'
            or 'x IN (1, 2, 3)'. The filter must not be empty, or it will error.
        values: dict, optional
            The values to update. The keys are the column names and the values
            are the values to set.
        values_sql: dict, optional
            The values to update, expressed as SQL expression strings. These can
            reference existing columns. For example, {"x": "x + 1"} will increment
            the x column by 1.

        Returns
        -------
        UpdateResult
            - rows_updated: The number of rows that were updated
            - version: The new version number of the table after the update

        Examples
        --------
        >>> import lancedb
        >>> import pandas as pd
        >>> data = pd.DataFrame({"x": [1, 2, 3], "vector": [[1.0, 2], [3, 4], [5, 6]]})
        >>> db = lancedb.connect("./.lancedb")
        >>> table = db.create_table("my_table", data)
        >>> table.to_pandas()
           x      vector
        0  1  [1.0, 2.0]
        1  2  [3.0, 4.0]
        2  3  [5.0, 6.0]
        >>> table.update(where="x = 2", values={"vector": [10.0, 10]})
        UpdateResult(rows_updated=1, version=2)
        >>> table.to_pandas()
           x        vector
        0  1    [1.0, 2.0]
        1  3    [5.0, 6.0]
        2  2  [10.0, 10.0]

        r  updates_sql)r   r  r  r  r  s       r]   r  zLanceTable.update  s,    f x**6J*WWXXXr_   r  rA   r  r  r   c               6   d| j         v r/| j        (ddlm} | j        | j        gz   } || j        ||          S t          j        | j        	                    |||                    fd}t          j                            j         |                      S )N
QueryTabler   )_execute_server_side_queryr  c               3     K   	 	 t          j                                                   V  )# t          $ r Y d S w xY wrp   )r   r  	__anext__StopAsyncIteration)
async_iters   r]   	iter_syncz,LanceTable._execute_query.<locals>.iter_sync%  sZ      ;(:#7#7#9#9:::::;%   s   */ 
==)r  r  lancedb.namespacer=  r  r   r   r  r  r  r   r   r   r   )rY  rF  r  r  r=  r  rB  rA  s          @r]   r  zLanceTable._execute_query  s     D555&2DDDDDD+tyk9H--d.DhPUVVVXK&&uW&UU
 

	 	 	 	 	 #001BIIKKPPPr_   r  c                \    t          j        | j                            ||                    S rp   )r   r  r  r   r  s      r]   r   zLanceTable._explain_plan.  s$    x11%AABBBr_   c                Z    t          j        | j                            |                    S rp   )r   r  r  r  r  s     r]   r  zLanceTable._analyze_plan1  s"    x11%88999r_   c                Z    t          j        | j                            |                    S rp   )r   r  r  r  r  s     r]   r  zLanceTable._output_schema4  s"    x22599:::r_   r  r2   r  r   c                `    t          j        | j                            ||||                    S rp   )r   r  r  r
  r	  s        r]   r
  zLanceTable._do_merge7  s1     xK!!%>:NN
 
 	
r_   c                    | j         j        S rp   )r  _innerrX  s    r]   rI  zLanceTable._innerB  s    {!!r_   z0.21.0zUse `Table.optimize` instead.)deprecated_incurrent_versiondetailsr  r  r  r  c               T    |                                                      ||          S )a  
        Clean up old versions of the table, freeing disk space.

        Parameters
        ----------
        older_than: timedelta, default None
            The minimum age of the version to delete. If None, then this defaults
            to two weeks.
        delete_unverified: bool, default False
            Because they may be part of an in-progress transaction, files newer
            than 7 days old are not deleted by default. If you are sure that
            there are no in-progress transactions, then you can set this to True
            to delete all files older than `older_than`.

        Returns
        -------
        CleanupStats
            The stats of the cleanup operation, including how many bytes were
            freed.
        r  )r|  r  r  s      r]   r  zLanceTable.cleanup_old_versionsF  s/    > }}33*; 4 
 
 	
r_   r|   c                v     |                                  j        j        |i |}|                                  |S )aa  
        Run the compaction process on the table.

        This can be run after making several small appends to optimize the table
        for faster reads.

        Arguments are passed onto `lance.dataset.DatasetOptimizer.compact_files`.
         (see Lance documentation for more details) For most cases, the default
        should be fine.
        )r|  r$  r  rA  )rY  rY   ru  r  s       r]   r  zLanceTable.compact_filesi  s=      7(6GGGr_   r  r  r   c               d    t          j        | j                            |||                     dS )r"  r  N)r   r  r  r$  r#  s       r]   r$  zLanceTable.optimize}  sF    Z 	K  #5"3 !  	
 	
 	
 	
 	
r_   r%  c                X    t          j        | j                                                  S )Q
        List all indices that have been created with Self::create_index
        )r   r  r  r'  rX  s    r]   r'  zLanceTable.list_indices  s"     x0022333r_   r(  r)  c                Z    t          j        | j                            |                    S )r+  )r   r  r  r-  r,  s     r]   r-  zLanceTable.index_stats  s$     x//
;;<<<r_   r.  6Dict[str, str] | pa.field | List[pa.field] | pa.Schemar~   c                Z    t          j        | j                            |                    S rp   )r   r  r  r2  r1  s     r]   r2  zLanceTable.add_columns  s$     x//
;;<<<r_   r3  r4  r   c                D    t          j         | j        j        |           S rp   )r   r  r  r7  r6  s     r]   r7  zLanceTable.alter_columns  s!     x11;?@@@r_   r   c                Z    t          j        | j                            |                    S rp   )r   r  r  r  r9  s     r]   r  zLanceTable.drop_columns  s"    x0099:::r_   r  c                Z    t          j        | j                            |                    S )zSet the unenforced primary key. See
        [`AsyncTable.set_unenforced_primary_key`][lancedb.AsyncTable.set_unenforced_primary_key].)r   r  r  set_unenforced_primary_keyr9  s     r]   rX  z%LanceTable.set_unenforced_primary_key  s$     x>>wGGHHHr_   spec'LsmWriteSpec'c                Z    t          j        | j                            |                    S )znInstall an LsmWriteSpec. See
        [`AsyncTable.set_lsm_write_spec`][lancedb.AsyncTable.set_lsm_write_spec].)r   r  r  set_lsm_write_specrY  rY  s     r]   r\  zLanceTable.set_lsm_write_spec  s$     x66t<<===r_   c                X    t          j        | j                                                  S )zrRemove the LsmWriteSpec. See
        [`AsyncTable.unset_lsm_write_spec`][lancedb.AsyncTable.unset_lsm_write_spec].)r   r  r  unset_lsm_write_specrX  s    r]   r_  zLanceTable.unset_lsm_write_spec  s"     x88::;;;r_   c                X    t          j        | j                                                  S )ra  )r   r  r  rb  rX  s    r]   rb  z!LanceTable.uses_v2_manifest_paths  s"     x::<<===r_   c                \    t          j        | j                                                   dS )az  
        Migrate the manifest paths to the new format.

        This will update the manifest to use the new v2 format for paths.

        This function is idempotent, and can be run multiple times without
        changing the state of the object store.

        !!! danger

            This should not be run while other concurrent operations are happening.
            And it should also run until completion before resuming other operations.

        You can use
        [LanceTable.uses_v2_manifest_paths][lancedb.table.LanceTable.uses_v2_manifest_paths]
        to check if the table is already using the new path style.
        N)r   r  r  rd  rX  s    r]   rd  z$LanceTable.migrate_v2_manifest_paths  s(    $ 	668899999r_   
field_namenew_metadataDict[str, str]c                `    t          j        | j                            ||                     dS z
        Replace the metadata of a field in the schema

        Parameters
        ----------
        field_name: str
            The name of the field to replace the metadata for
        new_metadata: dict
            The new metadata to set
        N)r   r  r  replace_field_metadatarY  rb  rc  s      r]   rg  z!LanceTable.replace_field_metadata  s,     	33JMMNNNNNr_   )r  r  r   rQ   r  r  r  r  r  r  r  rm   r  r  r  r  r  r  r  r  re  )rR   r  )r  r  )r  r  r  r  r  r  r  rm   r  r  r  r  r  r  rm  rg  r  rf  )r  r  rR   r@   )r  r  rR   r@   rh  r  rp   r  rj  r  rl  rk  ro  rp  rR   r  )rR   r  )r  r   r  rQ   r  ra   r  rm   r  r  r  r[  r  r  r  r[  r  r[  r  r[  r  r[  r   rm   r  ra   r  r  rn  r   r  rR   rS   ro  rp  rR   r  )r  rQ   r  ra   r  r   r   rm   )$r  r  r  r  r  ra   r  r  r  ra   r  rm   r  ra   r`   r   rl   rQ   r  r  r  ra   r  ra   r  ra   r  ra   r  r[  r  r[  r  ra   r   rm   )r  rQ   rR   r   rq  rr  r}  )r  r  r  rQ   r  rm   r   r  ri  )NNr  NN)rF  r  r  rm   r  r  r  rm   r  r  rR   r?   )NNrQ  NN)rF  r  r  rm   r  r   r  rm   r  r  rR   r<   )NNr"  NN)rF  r  r  rm   r  r#  r  rm   r  r  rR   r=   rt  )rF  rS   r  rm   r  r   r  rm   r  r  rR   r;   ru  )NNr(  Fr   r   N) r  rx   r   rQ   r   r+  r   r   r  r   r,  ra   r   r   r   r   ri  r-  r  r  r  r.  r)  rm   r*  r  r  rm   r  r  r  r  r|  r~  rv  rw  rx  ry  rz  r{  )rR   r  r  )rR   r|   r  r  r  )r.  rS  rR   r~   )r3  r4  rR   r   r  r   r  rR   rS   rY  rZ  rR   rS   rR   rS   r  )rb  rQ   rc  rd  )Mr   r   r  r  r  r  r   r  r  classmethodr  r  r	   r  r|  r   rH  r]  r  r  rc  r>  rA  rD  re  r  r  r  rs  r   r  r$   r  r  r  r  r   r  r  rJ  r  r  r  r  staticmethodr  r  r  ri  r   r  r(  r  r  r  r   r  r  r
  rI  deprecation
deprecatedr   r  r  r$  r'  r-  r2  r7  r  rX  r\  r_  rb  rd  rg  rV   r_   r]   r  r  J  s!	        & /348*."&*.-1-1!" " " " " "H       X  $ $ $ X$    X 	
 	
 	
 [	
  /348*."&*.-1-1# # # # # [#J 6 6 6 _6
 
 
 
6 . . . X.5 5 5 5 / / / X/H H H HH H H H $! $! $! X$!L$0 $0 $0 $0L0 0 0.( .( .( .( .(`8 8 8 8 8      - - - - -H H H H H,0 0 0 0
 
 
 
 
:  $"4%)*.  "-}
0 #/35}
 }
 }
 }
 }
 }
~	6 	6 	6 	69 9 9 9,; ; ; ; ;0 @IyQT?U?U?UJ J J J J
- - - - + + + X+? ? ? ?> > > >( &-"
 
 
 
 
 
0 AE*<!(,#,4!*,"&" ! !!"+S S S S S Sj <
 <
 <
 \<
B !+29=/! /! /! /! /!j #'9=> > > > >@ 
 
 
 _
  FJ,0(0-17;& & & & X&  FJ,0%*-17;# # # # X# 
 ,0(0-17;	& 	& 	& 	& X	&  ,0 &-17;% % % % X% ,0 &-17;\
 \
 \
 \
 \
| 
  $&*#+2GKw /3;?.237"&*.-1%w w w w w [wr3 3 3 3
  $!%3Y
 043Y 3Y 3Y 3Y 3Y 3Yr %)'+Q Q Q Q Q Q:C C C C C: : : :; ; ; ;	
 	
 	
 	
 " " " X" [#/   +/
 #(	
 
 
 
 
 

< [#/  
   
$ 37"'3
 3
 3
 3
 3
 3
j4 4 4 4= = = = = = = =
A A A A
; ; ; ;I I I I
> > > >
< < < <
	> 	> 	> 	>: : :(O O O O O Or_   r  (Literal['error', 'drop', 'fill', 'null']c                     t           j        |          s S t           j                   fd}t          j                             |                      S )Nc               3  d  K   D ](} g }
D ]_}|d         }	2|0t          | |d                            }|                    |           t          | |d         ||d                   } `|D ](}|d         t          | |d                            |d<   )| j                            d          r| V  t
          j                            | g                                        	                                }|r.t
          j
                            |d         j                  V  *d S )	Nexpected_dimr   expected_value_type)r  r   r   rw  rx  Tr   r   r   )_infer_vector_dimr   _handle_bad_vector_columnr   r   r   ry   r   r   r   r   r   r   )r   pending_dimsr5  dimr   r   r   output_schemar   r   vector_columnss        r]   r   z _handle_bad_vectors.<locals>.gen  s      	 	EL!/  #N3 ,+E-2G,HIIC ''6661'4V'<#1)!$(56K(L   ".   084EmF345 5M.1 |""="FF  %%ug..33MBBMMOO   n00 O+( 1     9	 	r_   )_find_vector_columnsr   _vector_output_schemar   r   r   )r   r   r   r   r   r   r}  r~  s   ````  @@r]   r   r     s     *&-QQN )&-HHM                   D ,,]CCEEBBBr_   reader_schema
List[dict]c                h   |g }| D ]}t          |j                  o8t          j                            |j        j                  o|j        t          k    }t          j                            |j                  o8t          j                            |j        j                  o|j        j	        dk    }|s|r|
                    |j        d d d           |S t          | j                  }t          |j        |          }t          t          j                                        |                                                    }	g }|D ]f}|j        |vrt          |j                  r)t          j                            |j        j                  sK|                     |j                  }
|j        |	v p>|j        t          k    p.|j        dk    o#t          j                            |j                  }t          j                            |
j                  o8t          j                            |
j        j                  o|
j        j	        dk    }|s|rX|
                    |j        t          j                            |j                  r|j        j	        nd |j        j        d           h|S )N
   )r   rw  rx  	embedding)r  r   r   r  is_floating
value_typer   r$   r
  r  r   r  r   r   r   r&   r;  r<  keysr   )r  r   r   r~  r   named_vector_collikely_vector_colreader_column_namesactive_metadataembedding_function_columnsreader_fieldtyped_fixed_vector_cols               r]   r  r  C  s   
 " 	 	Eej)) 5H(()>??5J"44  ++EJ77 1H(()>??1Z)R/ 
   #4 %% %
(,/3    m122%m&<hGGO!$!.00@@QQVVXX" " N  :000UZ(( 	0D0DJ!1
 1
 	 $**5:66J44 Wz//W
k)Ubh.I.I%*.U.U 	 H''(9:: 2$$\%6%ABB2!+r1 	  	5 	!!!J 866uzBB"
,,!+0:+@ 
 
 
 r_   r~  c           	     8   d |D             }g }| D ]o}|                     |j                  }||j        }nt          ||          }|                    t          j        |j        ||j        |j                             pt          j	        || j                  S )Nc                     i | ]}|d          |S r   rV   )rf   r  s     r]   r   z)_vector_output_schema.<locals>.<dictcomp>  s    KKK&vf~vKKKr_   r   )
r  r   r   _vector_output_typer   r   r   r  r   r   )r  r~  columns_by_namer   r   r  output_types          r]   r  r    s     LKNKKKOF Y Y $$UZ00>*KK-eV<<Kbhuz;WWXXXX9Vm&<====r_   r   pa.Fieldr5  r   pa.DataTypec                @   t          | j                  s| j        S |d         t          j                            | j        j                  sRt          j                            | j        j                  s)t          j                            | j        j                  rt          j        |d                   S |d         Xt          j        	                    | j                  r4| j        j
        |d         k    rt          j        | j        j                  S | j        S )Nrx  rw  )r  r   r   r  r/  r  
is_integeris_unsigned_integerr  r
  r  )r   r5  s     r]   r  r    s    $$ z*+7
.// 88uz455 8 8''
(=>> 8
 x&;<=== 	n%1H''
33 	2J M.$AAAx
-...:r_   r  r  rw  r  rx  Optional[pa.DataType]c                   | j                             |          }| |         }t          |j                  s| S |t          j                            |j                  rk|j        j        |k    r[t	          j        |	                                t	          j
        |j        j                            }|                     |||          } |t          j                            |j        j                  s)t          j                            |j        j                  rQt	          j        |	                                t	          j
        |                    }|                     |||          } t          j                            |j        j                  rt!          |          }n%t	          j        dgt#          |          z            }||}	nDt          j                            |j                  r|j        j        }	nt%          |          }	|	| S t'          j        t'          j        |          |	          }
t'          j        |                                          p%t'          j        |
                                          }|r)t'          j        ||
          }|dk    rLt'          j        |
                                          rt3          d| d          t3          d| d          |dk    r)t'          j        |t	          j        d          |          }n|d	k    r0|                     t'          j        |                    } | |         }n]|d
k    rE|t3          d          t'          j        |t	          j        |g|	z  |j                  |          }nt3          d|           |                     |||          S )a  
    Ensure that the vector column exists and has type fixed_size_list(float)

    Parameters
    ----------
    data: pa.Table
        The table to sanitize.
    vector_column_name: str
        The name of the vector column.
    on_bad_vectors: str, default "error"
        What to do if any of the vectors are not the same size or contains NaNs.
        One of "error", "drop", "fill", "null".
    fill_value: float, default 0.0
        The value to use when filling vectors. Only used if on_bad_vectors="fill".
    Nr+  Fr   zVector column 'z' has variable length vectors. Set on_bad_vectors='drop' to remove them, set on_bad_vectors='fill' and fill_value=<value> to replace them, or set on_bad_vectors='null' to replace them with null.z' has NaNs. Set on_bad_vectors='drop' to remove them, set on_bad_vectors='fill' and fill_value=<value> to replace them, or set on_bad_vectors='null' to replace them with null.nulldropfillz;`fill_value` must not be None if `on_bad_vectors` is 'fill'z"Invalid value for on_bad_vectors: )r#  r4  r  r   r   r  r
  r  r%  	to_pylistr  r  r3  r  r  r  has_nan_valuesr   ry  r-  	not_equallist_value_lengthri   r0  or_r   if_elsescalarrk  invert)r   r  r   r   rw  rx  positionvec_arrhas_nanr|  has_wrong_dimhas_bad_vectorsis_bads                r]   rz  rz    s   .  &&'9::H%&G&&  	 H''55 	!L"l22(7,,..RXgl>U5V5VWWWx);WEE&
GL344 '8''(?@@ ' (7,,..RX>Q5R5RSSSx);WEE	xGL344 3 ))(E7S\\122		$	$W\	2	2 l$((;KL!5g!>!>DDMfWoo++--N1F1F1L1L1N1NO %T//W$$vm$$**,,  N&8 N N N   !N&8 N N N   v%%j	$ GG
 v%%;;ry0011D-.GGv%%! Q   j	:,,7<@@@ GG R.RRSSS??8%7AAAr_   arr$Union[pa.ListArray, pa.ChunkedArray]pa.BooleanArrayc                   t          | t          j                  r$t          j        d | j        D                       }n|                                 }t          j                            |j                  r9t          j
        |                    t          j                                        }nt          j
        |          }t          j        |           }t          j        t          j        ||                    }t          j        t#          t%          |                     t          j                              }t          j        ||          S )Nc                6    g | ]}|                                 S rV   )flatten)rf   chunks     r]   r   z"has_nan_values.<locals>.<listcomp>  s     "K"K"Ku5==??"K"K"Kr_   r+  )rZ   r   ChunkedArraychunked_arraychunksr  r  
is_float16r   r-  is_nanr   r=  list_parent_indicesuniquerk  r%  r  r   uint32is_in)r  rs   values_has_nanvalues_indiceshas_nan_indicesr  s         r]   r  r    s    #r'' !"K"K
"K"K"KLL	x6;'' + 6;;rz||#<#<==6**+C00Ni	.. I IJJOhuSXXRY[[999G8G_---r_   	data_typec                    t           j                            |           p=t           j                            |           pt           j                            |           S rp   )r   r  r  r  r
  )r  s    r]   r  r  %  sJ    
## 	28!!),,	28&&y11r_   metadata_dictsc                     i }| D ]x}||                                 D ]^\  }}t          |t                    r|                    d          }t          |t                    r|                    d          }|||<   _y|S )Nzutf-8)r   rZ   rQ   encode)r  mergedr   r'  values        r]   r   r   -  s    F"    "..** 	  	 JC#s## *jj))%%% .W--F3KK	  Mr_   rb  c                :    |                                  }d|v pd|v S )z0Check if a field name indicates a vector column.r  r  )lower)rb  
name_lowers     r]   _name_suggests_vector_columnr  ;  s(    !!##Jz!>[J%>>r_   &Tuple[pa.Schema, pa.RecordBatchReader]c                   | j         }d }t          |          D ]\  }}t          j                            |j                  p#t          j                            |j                  }t          |j                  r?|r<|t          |           \  }} t          |                    |                    }t          j                            |j        j                  r(t          j        t          j                    |          }n~t          j                            |j        j                  rR|                    |          }t#          |t          j                  r|                                }|                                }	t+          j        |	d                                          }
|
dk    r't          j        t          j                    |          }nt+          j        |	          }|d                                         }|d                                         }||dk     s|-|dk    r't          j        t          j                    |          }n)t          j        t          j                    |          }nkt          j        |j        ||j                  }|                    ||          }|| fS )N
only_valid)r  r   minmax   )r  )r   	enumerater   r  r  r   r  r  r   r   _modal_list_sizer  r  r  r  r=  r  rZ   r  r  r  r-  countr0  uint8min_maxr   r  r  )r   r   peekedr  r   is_list_typer|  target_typers   	flattenedvalid_countr  	min_value	max_valuer  s                  r]   r   r   A  sU    ]FFf%% -. -.5x''
33 
rx7M7MJ8
 8
 (
33 (	. (	.~!,V!4!4"6==#3#344C x##EJ$9::  hrz||S99$$UZ%:;; q))fbo66 5#2244F"NN,,	 hy|DDDJJLL!##"$(28::s";";KK j33G ' 4 4 6 6I ' 4 4 6 6I!-)a--!-)c//&(hrz||S&A&A&(hrxzz3&?&?
  I ZZ9--F6>r_   r[  c                    t          j        t          j        |                     d                                         d         S Nr   r  )r-  r  r  r0  )r  s    r]   r  r  y  s3    72',,--a06688@@r_    Union[pa.Array, pa.ChunkedArray]c                4   t          | j                  sd S t          j        |           }t          j        |t          j        |d                    }t          |          dk    rd S t          j        |          d                                         d         S r  )	r  r   r-  r  rk  greaterr   r  r0  )r  lengthss     r]   ry  ry  ~  s    "" t"3''GiGQ!7!788G
7||qt77A$$&&v..r_   c                @    | j         t          | j                    dS dS )z.
    Make sure the metadata is valid utf8
    N)r   _validate_metadatar   s    r]   r   r     s*     "6?+++++ #"r_   c                   |                                  D ]v\  }}t          |t                    r8	 |                    d           1# t          $ r t          d| d          w xY wt          |t                    rt          |           wdS )zo
    Make sure the metadata values are valid utf8 (can be nested)

    Raises ValueError if not valid utf8
    utf8zMetadata key zG is not valid utf8. Consider base64 encode for generic binary metadata.N)r   rZ   rJ   decodeUnicodeDecodeErrorr   r   r  )r   r   r   s      r]   r  r    s        
" 
"1a 		"    %    JA J J J  
 4   	"q!!!
" 
"s   AA$c                     e Zd ZdZddZd Zd Zd Zdd
Zd Z	ddZ
ddZddZedd            ZddZddZdddZddd!Zdd#Zdd%Zddd*Zdd+Zddddd,d-dd7Zdd8Zdd9Zddd;Z ed<=          fddBZddDZddEZddGZddHZ dIddddJddVZ!ddYZ"e#	 	 	 	 	 dddc            Z$e#	 	 	 	 	 dddf            Z$e#	 	 	 	 	 dddi            Z$e#	 	 	 	 	 dddl            Z$e#	 	 	 	 	 dddo            Z$	 	 	 	 	 dddrZ$dduZ%ddxZ&dddydd}Z'ddZ(ddZ)ddZ*ddZ+ddZ,	 ddddddZ-ddZ.ddZ/ddZ0ddZ1d Z2ddZ3d Z4dddZ5ddZ6ddZ7edd            Z8ddddddZ9ddZ:ddZ;ddZ<d Z=ddZ>dS )r  a8
  
    An AsyncTable is a collection of Records in a LanceDB Database.

    An AsyncTable can be obtained from the
    [AsyncConnection.create_table][lancedb.AsyncConnection.create_table] and
    [AsyncConnection.open_table][lancedb.AsyncConnection.open_table] methods.

    An AsyncTable object is expected to be long lived and reused for multiple
    operations. AsyncTable objects will cache a certain amount of index data in memory.
    This cache will be freed when the Table is garbage collected.  To eagerly free the
    cache you can call the [close][lancedb.AsyncTable.close] method.  Once the
    AsyncTable is closed, it cannot be used for any further operations.

    An AsyncTable can also be used as a context manager, and will automatically close
    when the context is exited.  Closing a table is optional.  If you do not close the
    table, it will be closed when the AsyncTable object is garbage collected.

    Examples
    --------

    Create using [AsyncConnection.create_table][lancedb.AsyncConnection.create_table]
    (more examples in that method's documentation).

    >>> import lancedb
    >>> async def create_a_table():
    ...     db = await lancedb.connect_async("./.lancedb")
    ...     data = [{"vector": [1.1, 1.2], "b": 2}]
    ...     table = await db.create_table("my_table", data=data)
    ...     print(await table.query().limit(5).to_arrow())
    >>> import asyncio
    >>> asyncio.run(create_a_table())
    pyarrow.Table
    vector: fixed_size_list<item: float>[2]
      child 0, item: float
    b: int64
    ----
    vector: [[[1.1,1.2]]]
    b: [[2]]

    Can append new data with [AsyncTable.add()][lancedb.table.AsyncTable.add].

    >>> async def add_to_table():
    ...     db = await lancedb.connect_async("./.lancedb")
    ...     table = await db.open_table("my_table")
    ...     await table.add([{"vector": [0.5, 1.3], "b": 4}])
    >>> asyncio.run(add_to_table())

    Can query the table with
    [AsyncTable.vector_search][lancedb.table.AsyncTable.vector_search].

    >>> async def search_table_for_vector():
    ...     db = await lancedb.connect_async("./.lancedb")
    ...     table = await db.open_table("my_table")
    ...     results = (
    ...       await table.vector_search([0.4, 0.4]).select(["b", "vector"]).to_pandas()
    ...     )
    ...     print(results)
    >>> asyncio.run(search_table_for_vector())
       b      vector  _distance
    0  4  [0.5, 1.3]       0.82
    1  2  [1.1, 1.2]       1.13

    Search queries are much faster when an index is created. See
    [AsyncTable.create_index][lancedb.table.AsyncTable.create_index].
    r   r  c                    || _         dS )a  Create a new AsyncTable object.

        You should not create AsyncTable objects directly.

        Use [AsyncConnection.create_table][lancedb.AsyncConnection.create_table] and
        [AsyncConnection.open_table][lancedb.AsyncConnection.open_table] to obtain
        Table objects.N)rI  rY  r   s     r]   r  zAsyncTable.__init__  s     r_   c                4    | j                                         S rp   )rI  r  rX  s    r]   r  zAsyncTable.__repr__  s    {##%%%r_   c                    | S rp   rV   rX  s    r]   	__enter__zAsyncTable.__enter__  s    r_   c                .    |                                   d S rp   )r  )rY  r\  s     r]   __exit__zAsyncTable.__exit__  s    

r_   rR   ra   c                4    | j                                         S )z!Return True if the table is open.)rI  is_openrX  s    r]   r  zAsyncTable.is_open  s    {""$$$r_   c                4    | j                                         S )zClose the table and free any resources associated with it.

        It is safe to call this method multiple times.

        Any attempt to use the table after it has been closed will raise an error.)rI  r  rX  s    r]   r  zAsyncTable.close  s     {  """r_   r   r  rS   c                   K   t          |t                    r|g}nt          |          }| j                            |           d{V  dS )a  Set the unenforced primary key for this table to the given
        ordered list of columns.

        "Unenforced" means LanceDB does not check uniqueness on writes; the
        columns are recorded in the schema as the primary key so that
        features such as `merge_insert` can use them. Calling this again
        replaces any previously-set primary key.

        Parameters
        ----------
        columns : str or Iterable[str]
            Either a single column name (single-column key) or an ordered
            iterable of column names (composite key). Each column dtype
            must be one of: int32, int64, utf8, large_utf8, binary,
            large_binary, fixed_size_binary.
        N)rZ   rQ   r   rI  rX  r9  s     r]   rX  z%AsyncTable.set_unenforced_primary_key  s\      & gs## 	$iGG7mmGk44W===========r_   rY  rZ  c                J   K   | j                             |           d{V  dS )u(  Install an LsmWriteSpec on this table.

        The spec selects Lance's MemWAL LSM-style write path for future
        `merge_insert` calls. ``LsmWriteSpec`` chooses one of three sharding
        strategies:

        - ``LsmWriteSpec.bucket(column, num_buckets)`` — hash-bucket writes by
          the single-column unenforced primary key.
        - ``LsmWriteSpec.identity(column)`` — shard by the raw value of a
          scalar column.
        - ``LsmWriteSpec.unsharded()`` — route every write to a single shard.

        All variants require the table to have an unenforced primary key set
        via [`set_unenforced_primary_key`]; bucket sharding additionally
        requires it to be the single column being bucketed.

        Parameters
        ----------
        spec : LsmWriteSpec
            The sharding spec to install.

        Examples
        --------
        >>> from lancedb._lancedb import LsmWriteSpec
        >>> # table.set_unenforced_primary_key("id")
        >>> # table.set_lsm_write_spec(LsmWriteSpec.bucket("id", 16))
        N)rI  r\  r]  s     r]   r\  zAsyncTable.set_lsm_write_spec  s6      8 k,,T22222222222r_   c                H   K   | j                                          d{V  dS )zRemove the LsmWriteSpec from this table.

        Reverts to the standard `merge_insert` write path. Errors if no spec
        is currently set.
        N)rI  r_  rX  s    r]   r_  zAsyncTable.unset_lsm_write_spec<  s4       k..00000000000r_   rQ   c                4    | j                                         S )zThe name of the table.)rI  r   rX  s    r]   r   zAsyncTable.nameD  s     {!!!r_   r   c                D   K   | j                                          d{V S )r_  N)rI  r   rX  s    r]   r   zAsyncTable.schemaI  s.      
 ['')))))))))r_   rg  c                   K   |                                   d{V }t          j                                        |j                  S )r  N)r   r&   r;  r<  r   )rY  r   s     r]   ri  zAsyncTable.embedding_functionsP  sF       {{}}$$$$$$(577GGXXXr_   Nrk  rm   r[  c                F   K   | j                             |           d{V S )rm  N)rI  re  rn  s     r]   re  zAsyncTable.count_rows]  s0       [++F333333333r_   r  rv  c                   K   |                                                      |                                           d{V S )z
        Return the first `n` rows of the table.

        Parameters
        ----------
        n: int, default 5
            The number of rows to return.
        N)rF  limitr   r  s     r]   r  zAsyncTable.headh  sB       ZZ\\''**33555555555r_   r7   c                N    t          | j                                                  S )a]  
        Returns an [AsyncQuery][lancedb.query.AsyncQuery] that can be used
        to search the table.

        Use methods on the returned query to control query behavior.  The query
        can be executed with methods like [to_arrow][lancedb.query.AsyncQuery.to_arrow],
        [to_pandas][lancedb.query.AsyncQuery.to_pandas] and more.
        )r7   rI  rF  rX  s    r]   rF  zAsyncTable.querys  s      $+++--...r_   ry  c                   K   	 dd l }n# t          $ r t          d          w xY w |j        |                                  d {V f|                                  d {V |                                  d {V d|S )Nr   r  r  )r   r  r   rJ  r]  r  )rY  ru  r   s      r]   	_to_lancezAsyncTable._to_lance~  s      	LLLL 	 	 	=  	 u}((**
,,..(((((("&"="="?"???????
 
 	
 
 	
s   	 #rI   ro  rp  r  c                   K   |dk    r% |                                   d{V j        di |S  |                                  d{V j        dd|i|S )a;  Return the table as a pandas DataFrame.

        Parameters
        ----------
        blob_mode: str, default "lazy"
            Controls how Lance blob columns are returned.
        **kwargs
            Forwarded to PyArrow / Lance pandas conversion.

        Returns
        -------
        pd.DataFrame
        rI   Nro  rV   )r   rs  r  rt  s      r]   rs  zAsyncTable.to_pandas  s       4$--//))))))4>>v>>>1dnn&&&&&&&&1PPIPPPPr_   c                ^   K   |                                                                   d{V S )rx  N)rF  r   rX  s    r]   r   zAsyncTable.to_arrow  s4       ZZ\\**,,,,,,,,,r_   T)r  r  r  r   r  r  r  r  r  _Optional[Union[IvfFlat, IvfPq, IvfRq, HnswPq, HnswSq, HnswFlat, BTree, Bitmap, LabelList, FTS]]r  r  r   r  c                 K   |~t          |t          t          t          t          t
          t          t          t          t          t          t          f          s,t          dt          t          |                    z             	 | j                            ||||||           d{V  dS # t"          t$          f$ r8}t          |t                    rt'          ||j        |j                   |d}~ww xY w)a  Create an index to speed up queries

        Indices can be created on vector columns or scalar columns.
        Indices on vector columns will speed up vector searches.
        Indices on scalar columns will speed up filtering (in both
        vector and non-vector searches)

        Parameters
        ----------
        column: str
            The column to index.
        replace: bool, default True
            Whether to replace the existing index

            If this is false, and another index already exists on the same columns
            and the same name, then an error will be returned.  This is true even if
            that index is out of date.

            The default is True
        config: default None
            For advanced configuration you can specify the type of index you would
            like to create.   You can also specify index-specific parameters when
            creating an index object.
        wait_timeout: timedelta, optional
            The timeout to wait if indexing is asynchronous.
        name: str, optional
            The name of the index. If not provided, a default name will be generated.
        train: bool, default True
            Whether to train the index with existing data. Vector indices always train
            with existing data.
        Nzmconfig must be an instance of IvfSq, IvfPq, IvfRq, HnswPq, HnswSq, BTree, Bitmap, LabelList, or FTS, but got )r4  r  r  r   r  r  )rZ   r(   r*   r)   r,   r.   r/   r0   r'   r+   r-   r1   r   rQ   r   rI  r  r   r  rv   r`   rl   )rY  r  r  r  r  r   r  r  s           r]   r  zAsyncTable.create_index  s=     l      BDGVDUDUV  	+**) +            L) 	 	 	&#&& )#)#8#_   
 G	s   &B, ,C5=3C00C5c                J   K   | j                             |           d{V  dS )a  
        Drop an index from the table.

        Parameters
        ----------
        name: str
            The name of the index to drop.

        Notes
        -----
        This does not delete the index from disk, it just removes it from the table.
        To delete the index, run [optimize][lancedb.table.AsyncTable.optimize]
        after dropping the index.

        Use [list_indices][lancedb.table.AsyncTable.list_indices] to find the names
        of the indices.
        N)rI  r  r  s     r]   r  zAsyncTable.drop_index  s6      $ k$$T***********r_   c                J   K   | j                             |           d{V  dS )r  N)rI  r  r  s     r]   r  zAsyncTable.prewarm_index  s6      ( k''-----------r_   r  c                J   K   | j                             |           d{V  dS )r  N)rI  r  r9  s     r]   r  zAsyncTable.prewarm_data0  s6      * k&&w///////////r_   r  r  r  r  r  r   c                L   K   | j                             ||           d{V  dS )r  N)rI  r  r  s      r]   r  zAsyncTable.wait_for_indexG  s8       k((g>>>>>>>>>>>r_   r  c                D   K   | j                                          d{V S )r  N)rI  r  rX  s    r]   r  zAsyncTable.statsX  s.       [&&(((((((((r_   c                D   K   | j                                          d{V S )a8  
        Get the table URI (storage location).

        For remote tables, this fetches the location from the server via describe.
        For local tables, this returns the dataset URI.

        Returns
        -------
        str
            The full storage location of the table (e.g., S3/GCS path).
        N)rI  rJ  rX  s    r]   rJ  zAsyncTable.uri^  s,       [__&&&&&&&&&r_   r  c                D   K   | j                                          d{V S )r  N)rI  r  rX  s    r]   r  z"AsyncTable.initial_storage_optionsl  s.       [88:::::::::r_   c                D   K   | j                                          d{V S )r  N)rI  r  rX  s    r]   r  z!AsyncTable.latest_storage_options{  s.       [77999999999r_   r   r  r   r"   r  (Optional[Literal['append', 'overwrite']]r   Optional[OnBadVectorsType]r   Optional[float]rR  r  r   c                 K   |                                   d{V }|d}|d}|dk    rt          |d||          \  }}n0|dk    s|j        #d|j        v rt          |||j        ||d          }t	                       t          |          }t          |          \  }}	 | j                            ||pd	|
           d{V 	 |r|	                                 S S # t          $ rF}	dt          |	          v rt          |	          dt          |	          v rt          |	           d}	~	ww xY w# |r|	                                 w w xY w)ae  Add more data to the [Table](Table).

        Parameters
        ----------
        data: DATA
            The data to insert into the table. Acceptable types are:

            - list-of-dict

            - pandas.DataFrame

            - pyarrow.Table or pyarrow.RecordBatch
        mode: str
            The mode to use when writing the data. Valid values are
            "append" and "overwrite".
        on_bad_vectors: str, default "error"
            What to do if any of the vectors are not the same size or contains NaNs.
            One of "error", "drop", "fill", "null".
        fill_value: float, default 0.
            The value to use when filling vectors. Only used if on_bad_vectors="fill".
        progress: callable or tqdm-like, optional
            A callback or tqdm-compatible progress bar. See
            :meth:`Table.add` for details.

        Nr   r   	overwrite)r   r   s   embedding_functionsTr   r   r   r   r   )rR  z
Cast errorzVector column contains NaN)r   r  r   r   r   r   rS  rI  r  r  r  rQ   r   )
rY  r   r  r   r   rR  r   r\  r  r  s
             r]   r  zAsyncTable.add  s     D {{}}$$$$$$!$NJ ; ,d>j  GD!! w&&O',Bfo,U,U!-% $  D 	&'''D!!,X66$	!t/?x(SSSSSSSSS  !    !  	 	 	s1vv%% mm#-Q77 mm#	  !    !s%   "$C 
D/)AD**D//D2 2Er  r2   c                    t          |t                    r|gnt          t          |                    }t	          | |          S r  r  r  s     r]   r  zAsyncTable.merge_insert  r  r_   .rF  r  r  Literal['auto']r  r  r  8Union[AsyncHybridQuery, AsyncFTSQuery, AsyncVectorQuery]c                
   K   d S rp   rV   r  s         r]   r  zAsyncTable.search  s       DG3r_   r#  r6   c                
   K   d S rp   rV   r  s         r]   r  zAsyncTable.search         3r_   .Optional[Union[VEC, 'PIL.Image.Image', Tuple]]r9   c                
   K   d S rp   rV   r  s         r]   r  zAsyncTable.search(  r  r_   r   r5   c                
   K   d S rp   rV   r  s         r]   r  zAsyncTable.search2  s       r_   r  r  c                
   K   d S rp   rV   r  s         r]   r  zAsyncTable.search<  s       3r_   r  r   c                   K   d }d fd
}d }|dk    r ||          r|}	d}nkt          |t                    rd}nRt          |t                    r	 t          j                                          ||d|                     d{V \  }
\  }3 ||           d{V }	t          fd|
D                       rd}nd}nd}n# t          $ r1}dt          |          v rdt          |          v rd}n|Y d}~nd}~ww xY wd}n|dk    r7 ||          r|}	nq ||||           d{V \  } ||           d{V }	nH|dk    rB ||          rt          d           ||||           d{V \  } ||           d{V }	|dk    r@                                 	                    |	          }|r|
                    |          }|S |dk    r)                                                     ||          S |dk    rU                                 	                    |	          }|r|
                    |          }|                    ||          S t          d| d          )a   Create a search query to find the nearest neighbors
        of the given query vector. We currently support [vector search][search]
        and [full-text search][experimental-full-text-search].

        All query options are defined in [AsyncQuery][lancedb.query.AsyncQuery].

        Parameters
        ----------
        query: list/np.ndarray/str/PIL.Image.Image, default None
            The targetted vector to search for.

            - *default None*.
            Acceptable types are: list, np.ndarray, PIL.Image.Image

            - If None then the select/where/limit clauses are applied to filter
            the table
        vector_column_name: str, optional
            The name of the vector column to search.

            The vector column needs to be a pyarrow fixed size list type

            - If not specified then the vector column is inferred from
            the table schema

            - If the table has multiple vector columns then the *vector_column_name*
            needs to be specified. Otherwise, an error is raised.
        query_type: str
            *default "auto"*.
            Acceptable types are: "vector", "fts", "hybrid", or "auto"

            - If "auto" then the query type is inferred from the query;

                - If `query` is a list/np.ndarray then the query type is
                "vector";

                - If `query` is a PIL.Image.Image then either do vector search,
                or raise an error if no corresponding embedding function is found.

            - If `query` is a string, then the query type is "vector" if the
              table has embedding functions else the query type is "fts"

        Returns
        -------
        LanceQueryBuilder
            A query builder object representing the query.
        c                p    t          | t          t          j        t          j        t          j        f          S rp   )rZ   r   npndarrayr   Arrayr  rE  s    r]   is_embeddingz'AsyncTable.search.<locals>.is_embedding  s!    edBJ"/%RSSSr_   r  rm   r  r   rF  r  rR   #Tuple[str, EmbeddingFunctionConfig]c                  K   t          |t                    rd}                                 d {V }t          ||||           } t	          j                                        |j                  }|                    |           }|kt          d|  d          }t          |          dk    r3t          |dt          |                                                      nt          |d           || |fS )NrQ  r'  zColumn 'z'' has no registered embedding function.r   z0Embedding functions are registered for columns: z6No embedding functions are registered for any columns.)rZ   r:   r   rE   r&   r;  r<  r   r  r   r   rB   r   r  )r  r  rF  r   funcsr7  r   rY  s          r]   get_embedding_funcz-AsyncTable.search.<locals>.get_embedding_func  s=     
 %// #"
;;==((((((F!9%#5	" " " .:<<LL E 99/00D|"*1 * * *  u::>>0

--0 0    W   %t++r_   c                   K   | @t          j                    }|                    d | j        j        |           d {V d         S d S )Nr   )asyncioget_running_looprun_in_executorr,  #compute_query_embeddings_with_retry)r  rF  loops      r]   make_embeddingz)AsyncTable.search.<locals>.make_embedding  sr      $/11 ..!*N       
   tr_   r  r  rQ  Nc              3  Z   K   | ]%}|j         d          j        k    o
|j        dk    V  &dS )r   r1   N)r   r2  r  )rf   r  embedding_confs     r]   rh   z$AsyncTable.search.<locals>.<genexpr>  sV         !" IaLN,HH 6 ! 5     r_   r"  Columnz$has no registered embedding functionz#Hybrid search requires a text query)r   zUnknown query type: '')r  rm   r  r   rF  r  rR   r"  )rZ   r:   rQ   r'  gatherr'  ri   r   rF  
nearest_tor  nearest_to_text)rY  rF  r  r  r  r  r!  r%  r,  vector_queryr  r  builderr.  s   `            @r]   r  zAsyncTable.searchH  s     r	T 	T 	T"	, "	, "	, "	, "	, "	,H	 	 	 |E"" &&$%

E=11 #&"

E3'' !&+ &n))++**+=vuMM       <+^ &1-;^NE-R-R'R'R'R'R'R'R     &-     2
 *2JJ)1JJ%*

+ "      3$ $  @CFFJJ &+

 #



 0 &

8##|E"" K$;M;M&
E< < 6 6 6 6 6 62"N &4^NE%J%JJJJJJJ8##|E"" K !FGGG;M;M&
E< < 6 6 6 6 6 62"N &4^NE%J%JJJJJJJ!!jjll--l;;G! =!..);<<N5  ::<<//{/KKK8##jjll--l;;G! =!..);<<**5+*FFFBZBBBCCCs   >C 
D'DDquery_vectorUnion[VEC, Tuple]c                P    |                                                      |          S )a2  
        Search the table with a given query vector.
        This is a convenience method for preparing a vector query and
        is the same thing as calling `nearestTo` on the builder returned
        by `query`.  Seer [nearest_to][lancedb.query.AsyncQuery.nearest_to] for more
        details.
        )rF  r2  )rY  r6  s     r]   vector_searchzAsyncTable.vector_search   s      zz||&&|444r_   rA   @AsyncHybridQuery | AsyncFTSQuery | AsyncVectorQuery | AsyncQueryc                B   |                                  }|j        |                    |j                  }|j        |                    |j                  }|j        r|                    |j                  }|j        |                    |j                  }|j        r|                                }|j        r|                                }|j	        r|	                    |j	                  }|j
        r[|                    |j
                                      |j        |j                  }|j        |                    |j                  }|j        :|j        3|                    |j                                      |j                  }nC|j        |                    |j                  }n!|j        |                    |j                  }|j        |                    |j                  }|j        r|                    |j                  }|j        r|                    |j                  }|j        r|                                }|j        r|                                }|j        r*|                    |j        j         |j        j                  }|S rp   )rF  r  offsetr   r  rk  r  fast_searchr  order_byr  r2  distance_rangelower_boundupper_boundr  minimum_nprobesmaximum_nprobesnprobesrefine_factorr5  r  efbypass_vector_index
postfilterfull_text_queryr3  rY  rF  async_querys      r]   _sync_query_to_asynczAsyncTable._sync_query_to_async  s    jjll;"%++EK88K<#%,,U\::K= 	<%,,U];;K<#%++EL99K 	4%1133K 	4%1133K> 	?%..u~>>K< 	@%00>>MM!5#4 K ".)778KLL$0U5J5V)11) !/%"788  &2)99%:OPP&2)99%:OPP".)778KLL" F)001DEEx 7)nnUX66( @)==?? 	3%0022K  	%55%+U-B-J K r_   r  r  r  r   c               j   K   |                      |          }|                    ||           d {V S )N)max_batch_lengthr  )rL  r   )rY  rF  r  r  rK  s        r]   r  zAsyncTable._execute_queryB  s[       //66 ++' , 
 
 
 
 
 
 
 
 	
r_   r  c                f   K   |                      |          }|                    |           d {V S rp   )rL  explain_plan)rY  rF  r  rK  s       r]   r   zAsyncTable._explain_planS  s?      //66 --g666666666r_   c                d   K   |                      |          }|                                 d {V S rp   )rL  analyze_planrJ  s      r]   r  zAsyncTable._analyze_planX  s=      //66 --/////////r_   c                d   K   |                      |          }|                                 d {V S rp   )rL  r}  rJ  s      r]   r  zAsyncTable._output_schema]  s=      //66 ..000000000r_   r  r  r   r   r   c                  K   |                                   d {V }|d}|d}t          |||j        ||d          }t          |t          j                  r7t          j                            |j         |                                          }| j	        
                    |t          |j        |j        |j        |j        |j        |j        |j        |j                             d {V S )Nr   r   Tr  )r  when_matched_update_all!when_matched_update_all_conditionwhen_not_matched_insert_all!when_not_matched_by_source_delete$when_not_matched_by_source_conditionr  	use_index)r   r   r   rZ   r   ry   r   r   r   rI  execute_merge_insertr   _on_when_matched_update_all"_when_matched_update_all_condition_when_not_matched_insert_all"_when_not_matched_by_source_delete%_when_not_matched_by_source_condition_timeout
_use_index)rY  r  r  r   r   r   r   s          r]   r
  zAsyncTable._do_mergea  s      {{}}$$$$$$!$NJ_)! 
 
 
 dBH%% 	U'44T[$//BSBSTTD[559(-(F272Z,1,N272Z5:5`*	 	 	
 
 
 
 
 
 
 
 	
r_   r  r   c                F   K   | j                             |           d{V S )a  Delete rows from the table.

        This can be used to delete a single row, many rows, all rows, or
        sometimes no rows (if your predicate matches nothing).

        Parameters
        ----------
        where: str
            The SQL where clause to use when deleting rows.

            - For example, 'x = 2' or 'x IN (1, 2, 3)'.

            The filter must not be empty, or it will error.

        Examples
        --------
        >>> import lancedb
        >>> data = [
        ...    {"x": 1, "vector": [1.0, 2]},
        ...    {"x": 2, "vector": [3.0, 4]},
        ...    {"x": 3, "vector": [5.0, 6]}
        ... ]
        >>> db = lancedb.connect("./.lancedb")
        >>> table = db.create_table("my_table", data)
        >>> table.to_pandas()
           x      vector
        0  1  [1.0, 2.0]
        1  2  [3.0, 4.0]
        2  3  [5.0, 6.0]
        >>> table.delete("x = 2")
        DeleteResult(num_deleted_rows=1, version=2)
        >>> table.to_pandas()
           x      vector
        0  1  [1.0, 2.0]
        1  3  [5.0, 6.0]

        If you have a list of values to delete, you can combine them into a
        stringified list and use the `IN` operator:

        >>> to_remove = [1, 5]
        >>> to_remove = ", ".join([str(v) for v in to_remove])
        >>> to_remove
        '1, 5'
        >>> table.delete(f"x IN ({to_remove})")
        DeleteResult(num_deleted_rows=1, version=3)
        >>> table.to_pandas()
           x      vector
        0  3  [5.0, 6.0]
        N)rI  r  r  s     r]   r  zAsyncTable.delete  s1      d [''.........r_   r9  updatesOptional[Dict[str, Any]]r:  r   c                  K   ||t          d          ||t          d          |d |                                D             }| j                            ||           d{V S )a  
        This can be used to update zero to all rows in the table.

        If a filter is provided with `where` then only rows matching the
        filter will be updated.  Otherwise all rows will be updated.

        Parameters
        ----------
        updates: dict, optional
            The updates to apply.  The keys should be the name of the column to
            update.  The values should be the new values to assign.  This is
            required unless updates_sql is supplied.
        where: str, optional
            An SQL filter that controls which rows are updated. For example, 'x = 2'
            or 'x IN (1, 2, 3)'.  Only rows that satisfy this filter will be udpated.
        updates_sql: dict, optional
            The updates to apply, expressed as SQL expression strings.  The keys should
            be column names. The values should be SQL expressions.  These can be SQL
            literals (e.g. "7" or "'foo'") or they can be expressions based on the
            previous value of the row (e.g. "x + 1" to increment the x column by 1)

        Returns
        -------
        UpdateResult
            An object containing:
            - rows_updated: The number of rows that were updated
            - version: The new version number of the table after the update

        Examples
        --------
        >>> import asyncio
        >>> import lancedb
        >>> import pandas as pd
        >>> async def demo_update():
        ...     data = pd.DataFrame({"x": [1, 2], "vector": [[1, 2], [3, 4]]})
        ...     db = await lancedb.connect_async("./.lancedb")
        ...     table = await db.create_table("my_table", data)
        ...     # x is [1, 2], vector is [[1, 2], [3, 4]]
        ...     await table.update({"vector": [10, 10]}, where="x = 2")
        ...     # x is [1, 2], vector is [[1, 2], [10, 10]]
        ...     await table.update(updates_sql={"x": "x + 1"})
        ...     # x is [2, 3], vector is [[1, 2], [10, 10]]
        >>> asyncio.run(demo_update())
        Nz2Only one of updates or updates_sql can be providedz.Either updates or updates_sql must be providedc                4    i | ]\  }}|t          |          S rV   )rG   r   s      r]   r   z%AsyncTable.update.<locals>.<dictcomp>  s$    JJJ$!Q1l1ooJJJr_   )r   r   rI  r  )rY  re  r  r:  s       r]   r  zAsyncTable.update  s      f ;#:QRRR?{2MNNNJJ'--//JJJK[''U;;;;;;;;;r_   r.  6dict[str, str] | pa.field | List[pa.field] | pa.Schemar~   c                  K   t          |t          j                  r|g}t          |t                    r-t	          d |D                       rt          j        |          }t          |t          j                  r | j                            |           d{V S | j        	                    t          |
                                                     d{V S )aX  
        Add new columns with defined values.

        Parameters
        ----------
        transforms: Dict[str, str]
            A map of column name to a SQL expression to use to calculate the
            value of the new column. These expressions will be evaluated for
            each row in the table, and can reference existing columns.
            Alternatively, you can pass a pyarrow field or schema to add
            new columns with NULLs.

        Returns
        -------
        AddColumnsResult
            version: the new version number of the table after adding columns.

        c                B    h | ]}t          |t          j                  S rV   )rZ   r   Fieldr  s     r]   	<setcomp>z)AsyncTable.add_columns.<locals>.<setcomp>  s$    999Z28$$999r_   N)rZ   r   rl  r   r.  r   SchemarI  add_columns_with_schemar2  r   r1  s     r]   r2  zAsyncTable.add_columns  s      * j"(++ 	&$Jj$'' 	/C99j999-
 -
 	/ :..Jj"),, 	K<<ZHHHHHHHHH00j6F6F6H6H1I1IJJJJJJJJJr_   r3  Iterable[dict[str, Any]]r   c                F   K   | j                             |           d{V S )a  
        Alter column names and nullability.

        alterations : Iterable[Dict[str, Any]]
            A sequence of dictionaries, each with the following keys:
            - "path": str
                The column path to alter. For a top-level column, this is the name.
                For a nested column, this is the dot-separated path, e.g. "a.b.c".
            - "rename": str, optional
                The new name of the column. If not specified, the column name is
                not changed.
            - "data_type": pyarrow.DataType, optional
               The new data type of the column. Existing values will be casted
               to this type. If not specified, the column data type is not changed.
            - "nullable": bool, optional
                Whether the column should be nullable. If not specified, the column
                nullability is not changed. Only non-nullable columns can be changed
                to nullable. Currently, you cannot change a nullable column to
                non-nullable.

        Returns
        -------
        AlterColumnsResult
            version: the new version number of the table after the alteration.
        N)rI  r7  r6  s     r]   r7  zAsyncTable.alter_columns  s0      8 [..{;;;;;;;;;r_   c                F   K   | j                             |           d{V S )z
        Drop columns from the table.

        Parameters
        ----------
        columns : Iterable[str]
            The names of the columns to drop.
        N)rI  r  r9  s     r]   r  zAsyncTable.drop_columns4  s0       [--g666666666r_   c                D   K   | j                                          d{V S )a  
        Retrieve the version of the table

        LanceDb supports versioning.  Every operation that modifies the table increases
        version.  As long as a version hasn't been deleted you can `[Self::checkout]`
        that version to view the data at that point.  In addition, you can
        `[Self::restore]` the version to replace the current table with a previous
        version.
        N)rI  r]  rX  s    r]   r]  zAsyncTable.version?  s.       [((*********r_   c                   K   | j                                          d{V }|D ];}|d         }t          j        |dz            t	          |dz  dz            z   |d<   <|S )z0
        List all versions of the table
        N	timestampg    eAg     @@)microseconds)rI  rH  r   fromtimestampr   )rY  versionsr   ts_nanoss       r]   rH  zAsyncTable.list_versionsK  s       2244444444 	 	A~H%3HODDy&n4H H H AkNN r_   r]  	int | strc                   K   	 | j                             |           d{V  dS # t          $ r*}dt          |          v rt	          d| d           d}~ww xY w)r<  Nz	not foundzVersion z% no longer exists. Was it cleaned up?)rI  r>  r  rQ   r   )rY  r]  r  s      r]   r>  zAsyncTable.checkoutX  s      .	+&&w/////////// 	 	 	c!ff$$ MwMMM   	s    & 
A%AAc                H   K   | j                                          d{V  dS r@  )rI  rA  rX  s    r]   rA  zAsyncTable.checkout_latesty  s4       k))+++++++++++r_   Optional[int | str]c                J   K   | j                             |           d{V  dS )a  
        Restore the table to the currently checked out version

        This operation will fail if checkout has not been called previously

        This operation will overwrite the latest version of the table with a
        previous version.  Any changes made since the checked out version will
        no longer be visible.

        Once the operation concludes the table will no longer be in a checked
        out state and the read_consistency_interval, if any, will apply.
        N)rI  rD  r=  s     r]   rD  zAsyncTable.restore  s6       k!!'***********r_   r  r  r8   c                P    t          | j                            |                    S )a  
        Take a list of offsets from the table.

        Offsets are 0-indexed and relative to the current version of the table.  Offsets
        are not stable.  A row with an offset of N may have a different offset in a
        different version of the table (e.g. if an earlier row is deleted).

        Offsets are mostly useful for sampling as the set of all valid offsets is easily
        known in advance to be [0, len(table)).

        Parameters
        ----------
        offsets: list[int]
            The offsets to take.

        Returns
        -------
        pa.RecordBatch
            A record batch containing the rows at the given offsets.
        )r8   rI  r  r  s     r]   r  zAsyncTable.take_offsets  s"    * dk66w??@@@r_   r  c                P    t          | j                            |                    S )ay  
        Take a list of row ids from the table.

        Row ids are not stable and are relative to the current version of the table.
        They can change due to compaction and updates.

        Unlike offsets, row ids are not 0-indexed and no assumptions should be made
        about the possible range of row ids.  In order to use this method you must
        first obtain the row ids by scanning or searching the table.

        Even so, row ids are more stable than offsets and can be useful in some
        situations.

        There is an ongoing effort to make row ids stable which is tracked at
        https://github.com/lancedb/lancedb/issues/1120

        Parameters
        ----------
        row_ids: list[int]
            The row ids to take.

        Returns
        -------
        AsyncTakeQuery
            A query object that can be executed to get the rows.
        )r8   rI  r  r  s     r]   r  zAsyncTable.take_row_ids  s"    6 dk66w??@@@r_   	AsyncTagsc                *    t          | j                  S )a  Tag management for the dataset.

        Similar to Git, tags are a way to add metadata to a specific version of the
        dataset.

        .. warning::

            Tagged versions are exempted from the
            :py:meth:`optimize(cleanup_older_than)` process.

            To remove a version that has been tagged, you must first
            :py:meth:`~Tags.delete` the associated tag.

        )r  rI  rX  s    r]   rc  zAsyncTable.tags  s      %%%r_   Fr  r  r  rz   c                  K   d}|$t          |                                dz            }|rddl} |j        dt                     | j                            ||           d{V S )r"  Ni  r   zNThe 'retrain' parameter is deprecated and will be removed in a future version.)cleanup_since_msr  )roundtotal_secondsr3  r4  r5  rI  r$  )rY  r  r  r   r  r3  s         r]   r$  zAsyncTable.optimize  s      Z +/)$%7%E%E%G%G$%NOO 	OOOHM""   [))-/ * 
 
 
 
 
 
 
 
 	
r_   r%  c                D   K   | j                                          d{V S )rQ  N)rI  r'  rX  s    r]   r'  zAsyncTable.list_indices  s.       [--/////////r_   r(  r)  c                f   K   | j                             |           d{V }|dS t          di |S )r+  NrV   )rI  r-  IndexStatistics)rY  r(  r  s      r]   r-  zAsyncTable.index_stats  sM       k--j99999999=4"++U+++r_   c                D   K   | j                                          d{V S r`  )rI  rb  rX  s    r]   rb  z!AsyncTable.uses_v2_manifest_paths1  s.       [77999999999r_   c                H   K   | j                                          d{V  dS )az  
        Migrate the manifest paths to the new format.

        This will update the manifest to use the new v2 format for paths.

        This function is idempotent, and can be run multiple times without
        changing the state of the object store.

        !!! danger

            This should not be run while other concurrent operations are happening.
            And it should also run until completion before resuming other operations.

        You can use
        [AsyncTable.uses_v2_manifest_paths][lancedb.table.AsyncTable.uses_v2_manifest_paths]
        to check if the table is already using the new path style.
        N)rI  migrate_manifest_paths_v2rX  s    r]   r  z$AsyncTable.migrate_manifest_paths_v2<  s4      $ k3355555555555r_   rb  rc  dict[str, str]c                L   K   | j                             ||           d{V  dS rf  )rI  rg  rh  s      r]   rg  z!AsyncTable.replace_field_metadataP  s8       k00\JJJJJJJJJJJr_   )r   r  r  rm  rn  ro  re  rg  ri  rp   rj  ri  rl  )rR   r7   rm  rk  rj  )r  rQ   r  r  r  r  r  r  r   rm   r  ra   rn  rk  ro  rp  rl  )r   r"   r  r  r   r  r   r  rR  r  rR   r   rs  )NN.NN)rF  rm   r  rm   r  r  r  rm   r  r  rR   r  )rF  rm   r  rm   r  r#  r  rm   r  r  rR   r6   )rF  r  r  rm   r  r  r  rm   r  r  rR   r9   )rF  rm   r  rm   r  r   r  rm   r  r  rR   r5   )rF  r  r  rm   r  r  r  rm   r  r  rR   r9   rt  )rF  r  r  rm   r  r   r  rm   r  r  rR   r  )r6  r7  rR   r9   )rF  rA   rR   r:  rv  rx  ry  rz  r{  r|  )re  rf  r  rm   r:  r  rR   r   )r.  ri  rR   r~   )r3  rp  rR   r   )r   r  rf  )r]  rz  )r]  r}  )r  r  rR   r8   )r  r  rR   r8   )rR   r  )r  r  r  ra   rR   rz   r  r  )rb  rQ   rc  r  )?r   r   r  r  r  r  r  r  r  r  rX  r\  r_  r  r   r   ri  re  r  rF  r  rs  r   r  r  r  r  r   r  r  rJ  r  r  r  r  r   r  r9  rL  r  r   r  r  r
  r  r  r2  r7  r  r]  rH  r>  rA  rD  r  r  rc  r$  r'  r-  rb  r  rg  rV   r_   r]   r  r    s       @ @D   & & &    % % % %# # #> > > >23 3 3 3<1 1 1 1 " " " X"* * * *Y Y Y Y	4 	4 	4 	4 	4	6 	6 	6 	6 	6	/ 	/ 	/ 	/
 
 
 
 Q Q Q Q Q$- - - - #' ,0"+[ [ [ [ [ [z+ + + +(. . . .,0 0 0 0 00 @IyQT?U?U?U? ? ? ? ?") ) ) )' ' ' '; ; ; ;: : : :( :B59&*9=I! I! I! I! I! I!V<1 <1 <1 <1|   $,0&)-17;G G G G XG   $,0(+-17;    X  AE,0&)-17;    X   $,0%(-17;    X 
 ,0(+-17;	 	 	 	 X	 ,0 &-17;vD vD vD vD vDp5 5 5 53 3 3 3r %)'+
 
 
 
 
 
"7 7 7 7
0 0 0 0
1 1 1 1"
 "
 "
 "
H2/ 2/ 2/ 2/l -1;<  $04;< ;< ;< ;< ;< ;<zK K K K@< < < <<	7 	7 	7 	7
+ 
+ 
+ 
+     B, , ,+ + + + +A A A A.A A A A: & & & X&( 37"'=
 =
 =
 =
 =
 =
~0 0 0 0, , , ,(	: 	: 	: 	:6 6 6(K K K K K Kr_   r  c                  b    e Zd ZU dZded<   ded<   ded<   dZded	<   dZd
ed<   dZded<   d ZdS )r  a/  
    Statistics about an index.

    Attributes
    ----------
    num_indexed_rows: int
        The number of rows that are covered by this index.
    num_unindexed_rows: int
        The number of rows that are not covered by this index.
    index_type: str
        The type of index that was created.
    distance_type: Optional[str]
        The distance type used by the index.
    num_indices: Optional[int]
        The number of parts the index is split into.
    loss: Optional[float]
        The KMeans loss for the index, for only vector indices.
    r[  num_indexed_rowsnum_unindexed_rowszLiteral['IVF_FLAT', 'IVF_SQ', 'IVF_PQ', 'IVF_RQ', 'IVF_HNSW_SQ', 'IVF_HNSW_PQ', 'IVF_HNSW_FLAT', 'FTS', 'BTREE', 'BITMAP', 'LABEL_LIST']r  Nz(Optional[Literal['l2', 'cosine', 'dot']]r  r  num_indicesr  lossc                "    t          | |          S rp   )rX   )rY  r'  s     r]   __getitem__zIndexStatistics.__getitem__  s    tS!!!r_   )	r   r   r  r  __annotations__r  r  r  r  rV   r_   r]   r  r  `  s          &     ?CMBBBB!%K%%%% D    " " " " "r_   r  c                  <    e Zd ZU dZded<   ded<   ded<   ded<   dS )	r  au  
    Statistics about a table and fragments.

    Attributes
    ----------
    total_bytes: int
        The total number of bytes in the table.
    num_rows: int
        The total number of rows in the table.
    num_indices: int
        The total number of indices in the table.
    fragment_stats: FragmentStatistics
        Statistics about fragments in the table.
    r[  total_bytesr&  r  FragmentStatisticsfragment_statsNr   r   r  r  r  rV   r_   r]   r  r    sK           MMM&&&&&&r_   r  c                  2    e Zd ZU dZded<   ded<   ded<   dS )r  a  
    Statistics about fragments.

    Attributes
    ----------
    num_fragments: int
        The total number of fragments in the table.
    num_small_fragments: int
        The total number of small fragments in the table.
        Small fragments have low row counts and may need to be compacted.
    lengths: FragmentSummaryStats
        Statistics about the number of rows in the table fragments.
    r[  num_fragmentsnum_small_fragmentsFragmentSummaryStatsr  Nr  rV   r_   r]   r  r    sB           !!!!!!r_   r  c                  Z    e Zd ZU dZded<   ded<   ded<   ded<   ded<   ded<   ded	<   d
S )r  aW  
    Statistics about fragments sizes

    Attributes
    ----------
    min: int
        The number of rows in the fragment with the fewest rows.
    max: int
        The number of rows in the fragment with the most rows.
    mean: int
        The mean number of rows in the fragments.
    p25: int
        The 25th percentile of number of rows in the fragments.
    p50: int
        The 50th percentile of number of rows in the fragments.
    p75: int
        The 75th percentile of number of rows in the fragments.
    p99: int
        The 99th percentile of number of rows in the fragments.
    r[  r  r  meanp25p50p75p99Nr  rV   r_   r]   r  r    s[          * HHHHHHIIIHHHHHHHHHHHHHHr_   r  c                  @    e Zd ZdZd ZddZdd	ZddZddZddZ	dS )ra  z
    Table tag manager.
    c                    || _         d S rp   r  r  s     r]   r  zTags.__init__      r_   rR   Dict[str, Tag]c                b    t          j        | j        j                                                  S )
        List all table tags.

        Returns
        -------
        dict[str, Tag]
            A dictionary mapping tag names to version numbers.
        )r   r  r  rc  r   rX  s    r]   r   z	Tags.list  s%     x(--//000r_   tagrQ   r[  c                d    t          j        | j        j                            |                    S )
        Get the version of a tag.

        Parameters
        ----------
        tag: str,
            The name of the tag to get the version for.
        )r   r  r  rc  get_versionrY  r  s     r]   r  zTags.get_version  s'     x(44S99:::r_   r]  rS   c                j    t          j        | j        j                            ||                     dS a!  
        Create a tag for a given table version.

        Parameters
        ----------
        tag: str,
            The name of the tag to create. This name must be unique among all tag
            names for the table.
        version: int,
            The table version to tag.
        N)r   r  r  rc  r(  rY  r  r]  s      r]   r(  zTags.create  s/     	!((g6677777r_   c                h    t          j        | j        j                            |                     dS z
        Delete tag from the table.

        Parameters
        ----------
        tag: str,
            The name of the tag to delete.
        N)r   r  r  rc  r  r  s     r]   r  zTags.delete  s-     	!((--.....r_   c                j    t          j        | j        j                            ||                     dS z
        Update tag to a new version.

        Parameters
        ----------
        tag: str,
            The name of the tag to update.
        version: int,
            The new table version to tag.
        N)r   r  r  rc  r  r  s      r]   r  zTags.update  s/     	!((g6677777r_   NrR   r  r  rQ   rR   r[  r  rQ   r]  r[  rR   rS   r  rQ   rR   rS   
r   r   r  r  r  r   r  r(  r  r  rV   r_   r]   ra  ra    s           	1 	1 	1 	1	; 	; 	; 	;8 8 8 8	/ 	/ 	/ 	/8 8 8 8 8 8r_   ra  c                  @    e Zd ZdZd ZddZdd	ZddZddZddZ	dS )r  z"
    Async table tag manager.
    c                    || _         d S rp   r  r  s     r]   r  zAsyncTags.__init__$  r  r_   rR   r  c                N   K   | j         j                                         d{V S )r  N)r  rc  r   rX  s    r]   r   zAsyncTags.list'  s1       [%**,,,,,,,,,r_   r  rQ   r[  c                P   K   | j         j                            |           d{V S )r  N)r  rc  r  r  s     r]   r  zAsyncTags.get_version2  s3       [%11#666666666r_   r]  rS   c                V   K   | j         j                            ||           d{V  dS r  )r  rc  r(  r  s      r]   r(  zAsyncTags.create=  s;       k%%c733333333333r_   c                T   K   | j         j                            |           d{V  dS r  )r  rc  r  r  s     r]   r  zAsyncTags.deleteK  s9       k%%c***********r_   c                V   K   | j         j                            ||           d{V  dS r  )r  rc  r  r  s      r]   r  zAsyncTags.updateV  s;       k%%c733333333333r_   Nr  r  r  r  r  rV   r_   r]   r  r    s           	- 	- 	- 	-	7 	7 	7 	74 4 4 4	+ 	+ 	+ 	+4 4 4 4 4 4r_   r  )rN   rO   rP   rQ   rR   rS   )r`   rQ   rR   ra   )rN   rO   r`   rQ   rl   rm   rR   rS   rp   )r   r   rR   r   )r   r   rR   r   )NNr   r   )r   r   r   r   r   r   r   r   r   r   r   ra   rR   r   rw  )r   r   r   r   r   ra   rR   r   )r   r   r   r   rR   r   )r   r   r  r   rR   r   )Nr   r   )r   r  r   r   r   r   rg  )r   r   r   r   r   r   rR   r   )r@  rQ   rA  rQ   rR   rQ   )r   r   NN)r   r   r   rt  r   r   r   r   r   r   rR   r   )r  r   r   r   r   r   rR   r  )r  r   r~  r  rR   r   )r   r  r5  r   rR   r  )r   r  r  rQ   r   rQ   r   r   rw  r  rx  r  rR   r  )r  r  rR   r  )r  r  rR   ra   )r  r   rR   r   )rb  rQ   rR   ra   )r   r   rR   r  )r  r  rR   r[  )r  r  rR   r  )r   r   )r   r   )
__future__r   r'  r  rr  r3  abcr   r   dataclassesr   r   r   	functoolsr	   typingr
   r   r   r   r   r   r   r   r   r   r   urllib.parser   lancedb.scannabler   r   rW   r   lancedb.arrowr   lancedb.background_loopr   dependenciesr   r   r   r   r    r   r!   r  pyarrowr   pyarrow.datasetpyarrow.computecomputer-  
pyarrow.fsrX  rT  numpyr  commonr"   r#   r$   
embeddingsr%   r&   r4  r'   r(   r)   r*   r+   r,   r-   r.   r/   r0   r1   r  r2   pydanticr3   r4   rF  r5   r6   r7   r8   r9   r:   r;   r<   r=   r>   r?   r@   rA   utilrB   rC   rD   rE   rF   rG   rH   rp  rj   rt   r^   rk   rv   r  rx   _lancedbry   r  rz   r{   r|   r}   r~   r   r   r   r   r   r   r   r   PILr  r   r   r   r   r   r   r   r   r   r   r   r   r   r   r  r   r   r   rL  rG  rS  r  r   r  r  r  rz  r  r  r   r  r   r  ry  r   r  r  r  r  r  r  ra  r  rV   r_   r]   <module>r     sC
   # " " " " "        # # # # # # # # ! ! ! ! ! ! ( ( ( ( ( ( ( ( % % % % % %                          " ! ! ! ! ! I I I I I I I I       % % % % % % ( ( ( ( ( (                                        1 1 1 1 1 1 1 1 1 1 J J J J J J J J                          + * * * * * / / / / / / / /                                                   23#7  " " " " "    QU     4  %%%%%%                              #"""""MMMJJJ	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 )-J
 J
 J
 J
 J
Z< < < <4 *.#'.B "B B B B B BP "$F $F $F $F $FN: : : :z% % % %V '.* * * * *Z     $ #'4<  $	4< 4< 4< 4< 4< 4<n0 0 0 0 1 1 1 1  y y y y yC y y yx%DO DO DO DO DO DO DO DOR. @G)-#/C /C /C /C /CdC C C CL> > > >    0 ""&15cB cB cB cB cBL. . . ."      ? ? ? ?5 5 5 5pA A A A
/ / / /, , , ," " " "&zK zK zK zK zK zK zK zKz- *" *" *" *" *" *" *" *"Z ' ' ' ' ' ' ' ', " " " " " " " "(        >B8 B8 B8 B8 B8 B8 B8 B8JB4 B4 B4 B4 B4 B4 B4 B4 B4 B4r_   