Compare commits

..

1 Commits

Author SHA1 Message Date
Gatefixer 3f9ff474c9 test(rust): guard parallel compaction fragment reservation 2026-08-06 00:15:11 +00:00
6 changed files with 70 additions and 108 deletions
-19
View File
@@ -17,25 +17,6 @@ The general flow of using the API is:
pip install lancedb
```
The core package does not require PyLance. When you need access to the underlying
Lance dataset or GPU-accelerated indexing, add the `pylance` extra to the
distribution you already installed.
For the standard distribution:
```shell
pip install "lancedb[pylance]"
```
For the pre-Haswell compatibility distribution:
```shell
pip install "lancedb-compat[pylance]"
```
Use only the extra matching your installed distribution. Do not install both
distributions because they share the `lancedb` namespace.
The following methods describe the synchronous API client. There
is also an [asynchronous API client](#connections-asynchronous).
-19
View File
@@ -8,25 +8,6 @@ A Python library for [LanceDB](https://github.com/lancedb/lancedb).
pip install lancedb
```
The core package does not require PyLance. When you need access to the underlying
Lance dataset or GPU-accelerated indexing, add the `pylance` extra to the
distribution you already installed.
For the standard distribution:
```bash
pip install "lancedb[pylance]"
```
For the pre-Haswell compatibility distribution:
```bash
pip install "lancedb-compat[pylance]"
```
Use only the extra matching your installed distribution. Do not install both
distributions because they share the `lancedb` namespace.
### Pre-Haswell x86_64 hosts: `lancedb-compat`
The default `lancedb` wheel targets `x86-64-haswell` (AVX2 + FMA + F16C) for full performance on modern hardware. Pre-Haswell hosts — Intel Sandy Bridge / Ivy Bridge / Westmere; AMD Bulldozer / Piledriver / Steamroller — don't have AVX2 and crash with `Illegal instruction` at `import lancedb`.
+8 -10
View File
@@ -117,14 +117,6 @@ _MODEL_BACKED_TOKENIZER_ERRORS = (
"Failed to initialize default tokenizer",
)
_PYLANCE_INSTALL_ERROR = (
"The lance library is required to use this function. Install the PyLance "
"extra for the distribution already installed: "
'`pip install "lancedb[pylance]"` for `lancedb`, or '
'`pip install "lancedb-compat[pylance]"` for `lancedb-compat`. '
"Do not install both distributions because they share the `lancedb` namespace."
)
def _add_unique_note(exception: BaseException, note: str) -> None:
existing_notes = getattr(exception, "__notes__", ()) or ()
@@ -2257,7 +2249,10 @@ class LanceTable(Table):
try:
import lance
except ImportError:
raise ImportError(_PYLANCE_INSTALL_ERROR)
raise ImportError(
"The lance library is required to use this function. "
"Please install with `pip install pylance`."
)
branch = self.current_branch()
version = None if branch is not None else self.version
@@ -4762,7 +4757,10 @@ class AsyncTable:
try:
import lance
except ImportError:
raise ImportError(_PYLANCE_INSTALL_ERROR)
raise ImportError(
"The lance library is required to use this function. "
"Please install with `pip install pylance`."
)
# lance.dataset() can't open a branch directly, so open the base table
# and check out the branch ref (a None branch resolves to main).
-30
View File
@@ -1,30 +0,0 @@
# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright The LanceDB Authors
import subprocess
import sys
def test_import_lancedb_without_pylance():
script = """
import sys
class BlockLanceImports:
def find_spec(self, fullname, path=None, target=None):
if fullname == "lance" or fullname.startswith("lance."):
raise ModuleNotFoundError(f"blocked optional dependency: {fullname}")
return None
sys.meta_path.insert(0, BlockLanceImports())
import lancedb
"""
result = subprocess.run(
[sys.executable, "-c", script],
capture_output=True,
text=True,
)
assert result.returncode == 0, result.stderr
-30
View File
@@ -1258,24 +1258,6 @@ def test_branch_to_lance_targets_branch(tmp_path):
assert table.to_lance().count_rows() == 1
def _assert_pylance_install_error(error: ImportError):
message = str(error)
assert 'pip install "lancedb[pylance]"' in message
assert 'pip install "lancedb-compat[pylance]"' in message
assert "distribution already installed" in message
assert "Do not install both distributions" in message
def test_to_lance_recommends_pylance_extra(tmp_db):
table = tmp_db.create_table("t", [{"i": 1}])
with patch("builtins.__import__", side_effect=ImportError):
with pytest.raises(ImportError) as exc_info:
table.to_lance()
_assert_pylance_install_error(exc_info.value)
@pytest.mark.asyncio
async def test_async_to_lance(tmp_path):
pytest.importorskip("lance")
@@ -1287,18 +1269,6 @@ async def test_async_to_lance(tmp_path):
assert dataset.count_rows() == 1
@pytest.mark.asyncio
async def test_async_to_lance_recommends_pylance_extra(tmp_path):
db = await lancedb.connect_async(tmp_path)
table = await db.create_table("t", [{"i": 1}])
with patch("builtins.__import__", side_effect=ImportError):
with pytest.raises(ImportError) as exc_info:
await table.to_lance()
_assert_pylance_install_error(exc_info.value)
@pytest.mark.asyncio
async def test_async_branch_to_lance_targets_branch(tmp_path):
pytest.importorskip("lance")
+62
View File
@@ -304,6 +304,68 @@ mod tests {
assert_eq!(all_values, expected);
}
#[tokio::test]
async fn test_parallel_compaction_reserves_fragment_ids_once() {
let conn = connect("memory://").execute().await.unwrap();
let schema = Arc::new(Schema::new(vec![Field::new("i", DataType::Int32, false)]));
let batch =
RecordBatch::try_new(schema, vec![Arc::new(Int32Array::from_iter_values(0..10))])
.unwrap();
let table = conn
.create_table("test_parallel_compaction", batch.clone())
.execute()
.await
.unwrap();
// Create 64 fragments. With a 20-row target, compaction plans 32 tasks,
// which is more than the commit retry limit that used to be exhausted
// when each parallel task reserved fragment IDs independently.
for _ in 1..64 {
table.add(batch.clone()).execute().await.unwrap();
}
// Legacy row IDs require fragment IDs before an index can be remapped.
assert!(
!table
.as_native()
.unwrap()
.manifest()
.await
.unwrap()
.uses_stable_row_ids()
);
table
.create_index(&["i"], Index::BTree(BTreeIndexBuilder::default()))
.execute()
.await
.unwrap();
let version_before = table.version().await.unwrap();
let stats = table
.optimize(OptimizeAction::Compact {
options: CompactionOptions {
target_rows_per_fragment: 20,
num_threads: Some(64),
..Default::default()
},
remap_options: None,
})
.await
.unwrap()
.compaction
.unwrap();
assert_eq!(stats.fragments_removed, 64);
assert_eq!(stats.fragments_added, 32);
assert_eq!(table.count_rows(None).await.unwrap(), 640);
assert_eq!(
table.version().await.unwrap(),
version_before + 2,
"parallel compaction should use one fragment reservation commit and one rewrite commit"
);
}
#[tokio::test]
async fn test_optimize_prune_versions() {
let conn = connect("memory://").execute().await.unwrap();