From c429863122489cc19a47aabc187fad1f37ef9bfc Mon Sep 17 00:00:00 2001 From: Wyatt Alt Date: Fri, 14 Aug 2026 16:05:04 -0700 Subject: [PATCH 01/12] feat: refresh_column_async returns a job handle (#3939) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Mirrors create_index's dual surface: the blocking refresh_column keeps returning {rows_filled, version}, and refresh_column_async returns the same Job handle create_index uses, running the refresh as an in-process task. Invalid input is reported by the submitting call rather than by the job. --- Stack created with GitHub Stacks CLIGive Feedback 💬 --- docs/src/js/classes/Table.md | 33 ++++ nodejs/__test__/table.test.ts | 22 +++ nodejs/lancedb/table.ts | 22 +++ nodejs/src/table.rs | 10 ++ python/python/lancedb/_lancedb.pyi | 1 + python/python/lancedb/remote/table.py | 3 + python/python/lancedb/table.py | 59 ++++++ python/python/tests/test_table.py | 28 +++ python/src/table.rs | 11 ++ rust/lancedb/src/job.rs | 2 +- rust/lancedb/src/table.rs | 33 ++++ rust/lancedb/src/table/refresh.rs | 246 ++++++++++++++++++++++++++ 12 files changed, 469 insertions(+), 1 deletion(-) diff --git a/docs/src/js/classes/Table.md b/docs/src/js/classes/Table.md index 278559cc4..712c15ad0 100644 --- a/docs/src/js/classes/Table.md +++ b/docs/src/js/classes/Table.md @@ -770,6 +770,39 @@ number of rows filled and the new version number of the table. *** +### refreshColumnAsync() + +```ts +abstract refreshColumnAsync(column): Promise +``` + +Like [Table#refreshColumn](Table.md#refreshcolumn), but returns a handle to the refresh +job instead of blocking until it completes. + +The job may already be complete when returned; callers must not assume +the column is filled until [Job.wait](Job.md#wait) resolves. Invalid input -- +an unknown column, or one that is not computed -- rejects here rather +than failing the job. Local tables only. + +#### Parameters + +* **column**: `string` + The name of the computed column to fill. + +#### Returns + +`Promise`<[`Job`](Job.md)> + +#### Example + +```ts +const job = await table.refreshColumnAsync("doubled"); +await job.wait(); +console.log(await job.status()); // "finished" +``` + +*** + ### restore() ```ts diff --git a/nodejs/__test__/table.test.ts b/nodejs/__test__/table.test.ts index bc495d24b..5396a251a 100644 --- a/nodejs/__test__/table.test.ts +++ b/nodejs/__test__/table.test.ts @@ -3365,6 +3365,28 @@ describe("computed columns", () => { expect(rows.map((r) => r.doubled).sort()).toEqual([2, 4]); }); + it("returns a job handle from refreshColumnAsync", async () => { + const db = await connect(tmpDir.name); + const table = await db.createTable("computed_job", [{ x: 1 }, { x: 2 }]); + + await table.addColumns({ + computed: [{ name: "doubled", valueSql: "x * 2" }], + }); + + const job = await table.refreshColumnAsync("doubled"); + expect(job.id).toBeNull(); + await job.wait(); + expect(await job.status()).toBe("finished"); + + const rows = await table.query().toArray(); + expect(rows.map((r) => r.doubled).sort()).toEqual([2, 4]); + + // Bad input rejects at the call, not through the job. + await expect(table.refreshColumnAsync("x")).rejects.toThrow( + "not a computed column", + ); + }); + it("fills rows added since the last refresh", async () => { const db = await connect(tmpDir.name); const table = await db.createTable("computed_append", [{ x: 1 }]); diff --git a/nodejs/lancedb/table.ts b/nodejs/lancedb/table.ts index 5b8d00076..4469e41a0 100644 --- a/nodejs/lancedb/table.ts +++ b/nodejs/lancedb/table.ts @@ -574,6 +574,24 @@ export abstract class Table { */ abstract refreshColumn(column: string): Promise; + /** + * Like {@link Table#refreshColumn}, but returns a handle to the refresh + * job instead of blocking until it completes. + * + * The job may already be complete when returned; callers must not assume + * the column is filled until {@link Job.wait} resolves. Invalid input -- + * an unknown column, or one that is not computed -- rejects here rather + * than failing the job. Local tables only. + * @param {string} column The name of the computed column to fill. + * @example + * ```ts + * const job = await table.refreshColumnAsync("doubled"); + * await job.wait(); + * console.log(await job.status()); // "finished" + * ``` + */ + abstract refreshColumnAsync(column: string): Promise; + /** * Alter the name or nullability of columns. * @param {ColumnAlteration[]} columnAlterations One or more alterations to @@ -1179,6 +1197,10 @@ export class LocalTable extends Table { return await this.inner.refreshColumn(column); } + async refreshColumnAsync(column: string): Promise { + return await this.inner.refreshColumnAsync(column); + } + async alterColumns( columnAlterations: ColumnAlteration[], ): Promise { diff --git a/nodejs/src/table.rs b/nodejs/src/table.rs index 40ed7d9f0..4c45be668 100644 --- a/nodejs/src/table.rs +++ b/nodejs/src/table.rs @@ -371,6 +371,16 @@ impl Table { Ok(res.into()) } + #[napi(catch_unwind)] + pub async fn refresh_column_async(&self, column: String) -> napi::Result { + let job = self + .inner_ref()? + .refresh_column_async(column) + .await + .default_error()?; + Ok(crate::job::Job::new(job)) + } + #[napi(catch_unwind)] pub async fn add_columns_with_schema( &self, diff --git a/python/python/lancedb/_lancedb.pyi b/python/python/lancedb/_lancedb.pyi index 96bbecad8..22878fd85 100644 --- a/python/python/lancedb/_lancedb.pyi +++ b/python/python/lancedb/_lancedb.pyi @@ -342,6 +342,7 @@ class Table: self, columns: list[tuple[str, str]] ) -> AddColumnsResult: ... async def refresh_column(self, column: str) -> RefreshColumnResult: ... + async def refresh_column_async(self, column: str) -> Job: ... async def add_columns_with_schema(self, schema: pa.Schema) -> AddColumnsResult: ... async def alter_columns( self, columns: list[dict[str, Any]] diff --git a/python/python/lancedb/remote/table.py b/python/python/lancedb/remote/table.py index 5bd446775..b1bc5bded 100644 --- a/python/python/lancedb/remote/table.py +++ b/python/python/lancedb/remote/table.py @@ -973,6 +973,9 @@ class RemoteTable(Table): def refresh_column(self, column: str): raise NotImplementedError("computed columns are supported only on local tables") + def refresh_column_async(self, column: str) -> Job: + raise NotImplementedError("computed columns are supported only on local tables") + def alter_columns( self, *alterations: Iterable[Dict[str, str]] ) -> AlterColumnsResult: diff --git a/python/python/lancedb/table.py b/python/python/lancedb/table.py index db25c4ebc..9c5925cb7 100644 --- a/python/python/lancedb/table.py +++ b/python/python/lancedb/table.py @@ -2002,6 +2002,31 @@ class Table(ABC): version: the new version number of the table. """ + @abstractmethod + def refresh_column_async(self, column: str) -> Job: + """ + Like :meth:`refresh_column`, but returns a handle to the refresh job + instead of blocking until it completes. + + The job may already be complete when returned; callers must not assume + the column is filled until :meth:`Job.wait` returns. Invalid input -- + an unknown column, or one that is not computed -- raises here rather + than failing the job. Local tables only; LanceDB Cloud and Enterprise + raise ``NotImplementedError``. + + Examples + -------- + >>> import lancedb + >>> db = lancedb.connect("./.lancedb") + >>> table = db.create_table("computed_job_demo", [{"x": 1}, {"x": 2}]) + >>> table.add_columns(computed={"doubled": "x * 2"}) + AddColumnsResult(version=2) + >>> job = table.refresh_column_async("doubled") + >>> job.wait() + >>> job.status() + 'finished' + """ + @abstractmethod def alter_columns(self, *alterations: Iterable[Dict[str, str]]): """ @@ -4020,6 +4045,13 @@ class LanceTable(Table): [`AsyncTable.refresh_column`][lancedb.AsyncTable.refresh_column].""" return LOOP.run(self._table.refresh_column(column)) + def refresh_column_async(self, column: str) -> Job: + """Fill a computed column's unfilled rows, returning a handle to the + refresh job. See + [`Table.refresh_column_async`][lancedb.table.Table.refresh_column_async]. + """ + return Job(LOOP.run(self._table.refresh_column_async(column))) + def alter_columns( self, *alterations: Iterable[Dict[str, str]] ) -> AlterColumnsResult: @@ -6018,6 +6050,33 @@ class AsyncTable: """ return await self._inner.refresh_column(column) + async def refresh_column_async(self, column: str) -> AsyncJob: + """ + Like :meth:`refresh_column`, but returns a handle to the refresh job + instead of blocking until it completes. + + The job may already be complete when returned; callers must not assume + the column is filled until :meth:`AsyncJob.wait` resolves. Invalid + input -- an unknown column, or one that is not computed -- raises here + rather than failing the job. Local tables only; LanceDB Cloud and + Enterprise raise ``NotImplementedError``. + + Examples + -------- + >>> import asyncio + >>> import lancedb + >>> async def refresh_in_background(): + ... db = await lancedb.connect_async("./.lancedb") + ... table = await db.create_table("computed_job_async_demo", [{"x": 1}]) + ... await table.add_columns(computed={"doubled": "x * 2"}) + ... job = await table.refresh_column_async("doubled") + ... await job.wait() + ... return await job.status() + >>> asyncio.run(refresh_in_background()) + 'finished' + """ + return AsyncJob(await self._inner.refresh_column_async(column)) + async def alter_columns( self, *alterations: Iterable[dict[str, Any]] ) -> AlterColumnsResult: diff --git a/python/python/tests/test_table.py b/python/python/tests/test_table.py index ddc1f450f..bb011f8c0 100644 --- a/python/python/tests/test_table.py +++ b/python/python/tests/test_table.py @@ -3888,3 +3888,31 @@ async def test_computed_column_async(tmp_path): await table.refresh_column("tripled") assert (await table.to_arrow())["tripled"].to_pylist() == [9] + + +def test_refresh_column_async_returns_job(tmp_path): + db = lancedb.connect(tmp_path) + table = db.create_table("computed_job", [{"x": 1}, {"x": 2}]) + table.add_columns(computed={"doubled": "x * 2"}) + + job = table.refresh_column_async("doubled") + assert job.id is None # in-process jobs have no server id + job.wait() + assert job.status() == "finished" + assert sorted(table.to_arrow()["doubled"].to_pylist()) == [2, 4] + + # Bad input raises at the call, not through the job. + with pytest.raises(Exception, match="not a computed column"): + table.refresh_column_async("x") + + +@pytest.mark.asyncio +async def test_refresh_column_async_job_async_table(tmp_path): + db = await lancedb.connect_async(tmp_path) + table = await db.create_table("computed_job_async", [{"x": 3}]) + await table.add_columns(computed={"tripled": "x * 3"}) + + job = await table.refresh_column_async("tripled") + await job.wait() + assert await job.status() == "finished" + assert (await table.to_arrow())["tripled"].to_pylist() == [9] diff --git a/python/src/table.rs b/python/src/table.rs index a4c3c307a..35ee92dc4 100644 --- a/python/src/table.rs +++ b/python/src/table.rs @@ -1559,6 +1559,17 @@ impl Table { }) } + pub fn refresh_column_async( + self_: PyRef<'_, Self>, + column: String, + ) -> PyResult> { + let inner = self_.inner_ref()?.clone(); + future_into_py(self_.py(), async move { + let job = inner.refresh_column_async(column).await.infer_error()?; + Ok(crate::job::Job::new(job)) + }) + } + pub fn add_columns_with_schema( self_: PyRef<'_, Self>, schema: PyArrowType, diff --git a/rust/lancedb/src/job.rs b/rust/lancedb/src/job.rs index 789ce8312..d77dd6974 100644 --- a/rust/lancedb/src/job.rs +++ b/rust/lancedb/src/job.rs @@ -141,7 +141,7 @@ impl SpawnedJob { Ok(Err(err)) => Outcome::Failed(Arc::new(err)), Err(err) if err.is_cancelled() => Outcome::Cancelled, Err(err) => Outcome::Failed(Arc::new(Error::Runtime { - message: format!("index job task failed: {err}"), + message: format!("job task failed: {err}"), })), }; let _ = tx.send(Some(outcome)); diff --git a/rust/lancedb/src/table.rs b/rust/lancedb/src/table.rs index bfa060638..093d63438 100644 --- a/rust/lancedb/src/table.rs +++ b/rust/lancedb/src/table.rs @@ -764,6 +764,13 @@ pub trait BaseTable: std::fmt::Display + std::fmt::Debug + Send + Sync { message: "computed columns are supported only on local tables".into(), }) } + /// Fill a computed column's unfilled rows, returning a [`Job`] tracking + /// the operation. + async fn refresh_column_async(&self, _column: &str) -> Result { + Err(Error::NotSupported { + message: "computed columns are supported only on local tables".into(), + }) + } /// Alter columns in the table. async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result; /// Drop columns from the table. @@ -1679,6 +1686,28 @@ impl Table { self.inner.refresh_column(column.as_ref()).await } + /// Like [`Table::refresh_column`], but returns a [`Job`] tracking the + /// operation instead of blocking until it completes. + /// + /// The job may already be complete when returned, and callers must not + /// assume the column is filled until [`Job::wait`] returns. Invalid input + /// -- an unknown column, or one that is not computed -- is reported by + /// this call rather than by the job. Local tables only: LanceDB Cloud and + /// Enterprise reject with `NotSupported`. + /// + /// ``` + /// # use lancedb::Table; + /// # async fn refresh_in_background(table: &Table) -> Result<(), Box> { + /// let job = table.refresh_column_async("doubled").await?; + /// println!("refresh running: {:?}", job.status().await?); + /// job.wait().await?; + /// # Ok(()) + /// # } + /// ``` + pub async fn refresh_column_async(&self, column: impl AsRef) -> Result { + self.inner.refresh_column_async(column.as_ref()).await + } + /// Change a column's name or nullability. pub async fn alter_columns( &self, @@ -3392,6 +3421,10 @@ impl BaseTable for NativeTable { Ok(result) } + async fn refresh_column_async(&self, column: &str) -> Result { + refresh::execute_refresh_column_async(self, column).await + } + async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result { let result = schema_evolution::execute_alter_columns(self, alterations).await?; self.bump_freshness(); diff --git a/rust/lancedb/src/table/refresh.rs b/rust/lancedb/src/table/refresh.rs index ab4f8e452..edc78387e 100644 --- a/rust/lancedb/src/table/refresh.rs +++ b/rust/lancedb/src/table/refresh.rs @@ -35,6 +35,7 @@ use serde::{Deserialize, Serialize}; use super::computed_columns::{BoundExpression, ComputedColumnKind, computed_column_from_field}; use super::{BaseTable, NativeTable}; +use crate::job::Job; use crate::{Error, Result}; /// The result of refreshing a computed column. @@ -115,6 +116,25 @@ pub(crate) async fn execute_refresh_column( }) } +/// Run the refresh as a [`Job`] in this process. +pub(crate) async fn execute_refresh_column_async(table: &NativeTable, column: &str) -> Result { + // Validate before spawning so bad input is reported by this call rather + // than only by the job. + table.dataset.ensure_mutable()?; + ensure_no_lsm_write_spec(table).await?; + let dataset = table.dataset.get().await?; + declared_expression(&dataset, column)?; + drop(dataset); + + let table = table.clone(); + let column = column.to_string(); + Ok(Job::spawned(tokio::spawn(async move { + execute_refresh_column(&table, &column).await?; + table.bump_freshness(); + Ok(()) + }))) +} + /// Refuse to refresh under an LSM write spec. /// /// Refresh enumerates base fragments, and a write spec keeps visible rows in @@ -606,6 +626,230 @@ mod tests { assert!(Arc::ptr_eq(&dataset.session(), &session)); } + /// The async form's job settles with the fill visible, like + /// create_index's execute_async. + #[tokio::test] + async fn test_refresh_async_job_waits_for_the_fill() { + let table = table_with("refresh_async", vec![1, 2, 3]).await; + declare_doubled(&table).await.unwrap(); + + let job = table.refresh_column_async("doubled").await.unwrap(); + assert!(job.id().is_none(), "in-process jobs have no server id"); + job.wait().await.unwrap(); + assert_eq!(job.status().await.unwrap(), "finished"); + assert_eq!( + read(&table, "doubled").await, + vec![Some(2), Some(4), Some(6)] + ); + } + + /// Bad input is reported by the call, not by the job. + #[tokio::test] + async fn test_refresh_async_rejects_bad_input_before_spawning() { + let table = table_with("refresh_async_bad", vec![1, 2, 3]).await; + + let err = table.refresh_column_async("x").await.unwrap_err(); + assert!(matches!(err, Error::NotAComputedColumn { name } if name == "x")); + + let err = table.refresh_column_async("nope").await.unwrap_err(); + assert!(matches!(err, Error::ColumnNotFound { name } if name == "nope")); + } + + #[tokio::test] + async fn test_refresh_async_job_reports_success_to_every_waiter() { + let table = table_with("refresh_async_waiters", vec![1, 2]).await; + declare_doubled(&table).await.unwrap(); + + let job = table.refresh_column_async("doubled").await.unwrap(); + job.wait().await.unwrap(); + // A second wait after completion observes the same outcome. + job.wait().await.unwrap(); + assert_eq!(job.status().await.unwrap(), "finished"); + } + + #[tokio::test] + async fn test_refresh_rejects_a_plain_column() { + let table = table_with("refresh_plain", vec![1, 2, 3]).await; + let err = table.refresh_column("x").await.unwrap_err(); + assert!(matches!(err, Error::NotAComputedColumn { name } if name == "x")); + } + + #[tokio::test] + async fn test_refresh_rejects_an_unknown_column() { + let table = table_with("refresh_missing", vec![1, 2, 3]).await; + let err = table.refresh_column("nope").await.unwrap_err(); + assert!(matches!(err, Error::ColumnNotFound { name } if name == "nope")); + } + + /// The gate's reproducer: a poison value in a deleted row must not + /// abort filling the live rows, since nobody can read it. + #[tokio::test] + async fn test_a_deleted_rows_value_is_never_evaluated() { + let table = table_with("refresh_deleted_poison", vec![1, 0]).await; + table + .add_columns() + .computed("quotient", "10 / x") + .execute() + .await + .unwrap(); + table.delete("x = 0").await.unwrap(); + + let result = table.refresh_column("quotient").await.unwrap(); + assert_eq!(result.rows_filled, 1); + assert_eq!(read(&table, "quotient").await, vec![Some(10)]); + } + + /// The gate's reproducer: an already-filled row's value must not be + /// re-evaluated either -- its input may have mutated into one the + /// expression chokes on. + #[tokio::test] + async fn test_a_filled_rows_value_is_never_evaluated() { + let table = table_with("refresh_filled_poison", vec![1, 2]).await; + table + .add_columns() + .computed("quotient", "10 / x") + .execute() + .await + .unwrap(); + table.refresh_column("quotient").await.unwrap(); + + table + .update() + .column("x", "0") + .only_if("x = 1") + .execute() + .await + .unwrap(); + append(&table, vec![5]).await; + + let result = table.refresh_column("quotient").await.unwrap(); + assert_eq!(result.rows_filled, 1); + assert_eq!( + read(&table, "quotient").await, + vec![Some(2), Some(5), Some(10)] + ); + } + + /// The gate's reproducer: the old internal projection alias is an + /// ordinary column name; a computed column may use it. + #[tokio::test] + async fn test_refresh_a_column_named_like_the_old_alias() { + let table = table_with("refresh_alias_name", vec![1, 2]).await; + table + .add_columns() + .computed("__lancedb_computed", "x * 2") + .execute() + .await + .unwrap(); + + let result = table.refresh_column("__lancedb_computed").await.unwrap(); + assert_eq!(result.rows_filled, 2); + assert_eq!( + read(&table, "__lancedb_computed").await, + vec![Some(2), Some(4)] + ); + } + + /// The gate's reproducer: a late-gain fragment (filled, then one null row + /// compacted onto the end) fills without the old probe's buffering, which + /// this pins behaviorally; the memory bound is structural -- the fill + /// stream retains no batches at all. + #[tokio::test] + async fn test_refresh_fills_a_late_gain_fragment() { + let values: Vec = (0..20_000).collect(); + let table = table_with("refresh_late_gain", values).await; + declare_doubled(&table).await.unwrap(); + table.refresh_column("doubled").await.unwrap(); + + append(&table, vec![2_000_000]).await; + table + .optimize(crate::table::OptimizeAction::Compact { + options: crate::table::CompactionOptions::default(), + remap_options: None, + }) + .await + .unwrap(); + + let result = table.refresh_column("doubled").await.unwrap(); + assert_eq!(result.rows_filled, 1); + let read_back = read(&table, "doubled").await; + assert_eq!(read_back.len(), 20_001); + assert_eq!(read_back.last().unwrap(), &Some(4_000_000)); + } + + /// The gate's reproducer: a nested input declares, refreshes, and guards + /// its root against invalidating schema changes. + #[tokio::test] + async fn test_a_nested_input_declares_and_refreshes() { + use arrow_array::{Int32Array, StructArray}; + use arrow_schema::{DataType, Field, Fields}; + + let conn = connect("memory://").execute().await.unwrap(); + let age = Arc::new(Int32Array::from(vec![30, 40])); + let fields = Fields::from(vec![Field::new("age", DataType::Int32, true)]); + let metadata = StructArray::new(fields.clone(), vec![age as _], None); + let schema = Arc::new(arrow_schema::Schema::new(vec![Field::new( + "metadata", + DataType::Struct(fields), + true, + )])); + let batch = + arrow_array::RecordBatch::try_new(schema, vec![Arc::new(metadata) as _]).unwrap(); + let table = conn + .create_table("refresh_nested", batch) + .execute() + .await + .unwrap(); + + table + .add_columns() + .computed("next_age", "metadata.age + 1") + .execute() + .await + .unwrap(); + let declaration = + &crate::table::computed_columns(table.schema().await.unwrap().as_ref())[0]; + assert_eq!(declaration.inputs, vec!["metadata.age".to_string()]); + + let result = table.refresh_column("next_age").await.unwrap(); + assert_eq!(result.rows_filled, 2); + assert_eq!(read(&table, "next_age").await, vec![Some(31), Some(41)]); + + // The dotted input guards its root. + let err = table.drop_columns(&["metadata"]).await.unwrap_err(); + assert!( + matches!(&err, Error::InvalidInput { message } if message.contains("next_age")), + "{err:?}" + ); + + // Masking a struct input for a deleted row goes through the same + // nullif path as a primitive; a nested input plus deletions must not + // be the combination that breaks it. + table.delete("next_age = 31").await.unwrap(); + append_struct_row(&table, 50).await; + let result = table.refresh_column("next_age").await.unwrap(); + assert_eq!(result.rows_filled, 1); + assert_eq!(read(&table, "next_age").await, vec![Some(41), Some(51)]); + } + + /// Append one `metadata: {age}` row to the nested-input table. + async fn append_struct_row(table: &Table, age: i32) { + use arrow_array::{Int32Array, StructArray}; + use arrow_schema::{DataType, Field, Fields}; + + let ages = Arc::new(Int32Array::from(vec![age])); + let fields = Fields::from(vec![Field::new("age", DataType::Int32, true)]); + let metadata = StructArray::new(fields.clone(), vec![ages as _], None); + let schema = Arc::new(arrow_schema::Schema::new(vec![Field::new( + "metadata", + DataType::Struct(fields), + true, + )])); + let batch = + arrow_array::RecordBatch::try_new(schema, vec![Arc::new(metadata) as _]).unwrap(); + table.add(batch).execute().await.unwrap(); + } + /// Both orders of declare+spec are refused at the source (see the /// schema_evolution tests); refresh's own check covers a dataset another /// writer left in that state. @@ -639,6 +883,8 @@ mod tests { matches!(&err, Error::NotSupported { message } if message.contains("LSM")), "{err:?}" ); + let err = table.refresh_column_async("doubled").await.unwrap_err(); + assert!(matches!(err, Error::NotSupported { .. })); } /// After catch-up activation and unset, no spec remains but the catch-up From 980818df2659233df0b9500ca3a1bbc4857cc227 Mon Sep 17 00:00:00 2001 From: LanceDB Robot Date: Fri, 14 Aug 2026 16:20:39 -0700 Subject: [PATCH 02/12] chore: update lance dependency to v11.0.0-beta.13 (#3947) Updates the Lance Rust workspace dependencies and Java lance-core dependency to [v11.0.0-beta.13](https://github.com/lance-format/lance/releases/tag/v11.0.0-beta.13). Adds the required `ListTablesResponse.context` compatibility field and validates the workspace with Clippy warnings denied. --- Cargo.lock | 88 ++++++++++++++-------------- Cargo.toml | 28 ++++----- java/pom.xml | 2 +- rust/lancedb/src/database/listing.rs | 1 + 4 files changed, 60 insertions(+), 59 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index b4f11fb65..f0a213459 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3455,8 +3455,8 @@ checksum = "42703706b716c37f96a77aea830392ad231f44c9e9a67872fa5548707e11b11c" [[package]] name = "fsst" -version = "11.0.0-beta.11" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3" +version = "11.0.0-beta.13" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" dependencies = [ "arrow-array", "rand 0.9.5", @@ -4815,8 +4815,8 @@ checksum = "e037a2e1d8d5fdbd49b16a4ea09d5d6401c1f29eca5ff29d03d3824dba16256a" [[package]] name = "lance" -version = "11.0.0-beta.11" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3" +version = "11.0.0-beta.13" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" dependencies = [ "arc-swap", "arrow", @@ -4888,8 +4888,8 @@ dependencies = [ [[package]] name = "lance-arrow" -version = "11.0.0-beta.11" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3" +version = "11.0.0-beta.13" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" dependencies = [ "arrow-array", "arrow-buffer", @@ -4911,7 +4911,7 @@ dependencies = [ [[package]] name = "lance-arrow-scalar" version = "58.0.0" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" dependencies = [ "arrow-array", "arrow-buffer", @@ -4925,7 +4925,7 @@ dependencies = [ [[package]] name = "lance-arrow-stats" version = "58.0.0" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" dependencies = [ "arrow-array", "arrow-schema", @@ -4934,8 +4934,8 @@ dependencies = [ [[package]] name = "lance-bitpacking" -version = "11.0.0-beta.11" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3" +version = "11.0.0-beta.13" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" dependencies = [ "arrayref", "crunchy", @@ -4945,8 +4945,8 @@ dependencies = [ [[package]] name = "lance-core" -version = "11.0.0-beta.11" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3" +version = "11.0.0-beta.13" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" dependencies = [ "arrow-array", "arrow-buffer", @@ -4983,8 +4983,8 @@ dependencies = [ [[package]] name = "lance-datafusion" -version = "11.0.0-beta.11" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3" +version = "11.0.0-beta.13" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" dependencies = [ "arrow", "arrow-array", @@ -5013,8 +5013,8 @@ dependencies = [ [[package]] name = "lance-datagen" -version = "11.0.0-beta.11" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3" +version = "11.0.0-beta.13" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" dependencies = [ "arrow", "arrow-array", @@ -5031,8 +5031,8 @@ dependencies = [ [[package]] name = "lance-derive" -version = "11.0.0-beta.11" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3" +version = "11.0.0-beta.13" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" dependencies = [ "proc-macro2", "quote", @@ -5041,8 +5041,8 @@ dependencies = [ [[package]] name = "lance-encoding" -version = "11.0.0-beta.11" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3" +version = "11.0.0-beta.13" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" dependencies = [ "arrow-arith", "arrow-array", @@ -5075,8 +5075,8 @@ dependencies = [ [[package]] name = "lance-file" -version = "11.0.0-beta.11" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3" +version = "11.0.0-beta.13" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" dependencies = [ "arrow-arith", "arrow-array", @@ -5107,8 +5107,8 @@ dependencies = [ [[package]] name = "lance-index" -version = "11.0.0-beta.11" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3" +version = "11.0.0-beta.13" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" dependencies = [ "arc-swap", "arrow", @@ -5172,8 +5172,8 @@ dependencies = [ [[package]] name = "lance-index-core" -version = "11.0.0-beta.11" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3" +version = "11.0.0-beta.13" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" dependencies = [ "arrow-array", "arrow-schema", @@ -5195,8 +5195,8 @@ dependencies = [ [[package]] name = "lance-io" -version = "11.0.0-beta.11" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3" +version = "11.0.0-beta.13" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" dependencies = [ "arrow", "arrow-array", @@ -5232,8 +5232,8 @@ dependencies = [ [[package]] name = "lance-linalg" -version = "11.0.0-beta.11" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3" +version = "11.0.0-beta.13" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" dependencies = [ "arrow-array", "arrow-schema", @@ -5247,8 +5247,8 @@ dependencies = [ [[package]] name = "lance-namespace" -version = "11.0.0-beta.11" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3" +version = "11.0.0-beta.13" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" dependencies = [ "arrow", "async-trait", @@ -5260,8 +5260,8 @@ dependencies = [ [[package]] name = "lance-namespace-impls" -version = "11.0.0-beta.11" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3" +version = "11.0.0-beta.13" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" dependencies = [ "arrow", "arrow-ipc", @@ -5300,9 +5300,9 @@ dependencies = [ [[package]] name = "lance-namespace-reqwest-client" -version = "0.8.6" +version = "0.11.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ba3f0a235e3ed5f8805205649ccc7d7d0f3df23ce1294242c9265ad488d7f19d" +checksum = "0a030196da1c994b63a96a4f0bf5b0cfa459fe6dadc9e962320246ca328da22a" dependencies = [ "reqwest 0.12.28", "serde", @@ -5314,8 +5314,8 @@ dependencies = [ [[package]] name = "lance-select" -version = "11.0.0-beta.11" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3" +version = "11.0.0-beta.13" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" dependencies = [ "arrow-array", "arrow-buffer", @@ -5329,8 +5329,8 @@ dependencies = [ [[package]] name = "lance-table" -version = "11.0.0-beta.11" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3" +version = "11.0.0-beta.13" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" dependencies = [ "arrow", "arrow-array", @@ -5370,8 +5370,8 @@ dependencies = [ [[package]] name = "lance-testing" -version = "11.0.0-beta.11" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3" +version = "11.0.0-beta.13" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" dependencies = [ "arrow-array", "arrow-schema", @@ -5384,8 +5384,8 @@ dependencies = [ [[package]] name = "lance-tokenizer" -version = "11.0.0-beta.11" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.11#5ef1e969030b79fe8c6d31fff9e40d18b24ebbd3" +version = "11.0.0-beta.13" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" dependencies = [ "frostem", "icu_segmenter", diff --git a/Cargo.toml b/Cargo.toml index 3e332adfc..2a19cbb00 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -13,20 +13,20 @@ categories = ["database-implementations"] rust-version = "1.91.0" [workspace.dependencies] -lance = { "version" = "=11.0.0-beta.11", default-features = false, "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" } -lance-core = { "version" = "=11.0.0-beta.11", "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" } -lance-datagen = { "version" = "=11.0.0-beta.11", "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" } -lance-file = { "version" = "=11.0.0-beta.11", "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" } -lance-io = { "version" = "=11.0.0-beta.11", default-features = false, "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" } -lance-index = { "version" = "=11.0.0-beta.11", "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" } -lance-linalg = { "version" = "=11.0.0-beta.11", "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" } -lance-namespace = { "version" = "=11.0.0-beta.11", "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" } -lance-namespace-impls = { "version" = "=11.0.0-beta.11", default-features = false, "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" } -lance-table = { "version" = "=11.0.0-beta.11", "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" } -lance-testing = { "version" = "=11.0.0-beta.11", "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" } -lance-datafusion = { "version" = "=11.0.0-beta.11", "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" } -lance-encoding = { "version" = "=11.0.0-beta.11", "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" } -lance-arrow = { "version" = "=11.0.0-beta.11", "tag" = "v11.0.0-beta.11", "git" = "https://github.com/lance-format/lance.git" } +lance = { "version" = "=11.0.0-beta.13", default-features = false, "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } +lance-core = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } +lance-datagen = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } +lance-file = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } +lance-io = { "version" = "=11.0.0-beta.13", default-features = false, "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } +lance-index = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } +lance-linalg = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } +lance-namespace = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } +lance-namespace-impls = { "version" = "=11.0.0-beta.13", default-features = false, "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } +lance-table = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } +lance-testing = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } +lance-datafusion = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } +lance-encoding = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } +lance-arrow = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } ahash = "0.8" # Note that this one does not include pyarrow arrow = { version = "58.0.0", optional = false } diff --git a/java/pom.xml b/java/pom.xml index 9d9fe1f87..3d0682c46 100644 --- a/java/pom.xml +++ b/java/pom.xml @@ -28,7 +28,7 @@ UTF-8 15.0.0 - 11.0.0-beta.11 + 11.0.0-beta.13 false 2.30.0 1.7 diff --git a/rust/lancedb/src/database/listing.rs b/rust/lancedb/src/database/listing.rs index f284320c6..5ebbe6c5c 100644 --- a/rust/lancedb/src/database/listing.rs +++ b/rust/lancedb/src/database/listing.rs @@ -1032,6 +1032,7 @@ impl Database for ListingDatabase { }; Ok(ListTablesResponse { + context: None, tables: f, page_token: next_page_token, }) From 928c3dde2dd94173931632bde06062e786e495be Mon Sep 17 00:00:00 2001 From: Wyatt Alt Date: Fri, 14 Aug 2026 17:21:55 -0700 Subject: [PATCH 03/12] feat: computed columns on remote tables (#3941) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit LanceDB Cloud and Enterprise support computed columns through the REST API, so declaration dispatches per backend: local tables plan the expression themselves, remote ones send {name, computed} entries for the server to plan. A remote refresh is the server's backfill job -- refresh_column_async submits it and returns a handle whose successful wait establishes a read-freshness baseline on the submitting handle, unless a checkout has pinned the handle by the time the job completes; the blocking form refuses rather than invent a fill count the server does not report. Declaration entries are built from the namespace client's AddColumnsEntry model (lance-namespace 0.11.0, via the lance beta.13 pin), so the payload shape is compile-checked against the published contract. --- Stack created with GitHub Stacks CLIGive Feedback 💬 --- docs/src/js/classes/Table.md | 11 +- nodejs/lancedb/table.ts | 11 +- python/python/lancedb/remote/table.py | 10 +- python/python/lancedb/table.py | 26 +- rust/lancedb/src/remote/table.rs | 494 ++++++++++++++++++++++++-- rust/lancedb/src/table.rs | 12 +- rust/lancedb/src/table/add_columns.rs | 5 +- 7 files changed, 500 insertions(+), 69 deletions(-) diff --git a/docs/src/js/classes/Table.md b/docs/src/js/classes/Table.md index 712c15ad0..4479bf4e4 100644 --- a/docs/src/js/classes/Table.md +++ b/docs/src/js/classes/Table.md @@ -79,8 +79,9 @@ input leaves the value computed at fill time; recomputing means dropping the column and declaring it again. While a declaration reads a column, that column cannot be renamed, retyped or dropped. -Computed columns are local-only: LanceDB Cloud and Enterprise reject a -declaration. +On LanceDB Cloud and Enterprise the expression is planned by the +server, and the refresh runs as a server job -- see +[Table#refreshColumnAsync](Table.md#refreshcolumnasync). #### Parameters @@ -754,7 +755,8 @@ Fill the rows of a computed column that hold no value yet. Rows appended since the last refresh are filled by the next one; rows already filled are left as they are, so the call is idempotent and does -not observe a mutated input. Local tables only. +not observe a mutated input. Local tables only: a remote refresh runs +as a server job, through [Table#refreshColumnAsync](Table.md#refreshcolumnasync). #### Parameters @@ -782,7 +784,8 @@ job instead of blocking until it completes. The job may already be complete when returned; callers must not assume the column is filled until [Job.wait](Job.md#wait) resolves. Invalid input -- an unknown column, or one that is not computed -- rejects here rather -than failing the job. Local tables only. +than failing the job. On local tables the job runs in-process; on +LanceDB Cloud and Enterprise it is the server's backfill job. #### Parameters diff --git a/nodejs/lancedb/table.ts b/nodejs/lancedb/table.ts index 4469e41a0..a7dc8def1 100644 --- a/nodejs/lancedb/table.ts +++ b/nodejs/lancedb/table.ts @@ -537,8 +537,9 @@ export abstract class Table { * the column and declaring it again. While a declaration reads a column, * that column cannot be renamed, retyped or dropped. * - * Computed columns are local-only: LanceDB Cloud and Enterprise reject a - * declaration. + * On LanceDB Cloud and Enterprise the expression is planned by the + * server, and the refresh runs as a server job -- see + * {@link Table#refreshColumnAsync}. * @param {AddColumnsSql[] | Field | Field[] | Schema} newColumnTransforms Either: * - An array of objects with column names and SQL expressions to calculate values * - A single Arrow Field defining one column with its data type (column will be initialized with null values) @@ -567,7 +568,8 @@ export abstract class Table { * * Rows appended since the last refresh are filled by the next one; rows * already filled are left as they are, so the call is idempotent and does - * not observe a mutated input. Local tables only. + * not observe a mutated input. Local tables only: a remote refresh runs + * as a server job, through {@link Table#refreshColumnAsync}. * @param {string} column The name of the computed column to fill. * @returns {Promise} A promise that resolves to the * number of rows filled and the new version number of the table. @@ -581,7 +583,8 @@ export abstract class Table { * The job may already be complete when returned; callers must not assume * the column is filled until {@link Job.wait} resolves. Invalid input -- * an unknown column, or one that is not computed -- rejects here rather - * than failing the job. Local tables only. + * than failing the job. On local tables the job runs in-process; on + * LanceDB Cloud and Enterprise it is the server's backfill job. * @param {string} column The name of the computed column to fill. * @example * ```ts diff --git a/python/python/lancedb/remote/table.py b/python/python/lancedb/remote/table.py index b1bc5bded..aa822b913 100644 --- a/python/python/lancedb/remote/table.py +++ b/python/python/lancedb/remote/table.py @@ -964,17 +964,13 @@ class RemoteTable(Table): *, computed: Dict[str, str] | None = None, ) -> AddColumnsResult: - if computed: - raise NotImplementedError( - "computed columns are supported only on local tables" - ) - return LOOP.run(self._table.add_columns(transforms)) + return LOOP.run(self._table.add_columns(transforms, computed=computed)) def refresh_column(self, column: str): - raise NotImplementedError("computed columns are supported only on local tables") + return LOOP.run(self._table.refresh_column(column)) def refresh_column_async(self, column: str) -> Job: - raise NotImplementedError("computed columns are supported only on local tables") + return Job(LOOP.run(self._table.refresh_column_async(column))) def alter_columns( self, *alterations: Iterable[Dict[str, str]] diff --git a/python/python/lancedb/table.py b/python/python/lancedb/table.py index 9c5925cb7..4ecf6e836 100644 --- a/python/python/lancedb/table.py +++ b/python/python/lancedb/table.py @@ -1954,8 +1954,10 @@ class Table(ABC): dropping the column and declaring it again. While a declaration reads a column, that column cannot be renamed, retyped or dropped. - Local tables only; LanceDB Cloud and Enterprise raise - ``NotImplementedError``. Cannot be combined with ``transforms``. + On LanceDB Cloud and Enterprise the expression is planned by the + server, and the refresh runs as a server job -- see + [`refresh_column_async`][lancedb.table.Table.refresh_column_async]. + Cannot be combined with ``transforms``. Returns ------- @@ -1987,8 +1989,8 @@ class Table(ABC): by the next one; rows already filled are left as they are, so the call is idempotent and does not observe a mutated input. - Local tables only; LanceDB Cloud and Enterprise raise - ``NotImplementedError``. + Local tables only: a remote refresh runs as a server job, through + [`refresh_column_async`][lancedb.table.Table.refresh_column_async]. Parameters ---------- @@ -2011,8 +2013,8 @@ class Table(ABC): The job may already be complete when returned; callers must not assume the column is filled until :meth:`Job.wait` returns. Invalid input -- an unknown column, or one that is not computed -- raises here rather - than failing the job. Local tables only; LanceDB Cloud and Enterprise - raise ``NotImplementedError``. + than failing the job. On local tables the job runs in-process; on + LanceDB Cloud and Enterprise it is the server's backfill job. Examples -------- @@ -5999,7 +6001,8 @@ class AsyncTable: declaration reads a column, that column cannot be renamed, retyped or dropped. - Local tables only. Cannot be combined with ``transforms``. + On LanceDB Cloud and Enterprise the expression is planned by + the server. Cannot be combined with ``transforms``. Returns ------- @@ -6035,8 +6038,8 @@ class AsyncTable: by the next one; rows already filled are left as they are, so the call is idempotent and does not observe a mutated input. - Local tables only; LanceDB Cloud and Enterprise raise - ``NotImplementedError``. + Local tables only: a remote refresh runs as a server job, through + [`refresh_column_async`][lancedb.table.Table.refresh_column_async]. Parameters ---------- @@ -6058,8 +6061,9 @@ class AsyncTable: The job may already be complete when returned; callers must not assume the column is filled until :meth:`AsyncJob.wait` resolves. Invalid input -- an unknown column, or one that is not computed -- raises here - rather than failing the job. Local tables only; LanceDB Cloud and - Enterprise raise ``NotImplementedError``. + rather than failing the job. On local tables the job runs + in-process; on LanceDB Cloud and Enterprise it is the server's + backfill job. Examples -------- diff --git a/rust/lancedb/src/remote/table.rs b/rust/lancedb/src/remote/table.rs index 8ca84a520..a0a4cebc2 100644 --- a/rust/lancedb/src/remote/table.rs +++ b/rust/lancedb/src/remote/table.rs @@ -33,7 +33,9 @@ use crate::table::lsm_stats::GetLsmStatsResponse; use crate::table::merge::MergeFilter; use crate::table::query::create_multi_vector_plan; use crate::table::write_progress::FinishOnDrop; -use crate::table::{AlterColumnsResult, FieldMetadataUpdate, UpdateFieldMetadataResult}; +use crate::table::{ + AlterColumnsResult, FieldMetadataUpdate, RefreshColumnResult, UpdateFieldMetadataResult, +}; use crate::table::{AnyQuery, Filter, Predicate, PreprocessingOutput, TableStatistics}; use crate::utils::background_cache::BackgroundCache; use crate::utils::{ @@ -140,6 +142,40 @@ impl FreshnessHeaders { } } +/// A backfill job whose successful wait establishes a read-freshness +/// baseline on the submitting handle, so a later read cannot be served +/// from a cache older than the completed fill. A handle pinned by checkout +/// at completion keeps its time-travel view instead. +struct FreshnessJob { + inner: RemoteJob, + freshness: Arc>, + version: Arc>>, +} + +#[async_trait] +impl crate::job::JobHandle for FreshnessJob { + fn id(&self) -> Option<&str> { + crate::job::JobHandle::id(&self.inner) + } + + async fn status(&self) -> Result { + crate::job::JobHandle::status(&self.inner).await + } + + async fn wait(&self) -> Result<()> { + crate::job::JobHandle::wait(&self.inner).await?; + let version = self.version.read().await; + if version.is_none() { + self.freshness.lock().unwrap().checkout_baseline = Some(SystemTime::now()); + } + Ok(()) + } + + async fn cancel(&self) -> Result<()> { + crate::job::JobHandle::cancel(&self.inner).await + } +} + fn compute_min_timestamp( state: &FreshnessState, interval: Option, @@ -274,10 +310,10 @@ pub struct RemoteTable { identifier: String, server_version: ServerVersion, - version: RwLock>, + version: Arc>>, location: RwLock>, schema_cache: BackgroundCache, - freshness: Mutex, + freshness: Arc>, /// The branch this handle is scoped to, or `None` for the main branch. /// Stamped onto every branch-accepting request so reads and writes resolve /// on the branch's own version chain rather than main's. @@ -415,10 +451,10 @@ impl RemoteTable { namespace, identifier, server_version, - version: RwLock::new(None), + version: Arc::new(RwLock::new(None)), location: RwLock::new(None), schema_cache: BackgroundCache::new(SCHEMA_CACHE_TTL, SCHEMA_CACHE_REFRESH_WINDOW), - freshness: Mutex::new(FreshnessState::default()), + freshness: Arc::new(Mutex::new(FreshnessState::default())), branch: None, } } @@ -447,10 +483,10 @@ impl RemoteTable { namespace: self.namespace.clone(), identifier: self.identifier.clone(), server_version: self.server_version.clone(), - version: RwLock::new(None), + version: Arc::new(RwLock::new(None)), location: RwLock::new(None), schema_cache: BackgroundCache::new(SCHEMA_CACHE_TTL, SCHEMA_CACHE_REFRESH_WINDOW), - freshness: Mutex::new(FreshnessState::default()), + freshness: Arc::new(Mutex::new(FreshnessState::default())), branch, } } @@ -1268,10 +1304,10 @@ mod test_utils { namespace: vec![], identifier: name, server_version: version.map(ServerVersion).unwrap_or_default(), - version: RwLock::new(None), + version: Arc::new(RwLock::new(None)), location: RwLock::new(None), schema_cache: BackgroundCache::new(SCHEMA_CACHE_TTL, SCHEMA_CACHE_REFRESH_WINDOW), - freshness: Mutex::new(FreshnessState::default()), + freshness: Arc::new(Mutex::new(FreshnessState::default())), branch: None, } } @@ -1292,10 +1328,10 @@ mod test_utils { namespace: vec![], identifier: name, server_version: ServerVersion::default(), - version: RwLock::new(None), + version: Arc::new(RwLock::new(None)), location: RwLock::new(None), schema_cache: BackgroundCache::new(SCHEMA_CACHE_TTL, SCHEMA_CACHE_REFRESH_WINDOW), - freshness: Mutex::new(FreshnessState::default()), + freshness: Arc::new(Mutex::new(FreshnessState::default())), branch: None, } } @@ -1325,10 +1361,10 @@ mod test_utils { namespace: vec![], identifier: name, server_version: version.map(ServerVersion).unwrap_or_default(), - version: RwLock::new(None), + version: Arc::new(RwLock::new(None)), location: RwLock::new(None), schema_cache: BackgroundCache::new(SCHEMA_CACHE_TTL, SCHEMA_CACHE_REFRESH_WINDOW), - freshness: Mutex::new(FreshnessState::default()), + freshness: Arc::new(Mutex::new(FreshnessState::default())), branch: None, } } @@ -2700,13 +2736,6 @@ impl BaseTable for RemoteTable { Ok(result) } - // A declaration reaches here as AllNulls, which the remote protocol - // has no representation for. - NewColumnTransform::AllNulls(_) => { - return Err(Error::NotSupported { - message: "computed columns are supported only on local tables".into(), - }); - } _ => { return Err(Error::NotSupported { message: "Only SQL expressions are supported for adding columns".into(), @@ -2715,6 +2744,86 @@ impl BaseTable for RemoteTable { } } + async fn add_computed_columns(&self, columns: &[(String, String)]) -> Result { + self.check_mutable().await?; + // The server plans the declaration: expression validation, type + // inference and the persisted binding all happen there. + let entries = columns + .iter() + .map( + |(name, expression)| lance_namespace::models::AddColumnsEntry { + name: name.clone(), + computed: Some(Some(expression.clone())), + ..Default::default() + }, + ) + .collect::>(); + let mut body = serde_json::json!({ "new_columns": entries }); + self.apply_branch_body(&mut body); + let request = self + .client + .post(&format!("/v1/table/{}/add_columns/", self.identifier)) + .json(&body); + let (request_id, response) = self.send(request, true).await?; + let response = self.check_table_response(&request_id, response).await?; + let body = response.text().await.err_to_http(request_id.clone())?; + + if body.trim().is_empty() { + // Backward compatible with old servers + return Ok(AddColumnsResult { version: 0 }); + } + + let result: AddColumnsResult = serde_json::from_str(&body).map_err(|e| Error::Http { + source: format!("Failed to parse add_columns response: {}", e).into(), + request_id, + status_code: None, + })?; + + self.invalidate_schema_cache(); + self.track_write_version(result.version); + + Ok(result) + } + + async fn refresh_column(&self, _column: &str) -> Result { + // The server runs a refresh as a job and does not report a fill + // count, so the blocking form has no honest result to return. + Err(Error::NotSupported { + message: "a remote refresh runs as a server job; use refresh_column_async and \ + wait on the returned handle" + .into(), + }) + } + + async fn refresh_column_async(&self, column: &str) -> Result { + self.check_mutable().await?; + let mut body = serde_json::json!({ "column": column }); + self.apply_branch_body(&mut body); + let request = self + .client + .post(&format!("/v1/table/{}/backfill_column", self.identifier)) + .json(&body); + let (request_id, response) = self.send(request, true).await?; + let response = self.check_table_response(&request_id, response).await?; + let body = response.text().await.err_to_http(request_id.clone())?; + + #[derive(serde::Deserialize)] + struct BackfillResponse { + job_id: String, + } + let response: BackfillResponse = serde_json::from_str(&body).map_err(|e| Error::Http { + source: format!("Failed to parse backfill_column response: {}", e).into(), + request_id, + status_code: None, + })?; + + Ok(Job::new(Box::new(FreshnessJob { + inner: RemoteJob::new(self.client.clone(), response.job_id), + freshness: self.freshness.clone(), + version: self.version.clone(), + }))) + } + async fn alter_columns(&self, alterations: &[ColumnAlteration]) -> Result { self.check_mutable().await?; let body = alterations @@ -6456,37 +6565,346 @@ mod tests { assert_eq!(result.version, if old_server { 0 } else { 43 }); } - /// Computed columns are local-only. Both halves say so here rather than - /// reaching the wire and failing somewhere less legible. + /// A declaration is sent as `{name, computed}` entries for the server to + /// plan; the client never types the expression itself. #[tokio::test] - async fn test_computed_columns_are_refused() { - let table = Table::new_with_handler("my_table", |request| -> http::Response { - panic!("unexpected request: {}", request.url().path()) + async fn test_add_computed_columns_sends_the_expression() { + let table = Table::new_with_handler("my_table", |request| { + assert_eq!(request.method(), "POST"); + assert_eq!(request.url().path(), "/v1/table/my_table/add_columns/"); + let body = request.body().unwrap().as_bytes().unwrap(); + let value: serde_json::Value = serde_json::from_slice(body).unwrap(); + assert_eq!( + value["new_columns"], + serde_json::json!([{"name": "doubled", "computed": "x * 2"}]) + ); + http::Response::builder() + .status(200) + .body(r#"{"version": 7}"#) + .unwrap() }); - let declared = Arc::new(Schema::new(vec![Field::new( - "doubled", - DataType::Int32, - true, - )])); - let err = table + let result = table .add_columns() - .transform(NewColumnTransform::AllNulls(declared)) + .computed("doubled", "x * 2") .execute() .await - .unwrap_err(); - assert!( - matches!(&err, Error::NotSupported { message } if message.contains("local tables")), - "{err:?}" - ); + .unwrap(); + assert_eq!(result.version, 7); + } + + /// A remote refresh is a server job: the async form returns its handle, + /// and the blocking form refuses rather than invent a fill count. + #[tokio::test] + async fn test_refresh_column_async_submits_a_backfill_job() { + let table = Table::new_with_handler("my_table", |request| { + assert_eq!(request.method(), "POST"); + assert_eq!(request.url().path(), "/v1/table/my_table/backfill_column"); + let body = request.body().unwrap().as_bytes().unwrap(); + let value: serde_json::Value = serde_json::from_slice(body).unwrap(); + assert_eq!(value["column"], "doubled"); + http::Response::builder() + .status(202) + .body(r#"{"job_id": "j-42"}"#) + .unwrap() + }); + + let job = table.refresh_column_async("doubled").await.unwrap(); + assert_eq!(job.id(), Some("j-42")); let err = table.refresh_column("doubled").await.unwrap_err(); assert!( - matches!(&err, Error::NotSupported { message } if message.contains("local tables")), + matches!(&err, Error::NotSupported { message } + if message.contains("refresh_column_async")), "{err:?}" ); } + /// The gate's reproducer: after a successful wait, a same-handle read + /// must carry a freshness baseline so a stale server cache cannot serve + /// the pre-backfill snapshot. + #[tokio::test] + async fn test_backfill_wait_establishes_read_freshness() { + let saw_min_timestamp = Arc::new(std::sync::atomic::AtomicBool::new(false)); + let saw = saw_min_timestamp.clone(); + let table = + Table::new_with_handler("my_table", move |request| match request.url().path() { + "/v1/table/my_table/backfill_column" => http::Response::builder() + .status(202) + .body(r#"{"job_id": "j-7"}"#.to_string()) + .unwrap(), + "/v1/jobs/describe" => http::Response::builder() + .status(200) + .body(r#"{"job_id": "j-7", "job_state": "DONE"}"#.to_string()) + .unwrap(), + "/v1/table/my_table/count_rows/" => { + saw.store( + request.headers().contains_key("x-lancedb-min-timestamp"), + std::sync::atomic::Ordering::SeqCst, + ); + http::Response::builder() + .status(200) + .body("1".to_string()) + .unwrap() + } + path => panic!("unexpected request: {path}"), + }); + + let job = table.refresh_column_async("doubled").await.unwrap(); + job.wait().await.unwrap(); + table.count_rows(None).await.unwrap(); + assert!( + saw_min_timestamp.load(std::sync::atomic::Ordering::SeqCst), + "read after wait carried no freshness baseline" + ); + } + + /// A checkout after submission wins over the completion fence: the + /// pinned view must not regain a timestamp floor from the job. + #[tokio::test] + async fn test_checkout_after_submit_beats_the_completion_fence() { + let saw_min_timestamp = Arc::new(std::sync::atomic::AtomicBool::new(false)); + let saw = saw_min_timestamp.clone(); + let table = + Table::new_with_handler("my_table", move |request| match request.url().path() { + "/v1/table/my_table/backfill_column" => http::Response::builder() + .status(202) + .body(r#"{"job_id": "j-8"}"#.to_string()) + .unwrap(), + "/v1/jobs/describe" => http::Response::builder() + .status(200) + .body(r#"{"job_id": "j-8", "job_state": "DONE"}"#.to_string()) + .unwrap(), + "/v1/table/my_table/describe/" => { + let schema = Schema::new(vec![Field::new("x", DataType::Int32, true)]); + http::Response::builder() + .status(200) + .body(describe_response(&schema)) + .unwrap() + } + "/v1/table/my_table/count_rows/" => { + saw.store( + request.headers().contains_key("x-lancedb-min-timestamp"), + std::sync::atomic::Ordering::SeqCst, + ); + http::Response::builder() + .status(200) + .body("1".to_string()) + .unwrap() + } + path => panic!("unexpected request: {path}"), + }); + + let job = table.refresh_column_async("doubled").await.unwrap(); + table.checkout(3).await.unwrap(); + job.wait().await.unwrap(); + table.count_rows(None).await.unwrap(); + assert!( + !saw_min_timestamp.load(std::sync::atomic::Ordering::SeqCst), + "completion fence overrode an explicit checkout" + ); + } + + /// Tag checkout resets freshness state wholesale; the fence must not + /// survive it. + #[tokio::test] + async fn test_tag_checkout_after_submit_beats_the_completion_fence() { + let saw_min_timestamp = Arc::new(std::sync::atomic::AtomicBool::new(false)); + let saw = saw_min_timestamp.clone(); + let table = + Table::new_with_handler("my_table", move |request| match request.url().path() { + "/v1/table/my_table/backfill_column" => http::Response::builder() + .status(202) + .body(r#"{"job_id": "j-9"}"#.to_string()) + .unwrap(), + "/v1/jobs/describe" => http::Response::builder() + .status(200) + .body(r#"{"job_id": "j-9", "job_state": "DONE"}"#.to_string()) + .unwrap(), + "/v1/table/my_table/tags/version/" => http::Response::builder() + .status(200) + .body(r#"{"version": 5}"#.to_string()) + .unwrap(), + "/v1/table/my_table/describe/" => { + let schema = Schema::new(vec![Field::new("x", DataType::Int32, true)]); + http::Response::builder() + .status(200) + .body(describe_response(&schema)) + .unwrap() + } + "/v1/table/my_table/count_rows/" => { + saw.store( + request.headers().contains_key("x-lancedb-min-timestamp"), + std::sync::atomic::Ordering::SeqCst, + ); + http::Response::builder() + .status(200) + .body("1".to_string()) + .unwrap() + } + path => panic!("unexpected request: {path}"), + }); + + let job = table.refresh_column_async("doubled").await.unwrap(); + table.checkout_tag("v1").await.unwrap(); + job.wait().await.unwrap(); + table.count_rows(None).await.unwrap(); + assert!( + !saw_min_timestamp.load(std::sync::atomic::Ordering::SeqCst), + "completion fence overrode a tag checkout" + ); + } + + /// A checkout landing while the submission request is in flight advances + /// the epoch past the token captured at submit. + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn test_checkout_during_submission_beats_the_completion_fence() { + let saw_min_timestamp = Arc::new(std::sync::atomic::AtomicBool::new(false)); + let saw = saw_min_timestamp.clone(); + let (release_tx, release_rx) = std::sync::mpsc::channel::<()>(); + let release_rx = Arc::new(std::sync::Mutex::new(release_rx)); + let (arrived_tx, arrived_rx) = std::sync::mpsc::channel::<()>(); + let arrived_tx = Arc::new(std::sync::Mutex::new(arrived_tx)); + let table = Table::new_with_handler("my_table", move |request| { + match request.url().path() { + "/v1/table/my_table/backfill_column" => { + // Signal arrival, then hold the response until the + // test's checkout completes. + arrived_tx.lock().unwrap().send(()).unwrap(); + release_rx + .lock() + .unwrap() + .recv_timeout(std::time::Duration::from_secs(10)) + .unwrap(); + http::Response::builder() + .status(202) + .body(r#"{"job_id": "j-10"}"#.to_string()) + .unwrap() + } + "/v1/jobs/describe" => http::Response::builder() + .status(200) + .body(r#"{"job_id": "j-10", "job_state": "DONE"}"#.to_string()) + .unwrap(), + "/v1/table/my_table/describe/" => { + let schema = Schema::new(vec![Field::new("x", DataType::Int32, true)]); + http::Response::builder() + .status(200) + .body(describe_response(&schema)) + .unwrap() + } + "/v1/table/my_table/count_rows/" => { + saw.store( + request.headers().contains_key("x-lancedb-min-timestamp"), + std::sync::atomic::Ordering::SeqCst, + ); + http::Response::builder() + .status(200) + .body("1".to_string()) + .unwrap() + } + path => panic!("unexpected request: {path}"), + } + }); + + let submit = tokio::spawn({ + let table = table.clone(); + async move { table.refresh_column_async("doubled").await } + }); + tokio::task::spawn_blocking(move || { + arrived_rx + .recv_timeout(std::time::Duration::from_secs(10)) + .unwrap() + }) + .await + .unwrap(); + table.checkout(7).await.unwrap(); + release_tx.send(()).unwrap(); + + let job = submit.await.unwrap().unwrap(); + job.wait().await.unwrap(); + table.count_rows(None).await.unwrap(); + assert!( + !saw_min_timestamp.load(std::sync::atomic::Ordering::SeqCst), + "completion fence overrode a checkout that landed mid-submission" + ); + } + + /// checkout_latest keeps the handle on latest, so a completed backfill + /// must still establish its post-fill baseline -- strictly later than the + /// checkout's own, or a pre-fill cache could still serve. + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn test_checkout_latest_during_submission_keeps_the_fence() { + let seen_min_timestamp = Arc::new(std::sync::Mutex::new(None::)); + let saw = seen_min_timestamp.clone(); + let (release_tx, release_rx) = std::sync::mpsc::channel::<()>(); + let release_rx = Arc::new(std::sync::Mutex::new(release_rx)); + let (arrived_tx, arrived_rx) = std::sync::mpsc::channel::<()>(); + let arrived_tx = Arc::new(std::sync::Mutex::new(arrived_tx)); + let table = + Table::new_with_handler("my_table", move |request| match request.url().path() { + "/v1/table/my_table/backfill_column" => { + arrived_tx.lock().unwrap().send(()).unwrap(); + release_rx + .lock() + .unwrap() + .recv_timeout(std::time::Duration::from_secs(10)) + .unwrap(); + http::Response::builder() + .status(202) + .body(r#"{"job_id": "j-11"}"#.to_string()) + .unwrap() + } + "/v1/jobs/describe" => http::Response::builder() + .status(200) + .body(r#"{"job_id": "j-11", "job_state": "DONE"}"#.to_string()) + .unwrap(), + "/v1/table/my_table/count_rows/" => { + *saw.lock().unwrap() = request + .headers() + .get("x-lancedb-min-timestamp") + .map(|v| v.to_str().unwrap().to_string()); + http::Response::builder() + .status(200) + .body("1".to_string()) + .unwrap() + } + path => panic!("unexpected request: {path}"), + }); + + let submit = tokio::spawn({ + let table = table.clone(); + async move { table.refresh_column_async("doubled").await } + }); + tokio::task::spawn_blocking(move || { + arrived_rx + .recv_timeout(std::time::Duration::from_secs(10)) + .unwrap() + }) + .await + .unwrap(); + table.checkout_latest().await.unwrap(); + let after_checkout = SystemTime::now(); + // Real separation between the checkout baseline and completion. + tokio::time::sleep(std::time::Duration::from_millis(50)).await; + release_tx.send(()).unwrap(); + + let job = submit.await.unwrap().unwrap(); + job.wait().await.unwrap(); + table.count_rows(None).await.unwrap(); + let header = seen_min_timestamp + .lock() + .unwrap() + .clone() + .expect("no baseline"); + let sent: SystemTime = chrono::DateTime::parse_from_rfc3339(&header) + .unwrap() + .into(); + assert!( + sent > after_checkout, + "baseline {header} did not advance past the checkout" + ); + } + #[tokio::test] async fn test_prewarm_index() { let table = Table::new_with_handler("my_table", |request| { diff --git a/rust/lancedb/src/table.rs b/rust/lancedb/src/table.rs index 093d63438..2e16b0940 100644 --- a/rust/lancedb/src/table.rs +++ b/rust/lancedb/src/table.rs @@ -748,6 +748,10 @@ pub trait BaseTable: std::fmt::Display + std::fmt::Debug + Send + Sync { read_columns: Option>, ) -> Result; /// Declare computed columns, each defined by a SQL expression. + /// + /// Where the declaration is planned depends on the backend: a local table + /// validates and types the expression itself, a remote one sends the text + /// for the server to plan. async fn add_computed_columns( &self, _columns: &[(String, String)], @@ -1672,7 +1676,8 @@ impl Table { /// filled are left as they are, so the call is idempotent and does not /// observe a mutated input. /// - /// Local tables only. + /// Local tables only: a remote refresh runs as a server job, through + /// [`Table::refresh_column_async`]. /// /// ``` /// # use lancedb::Table; @@ -1692,8 +1697,9 @@ impl Table { /// The job may already be complete when returned, and callers must not /// assume the column is filled until [`Job::wait`] returns. Invalid input /// -- an unknown column, or one that is not computed -- is reported by - /// this call rather than by the job. Local tables only: LanceDB Cloud and - /// Enterprise reject with `NotSupported`. + /// this call rather than by the job. On local tables the job runs as an + /// in-process task; on LanceDB Cloud and Enterprise it is the server's + /// backfill job. /// /// ``` /// # use lancedb::Table; diff --git a/rust/lancedb/src/table/add_columns.rs b/rust/lancedb/src/table/add_columns.rs index 6aa2ce86a..67764c346 100644 --- a/rust/lancedb/src/table/add_columns.rs +++ b/rust/lancedb/src/table/add_columns.rs @@ -61,8 +61,9 @@ impl AddColumnsBuilder { /// column and declaring it again. An input cannot be renamed, retyped or /// dropped while a declaration reads it, since the expression names it. /// - /// Local tables only: LanceDB Cloud and Enterprise reject a declaration - /// with `NotSupported`. + /// On LanceDB Cloud and Enterprise the expression is planned by the + /// server, and the refresh runs as a server job -- see + /// [`Table::refresh_column_async`](super::Table::refresh_column_async). /// /// ``` /// # use lancedb::Table; From 040a4120c876dc105df18afc34c075a36fc64cb6 Mon Sep 17 00:00:00 2001 From: Lance Release Date: Mon, 17 Aug 2026 16:56:20 +0000 Subject: [PATCH 04/12] =?UTF-8?q?Bump=20version:=200.38.0-beta.0=20?= =?UTF-8?q?=E2=86=92=200.38.0-beta.1?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .bumpversion.toml | 2 +- Cargo.lock | 6 +++--- docs/src/java/java.md | 2 +- java/lancedb-core/pom.xml | 2 +- java/pom.xml | 2 +- nodejs/Cargo.toml | 2 +- nodejs/npm/darwin-arm64/package.json | 2 +- nodejs/npm/linux-arm64-gnu/package.json | 2 +- nodejs/npm/linux-arm64-musl/package.json | 2 +- nodejs/npm/linux-x64-gnu/package.json | 2 +- nodejs/npm/linux-x64-musl/package.json | 2 +- nodejs/npm/win32-arm64-msvc/package.json | 2 +- nodejs/npm/win32-x64-msvc/package.json | 2 +- nodejs/package-lock.json | 4 ++-- nodejs/package.json | 2 +- python/Cargo.toml | 2 +- rust/lancedb/Cargo.toml | 2 +- 17 files changed, 20 insertions(+), 20 deletions(-) diff --git a/.bumpversion.toml b/.bumpversion.toml index cab6bb104..e2e693549 100644 --- a/.bumpversion.toml +++ b/.bumpversion.toml @@ -1,5 +1,5 @@ [tool.bumpversion] -current_version = "0.38.0-beta.0" +current_version = "0.38.0-beta.1" parse = """(?x) (?P0|[1-9]\\d*)\\. (?P0|[1-9]\\d*)\\. diff --git a/Cargo.lock b/Cargo.lock index f0a213459..2f7499e91 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -5398,7 +5398,7 @@ dependencies = [ [[package]] name = "lancedb" -version = "0.38.0-beta.0" +version = "0.38.0-beta.1" dependencies = [ "ahash", "anyhow", @@ -5486,7 +5486,7 @@ dependencies = [ [[package]] name = "lancedb-nodejs" -version = "0.38.0-beta.0" +version = "0.38.0-beta.1" dependencies = [ "arrow-array", "arrow-buffer", @@ -5511,7 +5511,7 @@ dependencies = [ [[package]] name = "lancedb-python" -version = "0.38.0-beta.0" +version = "0.38.0-beta.1" dependencies = [ "arrow", "async-trait", diff --git a/docs/src/java/java.md b/docs/src/java/java.md index f9a0ea053..42bc06b9d 100644 --- a/docs/src/java/java.md +++ b/docs/src/java/java.md @@ -14,7 +14,7 @@ Add the following dependency to your `pom.xml`: com.lancedb lancedb-core - 0.38.0-beta.0 + 0.38.0-beta.1 ``` diff --git a/java/lancedb-core/pom.xml b/java/lancedb-core/pom.xml index 09b088e46..94c72b326 100644 --- a/java/lancedb-core/pom.xml +++ b/java/lancedb-core/pom.xml @@ -8,7 +8,7 @@ com.lancedb lancedb-parent - 0.38.0-beta.0 + 0.38.0-beta.1 ../pom.xml diff --git a/java/pom.xml b/java/pom.xml index 3d0682c46..90ad2f7f8 100644 --- a/java/pom.xml +++ b/java/pom.xml @@ -6,7 +6,7 @@ com.lancedb lancedb-parent - 0.38.0-beta.0 + 0.38.0-beta.1 pom ${project.artifactId} LanceDB Java SDK Parent POM diff --git a/nodejs/Cargo.toml b/nodejs/Cargo.toml index 2e9373b9b..3496e2839 100644 --- a/nodejs/Cargo.toml +++ b/nodejs/Cargo.toml @@ -1,7 +1,7 @@ [package] name = "lancedb-nodejs" edition.workspace = true -version = "0.38.0-beta.0" +version = "0.38.0-beta.1" publish = false license.workspace = true description.workspace = true diff --git a/nodejs/npm/darwin-arm64/package.json b/nodejs/npm/darwin-arm64/package.json index e0fd7426d..ad1503090 100644 --- a/nodejs/npm/darwin-arm64/package.json +++ b/nodejs/npm/darwin-arm64/package.json @@ -1,6 +1,6 @@ { "name": "@lancedb/lancedb-darwin-arm64", - "version": "0.38.0-beta.0", + "version": "0.38.0-beta.1", "os": ["darwin"], "cpu": ["arm64"], "main": "lancedb.darwin-arm64.node", diff --git a/nodejs/npm/linux-arm64-gnu/package.json b/nodejs/npm/linux-arm64-gnu/package.json index ef281de3d..e7455832d 100644 --- a/nodejs/npm/linux-arm64-gnu/package.json +++ b/nodejs/npm/linux-arm64-gnu/package.json @@ -1,6 +1,6 @@ { "name": "@lancedb/lancedb-linux-arm64-gnu", - "version": "0.38.0-beta.0", + "version": "0.38.0-beta.1", "os": ["linux"], "cpu": ["arm64"], "main": "lancedb.linux-arm64-gnu.node", diff --git a/nodejs/npm/linux-arm64-musl/package.json b/nodejs/npm/linux-arm64-musl/package.json index d535820fa..8269bc4ce 100644 --- a/nodejs/npm/linux-arm64-musl/package.json +++ b/nodejs/npm/linux-arm64-musl/package.json @@ -1,6 +1,6 @@ { "name": "@lancedb/lancedb-linux-arm64-musl", - "version": "0.38.0-beta.0", + "version": "0.38.0-beta.1", "os": ["linux"], "cpu": ["arm64"], "main": "lancedb.linux-arm64-musl.node", diff --git a/nodejs/npm/linux-x64-gnu/package.json b/nodejs/npm/linux-x64-gnu/package.json index 7aa21301e..7a7d0a097 100644 --- a/nodejs/npm/linux-x64-gnu/package.json +++ b/nodejs/npm/linux-x64-gnu/package.json @@ -1,6 +1,6 @@ { "name": "@lancedb/lancedb-linux-x64-gnu", - "version": "0.38.0-beta.0", + "version": "0.38.0-beta.1", "os": ["linux"], "cpu": ["x64"], "main": "lancedb.linux-x64-gnu.node", diff --git a/nodejs/npm/linux-x64-musl/package.json b/nodejs/npm/linux-x64-musl/package.json index d220991f7..ce95c0174 100644 --- a/nodejs/npm/linux-x64-musl/package.json +++ b/nodejs/npm/linux-x64-musl/package.json @@ -1,6 +1,6 @@ { "name": "@lancedb/lancedb-linux-x64-musl", - "version": "0.38.0-beta.0", + "version": "0.38.0-beta.1", "os": ["linux"], "cpu": ["x64"], "main": "lancedb.linux-x64-musl.node", diff --git a/nodejs/npm/win32-arm64-msvc/package.json b/nodejs/npm/win32-arm64-msvc/package.json index 519d7376a..2ae43e763 100644 --- a/nodejs/npm/win32-arm64-msvc/package.json +++ b/nodejs/npm/win32-arm64-msvc/package.json @@ -1,6 +1,6 @@ { "name": "@lancedb/lancedb-win32-arm64-msvc", - "version": "0.38.0-beta.0", + "version": "0.38.0-beta.1", "os": [ "win32" ], diff --git a/nodejs/npm/win32-x64-msvc/package.json b/nodejs/npm/win32-x64-msvc/package.json index 9f608d6d0..1030609fa 100644 --- a/nodejs/npm/win32-x64-msvc/package.json +++ b/nodejs/npm/win32-x64-msvc/package.json @@ -1,6 +1,6 @@ { "name": "@lancedb/lancedb-win32-x64-msvc", - "version": "0.38.0-beta.0", + "version": "0.38.0-beta.1", "os": ["win32"], "cpu": ["x64"], "main": "lancedb.win32-x64-msvc.node", diff --git a/nodejs/package-lock.json b/nodejs/package-lock.json index 9222bf582..26091b118 100644 --- a/nodejs/package-lock.json +++ b/nodejs/package-lock.json @@ -1,12 +1,12 @@ { "name": "@lancedb/lancedb", - "version": "0.38.0-beta.0", + "version": "0.38.0-beta.1", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "@lancedb/lancedb", - "version": "0.38.0-beta.0", + "version": "0.38.0-beta.1", "cpu": [ "x64", "arm64" diff --git a/nodejs/package.json b/nodejs/package.json index c87af926b..cfdc851ef 100644 --- a/nodejs/package.json +++ b/nodejs/package.json @@ -11,7 +11,7 @@ "ann" ], "private": false, - "version": "0.38.0-beta.0", + "version": "0.38.0-beta.1", "main": "dist/index.js", "exports": { ".": "./dist/index.js", diff --git a/python/Cargo.toml b/python/Cargo.toml index bede2bc37..745ca4ea2 100644 --- a/python/Cargo.toml +++ b/python/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "lancedb-python" -version = "0.38.0-beta.0" +version = "0.38.0-beta.1" publish = false edition.workspace = true description = "Python bindings for LanceDB" diff --git a/rust/lancedb/Cargo.toml b/rust/lancedb/Cargo.toml index 23c0dcfd0..92f020956 100644 --- a/rust/lancedb/Cargo.toml +++ b/rust/lancedb/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "lancedb" -version = "0.38.0-beta.0" +version = "0.38.0-beta.1" edition.workspace = true description = "LanceDB: A serverless, low-latency vector database for AI applications" license.workspace = true From a075aa62f8cdd87ef666eea2f3d7555507e8d773 Mon Sep 17 00:00:00 2001 From: Igor Ganapolsky Date: Mon, 17 Aug 2026 10:48:02 -0700 Subject: [PATCH 05/12] fix(python): treat naive lit(datetime) as UTC wall clock (#3262) (#3775) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## Summary Fixes naive `lit(datetime)` equality filters against table timestamp columns on non-UTC hosts, and adds the integration matrix from #3262. ## Failure (before) On a machine in US Eastern (UTC−4 / EDT), with PyPI `lancedb==0.36.0`: ```python from datetime import datetime import lancedb from lancedb.expr import col, lit db = lancedb.connect("memory://") ts = datetime(2024, 7, 1, 10, 0, 0) # naive table = db.create_table("t", [{"id": 1, "ts": ts}]) rows = table.search().where(col("ts") == lit(ts)).to_list() # actual: [] (0 rows) # expected: 1 row ``` ### Root cause In `python/src/expr.rs`, `expr_lit` converted every `datetime` via Python's `.timestamp()`: - **naive** `.timestamp()` = local wall → UTC epoch (shifted by host offset) - **PyArrow naive** storage = UTC wall-clock microseconds (no local shift) So `lit(naive)` became `CAST('2024-07-01 14:00:00' AS TIMESTAMP)` on EDT while the table held `10:00:00`. ## After Naive datetimes are interpreted as UTC wall clock (`replace(tzinfo=timezone.utc).timestamp()`), matching Arrow storage. Aware datetimes still use `.timestamp()` (correct epoch). Same repro on this branch: **1 matching row**. ## Tests Added `TestExprDatetimeTimezoneIntegration` covering: | Case | Result | |------|--------| | both naive | match | | both same TZ (UTC) | match | | different TZs, same instant | match | | table TZ + naive lit | match (wall clock) | | table naive + aware lit | match | | naive lit SQL is wall clock, not local-shifted | asserts `10:00:00` in SQL | ### Verification ```bash cd python maturin develop pytest python/tests/test_expr.py -v ``` **102 passed** (full `test_expr.py`, including the 6 new cases). Closes #3262 --------- Co-authored-by: Will Jones Co-authored-by: Claude Opus 5 (1M context) --- python/python/tests/test_expr.py | 98 ++++++++++++++++++++++++++++++++ python/src/expr.rs | 21 ++++++- 2 files changed, 118 insertions(+), 1 deletion(-) diff --git a/python/python/tests/test_expr.py b/python/python/tests/test_expr.py index 6aa78943e..0eb6f8929 100644 --- a/python/python/tests/test_expr.py +++ b/python/python/tests/test_expr.py @@ -632,3 +632,101 @@ class TestExprBytesIntegration: .to_arrow() ) assert result.num_rows == 2 + + +# ── datetime / timezone integration for lit() (issue #3262) ────────────────── + + +class TestExprDatetimeTimezoneIntegration: + """Integration coverage for lit(datetime) against table timestamp columns. + + PyArrow stores naive timestamps as UTC wall-clock microseconds. Python's + datetime.timestamp() treats naive values as *local* time, which used to + shift lit(naive) by the host UTC offset and break equality filters on + non-UTC machines. These cases lock the expected semantics. + """ + + def test_both_naive_match(self, tmp_path): + """Table naive + lit naive with the same wall clock must match.""" + db = lancedb.connect(str(tmp_path / "naive")) + ts = datetime(2024, 7, 1, 10, 0, 0) + table = db.create_table( + "t", [{"id": 1, "ts": ts}, {"id": 2, "ts": datetime(2024, 7, 2, 10, 0, 0)}] + ) + result = table.search().where(col("ts") == lit(ts)).to_list() + assert len(result) == 1 + assert result[0]["id"] == 1 + + def test_both_same_timezone_match(self, tmp_path): + """Table UTC + lit UTC for the same instant must match.""" + db = lancedb.connect(str(tmp_path / "utc")) + ts = datetime(2024, 7, 1, 10, 0, 0, tzinfo=timezone.utc) + table = db.create_table( + "t", + pa.table( + { + "id": [1, 2], + "ts": pa.array( + [ts, datetime(2024, 7, 2, 10, 0, 0, tzinfo=timezone.utc)], + type=pa.timestamp("us", tz="UTC"), + ), + } + ), + ) + result = table.search().where(col("ts") == lit(ts)).to_list() + assert len(result) == 1 + assert result[0]["id"] == 1 + + def test_different_timezones_same_instant(self, tmp_path): + """UTC table row equals lit of the same instant in a different zone.""" + db = lancedb.connect(str(tmp_path / "diff_tz")) + ts_utc = datetime(2024, 7, 1, 10, 0, 0, tzinfo=timezone.utc) + # Same instant as 06:00 in UTC-4 + ts_est = datetime(2024, 7, 1, 6, 0, 0, tzinfo=timezone(timedelta(hours=-4))) + table = db.create_table( + "t", + pa.table( + { + "id": [1], + "ts": pa.array([ts_utc], type=pa.timestamp("us", tz="UTC")), + } + ), + ) + result = table.search().where(col("ts") == lit(ts_est)).to_list() + assert len(result) == 1 + assert result[0]["id"] == 1 + + def test_table_tz_literal_naive(self, tmp_path): + """UTC table + naive lit uses wall-clock equality (10:00 == 10:00 UTC).""" + db = lancedb.connect(str(tmp_path / "tz_naive")) + ts_utc = datetime(2024, 7, 1, 10, 0, 0, tzinfo=timezone.utc) + ts_naive = datetime(2024, 7, 1, 10, 0, 0) + table = db.create_table( + "t", + pa.table( + { + "id": [1], + "ts": pa.array([ts_utc], type=pa.timestamp("us", tz="UTC")), + } + ), + ) + result = table.search().where(col("ts") == lit(ts_naive)).to_list() + assert len(result) == 1 + assert result[0]["id"] == 1 + + def test_table_naive_literal_aware(self, tmp_path): + """Naive table + UTC lit with the same wall clock must match.""" + db = lancedb.connect(str(tmp_path / "naive_aware")) + ts_naive = datetime(2024, 7, 1, 10, 0, 0) + ts_utc = datetime(2024, 7, 1, 10, 0, 0, tzinfo=timezone.utc) + table = db.create_table("t", [{"id": 1, "ts": ts_naive}]) + result = table.search().where(col("ts") == lit(ts_utc)).to_list() + assert len(result) == 1 + assert result[0]["id"] == 1 + + def test_naive_lit_sql_is_wall_clock_not_local_shifted(self): + """Regression: naive lit must not apply the host local UTC offset.""" + ts = datetime(2024, 7, 1, 10, 0, 0) + sql = lit(ts).to_sql() + # Must encode 10:00 wall clock, not 10:00+local_offset. + assert "2024-07-01 10:00:00" in sql diff --git a/python/src/expr.rs b/python/src/expr.rs index 242e88b05..eae1d96ec 100644 --- a/python/src/expr.rs +++ b/python/src/expr.rs @@ -191,8 +191,27 @@ pub fn expr_lit(value: Bound<'_, PyAny>) -> PyResult { } // datetime.datetime is a subclass of datetime.date, so it must be checked first. + // + // Python's datetime.timestamp() treats *naive* datetimes as local wall time. + // PyArrow (and therefore Lance table storage) encodes naive timestamps as + // UTC wall-clock microseconds. Using .timestamp() for naive values therefore + // shifts the literal by the local UTC offset on non-UTC machines, so + // `col("ts") == lit(naive_dt)` fails against a table that holds the same + // naive value. Fix: treat naive datetimes as UTC wall clock (match Arrow); + // keep aware datetimes on the real .timestamp() path (correct epoch). if let Ok(dt) = value.cast::() { - let ts: f64 = dt.call_method0("timestamp")?.extract()?; + let ts: f64 = if dt.getattr("tzinfo")?.is_none() { + // Force UTC interpretation of the naive wall clock. + let utc = pyo3::types::PyModule::import(value.py(), "datetime")? + .getattr("timezone")? + .getattr("utc")?; + let kwargs = pyo3::types::PyDict::new(value.py()); + kwargs.set_item("tzinfo", utc)?; + let aware = dt.call_method("replace", (), Some(&kwargs))?; + aware.call_method0("timestamp")?.extract()? + } else { + dt.call_method0("timestamp")?.extract()? + }; let micros = (ts * 1_000_000.0).round() as i64; return Ok(PyExpr(df_lit(ScalarValue::TimestampMicrosecond( Some(micros), From d742b174c4d5c10086694213e17f26bbee4c2dd2 Mon Sep 17 00:00:00 2001 From: Adityaj0 <93090622+Adityaj0@users.noreply.github.com> Date: Mon, 17 Aug 2026 11:38:38 -0700 Subject: [PATCH 06/12] fix: hybrid search silently ignores .offset() (#3769) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## Summary `LanceHybridQueryBuilder` (sync hybrid search, `table.search(query_type="hybrid")`) silently ignored `.offset()`. `self._offset` was never forwarded to the vector/FTS sub-queries and never applied when slicing the final combined/reranked result, so `.offset(N)` behaved identically to `.offset(0)` — no error, just wrong pagination. Fixes #3765 ## Changes - `_create_query_builders()`: each sub-query now fetches `limit + offset` rows so there's enough data to slice the correct window out of after combining/reranking. - `_combine_hybrid_results()` / `to_arrow()`: the final table is sliced with `offset=self._offset` instead of always starting at 0. ## Test plan - [x] New regression test `test_hybrid_query_offset` in `python/python/tests/test_hybrid_query.py` - [x] `uv run --extra tests pytest python/tests/test_hybrid_query.py -vv` — 13 passed - [x] `uv run --extra dev ruff format` / `ruff check` — clean Co-authored-by: Claude Sonnet 5 Co-authored-by: Will Jones --- python/python/lancedb/query.py | 12 +++++++++--- python/python/tests/test_hybrid_query.py | 25 ++++++++++++++++++++++++ 2 files changed, 34 insertions(+), 3 deletions(-) diff --git a/python/python/lancedb/query.py b/python/python/lancedb/query.py index 095a7b5ff..e2bb491ea 100644 --- a/python/python/lancedb/query.py +++ b/python/python/lancedb/query.py @@ -2235,6 +2235,7 @@ class LanceHybridQueryBuilder(LanceQueryBuilder): reranker=self._reranker, limit=self._limit, with_row_ids=True, + offset=self._offset, ) return self._finish_hybrid_results(results) @@ -2256,6 +2257,7 @@ class LanceHybridQueryBuilder(LanceQueryBuilder): reranker, limit: int, with_row_ids: bool, + offset: Optional[int] = None, ) -> pa.Table: if norm == "rank": vector_results = LanceHybridQueryBuilder._rank(vector_results, "_distance") @@ -2332,7 +2334,7 @@ class LanceHybridQueryBuilder(LanceQueryBuilder): score_i = results.column_names.index("_score") results = results.set_column(score_i, "_score", original_scores) - results = results.slice(length=limit) + results = results.slice(offset=offset or 0, length=limit) if not with_row_ids: results = results.drop(["_rowid"]) @@ -2679,8 +2681,12 @@ class LanceHybridQueryBuilder(LanceQueryBuilder): # Apply common configurations if self._limit: - self._vector_query.limit(self._limit) - self._fts_query.limit(self._limit) + # The final offset/limit window is sliced out of the combined, + # reranked results, so each sub-query must fetch enough rows to + # cover the skipped prefix as well as the window itself. + sub_query_limit = self._limit + (self._offset or 0) + self._vector_query.limit(sub_query_limit) + self._fts_query.limit(sub_query_limit) if self._columns: self._vector_query.select(self._columns) self._fts_query.select(self._columns) diff --git a/python/python/tests/test_hybrid_query.py b/python/python/tests/test_hybrid_query.py index 72dcaaa49..5e9b45ecb 100644 --- a/python/python/tests/test_hybrid_query.py +++ b/python/python/tests/test_hybrid_query.py @@ -203,6 +203,31 @@ async def test_async_hybrid_query_default_limit(table: AsyncTable): assert texts.count("a") == 1 +def test_hybrid_query_offset(sync_table: Table): + # The offset window of a hybrid query must be a suffix of the same query + # run without an offset -- it must not be silently ignored. + full = ( + sync_table.search(query_type="hybrid") + .vector([0.0, 0.4]) + .text("dog") + .limit(4) + .with_row_id(True) + .to_arrow() + ) + assert len(full) == 4 + + offset_result = ( + sync_table.search(query_type="hybrid") + .vector([0.0, 0.4]) + .text("dog") + .offset(2) + .limit(2) + .with_row_id(True) + .to_arrow() + ) + assert offset_result["_rowid"].to_pylist() == full["_rowid"].to_pylist()[2:] + + def test_hybrid_query_minimum_nprobes_zero_raises(sync_table: Table): # minimum_nprobes(0) must raise the same validation error a plain vector # query raises, not silently no-op because 0 is falsy. From 76942306b796b329f67a162beffc3e04c28acd1f Mon Sep 17 00:00:00 2001 From: Xuanwo Date: Tue, 18 Aug 2026 20:03:20 +0800 Subject: [PATCH 07/12] docs(java): add vended credentials example (#3958) ## Context Java users opening catalog-backed tables with vended credentials currently lack a documented workflow. Opening the catalog-returned URI directly drops the namespace-provided storage options and automatic credential refresh. Document the namespace-backed `Dataset.open()` path so temporary object store credentials are applied and refreshed transparently. --- docs/src/java/java.md | 32 ++++++++++++++++++++++++++++++++ 1 file changed, 32 insertions(+) diff --git a/docs/src/java/java.md b/docs/src/java/java.md index 42bc06b9d..df8f4a119 100644 --- a/docs/src/java/java.md +++ b/docs/src/java/java.md @@ -55,6 +55,38 @@ LanceNamespace namespaceClient = LanceDbNamespaceClientBuilder.newBuilder() | `region(String)` | AWS region (default: "us-east-1") | No | | `config(String, String)` | Additional configuration parameters | No | +### Opening a Table with Vended Credentials + +When the catalog vends temporary object store credentials, open the table through the +namespace client. The Lance dataset builder fetches the table location and storage options +from the catalog and refreshes the credentials when they expire. + +```java +import com.lancedb.LanceDbNamespaceClientBuilder; +import org.lance.Dataset; +import org.lance.namespace.LanceNamespace; + +import java.util.Arrays; + +LanceNamespace namespaceClient = LanceDbNamespaceClientBuilder.newBuilder() + .apiKey(System.getenv("LANCEDB_API_KEY")) + .database(System.getenv("LANCEDB_DATABASE")) + // Set the endpoint for a LanceDB Enterprise deployment. + // .endpoint("https://your-enterprise-endpoint") + .build(); + +try (Dataset dataset = Dataset.open() + .namespaceClient(namespaceClient) + .tableId(Arrays.asList("my_namespace", "my_table")) + .build()) { + System.out.println("Rows: " + dataset.countRows()); +} +``` + +Do not call `describeTable()` and then open the returned location with `Dataset.open(uri)`. +Opening through `namespaceClient()` is what applies the vended storage options and enables +automatic credential refresh. No object store credentials need to be passed by the application. + ## Metadata Operations ### Creating a Namespace Path From cdebea118d43e0bcc7ef3a31a959bda3c9956acf Mon Sep 17 00:00:00 2001 From: Dan Rammer Date: Tue, 18 Aug 2026 17:25:08 -0500 Subject: [PATCH 08/12] feat(python): expose LSM checkpoint and stats on sync RemoteTable (#3961) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## Summary The sync `RemoteTable` carried `set_lsm_write_spec`, `unset_lsm_write_spec`, `get_lsm_write_spec`, and `close_lsm_writers`, but not `checkpoint_lsm`, `flush_lsm`, `compact_lsm`, or `get_lsm_stats`. That left the four LSM control methods reachable from `AsyncTable` only. They are also the four that *only* work against a remote table — `NativeTable` does not override the `BaseTable` defaults, so on a local table they return `NotSupported` (`rust/lancedb/src/table.rs:679-701`). The net effect for sync users: | | `checkpoint_lsm` / `get_lsm_stats` | |---|---| | `LanceTable` (sync, local) | present, but always `NotSupported` | | `RemoteTable` (sync, remote) | `AttributeError` — method absent | | `AsyncTable` (remote) | works | So there was no working sync path at all, despite the Rust `RemoteTable` implementing every one of these against real endpoints. ## Changes * Add `checkpoint_lsm`, `flush_lsm`, `compact_lsm`, and `get_lsm_stats` to `lancedb.remote.table.RemoteTable`, mirroring the delegation style of their neighbours. * Correct the docstrings on `set_lsm_write_spec` / `unset_lsm_write_spec`, which read `"""Not supported on LanceDB Cloud."""` although `rust/lancedb/src/remote/table.rs:2549-2601` implements both against `/v1/table/{}/set_lsm_write_spec/` and `/unset_lsm_write_spec/`. They appear to have been copy-pasted from `set_unenforced_primary_key` directly above. No Rust or PyO3 changes — the bindings and the `AsyncTable` methods already existed. The `Table` ABC is left alone, matching how the existing `*_lsm_write_spec` methods are declared on the concrete classes only. ## Tests Four new tests in `python/python/tests/test_remote_db.py`, against the existing mock HTTP server: * `test_get_lsm_stats_sync` — the server payload round-trips into the dict, and `include_generation_rows` defaults to `False` and is forwarded when set. * `test_get_lsm_stats_sync_returns_none_when_lsm_disabled` — a `{"lsm_stats": null}` envelope yields `None` rather than an error. * `test_flush_and_compact_lsm_sync` — both are one-shot POSTs answered `202` with no body. * `test_checkpoint_lsm_sync` — pins the binding to the endpoints it drives (`flush_lsm` then `get_lsm_stats`); the convergence loop itself is already covered in Rust. 🤖 Generated with [Claude Code](https://claude.com/claude-code) --------- Co-authored-by: Claude Opus 5 (1M context) --- python/python/lancedb/remote/table.py | 26 +++++- python/python/lancedb/table.py | 6 +- python/python/tests/test_remote_db.py | 125 ++++++++++++++++++++++++++ 3 files changed, 152 insertions(+), 5 deletions(-) diff --git a/python/python/lancedb/remote/table.py b/python/python/lancedb/remote/table.py index aa822b913..25363cf8f 100644 --- a/python/python/lancedb/remote/table.py +++ b/python/python/lancedb/remote/table.py @@ -990,17 +990,39 @@ class RemoteTable(Table): return LOOP.run(self._table.set_unenforced_primary_key(columns)) def set_lsm_write_spec(self, spec: "LsmWriteSpec") -> None: - """Not supported on LanceDB Cloud.""" + """Install an LsmWriteSpec.""" return LOOP.run(self._table.set_lsm_write_spec(spec)) def unset_lsm_write_spec(self) -> None: - """Not supported on LanceDB Cloud.""" + """Remove the LsmWriteSpec.""" return LOOP.run(self._table.unset_lsm_write_spec()) def get_lsm_write_spec(self) -> Optional["LsmWriteSpec"]: """Read the installed LsmWriteSpec, or ``None``.""" return LOOP.run(self._table.get_lsm_write_spec()) + def checkpoint_lsm(self) -> None: + """Synchronous version of + [`AsyncTable.checkpoint_lsm`][lancedb.AsyncTable.checkpoint_lsm].""" + return LOOP.run(self._table.checkpoint_lsm()) + + def flush_lsm(self) -> None: + """Synchronous version of + [`AsyncTable.flush_lsm`][lancedb.AsyncTable.flush_lsm].""" + return LOOP.run(self._table.flush_lsm()) + + def compact_lsm(self) -> None: + """Synchronous version of + [`AsyncTable.compact_lsm`][lancedb.AsyncTable.compact_lsm].""" + return LOOP.run(self._table.compact_lsm()) + + def get_lsm_stats(self, *, include_generation_rows: bool = False) -> Optional[dict]: + """Synchronous version of + [`AsyncTable.get_lsm_stats`][lancedb.AsyncTable.get_lsm_stats].""" + return LOOP.run( + self._table.get_lsm_stats(include_generation_rows=include_generation_rows) + ) + def close_lsm_writers(self) -> None: """No-op on LanceDB Cloud (no local shard writers).""" return LOOP.run(self._table.close_lsm_writers()) diff --git a/python/python/lancedb/table.py b/python/python/lancedb/table.py index 4ecf6e836..f97d0331c 100644 --- a/python/python/lancedb/table.py +++ b/python/python/lancedb/table.py @@ -4846,7 +4846,7 @@ class AsyncTable: ``asyncio.wait_for`` for a wall-clock bound; abandoning it partway costs nothing. """ - return await self._inner.checkpoint_lsm() + await self._inner.checkpoint_lsm() async def flush_lsm(self) -> None: """Seal every bucket's active memtable into L0. @@ -4855,7 +4855,7 @@ class AsyncTable: `compact_lsm`. On a node that has not claimed this table, this claims it and replays its WAL log first. """ - return await self._inner.flush_lsm() + await self._inner.flush_lsm() async def compact_lsm(self) -> None: """Trigger a background L0 to base compaction pass per bucket. @@ -4864,7 +4864,7 @@ class AsyncTable: ``get_lsm_stats`` for progress, or use ``checkpoint_lsm`` to loop until the current L0 has reached base. """ - return await self._inner.compact_lsm() + await self._inner.compact_lsm() async def get_lsm_stats( self, *, include_generation_rows: bool = False diff --git a/python/python/tests/test_remote_db.py b/python/python/tests/test_remote_db.py index ce8d5bd6e..13ffc4415 100644 --- a/python/python/tests/test_remote_db.py +++ b/python/python/tests/test_remote_db.py @@ -1133,6 +1133,131 @@ def test_stats(): assert res == stats +@contextlib.contextmanager +def lsm_test_table(lsm_handler): + """A remote table whose LSM routes are served by ``lsm_handler``. + + ``lsm_handler(request, route)`` is called for ``/v1/table/test//`` + where route is one of flush_lsm, compact_lsm, get_lsm_stats, and is + responsible for writing the response. + """ + routes = ("flush_lsm", "compact_lsm", "get_lsm_stats") + + def handler(request): + match = re.fullmatch(r"/v1/table/test/(\w+)/", request.path) + route = match.group(1) if match else None + if route in routes: + lsm_handler(request, route) + elif route == "describe": + request.send_response(200) + request.send_header("Content-Type", "application/json") + request.end_headers() + request.wfile.write(b'{"version": 1, "schema": {"fields": []}}') + else: + request.send_response(404) + request.end_headers() + + with mock_lancedb_connection(handler) as db: + yield db.open_table("test") + + +def read_json_body(request): + content_len = int(request.headers.get("Content-Length")) + return json.loads(request.rfile.read(content_len)) + + +def send_json(request, payload, status=200): + request.send_response(status) + request.send_header("Content-Type", "application/json") + request.end_headers() + request.wfile.write(json.dumps(payload).encode()) + + +def test_get_lsm_stats_sync(): + """The sync wrapper round-trips the server payload into a dict.""" + bucket = { + "shard_id": "b0", + "status": "Active", + "writer_epoch": 3, + "manifest_version": 12, + "current_generation": 6, + "replay_after_wal_entry_position": 40, + "wal_entry_position_last_seen": 42, + "generations": [{"generation": 5, "bytes": 1024, "rows": 7}], + "compacting": False, + "memtables": [ + { + "generation": 6, + "rows": 2, + "bytes": 64, + "batches": 1, + "indexes": ["vec_idx"], + } + ], + } + seen_bodies = [] + + def lsm_handler(request, route): + assert route == "get_lsm_stats" + seen_bodies.append(read_json_body(request)) + send_json(request, {"lsm_stats": {"buckets": [bucket]}}) + + with lsm_test_table(lsm_handler) as table: + assert table.get_lsm_stats() == {"buckets": [bucket]} + # Off by default, and forwarded when asked for. + assert seen_bodies == [{"include_generation_rows": False}] + table.get_lsm_stats(include_generation_rows=True) + assert seen_bodies[-1] == {"include_generation_rows": True} + + +def test_get_lsm_stats_sync_returns_none_when_lsm_disabled(): + """A null envelope means the LSM write path is not enabled, not an error.""" + + def lsm_handler(request, route): + send_json(request, {"lsm_stats": None}) + + with lsm_test_table(lsm_handler) as table: + assert table.get_lsm_stats() is None + + +def test_flush_and_compact_lsm_sync(): + """Both are one-shot POSTs answered 202 with no body.""" + called = [] + + def lsm_handler(request, route): + called.append(route) + request.send_response(202) + request.end_headers() + + with lsm_test_table(lsm_handler) as table: + assert table.flush_lsm() is None + assert table.compact_lsm() is None + assert called == ["flush_lsm", "compact_lsm"] + + +def test_checkpoint_lsm_sync(): + """Seal, read the watermark, and return once L0 holds nothing. + + The convergence loop itself is covered in Rust; this pins the sync + binding to the endpoints it drives. + """ + called = [] + + def lsm_handler(request, route): + called.append(route) + if route == "get_lsm_stats": + # An empty L0 yields no target watermark, so the loop is done + # after the seal without ever polling compaction. + send_json(request, {"lsm_stats": {"buckets": []}}) + else: + request.send_response(202) + request.end_headers() + + with lsm_test_table(lsm_handler) as table: + assert table.checkpoint_lsm() is None + assert called == ["flush_lsm", "get_lsm_stats"] + + @contextlib.contextmanager def query_test_table(query_handler, *, server_version=Version("0.1.0")): def handler(request): From f6efdc9e9f2c705c4102db78b55cddddb50173bd Mon Sep 17 00:00:00 2001 From: Lance Release Date: Wed, 19 Aug 2026 01:58:48 +0000 Subject: [PATCH 09/12] =?UTF-8?q?Bump=20version:=200.38.0-beta.1=20?= =?UTF-8?q?=E2=86=92=200.38.0-beta.2?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .bumpversion.toml | 2 +- Cargo.lock | 6 +++--- docs/src/java/java.md | 2 +- java/lancedb-core/pom.xml | 2 +- java/pom.xml | 2 +- nodejs/Cargo.toml | 2 +- nodejs/npm/darwin-arm64/package.json | 2 +- nodejs/npm/linux-arm64-gnu/package.json | 2 +- nodejs/npm/linux-arm64-musl/package.json | 2 +- nodejs/npm/linux-x64-gnu/package.json | 2 +- nodejs/npm/linux-x64-musl/package.json | 2 +- nodejs/npm/win32-arm64-msvc/package.json | 2 +- nodejs/npm/win32-x64-msvc/package.json | 2 +- nodejs/package-lock.json | 4 ++-- nodejs/package.json | 2 +- python/Cargo.toml | 2 +- rust/lancedb/Cargo.toml | 2 +- 17 files changed, 20 insertions(+), 20 deletions(-) diff --git a/.bumpversion.toml b/.bumpversion.toml index e2e693549..b0be7dc82 100644 --- a/.bumpversion.toml +++ b/.bumpversion.toml @@ -1,5 +1,5 @@ [tool.bumpversion] -current_version = "0.38.0-beta.1" +current_version = "0.38.0-beta.2" parse = """(?x) (?P0|[1-9]\\d*)\\. (?P0|[1-9]\\d*)\\. diff --git a/Cargo.lock b/Cargo.lock index 2f7499e91..5bd5479fd 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -5398,7 +5398,7 @@ dependencies = [ [[package]] name = "lancedb" -version = "0.38.0-beta.1" +version = "0.38.0-beta.2" dependencies = [ "ahash", "anyhow", @@ -5486,7 +5486,7 @@ dependencies = [ [[package]] name = "lancedb-nodejs" -version = "0.38.0-beta.1" +version = "0.38.0-beta.2" dependencies = [ "arrow-array", "arrow-buffer", @@ -5511,7 +5511,7 @@ dependencies = [ [[package]] name = "lancedb-python" -version = "0.38.0-beta.1" +version = "0.38.0-beta.2" dependencies = [ "arrow", "async-trait", diff --git a/docs/src/java/java.md b/docs/src/java/java.md index df8f4a119..452bd54f4 100644 --- a/docs/src/java/java.md +++ b/docs/src/java/java.md @@ -14,7 +14,7 @@ Add the following dependency to your `pom.xml`: com.lancedb lancedb-core - 0.38.0-beta.1 + 0.38.0-beta.2 ``` diff --git a/java/lancedb-core/pom.xml b/java/lancedb-core/pom.xml index 94c72b326..fb0d2618f 100644 --- a/java/lancedb-core/pom.xml +++ b/java/lancedb-core/pom.xml @@ -8,7 +8,7 @@ com.lancedb lancedb-parent - 0.38.0-beta.1 + 0.38.0-beta.2 ../pom.xml diff --git a/java/pom.xml b/java/pom.xml index 90ad2f7f8..5f47d6f19 100644 --- a/java/pom.xml +++ b/java/pom.xml @@ -6,7 +6,7 @@ com.lancedb lancedb-parent - 0.38.0-beta.1 + 0.38.0-beta.2 pom ${project.artifactId} LanceDB Java SDK Parent POM diff --git a/nodejs/Cargo.toml b/nodejs/Cargo.toml index 3496e2839..9b9b56f7e 100644 --- a/nodejs/Cargo.toml +++ b/nodejs/Cargo.toml @@ -1,7 +1,7 @@ [package] name = "lancedb-nodejs" edition.workspace = true -version = "0.38.0-beta.1" +version = "0.38.0-beta.2" publish = false license.workspace = true description.workspace = true diff --git a/nodejs/npm/darwin-arm64/package.json b/nodejs/npm/darwin-arm64/package.json index ad1503090..c2be3aeac 100644 --- a/nodejs/npm/darwin-arm64/package.json +++ b/nodejs/npm/darwin-arm64/package.json @@ -1,6 +1,6 @@ { "name": "@lancedb/lancedb-darwin-arm64", - "version": "0.38.0-beta.1", + "version": "0.38.0-beta.2", "os": ["darwin"], "cpu": ["arm64"], "main": "lancedb.darwin-arm64.node", diff --git a/nodejs/npm/linux-arm64-gnu/package.json b/nodejs/npm/linux-arm64-gnu/package.json index e7455832d..901405fe4 100644 --- a/nodejs/npm/linux-arm64-gnu/package.json +++ b/nodejs/npm/linux-arm64-gnu/package.json @@ -1,6 +1,6 @@ { "name": "@lancedb/lancedb-linux-arm64-gnu", - "version": "0.38.0-beta.1", + "version": "0.38.0-beta.2", "os": ["linux"], "cpu": ["arm64"], "main": "lancedb.linux-arm64-gnu.node", diff --git a/nodejs/npm/linux-arm64-musl/package.json b/nodejs/npm/linux-arm64-musl/package.json index 8269bc4ce..415e60c78 100644 --- a/nodejs/npm/linux-arm64-musl/package.json +++ b/nodejs/npm/linux-arm64-musl/package.json @@ -1,6 +1,6 @@ { "name": "@lancedb/lancedb-linux-arm64-musl", - "version": "0.38.0-beta.1", + "version": "0.38.0-beta.2", "os": ["linux"], "cpu": ["arm64"], "main": "lancedb.linux-arm64-musl.node", diff --git a/nodejs/npm/linux-x64-gnu/package.json b/nodejs/npm/linux-x64-gnu/package.json index 7a7d0a097..22416dbcb 100644 --- a/nodejs/npm/linux-x64-gnu/package.json +++ b/nodejs/npm/linux-x64-gnu/package.json @@ -1,6 +1,6 @@ { "name": "@lancedb/lancedb-linux-x64-gnu", - "version": "0.38.0-beta.1", + "version": "0.38.0-beta.2", "os": ["linux"], "cpu": ["x64"], "main": "lancedb.linux-x64-gnu.node", diff --git a/nodejs/npm/linux-x64-musl/package.json b/nodejs/npm/linux-x64-musl/package.json index ce95c0174..77d6a5dd5 100644 --- a/nodejs/npm/linux-x64-musl/package.json +++ b/nodejs/npm/linux-x64-musl/package.json @@ -1,6 +1,6 @@ { "name": "@lancedb/lancedb-linux-x64-musl", - "version": "0.38.0-beta.1", + "version": "0.38.0-beta.2", "os": ["linux"], "cpu": ["x64"], "main": "lancedb.linux-x64-musl.node", diff --git a/nodejs/npm/win32-arm64-msvc/package.json b/nodejs/npm/win32-arm64-msvc/package.json index 2ae43e763..0dea90f81 100644 --- a/nodejs/npm/win32-arm64-msvc/package.json +++ b/nodejs/npm/win32-arm64-msvc/package.json @@ -1,6 +1,6 @@ { "name": "@lancedb/lancedb-win32-arm64-msvc", - "version": "0.38.0-beta.1", + "version": "0.38.0-beta.2", "os": [ "win32" ], diff --git a/nodejs/npm/win32-x64-msvc/package.json b/nodejs/npm/win32-x64-msvc/package.json index 1030609fa..0be0b457b 100644 --- a/nodejs/npm/win32-x64-msvc/package.json +++ b/nodejs/npm/win32-x64-msvc/package.json @@ -1,6 +1,6 @@ { "name": "@lancedb/lancedb-win32-x64-msvc", - "version": "0.38.0-beta.1", + "version": "0.38.0-beta.2", "os": ["win32"], "cpu": ["x64"], "main": "lancedb.win32-x64-msvc.node", diff --git a/nodejs/package-lock.json b/nodejs/package-lock.json index 26091b118..999b3f16f 100644 --- a/nodejs/package-lock.json +++ b/nodejs/package-lock.json @@ -1,12 +1,12 @@ { "name": "@lancedb/lancedb", - "version": "0.38.0-beta.1", + "version": "0.38.0-beta.2", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "@lancedb/lancedb", - "version": "0.38.0-beta.1", + "version": "0.38.0-beta.2", "cpu": [ "x64", "arm64" diff --git a/nodejs/package.json b/nodejs/package.json index cfdc851ef..8291d3dc8 100644 --- a/nodejs/package.json +++ b/nodejs/package.json @@ -11,7 +11,7 @@ "ann" ], "private": false, - "version": "0.38.0-beta.1", + "version": "0.38.0-beta.2", "main": "dist/index.js", "exports": { ".": "./dist/index.js", diff --git a/python/Cargo.toml b/python/Cargo.toml index 745ca4ea2..e41563266 100644 --- a/python/Cargo.toml +++ b/python/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "lancedb-python" -version = "0.38.0-beta.1" +version = "0.38.0-beta.2" publish = false edition.workspace = true description = "Python bindings for LanceDB" diff --git a/rust/lancedb/Cargo.toml b/rust/lancedb/Cargo.toml index 92f020956..69d07b2d8 100644 --- a/rust/lancedb/Cargo.toml +++ b/rust/lancedb/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "lancedb" -version = "0.38.0-beta.1" +version = "0.38.0-beta.2" edition.workspace = true description = "LanceDB: A serverless, low-latency vector database for AI applications" license.workspace = true From 11c1d816389cfbba78eaad42a464aed454d133bf Mon Sep 17 00:00:00 2001 From: LanceDB Robot Date: Wed, 19 Aug 2026 06:04:45 -0700 Subject: [PATCH 10/12] chore: update lance dependency to v11.0.0-beta.14 (#3965) Updates the Rust workspace Lance dependencies and Java lance-core dependency to v11.0.0-beta.14. No compatibility fixes were required; full workspace clippy with all features passes. Trigger: https://github.com/lance-format/lance/releases/tag/v11.0.0-beta.14 --------- Co-authored-by: Yang Cen --- Cargo.lock | 96 ++++++++++++++++++++++++++-------------------------- Cargo.toml | 28 +++++++-------- deny.toml | 6 ++++ java/pom.xml | 2 +- 4 files changed, 69 insertions(+), 63 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 5bd5479fd..3f4d6682c 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -959,7 +959,7 @@ dependencies = [ "aws-smithy-runtime-api", "aws-smithy-types", "h2 0.3.27", - "h2 0.4.14", + "h2 0.4.16", "http 0.2.12", "http 1.5.0", "http-body 0.4.6", @@ -3455,8 +3455,8 @@ checksum = "42703706b716c37f96a77aea830392ad231f44c9e9a67872fa5548707e11b11c" [[package]] name = "fsst" -version = "11.0.0-beta.13" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" +version = "11.0.0-beta.14" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" dependencies = [ "arrow-array", "rand 0.9.5", @@ -3877,9 +3877,9 @@ dependencies = [ [[package]] name = "h2" -version = "0.4.14" +version = "0.4.16" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "171fefbc92fe4a4de27e0698d6a5b392d6a0e333506bc49133760b3bcf948733" +checksum = "a9f37a958b41b3b19ee2707c06439c0e9e547e847223eb791ecb0cb821c65e27" dependencies = [ "atomic-waker", "bytes", @@ -4188,7 +4188,7 @@ dependencies = [ "bytes", "futures-channel", "futures-core", - "h2 0.4.14", + "h2 0.4.16", "http 1.5.0", "http-body 1.1.0", "httparse", @@ -4815,8 +4815,8 @@ checksum = "e037a2e1d8d5fdbd49b16a4ea09d5d6401c1f29eca5ff29d03d3824dba16256a" [[package]] name = "lance" -version = "11.0.0-beta.13" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" +version = "11.0.0-beta.14" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" dependencies = [ "arc-swap", "arrow", @@ -4888,8 +4888,8 @@ dependencies = [ [[package]] name = "lance-arrow" -version = "11.0.0-beta.13" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" +version = "11.0.0-beta.14" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" dependencies = [ "arrow-array", "arrow-buffer", @@ -4911,7 +4911,7 @@ dependencies = [ [[package]] name = "lance-arrow-scalar" version = "58.0.0" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" dependencies = [ "arrow-array", "arrow-buffer", @@ -4925,7 +4925,7 @@ dependencies = [ [[package]] name = "lance-arrow-stats" version = "58.0.0" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" dependencies = [ "arrow-array", "arrow-schema", @@ -4934,8 +4934,8 @@ dependencies = [ [[package]] name = "lance-bitpacking" -version = "11.0.0-beta.13" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" +version = "11.0.0-beta.14" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" dependencies = [ "arrayref", "crunchy", @@ -4945,8 +4945,8 @@ dependencies = [ [[package]] name = "lance-core" -version = "11.0.0-beta.13" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" +version = "11.0.0-beta.14" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" dependencies = [ "arrow-array", "arrow-buffer", @@ -4983,8 +4983,8 @@ dependencies = [ [[package]] name = "lance-datafusion" -version = "11.0.0-beta.13" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" +version = "11.0.0-beta.14" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" dependencies = [ "arrow", "arrow-array", @@ -5013,8 +5013,8 @@ dependencies = [ [[package]] name = "lance-datagen" -version = "11.0.0-beta.13" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" +version = "11.0.0-beta.14" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" dependencies = [ "arrow", "arrow-array", @@ -5031,8 +5031,8 @@ dependencies = [ [[package]] name = "lance-derive" -version = "11.0.0-beta.13" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" +version = "11.0.0-beta.14" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" dependencies = [ "proc-macro2", "quote", @@ -5041,8 +5041,8 @@ dependencies = [ [[package]] name = "lance-encoding" -version = "11.0.0-beta.13" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" +version = "11.0.0-beta.14" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" dependencies = [ "arrow-arith", "arrow-array", @@ -5075,8 +5075,8 @@ dependencies = [ [[package]] name = "lance-file" -version = "11.0.0-beta.13" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" +version = "11.0.0-beta.14" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" dependencies = [ "arrow-arith", "arrow-array", @@ -5107,8 +5107,8 @@ dependencies = [ [[package]] name = "lance-index" -version = "11.0.0-beta.13" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" +version = "11.0.0-beta.14" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" dependencies = [ "arc-swap", "arrow", @@ -5172,8 +5172,8 @@ dependencies = [ [[package]] name = "lance-index-core" -version = "11.0.0-beta.13" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" +version = "11.0.0-beta.14" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" dependencies = [ "arrow-array", "arrow-schema", @@ -5195,8 +5195,8 @@ dependencies = [ [[package]] name = "lance-io" -version = "11.0.0-beta.13" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" +version = "11.0.0-beta.14" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" dependencies = [ "arrow", "arrow-array", @@ -5232,8 +5232,8 @@ dependencies = [ [[package]] name = "lance-linalg" -version = "11.0.0-beta.13" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" +version = "11.0.0-beta.14" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" dependencies = [ "arrow-array", "arrow-schema", @@ -5247,8 +5247,8 @@ dependencies = [ [[package]] name = "lance-namespace" -version = "11.0.0-beta.13" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" +version = "11.0.0-beta.14" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" dependencies = [ "arrow", "async-trait", @@ -5260,8 +5260,8 @@ dependencies = [ [[package]] name = "lance-namespace-impls" -version = "11.0.0-beta.13" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" +version = "11.0.0-beta.14" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" dependencies = [ "arrow", "arrow-ipc", @@ -5314,8 +5314,8 @@ dependencies = [ [[package]] name = "lance-select" -version = "11.0.0-beta.13" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" +version = "11.0.0-beta.14" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" dependencies = [ "arrow-array", "arrow-buffer", @@ -5329,8 +5329,8 @@ dependencies = [ [[package]] name = "lance-table" -version = "11.0.0-beta.13" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" +version = "11.0.0-beta.14" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" dependencies = [ "arrow", "arrow-array", @@ -5370,8 +5370,8 @@ dependencies = [ [[package]] name = "lance-testing" -version = "11.0.0-beta.13" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" +version = "11.0.0-beta.14" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" dependencies = [ "arrow-array", "arrow-schema", @@ -5384,8 +5384,8 @@ dependencies = [ [[package]] name = "lance-tokenizer" -version = "11.0.0-beta.13" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.13#ee41152ceb9a78e5df4d2456fdbdb98542eb2059" +version = "11.0.0-beta.14" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" dependencies = [ "frostem", "icu_segmenter", @@ -8426,7 +8426,7 @@ dependencies = [ "encoding_rs", "futures-core", "futures-util", - "h2 0.4.14", + "h2 0.4.16", "http 1.5.0", "http-body 1.1.0", "http-body-util", @@ -10082,7 +10082,7 @@ dependencies = [ "async-trait", "base64 0.22.1", "bytes", - "h2 0.4.14", + "h2 0.4.16", "http 1.5.0", "http-body 1.1.0", "http-body-util", diff --git a/Cargo.toml b/Cargo.toml index 2a19cbb00..c90fb81d8 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -13,20 +13,20 @@ categories = ["database-implementations"] rust-version = "1.91.0" [workspace.dependencies] -lance = { "version" = "=11.0.0-beta.13", default-features = false, "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } -lance-core = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } -lance-datagen = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } -lance-file = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } -lance-io = { "version" = "=11.0.0-beta.13", default-features = false, "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } -lance-index = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } -lance-linalg = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } -lance-namespace = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } -lance-namespace-impls = { "version" = "=11.0.0-beta.13", default-features = false, "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } -lance-table = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } -lance-testing = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } -lance-datafusion = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } -lance-encoding = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } -lance-arrow = { "version" = "=11.0.0-beta.13", "tag" = "v11.0.0-beta.13", "git" = "https://github.com/lance-format/lance.git" } +lance = { "version" = "=11.0.0-beta.14", default-features = false, "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } +lance-core = { "version" = "=11.0.0-beta.14", "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } +lance-datagen = { "version" = "=11.0.0-beta.14", "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } +lance-file = { "version" = "=11.0.0-beta.14", "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } +lance-io = { "version" = "=11.0.0-beta.14", default-features = false, "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } +lance-index = { "version" = "=11.0.0-beta.14", "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } +lance-linalg = { "version" = "=11.0.0-beta.14", "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } +lance-namespace = { "version" = "=11.0.0-beta.14", "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } +lance-namespace-impls = { "version" = "=11.0.0-beta.14", default-features = false, "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } +lance-table = { "version" = "=11.0.0-beta.14", "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } +lance-testing = { "version" = "=11.0.0-beta.14", "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } +lance-datafusion = { "version" = "=11.0.0-beta.14", "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } +lance-encoding = { "version" = "=11.0.0-beta.14", "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } +lance-arrow = { "version" = "=11.0.0-beta.14", "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } ahash = "0.8" # Note that this one does not include pyarrow arrow = { version = "58.0.0", optional = false } diff --git a/deny.toml b/deny.toml index d94c9d536..cea2522fd 100644 --- a/deny.toml +++ b/deny.toml @@ -108,6 +108,12 @@ ignore = [ # compact_str/smol_str, so clearing this requires polars to migrate. # https://rustsec.org/advisories/RUSTSEC-2026-0249 { id = "RUSTSEC-2026-0249", reason = "smartstring unmaintained via polars; no fixed upstream release" }, + + # h2 0.3: empty DATA frames can be queued without limit. The patched + # h2 0.4 line is locked to 0.4.16, but no patched 0.3 release exists. + # The old copy is pulled in by aws-smithy's legacy hyper 0.14 client. + # https://rustsec.org/advisories/RUSTSEC-2026-0258 + { id = "RUSTSEC-2026-0258", reason = "h2 0.3 via legacy aws-smithy/hyper 0.14; no patched 0.3 release" }, ] # --------------------------------------------------------------------------- diff --git a/java/pom.xml b/java/pom.xml index 5f47d6f19..63711d0c9 100644 --- a/java/pom.xml +++ b/java/pom.xml @@ -28,7 +28,7 @@ UTF-8 15.0.0 - 11.0.0-beta.13 + 11.0.0-beta.14 false 2.30.0 1.7 From f1c4967eebf2c9a08e9bcce7a12fd10fd64a8740 Mon Sep 17 00:00:00 2001 From: Dan Rammer Date: Wed, 19 Aug 2026 11:44:46 -0500 Subject: [PATCH 11/12] feat: bring the MemWAL LSM surface to parity across the SDKs (#3962) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## Why Four of the eight LSM methods are **remote-only in the core**. `impl BaseTable for NativeTable` implements only `set`/`unset`/`get_lsm_write_spec` and `close_lsm_writers`; `flush_lsm`, `compact_lsm` and `get_lsm_stats` fall through to trait defaults returning `NotSupported` (`rust/lancedb/src/table.rs:679,687,696`), and `checkpoint_lsm` is built on all three. That explains the state of the bindings: Node had bound the four that work against a local table and stopped, so a Cloud user could install an LSM write spec but had no way to observe fresh-tier state or drive a checkpoint. Java had none of it at all. | SDK | set/unset/get spec | closeWriters | flush | compact | getStats | checkpoint | |---|---|---|---|---|---|---| | Rust core | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | Python | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | Node *(before)* | ✅ | ✅ | — | — | — | — | | **Node (after)** | ✅ | ✅ | **new** | **new** | **new** | **new** | | Java *(before)* | — | — | — | — | — | — | | **Java (after)** | **new** | n/a | **new** | **new** | **new** | **new** | Go and C are separate repos and are out of scope here. `closeLsmWriters` drains cached in-process shard writers, so it has no meaning for Java, which is a pure REST client. ## Node Adds napi bindings for `flushLsm`, `compactLsm`, `checkpointLsm` and `getLsmStats`, plus typed `LsmStats` / `BucketStats` / `GenerationStats` / `MemtableStats` objects — typed rather than a JSON blob, matching the existing `LsmWriteSpec` object in the same file, with `u64` cast to `i64` per that file's convention. Because these four are remote-only, the new tests assert each binding reaches the core and surfaces `NotSupported` against a local table. That covers the wiring; behavior against a real endpoint stays covered by the mocked-endpoint tests in `rust/lancedb/src/remote/table.rs`. ## Python No new methods. All eight are on `LanceTable`, `AsyncTable` and `RemoteTable` — the last four landed on the sync `RemoteTable` in #3961, which is merged into this branch. What was missing here was reachability. `LsmWriteSpec` was importable only from the private `lancedb._lancedb`, appearing in `table.py` solely under `if TYPE_CHECKING:`, and `docs/src/python/python.md` had no mention of it, which per the repo's docs guidance means it rendered nowhere in the API reference. It is now `lancedb.LsmWriteSpec`, in `__all__`, and documented. ## Java Java reaches LanceDB purely over REST through the generated Lance Namespace client, and these routes are not in that spec, so they are issued through a small dedicated client rather than added to the spec. That call is revisitable — LSM is one of four unspecified route families alongside `multipart_write`, `page_cache/prewarm` and `branches/diff|merge`. If those are ever regularized into the spec as a group, `LanceDbTableLsm` is one file that gets deleted. `LsmWriteSpec` here is deliberately **not** `org.lance.memwal.InitializeMemWalParams`. That type defaults to maintaining *no* indexes where a spec here defaults to maintaining *every* index, and it cannot express the `null` that asks the server to resolve the set: | Value | On the wire | Meaning | |---|---|---| | unset (null) | `null` | Server resolves **every** maintainable index | | `Collections.emptyList()` | `[]` | Maintain **none** | | `Arrays.asList("id_idx")` | `["id_idx"]` | Exactly those | A dedicated test pins null and `[]` as distinct on the wire, since collapsing them is the failure mode that motivated a LanceDB-owned type. `checkpointLsm` is ported from `rust/lancedb/src/table/checkpoint.rs` with its constants and status semantics intact: 429/503 retried in place against an 8-budget, 421 restarting from flush against a 3-budget, 5s poll, and a target watermark fixed after the seal so it terminates under write load. `getLsmStats` returns typed `LsmStats` / `BucketStats` / `GenerationStats` / `MemtableStats`, mirroring the Rust structs in `rust/lancedb/src/table/lsm_stats.rs` and the objects Node exposes. Decoding is strict — see below. ## Review feedback Both gatekeeper findings were real. Each was reproduced against the scripted test server first, and each fix ships with the reproducer as a regression test. **The transport was doubling every checkpoint retry budget.** `HttpClients.createDefault()` installs Apache's default response retry strategy, whose retryable-status list is exactly 429 and 503 — the two statuses `isRetryable` owns. A 429 held against `flush_lsm` issued **18** wire requests where the loop intends 9, and `compact_lsm` was retried in place despite the loop being built to fall through to a fresh stats poll instead. Timing confirmed the mechanism: that run took 25.4s ≈ 16.3s of the loop's own backoff plus 9 × the transport's 1s retry interval. Automatic retries are now disabled, so the checkpoint loop is the sole owner of the 421/429/503 transitions. A side effect worth noting: `testCheckpointRetriesRetryableStatusInPlace` was passing on a transport-absorbed 429 and never reaching `issue()`'s retry branch at all. It now exercises the real path. **Stats decoding failed open.** `getLsmStats` read the response with Jackson's `path()`, which yields a missing node that iterates as an empty array — making "malformed" indistinguishable from "no buckets", which is indistinguishable from "drained". Four separate payloads made `checkpointLsm()` report convergence for a checkpoint that never ran: | Response | Before | Now | |---|---|---| | `{"lsm_stats": null}` or absent key | disabled ✓ | disabled ✓ | | `{"lsm_stats": {}}` | **reported success** | `IllegalStateException` | | empty response body | **reported success** | `IllegalStateException` | | bucket missing required fields | **reported success** | `IllegalStateException` | The empty-body row is the one to weight: a proxy 200 with no body is a realistic production event, and it silently reported a checkpoint that never happened. Decoding is now strict and fails closed, matching the serde contract on the Rust side exactly. One deliberate deviation from the review comment, which asked that *only* explicit JSON `null` count as disabled: Rust has `#[serde(default)]` on `lsm_stats`, so an **absent key** decodes to `None` there too. Java now matches that. It is an absent-or-malformed **`buckets`** that fails closed, which is the case the comment was actually protecting. ## Testing - Java: **33 passing** (8 existing + 25 LSM) against a scripted `com.sun.net.httpserver.HttpServer` — no new test dependency. Wire assertions mirror `rust/lancedb/src/remote/table.rs:6581-6748`; checkpoint tests cover convergence, not piling onto a latched bucket, 421 restart-from-flush, 429 retry-in-place, terminal-status propagation, reissue exhaustion, the exact wire-request count against the retry budget, and five malformed stats payloads. - Node: **19 LSM tests passing**; `cargo check`, `npm run build`, `npm run tsc`, `npm run lint`, `npm run docs` all clean. - Python: `ruff format --check` and `ruff check` clean. - Java formatting: `./mvnw -pl lancedb-core spotless:apply` and `spotless:check` both clean under a JDK 11 toolchain. ## Note: spotless needs a pre-16 JDK `./mvnw spotless:apply` fails on JDK 16+ with `JCTree$JCImport.getQualifiedIdentifier()` — google-java-format 1.7, pinned at `java/pom.xml:34`, predates JDK 16's compiler API change. **This is pre-existing** and reproduces on a pristine `main` checkout. It is not a blocker, just a toolchain requirement. Spotless was run against these sources under JDK 11 and both `spotless:apply` and `spotless:check` pass on the whole module: ```shell JAVA_HOME=/path/to/jdk11 ./mvnw -pl lancedb-core spotless:apply ``` Bumping the plugin so it works on modern JDKs is still worth doing, but separately from this PR. 🤖 Generated with [Claude Code](https://claude.com/claude-code) --------- Co-authored-by: Claude Opus 5 (1M context) --- docs/src/js/classes/Table.md | 93 +++ docs/src/js/globals.md | 4 + docs/src/js/interfaces/BucketStats.md | 116 ++++ docs/src/js/interfaces/GenerationStats.md | 40 ++ docs/src/js/interfaces/LsmStats.md | 22 + docs/src/js/interfaces/MemtableStats.md | 60 ++ docs/src/python/python.md | 2 + java/README.md | 42 ++ java/lancedb-core/pom.xml | 14 + .../main/java/com/lancedb/BucketStats.java | 194 ++++++ .../java/com/lancedb/GenerationStats.java | 64 ++ .../src/main/java/com/lancedb/JsonFields.java | 109 ++++ .../LanceDbNamespaceClientBuilder.java | 49 +- .../java/com/lancedb/LanceDbRestClient.java | 119 ++++ .../java/com/lancedb/LanceDbTableLsm.java | 394 ++++++++++++ .../src/main/java/com/lancedb/LsmStats.java | 56 ++ .../main/java/com/lancedb/LsmWriteSpec.java | 260 ++++++++ .../main/java/com/lancedb/MemtableStats.java | 99 +++ .../java/com/lancedb/LanceDbTableLsmTest.java | 570 ++++++++++++++++++ nodejs/__test__/table.test.ts | 53 ++ nodejs/lancedb/index.ts | 4 + nodejs/lancedb/table.ts | 78 +++ nodejs/src/table.rs | 151 +++++ python/python/lancedb/__init__.py | 2 + python/python/lancedb/table.py | 2 +- 25 files changed, 2581 insertions(+), 16 deletions(-) create mode 100644 docs/src/js/interfaces/BucketStats.md create mode 100644 docs/src/js/interfaces/GenerationStats.md create mode 100644 docs/src/js/interfaces/LsmStats.md create mode 100644 docs/src/js/interfaces/MemtableStats.md create mode 100644 java/lancedb-core/src/main/java/com/lancedb/BucketStats.java create mode 100644 java/lancedb-core/src/main/java/com/lancedb/GenerationStats.java create mode 100644 java/lancedb-core/src/main/java/com/lancedb/JsonFields.java create mode 100644 java/lancedb-core/src/main/java/com/lancedb/LanceDbRestClient.java create mode 100644 java/lancedb-core/src/main/java/com/lancedb/LanceDbTableLsm.java create mode 100644 java/lancedb-core/src/main/java/com/lancedb/LsmStats.java create mode 100644 java/lancedb-core/src/main/java/com/lancedb/LsmWriteSpec.java create mode 100644 java/lancedb-core/src/main/java/com/lancedb/MemtableStats.java create mode 100644 java/lancedb-core/src/test/java/com/lancedb/LanceDbTableLsmTest.java diff --git a/docs/src/js/classes/Table.md b/docs/src/js/classes/Table.md index 4479bf4e4..06dc8479e 100644 --- a/docs/src/js/classes/Table.md +++ b/docs/src/js/classes/Table.md @@ -213,6 +213,39 @@ version of the table. *** +### checkpointLsm() + +```ts +abstract checkpointLsm(): Promise +``` + +Converge this table's LSM write path into its base table. + +Seals once, then triggers compaction and polls until the L0 that existed +at the start is gone. The target set is fixed at the start, so +generations created *during* the checkpoint are ignored — that is what +lets it terminate under write load, and what makes it best-effort: it +converges the fresh tier as of some instant. Idempotent, abandonable at +any point, and safe to run on a cadence. + +There is no liveness bound — the compactor pool is shared across tables, +so a checkpoint queued behind unrelated work looks exactly like one that +is merging. The caller owns the deadline. + +#### Returns + +`Promise`<`void`> + +#### Example + +```ts +const before = await table.getLsmStats(); +await table.checkpointLsm(); +const after = await table.getLsmStats(); +``` + +*** + ### close() ```ts @@ -250,6 +283,24 @@ It is a no-op when no writers are cached. *** +### compactLsm() + +```ts +abstract compactLsm(): Promise +``` + +Trigger a background L0 → base compaction pass per bucket. + +Returns once the passes are *dispatched*, not once they finish — watch +[Table#getLsmStats](Table.md#getlsmstats) for progress, or use +[Table#checkpointLsm](Table.md#checkpointlsm) to wait for convergence. + +#### Returns + +`Promise`<`void`> + +*** + ### countRows() ```ts @@ -448,6 +499,48 @@ Drop an index from the table. *** +### flushLsm() + +```ts +abstract flushLsm(): Promise +``` + +Seal every bucket's active memtable into a new L0 generation. + +Returns once the seal is committed. Sealing an empty memtable is a no-op, +so this is safe to call repeatedly. + +#### Returns + +`Promise`<`void`> + +*** + +### getLsmStats() + +```ts +abstract getLsmStats(includeGenerationRows?): Promise +``` + +Read live per-bucket LSM state. + +Answers "how far behind is my fresh tier", "which bucket is hot", and +"why is my fresh-tier vector search brute-force". Mutates no table state. + +Resolves to `undefined` only when the LSM write path is not enabled. + +#### Parameters + +* **includeGenerationRows?**: `boolean` + Also count rows per L0 generation. + Off by default because each count opens an uncached Lance dataset. + +#### Returns + +`Promise`<`undefined` \| [`LsmStats`](../interfaces/LsmStats.md)> + +*** + ### getLsmWriteSpec() ```ts diff --git a/docs/src/js/globals.md b/docs/src/js/globals.md index bd2ca54b5..462907cfd 100644 --- a/docs/src/js/globals.md +++ b/docs/src/js/globals.md @@ -58,6 +58,7 @@ - [BranchDiff](interfaces/BranchDiff.md) - [BranchIndexSummary](interfaces/BranchIndexSummary.md) - [BranchRowCountSummary](interfaces/BranchRowCountSummary.md) +- [BucketStats](interfaces/BucketStats.md) - [ClientConfig](interfaces/ClientConfig.md) - [ColumnAlteration](interfaces/ColumnAlteration.md) - [ColumnOrdering](interfaces/ColumnOrdering.md) @@ -81,6 +82,7 @@ - [FtsToken](interfaces/FtsToken.md) - [FullTextQuery](interfaces/FullTextQuery.md) - [FullTextSearchOptions](interfaces/FullTextSearchOptions.md) +- [GenerationStats](interfaces/GenerationStats.md) - [HnswPqOptions](interfaces/HnswPqOptions.md) - [HnswSqOptions](interfaces/HnswSqOptions.md) - [IndexConfig](interfaces/IndexConfig.md) @@ -94,7 +96,9 @@ - [JobInfo](interfaces/JobInfo.md) - [ListNamespacesOptions](interfaces/ListNamespacesOptions.md) - [ListNamespacesResponse](interfaces/ListNamespacesResponse.md) +- [LsmStats](interfaces/LsmStats.md) - [LsmWriteSpec](interfaces/LsmWriteSpec.md) +- [MemtableStats](interfaces/MemtableStats.md) - [MergeBlocker](interfaces/MergeBlocker.md) - [MergeBranchResult](interfaces/MergeBranchResult.md) - [MergePreview](interfaces/MergePreview.md) diff --git a/docs/src/js/interfaces/BucketStats.md b/docs/src/js/interfaces/BucketStats.md new file mode 100644 index 000000000..3f5095672 --- /dev/null +++ b/docs/src/js/interfaces/BucketStats.md @@ -0,0 +1,116 @@ +[**@lancedb/lancedb**](../README.md) • **Docs** + +*** + +[@lancedb/lancedb](../globals.md) / BucketStats + +# Interface: BucketStats + +Live state of one bucket. A table is N buckets on one node; flattening to a +single number hides the one hot bucket that is usually why someone opened +this endpoint. + +## Properties + +### compacting + +```ts +compacting: boolean; +``` + +Whether a pass owns this bucket's compaction latch right now. Says *a* +driver is running, not *whose*, and the latch is held from dispatch — +including while the pass queues for a pod-wide compactor permit. Read it +as "do not pile on", never as "mine is progressing". + +*** + +### currentGeneration + +```ts +currentGeneration: number; +``` + +The generation the active memtable will become. + +*** + +### generations + +```ts +generations: GenerationStats[]; +``` + +Flushed L0 generations not yet merged into the base table. + +*** + +### manifestVersion + +```ts +manifestVersion: number; +``` + +Version of the shard manifest these numbers were read from. + +*** + +### memtables? + +```ts +optional memtables: MemtableStats[]; +``` + +Oldest first, active last. Absent for a `"Sealed"` bucket, whose +in-memory state is torn down. + +*** + +### replayAfterWalEntryPosition + +```ts +replayAfterWalEntryPosition: number; +``` + +WAL position replay resumes from. + +*** + +### shardId + +```ts +shardId: string; +``` + +The shard this bucket writes. + +*** + +### status + +```ts +status: string; +``` + +`"Active"` or `"Sealed"` (drop-table 2PC in flight). + +*** + +### walEntryPositionLastSeen + +```ts +walEntryPositionLastSeen: number; +``` + +Highest WAL position the writer has seen. The difference against +`replayAfterWalEntryPosition` is the WAL lag. + +*** + +### writerEpoch + +```ts +writerEpoch: number; +``` + +Epoch of the writer that currently owns the shard. diff --git a/docs/src/js/interfaces/GenerationStats.md b/docs/src/js/interfaces/GenerationStats.md new file mode 100644 index 000000000..19dd2afda --- /dev/null +++ b/docs/src/js/interfaces/GenerationStats.md @@ -0,0 +1,40 @@ +[**@lancedb/lancedb**](../README.md) • **Docs** + +*** + +[@lancedb/lancedb](../globals.md) / GenerationStats + +# Interface: GenerationStats + +One flushed L0 generation. + +## Properties + +### bytes + +```ts +bytes: number; +``` + +On-disk size of the generation. + +*** + +### generation + +```ts +generation: number; +``` + +The generation number. Increases as memtables are sealed into L0. + +*** + +### rows? + +```ts +optional rows: number; +``` + +Present only when `includeGenerationRows` was requested. Off by default +because each count opens an uncached Lance dataset. diff --git a/docs/src/js/interfaces/LsmStats.md b/docs/src/js/interfaces/LsmStats.md new file mode 100644 index 000000000..76a2f50db --- /dev/null +++ b/docs/src/js/interfaces/LsmStats.md @@ -0,0 +1,22 @@ +[**@lancedb/lancedb**](../README.md) • **Docs** + +*** + +[@lancedb/lancedb](../globals.md) / LsmStats + +# Interface: LsmStats + +Live per-bucket LSM state, as returned by `Table#getLsmStats`. + +Nothing here is derived: sums and differences (total L0 bytes, WAL lag) are +the caller's to compute. + +## Properties + +### buckets + +```ts +buckets: BucketStats[]; +``` + +One entry per bucket backing this table. diff --git a/docs/src/js/interfaces/MemtableStats.md b/docs/src/js/interfaces/MemtableStats.md new file mode 100644 index 000000000..fdc1e4467 --- /dev/null +++ b/docs/src/js/interfaces/MemtableStats.md @@ -0,0 +1,60 @@ +[**@lancedb/lancedb**](../README.md) • **Docs** + +*** + +[@lancedb/lancedb](../globals.md) / MemtableStats + +# Interface: MemtableStats + +One in-memory memtable. + +## Properties + +### batches + +```ts +batches: number; +``` + +Record batches currently buffered. + +*** + +### bytes + +```ts +bytes: number; +``` + +Estimated in-memory size. + +*** + +### generation + +```ts +generation: number; +``` + +The generation this memtable will become once sealed. + +*** + +### indexes + +```ts +indexes: string[]; +``` + +Names of the indexes this memtable carries. An absent name is the whole +answer to "why is my fresh-tier search on that column brute-force". + +*** + +### rows + +```ts +rows: number; +``` + +Rows currently buffered. diff --git a/docs/src/python/python.md b/docs/src/python/python.md index 3dd6f59f4..1d5975dee 100644 --- a/docs/src/python/python.md +++ b/docs/src/python/python.md @@ -52,6 +52,8 @@ listing a storage directory. ::: lancedb.table.Branches +::: lancedb.LsmWriteSpec + ## Expressions Type-safe expression builder for filters and projections. Use these instead diff --git a/java/README.md b/java/README.md index d3560ba4d..c46c8174b 100644 --- a/java/README.md +++ b/java/README.md @@ -29,6 +29,48 @@ LanceNamespace namespaceClient = LanceDbNamespaceClientBuilder.newBuilder() .build(); ``` +## MemWAL LSM write path + +Most table operations reach LanceDB through the `LanceNamespace` above, which is +generated from the Lance Namespace specification. The MemWAL LSM routes are not part +of that specification, so they are issued through a separate client: + +```java +import com.lancedb.LanceDbRestClient; +import com.lancedb.LanceDbTableLsm; +import com.lancedb.LsmWriteSpec; + +LanceDbRestClient client = LanceDbNamespaceClientBuilder.newBuilder() + .apiKey("your_lancedb_cloud_api_key") + .database("your_database_name") + .buildRestClient(); + +LanceDbTableLsm lsm = new LanceDbTableLsm(client, "my_table"); + +// Route future merge_insert upserts through the MemWAL, hash-bucketed by `id`. +lsm.setLsmWriteSpec(LsmWriteSpec.bucket("id", 16)); + +// ... merge_insert traffic ... + +// Converge the fresh tier into the base table. +lsm.checkpointLsm(); + +// Inspect live per-bucket state. +lsm.getLsmStats().ifPresent(stats -> stats.buckets().forEach(bucket -> + System.out.println(bucket.shardId() + ": " + bucket.generations().size() + " L0 generations"))); + +client.close(); +``` + +`maintainedIndexes` is tri-state, and the null default is the opposite of what a Java +reader usually expects: + +| Value | Meaning | +| --- | --- | +| unset (null) | Maintain **every** index the MemWAL can, resolved on install | +| `Collections.emptyList()` | Maintain **none** | +| `Arrays.asList("id_idx")` | Maintain exactly those | + ## Development Build: diff --git a/java/lancedb-core/pom.xml b/java/lancedb-core/pom.xml index fb0d2618f..60c1549e3 100644 --- a/java/lancedb-core/pom.xml +++ b/java/lancedb-core/pom.xml @@ -33,6 +33,20 @@ arrow-memory-netty + + + org.apache.httpcomponents.client5 + httpclient5 + 5.2.1 + + + + com.fasterxml.jackson.core + jackson-databind + 2.17.1 + + org.junit.jupiter junit-jupiter diff --git a/java/lancedb-core/src/main/java/com/lancedb/BucketStats.java b/java/lancedb-core/src/main/java/com/lancedb/BucketStats.java new file mode 100644 index 000000000..2a8060c5d --- /dev/null +++ b/java/lancedb-core/src/main/java/com/lancedb/BucketStats.java @@ -0,0 +1,194 @@ +/* + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package com.lancedb; + +import com.fasterxml.jackson.databind.JsonNode; + +import java.util.ArrayList; +import java.util.Collections; +import java.util.List; +import java.util.Optional; +import java.util.OptionalLong; + +/** + * Live state of one bucket. A table is N buckets on one node; flattening to a single number hides + * the one hot bucket that is usually why someone opened this endpoint. + */ +public class BucketStats { + private static final String CONTEXT = "bucket stats"; + + private final String shardId; + private final String status; + private final long writerEpoch; + private final long manifestVersion; + private final long currentGeneration; + private final long replayAfterWalEntryPosition; + private final long walEntryPositionLastSeen; + private final List generations; + private final boolean compacting; + private final List memtables; + + BucketStats( + String shardId, + String status, + long writerEpoch, + long manifestVersion, + long currentGeneration, + long replayAfterWalEntryPosition, + long walEntryPositionLastSeen, + List generations, + boolean compacting, + List memtables) { + this.shardId = shardId; + this.status = status; + this.writerEpoch = writerEpoch; + this.manifestVersion = manifestVersion; + this.currentGeneration = currentGeneration; + this.replayAfterWalEntryPosition = replayAfterWalEntryPosition; + this.walEntryPositionLastSeen = walEntryPositionLastSeen; + this.generations = Collections.unmodifiableList(generations); + this.compacting = compacting; + this.memtables = memtables == null ? null : Collections.unmodifiableList(memtables); + } + + /** The shard this bucket writes. */ + public String shardId() { + return shardId; + } + + /** {@code "Active"} or {@code "Sealed"} (drop-table 2PC in flight). */ + public String status() { + return status; + } + + /** Epoch of the writer that currently owns the shard. */ + public long writerEpoch() { + return writerEpoch; + } + + /** Version of the shard manifest these numbers were read from. */ + public long manifestVersion() { + return manifestVersion; + } + + /** The generation the active memtable will become. */ + public long currentGeneration() { + return currentGeneration; + } + + /** WAL position replay resumes from. */ + public long replayAfterWalEntryPosition() { + return replayAfterWalEntryPosition; + } + + /** + * Highest WAL position the writer has seen. The difference against {@link + * #replayAfterWalEntryPosition()} is the WAL lag. + */ + public long walEntryPositionLastSeen() { + return walEntryPositionLastSeen; + } + + /** Flushed L0 generations not yet merged into the base table. */ + public List generations() { + return generations; + } + + /** + * Whether a pass owns this bucket's compaction latch right now. Says a driver is + * running, not whose, and the latch is held from dispatch — including while the pass + * queues for a pod-wide compactor permit. Read it as "do not pile on", never as "mine is + * progressing". + */ + public boolean compacting() { + return compacting; + } + + /** Oldest first, active last. Empty for a {@code "Sealed"} bucket, whose state is torn down. */ + public Optional> memtables() { + return Optional.ofNullable(memtables); + } + + /** The newest flushed generation, or empty when L0 is empty. */ + OptionalLong newestGeneration() { + OptionalLong newest = OptionalLong.empty(); + for (GenerationStats generation : generations) { + if (!newest.isPresent() || generation.generation() > newest.getAsLong()) { + newest = OptionalLong.of(generation.generation()); + } + } + return newest; + } + + /** + * How many generations at or below {@code target} are still in L0. + * + *

A count, not a boolean: one pass drains a bounded prefix rather than the whole target set, + * so a boolean would read as "no progress" for every pass but the last. Compaction drains + * oldest-first, so this decreases monotonically. + */ + long outstandingGenerations(long target) { + long count = 0; + for (GenerationStats generation : generations) { + if (generation.generation() <= target) { + count++; + } + } + return count; + } + + static BucketStats fromJson(JsonNode node) { + JsonFields.requiredObject(node, CONTEXT); + List generations = new ArrayList(); + for (JsonNode generation : JsonFields.requiredArray(node, "generations", CONTEXT)) { + generations.add(GenerationStats.fromJson(generation)); + } + + JsonNode memtablesNode = JsonFields.optionalArray(node, "memtables", CONTEXT); + List memtables = null; + if (memtablesNode != null) { + memtables = new ArrayList(); + for (JsonNode memtable : memtablesNode) { + memtables.add(MemtableStats.fromJson(memtable)); + } + } + + return new BucketStats( + JsonFields.requiredText(node, "shard_id", CONTEXT), + JsonFields.requiredText(node, "status", CONTEXT), + JsonFields.requiredLong(node, "writer_epoch", CONTEXT), + JsonFields.requiredLong(node, "manifest_version", CONTEXT), + JsonFields.requiredLong(node, "current_generation", CONTEXT), + JsonFields.requiredLong(node, "replay_after_wal_entry_position", CONTEXT), + JsonFields.requiredLong(node, "wal_entry_position_last_seen", CONTEXT), + generations, + JsonFields.requiredBoolean(node, "compacting", CONTEXT), + memtables); + } + + @Override + public String toString() { + return "BucketStats{shardId=" + + shardId + + ", status=" + + status + + ", currentGeneration=" + + currentGeneration + + ", generations=" + + generations + + ", compacting=" + + compacting + + "}"; + } +} diff --git a/java/lancedb-core/src/main/java/com/lancedb/GenerationStats.java b/java/lancedb-core/src/main/java/com/lancedb/GenerationStats.java new file mode 100644 index 000000000..12222407c --- /dev/null +++ b/java/lancedb-core/src/main/java/com/lancedb/GenerationStats.java @@ -0,0 +1,64 @@ +/* + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package com.lancedb; + +import com.fasterxml.jackson.databind.JsonNode; + +import java.util.OptionalLong; + +/** One flushed L0 generation. */ +public class GenerationStats { + private static final String CONTEXT = "generation stats"; + + private final long generation; + private final long bytes; + private final Long rows; + + GenerationStats(long generation, long bytes, Long rows) { + this.generation = generation; + this.bytes = bytes; + this.rows = rows; + } + + /** The generation number. Increases as memtables are sealed into L0. */ + public long generation() { + return generation; + } + + /** On-disk size of the generation. */ + public long bytes() { + return bytes; + } + + /** + * Rows in this generation, present only when {@code includeGenerationRows} was requested. Off by + * default because each count opens an uncached Lance dataset. + */ + public OptionalLong rows() { + return rows == null ? OptionalLong.empty() : OptionalLong.of(rows); + } + + static GenerationStats fromJson(JsonNode node) { + JsonFields.requiredObject(node, CONTEXT); + return new GenerationStats( + JsonFields.requiredLong(node, "generation", CONTEXT), + JsonFields.requiredLong(node, "bytes", CONTEXT), + JsonFields.optionalLong(node, "rows", CONTEXT)); + } + + @Override + public String toString() { + return "GenerationStats{generation=" + generation + ", bytes=" + bytes + ", rows=" + rows + "}"; + } +} diff --git a/java/lancedb-core/src/main/java/com/lancedb/JsonFields.java b/java/lancedb-core/src/main/java/com/lancedb/JsonFields.java new file mode 100644 index 000000000..b78e2411a --- /dev/null +++ b/java/lancedb-core/src/main/java/com/lancedb/JsonFields.java @@ -0,0 +1,109 @@ +/* + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package com.lancedb; + +import com.fasterxml.jackson.databind.JsonNode; + +/** + * Strict readers for decoding LanceDB JSON responses. + * + *

Every reader fails closed: a missing, null, or wrong-typed field throws rather than + * defaulting. That mirrors the serde decoding the Rust client applies to the same payloads in + * {@code rust/lancedb/src/table/lsm_stats.rs}, where a required field has no default and a + * malformed response is an error rather than a zero. + * + *

The alternative — Jackson's {@code path()}, which yields a missing node that reads as an empty + * array or a zero — is unsafe here because {@link LanceDbTableLsm#checkpointLsm()} decides + * convergence from these numbers. A defaulted {@code generations} array is indistinguishable from a + * drained one, so a malformed response would report a checkpoint that never happened. + */ +final class JsonFields { + private JsonFields() {} + + /** The node itself, once confirmed to be a JSON object. */ + static JsonNode requiredObject(JsonNode node, String context) { + if (node == null || !node.isObject()) { + throw new IllegalStateException(context + " is not a JSON object: " + node); + } + return node; + } + + static String requiredText(JsonNode owner, String field, String context) { + JsonNode value = required(owner, field, context); + if (!value.isTextual()) { + throw new IllegalStateException(fieldIs(context, field, "a string", value)); + } + return value.asText(); + } + + static long requiredLong(JsonNode owner, String field, String context) { + JsonNode value = required(owner, field, context); + if (!value.isIntegralNumber()) { + throw new IllegalStateException(fieldIs(context, field, "an integer", value)); + } + return value.asLong(); + } + + static boolean requiredBoolean(JsonNode owner, String field, String context) { + JsonNode value = required(owner, field, context); + if (!value.isBoolean()) { + throw new IllegalStateException(fieldIs(context, field, "a boolean", value)); + } + return value.asBoolean(); + } + + static JsonNode requiredArray(JsonNode owner, String field, String context) { + JsonNode value = required(owner, field, context); + if (!value.isArray()) { + throw new IllegalStateException(fieldIs(context, field, "an array", value)); + } + return value; + } + + /** Null when the field is absent or JSON null, mirroring a serde {@code Option}. */ + static Long optionalLong(JsonNode owner, String field, String context) { + JsonNode value = owner.get(field); + if (value == null || value.isNull()) { + return null; + } + if (!value.isIntegralNumber()) { + throw new IllegalStateException(fieldIs(context, field, "an integer", value)); + } + return value.asLong(); + } + + /** Null when the field is absent or JSON null, mirroring a serde {@code Option}. */ + static JsonNode optionalArray(JsonNode owner, String field, String context) { + JsonNode value = owner.get(field); + if (value == null || value.isNull()) { + return null; + } + if (!value.isArray()) { + throw new IllegalStateException(fieldIs(context, field, "an array", value)); + } + return value; + } + + private static JsonNode required(JsonNode owner, String field, String context) { + JsonNode value = owner.get(field); + if (value == null || value.isNull()) { + throw new IllegalStateException(context + " is missing required field '" + field + "'"); + } + return value; + } + + private static String fieldIs(String context, String field, String expected, JsonNode value) { + return context + " field '" + field + "' is not " + expected + ": " + value; + } +} diff --git a/java/lancedb-core/src/main/java/com/lancedb/LanceDbNamespaceClientBuilder.java b/java/lancedb-core/src/main/java/com/lancedb/LanceDbNamespaceClientBuilder.java index 5e31aaaa1..da241dfd5 100644 --- a/java/lancedb-core/src/main/java/com/lancedb/LanceDbNamespaceClientBuilder.java +++ b/java/lancedb-core/src/main/java/com/lancedb/LanceDbNamespaceClientBuilder.java @@ -136,29 +136,48 @@ public class LanceDbNamespaceClientBuilder { * @throws IllegalStateException if required parameters are missing */ public LanceNamespace build() { - // Validate required fields + validate(); + + // Build configuration map + Map config = new HashMap<>(additionalConfig); + config.put("header.x-lancedb-database", database); + config.put("header.x-api-key", apiKey); + config.put("uri", resolveUri()); + + return LanceNamespace.connect("rest", config, null); + } + + /** + * Build a {@link LanceDbRestClient} for the same endpoint. + * + *

Needed only for LanceDB routes that the Lance Namespace specification does not cover — the + * MemWAL LSM write path, reached through {@link LanceDbTableLsm}. Every other table operation + * belongs on the {@link LanceNamespace} from {@link #build()}. + * + *

The returned client owns an HTTP connection pool; close it when you are done with it. + * + * @return A configured LanceDbRestClient + * @throws IllegalStateException if required parameters are missing + */ + public LanceDbRestClient buildRestClient() { + validate(); + return new LanceDbRestClient(resolveUri(), apiKey, database); + } + + private void validate() { if (apiKey == null) { throw new IllegalStateException("API key is required"); } if (database == null) { throw new IllegalStateException("Database is required"); } + } - // Build configuration map - Map config = new HashMap<>(additionalConfig); - config.put("header.x-lancedb-database", database); - config.put("header.x-api-key", apiKey); - - // Determine base URL - String uri; + /** The custom endpoint when set, else the LanceDB Cloud URL for this database and region. */ + private String resolveUri() { if (endpoint.isPresent()) { - uri = endpoint.get(); - } else { - String effectiveRegion = region.orElse(DEFAULT_REGION); - uri = String.format(CLOUD_URL_PATTERN, database, effectiveRegion); + return endpoint.get(); } - config.put("uri", uri); - - return LanceNamespace.connect("rest", config, null); + return String.format(CLOUD_URL_PATTERN, database, region.orElse(DEFAULT_REGION)); } } diff --git a/java/lancedb-core/src/main/java/com/lancedb/LanceDbRestClient.java b/java/lancedb-core/src/main/java/com/lancedb/LanceDbRestClient.java new file mode 100644 index 000000000..baafbb9df --- /dev/null +++ b/java/lancedb-core/src/main/java/com/lancedb/LanceDbRestClient.java @@ -0,0 +1,119 @@ +/* + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package com.lancedb; + +import com.fasterxml.jackson.databind.JsonNode; +import com.fasterxml.jackson.databind.ObjectMapper; +import org.apache.hc.client5.http.classic.methods.HttpPost; +import org.apache.hc.client5.http.impl.classic.CloseableHttpClient; +import org.apache.hc.client5.http.impl.classic.HttpClients; +import org.apache.hc.core5.http.ContentType; +import org.apache.hc.core5.http.io.entity.EntityUtils; +import org.apache.hc.core5.http.io.entity.StringEntity; + +import java.io.Closeable; +import java.io.IOException; +import java.io.UncheckedIOException; + +/** + * Minimal HTTP client for LanceDB Cloud and Enterprise routes that the Lance Namespace + * specification does not cover. + * + *

Most table operations reach LanceDB through {@link org.lance.namespace.LanceNamespace}, which + * is generated from the namespace spec. A handful of routes — the MemWAL LSM write path in + * particular — are served by the same endpoint but are not part of that spec, so they are issued + * directly here. See {@link LanceDbTableLsm}. + * + *

Obtain one from {@link LanceDbNamespaceClientBuilder#buildRestClient()}. + */ +public class LanceDbRestClient implements Closeable { + private static final ObjectMapper MAPPER = new ObjectMapper(); + + private final String baseUri; + private final String apiKey; + private final String database; + private final CloseableHttpClient http; + + LanceDbRestClient(String baseUri, String apiKey, String database) { + this.baseUri = baseUri.endsWith("/") ? baseUri.substring(0, baseUri.length() - 1) : baseUri; + this.apiKey = apiKey; + this.database = database; + // Automatic retries off, deliberately. The default strategy retries 429 and 503 — + // exactly the two statuses LanceDbTableLsm.checkpointLsm() acts on — which would + // silently double its explicit retry budget and would also retry compact_lsm in + // place, where the loop is designed to fall through to a fresh stats poll instead. + // The checkpoint loop owns the 421/429/503 transitions; the transport must not. + this.http = HttpClients.custom().disableAutomaticRetries().build(); + } + + /** + * POST {@code path}, sending {@code body} as JSON when it is non-null. + * + * @param path Absolute request path, beginning with {@code /}. + * @param body Object to serialize as the request body, or null to send no body. + * @return The parsed response body, or null when the response carried no content. + * @throws HttpException if the server returned a non-2xx status. + */ + public JsonNode post(String path, Object body) { + HttpPost request = new HttpPost(baseUri + path); + request.setHeader("x-api-key", apiKey); + request.setHeader("x-lancedb-database", database); + try { + if (body != null) { + request.setEntity( + new StringEntity(MAPPER.writeValueAsString(body), ContentType.APPLICATION_JSON)); + } + return http.execute( + request, + response -> { + String text = + response.getEntity() == null ? "" : EntityUtils.toString(response.getEntity()); + int status = response.getCode(); + if (status < 200 || status >= 300) { + throw new HttpException(status, "LanceDB request to " + path + " failed: " + text); + } + return text.isEmpty() ? null : MAPPER.readTree(text); + }); + } catch (IOException e) { + throw new UncheckedIOException("LanceDB request to " + path + " failed", e); + } + } + + @Override + public void close() throws IOException { + http.close(); + } + + /** + * A non-2xx response. + * + *

The status is exposed because callers act on it: {@link LanceDbTableLsm#checkpointLsm()} + * treats 429 and 503 as retryable and 421 as a lost node claim. + */ + public static class HttpException extends RuntimeException { + private static final long serialVersionUID = 1L; + + private final int statusCode; + + public HttpException(int statusCode, String message) { + super(message); + this.statusCode = statusCode; + } + + /** The HTTP status the failed response carried. */ + public int statusCode() { + return statusCode; + } + } +} diff --git a/java/lancedb-core/src/main/java/com/lancedb/LanceDbTableLsm.java b/java/lancedb-core/src/main/java/com/lancedb/LanceDbTableLsm.java new file mode 100644 index 000000000..23b18199e --- /dev/null +++ b/java/lancedb-core/src/main/java/com/lancedb/LanceDbTableLsm.java @@ -0,0 +1,394 @@ +/* + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package com.lancedb; + +import com.fasterxml.jackson.databind.JsonNode; + +import java.util.HashMap; +import java.util.LinkedHashMap; +import java.util.Map; +import java.util.Optional; +import java.util.OptionalLong; + +/** + * The MemWAL LSM write path for one LanceDB Cloud or Enterprise table. + * + *

Installing an {@link LsmWriteSpec} routes {@code mergeInsert} upserts through Lance's MemWAL — + * an LSM-style append — instead of the standard merge path. Rows land in an in-memory memtable, + * seal into L0 generations, and are merged into the base table by compaction. + * + *

These routes are not part of the Lance Namespace specification, so they are issued directly + * rather than through {@link org.lance.namespace.LanceNamespace}. + * + *

{@code
+ * LanceDbRestClient client = LanceDbNamespaceClientBuilder.newBuilder()
+ *     .apiKey("your_lancedb_cloud_api_key")
+ *     .database("your_database_name")
+ *     .buildRestClient();
+ *
+ * LanceDbTableLsm lsm = new LanceDbTableLsm(client, "my_table");
+ * lsm.setLsmWriteSpec(LsmWriteSpec.bucket("id", 16));
+ * // ... merge_insert traffic ...
+ * lsm.checkpointLsm();
+ * }
+ */ +public class LanceDbTableLsm { + + /** + * Interval between {@code get_lsm_stats} polls during a checkpoint. One interval is roughly one + * compaction pass, the granularity at which the answer can change. + */ + private static final long POLL_INTERVAL_MS = 5_000L; + + /** + * Cap on re-issues from {@code flushLsm} after a 421, so a crash-looping node cannot turn flush → + * compact → 421 → flush into a spin. + * + *

Deliberately not shared with {@link #MAX_RETRIES}: a claim that keeps evaporating is a + * broken node, while contention is routine and wants a real budget. + */ + private static final int MAX_REISSUES = 3; + + /** + * Retryable faults tolerated on a single request, reset on every success — scattered + * contention across a long checkpoint must not accumulate toward a cap. + */ + private static final int MAX_RETRIES = 8; + + private static final long RETRY_BACKOFF_BASE_MS = 100L; + private static final long RETRY_BACKOFF_MAX_MS = 5_000L; + + private final LanceDbRestClient client; + private final String tableIdentifier; + + /** + * Bind the LSM routes for one table. + * + * @param client Transport for the LanceDB endpoint. + * @param tableIdentifier The table's full identifier, {@code $}-delimited when it sits inside a + * namespace, such as {@code analytics$events}. + */ + public LanceDbTableLsm(LanceDbRestClient client, String tableIdentifier) { + if (client == null) { + throw new IllegalArgumentException("Client cannot be null"); + } + if (tableIdentifier == null || tableIdentifier.trim().isEmpty()) { + throw new IllegalArgumentException("Table identifier cannot be null or empty"); + } + this.client = client; + this.tableIdentifier = tableIdentifier; + } + + /** + * Install an {@link LsmWriteSpec} on this table, selecting the MemWAL LSM write path for future + * {@code mergeInsert} calls. + * + *

All variants require the table to have an unenforced primary key; bucket sharding + * additionally requires it to be the single column being bucketed. + */ + public void setLsmWriteSpec(LsmWriteSpec spec) { + if (spec == null) { + throw new IllegalArgumentException("Spec cannot be null"); + } + client.post(route("set_lsm_write_spec"), spec.toRequestBody()); + } + + /** + * Remove the {@link LsmWriteSpec} from this table, reverting to the standard {@code mergeInsert} + * write path. + * + *

Errors if no spec is currently set. + */ + public void unsetLsmWriteSpec() { + client.post(route("unset_lsm_write_spec"), null); + } + + /** + * Read the {@link LsmWriteSpec} currently installed on this table. + * + *

Empty when the LSM write path is not enabled. The returned spec mirrors what was installed, + * except that {@link LsmWriteSpec#maintainedIndexes()} always reports the concrete list resolved + * when the spec was set — a null selection never round-trips. + */ + public Optional getLsmWriteSpec() { + JsonNode response = client.post(route("get_lsm_write_spec"), null); + if (response == null || !response.hasNonNull("lsm_write_spec")) { + return Optional.empty(); + } + return Optional.of(LsmWriteSpec.fromJson(response.get("lsm_write_spec"))); + } + + /** + * Seal every bucket's active memtable into a new L0 generation. + * + *

Returns once the seal is committed. Sealing an empty memtable is a no-op, so this is safe to + * call repeatedly. + */ + public void flushLsm() { + client.post(route("flush_lsm"), null); + } + + /** + * Trigger a background L0 → base compaction pass per bucket. + * + *

Returns once the passes are dispatched, not once they finish — watch {@link + * #getLsmStats}, or use {@link #checkpointLsm} to wait for convergence. + */ + public void compactLsm() { + client.post(route("compact_lsm"), null); + } + + /** + * Read live per-bucket LSM state. + * + *

Answers "how far behind is my fresh tier", "which bucket is hot", and "why is my fresh-tier + * vector search brute-force". Mutates no table state. + * + *

Empty only when the LSM write path is not enabled — that is, when the server sends an absent + * or null {@code lsm_stats}. A stats object that is present is decoded strictly, and a malformed + * one throws rather than decoding to something empty, because {@link #checkpointLsm} reads + * convergence out of these numbers and cannot tell a defaulted array from a drained one. + * + * @param includeGenerationRows Also count rows per L0 generation. Off by default because each + * count opens an uncached Lance dataset. + * @throws IllegalStateException if the response is absent or does not decode. + */ + public Optional getLsmStats(boolean includeGenerationRows) { + Map body = new LinkedHashMap(); + body.put("include_generation_rows", includeGenerationRows); + JsonNode response = client.post(route("get_lsm_stats"), body); + if (response == null) { + throw new IllegalStateException("get_lsm_stats returned an empty response body"); + } + JsonNode stats = response.get("lsm_stats"); + if (stats == null || stats.isNull()) { + return Optional.empty(); + } + return Optional.of(LsmStats.fromJson(stats)); + } + + /** Equivalent to {@code getLsmStats(false)}. */ + public Optional getLsmStats() { + return getLsmStats(false); + } + + /** + * Converge this table's LSM write path into its base table. + * + *

Seals once, fixes a target watermark from the resulting L0, then triggers compaction and + * polls until that L0 is gone. The target set is fixed at the start, so generations created + * during the checkpoint are ignored — that is what lets it terminate under write load, + * and what makes it best-effort: it converges the fresh tier as of some instant. Idempotent, + * abandonable at any point, safe on a cadence. + * + *

The loop runs here, not on the server: {@link #compactLsm} dispatches a pass and returns, so + * nothing holds a socket and a client can vanish mid-operation with nothing to reconcile. + * Completion is read from generation numbers in the shard manifest — durable state, unlike a + * count in a compact response, which a concurrent write invalidates. + * + *

No liveness bound — the caller owns the deadline. The compactor pool is shared across + * tables, so a checkpoint queued behind unrelated work looks exactly like one that is merging. + */ + public void checkpointLsm() { + for (int reissue = 0; reissue <= MAX_REISSUES; reissue++) { + // The seal turns everything written before this call into a generation, so the + // watermark has to be read after it. Idempotent: sealing an empty memtable is a + // no-op, so a re-issue does not churn empty generations. + if (issueVoid(this::flushLsm)) { + backoff(reissue); + continue; + } + + Attempt> stats = issue(() -> getLsmStats(false)); + if (stats.lostClaim) { + backoff(reissue); + continue; + } + if (!stats.value.isPresent()) { + // Not WAL-backed; flushLsm would have errored first but for a race. + return; + } + + Map targets = newestGenerations(stats.value.get()); + if (targets.isEmpty()) { + return; + } + + if (drainToTargets(targets)) { + return; + } + backoff(reissue); + } + throw new IllegalStateException( + "checkpointLsm: the owning node kept losing its claim; re-issued from flush the maximum " + + "number of times"); + } + + /** + * Trigger and poll until no bucket holds a generation at or below its target. + * + * @return true when the drain finished, false when the table needs re-claiming from flush. + */ + private boolean drainToTargets(Map targets) { + while (true) { + Attempt> stats = issue(() -> getLsmStats(false)); + if (stats.lostClaim) { + return false; + } + if (!stats.value.isPresent()) { + return true; + } + + // `compacting` is the bucket's compaction latch, held from dispatch until the pass + // ends — including while it waits on a pod-wide permit. So it answers one question + // only: do not pile on. Buckets with nothing outstanding are skipped, not counted + // as idle. + long outstanding = 0; + boolean allCompacting = true; + for (BucketStats bucket : stats.value.get().buckets()) { + Long target = targets.get(bucket.shardId()); + if (target == null) { + continue; + } + long remaining = bucket.outstandingGenerations(target); + if (remaining > 0) { + outstanding += remaining; + allCompacting &= bucket.compacting(); + } + } + if (outstanding == 0) { + return true; + } + + if (!allCompacting) { + try { + compactLsm(); + } catch (LanceDbRestClient.HttpException e) { + if (isLostClaim(e)) { + return false; + } + if (!isRetryable(e)) { + throw e; + } + // A 429 here means the server could latch no bucket at all, which the poll + // above already handles. Not retried in place: the latch it would contend for + // is the one doing the work, so fall through and re-read — POLL_INTERVAL_MS is + // the backoff. + } + } + sleep(POLL_INTERVAL_MS); + } + } + + /** The newest generation held by each bucket, skipping buckets holding none. */ + private static Map newestGenerations(LsmStats stats) { + Map targets = new HashMap(); + for (BucketStats bucket : stats.buckets()) { + OptionalLong newest = bucket.newestGeneration(); + if (newest.isPresent()) { + targets.put(bucket.shardId(), newest.getAsLong()); + } + } + return targets; + } + + /** + * 429 (latch held, pool saturated, or the pod replaying its WAL) and 503 (a draining node, or a + * proxy between here and it). + */ + private static boolean isRetryable(LanceDbRestClient.HttpException e) { + return e.statusCode() == 429 || e.statusCode() == 503; + } + + /** + * 421: the owning node holds no claim. Only {@code flush} re-claims and replays, so this cannot + * be retried in place — the caller has to start over. + */ + private static boolean isLostClaim(LanceDbRestClient.HttpException e) { + return e.statusCode() == 421; + } + + /** + * Issue one LSM request, retrying in place while the fault is retryable. + * + *

The two recoverable faults have separate budgets: contention clears on its own and retries + * here against {@link #MAX_RETRIES}, while a 421 needs {@code flush} to re-claim, which only the + * caller can drive. + * + *

An exhausted budget propagates the last error as itself rather than a synthesized one — "429 + * after nine tries" beats "checkpoint failed". + */ + private static Attempt issue(Call call) { + int retries = 0; + while (true) { + try { + return new Attempt(call.run(), false); + } catch (LanceDbRestClient.HttpException e) { + if (isLostClaim(e)) { + return new Attempt(null, true); + } + if (!isRetryable(e) || retries >= MAX_RETRIES) { + throw e; + } + backoff(retries); + retries++; + } + } + } + + /** {@link #issue} for a call with no return value. Returns true when the claim was lost. */ + private static boolean issueVoid(Runnable call) { + return issue( + () -> { + call.run(); + return Boolean.TRUE; + }) + .lostClaim; + } + + /** Sleep before re-issuing a retryable request. Doubles up to {@link #RETRY_BACKOFF_MAX_MS}. */ + private static void backoff(int attempt) { + long delay = RETRY_BACKOFF_BASE_MS << Math.min(attempt, 8); + sleep(Math.min(delay, RETRY_BACKOFF_MAX_MS)); + } + + private static void sleep(long millis) { + try { + Thread.sleep(millis); + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + throw new IllegalStateException("Interrupted while waiting on the LSM checkpoint", e); + } + } + + private String route(String operation) { + return "/v1/table/" + tableIdentifier + "/" + operation + "/"; + } + + /** What one LSM request produced: its value, or word that the owning node holds no claim. */ + private static final class Attempt { + private final T value; + private final boolean lostClaim; + + private Attempt(T value, boolean lostClaim) { + this.value = value; + this.lostClaim = lostClaim; + } + } + + @FunctionalInterface + private interface Call { + T run(); + } +} diff --git a/java/lancedb-core/src/main/java/com/lancedb/LsmStats.java b/java/lancedb-core/src/main/java/com/lancedb/LsmStats.java new file mode 100644 index 000000000..3496ebc96 --- /dev/null +++ b/java/lancedb-core/src/main/java/com/lancedb/LsmStats.java @@ -0,0 +1,56 @@ +/* + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package com.lancedb; + +import com.fasterxml.jackson.databind.JsonNode; + +import java.util.ArrayList; +import java.util.Collections; +import java.util.List; + +/** + * Live per-bucket LSM state, as returned by {@link LanceDbTableLsm#getLsmStats()}. + * + *

Nothing here is derived: sums and differences (total L0 bytes, WAL lag) are the caller's to + * compute. There is no "LSM is off" shape — that case is an empty {@link java.util.Optional}, + * because a stats object of zeros would read as measurements. + */ +public class LsmStats { + private static final String CONTEXT = "lsm stats"; + + private final List buckets; + + LsmStats(List buckets) { + this.buckets = Collections.unmodifiableList(buckets); + } + + /** One entry per bucket. */ + public List buckets() { + return buckets; + } + + static LsmStats fromJson(JsonNode node) { + JsonFields.requiredObject(node, CONTEXT); + List buckets = new ArrayList(); + for (JsonNode bucket : JsonFields.requiredArray(node, "buckets", CONTEXT)) { + buckets.add(BucketStats.fromJson(bucket)); + } + return new LsmStats(buckets); + } + + @Override + public String toString() { + return "LsmStats{buckets=" + buckets + "}"; + } +} diff --git a/java/lancedb-core/src/main/java/com/lancedb/LsmWriteSpec.java b/java/lancedb-core/src/main/java/com/lancedb/LsmWriteSpec.java new file mode 100644 index 000000000..da0966910 --- /dev/null +++ b/java/lancedb-core/src/main/java/com/lancedb/LsmWriteSpec.java @@ -0,0 +1,260 @@ +/* + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package com.lancedb; + +import com.fasterxml.jackson.databind.JsonNode; + +import java.util.ArrayList; +import java.util.Collections; +import java.util.HashMap; +import java.util.LinkedHashMap; +import java.util.List; +import java.util.Map; + +/** + * Specification selecting Lance's MemWAL LSM-style write path for {@code mergeInsert}. + * + *

Construct via {@link #bucket}, {@link #identity}, or {@link #unsharded}, then optionally chain + * {@link #withMaintainedIndexes} and {@link #withWriterConfigDefaults}. Install it with {@link + * LanceDbTableLsm#setLsmWriteSpec} and remove it with {@link LanceDbTableLsm#unsetLsmWriteSpec}. + * + *

This is deliberately not {@code org.lance.memwal.InitializeMemWalParams}. That type is Lance's + * own, and its maintained-index default is the opposite of this one: it defaults to maintaining + * nothing, while a fresh spec here maintains every index. It also cannot express + * the null that asks the server to resolve the set. + */ +public class LsmWriteSpec { + + /** How writes are routed to MemWAL shards. */ + public enum Sharding { + /** Hash-bucket writes by a scalar column. */ + BUCKET("bucket"), + /** Shard by the raw value of a scalar column. */ + IDENTITY("identity"), + /** Route every write to a single shard. */ + UNSHARDED("unsharded"); + + private final String wireName; + + Sharding(String wireName) { + this.wireName = wireName; + } + + String wireName() { + return wireName; + } + + static Sharding fromWireName(String name) { + for (Sharding s : values()) { + if (s.wireName.equals(name)) { + return s; + } + } + throw new IllegalArgumentException("Unknown sharding mode: " + name); + } + } + + private final Sharding sharding; + private final String column; + private final Integer numBuckets; + private final List maintainedIndexes; + private final Map writerConfigDefaults; + + private LsmWriteSpec( + Sharding sharding, + String column, + Integer numBuckets, + List maintainedIndexes, + Map writerConfigDefaults) { + this.sharding = sharding; + this.column = column; + this.numBuckets = numBuckets; + this.maintainedIndexes = maintainedIndexes; + this.writerConfigDefaults = writerConfigDefaults; + } + + /** + * Hash-bucket sharding by a scalar column, maintaining every index on the table. + * + *

Iceberg-compatible Murmur3-x86-32 (seed 0) is used, so each row's {@code bucket(column, + * numBuckets)} value is stable across processes. + * + * @param column A non-nested column with a supported scalar type. + * @param numBuckets The number of buckets, in {@code [1, 1024]}. + */ + public static LsmWriteSpec bucket(String column, int numBuckets) { + if (column == null || column.trim().isEmpty()) { + throw new IllegalArgumentException("Column cannot be null or empty"); + } + return new LsmWriteSpec( + Sharding.BUCKET, column, numBuckets, null, new HashMap()); + } + + /** + * Identity sharding — shard by the raw value of {@code column} — maintaining every index on the + * table. + * + *

{@code column} must be a deterministic function of the unenforced primary key: every row + * with a given primary key must always produce the same {@code column} value, or upserts of that + * key can land in different shards and a stale version can win. + */ + public static LsmWriteSpec identity(String column) { + if (column == null || column.trim().isEmpty()) { + throw new IllegalArgumentException("Column cannot be null or empty"); + } + return new LsmWriteSpec(Sharding.IDENTITY, column, null, null, new HashMap()); + } + + /** No sharding — every write goes to a single MemWAL shard — maintaining every index. */ + public static LsmWriteSpec unsharded() { + return new LsmWriteSpec(Sharding.UNSHARDED, null, null, null, new HashMap()); + } + + /** + * Set the indexes the MemWAL keeps up to date as rows are appended. + * + *

Pass {@code null} — the default for a fresh spec — to maintain every index the MemWAL can, + * resolved when the spec is installed. That is a snapshot: indexes created later are not + * maintained until the spec is unset and set again. Pass an empty list to maintain none. + * + *

Note that {@code null} and the empty list mean opposite things here. + */ + public LsmWriteSpec withMaintainedIndexes(List maintainedIndexes) { + return new LsmWriteSpec( + sharding, + column, + numBuckets, + maintainedIndexes == null ? null : new ArrayList(maintainedIndexes), + writerConfigDefaults); + } + + /** + * Set default {@code ShardWriter} configuration recorded in the MemWAL index. + * + *

A sparse override map — only the keys you set are recorded. Recognized keys include {@code + * durable_write}, {@code max_wal_buffer_size}, {@code max_memtable_size}, {@code + * max_memtable_rows}, {@code max_memtable_batches}, {@code manifest_scan_batch_size}, {@code + * max_unflushed_memtable_bytes}, and {@code enable_memtable}. Duration knobs carry an {@code _ms} + * suffix, such as {@code max_wal_flush_interval_ms}. + */ + public LsmWriteSpec withWriterConfigDefaults(Map writerConfigDefaults) { + if (writerConfigDefaults == null) { + throw new IllegalArgumentException("writerConfigDefaults cannot be null"); + } + return new LsmWriteSpec( + sharding, + column, + numBuckets, + maintainedIndexes, + new HashMap(writerConfigDefaults)); + } + + /** How writes are routed to shards. */ + public Sharding sharding() { + return sharding; + } + + /** The sharding column for {@link Sharding#BUCKET} and {@link Sharding#IDENTITY}, else null. */ + public String column() { + return column; + } + + /** The bucket count for {@link Sharding#BUCKET}, else null. */ + public Integer numBuckets() { + return numBuckets; + } + + /** + * The indexes the MemWAL maintains, or null to have the server resolve every maintainable index + * on install. An empty list means none. + */ + public List maintainedIndexes() { + return maintainedIndexes == null ? null : Collections.unmodifiableList(maintainedIndexes); + } + + /** Default {@code ShardWriter} configuration recorded in the MemWAL index. */ + public Map writerConfigDefaults() { + return Collections.unmodifiableMap(writerConfigDefaults); + } + + /** Render this spec as the {@code set_lsm_write_spec} request body. */ + Map toRequestBody() { + Map shardingBody = new LinkedHashMap(); + shardingBody.put("mode", sharding.wireName()); + if (column != null) { + shardingBody.put("column", column); + } + if (numBuckets != null) { + shardingBody.put("num_buckets", numBuckets); + } + + Map body = new LinkedHashMap(); + body.put("sharding", shardingBody); + // Null is meaningful: it asks the server to resolve every maintainable index. + body.put("maintained_indexes", maintainedIndexes); + body.put("writer_config_defaults", writerConfigDefaults); + return body; + } + + /** + * Rebuild a spec from a {@code get_lsm_write_spec} response body. + * + *

The server always reports a concrete maintained-index list, so a null selection never + * round-trips. + */ + static LsmWriteSpec fromJson(JsonNode node) { + JsonNode shardingNode = node.get("sharding"); + if (shardingNode == null || shardingNode.get("mode") == null) { + throw new IllegalStateException("get_lsm_write_spec response has no sharding mode"); + } + Sharding sharding = Sharding.fromWireName(shardingNode.get("mode").asText()); + + String column = shardingNode.hasNonNull("column") ? shardingNode.get("column").asText() : null; + Integer numBuckets = + shardingNode.hasNonNull("num_buckets") ? shardingNode.get("num_buckets").asInt() : null; + + List maintainedIndexes = new ArrayList(); + JsonNode indexesNode = node.get("maintained_indexes"); + if (indexesNode != null && indexesNode.isArray()) { + for (JsonNode index : indexesNode) { + maintainedIndexes.add(index.asText()); + } + } + + Map defaults = new HashMap(); + JsonNode defaultsNode = node.get("writer_config_defaults"); + if (defaultsNode != null && defaultsNode.isObject()) { + defaultsNode + .fieldNames() + .forEachRemaining(name -> defaults.put(name, defaultsNode.get(name).asText())); + } + + return new LsmWriteSpec(sharding, column, numBuckets, maintainedIndexes, defaults); + } + + @Override + public String toString() { + return "LsmWriteSpec{sharding=" + + sharding + + ", column=" + + column + + ", numBuckets=" + + numBuckets + + ", maintainedIndexes=" + + maintainedIndexes + + ", writerConfigDefaults=" + + writerConfigDefaults + + "}"; + } +} diff --git a/java/lancedb-core/src/main/java/com/lancedb/MemtableStats.java b/java/lancedb-core/src/main/java/com/lancedb/MemtableStats.java new file mode 100644 index 000000000..777e915aa --- /dev/null +++ b/java/lancedb-core/src/main/java/com/lancedb/MemtableStats.java @@ -0,0 +1,99 @@ +/* + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package com.lancedb; + +import com.fasterxml.jackson.databind.JsonNode; + +import java.util.ArrayList; +import java.util.Collections; +import java.util.List; + +/** One in-memory memtable. */ +public class MemtableStats { + private static final String CONTEXT = "memtable stats"; + + private final long generation; + private final long rows; + private final long bytes; + private final long batches; + private final List indexes; + + MemtableStats(long generation, long rows, long bytes, long batches, List indexes) { + this.generation = generation; + this.rows = rows; + this.bytes = bytes; + this.batches = batches; + this.indexes = Collections.unmodifiableList(indexes); + } + + /** The generation this memtable will become once sealed. */ + public long generation() { + return generation; + } + + /** Rows currently buffered. */ + public long rows() { + return rows; + } + + /** Estimated in-memory size. */ + public long bytes() { + return bytes; + } + + /** Record batches currently buffered. */ + public long batches() { + return batches; + } + + /** + * Names of the indexes this memtable carries. An absent name is the whole answer to "why is my + * fresh-tier search on that column brute-force". + */ + public List indexes() { + return indexes; + } + + static MemtableStats fromJson(JsonNode node) { + JsonFields.requiredObject(node, CONTEXT); + List indexes = new ArrayList(); + for (JsonNode index : JsonFields.requiredArray(node, "indexes", CONTEXT)) { + if (!index.isTextual()) { + throw new IllegalStateException(CONTEXT + " has a non-string index name: " + index); + } + indexes.add(index.asText()); + } + return new MemtableStats( + JsonFields.requiredLong(node, "generation", CONTEXT), + JsonFields.requiredLong(node, "rows", CONTEXT), + JsonFields.requiredLong(node, "bytes", CONTEXT), + JsonFields.requiredLong(node, "batches", CONTEXT), + indexes); + } + + @Override + public String toString() { + return "MemtableStats{generation=" + + generation + + ", rows=" + + rows + + ", bytes=" + + bytes + + ", batches=" + + batches + + ", indexes=" + + indexes + + "}"; + } +} diff --git a/java/lancedb-core/src/test/java/com/lancedb/LanceDbTableLsmTest.java b/java/lancedb-core/src/test/java/com/lancedb/LanceDbTableLsmTest.java new file mode 100644 index 000000000..e84fa5421 --- /dev/null +++ b/java/lancedb-core/src/test/java/com/lancedb/LanceDbTableLsmTest.java @@ -0,0 +1,570 @@ +/* + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package com.lancedb; + +import com.fasterxml.jackson.databind.JsonNode; +import com.fasterxml.jackson.databind.ObjectMapper; +import com.sun.net.httpserver.HttpServer; +import org.junit.jupiter.api.AfterEach; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; + +import java.io.ByteArrayOutputStream; +import java.io.IOException; +import java.io.InputStream; +import java.io.UncheckedIOException; +import java.net.InetSocketAddress; +import java.nio.charset.StandardCharsets; +import java.util.ArrayDeque; +import java.util.ArrayList; +import java.util.Arrays; +import java.util.Collections; +import java.util.Deque; +import java.util.HashMap; +import java.util.LinkedHashMap; +import java.util.List; +import java.util.Map; +import java.util.Optional; +import java.util.concurrent.ConcurrentHashMap; + +import static org.junit.jupiter.api.Assertions.*; + +/** + * Unit tests for the MemWAL LSM routes, run against a scripted local HTTP server. + * + *

The wire assertions mirror the Rust mocked-endpoint tests in {@code + * rust/lancedb/src/remote/table.rs}, which are the contract these routes have to match. + */ +public class LanceDbTableLsmTest { + private static final ObjectMapper MAPPER = new ObjectMapper(); + + private HttpServer server; + private LanceDbRestClient client; + private LanceDbTableLsm lsm; + + private final List requestPaths = Collections.synchronizedList(new ArrayList()); + private final List requestBodies = Collections.synchronizedList(new ArrayList()); + private final Map> replies = new ConcurrentHashMap>(); + + @BeforeEach + public void setUp() throws IOException { + start(); + } + + /** Tear down and restart the scripted server, for a test that scripts several exchanges. */ + private void setUpFresh() { + try { + client.close(); + server.stop(0); + requestPaths.clear(); + requestBodies.clear(); + replies.clear(); + start(); + } catch (IOException e) { + throw new UncheckedIOException(e); + } + } + + private void start() throws IOException { + server = HttpServer.create(new InetSocketAddress("127.0.0.1", 0), 0); + server.createContext( + "/", + exchange -> { + String path = exchange.getRequestURI().getPath(); + requestPaths.add(path); + requestBodies.add(readAll(exchange.getRequestBody())); + + Reply reply = nextReply(path); + byte[] out = reply.body.getBytes(StandardCharsets.UTF_8); + exchange.sendResponseHeaders(reply.status, out.length == 0 ? -1 : out.length); + if (out.length > 0) { + exchange.getResponseBody().write(out); + } + exchange.close(); + }); + server.start(); + + client = + LanceDbNamespaceClientBuilder.newBuilder() + .apiKey("test-key") + .database("test-db") + .endpoint("http://127.0.0.1:" + server.getAddress().getPort()) + .buildRestClient(); + lsm = new LanceDbTableLsm(client, "my_table"); + } + + @AfterEach + public void tearDown() throws IOException { + client.close(); + server.stop(0); + } + + // =========================================================================== + // set / unset / get spec + // =========================================================================== + + @Test + public void testSetLsmWriteSpecUnsharded() throws Exception { + enqueue("set_lsm_write_spec", 200, ""); + + lsm.setLsmWriteSpec(LsmWriteSpec.unsharded()); + + assertEquals("/v1/table/my_table/set_lsm_write_spec/", requestPaths.get(0)); + JsonNode body = MAPPER.readTree(requestBodies.get(0)); + assertEquals("unsharded", body.get("sharding").get("mode").asText()); + assertFalse(body.get("sharding").has("column")); + assertFalse(body.get("sharding").has("num_buckets")); + } + + @Test + public void testSetLsmWriteSpecBucket() throws Exception { + enqueue("set_lsm_write_spec", 200, ""); + + lsm.setLsmWriteSpec( + LsmWriteSpec.bucket("id", 16).withMaintainedIndexes(Arrays.asList("id_idx"))); + + JsonNode body = MAPPER.readTree(requestBodies.get(0)); + assertEquals("bucket", body.get("sharding").get("mode").asText()); + assertEquals("id", body.get("sharding").get("column").asText()); + assertEquals(16, body.get("sharding").get("num_buckets").asInt()); + assertEquals(1, body.get("maintained_indexes").size()); + assertEquals("id_idx", body.get("maintained_indexes").get(0).asText()); + } + + @Test + public void testSetLsmWriteSpecIdentity() throws Exception { + enqueue("set_lsm_write_spec", 200, ""); + + lsm.setLsmWriteSpec(LsmWriteSpec.identity("tenant")); + + JsonNode body = MAPPER.readTree(requestBodies.get(0)); + assertEquals("identity", body.get("sharding").get("mode").asText()); + assertEquals("tenant", body.get("sharding").get("column").asText()); + assertFalse(body.get("sharding").has("num_buckets")); + } + + /** + * The tri-state that motivated a LanceDB-owned spec type: a null selection asks the server to + * resolve every maintainable index, while an empty list asks for none. They must not collapse. + */ + @Test + public void testMaintainedIndexesNullAndEmptyAreDistinctOnTheWire() throws Exception { + enqueue("set_lsm_write_spec", 200, ""); + + lsm.setLsmWriteSpec(LsmWriteSpec.unsharded()); + JsonNode fresh = MAPPER.readTree(requestBodies.get(0)); + assertTrue(fresh.has("maintained_indexes"), "the key must be present"); + assertTrue(fresh.get("maintained_indexes").isNull(), "a fresh spec sends null, not []"); + + lsm.setLsmWriteSpec( + LsmWriteSpec.unsharded().withMaintainedIndexes(Collections.emptyList())); + JsonNode none = MAPPER.readTree(requestBodies.get(1)); + assertTrue(none.get("maintained_indexes").isArray()); + assertEquals(0, none.get("maintained_indexes").size()); + } + + @Test + public void testSetLsmWriteSpecWriterConfigDefaults() throws Exception { + enqueue("set_lsm_write_spec", 200, ""); + + Map defaults = new HashMap(); + defaults.put("max_memtable_rows", "50000"); + lsm.setLsmWriteSpec(LsmWriteSpec.unsharded().withWriterConfigDefaults(defaults)); + + JsonNode body = MAPPER.readTree(requestBodies.get(0)); + assertEquals("50000", body.get("writer_config_defaults").get("max_memtable_rows").asText()); + } + + @Test + public void testUnsetLsmWriteSpec() { + enqueue("unset_lsm_write_spec", 200, ""); + + lsm.unsetLsmWriteSpec(); + + assertEquals("/v1/table/my_table/unset_lsm_write_spec/", requestPaths.get(0)); + assertEquals("", requestBodies.get(0)); + } + + @Test + public void testGetLsmWriteSpec() { + enqueue( + "get_lsm_write_spec", + 200, + "{\"lsm_write_spec\":{\"sharding\":{\"mode\":\"bucket\",\"column\":\"id\"," + + "\"num_buckets\":16},\"maintained_indexes\":[\"id_idx\"]," + + "\"writer_config_defaults\":{\"durable_write\":\"true\"}}}"); + + Optional spec = lsm.getLsmWriteSpec(); + + assertTrue(spec.isPresent()); + assertEquals(LsmWriteSpec.Sharding.BUCKET, spec.get().sharding()); + assertEquals("id", spec.get().column()); + assertEquals(Integer.valueOf(16), spec.get().numBuckets()); + assertEquals(Arrays.asList("id_idx"), spec.get().maintainedIndexes()); + assertEquals("true", spec.get().writerConfigDefaults().get("durable_write")); + } + + @Test + public void testGetLsmWriteSpecAbsent() { + enqueue("get_lsm_write_spec", 200, "{\"lsm_write_spec\":null}"); + + assertFalse(lsm.getLsmWriteSpec().isPresent()); + } + + // =========================================================================== + // stats + // =========================================================================== + + @Test + public void testGetLsmStats() throws Exception { + enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false, 7L, 8L))); + + Optional got = lsm.getLsmStats(true); + + assertEquals("/v1/table/my_table/get_lsm_stats/", requestPaths.get(0)); + assertTrue(MAPPER.readTree(requestBodies.get(0)).get("include_generation_rows").asBoolean()); + assertTrue(got.isPresent()); + BucketStats decoded = got.get().buckets().get(0); + assertEquals("shard-0", decoded.shardId()); + assertEquals("Active", decoded.status()); + assertEquals(1, decoded.writerEpoch()); + assertEquals(2, decoded.manifestVersion()); + assertEquals(9, decoded.currentGeneration()); + assertFalse(decoded.compacting()); + assertEquals(Arrays.asList(7L, 8L), generationNumbers(decoded)); + assertEquals(1024, decoded.generations().get(0).bytes()); + assertFalse(decoded.generations().get(0).rows().isPresent(), "rows absent unless requested"); + assertFalse(decoded.memtables().isPresent(), "absent memtables stay absent"); + } + + /** The optional fields decode when the server does send them. */ + @Test + public void testGetLsmStatsDecodesOptionalFields() { + enqueue( + "get_lsm_stats", + 200, + "{\"lsm_stats\":{\"buckets\":[{\"shard_id\":\"shard-0\",\"status\":\"Active\"," + + "\"writer_epoch\":1,\"manifest_version\":2,\"current_generation\":9," + + "\"replay_after_wal_entry_position\":3,\"wal_entry_position_last_seen\":11," + + "\"generations\":[{\"generation\":7,\"bytes\":1024,\"rows\":42}]," + + "\"compacting\":true,\"memtables\":[{\"generation\":8,\"rows\":5," + + "\"bytes\":64,\"batches\":2,\"indexes\":[\"id_idx\"]}]}]}}"); + + BucketStats decoded = lsm.getLsmStats(true).get().buckets().get(0); + + assertEquals(3, decoded.replayAfterWalEntryPosition()); + assertEquals(11, decoded.walEntryPositionLastSeen()); + assertTrue(decoded.compacting()); + assertEquals(42, decoded.generations().get(0).rows().getAsLong()); + assertTrue(decoded.memtables().isPresent()); + MemtableStats memtable = decoded.memtables().get().get(0); + assertEquals(8, memtable.generation()); + assertEquals(5, memtable.rows()); + assertEquals(64, memtable.bytes()); + assertEquals(2, memtable.batches()); + assertEquals(Arrays.asList("id_idx"), memtable.indexes()); + } + + @Test + public void testGetLsmStatsAbsentWhenLsmDisabled() { + enqueue("get_lsm_stats", 200, "{\"lsm_stats\":null}"); + + assertFalse(lsm.getLsmStats().isPresent()); + } + + @Test + public void testGetLsmStatsDefaultsToExcludingGenerationRows() throws Exception { + enqueue("get_lsm_stats", 200, stats()); + + lsm.getLsmStats(); + + assertFalse(MAPPER.readTree(requestBodies.get(0)).get("include_generation_rows").asBoolean()); + } + + // =========================================================================== + // flush / compact + // =========================================================================== + + @Test + public void testFlushAndCompactRoutes() { + enqueue("flush_lsm", 200, ""); + enqueue("compact_lsm", 200, ""); + + lsm.flushLsm(); + lsm.compactLsm(); + + assertEquals("/v1/table/my_table/flush_lsm/", requestPaths.get(0)); + assertEquals("/v1/table/my_table/compact_lsm/", requestPaths.get(1)); + } + + @Test + public void testHttpErrorCarriesStatus() { + enqueue("flush_lsm", 404, "no such table"); + + LanceDbRestClient.HttpException e = + assertThrows(LanceDbRestClient.HttpException.class, () -> lsm.flushLsm()); + assertEquals(404, e.statusCode()); + } + + // =========================================================================== + // checkpoint + // =========================================================================== + + @Test + public void testCheckpointReturnsWhenLsmDisabled() { + enqueue("flush_lsm", 200, ""); + enqueue("get_lsm_stats", 200, "{\"lsm_stats\":null}"); + + lsm.checkpointLsm(); + + assertEquals(0, countCalls("compact_lsm"), "nothing to compact when the LSM path is off"); + } + + @Test + public void testCheckpointReturnsWhenNoGenerationsOutstanding() { + enqueue("flush_lsm", 200, ""); + // A bucket with no L0 generations yields no target, so the drain never starts. + enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false))); + + lsm.checkpointLsm(); + + assertEquals(0, countCalls("compact_lsm")); + } + + @Test + public void testCheckpointConvergesOnceTargetGenerationsAreGone() { + enqueue("flush_lsm", 200, ""); + // Watermark read: shard-0 holds generations 7 and 8, so target = 8. + enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false, 7L, 8L))); + // First drain poll: both still outstanding, nothing compacting -> dispatch a pass. + enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false, 7L, 8L))); + // Second drain poll: drained past the target -> done. + enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false, 9L))); + enqueue("compact_lsm", 200, ""); + + lsm.checkpointLsm(); + + assertEquals(1, countCalls("compact_lsm"), "one pass dispatched"); + assertEquals(3, countCalls("get_lsm_stats"), "watermark read plus two drain polls"); + } + + @Test + public void testCheckpointDoesNotPileOnWhileEveryTargetBucketIsCompacting() { + enqueue("flush_lsm", 200, ""); + enqueue("get_lsm_stats", 200, stats(bucket("shard-0", true, 4L))); + // Still compacting on the first poll, so no pass is dispatched; then it drains. + enqueue("get_lsm_stats", 200, stats(bucket("shard-0", true, 4L))); + enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false, 5L))); + + lsm.checkpointLsm(); + + assertEquals(0, countCalls("compact_lsm"), "a latched bucket is left alone"); + } + + @Test + public void testCheckpointRetriesFromFlushAfterLostClaim() { + // 421 on the watermark read: the node lost its claim, so the whole thing restarts + // from flush rather than retrying the read in place. + enqueue("flush_lsm", 200, ""); + enqueue("get_lsm_stats", 421, "no claim"); + enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false))); + + lsm.checkpointLsm(); + + assertEquals(2, countCalls("flush_lsm"), "re-issued from flush"); + } + + @Test + public void testCheckpointRetriesRetryableStatusInPlace() { + enqueue("flush_lsm", 429, "latch held"); + enqueue("flush_lsm", 200, ""); + enqueue("get_lsm_stats", 200, stats(bucket("shard-0", false))); + + lsm.checkpointLsm(); + + assertEquals(2, countCalls("flush_lsm"), "429 retried in place, not re-issued"); + } + + @Test + public void testCheckpointPropagatesTerminalStatus() { + enqueue("flush_lsm", 400, "bad request"); + + LanceDbRestClient.HttpException e = + assertThrows(LanceDbRestClient.HttpException.class, () -> lsm.checkpointLsm()); + assertEquals(400, e.statusCode()); + assertEquals(1, countCalls("flush_lsm"), "a terminal status is not retried"); + } + + @Test + public void testCheckpointGivesUpAfterRepeatedLostClaims() { + enqueue("flush_lsm", 421, "no claim"); + + IllegalStateException e = assertThrows(IllegalStateException.class, () -> lsm.checkpointLsm()); + assertTrue(e.getMessage().contains("kept losing its claim"), e.getMessage()); + assertEquals(4, countCalls("flush_lsm"), "the initial attempt plus MAX_REISSUES"); + } + + // =========================================================================== + // strict decoding + // =========================================================================== + + /** + * A stats payload that does not decode must fail closed. Every one of these bodies used to be + * read as "no buckets", which is indistinguishable from a drained table, so {@code checkpointLsm} + * reported convergence for a checkpoint that never ran. + */ + @Test + public void testCheckpointRejectsMalformedStats() { + Map malformed = new LinkedHashMap(); + malformed.put("no response body at all", ""); + malformed.put("stats object with no buckets", "{\"lsm_stats\":{}}"); + malformed.put("bucket missing its required fields", "{\"lsm_stats\":{\"buckets\":[{}]}}"); + malformed.put( + "bucket missing generations", + "{\"lsm_stats\":{\"buckets\":[{\"shard_id\":\"shard-0\",\"status\":\"Active\"," + + "\"writer_epoch\":1,\"manifest_version\":2,\"current_generation\":9," + + "\"replay_after_wal_entry_position\":0,\"wal_entry_position_last_seen\":0," + + "\"compacting\":false}]}}"); + malformed.put( + "generation with a non-numeric generation number", + "{\"lsm_stats\":{\"buckets\":[{\"shard_id\":\"shard-0\",\"status\":\"Active\"," + + "\"writer_epoch\":1,\"manifest_version\":2,\"current_generation\":9," + + "\"replay_after_wal_entry_position\":0,\"wal_entry_position_last_seen\":0," + + "\"generations\":[{\"generation\":\"7\",\"bytes\":1024}]," + + "\"compacting\":false}]}}"); + + for (Map.Entry each : malformed.entrySet()) { + setUpFresh(); + enqueue("flush_lsm", 200, ""); + enqueue("get_lsm_stats", 200, each.getValue()); + + assertThrows( + IllegalStateException.class, + () -> lsm.checkpointLsm(), + each.getKey() + " must not report convergence"); + } + } + + /** The one shape that legitimately means "this table has no LSM write path". */ + @Test + public void testCheckpointTreatsNullStatsAsNotWalBacked() { + enqueue("flush_lsm", 200, ""); + enqueue("get_lsm_stats", 200, "{\"lsm_stats\":null}"); + + lsm.checkpointLsm(); + + assertEquals(1, countCalls("get_lsm_stats")); + } + + // =========================================================================== + // retry budget + // =========================================================================== + + /** + * The transport must not retry on the checkpoint loop's behalf. Apache HttpClient's default + * strategy retries exactly 429 and 503 — the two statuses {@code isRetryable} owns — which + * doubled every budget here and also retried {@code compact_lsm} in place, where the loop is + * built to fall through to a fresh stats poll instead. + */ + @Test + public void testCheckpointRetryBudgetIsNotDoubledByTheTransport() { + enqueue("flush_lsm", 429, "latch held"); + + LanceDbRestClient.HttpException e = + assertThrows(LanceDbRestClient.HttpException.class, () -> lsm.checkpointLsm()); + + assertEquals(429, e.statusCode(), "the exhausted budget propagates the last error as itself"); + assertEquals(9, countCalls("flush_lsm"), "the initial request plus MAX_RETRIES, and no more"); + } + + // =========================================================================== + // harness + // =========================================================================== + + private static List generationNumbers(BucketStats bucket) { + List numbers = new ArrayList(); + for (GenerationStats generation : bucket.generations()) { + numbers.add(generation.generation()); + } + return numbers; + } + + /** Build an {@code lsm_stats} response body from bucket fragments. */ + private static String stats(String... buckets) { + return "{\"lsm_stats\":{\"buckets\":[" + String.join(",", buckets) + "]}}"; + } + + private static String bucket(String shardId, boolean compacting, Long... generations) { + StringBuilder gens = new StringBuilder(); + for (Long generation : generations) { + if (gens.length() > 0) { + gens.append(","); + } + gens.append("{\"generation\":").append(generation).append(",\"bytes\":1024}"); + } + return "{\"shard_id\":\"" + + shardId + + "\",\"status\":\"Active\",\"writer_epoch\":1,\"manifest_version\":2," + + "\"current_generation\":9,\"replay_after_wal_entry_position\":0," + + "\"wal_entry_position_last_seen\":0,\"generations\":[" + + gens + + "],\"compacting\":" + + compacting + + "}"; + } + + /** Queue a reply for an operation. The last queued reply repeats once the queue drains. */ + private void enqueue(String operation, int status, String body) { + replies.computeIfAbsent(operation, key -> new ArrayDeque()).add(new Reply(status, body)); + } + + private Reply nextReply(String path) { + String operation = operationOf(path); + Deque queued = replies.get(operation); + if (queued == null || queued.isEmpty()) { + return new Reply(200, ""); + } + return queued.size() > 1 ? queued.poll() : queued.peek(); + } + + private long countCalls(String operation) { + return requestPaths.stream().filter(path -> operationOf(path).equals(operation)).count(); + } + + /** {@code /v1/table/my_table/flush_lsm/} -> {@code flush_lsm}. */ + private static String operationOf(String path) { + String[] segments = path.split("/"); + return segments.length == 0 ? "" : segments[segments.length - 1]; + } + + private static String readAll(InputStream in) throws IOException { + ByteArrayOutputStream out = new ByteArrayOutputStream(); + byte[] buffer = new byte[4096]; + int read; + while ((read = in.read(buffer)) != -1) { + out.write(buffer, 0, read); + } + return new String(out.toByteArray(), StandardCharsets.UTF_8); + } + + private static final class Reply { + private final int status; + private final String body; + + private Reply(int status, String body) { + this.status = status; + this.body = body; + } + } +} diff --git a/nodejs/__test__/table.test.ts b/nodejs/__test__/table.test.ts index 5396a251a..80c50f1ac 100644 --- a/nodejs/__test__/table.test.ts +++ b/nodejs/__test__/table.test.ts @@ -3341,6 +3341,59 @@ describe("LSM merge insert", () => { }); }); +describe("LSM convergence and stats", () => { + let tmpDir: tmp.DirResult; + + beforeEach(() => { + tmpDir = tmp.dirSync({ unsafeCleanup: true }); + }); + afterEach(() => tmpDir.removeCallback()); + + async function lsmTable(conn: Connection): Promise { + const table = await conn.createEmptyTable( + "t", + new arrow.Schema([new arrow.Field("id", new arrow.Utf8(), false)]), + ); + await table.setUnenforcedPrimaryKey("id"); + await table.setLsmWriteSpec({ specType: "unsharded" }); + return table; + } + + // These four route through the server that owns the MemWAL, so a local table + // rejects them rather than answering. What is asserted here is that the + // bindings reach the core at all; the behavior against a real endpoint is + // covered by the mocked endpoint tests in rust/lancedb/src/remote/table.rs. + it("rejects flushLsm on a local table", async () => { + const conn = await connect(tmpDir.name); + const table = await lsmTable(conn); + + await expect(table.flushLsm()).rejects.toThrow(/not supported/i); + }); + + it("rejects compactLsm on a local table", async () => { + const conn = await connect(tmpDir.name); + const table = await lsmTable(conn); + + await expect(table.compactLsm()).rejects.toThrow(/not supported/i); + }); + + it("rejects getLsmStats on a local table", async () => { + const conn = await connect(tmpDir.name); + const table = await lsmTable(conn); + + await expect(table.getLsmStats()).rejects.toThrow(/not supported/i); + await expect(table.getLsmStats(true)).rejects.toThrow(/not supported/i); + }); + + it("rejects checkpointLsm on a local table", async () => { + const conn = await connect(tmpDir.name); + const table = await lsmTable(conn); + + // checkpointLsm seals first, so it surfaces flushLsm's rejection. + await expect(table.checkpointLsm()).rejects.toThrow(/not supported/i); + }); +}); + describe("computed columns", () => { let tmpDir: tmp.DirResult; beforeEach(() => { diff --git a/nodejs/lancedb/index.ts b/nodejs/lancedb/index.ts index 9f2e97989..6a5bfe3b4 100644 --- a/nodejs/lancedb/index.ts +++ b/nodejs/lancedb/index.ts @@ -147,6 +147,10 @@ export { FtsToken, TokenizeTableOptions, LsmWriteSpec, + LsmStats, + BucketStats, + GenerationStats, + MemtableStats, ColumnAlteration, FieldMetadataUpdate, } from "./table"; diff --git a/nodejs/lancedb/table.ts b/nodejs/lancedb/table.ts index a7dc8def1..964c2cea3 100644 --- a/nodejs/lancedb/table.ts +++ b/nodejs/lancedb/table.ts @@ -31,6 +31,7 @@ import { IndexConfig, IndexStatistics, Job, + LsmStats, Branches as NativeBranches, OptimizeStats, RefreshColumnResult, @@ -50,6 +51,12 @@ import { import { sanitizeType } from "./sanitize"; import { IntoSql, toSQL } from "./util"; export { IndexConfig } from "./native"; +export { + BucketStats, + GenerationStats, + LsmStats, + MemtableStats, +} from "./native"; /** * Progress snapshot for a write operation, delivered to the `progress` @@ -706,6 +713,59 @@ export abstract class Table { * @returns {Promise} */ abstract closeLsmWriters(): Promise; + /** + * Seal every bucket's active memtable into a new L0 generation. + * + * Returns once the seal is committed. Sealing an empty memtable is a no-op, + * so this is safe to call repeatedly. + * @returns {Promise} + */ + abstract flushLsm(): Promise; + /** + * Trigger a background L0 → base compaction pass per bucket. + * + * Returns once the passes are *dispatched*, not once they finish — watch + * {@link Table#getLsmStats} for progress, or use + * {@link Table#checkpointLsm} to wait for convergence. + * @returns {Promise} + */ + abstract compactLsm(): Promise; + /** + * Converge this table's LSM write path into its base table. + * + * Seals once, then triggers compaction and polls until the L0 that existed + * at the start is gone. The target set is fixed at the start, so + * generations created *during* the checkpoint are ignored — that is what + * lets it terminate under write load, and what makes it best-effort: it + * converges the fresh tier as of some instant. Idempotent, abandonable at + * any point, and safe to run on a cadence. + * + * There is no liveness bound — the compactor pool is shared across tables, + * so a checkpoint queued behind unrelated work looks exactly like one that + * is merging. The caller owns the deadline. + * @returns {Promise} + * @example + * ```ts + * const before = await table.getLsmStats(); + * await table.checkpointLsm(); + * const after = await table.getLsmStats(); + * ``` + */ + abstract checkpointLsm(): Promise; + /** + * Read live per-bucket LSM state. + * + * Answers "how far behind is my fresh tier", "which bucket is hot", and + * "why is my fresh-tier vector search brute-force". Mutates no table state. + * + * Resolves to `undefined` only when the LSM write path is not enabled. + * @param {boolean} includeGenerationRows Also count rows per L0 generation. + * Off by default because each count opens an uncached Lance dataset. + * @returns {Promise} + */ + abstract getLsmStats( + includeGenerationRows?: boolean, + ): Promise; /** Retrieve the version of the table */ abstract version(): Promise; @@ -1266,6 +1326,24 @@ export class LocalTable extends Table { return await this.inner.closeLsmWriters(); } + async flushLsm(): Promise { + return await this.inner.flushLsm(); + } + + async compactLsm(): Promise { + return await this.inner.compactLsm(); + } + + async checkpointLsm(): Promise { + return await this.inner.checkpointLsm(); + } + + async getLsmStats( + includeGenerationRows: boolean = false, + ): Promise { + return (await this.inner.getLsmStats(includeGenerationRows)) ?? undefined; + } + async version(): Promise { return await this.inner.version(); } diff --git a/nodejs/src/table.rs b/nodejs/src/table.rs index 4c45be668..b15491202 100644 --- a/nodejs/src/table.rs +++ b/nodejs/src/table.rs @@ -497,6 +497,34 @@ impl Table { self.inner_ref()?.close_lsm_writers().await.default_error() } + #[napi(catch_unwind)] + pub async fn flush_lsm(&self) -> napi::Result<()> { + self.inner_ref()?.flush_lsm().await.default_error() + } + + #[napi(catch_unwind)] + pub async fn compact_lsm(&self) -> napi::Result<()> { + self.inner_ref()?.compact_lsm().await.default_error() + } + + #[napi(catch_unwind)] + pub async fn checkpoint_lsm(&self) -> napi::Result<()> { + self.inner_ref()?.checkpoint_lsm().await.default_error() + } + + #[napi(catch_unwind)] + pub async fn get_lsm_stats( + &self, + include_generation_rows: bool, + ) -> napi::Result> { + let stats = self + .inner_ref()? + .get_lsm_stats(include_generation_rows) + .await + .default_error()?; + Ok(stats.map(LsmStats::from)) + } + #[napi(catch_unwind)] pub async fn version(&self) -> napi::Result { self.inner_ref()? @@ -889,6 +917,129 @@ impl From for LsmWriteSpec { } } +/// One flushed L0 generation. +#[napi(object)] +#[derive(Clone, Debug)] +pub struct GenerationStats { + /// The generation number. Increases as memtables are sealed into L0. + pub generation: i64, + /// On-disk size of the generation. + pub bytes: i64, + /// Present only when `includeGenerationRows` was requested. Off by default + /// because each count opens an uncached Lance dataset. + pub rows: Option, +} + +impl From for GenerationStats { + fn from(g: lancedb::table::GenerationStats) -> Self { + Self { + generation: g.generation as i64, + bytes: g.bytes as i64, + rows: g.rows.map(|r| r as i64), + } + } +} + +/// One in-memory memtable. +#[napi(object)] +#[derive(Clone, Debug)] +pub struct MemtableStats { + /// The generation this memtable will become once sealed. + pub generation: i64, + /// Rows currently buffered. + pub rows: i64, + /// Estimated in-memory size. + pub bytes: i64, + /// Record batches currently buffered. + pub batches: i64, + /// Names of the indexes this memtable carries. An absent name is the whole + /// answer to "why is my fresh-tier search on that column brute-force". + pub indexes: Vec, +} + +impl From for MemtableStats { + fn from(m: lancedb::table::MemtableStats) -> Self { + Self { + generation: m.generation as i64, + rows: m.rows as i64, + bytes: m.bytes as i64, + batches: m.batches as i64, + indexes: m.indexes, + } + } +} + +/// Live state of one bucket. A table is N buckets on one node; flattening to a +/// single number hides the one hot bucket that is usually why someone opened +/// this endpoint. +#[napi(object)] +#[derive(Clone, Debug)] +pub struct BucketStats { + /// The shard this bucket writes. + pub shard_id: String, + /// `"Active"` or `"Sealed"` (drop-table 2PC in flight). + pub status: String, + /// Epoch of the writer that currently owns the shard. + pub writer_epoch: i64, + /// Version of the shard manifest these numbers were read from. + pub manifest_version: i64, + /// The generation the active memtable will become. + pub current_generation: i64, + /// WAL position replay resumes from. + pub replay_after_wal_entry_position: i64, + /// Highest WAL position the writer has seen. The difference against + /// `replayAfterWalEntryPosition` is the WAL lag. + pub wal_entry_position_last_seen: i64, + /// Flushed L0 generations not yet merged into the base table. + pub generations: Vec, + /// Whether a pass owns this bucket's compaction latch right now. Says *a* + /// driver is running, not *whose*, and the latch is held from dispatch — + /// including while the pass queues for a pod-wide compactor permit. Read it + /// as "do not pile on", never as "mine is progressing". + pub compacting: bool, + /// Oldest first, active last. Absent for a `"Sealed"` bucket, whose + /// in-memory state is torn down. + pub memtables: Option>, +} + +impl From for BucketStats { + fn from(b: lancedb::table::BucketStats) -> Self { + Self { + shard_id: b.shard_id, + status: b.status, + writer_epoch: b.writer_epoch as i64, + manifest_version: b.manifest_version as i64, + current_generation: b.current_generation as i64, + replay_after_wal_entry_position: b.replay_after_wal_entry_position as i64, + wal_entry_position_last_seen: b.wal_entry_position_last_seen as i64, + generations: b.generations.into_iter().map(Into::into).collect(), + compacting: b.compacting, + memtables: b + .memtables + .map(|ms| ms.into_iter().map(Into::into).collect()), + } + } +} + +/// Live per-bucket LSM state, as returned by `Table#getLsmStats`. +/// +/// Nothing here is derived: sums and differences (total L0 bytes, WAL lag) are +/// the caller's to compute. +#[napi(object)] +#[derive(Clone, Debug)] +pub struct LsmStats { + /// One entry per bucket backing this table. + pub buckets: Vec, +} + +impl From for LsmStats { + fn from(stats: lancedb::table::LsmStats) -> Self { + Self { + buckets: stats.buckets.into_iter().map(Into::into).collect(), + } + } +} + /// Statistics about a compaction operation. #[napi(object)] #[derive(Clone, Debug)] diff --git a/python/python/lancedb/__init__.py b/python/python/lancedb/__init__.py index 235049f97..e12ef4e86 100644 --- a/python/python/lancedb/__init__.py +++ b/python/python/lancedb/__init__.py @@ -12,6 +12,7 @@ __version__ = importlib.metadata.version("lancedb") from ._lancedb import connect as lancedb_connect from ._lancedb import FtsToken +from ._lancedb import LsmWriteSpec from ._lancedb import tokenize as _tokenize from .common import URI, sanitize_uri from urllib.parse import urlparse @@ -518,6 +519,7 @@ __all__ = [ "Job", "LanceDBConnection", "LanceNamespaceDBConnection", + "LsmWriteSpec", "RemoteDBConnection", "Session", "Table", diff --git a/python/python/lancedb/table.py b/python/python/lancedb/table.py index f97d0331c..393b2eed3 100644 --- a/python/python/lancedb/table.py +++ b/python/python/lancedb/table.py @@ -4801,7 +4801,7 @@ class AsyncTable: Examples -------- - >>> from lancedb._lancedb import LsmWriteSpec + >>> from lancedb import LsmWriteSpec >>> # table.set_unenforced_primary_key("id") >>> # table.set_lsm_write_spec(LsmWriteSpec.bucket("id", 16)) """ From 27cea03b7d4a71665567e24a09abef30fa8b62d8 Mon Sep 17 00:00:00 2001 From: LanceDB Robot Date: Wed, 19 Aug 2026 13:18:27 -0700 Subject: [PATCH 12/12] chore: update lance dependency to v11.0.0-beta.15 (#3968) Bumps the Rust workspace Lance dependencies and Java lance-core to v11.0.0-beta.15. Updates the computed-column refresh path for the new `write_columns` API. Release: https://github.com/lance-format/lance/releases/tag/v11.0.0-beta.15 --- Cargo.lock | 84 +++++++++++++++---------------- Cargo.toml | 28 +++++------ java/pom.xml | 2 +- rust/lancedb/src/table/refresh.rs | 6 +-- 4 files changed, 60 insertions(+), 60 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 3f4d6682c..013850034 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3455,8 +3455,8 @@ checksum = "42703706b716c37f96a77aea830392ad231f44c9e9a67872fa5548707e11b11c" [[package]] name = "fsst" -version = "11.0.0-beta.14" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" +version = "11.0.0-beta.15" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb" dependencies = [ "arrow-array", "rand 0.9.5", @@ -4815,8 +4815,8 @@ checksum = "e037a2e1d8d5fdbd49b16a4ea09d5d6401c1f29eca5ff29d03d3824dba16256a" [[package]] name = "lance" -version = "11.0.0-beta.14" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" +version = "11.0.0-beta.15" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb" dependencies = [ "arc-swap", "arrow", @@ -4888,8 +4888,8 @@ dependencies = [ [[package]] name = "lance-arrow" -version = "11.0.0-beta.14" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" +version = "11.0.0-beta.15" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb" dependencies = [ "arrow-array", "arrow-buffer", @@ -4911,7 +4911,7 @@ dependencies = [ [[package]] name = "lance-arrow-scalar" version = "58.0.0" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb" dependencies = [ "arrow-array", "arrow-buffer", @@ -4925,7 +4925,7 @@ dependencies = [ [[package]] name = "lance-arrow-stats" version = "58.0.0" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb" dependencies = [ "arrow-array", "arrow-schema", @@ -4934,8 +4934,8 @@ dependencies = [ [[package]] name = "lance-bitpacking" -version = "11.0.0-beta.14" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" +version = "11.0.0-beta.15" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb" dependencies = [ "arrayref", "crunchy", @@ -4945,8 +4945,8 @@ dependencies = [ [[package]] name = "lance-core" -version = "11.0.0-beta.14" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" +version = "11.0.0-beta.15" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb" dependencies = [ "arrow-array", "arrow-buffer", @@ -4983,8 +4983,8 @@ dependencies = [ [[package]] name = "lance-datafusion" -version = "11.0.0-beta.14" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" +version = "11.0.0-beta.15" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb" dependencies = [ "arrow", "arrow-array", @@ -5013,8 +5013,8 @@ dependencies = [ [[package]] name = "lance-datagen" -version = "11.0.0-beta.14" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" +version = "11.0.0-beta.15" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb" dependencies = [ "arrow", "arrow-array", @@ -5031,8 +5031,8 @@ dependencies = [ [[package]] name = "lance-derive" -version = "11.0.0-beta.14" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" +version = "11.0.0-beta.15" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb" dependencies = [ "proc-macro2", "quote", @@ -5041,8 +5041,8 @@ dependencies = [ [[package]] name = "lance-encoding" -version = "11.0.0-beta.14" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" +version = "11.0.0-beta.15" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb" dependencies = [ "arrow-arith", "arrow-array", @@ -5075,8 +5075,8 @@ dependencies = [ [[package]] name = "lance-file" -version = "11.0.0-beta.14" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" +version = "11.0.0-beta.15" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb" dependencies = [ "arrow-arith", "arrow-array", @@ -5107,8 +5107,8 @@ dependencies = [ [[package]] name = "lance-index" -version = "11.0.0-beta.14" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" +version = "11.0.0-beta.15" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb" dependencies = [ "arc-swap", "arrow", @@ -5172,8 +5172,8 @@ dependencies = [ [[package]] name = "lance-index-core" -version = "11.0.0-beta.14" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" +version = "11.0.0-beta.15" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb" dependencies = [ "arrow-array", "arrow-schema", @@ -5195,8 +5195,8 @@ dependencies = [ [[package]] name = "lance-io" -version = "11.0.0-beta.14" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" +version = "11.0.0-beta.15" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb" dependencies = [ "arrow", "arrow-array", @@ -5232,8 +5232,8 @@ dependencies = [ [[package]] name = "lance-linalg" -version = "11.0.0-beta.14" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" +version = "11.0.0-beta.15" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb" dependencies = [ "arrow-array", "arrow-schema", @@ -5247,8 +5247,8 @@ dependencies = [ [[package]] name = "lance-namespace" -version = "11.0.0-beta.14" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" +version = "11.0.0-beta.15" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb" dependencies = [ "arrow", "async-trait", @@ -5260,8 +5260,8 @@ dependencies = [ [[package]] name = "lance-namespace-impls" -version = "11.0.0-beta.14" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" +version = "11.0.0-beta.15" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb" dependencies = [ "arrow", "arrow-ipc", @@ -5314,8 +5314,8 @@ dependencies = [ [[package]] name = "lance-select" -version = "11.0.0-beta.14" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" +version = "11.0.0-beta.15" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb" dependencies = [ "arrow-array", "arrow-buffer", @@ -5329,8 +5329,8 @@ dependencies = [ [[package]] name = "lance-table" -version = "11.0.0-beta.14" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" +version = "11.0.0-beta.15" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb" dependencies = [ "arrow", "arrow-array", @@ -5370,8 +5370,8 @@ dependencies = [ [[package]] name = "lance-testing" -version = "11.0.0-beta.14" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" +version = "11.0.0-beta.15" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb" dependencies = [ "arrow-array", "arrow-schema", @@ -5384,8 +5384,8 @@ dependencies = [ [[package]] name = "lance-tokenizer" -version = "11.0.0-beta.14" -source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.14#059f71af86c8e77d98bea2469e259e9a8363a460" +version = "11.0.0-beta.15" +source = "git+https://github.com/lance-format/lance.git?tag=v11.0.0-beta.15#8064b3a27dc4e05a6ab6ceb439fa1be9950e00eb" dependencies = [ "frostem", "icu_segmenter", diff --git a/Cargo.toml b/Cargo.toml index c90fb81d8..0a8c0d36e 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -13,20 +13,20 @@ categories = ["database-implementations"] rust-version = "1.91.0" [workspace.dependencies] -lance = { "version" = "=11.0.0-beta.14", default-features = false, "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } -lance-core = { "version" = "=11.0.0-beta.14", "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } -lance-datagen = { "version" = "=11.0.0-beta.14", "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } -lance-file = { "version" = "=11.0.0-beta.14", "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } -lance-io = { "version" = "=11.0.0-beta.14", default-features = false, "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } -lance-index = { "version" = "=11.0.0-beta.14", "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } -lance-linalg = { "version" = "=11.0.0-beta.14", "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } -lance-namespace = { "version" = "=11.0.0-beta.14", "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } -lance-namespace-impls = { "version" = "=11.0.0-beta.14", default-features = false, "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } -lance-table = { "version" = "=11.0.0-beta.14", "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } -lance-testing = { "version" = "=11.0.0-beta.14", "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } -lance-datafusion = { "version" = "=11.0.0-beta.14", "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } -lance-encoding = { "version" = "=11.0.0-beta.14", "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } -lance-arrow = { "version" = "=11.0.0-beta.14", "tag" = "v11.0.0-beta.14", "git" = "https://github.com/lance-format/lance.git" } +lance = { "version" = "=11.0.0-beta.15", default-features = false, "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" } +lance-core = { "version" = "=11.0.0-beta.15", "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" } +lance-datagen = { "version" = "=11.0.0-beta.15", "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" } +lance-file = { "version" = "=11.0.0-beta.15", "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" } +lance-io = { "version" = "=11.0.0-beta.15", default-features = false, "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" } +lance-index = { "version" = "=11.0.0-beta.15", "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" } +lance-linalg = { "version" = "=11.0.0-beta.15", "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" } +lance-namespace = { "version" = "=11.0.0-beta.15", "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" } +lance-namespace-impls = { "version" = "=11.0.0-beta.15", default-features = false, "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" } +lance-table = { "version" = "=11.0.0-beta.15", "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" } +lance-testing = { "version" = "=11.0.0-beta.15", "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" } +lance-datafusion = { "version" = "=11.0.0-beta.15", "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" } +lance-encoding = { "version" = "=11.0.0-beta.15", "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" } +lance-arrow = { "version" = "=11.0.0-beta.15", "tag" = "v11.0.0-beta.15", "git" = "https://github.com/lance-format/lance.git" } ahash = "0.8" # Note that this one does not include pyarrow arrow = { version = "58.0.0", optional = false } diff --git a/java/pom.xml b/java/pom.xml index 63711d0c9..92e6344f3 100644 --- a/java/pom.xml +++ b/java/pom.xml @@ -28,7 +28,7 @@ UTF-8 15.0.0 - 11.0.0-beta.14 + 11.0.0-beta.15 false 2.30.0 1.7 diff --git a/rust/lancedb/src/table/refresh.rs b/rust/lancedb/src/table/refresh.rs index edc78387e..b29c97e98 100644 --- a/rust/lancedb/src/table/refresh.rs +++ b/rust/lancedb/src/table/refresh.rs @@ -12,7 +12,7 @@ //! decides whether the fragment is staged at all -- a fragment where nothing //! would change stages nothing, which is what lets an expression yielding //! null settle instead of restaging forever. The second streams the -//! fragment's physical rows into `write_column` a batch at a time, so peak +//! fragment's physical rows into `write_columns` a batch at a time, so peak //! memory is bounded by a scan batch. The expression is evaluated by this //! module, never through a projection alias, and only over rows being //! filled: every other row -- deleted, or already holding a value -- has its @@ -67,7 +67,7 @@ pub(crate) async fn execute_refresh_column( .ok_or_else(|| Error::ColumnNotFound { name: column.to_string(), })?; - // The dataset's own field, so the identity write_column checks against the + // The dataset's own field, so the identity write_columns checks against the // manifest holds by construction. let column_schema = LanceSchema { fields: vec![field.clone()], @@ -83,7 +83,7 @@ pub(crate) async fn execute_refresh_column( } rows_filled += gained; let values = fill_stream(&dataset, &fragment, bound.clone(), column).await?; - replacements.push(fragment.write_column(values, &column_schema).await?); + replacements.push(fragment.write_columns(values, &column_schema).await?); } if replacements.is_empty() {