mirror of
https://github.com/lancedb/lancedb.git
synced 2026-09-12 00:02:21 +00:00
this PR blob v2 field helpers and reads to the Node SDK.
`blob()` marks a field as blob v2 and lets you set the storage
thresholds. Inputs can be bytes, a URI, or a data/uri struct.
Queries return descriptors. `fetchBlobs()` reads the bytes by row ID,
and `fetchBlobFiles()` gives you lazy handles for full or range reads.
`blobColumns()` lists the blob fields, including nested ones.
Fetch uses the table’s current checkout. It preserves order, duplicates,
and nulls. Holding row IDs across compaction still requires stable row
IDs.
```javascript
const db = await connect("./data");
const video = await readFile("clip.mp4");
const table = await db.createTable(
"videos",
[{ id: 1n, video }],
{
schema: new Schema([
new Field("id", new Int64()),
blob("video"),
]),
},
);
const rows = await table.query().select(["id"]).withRowId().toArray();
const rowIds = rows.map((row) => row._rowid as bigint);
const bytes = await table.fetchBlobs("video", rowIds);
const [handle] = await table.fetchBlobFiles("video", rowIds);
const header = await handle!.readRange(0n, 65536n);
```
### Testing
- cover input validation, thresholds, nested fields, fetch ordering,
nulls, and range reads.
96 lines
2.4 KiB
Rust
96 lines
2.4 KiB
Rust
// SPDX-License-Identifier: Apache-2.0
|
|
// SPDX-FileCopyrightText: Copyright The LanceDB Authors
|
|
|
|
use std::ops::Range;
|
|
use std::sync::Arc;
|
|
|
|
use arrow_array::{Array, LargeBinaryArray};
|
|
use lancedb::blob::BlobFile as LanceBlobFile;
|
|
use napi::bindgen_prelude::*;
|
|
use napi_derive::napi;
|
|
|
|
use crate::error::convert_error;
|
|
|
|
#[napi]
|
|
pub struct BlobFile {
|
|
inner: Arc<LanceBlobFile>,
|
|
}
|
|
|
|
impl BlobFile {
|
|
pub(crate) fn new(inner: LanceBlobFile) -> Self {
|
|
Self {
|
|
inner: Arc::new(inner),
|
|
}
|
|
}
|
|
}
|
|
|
|
#[napi]
|
|
impl BlobFile {
|
|
#[napi]
|
|
pub fn size(&self) -> BigInt {
|
|
BigInt::from(self.inner.size())
|
|
}
|
|
|
|
#[napi]
|
|
pub async fn read(&self) -> napi::Result<Buffer> {
|
|
let bytes = self.inner.read().await.map_err(|err| convert_error(&err))?;
|
|
Ok(Buffer::from(bytes.as_ref()))
|
|
}
|
|
|
|
#[napi]
|
|
pub async fn read_range(&self, start: BigInt, end: BigInt) -> napi::Result<Buffer> {
|
|
let range = bigint_range(start, end)?;
|
|
let bytes = self
|
|
.inner
|
|
.read_range(range)
|
|
.await
|
|
.map_err(|err| convert_error(&err))?;
|
|
Ok(Buffer::from(bytes.as_ref()))
|
|
}
|
|
}
|
|
|
|
fn bigint_range(start: BigInt, end: BigInt) -> napi::Result<Range<u64>> {
|
|
let start = parse_u64(start, "start")?;
|
|
let end = parse_u64(end, "end")?;
|
|
if start > end {
|
|
return Err(napi::Error::from_reason(format!(
|
|
"invalid blob range: start ({start}) > end ({end})"
|
|
)));
|
|
}
|
|
Ok(start..end)
|
|
}
|
|
|
|
pub(crate) fn parse_u64(value: BigInt, name: &str) -> napi::Result<u64> {
|
|
let (negative, value, lossless) = value.get_u64();
|
|
if negative {
|
|
return Err(napi::Error::from_reason(format!(
|
|
"{name} cannot be negative"
|
|
)));
|
|
}
|
|
if !lossless {
|
|
return Err(napi::Error::from_reason(format!(
|
|
"{name} is too large to fit in u64"
|
|
)));
|
|
}
|
|
Ok(value)
|
|
}
|
|
|
|
pub(crate) fn parse_row_ids(row_ids: Vec<BigInt>) -> napi::Result<Vec<u64>> {
|
|
row_ids
|
|
.into_iter()
|
|
.map(|id| parse_u64(id, "row id"))
|
|
.collect()
|
|
}
|
|
|
|
pub(crate) fn copy_blob_buffers(array: LargeBinaryArray) -> Vec<Option<Buffer>> {
|
|
(0..array.len())
|
|
.map(|i| {
|
|
if array.is_null(i) {
|
|
None
|
|
} else {
|
|
Some(Buffer::from(array.value(i).to_vec()))
|
|
}
|
|
})
|
|
.collect()
|
|
}
|