use std::borrow::Cow;
use std::sync::Arc;
use std::sync::atomic::{AtomicBool, AtomicU64, Ordering};
use std::time::Duration;
use anyhow::Result;
use chrono::Utc;
use surrealdb_kvs::TransactionType;
#[cfg(any(feature = "kv-surrealkv", feature = "kv-rocksdb"))]
use temp_dir::TempDir;
use tokio::time::{sleep, timeout};
use tokio_util::sync::CancellationToken;
use uuid::Uuid;
use web_time::Instant;
use super::builder::{Building, IndexKey};
use super::state::{build_owner_expired, report_status_from_phase};
use super::*;
use crate::catalog::providers::{
CatalogProvider, DatabaseProvider, NamespaceProvider, TableProvider,
};
use crate::catalog::{DatabaseId, Index, IndexDefinition, IndexId, NamespaceId, Record};
use crate::dbs::Session;
use crate::err::Error;
use crate::idx::IndexKeyBase;
use crate::idx::index::IndexOperation;
use crate::key::reclaim::{Expunge, ReclaimKind};
use crate::key::schema::{
DocKeyPrefix, DocLookupKey, DocLookupPrefix, DocPendingKey, DocPendingPrefix, EntryPrefix,
IdxRoot, IndexCountKey, IndexCountPrefix, ReclaimKey, ReclaimPrefix, RecordPrefix,
};
use crate::key::{KVKey, KVKeyDecode, KVSubspace, KVValue, Key};
use crate::kvs::index::CleanUncommittedBuild;
use crate::kvs::index::admission::CachedAdmission;
use crate::kvs::testing::{
NonRetryableErrorSite, RetryableConflictGuard, RetryableConflictSite,
inject_non_retryable_error, inject_retryable_conflict, inject_retryable_conflicts,
retryable_conflict_count,
};
use crate::kvs::tx::{
CachedIndexBuildReservationKey, CachedIndexBuildReservationLookup, IndexBuildReservationRelease,
};
use crate::kvs::{Datastore, DatastoreError, is_retryable_transaction_conflict, storage_error};
use crate::val::{RecordId, RecordIdKey, RecordIdentity, TableName, Value};
const REPEATED_RETRY_CONFLICTS: usize = 1000;
async fn new_index_test_ds() -> Result<(Arc<Datastore>, Session)> {
let ds = Datastore::builder().without_maintenance_tasks().build_with_path("memory").await?;
let session = Session::owner().with_ns("test").with_db("test");
let tx = ds.transaction(TransactionType::Write).await?;
tx.ensure_ns_db(None, "test", "test").await?;
tx.commit().await?;
Ok((ds, session))
}
#[cfg(feature = "kv-mem")]
async fn new_distributed_index_test_ds() -> Result<(Arc<Datastore>, Datastore, Session)> {
let (ds_a, session) = new_index_test_ds().await?;
let ds_b = ds_a.fork_for_test_with_node_id(uuid::Uuid::new_v4());
ds_a.insert_node().await?;
ds_b.insert_node().await?;
Ok((ds_a, ds_b, session))
}
async fn execute_all(ds: &Datastore, session: &Session, sql: &str) -> Result<()> {
for result in ds.execute(sql, session, None).await? {
result.result?;
}
Ok(())
}
async fn execute_cancelled_transaction(ds: &Datastore, session: &Session, sql: &str) -> Result<()> {
let results = ds.execute(sql, session, None).await?;
let error = results.into_iter().find_map(|result| result.result.err());
assert!(
error
.expect("transaction should be reported as cancelled")
.to_string()
.contains("cancelled transaction")
);
Ok(())
}
#[cfg(feature = "kv-mem")]
fn is_retryable_statement_conflict(err: &anyhow::Error) -> bool {
is_retryable_transaction_conflict(err)
|| err.to_string().starts_with("Transaction conflict:")
|| err.downcast_ref::<surrealdb_types::Error>().is_some_and(|err| {
matches!(
err.details(),
surrealdb_types::ErrorDetails::Query(Some(
surrealdb_types::QueryError::TransactionConflict
))
)
})
}
#[cfg(feature = "kv-mem")]
async fn execute_all_retrying_conflicts(
ds: &Datastore,
session: &Session,
sql: &str,
) -> Result<()> {
timeout(Duration::from_secs(10), async {
loop {
match execute_all(ds, session, sql).await {
Ok(()) => return Ok(()),
Err(err) if is_retryable_statement_conflict(&err) => {
sleep(Duration::from_millis(10)).await;
}
Err(err) => return Err(err),
}
}
})
.await
.map_err(|_| anyhow::anyhow!("timed out retrying statement during index build"))?
}
#[cfg(feature = "kv-mem")]
async fn execute_cancelled_transaction_retrying_conflicts(
ds: &Datastore,
session: &Session,
sql: &str,
) -> Result<()> {
timeout(Duration::from_secs(10), async {
loop {
match execute_cancelled_transaction(ds, session, sql).await {
Ok(()) => return Ok(()),
Err(err) if is_retryable_statement_conflict(&err) => {
sleep(Duration::from_millis(10)).await;
}
Err(err) => return Err(err),
}
}
})
.await
.map_err(|_| anyhow::anyhow!("timed out retrying cancelled statement during index build"))?
}
#[cfg(feature = "kv-mem")]
async fn execute_error_text_retrying_conflicts(
ds: &Datastore,
session: &Session,
sql: &str,
) -> Result<String> {
timeout(Duration::from_secs(10), async {
loop {
let results = ds.execute(sql, session, None).await?;
let error = results
.into_iter()
.find_map(|result| result.result.err())
.expect("transaction should report an error")
.to_string();
if error.starts_with("Transaction conflict:") {
sleep(Duration::from_millis(10)).await;
continue;
}
return Ok(error);
}
})
.await
.map_err(|_| anyhow::anyhow!("timed out retrying errored statement during index build"))?
}
async fn wait_for_index_ready(
ds: &Datastore,
session: &Session,
table: &str,
index: &str,
) -> Result<()> {
let sql = format!("INFO FOR INDEX {index} ON {table}");
timeout(Duration::from_secs(10), async {
loop {
let mut results = ds.execute(&sql, session, None).await?;
let value = results.remove(0).result?;
let json = value.into_json_value();
let status = json
.pointer("/building/status")
.and_then(|status| status.as_str())
.unwrap_or_default();
match status {
"ready" => return Ok(()),
"error" => anyhow::bail!("index build entered error state: {json}"),
_ => sleep(Duration::from_millis(20)).await,
}
}
})
.await
.map_err(|_| anyhow::anyhow!("timed out waiting for concurrent index build"))?
}
async fn index_building_json(
ds: &Datastore,
session: &Session,
table: &str,
index: &str,
) -> Result<serde_json::Value> {
let sql = format!("INFO FOR INDEX {index} ON {table}");
let mut results = ds.execute(&sql, session, None).await?;
let value = results.remove(0).result?;
let json = value.into_json_value();
json.get("building")
.cloned()
.ok_or_else(|| anyhow::anyhow!("index info did not include building status: {json}"))
}
async fn index_building_status(
ds: &Datastore,
session: &Session,
table: &str,
index: &str,
) -> Result<String> {
let building = index_building_json(ds, session, table, index).await?;
building
.get("status")
.and_then(|status| status.as_str())
.map(str::to_owned)
.ok_or_else(|| anyhow::anyhow!("index info did not include building.status: {building}"))
}
async fn durable_build_state(ds: &Datastore, ikb: &IndexKeyBase) -> Result<IndexBuildState> {
let tx = ds.transaction(TransactionType::Read).await?;
let state = catch!(tx, tx.get_key(&ikb.new_bs_key(), None).await)
.ok_or_else(|| anyhow::anyhow!("durable build state should exist"))?;
tx.cancel().await?;
Ok(state)
}
async fn durable_build_state_with_ticket_counter(
ds: &Datastore,
ikb: &IndexKeyBase,
) -> Result<(IndexBuildState, Option<BuildTicket>)> {
let tx = ds.transaction(TransactionType::Read).await?;
let state = catch!(tx, tx.get_key(&ikb.new_bs_key(), None).await)
.ok_or_else(|| anyhow::anyhow!("durable build state should exist"))?;
let counter = catch!(tx, tx.get_key(&ikb.new_bt_key(state.generation), None).await);
tx.cancel().await?;
Ok((state, counter))
}
async fn set_durable_build_state(
ds: &Datastore,
ikb: &IndexKeyBase,
state: IndexBuildState,
) -> Result<()> {
let tx = ds.transaction(TransactionType::Write).await?;
tx.set_key(&ikb.new_bs_key(), &state).await?;
tx.commit().await
}
fn durable_build_state_for_phase(
phase: IndexBuildPhase,
generation: BuildGeneration,
owner: Option<Uuid>,
) -> IndexBuildState {
let now = Utc::now();
IndexBuildState {
generation,
phase,
owner,
next_ticket: 0,
initial_complete: true,
updated_at: now,
owner_heartbeat_at: owner.map(|_| now),
error: None,
report_status: Some(report_status_from_phase(phase)),
initial: Some(1),
updated: Some(0),
pending: Some(0),
initial_cursor: None,
}
}
async fn durable_build_state_exists(ds: &Datastore, ikb: &IndexKeyBase) -> Result<bool> {
let tx = ds.transaction(TransactionType::Read).await?;
let state: Option<IndexBuildState> = catch!(tx, tx.get_key(&ikb.new_bs_key(), None).await);
tx.cancel().await?;
Ok(state.is_some())
}
async fn new_building_for_index(
ds: &Datastore,
session: &Session,
ns: NamespaceId,
db: DatabaseId,
table: &TableName,
ix: Arc<IndexDefinition>,
) -> Result<Building> {
let tx = ds.transaction(TransactionType::Read).await?;
let table_def = catch!(tx, tx.get_tb(ns, db, table, None).await).expect("table should exist");
let mut ctx = ds.setup_ctx()?;
let tx = Arc::new(tx);
ctx.set_transaction(Arc::clone(&tx));
let ctx = ctx.freeze();
let build = Building::new(
&ctx,
ds.transaction_factory().clone(),
ds.setup_options(session),
table_def.table_id,
Arc::clone(&ix),
Arc::new(IndexKey::new(ns, db, table, ix.index_id)),
)?;
tx.cancel().await?;
Ok(build)
}
#[cfg(feature = "kv-mem")]
async fn start_index_build_paused(
ds: &Datastore,
session: &Session,
sql: &str,
) -> Result<RetryableConflictGuard> {
let site = RetryableConflictSite::ConcurrentIndexInitialCleanup;
let node_id = ds.id();
let guard = inject_retryable_conflicts(site, node_id, REPEATED_RETRY_CONFLICTS);
execute_all(ds, session, sql).await?;
wait_for_retry_conflict(site, node_id, REPEATED_RETRY_CONFLICTS).await?;
Ok(guard)
}
#[cfg(feature = "kv-mem")]
struct PausedRemoveBuild {
ds: Arc<Datastore>,
session: Session,
guard: RetryableConflictGuard,
ns: NamespaceId,
db: DatabaseId,
table: TableName,
ix: Arc<IndexDefinition>,
builder: IndexBuilding,
}
#[cfg(feature = "kv-mem")]
async fn start_paused_remove_build() -> Result<PausedRemoveBuild> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
CREATE user:two SET email = 'two@example.com' RETURN NONE;
",
)
.await?;
let guard = start_index_build_paused(
&ds,
&session,
"DEFINE INDEX test ON user FIELDS email CONCURRENTLY",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let builder = local_builder_for_key(&ds, ns, db, &table, ix.index_id)
.await?
.expect("local builder should be running");
Ok(PausedRemoveBuild {
ds,
session,
guard,
ns,
db,
table,
ix,
builder,
})
}
#[cfg(feature = "kv-mem")]
async fn assert_cancelled_remove_keeps_local_builder(sql: &str) -> Result<()> {
let PausedRemoveBuild {
ds,
session,
guard,
ns,
db,
table,
ix,
builder,
} = start_paused_remove_build().await?;
execute_cancelled_transaction_retrying_conflicts(&ds, &session, sql).await?;
sleep(Duration::from_millis(200)).await;
assert!(
!builder.is_finished(),
"cancelled cascading remove must not abort the still-valid local builder"
);
assert!(
local_builder_for_key(&ds, ns, db, &table, ix.index_id).await?.is_some(),
"cancelled cascading remove must keep the builder map entry"
);
drop(guard);
wait_for_index_ready(&ds, &session, "user", "test").await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
async fn resume_paused_build_and_count(delete_build_state: bool) -> Result<(usize, bool)> {
let PausedRemoveBuild {
ds,
session: _session,
guard,
ns,
db,
table,
ix,
builder,
} = start_paused_remove_build().await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
assert!(
durable_build_state_exists(&ds, &ikb).await?,
"the parked build must have durable state for either arm to mean anything"
);
if delete_build_state {
let txn = ds.transaction(TransactionType::Write).await?;
catch!(txn, txn.del_key(&ikb.new_bs_key()).await);
txn.commit().await?;
assert!(
!builder.is_finished(),
"the builder must still be live when the state is deleted, or the window \
under test was not reproduced"
);
}
drop(guard);
timeout(Duration::from_secs(30), builder.wait_finished())
.await
.map_err(|_| anyhow::anyhow!("the resumed build task did not finish"))?;
Ok((
index_prefix_key_count(&ds, ns, db, &table, ix.index_id).await?,
durable_build_state_exists(&ds, &ikb).await?,
))
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn deleting_build_state_stops_a_running_builder_committing() -> Result<()> {
let (control_keys, _) = resume_paused_build_and_count(false).await?;
assert!(
control_keys > 0,
"the control arm must commit index data, or the fenced arm proves nothing"
);
let (fenced_keys, state_back) = resume_paused_build_and_count(true).await?;
assert_eq!(
fenced_keys, 0,
"a build whose durable state was deleted must not commit index data \
(control committed {control_keys})"
);
assert!(!state_back, "a stopped build must not recreate the durable state it was stopped by");
Ok(())
}
#[cfg(feature = "kv-mem")]
async fn assert_committed_remove_aborts_local_builder(sql: &str) -> Result<()> {
let PausedRemoveBuild {
ds,
session,
guard,
ns,
db,
table,
ix,
builder,
} = start_paused_remove_build().await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
seed_durable_queue_generation(&ds, &ikb, 99).await?;
assert_eq!(durable_queue_all_generations_count(&ds, &ikb).await?, 3);
execute_all_retrying_conflicts(&ds, &session, sql).await?;
wait_for_no_local_builder(&ds, ns, db, &table, ix.index_id).await?;
assert_no_index_build_artifacts(&ds, ns, db, &table, ix.index_id).await?;
drop(guard);
wait_for_finished_builder(&builder).await?;
assert_no_index_build_artifacts(&ds, ns, db, &table, ix.index_id).await?;
Ok(())
}
async fn query_array_len(ds: &Datastore, session: &Session, sql: &str) -> Result<usize> {
let mut results = ds.execute(sql, session, None).await?;
let value = results.remove(0).result?;
let surrealdb_types::Value::Array(rows) = value else {
anyhow::bail!("query returned non-array value: {value:?}");
};
Ok(rows.len())
}
async fn expect_indexed_query_len(
ds: &Datastore,
session: &Session,
sql: &str,
expected: usize,
) -> Result<()> {
let len = query_array_len(ds, session, sql).await?;
assert_eq!(len, expected, "unexpected row count for query: {sql}");
Ok(())
}
async fn get_table_index(
ds: &Datastore,
table: &str,
index: &str,
) -> Result<(NamespaceId, DatabaseId, TableName, Arc<IndexDefinition>)> {
let table = TableName::from(table);
let tx = ds.transaction(TransactionType::Read).await?;
let ns = catch!(tx, tx.get_ns_by_name("test", None).await).expect("namespace should exist");
let db =
catch!(tx, tx.get_db_by_name("test", "test", None).await).expect("database should exist");
let ix = catch!(tx, tx.expect_tb_index(ns.namespace_id, db.database_id, &table, index).await);
tx.cancel().await?;
Ok((ns.namespace_id, db.database_id, table, ix))
}
async fn get_table_ids(
ds: &Datastore,
table: &str,
) -> Result<(NamespaceId, DatabaseId, TableName)> {
let table = TableName::from(table);
let tx = ds.transaction(TransactionType::Read).await?;
let ns = catch!(tx, tx.get_ns_by_name("test", None).await).expect("namespace should exist");
let db =
catch!(tx, tx.get_db_by_name("test", "test", None).await).expect("database should exist");
catch!(tx, tx.get_tb(ns.namespace_id, db.database_id, &table, None).await)
.expect("table should exist");
tx.cancel().await?;
Ok((ns.namespace_id, db.database_id, table))
}
async fn local_builder_for_key(
ds: &Datastore,
ns: NamespaceId,
db: DatabaseId,
table: &TableName,
ix: IndexId,
) -> Result<Option<IndexBuilding>> {
let ctx = ds.setup_ctx()?;
let Some(index_builder) = ctx.get_index_builder() else {
return Ok(None);
};
let key = Arc::new(IndexKey::new(ns, db, table, ix));
Ok(index_builder.indexes.read().await.get(&key).cloned())
}
async fn wait_for_finished_builder(building: &IndexBuilding) -> Result<()> {
timeout(Duration::from_secs(10), async {
loop {
if building.is_finished() {
return Ok(());
}
sleep(Duration::from_millis(10)).await;
}
})
.await
.map_err(|_| anyhow::anyhow!("timed out waiting for the index builder task to exit"))?
}
async fn wait_for_no_local_builder(
ds: &Datastore,
ns: NamespaceId,
db: DatabaseId,
table: &TableName,
ix: IndexId,
) -> Result<()> {
timeout(Duration::from_secs(10), async {
loop {
if local_builder_for_key(ds, ns, db, table, ix).await?.is_none() {
return Ok(());
}
sleep(Duration::from_millis(10)).await;
}
})
.await
.map_err(|_| anyhow::anyhow!("timed out waiting for local index builder abort"))?
}
fn previous_index_id(ix: IndexId) -> IndexId {
assert!(ix.0 > 0, "test expected a previous allocated index id before {ix:?}");
IndexId(ix.0 - 1)
}
async fn assert_no_index_build_artifacts(
ds: &Datastore,
ns: NamespaceId,
db: DatabaseId,
table: &TableName,
ix: IndexId,
) -> Result<()> {
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix);
assert!(!durable_build_state_exists(ds, &ikb).await?);
assert_eq!(durable_queue_all_generations_count(ds, &ikb).await?, 0);
assert_eq!(
durable_ticket_counter_count(ds, &ikb).await?,
0,
"a retired or rolled-back build must not strand its generation ticket counter"
);
assert_eq!(index_prefix_key_count(ds, ns, db, table, ix).await?, 0);
assert!(local_builder_for_key(ds, ns, db, table, ix).await?.is_none());
Ok(())
}
async fn index_reclaim_modes(
ds: &Datastore,
ns: NamespaceId,
db: DatabaseId,
tb: &TableName,
ix: IndexId,
) -> Result<Vec<Expunge>> {
let tx = ds.transaction(TransactionType::Read).await?;
let entries = catch!(tx, tx.getr_raw(ReclaimPrefix {}.range()?, None).await);
tx.cancel().await?;
Ok(entries
.into_iter()
.filter_map(|(key, _)| {
let reclaim = ReclaimKey::decode_key(&key).ok()?;
(reclaim.kind == ReclaimKind::Index
&& reclaim.ns == ns
&& reclaim.db == db
&& reclaim.tb.as_ref() == tb
&& reclaim.ix == ix)
.then_some(reclaim.expunge)
})
.collect())
}
async fn index_prefix_key_count(
ds: &Datastore,
ns: NamespaceId,
db: DatabaseId,
table: &TableName,
ix: IndexId,
) -> Result<usize> {
let tx = ds.transaction(TransactionType::Read).await?;
let key = IdxRoot {
ns,
db,
tb: Cow::Borrowed(table),
ix,
};
let keys: Vec<(Vec<u8>, Vec<u8>)> = catch!(tx, tx.get_prefix_key(&key, None).await);
tx.cancel().await?;
Ok(keys.len())
}
async fn seed_durable_queue_generation(
ds: &Datastore,
ikb: &IndexKeyBase,
generation: BuildGeneration,
) -> Result<()> {
let tx = ds.transaction(TransactionType::Write).await?;
let id = RecordIdKey::from(format!("stale-{generation}"));
let ticket = generation;
let mutation_seq = 0;
tx.set_key(
&ikb.new_bg_key(generation, ticket, mutation_seq),
&Appending {
old_values: None,
new_values: None,
id: id.clone(),
count_cond_match: None,
},
)
.await?;
tx.set_key(
&ikb.new_bp_key(generation, &id),
&PrimaryAppendingTicket {
ticket,
mutation_seq,
},
)
.await?;
tx.set_key(
&ikb.new_br_key(generation, ticket),
&IndexBuildReservation {
node: ds.id(),
expires_at: Utc::now() + chrono::Duration::seconds(BUILD_RESERVATION_TTL_SECS),
},
)
.await?;
tx.commit().await?;
Ok(())
}
async fn seed_uncommitted_index_build_artifacts(
ds: &Datastore,
ns: NamespaceId,
db: DatabaseId,
table: &TableName,
ix: IndexId,
) -> Result<()> {
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix);
let tx = ds.transaction(TransactionType::Write).await?;
tx.set_key(
&ikb.new_bs_key(),
&durable_build_state_for_phase(IndexBuildPhase::Building, 1, Some(ds.id())),
)
.await?;
let id = RecordIdKey::from("orphan".to_owned());
let ticket = 1;
let mutation_seq = 0;
tx.set_key(
&ikb.new_bg_key(1, ticket, mutation_seq),
&Appending {
old_values: None,
new_values: None,
id: id.clone(),
count_cond_match: None,
},
)
.await?;
tx.set_key(
&ikb.new_bp_key(1, &id),
&PrimaryAppendingTicket {
ticket,
mutation_seq,
},
)
.await?;
tx.set_key(
&ikb.new_br_key(1, ticket),
&IndexBuildReservation {
node: ds.id(),
expires_at: Utc::now() + chrono::Duration::seconds(BUILD_RESERVATION_TTL_SECS),
},
)
.await?;
let index_data_key = IdxRoot {
ns,
db,
tb: Cow::Borrowed(table),
ix,
}
.encode_bound()?;
let mut idx_key = index_data_key.into_vec();
idx_key.extend_from_slice(b"orphan");
tx.set(Key::from(idx_key), b"orphan".to_vec()).await?;
tx.commit().await?;
Ok(())
}
async fn durable_queue_generation_count(
ds: &Datastore,
ikb: &IndexKeyBase,
generation: BuildGeneration,
) -> Result<usize> {
let tx = ds.transaction(TransactionType::Read).await?;
let bg = catch!(tx, tx.keys(ikb.new_bg_range(generation)?, u32::MAX, 0, None).await);
let bp = catch!(tx, tx.keys(ikb.new_bp_range(generation)?, u32::MAX, 0, None).await);
let br = catch!(tx, tx.keys(ikb.new_br_range(generation)?, u32::MAX, 0, None).await);
tx.cancel().await?;
Ok(bg.len() + bp.len() + br.len())
}
async fn durable_ticket_counter_count(ds: &Datastore, ikb: &IndexKeyBase) -> Result<usize> {
let tx = ds.transaction(TransactionType::Read).await?;
let bt = catch!(tx, tx.keys(ikb.new_bt_all_generations_range()?, u32::MAX, 0, None).await);
tx.cancel().await?;
Ok(bt.len())
}
async fn durable_queue_all_generations_count(ds: &Datastore, ikb: &IndexKeyBase) -> Result<usize> {
let tx = ds.transaction(TransactionType::Read).await?;
let bg = catch!(tx, tx.keys(ikb.new_bg_all_generations_range()?, u32::MAX, 0, None).await);
let bp = catch!(tx, tx.keys(ikb.new_bp_all_generations_range()?, u32::MAX, 0, None).await);
let br = catch!(tx, tx.keys(ikb.new_br_all_generations_range()?, u32::MAX, 0, None).await);
tx.cancel().await?;
Ok(bg.len() + bp.len() + br.len())
}
#[cfg(feature = "kv-mem")]
async fn expect_statement_error(ds: &Datastore, session: &Session, sql: &str) -> Result<()> {
let mut results = ds.execute(sql, session, None).await?;
let result = results.remove(0).result;
if result.is_ok() {
anyhow::bail!("statement unexpectedly succeeded: {sql}");
}
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn define_index_import_replay_preserves_existing_index_without_rebuild() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, old_ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table, old_ix.index_id);
let before = durable_build_state(&ds, &ikb).await?;
execute_all(&ds, &session, "OPTION IMPORT; DEFINE INDEX test ON user FIELDS email;").await?;
let (_, _, _, current_ix) = get_table_index(&ds, "user", "test").await?;
let after = durable_build_state(&ds, &ikb).await?;
assert_eq!(current_ix.index_id, old_ix.index_id);
assert_eq!(after.generation, before.generation);
assert_eq!(after.phase, IndexBuildPhase::Online);
assert_eq!(index_building_status(&ds, &session, "user", "test").await?, "ready");
expect_indexed_query_len(
&ds,
&session,
"SELECT * FROM user WITH INDEX test WHERE email = 'one@example.com'",
1,
)
.await?;
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn define_index_import_replay_updates_comment_without_rebuild() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email COMMENT 'old comment';
",
)
.await?;
let (ns, db, table, old_ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table, old_ix.index_id);
let before = durable_build_state(&ds, &ikb).await?;
execute_all(
&ds,
&session,
"OPTION IMPORT; DEFINE INDEX test ON user FIELDS email COMMENT 'new comment';",
)
.await?;
let (_, _, _, current_ix) = get_table_index(&ds, "user", "test").await?;
let after = durable_build_state(&ds, &ikb).await?;
assert_eq!(current_ix.index_id, old_ix.index_id);
assert_eq!(current_ix.comment.as_deref(), Some("new comment"));
assert_eq!(after.generation, before.generation);
assert_eq!(after.phase, IndexBuildPhase::Online);
expect_indexed_query_len(
&ds,
&session,
"SELECT * FROM user WITH INDEX test WHERE email = 'one@example.com'",
1,
)
.await?;
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn define_index_import_changed_definition_rebuilds_with_fresh_index_id() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com', account = 'apple' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, old_ix) = get_table_index(&ds, "user", "test").await?;
let old_ikb = IndexKeyBase::new(ns, db, table.clone(), old_ix.index_id);
let before = durable_build_state(&ds, &old_ikb).await?;
execute_all(&ds, &session, "OPTION IMPORT; DEFINE INDEX test ON user FIELDS account;").await?;
let (_, _, _, current_ix) = get_table_index(&ds, "user", "test").await?;
let current_ikb = IndexKeyBase::new(ns, db, table.clone(), current_ix.index_id);
let after = durable_build_state(&ds, ¤t_ikb).await?;
assert_ne!(current_ix.index_id, old_ix.index_id);
assert_eq!(after.generation, 1);
assert_eq!(after.phase, IndexBuildPhase::Online);
expect_indexed_query_len(
&ds,
&session,
"SELECT * FROM user WITH INDEX test WHERE account = 'apple'",
1,
)
.await?;
let tx = ds.transaction(TransactionType::Read).await?;
let old_lookup = catch!(tx, tx.get_tb_index_by_id(ns, db, &table, old_ix.index_id, None).await);
tx.cancel().await?;
assert!(old_lookup.is_none(), "old index id lookup should be retired");
assert_eq!(before.phase, IndexBuildPhase::Online);
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn remove_index_cancel_preserves_durable_ready_state() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
assert_eq!(index_building_status(&ds, &session, "user", "test").await?, "ready");
execute_cancelled_transaction(&ds, &session, "BEGIN; REMOVE INDEX test ON user; CANCEL;")
.await?;
assert_eq!(index_building_status(&ds, &session, "user", "test").await?, "ready");
expect_indexed_query_len(
&ds,
&session,
"SELECT * FROM user WITH INDEX test WHERE email = 'one@example.com'",
1,
)
.await?;
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn remove_index_committed_deletes_durable_queue_keys() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table, ix.index_id);
seed_durable_queue_generation(&ds, &ikb, 1).await?;
assert!(durable_build_state_exists(&ds, &ikb).await?);
assert_eq!(durable_queue_all_generations_count(&ds, &ikb).await?, 3);
execute_all(&ds, &session, "REMOVE INDEX test ON user").await?;
assert!(!durable_build_state_exists(&ds, &ikb).await?);
assert_eq!(durable_queue_all_generations_count(&ds, &ikb).await?, 0);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn remove_index_cancel_keeps_local_builder_running() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
CREATE user:two SET email = 'two@example.com' RETURN NONE;
",
)
.await?;
let guard = start_index_build_paused(
&ds,
&session,
"DEFINE INDEX test ON user FIELDS email CONCURRENTLY",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let builder = local_builder_for_key(&ds, ns, db, &table, ix.index_id)
.await?
.expect("local builder should be running");
execute_cancelled_transaction_retrying_conflicts(
&ds,
&session,
"BEGIN; REMOVE INDEX test ON user; CANCEL;",
)
.await?;
sleep(Duration::from_millis(200)).await;
assert!(
!builder.is_finished(),
"cancelled REMOVE INDEX must not abort the still-valid local builder"
);
assert!(
local_builder_for_key(&ds, ns, db, &table, ix.index_id).await?.is_some(),
"cancelled REMOVE INDEX must keep the builder map entry"
);
drop(guard);
wait_for_index_ready(&ds, &session, "user", "test").await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn remove_index_commit_aborts_local_builder_after_commit() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
CREATE user:two SET email = 'two@example.com' RETURN NONE;
",
)
.await?;
let guard = start_index_build_paused(
&ds,
&session,
"DEFINE INDEX test ON user FIELDS email CONCURRENTLY",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let builder = local_builder_for_key(&ds, ns, db, &table, ix.index_id)
.await?
.expect("local builder should be running");
execute_all_retrying_conflicts(&ds, &session, "REMOVE INDEX test ON user").await?;
wait_for_no_local_builder(&ds, ns, db, &table, ix.index_id).await?;
assert_no_index_build_artifacts(&ds, ns, db, &table, ix.index_id).await?;
drop(guard);
wait_for_finished_builder(&builder).await?;
assert_no_index_build_artifacts(&ds, ns, db, &table, ix.index_id).await?;
assert_eq!(
index_reclaim_modes(&ds, ns, db, &table, ix.index_id).await?.len(),
1,
"retirement must queue exactly the schema tombstone for the index data"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn remove_table_cancel_keeps_local_builder_running() -> Result<()> {
assert_cancelled_remove_keeps_local_builder("BEGIN; REMOVE TABLE user; CANCEL;").await
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn remove_database_cancel_keeps_local_builder_running() -> Result<()> {
assert_cancelled_remove_keeps_local_builder("BEGIN; REMOVE DATABASE test; CANCEL;").await
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn remove_namespace_cancel_keeps_local_builder_running() -> Result<()> {
assert_cancelled_remove_keeps_local_builder("BEGIN; REMOVE NAMESPACE test; CANCEL;").await
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn remove_table_commit_aborts_local_builder_after_commit() -> Result<()> {
assert_committed_remove_aborts_local_builder("REMOVE TABLE user").await
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn remove_database_commit_aborts_local_builder_after_commit() -> Result<()> {
assert_committed_remove_aborts_local_builder("REMOVE DATABASE test").await
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn remove_namespace_commit_aborts_local_builder_after_commit() -> Result<()> {
assert_committed_remove_aborts_local_builder("REMOVE NAMESPACE test").await
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn remove_table_expunge_aborts_local_builder_after_commit() -> Result<()> {
assert_committed_remove_aborts_local_builder("REMOVE TABLE AND EXPUNGE user").await
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn remove_database_expunge_aborts_local_builder_after_commit() -> Result<()> {
assert_committed_remove_aborts_local_builder("REMOVE DATABASE AND EXPUNGE test").await
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn remove_namespace_expunge_aborts_local_builder_after_commit() -> Result<()> {
assert_committed_remove_aborts_local_builder("REMOVE NAMESPACE AND EXPUNGE test").await
}
#[cfg(feature = "kv-surrealkv")]
async fn assert_removal_expunges_historical_build_state(sql: &str) -> Result<()> {
let dir = TempDir::new()?;
let path = format!("surrealkv://{}?versioned=true&retention=1h", dir.path().to_string_lossy());
let ds = Datastore::builder().without_maintenance_tasks().build_with_path(&path).await?;
let session = Session::owner().with_ns("test").with_db("test");
let tx = ds.transaction(TransactionType::Write).await?;
tx.ensure_ns_db(None, "test", "test").await?;
tx.commit().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let tx = ds.transaction(TransactionType::Read).await?;
assert!(tx.get_key(&ikb.new_bs_key(), None).await?.is_some());
let before_removal: u64 = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)?
.as_nanos()
.try_into()?;
assert!(tx.get_key(&ikb.new_bs_key(), Some(before_removal)).await?.is_some());
tx.cancel().await?;
execute_all(&ds, &session, sql).await?;
let tx = ds.transaction(TransactionType::Read).await?;
let historical = tx.get_key(&ikb.new_bs_key(), Some(before_removal)).await?;
tx.cancel().await?;
assert!(
historical.is_none(),
"`{sql}` must expunge the historical durable build state it retires"
);
Ok(())
}
#[cfg(feature = "kv-surrealkv")]
#[tokio::test]
async fn remove_table_expunge_removes_historical_durable_build_state() -> Result<()> {
assert_removal_expunges_historical_build_state("REMOVE TABLE AND EXPUNGE user").await
}
#[cfg(feature = "kv-surrealkv")]
#[tokio::test]
async fn remove_database_expunge_removes_historical_durable_build_state() -> Result<()> {
assert_removal_expunges_historical_build_state("REMOVE DATABASE AND EXPUNGE test").await
}
#[cfg(feature = "kv-surrealkv")]
#[tokio::test]
async fn remove_namespace_expunge_removes_historical_durable_build_state() -> Result<()> {
assert_removal_expunges_historical_build_state("REMOVE NAMESPACE AND EXPUNGE test").await
}
#[cfg(feature = "kv-surrealkv")]
#[tokio::test]
async fn expunge_removes_historical_durable_build_state() -> Result<()> {
let dir = TempDir::new()?;
let path = format!("surrealkv://{}?versioned=true&retention=1h", dir.path().to_string_lossy());
let ds = Datastore::builder().without_maintenance_tasks().build_with_path(&path).await?;
let ns = NamespaceId(1);
let db = DatabaseId(2);
let table = TableName::from("user");
let ix = IndexId(3);
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix);
let tx = ds.transaction(TransactionType::Write).await?;
tx.set_key(
&ikb.new_bs_key(),
&durable_build_state_for_phase(IndexBuildPhase::Building, 1, Some(ds.id())),
)
.await?;
tx.commit().await?;
let tx = ds.transaction(TransactionType::Read).await?;
assert!(tx.get_key(&ikb.new_bs_key(), None).await?.is_some());
let before_expunge = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)?
.as_nanos()
.try_into()?;
assert!(tx.get_key(&ikb.new_bs_key(), Some(before_expunge)).await?.is_some());
tx.cancel().await?;
let tx = ds.transaction(TransactionType::Write).await?;
retire_durable_index(&tx, ns, db, &table, ix, true).await?;
tx.commit().await?;
let tx = ds.transaction(TransactionType::Read).await?;
assert!(
tx.get_key(&ikb.new_bs_key(), Some(before_expunge)).await?.is_none(),
"EXPUNGE must remove historical durable build-state versions"
);
tx.cancel().await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn same_transaction_define_concurrent_index_write_uses_fresh_fence_snapshot() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'old@example.com' RETURN NONE;
",
)
.await?;
let site = RetryableConflictSite::ConcurrentIndexInitialCleanup;
let node_id = ds.id();
let guard = inject_retryable_conflicts(site, node_id, REPEATED_RETRY_CONFLICTS);
execute_all(
&ds,
&session,
"
BEGIN;
DEFINE INDEX test ON user FIELDS email CONCURRENTLY;
UPDATE user:one SET email = 'new@example.com' RETURN NONE;
COMMIT;
",
)
.await?;
wait_for_retry_conflict(site, node_id, REPEATED_RETRY_CONFLICTS).await?;
drop(guard);
wait_for_index_ready(&ds, &session, "user", "test").await?;
expect_indexed_query_len(
&ds,
&session,
"SELECT id FROM user WITH INDEX test WHERE email = 'new@example.com'",
1,
)
.await?;
expect_indexed_query_len(
&ds,
&session,
"SELECT id FROM user WITH INDEX test WHERE email = 'old@example.com'",
0,
)
.await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn same_transaction_overwrite_concurrent_index_write_uses_fresh_fence_snapshot() -> Result<()>
{
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'old@example.com', account = 'old-account' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let site = RetryableConflictSite::ConcurrentIndexInitialCleanup;
let node_id = ds.id();
let guard = inject_retryable_conflicts(site, node_id, REPEATED_RETRY_CONFLICTS);
execute_all(
&ds,
&session,
"
BEGIN;
DEFINE INDEX OVERWRITE test ON user FIELDS account CONCURRENTLY;
UPDATE user:one SET account = 'new-account' RETURN NONE;
COMMIT;
",
)
.await?;
wait_for_retry_conflict(site, node_id, REPEATED_RETRY_CONFLICTS).await?;
drop(guard);
wait_for_index_ready(&ds, &session, "user", "test").await?;
expect_indexed_query_len(
&ds,
&session,
"SELECT id FROM user WITH INDEX test WHERE account = 'new-account'",
1,
)
.await?;
expect_indexed_query_len(
&ds,
&session,
"SELECT id FROM user WITH INDEX test WHERE account = 'old-account'",
0,
)
.await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn define_index_overwrite_aborts_retired_local_builder_after_commit() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com', account = 'apple' RETURN NONE;
CREATE user:two SET email = 'two@example.com', account = 'banana' RETURN NONE;
",
)
.await?;
let guard = start_index_build_paused(
&ds,
&session,
"DEFINE INDEX test ON user FIELDS email CONCURRENTLY",
)
.await?;
let (ns, db, table, old_ix) = get_table_index(&ds, "user", "test").await?;
let old_builder = local_builder_for_key(&ds, ns, db, &table, old_ix.index_id)
.await?
.expect("retired local builder should be running");
execute_all_retrying_conflicts(
&ds,
&session,
"DEFINE INDEX OVERWRITE test ON user FIELDS account CONCURRENTLY",
)
.await?;
wait_for_no_local_builder(&ds, ns, db, &table, old_ix.index_id).await?;
assert_no_index_build_artifacts(&ds, ns, db, &table, old_ix.index_id).await?;
drop(guard);
wait_for_finished_builder(&old_builder).await?;
assert_no_index_build_artifacts(&ds, ns, db, &table, old_ix.index_id).await?;
wait_for_index_ready(&ds, &session, "user", "test").await?;
let (_, _, _, new_ix) = get_table_index(&ds, "user", "test").await?;
assert_ne!(new_ix.index_id, old_ix.index_id);
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn define_index_overwrite_cancel_preserves_previous_index() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com', account = 'apple' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (_, _, _, old_ix) = get_table_index(&ds, "user", "test").await?;
execute_cancelled_transaction(
&ds,
&session,
"BEGIN; DEFINE INDEX OVERWRITE test ON user FIELDS account; CANCEL;",
)
.await?;
let (_, _, _, current_ix) = get_table_index(&ds, "user", "test").await?;
assert_eq!(current_ix.index_id, old_ix.index_id);
assert_eq!(index_building_status(&ds, &session, "user", "test").await?, "ready");
expect_indexed_query_len(
&ds,
&session,
"SELECT * FROM user WITH INDEX test WHERE email = 'one@example.com'",
1,
)
.await?;
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn define_index_concurrent_cancel_cleans_uncommitted_build_artifacts() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
CREATE user:two SET email = 'two@example.com' RETURN NONE;
",
)
.await?;
execute_cancelled_transaction(
&ds,
&session,
"BEGIN; DEFINE INDEX test ON user FIELDS email CONCURRENTLY; CANCEL;",
)
.await?;
execute_all(&ds, &session, "DEFINE INDEX test ON user FIELDS email CONCURRENTLY").await?;
wait_for_index_ready(&ds, &session, "user", "test").await?;
let (ns, db, table, current_ix) = get_table_index(&ds, "user", "test").await?;
let cancelled_ix = previous_index_id(current_ix.index_id);
assert_no_index_build_artifacts(&ds, ns, db, &table, cancelled_ix).await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn remove_index_and_wait_returns_only_after_the_builder_task_exits() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
CREATE user:two SET email = 'two@example.com' RETURN NONE;
",
)
.await?;
let _guard = start_index_build_paused(
&ds,
&session,
"DEFINE INDEX test ON user FIELDS email CONCURRENTLY",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let building = local_builder_for_key(&ds, ns, db, &table, ix.index_id)
.await?
.expect("local builder should be running");
assert!(!building.is_finished(), "the paused builder should still be running");
ds.index_builder()
.remove_index_and_wait(ns, db, &table, ix.index_id, build_abort_deadline(Instant::now()))
.await;
assert!(
building.is_finished(),
"remove_index_and_wait returned while the builder task was still writing"
);
assert!(
local_builder_for_key(&ds, ns, db, &table, ix.index_id).await?.is_none(),
"the aborted builder must be gone from the local task map"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn remove_index_and_wait_stops_waiting_once_the_drain_budget_is_spent() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
",
)
.await?;
let _guard = start_index_build_paused(
&ds,
&session,
"DEFINE INDEX test ON user FIELDS email CONCURRENTLY",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let building = local_builder_for_key(&ds, ns, db, &table, ix.index_id)
.await?
.expect("local builder should be running");
let started = Instant::now();
ds.index_builder().remove_index_and_wait(ns, db, &table, ix.index_id, Instant::now()).await;
assert!(
started.elapsed() < Duration::from_secs(2),
"a spent budget must not buy another wait (took {:?})",
started.elapsed()
);
assert!(
building.aborted.load(Ordering::Relaxed),
"the builder must still be signalled to abort when the wait is skipped"
);
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn define_index_blocking_cancel_cleans_uncommitted_build_artifacts() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
CREATE user:two SET email = 'two@example.com' RETURN NONE;
",
)
.await?;
execute_cancelled_transaction(
&ds,
&session,
"BEGIN; DEFINE INDEX test ON user FIELDS email; CANCEL;",
)
.await?;
execute_all(&ds, &session, "DEFINE INDEX test ON user FIELDS email").await?;
let (ns, db, table, current_ix) = get_table_index(&ds, "user", "test").await?;
let cancelled_ix = previous_index_id(current_ix.index_id);
assert_no_index_build_artifacts(&ds, ns, db, &table, cancelled_ix).await?;
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn define_index_overwrite_cancel_cleans_new_build_artifacts() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com', account = 'apple' RETURN NONE;
CREATE user:two SET email = 'two@example.com', account = 'banana' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, old_ix) = get_table_index(&ds, "user", "test").await?;
execute_cancelled_transaction(
&ds,
&session,
"BEGIN; DEFINE INDEX OVERWRITE test ON user FIELDS account CONCURRENTLY; CANCEL;",
)
.await?;
let (_, _, _, preserved_ix) = get_table_index(&ds, "user", "test").await?;
assert_eq!(preserved_ix.index_id, old_ix.index_id);
execute_all(&ds, &session, "DEFINE INDEX OVERWRITE test ON user FIELDS account CONCURRENTLY")
.await?;
wait_for_index_ready(&ds, &session, "user", "test").await?;
let (_, _, _, current_ix) = get_table_index(&ds, "user", "test").await?;
let cancelled_ix = previous_index_id(current_ix.index_id);
assert_ne!(cancelled_ix, old_ix.index_id);
assert_no_index_build_artifacts(&ds, ns, db, &table, cancelled_ix).await?;
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn define_index_overwrite_commits_new_id_and_retires_old_lookup() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com', account = 'apple' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, old_ix) = get_table_index(&ds, "user", "test").await?;
let old_ikb = IndexKeyBase::new(ns, db, table.clone(), old_ix.index_id);
seed_durable_queue_generation(&ds, &old_ikb, 1).await?;
assert!(durable_build_state_exists(&ds, &old_ikb).await?);
assert_eq!(durable_queue_all_generations_count(&ds, &old_ikb).await?, 3);
execute_all(&ds, &session, "DEFINE INDEX OVERWRITE test ON user FIELDS account").await?;
let (_, _, _, current_ix) = get_table_index(&ds, "user", "test").await?;
assert_ne!(current_ix.index_id, old_ix.index_id);
expect_indexed_query_len(
&ds,
&session,
"SELECT * FROM user WITH INDEX test WHERE account = 'apple'",
1,
)
.await?;
let tx = ds.transaction(TransactionType::Read).await?;
let old_lookup = catch!(tx, tx.get_tb_index_by_id(ns, db, &table, old_ix.index_id, None).await);
let current_lookup =
catch!(tx, tx.get_tb_index_by_id(ns, db, &table, current_ix.index_id, None).await);
tx.cancel().await?;
assert!(old_lookup.is_none(), "old index id lookup should be retired");
assert!(current_lookup.is_some(), "current index id lookup should remain");
assert!(
!durable_build_state_exists(&ds, &old_ikb).await?,
"old durable build state should be retired"
);
assert!(
durable_build_state_exists(&ds, &IndexKeyBase::new(ns, db, table, current_ix.index_id))
.await?,
"current durable build state should remain"
);
assert_eq!(
durable_queue_all_generations_count(&ds, &old_ikb).await?,
0,
"old durable queue keys should be retired"
);
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn missing_durable_state_filters_retired_cached_index_definitions() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com', account = 'apple', name = 'one', age = 1 RETURN NONE;
DEFINE INDEX online ON user FIELDS email;
DEFINE INDEX building ON user FIELDS account;
DEFINE INDEX legacy ON user FIELDS name;
DEFINE INDEX stale ON user FIELDS age;
",
)
.await?;
let (ns, db, table, online_ix) = get_table_index(&ds, "user", "online").await?;
let (_, _, _, building_ix) = get_table_index(&ds, "user", "building").await?;
let (_, _, _, legacy_ix) = get_table_index(&ds, "user", "legacy").await?;
let (_, _, _, stale_ix) = get_table_index(&ds, "user", "stale").await?;
let online_ikb = IndexKeyBase::new(ns, db, table.clone(), online_ix.index_id);
let building_ikb = IndexKeyBase::new(ns, db, table.clone(), building_ix.index_id);
let legacy_ikb = IndexKeyBase::new(ns, db, table.clone(), legacy_ix.index_id);
let tx = ds.transaction(TransactionType::Write).await?;
tx.set_key(
&online_ikb.new_bs_key(),
&durable_build_state_for_phase(IndexBuildPhase::Online, 1, None),
)
.await?;
tx.set_key(
&building_ikb.new_bs_key(),
&durable_build_state_for_phase(IndexBuildPhase::Building, 1, Some(ds.id())),
)
.await?;
tx.del_key(&legacy_ikb.new_bs_key()).await?;
tx.commit().await?;
execute_all(&ds, &session, "DEFINE INDEX OVERWRITE stale ON user FIELDS score").await?;
let tx = ds.transaction(TransactionType::Read).await?;
let indexes: Arc<[IndexDefinition]> = Arc::from(vec![
online_ix.as_ref().clone(),
building_ix.as_ref().clone(),
legacy_ix.as_ref().clone(),
stale_ix.as_ref().clone(),
]);
let filtered = filter_online_indexes(&tx, ns, db, indexes).await?;
tx.cancel().await?;
let names: Vec<_> = filtered.iter().map(|ix| ix.name.as_str()).collect();
assert_eq!(
names,
vec!["online", "legacy"],
"only durable-online and catalog-reachable legacy indexes should remain"
);
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn filter_online_indexes_batches_durable_state_reads() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com', account = 'apple', name = 'one' RETURN NONE;
DEFINE INDEX one ON user FIELDS email;
DEFINE INDEX two ON user FIELDS account;
DEFINE INDEX three ON user FIELDS name;
",
)
.await?;
let (ns, db, table, one_ix) = get_table_index(&ds, "user", "one").await?;
let (_, _, _, two_ix) = get_table_index(&ds, "user", "two").await?;
let (_, _, _, three_ix) = get_table_index(&ds, "user", "three").await?;
let tx = ds.transaction(TransactionType::Write).await?;
for ix in [&one_ix, &two_ix, &three_ix] {
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
tx.set_key(
&ikb.new_bs_key(),
&durable_build_state_for_phase(IndexBuildPhase::Online, 1, None),
)
.await?;
}
tx.commit().await?;
let tx = ds.transaction(TransactionType::Read).await?;
let indexes: Arc<[IndexDefinition]> = Arc::from(vec![
one_ix.as_ref().clone(),
two_ix.as_ref().clone(),
three_ix.as_ref().clone(),
]);
let filtered = filter_online_indexes(&tx, ns, db, indexes).await?;
let metrics = tx.metrics_snapshot_for_test();
tx.cancel().await?;
assert_eq!(filtered.len(), 3);
assert_eq!(metrics.ops_get, 1, "durable build states should be read with one batched get");
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn consume_skips_retired_cached_index_definition() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com', account = 'apple' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (_, _, table, old_ix) = get_table_index(&ds, "user", "test").await?;
execute_all(&ds, &session, "DEFINE INDEX OVERWRITE test ON user FIELDS account").await?;
let tx = Arc::new(ds.transaction(TransactionType::Write).await?);
let db_def = tx.get_db_by_name("test", "test", None).await?.expect("database should exist");
let mut ctx = ds.setup_ctx()?;
ctx.set_transaction(Arc::clone(&tx));
let ctx = ctx.freeze();
let rid = RecordId {
table,
key: RecordIdKey::from("two".to_owned()),
};
let result = ctx
.get_index_builder()
.expect("index builder should be present")
.consume(
db_def.as_ref(),
&ctx,
&old_ix,
IndexMutation {
old_values: None,
new_values: None,
rid: &rid,
count_cond_match: None,
},
)
.await?;
tx.cancel().await?;
assert!(matches!(result, ConsumeResult::Retired));
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn consume_skips_an_index_retired_after_the_write_began() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let tx = Arc::new(ds.transaction(TransactionType::Write).await?);
let db_def = tx.get_db_by_name("test", "test", None).await?.expect("database should exist");
assert!(
tx.get_tb_index(ns, db, &table, ix.name.as_str(), None).await?.is_some(),
"the write's transaction must have the definition cached for this to mean anything"
);
execute_all(&ds, &session, "REMOVE INDEX test ON user").await?;
let mut ctx = ds.setup_ctx()?;
ctx.set_transaction(Arc::clone(&tx));
let ctx = ctx.freeze();
let rid = RecordId {
table,
key: RecordIdKey::from("two".to_owned()),
};
let result = ctx
.get_index_builder()
.expect("index builder should be present")
.consume(
db_def.as_ref(),
&ctx,
&ix,
IndexMutation {
old_values: None,
new_values: Some(vec![Value::from("two@example.com")]),
rid: &rid,
count_cond_match: None,
},
)
.await?;
tx.cancel().await?;
assert!(
matches!(result, ConsumeResult::Retired),
"a write must not index into an index retired after it began"
);
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn compaction_write_fence_rejects_committed_index_removal() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let generation = durable_build_state(&ds, &ikb).await?.generation;
let tx = ds.transaction(TransactionType::Read).await?;
let table_def = catch!(tx, tx.get_tb(ns, db, &table, None).await).expect("table should exist");
let mut ctx = ds.setup_ctx()?;
let tx = Arc::new(tx);
ctx.set_transaction(Arc::clone(&tx));
let ctx = ctx.freeze();
let build = Building::new(
&ctx,
ds.transaction_factory().clone(),
ds.setup_options(&session),
table_def.table_id,
Arc::clone(&ix),
Arc::new(IndexKey::new(ns, db, &table, ix.index_id)),
)?;
build.build_generation.store(generation, Ordering::Release);
tx.cancel().await?;
let tx = ds.transaction(TransactionType::Write).await?;
assert!(build.compaction_write_still_owns_index(&tx, generation).await?);
tx.cancel().await?;
execute_all(&ds, &session, "REMOVE INDEX test ON user").await?;
let tx = ds.transaction(TransactionType::Write).await?;
assert!(
!build.compaction_write_still_owns_index(&tx, generation).await?,
"retired indexes must not accept post-online builder compaction writes"
);
tx.cancel().await?;
Ok(())
}
#[cfg(feature = "kv-rocksdb")]
#[tokio::test(flavor = "multi_thread")]
async fn compaction_write_fence_conflicts_with_a_retirement_that_commits_after_it() -> Result<()> {
let dir = TempDir::new()?;
let ds = Datastore::builder()
.without_maintenance_tasks()
.build_with_path(&format!("rocksdb://{}", dir.path().to_string_lossy()))
.await?;
let session = Session::owner().with_ns("test").with_db("test");
let tx = ds.transaction(TransactionType::Write).await?;
tx.ensure_ns_db(None, "test", "test").await?;
tx.commit().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let state = durable_build_state(&ds, &ikb).await?;
let tx = ds.transaction(TransactionType::Write).await?;
let mut ctx = ds.setup_ctx()?;
ctx.set_transaction(Arc::new(ds.transaction(TransactionType::Read).await?));
let ctx = ctx.freeze();
let table_def = ds
.transaction(TransactionType::Read)
.await?
.get_tb(ns, db, &table, None)
.await?
.expect("table should exist");
let build = Building::new(
&ctx,
ds.transaction_factory().clone(),
ds.setup_options(&session),
table_def.table_id,
Arc::clone(&ix),
Arc::new(IndexKey::new(ns, db, &table, ix.index_id)),
)?;
assert!(build.compaction_write_still_owns_index(&tx, state.generation).await?);
execute_all(&ds, &session, "REMOVE INDEX test ON user").await?;
let mut entry = IdxRoot {
ns,
db,
tb: Cow::Borrowed(&table),
ix: ix.index_id,
}
.range()?
.start()
.to_vec();
entry.push(0x01);
tx.set(Key::from(entry), vec![0x01]).await?;
let err = tx.commit().await.expect_err("the retirement must take the compaction with it");
assert!(is_retryable_transaction_conflict(&err), "expected a retryable conflict, got: {err}");
let _ = tx.cancel().await;
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn compaction_write_fence_rejects_previous_rebuild_generation() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let generation = durable_build_state(&ds, &ikb).await?.generation;
let tx = ds.transaction(TransactionType::Read).await?;
let table_def = catch!(tx, tx.get_tb(ns, db, &table, None).await).expect("table should exist");
let mut ctx = ds.setup_ctx()?;
let tx = Arc::new(tx);
ctx.set_transaction(Arc::clone(&tx));
let ctx = ctx.freeze();
let build = Building::new(
&ctx,
ds.transaction_factory().clone(),
ds.setup_options(&session),
table_def.table_id,
Arc::clone(&ix),
Arc::new(IndexKey::new(ns, db, &table, ix.index_id)),
)?;
build.build_generation.store(generation, Ordering::Release);
tx.cancel().await?;
let tx = ds.transaction(TransactionType::Write).await?;
assert!(build.compaction_write_still_owns_index(&tx, generation).await?);
tx.cancel().await?;
execute_all(&ds, &session, "REBUILD INDEX test ON user").await?;
let (_, _, _, rebuilt_ix) = get_table_index(&ds, "user", "test").await?;
assert_eq!(rebuilt_ix.index_id, ix.index_id);
assert_eq!(rebuilt_ix.name, ix.name);
let rebuilt_state = durable_build_state(&ds, &ikb).await?;
assert_eq!(rebuilt_state.generation, generation.saturating_add(1));
assert_eq!(rebuilt_state.phase, IndexBuildPhase::Online);
let tx = ds.transaction(TransactionType::Write).await?;
assert!(
!build.compaction_write_still_owns_index(&tx, generation).await?,
"previous build generations must not accept post-online compaction writes after rebuild"
);
assert!(
build.compaction_write_still_owns_index(&tx, rebuilt_state.generation).await?,
"current online generation should still accept post-online compaction writes"
);
tx.cancel().await?;
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn fresh_build_cleans_stale_durable_queue_generations() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
for generation in 1..=3 {
seed_durable_queue_generation(&ds, &ikb, generation).await?;
}
let tx = ds.transaction(TransactionType::Write).await?;
tx.set_key(
&ikb.new_bs_key(),
&IndexBuildState {
generation: 3,
phase: IndexBuildPhase::Error,
owner: None,
next_ticket: 4,
initial_complete: false,
updated_at: Utc::now(),
owner_heartbeat_at: None,
error: Some("previous build failed".to_string()),
report_status: Some(IndexBuildReportStatus::Error),
initial: None,
updated: None,
pending: None,
initial_cursor: None,
},
)
.await?;
tx.commit().await?;
let tx = ds.transaction(TransactionType::Read).await?;
let table_def = tx.get_tb(ns, db, &table, None).await?.expect("table should exist");
let mut ctx = ds.setup_ctx()?;
let tx = Arc::new(tx);
ctx.set_transaction(Arc::clone(&tx));
let ctx = ctx.freeze();
let build = Building::new(
&ctx,
ds.transaction_factory().clone(),
ds.setup_options(&session),
table_def.table_id,
Arc::clone(&ix),
Arc::new(IndexKey::new(ns, db, &table, ix.index_id)),
)?;
let acquired = build.acquire_build_state().await?.expect("fresh build should start");
tx.cancel().await?;
assert_eq!(acquired.generation, 4);
for generation in 1..=3 {
assert_eq!(durable_queue_generation_count(&ds, &ikb, generation).await?, 0);
}
Ok(())
}
async fn count_query_value(ds: &Datastore, session: &Session, sql: &str) -> Result<i64> {
let mut results = ds.execute(sql, session, None).await?;
let value = results.remove(0).result?;
let surrealdb_types::Value::Array(rows) = value else {
anyhow::bail!("count query returned non-array value: {value:?}");
};
let Some(surrealdb_types::Value::Object(row)) = rows.first() else {
anyhow::bail!("count query returned no object row: {rows:?}");
};
let Some(surrealdb_types::Value::Number(count)) = row.get("count") else {
anyhow::bail!("count query returned no numeric count field: {row:?}");
};
count.to_int().ok_or_else(|| anyhow::anyhow!("count value is not an integer"))
}
async fn count_index_value(ds: &Datastore, session: &Session) -> Result<i64> {
count_query_value(ds, session, "SELECT count() FROM user GROUP ALL").await
}
#[cfg(feature = "kv-mem")]
async fn count_index_delta_keys(
ds: &Datastore,
ns: NamespaceId,
db: DatabaseId,
table: &TableName,
ix: IndexId,
) -> Result<usize> {
let tx = ds.transaction(TransactionType::Read).await?;
let rng = IndexCountPrefix {
ns,
db,
tb: Cow::Borrowed(table),
ix,
}
.range()?;
let keys = catch!(tx, tx.keys(rng, u32::MAX, 0, None).await);
tx.cancel().await?;
Ok(keys.len())
}
#[cfg(feature = "kv-mem")]
async fn claim_build_for_a_remote_owner(
ds: &Datastore,
ikb: &IndexKeyBase,
heartbeat_age: chrono::Duration,
) -> Result<()> {
let mut state = durable_build_state(ds, ikb).await?;
state.phase = IndexBuildPhase::Building;
state.owner = Some(Uuid::now_v7());
state.owner_heartbeat_at = Some(Utc::now() - heartbeat_age);
set_durable_build_state(ds, ikb, state).await
}
#[cfg(feature = "kv-mem")]
async fn count_hnsw_pendings(ds: &Datastore, ikb: &IndexKeyBase) -> Result<usize> {
let tx = ds.transaction(TransactionType::Read).await?;
let count = catch!(tx, tx.count(ikb.new_hr_range()?, None).await);
tx.cancel().await?;
Ok(count)
}
async fn wait_for_retry_conflict(
site: RetryableConflictSite,
node_id: uuid::Uuid,
initial_count: usize,
) -> Result<()> {
timeout(Duration::from_secs(10), async {
loop {
if retryable_conflict_count(site, node_id) < initial_count {
return Ok(());
}
sleep(Duration::from_millis(10)).await;
}
})
.await
.map_err(|_| anyhow::anyhow!("timed out waiting for injected retry conflict"))?
}
async fn wait_for_retry_conflict_count_to_stabilize(
site: RetryableConflictSite,
node_id: uuid::Uuid,
) -> Result<usize> {
timeout(Duration::from_secs(10), async {
let mut previous = retryable_conflict_count(site, node_id);
loop {
sleep(Duration::from_millis(250)).await;
let current = retryable_conflict_count(site, node_id);
if current == previous {
return Ok(current);
}
previous = current;
}
})
.await
.map_err(|_| anyhow::anyhow!("timed out waiting for retry conflict count to stabilize"))?
}
#[tokio::test(flavor = "multi_thread")]
async fn count_index_duplicate_initial_build_does_not_overcount() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:1 RETURN NONE;
CREATE user:2 RETURN NONE;
",
)
.await?;
let table_name = TableName::from("user");
let (ns_id, db_id, table_id, index) = {
let tx = ds.transaction(TransactionType::Write).await?;
let ns = tx.get_ns_by_name("test", None).await?.expect("namespace should exist");
let db = tx.get_db_by_name("test", "test", None).await?.expect("database should exist");
let table = tx
.get_tb(ns.namespace_id, db.database_id, &table_name, None)
.await?
.expect("table should exist");
let index = IndexDefinition {
index_id: IndexId(1),
name: "test".into(),
table_name: table_name.clone(),
cols: Vec::new(),
index: Index::Count(None),
count_cond: None,
comment: None,
prepare_remove: false,
format_version: 1,
};
tx.put_tb_index(ns.namespace_id, db.database_id, &table_name, &index).await?;
tx.commit().await?;
(ns.namespace_id, db.database_id, table.table_id, Arc::new(index))
};
let index_key = Arc::new(IndexKey::new(ns_id, db_id, &table_name, index.index_id));
let opt = ds.setup_options(&session);
let mut ctx = ds.setup_ctx()?;
let read_tx = Arc::new(ds.transaction(TransactionType::Read).await?);
ctx.set_transaction(Arc::clone(&read_tx));
let ctx = ctx.freeze();
let build_a = Building::new(
&ctx,
ds.transaction_factory().clone(),
opt.clone(),
table_id,
Arc::clone(&index),
Arc::clone(&index_key),
)?;
let build_b =
Building::new(&ctx, ds.transaction_factory().clone(), opt, table_id, index, index_key)?;
read_tx.cancel().await?;
let (a, b) = tokio::join!(build_a.run(), build_b.run());
a?;
b?;
assert_eq!(count_index_value(&ds, &session).await?, 2);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn count_index_initial_scan_preserves_where_count_baseline() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET status = 'active' RETURN NONE;
CREATE user:two SET status = 'active' RETURN NONE;
CREATE user:three SET status = 'inactive' RETURN NONE;
DEFINE INDEX test ON user COUNT WHERE status = 'active' CONCURRENTLY;
",
)
.await?;
wait_for_index_ready(&ds, &session, "user", "test").await?;
assert_eq!(
count_query_value(
&ds,
&session,
"SELECT count() FROM user WHERE status = 'active' GROUP ALL"
)
.await?,
2
);
execute_all_retrying_conflicts(
&ds,
&session,
"CREATE user:four SET status = 'active' RETURN NONE",
)
.await?;
assert_eq!(
count_query_value(
&ds,
&session,
"SELECT count() FROM user WHERE status = 'active' GROUP ALL"
)
.await?,
3
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn count_index_delete_before_scan_preserves_plain_count_baseline() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one RETURN NONE;
",
)
.await?;
let guard =
start_index_build_paused(&ds, &session, "DEFINE INDEX test ON user COUNT CONCURRENTLY")
.await?;
execute_all_retrying_conflicts(&ds, &session, "DELETE user:one RETURN NONE").await?;
drop(guard);
wait_for_index_ready(&ds, &session, "user", "test").await?;
execute_all_retrying_conflicts(&ds, &session, "CREATE user:two RETURN NONE").await?;
assert_eq!(count_index_value(&ds, &session).await?, 1);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn count_index_delete_before_scan_preserves_where_count_baseline() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET status = 'active' RETURN NONE;
",
)
.await?;
let guard = start_index_build_paused(
&ds,
&session,
"DEFINE INDEX test ON user COUNT WHERE status = 'active' CONCURRENTLY",
)
.await?;
execute_all_retrying_conflicts(&ds, &session, "DELETE user:one RETURN NONE").await?;
drop(guard);
wait_for_index_ready(&ds, &session, "user", "test").await?;
execute_all_retrying_conflicts(
&ds,
&session,
"CREATE user:two SET status = 'active' RETURN NONE",
)
.await?;
assert_eq!(
count_query_value(
&ds,
&session,
"SELECT count() FROM user WHERE status = 'active' GROUP ALL"
)
.await?,
1
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn count_index_updates_before_scan_preserve_where_count_baseline() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET status = 'active' RETURN NONE;
CREATE user:two SET status = 'inactive' RETURN NONE;
",
)
.await?;
let guard = start_index_build_paused(
&ds,
&session,
"DEFINE INDEX test ON user COUNT WHERE status = 'active' CONCURRENTLY",
)
.await?;
execute_all_retrying_conflicts(
&ds,
&session,
"
UPDATE user:one SET status = 'inactive' RETURN NONE;
UPDATE user:two SET status = 'active' RETURN NONE;
",
)
.await?;
drop(guard);
wait_for_index_ready(&ds, &session, "user", "test").await?;
assert_eq!(
count_query_value(
&ds,
&session,
"SELECT count() FROM user WHERE status = 'active' GROUP ALL"
)
.await?,
1
);
execute_all_retrying_conflicts(
&ds,
&session,
"CREATE user:three SET status = 'active' RETURN NONE",
)
.await?;
assert_eq!(
count_query_value(
&ds,
&session,
"SELECT count() FROM user WHERE status = 'active' GROUP ALL"
)
.await?,
2
);
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn takeover_preserves_durable_progress_counts() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
CREATE user:two SET email = 'two@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let cases = [(IndexBuildPhase::Building, 2, 42, 7), (IndexBuildPhase::Closing, 3, 84, 11)];
for (phase, generation, initial, updated) in cases {
let expired = Utc::now() - chrono::Duration::seconds(BUILD_OWNER_LEASE_SECS + 5);
let tx = ds.transaction(TransactionType::Write).await?;
tx.set_key(
&ikb.new_bs_key(),
&IndexBuildState {
generation,
phase,
owner: Some(uuid::Uuid::new_v4()),
next_ticket: 0,
initial_complete: true,
updated_at: expired,
owner_heartbeat_at: Some(expired),
error: None,
report_status: Some(IndexBuildReportStatus::Indexing),
initial: Some(initial),
updated: Some(updated),
pending: Some(5),
initial_cursor: None,
},
)
.await?;
tx.commit().await?;
let build = new_building_for_index(&ds, &session, ns, db, &table, Arc::clone(&ix)).await?;
let acquired = build
.acquire_build_state()
.await?
.expect("expired build state should be available for takeover");
build.run_acquired(acquired).await?;
let state = durable_build_state(&ds, &ikb).await?;
assert_eq!(state.phase, IndexBuildPhase::Online);
assert_eq!(state.initial, Some(initial));
assert_eq!(state.updated, Some(updated));
assert_eq!(state.pending, Some(0));
let building = index_building_json(&ds, &session, "user", "test").await?;
assert_eq!(building.get("status").and_then(|status| status.as_str()), Some("ready"));
assert_eq!(building.get("initial").and_then(|initial| initial.as_u64()), Some(initial));
assert_eq!(building.get("updated").and_then(|updated| updated.as_u64()), Some(updated));
}
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn takeover_resumes_initial_scan_from_checkpoint() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:a SET email = 'a@example.com' RETURN NONE;
CREATE user:b SET email = 'b@example.com' RETURN NONE;
CREATE user:c SET email = 'c@example.com' RETURN NONE;
CREATE user:d SET email = 'd@example.com' RETURN NONE;
CREATE user:e SET email = 'e@example.com' RETURN NONE;
CREATE user:f SET email = 'f@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let expired = Utc::now() - chrono::Duration::seconds(BUILD_OWNER_LEASE_SECS + 5);
let tx = ds.transaction(TransactionType::Write).await?;
tx.del_prefix_key(&IdxRoot {
ns,
db,
tb: Cow::Borrowed(&table),
ix: ix.index_id,
})
.await?;
tx.set_key(
&ikb.new_bs_key(),
&IndexBuildState {
generation: 2,
phase: IndexBuildPhase::Building,
owner: Some(Uuid::new_v4()),
next_ticket: 0,
initial_complete: false,
updated_at: expired,
owner_heartbeat_at: Some(expired),
error: None,
report_status: Some(IndexBuildReportStatus::Indexing),
initial: Some(42),
updated: None,
pending: None,
initial_cursor: Some(RecordIdKey::from("c".to_string())),
},
)
.await?;
tx.commit().await?;
let build = new_building_for_index(&ds, &session, ns, db, &table, Arc::clone(&ix)).await?;
let acquired = build
.acquire_build_state()
.await?
.expect("expired build state should be available for takeover");
build.run_acquired(acquired).await?;
let state = durable_build_state(&ds, &ikb).await?;
assert_eq!(state.phase, IndexBuildPhase::Online);
assert_eq!(state.initial, Some(45));
assert_eq!(state.initial_cursor, None);
assert_eq!(
index_prefix_key_count(&ds, ns, db, &table, ix.index_id).await?,
3,
"only user:d..f should be indexed after a resumed scan"
);
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn takeover_resumes_count_initial_scan_from_checkpoint() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:a RETURN NONE;
CREATE user:b RETURN NONE;
CREATE user:c RETURN NONE;
CREATE user:d RETURN NONE;
CREATE user:e RETURN NONE;
CREATE user:f RETURN NONE;
DEFINE INDEX test ON user COUNT;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let expired = Utc::now() - chrono::Duration::seconds(BUILD_OWNER_LEASE_SECS + 5);
let tx = ds.transaction(TransactionType::Write).await?;
tx.del_prefix_key(&IdxRoot {
ns,
db,
tb: Cow::Borrowed(&table),
ix: ix.index_id,
})
.await?;
tx.set_key(
&ikb.new_bs_key(),
&IndexBuildState {
generation: 2,
phase: IndexBuildPhase::Building,
owner: Some(Uuid::new_v4()),
next_ticket: 0,
initial_complete: false,
updated_at: expired,
owner_heartbeat_at: Some(expired),
error: None,
report_status: Some(IndexBuildReportStatus::Indexing),
initial: Some(4),
updated: None,
pending: None,
initial_cursor: Some(RecordIdKey::from("c".to_string())),
},
)
.await?;
tx.commit().await?;
let build = new_building_for_index(&ds, &session, ns, db, &table, Arc::clone(&ix)).await?;
let acquired = build
.acquire_build_state()
.await?
.expect("expired build state should be available for takeover");
build.run_acquired(acquired).await?;
let state = durable_build_state(&ds, &ikb).await?;
assert_eq!(state.phase, IndexBuildPhase::Online);
assert_eq!(state.initial, Some(7));
assert_eq!(state.initial_cursor, None);
assert_eq!(count_index_value(&ds, &session).await?, 3);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn count_tail_crash_after_commit_does_not_double_count_on_takeover() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one RETURN NONE;
CREATE user:two RETURN NONE;
CREATE user:three RETURN NONE;
CREATE user:four RETURN NONE;
CREATE user:five RETURN NONE;
CREATE user:six RETURN NONE;
DEFINE INDEX test ON user COUNT;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let expired = Utc::now() - chrono::Duration::seconds(BUILD_OWNER_LEASE_SECS + 5);
let tx = ds.transaction(TransactionType::Write).await?;
tx.del_prefix_key(&IdxRoot {
ns,
db,
tb: Cow::Borrowed(&table),
ix: ix.index_id,
})
.await?;
tx.set_key(
&ikb.new_bs_key(),
&IndexBuildState {
generation: 2,
phase: IndexBuildPhase::Building,
owner: Some(Uuid::new_v4()),
next_ticket: 0,
initial_complete: false,
updated_at: expired,
owner_heartbeat_at: Some(expired),
error: None,
report_status: Some(IndexBuildReportStatus::Indexing),
initial: None,
updated: None,
pending: None,
initial_cursor: None,
},
)
.await?;
tx.commit().await?;
execute_all_retrying_conflicts(&ds, &session, "DELETE user:two RETURN NONE").await?;
let _release_guard = inject_non_retryable_error(
NonRetryableErrorSite::ConcurrentIndexCountTailCommitted,
ds.id(),
);
let build = new_building_for_index(&ds, &session, ns, db, &table, Arc::clone(&ix)).await?;
let acquired = build
.acquire_build_state()
.await?
.expect("expired build state should be available for takeover");
let err = build
.run_acquired(acquired)
.await
.expect_err("builder should die at the injected crash site");
assert!(
err.to_string().contains("injected non-retryable error"),
"unexpected builder error: {err}"
);
let state = durable_build_state(&ds, &ikb).await?;
assert_eq!(state.phase, IndexBuildPhase::Building);
assert!(state.initial_complete, "tail commit must complete the scan atomically");
assert_eq!(state.initial, Some(6));
assert_eq!(state.initial_cursor, None);
let tx = ds.transaction(TransactionType::Write).await?;
let mut state = tx
.get_key(&ikb.new_bs_key(), None)
.await?
.ok_or_else(|| anyhow::anyhow!("durable build state should exist"))?;
state.updated_at = expired;
state.owner_heartbeat_at = Some(expired);
tx.set_key(&ikb.new_bs_key(), &state).await?;
tx.commit().await?;
let build = new_building_for_index(&ds, &session, ns, db, &table, Arc::clone(&ix)).await?;
let acquired = build
.acquire_build_state()
.await?
.expect("expired build state should be available for takeover");
build.run_acquired(acquired).await?;
let state = durable_build_state(&ds, &ikb).await?;
assert_eq!(state.phase, IndexBuildPhase::Online);
assert_eq!(count_index_value(&ds, &session).await?, 5);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn abort_mid_scan_never_checkpoints_unindexed_records() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE |user:1..=1000| SET email = 'user@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let expired = Utc::now() - chrono::Duration::seconds(BUILD_OWNER_LEASE_SECS + 5);
let tx = ds.transaction(TransactionType::Write).await?;
tx.del_prefix_key(&IdxRoot {
ns,
db,
tb: Cow::Borrowed(&table),
ix: ix.index_id,
})
.await?;
tx.set_key(
&ikb.new_bs_key(),
&IndexBuildState {
generation: 2,
phase: IndexBuildPhase::Building,
owner: Some(Uuid::new_v4()),
next_ticket: 0,
initial_complete: false,
updated_at: expired,
owner_heartbeat_at: Some(expired),
error: None,
report_status: Some(IndexBuildReportStatus::Indexing),
initial: Some(1),
updated: None,
pending: None,
initial_cursor: Some(RecordIdKey::from(1i64)),
},
)
.await?;
tx.commit().await?;
let building =
Arc::new(new_building_for_index(&ds, &session, ns, db, &table, Arc::clone(&ix)).await?);
let acquired = building
.acquire_build_state()
.await?
.expect("expired build state should be available for takeover");
let aborter = Arc::clone(&building);
tokio::spawn(async move {
sleep(Duration::from_millis(3)).await;
aborter.abort();
});
building.run_acquired(acquired).await?;
let state = durable_build_state(&ds, &ikb).await?;
let expected = if state.initial_complete {
999
} else {
match &state.initial_cursor {
Some(RecordIdKey::Number(n)) => usize::try_from(n - 1).unwrap(),
other => panic!("unexpected checkpoint cursor after abort: {other:?}"),
}
};
assert_eq!(
index_prefix_key_count(&ds, ns, db, &table, ix.index_id).await?,
expected,
"durable checkpoint must cover exactly the indexed records (state: {state:?})"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn abort_during_count_tail_pass_never_commits_partial_baselines() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE |user:1..=3000| RETURN NONE;
DEFINE INDEX test ON user COUNT;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let expired = Utc::now() - chrono::Duration::seconds(BUILD_OWNER_LEASE_SECS + 5);
let tx = ds.transaction(TransactionType::Write).await?;
tx.del_prefix_key(&IdxRoot {
ns,
db,
tb: Cow::Borrowed(&table),
ix: ix.index_id,
})
.await?;
tx.set_key(
&ikb.new_bs_key(),
&IndexBuildState {
generation: 2,
phase: IndexBuildPhase::Building,
owner: Some(Uuid::new_v4()),
next_ticket: 0,
initial_complete: false,
updated_at: expired,
owner_heartbeat_at: Some(expired),
error: None,
report_status: Some(IndexBuildReportStatus::Indexing),
initial: Some(500),
updated: None,
pending: None,
initial_cursor: Some(RecordIdKey::from(500i64)),
},
)
.await?;
tx.commit().await?;
execute_all_retrying_conflicts(&ds, &session, "DELETE user:501..=3000 RETURN NONE").await?;
let building =
Arc::new(new_building_for_index(&ds, &session, ns, db, &table, Arc::clone(&ix)).await?);
let acquired = building
.acquire_build_state()
.await?
.expect("expired build state should be available for takeover");
let aborter = Arc::clone(&building);
tokio::spawn(async move {
sleep(Duration::from_millis(4)).await;
aborter.abort();
});
building.run_acquired(acquired).await?;
let tx = ds.transaction(TransactionType::Read).await?;
let rng = IndexCountPrefix {
ns,
db,
tb: Cow::Borrowed(&table),
ix: ix.index_id,
}
.range()?;
let keys = catch!(tx, tx.keys(rng, u32::MAX, 0, None).await);
let mut sum: i64 = 0;
for key in &keys {
let iu = IndexCountKey::decode_key(key)?;
let delta = i64::try_from(iu.count).expect("count delta out of range");
sum += if iu.pos {
delta
} else {
-delta
};
}
let pending = catch!(tx, tx.keys(ikb.new_bg_range(2)?, u32::MAX, 0, None).await).len();
tx.cancel().await?;
let state = durable_build_state(&ds, &ikb).await?;
if state.initial_complete {
assert_eq!(
sum,
i64::try_from(pending).unwrap(),
"completion marker committed over a partial tail (state: {state:?})"
);
} else {
assert_eq!(sum, 0, "partial tail baselines committed (state: {state:?})");
assert_eq!(pending, 2500, "queued deletes must be untouched (state: {state:?})");
}
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn resume_scan_adopts_stalled_concurrent_build() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
CREATE user:two SET email = 'two@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let expired = Utc::now() - chrono::Duration::seconds(BUILD_OWNER_LEASE_SECS + 5);
let tx = ds.transaction(TransactionType::Write).await?;
tx.set_key(
&ikb.new_bs_key(),
&IndexBuildState {
generation: 2,
phase: IndexBuildPhase::Building,
owner: Some(Uuid::new_v4()),
next_ticket: 0,
initial_complete: false,
updated_at: expired,
owner_heartbeat_at: Some(expired),
error: None,
report_status: Some(IndexBuildReportStatus::Indexing),
initial: Some(0),
updated: None,
pending: None,
initial_cursor: None,
},
)
.await?;
tx.commit().await?;
let resumed = ds
.resume_stalled_index_builds(
Duration::from_secs(30),
tokio_util::sync::CancellationToken::new(),
)
.await?;
assert_eq!(resumed, 1, "scan should adopt the stalled build");
let deadline = Instant::now() + Duration::from_secs(30);
loop {
let state = durable_build_state(&ds, &ikb).await?;
if state.phase == IndexBuildPhase::Online {
break;
}
assert!(
Instant::now() < deadline,
"resumed build did not complete; phase={:?}",
state.phase
);
sleep(Duration::from_millis(100)).await;
}
let building = index_building_json(&ds, &session, "user", "test").await?;
assert_eq!(building.get("status").and_then(|status| status.as_str()), Some("ready"));
let resumed_again = ds
.resume_stalled_index_builds(
Duration::from_secs(30),
tokio_util::sync::CancellationToken::new(),
)
.await?;
assert_eq!(resumed_again, 0, "healthy index must not be re-adopted");
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn periodic_task_resumes_stalled_build() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
CREATE user:two SET email = 'two@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let expired = Utc::now() - chrono::Duration::seconds(BUILD_OWNER_LEASE_SECS + 5);
let tx = ds.transaction(TransactionType::Write).await?;
tx.set_key(
&ikb.new_bs_key(),
&IndexBuildState {
generation: 2,
phase: IndexBuildPhase::Building,
owner: Some(Uuid::new_v4()),
next_ticket: 0,
initial_complete: false,
updated_at: expired,
owner_heartbeat_at: Some(expired),
error: None,
report_status: Some(IndexBuildReportStatus::Indexing),
initial: Some(0),
updated: None,
pending: None,
initial_cursor: None,
},
)
.await?;
tx.commit().await?;
let canceller = tokio_util::sync::CancellationToken::new();
let task = {
let ds = Arc::clone(&ds);
let canceller = canceller.clone();
tokio::spawn(async move {
let tick = Duration::from_millis(200);
let mut interval = tokio::time::interval(tick);
loop {
tokio::select! {
biased;
_ = canceller.cancelled() => break,
_ = interval.tick() => {
let _ = ds.resume_stalled_index_builds(tick, canceller.clone()).await;
}
}
}
})
};
let deadline = Instant::now() + Duration::from_secs(30);
loop {
if durable_build_state(&ds, &ikb).await?.phase == IndexBuildPhase::Online {
break;
}
assert!(
Instant::now() < deadline,
"the periodic resume task did not adopt and complete the stalled build"
);
sleep(Duration::from_millis(100)).await;
}
canceller.cancel();
let _ = task.await;
let building = index_building_json(&ds, &session, "user", "test").await?;
assert_eq!(building.get("status").and_then(|status| status.as_str()), Some("ready"));
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn distributed_writer_admission_does_not_extend_builder_lease() -> Result<()> {
let (ds_a, ds_b, session) = new_distributed_index_test_ds().await?;
execute_all(
&ds_a,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:seed SET email = 'seed@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds_a, "user", "test").await?;
let generation = 2;
let expired = Utc::now() - chrono::Duration::seconds(BUILD_OWNER_LEASE_SECS + 5);
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let stale_state = IndexBuildState {
generation,
phase: IndexBuildPhase::Building,
owner: Some(uuid::Uuid::new_v4()),
next_ticket: 0,
initial_complete: false,
updated_at: expired,
owner_heartbeat_at: Some(expired),
error: None,
report_status: Some(IndexBuildReportStatus::Indexing),
initial: None,
updated: None,
pending: None,
initial_cursor: None,
};
let tx = ds_a.transaction(TransactionType::Write).await?;
tx.set_key(&ikb.new_bs_key(), &stale_state).await?;
tx.commit().await?;
for record in ["one", "two", "three"] {
execute_all(
&ds_b,
&session,
&format!("CREATE user:{record} SET email = '{record}@example.com' RETURN NONE"),
)
.await?;
}
let tx = ds_a.transaction(TransactionType::Read).await?;
let admitted_state: IndexBuildState =
tx.get_key(&ikb.new_bs_key(), None).await?.expect("build state should exist");
tx.cancel().await?;
assert_eq!(admitted_state.next_ticket, 3);
assert_eq!(admitted_state.owner_heartbeat_at, Some(expired));
assert!(
admitted_state.updated_at > expired,
"writer admission should update durable state metadata"
);
assert!(
build_owner_expired(&admitted_state, Utc::now()),
"writer admission must not refresh builder lease"
);
let tx = ds_a.transaction(TransactionType::Read).await?;
let table_def = tx.get_tb(ns, db, &table, None).await?.expect("table should exist");
let mut ctx = ds_a.setup_ctx()?;
let tx = Arc::new(tx);
ctx.set_transaction(Arc::clone(&tx));
let ctx = ctx.freeze();
let build = Building::new(
&ctx,
ds_a.transaction_factory().clone(),
ds_a.setup_options(&session),
table_def.table_id,
Arc::clone(&ix),
Arc::new(IndexKey::new(ns, db, &table, ix.index_id)),
)?;
let acquired = build
.acquire_build_state()
.await?
.expect("expired builder lease should be available for takeover");
tx.cancel().await?;
assert_eq!(acquired.generation, generation);
assert_eq!(acquired.phase, IndexBuildPhase::Building);
assert!(!acquired.initial_complete);
let tx = ds_a.transaction(TransactionType::Read).await?;
let taken_over: IndexBuildState =
tx.get_key(&ikb.new_bs_key(), None).await?.expect("build state should exist");
tx.cancel().await?;
assert_eq!(taken_over.owner, Some(build.owner));
assert!(taken_over.owner_heartbeat_at.is_some());
assert!(!build_owner_expired(&taken_over, Utc::now()));
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn writer_admission_batches_reservations_per_user_transaction() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:seed SET email = 'seed@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let generation = 2;
let building = IndexBuildState {
generation,
phase: IndexBuildPhase::Building,
owner: Some(ds.id()),
next_ticket: 0,
initial_complete: false,
updated_at: Utc::now(),
owner_heartbeat_at: Some(Utc::now()),
error: None,
report_status: Some(IndexBuildReportStatus::Indexing),
initial: None,
updated: None,
pending: None,
initial_cursor: None,
};
let tx = ds.transaction(TransactionType::Write).await?;
tx.set_key(&ikb.new_bs_key(), &building).await?;
tx.set_key(&ikb.new_bt_key(generation), &0u64).await?;
tx.commit().await?;
execute_all(
&ds,
&session,
"
BEGIN;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
CREATE user:two SET email = 'two@example.com' RETURN NONE;
CREATE user:three SET email = 'three@example.com' RETURN NONE;
CREATE user:four SET email = 'four@example.com' RETURN NONE;
CREATE user:five SET email = 'five@example.com' RETURN NONE;
COMMIT;
",
)
.await?;
assert_eq!(
durable_ticket_counter(&ds, &ikb, generation).await?,
Some(1),
"a single user transaction must consume exactly one durable ticket regardless of mutation count"
);
let tx = ds.transaction(TransactionType::Read).await?;
let bg_keys = tx.keys(ikb.new_bg_range(generation)?, u32::MAX, 0, None).await?;
let bp_keys = tx.keys(ikb.new_bp_range(generation)?, u32::MAX, 0, None).await?;
let br_keys = tx.keys(ikb.new_br_range(generation)?, u32::MAX, 0, None).await?;
tx.cancel().await?;
assert_eq!(
bg_keys.len(),
5,
"each indexed mutation should produce one `!bg` entry; got {} for 5 mutations",
bg_keys.len()
);
assert_eq!(
bp_keys.len(),
5,
"first-time-per-record admission during initial scan should produce one `!bp` per record; got {} for 5 records",
bp_keys.len()
);
assert!(
br_keys.is_empty(),
"the user transaction's close-time release should have removed the `!br`; found {} stranded reservation keys",
br_keys.len()
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn writer_admission_cancelled_batch_clears_durable_queue() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let generation = 2;
let building = IndexBuildState {
generation,
phase: IndexBuildPhase::Building,
owner: Some(ds.id()),
next_ticket: 0,
initial_complete: false,
updated_at: Utc::now(),
owner_heartbeat_at: Some(Utc::now()),
error: None,
report_status: Some(IndexBuildReportStatus::Indexing),
initial: None,
updated: None,
pending: None,
initial_cursor: None,
};
let tx = ds.transaction(TransactionType::Write).await?;
tx.set_key(&ikb.new_bs_key(), &building).await?;
tx.set_key(&ikb.new_bt_key(generation), &0u64).await?;
tx.commit().await?;
execute_cancelled_transaction(
&ds,
&session,
"
BEGIN;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
CREATE user:two SET email = 'two@example.com' RETURN NONE;
CREATE user:three SET email = 'three@example.com' RETURN NONE;
CANCEL;
",
)
.await?;
let tx = ds.transaction(TransactionType::Read).await?;
let bg_keys = tx.keys(ikb.new_bg_range(generation)?, u32::MAX, 0, None).await?;
let bp_keys = tx.keys(ikb.new_bp_range(generation)?, u32::MAX, 0, None).await?;
let br_keys = tx.keys(ikb.new_br_range(generation)?, u32::MAX, 0, None).await?;
tx.cancel().await?;
assert_eq!(
durable_ticket_counter(&ds, &ikb, generation).await?,
Some(1),
"cancelled batch still consumes one durable ticket"
);
assert!(
bg_keys.is_empty(),
"cancelled user transaction must not leave any `!bg` entries; found {}",
bg_keys.len()
);
assert!(
bp_keys.is_empty(),
"cancelled user transaction must not leave any `!bp` entries; found {}",
bp_keys.len()
);
assert!(
br_keys.is_empty(),
"cancel path must release the durable reservation; found {} stranded `!br` keys",
br_keys.len()
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn cached_index_build_reservation_remove_clears_entry() -> Result<()> {
let (ds, _) = new_index_test_ds().await?;
let tx = ds.transaction(TransactionType::Write).await?;
let key = CachedIndexBuildReservationKey {
ns: NamespaceId(1),
db: DatabaseId(1),
tb: TableName::from("user"),
ix: IndexId(1),
};
assert!(
tx.lookup_cached_index_build_reservation(&key).await?.is_none(),
"empty cache should miss"
);
let first = tx.insert_cached_index_build_reservation(key.clone(), 1, 7, false).await;
match first {
CachedIndexBuildReservationLookup::FirstUse {
generation,
ticket,
mutation_seq,
initial_complete,
} => {
assert_eq!(generation, 1);
assert_eq!(ticket, 7);
assert_eq!(mutation_seq, 0);
assert!(!initial_complete);
}
CachedIndexBuildReservationLookup::Reused {
..
} => panic!("first admission must return FirstUse, not Reused"),
}
let reused = tx
.lookup_cached_index_build_reservation(&key)
.await?
.expect("cache should hit after insert");
match reused {
CachedIndexBuildReservationLookup::Reused {
generation,
ticket,
mutation_seq,
initial_complete,
} => {
assert_eq!(generation, 1);
assert_eq!(ticket, 7);
assert_eq!(mutation_seq, 1, "second mutation should consume seq 1");
assert!(!initial_complete);
}
CachedIndexBuildReservationLookup::FirstUse {
..
} => panic!("subsequent admission must return Reused, not FirstUse"),
}
tx.remove_cached_index_build_reservation(&key).await;
assert!(
tx.lookup_cached_index_build_reservation(&key).await?.is_none(),
"cache should miss after removal so subsequent mutations re-enter reservation"
);
tx.cancel().await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn cached_index_build_reservation_lookup_errors_on_seq_overflow() -> Result<()> {
let (ds, _) = new_index_test_ds().await?;
let tx = ds.transaction(TransactionType::Write).await?;
let key = CachedIndexBuildReservationKey {
ns: NamespaceId(1),
db: DatabaseId(1),
tb: TableName::from("user"),
ix: IndexId(1),
};
tx.seed_cached_index_build_reservation_for_test(key.clone(), 1, 0, false, u32::MAX).await;
let err = tx
.lookup_cached_index_build_reservation(&key)
.await
.expect_err("lookup at u32::MAX must surface an overflow error");
let downcast = err
.downcast_ref::<DatastoreError>()
.expect("error should be the typed IndexingBuildingCancelled");
assert!(
matches!(downcast, DatastoreError::IndexingBuildingCancelled { .. }),
"expected IndexingBuildingCancelled, got {downcast:?}"
);
assert!(
err.to_string().contains("mutation sequence overflowed"),
"unexpected error message: {err}"
);
tx.cancel().await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn recheck_cached_admission_rejects_mid_transaction_state_changes() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
seed_build_state(&ds, &ikb, IndexBuildPhase::Building, 1).await?;
let outcome = run_recheck_cached_admission(&ds, ns, db, &ikb, ix.as_ref(), 1)
.await?
.expect("recheck on matching Building state should succeed");
assert!(matches!(outcome, CachedAdmission::Valid));
seed_build_state(&ds, &ikb, IndexBuildPhase::Closing, 1).await?;
let outcome = run_recheck_cached_admission(&ds, ns, db, &ikb, ix.as_ref(), 1)
.await?
.expect("recheck on matching Closing state should succeed");
assert!(matches!(outcome, CachedAdmission::Valid));
seed_build_state(&ds, &ikb, IndexBuildPhase::Building, 2).await?;
let err = run_recheck_cached_admission(&ds, ns, db, &ikb, ix.as_ref(), 1)
.await?
.expect_err("generation mismatch must abort cached admission");
assert!(err.to_string().contains("generation"), "unexpected error: {err}");
seed_build_state(&ds, &ikb, IndexBuildPhase::Online, 1).await?;
let err = run_recheck_cached_admission(&ds, ns, db, &ikb, ix.as_ref(), 1)
.await?
.expect_err("online phase must abort cached admission");
assert!(err.to_string().contains("online"), "unexpected error: {err}");
seed_build_state(&ds, &ikb, IndexBuildPhase::Error, 1).await?;
let outcome = run_recheck_cached_admission(&ds, ns, db, &ikb, ix.as_ref(), 1)
.await?
.expect("recheck on matching Error state should keep queueing");
assert!(matches!(outcome, CachedAdmission::Valid));
let tx = ds.transaction(TransactionType::Write).await?;
tx.del_key(&ikb.new_bs_key()).await?;
tx.commit().await?;
let err = run_recheck_cached_admission(&ds, ns, db, &ikb, ix.as_ref(), 1)
.await?
.expect_err("missing build state must abort cached admission");
assert!(err.to_string().contains("no longer exists"), "unexpected error: {err}");
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn recheck_cached_admission_skips_a_retired_index_instead_of_failing_the_write() -> Result<()>
{
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
seed_build_state(&ds, &ikb, IndexBuildPhase::Building, 1).await?;
let cached_ix = Arc::clone(&ix);
execute_all(&ds, &session, "REMOVE INDEX test ON user").await?;
let outcome = run_recheck_cached_admission(&ds, ns, db, &ikb, cached_ix.as_ref(), 1)
.await?
.expect("a retired index must not fail the write that was mid-transaction");
assert!(
matches!(outcome, CachedAdmission::Retired),
"the recheck must report the index retired so the mutation skips it"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
async fn seed_build_state(
ds: &Datastore,
ikb: &IndexKeyBase,
phase: IndexBuildPhase,
generation: u64,
) -> Result<()> {
let mut state = IndexBuildState {
generation,
phase,
owner: Some(ds.id()),
next_ticket: 0,
initial_complete: false,
updated_at: Utc::now(),
owner_heartbeat_at: Some(Utc::now()),
error: None,
report_status: Some(report_status_from_phase(phase)),
initial: None,
updated: None,
pending: None,
initial_cursor: None,
};
if matches!(phase, IndexBuildPhase::Error) {
state.error = Some("seeded test failure".to_string());
state.report_status = Some(IndexBuildReportStatus::Error);
}
let tx = ds.transaction(TransactionType::Write).await?;
tx.set_key(&ikb.new_bs_key(), &state).await?;
tx.commit().await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn recheck_cached_admission_sees_a_retirement_the_user_transaction_cannot() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
seed_build_state(&ds, &ikb, IndexBuildPhase::Building, 1).await?;
let tx = Arc::new(ds.transaction(TransactionType::Read).await?);
assert!(
tx.get_tb_index(ns, db, &table, ix.name.as_str(), None).await?.is_some(),
"the write's transaction must have the definition cached for this to mean anything"
);
execute_all(&ds, &session, "REMOVE INDEX test ON user").await?;
let mut ctx = ds.setup_ctx()?;
ctx.set_transaction(Arc::clone(&tx));
let frozen = ctx.freeze();
let builder = frozen
.get_index_builder()
.expect("index builder should be available on the configured Datastore")
.clone();
let outcome = builder.recheck_cached_admission(&frozen, &ikb, ix.as_ref(), ns, db, 1).await;
tx.cancel().await?;
assert!(
matches!(outcome?, CachedAdmission::Retired),
"a retirement that commits mid-write must still be reported as retired"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn recheck_cached_admission_rejects_a_replacement_the_write_cannot_see() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com', account = 'apple' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, old_ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), old_ix.index_id);
seed_build_state(&ds, &ikb, IndexBuildPhase::Building, 1).await?;
let tx = Arc::new(ds.transaction(TransactionType::Read).await?);
assert!(
tx.get_tb_index(ns, db, &table, old_ix.name.as_str(), None).await?.is_some(),
"the write's transaction must have the old definition cached"
);
execute_all(&ds, &session, "DEFINE INDEX OVERWRITE test ON user FIELDS account").await?;
let (_, _, _, new_ix) = get_table_index(&ds, "user", "test").await?;
assert_ne!(new_ix.index_id, old_ix.index_id, "the overwrite must allocate a fresh id");
let mut ctx = ds.setup_ctx()?;
ctx.set_transaction(Arc::clone(&tx));
let frozen = ctx.freeze();
let builder = frozen
.get_index_builder()
.expect("index builder should be available on the configured Datastore")
.clone();
let outcome = builder.recheck_cached_admission(&frozen, &ikb, old_ix.as_ref(), ns, db, 1).await;
tx.cancel().await?;
let err = outcome.expect_err("a replacement the write cannot see must abort it");
assert!(err.to_string().contains("replaced"), "unexpected error: {err}");
Ok(())
}
#[cfg(feature = "kv-mem")]
async fn run_recheck_cached_admission(
ds: &Datastore,
ns: NamespaceId,
db: DatabaseId,
ikb: &IndexKeyBase,
ix: &IndexDefinition,
cached_generation: u64,
) -> Result<Result<CachedAdmission>> {
let tx = ds.transaction(TransactionType::Read).await?;
let mut ctx = ds.setup_ctx()?;
let tx = Arc::new(tx);
ctx.set_transaction(Arc::clone(&tx));
let frozen = ctx.freeze();
let builder = frozen
.get_index_builder()
.expect("index builder should be available on the configured Datastore")
.clone();
let result =
builder.recheck_cached_admission(&frozen, ikb, ix, ns, db, cached_generation).await;
tx.cancel().await?;
Ok(result)
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn acquire_build_state_waits_for_prior_generation_reservations() -> Result<()> {
let (ds_a, ds_b, session) = new_distributed_index_test_ds().await?;
execute_all(
&ds_a,
&session,
"
DEFINE TABLE user SCHEMALESS;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds_a, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let errored = IndexBuildState {
generation: 1,
phase: IndexBuildPhase::Error,
owner: Some(ds_a.id()),
next_ticket: 1,
initial_complete: false,
updated_at: Utc::now(),
owner_heartbeat_at: Some(Utc::now()),
error: Some("seeded test failure".to_string()),
report_status: Some(IndexBuildReportStatus::Error),
initial: None,
updated: None,
pending: None,
initial_cursor: None,
};
let seed_tx = ds_a.transaction(TransactionType::Write).await?;
seed_tx.set_key(&ikb.new_bs_key(), &errored).await?;
let reservation = IndexBuildReservation {
node: ds_a.id(),
expires_at: Utc::now() + chrono::Duration::seconds(BUILD_RESERVATION_TTL_SECS),
};
let br_key = ikb.new_br_key(1, 0);
seed_tx.set_key(&br_key, &reservation).await?;
seed_tx.commit().await?;
let building = new_building_for_index(&ds_b, &session, ns, db, &table, Arc::clone(&ix)).await?;
let timeout_result =
timeout(Duration::from_millis(200), building.wait_for_prior_generation_reservations(2))
.await;
assert!(
timeout_result.is_err(),
"drain must block while another node's !br is alive in durable membership"
);
let release_tx = ds_a.transaction(TransactionType::Write).await?;
release_tx.del_key(&br_key).await?;
release_tx.commit().await?;
timeout(Duration::from_secs(5), building.wait_for_prior_generation_reservations(2))
.await
.expect("drain must complete after !br is removed")?;
let acquired =
building.acquire_build_state().await?.expect("takeover should succeed after drain");
assert_eq!(acquired.generation, 2);
assert!(matches!(acquired.phase, IndexBuildPhase::Building));
let tx = ds_a.transaction(TransactionType::Read).await?;
let br_keys = tx.keys(ikb.new_br_all_generations_range()?, u32::MAX, 0, None).await?;
tx.cancel().await?;
assert!(br_keys.is_empty(), "no `!br` should remain after a clean takeover");
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn drain_prior_generation_reservations_cleans_dead_writers() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let stale_node = Uuid::new_v4();
let reservation = IndexBuildReservation {
node: stale_node,
expires_at: Utc::now() - chrono::Duration::seconds(1),
};
let br_key = ikb.new_br_key(1, 0);
let seed_tx = ds.transaction(TransactionType::Write).await?;
seed_tx.set_key(&br_key, &reservation).await?;
seed_tx.commit().await?;
let building = new_building_for_index(&ds, &session, ns, db, &table, Arc::clone(&ix)).await?;
timeout(Duration::from_secs(5), building.wait_for_prior_generation_reservations(2))
.await
.expect("drain must complete promptly for a dead writer's reservation")?;
let tx = ds.transaction(TransactionType::Read).await?;
let br_keys = tx.keys(ikb.new_br_all_generations_range()?, u32::MAX, 0, None).await?;
tx.cancel().await?;
assert!(br_keys.is_empty(), "stale `!br` from a dead writer should have been removed");
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn distributed_info_uses_durable_state_from_second_node() -> Result<()> {
let (ds_a, ds_b, session) = new_distributed_index_test_ds().await?;
execute_all(
&ds_a,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
CREATE user:two SET email = 'two@example.com' RETURN NONE;
",
)
.await?;
let guard = start_index_build_paused(
&ds_a,
&session,
"DEFINE INDEX test ON user FIELDS email CONCURRENTLY",
)
.await?;
assert_eq!(index_building_status(&ds_b, &session, "user", "test").await?, "cleaning");
drop(guard);
wait_for_index_ready(&ds_a, &session, "user", "test").await?;
assert_eq!(index_building_status(&ds_b, &session, "user", "test").await?, "ready");
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn distributed_blocking_rebuild_waits_for_remote_owner() -> Result<()> {
let (ds_a, ds_b, session) = new_distributed_index_test_ds().await?;
execute_all(
&ds_a,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
CREATE user:two SET email = 'two@example.com' RETURN NONE;
",
)
.await?;
let guard = start_index_build_paused(
&ds_a,
&session,
"DEFINE INDEX test ON user FIELDS email CONCURRENTLY",
)
.await?;
let session_rebuild = session.clone();
let rebuild = tokio::spawn(async move {
execute_all(&ds_b, &session_rebuild, "REBUILD INDEX test ON user").await
});
sleep(Duration::from_millis(200)).await;
assert!(
!rebuild.is_finished(),
"blocking REBUILD INDEX returned while the remote build was still paused"
);
drop(guard);
timeout(Duration::from_secs(10), rebuild)
.await
.map_err(|_| anyhow::anyhow!("timed out waiting for blocking rebuild"))???;
assert_eq!(index_building_status(&ds_a, &session, "user", "test").await?, "ready");
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn distributed_blocking_rebuild_takes_over_expired_remote_owner() -> Result<()> {
let (ds_a, ds_b, session) = new_distributed_index_test_ds().await?;
execute_all(
&ds_a,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds_a, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table, ix.index_id);
let generation = 2;
let now = Utc::now();
let tx = ds_a.transaction(TransactionType::Write).await?;
tx.set_key(
&ikb.new_bs_key(),
&IndexBuildState {
generation,
phase: IndexBuildPhase::Building,
owner: Some(uuid::Uuid::new_v4()),
next_ticket: 0,
initial_complete: true,
updated_at: now,
owner_heartbeat_at: Some(now),
error: None,
report_status: Some(IndexBuildReportStatus::Indexing),
initial: Some(1),
updated: Some(0),
pending: Some(0),
initial_cursor: None,
},
)
.await?;
tx.commit().await?;
let session_rebuild = session.clone();
let rebuild = tokio::spawn(async move {
execute_all(&ds_b, &session_rebuild, "REBUILD INDEX test ON user").await
});
sleep(Duration::from_millis(200)).await;
assert!(
!rebuild.is_finished(),
"blocking REBUILD INDEX returned before the remote lease expired"
);
let expired = Utc::now() - chrono::Duration::seconds(BUILD_OWNER_LEASE_SECS + 5);
let tx = ds_a.transaction(TransactionType::Write).await?;
let mut state: IndexBuildState =
tx.get_key(&ikb.new_bs_key(), None).await?.expect("build state should exist");
state.updated_at = expired;
state.owner_heartbeat_at = Some(expired);
tx.set_key(&ikb.new_bs_key(), &state).await?;
tx.commit().await?;
timeout(Duration::from_secs(10), rebuild)
.await
.map_err(|_| anyhow::anyhow!("timed out waiting for takeover rebuild"))???;
let state = durable_build_state(&ds_a, &ikb).await?;
assert_eq!(state.phase, IndexBuildPhase::Online);
assert_eq!(state.generation, generation);
assert_eq!(state.owner, None);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn distributed_info_reports_durable_error_from_second_node() -> Result<()> {
let (ds_a, ds_b, session) = new_distributed_index_test_ds().await?;
execute_all(
&ds_a,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET account = 'apple', email = 'test@surrealdb.com' RETURN NONE;
CREATE user:two SET account = 'apple', email = 'test@surrealdb.com' RETURN NONE;
",
)
.await?;
execute_all(
&ds_a,
&session,
"DEFINE INDEX test ON user FIELDS account, email UNIQUE CONCURRENTLY",
)
.await?;
let building = timeout(Duration::from_secs(10), async {
loop {
let building = index_building_json(&ds_b, &session, "user", "test").await?;
if building.get("status").and_then(|status| status.as_str()) == Some("error") {
return Ok::<_, anyhow::Error>(building);
}
sleep(Duration::from_millis(20)).await;
}
})
.await
.map_err(|_| anyhow::anyhow!("timed out waiting for durable index build error"))??;
assert_eq!(building.get("status").and_then(|status| status.as_str()), Some("error"));
let error_reason = building
.get("error")
.and_then(|error| error.as_str())
.ok_or_else(|| anyhow::anyhow!("index info did not include building.error: {building}"))?;
assert!(
error_reason.contains("already contains"),
"unexpected durable index build error: {building}"
);
execute_all_retrying_conflicts(
&ds_b,
&session,
"CREATE user:three SET account = 'tesla', email = 'three@surrealdb.com' RETURN NONE",
)
.await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn distributed_standard_index_replays_second_node_update() -> Result<()> {
let (ds_a, ds_b, session) = new_distributed_index_test_ds().await?;
execute_all(
&ds_a,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'old@example.com' RETURN NONE;
CREATE user:two SET email = 'steady@example.com' RETURN NONE;
",
)
.await?;
let guard = start_index_build_paused(
&ds_a,
&session,
"DEFINE INDEX test ON user FIELDS email CONCURRENTLY",
)
.await?;
execute_all_retrying_conflicts(
&ds_b,
&session,
"UPDATE user:one SET email = 'new@example.com' RETURN NONE",
)
.await?;
execute_all_retrying_conflicts(&ds_b, &session, "DELETE user:two RETURN NONE").await?;
drop(guard);
wait_for_index_ready(&ds_a, &session, "user", "test").await?;
let building = index_building_json(&ds_a, &session, "user", "test").await?;
assert_eq!(building.get("pending").and_then(|pending| pending.as_u64()), Some(0));
let updated =
building.get("updated").and_then(|updated| updated.as_u64()).ok_or_else(|| {
anyhow::anyhow!("index info did not include building.updated: {building}")
})?;
assert!(
updated >= 2,
"expected queued update and delete to be replayed in index status: {building}"
);
assert_eq!(
query_array_len(
&ds_a,
&session,
"SELECT id FROM user WITH INDEX test WHERE email = 'new@example.com'"
)
.await?,
1
);
assert_eq!(
query_array_len(
&ds_a,
&session,
"SELECT id FROM user WITH INDEX test WHERE email = 'old@example.com'"
)
.await?,
0
);
assert_eq!(
query_array_len(
&ds_a,
&session,
"SELECT id FROM user WITH INDEX test WHERE email = 'steady@example.com'"
)
.await?,
0
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn distributed_local_reservation_release_retries_conflict() -> Result<()> {
let (ds_a, ds_b, session) = new_distributed_index_test_ds().await?;
execute_all(
&ds_a,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'old@example.com' RETURN NONE;
",
)
.await?;
let build_guard = start_index_build_paused(
&ds_a,
&session,
"DEFINE INDEX test ON user FIELDS email CONCURRENTLY",
)
.await?;
let release_site = RetryableConflictSite::ConcurrentIndexReservationRelease;
let writer_node = ds_b.id();
let _release_guard = inject_retryable_conflict(release_site, writer_node);
execute_all_retrying_conflicts(
&ds_b,
&session,
"UPDATE user:one SET email = 'new@example.com' RETURN NONE",
)
.await?;
assert_eq!(retryable_conflict_count(release_site, writer_node), 0);
drop(build_guard);
wait_for_index_ready(&ds_a, &session, "user", "test").await?;
expect_indexed_query_len(
&ds_a,
&session,
"SELECT id FROM user WITH INDEX test WHERE email = 'new@example.com'",
1,
)
.await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn commit_failure_preserves_primary_error_when_reservation_cleanup_fails() -> Result<()> {
let (ds, _) = new_index_test_ds().await?;
let ikb = IndexKeyBase::new(NamespaceId(1), DatabaseId(1), TableName::from("user"), IndexId(1));
let generation = 1;
let ticket = 1;
let reservation = IndexBuildReservation {
node: ds.id(),
expires_at: Utc::now() + chrono::Duration::seconds(BUILD_RESERVATION_TTL_SECS),
};
let br = ikb.new_br_key(generation, ticket);
let br_key = br.encode_key()?;
let br_val = reservation.kv_encode_value()?;
let seed_tx = ds.transaction(TransactionType::Write).await?;
seed_tx.set_key(&br, &reservation).await?;
(*seed_tx).set("commit-conflict".as_bytes().into(), b"initial".as_slice()).await?;
seed_tx.commit().await?;
let user_tx = ds.transaction(TransactionType::Write).await?;
(*user_tx).set("commit-conflict".as_bytes().into(), b"user-write".as_slice()).await?;
user_tx
.register_index_build_reservation_release(IndexBuildReservationRelease::new(
ds.transaction_factory().clone(),
ds.sequences().clone(),
ds.id(),
br_key,
br_val,
))
.await;
let conflicting_tx = ds.transaction(TransactionType::Write).await?;
(*conflicting_tx)
.set("commit-conflict".as_bytes().into(), b"conflicting-write".as_slice())
.await?;
conflicting_tx.commit().await?;
let _release_guard = inject_non_retryable_error(
NonRetryableErrorSite::ConcurrentIndexReservationRelease,
ds.id(),
);
let err = user_tx
.commit()
.await
.expect_err("commit conflict should remain visible when cleanup also fails");
assert!(
is_retryable_transaction_conflict(&err),
"primary commit error was not preserved: {err}"
);
assert!(
!err.to_string().contains("injected non-retryable error"),
"cleanup error replaced primary commit error: {err}"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn commit_failure_cleans_uncommitted_index_build_artifacts() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
",
)
.await?;
let (ns, db, table) = get_table_ids(&ds, "user").await?;
let ix = IndexId(42);
seed_uncommitted_index_build_artifacts(&ds, ns, db, &table, ix).await?;
let user_tx = ds.transaction(TransactionType::Write).await?;
(*user_tx).set("commit-conflict".as_bytes().into(), b"user-write".as_slice()).await?;
let ctx = ds.setup_ctx()?;
let builder = ctx.get_index_builder().expect("index builder should exist").clone();
user_tx
.on_rollback(CleanUncommittedBuild::boxed(
builder.clone(),
builder.transaction_factory(),
user_tx.sequences(),
ns,
db,
table.clone(),
ix,
))
.await;
let conflicting_tx = ds.transaction(TransactionType::Write).await?;
(*conflicting_tx)
.set("commit-conflict".as_bytes().into(), b"conflicting-write".as_slice())
.await?;
conflicting_tx.commit().await?;
let err = user_tx.commit().await.expect_err("commit conflict should remain visible");
assert!(
is_retryable_transaction_conflict(&err),
"primary commit error was not preserved: {err}"
);
assert_no_index_build_artifacts(&ds, ns, db, &table, ix).await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn store_changes_failure_preserves_primary_error_when_cleanup_fails() -> Result<()> {
let (ds, _) = new_index_test_ds().await?;
let ikb = IndexKeyBase::new(NamespaceId(1), DatabaseId(1), TableName::from("user"), IndexId(1));
let generation = 1;
let ticket = 1;
let reservation = IndexBuildReservation {
node: ds.id(),
expires_at: Utc::now() + chrono::Duration::seconds(BUILD_RESERVATION_TTL_SECS),
};
let br = ikb.new_br_key(generation, ticket);
let br_key = br.encode_key()?;
let br_val = reservation.kv_encode_value()?;
let seed_tx = ds.transaction(TransactionType::Write).await?;
seed_tx.set_key(&br, &reservation).await?;
seed_tx.commit().await?;
let user_tx = ds.transaction(TransactionType::Read).await?;
let table = TableName::from("user");
let record = RecordId {
table: table.clone(),
key: RecordIdKey::from("one".to_owned()),
};
let current: Value = "new@example.com".into();
user_tx.changefeed_buffer_record_change(
NamespaceId(1),
DatabaseId(1),
&table,
&record,
Record::new(Value::None).into_read_only(),
Record::new(current).into_read_only(),
false,
);
user_tx
.register_index_build_reservation_release(IndexBuildReservationRelease::new(
ds.transaction_factory().clone(),
ds.sequences().clone(),
ds.id(),
br_key,
br_val,
))
.await;
let _release_guard = inject_non_retryable_error(
NonRetryableErrorSite::ConcurrentIndexReservationRelease,
ds.id(),
);
let err = user_tx
.commit()
.await
.expect_err("store_changes failure should remain visible when cleanup also fails");
assert!(
matches!(storage_error(&err), Some(crate::kvs::Error::TransactionReadonly)),
"primary store_changes error was not preserved: {err}"
);
assert!(
!err.to_string().contains("injected non-retryable error"),
"cleanup error replaced primary store_changes error: {err}"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn distributed_committed_cleanup_failure_recovers_from_appending() -> Result<()> {
let (ds_a, ds_b, session) = new_distributed_index_test_ds().await?;
execute_all(
&ds_a,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'old@example.com' RETURN NONE;
",
)
.await?;
let build_guard = start_index_build_paused(
&ds_a,
&session,
"DEFINE INDEX test ON user FIELDS email CONCURRENTLY",
)
.await?;
let _release_guard = inject_non_retryable_error(
NonRetryableErrorSite::ConcurrentIndexReservationRelease,
ds_b.id(),
);
execute_all_retrying_conflicts(
&ds_b,
&session,
"UPDATE user:one SET email = 'new@example.com' RETURN NONE",
)
.await?;
drop(build_guard);
wait_for_index_ready(&ds_a, &session, "user", "test").await?;
expect_indexed_query_len(
&ds_a,
&session,
"SELECT id FROM user WITH INDEX test WHERE email = 'new@example.com'",
1,
)
.await?;
expect_indexed_query_len(
&ds_a,
&session,
"SELECT id FROM user WITH INDEX test WHERE email = 'old@example.com'",
0,
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds_a, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table, ix.index_id);
assert_eq!(durable_queue_all_generations_count(&ds_a, &ikb).await?, 0);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn distributed_writer_admission_honors_statement_timeout_while_closing() -> Result<()> {
let (ds_a, ds_b, session) = new_distributed_index_test_ds().await?;
execute_all(
&ds_a,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'old@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds_a, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table, ix.index_id);
set_durable_build_state(
&ds_a,
&ikb,
durable_build_state_for_phase(IndexBuildPhase::Closing, 2, Some(ds_a.id())),
)
.await?;
let started = Instant::now();
let mut results = timeout(
Duration::from_secs(5),
ds_b.execute("UPDATE user:one SET email = 'new@example.com' TIMEOUT 50ms", &session, None),
)
.await
.map_err(|_| anyhow::anyhow!("writer admission ignored the statement timeout"))??;
let error = results
.remove(0)
.result
.expect_err("write should time out while durable state is Closing")
.to_string();
assert!(
started.elapsed() < Duration::from_secs(5),
"write waited for an internal timeout instead of the statement timeout"
);
assert!(error.contains("exceeded the timeout: 50ms"), "unexpected timeout error: {error}");
assert_eq!(durable_build_state(&ds_a, &ikb).await?.phase, IndexBuildPhase::Closing);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn distributed_writer_admission_waits_until_closing_online() -> Result<()> {
let (ds_a, ds_b, session) = new_distributed_index_test_ds().await?;
execute_all(
&ds_a,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'old@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds_a, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table, ix.index_id);
set_durable_build_state(
&ds_a,
&ikb,
durable_build_state_for_phase(IndexBuildPhase::Closing, 2, Some(ds_a.id())),
)
.await?;
let session_write = session.clone();
let writer = tokio::spawn(async move {
execute_all(
&ds_b,
&session_write,
"UPDATE user:one SET email = 'new@example.com' RETURN NONE",
)
.await
});
sleep(Duration::from_millis(250)).await;
assert!(!writer.is_finished(), "write returned before durable Closing became Online");
set_durable_build_state(
&ds_a,
&ikb,
durable_build_state_for_phase(IndexBuildPhase::Online, 2, None),
)
.await?;
timeout(Duration::from_secs(10), writer)
.await
.map_err(|_| anyhow::anyhow!("timed out waiting for Closing admission write"))???;
expect_indexed_query_len(
&ds_a,
&session,
"SELECT id FROM user WITH INDEX test WHERE email = 'new@example.com'",
1,
)
.await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn distributed_rolled_back_writer_releases_reservation_after_close() -> Result<()> {
let (ds_a, ds_b, session) = new_distributed_index_test_ds().await?;
execute_all(
&ds_a,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'old@example.com' RETURN NONE;
",
)
.await?;
let build_guard = start_index_build_paused(
&ds_a,
&session,
"DEFINE INDEX test ON user FIELDS email CONCURRENTLY",
)
.await?;
execute_cancelled_transaction_retrying_conflicts(
&ds_b,
&session,
"BEGIN; UPDATE user:one SET email = 'new@example.com'; CANCEL;",
)
.await?;
drop(build_guard);
wait_for_index_ready(&ds_a, &session, "user", "test").await?;
expect_indexed_query_len(
&ds_a,
&session,
"SELECT id FROM user WITH INDEX test WHERE email = 'old@example.com'",
1,
)
.await?;
expect_indexed_query_len(
&ds_a,
&session,
"SELECT id FROM user WITH INDEX test WHERE email = 'new@example.com'",
0,
)
.await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn distributed_admission_error_after_registration_releases_reservation() -> Result<()> {
let (ds_a, ds_b, session) = new_distributed_index_test_ds().await?;
execute_all(
&ds_a,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'old@example.com' RETURN NONE;
",
)
.await?;
let build_guard = start_index_build_paused(
&ds_a,
&session,
"DEFINE INDEX test ON user FIELDS email CONCURRENTLY",
)
.await?;
let _guard = inject_non_retryable_error(
NonRetryableErrorSite::ConcurrentIndexAfterReservationRegistration,
ds_b.id(),
);
let error = execute_error_text_retrying_conflicts(
&ds_b,
&session,
"UPDATE user:one SET email = 'new@example.com' RETURN NONE",
)
.await?;
assert!(error.contains("injected non-retryable error"), "unexpected error: {error}");
drop(build_guard);
wait_for_index_ready(&ds_a, &session, "user", "test").await?;
expect_indexed_query_len(
&ds_a,
&session,
"SELECT id FROM user WITH INDEX test WHERE email = 'old@example.com'",
1,
)
.await?;
expect_indexed_query_len(
&ds_a,
&session,
"SELECT id FROM user WITH INDEX test WHERE email = 'new@example.com'",
0,
)
.await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn distributed_rollback_reservation_cleanup_failure_marks_build_error() -> Result<()> {
let (ds_a, ds_b, session) = new_distributed_index_test_ds().await?;
execute_all(
&ds_a,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'old@example.com' RETURN NONE;
",
)
.await?;
let build_guard = start_index_build_paused(
&ds_a,
&session,
"DEFINE INDEX test ON user FIELDS email CONCURRENTLY",
)
.await?;
let _release_guard = inject_non_retryable_error(
NonRetryableErrorSite::ConcurrentIndexReservationRelease,
ds_b.id(),
);
let error = execute_error_text_retrying_conflicts(
&ds_b,
&session,
"BEGIN; UPDATE user:one SET email = 'new@example.com'; CANCEL;",
)
.await?;
assert!(
error.contains("injected non-retryable error")
|| error.contains("durable index-build reservation")
|| error.contains("cancelled transaction"),
"unexpected transaction error: {error}"
);
let building = index_building_json(&ds_a, &session, "user", "test").await?;
assert_eq!(building.get("status").and_then(|status| status.as_str()), Some("error"));
let reason = building.get("error").and_then(|error| error.as_str()).unwrap_or_default();
assert!(
reason.contains("Failed to release durable index-build reservation"),
"unexpected durable error reason: {building}"
);
drop(build_guard);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn distributed_unique_index_replays_second_node_insert() -> Result<()> {
let (ds_a, ds_b, session) = new_distributed_index_test_ds().await?;
execute_all(
&ds_a,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
CREATE user:two SET email = 'two@example.com' RETURN NONE;
",
)
.await?;
let guard = start_index_build_paused(
&ds_a,
&session,
"DEFINE INDEX test ON user FIELDS email UNIQUE CONCURRENTLY",
)
.await?;
execute_all_retrying_conflicts(
&ds_b,
&session,
"CREATE user:queued SET email = 'queued@example.com' RETURN NONE",
)
.await?;
drop(guard);
wait_for_index_ready(&ds_a, &session, "user", "test").await?;
assert_eq!(
query_array_len(
&ds_a,
&session,
"SELECT id FROM user WITH INDEX test WHERE email = 'queued@example.com'"
)
.await?,
1
);
expect_statement_error(
&ds_b,
&session,
"CREATE user:duplicate SET email = 'queued@example.com' RETURN NONE",
)
.await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn distributed_fulltext_index_replays_second_node_update() -> Result<()> {
let (ds_a, ds_b, session) = new_distributed_index_test_ds().await?;
execute_all(
&ds_a,
&session,
"
DEFINE ANALYZER simple TOKENIZERS blank FILTERS lowercase;
DEFINE TABLE doc SCHEMALESS;
CREATE doc:one SET text = 'old phrase' RETURN NONE;
CREATE doc:two SET text = 'stable text' RETURN NONE;
",
)
.await?;
let guard = start_index_build_paused(
&ds_a,
&session,
"DEFINE INDEX test ON doc FIELDS text FULLTEXT ANALYZER simple BM25 HIGHLIGHTS CONCURRENTLY",
)
.await?;
execute_all_retrying_conflicts(
&ds_b,
&session,
"UPDATE doc:one SET text = 'queued phrase' RETURN NONE",
)
.await?;
drop(guard);
wait_for_index_ready(&ds_a, &session, "doc", "test").await?;
let building = index_building_json(&ds_a, &session, "doc", "test").await?;
assert_eq!(building.get("pending").and_then(|pending| pending.as_u64()), Some(0));
let updated =
building.get("updated").and_then(|updated| updated.as_u64()).ok_or_else(|| {
anyhow::anyhow!("index info did not include building.updated: {building}")
})?;
assert!(
updated >= 1,
"expected queued full-text update to be replayed in index status: {building}"
);
assert_eq!(
query_array_len(
&ds_a,
&session,
"SELECT id FROM doc WITH INDEX test WHERE text @@ 'queued'"
)
.await?,
1
);
assert_eq!(
query_array_len(&ds_a, &session, "SELECT id FROM doc WITH INDEX test WHERE text @@ 'old'")
.await?,
0
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn distributed_hnsw_index_replays_second_node_insert() -> Result<()> {
let (ds_a, ds_b, session) = new_distributed_index_test_ds().await?;
execute_all(
&ds_a,
&session,
"
DEFINE TABLE vec SCHEMALESS;
CREATE vec:one SET vector = [0f, 0f] RETURN NONE;
CREATE vec:two SET vector = [20f, 20f] RETURN NONE;
",
)
.await?;
let guard = start_index_build_paused(
&ds_a,
&session,
"DEFINE INDEX test ON vec FIELDS vector HNSW DIMENSION 2 DIST EUCLIDEAN TYPE F32 EFC 16 M 4 CONCURRENTLY",
)
.await?;
execute_all_retrying_conflicts(
&ds_b,
&session,
"CREATE vec:queued SET vector = [10f, 10f] RETURN NONE",
)
.await?;
drop(guard);
wait_for_index_ready(&ds_a, &session, "vec", "test").await?;
assert_eq!(
query_array_len(
&ds_a,
&session,
"SELECT id FROM vec WITH INDEX test WHERE vector <|1,40|> [10f, 10f] AND id = vec:queued"
)
.await?,
1
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[derive(Clone, Copy, Debug)]
enum BuildPause {
BeforeScan,
InScanBatch,
}
#[cfg(feature = "kv-mem")]
impl BuildPause {
const ALL: [Self; 2] = [Self::BeforeScan, Self::InScanBatch];
fn site(self) -> RetryableConflictSite {
match self {
Self::BeforeScan => RetryableConflictSite::ConcurrentIndexInitialCleanup,
Self::InScanBatch => RetryableConflictSite::ConcurrentIndexInitialBatch,
}
}
}
#[cfg(feature = "kv-mem")]
const TWICE_MODIFIED_SEED: [(&str, &str); 5] =
[("twice", "v0"), ("in_one", "w0"), ("gone", "x0"), ("back", "y0"), ("still", "z0")];
#[cfg(feature = "kv-mem")]
const TWICE_MODIFIED_STEPS: [&str; 7] = [
"UPDATE t:twice SET f = {v1} RETURN NONE",
"UPDATE t:twice SET f = {v2} RETURN NONE",
"BEGIN; UPDATE t:in_one SET f = {w1} RETURN NONE; UPDATE t:in_one SET f = {w2} RETURN NONE; COMMIT",
"UPDATE t:gone SET f = {x1} RETURN NONE",
"DELETE t:gone RETURN NONE",
"DELETE t:back RETURN NONE",
"CREATE t:back SET f = {y1} RETURN NONE",
];
#[cfg(feature = "kv-mem")]
const TWICE_MODIFIED_FINAL: [(&str, &str); 4] =
[("back", "y1"), ("in_one", "w2"), ("still", "z0"), ("twice", "v2")];
#[cfg(feature = "kv-mem")]
const TWICE_MODIFIED_HISTORY: [&str; 11] =
["v0", "v1", "v2", "w0", "w1", "w2", "x0", "x1", "y0", "y1", "z0"];
#[cfg(feature = "kv-mem")]
fn word(name: &str) -> String {
format!("'{name}'")
}
#[cfg(feature = "kv-mem")]
fn point(name: &str) -> String {
let (x, y) = match name {
"v0" => (0, 0),
"v1" => (10, 0),
"v2" => (20, 0),
"w0" => (0, 10),
"w1" => (0, 20),
"w2" => (0, 30),
"x0" => (-10, 0),
"x1" => (-20, 0),
"y0" => (0, -10),
"y1" => (0, -25),
"z0" => (50, 50),
other => unreachable!("no point for {other}"),
};
format!("[{x}f, {y}f]")
}
#[cfg(feature = "kv-mem")]
fn spell(template: &str, literal: fn(&str) -> String) -> String {
TWICE_MODIFIED_HISTORY
.iter()
.fold(template.to_owned(), |sql, name| sql.replace(&format!("{{{name}}}"), &literal(name)))
}
#[cfg(feature = "kv-mem")]
fn twice_modified_holders(value: &str) -> Vec<String> {
TWICE_MODIFIED_FINAL
.iter()
.filter(|(_, held)| *held == value)
.map(|(id, _)| (*id).to_owned())
.collect()
}
#[cfg(feature = "kv-mem")]
async fn durable_appending_count(ds: &Datastore, ikb: &IndexKeyBase) -> Result<usize> {
let tx = ds.transaction(TransactionType::Read).await?;
let bg = catch!(tx, tx.keys(ikb.new_bg_all_generations_range()?, u32::MAX, 0, None).await);
tx.cancel().await?;
Ok(bg.len())
}
#[cfg(feature = "kv-mem")]
async fn query_json(ds: &Datastore, session: &Session, sql: &str) -> Result<serde_json::Value> {
let mut results = ds.execute(sql, session, None).await?;
Ok(results.remove(0).result?.into_json_value())
}
#[cfg(feature = "kv-mem")]
async fn build_over_records_modified_twice(
define: &str,
pause: BuildPause,
literal: fn(&str) -> String,
queued: usize,
) -> Result<(Arc<Datastore>, Session)> {
let (ds_a, ds_b, session) = new_distributed_index_test_ds().await?;
let seed: String = TWICE_MODIFIED_SEED
.iter()
.map(|(id, value)| format!("CREATE t:{id} SET f = {} RETURN NONE;", literal(value)))
.collect();
execute_all(&ds_a, &session, &format!("DEFINE TABLE t SCHEMALESS; {seed}")).await?;
let site = pause.site();
let node_id = ds_a.id();
let guard = inject_retryable_conflicts(site, node_id, REPEATED_RETRY_CONFLICTS);
execute_all(&ds_a, &session, define).await?;
wait_for_retry_conflict(site, node_id, REPEATED_RETRY_CONFLICTS).await?;
for step in TWICE_MODIFIED_STEPS {
execute_all_retrying_conflicts(&ds_b, &session, &spell(step, literal)).await?;
}
let (ns, db, table, ix) = get_table_index(&ds_a, "t", "ix").await?;
let ikb = IndexKeyBase::new(ns, db, table, ix.index_id);
assert_eq!(
durable_appending_count(&ds_a, &ikb).await?,
queued,
"{pause:?}: one entry per change"
);
assert!(
retryable_conflict_count(site, node_id) > 0,
"{pause:?}: the build must still be held when the last write commits"
);
drop(guard);
wait_for_index_ready(&ds_a, &session, "t", "ix").await?;
Ok((ds_a, session))
}
#[cfg(feature = "kv-mem")]
async fn assert_equality_lookups_match_the_rows(
ds: &Datastore,
session: &Session,
pause: BuildPause,
) -> Result<()> {
for value in TWICE_MODIFIED_HISTORY {
let indexed = format!(
"SELECT VALUE record::id(id) FROM t WITH INDEX ix WHERE f = '{value}' ORDER BY id"
);
let scanned = indexed.replace("WITH INDEX ix", "WITH NOINDEX");
let got = query_json(ds, session, &indexed).await?;
assert_eq!(got, serde_json::json!(twice_modified_holders(value)), "{pause:?}: {indexed}");
assert_eq!(got, query_json(ds, session, &scanned).await?, "{pause:?}: {indexed}");
}
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn concurrent_standard_index_replays_records_modified_twice() -> Result<()> {
for pause in BuildPause::ALL {
let (ds, session) = build_over_records_modified_twice(
"DEFINE INDEX ix ON t FIELDS f CONCURRENTLY",
pause,
word,
TWICE_MODIFIED_STEPS.len() + 1,
)
.await?;
assert_equality_lookups_match_the_rows(&ds, &session, pause).await?;
let all = query_json(
&ds,
&session,
"SELECT VALUE record::id(id) FROM t WITH INDEX ix WHERE f >= 'a' ORDER BY id",
)
.await?;
let finals: Vec<&str> = TWICE_MODIFIED_FINAL.iter().map(|(id, _)| *id).collect();
assert_eq!(all, serde_json::json!(finals), "{pause:?}: one entry per live record");
}
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn concurrent_unique_index_replays_records_modified_twice() -> Result<()> {
for pause in BuildPause::ALL {
let (ds, session) = build_over_records_modified_twice(
"DEFINE INDEX ix ON t FIELDS f UNIQUE CONCURRENTLY",
pause,
word,
TWICE_MODIFIED_STEPS.len() + 1,
)
.await?;
assert_equality_lookups_match_the_rows(&ds, &session, pause).await?;
for value in TWICE_MODIFIED_HISTORY {
let sql = format!("CREATE t:{value}_again SET f = '{value}' RETURN NONE");
let created = execute_all(&ds, &session, &sql).await;
let held = !twice_modified_holders(value).is_empty();
assert_eq!(
created.is_err(),
held,
"{pause:?}: {sql} must fail exactly when a record holds '{value}': {created:?}"
);
}
}
Ok(())
}
#[cfg(feature = "kv-mem")]
async fn count_index_total(
ds: &Datastore,
ns: NamespaceId,
db: DatabaseId,
table: &TableName,
ix: IndexId,
) -> Result<i64> {
let tx = ds.transaction(TransactionType::Read).await?;
let rng = IndexCountPrefix {
ns,
db,
tb: Cow::Borrowed(table),
ix,
}
.range()?;
let keys = catch!(tx, tx.keys(rng, u32::MAX, 0, None).await);
tx.cancel().await?;
let mut total = 0;
for key in keys {
let iu = IndexCountKey::decode_key(&key)?;
total += if iu.pos {
iu.count as i64
} else {
-(iu.count as i64)
};
}
Ok(total)
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn concurrent_count_index_replays_records_modified_twice() -> Result<()> {
let cases = [
("DEFINE INDEX ix ON t COUNT CONCURRENTLY", "SELECT count() FROM t GROUP ALL", 4, 3),
(
"DEFINE INDEX ix ON t COUNT WHERE f IN ['v1', 'w1', 'x1', 'y1'] CONCURRENTLY",
"SELECT count() FROM t WHERE f IN ['v1', 'w1', 'x1', 'y1'] GROUP ALL",
1,
TWICE_MODIFIED_STEPS.len() + 1,
),
];
for (define, count, expected, queued) in cases {
for pause in BuildPause::ALL {
let (ds, session) =
build_over_records_modified_twice(define, pause, word, queued).await?;
let (ns, db, table, ix) = get_table_index(&ds, "t", "ix").await?;
assert_eq!(
count_index_total(&ds, ns, db, &table, ix.index_id).await?,
expected,
"{pause:?}: {define}"
);
assert_eq!(
count_query_value(&ds, &session, count).await?,
expected,
"{pause:?}: {count}"
);
let scanned = count.replace("FROM t", "FROM t WITH NOINDEX");
assert_eq!(
count_query_value(&ds, &session, &scanned).await?,
expected,
"{pause:?}: {scanned}"
);
}
}
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn concurrent_fulltext_index_replays_records_modified_twice() -> Result<()> {
const ANALYZER: &str = "DEFINE ANALYZER simple TOKENIZERS blank FILTERS lowercase;";
for pause in BuildPause::ALL {
let (ds, session) = build_over_records_modified_twice(
&format!(
"{ANALYZER} DEFINE INDEX ix ON t FIELDS f FULLTEXT ANALYZER simple BM25 CONCURRENTLY"
),
pause,
word,
TWICE_MODIFIED_STEPS.len() + 1,
)
.await?;
execute_all(
&ds,
&session,
"DEFINE INDEX reference ON t FIELDS f FULLTEXT ANALYZER simple BM25",
)
.await?;
for value in TWICE_MODIFIED_HISTORY {
let indexed = format!(
"SELECT record::id(id) AS rid, search::score(0) AS score FROM t WITH INDEX ix WHERE f @0@ '{value}' ORDER BY rid"
);
let got = query_json(&ds, &session, &indexed).await?;
let ids: Vec<&str> = got
.as_array()
.into_iter()
.flatten()
.filter_map(|row| row["rid"].as_str())
.collect();
assert_eq!(ids, twice_modified_holders(value), "{pause:?}: {indexed}");
let reference = indexed.replace("WITH INDEX ix", "WITH INDEX reference");
assert_eq!(got, query_json(&ds, &session, &reference).await?, "{pause:?}: {indexed}");
}
}
Ok(())
}
#[cfg(feature = "kv-mem")]
async fn wait_for_index_drained(ds: &Datastore, session: &Session, index: &str) -> Result<()> {
timeout(Duration::from_secs(10), async {
loop {
let building = index_building_json(ds, session, "t", index).await?;
if building.get("compacting").and_then(|compacting| compacting.as_bool()) == Some(false)
{
return Ok(());
}
sleep(Duration::from_millis(20)).await;
}
})
.await
.map_err(|_| anyhow::anyhow!("timed out waiting for {index} to drain its pending updates"))?
}
#[cfg(feature = "kv-mem")]
async fn assert_nearest_neighbours_match_brute_force(
ds: &Datastore,
session: &Session,
context: &str,
points: &[String],
live: &[&str],
) -> Result<()> {
for at in points {
let indexed = format!(
"SELECT record::id(id) AS id, math::fixed(vector::distance::knn(), 3) AS distance FROM t WITH INDEX ix WHERE f <|1,40|> {at}"
);
let brute = format!(
"SELECT record::id(id) AS id, math::fixed(vector::distance::knn(), 3) AS distance FROM t WITH NOINDEX WHERE f <|1,EUCLIDEAN|> {at}"
);
assert_eq!(
query_json(ds, session, &indexed).await?,
query_json(ds, session, &brute).await?,
"{context}: nearest neighbour of {at}"
);
}
let wide = query_json(
ds,
session,
"SELECT VALUE record::id(id) FROM t WITH INDEX ix WHERE f <|10,40|> [0f, 0f] ORDER BY id",
)
.await?;
assert_eq!(wide, serde_json::json!(live), "{context}: each live record exactly once");
Ok(())
}
#[cfg(feature = "kv-mem")]
async fn assert_vector_index_replays_records_modified_twice(define: &str) -> Result<()> {
for pause in BuildPause::ALL {
let (ds, session) =
build_over_records_modified_twice(define, pause, point, TWICE_MODIFIED_STEPS.len() + 1)
.await?;
let history: Vec<String> = TWICE_MODIFIED_HISTORY.iter().map(|name| point(name)).collect();
let finals: Vec<&str> = TWICE_MODIFIED_FINAL.iter().map(|(id, _)| *id).collect();
let context = format!("{pause:?} after the build");
assert_nearest_neighbours_match_brute_force(&ds, &session, &context, &history, &finals)
.await?;
wait_for_index_drained(&ds, &session, "ix").await?;
let context = format!("{pause:?} after the drain");
assert_nearest_neighbours_match_brute_force(&ds, &session, &context, &history, &finals)
.await?;
for step in [
"UPDATE t:twice SET f = [30f, 0f] RETURN NONE",
"UPDATE t:twice SET f = [40f, 0f] RETURN NONE",
"BEGIN; UPDATE t:in_one SET f = [0f, 40f] RETURN NONE; UPDATE t:in_one SET f = [0f, 44f] RETURN NONE; COMMIT",
"UPDATE t:still SET f = [60f, 60f] RETURN NONE",
"DELETE t:still RETURN NONE",
] {
execute_all(&ds, &session, step).await?;
}
let online: Vec<String> = [
"[20f, 0f]",
"[30f, 0f]",
"[40f, 0f]",
"[0f, 30f]",
"[0f, 40f]",
"[0f, 44f]",
"[50f, 50f]",
"[60f, 60f]",
]
.into_iter()
.chain(history.iter().map(String::as_str))
.map(str::to_owned)
.collect();
let live = ["back", "in_one", "twice"];
let context = format!("{pause:?} with online changes pending");
assert_nearest_neighbours_match_brute_force(&ds, &session, &context, &online, &live)
.await?;
Datastore::index_compaction(
Arc::clone(&ds),
Duration::from_secs(1),
CancellationToken::new(),
)
.await?;
wait_for_index_drained(&ds, &session, "ix").await?;
let context = format!("{pause:?} after compacting the online changes");
assert_nearest_neighbours_match_brute_force(&ds, &session, &context, &online, &live)
.await?;
}
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn concurrent_hnsw_index_replays_records_modified_twice() -> Result<()> {
assert_vector_index_replays_records_modified_twice(
"DEFINE INDEX ix ON t FIELDS f HNSW DIMENSION 2 DIST EUCLIDEAN TYPE F32 EFC 16 M 4 CONCURRENTLY",
)
.await
}
#[cfg(all(feature = "kv-mem", diskann))]
#[tokio::test(flavor = "multi_thread")]
async fn concurrent_diskann_index_replays_records_modified_twice() -> Result<()> {
assert_vector_index_replays_records_modified_twice(
"DEFINE INDEX ix ON t FIELDS f DISKANN DIMENSION 2 DIST EUCLIDEAN TYPE F32 CONCURRENTLY",
)
.await
}
#[cfg(feature = "kv-mem")]
async fn assert_vector_drain_folds_records_modified_twice(
define: &str,
site: RetryableConflictSite,
) -> Result<()> {
let (ds_a, ds_b, session) = new_distributed_index_test_ds().await?;
let seed: String = TWICE_MODIFIED_SEED
.iter()
.map(|(id, value)| format!("CREATE t:{id} SET f = {} RETURN NONE;", point(value)))
.collect();
execute_all(&ds_a, &session, &format!("DEFINE TABLE t SCHEMALESS; {seed}")).await?;
let node_id = ds_a.id();
let guard = inject_retryable_conflicts(site, node_id, REPEATED_RETRY_CONFLICTS);
execute_all(&ds_a, &session, define).await?;
wait_for_retry_conflict(site, node_id, REPEATED_RETRY_CONFLICTS).await?;
for step in TWICE_MODIFIED_STEPS {
execute_all_retrying_conflicts(&ds_b, &session, &spell(step, point)).await?;
}
assert!(
retryable_conflict_count(site, node_id) > 0,
"the drain must still be retrying when the last write commits"
);
drop(guard);
wait_for_index_drained(&ds_a, &session, "ix").await?;
let history: Vec<String> = TWICE_MODIFIED_HISTORY.iter().map(|name| point(name)).collect();
let finals: Vec<&str> = TWICE_MODIFIED_FINAL.iter().map(|(id, _)| *id).collect();
assert_nearest_neighbours_match_brute_force(
&ds_a,
&session,
"after the drain",
&history,
&finals,
)
.await
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn hnsw_build_drain_folds_records_modified_twice() -> Result<()> {
assert_vector_drain_folds_records_modified_twice(
"DEFINE INDEX ix ON t FIELDS f HNSW DIMENSION 2 DIST EUCLIDEAN TYPE F32 EFC 16 M 4 CONCURRENTLY",
RetryableConflictSite::ConcurrentIndexHnswPendingCompaction,
)
.await
}
#[cfg(all(feature = "kv-mem", diskann))]
#[tokio::test(flavor = "multi_thread")]
async fn diskann_build_drain_folds_records_modified_twice() -> Result<()> {
assert_vector_drain_folds_records_modified_twice(
"DEFINE INDEX ix ON t FIELDS f DISKANN DIMENSION 2 DIST EUCLIDEAN TYPE F32 CONCURRENTLY",
RetryableConflictSite::ConcurrentIndexDiskAnnPendingCompaction,
)
.await
}
#[tokio::test(flavor = "multi_thread")]
async fn concurrent_indexing_retries_initial_cleanup_commit_conflict() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
let site = RetryableConflictSite::ConcurrentIndexInitialCleanup;
let node_id = ds.id();
let _guard = inject_retryable_conflict(site, node_id);
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
DEFINE INDEX test ON user FIELDS email CONCURRENTLY;
",
)
.await?;
wait_for_index_ready(&ds, &session, "user", "test").await?;
assert_eq!(retryable_conflict_count(site, node_id), 0);
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn concurrent_indexing_retries_initial_batch_commit_conflict() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:1 SET email = 'one@example.com' RETURN NONE;
CREATE user:2 SET email = 'two@example.com' RETURN NONE;
",
)
.await?;
let site = RetryableConflictSite::ConcurrentIndexInitialBatch;
let node_id = ds.id();
let _guard = inject_retryable_conflict(site, node_id);
execute_all(&ds, &session, "DEFINE INDEX test ON user FIELDS email CONCURRENTLY").await?;
wait_for_index_ready(&ds, &session, "user", "test").await?;
assert_eq!(retryable_conflict_count(site, node_id), 0);
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn concurrent_indexing_retries_initial_cleanup_stops_after_abort() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
let site = RetryableConflictSite::ConcurrentIndexInitialCleanup;
let node_id = ds.id();
let _guard = inject_retryable_conflicts(site, node_id, REPEATED_RETRY_CONFLICTS);
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
DEFINE INDEX test ON user FIELDS email CONCURRENTLY;
",
)
.await?;
wait_for_retry_conflict(site, node_id, REPEATED_RETRY_CONFLICTS).await?;
execute_all(&ds, &session, "REMOVE INDEX test ON user").await?;
let remaining = wait_for_retry_conflict_count_to_stabilize(site, node_id).await?;
assert!(remaining > 0, "abort should stop retries before all conflicts are consumed");
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn rebuild_index_retries_a_conflicted_hnsw_pending_compaction() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE INDEX hx ON t FIELDS vec HNSW DIMENSION 2 DIST EUCLIDEAN TYPE F32;
CREATE t:1 SET vec = [1.0, 0.0];
CREATE t:2 SET vec = [0.0, 1.0];",
)
.await?;
let site = RetryableConflictSite::ConcurrentIndexHnswPendingCompaction;
let node_id = ds.id();
let _guard = inject_retryable_conflict(site, node_id);
execute_all(&ds, &session, "REBUILD INDEX hx ON t").await?;
assert_eq!(
retryable_conflict_count(site, node_id),
0,
"the rebuild must consume the injected conflict by retrying the drain"
);
expect_indexed_query_len(&ds, &session, "SELECT id FROM t WHERE vec <|2,40|> [1.0f, 0.0f]", 2)
.await
}
#[cfg(all(feature = "kv-mem", diskann))]
#[tokio::test(flavor = "multi_thread")]
async fn rebuild_index_retries_a_conflicted_diskann_pending_compaction() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE INDEX dx ON t FIELDS vec DISKANN DIMENSION 2 DIST EUCLIDEAN TYPE F32;
CREATE t:1 SET vec = [1.0, 0.0];
CREATE t:2 SET vec = [0.0, 1.0];",
)
.await?;
let site = RetryableConflictSite::ConcurrentIndexDiskAnnPendingCompaction;
let node_id = ds.id();
let _guard = inject_retryable_conflict(site, node_id);
execute_all(&ds, &session, "REBUILD INDEX dx ON t").await?;
assert_eq!(
retryable_conflict_count(site, node_id),
0,
"the rebuild must consume the injected conflict by retrying the drain"
);
expect_indexed_query_len(&ds, &session, "SELECT id FROM t WHERE vec <|2,40|> [1.0f, 0.0f]", 2)
.await
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn index_compaction_leaves_an_index_its_builder_still_owns() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE INDEX hx ON t FIELDS vec HNSW DIMENSION 2 DIST EUCLIDEAN TYPE F32;
CREATE t:1 SET vec = [1.0, 0.0];
CREATE t:2 SET vec = [0.0, 1.0];",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "t", "hx").await?;
let ikb = IndexKeyBase::new(ns, db, table, ix.index_id);
let pendings = count_hnsw_pendings(&ds, &ikb).await?;
assert!(pendings > 0, "the fixture must leave pendings for a compaction to consume");
let site = RetryableConflictSite::ConcurrentIndexInitialCleanup;
let node_id = ds.id();
let _guard = inject_retryable_conflicts(site, node_id, REPEATED_RETRY_CONFLICTS);
execute_all(&ds, &session, "REBUILD INDEX hx ON t CONCURRENTLY").await?;
wait_for_retry_conflict(site, node_id, REPEATED_RETRY_CONFLICTS).await?;
Datastore::index_compaction(Arc::clone(&ds), Duration::from_secs(1), CancellationToken::new())
.await?;
assert_eq!(
count_hnsw_pendings(&ds, &ikb).await?,
pendings,
"compaction must leave the pendings to the builder that owns the index"
);
execute_all(&ds, &session, "REMOVE INDEX hx ON t").await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn info_for_index_reports_an_undrained_compaction_queue() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE INDEX hx ON t FIELDS vec HNSW DIMENSION 2 DIST EUCLIDEAN TYPE F32;
CREATE t:1 SET vec = [1.0, 0.0];
CREATE t:2 SET vec = [0.0, 1.0];",
)
.await?;
let building = index_building_json(&ds, &session, "t", "hx").await?;
assert_eq!(
building.get("compacting").and_then(|v| v.as_bool()),
Some(true),
"records were written and not yet folded into the graph: {building}"
);
assert_eq!(building.get("status").and_then(|v| v.as_str()), Some("ready"), "{building}");
assert_eq!(building.get("pending").and_then(|v| v.as_u64()), Some(0), "{building}");
Datastore::index_compaction(Arc::clone(&ds), Duration::from_secs(1), CancellationToken::new())
.await?;
let building = index_building_json(&ds, &session, "t", "hx").await?;
assert_eq!(
building.get("compacting").and_then(|v| v.as_bool()),
Some(false),
"the queue drained, so the index now answers at index speed: {building}"
);
Ok(())
}
#[cfg(all(feature = "kv-mem", diskann))]
#[tokio::test(flavor = "multi_thread")]
async fn info_for_index_reports_an_undrained_diskann_pending_set() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE INDEX dx ON t FIELDS vec DISKANN DIMENSION 2 DIST EUCLIDEAN TYPE F32;
CREATE t:1 SET vec = [1.0, 0.0];
CREATE t:2 SET vec = [0.0, 1.0];",
)
.await?;
let building = index_building_json(&ds, &session, "t", "dx").await?;
assert_eq!(
building.get("compacting").and_then(|v| v.as_bool()),
Some(true),
"records were written and not yet folded into the graph: {building}"
);
assert_eq!(building.get("status").and_then(|v| v.as_str()), Some("ready"), "{building}");
assert_eq!(building.get("pending").and_then(|v| v.as_u64()), Some(0), "{building}");
Datastore::index_compaction(Arc::clone(&ds), Duration::from_secs(1), CancellationToken::new())
.await?;
let building = index_building_json(&ds, &session, "t", "dx").await?;
assert_eq!(
building.get("compacting").and_then(|v| v.as_bool()),
Some(false),
"the pending set drained, so the index now answers at index speed: {building}"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn index_compaction_leaves_an_index_a_remote_builder_owns() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE INDEX hx ON t FIELDS vec HNSW DIMENSION 2 DIST EUCLIDEAN TYPE F32;
CREATE t:1 SET vec = [1.0, 0.0];
CREATE t:2 SET vec = [0.0, 1.0];",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "t", "hx").await?;
let ikb = IndexKeyBase::new(ns, db, table, ix.index_id);
let pendings = count_hnsw_pendings(&ds, &ikb).await?;
assert!(pendings > 0, "the fixture must leave pendings for a compaction to consume");
claim_build_for_a_remote_owner(&ds, &ikb, chrono::Duration::zero()).await?;
Datastore::index_compaction(Arc::clone(&ds), Duration::from_secs(1), CancellationToken::new())
.await?;
assert_eq!(
count_hnsw_pendings(&ds, &ikb).await?,
pendings,
"compaction must leave the pendings to the builder that owns the index"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn index_compaction_still_defers_to_a_build_whose_heartbeat_lapsed() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE INDEX hx ON t FIELDS vec HNSW DIMENSION 2 DIST EUCLIDEAN TYPE F32;
CREATE t:1 SET vec = [1.0, 0.0];
CREATE t:2 SET vec = [0.0, 1.0];",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "t", "hx").await?;
let ikb = IndexKeyBase::new(ns, db, table, ix.index_id);
let pendings = count_hnsw_pendings(&ds, &ikb).await?;
assert!(pendings > 0, "the fixture must leave pendings for a compaction to consume");
claim_build_for_a_remote_owner(
&ds,
&ikb,
chrono::Duration::seconds(BUILD_OWNER_LEASE_SECS * 4),
)
.await?;
Datastore::index_compaction(Arc::clone(&ds), Duration::from_secs(1), CancellationToken::new())
.await?;
assert_eq!(
count_hnsw_pendings(&ds, &ikb).await?,
pendings,
"an expired heartbeat is a takeover trigger, not evidence that nobody is writing"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn index_compaction_defers_to_an_online_build_that_has_not_released() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE INDEX hx ON t FIELDS vec HNSW DIMENSION 2 DIST EUCLIDEAN TYPE F32;
CREATE t:1 SET vec = [1.0, 0.0];
CREATE t:2 SET vec = [0.0, 1.0];",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "t", "hx").await?;
let ikb = IndexKeyBase::new(ns, db, table, ix.index_id);
let pendings = count_hnsw_pendings(&ds, &ikb).await?;
assert!(pendings > 0, "the fixture must leave pendings for a compaction to consume");
let mut state = durable_build_state(&ds, &ikb).await?;
assert_eq!(state.phase, IndexBuildPhase::Online);
assert_eq!(state.owner, None, "a finished build must have released ownership");
state.owner = Some(Uuid::now_v7());
state.owner_heartbeat_at = Some(Utc::now());
set_durable_build_state(&ds, &ikb, state).await?;
Datastore::index_compaction(Arc::clone(&ds), Duration::from_secs(1), CancellationToken::new())
.await?;
assert_eq!(
count_hnsw_pendings(&ds, &ikb).await?,
pendings,
"an Online generation still owned by its builder must be left to that builder"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
async fn build_index_inside_an_open_statement_transaction(
ds: &Arc<Datastore>,
session: &Session,
ns: NamespaceId,
db: DatabaseId,
table: &TableName,
ix: Arc<IndexDefinition>,
blocking: bool,
) -> Result<Arc<crate::kvs::Transaction>> {
let txn = Arc::new(ds.transaction(TransactionType::Write).await?);
let table_def = catch!(txn, txn.get_tb(ns, db, table, None).await).expect("table should exist");
let mut ctx = ds.setup_ctx()?;
ctx.set_transaction(Arc::clone(&txn));
let ctx = ctx.freeze();
let builder = ctx.get_index_builder().expect("index builder should exist").clone();
let index_id = ix.index_id;
let rcv =
builder.build(&ctx, ds.setup_options(session), table_def.table_id, ix, blocking).await?;
match rcv {
Some(rcv) => rcv.await.expect("the build task must report its result")?,
None => {
let building = local_builder_for_key(ds, ns, db, table, index_id)
.await?
.expect("the build registers a local builder");
wait_for_finished_builder(&building).await?;
}
}
Ok(txn)
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn build_ownership_outlives_a_concurrent_build_until_its_statement_commits() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE INDEX hx ON t FIELDS vec HNSW DIMENSION 2 DIST EUCLIDEAN TYPE F32;
CREATE t:1 SET vec = [1.0, 0.0];",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "t", "hx").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let txn = build_index_inside_an_open_statement_transaction(
&ds,
&session,
ns,
db,
&table,
Arc::clone(&ix),
false,
)
.await?;
let state = durable_build_state(&ds, &ikb).await?;
assert_eq!(state.phase, IndexBuildPhase::Online);
assert!(
state.owner.is_some(),
"a concurrent build must hold the fence until its statement transaction closes"
);
txn.commit().await?;
let state = durable_build_state(&ds, &ikb).await?;
assert_eq!(state.owner, None, "the commit must release the fence");
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn build_ownership_outlives_the_builder_until_its_statement_commits() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE INDEX hx ON t FIELDS vec HNSW DIMENSION 2 DIST EUCLIDEAN TYPE F32;
CREATE t:1 SET vec = [1.0, 0.0];
CREATE t:2 SET vec = [0.0, 1.0];",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "t", "hx").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let txn = build_index_inside_an_open_statement_transaction(
&ds,
&session,
ns,
db,
&table,
Arc::clone(&ix),
true,
)
.await?;
let building = local_builder_for_key(&ds, ns, db, &table, ix.index_id)
.await?
.expect("the build registers a local builder");
assert!(building.is_finished(), "a blocking build reports only after its task has exited");
let state = durable_build_state(&ds, &ikb).await?;
assert_eq!(state.phase, IndexBuildPhase::Online);
assert!(
state.owner.is_some(),
"ownership must still fence compaction while the statement transaction is open"
);
txn.commit().await?;
let state = durable_build_state(&ds, &ikb).await?;
assert_eq!(
state.owner, None,
"the commit that publishes the definition must release the fence"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn releasing_build_ownership_requeues_compaction_skipped_by_the_fence() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE INDEX hx ON t FIELDS vec HNSW DIMENSION 2 DIST EUCLIDEAN TYPE F32;
CREATE t:1 SET vec = [1.0, 0.0];",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "t", "hx").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let txn = build_index_inside_an_open_statement_transaction(
&ds,
&session,
ns,
db,
&table,
Arc::clone(&ix),
true,
)
.await?;
assert_eq!(
count_hnsw_pendings(&ds, &ikb).await?,
0,
"the build drains the index's pendings before it reports"
);
execute_all(&ds, &session, "CREATE t:2 SET vec = [0.0, 1.0]").await?;
let pendings = count_hnsw_pendings(&ds, &ikb).await?;
assert!(pendings > 0, "the write must queue a pending for compaction to consume");
Datastore::index_compaction(Arc::clone(&ds), Duration::from_secs(1), CancellationToken::new())
.await?;
assert_eq!(
count_hnsw_pendings(&ds, &ikb).await?,
pendings,
"the fence must leave the pending to the builder"
);
txn.commit().await?;
Datastore::index_compaction(Arc::clone(&ds), Duration::from_secs(1), CancellationToken::new())
.await?;
assert_eq!(
count_hnsw_pendings(&ds, &ikb).await?,
0,
"the released fence must queue the compaction it turned away"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn build_ownership_is_released_when_its_statement_rolls_back() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE INDEX hx ON t FIELDS vec HNSW DIMENSION 2 DIST EUCLIDEAN TYPE F32;
CREATE t:1 SET vec = [1.0, 0.0];",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "t", "hx").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let txn = build_index_inside_an_open_statement_transaction(
&ds,
&session,
ns,
db,
&table,
Arc::clone(&ix),
true,
)
.await?;
txn.cancel().await?;
let state = durable_build_state(&ds, &ikb).await?;
assert_eq!(state.owner, None, "a rolled-back statement must not leave the index fenced");
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
async fn index_compaction_still_drains_a_count_index_under_build() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE INDEX cx ON t COUNT;
CREATE t:1 SET n = 1;
CREATE t:2 SET n = 2;
CREATE t:3 SET n = 3;",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "t", "cx").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let deltas = count_index_delta_keys(&ds, ns, db, &table, ix.index_id).await?;
assert!(deltas > 1, "the fixture must leave deltas for a compaction to fold");
claim_build_for_a_remote_owner(&ds, &ikb, chrono::Duration::zero()).await?;
Datastore::index_compaction(Arc::clone(&ds), Duration::from_secs(1), CancellationToken::new())
.await?;
assert!(
count_index_delta_keys(&ds, ns, db, &table, ix.index_id).await? < deltas,
"a count index under build must still be compacted"
);
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn concurrent_indexing_retries_initial_batch_stops_after_abort() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:1 SET email = 'one@example.com' RETURN NONE;
CREATE user:2 SET email = 'two@example.com' RETURN NONE;
",
)
.await?;
let site = RetryableConflictSite::ConcurrentIndexInitialBatch;
let node_id = ds.id();
let _guard = inject_retryable_conflicts(site, node_id, REPEATED_RETRY_CONFLICTS);
execute_all(&ds, &session, "DEFINE INDEX test ON user FIELDS email CONCURRENTLY").await?;
wait_for_retry_conflict(site, node_id, REPEATED_RETRY_CONFLICTS).await?;
execute_all(&ds, &session, "REMOVE INDEX test ON user").await?;
let remaining = wait_for_retry_conflict_count_to_stabilize(site, node_id).await?;
assert!(remaining > 0, "abort should stop retries before all conflicts are consumed");
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn datastore_drop_releases_index_builder_after_build() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
DEFINE INDEX test ON user FIELDS email CONCURRENTLY;
",
)
.await?;
wait_for_index_ready(&ds, &session, "user", "test").await?;
let weak_indexes = Arc::downgrade(&ds.index_builder().indexes);
drop(ds);
timeout(Duration::from_secs(5), async {
loop {
if weak_indexes.upgrade().is_none() {
return;
}
sleep(Duration::from_millis(20)).await;
}
})
.await
.map_err(|_| {
anyhow::anyhow!(
"IndexBuilder::indexes Arc not released after Datastore drop — \
index Building still pins the back-reference (regression of #7304)"
)
})?;
Ok(())
}
async fn drain_reclaim_queue(ds: &Arc<Datastore>) {
const MAX_PASSES: usize = 64;
for _ in 0..MAX_PASSES {
let (batches, _) = Datastore::reclaim_tombstones(
Arc::clone(ds),
web_time::Duration::from_secs(1),
web_time::Duration::ZERO,
tokio_util::sync::CancellationToken::new(),
)
.await
.expect("the reclaim pass must not fail");
if batches == 0 {
return;
}
}
}
async fn count_doc_id_mappings(ds: &Datastore, table: &str) -> Result<(usize, usize)> {
let tx = ds.transaction(TransactionType::Read).await?;
let ns = tx.get_ns_by_name("test", None).await?.expect("namespace should exist");
let db = tx.get_db_by_name("test", "test", None).await?.expect("database should exist");
let tb: TableName = table.into();
let di = DocLookupPrefix::new(ns.namespace_id, db.database_id, Cow::Borrowed(&tb)).range()?;
let dd = DocKeyPrefix::new(ns.namespace_id, db.database_id, Cow::Borrowed(&tb)).range()?;
let di_keys = tx.keys(di, u32::MAX, 0, None).await?;
let dd_keys = tx.keys(dd, u32::MAX, 0, None).await?;
tx.cancel().await?;
Ok((di_keys.len(), dd_keys.len()))
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn table_doc_ids_purged_only_when_last_consumer_removed() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE ANALYZER simple TOKENIZERS blank;
DEFINE INDEX ft1 ON t FIELDS a FULLTEXT ANALYZER simple BM25;
DEFINE INDEX ft2 ON t FIELDS b FULLTEXT ANALYZER simple BM25;
CREATE t:1 SET a = 'alpha', b = 'one';
CREATE t:2 SET a = 'beta', b = 'two';
CREATE t:3 SET a = 'gamma', b = 'three';",
)
.await?;
assert_eq!(count_doc_id_mappings(&ds, "t").await?, (3, 3), "one shared doc-ID per record");
execute_all(&ds, &session, "REMOVE INDEX ft1 ON t;").await?;
assert_eq!(
count_doc_id_mappings(&ds, "t").await?,
(3, 3),
"mappings survive while another doc-ID index remains"
);
execute_all(&ds, &session, "REMOVE INDEX ft2 ON t;").await?;
drain_reclaim_queue(&ds).await;
assert_eq!(
count_doc_id_mappings(&ds, "t").await?,
(0, 0),
"mappings purged once the last doc-ID index is removed"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn btree_index_entries_carry_doc_ids() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE INDEX idx_a ON t FIELDS a;
DEFINE INDEX uniq_b ON t FIELDS b UNIQUE;
CREATE t:1 SET a = 'alpha', b = 1;
CREATE t:2 SET a = 'beta', b = 2;",
)
.await?;
assert_eq!(count_doc_id_mappings(&ds, "t").await?, (2, 2), "one shared doc-ID per record");
let tx = ds.transaction(TransactionType::Read).await?;
let ns = tx.get_ns_by_name("test", None).await?.expect("namespace should exist");
let db = tx.get_db_by_name("test", "test", None).await?.expect("database should exist");
let tb: TableName = "t".into();
let docids = crate::idx::docids::TableDocIds::new(ns.namespace_id, db.database_id, tb.clone());
let indexes = tx.all_tb_indexes(ns.namespace_id, db.database_id, &tb, None).await?;
assert_eq!(indexes.len(), 2);
let mut entries = 0;
for ix in indexes.iter() {
assert!(ix.has_entry_doc_ids(), "fresh b-tree indexes carry entry doc-IDs");
let rng = EntryPrefix {
ns: ns.namespace_id,
db: db.database_id,
tb: Cow::Borrowed(&tb),
ix: ix.index_id,
}
.range()?;
for (_, entry) in tx.scan(rng, u32::MAX, 0, None).await? {
let expected = docids.get_doc_id(&tx, &entry.rid.key).await?;
assert!(expected.is_some(), "indexed record has a shared doc-ID mapping");
assert_eq!(entry.doc_id, expected, "entry doc-ID matches the shared mapping");
entries += 1;
}
}
assert_eq!(entries, 4, "one entry per record per index");
tx.cancel().await?;
execute_all(&ds, &session, "DELETE t:1;").await?;
assert_eq!(count_doc_id_mappings(&ds, "t").await?, (1, 1), "mapping reclaimed on delete");
execute_all(&ds, &session, "REMOVE INDEX idx_a ON t;").await?;
assert_eq!(
count_doc_id_mappings(&ds, "t").await?,
(1, 1),
"mappings survive while another b-tree consumer remains"
);
execute_all(&ds, &session, "REMOVE INDEX uniq_b ON t;").await?;
drain_reclaim_queue(&ds).await;
assert_eq!(
count_doc_id_mappings(&ds, "t").await?,
(0, 0),
"mappings purged once the last consumer is removed"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn bitmap_count_performs_zero_record_fetches() -> Result<()> {
use surrealdb_types::Value as PublicValue;
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE FIELD a ON t TYPE string;
DEFINE FIELD b ON t TYPE bool;
DEFINE INDEX idx_a ON t FIELDS a;
DEFINE INDEX idx_b ON t FIELDS b;
CREATE t:1 SET a = 'x', b = true;
CREATE t:2 SET a = 'x', b = true;
CREATE t:3 SET a = 'x', b = false;
CREATE t:4 SET a = 'y', b = true;",
)
.await?;
let count = |sql: &'static str| {
let ds = &ds;
let session = &session;
async move {
let mut results = ds.execute(sql, session, None).await?;
let value = results.remove(0).result?;
match value {
PublicValue::Array(rows) if rows.len() == 1 => match rows.into_iter().next() {
Some(PublicValue::Object(obj)) => match obj.get("count") {
Some(PublicValue::Number(n)) => {
n.to_int().ok_or_else(|| anyhow::anyhow!("non-integer count"))
}
other => anyhow::bail!("unexpected count value: {other:?}"),
},
other => anyhow::bail!("unexpected count row: {other:?}"),
},
other => anyhow::bail!("unexpected count result: {other:?}"),
}
}
};
const BITMAP_COUNT: &str = "SELECT count() FROM t WHERE a = 'x' AND b = true GROUP ALL";
const NOINDEX_COUNT: &str =
"SELECT count() FROM t WITH NOINDEX WHERE a = 'x' AND b = true GROUP ALL";
assert_eq!(count(BITMAP_COUNT).await?, 2);
assert_eq!(count(NOINDEX_COUNT).await?, 2);
{
let tx = ds.transaction(TransactionType::Write).await?;
let ns = tx.get_ns_by_name("test", None).await?.expect("namespace should exist");
let db = tx.get_db_by_name("test", "test", None).await?.expect("database should exist");
let tb: TableName = "t".into();
let rng = RecordPrefix {
ns: ns.namespace_id,
db: db.database_id,
tb: std::borrow::Cow::Borrowed(&tb),
}
.range()?;
for key in tx.keys_raw(rng, u32::MAX, 0, None).await? {
tx.del(key.into()).await?;
}
tx.commit().await?;
}
assert_eq!(count(BITMAP_COUNT).await?, 2, "bitmap count reads index entries only");
assert_eq!(count(NOINDEX_COUNT).await?, 0, "record-fetching count sees the deletions");
Ok(())
}
#[cfg(feature = "kv-mem")]
async fn downgrade_index_format(
ds: &Datastore,
table: &str,
index: &str,
format_version: u16,
) -> Result<()> {
let tx = ds.transaction(TransactionType::Write).await?;
let ns = tx.get_ns_by_name("test", None).await?.expect("namespace should exist");
let db = tx.get_db_by_name("test", "test", None).await?.expect("database should exist");
let tb: TableName = table.into();
let ix = tx
.get_tb_index(ns.namespace_id, db.database_id, &tb, index, None)
.await?
.expect("index should exist");
let mut old = (*ix).clone();
old.format_version = format_version;
tx.put_tb_index(ns.namespace_id, db.database_id, &tb, &old).await?;
let tb_def = tx.expect_tb(ns.namespace_id, db.database_id, &tb).await?;
tx.put_tb("test", "test", &tb_def).await?;
tx.commit().await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
async fn planner_result_ids(
ds: &Datastore,
strategy: crate::dbs::NewPlannerStrategy,
sql: &str,
) -> Result<Vec<String>> {
use surrealdb_types::ToSql;
let session = Session::owner().with_ns("test").with_db("test").new_planner_strategy(strategy);
let mut results = ds.execute(sql, &session, None).await?;
let value = results.remove(0).result?;
let surrealdb_types::Value::Array(rows) = value else {
anyhow::bail!("query returned non-array value: {value:?}");
};
let mut ids = Vec::with_capacity(rows.len());
for row in rows.iter() {
let id = match row {
surrealdb_types::Value::Object(obj) => obj.get("id"),
other => Some(other),
};
match id {
Some(surrealdb_types::Value::RecordId(rid)) => ids.push(rid.to_sql()),
other => anyhow::bail!("row without a record id: {other:?}"),
}
}
ids.sort();
Ok(ids)
}
#[cfg(feature = "kv-mem")]
async fn assert_planner_returns(
ds: &Datastore,
strategy: crate::dbs::NewPlannerStrategy,
sql: &str,
expected: &[&str],
) -> Result<()> {
let ids = planner_result_ids(ds, strategy, sql).await?;
assert_eq!(ids, expected, "{strategy} engine: {sql}");
Ok(())
}
#[cfg(feature = "kv-mem")]
async fn assert_both_planners_return(ds: &Datastore, sql: &str, expected: &[&str]) -> Result<()> {
use crate::dbs::NewPlannerStrategy;
assert_planner_returns(ds, NewPlannerStrategy::ComputeOnly, sql, expected).await?;
assert_planner_returns(ds, NewPlannerStrategy::AllReadOnlyStatements, sql, expected).await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn btree_index_pre_doc_id_format_stays_maintainable() -> Result<()> {
use crate::catalog::BTREE_ENTRY_DOC_IDS_FORMAT_VERSION;
let (ds, session) = new_index_test_ds().await?;
execute_all(&ds, &session, "DEFINE INDEX idx_a ON t FIELDS a;").await?;
downgrade_index_format(&ds, "t", "idx_a", BTREE_ENTRY_DOC_IDS_FORMAT_VERSION - 1).await?;
let (_, _, _, ix) = get_table_index(&ds, "t", "idx_a").await?;
assert!(!ix.uses_doc_ids());
execute_all(
&ds,
&session,
"CREATE t:1 SET a = 'alpha';
CREATE t:2 SET a = 'gamma';
UPDATE t:1 SET a = 'beta';
DELETE t:2;",
)
.await?;
assert_eq!(query_array_len(&ds, &session, "SELECT * FROM t WHERE a = 'beta'").await?, 1);
assert_eq!(query_array_len(&ds, &session, "SELECT * FROM t WHERE a = 'alpha'").await?, 0);
assert_eq!(query_array_len(&ds, &session, "SELECT * FROM t WHERE a = 'gamma'").await?, 0);
assert_eq!(count_doc_id_mappings(&ds, "t").await?, (0, 0));
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn legacy_fulltext_index_serves_matches_under_both_planners() -> Result<()> {
const MATCHES: &str = "SELECT VALUE id FROM t WHERE content @@ 'bravo'";
const CARRIERS: &[&str] = &["t:1", "t:2", "t:3"];
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE ANALYZER simple TOKENIZERS blank;
DEFINE INDEX ft ON t FIELDS content FULLTEXT ANALYZER simple BM25;",
)
.await?;
downgrade_index_format(&ds, "t", "ft", 0).await?;
let (_, _, _, ix) = get_table_index(&ds, "t", "ft").await?;
assert!(ix.uses_doc_ids() && !ix.uses_shared_doc_ids());
execute_all(
&ds,
&session,
"CREATE t:1 SET content = 'alpha bravo';
CREATE t:2 SET content = 'bravo charlie';
CREATE t:3 SET content = 'delta';
CREATE t:4 SET content = 'bravo echo';
UPDATE t:3 SET content = 'bravo foxtrot';
DELETE t:4;",
)
.await?;
assert_eq!(
count_doc_id_mappings(&ds, "t").await?,
(0, 0),
"a legacy full-text index keeps its private doc-ID space"
);
assert_both_planners_return(&ds, MATCHES, CARRIERS).await?;
let (batches, errors) = Datastore::index_compaction(
Arc::clone(&ds),
Duration::from_secs(60),
tokio_util::sync::CancellationToken::new(),
)
.await?;
assert_eq!(errors, 0, "the legacy layout must compact cleanly");
assert!(batches > 0, "the writes must leave the index queued for compaction");
assert_eq!(count_doc_id_mappings(&ds, "t").await?, (0, 0));
assert_both_planners_return(&ds, MATCHES, CARRIERS).await?;
execute_all(&ds, &session, "REBUILD INDEX ft ON t;").await?;
let (_, _, _, ix) = get_table_index(&ds, "t", "ft").await?;
assert!(ix.uses_shared_doc_ids(), "the rebuild must stamp the current format");
assert_eq!(
count_doc_id_mappings(&ds, "t").await?,
(3, 3),
"the rebuilt index allocates one shared doc-ID per surviving record"
);
assert_both_planners_return(&ds, MATCHES, CARRIERS).await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
async fn legacy_vector_index_knn_under_both_planners(define_index: &str) -> Result<()> {
use crate::dbs::NewPlannerStrategy::{AllReadOnlyStatements, ComputeOnly};
const KNN: &str = "SELECT VALUE id FROM t WHERE pt <|2,40|> [0.0, 0.0]";
const OR_MATCHES_KNN: &str = "SELECT VALUE id FROM t \
WHERE (content @@ 'bravo' OR content = 'unmatched') AND pt <|2,EUCLIDEAN|> [0.0, 0.0]";
const MATCHES_KNN: &str =
"SELECT VALUE id FROM t WHERE content @@ 'bravo' AND pt <|2,40|> [0.0, 0.0]";
const NEAREST_TWO: &[&str] = &["t:1", "t:6"];
const NEAREST_TWO_MATCHING: &[&str] = &["t:1", "t:2"];
let (ds, session) = new_index_test_ds().await?;
execute_all(&ds, &session, define_index).await?;
downgrade_index_format(&ds, "t", "vx", 0).await?;
let (_, _, _, ix) = get_table_index(&ds, "t", "vx").await?;
assert!(ix.uses_doc_ids() && !ix.uses_shared_doc_ids());
execute_all(
&ds,
&session,
"CREATE t:1 SET pt = [1.0, 0.0], content = 'alpha bravo';
CREATE t:2 SET pt = [0.0, 2.0], content = 'bravo charlie';
CREATE t:3 SET pt = [3.0, 0.0], content = 'bravo delta';
CREATE t:4 SET pt = [0.5, 0.0], content = 'echo';
CREATE t:5 SET pt = [0.0, 0.75], content = 'foxtrot';
CREATE t:6 SET pt = [1.5, 0.0], content = 'zulu golf';
UPDATE t:5 SET pt = [0.0, 5.0];
DELETE t:4;",
)
.await?;
assert_eq!(
count_doc_id_mappings(&ds, "t").await?,
(0, 0),
"a legacy vector index keeps its private doc-ID space"
);
assert_both_planners_return(&ds, KNN, NEAREST_TWO).await?;
let (batches, errors) = Datastore::index_compaction(
Arc::clone(&ds),
Duration::from_secs(60),
tokio_util::sync::CancellationToken::new(),
)
.await?;
assert_eq!(errors, 0, "the legacy layout must compact cleanly");
assert!(batches > 0, "the writes must leave the index queued for compaction");
assert_eq!(count_doc_id_mappings(&ds, "t").await?, (0, 0));
assert_both_planners_return(&ds, KNN, NEAREST_TWO).await?;
execute_all(&ds, &session, "DEFINE ANALYZER simple TOKENIZERS blank;").await?;
execute_all(
&ds,
&session,
"DEFINE INDEX ft ON t FIELDS content FULLTEXT ANALYZER simple BM25;",
)
.await?;
assert_eq!(
count_doc_id_mappings(&ds, "t").await?,
(5, 5),
"the current-format full-text index allocates one shared doc-ID per record"
);
assert_both_planners_return(&ds, KNN, NEAREST_TWO).await?;
assert_both_planners_return(&ds, OR_MATCHES_KNN, NEAREST_TWO_MATCHING).await?;
assert_planner_returns(&ds, AllReadOnlyStatements, MATCHES_KNN, NEAREST_TWO_MATCHING).await?;
const UNION_MATCHES_KNN: &str = "SELECT VALUE id FROM t \
WHERE (content @@ 'bravo' AND pt <|2,40|> [0.0, 0.0]) OR content @@ 'zulu'";
let err = planner_result_ids(&ds, AllReadOnlyStatements, UNION_MATCHES_KNN)
.await
.expect_err("KNN under OR must be rejected by the exec planner");
assert!(err.to_string().contains("top level of the WHERE clause"), "unexpected error: {err}");
assert_planner_returns(&ds, ComputeOnly, UNION_MATCHES_KNN, &["t:6"]).await?;
execute_all(
&ds,
&session,
"DEFINE FUNCTION fn::knn_plain() { \
RETURN SELECT VALUE id FROM t WHERE pt <|2,40|> [0.0, 0.0]; };
DEFINE FUNCTION fn::knn_matches() { \
RETURN SELECT VALUE id FROM t \
WHERE content @@ 'bravo' AND pt <|2,40|> [0.0, 0.0]; };",
)
.await?;
assert_planner_returns(&ds, AllReadOnlyStatements, "RETURN fn::knn_plain()", NEAREST_TWO)
.await?;
assert_planner_returns(
&ds,
AllReadOnlyStatements,
"RETURN fn::knn_matches()",
NEAREST_TWO_MATCHING,
)
.await?;
assert_planner_returns(&ds, ComputeOnly, MATCHES_KNN, &[]).await?;
execute_all(&ds, &session, "REBUILD INDEX vx ON t;").await?;
let (_, _, _, ix) = get_table_index(&ds, "t", "vx").await?;
assert!(ix.uses_shared_doc_ids(), "the rebuild must stamp the current format");
assert_eq!(count_doc_id_mappings(&ds, "t").await?, (5, 5));
assert_both_planners_return(&ds, KNN, NEAREST_TWO).await?;
assert_planner_returns(&ds, AllReadOnlyStatements, MATCHES_KNN, NEAREST_TWO_MATCHING).await?;
assert_planner_returns(&ds, AllReadOnlyStatements, OR_MATCHES_KNN, NEAREST_TWO_MATCHING)
.await?;
assert_planner_returns(&ds, ComputeOnly, MATCHES_KNN, &[]).await?;
assert_planner_returns(&ds, ComputeOnly, OR_MATCHES_KNN, &[]).await?;
assert_planner_returns(&ds, AllReadOnlyStatements, "RETURN fn::knn_plain()", NEAREST_TWO)
.await?;
assert_planner_returns(
&ds,
AllReadOnlyStatements,
"RETURN fn::knn_matches()",
NEAREST_TWO_MATCHING,
)
.await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn uncovered_matches_restricts_current_format_knn() -> Result<()> {
use crate::dbs::NewPlannerStrategy::AllReadOnlyStatements;
const MATCHES_KNN: &str =
"SELECT VALUE id FROM t WHERE content @@ 'bravo' AND pt <|2,40|> [0.0, 0.0]";
const COVERED_AND_MATCHES_KNN: &str = "SELECT VALUE id FROM t \
WHERE flag = true AND content @@ 'bravo' AND pt <|2,40|> [0.0, 0.0]";
const NEAREST_TWO_MATCHING: &[&str] = &["t:1", "t:2"];
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE ANALYZER simple TOKENIZERS blank;
DEFINE INDEX ft ON t FIELDS content FULLTEXT ANALYZER simple BM25;",
)
.await?;
downgrade_index_format(&ds, "t", "ft", 0).await?;
execute_all(
&ds,
&session,
"DEFINE INDEX vx ON t FIELDS pt HNSW DIMENSION 2 DIST EUCLIDEAN TYPE F32 EFC 16 M 4;
DEFINE INDEX flag_idx ON t FIELDS flag;
CREATE t:1 SET pt = [1.0, 0.0], content = 'alpha bravo', flag = true;
CREATE t:2 SET pt = [0.0, 2.0], content = 'bravo charlie', flag = true;
CREATE t:3 SET pt = [3.0, 0.0], content = 'bravo delta', flag = true;
CREATE t:6 SET pt = [1.5, 0.0], content = 'zulu golf', flag = true;",
)
.await?;
let (_, _, _, vx) = get_table_index(&ds, "t", "vx").await?;
assert!(vx.uses_shared_doc_ids(), "the vector index under test is current-format");
assert_planner_returns(&ds, AllReadOnlyStatements, MATCHES_KNN, NEAREST_TWO_MATCHING).await?;
assert_planner_returns(
&ds,
AllReadOnlyStatements,
COVERED_AND_MATCHES_KNN,
NEAREST_TWO_MATCHING,
)
.await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn legacy_hnsw_index_serves_knn_under_both_planners() -> Result<()> {
legacy_vector_index_knn_under_both_planners(
"DEFINE INDEX vx ON t FIELDS pt HNSW DIMENSION 2 DIST EUCLIDEAN TYPE F32 EFC 16 M 4;",
)
.await
}
#[cfg(all(feature = "kv-mem", diskann))]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn legacy_diskann_index_serves_knn_under_both_planners() -> Result<()> {
legacy_vector_index_knn_under_both_planners(
"DEFINE INDEX vx ON t FIELDS pt DISKANN DIMENSION 2 DIST EUCLIDEAN TYPE F32;",
)
.await
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
#[test_log::test]
async fn concurrent_removal_of_last_doc_id_indexes_purges_mappings() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE ANALYZER simple TOKENIZERS blank;
DEFINE INDEX ft1 ON t FIELDS a FULLTEXT ANALYZER simple BM25;
DEFINE INDEX ft2 ON t FIELDS b FULLTEXT ANALYZER simple BM25;
CREATE t:1 SET a = 'alpha', b = 'one';
CREATE t:2 SET a = 'beta', b = 'two';
CREATE t:3 SET a = 'gamma', b = 'three';",
)
.await?;
assert_eq!(count_doc_id_mappings(&ds, "t").await?, (3, 3));
let task_a = {
let (ds, session) = (Arc::clone(&ds), session.clone());
tokio::spawn(async move {
execute_all_retrying_conflicts(&ds, &session, "REMOVE INDEX ft1 ON t;").await
})
};
let task_b = {
let (ds, session) = (Arc::clone(&ds), session.clone());
tokio::spawn(async move {
execute_all_retrying_conflicts(&ds, &session, "REMOVE INDEX ft2 ON t;").await
})
};
let (ra, rb) = tokio::join!(task_a, task_b);
ra.expect("task A panicked")?;
rb.expect("task B panicked")?;
drain_reclaim_queue(&ds).await;
assert_eq!(
count_doc_id_mappings(&ds, "t").await?,
(0, 0),
"no leaked mappings after concurrent last-consumer removal"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
#[test_log::test]
async fn concurrent_definition_of_doc_id_indexes_shares_one_space() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE ANALYZER simple TOKENIZERS blank;
CREATE t:1 SET a = 'alpha', b = 'one';
CREATE t:2 SET a = 'beta', b = 'two';
CREATE t:3 SET a = 'gamma', b = 'three';",
)
.await?;
assert_eq!(
count_doc_id_mappings(&ds, "t").await?,
(0, 0),
"no mappings before any doc-ID index"
);
let task_a = {
let (ds, session) = (Arc::clone(&ds), session.clone());
tokio::spawn(async move {
execute_all_retrying_conflicts(
&ds,
&session,
"DEFINE INDEX ft1 ON t FIELDS a FULLTEXT ANALYZER simple BM25;",
)
.await
})
};
let task_b = {
let (ds, session) = (Arc::clone(&ds), session.clone());
tokio::spawn(async move {
execute_all_retrying_conflicts(
&ds,
&session,
"DEFINE INDEX ft2 ON t FIELDS b FULLTEXT ANALYZER simple BM25;",
)
.await
})
};
let (ra, rb) = tokio::join!(task_a, task_b);
ra.expect("task A panicked")?;
rb.expect("task B panicked")?;
assert_eq!(
count_doc_id_mappings(&ds, "t").await?,
(3, 3),
"one shared doc-ID per record after concurrent index builds"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
#[test_log::test]
async fn distributed_concurrent_removal_of_last_doc_id_indexes_purges_mappings() -> Result<()> {
let (ds_a, ds_b, session) = new_distributed_index_test_ds().await?;
execute_all(
&ds_a,
&session,
"DEFINE ANALYZER simple TOKENIZERS blank;
DEFINE INDEX ft1 ON t FIELDS a FULLTEXT ANALYZER simple BM25;
DEFINE INDEX ft2 ON t FIELDS b FULLTEXT ANALYZER simple BM25;
CREATE t:1 SET a = 'alpha', b = 'one';
CREATE t:2 SET a = 'beta', b = 'two';
CREATE t:3 SET a = 'gamma', b = 'three';",
)
.await?;
assert_eq!(count_doc_id_mappings(&ds_a, "t").await?, (3, 3));
let ds_a = Arc::new(ds_a);
let ds_b = Arc::new(ds_b);
let task_a = {
let (ds, session) = (Arc::clone(&ds_a), session.clone());
tokio::spawn(async move {
execute_all_retrying_conflicts(&ds, &session, "REMOVE INDEX ft1 ON t;").await
})
};
let task_b = {
let (ds, session) = (Arc::clone(&ds_b), session.clone());
tokio::spawn(async move {
execute_all_retrying_conflicts(&ds, &session, "REMOVE INDEX ft2 ON t;").await
})
};
let (ra, rb) = tokio::join!(task_a, task_b);
ra.expect("node A task panicked")?;
rb.expect("node B task panicked")?;
drain_reclaim_queue(&ds_a).await;
assert_eq!(
count_doc_id_mappings(&ds_a, "t").await?,
(0, 0),
"no leaked mappings after concurrent cross-instance last-consumer removal"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn overwrite_last_doc_id_index_with_plain_index_purges_mappings() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE ANALYZER simple TOKENIZERS blank;
DEFINE INDEX ix ON t FIELDS a FULLTEXT ANALYZER simple BM25;
CREATE t:1 SET a = 'alpha';
CREATE t:2 SET a = 'beta';",
)
.await?;
assert_eq!(count_doc_id_mappings(&ds, "t").await?, (2, 2));
execute_all(&ds, &session, "DEFINE INDEX OVERWRITE ix ON t FIELDS a;").await?;
assert_eq!(
count_doc_id_mappings(&ds, "t").await?,
(2, 2),
"mappings kept when OVERWRITE replaces the consumer with a b-tree consumer"
);
execute_all(&ds, &session, "DEFINE INDEX OVERWRITE ix ON t COUNT;").await?;
drain_reclaim_queue(&ds).await;
assert_eq!(
count_doc_id_mappings(&ds, "t").await?,
(0, 0),
"mappings purged when OVERWRITE drops the last doc-ID consumer"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn overwrite_doc_id_index_keeps_mappings_when_another_consumer_survives() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE ANALYZER simple TOKENIZERS blank;
DEFINE INDEX ft1 ON t FIELDS a FULLTEXT ANALYZER simple BM25;
DEFINE INDEX ft2 ON t FIELDS b FULLTEXT ANALYZER simple BM25;
CREATE t:1 SET a = 'alpha', b = 'one';
CREATE t:2 SET a = 'beta', b = 'two';",
)
.await?;
assert_eq!(count_doc_id_mappings(&ds, "t").await?, (2, 2));
execute_all(&ds, &session, "DEFINE INDEX OVERWRITE ft1 ON t FIELDS a;").await?;
assert_eq!(
count_doc_id_mappings(&ds, "t").await?,
(2, 2),
"mappings survive OVERWRITE while another doc-ID index remains"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn overwrite_last_doc_id_index_with_doc_id_index_keeps_mappings() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE ANALYZER simple TOKENIZERS blank;
DEFINE INDEX ix ON t FIELDS a FULLTEXT ANALYZER simple BM25;
CREATE t:1 SET a = 'alpha';
CREATE t:2 SET a = 'beta';",
)
.await?;
assert_eq!(count_doc_id_mappings(&ds, "t").await?, (2, 2));
execute_all(
&ds,
&session,
"DEFINE ANALYZER simple2 TOKENIZERS class;
DEFINE INDEX OVERWRITE ix ON t FIELDS a FULLTEXT ANALYZER simple2 BM25;",
)
.await?;
assert_eq!(
count_doc_id_mappings(&ds, "t").await?,
(2, 2),
"mappings kept when the replacement is itself a doc-ID index"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn delete_during_doc_id_index_build_keeps_shared_mapping_consistent() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE ANALYZER simple TOKENIZERS blank;
CREATE t:1 SET a = 'alpha', b = 'x';
CREATE t:2 SET a = 'beta', b = 'y';
CREATE t:3 SET a = 'gamma', b = 'z';
DEFINE INDEX ft1 ON t FIELDS a FULLTEXT ANALYZER simple BM25;",
)
.await?;
assert_eq!(count_doc_id_mappings(&ds, "t").await?, (3, 3));
let guard = start_index_build_paused(
&ds,
&session,
"DEFINE INDEX ft2 ON t FIELDS b FULLTEXT ANALYZER simple BM25 CONCURRENTLY",
)
.await?;
execute_all_retrying_conflicts(&ds, &session, "DELETE t:2;").await?;
drop(guard);
wait_for_index_ready(&ds, &session, "t", "ft2").await?;
assert_eq!(
count_doc_id_mappings(&ds, "t").await?,
(2, 2),
"delete during an index build must not leave a stale shared doc-ID mapping"
);
Ok(())
}
async fn read_shared_doc_id(
ds: &Datastore,
docs: &crate::idx::docids::TableDocIds,
id: &RecordIdKey,
) -> Result<Option<crate::idx::docids::DocId>> {
let tx = ds.transaction(TransactionType::Read).await?;
let d = docs.get_doc_id(&tx, id).await?;
tx.cancel().await?;
Ok(d)
}
async fn read_shared_record_id(
ds: &Datastore,
docs: &crate::idx::docids::TableDocIds,
doc_id: crate::idx::docids::DocId,
) -> Result<Option<RecordIdKey>> {
let tx = ds.transaction(TransactionType::Read).await?;
let id = docs.get_record_id(&tx, doc_id).await?;
tx.cancel().await?;
Ok(id)
}
async fn assign_shared_doc_id(
ds: &Datastore,
docs: &crate::idx::docids::TableDocIds,
id: &RecordIdKey,
) -> Result<crate::idx::docids::DocId> {
let mut ctx = ds.setup_ctx()?;
ctx.set_transaction(ds.transaction(TransactionType::Write).await?.into());
let ctx = ctx.freeze();
let d = docs.resolve_or_assign(&ctx, id).await?;
ctx.tx().commit().await?;
Ok(d)
}
async fn seed_pending_reclaim(
ds: &Datastore,
ns: NamespaceId,
db: DatabaseId,
table: &TableName,
id: &RecordIdKey,
) -> Result<()> {
let tx = ds.transaction(TransactionType::Write).await?;
tx.set_key(&DocPendingKey::new(ns, db, Cow::Borrowed(table), RecordIdentity(id.clone())), &())
.await?;
tx.commit().await
}
async fn seed_pending_reclaim_in_previous_encoding(
ds: &Datastore,
ns: NamespaceId,
db: DatabaseId,
table: &TableName,
id: &RecordIdKey,
) -> Result<()> {
let mut key =
DocLookupKey::new(ns, db, Cow::Borrowed(table), Cow::Borrowed(id)).encode_key()?.to_vec();
let tag = key
.windows(3)
.position(|w| w == b"!di")
.expect("the erased forward key carries its route tag");
key[tag..tag + 3].copy_from_slice(b"!dp");
let tx = ds.transaction(TransactionType::Write).await?;
tx.set(Key::from(key), Vec::new()).await?;
tx.commit().await
}
async fn count_pending_reclaims(
ds: &Datastore,
ns: NamespaceId,
db: DatabaseId,
table: &TableName,
) -> Result<usize> {
let tx = ds.transaction(TransactionType::Read).await?;
let rng = DocPendingPrefix::new(ns, db, Cow::Borrowed(table)).range()?;
let keys = tx.keys(rng, u32::MAX, 0, None).await?;
tx.cancel().await?;
Ok(keys.len())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn concurrent_doc_id_index_build_reclaims_deferred_mapping() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE TABLE t SCHEMALESS;
DEFINE ANALYZER simple TOKENIZERS blank;
DEFINE INDEX ia ON t FIELDS a FULLTEXT ANALYZER simple BM25;
DEFINE INDEX ib ON t FIELDS b FULLTEXT ANALYZER simple BM25;",
)
.await?;
let (ns, db, table, ia) = get_table_index(&ds, "t", "ia").await?;
let (_, _, _, ib) = get_table_index(&ds, "t", "ib").await?;
let ib_ikb = IndexKeyBase::new(ns, db, table.clone(), ib.index_id);
let docs = crate::idx::docids::TableDocIds::new(ns, db, table.clone());
let gone = RecordIdKey::from("gone".to_owned());
let assigned = assign_shared_doc_id(&ds, &docs, &gone).await?;
assert_eq!(
read_shared_doc_id(&ds, &docs, &gone).await?,
Some(assigned),
"the deleted record must start with a shared mapping"
);
seed_pending_reclaim(&ds, ns, db, &table, &gone).await?;
let ia_building = new_building_for_index(&ds, &session, ns, db, &table, ia).await?;
set_durable_build_state(
&ds,
&ib_ikb,
durable_build_state_for_phase(IndexBuildPhase::Building, 1, Some(Uuid::now_v7())),
)
.await?;
ia_building.reclaim_deferred_doc_ids().await?;
assert_eq!(
read_shared_doc_id(&ds, &docs, &gone).await?,
Some(assigned),
"mapping must survive while a sibling doc-ID index is still building"
);
assert_eq!(
count_pending_reclaims(&ds, ns, db, &table).await?,
1,
"the durable marker must survive a sweep that bails on a building sibling"
);
set_durable_build_state(
&ds,
&ib_ikb,
durable_build_state_for_phase(IndexBuildPhase::Online, 1, None),
)
.await?;
ia_building.reclaim_deferred_doc_ids().await?;
assert_eq!(
read_shared_doc_id(&ds, &docs, &gone).await?,
None,
"the sweep must reclaim the forward mapping once no sibling is building"
);
assert_eq!(
read_shared_record_id(&ds, &docs, assigned).await?,
None,
"the sweep must reclaim the reverse mapping too"
);
assert_eq!(
count_pending_reclaims(&ds, ns, db, &table).await?,
0,
"the sweep must consume the pending-reclaim marker"
);
let recreated = assign_shared_doc_id(&ds, &docs, &gone).await?;
assert_ne!(recreated, assigned, "re-create after reclaim must allocate a fresh doc-ID");
execute_all(&ds, &session, "CREATE t:live SET a = 'alpha', b = 'beta';").await?;
let live = RecordIdKey::from("live".to_owned());
let live_id = read_shared_doc_id(&ds, &docs, &live).await?;
assert!(live_id.is_some(), "the live record must have a shared mapping");
seed_pending_reclaim(&ds, ns, db, &table, &live).await?;
ia_building.reclaim_deferred_doc_ids().await?;
assert_eq!(
read_shared_doc_id(&ds, &docs, &live).await?,
live_id,
"a still-present (re-created) record must keep its live mapping"
);
assert_eq!(
count_pending_reclaims(&ds, ns, db, &table).await?,
0,
"a stale marker for a live record must still be consumed"
);
let nested = RecordIdKey::Array(vec![Value::Number(crate::val::Number::Int(1))].into());
let nested_id = assign_shared_doc_id(&ds, &docs, &nested).await?;
seed_pending_reclaim(&ds, ns, db, &table, &nested).await?;
ia_building.reclaim_deferred_doc_ids().await?;
assert_eq!(
read_shared_doc_id(&ds, &docs, &nested).await?,
None,
"a nested-number id must have its forward mapping reclaimed like any other"
);
assert_eq!(
read_shared_record_id(&ds, &docs, nested_id).await?,
None,
"and its reverse mapping too"
);
assert_eq!(
count_pending_reclaims(&ds, ns, db, &table).await?,
0,
"its marker must be consumed"
);
let legacy = RecordIdKey::Array(vec![Value::Number(crate::val::Number::Int(2))].into());
let legacy_id = assign_shared_doc_id(&ds, &docs, &legacy).await?;
seed_pending_reclaim_in_previous_encoding(&ds, ns, db, &table, &legacy).await?;
let readable = RecordIdKey::from("also-gone".to_owned());
let readable_id = assign_shared_doc_id(&ds, &docs, &readable).await?;
seed_pending_reclaim(&ds, ns, db, &table, &readable).await?;
assert_eq!(count_pending_reclaims(&ds, ns, db, &table).await?, 2);
ia_building.reclaim_deferred_doc_ids().await?;
assert_eq!(
count_pending_reclaims(&ds, ns, db, &table).await?,
0,
"both spellings of a marker must be consumed, whichever the scan returned"
);
assert_eq!(
read_shared_doc_id(&ds, &docs, &legacy).await?,
None,
"a marker in the previous spelling must still get its mapping reclaimed"
);
assert_eq!(read_shared_record_id(&ds, &docs, legacy_id).await?, None, "in both directions");
assert_eq!(
read_shared_doc_id(&ds, &docs, &readable).await?,
None,
"and the markers beside it are unaffected"
);
assert_eq!(read_shared_record_id(&ds, &docs, readable_id).await?, None);
for float in [0.1, 0.7] {
let id = RecordIdKey::Array(vec![Value::Number(crate::val::Number::Float(float))].into());
let doc_id = assign_shared_doc_id(&ds, &docs, &id).await?;
seed_pending_reclaim_in_previous_encoding(&ds, ns, db, &table, &id).await?;
ia_building.reclaim_deferred_doc_ids().await?;
assert_eq!(
read_shared_doc_id(&ds, &docs, &id).await?,
None,
"a float id's marker in the previous spelling must reclaim its mapping"
);
assert_eq!(read_shared_record_id(&ds, &docs, doc_id).await?, None, "in both directions");
assert_eq!(count_pending_reclaims(&ds, ns, db, &table).await?, 0);
}
let displaced = RecordIdKey::Array(vec![Value::Number(crate::val::Number::Int(7))].into());
let displaced_id = assign_shared_doc_id(&ds, &docs, &displaced).await?;
execute_all(&ds, &session, "CREATE t:[7dec] SET a = 'alpha', b = 'beta';").await?;
let twin = RecordIdKey::Array(
vec![Value::Number(crate::val::Number::Decimal(rust_decimal::Decimal::from(7)))].into(),
);
let twin_id = read_shared_doc_id(&ds, &docs, &twin).await?.expect("the live twin is indexed");
{
let tx = ds.transaction(TransactionType::Read).await?;
let erased = DocLookupKey::new(ns, db, Cow::Borrowed(&table), Cow::Borrowed(&displaced));
let held = tx.get_key(&erased, None).await?;
tx.cancel().await?;
assert_eq!(held, Some(twin_id), "the premise: the twin holds the erased key");
}
seed_pending_reclaim_in_previous_encoding(&ds, ns, db, &table, &displaced).await?;
ia_building.reclaim_deferred_doc_ids().await?;
assert_eq!(
read_shared_doc_id(&ds, &docs, &displaced).await?,
None,
"the marked record's own mapping must be reclaimed"
);
assert_eq!(read_shared_record_id(&ds, &docs, displaced_id).await?, None, "in both directions");
assert_eq!(
read_shared_doc_id(&ds, &docs, &twin).await?,
Some(twin_id),
"and the live record the chain reached keeps its own"
);
assert_eq!(count_pending_reclaims(&ds, ns, db, &table).await?, 0);
seed_pending_reclaim_in_previous_encoding(&ds, ns, db, &table, &legacy).await?;
ia_building.reclaim_deferred_doc_ids().await?;
assert_eq!(
count_pending_reclaims(&ds, ns, db, &table).await?,
0,
"an unresolvable marker must be consumed, not left to wedge the range"
);
let aliased =
RecordIdKey::Array(vec![Value::Number(crate::val::Number::Int(2)), Value::None].into());
let aliased_id = assign_shared_doc_id(&ds, &docs, &aliased).await?;
seed_pending_reclaim_in_previous_encoding(&ds, ns, db, &table, &aliased).await?;
{
let tx = ds.transaction(TransactionType::Read).await?;
let rng = DocPendingPrefix::new(ns, db, Cow::Borrowed(&table)).range()?;
let keys = tx.keys(rng, u32::MAX, 0, None).await?;
tx.cancel().await?;
let decoded = DocPendingKey::decode_key(&keys[0]).expect("the aliased spelling decodes");
assert!(
!decoded.id.0.addresses_same_record(&aliased),
"and decodes as a different record, which is what makes it dangerous"
);
}
ia_building.reclaim_deferred_doc_ids().await?;
assert_eq!(
read_shared_doc_id(&ds, &docs, &aliased).await?,
None,
"an aliased marker must still reclaim the mapping of the record it names"
);
assert_eq!(read_shared_record_id(&ds, &docs, aliased_id).await?, None);
assert_eq!(count_pending_reclaims(&ds, ns, db, &table).await?, 0);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn deferred_doc_id_reclaim_survives_across_builds() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE ANALYZER simple TOKENIZERS blank;
CREATE t:1 SET a = 'alpha', b = 'x';
CREATE t:2 SET a = 'beta', b = 'y';
CREATE t:3 SET a = 'gamma', b = 'z';
DEFINE INDEX ft1 ON t FIELDS a FULLTEXT ANALYZER simple BM25;
DEFINE INDEX ft3 ON t FIELDS b FULLTEXT ANALYZER simple BM25;",
)
.await?;
assert_eq!(count_doc_id_mappings(&ds, "t").await?, (3, 3));
let (ns, db, table, _) = get_table_index(&ds, "t", "ft3").await?;
let ft3_ikb = {
let (_, _, _, ft3) = get_table_index(&ds, "t", "ft3").await?;
IndexKeyBase::new(ns, db, table.clone(), ft3.index_id)
};
let ft3_online = durable_build_state(&ds, &ft3_ikb).await?;
let guard = start_index_build_paused(
&ds,
&session,
"DEFINE INDEX ft2 ON t FIELDS b FULLTEXT ANALYZER simple BM25 CONCURRENTLY",
)
.await?;
set_durable_build_state(
&ds,
&ft3_ikb,
durable_build_state_for_phase(
IndexBuildPhase::Building,
ft3_online.generation,
Some(Uuid::now_v7()),
),
)
.await?;
execute_all_retrying_conflicts(&ds, &session, "DELETE t:2;").await?;
drop(guard);
wait_for_index_ready(&ds, &session, "t", "ft2").await?;
assert_eq!(
count_doc_id_mappings(&ds, "t").await?,
(3, 3),
"the deleted record's mapping must be retained while a sibling builds"
);
assert_eq!(
count_pending_reclaims(&ds, ns, db, &table).await?,
1,
"the replayed delete must leave a durable pending-reclaim marker"
);
set_durable_build_state(&ds, &ft3_ikb, ft3_online).await?;
execute_all(&ds, &session, "REBUILD INDEX ft2 ON t;").await?;
assert_eq!(
count_doc_id_mappings(&ds, "t").await?,
(2, 2),
"the next doc-ID index build must reclaim the deferred mapping"
);
assert_eq!(
count_pending_reclaims(&ds, ns, db, &table).await?,
0,
"the reclaiming sweep must consume the durable marker"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn deferred_doc_id_reclaim_survives_builder_removal() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE ANALYZER simple TOKENIZERS blank;
CREATE t:1 SET a = 'alpha', b = 'x';
CREATE t:2 SET a = 'beta', b = 'y';
CREATE t:3 SET a = 'gamma', b = 'z';
DEFINE INDEX ft1 ON t FIELDS a FULLTEXT ANALYZER simple BM25;",
)
.await?;
assert_eq!(count_doc_id_mappings(&ds, "t").await?, (3, 3));
let (ns, db, table, _) = get_table_index(&ds, "t", "ft1").await?;
let guard = start_index_build_paused(
&ds,
&session,
"DEFINE INDEX ft2 ON t FIELDS b FULLTEXT ANALYZER simple BM25 CONCURRENTLY",
)
.await?;
execute_all_retrying_conflicts(&ds, &session, "DELETE t:2;").await?;
assert_eq!(
count_pending_reclaims(&ds, ns, db, &table).await?,
1,
"the delete transaction must write the durable marker at deferral time"
);
assert_eq!(count_doc_id_mappings(&ds, "t").await?, (3, 3));
execute_all_retrying_conflicts(&ds, &session, "REMOVE INDEX ft2 ON t;").await?;
drop(guard);
assert_eq!(
count_pending_reclaims(&ds, ns, db, &table).await?,
1,
"the marker must survive a builder that never replayed the delete"
);
assert_eq!(count_doc_id_mappings(&ds, "t").await?, (3, 3));
execute_all(&ds, &session, "REBUILD INDEX ft1 ON t;").await?;
assert_eq!(
count_doc_id_mappings(&ds, "t").await?,
(2, 2),
"the next doc-ID index build must reclaim the never-replayed delete's mapping"
);
assert_eq!(count_pending_reclaims(&ds, ns, db, &table).await?, 0);
Ok(())
}
#[cfg(feature = "kv-mem")]
async fn knn_result(ds: &Datastore, session: &Session, sql: &str) -> Result<String> {
let mut responses = ds.execute(sql, session, None).await?;
let value = responses.remove(0).result?;
Ok(value.into_json_value().to_string())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn hnsw_pending_search_returns_recreated_record() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE ANALYZER simple TOKENIZERS blank;
DEFINE INDEX ft ON t FIELDS a FULLTEXT ANALYZER simple BM25;
DEFINE INDEX hx ON t FIELDS vec HNSW DIMENSION 4 DIST EUCLIDEAN TYPE F32;
CREATE t:1 SET a = 'alpha', vec = [1.0, 0.0, 0.0, 0.0];",
)
.await?;
execute_all(
&ds,
&session,
"DELETE t:1;
CREATE t:1 SET a = 'alpha', vec = [1.0, 0.0, 0.0, 0.0];",
)
.await?;
let knn = "SELECT id FROM t WHERE vec <|1,40|> [1.0, 0.0, 0.0, 0.0];";
assert!(
knn_result(&ds, &session, knn).await?.contains("t:1"),
"a record re-created before compaction must remain visible to KNN"
);
Ok(())
}
#[cfg(all(feature = "kv-mem", diskann))]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn diskann_pending_search_returns_recreated_record() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE ANALYZER simple TOKENIZERS blank;
DEFINE INDEX ft ON t FIELDS a FULLTEXT ANALYZER simple BM25;
DEFINE INDEX dx ON t FIELDS vec DISKANN DIMENSION 4 DIST EUCLIDEAN TYPE F32;
CREATE t:1 SET a = 'alpha', vec = [1.0, 0.0, 0.0, 0.0];",
)
.await?;
execute_all(
&ds,
&session,
"DELETE t:1;
CREATE t:1 SET a = 'alpha', vec = [1.0, 0.0, 0.0, 0.0];",
)
.await?;
let knn = "SELECT id FROM t WHERE vec <|1,40|> [1.0, 0.0, 0.0, 0.0];";
assert!(
knn_result(&ds, &session, knn).await?.contains("t:1"),
"a record re-created before compaction must remain visible to KNN"
);
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn shutdown_error_does_not_poison_durable_build_state() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
CREATE user:two SET email = 'two@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email CONCURRENTLY;
",
)
.await?;
wait_for_index_ready(&ds, &session, "user", "test").await?;
let _guard = inject_non_retryable_error(
NonRetryableErrorSite::ConcurrentIndexInitialBatchShutdown,
ds.id(),
);
let err = ds
.execute("REBUILD INDEX test ON user", &session, None)
.await?
.remove(0)
.result
.expect_err("rebuild should fail with the injected shutdown error");
assert!(err.to_string().contains("shutting down"), "unexpected rebuild error: {err}");
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let state = durable_build_state(&ds, &ikb).await?;
assert_eq!(state.phase, IndexBuildPhase::Building);
assert_eq!(state.error, None);
assert_eq!(state.report_status, Some(IndexBuildReportStatus::Indexing));
execute_all_retrying_conflicts(
&ds,
&session,
"CREATE user:three SET email = 'three@example.com' RETURN NONE",
)
.await?;
let expired = Utc::now() - chrono::Duration::seconds(BUILD_OWNER_LEASE_SECS + 5);
let mut stranded = durable_build_state(&ds, &ikb).await?;
stranded.updated_at = expired;
stranded.owner_heartbeat_at = Some(expired);
set_durable_build_state(&ds, &ikb, stranded).await?;
let resumed = ds
.resume_stalled_index_builds(
Duration::from_secs(30),
tokio_util::sync::CancellationToken::new(),
)
.await?;
assert_eq!(resumed, 1, "the interrupted build should be adopted");
wait_for_index_ready(&ds, &session, "user", "test").await?;
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn non_shutdown_build_error_still_publishes_durable_error() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
CREATE user:two SET email = 'two@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email CONCURRENTLY;
",
)
.await?;
wait_for_index_ready(&ds, &session, "user", "test").await?;
let _guard = inject_non_retryable_error(
NonRetryableErrorSite::ConcurrentIndexInitialBatchCommit,
ds.id(),
);
let err = ds
.execute("REBUILD INDEX test ON user", &session, None)
.await?
.remove(0)
.result
.expect_err("rebuild should fail with the injected error");
assert!(err.to_string().contains("injected non-retryable error"), "unexpected error: {err}");
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let state = durable_build_state(&ds, &ikb).await?;
assert_eq!(state.phase, IndexBuildPhase::Error);
assert_eq!(state.report_status, Some(IndexBuildReportStatus::Error));
assert!(
state.error.as_deref().is_some_and(|e| e.contains("injected non-retryable error")),
"durable error should carry the failure reason: {:?}",
state.error
);
execute_all_retrying_conflicts(
&ds,
&session,
"CREATE user:three SET email = 'three@example.com' RETURN NONE",
)
.await?;
execute_all(&ds, &session, "REBUILD INDEX test ON user").await?;
assert_eq!(index_building_status(&ds, &session, "user", "test").await?, "ready");
assert_eq!(
index_prefix_key_count(&ds, ns, db, &table, ix.index_id).await?,
3,
"the rebuild rescan must index the write admitted during the error state"
);
Ok(())
}
#[test]
fn shutdown_error_classification() {
use crate::kvs::is_shutdown_error;
let direct: anyhow::Error = crate::kvs::Error::Shutdown.into();
assert!(is_shutdown_error(&direct));
let wrapped: anyhow::Error = Error::Kvs(crate::kvs::Error::Shutdown).into();
assert!(is_shutdown_error(&wrapped));
let internal: anyhow::Error = crate::kvs::Error::Internal("boom".to_string()).into();
assert!(!is_shutdown_error(&internal));
let conflict: anyhow::Error = crate::kvs::Error::TransactionConflict("busy".to_string()).into();
assert!(!is_shutdown_error(&conflict));
}
#[tokio::test(flavor = "multi_thread")]
async fn memory_threshold_error_does_not_poison_durable_build_state() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
CREATE user:two SET email = 'two@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email CONCURRENTLY;
",
)
.await?;
wait_for_index_ready(&ds, &session, "user", "test").await?;
let _guard = inject_non_retryable_error(
NonRetryableErrorSite::ConcurrentIndexInitialBatchMemoryThreshold,
ds.id(),
);
let err = ds
.execute("REBUILD INDEX test ON user", &session, None)
.await?
.remove(0)
.result
.expect_err("rebuild should fail with the injected memory threshold error");
assert!(err.to_string().contains("memory threshold"), "unexpected rebuild error: {err}");
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let state = durable_build_state(&ds, &ikb).await?;
assert_eq!(state.phase, IndexBuildPhase::Building);
assert_eq!(state.error, None);
assert_eq!(state.report_status, Some(IndexBuildReportStatus::Indexing));
let expired = Utc::now() - chrono::Duration::seconds(BUILD_OWNER_LEASE_SECS + 5);
let mut stranded = durable_build_state(&ds, &ikb).await?;
stranded.updated_at = expired;
stranded.owner_heartbeat_at = Some(expired);
set_durable_build_state(&ds, &ikb, stranded).await?;
let resumed = ds
.resume_stalled_index_builds(
Duration::from_secs(30),
tokio_util::sync::CancellationToken::new(),
)
.await?;
assert_eq!(resumed, 1, "the interrupted build should be adopted");
wait_for_index_ready(&ds, &session, "user", "test").await?;
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn resume_scan_adopts_one_stalled_build_at_a_time() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com', name = 'One' RETURN NONE;
CREATE user:two SET email = 'two@example.com', name = 'Two' RETURN NONE;
DEFINE INDEX test_email ON user FIELDS email;
DEFINE INDEX test_name ON user FIELDS name;
",
)
.await?;
let expired = Utc::now() - chrono::Duration::seconds(BUILD_OWNER_LEASE_SECS + 5);
let mut ikbs = Vec::new();
for index in ["test_email", "test_name"] {
let (ns, db, table, ix) = get_table_index(&ds, "user", index).await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let tx = ds.transaction(TransactionType::Write).await?;
tx.del_prefix_key(&IdxRoot {
ns,
db,
tb: Cow::Borrowed(&table),
ix: ix.index_id,
})
.await?;
tx.set_key(
&ikb.new_bs_key(),
&IndexBuildState {
generation: 2,
phase: IndexBuildPhase::Building,
owner: Some(Uuid::new_v4()),
next_ticket: 0,
initial_complete: false,
updated_at: expired,
owner_heartbeat_at: Some(expired),
error: None,
report_status: Some(IndexBuildReportStatus::Indexing),
initial: Some(0),
updated: None,
pending: None,
initial_cursor: None,
},
)
.await?;
tx.commit().await?;
ikbs.push(ikb);
}
let guard = inject_retryable_conflicts(
RetryableConflictSite::ConcurrentIndexInitialBatch,
ds.id(),
REPEATED_RETRY_CONFLICTS,
);
let resumed = ds
.resume_stalled_index_builds(
Duration::from_secs(30),
tokio_util::sync::CancellationToken::new(),
)
.await?;
assert_eq!(resumed, 1, "a pass must adopt at most one stalled build");
timeout(Duration::from_secs(10), async {
while retryable_conflict_count(RetryableConflictSite::ConcurrentIndexInitialBatch, ds.id())
== REPEATED_RETRY_CONFLICTS
{
sleep(Duration::from_millis(5)).await;
}
})
.await
.map_err(|_| anyhow::anyhow!("adopted build never reached its batch commit"))?;
let resumed_while_running = ds
.resume_stalled_index_builds(
Duration::from_secs(30),
tokio_util::sync::CancellationToken::new(),
)
.await?;
assert_eq!(resumed_while_running, 0, "no adoption while a local build is running");
drop(guard);
let deadline = Instant::now() + Duration::from_secs(30);
loop {
let mut online = 0;
for ikb in &ikbs {
if durable_build_state(&ds, ikb).await?.phase == IndexBuildPhase::Online {
online += 1;
}
}
if online == 2 {
break;
}
assert!(Instant::now() < deadline, "stranded builds were not recovered sequentially");
let _ = ds
.resume_stalled_index_builds(
Duration::from_secs(30),
tokio_util::sync::CancellationToken::new(),
)
.await?;
sleep(Duration::from_millis(20)).await;
}
assert_eq!(index_building_status(&ds, &session, "user", "test_email").await?, "ready");
assert_eq!(index_building_status(&ds, &session, "user", "test_name").await?, "ready");
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn takeover_installs_generation_before_draining_reservations() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let tx = ds.transaction(TransactionType::Write).await?;
tx.set_key(
&ikb.new_bs_key(),
&IndexBuildState {
generation: 1,
phase: IndexBuildPhase::Error,
owner: None,
next_ticket: 1,
initial_complete: false,
updated_at: Utc::now(),
owner_heartbeat_at: None,
error: Some("seeded test failure".to_string()),
report_status: Some(IndexBuildReportStatus::Error),
initial: None,
updated: None,
pending: None,
initial_cursor: None,
},
)
.await?;
let br_key = ikb.new_br_key(1, 0);
tx.set_key(
&br_key,
&IndexBuildReservation {
node: ds.id(),
expires_at: Utc::now() + chrono::Duration::seconds(BUILD_RESERVATION_TTL_SECS),
},
)
.await?;
tx.set_key(
&ikb.new_bg_key(1, 5, 0),
&Appending {
old_values: None,
new_values: None,
id: RecordIdKey::from("one".to_string()),
count_cond_match: None,
},
)
.await?;
tx.commit().await?;
let building =
Arc::new(new_building_for_index(&ds, &session, ns, db, &table, Arc::clone(&ix)).await?);
let acquire = {
let building = Arc::clone(&building);
tokio::spawn(async move { building.acquire_build_state().await })
};
timeout(Duration::from_secs(5), async {
while durable_build_state(&ds, &ikb).await?.generation != 2 {
sleep(Duration::from_millis(10)).await;
}
Ok::<_, anyhow::Error>(())
})
.await
.map_err(|_| anyhow::anyhow!("takeover never installed the next generation"))??;
let state = durable_build_state(&ds, &ikb).await?;
assert_eq!(state.phase, IndexBuildPhase::Building);
sleep(Duration::from_millis(200)).await;
assert!(
!acquire.is_finished(),
"acquire must keep draining the live prior-generation reservation"
);
let tx = ds.transaction(TransactionType::Write).await?;
tx.del_key(&br_key).await?;
tx.commit().await?;
let acquired = timeout(Duration::from_secs(5), acquire)
.await
.map_err(|_| {
anyhow::anyhow!("acquire did not finish after the reservation was released")
})???
.expect("takeover should acquire the new generation");
assert_eq!(acquired.generation, 2);
let tx = ds.transaction(TransactionType::Read).await?;
let stale_bg = tx.keys(ikb.new_bg_all_generations_range()?, u32::MAX, 0, None).await?;
let stale_br = tx.keys(ikb.new_br_all_generations_range()?, u32::MAX, 0, None).await?;
tx.cancel().await?;
assert!(stale_bg.is_empty(), "stale generation-1 queue entries should be wiped");
assert!(stale_br.is_empty(), "no reservations should remain after the takeover");
Ok(())
}
const BUILD_UNDER_WRITES_TIMEOUT: Duration = Duration::from_secs(45);
const BUILD_PUBLISH_TIMEOUT: Duration = Duration::from_secs(30);
struct BuildUnderWritesOutcome {
scan_progressed: bool,
published: bool,
max_initial: u64,
max_ticket: u64,
committed: u64,
}
async fn concurrent_build_under_table_writes(
writer_sql: fn(u64) -> String,
scan_timeout: Duration,
) -> Result<BuildUnderWritesOutcome> {
const RECORDS: usize = 4000;
const WRITES_BEFORE_BUILD: u64 = 50;
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE doc SCHEMALESS;
DEFINE ANALYZER simple TOKENIZERS blank,class FILTERS lowercase;
",
)
.await?;
execute_all(
&ds,
&session,
&format!(
"CREATE |doc:1..{RECORDS}| SET text = string::repeat('lorem ipsum dolor sit amet \
consectetur adipiscing elit sed do eiusmod tempor incididunt ut labore ', 4) + \
<string>id RETURN NONE"
),
)
.await?;
let stop = Arc::new(AtomicBool::new(false));
let committed = Arc::new(AtomicU64::new(0));
let writer = {
let ds = Arc::clone(&ds);
let session = session.clone();
let stop = Arc::clone(&stop);
let committed = Arc::clone(&committed);
tokio::spawn(async move {
let mut n = 0u64;
while !stop.load(Ordering::Relaxed) {
n += 1;
if execute_all(&ds, &session, &writer_sql(n)).await.is_ok() {
committed.fetch_add(1, Ordering::Relaxed);
}
sleep(Duration::from_millis(1)).await;
}
})
};
timeout(Duration::from_secs(10), async {
while committed.load(Ordering::Relaxed) < WRITES_BEFORE_BUILD {
sleep(Duration::from_millis(5)).await;
}
})
.await
.map_err(|_| anyhow::anyhow!("the writer never reached the pre-build write threshold"))?;
execute_all_retrying_conflicts(
&ds,
&session,
"DEFINE INDEX ft ON doc FIELDS text FULLTEXT ANALYZER simple BM25(1.2,0.75) HIGHLIGHTS \
CONCURRENTLY",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "doc", "ft").await?;
let ikb = IndexKeyBase::new(ns, db, table, ix.index_id);
let mut max_initial = 0u64;
let mut max_ticket = 0u64;
let progressed = timeout(scan_timeout, async {
loop {
let (state, ticket) = durable_build_state_with_ticket_counter(&ds, &ikb).await?;
max_initial = max_initial.max(state.initial.unwrap_or(0));
if state.phase == IndexBuildPhase::Building {
max_ticket = max_ticket.max(ticket.unwrap_or(state.next_ticket));
}
if max_initial >= RECORDS as u64 || state.phase == IndexBuildPhase::Online {
return Ok::<_, anyhow::Error>(());
}
sleep(Duration::from_millis(20)).await;
}
})
.await;
stop.store(true, Ordering::Relaxed);
let _ = writer.await;
let scan_progressed = match progressed {
Ok(Ok(())) => true,
Ok(Err(err)) => return Err(err),
Err(_elapsed) => false,
};
let published = timeout(BUILD_PUBLISH_TIMEOUT, async {
loop {
let state = durable_build_state(&ds, &ikb).await?;
max_initial = max_initial.max(state.initial.unwrap_or(0));
if state.phase == IndexBuildPhase::Online {
return Ok::<_, anyhow::Error>(());
}
sleep(Duration::from_millis(20)).await;
}
})
.await;
let published = match published {
Ok(Ok(())) => true,
Ok(Err(err)) => return Err(err),
Err(_elapsed) => false,
};
let committed = committed.load(Ordering::Relaxed);
assert!(
committed > 100,
"the writer only committed {committed} writes — too little load for this test to mean \
anything"
);
Ok(BuildUnderWritesOutcome {
scan_progressed,
published,
max_initial,
max_ticket,
committed,
})
}
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
async fn concurrent_build_is_unaffected_by_writes_that_miss_the_index() -> Result<()> {
let outcome = concurrent_build_under_table_writes(
|n| format!("UPDATE doc:{} SET untracked = {n} RETURN NONE", n % 25 + 1),
BUILD_UNDER_WRITES_TIMEOUT,
)
.await?;
assert_eq!(
outcome.max_ticket, 0,
"writes that miss the index must not enter admission (allocated {} tickets)",
outcome.max_ticket
);
assert!(
outcome.scan_progressed,
"the scan stalled under {} writes that miss the index: {}/4000 records",
outcome.committed, outcome.max_initial
);
assert!(outcome.published, "the build did not publish after the writer stopped");
Ok(())
}
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
async fn concurrent_build_publishes_under_indexed_updates() -> Result<()> {
let outcome = concurrent_build_under_table_writes(
|n| format!("UPDATE doc:{} SET text = 'changed payload {n}' RETURN NONE", n % 25 + 1),
BUILD_UNDER_WRITES_TIMEOUT,
)
.await?;
assert!(
outcome.max_ticket > 0,
"indexed updates must enter writer admission, otherwise this test exercises nothing"
);
assert!(
outcome.scan_progressed,
"the scan stalled under {} indexed updates ({} tickets allocated): it indexed {}/4000 \
records",
outcome.committed, outcome.max_ticket, outcome.max_initial
);
assert!(outcome.published, "the build did not publish after the writer stopped");
Ok(())
}
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
async fn concurrent_build_publishes_under_indexed_inserts() -> Result<()> {
let outcome = concurrent_build_under_table_writes(
|n| format!("CREATE doc:w{n} SET text = 'writer payload {n}' RETURN NONE"),
BUILD_UNDER_WRITES_TIMEOUT,
)
.await?;
assert!(
outcome.max_ticket > 0,
"indexed inserts must enter writer admission, otherwise this test exercises nothing"
);
assert!(
outcome.scan_progressed,
"the scan stalled under {} indexed inserts ({} tickets allocated): it indexed {}/4000 \
records",
outcome.committed, outcome.max_ticket, outcome.max_initial
);
assert!(outcome.published, "the build did not publish after the writer stopped");
Ok(())
}
async fn durable_ticket_counter(
ds: &Datastore,
ikb: &IndexKeyBase,
generation: BuildGeneration,
) -> Result<Option<BuildTicket>> {
let tx = ds.transaction(TransactionType::Read).await?;
let counter = catch!(tx, tx.get_key(&ikb.new_bt_key(generation), None).await);
tx.cancel().await?;
Ok(counter)
}
#[tokio::test(flavor = "multi_thread")]
async fn generation_flip_rotates_the_ticket_counter() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table, ix.index_id);
let first = durable_build_state(&ds, &ikb).await?.generation;
assert!(
durable_ticket_counter(&ds, &ikb, first).await?.is_some(),
"the active generation must own a ticket counter for admission to CAS"
);
execute_all(&ds, &session, "REBUILD INDEX test ON user").await?;
let second = durable_build_state(&ds, &ikb).await?.generation;
assert_eq!(second, first.saturating_add(1));
assert!(
durable_ticket_counter(&ds, &ikb, second).await?.is_some(),
"the rebuilt generation must own a ticket counter"
);
assert!(
durable_ticket_counter(&ds, &ikb, first).await?.is_none(),
"the flip must remove the previous generation's counter, which is what fences writers \
still admitting under it"
);
execute_all(&ds, &session, "REMOVE INDEX test ON user").await?;
assert!(durable_ticket_counter(&ds, &ikb, second).await?.is_none());
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn admission_falls_back_to_build_state_ticket_without_a_counter() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table, ix.index_id);
let generation = durable_build_state(&ds, &ikb).await?.generation;
let mut legacy = durable_build_state_for_phase(IndexBuildPhase::Building, generation, None);
legacy.next_ticket = 7;
set_durable_build_state(&ds, &ikb, legacy).await?;
let tx = ds.transaction(TransactionType::Write).await?;
tx.del_key(&ikb.new_bt_key(generation)).await?;
tx.commit().await?;
execute_all_retrying_conflicts(
&ds,
&session,
"UPDATE user:one SET email = 'changed@example.com' RETURN NONE",
)
.await?;
let state = durable_build_state(&ds, &ikb).await?;
assert_eq!(
state.next_ticket, 8,
"a generation without a counter must advance the build state's own ticket"
);
assert!(
durable_ticket_counter(&ds, &ikb, generation).await?.is_none(),
"the legacy path must not create a counter mid-generation: nodes still running the \
previous version would keep allocating from `!bs.next_ticket` and collide with it"
);
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn closing_transition_fences_in_flight_ticket_allocation() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
const GENERATION: BuildGeneration = 9;
let expired = Utc::now() - chrono::Duration::seconds(BUILD_OWNER_LEASE_SECS + 5);
let mut state = durable_build_state_for_phase(IndexBuildPhase::Building, GENERATION, None);
state.updated_at = expired;
state.owner_heartbeat_at = Some(expired);
set_durable_build_state(&ds, &ikb, state).await?;
let tx = ds.transaction(TransactionType::Write).await?;
tx.set_key(&ikb.new_bt_key(GENERATION), &5u64).await?;
tx.commit().await?;
let build = new_building_for_index(&ds, &session, ns, db, &table, Arc::clone(&ix)).await?;
let acquired = build
.acquire_build_state()
.await?
.expect("the expired generation should be available for takeover");
assert_eq!(acquired.generation, GENERATION);
let writer = ds.transaction(TransactionType::Write).await?;
let bt = ikb.new_bt_key(GENERATION);
let ticket = catch!(writer, writer.get_key(&bt, None).await).expect("counter should exist");
assert_eq!(ticket, 5);
catch!(writer, writer.put_compare_key(&bt, &(ticket + 1), Some(&ticket)).await);
build.mark_durable_closing(GENERATION).await?;
assert_eq!(durable_build_state(&ds, &ikb).await?.phase, IndexBuildPhase::Closing);
assert!(
writer.commit().await.is_err(),
"a ticket allocation in flight across the `Closing` transition must not commit: its \
reservation could land after the drain and be missed"
);
Ok(())
}
#[allow(clippy::too_many_arguments)]
async fn run_takeover_of_generation(
ds: &Datastore,
session: &Session,
ns: NamespaceId,
db: DatabaseId,
table: &TableName,
ix: Arc<IndexDefinition>,
generation: BuildGeneration,
next_ticket: BuildTicket,
counter: Option<BuildTicket>,
) -> Result<Result<()>> {
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let expired = Utc::now() - chrono::Duration::seconds(BUILD_OWNER_LEASE_SECS + 5);
let mut state = durable_build_state_for_phase(IndexBuildPhase::Building, generation, None);
state.next_ticket = next_ticket;
state.initial_complete = false;
state.updated_at = expired;
state.owner_heartbeat_at = Some(expired);
set_durable_build_state(ds, &ikb, state).await?;
let tx = ds.transaction(TransactionType::Write).await?;
match counter {
Some(counter) => tx.set_key(&ikb.new_bt_key(generation), &counter).await?,
None => tx.del_key(&ikb.new_bt_key(generation)).await?,
}
tx.commit().await?;
let build = new_building_for_index(ds, session, ns, db, table, ix).await?;
let acquired = build
.acquire_build_state()
.await?
.expect("the expired generation should be available for takeover");
assert_eq!(acquired.generation, generation);
Ok(build.run_acquired(acquired).await)
}
#[tokio::test(flavor = "multi_thread")]
async fn cross_version_ticket_allocation_blocks_publishing() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let result =
run_takeover_of_generation(&ds, &session, ns, db, &table, ix, 4, 3, Some(7)).await?;
let err = result.expect_err("a generation with two ticket allocators must not publish");
let message = err.to_string();
assert!(
message.contains("REBUILD INDEX test ON user"),
"the failure must tell the operator how to recover, got: {message}"
);
assert_ne!(
durable_build_state(&ds, &ikb).await?.phase,
IndexBuildPhase::Online,
"the index must not be queryable when its queue may have lost mutations"
);
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn legacy_generation_with_advanced_ticket_still_publishes() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
run_takeover_of_generation(&ds, &session, ns, db, &table, ix, 4, 3, None).await??;
assert_eq!(durable_build_state(&ds, &ikb).await?.phase, IndexBuildPhase::Online);
Ok(())
}
#[tokio::test(flavor = "multi_thread")]
async fn generation_flip_fences_in_flight_ticket_allocation() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"
DEFINE TABLE user SCHEMALESS;
CREATE user:one SET email = 'one@example.com' RETURN NONE;
DEFINE INDEX test ON user FIELDS email;
",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "user", "test").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let br_key = ikb.new_br_key(1, 0);
let tx = ds.transaction(TransactionType::Write).await?;
tx.set_key(
&ikb.new_bs_key(),
&IndexBuildState {
generation: 1,
phase: IndexBuildPhase::Error,
owner: None,
next_ticket: 0,
initial_complete: false,
updated_at: Utc::now(),
owner_heartbeat_at: None,
error: Some("seeded test failure".to_string()),
report_status: Some(IndexBuildReportStatus::Error),
initial: None,
updated: None,
pending: None,
initial_cursor: None,
},
)
.await?;
tx.set_key(&ikb.new_bt_key(1), &5u64).await?;
tx.set_key(
&br_key,
&IndexBuildReservation {
node: ds.id(),
expires_at: Utc::now() + chrono::Duration::seconds(BUILD_RESERVATION_TTL_SECS),
},
)
.await?;
tx.commit().await?;
let writer = ds.transaction(TransactionType::Write).await?;
let bt = ikb.new_bt_key(1);
let ticket = catch!(writer, writer.get_key(&bt, None).await).expect("counter should exist");
catch!(writer, writer.put_compare_key(&bt, &(ticket + 1), Some(&ticket)).await);
let building =
Arc::new(new_building_for_index(&ds, &session, ns, db, &table, Arc::clone(&ix)).await?);
let acquire = {
let building = Arc::clone(&building);
tokio::spawn(async move { building.acquire_build_state().await })
};
timeout(Duration::from_secs(5), async {
while durable_build_state(&ds, &ikb).await?.generation != 2 {
sleep(Duration::from_millis(10)).await;
}
Ok::<_, anyhow::Error>(())
})
.await
.map_err(|_| anyhow::anyhow!("takeover never installed the next generation"))??;
assert!(
!acquire.is_finished(),
"the takeover should still be draining, which is the window under test"
);
assert!(
writer.commit().await.is_err(),
"an allocation in flight across the generation flip must not commit: its reservation \
would land after the drain and its queued mutation would be wiped"
);
let tx = ds.transaction(TransactionType::Write).await?;
tx.del_key(&br_key).await?;
tx.commit().await?;
timeout(Duration::from_secs(5), acquire)
.await
.map_err(|_| {
anyhow::anyhow!("takeover did not finish after the reservation was released")
})???
.expect("takeover should acquire the new generation");
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn a_compaction_plan_prepared_before_a_wipe_cannot_apply() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE INDEX hx ON t FIELDS vec HNSW DIMENSION 2 DIST EUCLIDEAN TYPE F32;
CREATE t:1 SET vec = [1.0, 0.0];
CREATE t:2 SET vec = [0.0, 1.0];",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "t", "hx").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let Index::Hnsw(params) = &ix.index else {
panic!("the fixture must define an HNSW index");
};
let tx = Arc::new(ds.transaction(TransactionType::Read).await?);
let pendings = catch!(tx, tx.scan_raw(ikb.new_hr_range()?, u32::MAX, 0, None).await);
let mut ctx = ds.setup_ctx()?;
ctx.set_transaction(Arc::clone(&tx));
let ctx = ctx.freeze();
let plan = IndexOperation::prepare_hnsw_compaction(&ctx, &ikb).await?;
tx.cancel().await?;
assert!(plan.has_work(), "the fixture must leave pendings for the plan to capture");
assert_eq!(pendings.len(), 2, "one pending per indexed record");
let tx = ds.transaction(TransactionType::Write).await?;
catch!(tx, crate::idx::wipe_index_data(&tx, &ikb, &ix.index).await);
for (key, value) in &pendings {
catch!(tx, tx.set(Key::from(key.clone()), value.clone()).await);
}
tx.commit().await?;
let tx = Arc::new(ds.transaction(TransactionType::Write).await?);
let mut ctx = ds.setup_ctx()?;
ctx.set_transaction(Arc::clone(&tx));
let ctx = ctx.freeze();
let applied = IndexOperation::apply_hnsw_compaction(
&ctx,
ctx.get_index_stores(),
&ikb,
params,
crate::catalog::DOC_IDS_FORMAT_VERSION,
plan,
)
.await?;
tx.cancel().await?;
assert!(!applied.applied(), "a plan prepared before the wipe must be rejected, not applied");
let tx = ds.transaction(TransactionType::Read).await?;
let surviving = catch!(tx, tx.count(ikb.new_hr_range()?, None).await);
tx.cancel().await?;
assert_eq!(surviving, pendings.len(), "the rejected plan must leave every pending in place");
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn knn_prefilter_performs_zero_in_traversal_fetches() -> Result<()> {
use surrealdb_types::Value as PublicValue;
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE FIELD category ON t TYPE string;
DEFINE INDEX idx_category ON t FIELDS category;
DEFINE INDEX idx_rawcat ON t FIELDS rawcat;
DEFINE INDEX hn_pt ON t FIELDS point HNSW DIMENSION 1;",
)
.await?;
execute_all(
&ds,
&session,
"FOR $i IN 0..2400 {
CREATE type::record('t', $i) SET
point = [ <float> $i ],
category = IF $i % 12 < 11 { 'a' } ELSE { 'b' },
rawcat = IF $i % 12 < 11 { 'a' } ELSE { 'b' };
};",
)
.await?;
let explain = |sql: &'static str| {
let ds = &ds;
let session = &session;
async move {
let mut results = ds.execute(sql, session, None).await?;
let value = results.remove(0).result?;
match value {
PublicValue::String(plan) => Ok::<String, anyhow::Error>(plan),
other => anyhow::bail!("unexpected EXPLAIN result: {other:?}"),
}
}
};
let plan =
explain("EXPLAIN ANALYZE SELECT id FROM t WHERE category = 'b' AND point <|3,40|> [0f]")
.await?;
assert!(plan.contains("prefilter_tier: exact"), "expected exact tier:\n{plan}");
assert!(!plan.contains("fetched:"), "no in-traversal fetches expected:\n{plan}");
let plan =
explain("EXPLAIN ANALYZE SELECT id FROM t WHERE category = 'a' AND point <|3,40|> [0f]")
.await?;
assert!(plan.contains("prefilter_tier: graph"), "expected graph tier:\n{plan}");
assert!(!plan.contains("fetched:"), "no in-traversal fetches expected:\n{plan}");
let plan =
explain("EXPLAIN ANALYZE SELECT id FROM t WHERE rawcat = 'b' AND point <|3,40|> [0f]")
.await?;
assert!(!plan.contains("prefilter_tier"), "no prefilter expected:\n{plan}");
assert!(plan.contains("fetched:"), "in-traversal fetches expected:\n{plan}");
Ok(())
}
fn primary_append_key_in_previous_encoding(
ikb: &IndexKeyBase,
generation: BuildGeneration,
id: &RecordIdKey,
) -> Result<Vec<u8>> {
use crate::key::schema::BuildPrimaryGenerationPrefix;
let (ns, db, table) = (ikb.ns(), ikb.db(), ikb.table());
let head = BuildPrimaryGenerationPrefix {
ns,
db,
tb: Cow::Borrowed(table),
ix: ikb.index(),
generation,
}
.encode_bound()?
.to_vec();
let erased = DocLookupPrefix::new(ns, db, Cow::Borrowed(table)).encode_bound()?.to_vec();
let di = DocLookupKey::new(ns, db, Cow::Borrowed(table), Cow::Borrowed(id)).encode_key()?;
let mut key = head;
key.extend_from_slice(&di[erased.len()..]);
Ok(key)
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn primary_append_markers_from_a_previous_release_are_rebuilt() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE TABLE t SCHEMALESS;
DEFINE ANALYZER simple TOKENIZERS blank;
DEFINE INDEX ia ON t FIELDS a FULLTEXT ANALYZER simple BM25;",
)
.await?;
let (ns, db, table, ia) = get_table_index(&ds, "t", "ia").await?;
let ix_id = ia.index_id;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix_id);
let building = new_building_for_index(&ds, &session, ns, db, &table, ia).await?;
let generation: BuildGeneration = 1;
let id = RecordIdKey::Array(vec![Value::Number(crate::val::Number::Int(1))].into());
let (ticket, mutation_seq) = (7, 0);
{
let tx = ds.transaction(TransactionType::Write).await?;
tx.set_key(
&ikb.new_bg_key(generation, ticket, mutation_seq),
&Appending {
old_values: None,
new_values: None,
id: id.clone(),
count_cond_match: None,
},
)
.await?;
let legacy = primary_append_key_in_previous_encoding(&ikb, generation, &id)?;
tx.set(
Key::from(legacy),
PrimaryAppendingTicket {
ticket,
mutation_seq,
}
.kv_encode_value()?,
)
.await?;
tx.commit().await?;
}
{
let tx = ds.transaction(TransactionType::Read).await?;
assert!(
tx.get_key(&ikb.new_bp_key(generation, &id), None).await?.is_none(),
"the marker must start out invisible to a lookup by record"
);
tx.cancel().await?;
}
building.build_generation.store(generation, Ordering::Release);
own_build(&ds, &ikb, &building, generation).await?;
building.rebuild_primary_appendings().await?;
let tx = ds.transaction(TransactionType::Read).await?;
let ptr = tx.get_key(&ikb.new_bp_key(generation, &id), None).await?;
let ptr = ptr.expect("the marker must now be found by a lookup for its own record");
assert_eq!(
(ptr.ticket, ptr.mutation_seq),
(ticket, mutation_seq),
"and still point at the same queued mutation"
);
let all = tx.keys(ikb.new_bp_range(generation)?, u32::MAX, 0, None).await?;
assert_eq!(all.len(), 1, "the old spelling must not be left beside the new one");
tx.cancel().await?;
let span = surrealdb_kvs::consts::INDEXING_BATCH_SIZE as usize + 50;
{
let tx = ds.transaction(TransactionType::Write).await?;
for i in 0..span {
let id = RecordIdKey::Array(
vec![Value::Number(crate::val::Number::Int(1000 + i as i64))].into(),
);
let (ticket, mutation_seq) = (1000 + i as u64, 0);
tx.set_key(
&ikb.new_bg_key(generation, ticket, mutation_seq),
&Appending {
old_values: None,
new_values: None,
id: id.clone(),
count_cond_match: None,
},
)
.await?;
tx.set(
Key::from(primary_append_key_in_previous_encoding(&ikb, generation, &id)?),
PrimaryAppendingTicket {
ticket,
mutation_seq,
}
.kv_encode_value()?,
)
.await?;
}
tx.commit().await?;
}
building.rebuild_primary_appendings().await?;
let tx = ds.transaction(TransactionType::Read).await?;
for i in 0..span {
let id = RecordIdKey::Array(
vec![Value::Number(crate::val::Number::Int(1000 + i as i64))].into(),
);
assert!(
tx.get_key(&ikb.new_bp_key(generation, &id), None).await?.is_some(),
"marker {i} must be found by a lookup for its own record"
);
}
let all = tx.keys(ikb.new_bp_range(generation)?, u32::MAX, 0, None).await?;
assert_eq!(
all.len(),
span + 1,
"every marker must end up in exactly one spelling, across page boundaries"
);
tx.cancel().await?;
let three = RecordIdKey::Array(vec![Value::Number(crate::val::Number::Int(3))].into());
let aliased =
RecordIdKey::Array(vec![Value::Number(crate::val::Number::Int(3)), Value::None].into());
let (ticket, mutation_seq) = (99, 0);
{
let tx = ds.transaction(TransactionType::Write).await?;
tx.set_key(
&ikb.new_bg_key(generation, ticket, mutation_seq),
&Appending {
old_values: None,
new_values: None,
id: aliased.clone(),
count_cond_match: None,
},
)
.await?;
let legacy = primary_append_key_in_previous_encoding(&ikb, generation, &aliased)?;
let decoded = crate::key::schema::BuildPrimaryKey::decode_key(&legacy)
.expect("the aliased spelling decodes cleanly");
assert!(
!decoded.id.0.addresses_same_record(&aliased),
"and decodes as a different record, which is what makes it dangerous"
);
tx.set(
Key::from(legacy),
PrimaryAppendingTicket {
ticket,
mutation_seq,
}
.kv_encode_value()?,
)
.await?;
tx.commit().await?;
}
building.rebuild_primary_appendings().await?;
let tx = ds.transaction(TransactionType::Read).await?;
let ptr = tx.get_key(&ikb.new_bp_key(generation, &aliased), None).await?;
let ptr = ptr.expect("an aliased marker must be rebuilt at its own record's key");
assert_eq!((ptr.ticket, ptr.mutation_seq), (ticket, mutation_seq));
assert!(
tx.get_key(&ikb.new_bp_key(generation, &three), None).await?.is_none(),
"the record the old key decodes as has no queued mutation, so no marker"
);
let all = tx.keys(ikb.new_bp_range(generation)?, u32::MAX, 0, None).await?;
assert_eq!(all.len(), span + 2, "one marker per record with a queued mutation");
tx.cancel().await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn aliasing_markers_from_a_previous_release_both_survive_the_rebuild() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE TABLE t SCHEMALESS;
DEFINE ANALYZER simple TOKENIZERS blank;
DEFINE INDEX ia ON t FIELDS a FULLTEXT ANALYZER simple BM25;",
)
.await?;
let (ns, db, table, ia) = get_table_index(&ds, "t", "ia").await?;
let ix_id = ia.index_id;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix_id);
let building = new_building_for_index(&ds, &session, ns, db, &table, ia).await?;
let generation: BuildGeneration = 1;
let one = RecordIdKey::Array(vec![Value::Number(crate::val::Number::Int(1))].into());
let one_none =
RecordIdKey::Array(vec![Value::Number(crate::val::Number::Int(1)), Value::None].into());
assert_eq!(
*ikb.new_bp_key(generation, &one).encode_key()?,
*primary_append_key_in_previous_encoding(&ikb, generation, &one_none)?,
"the two spellings must alias, or this test guards nothing"
);
for (id, ticket) in [(&one, 7u64), (&one_none, 9u64)] {
let tx = ds.transaction(TransactionType::Write).await?;
tx.set_key(
&ikb.new_bg_key(generation, ticket, 0),
&Appending {
old_values: None,
new_values: None,
id: id.clone(),
count_cond_match: None,
},
)
.await?;
tx.set(
Key::from(primary_append_key_in_previous_encoding(&ikb, generation, id)?),
PrimaryAppendingTicket {
ticket,
mutation_seq: 0,
}
.kv_encode_value()?,
)
.await?;
tx.commit().await?;
}
building.build_generation.store(generation, Ordering::Release);
own_build(&ds, &ikb, &building, generation).await?;
building.rebuild_primary_appendings().await?;
let tx = ds.transaction(TransactionType::Read).await?;
assert_eq!(
tx.keys(ikb.new_bp_range(generation)?, u32::MAX, 0, None).await?.len(),
2,
"neither marker may be lost"
);
for (id, ticket) in [(&one, 7u64), (&one_none, 9u64)] {
let ptr = tx
.get_key(&ikb.new_bp_key(generation, id), None)
.await?
.expect("each record keeps a marker at its own key");
assert_eq!(ptr.ticket, ticket, "and it still names that record's own mutation");
}
tx.cancel().await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn the_rebuild_keeps_the_earlier_of_two_markers_for_one_record() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE TABLE t SCHEMALESS;
DEFINE ANALYZER simple TOKENIZERS blank;
DEFINE INDEX ia ON t FIELDS a FULLTEXT ANALYZER simple BM25;",
)
.await?;
let (ns, db, table, ia) = get_table_index(&ds, "t", "ia").await?;
let ix_id = ia.index_id;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix_id);
let building = new_building_for_index(&ds, &session, ns, db, &table, ia).await?;
let generation: BuildGeneration = 1;
let id = RecordIdKey::Array(vec![Value::Number(crate::val::Number::Int(5))].into());
let tx = ds.transaction(TransactionType::Write).await?;
for ticket in [5u64, 9u64] {
tx.set_key(
&ikb.new_bg_key(generation, ticket, 0),
&Appending {
old_values: None,
new_values: None,
id: id.clone(),
count_cond_match: None,
},
)
.await?;
}
tx.set_key(
&ikb.new_bp_key(generation, &id),
&PrimaryAppendingTicket {
ticket: 9,
mutation_seq: 0,
},
)
.await?;
tx.set(
Key::from(primary_append_key_in_previous_encoding(&ikb, generation, &id)?),
PrimaryAppendingTicket {
ticket: 5,
mutation_seq: 0,
}
.kv_encode_value()?,
)
.await?;
tx.commit().await?;
building.build_generation.store(generation, Ordering::Release);
own_build(&ds, &ikb, &building, generation).await?;
building.rebuild_primary_appendings().await?;
let tx = ds.transaction(TransactionType::Read).await?;
let ptr = tx
.get_key(&ikb.new_bp_key(generation, &id), None)
.await?
.expect("the record keeps a marker");
assert_eq!(ptr.ticket, 5, "the earlier queued mutation is the one a marker must name");
assert_eq!(
tx.keys(ikb.new_bp_range(generation)?, u32::MAX, 0, None).await?.len(),
1,
"and the other spelling is consumed"
);
tx.cancel().await?;
let far = RecordIdKey::Array(vec![Value::Number(crate::val::Number::Int(6))].into());
let page = surrealdb_kvs::consts::INDEXING_BATCH_SIZE as u64;
let tx = ds.transaction(TransactionType::Write).await?;
for (ticket, id) in
[(20, far.clone()), (21 + page, far.clone())].into_iter().chain((0..page).map(|i| {
(21 + i, RecordIdKey::Array(vec![Value::Number(crate::val::Number::Int(7))].into()))
})) {
tx.set_key(
&ikb.new_bg_key(generation, ticket, 0),
&Appending {
old_values: None,
new_values: None,
id,
count_cond_match: None,
},
)
.await?;
}
tx.commit().await?;
building.rebuild_primary_appendings().await?;
let tx = ds.transaction(TransactionType::Read).await?;
let ptr = tx
.get_key(&ikb.new_bp_key(generation, &far), None)
.await?
.expect("the record keeps a marker");
assert_eq!(ptr.ticket, 20, "a later page must not replace an earlier page's marker");
tx.cancel().await?;
Ok(())
}
async fn own_build(
ds: &Datastore,
ikb: &IndexKeyBase,
building: &Building,
generation: BuildGeneration,
) -> Result<()> {
set_durable_build_state(
ds,
ikb,
durable_build_state_for_phase(IndexBuildPhase::Building, generation, Some(building.owner)),
)
.await
}
async fn queue_with_marker_in_previous_encoding(
tx: &crate::kvs::Transaction,
ikb: &IndexKeyBase,
generation: BuildGeneration,
ticket: BuildTicket,
id: &RecordIdKey,
old_values: Option<Vec<Value>>,
marker: bool,
) -> Result<()> {
let appending = Appending {
old_values,
new_values: None,
id: id.clone(),
count_cond_match: None,
};
queue_in_previous_encoding(tx, ikb, generation, ticket, appending, marker).await
}
async fn queue_in_previous_encoding(
tx: &crate::kvs::Transaction,
ikb: &IndexKeyBase,
generation: BuildGeneration,
ticket: BuildTicket,
appending: Appending,
marker: bool,
) -> Result<()> {
if marker {
tx.set(
Key::from(primary_append_key_in_previous_encoding(ikb, generation, &appending.id)?),
PrimaryAppendingTicket {
ticket,
mutation_seq: 0,
}
.kv_encode_value()?,
)
.await?;
}
tx.set_key(&ikb.new_bg_key(generation, ticket, 0), &appending).await
}
fn aliasing_chain(len: usize) -> Vec<RecordIdKey> {
(0..len)
.map(|nones| {
let mut id = vec![Value::Number(crate::val::Number::Int(1))];
id.extend(std::iter::repeat_n(Value::None, nones));
RecordIdKey::Array(id.into())
})
.collect()
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn a_chain_of_aliasing_markers_is_rebuilt() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE TABLE t SCHEMALESS;
DEFINE ANALYZER simple TOKENIZERS blank;
DEFINE INDEX ia ON t FIELDS a FULLTEXT ANALYZER simple BM25;",
)
.await?;
let (ns, db, table, ia) = get_table_index(&ds, "t", "ia").await?;
let ix_id = ia.index_id;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix_id);
let building = new_building_for_index(&ds, &session, ns, db, &table, ia).await?;
let generation: BuildGeneration = 1;
let chain = aliasing_chain(5);
for link in chain.windows(2) {
assert_eq!(
*ikb.new_bp_key(generation, &link[0]).encode_key()?,
*primary_append_key_in_previous_encoding(&ikb, generation, &link[1])?,
"each link must occupy the key the one before it belongs at"
);
}
let tx = ds.transaction(TransactionType::Write).await?;
for (ticket, id) in chain.iter().enumerate() {
queue_with_marker_in_previous_encoding(
&tx,
&ikb,
generation,
ticket as BuildTicket + 1,
id,
None,
true,
)
.await?;
}
tx.commit().await?;
building.build_generation.store(generation, Ordering::Release);
own_build(&ds, &ikb, &building, generation).await?;
building.rebuild_primary_appendings().await?;
let tx = ds.transaction(TransactionType::Read).await?;
for (ticket, id) in chain.iter().enumerate() {
let ptr = tx
.get_key(&ikb.new_bp_key(generation, id), None)
.await?
.expect("every link keeps a marker at its own key");
assert_eq!(ptr.ticket, ticket as BuildTicket + 1, "naming its own mutation");
}
assert_eq!(
tx.keys(ikb.new_bp_range(generation)?, u32::MAX, 0, None).await?.len(),
chain.len(),
"and nothing else is left behind"
);
tx.cancel().await?;
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn a_record_left_without_a_marker_gets_one_from_its_queued_mutation() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE TABLE t SCHEMALESS;
DEFINE ANALYZER simple TOKENIZERS blank;
DEFINE INDEX ia ON t FIELDS a FULLTEXT ANALYZER simple BM25;",
)
.await?;
let (ns, db, table, ia) = get_table_index(&ds, "t", "ia").await?;
let ix_id = ia.index_id;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix_id);
let building = new_building_for_index(&ds, &session, ns, db, &table, ia).await?;
let generation: BuildGeneration = 1;
let [one, one_none] = <[RecordIdKey; 2]>::try_from(aliasing_chain(2)).expect("two links");
let tx = ds.transaction(TransactionType::Write).await?;
queue_with_marker_in_previous_encoding(&tx, &ikb, generation, 9, &one_none, None, true).await?;
queue_with_marker_in_previous_encoding(&tx, &ikb, generation, 11, &one, None, false).await?;
tx.commit().await?;
building.build_generation.store(generation, Ordering::Release);
own_build(&ds, &ikb, &building, generation).await?;
building.rebuild_primary_appendings().await?;
let tx = ds.transaction(TransactionType::Read).await?;
for (id, ticket) in [(&one, 11), (&one_none, 9)] {
let ptr = tx
.get_key(&ikb.new_bp_key(generation, id), None)
.await?
.expect("each record with a queued mutation has a marker");
assert_eq!(ptr.ticket, ticket, "naming its own mutation");
}
tx.cancel().await?;
Ok(())
}
async fn scan_records(
ds: &Datastore,
ns: NamespaceId,
db: DatabaseId,
table: &TableName,
) -> Result<Vec<(Vec<u8>, crate::kvs::Val)>> {
let tx = ds.transaction(TransactionType::Read).await?;
let rng = RecordPrefix {
ns,
db,
tb: Cow::Borrowed(table),
}
.range()?;
let values = catch!(tx, tx.batch_keys_vals_raw(rng, u32::MAX, None).await).result;
tx.cancel().await?;
Ok(values)
}
fn mutation(id: &RecordIdKey, old: bool, new: bool) -> Appending {
Appending {
old_values: old.then(Vec::new),
new_values: new.then(Vec::new),
id: id.clone(),
count_cond_match: None,
}
}
fn number_then_string(n: i64, s: &str) -> RecordIdKey {
RecordIdKey::Array(vec![Value::Number(crate::val::Number::Int(n)), Value::from(s)].into())
}
const COUNT_GENERATION: BuildGeneration = 1;
struct CountBuild {
ds: Arc<Datastore>,
session: Session,
ikb: IndexKeyBase,
building: Building,
}
impl CountBuild {
async fn new(sql: &str) -> Result<Self> {
let (ds, session) = new_index_test_ds().await?;
execute_all(&ds, &session, sql).await?;
let (ns, db, table, ic) = get_table_index(&ds, "t", "ic").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ic.index_id);
let building = new_building_for_index(&ds, &session, ns, db, &table, ic).await?;
building.build_generation.store(COUNT_GENERATION, Ordering::Release);
Ok(Self {
ds,
session,
ikb,
building,
})
}
fn own_key(&self, id: &RecordIdKey) -> Result<Vec<u8>> {
Ok(self.ikb.new_bp_key(COUNT_GENERATION, id).encode_key()?.to_vec())
}
fn older_key(&self, id: &RecordIdKey) -> Result<Vec<u8>> {
primary_append_key_in_previous_encoding(&self.ikb, COUNT_GENERATION, id)
}
async fn queue_older(
&self,
tx: &crate::kvs::Transaction,
ticket: BuildTicket,
appending: Appending,
) -> Result<()> {
queue_in_previous_encoding(tx, &self.ikb, COUNT_GENERATION, ticket, appending, true).await
}
async fn queue_own(
&self,
tx: &crate::kvs::Transaction,
ticket: BuildTicket,
appending: Appending,
) -> Result<()> {
let ptr = PrimaryAppendingTicket {
ticket,
mutation_seq: 0,
};
tx.set_key(&self.ikb.new_bp_key(COUNT_GENERATION, &appending.id), &ptr).await?;
self.queue_unmarked(tx, ticket, appending).await
}
async fn queue_unmarked(
&self,
tx: &crate::kvs::Transaction,
ticket: BuildTicket,
appending: Appending,
) -> Result<()> {
tx.set_key(&self.ikb.new_bg_key(COUNT_GENERATION, ticket, 0), &appending).await
}
async fn records(&self) -> Result<Vec<(Vec<u8>, crate::kvs::Val)>> {
scan_records(&self.ds, self.ikb.ns(), self.ikb.db(), self.ikb.table()).await
}
async fn scan_in_batches(
&self,
batches: &[&[(Vec<u8>, crate::kvs::Val)]],
) -> Result<(Vec<usize>, i64)> {
let mut scan = CountScan::start(&self.building).await?;
let mut indexed = Vec::new();
for values in batches {
indexed.push(scan.batch(values).await?);
}
indexed.push(scan.tail().await?);
let baseline = scan.baseline();
scan.cancel().await?;
Ok((indexed, baseline))
}
}
struct CountScan<'a> {
building: &'a Building,
ctx: crate::ctx::FrozenContext,
progress: Option<super::replay::CountPrimaryProgress>,
indexed: usize,
}
impl<'a> CountScan<'a> {
async fn start(building: &'a Building) -> Result<Self> {
Ok(Self {
building,
ctx: building.new_write_tx_ctx().await?,
progress: Some(super::replay::CountPrimaryProgress::default()),
indexed: 0,
})
}
async fn batch(&mut self, values: &[(Vec<u8>, crate::kvs::Val)]) -> Result<usize> {
let tx = self.ctx.tx();
let n = self
.building
.index_initial_batch(
&self.ctx,
&tx,
values,
self.indexed,
&mut false,
&mut self.progress,
)
.await?;
self.indexed += n;
Ok(n)
}
async fn tail(&mut self) -> Result<usize> {
let tx = self.ctx.tx();
let n = self
.building
.index_remaining_count_primary_appendings(
&self.ctx,
&tx,
&mut self.progress,
self.indexed,
)
.await?;
self.indexed += n;
Ok(n)
}
fn baseline(&self) -> i64 {
let building = self.building;
self.ctx.tx().pending_count_delta(
building.ix_key.ns,
building.ix_key.db,
building.ikb.table(),
building.ix.index_id,
)
}
async fn cancel(self) -> Result<()> {
self.ctx.tx().cancel().await
}
}
#[test]
fn a_markers_older_spelling_is_the_key_an_older_node_writes() -> Result<()> {
let ikb = IndexKeyBase::new(NamespaceId(1), DatabaseId(2), TableName::from("t"), IndexId(3));
let older = |id: &RecordIdKey| -> Result<Option<Vec<u8>>> {
let key = ikb.new_bp_key_in_previous_spelling(COUNT_GENERATION, id)?;
Ok(key.map(|key| key.as_bytes().to_vec()))
};
for id in [&aliasing_chain(1)[0], &aliasing_chain(2)[1], &number_then_string(42, "intro")] {
let written = primary_append_key_in_previous_encoding(&ikb, COUNT_GENERATION, id)?;
assert_eq!(older(id)?, Some(written), "{id:?}");
}
assert_eq!(
older(&number_then_string(42, "intro"))?,
Some(
b"/*\x00\x00\x00\x01*\x00\x00\x00\x02*t\x00!bp\x00\x00\x00\x03\x00\x00\x00\x00\x00\x00\x00\x01\
\x05\x05\xa0\x19\x1aS\x0f\x01\x00\x00\x06intro\x00\x00"
.to_vec()
)
);
let plain = [
RecordIdKey::String("plain".into()),
RecordIdKey::Number(7),
RecordIdKey::Array(vec![Value::from("a"), Value::None].into()),
];
for id in &plain {
assert_eq!(older(id)?, None, "{id:?}");
}
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn a_count_build_merges_older_markers_by_their_queued_mutation() -> Result<()> {
let b = CountBuild::new(
"DEFINE TABLE t SCHEMALESS;
DEFINE INDEX ic ON t COUNT;
CREATE t:[1] SET a = 1;",
)
.await?;
let (one_none, intro) =
(aliasing_chain(2).pop().expect("two links"), number_then_string(42, "intro"));
assert_eq!(b.own_key(&aliasing_chain(1)[0])?, b.older_key(&one_none)?);
let tx = b.ds.transaction(TransactionType::Write).await?;
b.queue_older(&tx, 9, mutation(&one_none, true, false)).await?;
b.queue_older(&tx, 10, mutation(&intro, true, false)).await?;
tx.commit().await?;
let values = b.records().await?;
assert_eq!(values.len(), 1, "only `t:[1]` is live");
let mut scan = CountScan::start(&b.building).await?;
assert_eq!(
scan.batch(&values).await?,
1,
"the live record only: `[1, NONE]`'s marker reads as `[1]` but waits for its own span"
);
assert_eq!(
scan.tail().await?,
2,
"the deleted `[1, NONE]` in its own span, and the deleted `[42, 'intro']`, whose key \
does not decode"
);
assert_eq!(scan.baseline(), 3);
scan.cancel().await
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn a_count_build_takes_an_early_marker_up_in_its_records_span() -> Result<()> {
let b = CountBuild::new(
"DEFINE TABLE t SCHEMALESS;
DEFINE INDEX ic ON t COUNT;
CREATE t:[1] SET a = 1;
CREATE t:[1, NONE] SET a = 1;",
)
.await?;
let one_none = aliasing_chain(2).pop().expect("two links");
let tx = b.ds.transaction(TransactionType::Write).await?;
b.queue_older(&tx, 9, mutation(&one_none, true, true)).await?;
tx.commit().await?;
let values = b.records().await?;
assert_eq!(values.len(), 2, "`[1]` and `[1, NONE]` are both live");
let (indexed, baseline) = b.scan_in_batches(&[&values[..1], &values[1..]]).await?;
assert_eq!(indexed, vec![1, 1, 0], "each live record once, and the marker not at all");
assert_eq!(baseline, 2);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn a_count_build_counts_a_late_marker_only_for_a_record_that_is_gone() -> Result<()> {
let b = CountBuild::new(
"DEFINE TABLE t SCHEMALESS;
DEFINE INDEX ic ON t COUNT;
CREATE t:[42, 'intro'] SET a = 1;
CREATE t:[42, 'k'] SET a = 1;
CREATE t:[43, 'x'] SET a = 1;",
)
.await?;
let (live, gone) = (number_then_string(42, "intro"), number_then_string(42, "b"));
for marked in [&live, &gone] {
let old = b.older_key(marked)?;
assert!(old > b.own_key(&number_then_string(42, "k"))?);
assert!(old < b.own_key(&number_then_string(43, "x"))?);
}
let tx = b.ds.transaction(TransactionType::Write).await?;
b.queue_older(&tx, 9, mutation(&live, true, true)).await?;
b.queue_older(&tx, 10, mutation(&gone, true, false)).await?;
tx.commit().await?;
let values = b.records().await?;
assert_eq!(values.len(), 3);
let (indexed, baseline) = b.scan_in_batches(&[&values[..2], &values[2..]]).await?;
assert_eq!(
indexed,
vec![2, 2, 0],
"the two live `[42, …]`, then `[43, 'x']` and the gone `[42, 'b']`, and nothing twice"
);
assert_eq!(baseline, 4);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn a_count_build_trusts_an_own_key_marker_only_for_its_own_record() -> Result<()> {
let b = CountBuild::new("DEFINE TABLE t SCHEMALESS; DEFINE INDEX ic ON t COUNT;").await?;
let [one, one_none] = <[RecordIdKey; 2]>::try_from(aliasing_chain(2)).expect("two links");
let both = number_then_string(42, "b");
let tx = b.ds.transaction(TransactionType::Write).await?;
b.queue_older(&tx, 1, mutation(&one, true, false)).await?;
b.queue_older(&tx, 2, mutation(&one_none, true, false)).await?;
b.queue_own(&tx, 3, mutation(&both, true, false)).await?;
b.queue_older(&tx, 4, mutation(&both, true, false)).await?;
tx.commit().await?;
let (indexed, baseline) = b.scan_in_batches(&[]).await?;
assert_eq!(indexed, vec![3], "`[1]`, `[1, NONE]` and `[42, 'b']`, each once");
assert_eq!(baseline, 3);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn a_count_where_build_baselines_a_changed_record_from_its_queued_state() -> Result<()> {
let b = CountBuild::new(
"DEFINE TABLE t SCHEMALESS;
DEFINE INDEX ic ON t COUNT WHERE a = 1;
CREATE t:[1] SET a = 1;
CREATE t:[1, NONE] SET a = 2;
CREATE t:[42, 'intro'] SET a = 2;
CREATE t:[42, 'k'] SET a = 1;
CREATE t:[43, 'x'] SET a = 1;
CREATE t:[45, 'z'] SET a = 1;",
)
.await?;
let one_none = aliasing_chain(2).pop().expect("two links");
let intro = number_then_string(42, "intro");
let (late, in_span) = (number_then_string(42, "e"), number_then_string(44, "d"));
let stopped_matching = |id: &RecordIdKey| Appending {
count_cond_match: Some((true, false)),
..mutation(id, true, true)
};
let deleted = |id: &RecordIdKey| Appending {
count_cond_match: Some((false, false)),
..mutation(id, true, false)
};
let tx = b.ds.transaction(TransactionType::Write).await?;
for (ticket, changed) in [(1, &one_none), (2, &intro), (3, &late), (5, &in_span)] {
b.queue_older(&tx, ticket, stopped_matching(changed)).await?;
}
b.queue_own(&tx, 4, deleted(&late)).await?;
b.queue_own(&tx, 6, deleted(&in_span)).await?;
tx.commit().await?;
let values = b.records().await?;
assert_eq!(values.len(), 6, "every record but `[42, 'e']` and `[44, 'd']` is live");
let (indexed, baseline) =
b.scan_in_batches(&[&values[..1], &values[1..4], &values[4..]]).await?;
assert_eq!(baseline, 8, "each record, as it stood before its queued mutations");
assert_eq!(
indexed,
vec![1, 4, 3, 0],
"`[1]`; `[1, NONE]`, `[42, 'intro']`, `[42, 'k']` and the gone `[42, 'e']`; \
`[43, 'x']`, `[45, 'z']` and the gone `[44, 'd']`"
);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn a_count_build_baselines_from_the_first_of_two_markers() -> Result<()> {
let b = CountBuild::new(
"DEFINE TABLE t SCHEMALESS;
DEFINE INDEX ic ON t COUNT;
CREATE t:plain SET a = 1;
CREATE t:[42, 'b'] SET a = 1;
CREATE t:[42, 'c'] SET a = 1;
CREATE t:[42, 'k'] SET a = 1;",
)
.await?;
let (recreated, created) = (number_then_string(42, "b"), number_then_string(42, "c"));
let tx = b.ds.transaction(TransactionType::Write).await?;
b.queue_older(&tx, 1, mutation(&recreated, true, false)).await?;
b.queue_own(&tx, 2, mutation(&recreated, false, true)).await?;
b.queue_own(&tx, 3, mutation(&created, false, true)).await?;
b.queue_older(&tx, 4, mutation(&created, true, true)).await?;
tx.commit().await?;
let values = b.records().await?;
assert_eq!(values.len(), 4);
let (indexed, baseline) = b.scan_in_batches(&[&values]).await?;
assert_eq!(baseline, 3, "all but `[42, 'c']`, which did not exist yet");
assert_eq!(indexed, vec![4, 0], "the live records once, and the older markers not at all");
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn a_count_build_baselines_a_record_an_older_node_created_as_absent() -> Result<()> {
let b = CountBuild::new(
"DEFINE TABLE t SCHEMALESS;
DEFINE INDEX ic ON t COUNT;
CREATE t:[42, 'k'] SET a = 1;
CREATE t:[42, 'n'] SET a = 1;",
)
.await?;
let tx = b.ds.transaction(TransactionType::Write).await?;
b.queue_older(&tx, 1, mutation(&number_then_string(42, "n"), false, true)).await?;
tx.commit().await?;
let values = b.records().await?;
assert_eq!(values.len(), 2);
let (indexed, baseline) = b.scan_in_batches(&[&values]).await?;
assert_eq!(baseline, 1, "`[42, 'k']` only");
assert_eq!(indexed, vec![2, 0]);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn a_count_build_skips_a_late_marker_its_records_span_settled() -> Result<()> {
let b = CountBuild::new(
"DEFINE TABLE t SCHEMALESS;
DEFINE INDEX ic ON t COUNT;
CREATE t:[42, 'a'] SET a = 1;
CREATE t:[42, 'c'] SET a = 1;
CREATE t:[42, 'k'] SET a = 1;",
)
.await?;
let (taken, updated) = (number_then_string(42, "a"), number_then_string(42, "c"));
let tx = b.ds.transaction(TransactionType::Write).await?;
b.queue_older(&tx, 1, mutation(&taken, true, true)).await?;
tx.commit().await?;
let values = b.records().await?;
assert_eq!(values.len(), 3);
let mut scan = CountScan::start(&b.building).await?;
assert_eq!(scan.batch(&values).await?, 3);
let tx = b.ds.transaction(TransactionType::Write).await?;
b.queue_unmarked(&tx, 2, mutation(&taken, true, false)).await?;
b.queue_own(&tx, 3, mutation(&updated, true, true)).await?;
b.queue_older(&tx, 4, mutation(&updated, true, false)).await?;
tx.commit().await?;
execute_all(&b.ds, &b.session, "DELETE t:[42, 'a']; DELETE t:[42, 'c'];").await?;
assert_eq!(scan.tail().await?, 0, "neither older marker counts where it sorts");
assert_eq!(scan.baseline(), 3, "each record once, as its span had it");
scan.cancel().await
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn a_count_build_counts_a_late_marker_whose_own_marker_came_after() -> Result<()> {
let b = CountBuild::new(
"DEFINE TABLE t SCHEMALESS;
DEFINE INDEX ic ON t COUNT;
CREATE t:[42, 'k'] SET a = 1;",
)
.await?;
let gone = number_then_string(42, "g");
let tx = b.ds.transaction(TransactionType::Write).await?;
b.queue_older(&tx, 1, mutation(&gone, true, false)).await?;
tx.commit().await?;
let values = b.records().await?;
assert_eq!(values.len(), 1, "only `[42, 'k']` is live");
let mut scan = CountScan::start(&b.building).await?;
assert_eq!(scan.batch(&values).await?, 1);
let tx = b.ds.transaction(TransactionType::Write).await?;
b.queue_own(&tx, 2, mutation(&gone, false, true)).await?;
b.queue_unmarked(&tx, 3, mutation(&gone, true, false)).await?;
tx.commit().await?;
assert_eq!(scan.tail().await?, 1, "`[42, 'g']`, as it stood before the older node's delete");
assert_eq!(scan.baseline(), 2);
scan.cancel().await
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn a_count_build_takes_an_early_marker_up_once_its_record_has_its_own() -> Result<()> {
let b = CountBuild::new(
"DEFINE TABLE t SCHEMALESS;
DEFINE INDEX ic ON t COUNT;
CREATE t:[1] SET a = 1;",
)
.await?;
let one_none = aliasing_chain(2).pop().expect("two links");
let tx = b.ds.transaction(TransactionType::Write).await?;
b.queue_older(&tx, 1, mutation(&one_none, true, false)).await?;
tx.commit().await?;
let values = b.records().await?;
assert_eq!(values.len(), 1, "only `[1]` is live");
let mut scan = CountScan::start(&b.building).await?;
assert_eq!(scan.batch(&values).await?, 1, "`[1]`, with `[1, NONE]`'s marker set aside");
let tx = b.ds.transaction(TransactionType::Write).await?;
b.queue_own(&tx, 2, mutation(&one_none, false, true)).await?;
b.queue_unmarked(&tx, 3, mutation(&one_none, true, false)).await?;
tx.commit().await?;
assert_eq!(scan.tail().await?, 1, "`[1, NONE]` once");
assert_eq!(scan.baseline(), 2);
scan.cancel().await
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn a_count_build_skips_a_late_marker_written_after_its_records_span() -> Result<()> {
let b = CountBuild::new(
"DEFINE TABLE t SCHEMALESS;
DEFINE INDEX ic ON t COUNT;
CREATE t:[42, 'a'] SET a = 1;
CREATE t:[42, 'k'] SET a = 1;",
)
.await?;
let values = b.records().await?;
let mut scan = CountScan::start(&b.building).await?;
assert_eq!(scan.batch(&values).await?, 2);
let tx = b.ds.transaction(TransactionType::Write).await?;
b.queue_older(&tx, 1, mutation(&number_then_string(42, "a"), true, true)).await?;
tx.commit().await?;
assert_eq!(scan.tail().await?, 0, "`[42, 'a']` was there when its span was scanned");
assert_eq!(scan.baseline(), 2);
scan.cancel().await
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn a_set_aside_marker_defers_only_to_its_own_records_marker() -> Result<()> {
let b = CountBuild::new(
"DEFINE TABLE t SCHEMALESS;
DEFINE INDEX ic ON t COUNT;
CREATE t:[1] SET a = 1;",
)
.await?;
let [_, one_none, one_none_none] =
<[RecordIdKey; 3]>::try_from(aliasing_chain(3)).expect("three links");
assert_eq!(b.older_key(&one_none_none)?, b.own_key(&one_none)?);
let tx = b.ds.transaction(TransactionType::Write).await?;
b.queue_older(&tx, 1, mutation(&one_none, true, false)).await?;
b.queue_older(&tx, 2, mutation(&one_none_none, true, false)).await?;
tx.commit().await?;
let values = b.records().await?;
assert_eq!(values.len(), 1, "only `[1]` is live");
let (indexed, baseline) = b.scan_in_batches(&[&values]).await?;
assert_eq!(indexed, vec![1, 2], "`[1]`, then `[1, NONE]` and `[1, NONE, NONE]`");
assert_eq!(baseline, 3);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn a_late_marker_defers_only_to_its_own_records_marker() -> Result<()> {
let b = CountBuild::new(
"DEFINE TABLE t SCHEMALESS;
DEFINE INDEX ic ON t COUNT;
CREATE t:[42, 'k'] SET a = 1;",
)
.await?;
let late = number_then_string(42, "a");
let alias = RecordIdKey::Array(
vec![Value::Number(crate::val::Number::Int(42)), Value::None, Value::from("a")].into(),
);
assert_eq!(b.older_key(&alias)?, b.own_key(&late)?);
let tx = b.ds.transaction(TransactionType::Write).await?;
b.queue_older(&tx, 1, mutation(&alias, true, false)).await?;
b.queue_older(&tx, 2, mutation(&late, true, false)).await?;
tx.commit().await?;
let values = b.records().await?;
assert_eq!(values.len(), 1, "only `[42, 'k']` is live");
let (indexed, baseline) = b.scan_in_batches(&[&values]).await?;
assert_eq!(indexed, vec![2, 1], "`[42, 'k']` and `[42, NONE, 'a']`, then `[42, 'a']`");
assert_eq!(baseline, 3);
Ok(())
}
async fn strand_count_build(
ds: &Datastore,
ikb: &IndexKeyBase,
cursor: Option<RecordIdKey>,
) -> Result<()> {
let expired = Utc::now() - chrono::Duration::seconds(BUILD_OWNER_LEASE_SECS + 5);
let tx = ds.transaction(TransactionType::Write).await?;
tx.del_prefix_key(&IdxRoot {
ns: ikb.ns(),
db: ikb.db(),
tb: Cow::Borrowed(ikb.table()),
ix: ikb.index(),
})
.await?;
tx.set_key(
&ikb.new_bs_key(),
&IndexBuildState {
generation: 2,
phase: IndexBuildPhase::Building,
owner: Some(Uuid::new_v4()),
next_ticket: 0,
initial_complete: false,
updated_at: expired,
owner_heartbeat_at: Some(expired),
error: None,
report_status: Some(IndexBuildReportStatus::Indexing),
initial: cursor.as_ref().map(|_| 1),
updated: None,
pending: None,
initial_cursor: cursor,
},
)
.await?;
tx.commit().await
}
async fn take_over_build(
ds: &Datastore,
session: &Session,
(ns, db, table, ix): (NamespaceId, DatabaseId, &TableName, &Arc<IndexDefinition>),
) -> Result<()> {
let build = new_building_for_index(ds, session, ns, db, table, Arc::clone(ix)).await?;
let acquired = build
.acquire_build_state()
.await?
.expect("expired build state should be available for takeover");
build.run_acquired(acquired).await
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn a_count_takeover_rescans_rather_than_leave_an_older_marker_unmet() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE TABLE t SCHEMALESS;
CREATE t:[42, 'k'] SET a = 1;
CREATE t:[43, 'x'] SET a = 1;
DEFINE INDEX ic ON t COUNT;",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "t", "ic").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
strand_count_build(&ds, &ikb, Some(number_then_string(42, "k"))).await?;
let tx = ds.transaction(TransactionType::Write).await?;
let delete = mutation(&number_then_string(42, "intro"), true, false);
queue_in_previous_encoding(&tx, &ikb, 2, 1, delete, true).await?;
tx.commit().await?;
take_over_build(&ds, &session, (ns, db, &table, &ix)).await?;
let state = durable_build_state(&ds, &ikb).await?;
assert_eq!(state.phase, IndexBuildPhase::Online);
assert_eq!(state.initial, Some(3), "both live records and `[42, 'intro']`, rescanned");
assert_eq!(count_query_value(&ds, &session, "SELECT count() FROM t GROUP ALL").await?, 2);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn a_count_takeover_rescans_a_checkpoint_that_would_lose_a_record() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE TABLE t SCHEMALESS;
CREATE t:[1, 'a'] SET a = 1;
CREATE t:[1f] SET a = 1;
CREATE t:[2] SET a = 1;
DEFINE INDEX ic ON t COUNT;",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "t", "ic").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
let float = RecordIdKey::Array(vec![Value::Number(crate::val::Number::Float(1.0))].into());
let gone = RecordIdKey::Array(
vec![Value::Number(crate::val::Number::Int(1)), Value::from("z")].into(),
);
assert!(RecordIdentity(gone.clone()) < RecordIdentity(float.clone()));
assert!(
primary_append_key_in_previous_encoding(&ikb, 2, &gone)?
> ikb.new_bp_key(2, &float).encode_key()?.to_vec()
);
strand_count_build(&ds, &ikb, Some(float)).await?;
let tx = ds.transaction(TransactionType::Write).await?;
queue_in_previous_encoding(&tx, &ikb, 2, 1, mutation(&gone, true, false), true).await?;
tx.commit().await?;
take_over_build(&ds, &session, (ns, db, &table, &ix)).await?;
let state = durable_build_state(&ds, &ikb).await?;
assert_eq!(state.phase, IndexBuildPhase::Online);
assert_eq!(state.initial, Some(4), "the three live records and `[1, 'z']`, rescanned");
assert_eq!(count_query_value(&ds, &session, "SELECT count() FROM t GROUP ALL").await?, 3);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn a_retried_count_batch_merges_its_span_again() -> Result<()> {
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE TABLE t SCHEMALESS;
CREATE t:b SET a = 1;
CREATE t:k SET a = 1;
DEFINE INDEX ic ON t COUNT;",
)
.await?;
let (ns, db, table, ix) = get_table_index(&ds, "t", "ic").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix.index_id);
strand_count_build(&ds, &ikb, None).await?;
execute_all_retrying_conflicts(&ds, &session, "DELETE t:b RETURN NONE").await?;
let site = RetryableConflictSite::ConcurrentIndexInitialBatch;
let _guard = inject_retryable_conflict(site, ds.id());
take_over_build(&ds, &session, (ns, db, &table, &ix)).await?;
assert_eq!(retryable_conflict_count(site, ds.id()), 0, "the batch was retried");
assert_eq!(count_query_value(&ds, &session, "SELECT count() FROM t GROUP ALL").await?, 1);
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn a_build_of_any_kind_baselines_from_an_older_marker() -> Result<()> {
use surrealdb_types::Value as PublicValue;
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE TABLE t SCHEMALESS;
DEFINE ANALYZER simple TOKENIZERS blank;
DEFINE INDEX ia ON t FIELDS a FULLTEXT ANALYZER simple BM25;
CREATE t:[42, 'intro'] SET a = 'new';",
)
.await?;
let (ns, db, table, ia) = get_table_index(&ds, "t", "ia").await?;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ia.index_id);
let building = new_building_for_index(&ds, &session, ns, db, &table, ia).await?;
let generation: BuildGeneration = 1;
let tx = ds.transaction(TransactionType::Write).await?;
let update = Appending {
old_values: Some(vec![Value::from("old")]),
new_values: Some(vec![Value::from("new")]),
..mutation(&number_then_string(42, "intro"), true, true)
};
queue_in_previous_encoding(&tx, &ikb, generation, 9, update, true).await?;
tx.commit().await?;
building.build_generation.store(generation, Ordering::Release);
let values = scan_records(&ds, ns, db, &table).await?;
let ctx = building.new_write_tx_ctx().await?;
let tx = ctx.tx();
building.index_initial_batch(&ctx, &tx, &values, 0, &mut false, &mut None).await?;
tx.commit().await?;
let res = ds.execute("SELECT VALUE id FROM t WHERE a @@ 'old'", &session, None).await?;
let PublicValue::Array(rows) = res.into_iter().next().expect("one result").result? else {
anyhow::bail!("unexpected result shape");
};
assert_eq!(rows.len(), 1, "`[42, 'intro']` is baselined from its queued old state");
Ok(())
}
#[cfg(feature = "kv-mem")]
#[tokio::test(flavor = "multi_thread")]
#[test_log::test]
async fn a_marker_naming_another_record_is_not_the_scanned_records_baseline() -> Result<()> {
use surrealdb_types::Value as PublicValue;
let (ds, session) = new_index_test_ds().await?;
execute_all(
&ds,
&session,
"DEFINE TABLE t SCHEMALESS;
DEFINE ANALYZER simple TOKENIZERS blank;
DEFINE INDEX ia ON t FIELDS a FULLTEXT ANALYZER simple BM25;
CREATE t:[1] SET a = 'live';",
)
.await?;
let (ns, db, table, ia) = get_table_index(&ds, "t", "ia").await?;
let ix_id = ia.index_id;
let ikb = IndexKeyBase::new(ns, db, table.clone(), ix_id);
let building = new_building_for_index(&ds, &session, ns, db, &table, ia).await?;
let generation: BuildGeneration = 1;
let one_none = aliasing_chain(2).pop().expect("two links");
let tx = ds.transaction(TransactionType::Write).await?;
queue_with_marker_in_previous_encoding(
&tx,
&ikb,
generation,
9,
&one_none,
Some(vec![Value::from("queued")]),
true,
)
.await?;
tx.commit().await?;
building.build_generation.store(generation, Ordering::Release);
let values = scan_records(&ds, ns, db, &table).await?;
let ctx = building.new_write_tx_ctx().await?;
let tx = ctx.tx();
building.index_initial_batch(&ctx, &tx, &values, 0, &mut false, &mut None).await?;
tx.commit().await?;
let matches = |sql: &'static str| {
let ds = &ds;
let session = &session;
async move {
match ds.execute(sql, session, None).await?.remove(0).result? {
PublicValue::Array(rows) => Ok(rows.len()),
other => anyhow::bail!("unexpected result: {other:?}"),
}
}
};
assert_eq!(
matches("SELECT VALUE id FROM t WHERE a @@ 'queued'").await?,
0,
"`t:[1]` must not be indexed with `[1, NONE]`'s queued state"
);
assert_eq!(matches("SELECT VALUE id FROM t WHERE a @@ 'live'").await?, 1);
Ok(())
}