fix(raft): multi-stmt write gate, rich apply, delete kv convention
CI / test (push) Has been cancelled
CI / verify (push) Has been cancelled
Clients CI / build-server (push) Has been cancelled
Clients CI / test-python (push) Has been cancelled
Clients CI / test-javascript (push) Has been cancelled
Clients CI / test-nim (push) Has been cancelled
Clients CI / test-rust (push) Has been cancelled
CI / test (push) Has been cancelled
CI / verify (push) Has been cancelled
Clients CI / build-server (push) Has been cancelled
Clients CI / test-python (push) Has been cancelled
Clients CI / test-javascript (push) Has been cancelled
Clients CI / test-nim (push) Has been cancelled
Clients CI / test-rust (push) Has been cancelled
- Reject follower writes if any statement in the batch is DML/COMMIT (not only stmts[0]). - COMMIT always emits empty-valued kvPairs for isDelete entries. - applyCommand updates LSM plus secondary B-tree/FTS/HNSW indexes (applyReplicatedPut/Delete) so follower index scans see replicated rows. - Tests: not-leader append, commit timeout, index apply unit, E2E index-backed SELECT on follower.
This commit is contained in:
@@ -253,8 +253,14 @@ proc executeQuery(db: LSMTree, ctx: ExecutionContext, query: string, params: seq
|
|||||||
return (true, QueryResult(), "")
|
return (true, QueryResult(), "")
|
||||||
|
|
||||||
# C3b: writes go through the Raft log — only the leader may accept them.
|
# C3b: writes go through the Raft log — only the leader may accept them.
|
||||||
if raftNode != nil and isWrite(astNode.stmts[0]):
|
# Inspect every statement so "SELECT 1; INSERT ..." cannot bypass the gate.
|
||||||
if raftNode.state != rsLeader:
|
if raftNode != nil:
|
||||||
|
var hasWrite = false
|
||||||
|
for stmt in astNode.stmts:
|
||||||
|
if isWrite(stmt):
|
||||||
|
hasWrite = true
|
||||||
|
break
|
||||||
|
if hasWrite and raftNode.state != rsLeader:
|
||||||
let who = if raftNode.leaderId.len > 0: raftNode.leaderId else: "none elected"
|
let who = if raftNode.leaderId.len > 0: raftNode.leaderId else: "none elected"
|
||||||
return (false, QueryResult(), "not leader; leader is '" & who & "'")
|
return (false, QueryResult(), "not leader; leader is '" & who & "'")
|
||||||
|
|
||||||
|
|||||||
@@ -340,3 +340,100 @@ proc execUpdateRow*(ctx: ExecutionContext, table: string, key: string, sets: Tab
|
|||||||
meta["key"] = fullKey
|
meta["key"] = fullKey
|
||||||
vengine.insert(vecIdx, docId, vec, meta)
|
vengine.insert(vecIdx, docId, vec, meta)
|
||||||
return 1
|
return 1
|
||||||
|
|
||||||
|
# ----------------------------------------------------------------------
|
||||||
|
# Raft / replication apply — keep secondary engines in sync with LSM
|
||||||
|
# ----------------------------------------------------------------------
|
||||||
|
|
||||||
|
proc docIdFromLsmKey(fullKey: string): uint64 =
|
||||||
|
result = 0
|
||||||
|
for ch in fullKey:
|
||||||
|
result = result * 31 + uint64(ord(ch))
|
||||||
|
|
||||||
|
proc injectPkFromKey(row: var Row, keyRest: string) =
|
||||||
|
## Decode `id=1` or `a=1:b=2` key tails into row columns (same as scan).
|
||||||
|
for part in keyRest.split(':'):
|
||||||
|
let eqPos = part.find('=')
|
||||||
|
if eqPos > 0:
|
||||||
|
row[part[0..<eqPos]] = part[eqPos+1..^1]
|
||||||
|
|
||||||
|
proc removeIndexesForRow(ctx: ExecutionContext, table: string, fullKey: string,
|
||||||
|
valStr: string) =
|
||||||
|
var oldRow = parseRowDataToValueRow(valStr)
|
||||||
|
let keyRest = if '.' in fullKey: fullKey[fullKey.find('.')+1..^1] else: ""
|
||||||
|
injectPkFromKey(oldRow, keyRest)
|
||||||
|
for colName in ctx.btrees.keys.toSeq():
|
||||||
|
if not colName.startsWith(table & "."): continue
|
||||||
|
let colsPart = colName[table.len + 1..^1]
|
||||||
|
let idxCols = colsPart.split(".")
|
||||||
|
var oldVals: seq[string] = @[]
|
||||||
|
for c in idxCols:
|
||||||
|
if c in oldRow: oldVals.add(valueToString(oldRow[c]))
|
||||||
|
else: oldVals.add("\\N")
|
||||||
|
let oldIdxVal = oldVals.join("|")
|
||||||
|
if oldIdxVal.len > 0 and not isNull(oldIdxVal):
|
||||||
|
ctx.btrees[colName].remove(oldIdxVal,
|
||||||
|
IndexEntry(lsmKey: fullKey, rowValue: valStr))
|
||||||
|
let docId = docIdFromLsmKey(fullKey)
|
||||||
|
for ftsKey, ftsIdx in ctx.ftsIndexes:
|
||||||
|
if ftsKey.startsWith(table & "."):
|
||||||
|
ftsIdx.removeDocument(docId)
|
||||||
|
|
||||||
|
proc insertIndexesForRow(ctx: ExecutionContext, table: string, fullKey: string,
|
||||||
|
valStr: string) =
|
||||||
|
var newRow = parseRowDataToValueRow(valStr)
|
||||||
|
let keyRest = if '.' in fullKey: fullKey[fullKey.find('.')+1..^1] else: ""
|
||||||
|
injectPkFromKey(newRow, keyRest)
|
||||||
|
for colName in ctx.btrees.keys.toSeq():
|
||||||
|
if not colName.startsWith(table & "."): continue
|
||||||
|
let colsPart = colName[table.len + 1..^1]
|
||||||
|
let idxCols = colsPart.split(".")
|
||||||
|
var colVals: seq[string] = @[]
|
||||||
|
for c in idxCols:
|
||||||
|
if c in newRow: colVals.add(valueToString(newRow[c]))
|
||||||
|
else: colVals.add("\\N")
|
||||||
|
let idxVal = colVals.join("|")
|
||||||
|
if idxVal.len > 0 and not isNull(idxVal):
|
||||||
|
ctx.btrees[colName].insert(idxVal,
|
||||||
|
IndexEntry(lsmKey: fullKey, rowValue: valStr))
|
||||||
|
let docId = docIdFromLsmKey(fullKey)
|
||||||
|
for ftsKey, ftsIdx in ctx.ftsIndexes:
|
||||||
|
if not ftsKey.startsWith(table & "."): continue
|
||||||
|
let colName = ftsKey[table.len + 1..^1]
|
||||||
|
if colName in newRow:
|
||||||
|
let text = valueToString(newRow[colName])
|
||||||
|
if text.len > 0:
|
||||||
|
ftsIdx.addDocument(docId, text)
|
||||||
|
for vecKey, vecIdx in ctx.vectorIndexes:
|
||||||
|
if not vecKey.startsWith(table & "."): continue
|
||||||
|
let colName = vecKey[table.len + 1..^1]
|
||||||
|
if colName notin newRow: continue
|
||||||
|
let vecStr = valueToString(newRow[colName])
|
||||||
|
let vec = parseVectorString(vecStr)
|
||||||
|
if vec.len > 0:
|
||||||
|
var meta = initTable[string, string]()
|
||||||
|
meta["key"] = fullKey
|
||||||
|
for col, val in newRow:
|
||||||
|
meta[col] = valueToString(val)
|
||||||
|
vengine.insert(vecIdx, docId, vec, meta)
|
||||||
|
|
||||||
|
proc applyReplicatedPut*(ctx: ExecutionContext, fullKey: string, value: seq[byte]) =
|
||||||
|
## Apply a raft/replication put: LSM write + secondary B-tree/FTS/HNSW.
|
||||||
|
## Idempotent on the leader (local DML already applied the same engines).
|
||||||
|
let dot = fullKey.find('.')
|
||||||
|
let table = if dot > 0: fullKey[0..<dot] else: ""
|
||||||
|
let (found, existing) = ctx.db.get(fullKey)
|
||||||
|
if found and table.len > 0:
|
||||||
|
removeIndexesForRow(ctx, table, fullKey, cast[string](existing))
|
||||||
|
ctx.db.put(fullKey, value)
|
||||||
|
if table.len > 0 and value.len > 0:
|
||||||
|
insertIndexesForRow(ctx, table, fullKey, cast[string](value))
|
||||||
|
|
||||||
|
proc applyReplicatedDelete*(ctx: ExecutionContext, fullKey: string) =
|
||||||
|
## Apply a raft/replication delete: LSM delete + drop secondary index entries.
|
||||||
|
let dot = fullKey.find('.')
|
||||||
|
let table = if dot > 0: fullKey[0..<dot] else: ""
|
||||||
|
let (found, existing) = ctx.db.get(fullKey)
|
||||||
|
if found and table.len > 0:
|
||||||
|
removeIndexesForRow(ctx, table, fullKey, cast[string](existing))
|
||||||
|
ctx.db.delete(fullKey)
|
||||||
|
|||||||
@@ -971,6 +971,9 @@ proc executeQueryImpl(ctx: ExecutionContext, astNode: Node, params: seq[WireValu
|
|||||||
for key, version in ctx.pendingTxn.writeSet:
|
for key, version in ctx.pendingTxn.writeSet:
|
||||||
if version.isDelete:
|
if version.isDelete:
|
||||||
ctx.db.delete(key)
|
ctx.db.delete(key)
|
||||||
|
# Empty value is the raft/replication "delete" convention — never
|
||||||
|
# ship a non-empty body for isDelete or followers will resurrect.
|
||||||
|
kvPairs.add((key, @[]))
|
||||||
else:
|
else:
|
||||||
ctx.db.put(key, version.value)
|
ctx.db.put(key, version.value)
|
||||||
kvPairs.add((key, version.value))
|
kvPairs.add((key, version.value))
|
||||||
|
|||||||
+5
-3
@@ -344,16 +344,18 @@ proc main() =
|
|||||||
dataDir = raftDataDir)
|
dataDir = raftDataDir)
|
||||||
raftNode.peerAddrs = config.raftPeerAddrs
|
raftNode.peerAddrs = config.raftPeerAddrs
|
||||||
tcpServer.raftNode = raftNode # C3b: executeQuery rejects writes on followers
|
tcpServer.raftNode = raftNode # C3b: executeQuery rejects writes on followers
|
||||||
# Wire state machine to apply committed entries to the default database
|
# Wire state machine: committed entries update LSM + secondary indexes
|
||||||
|
# (B-tree/FTS/HNSW) on every node via applyReplicatedPut/Delete.
|
||||||
let defaultDbInfo = getDatabaseInfo(registry, "default")
|
let defaultDbInfo = getDatabaseInfo(registry, "default")
|
||||||
raftNode.applyCommand = proc(cmd: string, data: seq[byte]) {.gcsafe.} =
|
raftNode.applyCommand = proc(cmd: string, data: seq[byte]) {.gcsafe.} =
|
||||||
withStorageGate:
|
withStorageGate:
|
||||||
|
let ctx = cast[ExecutionContext](cast[pointer](defaultDbInfo.ctx))
|
||||||
if cmd == "put":
|
if cmd == "put":
|
||||||
let parts = cast[string](data).split("\x00")
|
let parts = cast[string](data).split("\x00")
|
||||||
if parts.len >= 2:
|
if parts.len >= 2:
|
||||||
defaultDbInfo.db.put(parts[0], cast[seq[byte]](parts[1]))
|
applyReplicatedPut(ctx, parts[0], cast[seq[byte]](parts[1]))
|
||||||
elif cmd == "delete":
|
elif cmd == "delete":
|
||||||
defaultDbInfo.db.delete(cast[string](data))
|
applyReplicatedDelete(ctx, cast[string](data))
|
||||||
|
|
||||||
# Wire RAFT ↔ DistTxn
|
# Wire RAFT ↔ DistTxn
|
||||||
wireRaftDistTxn(raftNode, tcpServer)
|
wireRaftDistTxn(raftNode, tcpServer)
|
||||||
|
|||||||
@@ -396,3 +396,15 @@ suite "Raft write classification":
|
|||||||
check not isWrite(parse("CREATE TABLE t (id INT)").stmts[0])
|
check not isWrite(parse("CREATE TABLE t (id INT)").stmts[0])
|
||||||
check not isWrite(parse("BEGIN").stmts[0])
|
check not isWrite(parse("BEGIN").stmts[0])
|
||||||
check not isWrite(parse("ROLLBACK").stmts[0])
|
check not isWrite(parse("ROLLBACK").stmts[0])
|
||||||
|
|
||||||
|
test "multi-statement queries with a trailing write are still writes":
|
||||||
|
## Server rejection must not look only at stmts[0] — a SELECT first
|
||||||
|
## would otherwise let a follower execute the INSERT.
|
||||||
|
let ast = parse("SELECT 1; INSERT INTO t (id) VALUES (1)")
|
||||||
|
check ast.stmts.len == 2
|
||||||
|
check not isWrite(ast.stmts[0])
|
||||||
|
check isWrite(ast.stmts[1])
|
||||||
|
var anyWrite = false
|
||||||
|
for s in ast.stmts:
|
||||||
|
if isWrite(s): anyWrite = true
|
||||||
|
check anyWrite
|
||||||
|
|||||||
@@ -211,14 +211,15 @@ proc runWritesScenario() =
|
|||||||
return
|
return
|
||||||
echo "leader elected: ", nodes[leaderIdx].id, " (term ", leaderTerm, ")"
|
echo "leader elected: ", nodes[leaderIdx].id, " (term ", leaderTerm, ")"
|
||||||
|
|
||||||
# Schema: CREATE TABLE is not a raft write (DML only is), and its _schema
|
# Schema: CREATE TABLE / INDEX are not raft writes (no kvPairs), and
|
||||||
# keys are not replicated — create the table locally on every node.
|
# _schema keys are not replicated — create them locally on every node.
|
||||||
for i in 0 ..< nodes.len:
|
for i in 0 ..< nodes.len:
|
||||||
let db = openClient(nodes[i].clientPort)
|
let db = openClient(nodes[i].clientPort)
|
||||||
try:
|
try:
|
||||||
db.exec(sql"CREATE TABLE rw_test (id INT PRIMARY KEY, name STRING)")
|
db.exec(sql"CREATE TABLE rw_test (id INT PRIMARY KEY, name STRING)")
|
||||||
|
db.exec(sql"CREATE INDEX idx_rw_name ON rw_test (name)")
|
||||||
except CatchableError as e:
|
except CatchableError as e:
|
||||||
echo "CREATE TABLE failed on ", nodes[i].id, ": ", e.msg
|
echo "CREATE TABLE/INDEX failed on ", nodes[i].id, ": ", e.msg
|
||||||
dumpAll(nodes)
|
dumpAll(nodes)
|
||||||
fail()
|
fail()
|
||||||
return
|
return
|
||||||
@@ -250,6 +251,31 @@ proc runWritesScenario() =
|
|||||||
return
|
return
|
||||||
echo "row replicated to follower ", nodes[followerIdx].id
|
echo "row replicated to follower ", nodes[followerIdx].id
|
||||||
|
|
||||||
|
# Index path: rich apply must have updated the follower B-tree so a
|
||||||
|
# filtered SELECT (planner prefers secondary index) still finds the row.
|
||||||
|
block:
|
||||||
|
let db = openClient(nodes[followerIdx].clientPort)
|
||||||
|
var saw = false
|
||||||
|
try:
|
||||||
|
let rows = db.getAllRows(
|
||||||
|
sql"SELECT id, name FROM rw_test WHERE name = 'raft-row'")
|
||||||
|
for row in rows:
|
||||||
|
if row.len >= 2 and row[1] == "raft-row":
|
||||||
|
saw = true
|
||||||
|
except CatchableError as e:
|
||||||
|
echo "follower index SELECT failed: ", e.msg
|
||||||
|
dumpAll(nodes)
|
||||||
|
fail()
|
||||||
|
return
|
||||||
|
db.close()
|
||||||
|
if not saw:
|
||||||
|
echo "follower ", nodes[followerIdx].id,
|
||||||
|
" index-backed SELECT missed the replicated row"
|
||||||
|
dumpAll(nodes)
|
||||||
|
fail()
|
||||||
|
return
|
||||||
|
echo "follower index-backed SELECT saw the row"
|
||||||
|
|
||||||
# Follower rejection: DML on a follower must fail with "not leader".
|
# Follower rejection: DML on a follower must fail with "not leader".
|
||||||
block:
|
block:
|
||||||
let db = openClient(nodes[followerIdx].clientPort)
|
let db = openClient(nodes[followerIdx].clientPort)
|
||||||
|
|||||||
@@ -2706,6 +2706,44 @@ suite "Raft SQL Write Path":
|
|||||||
check res.keyValuePairs.len == 1
|
check res.keyValuePairs.len == 1
|
||||||
check res.keyValuePairs[0][1].len == 0
|
check res.keyValuePairs[0][1].len == 0
|
||||||
|
|
||||||
|
test "appendWriteToRaft fails when node is not leader":
|
||||||
|
var n = newRaftNode("n1", @["n2"], raftPort = 29111)
|
||||||
|
# Still a follower — appendLog returns index 0.
|
||||||
|
let (ok, err) = waitFor appendWriteToRaft(n,
|
||||||
|
@[("k", cast[seq[byte]]("v"))], timeoutMs = 200)
|
||||||
|
check not ok
|
||||||
|
check "lost leadership" in err
|
||||||
|
|
||||||
|
test "appendWriteToRaft times out without majority replies":
|
||||||
|
# Leader with peers but no network — commitIndex never advances.
|
||||||
|
var n = newRaftNode("n1", @["n2", "n3"], raftPort = 29112)
|
||||||
|
n.becomeLeader()
|
||||||
|
let (ok, err) = waitFor appendWriteToRaft(n,
|
||||||
|
@[("k", cast[seq[byte]]("v"))], timeoutMs = 300)
|
||||||
|
check not ok
|
||||||
|
check "raft commit timeout" in err
|
||||||
|
|
||||||
|
test "applyReplicatedPut updates secondary B-tree indexes":
|
||||||
|
var testDir = getTempDir() / "baradb_raft_apply_idx_" & $getCurrentProcessId() & "_" & $getMonoTime().ticks
|
||||||
|
createDir(testDir)
|
||||||
|
var db = newLSMTree(testDir)
|
||||||
|
var ctx = qexec.newExecutionContext(db)
|
||||||
|
discard qexec.executeQuery(ctx, parse(
|
||||||
|
"CREATE TABLE t (id INT PRIMARY KEY, name STRING)"))
|
||||||
|
discard qexec.executeQuery(ctx, parse(
|
||||||
|
"CREATE INDEX idx_name ON t (name)"))
|
||||||
|
# Simulate follower apply of a leader-replicated INSERT (no execInsert).
|
||||||
|
applyReplicatedPut(ctx, "t.id=1", cast[seq[byte]]("name=alice"))
|
||||||
|
check "t.name" in ctx.btrees
|
||||||
|
let entries = ctx.btrees["t.name"].get("alice")
|
||||||
|
check entries.len >= 1
|
||||||
|
check entries[0].lsmKey == "t.id=1"
|
||||||
|
# Delete must drop the index entry too.
|
||||||
|
applyReplicatedDelete(ctx, "t.id=1")
|
||||||
|
check ctx.btrees["t.name"].get("alice").len == 0
|
||||||
|
let (found, _) = db.get("t.id=1")
|
||||||
|
check not found
|
||||||
|
|
||||||
suite "CLI Autocomplete":
|
suite "CLI Autocomplete":
|
||||||
test "Autocomplete commands":
|
test "Autocomplete commands":
|
||||||
let res = autocomplete("he")
|
let res = autocomplete("he")
|
||||||
|
|||||||
Reference in New Issue
Block a user