diff --git a/docs/bg/known-limitations.md b/docs/bg/known-limitations.md index 3464f81..06d60ee 100644 --- a/docs/bg/known-limitations.md +++ b/docs/bg/known-limitations.md @@ -35,6 +35,9 @@ Crash recovery с WAL, schema/index persist, `/health` + `/metrics`, offline bac - **Legacy REP replication (без raft)** — пътят още извежда delete от празна стойност; insert в PK-only таблица се прилага грешно по него (редът изчезва). Използвай raft. - **Snapshot-restore ctx** — след InstallSnapshot restore HTTP endpoints със startup-captured ctx може да сервират стари данни до рестарт (`/query` е свеж per-request); съществуващите клиентски връзки виждат pre-restore състояние — reconnect след restore. +- **FK-cascade дивергенция под raft** — ефектите на `ON DELETE/UPDATE CASCADE` (и `SET NULL`) не се реплицират през raft: followers прилагат само KV промяната на родителския ред, така че каскадираните дъщерни редове остават на followers. Избягвай FK actions върху raft-реплицирани таблици или приеми периодичен snapshot resync. +- **Непотвърдени записи в snapshots** — leader прилага записите локално преди raft majority commit; snapshot, направен в този прозорец, може да включи записи, които никога не се commit-ват (фантомни редове след restore + смяна на leadership). Тесен прозорец; поправката е планирана за следващ release. +- **Блокиране на event loop при snapshot build/restore** — snapshot build/restore изпълнява блокиращ tar/gzip на event loop на възела; големи data dirs могат да забавят heartbeats и да предизвикат election по средата на трансфер. ## Виж също diff --git a/docs/en/known-limitations.md b/docs/en/known-limitations.md index 4c00bd0..47597c4 100644 --- a/docs/en/known-limitations.md +++ b/docs/en/known-limitations.md @@ -48,6 +48,9 @@ Documented in [distributed.md](distributed.md). Supported scope: - **Legacy non-raft REP replication infers delete from empty value** — the non-raft replication path still treats an empty value as a delete, so inserts into a PK-only table are misapplied over that path (the row vanishes). Use raft replication instead. - **Snapshot-restore ctx staleness** — after an InstallSnapshot restore, HTTP endpoints using the startup-captured ctx may serve stale data until the node is restarted; the `/query` path is fresh per-request. Pre-existing client connections likewise see pre-restore state — reconnect after a restore. +- **FK-cascade divergence under raft** — `ON DELETE/UPDATE CASCADE` (and `SET NULL`) effects are not raft-replicated: followers only apply the parent row's KV change, so cascaded child rows persist on followers. Avoid FK actions on raft-replicated tables, or accept periodic snapshot resync. +- **Uncommitted writes in snapshots** — the leader applies writes locally before raft majority commit; a snapshot taken in that window can include writes that never commit (phantom rows after restore + leadership change). Narrow window; fix tracked for a later release. +- **Event-loop stall during snapshot build/restore** — snapshot build/restore performs blocking tar/gzip on the node's event loop; large data dirs can stall heartbeats and trigger an election mid-transfer. ## Operational requirements diff --git a/docs/en/release-checklist.md b/docs/en/release-checklist.md index cd852d8..db39b01 100644 --- a/docs/en/release-checklist.md +++ b/docs/en/release-checklist.md @@ -1,4 +1,4 @@ -# Release checklist — v1.2.0 Production GA +# Release checklist — v1.3.0 raft-supported Use before tagging and publishing artifacts. @@ -6,8 +6,8 @@ Use before tagging and publishing artifacts. - [ ] Working tree clean on `main` - [ ] [Known limitations](known-limitations.md) accurate -- [ ] `CHANGELOG.md` has dated `## [1.2.0]` (not Unreleased for shipped items) -- [ ] `baradadb.nimble` version `1.2.0` +- [ ] `CHANGELOG.md` has dated `## [1.3.0]` (not Unreleased for shipped items) +- [ ] `baradadb.nimble` version `1.3.0` ## Tests @@ -44,17 +44,17 @@ docker compose -f docker-compose.prod.yml config >/dev/null ```bash nimble build_release # or: nim c -d:release -o:build/baradadb src/baradadb.nim -docker build -t baradb:1.2.0 -t baradb:latest . +docker build -t baradb:1.3.0 -t baradb:latest . ``` ## Tag ```bash -git tag -a v1.2.0 -m "BaraDB v1.2.0 Production GA (single-node)" +git tag -a v1.3.0 -m "BaraDB v1.3.0 raft-supported" git push origin main --tags ``` ## Post-release - [ ] Smoke: start prod compose, `/health` → ok, auth required for `/query` -- [ ] Announce: single-node GA; Raft experimental (link known-limitations) +- [ ] Announce: raft-supported release (3-node, `default` DB); link known-limitations diff --git a/src/barabadb/core/httpserver.nim b/src/barabadb/core/httpserver.nim index 09c72cf..e717831 100644 --- a/src/barabadb/core/httpserver.nim +++ b/src/barabadb/core/httpserver.nim @@ -31,7 +31,7 @@ type config: BaraConfig running: bool db*: LSMTree - ctx: ExecutionContext + ctx*: ExecutionContext # read/write only under the storage gate registry*: DatabaseRegistry metrics*: Metrics secretKey*: string diff --git a/src/baradadb.nim b/src/baradadb.nim index c7aa166..e15d191 100644 --- a/src/baradadb.nim +++ b/src/baradadb.nim @@ -404,30 +404,40 @@ proc main() = # the rest of the server); the registry ctxFactory type is not marked # gcsafe, which would otherwise reject the call. {.cast(gcsafe).}: - try: - defaultDbInfo.db.close() - # restoreDataDir moves the old dir aside and extracts the archive; on - # extraction failure it rolls back automatically. Reopen whatever is - # on disk either way so the node is not left with a closed DB. - let restored = restoreDataDir(archivePath, defaultDbDir) - let reopened = registry.reopenDatabase("default") - if reopened: - # Serve the (re)opened data. Client connections clone tcpServer.ctx - # on accept (cloneForConnection), and reopenDatabase installs a NEW - # ctx object in the registry slot — without repointing, queries - # keep reading the closed pre-restore LSM (empty results, no - # error). The websocket change hook was installed on the previous - # ctx object; carry it over. - let oldCtx = tcpServer.ctx - let newCtx = cast[ExecutionContext](cast[pointer](defaultDbInfo.ctx)) - newCtx.onChange = oldCtx.onChange - tcpServer.db = defaultDbInfo.db - tcpServer.ctx = newCtx - tcpServer.txnManager = newCtx.txnManager - result = reopened and restored - except CatchableError as e: - echo "[raft] Snapshot restore failed: ", e.msg - result = false + # Hold the storage gate for the whole close/extract/reopen/repoint + # sequence: HTTP workers run queries under the same gate, so this + # cannot close the LSM out from under an in-flight /query. + withStorageGate: + try: + defaultDbInfo.db.close() + # restoreDataDir moves the old dir aside and extracts the archive; on + # extraction failure it rolls back automatically. Reopen whatever is + # on disk either way so the node is not left with a closed DB. + let restored = restoreDataDir(archivePath, defaultDbDir) + let reopened = registry.reopenDatabase("default") + if reopened: + # Serve the (re)opened data. Client connections clone tcpServer.ctx + # on accept (cloneForConnection), and reopenDatabase installs a NEW + # ctx object in the registry slot — without repointing, queries + # keep reading the closed pre-restore LSM (empty results, no + # error). The websocket change hook was installed on the previous + # ctx object; carry it over. + let oldCtx = tcpServer.ctx + let newCtx = cast[ExecutionContext](cast[pointer](defaultDbInfo.ctx)) + newCtx.onChange = oldCtx.onChange + tcpServer.db = defaultDbInfo.db + tcpServer.ctx = newCtx + tcpServer.txnManager = newCtx.txnManager + # The HTTP thread reads server.db/ctx only inside + # withStorageGate (getRequestDatabaseContext in the /query and + # /tables handlers), so repointing them here — while this thread + # holds the gate — is race-free against request handlers. + httpServer.db = defaultDbInfo.db + httpServer.ctx = newCtx + result = reopened and restored + except CatchableError as e: + echo "[raft] Snapshot restore failed: ", e.msg + result = false # Leader InstallSnapshot send: archive the default DB's data directory # into the path raft picks (dataDir/raft/snap_out_.tar.gz). Like @@ -436,11 +446,15 @@ proc main() = raftNode.buildSnapshot = proc(destPath: string): bool {.gcsafe.} = echo "[raft] Building snapshot archive ", destPath {.cast(gcsafe).}: - try: - result = backupDataDir(defaultDbDir, destPath) - except CatchableError as e: - echo "[raft] Snapshot build failed: ", e.msg - result = false + # Hold the storage gate while tarring the data dir so a concurrent + # memtable flush (HTTP /query path) cannot write an SSTable + # mid-archive. + withStorageGate: + try: + result = backupDataDir(defaultDbDir, destPath) + except CatchableError as e: + echo "[raft] Snapshot build failed: ", e.msg + result = false # Wire RAFT ↔ DistTxn wireRaftDistTxn(raftNode, tcpServer)