From 5c538232f67fc8c3e9e638a480640c4f295ac220 Mon Sep 17 00:00:00 2001 From: Zaid-edge Date: Fri, 17 Jul 2026 14:15:53 +0500 Subject: [PATCH 1/2] fix(mtree): detect UPDATE-UPDATE conflicts missed by overlapping leaves (ACE-205) mtree table-diff could report divergent nodes as identical: the same primary key holding different non-key data on two nodes produced matching Merkle-tree root hashes, so the diff printed "Merkle trees are identical" while table-diff correctly found the difference. This is the core UPDATE-UPDATE conflict that active-active clusters produce, silently reported as consistent. Root cause: leaf hashing used a closed ("<=") upper bound on block ranges, but a block's range_end is the EXCLUSIVE start of the next block -- the build offsets query pairs bounds with LEAD (each range_end is the following block's range_start) and splitBlocks sets range_end to a split point that GetBulkSplitPoints returns as the first row of the next chunk. The closed bound double-counted every boundary row into two adjacent leaves. When a table's rows collapsed into overlapping leaves that hashed identically (most visibly a single-row table, where the row lands in both [pk,pk] and [pk,inf)), the XOR-based parent hash cancelled the duplicate siblings to zero on every node, so divergent data yielded matching roots. Leaf hashing now uses an exclusive ("<") upper bound, matching the boundaries the build and split paths already produce, so each row belongs to exactly one leaf. Merkle trees built before this fix must be rebuilt (mtree build) to pick up the corrected layout. Adds an integration regression test (single-row and multi-row-boundary cases) asserting mtree diff agrees with table-diff. Verified against the full mtree integration suite plus db/queries and consistency unit tests. --- db/queries/queries.go | 10 +- docs/CHANGELOG.md | 8 ++ .../integration/mtree_update_conflict_test.go | 130 ++++++++++++++++++ 3 files changed, 144 insertions(+), 4 deletions(-) create mode 100644 tests/integration/mtree_update_conflict_test.go diff --git a/db/queries/queries.go b/db/queries/queries.go index 357410c..dbb6f15 100644 --- a/db/queries/queries.go +++ b/db/queries/queries.go @@ -671,10 +671,12 @@ func BlockHashSQL(schema, table string, primaryKeyCols []string, mode string, in endPlaceholders[i] = fmt.Sprintf("$%d", paramIndex) paramIndex++ } - operator := "<=" - if mode == "TD_BLOCK_HASH" { - operator = "<" - } + // Upper bound is always EXCLUSIVE: a block's range_end is the next + // block's range_start (from LEAD in the build offsets and from split + // points), so a closed "<=" would hash boundary rows into two adjacent + // leaves. XOR parent hashing then cancels the duplicate siblings, letting + // divergent data produce matching root hashes and hiding real conflicts. + operator := "<" var upperExpr string if len(primaryKeyCols) == 1 { upperExpr = fmt.Sprintf("%s %s %s", pkComparisonExpression, operator, endPlaceholders[0]) diff --git a/docs/CHANGELOG.md b/docs/CHANGELOG.md index a68d209..df4ef21 100644 --- a/docs/CHANGELOG.md +++ b/docs/CHANGELOG.md @@ -54,6 +54,14 @@ This release focuses primarily on Merkle tree functionality. means "re-run or raise", since drain progress is durable). ### Fixed +- **`mtree table-diff` could report divergent nodes as identical (silent + false-negative).** Leaf hashing used a closed (`<=`) upper bound, so rows on + block boundaries were hashed into two adjacent leaves; the XOR-based parent + hash then cancelled the duplicate siblings, letting the same primary key with + *different* non-key data produce matching root hashes — the core UPDATE-UPDATE + conflict active-active clusters produce. Leaf hashing now uses an exclusive + (`<`) upper bound, so each row belongs to exactly one leaf. **Merkle trees + built before this fix must be rebuilt (`mtree build`).** - **A bounded CDC drain could silently under-report divergence.** The drain treated a 1-second idle timeout as "caught up", so a node with a large backlog could leave its Merkle tree stale and the diff would report the diff --git a/tests/integration/mtree_update_conflict_test.go b/tests/integration/mtree_update_conflict_test.go new file mode 100644 index 0000000..53ca910 --- /dev/null +++ b/tests/integration/mtree_update_conflict_test.go @@ -0,0 +1,130 @@ +// /////////////////////////////////////////////////////////////////////////// +// +// # ACE - Active Consistency Engine +// +// Copyright (C) 2023 - 2026, pgEdge (https://www.pgedge.com/) +// +// This software is released under the PostgreSQL License: +// https://opensource.org/license/postgresql +// +// /////////////////////////////////////////////////////////////////////////// + +package integration + +import ( + "context" + "fmt" + "testing" + + "github.com/jackc/pgx/v5/pgxpool" + "github.com/stretchr/testify/require" +) + +// Regression: same PK exists on both nodes with different non-key data at BUILD +// time (an UPDATE-UPDATE conflict). mtree diff must report it, matching plain +// table-diff. +// +// The divergence exists before the tree is built -- unlike +// testMerkleTreeDiffModifiedRows, which builds on identical data and mutates +// afterward, so it never exercised the degenerate build-time range layout. +// +// The bug: a block's range_end is the EXCLUSIVE start of the next block, but +// leaf hashing used a closed "<=" upper bound, double-counting boundary rows +// into two adjacent leaves. A single-row table collapses into two identical +// overlapping leaves ([pk,pk] and [pk,inf)); the XOR-based parent hash cancels +// the duplicate siblings to zero on every node, so divergent data yielded +// matching root hashes and the diff reported "trees identical". The single-row +// case below reproduced it; the multi-row case guards the boundary. +func TestMtreeDiff_UpdateUpdateConflictAtBuildTime(t *testing.T) { + t.Run("SingleRow", func(t *testing.T) { + runMtreeUpdateConflictCase(t, "mtree_uu_conflict_1", []int{1}, 1) + }) + t.Run("MultiRowBoundary", func(t *testing.T) { + // The divergent row (id=3) is the middle block boundary under BlockSize=2. + runMtreeUpdateConflictCase(t, "mtree_uu_conflict_n", []int{1, 2, 3, 4, 5}, 3) + }) +} + +// runMtreeUpdateConflictCase seeds ids on both nodes, diverging divergentID's +// non-key column, then asserts both table-diff and mtree diff report exactly +// one divergent row. +func runMtreeUpdateConflictCase(t *testing.T, tableName string, ids []int, divergentID int) { + t.Helper() + ctx := context.Background() + qualifiedTable := fmt.Sprintf("%s.%s", testSchema, tableName) + nodes := []string{serviceN1, serviceN2} + + for i, pool := range []*pgxpool.Pool{pgCluster.Node1Pool, pgCluster.Node2Pool} { + nodeName := pgCluster.ClusterNodes[i]["Name"].(string) + _, err := pool.Exec(ctx, fmt.Sprintf( // nosemgrep + "CREATE TABLE IF NOT EXISTS %s (id INT PRIMARY KEY, name VARCHAR)", qualifiedTable)) + require.NoError(t, err, "create table on %s", nodeName) + _, err = pool.Exec(ctx, fmt.Sprintf( // nosemgrep + "SELECT spock.repset_add_table('default', '%s')", qualifiedTable)) + require.NoError(t, err, "add to repset on %s", nodeName) + } + t.Cleanup(func() { + for _, pool := range []*pgxpool.Pool{pgCluster.Node1Pool, pgCluster.Node2Pool} { + pool.Exec(ctx, fmt.Sprintf("DROP TABLE IF EXISTS %s CASCADE", qualifiedTable)) // nosemgrep + } + }) + + // Seed identical rows on both nodes, then diverge divergentID's non-key + // column. repair_mode(true) keeps these writes from replicating. + seed := func(pool *pgxpool.Pool, divergentName string) { + t.Helper() + tx, err := pool.Begin(ctx) + require.NoError(t, err) + defer tx.Rollback(ctx) // safe to call even after Commit() in pgx + _, err = tx.Exec(ctx, "SELECT spock.repair_mode(true)") + require.NoError(t, err) + for _, id := range ids { + name := fmt.Sprintf("name-%d", id) + if id == divergentID { + name = divergentName + } + _, err = tx.Exec(ctx, fmt.Sprintf( // nosemgrep + "INSERT INTO %s (id, name) VALUES ($1, $2)", qualifiedTable), id, name) + require.NoError(t, err) + } + _, err = tx.Exec(ctx, "SELECT spock.repair_mode(false)") + require.NoError(t, err) + require.NoError(t, tx.Commit(ctx)) + } + seed(pgCluster.Node1Pool, "zaid") + seed(pgCluster.Node2Pool, "shabbir") + + // Baseline: plain table-diff must find the divergence. + tdTask := newTestTableDiffTask(t, qualifiedTable, nodes) + require.NoError(t, tdTask.RunChecks(false)) + require.NoError(t, tdTask.ExecuteTask()) + require.Equal(t, 1, sumDiffRows(tdTask.DiffResult.Summary.DiffRowsCount), + "table-diff must find the id=%d conflict", divergentID) + + // mtree diff on the same build-time divergence must ALSO find it. + mtreeTask := newTestMerkleTreeTask(t, qualifiedTable, nodes) + mtreeTask.BlockSize = 2 + mtreeTask.OverrideBlockSize = true + require.NoError(t, mtreeTask.RunChecks(false)) + require.NoError(t, mtreeTask.MtreeInit()) + t.Cleanup(func() { _ = mtreeTask.MtreeTeardown() }) + require.NoError(t, mtreeTask.BuildMtree()) + + diffTask := newTestMerkleTreeTask(t, qualifiedTable, nodes) + diffTask.Mode = "diff" + diffTask.Output = "json" + diffTask.BlockSize = 2 + diffTask.OverrideBlockSize = true + require.NoError(t, diffTask.RunChecks(false)) + require.NoError(t, diffTask.DiffMtree()) + require.Equal(t, 1, sumDiffRows(diffTask.DiffResult.Summary.DiffRowsCount), + "mtree diff must ALSO find the id=%d conflict", divergentID) +} + +func sumDiffRows(counts map[string]int) int { + total := 0 + for _, c := range counts { + total += c + } + return total +} From f2262361430e1dd18cdf32e8bc4ae5f9d96adf8b Mon Sep 17 00:00:00 2001 From: Mason Sharp Date: Tue, 21 Jul 2026 09:27:06 -0700 Subject: [PATCH 2/2] docs(changelog): add v2.1.1 entries for repair commit_ts and empty-table build Document pick_freshest latest-commit-wins (new capability) and the clearer mtree build empty-table error. --- docs/CHANGELOG.md | 27 +++++++++++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/docs/CHANGELOG.md b/docs/CHANGELOG.md index df4ef21..a9e4b1c 100644 --- a/docs/CHANGELOG.md +++ b/docs/CHANGELOG.md @@ -2,6 +2,33 @@ All notable changes to ACE will be captured in this document. This project follows semantic versioning; the latest changes appear first. +## [v2.1.1] + +### Added +- **`repair` can now resolve `pick_freshest` by spock `commit_ts` + (latest-commit-wins).** `pick_freshest` previously only compared ordinary data + columns (e.g. `updated_at`); keying it on `commit_ts` silently fell back to + `tie` because that field is held in the diff's spock metadata, not the row. It + now resolves the key from row data or metadata, so a plan can keep the side + with the newer commit timestamp. See the latest-commit-wins repair example. + +### Changed +- **`mtree build` on an empty table now fails with a clear message.** A table + with 0 rows on every node previously errored with a misleading "could not + determine a reference node" (as if row-estimate collection had failed). It now + reports that the table is empty and to add data before building, and correctly + distinguishes an all-empty table from an actual estimate failure. + +### Fixed +- **`mtree table-diff` could report divergent nodes as identical (silent + false-negative).** Leaf hashing used a closed (`<=`) upper bound, so rows on + block boundaries were hashed into two adjacent leaves; the XOR-based parent + hash then cancelled the duplicate siblings, letting the same primary key with + *different* non-key data produce matching root hashes — the core UPDATE-UPDATE + conflict active-active clusters produce. Leaf hashing now uses an exclusive + (`<`) upper bound, so each row belongs to exactly one leaf. **Merkle trees + built before this fix must be rebuilt (`mtree build`).** + ## [v2.1.0] This release focuses primarily on Merkle tree functionality.