Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 16 additions & 0 deletions src/backend/distributed/planner/distributed_planner.c
Original file line number Diff line number Diff line change
Expand Up @@ -2085,6 +2085,22 @@ multi_join_restriction_hook(PlannerInfo *root,
JoinType jointype,
JoinPathExtraData *extra)
{
#if PG_VERSION_NUM >= PG_VERSION_19

/*
* PG19's eager aggregation builds a second, grouped RelOptInfo for a join
* relation and runs add_paths_to_joinrel() over it again, so this hook
* fires a second time for the same relids and would append a duplicate
* JoinRestriction. Grouped join relations are an intra-node costing
* artifact that no consumer of joinRestrictionList needs, so skip them and
* keep the recorded context describing each join exactly once.
*/
if (IS_GROUPED_REL(joinrel))
{
return;
}
#endif

if (bms_is_empty(innerrel->relids) || bms_is_empty(outerrel->relids))
{
/*
Expand Down
86 changes: 86 additions & 0 deletions src/test/regress/expected/pg19.out
Original file line number Diff line number Diff line change
Expand Up @@ -987,3 +987,89 @@ SET client_min_messages TO WARNING;
DROP SCHEMA pg19_repack CASCADE;
RESET client_min_messages;
RESET search_path;
-- PG19 eager aggregation can build a second, grouped RelOptInfo for a join
-- relation and re-run path generation over it, which makes the Citus join
-- restriction hook fire twice for the same relids. Citus skips those grouped
-- rels; make sure queries that reach that path still return correct results.
CREATE SCHEMA pg19_eager_agg;
SET search_path TO pg19_eager_agg;
SET citus.shard_count TO 4;
CREATE TABLE eager_a (id int, bid int, val int);
CREATE TABLE eager_b (id int, cid int);
CREATE TABLE eager_c (id int, c int);
SELECT create_distributed_table('eager_a', 'id');
create_distributed_table
---------------------------------------------------------------------

(1 row)

SELECT create_distributed_table('eager_b', 'id');
create_distributed_table
---------------------------------------------------------------------

(1 row)

SELECT create_distributed_table('eager_c', 'id');
create_distributed_table
---------------------------------------------------------------------

(1 row)

CREATE TABLE eager_ref (id int, label text);
SELECT create_reference_table('eager_ref');
create_reference_table
---------------------------------------------------------------------

(1 row)

INSERT INTO eager_a SELECT i, i % 50, i FROM generate_series(1, 500) i;
INSERT INTO eager_b SELECT i, i % 10 FROM generate_series(1, 500) i;
INSERT INTO eager_c SELECT i, i % 5 FROM generate_series(1, 500) i;
INSERT INTO eager_ref SELECT i, 'l' || (i % 3) FROM generate_series(1, 50) i;
-- min_eager_agg_group_size = 0 removes the eligibility gate, so the planner
-- builds grouped join relations it would otherwise skip.
SET enable_eager_aggregate TO on;
SET min_eager_agg_group_size TO 0;
-- colocated three-way join with grouping
SELECT c.c, count(*) AS n, sum(a.val) AS s
FROM eager_a a JOIN eager_b b ON a.id = b.id JOIN eager_c c ON b.id = c.id
GROUP BY c.c ORDER BY c.c;
c | n | s
---------------------------------------------------------------------
0 | 100 | 25250
1 | 100 | 24850
2 | 100 | 24950
3 | 100 | 25050
4 | 100 | 25150
(5 rows)

-- join against a reference table
SELECT r.label, count(*) AS n
FROM eager_a a JOIN eager_b b ON a.id = b.id JOIN eager_ref r ON a.bid = r.id
GROUP BY r.label ORDER BY r.label;
label | n
---------------------------------------------------------------------
l0 | 160
l1 | 170
l2 | 160
(3 rows)

-- outer join must keep the same restriction context
SELECT c.c, count(b.id) AS n
FROM eager_c c LEFT JOIN eager_b b ON c.id = b.id
GROUP BY c.c ORDER BY c.c;
c | n
---------------------------------------------------------------------
0 | 100
1 | 100
2 | 100
3 | 100
4 | 100
(5 rows)

RESET min_eager_agg_group_size;
RESET enable_eager_aggregate;
SET client_min_messages TO WARNING;
DROP SCHEMA pg19_eager_agg CASCADE;
RESET client_min_messages;
RESET search_path;
42 changes: 42 additions & 0 deletions src/test/regress/sql/pg19.sql
Original file line number Diff line number Diff line change
Expand Up @@ -749,3 +749,45 @@ SET client_min_messages TO WARNING;
DROP SCHEMA pg19_repack CASCADE;
RESET client_min_messages;
RESET search_path;

-- PG19 eager aggregation can build a second, grouped RelOptInfo for a join
-- relation and re-run path generation over it, which makes the Citus join
-- restriction hook fire twice for the same relids. Citus skips those grouped
-- rels; make sure queries that reach that path still return correct results.
CREATE SCHEMA pg19_eager_agg;
SET search_path TO pg19_eager_agg;
SET citus.shard_count TO 4;
CREATE TABLE eager_a (id int, bid int, val int);
CREATE TABLE eager_b (id int, cid int);
CREATE TABLE eager_c (id int, c int);
SELECT create_distributed_table('eager_a', 'id');
SELECT create_distributed_table('eager_b', 'id');
SELECT create_distributed_table('eager_c', 'id');
CREATE TABLE eager_ref (id int, label text);
SELECT create_reference_table('eager_ref');
INSERT INTO eager_a SELECT i, i % 50, i FROM generate_series(1, 500) i;
INSERT INTO eager_b SELECT i, i % 10 FROM generate_series(1, 500) i;
INSERT INTO eager_c SELECT i, i % 5 FROM generate_series(1, 500) i;
INSERT INTO eager_ref SELECT i, 'l' || (i % 3) FROM generate_series(1, 50) i;
-- min_eager_agg_group_size = 0 removes the eligibility gate, so the planner
-- builds grouped join relations it would otherwise skip.
SET enable_eager_aggregate TO on;
SET min_eager_agg_group_size TO 0;
-- colocated three-way join with grouping
SELECT c.c, count(*) AS n, sum(a.val) AS s
FROM eager_a a JOIN eager_b b ON a.id = b.id JOIN eager_c c ON b.id = c.id
GROUP BY c.c ORDER BY c.c;
-- join against a reference table
SELECT r.label, count(*) AS n
FROM eager_a a JOIN eager_b b ON a.id = b.id JOIN eager_ref r ON a.bid = r.id
GROUP BY r.label ORDER BY r.label;
-- outer join must keep the same restriction context
SELECT c.c, count(b.id) AS n
FROM eager_c c LEFT JOIN eager_b b ON c.id = b.id
GROUP BY c.c ORDER BY c.c;
RESET min_eager_agg_group_size;
RESET enable_eager_aggregate;
SET client_min_messages TO WARNING;
DROP SCHEMA pg19_eager_agg CASCADE;
RESET client_min_messages;
RESET search_path;
Loading