Improve the speed of the aggregation dataview
Improve the performance of the aggregation dataview. Instead of using a CTE (WITH) for filtered_source, which is only used in one place to calculate ranks, inject it as a subquery. This way the planner has a chance to ignore uneeded columns as well as to parallelize the exectution of the window function (WindowAgg in the query plan). That is the part that takes most of the time of the query. The improvement is about 20-40% in speed on PG10 with 4 cores.
This commit is contained in:
@@ -2,7 +2,7 @@ const BaseDataview = require('./base');
|
||||
const debug = require('debug')('windshaft:dataview:aggregation');
|
||||
|
||||
const filteredQueryTpl = ctx => `
|
||||
filtered_source AS (
|
||||
(
|
||||
SELECT *
|
||||
FROM (${ctx.query}) _cdb_filtered_source
|
||||
${ctx.aggregationColumn && ctx.isFloatColumn ? `
|
||||
@@ -14,7 +14,7 @@ const filteredQueryTpl = ctx => `
|
||||
${ctx.aggregationColumn} != 'NaN'::float` :
|
||||
''
|
||||
}
|
||||
)
|
||||
) filtered_source
|
||||
`;
|
||||
|
||||
const summaryQueryTpl = ctx => `
|
||||
@@ -43,7 +43,7 @@ const rankedCategoriesQueryTpl = ctx => `
|
||||
${ctx.column} AS category,
|
||||
${ctx.aggregationFn} AS value,
|
||||
row_number() OVER (ORDER BY ${ctx.aggregationFn} desc) as rank
|
||||
FROM filtered_source
|
||||
FROM ${filteredQueryTpl(ctx)}
|
||||
${ctx.aggregationColumn !== null ? `WHERE ${ctx.aggregationColumn} IS NOT NULL` : ''}
|
||||
GROUP BY ${ctx.column}
|
||||
ORDER BY 2 DESC
|
||||
@@ -134,7 +134,6 @@ const aggregationFnQueryTpl = ctx => `${ctx.aggregation}(${ctx.aggregationColumn
|
||||
|
||||
const aggregationDataviewQueryTpl = ctx => `
|
||||
WITH
|
||||
${filteredQueryTpl(ctx)},
|
||||
${summaryQueryTpl(ctx)},
|
||||
${rankedCategoriesQueryTpl(ctx)},
|
||||
${categoriesSummaryMinMaxQueryTpl(ctx)},
|
||||
|
||||
Reference in New Issue
Block a user