From 1b405e42c2804f2ab85ae200be077c9f3ad4a0d5 Mon Sep 17 00:00:00 2001 From: Raul Marin Date: Mon, 4 Dec 2017 17:03:31 +0100 Subject: [PATCH] Date histogram optimizations --- .../dataview/histograms/date-histogram.js | 229 +++++++++--------- 1 file changed, 121 insertions(+), 108 deletions(-) diff --git a/lib/cartodb/models/dataview/histograms/date-histogram.js b/lib/cartodb/models/dataview/histograms/date-histogram.js index f1790b15..785d132a 100644 --- a/lib/cartodb/models/dataview/histograms/date-histogram.js +++ b/lib/cartodb/models/dataview/histograms/date-histogram.js @@ -1,5 +1,102 @@ const BaseHistogram = require('./base-histogram'); const debug = require('debug')('windshaft:dataview:date-histogram'); +const utils = require('../../../utils/query-utils'); + +/** + * Gets the name of a timezone with the same offset as the required + * using the pg_timezone_names table. We do this because it's simpler to pass + * the name than to pass the offset itself as PostgreSQL uses different + * sign convention. For example: TIME ZONE 'CET' is equal to TIME ZONE 'UTC-1', + * not 'UTC+1' which would be expected. + * Gives priority to Etc/GMT±N timezones but still support odd offsets like 8.5 + * hours for Asia/Pyongyang. + * It also makes it easier to, in the future, support the input of expected timezone + * instead of the offset; that is using 'Europe/Madrid' instead of + * '+3600' or '+7200'. The daylight saving status can be handled by postgres. + */ +const offsetNameQueryTpl = ctx => ` +WITH __wd_tz AS +( + SELECT name + FROM pg_timezone_names + WHERE utc_offset = interval '${ctx.offset} hours' + ORDER BY CASE WHEN name LIKE 'Etc/GMT%' THEN 0 ELSE 1 END + LIMIT 1 +),`; + +/** + * Function to get the subquery that places each row in its bin depending on + * the aggregation. Since the data stored is in epoch we need to adapt it to + * our timezone so when calling date_trunc it falls into the correct bin + */ +function dataBucketsQuery(ctx) { + var condition_str = ''; + + if (ctx.start !== 0) { + condition_str = `WHERE ${ctx.column} >= to_timestamp(${ctx.start})`; + } + if (ctx.end !== 0) { + if (condition_str === '') { + condition_str = `WHERE ${ctx.column} <= to_timestamp(${ctx.end})`; + } + else { + condition_str += ` and ${ctx.column} <= to_timestamp(${ctx.end})`; + } + } + + return ` +__wd_buckets AS +( + SELECT + date_trunc('${ctx.aggregation}', timezone(__wd_tz.name, ${ctx.column}::timestamptz)) as timestamp, + count(*) as freq, + ${utils.countNULLs(ctx)} as nulls_count + FROM + ( + ${ctx.query} + ) __source, __wd_tz + ${condition_str} + GROUP BY timestamp +),`; +} + +/** + * Function that generates an array with all the possible bins between the + * start and end date. If not provided we use the min and max generated from + * the dataBucketsQuery + */ +function allBucketsArrayQuery(ctx) { + var extra_from = ``; + var series_start = ``; + var series_end = ``; + + if (ctx.start === 0) { + extra_from = `, __wd_buckets GROUP BY __wd_tz.name`; + series_start = `min(__wd_buckets.timestamp)`; + } else { + series_start = `date_trunc('${ctx.aggregation}', timezone(__wd_tz.name, to_timestamp(${ctx.start})))`; + } + + if (ctx.end === 0) { + extra_from = `, __wd_buckets GROUP BY __wd_tz.name`; + series_end = `max(__wd_buckets.timestamp)`; + } else { + series_end = `date_trunc('${ctx.aggregation}', timezone(__wd_tz.name, to_timestamp(${ctx.end})))`; + } + + return ` +__wd_all_buckets AS +( + SELECT ARRAY( + SELECT + generate_series( + ${series_start}, + ${series_end}, + interval '${ctx.interval}') as bin_start + FROM __wd_tz${extra_from} + ) as bins +)`; +} const dateIntervalQueryTpl = ctx => ` WITH @@ -41,107 +138,6 @@ const dateIntervalQueryTpl = ctx => ` FROM __cdb_interval_in_days, __cdb_interval_in_hours, __cdb_interval_in_minutes, __cdb_interval_in_seconds `; -const nullsQueryTpl = ctx => ` - __cdb_nulls AS ( - SELECT - count(*) AS __cdb_nulls_count - FROM (${ctx.query}) __cdb_histogram_nulls - WHERE ${ctx.column} IS NULL - ) -`; - -const dateBasicsQueryTpl = ctx => ` - __cdb_basics AS ( - SELECT - max(date_part('epoch', ${ctx.column})) AS __cdb_max_val, - min(date_part('epoch', ${ctx.column})) AS __cdb_min_val, - avg(date_part('epoch', ${ctx.column})) AS __cdb_avg_val, - min( - date_trunc( - '${ctx.aggregation}', ${ctx.column}::timestamp AT TIME ZONE '${ctx.offset}' - ) - ) AS __cdb_start_date, - max(${ctx.column}::timestamp AT TIME ZONE '${ctx.offset}') AS __cdb_end_date, - count(1) AS __cdb_total_rows - FROM (${ctx.query}) __cdb_basics_query - ) -`; - -const dateOverrideBasicsQueryTpl = ctx => ` - __cdb_basics AS ( - SELECT - max(${ctx.end})::float AS __cdb_max_val, - min(${ctx.start})::float AS __cdb_min_val, - avg(date_part('epoch', ${ctx.column})) AS __cdb_avg_val, - min( - date_trunc( - '${ctx.aggregation}', - TO_TIMESTAMP(${ctx.start})::timestamp AT TIME ZONE '${ctx.offset}' - ) - ) AS __cdb_start_date, - max( - TO_TIMESTAMP(${ctx.end})::timestamp AT TIME ZONE '${ctx.offset}' - ) AS __cdb_end_date, - count(1) AS __cdb_total_rows - FROM (${ctx.query}) __cdb_basics_query - ) -`; - -const dateBinsQueryTpl = ctx => ` - __cdb_bins AS ( - SELECT - __cdb_bins_array, - ARRAY_LENGTH(__cdb_bins_array, 1) AS __cdb_bins_number - FROM ( - SELECT - ARRAY( - SELECT GENERATE_SERIES( - __cdb_start_date::timestamptz, - __cdb_end_date::timestamptz, - ${ctx.aggregation === 'quarter' ? `'3 month'::interval` : `'1 ${ctx.aggregation}'::interval`} - ) - ) AS __cdb_bins_array - FROM __cdb_basics - ) __cdb_bins_array_query - ) -`; - -const dateHistogramQueryTpl = ctx => ` - SELECT - (__cdb_max_val - __cdb_min_val) / cast(__cdb_bins_number as float) AS bin_width, - __cdb_bins_number AS bins_number, - __cdb_nulls_count AS nulls_count, - CASE WHEN __cdb_min_val = __cdb_max_val - THEN 0 - ELSE GREATEST( - 1, - LEAST( - WIDTH_BUCKET( - ${ctx.column}::timestamp AT TIME ZONE '${ctx.offset}', - __cdb_bins_array - ), - __cdb_bins_number - ) - ) - 1 - END AS bin, - min( - date_part( - 'epoch', - date_trunc( - '${ctx.aggregation}', ${ctx.column}::timestamp AT TIME ZONE '${ctx.offset}' - ) AT TIME ZONE '${ctx.offset}' - ) - )::numeric AS timestamp, - date_part('epoch', __cdb_start_date)::numeric AS timestamp_start, - min(date_part('epoch', ${ctx.column}))::numeric AS min, - max(date_part('epoch', ${ctx.column}))::numeric AS max, - avg(date_part('epoch', ${ctx.column}))::numeric AS avg, - count(*) AS freq - FROM (${ctx.query}) __cdb_histogram, __cdb_basics, __cdb_bins, __cdb_nulls - WHERE date_part('epoch', ${ctx.column}) IS NOT NULL - GROUP BY bin, bins_number, bin_width, nulls_count, timestamp_start - ORDER BY bin -`; const MAX_INTERVAL_VALUE = 366; @@ -176,12 +172,21 @@ module.exports = class DateHistogram extends BaseHistogram { _buildQueryTpl (ctx) { return ` - WITH - ${this._hasOverridenRange(ctx.override) ? dateOverrideBasicsQueryTpl(ctx) : dateBasicsQueryTpl(ctx)}, - ${dateBinsQueryTpl(ctx)}, - ${nullsQueryTpl(ctx)} - ${dateHistogramQueryTpl(ctx)} - `; +${offsetNameQueryTpl(ctx)} +${dataBucketsQuery(ctx)} +${allBucketsArrayQuery(ctx)} +SELECT + array_position(__wd_all_buckets.bins, __wd_buckets.timestamp) - 1 as bin, + date_part('epoch', timezone(__wd_tz.name, __wd_buckets.timestamp)) AS timestamp, + __wd_buckets.freq as freq, + date_part('epoch', timezone(__wd_tz.name, (__wd_all_buckets.bins)[1])) as timestamp_start, + array_length(__wd_all_buckets.bins, 1) as bins_number, + date_part('epoch', interval '${ctx.interval}') as bin_width, + __wd_buckets.nulls_count as nulls_count +FROM __wd_buckets, __wd_all_buckets, __wd_tz +GROUP BY __wd_tz.name, __wd_all_buckets.bins, __wd_buckets.timestamp, __wd_buckets.nulls_count, __wd_buckets.freq +ORDER BY bin ASC; +`; } _buildQuery (psql, override, callback) { @@ -204,6 +209,9 @@ module.exports = class DateHistogram extends BaseHistogram { return null; } + var interval = this._getAggregation(override) === 'quarter' ? + '3 months' : '1 ' + this._getAggregation(override); + const histogramSql = this._buildQueryTpl({ override: override, query: this.query, @@ -211,7 +219,8 @@ module.exports = class DateHistogram extends BaseHistogram { aggregation: this._getAggregation(override), start: this._getBinStart(override), end: this._getBinEnd(override), - offset: this._parseOffset(override) + offset: this._parseOffset(override), + interval: interval }); debug(histogramSql); @@ -275,6 +284,10 @@ module.exports = class DateHistogram extends BaseHistogram { } _getBuckets (result) { + result.rows.forEach(function(row) { + row.min = row.max = row.avg = row.timestamp; + }); + return result.rows.map(({ bin, min, max, avg, freq, timestamp }) => ({ bin, min, max, avg, freq, timestamp })); }