From 67331232754d45d6de713df3721dae7c10a5611b Mon Sep 17 00:00:00 2001 From: javier Date: Thu, 23 Apr 2026 00:59:39 +0200 Subject: [PATCH 01/14] Add SUBSAMPLE keyword documentation page New SQL reference page for the SUBSAMPLE clause covering LTTB, M4, and MinMax downsampling algorithms with SVG diagrams, runnable examples on fx_trades, gap-preserving LTTB, and configuration reference. --- .../configuration-utils/_cairo.config.json | 4 + documentation/query/sql/subsample.md | 354 ++++++++++++++++++ documentation/sidebars.js | 1 + scripts/gen_subsample_svgs.py | 263 +++++++++++++ src/css/_global.css | 5 + static/images/docs/subsample/lttb-gap.svg | 60 +++ static/images/docs/subsample/lttb.svg | 37 ++ static/images/docs/subsample/m4.svg | 41 ++ static/images/docs/subsample/minmax.svg | 42 +++ static/images/docs/subsample/raw.svg | 50 +++ 10 files changed, 857 insertions(+) create mode 100644 documentation/query/sql/subsample.md create mode 100644 scripts/gen_subsample_svgs.py create mode 100644 static/images/docs/subsample/lttb-gap.svg create mode 100644 static/images/docs/subsample/lttb.svg create mode 100644 static/images/docs/subsample/m4.svg create mode 100644 static/images/docs/subsample/minmax.svg create mode 100644 static/images/docs/subsample/raw.svg diff --git a/documentation/configuration/configuration-utils/_cairo.config.json b/documentation/configuration/configuration-utils/_cairo.config.json index bed859b73b..5b95a8393f 100644 --- a/documentation/configuration/configuration-utils/_cairo.config.json +++ b/documentation/configuration/configuration-utils/_cairo.config.json @@ -319,6 +319,10 @@ "default": "0", "description": "SampleBy default alignment behaviour. true corresponds to ALIGN TO CALENDAR, false corresponds to ALIGN TO FIRST OBSERVATION." }, + "cairo.sql.subsample.max.rows": { + "default": "100000000", + "description": "Maximum number of input rows SUBSAMPLE will buffer. Exceeding this limit returns an error. Must be between 1 and 2,147,483,647." + }, "cairo.date.locale": { "default": "en", "description": "The locale to handle date types." diff --git a/documentation/query/sql/subsample.md b/documentation/query/sql/subsample.md new file mode 100644 index 0000000000..706fcfafbf --- /dev/null +++ b/documentation/query/sql/subsample.md @@ -0,0 +1,354 @@ +--- +title: SUBSAMPLE keyword +sidebar_label: SUBSAMPLE +description: SUBSAMPLE SQL keyword reference for time-series downsampling using LTTB, M4, and MinMax algorithms. +--- + +`SUBSAMPLE` reduces the number of rows in a query result while preserving the +visual shape of the data. It selects the most representative points from a +time-ordered dataset, making it ideal for rendering charts at screen resolution +without transferring millions of rows to the client. + +Unlike [SAMPLE BY](/docs/query/sql/sample-by/), which computes new aggregate +values at synthetic bucket boundaries, `SUBSAMPLE` selects actual rows from +the input. Every output row exists in the source table with its original +timestamp and values. This means output timestamps match real rows (useful for +joins), and users can drill down to the exact source record behind any point +on a chart. + +Requires a [designated timestamp](/docs/concepts/designated-timestamp/) column. + +## Syntax + +```questdb-sql +SELECT columns +FROM table +[WHERE conditions] +[SAMPLE BY ...] +SUBSAMPLE { lttb | m4 | minmax }(valueColumn, targetPoints [, gapThreshold]) +[ORDER BY ...] +[LIMIT ...] +``` + +Where: + +- **`valueColumn`** - the numeric column used to decide which points are + visually significant. All other columns pass through for selected rows. +- **`targetPoints`** - target number of output rows. Supports integer + literals, [DECLARE](/docs/query/sql/declare/) variables, and bind + variables (`$1`). Must be at least 2. Maximum is 2,147,483,647. +- **`gapThreshold`** - (LTTB only) optional interval that enables + gap-preserving mode. See [gap-preserving LTTB](#gap-preserving-lttb). + +### Execution order + +`SUBSAMPLE` runs after `SAMPLE BY`, `GROUP BY`, and window functions, but +before `ORDER BY` and `LIMIT`. All value computations are complete before +downsampling decides which rows to keep. `SUBSAMPLE` only selects rows - it +never modifies computed values. + +All three algorithms execute serially. `SUBSAMPLE` buffers its entire input, +runs the selected algorithm, then emits the chosen rows. It does not block +upstream parallel execution - for example, a parallel `SAMPLE BY` completes +before `SUBSAMPLE` buffers its output. + +### Supported value types + +The value column must be a numeric type: `DOUBLE`, `FLOAT`, `INT`, `LONG`, +`SHORT`, or `BYTE`. `NULL` values in the value column are skipped during +downsampling. + +## Algorithms + +Three algorithms are available. Each one selects real rows from the input - +no values are ever interpolated or computed. The diagrams below all use the +same 24-point series as input (think 24 hourly bars over one day): + +![Raw time series](/images/docs/subsample/raw.svg) + +### lttb - Largest Triangle Three Buckets + +Divides the data into equal-sized row-count buckets and selects the point in +each bucket that forms the largest triangle with its neighbors. The first and +last points are always kept. Output is exactly N points. + +Best for line charts where preserving the visual shape (spikes, valleys, +trend changes) matters most. + +![LTTB downsampling](/images/docs/subsample/lttb.svg) + +How it works: + +1. First and last points are always selected. +2. Remaining data is divided into N-2 equal-sized buckets by row count. +3. For each bucket, the point creating the largest triangle area with the + previously selected point and the average of the next bucket is chosen. +4. Output preserves the original timestamp order. + +```questdb-sql title="Aggregate to hourly bars, then pick the 8 most representative" demo +SELECT timestamp, avg(price) avg_price +FROM fx_trades +WHERE symbol = 'EURUSD' + AND timestamp IN '$today' +SAMPLE BY 1h +SUBSAMPLE lttb(avg_price, 8) +``` + +### m4 - Min/Max/First/Last per time interval + +Divides the time range into equal time intervals and selects up to 4 points +per interval: the first, last, minimum, and maximum values. Empty intervals +produce no output, naturally preserving data gaps. + +Best for monitoring dashboards where you must not miss spikes or drops. The +min/max envelope is pixel-accurate to the full dataset. + +![M4 downsampling](/images/docs/subsample/m4.svg) + +How it works: + +1. The total time range is divided into N/4 equal time intervals. +2. For each interval, up to 4 points are selected: first, last, min, max. +3. When multiple roles resolve to the same physical row (e.g., the minimum + value is also the first row), duplicates are removed. A bucket emits + between 1 and 4 rows depending on the data. +4. Empty intervals produce no output. + +Output is up to N points (N/4 buckets, up to 4 points each). In the diagram +above, target 8 creates 2 time buckets. The first row happens to also be +the minimum in bucket 1, so each bucket emits 3-4 distinct rows instead of 4, +giving 7 total. + +```questdb-sql title="Hourly bars reduced to 8 with M4 - spike and trough guaranteed" demo +SELECT timestamp, avg(price) avg_price +FROM fx_trades +WHERE symbol = 'EURUSD' + AND timestamp IN '$today' +SAMPLE BY 1h +SUBSAMPLE m4(avg_price, 8) +``` + +:::tip + +When sizing `targetPoints` for a pixel-wide chart, remember that N/4 gives +the number of time buckets. A 1920-pixel-wide chart needs +`SUBSAMPLE m4(col, 1920)` to get 480 time buckets with up to 4 points each. + +::: + +### minmax - Min/Max per time interval + +Divides the time range into equal time intervals and selects up to 2 points +per interval: the minimum and maximum values. Lighter than M4 (no first/last +tracking), producing roughly half the output. Empty intervals produce no +output. + +Best for simple envelope visualization where you only need the value range +per bucket, not entry/exit points. + +![MinMax downsampling](/images/docs/subsample/minmax.svg) + +How it works: + +1. The total time range is divided into N/2 equal time intervals. +2. For each interval, up to 2 points are selected: min, max. +3. Duplicate points are removed (if min and max are the same row). +4. Empty intervals produce no output. + +Output is up to N points (N/2 buckets, up to 2 points each). + +```questdb-sql title="Hourly bars reduced to 8 with MinMax - min/max per bucket" demo +SELECT timestamp, avg(price) avg_price +FROM fx_trades +WHERE symbol = 'EURUSD' + AND timestamp IN '$today' +SAMPLE BY 1h +SUBSAMPLE minmax(avg_price, 8) +``` + +### Gap-preserving LTTB + +Standard LTTB divides data by row count, so it connects across time gaps. An +optional third parameter enables gap detection: + +```questdb-sql +SUBSAMPLE lttb(price, 12, '1h') +``` + +When specified, LTTB splits data into contiguous segments where consecutive +timestamps are within the gap threshold. Each segment is downsampled +independently with its proportional share of the target points. Gaps between +segments are preserved in the output. + +![LTTB gap handling comparison](/images/docs/subsample/lttb-gap.svg) + +Without gap detection, LTTB draws a straight line across the gap. With gap +detection enabled, each segment is downsampled independently and the gap is +visible in the output. + +Supported interval units: `s` (seconds), `m` (minutes), `h` (hours), +`d` (days). + +Examples: `'30s'`, `'5m'`, `'1h'`, `'7d'` + +```questdb-sql title="Preserve gaps larger than 1 hour in the output" demo +SELECT timestamp, price +FROM fx_trades +WHERE symbol = 'EURUSD' +SUBSAMPLE lttb(price, 12, '1h') +``` + +:::note + +Gap-preserving LTTB uses a soft target. Each segment receives at least its +first and last points. When many segments are detected, the total output may +exceed `targetPoints`. This is by design so that the same query does not fail +for one time range and succeed for another. Non-gap LTTB, M4, and MinMax +treat `targetPoints` as a hard maximum. + +::: + +### Algorithm comparison + +| Property | lttb | m4 | minmax | +|----------|------|----|--------| +| Bucket type | Equal row count | Equal time intervals | Equal time intervals | +| Points per bucket | Exactly 1 | Up to 4 (first, last, min, max) | Up to 2 (min, max) | +| Output count | Exactly N (non-gap mode) | Up to N | Up to N | +| Gap handling | Connects across gaps (use 3rd parameter to preserve) | Naturally preserves gaps | Naturally preserves gaps | +| Best use case | Line charts, shape preservation | Monitoring, spike detection | Lightweight envelope | + +## Examples + +### Chart-ready downsampling + +```questdb-sql title="LTTB: 500 representative points for a line chart" demo +SELECT timestamp, price +FROM fx_trades +WHERE symbol = 'EURUSD' +SUBSAMPLE lttb(price, 500) +``` + +```questdb-sql title="M4: pixel-accurate envelope for a 1920px-wide chart" demo +SELECT timestamp, price +FROM fx_trades +WHERE symbol = 'EURUSD' +SUBSAMPLE m4(price, 1920) +``` + +```questdb-sql title="MinMax: lightweight envelope at half the output of M4" demo +SELECT timestamp, price +FROM fx_trades +WHERE symbol = 'EURUSD' +SUBSAMPLE minmax(price, 500) +``` + +### Composing with SAMPLE BY + +```questdb-sql title="Aggregate to 1-minute bars, then downsample" demo +SELECT timestamp, avg(price) avg_price +FROM fx_trades +WHERE symbol = 'EURUSD' +SAMPLE BY 1m +SUBSAMPLE lttb(avg_price, 500) +``` + +`SAMPLE BY` computes aggregate values at bucket boundaries. `SUBSAMPLE` then +selects the most representative rows from that output. The two operations +complement each other: aggregate first, then reduce for display. + +### Multiple columns pass through + +```questdb-sql title="LTTB selects rows by price; all columns emit" demo +SELECT timestamp, symbol, side, price, quantity +FROM fx_trades +WHERE symbol = 'GBPUSD' +SUBSAMPLE lttb(price, 500) +``` + +### After window functions + +```questdb-sql title="Window functions see all rows before SUBSAMPLE selects" demo +SELECT timestamp, price, + avg(price) OVER (ROWS 10 PRECEDING) ma +FROM fx_trades +WHERE symbol = 'EURUSD' +SUBSAMPLE lttb(price, 500) +``` + +Window functions compute on the full dataset. `SUBSAMPLE` then selects from +the result, so the moving average values are accurate. + +### With DECLARE variable + +```questdb-sql title="Parameterized target point count" +DECLARE @points := 500 +SELECT timestamp, price +FROM fx_trades +WHERE symbol = 'EURUSD' +SUBSAMPLE lttb(price, @points) +``` + +### With bind variable + +```questdb-sql title="Grafana integration - screen width as bind variable" +SELECT timestamp, price +FROM fx_trades +WHERE symbol = 'EURUSD' +SUBSAMPLE lttb(price, $1) +``` + +### With ORDER BY and LIMIT + +```questdb-sql title="Downsample, then sort by price" demo +SELECT timestamp, price +FROM fx_trades +WHERE symbol = 'EURUSD' +SUBSAMPLE lttb(price, 100) +ORDER BY price DESC +LIMIT 10 +``` + +### Inside subqueries + +```questdb-sql title="SUBSAMPLE works inside parenthesized subqueries" demo +SELECT count() FROM ( + SELECT timestamp, price + FROM fx_trades + WHERE symbol = 'EURUSD' + SUBSAMPLE lttb(price, 500) +) +``` + +## Behavior notes + +- If the input has fewer rows than the target, all rows are returned unchanged. +- Output rows are always in timestamp-ascending order. +- All columns from the `SELECT` clause pass through for selected rows. +- `SUBSAMPLE` works with `WHERE`, `SAMPLE BY`, `GROUP BY`, CTEs, subqueries, + `ORDER BY`, and `LIMIT`. +- `SUBSAMPLE` inside a parenthesized subquery applies inside that subquery, + not the outer query. + +## Configuration + +| Property | Default | Description | +|----------|---------|-------------| +| `cairo.sql.subsample.max.rows` | 100,000,000 | Maximum input rows SUBSAMPLE will buffer. Exceeding this limit returns an error. | + +`SUBSAMPLE` buffers its entire input before running the algorithm. For direct +table scans, memory usage is 24 bytes per row. For queries involving +`SAMPLE BY`, `GROUP BY`, or subqueries, memory also scales with the projected +row width. At the default limit, the base buffer is approximately 2.4 GB. + +## See also + +- [SAMPLE BY](/docs/query/sql/sample-by/) - time-based aggregation + (computes new values at bucket boundaries, while `SUBSAMPLE` selects + existing rows) +- [Designated timestamp](/docs/concepts/designated-timestamp/) - required + for `SUBSAMPLE` to operate +- [Steinarsson, S. (2013). "Downsampling Time Series for Visual Representation"](https://github.com/sveinn-steinarsson/flot-downsample) - + the original LTTB algorithm and thesis reference +- [Jugel, U. et al. (2014). "M4: A Visualization-Oriented Time Series Data Aggregation"](https://www.vldb.org/pvldb/vol7/p797-jugel.pdf) - + the M4 paper diff --git a/documentation/sidebars.js b/documentation/sidebars.js index bf61c76207..ed60e5dc83 100644 --- a/documentation/sidebars.js +++ b/documentation/sidebars.js @@ -427,6 +427,7 @@ module.exports = { "query/sql/order-by", "query/sql/pivot", "query/sql/sample-by", + "query/sql/subsample", "query/sql/unnest", "query/sql/where", "query/sql/window-join", diff --git a/scripts/gen_subsample_svgs.py b/scripts/gen_subsample_svgs.py new file mode 100644 index 0000000000..0df09e9dcd --- /dev/null +++ b/scripts/gen_subsample_svgs.py @@ -0,0 +1,263 @@ +"""Generate SVG diagrams for the SUBSAMPLE documentation page. + +Uses @media (prefers-color-scheme) for light/dark theme support since SVGs +loaded via tags don't inherit CSS from the parent document. +ViewBox width is ~600 to match typical content width so 1 unit ~ 1px. +""" + +import os + +OUT_DIR = os.path.join(os.path.dirname(__file__), "..", "static", "images", "docs", "subsample") + +# QuestDB palette +PINK = "#e289a4" # algorithm lines +CYAN = "#0cc0df" # titles, M4/MinMax min/max role dots +GRAY = "#888" # default dots (real rows from raw data) + +# Segment A: 24 points, i=0..23 (represents 24 hourly bars) +SEG_A = [ + 0.50, 0.55, 0.60, 0.65, 0.70, 0.95, 0.85, 0.70, 0.60, 0.55, + 0.50, 0.45, 0.40, 0.35, 0.28, 0.20, 0.25, 0.30, 0.35, 0.40, + 0.45, 0.50, 0.48, 0.46, +] + +SEG_B_START = 48 +SEG_B = [ + 0.45, 0.50, 0.55, 0.58, 0.60, 0.65, 0.70, 0.75, 0.70, 0.55, + 0.40, 0.25, 0.15, 0.25, 0.40, 0.55, 0.60, 0.62, 0.60, 0.58, 0.55, +] + +# Single panel layout (viewBox units - keep at 600 for good proportions) +W = 600 +H = 300 +XL, XR = 10, 590 # plot x range - use full width +YT, YB = 60, 240 # plot y range (180px tall) +LY = 275 # legend baseline + +# Gap SVG layout (two panels) +GH = 530 +G1T, G1B = 60, 210 # panel 1 plot area +G2T, G2B = 300, 450 # panel 2 plot area +GLY = 510 + +# Intrinsic pixel width - set larger than container so max-width:100% fills it +PX_W = 1400 + +# Sizes (viewBox units - rendered ~1.3x on screen) +TITLE_SZ = 12 +LEG_SZ = 11 +REF_SW = 1.0 +ALGO_SW = 2.0 +DOT_R = 4.5 +LEG_DOT = 3.5 +BK_SW = 0.8 + +STYLE = f"""""" + + +def xp(i, imin, imax): + if imax == imin: + return (XL + XR) / 2 + return XL + (i - imin) / (imax - imin) * (XR - XL) + + +def yp(v, yt, yb): + return yt + (1 - v) * (yb - yt) + + +def pl(ii, vv, imin, imax, yt, yb): + return " ".join(f"{xp(i,imin,imax):.1f},{yp(v,yt,yb):.1f}" for i, v in zip(ii, vv)) + + +def cd(ii, vv, imin, imax, yt, yb, fill): + return "\n".join( + f'' + for i, v in zip(ii, vv)) + + +def cdm(pcs, imin, imax, yt, yb): + return "\n".join( + f'' + for i, v, c in pcs) + + +def rpl(ii, vv, imin, imax, yt, yb): + return f'' + + +def bkl(bounds, imin, imax, yt, yb): + return "\n".join( + f'' + for b in bounds) + + +def hdr(w, h, title, desc): + px_h = int(h * PX_W / w) + return (f'\n' + f'{title}\n{desc}\n{STYLE}') + + +def gen_raw(): + """Raw data panel - 24 hourly bars.""" + N = len(SEG_A) + im, ix = 0, N - 1 + ri = list(range(N)) + h = 260 + yt, yb = 55, 200 + ly = 240 + raw_color = "#888" + raw_dots = "\n".join( + f'' + for i, v in zip(ri, SEG_A) + ) + + return f"""{hdr(W, h, "Raw time series", "24 hourly data points with a spike and a trough.")} +Raw time series: 24 hourly bars + +{raw_dots} + +Hourly bars (24) +""" + + +def gen_lttb(): + N = len(SEG_A) + im, ix = 0, N - 1 + ri = list(range(N)) + # LTTB target 8: first + last always kept, 6 interior buckets + si = [0, 4, 5, 8, 15, 19, 22, 23] + sv = [SEG_A[i] for i in si] + return f"""{hdr(W, H, "LTTB downsampling", "LTTB selects 8 points from 24.")} +LTTB: 24 hourly bars reduced to 8 +{rpl(ri, SEG_A, im, ix, YT, YB)} + +{cd(si,sv,im,ix,YT,YB,GRAY)} + +Raw data + +Selected points (8 of 24) +""" + + +def gen_m4(): + N = len(SEG_A) + im, ix = 0, N - 1 + ri = list(range(N)) + # M4 target 8 -> 2 time buckets (0..11, 12..23) + # Bucket 1: first=0(.50), last=11(.45), min=0(.50)->dup, max=5(.95) -> 3 pts + # Bucket 2: first=12(.40), last=23(.46), min=15(.20), max=21(.50) -> 4 pts + m4 = [ + (0,.50,GRAY),(5,.95,CYAN),(11,.45,GRAY), + (12,.40,GRAY),(15,.20,CYAN),(21,.50,CYAN),(23,.46,GRAY), + ] + mi = [p[0] for p in m4] + mv = [p[1] for p in m4] + return f"""{hdr(W, H, "M4 downsampling", "M4 selects 7 points from 24.")} +M4: target 8, emitted 7 (2 time buckets) +{bkl([12], im, ix, YT, YB)} +{rpl(ri, SEG_A, im, ix, YT, YB)} + +{cdm(m4, im, ix, YT, YB)} + +Raw data + +First / Last + +Min / Max + +Bucket boundary +""" + + +def gen_minmax(): + N = len(SEG_A) + im, ix = 0, N - 1 + ri = list(range(N)) + # MinMax target 8 -> 4 time buckets of 6 (0..5, 6..11, 12..17, 18..23) + # Bucket 1: min=0(.50), max=5(.95) + # Bucket 2: min=11(.45), max=6(.85) + # Bucket 3: min=15(.20), max=12(.40) + # Bucket 4: min=23(.46), max=21(.50) + mi = [0, 5, 6, 11, 12, 15, 21, 23] + mv = [.50, .95, .85, .45, .40, .20, .50, .46] + return f"""{hdr(W, H, "MinMax downsampling", "MinMax selects 8 points from 24.")} +MinMax: target 8, emitted 8 (4 time buckets) +{bkl([6, 12, 18], im, ix, YT, YB)} +{rpl(ri, SEG_A, im, ix, YT, YB)} + +{cd(mi,mv,im,ix,YT,YB,GRAY)} + +Raw data + +Selected points (8 of 24) + +Bucket boundary +""" + + +def gen_lttb_gap(): + N_A = len(SEG_A) + N_B = len(SEG_B) + im, ix = 0, SEG_B_START + N_B - 1 # 0..68 + rai = list(range(N_A)) + rbi = list(range(SEG_B_START, SEG_B_START + N_B)) + # LTTB no gap, target 12 on 45 total points + li = [0,4,5,12,15,23,51,55,59,60,65,68] + lv = [.50,.70,.95,.40,.20,.46,.25,.55,.60,.62,.58,.55] + # LTTB with gap detection, 6 per segment + g1i = [0,4,5,15,19,23] + g1v = [.50,.70,.95,.20,.40,.46] + g2i = [48,53,55,60,65,68] + g2v = [.45,.65,.75,.15,.62,.55] + + def refs(yt, yb): + return f"{rpl(rai, SEG_A, im, ix, yt, yb)}\n{rpl(rbi, SEG_B, im, ix, yt, yb)}" + + total = N_A + N_B + return f"""{hdr(W, GH, "LTTB gap handling", "Comparing LTTB with and without gap detection.")} +LTTB without gap detection: line connects across the gap +{refs(G1T, G1B)} + +{cd(li,lv,im,ix,G1T,G1B,GRAY)} + +LTTB with gap detection: each segment downsampled +{refs(G2T, G2B)} + + +{cd(g1i,g1v,im,ix,G2T,G2B,GRAY)} +{cd(g2i,g2v,im,ix,G2T,G2B,GRAY)} + +Raw data ({total} points with gap) + +Selected points (12) +""" + + +if __name__ == "__main__": + os.makedirs(OUT_DIR, exist_ok=True) + for name, fn in [("raw.svg", gen_raw), ("lttb.svg", gen_lttb), ("m4.svg", gen_m4), + ("minmax.svg", gen_minmax), ("lttb-gap.svg", gen_lttb_gap)]: + path = os.path.join(OUT_DIR, name) + with open(path, "w") as f: + f.write(fn()) + print(f"Wrote {path}") diff --git a/src/css/_global.css b/src/css/_global.css index 0f5ce0d7c4..172bebcc92 100644 --- a/src/css/_global.css +++ b/src/css/_global.css @@ -485,3 +485,8 @@ html[data-theme="dark"] .DocSearch { font-family: SegoeUI, -apple-system, BlinkMacSystemFont, Ubuntu, sans-serif; font-size: var(--font-size-small); } + +/* Make doc article images fill the content width */ +article img { + width: 100%; +} diff --git a/static/images/docs/subsample/lttb-gap.svg b/static/images/docs/subsample/lttb-gap.svg new file mode 100644 index 0000000000..e45cac3317 --- /dev/null +++ b/static/images/docs/subsample/lttb-gap.svg @@ -0,0 +1,60 @@ + +LTTB gap handling +Comparing LTTB with and without gap detection. + +LTTB without gap detection: line connects across the gap + + + + + + + + + + + + + + + + +LTTB with gap detection: each segment downsampled + + + + + + + + + + + + + + + + + +Raw data (45 points with gap) + +Selected points (12) + \ No newline at end of file diff --git a/static/images/docs/subsample/lttb.svg b/static/images/docs/subsample/lttb.svg new file mode 100644 index 0000000000..400c3fc3ec --- /dev/null +++ b/static/images/docs/subsample/lttb.svg @@ -0,0 +1,37 @@ + +LTTB downsampling +LTTB selects 8 points from 24. + +LTTB: 24 hourly bars reduced to 8 + + + + + + + + + + + +Raw data + +Selected points (8 of 24) + \ No newline at end of file diff --git a/static/images/docs/subsample/m4.svg b/static/images/docs/subsample/m4.svg new file mode 100644 index 0000000000..959e254988 --- /dev/null +++ b/static/images/docs/subsample/m4.svg @@ -0,0 +1,41 @@ + +M4 downsampling +M4 selects 7 points from 24. + +M4: target 8, emitted 7 (2 time buckets) + + + + + + + + + + + +Raw data + +First / Last + +Min / Max + +Bucket boundary + \ No newline at end of file diff --git a/static/images/docs/subsample/minmax.svg b/static/images/docs/subsample/minmax.svg new file mode 100644 index 0000000000..45aa7ad77b --- /dev/null +++ b/static/images/docs/subsample/minmax.svg @@ -0,0 +1,42 @@ + +MinMax downsampling +MinMax selects 8 points from 24. + +MinMax: target 8, emitted 8 (4 time buckets) + + + + + + + + + + + + + + +Raw data + +Selected points (8 of 24) + +Bucket boundary + \ No newline at end of file diff --git a/static/images/docs/subsample/raw.svg b/static/images/docs/subsample/raw.svg new file mode 100644 index 0000000000..31c108a40e --- /dev/null +++ b/static/images/docs/subsample/raw.svg @@ -0,0 +1,50 @@ + +Raw time series +24 hourly data points with a spike and a trough. + +Raw time series: 24 hourly bars + + + + + + + + + + + + + + + + + + + + + + + + + + +Hourly bars (24) + \ No newline at end of file From e1187bc8c2d9c2e270d2f24e03fe98b23ca7b59e Mon Sep 17 00:00:00 2001 From: javier Date: Thu, 23 Apr 2026 01:04:24 +0200 Subject: [PATCH 02/14] Scope full-width CSS rule to subsample diagrams only Avoid stretching all doc images by targeting only img[src*="/subsample/"] instead of article img. --- src/css/_global.css | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/css/_global.css b/src/css/_global.css index 172bebcc92..c82e2a1f9f 100644 --- a/src/css/_global.css +++ b/src/css/_global.css @@ -486,7 +486,7 @@ html[data-theme="dark"] .DocSearch { font-size: var(--font-size-small); } -/* Make doc article images fill the content width */ -article img { +/* Make subsample diagram SVGs fill the content width */ +article img[src*="/subsample/"] { width: 100%; } From 33de85028e7f9cf367d05f0f2a61b6df285f1259 Mon Sep 17 00:00:00 2001 From: javier Date: Thu, 23 Apr 2026 01:20:23 +0200 Subject: [PATCH 03/14] Improve SUBSAMPLE diagrams and gap-preserving LTTB section Split gap SVG into three separate charts with individual legends. Add small gap to dataset to illustrate threshold behavior. Use QuestDB color palette (pink lines, cyan titles, gray dots). Match inline SQL examples to chart storytelling (24 hourly bars). Add boundary markers distinguishing gaps from bucket boundaries. --- documentation/query/sql/subsample.md | 35 ++-- scripts/gen_subsample_svgs.py | 158 +++++++++++++----- static/images/docs/subsample/gap-detect.svg | 47 ++++++ .../images/docs/subsample/gap-no-detect.svg | 43 +++++ static/images/docs/subsample/gap-raw.svg | 74 ++++++++ static/images/docs/subsample/lttb-gap.svg | 60 ------- static/images/docs/subsample/lttb.svg | 6 +- static/images/docs/subsample/m4.svg | 6 +- static/images/docs/subsample/minmax.svg | 6 +- static/images/docs/subsample/raw.svg | 6 +- 10 files changed, 311 insertions(+), 130 deletions(-) create mode 100644 static/images/docs/subsample/gap-detect.svg create mode 100644 static/images/docs/subsample/gap-no-detect.svg create mode 100644 static/images/docs/subsample/gap-raw.svg delete mode 100644 static/images/docs/subsample/lttb-gap.svg diff --git a/documentation/query/sql/subsample.md b/documentation/query/sql/subsample.md index 706fcfafbf..9744e021b2 100644 --- a/documentation/query/sql/subsample.md +++ b/documentation/query/sql/subsample.md @@ -169,33 +169,44 @@ SUBSAMPLE minmax(avg_price, 8) ### Gap-preserving LTTB Standard LTTB divides data by row count, so it connects across time gaps. An -optional third parameter enables gap detection: +optional third parameter sets a gap threshold: ```questdb-sql -SUBSAMPLE lttb(price, 12, '1h') +SUBSAMPLE lttb(price, 12, '6h') ``` -When specified, LTTB splits data into contiguous segments where consecutive -timestamps are within the gap threshold. Each segment is downsampled -independently with its proportional share of the target points. Gaps between -segments are preserved in the output. +When specified, LTTB scans for gaps where consecutive timestamps are further +apart than the threshold. Gaps below the threshold are ignored - the data is +treated as continuous. Gaps above the threshold split the data into separate +segments, each downsampled independently with its proportional share of the +target points. -![LTTB gap handling comparison](/images/docs/subsample/lttb-gap.svg) +The diagrams below show a dataset with two gaps - a small one (3 hours) and +a large one (24 hours): -Without gap detection, LTTB draws a straight line across the gap. With gap -detection enabled, each segment is downsampled independently and the gap is -visible in the output. +![Raw data with gaps](/images/docs/subsample/gap-raw.svg) + +Without gap detection, LTTB treats all points as continuous and connects +across both gaps: + +![LTTB without gap detection](/images/docs/subsample/gap-no-detect.svg) + +With a threshold of `'6h'`, the small gap (3h) is below the threshold so +segments A and B are treated as continuous. The large gap (24h) exceeds the +threshold, so segment C is downsampled separately and the gap is preserved: + +![LTTB with gap detection](/images/docs/subsample/gap-detect.svg) Supported interval units: `s` (seconds), `m` (minutes), `h` (hours), `d` (days). Examples: `'30s'`, `'5m'`, `'1h'`, `'7d'` -```questdb-sql title="Preserve gaps larger than 1 hour in the output" demo +```questdb-sql title="Preserve gaps larger than 6 hours in the output" demo SELECT timestamp, price FROM fx_trades WHERE symbol = 'EURUSD' -SUBSAMPLE lttb(price, 12, '1h') +SUBSAMPLE lttb(price, 12, '6h') ``` :::note diff --git a/scripts/gen_subsample_svgs.py b/scripts/gen_subsample_svgs.py index 0df09e9dcd..bd4ef4af55 100644 --- a/scripts/gen_subsample_svgs.py +++ b/scripts/gen_subsample_svgs.py @@ -21,8 +21,16 @@ 0.45, 0.50, 0.48, 0.46, ] -SEG_B_START = 48 -SEG_B = [ +# Gap dataset: 3 data segments, 1 small gap (3h), 1 big gap (24h) +# Seg A: i=0..10, Seg B: i=14..23 (small gap 11-13), Seg C: i=48..68 (big gap 24-47) +GAP_SEG_A_I = list(range(0, 11)) +GAP_SEG_A_V = [0.50, 0.55, 0.60, 0.65, 0.70, 0.95, 0.85, 0.70, 0.60, 0.55, 0.50] + +GAP_SEG_B_I = list(range(14, 24)) +GAP_SEG_B_V = [0.42, 0.38, 0.35, 0.28, 0.20, 0.25, 0.30, 0.35, 0.40, 0.45] + +GAP_SEG_C_I = list(range(48, 69)) +GAP_SEG_C_V = [ 0.45, 0.50, 0.55, 0.58, 0.60, 0.65, 0.70, 0.75, 0.70, 0.55, 0.40, 0.25, 0.15, 0.25, 0.40, 0.55, 0.60, 0.62, 0.60, 0.58, 0.55, ] @@ -34,11 +42,12 @@ YT, YB = 60, 240 # plot y range (180px tall) LY = 275 # legend baseline -# Gap SVG layout (two panels) -GH = 530 -G1T, G1B = 60, 210 # panel 1 plot area -G2T, G2B = 300, 450 # panel 2 plot area -GLY = 510 +# Gap SVG layout (three panels: raw, no-gap LTTB, gap LTTB) +GH = 600 +G0T, G0B = 50, 140 # panel 0: raw data with gap +G1T, G1B = 200, 310 # panel 1: LTTB without gap detection +G2T, G2B = 370, 480 # panel 2: LTTB with gap detection +GLY = 520 # legend # Intrinsic pixel width - set larger than container so max-width:100% fills it PX_W = 1400 @@ -56,17 +65,17 @@ .t {{ font-size: {TITLE_SZ}px; font-weight: 600; }} .l {{ font-size: {LEG_SZ}px; }} .ref {{ stroke-width: {REF_SW}; stroke-dasharray: 6 5; fill: none; }} - .bk {{ stroke-width: {BK_SW}; stroke-dasharray: 6 5; }} + .bk {{ stroke-width: {BK_SW}; stroke-dasharray: 2 3; }} .t {{ fill: {CYAN}; }} .l {{ fill: #64748b; }} .ref {{ stroke: #bbb; }} - .bk {{ stroke: #aaa; }} + .bk {{ stroke: #5a9aa8; }} .sep {{ stroke: #ccc; }} @media (prefers-color-scheme: dark) {{ .t {{ fill: {CYAN}; }} .l {{ fill: #b1b5d3; }} .ref {{ stroke: #555; }} - .bk {{ stroke: #4a4a4a; }} + .bk {{ stroke: #2a7a8a; }} .sep {{ stroke: #3a3a3a; }} }} """ @@ -215,48 +224,105 @@ def gen_minmax(): """ -def gen_lttb_gap(): - N_A = len(SEG_A) - N_B = len(SEG_B) - im, ix = 0, SEG_B_START + N_B - 1 # 0..68 - rai = list(range(N_A)) - rbi = list(range(SEG_B_START, SEG_B_START + N_B)) - # LTTB no gap, target 12 on 45 total points - li = [0,4,5,12,15,23,51,55,59,60,65,68] - lv = [.50,.70,.95,.40,.20,.46,.25,.55,.60,.62,.58,.55] - # LTTB with gap detection, 6 per segment - g1i = [0,4,5,15,19,23] - g1v = [.50,.70,.95,.20,.40,.46] - g2i = [48,53,55,60,65,68] - g2v = [.45,.65,.75,.15,.62,.55] - - def refs(yt, yb): - return f"{rpl(rai, SEG_A, im, ix, yt, yb)}\n{rpl(rbi, SEG_B, im, ix, yt, yb)}" - - total = N_A + N_B - return f"""{hdr(W, GH, "LTTB gap handling", "Comparing LTTB with and without gap detection.")} -LTTB without gap detection: line connects across the gap -{refs(G1T, G1B)} - -{cd(li,lv,im,ix,G1T,G1B,GRAY)} - -LTTB with gap detection: each segment downsampled -{refs(G2T, G2B)} - - -{cd(g1i,g1v,im,ix,G2T,G2B,GRAY)} -{cd(g2i,g2v,im,ix,G2T,G2B,GRAY)} - -Raw data ({total} points with gap) - -Selected points (12) +def _gap_helpers(): + """Shared helpers for the three gap SVGs.""" + im, ix = 0, 68 + raw_color = "#888" + small_gap_mid = 12 + big_gap_mid = 35.5 + + def raw_pls(yt, yb): + return (f"{rpl(GAP_SEG_A_I, GAP_SEG_A_V, im, ix, yt, yb)}\n" + f"{rpl(GAP_SEG_B_I, GAP_SEG_B_V, im, ix, yt, yb)}\n" + f"{rpl(GAP_SEG_C_I, GAP_SEG_C_V, im, ix, yt, yb)}") + + def raw_dots_str(yt, yb): + parts = [] + for si, sv in [(GAP_SEG_A_I, GAP_SEG_A_V), + (GAP_SEG_B_I, GAP_SEG_B_V), + (GAP_SEG_C_I, GAP_SEG_C_V)]: + parts.extend( + f'' + for i, v in zip(si, sv)) + return "\n".join(parts) + + def raw_lines_str(yt, yb): + return ( + f'\n' + f'\n' + f'') + + return im, ix, raw_color, small_gap_mid, big_gap_mid, raw_pls, raw_dots_str, raw_lines_str + + +def gen_gap_raw(): + """Raw data with gaps - shows where the gaps are.""" + im, ix, raw_color, sg, bg, _, raw_dots_str, raw_lines_str = _gap_helpers() + total = len(GAP_SEG_A_V) + len(GAP_SEG_B_V) + len(GAP_SEG_C_V) + return f"""{hdr(W, H, "Raw data with gaps", "42 points with a small and large gap.")} +Raw data: {total} points, small gap (3h) and large gap (24h) +{bkl([sg, bg], im, ix, YT, YB)} +{raw_lines_str(YT, YB)} +{raw_dots_str(YT, YB)} + +Data points ({total}) + +Gap boundary +""" + + +def gen_gap_no_detect(): + """LTTB without gap detection - connects across all gaps.""" + im, ix, _, sg, bg, raw_pls, _, _ = _gap_helpers() + ng_i = [0, 4, 5, 10, 18, 23, 51, 55, 60, 64, 67, 68] + ng_v = [.50, .70, .95, .50, .20, .45, .55, .75, .15, .55, .60, .55] + return f"""{hdr(W, H, "LTTB without gap detection", "LTTB connects across all gaps.")} +LTTB without gap detection: connects across all gaps +{raw_pls(YT, YB)} + +{cd(ng_i,ng_v,im,ix,YT,YB,GRAY)} + +Raw data + +Selected points (12 of {len(GAP_SEG_A_V)+len(GAP_SEG_B_V)+len(GAP_SEG_C_V)}) +""" + + +def gen_gap_detect(): + """LTTB with gap detection - small gap connected, large gap preserved.""" + im, ix, _, sg, bg, raw_pls, _, _ = _gap_helpers() + g_ab_i = [0, 5, 10, 18, 22, 23] + g_ab_v = [.50, .95, .50, .20, .40, .45] + g_c_i = [48, 55, 58, 60, 65, 68] + g_c_v = [.45, .75, .55, .15, .60, .55] + return f"""{hdr(W, H, "LTTB with gap detection", "Small gap connected, large gap preserved.")} +LTTB with gap threshold '6h': small gap connected, large gap preserved +{bkl([bg], im, ix, YT, YB)} +{raw_pls(YT, YB)} + + +{cd(g_ab_i,g_ab_v,im,ix,YT,YB,GRAY)} +{cd(g_c_i,g_c_v,im,ix,YT,YB,GRAY)} + +Raw data + +Selected points (12) + +Gap boundary """ if __name__ == "__main__": os.makedirs(OUT_DIR, exist_ok=True) for name, fn in [("raw.svg", gen_raw), ("lttb.svg", gen_lttb), ("m4.svg", gen_m4), - ("minmax.svg", gen_minmax), ("lttb-gap.svg", gen_lttb_gap)]: + ("minmax.svg", gen_minmax), + ("gap-raw.svg", gen_gap_raw), + ("gap-no-detect.svg", gen_gap_no_detect), + ("gap-detect.svg", gen_gap_detect)]: path = os.path.join(OUT_DIR, name) with open(path, "w") as f: f.write(fn()) diff --git a/static/images/docs/subsample/gap-detect.svg b/static/images/docs/subsample/gap-detect.svg new file mode 100644 index 0000000000..8afadec025 --- /dev/null +++ b/static/images/docs/subsample/gap-detect.svg @@ -0,0 +1,47 @@ + +LTTB with gap detection +Small gap connected, large gap preserved. + +LTTB with gap threshold '6h': small gap connected, large gap preserved + + + + + + + + + + + + + + + + + + + +Raw data + +Selected points (12) + +Gap boundary + \ No newline at end of file diff --git a/static/images/docs/subsample/gap-no-detect.svg b/static/images/docs/subsample/gap-no-detect.svg new file mode 100644 index 0000000000..9d9c03e231 --- /dev/null +++ b/static/images/docs/subsample/gap-no-detect.svg @@ -0,0 +1,43 @@ + +LTTB without gap detection +LTTB connects across all gaps. + +LTTB without gap detection: connects across all gaps + + + + + + + + + + + + + + + + + +Raw data + +Selected points (12 of 42) + \ No newline at end of file diff --git a/static/images/docs/subsample/gap-raw.svg b/static/images/docs/subsample/gap-raw.svg new file mode 100644 index 0000000000..26bf617a6b --- /dev/null +++ b/static/images/docs/subsample/gap-raw.svg @@ -0,0 +1,74 @@ + +Raw data with gaps +42 points with a small and large gap. + +Raw data: 42 points, small gap (3h) and large gap (24h) + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +Data points (42) + +Gap boundary + \ No newline at end of file diff --git a/static/images/docs/subsample/lttb-gap.svg b/static/images/docs/subsample/lttb-gap.svg deleted file mode 100644 index e45cac3317..0000000000 --- a/static/images/docs/subsample/lttb-gap.svg +++ /dev/null @@ -1,60 +0,0 @@ - -LTTB gap handling -Comparing LTTB with and without gap detection. - -LTTB without gap detection: line connects across the gap - - - - - - - - - - - - - - - - -LTTB with gap detection: each segment downsampled - - - - - - - - - - - - - - - - - -Raw data (45 points with gap) - -Selected points (12) - \ No newline at end of file diff --git a/static/images/docs/subsample/lttb.svg b/static/images/docs/subsample/lttb.svg index 400c3fc3ec..693ece8442 100644 --- a/static/images/docs/subsample/lttb.svg +++ b/static/images/docs/subsample/lttb.svg @@ -5,17 +5,17 @@ .t { font-size: 12px; font-weight: 600; } .l { font-size: 11px; } .ref { stroke-width: 1.0; stroke-dasharray: 6 5; fill: none; } - .bk { stroke-width: 0.8; stroke-dasharray: 6 5; } + .bk { stroke-width: 0.8; stroke-dasharray: 2 3; } .t { fill: #0cc0df; } .l { fill: #64748b; } .ref { stroke: #bbb; } - .bk { stroke: #aaa; } + .bk { stroke: #5a9aa8; } .sep { stroke: #ccc; } @media (prefers-color-scheme: dark) { .t { fill: #0cc0df; } .l { fill: #b1b5d3; } .ref { stroke: #555; } - .bk { stroke: #4a4a4a; } + .bk { stroke: #2a7a8a; } .sep { stroke: #3a3a3a; } } diff --git a/static/images/docs/subsample/m4.svg b/static/images/docs/subsample/m4.svg index 959e254988..9fcfaa6d99 100644 --- a/static/images/docs/subsample/m4.svg +++ b/static/images/docs/subsample/m4.svg @@ -5,17 +5,17 @@ .t { font-size: 12px; font-weight: 600; } .l { font-size: 11px; } .ref { stroke-width: 1.0; stroke-dasharray: 6 5; fill: none; } - .bk { stroke-width: 0.8; stroke-dasharray: 6 5; } + .bk { stroke-width: 0.8; stroke-dasharray: 2 3; } .t { fill: #0cc0df; } .l { fill: #64748b; } .ref { stroke: #bbb; } - .bk { stroke: #aaa; } + .bk { stroke: #5a9aa8; } .sep { stroke: #ccc; } @media (prefers-color-scheme: dark) { .t { fill: #0cc0df; } .l { fill: #b1b5d3; } .ref { stroke: #555; } - .bk { stroke: #4a4a4a; } + .bk { stroke: #2a7a8a; } .sep { stroke: #3a3a3a; } } diff --git a/static/images/docs/subsample/minmax.svg b/static/images/docs/subsample/minmax.svg index 45aa7ad77b..0ab22a13b7 100644 --- a/static/images/docs/subsample/minmax.svg +++ b/static/images/docs/subsample/minmax.svg @@ -5,17 +5,17 @@ .t { font-size: 12px; font-weight: 600; } .l { font-size: 11px; } .ref { stroke-width: 1.0; stroke-dasharray: 6 5; fill: none; } - .bk { stroke-width: 0.8; stroke-dasharray: 6 5; } + .bk { stroke-width: 0.8; stroke-dasharray: 2 3; } .t { fill: #0cc0df; } .l { fill: #64748b; } .ref { stroke: #bbb; } - .bk { stroke: #aaa; } + .bk { stroke: #5a9aa8; } .sep { stroke: #ccc; } @media (prefers-color-scheme: dark) { .t { fill: #0cc0df; } .l { fill: #b1b5d3; } .ref { stroke: #555; } - .bk { stroke: #4a4a4a; } + .bk { stroke: #2a7a8a; } .sep { stroke: #3a3a3a; } } diff --git a/static/images/docs/subsample/raw.svg b/static/images/docs/subsample/raw.svg index 31c108a40e..c9fe3c6f8c 100644 --- a/static/images/docs/subsample/raw.svg +++ b/static/images/docs/subsample/raw.svg @@ -5,17 +5,17 @@ .t { font-size: 12px; font-weight: 600; } .l { font-size: 11px; } .ref { stroke-width: 1.0; stroke-dasharray: 6 5; fill: none; } - .bk { stroke-width: 0.8; stroke-dasharray: 6 5; } + .bk { stroke-width: 0.8; stroke-dasharray: 2 3; } .t { fill: #0cc0df; } .l { fill: #64748b; } .ref { stroke: #bbb; } - .bk { stroke: #aaa; } + .bk { stroke: #5a9aa8; } .sep { stroke: #ccc; } @media (prefers-color-scheme: dark) { .t { fill: #0cc0df; } .l { fill: #b1b5d3; } .ref { stroke: #555; } - .bk { stroke: #4a4a4a; } + .bk { stroke: #2a7a8a; } .sep { stroke: #3a3a3a; } } From 6e724d4f4b12ead3f58a228af5f2ec4d1a838ee3 Mon Sep 17 00:00:00 2001 From: javier Date: Thu, 23 Apr 2026 09:56:13 +0200 Subject: [PATCH 04/14] Reorder MinMax before M4 and improve algorithm explanations Tweak dataset so M4 visibly outperforms MinMax (late spike with pullback). Explain envelope, triangle method, and first/last advantage for users unfamiliar with downsampling. --- documentation/query/sql/subsample.md | 97 +++++++++++++------------ scripts/gen_subsample_svgs.py | 27 ++++--- static/images/docs/subsample/lttb.svg | 12 +-- static/images/docs/subsample/m4.svg | 10 +-- static/images/docs/subsample/minmax.svg | 10 +-- static/images/docs/subsample/raw.svg | 8 +- 6 files changed, 86 insertions(+), 78 deletions(-) diff --git a/documentation/query/sql/subsample.md b/documentation/query/sql/subsample.md index 9744e021b2..a97e7c012c 100644 --- a/documentation/query/sql/subsample.md +++ b/documentation/query/sql/subsample.md @@ -69,11 +69,15 @@ same 24-point series as input (think 24 hourly bars over one day): ### lttb - Largest Triangle Three Buckets Divides the data into equal-sized row-count buckets and selects the point in -each bucket that forms the largest triangle with its neighbors. The first and +each bucket that forms the largest triangle with its neighbors. The idea is +that points where the line changes direction sharply (a spike, a valley, a +sudden trend shift) form large triangles and get kept, while points in the +middle of a smooth trend form small triangles and get dropped. The first and last points are always kept. Output is exactly N points. -Best for line charts where preserving the visual shape (spikes, valleys, -trend changes) matters most. +Best for line charts where the visual shape matters most - a chart drawn +from the LTTB output looks nearly identical to one drawn from the full +dataset, despite using far fewer points. ![LTTB downsampling](/images/docs/subsample/lttb.svg) @@ -94,14 +98,45 @@ SAMPLE BY 1h SUBSAMPLE lttb(avg_price, 8) ``` +### minmax - Min/Max per time interval + +Divides the time range into equal time intervals and selects up to 2 points +per interval: the row with the minimum value and the row with the maximum +value. This creates a visual envelope - at any point on the chart, you can +see the full range the data covered during that interval. No spike or drop +is ever hidden, even under heavy compression. Empty intervals produce no +output, naturally preserving data gaps. + +![MinMax downsampling](/images/docs/subsample/minmax.svg) + +How it works: + +1. The total time range is divided into N/2 equal time intervals. +2. For each interval, up to 2 points are selected: min, max. +3. Duplicate points are removed (if min and max are the same row). +4. Empty intervals produce no output. + +Output is up to N points (N/2 buckets, up to 2 points each). + +```questdb-sql title="Hourly bars reduced to 8 with MinMax - min/max per bucket" demo +SELECT timestamp, avg(price) avg_price +FROM fx_trades +WHERE symbol = 'EURUSD' + AND timestamp IN '$today' +SAMPLE BY 1h +SUBSAMPLE minmax(avg_price, 8) +``` + ### m4 - Min/Max/First/Last per time interval -Divides the time range into equal time intervals and selects up to 4 points -per interval: the first, last, minimum, and maximum values. Empty intervals -produce no output, naturally preserving data gaps. +Builds on MinMax by also capturing the first and last rows in each time +interval. Where MinMax shows you the range of values in a bucket, M4 also +shows you where the data entered and exited - the opening and closing levels. +This matters when trends within a bucket are important: a price that opens +high, dips, then recovers looks different from one that opens low and climbs. +MinMax would show the same min/max range for both; M4 distinguishes them. -Best for monitoring dashboards where you must not miss spikes or drops. The -min/max envelope is pixel-accurate to the full dataset. +Empty intervals produce no output, naturally preserving data gaps. ![M4 downsampling](/images/docs/subsample/m4.svg) @@ -115,11 +150,11 @@ How it works: 4. Empty intervals produce no output. Output is up to N points (N/4 buckets, up to 4 points each). In the diagram -above, target 8 creates 2 time buckets. The first row happens to also be -the minimum in bucket 1, so each bucket emits 3-4 distinct rows instead of 4, -giving 7 total. +above, compare the right side with MinMax: M4 captures the exit at i=23 +(the pullback after the late spike), while MinMax ends at the peak. M4 +gives a more faithful picture of where the data actually settled. -```questdb-sql title="Hourly bars reduced to 8 with M4 - spike and trough guaranteed" demo +```questdb-sql title="Hourly bars reduced to 8 with M4 - captures entry/exit levels" demo SELECT timestamp, avg(price) avg_price FROM fx_trades WHERE symbol = 'EURUSD' @@ -136,36 +171,6 @@ the number of time buckets. A 1920-pixel-wide chart needs ::: -### minmax - Min/Max per time interval - -Divides the time range into equal time intervals and selects up to 2 points -per interval: the minimum and maximum values. Lighter than M4 (no first/last -tracking), producing roughly half the output. Empty intervals produce no -output. - -Best for simple envelope visualization where you only need the value range -per bucket, not entry/exit points. - -![MinMax downsampling](/images/docs/subsample/minmax.svg) - -How it works: - -1. The total time range is divided into N/2 equal time intervals. -2. For each interval, up to 2 points are selected: min, max. -3. Duplicate points are removed (if min and max are the same row). -4. Empty intervals produce no output. - -Output is up to N points (N/2 buckets, up to 2 points each). - -```questdb-sql title="Hourly bars reduced to 8 with MinMax - min/max per bucket" demo -SELECT timestamp, avg(price) avg_price -FROM fx_trades -WHERE symbol = 'EURUSD' - AND timestamp IN '$today' -SAMPLE BY 1h -SUBSAMPLE minmax(avg_price, 8) -``` - ### Gap-preserving LTTB Standard LTTB divides data by row count, so it connects across time gaps. An @@ -221,13 +226,13 @@ treat `targetPoints` as a hard maximum. ### Algorithm comparison -| Property | lttb | m4 | minmax | -|----------|------|----|--------| +| Property | lttb | minmax | m4 | +|----------|------|--------|-----| | Bucket type | Equal row count | Equal time intervals | Equal time intervals | -| Points per bucket | Exactly 1 | Up to 4 (first, last, min, max) | Up to 2 (min, max) | +| Points per bucket | Exactly 1 | Up to 2 (min, max) | Up to 4 (first, last, min, max) | | Output count | Exactly N (non-gap mode) | Up to N | Up to N | | Gap handling | Connects across gaps (use 3rd parameter to preserve) | Naturally preserves gaps | Naturally preserves gaps | -| Best use case | Line charts, shape preservation | Monitoring, spike detection | Lightweight envelope | +| Best use case | Line charts, shape preservation | Quick value range overview | Dashboards, SLA compliance | ## Examples diff --git a/scripts/gen_subsample_svgs.py b/scripts/gen_subsample_svgs.py index bd4ef4af55..07c1e6bc9a 100644 --- a/scripts/gen_subsample_svgs.py +++ b/scripts/gen_subsample_svgs.py @@ -15,10 +15,12 @@ GRAY = "#888" # default dots (real rows from raw data) # Segment A: 24 points, i=0..23 (represents 24 hourly bars) +# Late spike at i=22 (0.65) with pullback at i=23 (0.60) makes M4 visibly +# better than MinMax: M4 captures the exit at 0.60, MinMax only sees the peak. SEG_A = [ 0.50, 0.55, 0.60, 0.65, 0.70, 0.95, 0.85, 0.70, 0.60, 0.55, - 0.50, 0.45, 0.40, 0.35, 0.28, 0.20, 0.25, 0.30, 0.35, 0.40, - 0.45, 0.50, 0.48, 0.46, + 0.50, 0.45, 0.55, 0.35, 0.28, 0.20, 0.25, 0.30, 0.35, 0.40, + 0.45, 0.50, 0.65, 0.60, ] # Gap dataset: 3 data segments, 1 small gap (3h), 1 big gap (24h) @@ -154,7 +156,7 @@ def gen_lttb(): im, ix = 0, N - 1 ri = list(range(N)) # LTTB target 8: first + last always kept, 6 interior buckets - si = [0, 4, 5, 8, 15, 19, 22, 23] + si = [0, 4, 5, 11, 15, 18, 22, 23] sv = [SEG_A[i] for i in si] return f"""{hdr(W, H, "LTTB downsampling", "LTTB selects 8 points from 24.")} LTTB: 24 hourly bars reduced to 8 @@ -173,11 +175,12 @@ def gen_m4(): im, ix = 0, N - 1 ri = list(range(N)) # M4 target 8 -> 2 time buckets (0..11, 12..23) - # Bucket 1: first=0(.50), last=11(.45), min=0(.50)->dup, max=5(.95) -> 3 pts - # Bucket 2: first=12(.40), last=23(.46), min=15(.20), max=21(.50) -> 4 pts + # Bucket 1: first=0(.50), last=11(.45), min=0(.50)->dup first, max=5(.95) -> 3 pts + # Bucket 2: first=12(.55), last=23(.60), min=15(.20), max=22(.65) -> 4 pts + # Key: M4 catches the exit at i=23 (0.60) that MinMax misses m4 = [ (0,.50,GRAY),(5,.95,CYAN),(11,.45,GRAY), - (12,.40,GRAY),(15,.20,CYAN),(21,.50,CYAN),(23,.46,GRAY), + (12,.55,GRAY),(15,.20,CYAN),(22,.65,CYAN),(23,.60,GRAY), ] mi = [p[0] for p in m4] mv = [p[1] for p in m4] @@ -205,10 +208,10 @@ def gen_minmax(): # MinMax target 8 -> 4 time buckets of 6 (0..5, 6..11, 12..17, 18..23) # Bucket 1: min=0(.50), max=5(.95) # Bucket 2: min=11(.45), max=6(.85) - # Bucket 3: min=15(.20), max=12(.40) - # Bucket 4: min=23(.46), max=21(.50) - mi = [0, 5, 6, 11, 12, 15, 21, 23] - mv = [.50, .95, .85, .45, .40, .20, .50, .46] + # Bucket 3: min=15(.20), max=12(.55) + # Bucket 4: min=18(.35), max=22(.65) -- misses the exit at i=23 (0.60) + mi = [0, 5, 6, 11, 12, 15, 18, 22] + mv = [.50, .95, .85, .45, .55, .20, .35, .65] return f"""{hdr(W, H, "MinMax downsampling", "MinMax selects 8 points from 24.")} MinMax: target 8, emitted 8 (4 time buckets) {bkl([6, 12, 18], im, ix, YT, YB)} @@ -318,8 +321,8 @@ def gen_gap_detect(): if __name__ == "__main__": os.makedirs(OUT_DIR, exist_ok=True) - for name, fn in [("raw.svg", gen_raw), ("lttb.svg", gen_lttb), ("m4.svg", gen_m4), - ("minmax.svg", gen_minmax), + for name, fn in [("raw.svg", gen_raw), ("lttb.svg", gen_lttb), + ("minmax.svg", gen_minmax), ("m4.svg", gen_m4), ("gap-raw.svg", gen_gap_raw), ("gap-no-detect.svg", gen_gap_no_detect), ("gap-detect.svg", gen_gap_detect)]: diff --git a/static/images/docs/subsample/lttb.svg b/static/images/docs/subsample/lttb.svg index 693ece8442..03d7ee4c83 100644 --- a/static/images/docs/subsample/lttb.svg +++ b/static/images/docs/subsample/lttb.svg @@ -20,16 +20,16 @@ } LTTB: 24 hourly bars reduced to 8 - - + + - + - - - + + + Raw data diff --git a/static/images/docs/subsample/m4.svg b/static/images/docs/subsample/m4.svg index 9fcfaa6d99..180baec1ea 100644 --- a/static/images/docs/subsample/m4.svg +++ b/static/images/docs/subsample/m4.svg @@ -21,15 +21,15 @@ M4: target 8, emitted 7 (2 time buckets) - - + + - + - - + + Raw data diff --git a/static/images/docs/subsample/minmax.svg b/static/images/docs/subsample/minmax.svg index 0ab22a13b7..6e4acc54ee 100644 --- a/static/images/docs/subsample/minmax.svg +++ b/static/images/docs/subsample/minmax.svg @@ -23,16 +23,16 @@ - - + + - + - - + + Raw data diff --git a/static/images/docs/subsample/raw.svg b/static/images/docs/subsample/raw.svg index c9fe3c6f8c..8d9cb1adda 100644 --- a/static/images/docs/subsample/raw.svg +++ b/static/images/docs/subsample/raw.svg @@ -20,7 +20,7 @@ } Raw time series: 24 hourly bars - + @@ -33,7 +33,7 @@ - + @@ -43,8 +43,8 @@ - - + + Hourly bars (24) \ No newline at end of file From 18efc4dd3660798c15f4f70f5502266c0ad76fd3 Mon Sep 17 00:00:00 2001 From: javier Date: Thu, 23 Apr 2026 11:28:06 +0200 Subject: [PATCH 05/14] Add uniform and cadence algorithms to SUBSAMPLE page Move gap-preserving LTTB right after LTTB. Add uniform (evenly spaced rows) and cadence (every Nth row with optional random offset) sections. Update syntax block and comparison table. --- documentation/query/sql/subsample.md | 198 ++++++++++++++++++++------- 1 file changed, 145 insertions(+), 53 deletions(-) diff --git a/documentation/query/sql/subsample.md b/documentation/query/sql/subsample.md index a97e7c012c..5de4ada906 100644 --- a/documentation/query/sql/subsample.md +++ b/documentation/query/sql/subsample.md @@ -20,24 +20,28 @@ Requires a [designated timestamp](/docs/concepts/designated-timestamp/) column. ## Syntax -```questdb-sql -SELECT columns -FROM table -[WHERE conditions] -[SAMPLE BY ...] +```questdb-sql title="Value-based algorithms" SUBSAMPLE { lttb | m4 | minmax }(valueColumn, targetPoints [, gapThreshold]) -[ORDER BY ...] -[LIMIT ...] +``` + +```questdb-sql title="Position-based algorithms" +SUBSAMPLE uniform(targetPoints) +SUBSAMPLE cadence(stride [, seed]) ``` Where: - **`valueColumn`** - the numeric column used to decide which points are - visually significant. All other columns pass through for selected rows. + visually significant. Required for `lttb`, `m4`, and `minmax`. Not used + by `uniform` or `cadence`. - **`targetPoints`** - target number of output rows. Supports integer literals, [DECLARE](/docs/query/sql/declare/) variables, and bind variables (`$1`). Must be at least 2. Maximum is 2,147,483,647. -- **`gapThreshold`** - (LTTB only) optional interval that enables +- **`stride`** - (`cadence` only) step distance between emitted rows. This + is not an output count: `cadence(500)` emits one row out of every 500. +- **`seed`** - (`cadence` only) optional integer seed or `NULL`. See + [cadence](#cadence---every-nth-row). +- **`gapThreshold`** - (`lttb` only) optional interval that enables gap-preserving mode. See [gap-preserving LTTB](#gap-preserving-lttb). ### Execution order @@ -60,9 +64,14 @@ downsampling. ## Algorithms -Three algorithms are available. Each one selects real rows from the input - -no values are ever interpolated or computed. The diagrams below all use the -same 24-point series as input (think 24 hourly bars over one day): +Five algorithms are available. The first three (`lttb`, `minmax`, `m4`) +inspect values to decide which rows are visually significant. The last two +(`uniform`, `cadence`) ignore values and select rows purely by position - +they are cheaper and useful when the input is dense or as a baseline. + +All five select real rows from the input - no values are ever interpolated +or computed. The diagrams below use a 24-point series as input (think 24 +hourly bars over one day): ![Raw time series](/images/docs/subsample/raw.svg) @@ -98,6 +107,59 @@ SAMPLE BY 1h SUBSAMPLE lttb(avg_price, 8) ``` +### Gap-preserving LTTB + +Standard LTTB divides data by row count, so it connects across time gaps. An +optional third parameter sets a gap threshold: + +```questdb-sql +SUBSAMPLE lttb(price, 12, '6h') +``` + +When specified, LTTB scans for gaps where consecutive timestamps are further +apart than the threshold. Gaps below the threshold are ignored - the data is +treated as continuous. Gaps above the threshold split the data into separate +segments, each downsampled independently with its proportional share of the +target points. + +The diagrams below show a dataset with two gaps - a small one (3 hours) and +a large one (24 hours): + +![Raw data with gaps](/images/docs/subsample/gap-raw.svg) + +Without gap detection, LTTB treats all points as continuous and connects +across both gaps: + +![LTTB without gap detection](/images/docs/subsample/gap-no-detect.svg) + +With a threshold of `'6h'`, the small gap (3h) is below the threshold so +segments A and B are treated as continuous. The large gap (24h) exceeds the +threshold, so segment C is downsampled separately and the gap is preserved: + +![LTTB with gap detection](/images/docs/subsample/gap-detect.svg) + +Supported interval units: `s` (seconds), `m` (minutes), `h` (hours), +`d` (days). + +Examples: `'30s'`, `'5m'`, `'1h'`, `'7d'` + +```questdb-sql title="Preserve gaps larger than 6 hours in the output" demo +SELECT timestamp, price +FROM fx_trades +WHERE symbol = 'EURUSD' +SUBSAMPLE lttb(price, 12, '6h') +``` + +:::note + +Gap-preserving LTTB uses a soft target. Each segment receives at least its +first and last points. When many segments are detected, the total output may +exceed `targetPoints`. This is by design so that the same query does not fail +for one time range and succeed for another. Non-gap LTTB, M4, and MinMax +treat `targetPoints` as a hard maximum. + +::: + ### minmax - Min/Max per time interval Divides the time range into equal time intervals and selects up to 2 points @@ -171,68 +233,98 @@ the number of time buckets. A 1920-pixel-wide chart needs ::: -### Gap-preserving LTTB +### uniform - Evenly spaced rows -Standard LTTB divides data by row count, so it connects across time gaps. An -optional third parameter sets a gap threshold: +Selects a target number of rows spaced evenly across the input. First and +last rows are always kept, interior rows are picked at regular positions +between them. Unlike the previous algorithms, `uniform` does not inspect +values - it reduces row count purely by position in the time-ordered input. -```questdb-sql -SUBSAMPLE lttb(price, 12, '6h') +Use `uniform` when the input is dense and you care about reducing transfer +size more than preserving spikes or troughs. For a line chart where visual +fidelity matters, `lttb` or `m4` produce better results at the same target +count. For a heatmap, scatter plot, or tabular display where every row looks +similar, `uniform` is faster and the output is indistinguishable from +value-aware methods. + +How it works: + +1. First and last rows are always selected. +2. Remaining `targetPoints - 2` rows are selected at evenly spaced positions + between first and last. +3. Output is exactly `targetPoints` rows when the input is larger than the + target, otherwise all input rows are returned unchanged. + +```questdb-sql title="500 evenly spaced rows from a dense tick table" demo +SELECT timestamp, price +FROM fx_trades +WHERE symbol = 'EURUSD' +SUBSAMPLE uniform(500) ``` -When specified, LTTB scans for gaps where consecutive timestamps are further -apart than the threshold. Gaps below the threshold are ignored - the data is -treated as continuous. Gaps above the threshold split the data into separate -segments, each downsampled independently with its proportional share of the -target points. +### cadence - Every Nth row -The diagrams below show a dataset with two gaps - a small one (3 hours) and -a large one (24 hours): +Selects one row out of every N, starting from a configurable offset. Like +`uniform`, `cadence` does not inspect values - it reduces row count by +stepping through the input at a fixed rhythm. -![Raw data with gaps](/images/docs/subsample/gap-raw.svg) +The `stride` parameter is the step distance, not the output count. To keep +500 rows, use `uniform(500)` or `lttb(col, 500)`. `cadence(500)` emits one +row out of every 500, which is a different (and input-dependent) number. -Without gap detection, LTTB treats all points as continuous and connects -across both gaps: +How it works: -![LTTB without gap detection](/images/docs/subsample/gap-no-detect.svg) +1. First and last rows are always selected (except when stride exceeds the + input size, in which case only the first row is emitted). +2. From the offset position, emit one row every `stride` rows. +3. Output is in timestamp-ascending order. -With a threshold of `'6h'`, the small gap (3h) is below the threshold so -segments A and B are treated as continuous. The large gap (24h) exceeds the -threshold, so segment C is downsampled separately and the gap is preserved: +| Form | Behavior | +|------|----------| +| `cadence(N)` | Every Nth row, deterministic, offset 0 | +| `cadence(N, seed)` | Random offset in [0, N), reproducible given seed | +| `cadence(N, NULL)` | Random offset in [0, N), fresh each run | -![LTTB with gap detection](/images/docs/subsample/gap-detect.svg) +The seeded and NULL forms exist to avoid phase-lock with periodic signals. +If the input has a 1000-row period and you stride by 1000 with offset 0, +every emitted row hits the same phase of the period and the chart loses the +periodic structure. A random offset breaks this alignment. -Supported interval units: `s` (seconds), `m` (minutes), `h` (hours), -`d` (days). +:::note -Examples: `'30s'`, `'5m'`, `'1h'`, `'7d'` +Randomizing the offset helps with aliasing on periodic signals, but it does +not make `cadence` a statistical sampler. It does not produce unbiased +estimates of aggregates like mean or percentile. For those, use +[SAMPLE BY](/docs/query/sql/sample-by/) with the appropriate aggregate +function. -```questdb-sql title="Preserve gaps larger than 6 hours in the output" demo +::: + +```questdb-sql title="Every 1000th row - simple decimation" demo SELECT timestamp, price FROM fx_trades WHERE symbol = 'EURUSD' -SUBSAMPLE lttb(price, 12, '6h') +SUBSAMPLE cadence(1000) ``` -:::note - -Gap-preserving LTTB uses a soft target. Each segment receives at least its -first and last points. When many segments are detected, the total output may -exceed `targetPoints`. This is by design so that the same query does not fail -for one time range and succeed for another. Non-gap LTTB, M4, and MinMax -treat `targetPoints` as a hard maximum. - -::: +```questdb-sql title="Anti-aliasing with reproducible seed" +SELECT timestamp, price +FROM fx_trades +WHERE symbol = 'EURUSD' +SUBSAMPLE cadence(1000, 42) +``` ### Algorithm comparison -| Property | lttb | minmax | m4 | -|----------|------|--------|-----| -| Bucket type | Equal row count | Equal time intervals | Equal time intervals | -| Points per bucket | Exactly 1 | Up to 2 (min, max) | Up to 4 (first, last, min, max) | -| Output count | Exactly N (non-gap mode) | Up to N | Up to N | -| Gap handling | Connects across gaps (use 3rd parameter to preserve) | Naturally preserves gaps | Naturally preserves gaps | -| Best use case | Line charts, shape preservation | Quick value range overview | Dashboards, SLA compliance | +| Property | lttb | minmax | m4 | uniform | cadence | +|----------|------|--------|-----|---------|---------| +| Parameter | targetPoints | targetPoints | targetPoints | targetPoints | stride | +| Inspects values | Yes | Yes | Yes | No | No | +| Bucket type | Equal row count | Equal time intervals | Equal time intervals | Equal row spacing | Fixed row stride | +| Points per bucket | Exactly 1 | Up to 2 (min, max) | Up to 4 (first, last, min, max) | N/A | N/A | +| Output count | Exactly N | Up to N | Up to N | Exactly N | ~rowCount/stride | +| Gap handling | Connects across (use threshold) | Naturally preserves | Naturally preserves | Connects across | Connects across | +| Best use case | Line charts | Value range overview | Dashboards, SLA | Dense uniform data | Decimation, anti-aliasing | ## Examples From 5bc7d4b4fee1cd86648ec09369f9d4b2f9b7b8b2 Mon Sep 17 00:00:00 2001 From: javier Date: Thu, 23 Apr 2026 11:45:36 +0200 Subject: [PATCH 06/14] Add uniform and cadence algorithms, charts, and comparison table Add uniform (evenly spaced) and cadence (every Nth row) sections with SVG diagrams. Move gap-preserving LTTB under LTTB heading. Reorder MinMax before M4. Add relative cost row to comparison table. --- documentation/query/sql/subsample.md | 9 +++-- scripts/gen_subsample_svgs.py | 44 ++++++++++++++++++++++++ static/images/docs/subsample/cadence.svg | 38 ++++++++++++++++++++ static/images/docs/subsample/uniform.svg | 37 ++++++++++++++++++++ 4 files changed, 126 insertions(+), 2 deletions(-) create mode 100644 static/images/docs/subsample/cadence.svg create mode 100644 static/images/docs/subsample/uniform.svg diff --git a/documentation/query/sql/subsample.md b/documentation/query/sql/subsample.md index 5de4ada906..24454c171f 100644 --- a/documentation/query/sql/subsample.md +++ b/documentation/query/sql/subsample.md @@ -107,7 +107,7 @@ SAMPLE BY 1h SUBSAMPLE lttb(avg_price, 8) ``` -### Gap-preserving LTTB +#### Gap-preserving LTTB Standard LTTB divides data by row count, so it connects across time gaps. An optional third parameter sets a gap threshold: @@ -247,6 +247,8 @@ count. For a heatmap, scatter plot, or tabular display where every row looks similar, `uniform` is faster and the output is indistinguishable from value-aware methods. +![Uniform downsampling](/images/docs/subsample/uniform.svg) + How it works: 1. First and last rows are always selected. @@ -272,6 +274,8 @@ The `stride` parameter is the step distance, not the output count. To keep 500 rows, use `uniform(500)` or `lttb(col, 500)`. `cadence(500)` emits one row out of every 500, which is a different (and input-dependent) number. +![Cadence downsampling](/images/docs/subsample/cadence.svg) + How it works: 1. First and last rows are always selected (except when stride exceeds the @@ -322,9 +326,10 @@ SUBSAMPLE cadence(1000, 42) | Inspects values | Yes | Yes | Yes | No | No | | Bucket type | Equal row count | Equal time intervals | Equal time intervals | Equal row spacing | Fixed row stride | | Points per bucket | Exactly 1 | Up to 2 (min, max) | Up to 4 (first, last, min, max) | N/A | N/A | -| Output count | Exactly N | Up to N | Up to N | Exactly N | ~rowCount/stride | +| Output count | Exactly N (or all rows if fewer) | Up to N | Up to N | Exactly N (or all rows if fewer) | ~rowCount/stride | | Gap handling | Connects across (use threshold) | Naturally preserves | Naturally preserves | Connects across | Connects across | | Best use case | Line charts | Value range overview | Dashboards, SLA | Dense uniform data | Decimation, anti-aliasing | +| Relative cost | Higher: triangle area per point | Low: min/max per bucket | Medium: first/last/min/max per bucket | Lowest: position arithmetic | Lowest: stride arithmetic | ## Examples diff --git a/scripts/gen_subsample_svgs.py b/scripts/gen_subsample_svgs.py index 07c1e6bc9a..aeacf3fe54 100644 --- a/scripts/gen_subsample_svgs.py +++ b/scripts/gen_subsample_svgs.py @@ -227,6 +227,49 @@ def gen_minmax(): """ +def gen_uniform(): + N = len(SEG_A) + im, ix = 0, N - 1 + ri = list(range(N)) + # uniform(8): evenly spaced, first and last pinned + # positions: round(i * 23 / 7) for i in 0..7 = 0, 3, 7, 10, 13, 16, 20, 23 + si = [round(i * (N - 1) / 7) for i in range(8)] + sv = [SEG_A[i] for i in si] + return f"""{hdr(W, H, "Uniform downsampling", "Uniform selects 8 evenly spaced points from 24.")} +Uniform: 8 evenly spaced from 24 +{rpl(ri, SEG_A, im, ix, YT, YB)} + +{cd(si,sv,im,ix,YT,YB,GRAY)} + +Raw data + +Selected points (8 of 24) +""" + + +def gen_cadence(): + N = len(SEG_A) + im, ix = 0, N - 1 + ri = list(range(N)) + # cadence(3): every 3rd row from offset 0, plus last row pinned + # positions: 0, 3, 6, 9, 12, 15, 18, 21, 23(pinned) + stride = 3 + si = list(range(0, N, stride)) + if si[-1] != N - 1: + si.append(N - 1) + sv = [SEG_A[i] for i in si] + return f"""{hdr(W, H, "Cadence downsampling", "Cadence selects every 3rd row from 24.")} +Cadence: stride 3, emitted {len(si)} from 24 +{rpl(ri, SEG_A, im, ix, YT, YB)} + +{cd(si,sv,im,ix,YT,YB,GRAY)} + +Raw data + +Selected points ({len(si)} of 24) +""" + + def _gap_helpers(): """Shared helpers for the three gap SVGs.""" im, ix = 0, 68 @@ -323,6 +366,7 @@ def gen_gap_detect(): os.makedirs(OUT_DIR, exist_ok=True) for name, fn in [("raw.svg", gen_raw), ("lttb.svg", gen_lttb), ("minmax.svg", gen_minmax), ("m4.svg", gen_m4), + ("uniform.svg", gen_uniform), ("cadence.svg", gen_cadence), ("gap-raw.svg", gen_gap_raw), ("gap-no-detect.svg", gen_gap_no_detect), ("gap-detect.svg", gen_gap_detect)]: diff --git a/static/images/docs/subsample/cadence.svg b/static/images/docs/subsample/cadence.svg new file mode 100644 index 0000000000..1666f2dc8c --- /dev/null +++ b/static/images/docs/subsample/cadence.svg @@ -0,0 +1,38 @@ + +Cadence downsampling +Cadence selects every 3rd row from 24. + +Cadence: stride 3, emitted 9 from 24 + + + + + + + + + + + + +Raw data + +Selected points (9 of 24) + \ No newline at end of file diff --git a/static/images/docs/subsample/uniform.svg b/static/images/docs/subsample/uniform.svg new file mode 100644 index 0000000000..adcf661bba --- /dev/null +++ b/static/images/docs/subsample/uniform.svg @@ -0,0 +1,37 @@ + +Uniform downsampling +Uniform selects 8 evenly spaced points from 24. + +Uniform: 8 evenly spaced from 24 + + + + + + + + + + + +Raw data + +Selected points (8 of 24) + \ No newline at end of file From 4c18b238d679cd327ba299bfab293492bdfeb3c1 Mon Sep 17 00:00:00 2001 From: javier Date: Thu, 23 Apr 2026 11:50:58 +0200 Subject: [PATCH 07/14] Mention optional offset parameter in cadence intro --- documentation/query/sql/subsample.md | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/documentation/query/sql/subsample.md b/documentation/query/sql/subsample.md index 24454c171f..627e53168a 100644 --- a/documentation/query/sql/subsample.md +++ b/documentation/query/sql/subsample.md @@ -268,7 +268,9 @@ SUBSAMPLE uniform(500) Selects one row out of every N, starting from a configurable offset. Like `uniform`, `cadence` does not inspect values - it reduces row count by -stepping through the input at a fixed rhythm. +stepping through the input at a fixed rhythm. An optional second parameter +sets the starting offset, either as a fixed seed for reproducible results or +as `NULL` for a fresh random offset each run. The `stride` parameter is the step distance, not the output count. To keep 500 rows, use `uniform(500)` or `lttb(col, 500)`. `cadence(500)` emits one From 153a1d02155b7045d7574f2b3a631bdd5df76b05 Mon Sep 17 00:00:00 2001 From: javier Date: Thu, 23 Apr 2026 12:01:05 +0200 Subject: [PATCH 08/14] Add chart-ready examples for all algorithms and minor fixes Add uniform, cadence, and gap-preserving LTTB to the chart-ready examples section. Make DECLARE example demoable. Replace Grafana reference with generic programmatic integration. --- documentation/query/sql/subsample.md | 25 +++++++++++++++++++++++-- 1 file changed, 23 insertions(+), 2 deletions(-) diff --git a/documentation/query/sql/subsample.md b/documentation/query/sql/subsample.md index 627e53168a..c12449867d 100644 --- a/documentation/query/sql/subsample.md +++ b/documentation/query/sql/subsample.md @@ -344,6 +344,13 @@ WHERE symbol = 'EURUSD' SUBSAMPLE lttb(price, 500) ``` +```questdb-sql title="LTTB with gap detection: preserve gaps larger than 1 hour" demo +SELECT timestamp, price +FROM fx_trades +WHERE symbol = 'EURUSD' +SUBSAMPLE lttb(price, 500, '1h') +``` + ```questdb-sql title="M4: pixel-accurate envelope for a 1920px-wide chart" demo SELECT timestamp, price FROM fx_trades @@ -358,6 +365,20 @@ WHERE symbol = 'EURUSD' SUBSAMPLE minmax(price, 500) ``` +```questdb-sql title="Uniform: 500 evenly spaced rows for a dense table" demo +SELECT timestamp, price +FROM fx_trades +WHERE symbol = 'EURUSD' +SUBSAMPLE uniform(500) +``` + +```questdb-sql title="Cadence: every 1000th row for quick decimation" demo +SELECT timestamp, price +FROM fx_trades +WHERE symbol = 'EURUSD' +SUBSAMPLE cadence(1000) +``` + ### Composing with SAMPLE BY ```questdb-sql title="Aggregate to 1-minute bars, then downsample" demo @@ -396,7 +417,7 @@ the result, so the moving average values are accurate. ### With DECLARE variable -```questdb-sql title="Parameterized target point count" +```questdb-sql title="Parameterized target point count" demo DECLARE @points := 500 SELECT timestamp, price FROM fx_trades @@ -406,7 +427,7 @@ SUBSAMPLE lttb(price, @points) ### With bind variable -```questdb-sql title="Grafana integration - screen width as bind variable" +```questdb-sql title="Programmatic integration - target as bind variable" SELECT timestamp, price FROM fx_trades WHERE symbol = 'EURUSD' From d77aa8548ed35c6609416d00022a66ac86a2f0e6 Mon Sep 17 00:00:00 2001 From: javier Date: Thu, 23 Apr 2026 12:03:57 +0200 Subject: [PATCH 09/14] Explain pass-through columns carry original values --- documentation/query/sql/subsample.md | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/documentation/query/sql/subsample.md b/documentation/query/sql/subsample.md index c12449867d..54c2a8da5b 100644 --- a/documentation/query/sql/subsample.md +++ b/documentation/query/sql/subsample.md @@ -395,6 +395,12 @@ complement each other: aggregate first, then reduce for display. ### Multiple columns pass through +Because `SUBSAMPLE` selects real rows rather than computing new ones, every +column in the output carries its original value from the source table. In +the query below, `side` and `quantity` are not involved in the downsampling +decision, but each output row is a real trade with the actual side and +quantity that occurred at that timestamp. + ```questdb-sql title="LTTB selects rows by price; all columns emit" demo SELECT timestamp, symbol, side, price, quantity FROM fx_trades From 50db1af69f685a21ae652257f3e923f5238e1ce8 Mon Sep 17 00:00:00 2001 From: javier Date: Fri, 24 Apr 2026 11:14:12 +0200 Subject: [PATCH 10/14] Highlight first/last dots in M4 chart to distinguish from MinMax --- scripts/gen_subsample_svgs.py | 8 ++++---- static/images/docs/subsample/m4.svg | 18 +++++++++--------- 2 files changed, 13 insertions(+), 13 deletions(-) diff --git a/scripts/gen_subsample_svgs.py b/scripts/gen_subsample_svgs.py index aeacf3fe54..c9cdd681d1 100644 --- a/scripts/gen_subsample_svgs.py +++ b/scripts/gen_subsample_svgs.py @@ -179,8 +179,8 @@ def gen_m4(): # Bucket 2: first=12(.55), last=23(.60), min=15(.20), max=22(.65) -> 4 pts # Key: M4 catches the exit at i=23 (0.60) that MinMax misses m4 = [ - (0,.50,GRAY),(5,.95,CYAN),(11,.45,GRAY), - (12,.55,GRAY),(15,.20,CYAN),(22,.65,CYAN),(23,.60,GRAY), + (0,.50,CYAN),(5,.95,GRAY),(11,.45,CYAN), + (12,.55,CYAN),(15,.20,GRAY),(22,.65,GRAY),(23,.60,CYAN), ] mi = [p[0] for p in m4] mv = [p[1] for p in m4] @@ -192,9 +192,9 @@ def gen_m4(): {cdm(m4, im, ix, YT, YB)} Raw data - + First / Last - + Min / Max Bucket boundary diff --git a/static/images/docs/subsample/m4.svg b/static/images/docs/subsample/m4.svg index 180baec1ea..3e2ca661b7 100644 --- a/static/images/docs/subsample/m4.svg +++ b/static/images/docs/subsample/m4.svg @@ -23,18 +23,18 @@ - - - - - - - + + + + + + + Raw data - + First / Last - + Min / Max Bucket boundary From a865124509289c87d024fdaa3afd92d12cdd7111 Mon Sep 17 00:00:00 2001 From: javier Date: Mon, 21 Sep 2026 13:29:05 +0200 Subject: [PATCH 11/14] Align SUBSAMPLE docs with merged engine behavior Correct the source-row guarantee, designated timestamp requirement, output ordering, NULL handling, LTTB output count and gap allocation, M4 and MinMax gap and sizing claims, uniform and cadence edge cases, and the execution and memory model. Update cairo.sql.subsample.max.rows wording and mark the gap-detect diagram's line break as client-rendered. --- documentation/configuration/cairo-engine.md | 7 +- documentation/query/sql/subsample.md | 227 +++++++++++++------- scripts/gen_subsample_svgs.py | 17 +- static/images/docs/subsample/gap-detect.svg | 8 +- 4 files changed, 175 insertions(+), 84 deletions(-) diff --git a/documentation/configuration/cairo-engine.md b/documentation/configuration/cairo-engine.md index d7d1978c4a..7d10089aa4 100644 --- a/documentation/configuration/cairo-engine.md +++ b/documentation/configuration/cairo-engine.md @@ -875,8 +875,11 @@ SAMPLE BY index query page size (maximum values returned in a single scan). - **Default**: `100000000` - **Reloadable**: no -Maximum number of input rows SUBSAMPLE will buffer. Exceeding this limit -returns an error. Must be between 1 and 2,147,483,647. +Maximum number of input rows a +[SUBSAMPLE](/docs/query/sql/subsample/) query accepts. Exceeding this limit +returns an error. Every input row counts, including rows that a value-based +method skips because of a `NULL` value. The limit is independent of the +`targetPoints` argument. Must be between 1 and 2,147,483,647. ## Window functions diff --git a/documentation/query/sql/subsample.md b/documentation/query/sql/subsample.md index 54c2a8da5b..c1e90b12e0 100644 --- a/documentation/query/sql/subsample.md +++ b/documentation/query/sql/subsample.md @@ -1,7 +1,7 @@ --- title: SUBSAMPLE keyword sidebar_label: SUBSAMPLE -description: SUBSAMPLE SQL keyword reference for time-series downsampling using LTTB, M4, and MinMax algorithms. +description: SUBSAMPLE SQL keyword reference for time-series downsampling using the LTTB, M4, MinMax, uniform, and cadence algorithms. --- `SUBSAMPLE` reduces the number of rows in a query result while preserving the @@ -10,18 +10,26 @@ time-ordered dataset, making it ideal for rendering charts at screen resolution without transferring millions of rows to the client. Unlike [SAMPLE BY](/docs/query/sql/sample-by/), which computes new aggregate -values at synthetic bucket boundaries, `SUBSAMPLE` selects actual rows from -the input. Every output row exists in the source table with its original -timestamp and values. This means output timestamps match real rows (useful for -joins), and users can drill down to the exact source record behind any point -on a chart. - -Requires a [designated timestamp](/docs/concepts/designated-timestamp/) column. +values at synthetic bucket boundaries, `SUBSAMPLE` selects rows from its +immediate query input and never alters their values. Over a direct table +scan, every output row is a physical table row with its original timestamp, +so output timestamps can be used in joins and users can drill down to the +exact source record behind any point on a chart. After `SAMPLE BY`, +`GROUP BY`, a join, or a computed projection, the selected rows are the +derived rows of that query, which can carry aggregate values or synthetic +timestamps. + +The query source must provide a +[designated timestamp](/docs/concepts/designated-timestamp/), and the `SELECT` +list must preserve it. A table with a designated timestamp is not enough if +the projection omits that column or replaces it with an expression that loses +the designation. ## Syntax ```questdb-sql title="Value-based algorithms" -SUBSAMPLE { lttb | m4 | minmax }(valueColumn, targetPoints [, gapThreshold]) +SUBSAMPLE lttb(valueColumn, targetPoints [, gapThreshold]) +SUBSAMPLE { m4 | minmax }(valueColumn, targetPoints) ``` ```questdb-sql title="Position-based algorithms" @@ -48,30 +56,47 @@ Where: `SUBSAMPLE` runs after `SAMPLE BY`, `GROUP BY`, and window functions, but before `ORDER BY` and `LIMIT`. All value computations are complete before -downsampling decides which rows to keep. `SUBSAMPLE` only selects rows - it -never modifies computed values. - -All three algorithms execute serially. `SUBSAMPLE` buffers its entire input, -runs the selected algorithm, then emits the chosen rows. It does not block -upstream parallel execution - for example, a parallel `SAMPLE BY` completes -before `SUBSAMPLE` buffers its output. +downsampling decides which rows to keep, and a final `ORDER BY` or `LIMIT` +operates on the selected rows. `SUBSAMPLE` only selects rows. It never +modifies computed values. + +Internally, `SUBSAMPLE` computes a keep-or-drop flag for every input row, +the same way a window function computes one value per row, and then filters +the input down to the flagged rows. This selection stage makes two passes +over its input and runs serially. It does not block upstream parallel +execution. For example, a parallel `SAMPLE BY` completes before `SUBSAMPLE` +reads its output. + +### Output order + +Every algorithm computes its selection against an ascending +designated-timestamp traversal of the input, but the query returns the +selected rows in the order of the incoming query. A descending timestamp +input stays descending, and an input explicitly ordered by another column +keeps that order. Add a final `ORDER BY` when you need a specific output +order. ### Supported value types -The value column must be a numeric type: `DOUBLE`, `FLOAT`, `INT`, `LONG`, -`SHORT`, or `BYTE`. `NULL` values in the value column are skipped during -downsampling. +The value column of `lttb`, `m4`, and `minmax` must be a numeric type: +`DOUBLE`, `FLOAT`, `INT`, `LONG`, `SHORT`, or `BYTE`. For these three +methods, a row is not eligible for selection when its value is `NULL` or +non-finite, or when its timestamp is `NULL`. Rows skipped this way still +count toward the [input row limit](#configuration). + +`uniform` and `cadence` take no value column. `NULL` values in any projected +column do not prevent a row from being selected. ## Algorithms Five algorithms are available. The first three (`lttb`, `minmax`, `m4`) inspect values to decide which rows are visually significant. The last two -(`uniform`, `cadence`) ignore values and select rows purely by position - -they are cheaper and useful when the input is dense or as a baseline. +(`uniform`, `cadence`) ignore values and select rows purely by position. +They are useful when the input is dense or as a baseline. -All five select real rows from the input - no values are ever interpolated -or computed. The diagrams below use a 24-point series as input (think 24 -hourly bars over one day): +All five select existing rows from their input. No values are ever +interpolated or computed. The diagrams below use a 24-point series as input +(think 24 hourly bars over one day): ![Raw time series](/images/docs/subsample/raw.svg) @@ -82,7 +107,9 @@ each bucket that forms the largest triangle with its neighbors. The idea is that points where the line changes direction sharply (a spike, a valley, a sudden trend shift) form large triangles and get kept, while points in the middle of a smooth trend form small triangles and get dropped. The first and -last points are always kept. Output is exactly N points. +last points are always kept. Output is exactly N points when at least N +[eligible rows](#supported-value-types) exist. With fewer eligible rows, all +of them are returned. Best for line charts where the visual shape matters most - a chart drawn from the LTTB output looks nearly identical to one drawn from the full @@ -96,7 +123,13 @@ How it works: 2. Remaining data is divided into N-2 equal-sized buckets by row count. 3. For each bucket, the point creating the largest triangle area with the previously selected point and the average of the next bucket is chosen. -4. Output preserves the original timestamp order. + +When the input is much larger than the target (more than about 8 eligible +rows per target point), QuestDB uses a two-stage variant known as +MinMaxLTTB. It first preselects the local minima and maxima from row-count +bins, then runs the triangle stage on those candidates only. The output +count is the same, but the selected points can differ from classic LTTB run +over every raw row. Smaller inputs use classic LTTB directly. ```questdb-sql title="Aggregate to hourly bars, then pick the 8 most representative" demo SELECT timestamp, avg(price) avg_price @@ -119,8 +152,10 @@ SUBSAMPLE lttb(price, 12, '6h') When specified, LTTB scans for gaps where consecutive timestamps are further apart than the threshold. Gaps below the threshold are ignored - the data is treated as continuous. Gaps above the threshold split the data into separate -segments, each downsampled independently with its proportional share of the -target points. +segments, each downsampled independently. Each segment receives an integer +share of the target proportional to its row count, with a minimum of two +points (one for a single-row segment) so that a multi-row segment keeps its +endpoints. The diagrams below show a dataset with two gaps - a small one (3 hours) and a large one (24 hours): @@ -134,10 +169,19 @@ across both gaps: With a threshold of `'6h'`, the small gap (3h) is below the threshold so segments A and B are treated as continuous. The large gap (24h) exceeds the -threshold, so segment C is downsampled separately and the gap is preserved: +threshold, so segment C is downsampled separately and both edges of the gap +are retained: ![LTTB with gap detection](/images/docs/subsample/gap-detect.svg) +The diagram draws each segment as a separate line, which is what a client +that breaks lines on large timestamp gaps would render. The SQL result +itself is a flat list of rows with no `NULL` separator row and no segment +identifier, so a renderer that connects consecutive points still joins the +last point of one segment to the first point of the next. To show the +discontinuity on a chart, apply a timestamp-gap or segment-breaking rule in +the client. + Supported interval units: `s` (seconds), `m` (minutes), `h` (hours), `d` (days). @@ -152,11 +196,13 @@ SUBSAMPLE lttb(price, 12, '6h') :::note -Gap-preserving LTTB uses a soft target. Each segment receives at least its -first and last points. When many segments are detected, the total output may -exceed `targetPoints`. This is by design so that the same query does not fail -for one time range and succeed for another. Non-gap LTTB, M4, and MinMax -treat `targetPoints` as a hard maximum. +Gap-preserving LTTB treats `targetPoints` as a goal, not an exact count. +Integer rounding of the proportional shares can leave part of the target +unused, so the output can be below `targetPoints`. When many segments are +detected, the per-segment minimum can push the total output above +`targetPoints`. This is by design so that the same query does not fail for +one time range and succeed for another. Non-gap LTTB, M4, and MinMax treat +`targetPoints` as a hard maximum. ::: @@ -165,9 +211,15 @@ treat `targetPoints` as a hard maximum. Divides the time range into equal time intervals and selects up to 2 points per interval: the row with the minimum value and the row with the maximum value. This creates a visual envelope - at any point on the chart, you can -see the full range the data covered during that interval. No spike or drop -is ever hidden, even under heavy compression. Empty intervals produce no -output, naturally preserving data gaps. +see the full range the data covered during that interval. The minimum and +maximum rows of every non-empty interval are always retained, and when both +resolve to the same row it is emitted once. + +Empty intervals emit no rows, so the result retains the absence of samples +in those intervals. That alone does not make a line renderer break the line +across the gap. As with +[gap-preserving LTTB](#gap-preserving-lttb), the client must apply a +timestamp-gap rule to show the discontinuity. ![MinMax downsampling](/images/docs/subsample/minmax.svg) @@ -198,7 +250,9 @@ This matters when trends within a bucket are important: a price that opens high, dips, then recovers looks different from one that opens low and climbs. MinMax would show the same min/max range for both; M4 distinguishes them. -Empty intervals produce no output, naturally preserving data gaps. +Empty intervals emit no rows, with the same rendering caveat as MinMax and +[gap-preserving LTTB](#gap-preserving-lttb): a line renderer can still +bridge the gap unless the client breaks the line. ![M4 downsampling](/images/docs/subsample/m4.svg) @@ -227,9 +281,11 @@ SUBSAMPLE m4(avg_price, 8) :::tip -When sizing `targetPoints` for a pixel-wide chart, remember that N/4 gives -the number of time buckets. A 1920-pixel-wide chart needs -`SUBSAMPLE m4(col, 1920)` to get 480 time buckets with up to 4 points each. +`targetPoints` is a row budget, not a bucket count: N/4 gives the number of +time buckets. `SUBSAMPLE m4(col, 1920)` creates 480 time buckets and returns +up to 1,920 rows. For one time bucket per pixel column on a 1920-pixel-wide +chart, use `SUBSAMPLE m4(col, 7680)`. Empty buckets and role deduplication +can reduce the returned row count. ::: @@ -238,22 +294,26 @@ the number of time buckets. A 1920-pixel-wide chart needs Selects a target number of rows spaced evenly across the input. First and last rows are always kept, interior rows are picked at regular positions between them. Unlike the previous algorithms, `uniform` does not inspect -values - it reduces row count purely by position in the time-ordered input. +values. It reduces row count purely by position in the timestamp-ordered +traversal of the input. Use `uniform` when the input is dense and you care about reducing transfer -size more than preserving spikes or troughs. For a line chart where visual -fidelity matters, `lttb` or `m4` produce better results at the same target -count. For a heatmap, scatter plot, or tabular display where every row looks -similar, `uniform` is faster and the output is indistinguishable from -value-aware methods. +size more than preserving spikes or troughs. Because it ignores values, it +can visibly miss spikes and troughs. For a line chart where visual fidelity +matters, `lttb` or `m4` produce better results at the same target count. +`uniform` fits a heatmap, scatter plot, or tabular display where every row +looks similar. It avoids value inspection and the timestamp/value buffer of +the value-based methods, but full-query performance depends on the +surrounding plan. ![Uniform downsampling](/images/docs/subsample/uniform.svg) How it works: -1. First and last rows are always selected. +1. The first and last rows in timestamp order are always selected. 2. Remaining `targetPoints - 2` rows are selected at evenly spaced positions - between first and last. + between first and last. Fractional positions round half up, so the + selection is deterministic. 3. Output is exactly `targetPoints` rows when the input is larger than the target, otherwise all input rows are returned unchanged. @@ -267,8 +327,9 @@ SUBSAMPLE uniform(500) ### cadence - Every Nth row Selects one row out of every N, starting from a configurable offset. Like -`uniform`, `cadence` does not inspect values - it reduces row count by -stepping through the input at a fixed rhythm. An optional second parameter +`uniform`, `cadence` does not inspect values. It reduces row count by +stepping through the timestamp-ordered traversal of the input at a fixed +rhythm. An optional second parameter sets the starting offset, either as a fixed seed for reproducible results or as `NULL` for a fresh random offset each run. @@ -280,10 +341,13 @@ row out of every 500, which is a different (and input-dependent) number. How it works: -1. First and last rows are always selected (except when stride exceeds the - input size, in which case only the first row is emitted). -2. From the offset position, emit one row every `stride` rows. -3. Output is in timestamp-ascending order. +1. When `stride` is greater than 1 and no larger than the input row count, + the first and last rows in timestamp order are always selected. +2. Between them, one row is selected every `stride` rows, starting from the + offset position. +3. `cadence(1)` returns every row. +4. When `stride` exceeds the input row count, only the first row is + selected. The last row is not pinned in this case. | Form | Behavior | |------|----------| @@ -328,10 +392,9 @@ SUBSAMPLE cadence(1000, 42) | Inspects values | Yes | Yes | Yes | No | No | | Bucket type | Equal row count | Equal time intervals | Equal time intervals | Equal row spacing | Fixed row stride | | Points per bucket | Exactly 1 | Up to 2 (min, max) | Up to 4 (first, last, min, max) | N/A | N/A | -| Output count | Exactly N (or all rows if fewer) | Up to N | Up to N | Exactly N (or all rows if fewer) | ~rowCount/stride | -| Gap handling | Connects across (use threshold) | Naturally preserves | Naturally preserves | Connects across | Connects across | +| Output count | Exactly N when N or more eligible rows exist, otherwise all eligible rows. Gap mode can return fewer or more than N | Up to N | Up to N | Exactly N (or all rows if fewer) | ~rowCount/stride | +| Gap handling | Connects across. With a threshold, segments are selected independently; a line renderer may still bridge the gap | Empty buckets emit no rows; a line renderer may still bridge the gap | Empty buckets emit no rows; a line renderer may still bridge the gap | Connects across | Connects across | | Best use case | Line charts | Value range overview | Dashboards, SLA | Dense uniform data | Decimation, anti-aliasing | -| Relative cost | Higher: triangle area per point | Low: min/max per bucket | Medium: first/last/min/max per bucket | Lowest: position arithmetic | Lowest: stride arithmetic | ## Examples @@ -351,14 +414,14 @@ WHERE symbol = 'EURUSD' SUBSAMPLE lttb(price, 500, '1h') ``` -```questdb-sql title="M4: pixel-accurate envelope for a 1920px-wide chart" demo +```questdb-sql title="M4: first/last/min/max envelope in up to 1,920 rows" demo SELECT timestamp, price FROM fx_trades WHERE symbol = 'EURUSD' SUBSAMPLE m4(price, 1920) ``` -```questdb-sql title="MinMax: lightweight envelope at half the output of M4" demo +```questdb-sql title="MinMax: min/max envelope in up to 500 rows" demo SELECT timestamp, price FROM fx_trades WHERE symbol = 'EURUSD' @@ -395,11 +458,13 @@ complement each other: aggregate first, then reduce for display. ### Multiple columns pass through -Because `SUBSAMPLE` selects real rows rather than computing new ones, every -column in the output carries its original value from the source table. In -the query below, `side` and `quantity` are not involved in the downsampling -decision, but each output row is a real trade with the actual side and -quantity that occurred at that timestamp. +Because `SUBSAMPLE` selects existing rows rather than computing new ones, +every selected row retains all the values of its immediate input row. The +query below reads directly from a table, so although `side` and `quantity` +are not involved in the downsampling decision, each output row is a trade +with the side and quantity recorded at that timestamp. When the input is an +aggregation, a join, or a computed projection, the pass-through values are +those of the derived row, not of a physical source record. ```questdb-sql title="LTTB selects rows by price; all columns emit" demo SELECT timestamp, symbol, side, price, quantity @@ -464,11 +529,16 @@ SELECT count() FROM ( ## Behavior notes -- If the input has fewer rows than the target, all rows are returned unchanged. -- Output rows are always in timestamp-ascending order. +- For the target-based methods (`lttb`, `minmax`, `m4`, `uniform`), if the + input has fewer eligible rows than the target, all of them are returned + unchanged. `cadence` uses a stride rather than a target. +- Selected rows are returned in the order of the incoming query, not + necessarily in timestamp-ascending order. See + [output order](#output-order). - All columns from the `SELECT` clause pass through for selected rows. -- `SUBSAMPLE` works with `WHERE`, `SAMPLE BY`, `GROUP BY`, CTEs, subqueries, - `ORDER BY`, and `LIMIT`. +- `SUBSAMPLE` works with `WHERE`, `SAMPLE BY`, `GROUP BY`, `PIVOT`, joins, + `UNION`, CTEs, subqueries, window functions, `ORDER BY`, and `LIMIT`. +- A final `ORDER BY` and `LIMIT` operate on the selected rows. - `SUBSAMPLE` inside a parenthesized subquery applies inside that subquery, not the outer query. @@ -476,12 +546,21 @@ SELECT count() FROM ( | Property | Default | Description | |----------|---------|-------------| -| `cairo.sql.subsample.max.rows` | 100,000,000 | Maximum input rows SUBSAMPLE will buffer. Exceeding this limit returns an error. | +| `cairo.sql.subsample.max.rows` | 100,000,000 | Maximum number of input rows `SUBSAMPLE` accepts. Exceeding this limit returns an error. | + +The limit counts every input row, including rows that `lttb`, `m4`, or +`minmax` skip because of a `NULL` or non-finite value. It is independent of +the `targetPoints` maximum. + +Memory use depends on the method: + +- `uniform` and `cadence` count the input rows and store only the positions + of the selected rows. They do not buffer a timestamp/value pair per row. +- `lttb`, `m4`, and `minmax` buffer a 16-byte timestamp/value entry for each + eligible row, plus bookkeeping for skipped rows and selected positions. -`SUBSAMPLE` buffers its entire input before running the algorithm. For direct -table scans, memory usage is 24 bytes per row. For queries involving -`SAMPLE BY`, `GROUP BY`, or subqueries, memory also scales with the projected -row width. At the default limit, the base buffer is approximately 2.4 GB. +Depending on input order and query shape, the query can need additional row +or sort storage, so no single bytes-per-row figure describes a whole query. ## See also diff --git a/scripts/gen_subsample_svgs.py b/scripts/gen_subsample_svgs.py index c9cdd681d1..5d8b2b47a1 100644 --- a/scripts/gen_subsample_svgs.py +++ b/scripts/gen_subsample_svgs.py @@ -339,14 +339,23 @@ def gen_gap_no_detect(): def gen_gap_detect(): - """LTTB with gap detection - small gap connected, large gap preserved.""" + """LTTB with gap detection - two segments selected independently. + + The two polylines show how a client that breaks lines on timestamp gaps + renders the result. The SQL output has no break marker, so the title, + description, and legend all attribute the line break to the client. + """ im, ix, _, sg, bg, raw_pls, _, _ = _gap_helpers() g_ab_i = [0, 5, 10, 18, 22, 23] g_ab_v = [.50, .95, .50, .20, .40, .45] g_c_i = [48, 55, 58, 60, 65, 68] g_c_v = [.45, .75, .55, .15, .60, .55] - return f"""{hdr(W, H, "LTTB with gap detection", "Small gap connected, large gap preserved.")} -LTTB with gap threshold '6h': small gap connected, large gap preserved + desc = ("The small gap is below the threshold and stays connected. " + "Rows on each side of the large gap are selected as separate " + "segments. The line break between them is drawn by the client: " + "the SQL result contains no break marker.") + return f"""{hdr(W, H, "LTTB with gap detection, client-rendered gap break", desc)} +LTTB with gap threshold '6h': two segments, line break drawn by the client {bkl([bg], im, ix, YT, YB)} {raw_pls(YT, YB)} @@ -358,7 +367,7 @@ def gen_gap_detect(): Selected points (12) -Gap boundary +Gap boundary (client-rendered break) """ diff --git a/static/images/docs/subsample/gap-detect.svg b/static/images/docs/subsample/gap-detect.svg index 8afadec025..2280a7a9da 100644 --- a/static/images/docs/subsample/gap-detect.svg +++ b/static/images/docs/subsample/gap-detect.svg @@ -1,6 +1,6 @@ -LTTB with gap detection -Small gap connected, large gap preserved. +LTTB with gap detection, client-rendered gap break +The small gap is below the threshold and stays connected. Rows on each side of the large gap are selected as separate segments. The line break between them is drawn by the client: the SQL result contains no break marker. -LTTB with gap threshold '6h': small gap connected, large gap preserved +LTTB with gap threshold '6h': two segments, line break drawn by the client @@ -43,5 +43,5 @@ Selected points (12) -Gap boundary +Gap boundary (client-rendered break) \ No newline at end of file From 45e91d09e75a572a44f618952225e580c22ac275 Mon Sep 17 00:00:00 2001 From: javier Date: Mon, 21 Sep 2026 14:25:48 +0200 Subject: [PATCH 12/14] Document SDT in the SUBSAMPLE page Add the sdt(valueColumn, compdev) method: compression deviation tolerance, data-dependent output count, swinging-door mechanism, 2 * compdev reconstruction bound, NULL run boundaries, timestamp-gap behavior, query shape restrictions, and memory behavior. Add a static chart, a gap example chart, and the swinging-door animation. Show where SUBSAMPLE goes in a SELECT statement. State that cairo.sql.subsample.max.rows applies to lttb, m4, minmax, uniform, and cadence, and not to sdt. --- documentation/configuration/cairo-engine.md | 14 +- documentation/query/sql/subsample.md | 307 ++++++++++++++-- scripts/gen_subsample_svgs.py | 70 +++- static/images/docs/subsample/sdt-gap.svg | 44 +++ .../docs/subsample/sdt-swinging-door.svg | 345 ++++++++++++++++++ static/images/docs/subsample/sdt.svg | 39 ++ 6 files changed, 790 insertions(+), 29 deletions(-) create mode 100644 static/images/docs/subsample/sdt-gap.svg create mode 100644 static/images/docs/subsample/sdt-swinging-door.svg create mode 100644 static/images/docs/subsample/sdt.svg diff --git a/documentation/configuration/cairo-engine.md b/documentation/configuration/cairo-engine.md index 7d10089aa4..d4b1b2262c 100644 --- a/documentation/configuration/cairo-engine.md +++ b/documentation/configuration/cairo-engine.md @@ -875,11 +875,15 @@ SAMPLE BY index query page size (maximum values returned in a single scan). - **Default**: `100000000` - **Reloadable**: no -Maximum number of input rows a -[SUBSAMPLE](/docs/query/sql/subsample/) query accepts. Exceeding this limit -returns an error. Every input row counts, including rows that a value-based -method skips because of a `NULL` value. The limit is independent of the -`targetPoints` argument. Must be between 1 and 2,147,483,647. +Maximum number of input rows accepted by the count-based and stride-based +[SUBSAMPLE](/docs/query/sql/subsample/) methods: `lttb`, `m4`, `minmax`, +`uniform`, and `cadence`. Exceeding this limit returns an error. Every input +row counts, including rows that `lttb`, `m4`, or `minmax` skip because of a +`NULL` value. The limit is independent of the `targetPoints` argument. Must +be between 1 and 2,147,483,647. + +This limit does not apply to `sdt`, which remains subject to the query's +normal memory limits. ## Window functions diff --git a/documentation/query/sql/subsample.md b/documentation/query/sql/subsample.md index c1e90b12e0..bba9f5ea54 100644 --- a/documentation/query/sql/subsample.md +++ b/documentation/query/sql/subsample.md @@ -1,13 +1,15 @@ --- title: SUBSAMPLE keyword sidebar_label: SUBSAMPLE -description: SUBSAMPLE SQL keyword reference for time-series downsampling using the LTTB, M4, MinMax, uniform, and cadence algorithms. +description: SUBSAMPLE SQL keyword reference for time-series downsampling using the LTTB, M4, MinMax, uniform, cadence, and SDT algorithms. --- `SUBSAMPLE` reduces the number of rows in a query result while preserving the visual shape of the data. It selects the most representative points from a time-ordered dataset, making it ideal for rendering charts at screen resolution -without transferring millions of rows to the client. +without transferring millions of rows to the client. One method, `sdt`, +reduces rows to within a value tolerance instead of to a row budget, which +suits error-bounded telemetry compression. Unlike [SAMPLE BY](/docs/query/sql/sample-by/), which computes new aggregate values at synthetic bucket boundaries, `SUBSAMPLE` selects rows from its @@ -27,21 +29,47 @@ the designation. ## Syntax -```questdb-sql title="Value-based algorithms" +```questdb-sql +SELECT columns +FROM table +[WHERE conditions] +[LATEST ON ...] +[SAMPLE BY ... | GROUP BY ...] +[WINDOW ...] +SUBSAMPLE method(arguments) +[ORDER BY ...] +[LIMIT ...] +``` + +`SUBSAMPLE` goes after the `WHERE`, `LATEST ON`, `SAMPLE BY`, `GROUP BY`, +and `WINDOW` clauses, and before `ORDER BY` and `LIMIT`. The `SELECT` list +must include the designated timestamp, and for the value-based and +tolerance-based methods it must also include `valueColumn`. + +`method(arguments)` is one of: + +```questdb-sql title="Value-based methods" SUBSAMPLE lttb(valueColumn, targetPoints [, gapThreshold]) SUBSAMPLE { m4 | minmax }(valueColumn, targetPoints) ``` -```questdb-sql title="Position-based algorithms" +```questdb-sql title="Position-based methods" SUBSAMPLE uniform(targetPoints) SUBSAMPLE cadence(stride [, seed]) ``` +```questdb-sql title="Tolerance-based method" +SUBSAMPLE sdt(valueColumn, compdev) +``` + +`sdt` cannot share a query level with `SAMPLE BY`, `GROUP BY`, `DISTINCT`, +or a join. See [query shape restrictions](#query-shape-restrictions). + Where: - **`valueColumn`** - the numeric column used to decide which points are - visually significant. Required for `lttb`, `m4`, and `minmax`. Not used - by `uniform` or `cadence`. + visually significant. Required for `lttb`, `m4`, `minmax`, and `sdt`. Not + used by `uniform` or `cadence`. - **`targetPoints`** - target number of output rows. Supports integer literals, [DECLARE](/docs/query/sql/declare/) variables, and bind variables (`$1`). Must be at least 2. Maximum is 2,147,483,647. @@ -51,6 +79,10 @@ Where: [cadence](#cadence---every-nth-row). - **`gapThreshold`** - (`lttb` only) optional interval that enables gap-preserving mode. See [gap-preserving LTTB](#gap-preserving-lttb). +- **`compdev`** - (`sdt` only) the compression deviation: a constant, + finite, non-negative error tolerance in the units of `valueColumn`. This + is not an output count: the data decides how many rows `sdt` retains. See + [sdt](#sdt---swinging-door-trending). ### Execution order @@ -84,17 +116,24 @@ methods, a row is not eligible for selection when its value is `NULL` or non-finite, or when its timestamp is `NULL`. Rows skipped this way still count toward the [input row limit](#configuration). +The value column of `sdt` accepts the same numeric types and is compared as +`DOUBLE`. Unlike the three methods above, `sdt` does not skip a `NULL` or +non-finite value. It retains that row as a run boundary. See +[NULLs and run boundaries](#nulls-and-run-boundaries). + `uniform` and `cadence` take no value column. `NULL` values in any projected column do not prevent a row from being selected. ## Algorithms -Five algorithms are available. The first three (`lttb`, `minmax`, `m4`) -inspect values to decide which rows are visually significant. The last two +Six algorithms are available. The first three (`lttb`, `minmax`, `m4`) +inspect values to decide which rows are visually significant. The next two (`uniform`, `cadence`) ignore values and select rows purely by position. -They are useful when the input is dense or as a baseline. +They are useful when the input is dense or as a baseline. The last one +(`sdt`) also inspects values, but takes an error tolerance instead of a +target row count or a stride, so the data decides how many rows it keeps. -All five select existing rows from their input. No values are ever +All six select existing rows from their input. No values are ever interpolated or computed. The diagrams below use a 24-point series as input (think 24 hourly bars over one day): @@ -384,17 +423,209 @@ WHERE symbol = 'EURUSD' SUBSAMPLE cadence(1000, 42) ``` +### sdt - Swinging Door Trending + +Swinging Door Trending (SDT) is an error-bounded compression method. It +replaces a run of samples with a smaller set of retained samples whose +connecting line approximates the original values within a known bound. +Instead of a target row count or a stride, you supply `compdev`, short for +compression deviation, a tolerance in the units of the value column, and the +data decides how many rows are retained. A flat signal keeps few rows. A noisy signal, or a smaller +tolerance, keeps more. + +SDT suits historian, telemetry, and industrial-sensor workloads, where the +acceptable error is known in engineering units (for example, half a degree) +and the right number of points is not. + +![SDT downsampling](/images/docs/subsample/sdt.svg) + +On the same 24-point series, `sdt` with `compdev = 0.05` retains 10 rows. +That count was not requested: it is what the tolerance allows on this data. +Near-straight stretches, such as the climb from i=0 to i=4 and the recovery from +i=16 to i=22, collapse to their two endpoints, while the sharp turns around +the spike keep more points. Unlike `minmax` and `m4`, `sdt` does not pin +extremes. The trough at i=15 is dropped because the line from i=14 to i=16 +stays within `2 * compdev` of it. + +How it works: + +1. The first eligible sample is retained and becomes the current anchor. +2. Each later eligible sample constrains a lower and an upper permissible + slope from that anchor, `compdev` below and above the sample. +3. The intersection of those slope constraints forms a narrowing corridor, + the swinging door. +4. When a new sample makes the corridor empty, the previous eligible sample + is retained and becomes the next anchor. +5. Processing resumes from the new anchor. +6. The final eligible sample is retained when the input ends. + +The animation below steps through the mechanism on a separate 20-sample +series with `compdev = 0.05`. It retains 5 samples (0, 6, 8, 14, and 19), +and the reconstruction error is bounded by `2 * compdev = 0.10`. + +![SDT swinging door animation](/images/docs/subsample/sdt-swinging-door.svg) + +In the animation, the corridor closes when samples 7, 9, and 15 arrive, so +samples 6, 8, and 14 are retained. Sample 19 is retained because the input +ends there, not because a corridor closed. + +```questdb-sql title="Compress a temperature series to within a known error" +SELECT ts, temperature, device_id +FROM sensor_readings +SUBSAMPLE sdt(temperature, 0.5); +``` + +```questdb-sql title="One-pip tolerance on a tick series" demo +SELECT timestamp, price +FROM fx_trades +WHERE symbol = 'EURUSD' + AND timestamp IN '$today' +SUBSAMPLE sdt(price, 0.0001) +``` + +#### The compdev tolerance + +`compdev` means compression deviation. It is the value-domain tolerance +that `sdt` uses to constrain the swinging-door corridor. Because the +retained endpoints are original samples, the end-to-end reconstruction bound +is `2 * compdev`. See [error guarantee](#error-guarantee). + +- `compdev` must be a constant, finite, non-negative numeric expression. A + numeric literal is the normal form. A constant expression such as + `abs(-0.5)` and a [DECLARE](/docs/query/sql/declare/) variable holding a + constant also work. +- Bind variables and row-dependent expressions, such as a column reference, + are rejected. +- A negative, `NULL`, `NaN`, or infinite `compdev` is invalid. +- `compdev = 0` retains a point whenever finite-precision arithmetic finds a + departure from exact collinearity. It is not lossless compression of + arbitrary floating-point input. +- `valueColumn` must be a numeric column that appears directly in the + `SELECT` list. + +`compdev` controls fidelity, not row count. The output count is +data-dependent and there is no way to request a fixed number of points from +`sdt`. When you need a row budget, use `lttb`, `m4`, `minmax`, or `uniform`. + +#### Error guarantee + +For finite values and strictly increasing timestamps within an SDT run, +linear interpolation between consecutive retained samples differs from every +original eligible sample in that run by no more than `2 * compdev`, apart +from normal floating-point rounding. + +:::warning + +The bound is `2 * compdev`, not `compdev`. The corridor extends `compdev` on +each side, and the retained endpoints are original samples rather than +points shifted to the center of the corridor. With `compdev = 0.05`, the +maximum reconstruction error is `0.10`. To guarantee a maximum error of `E`, +use `compdev = E / 2`. + +::: + +Comparisons are conservative in floating-point arithmetic. Near a numerical +boundary, `sdt` can retain an extra point rather than risk violating the +bound. + +#### NULLs and run boundaries + +`sdt` handles ineligible values differently from `lttb`, `m4`, and `minmax`, +which skip them: + +- A row with a `NULL` or non-finite value is a hard boundary, and the row + itself is retained. +- The last eligible sample before the boundary is retained, as it would be + at the end of the input. +- The next eligible finite row starts a new run with a fresh anchor. +- A `NULL` designated timestamp also interrupts normal processing of finite + samples. + +The error guarantee applies within each run. + +#### Timestamp gaps + +A timestamp gap on its own is not a boundary. `sdt` uses the actual +timestamp distance in its slope calculations, so a long gap influences which +points the corridor retains, but `sdt` does not promise to retain both sides +of the gap. The result contains no `NULL` separator row and no segment +identifier. As with +[gap-preserving LTTB](#gap-preserving-lttb), a chart that must show a +discontinuity needs a timestamp-gap rule in the client. + +The example below has 41 samples at positions 0 to 19 and 40 to 60, with no +data in between: + +```text +(0, 0.50), (1, 0.55), (2, 0.60), (3, 0.65), (4, 0.70), +(5, 0.95), (6, 0.85), (7, 0.70), (8, 0.60), (9, 0.55), +(10, 0.50), (11, 0.45), (12, 0.40), (13, 0.35), (14, 0.28), +(15, 0.20), (16, 0.25), (17, 0.30), (18, 0.35), (19, 0.40), +(40, 0.45), (41, 0.50), (42, 0.55), (43, 0.58), (44, 0.60), +(45, 0.65), (46, 0.70), (47, 0.75), (48, 0.70), (49, 0.55), +(50, 0.40), (51, 0.25), (52, 0.15), (53, 0.25), (54, 0.40), +(55, 0.55), (56, 0.60), (57, 0.62), (58, 0.60), (59, 0.58), +(60, 0.55) +``` + +With `compdev = 0.05`, `sdt` retains 11 rows: + +```text +(0, 0.50), (4, 0.70), (5, 0.95), (9, 0.55), (15, 0.20), +(19, 0.40), (42, 0.55), (48, 0.70), (52, 0.15), (56, 0.60), +(60, 0.55) +``` + +![SDT across a timestamp gap](/images/docs/subsample/sdt-gap.svg) + +The jump from 19 to 40 does not create a boundary. Sample 19 is retained +because the corridor closes when sample 40 arrives, and sample 40 itself is +not retained: the first retained row after the gap is 42. The largest +reconstruction error in this example is 0.087, at sample 40. That is above +`compdev` and within the `2 * compdev = 0.10` bound. + +#### Query shape restrictions + +`sdt` accepts a narrower set of query shapes than the other five methods. +It is rejected when the same query level contains: + +- aggregate functions or `GROUP BY` +- `SAMPLE BY` +- `DISTINCT` +- a join + +Filters with `WHERE`, additional plain columns in the `SELECT` list, and a +final `ORDER BY` or `LIMIT` are all supported. To apply `sdt` to aggregated +data, compute the aggregation in a subquery or CTE and apply `sdt` outside +it: + +```questdb-sql title="Aggregate in a CTE, then apply SDT to the result" demo +WITH bars AS ( + SELECT timestamp, avg(price) avg_price + FROM fx_trades + WHERE symbol = 'EURUSD' + AND timestamp IN '$today' + SAMPLE BY 1m +) +SELECT timestamp, avg_price +FROM bars +SUBSAMPLE sdt(avg_price, 0.0001) +``` + ### Algorithm comparison -| Property | lttb | minmax | m4 | uniform | cadence | -|----------|------|--------|-----|---------|---------| -| Parameter | targetPoints | targetPoints | targetPoints | targetPoints | stride | -| Inspects values | Yes | Yes | Yes | No | No | -| Bucket type | Equal row count | Equal time intervals | Equal time intervals | Equal row spacing | Fixed row stride | -| Points per bucket | Exactly 1 | Up to 2 (min, max) | Up to 4 (first, last, min, max) | N/A | N/A | -| Output count | Exactly N when N or more eligible rows exist, otherwise all eligible rows. Gap mode can return fewer or more than N | Up to N | Up to N | Exactly N (or all rows if fewer) | ~rowCount/stride | -| Gap handling | Connects across. With a threshold, segments are selected independently; a line renderer may still bridge the gap | Empty buckets emit no rows; a line renderer may still bridge the gap | Empty buckets emit no rows; a line renderer may still bridge the gap | Connects across | Connects across | -| Best use case | Line charts | Value range overview | Dashboards, SLA | Dense uniform data | Decimation, anti-aliasing | +| Property | lttb | minmax | m4 | uniform | cadence | sdt | +|----------|------|--------|-----|---------|---------|-----| +| Parameter | targetPoints | targetPoints | targetPoints | targetPoints | stride | compdev (value tolerance) | +| Inspects values | Yes | Yes | Yes | No | No | Yes, as `DOUBLE` | +| Bucket type | Equal row count | Equal time intervals | Equal time intervals | Equal row spacing | Fixed row stride | None: adaptive swinging corridor | +| Points per bucket | Exactly 1 | Up to 2 (min, max) | Up to 4 (first, last, min, max) | N/A | N/A | N/A | +| Output count | Exactly N when N or more eligible rows exist, otherwise all eligible rows. Gap mode can return fewer or more than N | Up to N | Up to N | Exactly N (or all rows if fewer) | ~rowCount/stride | Data-dependent, no target | +| Error bound | None | None | None | None | None | Linear reconstruction within `2 * compdev`, for finite values with strictly increasing timestamps | +| Gap handling | Connects across. With a threshold, segments are selected independently; a line renderer may still bridge the gap | Empty buckets emit no rows; a line renderer may still bridge the gap | Empty buckets emit no rows; a line renderer may still bridge the gap | Connects across | Connects across | A gap is not a boundary and both sides are not guaranteed; no automatic visual break | +| `NULL` values | Skipped | Skipped | Skipped | Not inspected | Not inspected | Retained as run boundaries | +| Best use case | Line charts | Value range overview | Dashboards, SLA | Dense uniform data | Decimation, anti-aliasing | Error-bounded telemetry and historian compression | +| Row limit applies | Yes | Yes | Yes | Yes | Yes | No | ## Examples @@ -442,6 +673,13 @@ WHERE symbol = 'EURUSD' SUBSAMPLE cadence(1000) ``` +```questdb-sql title="SDT: error within 2 pips, row count decided by the data" demo +SELECT timestamp, price +FROM fx_trades +WHERE symbol = 'EURUSD' +SUBSAMPLE sdt(price, 0.0001) +``` + ### Composing with SAMPLE BY ```questdb-sql title="Aggregate to 1-minute bars, then downsample" demo @@ -456,6 +694,10 @@ SUBSAMPLE lttb(avg_price, 500) selects the most representative rows from that output. The two operations complement each other: aggregate first, then reduce for display. +`sdt` cannot share a query level with `SAMPLE BY`. Put the aggregation in a +subquery or CTE, as shown in +[query shape restrictions](#query-shape-restrictions). + ### Multiple columns pass through Because `SUBSAMPLE` selects existing rows rather than computing new ones, @@ -531,13 +773,16 @@ SELECT count() FROM ( - For the target-based methods (`lttb`, `minmax`, `m4`, `uniform`), if the input has fewer eligible rows than the target, all of them are returned - unchanged. `cadence` uses a stride rather than a target. + unchanged. `cadence` uses a stride rather than a target, and `sdt` uses a + tolerance: its output count is data-dependent. - Selected rows are returned in the order of the incoming query, not necessarily in timestamp-ascending order. See [output order](#output-order). - All columns from the `SELECT` clause pass through for selected rows. -- `SUBSAMPLE` works with `WHERE`, `SAMPLE BY`, `GROUP BY`, `PIVOT`, joins, - `UNION`, CTEs, subqueries, window functions, `ORDER BY`, and `LIMIT`. +- `lttb`, `minmax`, `m4`, `uniform`, and `cadence` work with `WHERE`, + `SAMPLE BY`, `GROUP BY`, `PIVOT`, joins, `UNION`, CTEs, subqueries, window + functions, `ORDER BY`, and `LIMIT`. `sdt` accepts fewer shapes. See + [query shape restrictions](#query-shape-restrictions). - A final `ORDER BY` and `LIMIT` operate on the selected rows. - `SUBSAMPLE` inside a parenthesized subquery applies inside that subquery, not the outer query. @@ -546,18 +791,24 @@ SELECT count() FROM ( | Property | Default | Description | |----------|---------|-------------| -| `cairo.sql.subsample.max.rows` | 100,000,000 | Maximum number of input rows `SUBSAMPLE` accepts. Exceeding this limit returns an error. | +| `cairo.sql.subsample.max.rows` | 100,000,000 | Maximum number of input rows accepted by the count-based and stride-based methods: `lttb`, `m4`, `minmax`, `uniform`, and `cadence`. Exceeding this limit returns an error. Does not apply to `sdt`. | The limit counts every input row, including rows that `lttb`, `m4`, or `minmax` skip because of a `NULL` or non-finite value. It is independent of the `targetPoints` maximum. +`sdt` is not governed by this limit. It remains subject to the query's +normal memory limits. + Memory use depends on the method: - `uniform` and `cadence` count the input rows and store only the positions of the selected rows. They do not buffer a timestamp/value pair per row. - `lttb`, `m4`, and `minmax` buffer a 16-byte timestamp/value entry for each eligible row, plus bookkeeping for skipped rows and selected positions. +- `sdt` keeps approximately one byte per input row for its keep flags. It + runs through the same two-pass window execution as the other methods, so + it is not a constant-memory streaming implementation. Depending on input order and query shape, the query can need additional row or sort storage, so no single bytes-per-row figure describes a whole query. @@ -573,3 +824,13 @@ or sort storage, so no single bytes-per-row figure describes a whole query. the original LTTB algorithm and thesis reference - [Jugel, U. et al. (2014). "M4: A Visualization-Oriented Time Series Data Aggregation"](https://www.vldb.org/pvldb/vol7/p797-jugel.pdf) - the M4 paper +- [Bristol, E. H. (1990). "Swinging Door Trending: Adaptive Trend Recording?"](https://cir.nii.ac.jp/crid/1574231875546173824) - + ISA National Conference Proceedings, pp. 749-754. The original SDT + description +- [Khan, M. A. et al. (2020). "Impacts of swinging door lossy compression of synchrophasor data"](https://doi.org/10.1016/j.ijepes.2020.106182) - + a peer-reviewed explanation of the slope corridor and the compression + deviation concept + +The SDT references are background only. They are not the normative +specification of QuestDB's implementation, whose behavior is described on +this page. diff --git a/scripts/gen_subsample_svgs.py b/scripts/gen_subsample_svgs.py index 5d8b2b47a1..8bf5dff048 100644 --- a/scripts/gen_subsample_svgs.py +++ b/scripts/gen_subsample_svgs.py @@ -371,6 +371,72 @@ def gen_gap_detect(): """ +def gen_sdt(): + N = len(SEG_A) + im, ix = 0, N - 1 + ri = list(range(N)) + # sdt(v, 0.05) on SEG_A, rows returned by QuestDB master 6f1d196f. + # The count is decided by the data, not requested. SDT does not pin + # extremes: the trough at i=15 (0.20) is dropped because the line from + # i=14 to i=16 passes within 2 * compdev of it. + si = [0, 4, 5, 9, 11, 12, 14, 16, 22, 23] + sv = [SEG_A[i] for i in si] + return f"""{hdr(W, H, "SDT downsampling", f"SDT with compdev 0.05 retains {len(si)} points from 24.")} +SDT: compdev 0.05, retained {len(si)} of 24 (count decided by the data) +{rpl(ri, SEG_A, im, ix, YT, YB)} + +{cd(si,sv,im,ix,YT,YB,GRAY)} + +Raw data + +Retained points ({len(si)} of 24) +""" + + +# SDT gap dataset: 41 points, i=0..19 and i=40..60, no data for i=20..39. +SDT_SEG_A_I = list(range(0, 20)) +SDT_SEG_A_V = [ + 0.50, 0.55, 0.60, 0.65, 0.70, 0.95, 0.85, 0.70, 0.60, 0.55, + 0.50, 0.45, 0.40, 0.35, 0.28, 0.20, 0.25, 0.30, 0.35, 0.40, +] +SDT_SEG_B_I = list(range(40, 61)) +SDT_SEG_B_V = [ + 0.45, 0.50, 0.55, 0.58, 0.60, 0.65, 0.70, 0.75, 0.70, 0.55, + 0.40, 0.25, 0.15, 0.25, 0.40, 0.55, 0.60, 0.62, 0.60, 0.58, 0.55, +] + + +def gen_sdt_gap(): + """SDT at compdev 0.05 across a timestamp gap. + + Retained rows come from QuestDB master 6f1d196f. The gap is not a run + boundary: i=19 is retained because the corridor closes when i=40 arrives, + and i=40 itself is not retained. The output is one connected line. + """ + im, ix = 0, 60 + si = [0, 4, 5, 9, 15, 19, 42, 48, 52, 56, 60] + raw = dict(zip(SDT_SEG_A_I + SDT_SEG_B_I, SDT_SEG_A_V + SDT_SEG_B_V)) + sv = [raw[i] for i in si] + total = len(raw) + skipped_x, skipped_y = xp(40, im, ix), yp(raw[40], YT, YB) + desc = (f"SDT retains {len(si)} of {total} points. The timestamp gap is not a boundary: " + "the first point after the gap is not retained and the line crosses the gap.") + return f"""{hdr(W, H, "SDT across a timestamp gap", desc)} +SDT, compdev 0.05: {total} points reduced to {len(si)}, the gap is not a boundary +{rpl(SDT_SEG_A_I, SDT_SEG_A_V, im, ix, YT, YB)} +{rpl(SDT_SEG_B_I, SDT_SEG_B_V, im, ix, YT, YB)} + +{cd(si,sv,im,ix,YT,YB,GRAY)} + + +Raw data + +Retained points ({len(si)} of {total}) + +First point after the gap, not retained +""" + + if __name__ == "__main__": os.makedirs(OUT_DIR, exist_ok=True) for name, fn in [("raw.svg", gen_raw), ("lttb.svg", gen_lttb), @@ -378,7 +444,9 @@ def gen_gap_detect(): ("uniform.svg", gen_uniform), ("cadence.svg", gen_cadence), ("gap-raw.svg", gen_gap_raw), ("gap-no-detect.svg", gen_gap_no_detect), - ("gap-detect.svg", gen_gap_detect)]: + ("gap-detect.svg", gen_gap_detect), + ("sdt.svg", gen_sdt), + ("sdt-gap.svg", gen_sdt_gap)]: path = os.path.join(OUT_DIR, name) with open(path, "w") as f: f.write(fn()) diff --git a/static/images/docs/subsample/sdt-gap.svg b/static/images/docs/subsample/sdt-gap.svg new file mode 100644 index 0000000000..81208fcc3b --- /dev/null +++ b/static/images/docs/subsample/sdt-gap.svg @@ -0,0 +1,44 @@ + +SDT across a timestamp gap +SDT retains 11 of 41 points. The timestamp gap is not a boundary: the first point after the gap is not retained and the line crosses the gap. + +SDT, compdev 0.05: 41 points reduced to 11, the gap is not a boundary + + + + + + + + + + + + + + + + +Raw data + +Retained points (11 of 41) + +First point after the gap, not retained + \ No newline at end of file diff --git a/static/images/docs/subsample/sdt-swinging-door.svg b/static/images/docs/subsample/sdt-swinging-door.svg new file mode 100644 index 0000000000..bfe439ad08 --- /dev/null +++ b/static/images/docs/subsample/sdt-swinging-door.svg @@ -0,0 +1,345 @@ +SDT swinging door animation +Animated illustration of the Swinging Door Trending algorithm walking through a 20-point time series. The doors narrow as each sample arrives, and a new anchor is placed whenever the doors would cross. Retained points are 0, 6, 8, 14, and 19 at compdev 0.05. + + + +SDT: swinging door compression, step by step +compdev = 0.05 • 20 samples → 5 retained • reconstruction error bounded by 2 × compdev = 0.10 + + + + + + +1.0 +0.0 + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + anchor + + + + anchor + + + + anchor + + + + anchor + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + Phase 1. + Anchor at point 0. Doors narrow as each sample lands inside the corridor. + + + Sample 7 closes the doors. + Retain point 6. New anchor at point 6, doors reopen. + + + Sample 9 closes the doors. + Retain point 8. New anchor at point 8, doors reopen for the descent. + + + Sample 15 closes the doors. + Retain point 14. New anchor at point 14, doors reopen for the recovery. + + + End of input. + Point 19 retained as the final sample. Five points retained in total. + + + + + Retained set at compdev 0.05: + points 0, 6, 8, 14, 19. Reconstruction error is bounded by 2 × compdev = 0.10. + + + + + + sample + + retained + + door bounds + + sample closes the doors + + + diff --git a/static/images/docs/subsample/sdt.svg b/static/images/docs/subsample/sdt.svg new file mode 100644 index 0000000000..2ee264b702 --- /dev/null +++ b/static/images/docs/subsample/sdt.svg @@ -0,0 +1,39 @@ + +SDT downsampling +SDT with compdev 0.05 retains 10 points from 24. + +SDT: compdev 0.05, retained 10 of 24 (count decided by the data) + + + + + + + + + + + + + +Raw data + +Retained points (10 of 24) + \ No newline at end of file From 0d51ab04514166d9449508dda5f68df4cbbcef33 Mon Sep 17 00:00:00 2001 From: javier Date: Mon, 21 Sep 2026 16:07:30 +0200 Subject: [PATCH 13/14] Document the SUBSAMPLE window functions Add a SUBSAMPLE window functions category to the window-function reference with cadence, lttb, m4, minmax, sdt, and uniform: signatures, the Boolean keep flag, argument validation, ordering, framing, partition, and NULL rules, and an outer-filter example each. Add a window-function form section to the SUBSAMPLE page explaining the two interfaces and when to use each. List the six functions in the window functions quick reference and correct the frame explanation. Rename the reference page's Examples section to Window function examples, keeping its anchor. --- .../functions/window-functions/overview.md | 8 +- .../functions/window-functions/reference.md | 388 +++++++++++++++++- documentation/query/sql/subsample.md | 78 ++++ 3 files changed, 470 insertions(+), 4 deletions(-) diff --git a/documentation/query/functions/window-functions/overview.md b/documentation/query/functions/window-functions/overview.md index 6869de65a4..1b4fcb98dc 100644 --- a/documentation/query/functions/window-functions/overview.md +++ b/documentation/query/functions/window-functions/overview.md @@ -44,6 +44,7 @@ Arithmetic operations on window functions (e.g., `sum(...) OVER (...) / sum(...) | Function | Description | Respects Frame | |----------|-------------|----------------| | [`avg()`](reference.md#avg) | Average value in window (also supports EMA and VWEMA) | Yes (standard) / No (EMA/VWEMA) | +| [`cadence()`](reference.md#cadence) | SUBSAMPLE keep flag: every Nth row | No (frame clause rejected) | | [`corr()`](reference.md#corr) | Pearson correlation coefficient | Yes | | [`count()`](reference.md#count) | Count rows or non-null values | Yes | | [`covar_pop()` / `covar_samp()`](reference.md#covariance) | Covariance between two columns | Yes | @@ -54,18 +55,23 @@ Arithmetic operations on window functions (e.g., `sum(...) OVER (...) / sum(...) | [`lag()`](reference.md#lag) | Value from previous row | No | | [`last_value()`](reference.md#last_value) | Last value in window | Yes | | [`lead()`](reference.md#lead) | Value from following row | No | +| [`lttb()`](reference.md#lttb) | SUBSAMPLE keep flag: Largest Triangle Three Buckets | No (frame clause rejected) | +| [`m4()`](reference.md#m4) | SUBSAMPLE keep flag: first, last, min, max per time bucket | No (frame clause rejected) | | [`max()`](reference.md#max) | Maximum value in window | Yes | | [`min()`](reference.md#min) | Minimum value in window | Yes | +| [`minmax()`](reference.md#minmax) | SUBSAMPLE keep flag: min and max per time bucket | No (frame clause rejected) | | [`nth_value()`](reference.md#nth_value) | N-th value in window | Yes | | [`ntile()`](reference.md#ntile) | Bucket number from 1 to N | No | | [`percent_rank()`](reference.md#percent_rank) | Relative rank (0 to 1) | No | | [`rank()`](reference.md#rank) | Rank with gaps for ties | No | | [`row_number()`](reference.md#row_number) | Sequential row number | No | +| [`sdt()`](reference.md#sdt) | SUBSAMPLE keep flag: Swinging Door Trending, error-bounded | No (frame clause rejected) | | [`stddev_pop()` / `stddev_samp()` / `stddev()`](reference.md#stddev) | Standard deviation (population or sample) | Yes | | [`sum()`](reference.md#sum) | Sum of values in window | Yes | +| [`uniform()`](reference.md#uniform) | SUBSAMPLE keep flag: evenly spaced rows | No (frame clause rejected) | | [`var_pop()` / `var_samp()` / `variance()`](reference.md#variance) | Variance (population or sample) | Yes | -**Respects Frame**: Functions marked "Yes" use the frame clause (`ROWS`/`RANGE BETWEEN`). Functions marked "No" operate on the entire partition regardless of frame specification. +**Respects Frame**: Functions marked "Yes" use the frame clause (`ROWS`/`RANGE BETWEEN`). Functions marked "No" either operate on the entire partition or reject explicit frame specifications. See each function reference for its behavior. ## When to use window functions diff --git a/documentation/query/functions/window-functions/reference.md b/documentation/query/functions/window-functions/reference.md index 75e70b286b..07c878a85f 100644 --- a/documentation/query/functions/window-functions/reference.md +++ b/documentation/query/functions/window-functions/reference.md @@ -1,8 +1,8 @@ --- title: Window Functions Reference sidebar_label: Function Reference -description: Complete reference for all window functions in QuestDB including avg, sum, ksum, count, stddev, variance, covariance, correlation, rank, dense_rank, percent_rank, ntile, cume_dist, row_number, lag, lead, nth_value, EMA, VWEMA, and more. -keywords: [window functions, avg, sum, ksum, count, stddev, stddev_pop, stddev_samp, var_pop, var_samp, variance, covar_pop, covar_samp, corr, correlation, rank, dense_rank, percent_rank, ntile, cume_dist, row_number, lag, lead, first_value, last_value, nth_value, min, max, ema, vwema, exponential moving average] +description: Complete reference for all window functions in QuestDB including avg, sum, ksum, count, stddev, variance, covariance, correlation, rank, dense_rank, percent_rank, ntile, cume_dist, row_number, lag, lead, nth_value, EMA, VWEMA, the SUBSAMPLE keep-flag functions lttb, m4, minmax, uniform, cadence, sdt, and more. +keywords: [subsample, lttb, m4, minmax, uniform, cadence, sdt, downsampling, window functions, avg, sum, ksum, count, stddev, stddev_pop, stddev_samp, var_pop, var_samp, variance, covar_pop, covar_samp, corr, correlation, rank, dense_rank, percent_rank, ntile, cume_dist, row_number, lag, lead, first_value, last_value, nth_value, min, max, ema, vwema, exponential moving average] --- This page provides detailed documentation for each window function. For an introduction to window functions and how they work, see the [Overview](overview.md). For syntax details on the `OVER` clause, see [OVER Clause Syntax](syntax.md). @@ -1251,7 +1251,388 @@ This example: --- -## Examples +## SUBSAMPLE window functions + +These functions expose the [SUBSAMPLE](/docs/query/sql/subsample/) +downsampling algorithms as window functions. They do not return the reduced +row set. Each one returns a `boolean` keep flag for every input row: `true` +means the row is selected, `false` means it is discarded. Filter on the flag +in an outer query to get the selected rows: + +```questdb-sql title="Filter on the keep flag in an outer query" +SELECT * +FROM ( + SELECT + ts, + price, + lttb(ts, price, 500) OVER (ORDER BY ts) AS keep + FROM trades +) +WHERE keep; +``` + +When the window uses the same ascending timestamp order as the clause form, +the rows flagged `true` are the rows the clause form returns. See +[window-function form](/docs/query/sql/subsample/#window-function-form) on +the SUBSAMPLE page for when to prefer each interface, and the +[algorithm sections](/docs/query/sql/subsample/#algorithms) for diagrams and +method selection. + +**Common rules:** + +- Window functions are not allowed in a `WHERE` clause at the same query + level. Filter on the flag from a subquery or CTE +- `ORDER BY` is required in the `OVER` clause +- `m4()`, `minmax()`, `lttb()`, and `sdt()` require an ascending `ORDER BY`. + `uniform()` and `cadence()` select by position in whatever window order + they are given. Order by the designated timestamp, ascending, to match + the clause form +- Explicit `ROWS` and `RANGE` framing is rejected +- `PARTITION BY` is supported by `sdt()` only. The other five functions + reject it +- `RESPECT NULLS` and `IGNORE NULLS` are supported by `sdt()` only +- The value-based functions take the timestamp as an explicit first + argument. The value argument can be an expression, such as `price * 2`, + which the clause form does not accept +- Targets, strides, and seeds accept an integer constant, a + [DECLARE](/docs/query/sql/declare/) variable, or a bind variable +- The surrounding query controls the final output order. Add an outer + `ORDER BY` when you need a specific order +- [`cairo.sql.subsample.max.rows`](/docs/configuration/cairo-engine/#cairosqlsubsamplemaxrows) + limits the input of every function except `sdt()` + +### cadence() {#cadence} + +Flags one row out of every `stride` rows, by position in the window order. + +**Syntax:** +```questdb-sql +cadence(stride [, seed]) OVER (ORDER BY ts) +``` + +**Arguments:** +- `stride`: Integer constant or bind variable, at least `1`. The step + distance between flagged rows, not an output count +- `seed` (optional): Integer constant, bind variable, or `NULL`. An integer + gives a reproducible random starting offset in `[0, stride)`. `NULL` gives + a fresh random offset on every run. Without `seed` the offset is `0` + +**Return value:** +- `boolean`. `true` for selected rows, `false` for the rest + +**Behavior:** +- The number of flagged rows depends on the input size, approximately + `rowCount / stride` +- When `stride` is greater than `1` and no larger than the input row count, + the first and last rows in window order are always flagged +- `cadence(1)` flags every row +- When `stride` exceeds the input row count, only the first row is flagged. + The last row is not pinned in this case +- No value column is inspected, so `NULL` and non-finite values in other + columns have no effect on the selection +- `ORDER BY` is required. `PARTITION BY` and `ROWS`/`RANGE` framing are + rejected + +**Example:** +```questdb-sql title="Every 1000th trade" demo +SELECT timestamp, price +FROM ( + SELECT + timestamp, + price, + cadence(1000) OVER (ORDER BY timestamp) AS keep + FROM fx_trades + WHERE symbol = 'EURUSD' + AND timestamp IN '$today' +) +WHERE keep; +``` + +See [cadence](/docs/query/sql/subsample/#cadence---every-nth-row) on the +SUBSAMPLE page for the algorithm, the diagram, and the anti-aliasing use of +`seed`. + +--- + +### lttb() {#lttb} + +Flags the rows chosen by the Largest Triangle Three Buckets algorithm, which +preserves the visual shape of a line chart. + +**Syntax:** +```questdb-sql +lttb(ts, value, target [, gapThreshold]) OVER (ORDER BY ts) +``` + +**Arguments:** +- `ts`: Timestamp of each row. It must be the ascending `ORDER BY` column +- `value`: Numeric column or expression used to measure triangle areas +- `target`: Integer constant or bind variable, at least `2`. The number of + rows to flag +- `gapThreshold` (optional): String constant such as `'30s'`, `'5m'`, `'1h'`, + or `'1d'`, greater than zero. Supported units are `s`, `m`, `h`, and `d`. + Enables gap-aware mode + +**Return value:** +- `boolean`. `true` for selected rows, `false` for the rest + +**Behavior:** +- Without `gapThreshold`, exactly `target` rows are flagged when at least + that many eligible rows exist. With fewer eligible rows, all of them are + flagged +- With `gapThreshold`, the input is split wherever consecutive timestamps + are further apart than the threshold, and each segment is selected + independently. `target` becomes a goal: the number of flagged rows can be + below or above it +- Gap-aware mode adds no `NULL` separator row and no segment identifier. + The result is still one flag per input row +- A row with a `NULL` or non-finite `value`, or a `NULL` timestamp, is not + eligible and is always flagged `false`. It still counts toward the input + row limit +- The `ORDER BY` must be ascending and must order by the `ts` argument. + `PARTITION BY` and `ROWS`/`RANGE` framing are rejected + +**Example:** +```questdb-sql title="500 representative points, with gaps over 1 hour kept apart" demo +SELECT timestamp, price +FROM ( + SELECT + timestamp, + price, + lttb(timestamp, price, 500, '1h') + OVER (ORDER BY timestamp) AS keep + FROM fx_trades + WHERE symbol = 'EURUSD' +) +WHERE keep; +``` + +See [lttb](/docs/query/sql/subsample/#lttb---largest-triangle-three-buckets) +and [gap-preserving LTTB](/docs/query/sql/subsample/#gap-preserving-lttb) on +the SUBSAMPLE page for the algorithm, the diagrams, and the client-side +rendering caveat for gaps. + +--- + +### m4() {#m4} + +Flags the first, last, minimum, and maximum rows of each M4 time bucket. + +**Syntax:** +```questdb-sql +m4(ts, value, target) OVER (ORDER BY ts) +``` + +**Arguments:** +- `ts`: Timestamp of each row. It must be the ascending `ORDER BY` column +- `value`: Numeric column or expression used to find the minimum and + maximum rows +- `target`: Integer constant or bind variable, at least `2`. The row budget: + the time range is divided into `target / 4` equal time buckets + +**Return value:** +- `boolean`. `true` for selected rows, `false` for the rest + +**Behavior:** +- Up to `target` rows are flagged: up to 4 per time bucket +- When several roles resolve to the same row, that row is flagged once, so + a bucket contributes between 1 and 4 rows +- Empty time buckets contribute no rows +- A row with a `NULL` or non-finite `value`, or a `NULL` timestamp, is not + eligible and is always flagged `false`. It still counts toward the input + row limit +- The `ORDER BY` must be ascending and must order by the `ts` argument. + `PARTITION BY` and `ROWS`/`RANGE` framing are rejected + +**Example:** +```questdb-sql title="First, last, min, and max per time bucket" demo +SELECT timestamp, price +FROM ( + SELECT + timestamp, + price, + m4(timestamp, price, 1920) OVER (ORDER BY timestamp) AS keep + FROM fx_trades + WHERE symbol = 'EURUSD' +) +WHERE keep; +``` + +See [m4](/docs/query/sql/subsample/#m4---minmaxfirstlast-per-time-interval) +on the SUBSAMPLE page for the algorithm, the diagram, and how to size +`target` for a pixel width. + +--- + +### minmax() {#minmax} + +Flags the minimum and maximum rows of each non-empty time bucket. + +**Syntax:** +```questdb-sql +minmax(ts, value, target) OVER (ORDER BY ts) +``` + +**Arguments:** +- `ts`: Timestamp of each row. It must be the ascending `ORDER BY` column +- `value`: Numeric column or expression used to find the minimum and + maximum rows +- `target`: Integer constant or bind variable, at least `2`. The row budget: + the time range is divided into `target / 2` equal time buckets + +**Return value:** +- `boolean`. `true` for selected rows, `false` for the rest + +**Behavior:** +- Up to `target` rows are flagged: up to 2 per time bucket +- When the minimum and the maximum resolve to the same row, that row is + flagged once +- Empty time buckets contribute no rows +- A row with a `NULL` or non-finite `value`, or a `NULL` timestamp, is not + eligible and is always flagged `false`. It still counts toward the input + row limit +- The `ORDER BY` must be ascending and must order by the `ts` argument. + `PARTITION BY` and `ROWS`/`RANGE` framing are rejected + +**Example:** +```questdb-sql title="Min/max envelope" demo +SELECT timestamp, price +FROM ( + SELECT + timestamp, + price, + minmax(timestamp, price, 500) OVER (ORDER BY timestamp) AS keep + FROM fx_trades + WHERE symbol = 'EURUSD' +) +WHERE keep; +``` + +See [minmax](/docs/query/sql/subsample/#minmax---minmax-per-time-interval) +on the SUBSAMPLE page for the algorithm and the diagram. + +--- + +### sdt() {#sdt} + +Flags the rows retained by Swinging Door Trending, an error-bounded +compression method. The number of flagged rows is decided by the data, not +requested. + +**Syntax:** +```questdb-sql +sdt(ts, value, compdev) + [RESPECT NULLS | IGNORE NULLS] + OVER ([PARTITION BY columns] ORDER BY ts) +``` + +**Arguments:** +- `ts`: Timestamp of each row, used for the slope calculations +- `value`: Numeric column or expression, compared as `double` +- `compdev`: The compression deviation. A constant, finite, non-negative + tolerance in the units of `value`. A numeric literal, a constant + expression such as `abs(-0.5)`, and a `DECLARE` variable holding a + constant are accepted. Bind variables, row-dependent expressions, + negative values, `NULL`, `NaN`, and infinity are rejected + +**Return value:** +- `boolean`. `true` for retained rows, `false` for the rest + +**Behavior:** +- Within a run, linear interpolation between consecutive retained rows + differs from every original eligible row by no more than `2 * compdev`, + not `compdev` +- With `RESPECT NULLS`, the default, a row with a `NULL` or non-finite + `value` is a run boundary. That row is flagged `true`, the last eligible + row before it is flagged `true`, and the next eligible row starts a new + run +- With `IGNORE NULLS`, a row with a `NULL` or non-finite `value` is flagged + `false` and the current run continues across it +- A timestamp that does not increase, including an equal timestamp, is also + a run boundary +- A timestamp gap on its own is not a boundary +- With `PARTITION BY`, every partition keeps its own anchor and corridor + state, so each series is compressed independently +- The `ORDER BY` must be ascending. `ROWS`/`RANGE` framing is rejected +- `sdt()` is not governed by `cairo.sql.subsample.max.rows`. It remains + subject to the query's normal memory limits + +**Example:** +```questdb-sql title="Compress every symbol independently, within 2 pips" demo +SELECT timestamp, symbol, price +FROM ( + SELECT + timestamp, + symbol, + price, + sdt(timestamp, price, 0.0001) + OVER (PARTITION BY symbol ORDER BY timestamp) AS keep + FROM fx_trades + WHERE timestamp IN '$today' +) +WHERE keep; +``` + +See [sdt](/docs/query/sql/subsample/#sdt---swinging-door-trending) on the +SUBSAMPLE page for the swinging-door mechanism, the animation, the +[error guarantee](/docs/query/sql/subsample/#error-guarantee), and +[timestamp-gap behavior](/docs/query/sql/subsample/#timestamp-gaps). + +--- + +### uniform() {#uniform} + +Flags a target number of evenly spaced rows, by position in the window +order. It does not inspect a value column. + +**Syntax:** +```questdb-sql +uniform(target) OVER (ORDER BY ts) +``` + +**Arguments:** +- `target`: Integer constant or bind variable, at least `2`. The number of + rows to flag + +**Return value:** +- `boolean`. `true` for selected rows, `false` for the rest + +**Behavior:** +- When the input has more rows than `target`, exactly `target` rows are + flagged. Otherwise every row is flagged +- The first and last rows in window order are always flagged. Interior + positions round half up, so the selection is deterministic +- No value column is inspected, so `NULL` and non-finite values in other + columns have no effect on the selection +- `ORDER BY` is required. `PARTITION BY` and `ROWS`/`RANGE` framing are + rejected + +**Example:** +```questdb-sql title="500 evenly spaced trades" demo +SELECT timestamp, price +FROM ( + SELECT + timestamp, + price, + uniform(500) OVER (ORDER BY timestamp) AS keep + FROM fx_trades + WHERE symbol = 'EURUSD' + AND timestamp IN '$today' +) +WHERE keep; +``` + +See [uniform](/docs/query/sql/subsample/#uniform---evenly-spaced-rows) on +the SUBSAMPLE page for the algorithm and the diagram. + +--- + +## Window function examples {#examples} + +These examples combine the aggregate, ranking, and offset functions from the +first three categories. Each +[SUBSAMPLE window function](#subsample-window-functions) has its own example +in its section. ### Moving average of best bid price @@ -1326,5 +1707,6 @@ WINDOW w AS (ORDER BY timestamp RANGE BETWEEN 60000000 PRECEDING AND CURRENT ROW - The order of rows in the result set is not guaranteed to be consistent across query executions. Use an `ORDER BY` clause outside the `OVER` clause to ensure consistent ordering. - Ranking functions (`row_number`, `rank`, `dense_rank`, `percent_rank`, `cume_dist`, `ntile`) and offset functions (`lag`, `lead`) ignore frame specifications. +- SUBSAMPLE window functions (`cadence`, `lttb`, `m4`, `minmax`, `sdt`, `uniform`) reject explicit `ROWS` and `RANGE` frame specifications instead of ignoring them. - For time-based calculations, consider using `RANGE` frames with timestamp columns. - Aggregate window functions (`avg`, `sum`, `ksum`, `count`, `min`, `max`) support numeric types: `short`, `int`, `long`, `float`, `double`. The `decimal` type is not supported. diff --git a/documentation/query/sql/subsample.md b/documentation/query/sql/subsample.md index bba9f5ea54..51e4bca080 100644 --- a/documentation/query/sql/subsample.md +++ b/documentation/query/sql/subsample.md @@ -27,6 +27,10 @@ list must preserve it. A table with a designated timestamp is not enough if the projection omits that column or replaces it with an expression that loses the designation. +Every method is also available as a window function that returns a keep flag +for each row instead of the reduced row set. See +[window-function form](#window-function-form). + ## Syntax ```questdb-sql @@ -612,6 +616,11 @@ FROM bars SUBSAMPLE sdt(avg_price, 0.0001) ``` +The clause form treats its input as a single series. To compress several +series independently in one query, use the +[`sdt()` window function](/docs/query/functions/window-functions/reference/#sdt) +with `PARTITION BY`. + ### Algorithm comparison | Property | lttb | minmax | m4 | uniform | cadence | sdt | @@ -769,6 +778,75 @@ SELECT count() FROM ( ) ``` +## Window-function form + +`SUBSAMPLE` has two interfaces: + +- The **clause form**, such as `SUBSAMPLE lttb(price, 500)`, directly + returns the selected rows. +- The **window-function form**, such as + `lttb(ts, price, 500) OVER (ORDER BY ts)`, returns one `BOOLEAN` keep flag + for every input row. `true` means the row is selected and `false` means + it is discarded. + +Both forms select existing rows. When the window uses the same ascending +timestamp order as the clause form, the rows flagged `true` are the rows the +clause form returns. Neither form interpolates values or creates replacement +rows. + +```questdb-sql title="Filter on the keep flag in an outer query" +SELECT * +FROM ( + SELECT + ts, + price, + lttb(ts, price, 500) OVER (ORDER BY ts) AS keep + FROM trades +) +WHERE keep; +``` + +Window functions cannot be used directly in a `WHERE` clause at the same +query level, so filtering on the keep flag needs a subquery or a CTE. The +value-based functions take the timestamp as an explicit first argument, so +their argument order differs from the clause form: + +| Clause form | Window-function form | +|-------------|----------------------| +| `SUBSAMPLE lttb(value, target [, gapThreshold])` | [`lttb(ts, value, target [, gapThreshold]) OVER (ORDER BY ts)`](/docs/query/functions/window-functions/reference/#lttb) | +| `SUBSAMPLE m4(value, target)` | [`m4(ts, value, target) OVER (ORDER BY ts)`](/docs/query/functions/window-functions/reference/#m4) | +| `SUBSAMPLE minmax(value, target)` | [`minmax(ts, value, target) OVER (ORDER BY ts)`](/docs/query/functions/window-functions/reference/#minmax) | +| `SUBSAMPLE uniform(target)` | [`uniform(target) OVER (ORDER BY ts)`](/docs/query/functions/window-functions/reference/#uniform) | +| `SUBSAMPLE cadence(stride [, seed])` | [`cadence(stride [, seed]) OVER (ORDER BY ts)`](/docs/query/functions/window-functions/reference/#cadence) | +| `SUBSAMPLE sdt(value, compdev)` | [`sdt(ts, value, compdev) OVER (ORDER BY ts)`](/docs/query/functions/window-functions/reference/#sdt) | + +### When to use which form + +Prefer the clause form when you simply want the reduced row set, for +charting or to cut the size of a result. It is shorter and clearer. + +Prefer the window-function form when the keep or drop decision must be: + +- exposed as a column, for example to inspect or debug a selection +- composed with other window calculations in the same query +- filtered at another query level +- computed per series with `PARTITION BY`, which only + [`sdt()`](/docs/query/functions/window-functions/reference/#sdt) supports + +| Need | Preferred form | +|------|----------------| +| Return only the downsampled rows | Clause form: `SUBSAMPLE ...` | +| Keep the selection decision as a column | Window form: `... OVER (...) AS keep` | +| Filter the decision in another query level | Window form inside a subquery or CTE | +| Straightforward chart downsampling | Clause form | + +The window-function form also lifts two clause-form restrictions. The value +argument can be an expression instead of a directly selected column, and +`sdt()` can run over several series at once with `PARTITION BY`. See +[SUBSAMPLE window functions](/docs/query/functions/window-functions/reference/#subsample-window-functions) +for the signatures, ordering, framing, partition, and `NULL` rules of each +function. + ## Behavior notes - For the target-based methods (`lttb`, `minmax`, `m4`, `uniform`), if the From ba17488bda1601830ddce8a6b2171d6b0051116d Mon Sep 17 00:00:00 2001 From: javier Date: Mon, 21 Sep 2026 16:47:56 +0200 Subject: [PATCH 14/14] Add SUBSAMPLE changelog entry and make examples demo-ready Add September 2026 changelog entries for the SUBSAMPLE page, the SUBSAMPLE window functions category, and cairo.sql.subsample.max.rows. Switch the outer-filter pattern to fx_trades with the demo column names and tag it as a demo query. Link the SUBSAMPLE page to the configuration reference instead of repeating the property's default, and link to the window functions section from See also. --- documentation/changelog.mdx | 6 +++++ .../functions/window-functions/reference.md | 10 +++++--- documentation/query/sql/subsample.md | 25 +++++++++++++------ 3 files changed, 29 insertions(+), 12 deletions(-) diff --git a/documentation/changelog.mdx b/documentation/changelog.mdx index 7ae494d1f4..94f165ad07 100644 --- a/documentation/changelog.mdx +++ b/documentation/changelog.mdx @@ -21,6 +21,12 @@ This page tracks significant updates to the QuestDB documentation. - [ALTER GROUP](/docs/query/sql/acl/alter-group/) - New reference page covering `SET MEMORY LIMIT` and external alias mapping - [Migrate QuestDB onto the Kubernetes Operator](/docs/enterprise-kubernetes-operator/getting-started/migrate/) - Move an existing Enterprise deployment onto the Operator with a replica-first cutover: restore the source backup, consume replication WAL, then promote after a controlled source drain - [Copy a schema to another instance](/docs/cookbook/operations/copy-schema-between-instances/) - Recreate one instance's tables, views, and materialized views on another from `SHOW CREATE DATABASE`, either by replaying the statements over the REST API or by dumping them to a `.sql` file, with `INCLUDE (SCHEMA)` and `INCLUDE (ACL)` for separating structure from permissions on Enterprise, plus the Web Console schema explorer as a manual alternative and the ordering caveat that comes with it +- [SUBSAMPLE](/docs/query/sql/subsample/) - New SQL keyword that reduces a query result to a subset of its existing rows, without interpolating values. Covers the six methods with a diagram each: `lttb` for line charts, with a gap-preserving mode, `minmax` and `m4` for per-time-bucket envelopes, `uniform` and `cadence` for position-based selection, and `sdt` (Swinging Door Trending) for error-bounded compression with a `2 * compdev` reconstruction bound and an animated walkthrough. Also covers where the clause goes in a query, output order, `NULL` handling, and which query shapes each method accepts +- [SUBSAMPLE window functions](/docs/query/functions/window-functions/reference/#subsample-window-functions) - New category with `cadence()`, `lttb()`, `m4()`, `minmax()`, `sdt()`, and `uniform()`, which return a Boolean keep flag per row instead of the reduced row set. Covers the outer-query filter pattern, the ordering, framing, and `NULL` rules of each function, `PARTITION BY` and `IGNORE NULLS` support on `sdt()`, and [when to use each form](/docs/query/sql/subsample/#window-function-form) + +### Reference + +- Added [`cairo.sql.subsample.max.rows`](/docs/configuration/cairo-engine/#cairosqlsubsamplemaxrows), the input row limit for the `lttb`, `m4`, `minmax`, `uniform`, and `cadence` methods of `SUBSAMPLE`. It does not apply to `sdt` ### Updated diff --git a/documentation/query/functions/window-functions/reference.md b/documentation/query/functions/window-functions/reference.md index 07c878a85f..4ef90af291 100644 --- a/documentation/query/functions/window-functions/reference.md +++ b/documentation/query/functions/window-functions/reference.md @@ -1259,14 +1259,16 @@ row set. Each one returns a `boolean` keep flag for every input row: `true` means the row is selected, `false` means it is discarded. Filter on the flag in an outer query to get the selected rows: -```questdb-sql title="Filter on the keep flag in an outer query" +```questdb-sql title="Filter on the keep flag in an outer query" demo SELECT * FROM ( SELECT - ts, + timestamp, price, - lttb(ts, price, 500) OVER (ORDER BY ts) AS keep - FROM trades + lttb(timestamp, price, 500) OVER (ORDER BY timestamp) AS keep + FROM fx_trades + WHERE symbol = 'EURUSD' + AND timestamp IN '$today' ) WHERE keep; ``` diff --git a/documentation/query/sql/subsample.md b/documentation/query/sql/subsample.md index 51e4bca080..35c32bf152 100644 --- a/documentation/query/sql/subsample.md +++ b/documentation/query/sql/subsample.md @@ -420,7 +420,7 @@ WHERE symbol = 'EURUSD' SUBSAMPLE cadence(1000) ``` -```questdb-sql title="Anti-aliasing with reproducible seed" +```questdb-sql title="Anti-aliasing with reproducible seed" demo SELECT timestamp, price FROM fx_trades WHERE symbol = 'EURUSD' @@ -794,14 +794,16 @@ timestamp order as the clause form, the rows flagged `true` are the rows the clause form returns. Neither form interpolates values or creates replacement rows. -```questdb-sql title="Filter on the keep flag in an outer query" +```questdb-sql title="Filter on the keep flag in an outer query" demo SELECT * FROM ( SELECT - ts, + timestamp, price, - lttb(ts, price, 500) OVER (ORDER BY ts) AS keep - FROM trades + lttb(timestamp, price, 500) OVER (ORDER BY timestamp) AS keep + FROM fx_trades + WHERE symbol = 'EURUSD' + AND timestamp IN '$today' ) WHERE keep; ``` @@ -867,9 +869,11 @@ function. ## Configuration -| Property | Default | Description | -|----------|---------|-------------| -| `cairo.sql.subsample.max.rows` | 100,000,000 | Maximum number of input rows accepted by the count-based and stride-based methods: `lttb`, `m4`, `minmax`, `uniform`, and `cadence`. Exceeding this limit returns an error. Does not apply to `sdt`. | +[`cairo.sql.subsample.max.rows`](/docs/configuration/cairo-engine/#cairosqlsubsamplemaxrows) +caps the number of input rows that `lttb`, `m4`, `minmax`, `uniform`, and +`cadence` accept, in both the clause form and the window-function form. A +query that exceeds it returns an error. See the configuration reference for +the default and the valid range. The limit counts every input row, including rows that `lttb`, `m4`, or `minmax` skip because of a `NULL` or non-finite value. It is independent of @@ -878,6 +882,8 @@ the `targetPoints` maximum. `sdt` is not governed by this limit. It remains subject to the query's normal memory limits. +### Memory use + Memory use depends on the method: - `uniform` and `cadence` count the input rows and store only the positions @@ -896,6 +902,9 @@ or sort storage, so no single bytes-per-row figure describes a whole query. - [SAMPLE BY](/docs/query/sql/sample-by/) - time-based aggregation (computes new values at bucket boundaries, while `SUBSAMPLE` selects existing rows) +- [SUBSAMPLE window functions](/docs/query/functions/window-functions/reference/#subsample-window-functions) - + the same six algorithms as window functions that return a keep flag per + row - [Designated timestamp](/docs/concepts/designated-timestamp/) - required for `SUBSAMPLE` to operate - [Steinarsson, S. (2013). "Downsampling Time Series for Visual Representation"](https://github.com/sveinn-steinarsson/flot-downsample) -