From 37a825e0cc8d242c935d65e7f86c9455e1fcb73a Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 23:14:32 +0000 Subject: [PATCH 1/4] fix(core,driver-memory,service-analytics): every platform fold adds sum / avg with the one compensated sum `compensatedSum` moves from objectql's rows path to `@objectstack/core` as a root export, as `bucketDateKey` did, and the three folds that still added naively call it: driver-memory's native `sum` / `avg` arm, its analytics face (a `$group` `$accumulator` in place of mingo's `$sum` / `$avg`), and service-analytics' draft preview. objectql imports the helper instead of keeping its own copy. Claude-Session: https://claude.ai/code/session_01DEvba2nBuD4tWzfq8r8NFY Co-authored-by: Claude --- packages/core/src/index.ts | 8 ++ packages/core/src/utils/compensated-sum.ts | 60 ++++++++++++ .../driver-memory/src/memory-analytics.ts | 92 ++++++++++++++++++- .../driver-memory/src/memory-driver.ts | 13 ++- .../objectql/src/in-memory-aggregation.ts | 42 +-------- .../src/preview-evaluator.ts | 11 ++- 6 files changed, 180 insertions(+), 46 deletions(-) create mode 100644 packages/core/src/utils/compensated-sum.ts diff --git a/packages/core/src/index.ts b/packages/core/src/index.ts index 61f02b9828f..1888aa9f931 100644 --- a/packages/core/src/index.ts +++ b/packages/core/src/index.ts @@ -49,6 +49,14 @@ export * from './utils/env.js'; // Export timezone-aware calendar utilities (ADR-0053 Phase 2) export * from './utils/datetime.js'; +// [#20544] The ONE compensated fold for `sum` / `avg`, hoisted from +// `@objectstack/objectql`'s rows path for the reason `bucketDateKey` above +// was: `driver-memory`'s two faces and `service-analytics`' draft preview add +// a group's values in JavaScript too, neither package has objectql among its +// runtime dependencies, and a copy per face is how one `sum` came to answer +// two doubles. +export * from './utils/compensated-sum.js'; + // Export the shared batched-write helper (framework#2678) export * from './utils/bulk-write.js'; diff --git a/packages/core/src/utils/compensated-sum.ts b/packages/core/src/utils/compensated-sum.ts new file mode 100644 index 00000000000..30e811c7004 --- /dev/null +++ b/packages/core/src/utils/compensated-sum.ts @@ -0,0 +1,60 @@ +// Copyright (c) 2026 ObjectStack. Licensed under the Apache-2.0 license. + +/** + * [#20489, #20544] The sum of `nums`, added in order with + * Kahan-Babuska-Neumaier compensation — the summation SQLite (3.43 and later) + * uses for its own `sum` and `avg`, transcribed from its + * `kahanBabuskaNeumaierStep` and the finalizers' overflow guard. + * + * Why: a naive fold (`reduce((a, b) => a + b, 0)`) and SQLite answer two + * different doubles for the same values. A `number` column holding `0.1`, + * `0.2` and `0.3` sums to `0.6` on SQLite and to `0.6000000000000001` naively + * (`avg` `0.19999999999999998` against `0.20000000000000004`), so + * `having { s: { $eq: 0.6 } }` kept a group on one face and dropped it on + * another. Compensated, the faces agree, and the answer is the more accurate + * one (`1e16 + 1 - 1e16` is `1`, not `0`). + * + * ## Why it lives here + * + * Every platform fold that adds a group's values in JavaScript calls this one + * function, so a `sum` is the same double on every face the platform owns: + * + * - `@objectstack/objectql`'s rows path (`in-memory-aggregation.ts`, the fold + * `engine.aggregate` runs itself), where it was written; + * - `@objectstack/driver-memory`'s native `aggregate` (`memory-driver.ts`, + * `computeAggregate`) and its analytics face (`memory-analytics.ts`, the + * `sum` / `avg` measure accumulator); + * - `@objectstack/service-analytics`' draft preview (`preview-evaluator.ts`, + * the `sum` / `avg` arms). + * + * Neither `driver-memory` nor `service-analytics` has objectql among its + * runtime dependencies, and this is the package all three already stand on — + * the same reason `bucketDateKey` lives here. A second transcription is how a + * face comes to answer its own double again. + * + * ## What does not move + * + * Two addends (the compensated `a + b` IS the naive one), integers whose + * partial sums stay within 2^53 (every addition is exact), and a non-finite + * total. `s` below is exactly the naive running sum; once it overflows or + * meets a NaN, the error term is non-finite and the naive answer is returned + * as it was, which is SQLite's rule too. An empty list sums to `0`. Which + * values count as addends, and what an empty group answers, stay each + * caller's own rule: this function only adds. + * + * ⚠️ Residual, stated: PostgreSQL and MySQL add their doubles natively without + * compensation, and the platform does not wrap that arithmetic, so over three + * or more fractions their native path can still differ from this one in the + * last place. An exact `$eq` on a fractional sum compares doubles; compare + * with a range. + */ +export function compensatedSum(nums: readonly number[]): number { + let s = 0; + let c = 0; + for (const r of nums) { + const t = s + r; + c += Math.abs(s) > Math.abs(r) ? (s - t) + r : (r - t) + s; + s = t; + } + return Number.isFinite(c) ? s + c : s; +} diff --git a/packages/drivers/driver-memory/src/memory-analytics.ts b/packages/drivers/driver-memory/src/memory-analytics.ts index 014497594cc..376acdfa4f6 100644 --- a/packages/drivers/driver-memory/src/memory-analytics.ts +++ b/packages/drivers/driver-memory/src/memory-analytics.ts @@ -28,6 +28,10 @@ import { bucketDateKey, isBucketGranularity, type BucketGranularity, + // [#20544] The ONE compensated fold for `sum` / `avg`, the one objectql's + // rows path, this package's data face and SQLite add with — see + // {@link compensatedAddendAccumulator}. + compensatedSum, } from '@objectstack/core'; // [#16178] The pipeline below is split at its `$group` when a time dimension // buckets, so the bucket key can be folded in JS between the two halves — mingo @@ -647,6 +651,80 @@ function numericAggregandExpr(path: string): Record { return { $cond: [{ $eq: [{ $type: path }, 'bool'] }, { $cond: [path, 1, 0] }, path] }; } +/** + * [#20544] A `sum` or `avg` measure as ONE `$group` accumulator that adds with + * `@objectstack/core`'s {@link compensatedSum} — the fold objectql's rows path, + * this package's data face (`memory-driver.ts`, `computeAggregate`) and SQLite + * add with. + * + * ## What it replaced + * + * mingo's `$sum` and `$avg` add in a plain loop, so this face answered + * `0.6000000000000001` / `0.20000000000000004` over `0.1`, `0.2` and `0.3` + * where the rows path and SQLite answer `0.6` / `0.19999999999999998`, and + * `1e16, 1, -1e16` summed to `0` rather than `1`. + * + * ## Why an `$accumulator`, measured against the other two routes (mingo 7.2.4) + * + * - **A post-group recompute** is ruled out by the reason in + * {@link numericAggregandExpr}'s header: it runs after the pipeline's own + * `$sort` and `$limit`, so `order` over a `sum` measure would rank the value + * the measure does not answer. + * - **A custom accumulator operator** cannot replace `$sum` / `$avg` through + * the `mingo` entry point this package imports: its `Aggregator` merges the + * default operators first (`Context.from`), and `addOps` keeps an operator + * already present, so a caller's context can only ADD a name. A new name + * would have to be registered where each `Aggregator` is built — the + * driver's public `aggregate()` and {@link MemoryAnalyticsService}'s own + * time-bucket half — widening the pipeline dialect the driver accepts. + * - **`$accumulator`** is in mingo's default operator set and needs + * `scriptEnabled`, which `ComputeOptions.init` defaults to `true`; both + * `Aggregator`s this face runs take the default options. It stays inside the + * `$group` stage, so every later stage sees the finished number. + * + * ## What does not move + * + * Only the addition. The aggregand is {@link numericAggregandExpr}, as before, + * and the addends are the values mingo's own `$sum` / `$avg` add: numbers, + * NaN excluded (mingo's `isNumber`), so null, a missing key and a non-numeric + * string stay ignored. `avg` over no addend is `null`, `sum` over none is `0`, + * exactly as mingo answered. The values are added in the group's row order, + * the order mingo's `$push` collects them in, so the naive running sum inside + * {@link compensatedSum} is the one `$sum` computed. + * + * The functions are named, and {@link pipelineDumpReplacer} renders a function + * by its name, so the pipeline dump still says which fold a measure runs. + */ +function compensatedAddendAccumulator(path: string, fn: 'sum' | 'avg'): Record { + return { + $accumulator: { + init: startAddends, + accumulateArgs: [numericAggregandExpr(path)], + accumulate: collectAddend, + finalize: fn === 'sum' ? compensatedSumOfAddends : compensatedMeanOfAddends, + lang: 'js', + }, + }; +} + +function startAddends(): number[] { + return []; +} + +/** mingo's `isNumber`: the values its `$sum` and `$avg` add. */ +function collectAddend(addends: number[], value: unknown): number[] { + if (typeof value === 'number' && !Number.isNaN(value)) addends.push(value); + return addends; +} + +function compensatedSumOfAddends(addends: readonly number[]): number { + return compensatedSum(addends); +} + +function compensatedMeanOfAddends(addends: readonly number[]): number | null { + return addends.length === 0 ? null : compensatedSum(addends) / addends.length; +} + /** * [#7853] A `JSON.stringify` replacer that renders a `RegExp` operand instead of * dropping it — the one value type the pipeline dump carries that @@ -700,7 +778,13 @@ function numericAggregandExpr(path: string): Record { * EXECUTION, before this dump is ever built, so no replacer here reaches it. */ function pipelineDumpReplacer(_key: string, value: unknown): unknown { - return value instanceof RegExp ? `/${value.source}/${value.flags}` : value; + if (value instanceof RegExp) return `/${value.source}/${value.flags}`; + // [#20544] A function is the other value `JSON.stringify` erases, and the + // `sum` / `avg` `$accumulator` carries three ({@link + // compensatedAddendAccumulator}). Dropped, the two measures dump identically; + // by name, the dump still says which fold each one runs. + if (typeof value === 'function') return `[function ${value.name}]`; + return value; } /** @@ -1728,10 +1812,12 @@ export class MemoryAnalyticsService implements IAnalyticsService { switch (measure.type) { case 'count': return { $sum: 1 }; + // [#20544] Compensated, as every other face the platform owns adds — + // see {@link compensatedAddendAccumulator}. case 'sum': - return { $sum: numericAggregandExpr(`$${fieldPath}`) }; + return compensatedAddendAccumulator(`$${fieldPath}`, 'sum'); case 'avg': - return { $avg: numericAggregandExpr(`$${fieldPath}`) }; + return compensatedAddendAccumulator(`$${fieldPath}`, 'avg'); // [#11152] `min`/`max` take the SAME boolean coercion as `sum`/`avg` — // maintainer ruling 2026-08-28 (superseding #11249's `false`/`true`): // booleans aggregate as NUMBERS on every face, no per-aggregate diff --git a/packages/drivers/driver-memory/src/memory-driver.ts b/packages/drivers/driver-memory/src/memory-driver.ts index d1e188562d5..7efbcc6109f 100644 --- a/packages/drivers/driver-memory/src/memory-driver.ts +++ b/packages/drivers/driver-memory/src/memory-driver.ts @@ -14,7 +14,7 @@ import { hasDanglingLikeEscape, hasNulInLikePattern, likePatternToRegExp } from // the ruled 「is empty」 table, asked of the spec by the live query path. import { expandEmptyOperator, type ValueShapeFieldDef } from '@objectstack/spec/data'; import type { DriverQuery, IDataDriver } from '@objectstack/spec/contracts'; -import { Logger, createLogger, nextUtcCalendarDay, isUnboundedAbove } from '@objectstack/core'; +import { Logger, createLogger, nextUtcCalendarDay, isUnboundedAbove, compensatedSum } from '@objectstack/core'; import { Query, Aggregator } from 'mingo'; import { assertSingleTenantPosture, @@ -2014,12 +2014,21 @@ export class InMemoryDriver implements IDataDriver { // because "the faces disagree" is this package's recurring defect // class (#5374, #6814) and one face aligned alone leaves the other // free to keep its own answer. + // + // [#20544] The addition is `@objectstack/core`'s `compensatedSum`, + // the fold objectql's rows path and SQLite add with. It used to be a + // naive `reduce`, so on this driver `engine.aggregate` answered two + // doubles by path — `0.1 + 0.2 + 0.3` was `0.6000000000000001` here + // and `0.6` on the rows path a filtered sibling aggregation forces, + // and `having { s: { $eq: 0.6 } }` kept the group on that path alone. + // Which values count as addends is untouched: the boolean rule above + // and the `typeof === 'number'` gate decide that, the fold only adds. case 'sum': case 'avg': { const nums = values .map(v => (typeof v === 'boolean' ? (v ? 1 : 0) : v)) .filter(v => typeof v === 'number'); - const sum = nums.reduce((a, b) => a + b, 0); + const sum = compensatedSum(nums); if (func === 'sum') return sum; return nums.length > 0 ? sum / nums.length : null; } diff --git a/packages/objectql/src/in-memory-aggregation.ts b/packages/objectql/src/in-memory-aggregation.ts index 4fc9005fc39..1824751db00 100644 --- a/packages/objectql/src/in-memory-aggregation.ts +++ b/packages/objectql/src/in-memory-aggregation.ts @@ -84,7 +84,7 @@ // every entry: labels fell back to raw ids, and cross-object rebucketing filed // every row under `'(restricted)'` while the grand total still reconciled. -import { bucketDateKey } from '@objectstack/core'; +import { bucketDateKey, compensatedSum } from '@objectstack/core'; import type { QueryAST, GroupByNode, AggregationNode, DateGranularityValue } from '@objectstack/spec/data'; import { declaredFieldClasses, matchesAggregationFilter } from './having-filter.js'; @@ -231,6 +231,8 @@ function aggregateBucket( // [#20489] Both arms add through ONE compensated fold // ({@link compensatedSum}) — the summation SQLite's own `sum` / `avg` // use — so the rows path and SQLite's native path answer the same double. + // [#20544] The fold is `@objectstack/core`'s now, and driver-memory's + // two faces and the analytics draft preview call the same one. case 'sum': out[alias] = compensatedSum(values.map(toNumber)); break; @@ -287,44 +289,6 @@ function toNumber(v: any): number { return Number.isFinite(n) ? n : 0; } -/** - * [#20489] The sum of `nums`, added in order with Kahan-Babuska-Neumaier - * compensation — the summation SQLite (3.43 and later) uses for its own `sum` - * and `avg`, transcribed from its `kahanBabuskaNeumaierStep` and the - * finalizers' overflow guard. - * - * Why: this fold used to add naively (`reduce((a, b) => a + b, 0)`), so one - * query answered two doubles on SQLite depending on the path `engine.aggregate` - * took. A `number` column holding `0.1`, `0.2` and `0.3` summed to `0.6` - * natively and to `0.6000000000000001` here (`avg` `0.19999999999999998` - * against `0.20000000000000004`), and `having { s: { $eq: 0.6 } }` kept the - * group on the native path alone. Compensated, the two paths agree, and the - * answer is the more accurate one (`1e16 + 1 - 1e16` is `1`, not `0`). - * - * What does not move: two addends (the compensated `a + b` IS the naive one), - * integers whose partial sums stay within 2^53 (every addition is exact), and - * a non-finite total. `s` below is exactly the naive running sum; once it - * overflows or meets a NaN, the error term is non-finite and the naive answer - * is returned as it was, which is SQLite's rule too. - * - * ⚠️ Residual, stated: PostgreSQL and MySQL add their doubles natively without - * compensation, so over three or more fractions their native path can still - * differ from this one in the last place. So can the folds that do not call - * this one: `driver-memory`'s `aggregate` and analytics faces, and - * `service-analytics`' draft preview (`preview-evaluator.ts`). An exact `$eq` - * on a fractional sum compares doubles; compare with a range. - */ -function compensatedSum(nums: readonly number[]): number { - let s = 0; - let c = 0; - for (const r of nums) { - const t = s + r; - c += Math.abs(s) > Math.abs(r) ? (s - t) + r : (r - t) + s; - s = t; - } - return Number.isFinite(c) ? s + c : s; -} - /** * Bucket a date-like value into an ISO-formatted period label. Weeks start * Monday and use ISO week numbering. diff --git a/packages/services/service-analytics/src/preview-evaluator.ts b/packages/services/service-analytics/src/preview-evaluator.ts index b5949b8e98c..159f321b0f4 100644 --- a/packages/services/service-analytics/src/preview-evaluator.ts +++ b/packages/services/service-analytics/src/preview-evaluator.ts @@ -36,6 +36,7 @@ import { resolveAnalyticsDateRangeString, utcInstantMs, isUnboundedAbove, + compensatedSum, } from '@objectstack/core'; import { explicitDateRangeWindow } from './date-range-array-arm.js'; // [#19810] The `where` door's refusal envelope — `INVALID_FILTER` / 400, @@ -511,7 +512,13 @@ function aggregate(rows: Row[], metricType: string, field: string): unknown { // numeric `default` below, answering a sum of coerced values (or a row // count) under the author's `count_distinct` name. case 'count_distinct': return new Set(rows.map((r) => r[field]).filter((v) => v != null)).size; - case 'sum': return nums.reduce((a, b) => a + b, 0); + // [#20544] `sum` and `avg` add with `@objectstack/core`'s `compensatedSum`, + // the fold SQLite and the engine's rows path add with. A naive `reduce` + // here answered `0.1 + 0.2 + 0.3` as `0.6000000000000001` over a draft + // while the published chart on SQLite read `0.6`. Only the addition moved: + // which rows are operands is each arm's own rule, unchanged, and a `0` + // operand leaves a compensated sum as it leaves a naive one. + case 'sum': return compensatedSum(nums); // [#16219] Averaging NOTHING has no answer, and this arm used to invent // one. The live faces answer SQL NULL over the same rows (`AVG(col)` is // defined over non-null values in every dialect), so a drafted chart read @@ -547,7 +554,7 @@ function aggregate(rows: Row[], metricType: string, field: string): unknown { case 'avg': { const present = rows.filter((r) => r[field] != null); const operands = present.map((r) => Number(r[field])).filter((n) => Number.isFinite(n)); - if (operands.length) return operands.reduce((a, b) => a + b, 0) / operands.length; + if (operands.length) return compensatedSum(operands) / operands.length; // ⭐ No numeric operand is TWO different situations, and only one of them // is "averaging nothing": // From 1c3ad8079a8164525298b94207bc3b7670a1c058 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 23:17:30 +0000 Subject: [PATCH 2/4] test(core,driver-memory,service-analytics): pin the compensated sum / avg on every face The card's `0.1 + 0.2 + 0.3` fixture, the `1e16` cancellation, a two-addend control and integers: on the core helper itself, on driver-memory's data face (both doors) and analytics face (plain, time-bucketed, ordered, and its pipeline dump), and on the draft preview as a differential against the live face on a real SQLite. Claude-Session: https://claude.ai/code/session_01DEvba2nBuD4tWzfq8r8NFY Co-authored-by: Claude --- .../core/src/utils/compensated-sum.test.ts | 68 ++++++ .../src/memory-compensated-sum.test.ts | 205 ++++++++++++++++++ .../__tests__/preview-compensated-sum.test.ts | 124 +++++++++++ 3 files changed, 397 insertions(+) create mode 100644 packages/core/src/utils/compensated-sum.test.ts create mode 100644 packages/drivers/driver-memory/src/memory-compensated-sum.test.ts create mode 100644 packages/services/service-analytics/src/__tests__/preview-compensated-sum.test.ts diff --git a/packages/core/src/utils/compensated-sum.test.ts b/packages/core/src/utils/compensated-sum.test.ts new file mode 100644 index 00000000000..bbd99712e69 --- /dev/null +++ b/packages/core/src/utils/compensated-sum.test.ts @@ -0,0 +1,68 @@ +// Copyright (c) 2026 ObjectStack. Licensed under the Apache-2.0 license. + +/** + * [#20544] `compensatedSum` — the one compensated fold every platform face that + * adds a group's values in JavaScript calls (objectql's rows path, + * driver-memory's data and analytics faces, service-analytics' draft preview). + * + * The expected values are SQLite's own `sum` answers, measured by #20489 with + * better-sqlite3 (SQLite 3.53.4), sql.js (3.49.1) and @libsql/client (3.45.1); + * the naive fold each face used before is asserted beside them, so a fixture + * that cannot tell the two folds apart fails here instead of passing vacuously. + * Each face pins the same fixtures through its own door, in its own package. + */ + +import { describe, it, expect } from 'vitest'; +import { compensatedSum } from './compensated-sum'; + +/** The fold the faces used before: in order, one addition at a time. */ +const naiveSum = (xs: readonly number[]) => xs.reduce((a, b) => a + b, 0); + +describe('[#20544] compensatedSum — the one compensated fold', () => { + it("the card's fixture: 0.1 + 0.2 + 0.3 is SQLite's 0.6, not the naive 0.6000000000000001", () => { + expect(naiveSum([0.1, 0.2, 0.3])).toBe(0.6000000000000001); + expect(compensatedSum([0.1, 0.2, 0.3])).toBe(0.6); + // The mean every avg arm divides out of it: SQLite's 0.19999999999999998. + expect(compensatedSum([0.1, 0.2, 0.3]) / 3).toBe(0.19999999999999998); + }); + + it('a large cancellation keeps the small addend: 1e16 + 1 - 1e16 is 1, not 0', () => { + expect(naiveSum([1e16, 1, -1e16])).toBe(0); + expect(compensatedSum([1e16, 1, -1e16])).toBe(1); + expect(compensatedSum([1e16, 0.5, -1e16])).toBe(0.5); + }); + + it('two addends are unchanged: the compensated a + b is the naive one', () => { + for (const pair of [[0.1, 0.2], [0.7, 0.1], [1e16, 1], [-0.3, 0.1]]) { + expect(compensatedSum(pair), `${pair}`).toBe(naiveSum(pair)); + } + expect(compensatedSum([0.1, 0.2])).toBe(0.30000000000000004); + }); + + it('integers whose partial sums stay within 2^53 are unchanged', () => { + for (const ints of [[1, 2, 3, 40, 500], [-7, 3, 12, 0, 9_000_000_000]]) { + const s = compensatedSum(ints); + expect(s, `${ints}`).toBe(naiveSum(ints)); + expect(Number.isInteger(s)).toBe(true); + } + expect(compensatedSum([1, 2, 3, 40, 500])).toBe(546); + }); + + it('an empty list is 0, and one addend is itself', () => { + expect(compensatedSum([])).toBe(0); + expect(compensatedSum([0.1])).toBe(0.1); + expect(compensatedSum([-2.5])).toBe(-2.5); + }); + + it('a non-finite total is the naive one, as SQLite returns its running sum when the error term overflows', () => { + for (const values of [ + [Infinity, 1, 2], + [1, -Infinity, 0.3], + [1e308, 1e308, -1e308], + [Infinity, -Infinity, 1], + [NaN, 0.1, 0.2], + ]) { + expect(Object.is(compensatedSum(values), naiveSum(values)), `${values}`).toBe(true); + } + }); +}); diff --git a/packages/drivers/driver-memory/src/memory-compensated-sum.test.ts b/packages/drivers/driver-memory/src/memory-compensated-sum.test.ts new file mode 100644 index 00000000000..0f3db780984 --- /dev/null +++ b/packages/drivers/driver-memory/src/memory-compensated-sum.test.ts @@ -0,0 +1,205 @@ +// Copyright (c) 2026 ObjectStack. Licensed under the Apache-2.0 license. + +/** + * [#20544] driver-memory adds `sum` / `avg` with `@objectstack/core`'s + * `compensatedSum`, on both of its faces, as SQLite and objectql's rows path do. + * + * Measured on the base (`d2820876f`), one `number` column, one group per + * fixture: + * + * | fixture | data face / analytics face (base) | rows path and SQLite | + * |:--|:--|:--| + * | `0.1, 0.2, 0.3` | `0.6000000000000001` / `0.20000000000000004` | `0.6` / `0.19999999999999998` | + * | `1e16, 1, -1e16` | `0` / `0` | `1` / `0.3333333333333333` | + * | `0.1, 0.2` (two addends) | `0.30000000000000004` / `0.15000000000000002` | the same | + * | `1, 2, 3, 40, 500` (integers) | `546` / `109.2` | the same | + * + * So `engine.aggregate` on this driver answered two doubles by path: the + * native path (`aggregate(object, AST)`, driven below) read the naive column, + * the rows path a filtered sibling aggregation forces read the compensated + * one, and `having { s: { $eq: 0.6 } }` kept the group on the rows path alone. + * The rows path's own answer is pinned in objectql + * (`in-memory-aggregation-compensated-sum.test.ts`) against the same literals. + * + * Both faces, and both doors of the data face, are driven: one face aligned + * alone is how this package's faces come to disagree (#5374, #6814, #11065). + */ + +import { describe, it, expect, beforeEach } from 'vitest'; +import type { DriverQuery } from '@objectstack/spec/contracts'; +import type { Cube } from '@objectstack/spec/data'; +import { InMemoryDriver } from './memory-driver.js'; +import { MemoryAnalyticsService } from './memory-analytics.js'; + +const TABLE = 'ledger_20544'; + +/** group → its values, and the compensated `sum` / `avg` over them. */ +const GROUPS: Record = { + card: { values: [0.1, 0.2, 0.3], s: 0.6, a: 0.19999999999999998 }, + cancel: { values: [1e16, 1, -1e16], s: 1, a: 0.3333333333333333 }, + two: { values: [0.1, 0.2], s: 0.30000000000000004, a: 0.15000000000000002 }, + ints: { values: [1, 2, 3, 40, 500], s: 546, a: 109.2 }, +}; + +/** The fold both faces used before #20544: in order, one addition at a time. */ +const naiveSum = (xs: readonly number[]) => xs.reduce((a, b) => a + b, 0); + +async function seed(): Promise { + const driver = new InMemoryDriver(); + let i = 0; + for (const [g, { values }] of Object.entries(GROUPS)) { + for (const w of values) await driver.create(TABLE, { id: `r${i++}`, g, w, at: '2026-03-04T10:00:00Z' }); + } + return driver; +} + +const expected = Object.fromEntries(Object.entries(GROUPS).map(([g, { s, a }]) => [g, { s, a }])); + +describe('[#20544] the fixtures discriminate the two folds', () => { + it('the naive fold answers another double on the card fixture and the cancellation, and the same on the controls', () => { + expect(naiveSum(GROUPS.card.values)).toBe(0.6000000000000001); + expect(naiveSum(GROUPS.cancel.values)).toBe(0); + expect(naiveSum(GROUPS.two.values)).toBe(GROUPS.two.s); + expect(naiveSum(GROUPS.ints.values)).toBe(GROUPS.ints.s); + }); +}); + +describe('[#20544] data face — sum / avg add with the compensated fold', () => { + let driver: InMemoryDriver; + beforeEach(async () => { driver = await seed(); }); + + const grouped: DriverQuery = { + groupBy: ['g'], + aggregations: [ + { function: 'sum', field: 'w', alias: 's' }, + { function: 'avg', field: 'w', alias: 'a' }, + ], + } as DriverQuery; + const byGroup = (rows: any[]) => Object.fromEntries(rows.map((r) => [r.g, { s: r.s, a: r.a }])); + + it("aggregate(AST) — the door engine.aggregate's native path takes — answers every group's compensated sum and mean", async () => { + expect(byGroup(await driver.aggregate(TABLE, grouped) as any[])).toStrictEqual(expected); + }); + + it('find() — the other door onto the same fold — answers the same numbers', async () => { + expect(byGroup(await driver.find(TABLE, grouped) as any[])).toStrictEqual(expected); + }); + + it("the card's having reads: s === 0.6 holds for the card group, so `having { s: { $eq: 0.6 } }` keeps it", async () => { + const rows = await driver.aggregate(TABLE, { ...grouped, where: { g: 'card' } } as DriverQuery) as any[]; + expect(rows).toHaveLength(1); + expect(rows[0].s === 0.6).toBe(true); + }); + + it('what the addends are does not move: a boolean is 1 / 0, null and a non-numeric string are left out', async () => { + const d = new InMemoryDriver(); + const cells: unknown[] = [0.1, null, 0.2, 'abc', true, 0.3, undefined]; + for (const [i, w] of cells.entries()) await d.create(TABLE, { id: `m${i}`, w }); + const [row] = await d.aggregate(TABLE, { + aggregations: [ + { function: 'sum', field: 'w', alias: 's' }, + { function: 'avg', field: 'w', alias: 'a' }, + ], + } as DriverQuery) as any[]; + // Addends 0.1, 0.2, 1, 0.3 — four of them, in row order. + expect(row).toStrictEqual({ s: 1.6, a: 0.4 }); + }); + + it('the empty group: sum 0, avg null', async () => { + const d = new InMemoryDriver(); + await d.create(TABLE, { id: 'n1', w: null }); + const [row] = await d.aggregate(TABLE, { + aggregations: [ + { function: 'sum', field: 'w', alias: 's' }, + { function: 'avg', field: 'w', alias: 'a' }, + ], + } as DriverQuery) as any[]; + expect(row).toStrictEqual({ s: 0, a: null }); + }); +}); + +describe('[#20544] analytics face — sum / avg measures add with the compensated fold', () => { + const cube = { + name: 'ledger', + title: 'Ledger', + sql: TABLE, + measures: { + total: { label: 'Total', type: 'sum', sql: 'w' }, + mean: { label: 'Mean', type: 'avg', sql: 'w' }, + }, + dimensions: { + g: { label: 'Group', type: 'string', sql: 'g' }, + at: { label: 'At', type: 'time', sql: 'at' }, + }, + } as unknown as Cube; + + let service: MemoryAnalyticsService; + beforeEach(async () => { service = new MemoryAnalyticsService({ driver: await seed(), cubes: [cube] }); }); + + const byGroup = (rows: any[]) => + Object.fromEntries(rows.map((r) => [r['ledger.g'], { s: r['ledger.total'], a: r['ledger.mean'] }])); + + it("grouped: every group's compensated sum and mean", async () => { + const result = await service.query({ + cube: 'ledger', measures: ['ledger.total', 'ledger.mean'], dimensions: ['ledger.g'], + } as any); + expect(byGroup(result.rows as any[])).toStrictEqual(expected); + }); + + /** + * The time-bucket half runs its own `Aggregator` over the `$group` onward + * (`aggregateWithTimeBuckets`), so it is a second place the accumulator has + * to work — measured, not assumed. + */ + it('under a granular time dimension — the split pipeline — the same numbers', async () => { + const result = await service.query({ + cube: 'ledger', + measures: ['ledger.total', 'ledger.mean'], + dimensions: ['ledger.g'], + timeDimensions: [{ dimension: 'ledger.at', granularity: 'day' }], + } as any); + expect(byGroup(result.rows as any[])).toStrictEqual(expected); + for (const row of result.rows as any[]) expect(row['ledger.at']).toBe('2026-03-04'); + }); + + /** + * `$sort` runs after `$group`, so ordering by the measure ranks the number + * the measure answers — the reason the fold lives inside `$group` rather + * than in a post-processing step. + */ + it('ordered by the sum measure, the rows rank by the compensated values', async () => { + const result = await service.query({ + cube: 'ledger', measures: ['ledger.total'], dimensions: ['ledger.g'], order: { 'ledger.total': 'asc' }, + } as any); + expect((result.rows as any[]).map((r) => [r['ledger.g'], r['ledger.total']])).toStrictEqual([ + ['two', 0.30000000000000004], ['card', 0.6], ['cancel', 1], ['ints', 546], + ]); + }); + + it('what the addends are does not move: a boolean is 1 / 0, null, a missing key and a non-numeric string are left out', async () => { + const d = new InMemoryDriver(); + const cells: unknown[] = [0.1, null, 0.2, 'abc', true, 0.3]; + for (const [i, w] of cells.entries()) await d.create(TABLE, { id: `m${i}`, w }); + await d.create(TABLE, { id: 'missing' }); + const svc = new MemoryAnalyticsService({ driver: d, cubes: [cube] }); + const result = await svc.query({ cube: 'ledger', measures: ['ledger.total', 'ledger.mean'] } as any); + expect(result.rows[0]).toStrictEqual({ 'ledger.total': 1.6, 'ledger.mean': 0.4 }); + }); + + it('the empty group: sum 0, avg null', async () => { + const d = new InMemoryDriver(); + await d.create(TABLE, { id: 'n1', w: null }); + const svc = new MemoryAnalyticsService({ driver: d, cubes: [cube] }); + const result = await svc.query({ cube: 'ledger', measures: ['ledger.total', 'ledger.mean'] } as any); + expect(result.rows[0]).toStrictEqual({ 'ledger.total': 0, 'ledger.mean': null }); + }); + + /** The debugging dump names each measure's fold: a function the replacer + * dropped would make a `sum` and an `avg` measure dump identically. */ + it('the pipeline dump still tells a sum measure from an avg measure', async () => { + const result = await service.query({ cube: 'ledger', measures: ['ledger.total', 'ledger.mean'] } as any); + const sql = String(result.sql); + expect(sql).toContain('"finalize":"[function compensatedSumOfAddends]"'); + expect(sql).toContain('"finalize":"[function compensatedMeanOfAddends]"'); + }); +}); diff --git a/packages/services/service-analytics/src/__tests__/preview-compensated-sum.test.ts b/packages/services/service-analytics/src/__tests__/preview-compensated-sum.test.ts new file mode 100644 index 00000000000..1d1172f31bb --- /dev/null +++ b/packages/services/service-analytics/src/__tests__/preview-compensated-sum.test.ts @@ -0,0 +1,124 @@ +// Copyright (c) 2026 ObjectStack. Licensed under the Apache-2.0 license. + +/** + * [#20544] The draft preview (`preview-evaluator.ts`) adds `sum` / `avg` with + * `@objectstack/core`'s `compensatedSum`, so a drafted chart reads the double + * the published chart reads on SQLite. + * + * Measured on the base (`d2820876f`): over `0.1`, `0.2` and `0.3` the preview + * answered `0.6000000000000001` / `0.20000000000000004` where the live face — + * `NativeSQLStrategy`'s SQL on a real SQLite (sql.js), which adds with + * Kahan-Babuska-Neumaier compensation since 3.43 — answered `0.6` / + * `0.19999999999999998`; over `1e16, 1, -1e16` it answered `0` / `0` against + * `1` / `0.3333333333333333`. + * + * ## The instrument — the differential #16203 built + * + * One dataset, one row set, two `AnalyticsService` instances differing in + * exactly one key (`draftRowsResolver`), so a difference between the two + * responses is one the preview evaluator caused. Every group is a control + * or a case; the two controls (two addends, integers) answer the same double + * under either fold, so a preview that stopped adding cannot pass on them by + * coincidence either. + */ + +import { describe, it, expect, beforeAll, afterAll } from 'vitest'; +import { DatasetSchema } from '@objectstack/spec/ui'; +import { AnalyticsService } from '../analytics-service.js'; + +/** group → its values, and SQLite's `sum` / `avg` over them (#20489's readings). */ +const GROUPS: Record = { + card: { values: [0.1, 0.2, 0.3], s: 0.6, a: 0.19999999999999998 }, + cancel: { values: [1e16, 1, -1e16], s: 1, a: 0.3333333333333333 }, + two: { values: [0.1, 0.2], s: 0.30000000000000004, a: 0.15000000000000002 }, + ints: { values: [1, 2, 3, 40, 500], s: 546, a: 109.2 }, +}; + +const ROWS: Record[] = Object.entries(GROUPS).flatMap(([g, { values }], gi) => + values.map((w, i) => ({ id: `${gi}-${i}`, g, w })), +); + +const DATASET = DatasetSchema.parse({ + name: 'ledger_ds', + label: 'Ledger', + object: 'ledger', + dimensions: [{ name: 'g', field: 'g', type: 'string', label: 'Group' }], + measures: [ + { name: 's', aggregate: 'sum', field: 'w' }, + { name: 'a', aggregate: 'avg', field: 'w' }, + ], +}); + +let db: any; + +const runSql = (sql: string, params: unknown[]): Record[] => { + const stmt = db.prepare(sql.replace(/\$\d+/g, '?')); + stmt.bind(params as any[]); + const rows: Record[] = []; + while (stmt.step()) rows.push(stmt.getAsObject()); + stmt.free(); + return rows; +}; + +async function locateWasm(): Promise<((file: string) => string) | undefined> { + try { + const { createRequire } = await import('node:module'); + const require = createRequire(import.meta.url); + const pkgJsonPath = require.resolve('sql.js/package.json'); + const { dirname, join } = await import('node:path'); + return (file: string) => join(dirname(pkgJsonPath), 'dist', file); + } catch { + return undefined; + } +} + +function svc(preview: boolean) { + return new AnalyticsService({ + queryCapabilities: () => ({ nativeSql: true, objectqlAggregate: false, inMemory: false }), + executeRawSql: async (_object: string, sql: string, params: unknown[]) => runSql(sql, params), + ...(preview ? { draftRowsResolver: async () => ROWS.map((r) => ({ ...r })) } : {}), + }); +} + +async function grid(preview: boolean): Promise> { + const result = await svc(preview).queryDataset( + DATASET, + { dimensions: ['g'], measures: ['s', 'a'] }, + undefined, + preview ? { previewDrafts: true } : undefined, + ); + return Object.fromEntries(result.rows.map((r) => [String(r.g), { s: r.s, a: r.a }])); +} + +const expected = Object.fromEntries(Object.entries(GROUPS).map(([g, { s, a }]) => [g, { s, a }])); + +beforeAll(async () => { + const mod: any = await import('sql.js'); + const initSqlJs = mod.default ?? mod; + const locateFile = await locateWasm(); + const SQL = await initSqlJs(locateFile ? { locateFile } : undefined); + db = new SQL.Database(); + db.run(`CREATE TABLE "ledger" ("id" TEXT PRIMARY KEY, "g" TEXT, "w" REAL);`); + const insert = db.prepare(`INSERT INTO "ledger" ("id","g","w") VALUES (?,?,?)`); + for (const r of ROWS) insert.run([r.id, r.g, r.w] as any[]); + insert.free(); +}); + +afterAll(() => db?.close()); + +describe('[#20544] draft preview — sum / avg read the published double', () => { + it("the live face is the standard: SQLite answers the card's 0.6 and the cancellation's 1", async () => { + expect(await grid(false)).toStrictEqual(expected); + }); + + it('the preview answers the same double in every group — the card, the cancellation and both controls', async () => { + const live = await grid(false); + const preview = await grid(true); + expect(preview).toStrictEqual(expected); + expect(preview).toStrictEqual(live); + }); + + it("so the card's exact comparison holds on the draft: s === 0.6", async () => { + expect((await grid(true)).card.s === 0.6).toBe(true); + }); +}); From e07690e1de935757a1c4aebd3a6469bad17fbe4a Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 23:18:23 +0000 Subject: [PATCH 3/4] test(service-analytics): name the preview differential's dataset members with valid identifiers Claude-Session: https://claude.ai/code/session_01DEvba2nBuD4tWzfq8r8NFY Co-authored-by: Claude --- .../__tests__/preview-compensated-sum.test.ts | 18 +++++++++--------- 1 file changed, 9 insertions(+), 9 deletions(-) diff --git a/packages/services/service-analytics/src/__tests__/preview-compensated-sum.test.ts b/packages/services/service-analytics/src/__tests__/preview-compensated-sum.test.ts index 1d1172f31bb..c7c09e128e0 100644 --- a/packages/services/service-analytics/src/__tests__/preview-compensated-sum.test.ts +++ b/packages/services/service-analytics/src/__tests__/preview-compensated-sum.test.ts @@ -35,17 +35,17 @@ const GROUPS: Record = { }; const ROWS: Record[] = Object.entries(GROUPS).flatMap(([g, { values }], gi) => - values.map((w, i) => ({ id: `${gi}-${i}`, g, w })), + values.map((amt, i) => ({ id: `${gi}-${i}`, grp: g, amt })), ); const DATASET = DatasetSchema.parse({ name: 'ledger_ds', label: 'Ledger', object: 'ledger', - dimensions: [{ name: 'g', field: 'g', type: 'string', label: 'Group' }], + dimensions: [{ name: 'grp', field: 'grp', type: 'string', label: 'Group' }], measures: [ - { name: 's', aggregate: 'sum', field: 'w' }, - { name: 'a', aggregate: 'avg', field: 'w' }, + { name: 'sum_amt', aggregate: 'sum', field: 'amt' }, + { name: 'avg_amt', aggregate: 'avg', field: 'amt' }, ], }); @@ -83,11 +83,11 @@ function svc(preview: boolean) { async function grid(preview: boolean): Promise> { const result = await svc(preview).queryDataset( DATASET, - { dimensions: ['g'], measures: ['s', 'a'] }, + { dimensions: ['grp'], measures: ['sum_amt', 'avg_amt'] }, undefined, preview ? { previewDrafts: true } : undefined, ); - return Object.fromEntries(result.rows.map((r) => [String(r.g), { s: r.s, a: r.a }])); + return Object.fromEntries(result.rows.map((r) => [String(r.grp), { s: r.sum_amt, a: r.avg_amt }])); } const expected = Object.fromEntries(Object.entries(GROUPS).map(([g, { s, a }]) => [g, { s, a }])); @@ -98,9 +98,9 @@ beforeAll(async () => { const locateFile = await locateWasm(); const SQL = await initSqlJs(locateFile ? { locateFile } : undefined); db = new SQL.Database(); - db.run(`CREATE TABLE "ledger" ("id" TEXT PRIMARY KEY, "g" TEXT, "w" REAL);`); - const insert = db.prepare(`INSERT INTO "ledger" ("id","g","w") VALUES (?,?,?)`); - for (const r of ROWS) insert.run([r.id, r.g, r.w] as any[]); + db.run(`CREATE TABLE "ledger" ("id" TEXT PRIMARY KEY, "grp" TEXT, "amt" REAL);`); + const insert = db.prepare(`INSERT INTO "ledger" ("id","grp","amt") VALUES (?,?,?)`); + for (const r of ROWS) insert.run([r.id, r.grp, r.amt] as any[]); insert.free(); }); From 7b2740bee43d85f72fdbb7be6e3b58cdd946556b Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 23:39:13 +0000 Subject: [PATCH 4/4] chore: changeset for the compensated sum / avg on every face Claude-Session: https://claude.ai/code/session_01DEvba2nBuD4tWzfq8r8NFY Co-authored-by: Claude --- .../20544-compensated-sum-every-face.md | 47 +++++++++++++++++++ 1 file changed, 47 insertions(+) create mode 100644 .changeset/20544-compensated-sum-every-face.md diff --git a/.changeset/20544-compensated-sum-every-face.md b/.changeset/20544-compensated-sum-every-face.md new file mode 100644 index 00000000000..2e7eb7e4a6b --- /dev/null +++ b/.changeset/20544-compensated-sum-every-face.md @@ -0,0 +1,47 @@ +--- +'@objectstack/core': minor +'@objectstack/objectql': patch +'@objectstack/driver-memory': patch +'@objectstack/service-analytics': patch +--- + +fix: `sum` / `avg` answer the same double on every face the platform owns, added with one compensated fold that `@objectstack/core` now exports as `compensatedSum` (#20544) + +Clause-②: yes + +**New export.** `@objectstack/core` exports `compensatedSum(nums)`: the sum of +`nums`, added in order with Kahan-Babuska-Neumaier compensation, which is the +summation SQLite (3.43 and later) uses for its own `sum` and `avg`. It moved +here from `@objectstack/objectql`'s rows path (`in-memory-aggregation.ts`), +which now imports it instead of keeping a private copy. + +**What changed.** Three folds still added a group's values naively, and now call +the same function: + +- `@objectstack/driver-memory`'s `aggregate()` and `find()` with aggregations, + the path `engine.aggregate` takes on an in-memory datasource; +- `@objectstack/driver-memory`'s analytics face (`MemoryAnalyticsService`), + whose `sum` / `avg` measures are now a `$group` `$accumulator` in place of + mingo's `$sum` / `$avg`; +- `@objectstack/service-analytics`' draft preview. + +Over a `number` column holding `0.1`, `0.2` and `0.3`, each of them answered +`0.6000000000000001` / `0.20000000000000004`. They now answer `0.6` / +`0.19999999999999998`, as SQLite and the engine's rows path do. Over +`1e16, 1, -1e16` they answered `0` and now answer `1`. On driver-memory, +`engine.aggregate` gave two answers depending on its path: `having { s: { $eq: +0.6 } }` kept the group on the rows path and dropped it on the native path. It +now keeps it on both. + +**What did not move.** Two addends, integers whose running total stays within +2^53, and a non-finite total give the same answer as before. Which values count +as addends did not change either: booleans as 1 / 0, and nulls and non-numeric +strings left out, as each face already had it. `count`, `min` and `max` are +untouched. The analytics face's pipeline dump (`result.sql`) now renders the +accumulator's functions by name, so a `sum` measure and an `avg` measure still +dump differently. + +**Residual.** PostgreSQL and MySQL add their doubles natively without +compensation, and the platform does not wrap that arithmetic. So over three or +more fractions their native path can still differ from these faces in the last +place. An exact `$eq` on a fractional sum compares doubles; compare with a range.