From 0f3adc8114db4a49a974e0ddfa965c764508b4ac Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 2 Oct 2026 14:06:21 +0000 Subject: [PATCH 1/6] test(rest): pin the dataset door's answer to a non-count measure over '*' The door cell the narrowing moves: a dataset measure aggregating the row wildcard under any aggregate other than count, inline and saved, on both strategies, beside the count-over-'*' controls. Claude-Session: https://claude.ai/code/session_01UtnxvdiN376GF3sgXwAw4d Co-authored-by: Claude --- ...nalytics-dataset-row-wildcard-door.test.ts | 247 ++++++++++++++++++ 1 file changed, 247 insertions(+) create mode 100644 packages/rest/src/analytics-dataset-row-wildcard-door.test.ts diff --git a/packages/rest/src/analytics-dataset-row-wildcard-door.test.ts b/packages/rest/src/analytics-dataset-row-wildcard-door.test.ts new file mode 100644 index 00000000000..23ab7faa73f --- /dev/null +++ b/packages/rest/src/analytics-dataset-row-wildcard-door.test.ts @@ -0,0 +1,247 @@ +// Copyright (c) 2026 ObjectStack. Licensed under the Apache-2.0 license. + +/** + * [#21409] `POST /api/v1/analytics/dataset/query` — a dataset measure that + * aggregates the row wildcard `'*'` under any aggregate other than `count` is + * refused at the route's parse with `400 VALIDATION_FAILED`, before any + * strategy or driver runs; `count` over `'*'` still runs, on both strategies. + * + * ## The cell this pins, and what it answered before + * + * `'*'` is what a `count` aggregates (`COUNT(*)`), reading no field value. The + * contract admitted it in a dataset measure's `field` under ANY aggregate, so + * `{ aggregate: 'sum', field: '*' }` parsed and the strategies passed it on + * verbatim. Measured on this route before the narrowing, over a real + * better-sqlite3 `SqlDriver`: `500 DATABASE_ERROR` on the native-SQL strategy + * (`SUM(*)`) and on the ObjectQL strategy (the engine aggregate over `'*'`). + * A server fault for an authoring mistake the contract admitted. + * + * Now `DatasetMeasureSchema` refuses it at `measures.N.field`, and this route + * parses every dataset it is handed — inline or saved by name — through + * `DatasetSchema` before calling `queryDataset`, so the cell answers the + * route's own `400 VALIDATION_FAILED` naming the path. Nothing reaches a + * strategy, so nothing reaches the driver: the read counters below stay at 0. + * + * The SAVED branch is the stored-document half of the same question: a dataset + * stored before the narrowing is read back as stored (the metadata read path + * does not re-validate it) and is refused HERE, at its first query, with the + * same 400 — fail closed, never a stand-down. + * + * ## The controls + * + * `count` over `'*'` — spelled out, and as a count with no `field` (which the + * dataset compiler lowers to the same `'*'`) — answers 200 with the row count + * on BOTH strategies, and the counters prove which strategy answered. + * + * This file exercises the BUILT `@objectstack/spec` (the route imports + * `@objectstack/spec/ui` through its `exports`, so `dist/`), the built + * `@objectstack/service-analytics`, `@objectstack/objectql` and + * `@objectstack/driver-sql`: mutating the spec source without rebuilding proves + * nothing here. + * + * Assertion set (ADR-0112): the route's own envelope `code` and HTTP status, + * plus the parse's issue `code` and `path` read out of `detail` — never the + * prescription prose, which no consumer of this route parses. + */ + +import { describe, it, expect, beforeAll, afterAll } from 'vitest'; +import { ObjectQL } from '@objectstack/objectql'; +import { SqlDriver } from '@objectstack/driver-sql'; +import { AnalyticsServicePlugin, type AnalyticsService } from '@objectstack/service-analytics'; +import { RestServer } from './rest-server'; + +const OBJECT = 'rest_dataset_row_wildcard_ledger'; + +const LEDGER = { + name: OBJECT, + label: 'Dataset row wildcard ledger', + fields: { + category: { name: 'category', type: 'text' as const }, + amount: { name: 'amount', type: 'number' as const }, + }, +}; + +const ROWS = [ + { id: 'a1', category: 'a', amount: 100 }, + { id: 'a2', category: 'a', amount: 400 }, + { id: 'b1', category: 'b', amount: 900 }, +] as const; + +/** The non-count aggregates a dataset measure declares — each one over `'*'` is the refused cell. */ +const NON_COUNT_AGGREGATES = ['sum', 'avg', 'min', 'max', 'count_distinct'] as const; + +/** An inline dataset whose measure `wildcard` aggregates `'*'` under `aggregate`, beside two count controls. */ +const datasetWith = (aggregate: string) => ({ + name: 'row_wildcard_inline', + label: 'Row wildcard inline', + object: OBJECT, + dimensions: [{ name: 'category', field: 'category', type: 'string' }], + measures: [ + { name: 'row_count', aggregate: 'count' }, + { name: 'star_count', aggregate: 'count', field: '*' }, + { name: 'wildcard', aggregate, field: '*' }, + ], +}); + +/** The same dataset with only the two count controls — what the 200 cells post. */ +const COUNT_ONLY = { + name: 'row_wildcard_counts', + label: 'Row wildcard counts', + object: OBJECT, + dimensions: [{ name: 'category', field: 'category', type: 'string' }], + measures: [ + { name: 'row_count', aggregate: 'count' }, + { name: 'star_count', aggregate: 'count', field: '*' }, + ], +}; + +const quiet = { debug() {}, info() {}, warn() {}, error() {}, child() { return quiet; } }; + +function createMockServer() { + const noop = () => {}; + return { get: noop, post: noop, put: noop, delete: noop, patch: noop, use: noop, listen: async () => {}, close: async () => {} }; +} + +/** `saved` is what the route's `body.datasetName` branch loads from metadata, as stored. */ +function mockProtocol(saved: unknown[]) { + return { + getDiscovery: async () => ({ version: 'v0', routes: { data: '', metadata: '' } }), + getMetaTypes: async () => [], + getMetaItems: async () => saved, + }; +} + +function makeRes() { + const res: any = { + statusCode: 200, + body: undefined as any, + header: () => res, + status: (code: number) => { res.statusCode = code; return res; }, + json: (body: unknown) => { res.body = body; return res; }, + end: () => res, + }; + return res; +} + +/** The stored copies the saved branch reads back — one per refused aggregate, plus the count control. */ +const SAVED = [ + ...NON_COUNT_AGGREGATES.map((aggregate) => ({ ...datasetWith(aggregate), name: `stored_star_${aggregate}` })), + COUNT_ONLY, +]; + +for (const strategy of ['native', 'objectql'] as const) { + describe(`[#21409] POST /api/v1/analytics/dataset/query — '*' runs only under count — ${strategy} strategy`, () => { + let engine: ObjectQL; + /** Raw-SQL statements (native strategy) and engine aggregates (ObjectQL strategy) that read THIS object. */ + const reads = { rawSql: 0, aggregate: 0 }; + let post: (body: Record) => Promise<{ status: number; body: any }>; + + beforeAll(async () => { + engine = new ObjectQL({ logger: quiet } as any); + engine.registerDriver( + new SqlDriver({ client: 'better-sqlite3', connection: { filename: ':memory:' }, useNullAsDefault: true } as any), + true, + ); + await engine.init(); + engine.registry.registerObject(LEDGER as any); + await engine.syncSchemas(); + for (const row of ROWS) await engine.insert(OBJECT, { ...row } as any); + + const realExecute = (engine as any).execute.bind(engine); + (engine as any).execute = (sql: unknown, opts?: { object?: string }) => { + if (opts?.object === OBJECT) reads.rawSql += 1; + return realExecute(sql, opts); + }; + const realAggregate = engine.aggregate.bind(engine); + (engine as any).aggregate = (object: string, ...rest: unknown[]) => { + if (object === OBJECT) reads.aggregate += 1; + return (realAggregate as any)(object, ...rest); + }; + + // The plugin's own composition over the real engine. The ObjectQL cell + // states the capability probe the native cell lets the plugin derive. + const registered: Record = {}; + await new AnalyticsServicePlugin( + strategy === 'objectql' + ? { queryCapabilities: () => ({ nativeSql: false, objectqlAggregate: true, inMemory: false }) } + : {}, + ).init({ + getService: (name: string) => (name === 'data' ? engine : registered[name]), + registerService: (name: string, svc: unknown) => { registered[name] = svc; }, + replaceService: (name: string, svc: unknown) => { registered[name] = svc; }, + hook: () => {}, + logger: quiet, + } as never); + const service = registered.analytics as AnalyticsService; + + const rest = new RestServer( + createMockServer() as any, mockProtocol(SAVED) as any, { api: { requireAuth: false } } as any, + undefined, undefined, undefined, undefined, undefined, undefined, undefined, + undefined, undefined, undefined, undefined, + async () => service, + ); + (rest as any).resolveExecCtx = async () => ({ userId: 'test-user' }); + rest.registerRoutes(); + const route = rest.getRoutes().find((r: any) => r.method === 'POST' && r.path === '/api/v1/analytics/dataset/query'); + expect(route).toBeDefined(); + post = async (body) => { + const res = makeRes(); + // What the wire carries: JSON, both ways. + await route!.handler({ method: 'POST', params: {}, headers: {}, body: JSON.parse(JSON.stringify(body)), query: {} } as any, res); + return { status: res.statusCode, body: JSON.parse(JSON.stringify(res.body ?? null)) }; + }; + }); + + afterAll(async () => { + try { await engine?.destroy(); } catch { /* noop */ } + }); + + const expectRefusedBeforeAnyRead = (res: { status: number; body: any }, before: typeof reads, label: string) => { + expect(res.status, `${label}: ${JSON.stringify(res.body)}`).toBe(400); + expect(res.body.code, label).toBe('VALIDATION_FAILED'); + // The parse's issue, at the measure's own `field` (`detail` is the + // issue list, cut at 1000 characters — read, not re-parsed). + expect(res.body.detail, label).toMatch(/"code":\s*"custom"/); + expect(res.body.detail, label).toMatch(/"path":\s*\[\s*"measures",\s*2,\s*"field"\s*\]/); + // Refused at the door: no strategy ran, so nothing read the object. + expect(reads.rawSql - before.rawSql, `${label}: no raw SQL`).toBe(0); + expect(reads.aggregate - before.aggregate, `${label}: no engine aggregate`).toBe(0); + }; + + it.each(NON_COUNT_AGGREGATES)("an INLINE dataset measure aggregating '*' under %s → 400 VALIDATION_FAILED at measures.2.field, nothing read", async (aggregate) => { + const before = { ...reads }; + const res = await post({ dataset: datasetWith(aggregate), selection: { measures: ['wildcard'], dimensions: ['category'] } }); + expectRefusedBeforeAnyRead(res, before, aggregate); + }); + + it.each(NON_COUNT_AGGREGATES)("a SAVED dataset (body.datasetName) stored with '*' under %s → the same 400 at its first query, nothing read", async (aggregate) => { + const before = { ...reads }; + const res = await post({ datasetName: `stored_star_${aggregate}`, selection: { measures: ['row_count'], dimensions: ['category'] } }); + expectRefusedBeforeAnyRead(res, before, `stored ${aggregate}`); + }); + + it("CONTROL: count over '*' and a count with no field both answer the row count — 200, through this strategy", async () => { + const before = { ...reads }; + const res = await post({ dataset: COUNT_ONLY, selection: { measures: ['row_count', 'star_count'], dimensions: ['category'] } }); + expect(res.status, JSON.stringify(res.body)).toBe(200); + const rows = [...(res.body.rows as Array>)].sort((x, y) => String(x.category).localeCompare(String(y.category))); + expect(rows).toEqual([ + { category: 'a', row_count: 2, star_count: 2 }, + { category: 'b', row_count: 1, star_count: 1 }, + ]); + if (strategy === 'native') { + expect(reads.rawSql - before.rawSql, 'NativeSQLStrategy answered').toBeGreaterThanOrEqual(1); + expect(reads.aggregate - before.aggregate, 'no engine aggregate').toBe(0); + } else { + expect(reads.aggregate - before.aggregate, 'ObjectQLStrategy answered').toBeGreaterThanOrEqual(1); + expect(reads.rawSql - before.rawSql, 'no raw SQL').toBe(0); + } + }); + + it("CONTROL: the SAVED count-only dataset answers the same 200", async () => { + const res = await post({ datasetName: COUNT_ONLY.name, selection: { measures: ['star_count'] } }); + expect(res.status, JSON.stringify(res.body)).toBe(200); + expect(res.body.rows).toEqual([{ star_count: 3 }]); + }); + }); +} From e4b7ec949d663150d931ce34e322bbd72df5506c Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 2 Oct 2026 14:32:33 +0000 Subject: [PATCH 2/6] feat(spec)!: the analytics row wildcard is admitted only where a count consumes it A cube measure's sql and a dataset measure's field admit '*' only under count, by one shared predicate (rowWildcardOutsideCount) both measure refinements call; a cube dimension's sql takes the column path without the wildcard arm, the dataset dimension's own pattern. One ADR-0087 D3 entry, the regenerated registry region, and the liveness notes re-pointed here. Generated artifacts and the dropped-refinement ledger follow in the next commit. Claude-Session: https://claude.ai/code/session_01UtnxvdiN376GF3sgXwAw4d Co-authored-by: Claude --- ...nalytics-dataset-row-wildcard-door.test.ts | 14 +- packages/spec/liveness/analytics_cube.json | 8 +- packages/spec/liveness/dataset.json | 4 +- .../src/data/analytics-column-reference.ts | 105 +++++-- .../analytics-row-wildcard-count-only.test.ts | 272 ++++++++++++++++++ packages/spec/src/data/analytics.zod.ts | 49 +++- .../cube-member-sql-column-reference.test.ts | 33 ++- ...tics-row-wildcard-outside-count-refused.ts | 64 +++++ packages/spec/src/migrations/registry.ts | 60 ++++ packages/spec/src/ui/dataset.zod.ts | 44 ++- 10 files changed, 596 insertions(+), 57 deletions(-) create mode 100644 packages/spec/src/data/analytics-row-wildcard-count-only.test.ts create mode 100644 packages/spec/src/migrations/entries/semantic/18.analytics-row-wildcard-outside-count-refused.ts diff --git a/packages/rest/src/analytics-dataset-row-wildcard-door.test.ts b/packages/rest/src/analytics-dataset-row-wildcard-door.test.ts index 23ab7faa73f..95ec535503f 100644 --- a/packages/rest/src/analytics-dataset-row-wildcard-door.test.ts +++ b/packages/rest/src/analytics-dataset-row-wildcard-door.test.ts @@ -214,12 +214,22 @@ for (const strategy of ['native', 'objectql'] as const) { expectRefusedBeforeAnyRead(res, before, aggregate); }); - it.each(NON_COUNT_AGGREGATES)("a SAVED dataset (body.datasetName) stored with '*' under %s → the same 400 at its first query, nothing read", async (aggregate) => { + it.each(NON_COUNT_AGGREGATES)("a SAVED dataset (body.datasetName) stored with '*' under %s → the same 400 when the query selects that measure, nothing read", async (aggregate) => { const before = { ...reads }; - const res = await post({ datasetName: `stored_star_${aggregate}`, selection: { measures: ['row_count'], dimensions: ['category'] } }); + const res = await post({ datasetName: `stored_star_${aggregate}`, selection: { measures: ['wildcard'], dimensions: ['category'] } }); expectRefusedBeforeAnyRead(res, before, `stored ${aggregate}`); }); + // Fail closed, never a stand-down: the route parses the whole stored + // definition, so a query that selects only the dataset's healthy count — + // which answered 200 before the narrowing — is refused too, until the + // member is fixed. The blast radius is the dataset, by design. + it.each(NON_COUNT_AGGREGATES)("a SAVED dataset stored with '*' under %s → 400 even when the query selects only its count", async (aggregate) => { + const before = { ...reads }; + const res = await post({ datasetName: `stored_star_${aggregate}`, selection: { measures: ['row_count'], dimensions: ['category'] } }); + expectRefusedBeforeAnyRead(res, before, `stored ${aggregate}, count selected`); + }); + it("CONTROL: count over '*' and a count with no field both answer the row count — 200, through this strategy", async () => { const before = { ...reads }; const res = await post({ dataset: COUNT_ONLY, selection: { measures: ['row_count', 'star_count'], dimensions: ['category'] } }); diff --git a/packages/spec/liveness/analytics_cube.json b/packages/spec/liveness/analytics_cube.json index 10a2c0e7fc5..0da5164142d 100644 --- a/packages/spec/liveness/analytics_cube.json +++ b/packages/spec/liveness/analytics_cube.json @@ -60,10 +60,10 @@ }, "sql": { "status": "live", - "verifiedAt": "2026-09-30", + "verifiedAt": "2026-10-02", "evidence": "packages/services/service-analytics/src/strategies/native-sql-strategy.ts#resolveMeasureSql — `measure.sql` is the column the aggregate is applied to (`'*'` the COUNT(*) form), and `qualifyAndRegisterJoin(measure.sql, …)` is what lowers a dotted reference into a LEFT JOIN chain; packages/services/service-analytics/src/strategies/objectql-strategy.ts#resolveFieldName reads `measure.sql.replace(/^\\$/, '')` as the aggregate field for the `engine.aggregate` path; packages/services/service-analytics/src/analytics-service.ts#fieldsOfColumnSql resolves it to the field(s) the analytics door's field-level read gate judges.", "producer": "packages/cli/src/commands/serve.ts#CAPABILITY_PROVIDERS — the `analytics` entry declares `configKey: 'analyticsCubes'` and the capability resolver threads it into the plugin (`const cubes = (config as any).analyticsCubes ?? (config as any).cubes ?? []; arg = { cubes }`); packages/services/service-analytics/src/analytics-service.ts#registerAll (`if (config.cubes) this.cubeRegistry.registerAll(config.cubes)`) is where the authored array becomes the registry every consumer below resolves through. Without this thread an authored cube reaches no reader at all — the `seed.env` shape (#4837).", - "note": "REQUIRED. NARROWED 2026-09-30 (#20943, maintainer ruling D; ADR-0021 zero raw expressions, ADR-0049 enforce-or-remove): the value is a COLUMN REFERENCE — a bare identifier, a dotted identifier path (relationship hops, then the column), or `'*'`, which the schema admits on any measure (the count-only boundary is #21000's) — and any SQL expression is refused at parse with a prescription naming the ADR-0021 dataset form (a measure-scoped `filter`, `derived: { op, of }`). The admitted pattern is the one `IDENTIFIER_PATH` (native-sql-strategy.ts) and the read gate already use to tell a column path from an expression (#4157), so every admitted value other than `'*'` (which reads no field value) resolves to a field the gate can judge. Still LIVE: the key itself is unchanged and read at the three sites above. The runtime's expression branches (verbatim emit on the raw-SQL path, the gate's stand-down) remain for a cube that reaches the service without meeting the parse; their deletion is the services-lane follow-up. The D3 entry is `cube-member-sql-expression-retired`; there is no D2 conversion, because an expression has no mechanical rewrite into a dataset." + "note": "REQUIRED. NARROWED 2026-09-30 (#20943, maintainer ruling D; ADR-0021 zero raw expressions, ADR-0049 enforce-or-remove): the value is a COLUMN REFERENCE — a bare identifier, a dotted identifier path (relationship hops, then the column), or `'*'` — and any SQL expression is refused at parse with a prescription naming the ADR-0021 dataset form (a measure-scoped `filter`, `derived: { op, of }`). The admitted pattern is the one `IDENTIFIER_PATH` (native-sql-strategy.ts) and the read gate already use to tell a column path from an expression (#4157), so every admitted value other than `'*'` (which reads no field value) resolves to a field the gate can judge. Still LIVE: the key itself is unchanged and read at the three sites above. The runtime's expression branches (verbatim emit on the raw-SQL path, the gate's stand-down) remain for a cube that reaches the service without meeting the parse; their deletion is the services-lane follow-up. The D3 entry is `cube-member-sql-expression-retired`; there is no D2 conversion, because an expression has no mechanical rewrite into a dataset. NARROWED again 2026-10-02 (#21409, the count-only boundary): `'*'` is admitted only under `type: 'count'` — the measure's refinement asks the one predicate it shares with the dataset measure (`data/analytics-column-reference.ts#rowWildcardOutsideCount`), and refuses `'*'` under any other type at `sql`, naming the slot and prescribing a `count` or a column; a dataset `sum` over `'*'`, which compiles to this very member, answered 500 DATABASE_ERROR at `POST /analytics/dataset/query` on both strategies before the rule. Cross-field, so a declared dropped-refinement site (`data/Metric` in `dropped-refinements.baseline.json`). D3 entry `analytics-row-wildcard-outside-count-refused`; no D2 conversion." }, "format": { "status": "live", @@ -104,10 +104,10 @@ }, "sql": { "status": "live", - "verifiedAt": "2026-09-30", + "verifiedAt": "2026-10-02", "evidence": "packages/services/service-analytics/src/strategies/native-sql-strategy.ts#resolveDimensionSql — `dim.sql` is the GROUP BY / SELECT column, and `qualifyAndRegisterJoin(dim.sql, …)` lowers a dotted path into the LEFT JOIN chain; packages/services/service-analytics/src/strategies/objectql-strategy.ts#resolveFieldName reads `dim.sql.replace(/^\\$/, '')` as the group-by field for the `engine.aggregate` path; packages/services/service-analytics/src/analytics-service.ts#fieldsOfColumnSql resolves it to the field(s) the analytics door's field-level read gate judges.", "producer": "packages/cli/src/commands/serve.ts#CAPABILITY_PROVIDERS — the `analytics` entry declares `configKey: 'analyticsCubes'` and the capability resolver threads it into the plugin (`const cubes = (config as any).analyticsCubes ?? (config as any).cubes ?? []; arg = { cubes }`); packages/services/service-analytics/src/analytics-service.ts#registerAll (`if (config.cubes) this.cubeRegistry.registerAll(config.cubes)`) is where the authored array becomes the registry every consumer below resolves through. Without this thread an authored cube reaches no reader at all — the `seed.env` shape (#4837).", - "note": "REQUIRED. NARROWED 2026-09-30 with `measures.sql` (#20943, maintainer ruling D): a bare identifier or a dotted identifier path, and `'*'`, which the schema admits on a dimension too (the accept set ruling D's execution parameters name for both members; it never made a working query, and narrowing it is #21000's); a SQL expression — a CASE bucket included — is refused at parse. A bucket computed over a column's values has no expression form in the cube or the dataset layer; the prescription sends it to a field of the object that a dimension names. Still LIVE, at the sites above." + "note": "REQUIRED. NARROWED 2026-09-30 with `measures.sql` (#20943, maintainer ruling D): a bare identifier or a dotted identifier path; a SQL expression — a CASE bucket included — is refused at parse. NARROWED again 2026-10-02 (#21409, the count-only boundary): `'*'`, which ruling D's execution parameters had named in the accept set of both members and which never made a working query on a dimension (`GROUP BY *`), is refused too — no aggregate consumes it on a dimension, so the slot takes `ANALYTICS_COLUMN_PATH` (`data/analytics-column-reference.ts`), the dataset dimension's own pattern, and the prescription sends a row count to a `count` measure. A pattern, so the published JSON Schema carries it. D3 entry `analytics-row-wildcard-outside-count-refused`. A bucket computed over a column's values has no expression form in the cube or the dataset layer; the prescription sends it to a field of the object that a dimension names. Still LIVE, at the sites above." }, "granularities": { "status": "live", diff --git a/packages/spec/liveness/dataset.json b/packages/spec/liveness/dataset.json index cc02fd29415..f48c30aecd0 100644 --- a/packages/spec/liveness/dataset.json +++ b/packages/spec/liveness/dataset.json @@ -94,8 +94,8 @@ "field": { "status": "live", "evidence": "packages/services/service-analytics/src/dataset-compiler.ts#compileDataset (`sql: m.field ?? '*'` — `count` with no field aggregates over rows — guarded by `assertDeclared(m.field, 'measure', m.name)`)", - "note": "SQL aggregate operand (count omits field → '*'); relationship-validated. NARROWED 2026-10-01 (#21220, with `dimensions.field`): the value is a COLUMN REFERENCE — a bare identifier, a dotted identifier path, or `'*'` — exactly the cube member `sql` accept set, from the one shared declaration in `data/analytics-column-reference.ts`. A SQL expression and a padded or empty string are refused at parse (a count spells \"no field\" by omitting the key), with a prescription naming the ADR-0021 form (a measure-scoped `filter`, `derived: { op, of }`). The one lossless repair is the D2 conversion `dataset-count-measure-empty-field-removed`: a `count` measure's empty `field` is dropped from stored rows and sources and still counts rows; the D3 entry `dataset-member-field-expression-refused` carries the rest. Still LIVE, at the site above; the dataset door's 403 for an expression (#21190) stays as defence in depth. 2026-08-28: RE-ANCHORED (commit 9ee2dcfbd) and REPOINTED — `:170` had rotted into a comment about members the spec now refuses at parse. Re-closed by hand against c459da6bc; DATED for the first time (this file carried no `verifiedAt` anywhere, which the review accepting that re-anchoring sorts as OLDEST).", - "verifiedAt": "2026-10-01" + "note": "SQL aggregate operand (count omits field → '*'); relationship-validated. NARROWED 2026-10-01 (#21220, with `dimensions.field`): the value is a COLUMN REFERENCE — a bare identifier, a dotted identifier path, or `'*'` — exactly the cube member `sql` accept set, from the one shared declaration in `data/analytics-column-reference.ts`. NARROWED again 2026-10-02 (#21409): `'*'` only under `aggregate: 'count'` — the measure's refinement asks the one predicate it shares with the cube measure (`rowWildcardOutsideCount`), since `sum` / `avg` / `min` / `max` / `count_distinct` over `'*'` answered 500 DATABASE_ERROR at `POST /analytics/dataset/query` on both strategies; a declared dropped-refinement site (`ui/DatasetMeasure`), D3 entry `analytics-row-wildcard-outside-count-refused`. A SQL expression and a padded or empty string are refused at parse (a count spells \"no field\" by omitting the key), with a prescription naming the ADR-0021 form (a measure-scoped `filter`, `derived: { op, of }`). The one lossless repair is the D2 conversion `dataset-count-measure-empty-field-removed`: a `count` measure's empty `field` is dropped from stored rows and sources and still counts rows; the D3 entry `dataset-member-field-expression-refused` carries the rest. Still LIVE, at the site above; the dataset door's 403 for an expression (#21190) stays as defence in depth. 2026-08-28: RE-ANCHORED (commit 9ee2dcfbd) and REPOINTED — `:170` had rotted into a comment about members the spec now refuses at parse. Re-closed by hand against c459da6bc; DATED for the first time (this file carried no `verifiedAt` anywhere, which the review accepting that re-anchoring sorts as OLDEST).", + "verifiedAt": "2026-10-02" }, "filter": { "status": "live", diff --git a/packages/spec/src/data/analytics-column-reference.ts b/packages/spec/src/data/analytics-column-reference.ts index 436b0518c4b..2bb692602dc 100644 --- a/packages/spec/src/data/analytics-column-reference.ts +++ b/packages/spec/src/data/analytics-column-reference.ts @@ -9,8 +9,8 @@ * `sql` of the cube member it compiles to, verbatim. So the two slots take one * accept set, from this module: * - * - the cube layer — `MetricSchema.sql` / `DimensionSchema.sql` - * (`./analytics.zod.ts`, as its `CUBE_MEMBER_SQL`; #20943); + * - the cube layer — `MetricSchema.sql` (`./analytics.zod.ts`, as its + * `CUBE_MEMBER_SQL`; #20943) / `DimensionSchema.sql`; * - the dataset layer — `DatasetMeasureSchema.field` / * `DatasetDimensionSchema.field` (`../ui/dataset.zod.ts`; #21220). * @@ -28,25 +28,42 @@ * string. Such a value names no single field, so no platform check could judge * which fields it reads. * - * ## The row wildcard, and the one restriction stated here + * ## The row wildcard: admitted only where a `count` consumes it (#21409) * * `'*'` is the row wildcard — what a `count` aggregates (`COUNT(*)`), reading - * no field value. {@link ANALYTICS_COLUMN_REFERENCE} admits it, for the slots - * whose ruling admits it: both cube members (maintainer ruling D on #20943 named - * one accept set "on a measure and a dimension alike") and a dataset MEASURE. - * {@link ANALYTICS_COLUMN_PATH} is the same path WITHOUT that arm, for the one - * slot where the wildcard has no meaning: a dataset DIMENSION. Grouping by - * every column at once is not an axis, and the runtime never answered one — - * measured on #21220 at `POST /analytics/dataset/query`, a dimension whose - * `field` is `'*'` compiled to `SELECT * AS … GROUP BY *` on the native-SQL - * strategy and to `groupBy: ['*']` on the ObjectQL one, and was answered - * `500 DATABASE_ERROR` on both. Both patterns are built from the one - * {@link COLUMN_PATH} source below: one pattern, one stated restriction, never a - * second copy that can drift. + * no field value. It is admitted in exactly one place: a MEASURE whose + * aggregate is `count`. Everywhere else it names nothing the runtime can + * answer, and it is refused at parse. Two layers of the rule, one per kind of + * slot: * - * They are `RegExp`s for `.regex()`, never refinements, so the published JSON - * Schema carries each as a `pattern`: a document validated against - * `json-schema/**` is judged as the parse judges it. + * - **A dimension** — a cube dimension's `sql` and a dataset dimension's + * `field` — has no aggregate at all, so no `count` can ever consume the + * wildcard there. Both take {@link ANALYTICS_COLUMN_PATH}, the column path + * WITHOUT the `'*'` arm. Grouping by every column at once is not an axis, + * and the runtime never answered one — measured on #21220 at + * `POST /analytics/dataset/query`, a dimension whose `field` is `'*'` + * compiled to `SELECT * AS … GROUP BY *` on the native-SQL strategy and to + * `groupBy: ['*']` on the ObjectQL one, and was answered + * `500 DATABASE_ERROR` on both. A pattern, so the published JSON Schema + * carries this half as written. + * - **A measure** — a cube measure's `sql` and a dataset measure's `field` — + * takes {@link ANALYTICS_COLUMN_REFERENCE}, the path WITH the `'*'` arm, + * and the wildcard is then judged against the measure's aggregate by the ONE + * predicate {@link rowWildcardOutsideCount}, which both measure schemas call + * from a refinement and neither restates. Under any aggregate other than + * `count` the strategies emitted `SUM(*)`, `AVG(*)`, `MIN(*)` / `MAX(*)` or + * `COUNT(DISTINCT *)` — measured on a dataset `sum` over `'*'` at the same + * door: `500 DATABASE_ERROR` on both strategies. + * + * The measure half is CROSS-FIELD (the slot and the aggregate beside it), so it + * is a refinement rather than a pattern, and a refinement does not reach the + * published JSON Schema: each site is declared in + * `dropped-refinements.baseline.json` and named on the artifact as + * `x-dropped-refinements`. A document a JSON-Schema validator accepts with + * `'*'` under a non-count aggregate is refused at parse. + * + * Both patterns are built from the one {@link COLUMN_PATH} source below: one + * pattern, one stated restriction, never a second copy that can drift. * * A module of its own, and outside the `data` barrel, so the two layers share * one declaration without it becoming published API (the @@ -58,12 +75,60 @@ const COLUMN_PATH = '[A-Za-z_][A-Za-z0-9_]*(?:\\.[A-Za-z_][A-Za-z0-9_]*)*'; /** * A column of the object, a relationship path ending in one, or the row - * wildcard `'*'` — a cube member's `sql` and a dataset measure's `field`. + * wildcard `'*'` — a MEASURE's slot: a cube measure's `sql` and a dataset + * measure's `field`. The wildcard arm is admitted by the pattern and then held + * to a `count` by {@link rowWildcardOutsideCount}. */ export const ANALYTICS_COLUMN_REFERENCE = new RegExp(`^(?:\\*|${COLUMN_PATH})$`); /** * {@link ANALYTICS_COLUMN_REFERENCE} without the row wildcard — a column of the - * object or a relationship path ending in one: a dataset dimension's `field`. + * object or a relationship path ending in one: a DIMENSION's slot, a cube + * dimension's `sql` and a dataset dimension's `field`. */ export const ANALYTICS_COLUMN_PATH = new RegExp(`^${COLUMN_PATH}$`); + +/** The row wildcard — what a `count` aggregates (`COUNT(*)`). */ +const ROW_WILDCARD = '*'; + +/** The one aggregate that consumes {@link ROW_WILDCARD}. */ +const ROW_WILDCARD_AGGREGATE = 'count'; + +/** + * THE rule for the row wildcard in a measure (#21409): `true` when `reference` + * is `'*'` and the measure's `aggregate` is anything but `count` — another + * aggregate, or none at all. Called from the refinement of both measure + * schemas (`MetricSchema`, with its `type`; `DatasetMeasureSchema`, with its + * `aggregate`); neither restates it. + * + * Both arguments are `unknown` on purpose: a refinement runs on a value whose + * other keys may already have failed their own checks, and the predicate + * answers only the one question it is asked. + */ +export function rowWildcardOutsideCount(reference: unknown, aggregate: unknown): boolean { + return reference === ROW_WILDCARD && aggregate !== ROW_WILDCARD_AGGREGATE; +} + +/** + * The refusal {@link rowWildcardOutsideCount} is answered with, worded once for + * both measure slots: it names the slot and the aggregate the author wrote, and + * prescribes the two ways out — a `count`, or a column. + * + * @param slot - the slot as the author reads it, e.g. `measures..sql` + * @param aggregateKey - the key that names the measure's aggregate there + * (`type` on a cube measure, `aggregate` on a dataset measure) + * @param aggregate - the value the author wrote under `aggregateKey` + */ +export function rowWildcardOutsideCountRefusal(slot: string, aggregateKey: string, aggregate: unknown): string { + const under = typeof aggregate === 'string' + ? `under \`${aggregateKey}: '${aggregate}'\`` + : `with no \`${aggregateKey}\``; + return ( + `\`${slot}\` is the row wildcard \`'*'\` ${under}. \`'*'\` is what a \`count\` aggregates ` + + '(`COUNT(*)`): it reads no field value, so it is admitted only on a `count` measure, and any other ' + + 'aggregate over it names no column to read — the analytics strategies sent it to the database ' + + 'as written, and the query failed there. ' + + `Declare \`${aggregateKey}: '${ROW_WILDCARD_AGGREGATE}'\` to count rows, or name the column this measure ` + + 'aggregates: a field of the object (`amount`) or a relationship path ending in one (`account.amount`).' + ); +} diff --git a/packages/spec/src/data/analytics-row-wildcard-count-only.test.ts b/packages/spec/src/data/analytics-row-wildcard-count-only.test.ts new file mode 100644 index 00000000000..edf638c235f --- /dev/null +++ b/packages/spec/src/data/analytics-row-wildcard-count-only.test.ts @@ -0,0 +1,272 @@ +// Copyright (c) 2026 ObjectStack. Licensed under the Apache-2.0 license. + +/** + * The row wildcard `'*'` is admitted only where a `count` consumes it (#21409; + * ADR-0049 enforce-or-remove) — the shared rule in + * `./analytics-column-reference.ts`. + * + * What is pinned here, position by position: + * 1. A cube measure's `sql` refuses `'*'` under every `type` but `count`, at + * `sql` (code `custom`, the measure's refinement). + * 2. A cube dimension's `sql` refuses `'*'` under every dimension `type`, at + * `sql` (code `invalid_format`, the column-path pattern without the + * wildcard arm — the dataset dimension's own pattern). + * 3. A dataset measure's `field` refuses `'*'` under every aggregate but + * `count`, and on a measure with no aggregate (a `derived` one), at + * `field` (code `custom`). + * CONTROL: a dataset dimension's `field` already refused `'*'`. + * 4. `count` over `'*'` is admitted byte-identically on both measure slots, + * and so is a dataset count with no `field`. + * 5. ONE predicate: both measure refinements ask `rowWildcardOutsideCount` + * and word the refusal with the one shared builder — the issue each + * raises is exactly the builder's output for its slot. + * 6. Every door that carries a cube or a dataset refuses it: the two + * container schemas, the `analytics_cube` and `dataset` write-door + * bindings, `defineCube()` and `defineStack()` (with its + * STACK_SCHEMA_INVALID / 422 envelope). + * 7. The measure half is cross-field, so it does not reach the published + * JSON Schema: the dimension's published `pattern` refuses `'*'`, the + * measure slots' still admits it, and the two measure schemas' root + * refinements are declared in `dropped-refinements.baseline.json`. + * 8. ADR-0087: the family's D3 entry is registered under step 18, with no D2 + * conversion and no retired-key row. + * + * Runtime half — that `count` over `'*'` still RUNS on both strategies, and that + * the measured `500` at the dataset door is now a `400` — is pinned over the + * real route in `packages/rest/src/analytics-dataset-row-wildcard-door.test.ts`. + * + * On the assertion set: a schema refusal raises a `ZodError` whose issues carry + * `code` and `path` but no ADR-0112 `status` — that envelope belongs to the + * authoring door, `defineStack`, pinned with its `code` and `status`. + */ + +import { readFileSync } from 'node:fs'; +import { describe, expect, it } from 'vitest'; +import { z } from 'zod'; + +import { getMetadataTypeSchema } from '../kernel/metadata-type-schemas'; +import { MIGRATIONS_BY_MAJOR, RETIRED_KEYS_BY_MAJOR } from '../migrations/registry'; +import { defineStack } from '../stack.zod'; +import { DatasetDimensionSchema, DatasetMeasureSchema, DatasetSchema } from '../ui/dataset.zod'; +import { rowWildcardOutsideCount, rowWildcardOutsideCountRefusal } from './analytics-column-reference'; +import { + AggregationMetricType, + CubeSchema, + DimensionSchema, + DimensionType, + MetricSchema, + defineCube, +} from './analytics.zod'; +import { AggregationFunction } from './query.zod'; + +const D3_ID = 'analytics-row-wildcard-outside-count-refused'; + +/** Every cube measure `type` but `count` — derived from the enum, never restated. */ +const NON_COUNT_METRIC_TYPES = AggregationMetricType.options.filter((t) => t !== 'count'); +/** Every dataset measure `aggregate` but `count` — derived from the enum, never restated. */ +const NON_COUNT_AGGREGATES = AggregationFunction.options.filter((a) => a !== 'count'); + +const COUNT_METRIC = { label: 'Orders', type: 'count', sql: '*' } as const; +const STATUS = { label: 'Status', type: 'string', sql: 'status' } as const; +const CUBE = { name: 'orders', sql: 'order', measures: { count: COUNT_METRIC }, dimensions: { status: STATUS } } as const; + +const BASE = { name: 'sales', label: 'Sales', object: 'opportunity' } as const; +const STAGE = { name: 'stage', field: 'stage', type: 'string' } as const; +const COUNT = { name: 'opp_count', aggregate: 'count' } as const; + +type Issue = { code: string; path: PropertyKey[]; message: string }; +const refusalOf = (schema: { safeParse: (v: unknown) => { success: boolean; error?: { issues: Issue[] } } }, value: unknown) => { + const r = schema.safeParse(value); + expect(r.success, JSON.stringify(value)).toBe(false); + return r.error!.issues; +}; + +describe('the one predicate', () => { + it("is true only for '*' outside a count — another aggregate, or none", () => { + expect(rowWildcardOutsideCount('*', 'count')).toBe(false); + for (const aggregate of [...NON_COUNT_AGGREGATES, ...NON_COUNT_METRIC_TYPES, undefined]) { + expect(rowWildcardOutsideCount('*', aggregate), String(aggregate)).toBe(true); + } + // A column is never its business, under any aggregate. + for (const aggregate of ['count', 'sum', undefined]) { + expect(rowWildcardOutsideCount('amount', aggregate)).toBe(false); + expect(rowWildcardOutsideCount(undefined, aggregate)).toBe(false); + } + }); +}); + +describe("position 1 — a cube measure's `sql` admits '*' only under `type: 'count'`", () => { + it.each(NON_COUNT_METRIC_TYPES)("refuses '*' under `type: '%s'` at `sql`, with the shared refusal", (type) => { + const issues = refusalOf(MetricSchema, { label: 'M', type, sql: '*' }); + expect(issues.map((i) => [i.code, i.path])).toEqual([['custom', ['sql']]]); + // ONE spelling: the issue is the shared builder's output for this slot. + expect(issues[0]!.message).toBe(rowWildcardOutsideCountRefusal('measures..sql', 'type', type)); + // It names the slot and the aggregate written, and prescribes both ways out. + expect(issues[0]!.message).toContain('`measures..sql`'); + expect(issues[0]!.message).toContain(`\`type: '${type}'\``); + expect(issues[0]!.message).toContain("`type: 'count'`"); + expect(issues[0]!.message).toContain('name the column'); + }); + + it("CONTROL: `count` over '*' and every non-count type over a column parse byte-identically", () => { + const r = MetricSchema.safeParse(COUNT_METRIC); + expect(r.success).toBe(true); + if (r.success) expect(r.data).toEqual(COUNT_METRIC); + for (const type of NON_COUNT_METRIC_TYPES) { + const metric = { label: 'M', type, sql: 'account.amount' }; + const parsed = MetricSchema.safeParse(metric); + expect(parsed.success, type).toBe(true); + if (parsed.success) expect(parsed.data).toEqual(metric); + } + }); +}); + +describe("position 2 — a cube dimension's `sql` never admits '*'", () => { + it.each(DimensionType.options)("refuses '*' on a `%s` dimension at `sql`, prescribing a count measure", (type) => { + const issues = refusalOf(DimensionSchema, { label: 'D', type, sql: '*' }); + expect(issues.map((i) => [i.code, i.path])).toEqual([['invalid_format', ['sql']]]); + expect(issues[0]!.message).toContain("`'*'` is no dimension"); + expect(issues[0]!.message).toContain("`type: 'count'`"); + }); + + it('CONTROL: a column and a relationship path still parse byte-identically', () => { + for (const sql of ['status', 'account.owner.region']) { + const dim = { label: 'D', type: 'string', sql }; + const r = DimensionSchema.safeParse(dim); + expect(r.success, sql).toBe(true); + if (r.success) expect(r.data).toEqual(dim); + } + }); +}); + +describe("position 3 — a dataset measure's `field` admits '*' only under `aggregate: 'count'`", () => { + it.each(NON_COUNT_AGGREGATES)("refuses '*' under `aggregate: '%s'` at `field`, with the shared refusal", (aggregate) => { + const issues = refusalOf(DatasetMeasureSchema, { name: 'wildcard', aggregate, field: '*' }); + expect(issues.map((i) => [i.code, i.path])).toEqual([['custom', ['field']]]); + expect(issues[0]!.message).toBe(rowWildcardOutsideCountRefusal('measures[].field', 'aggregate', aggregate)); + expect(issues[0]!.message).toContain('`measures[].field`'); + expect(issues[0]!.message).toContain(`\`aggregate: '${aggregate}'\``); + expect(issues[0]!.message).toContain("`aggregate: 'count'`"); + }); + + it("refuses '*' on a measure with no aggregate — a `derived` one, which reads no `field`", () => { + const issues = refusalOf(DatasetMeasureSchema, { + name: 'done_rate', + derived: { op: 'ratio', of: ['done_count', 'task_count'] }, + field: '*', + }); + expect(issues.map((i) => [i.code, i.path])).toEqual([['custom', ['field']]]); + expect(issues[0]!.message).toBe(rowWildcardOutsideCountRefusal('measures[].field', 'aggregate', undefined)); + }); + + it("CONTROL: a count over '*', a count with no `field`, and a non-count over a column parse byte-identically", () => { + const measures = [ + COUNT, + { name: 'star_count', aggregate: 'count', field: '*' }, + ...NON_COUNT_AGGREGATES.map((aggregate) => ({ name: `amount_${aggregate}`, aggregate, field: 'amount' })), + ]; + const dataset = { ...BASE, dimensions: [STAGE], measures }; + const r = DatasetSchema.safeParse(dataset); + expect(r.success, JSON.stringify(r.error?.issues)).toBe(true); + if (r.success) expect(r.data).toEqual(dataset); + }); + + it("CONTROL: a dataset dimension's `field` already refused '*' — unchanged", () => { + const issues = refusalOf(DatasetDimensionSchema, { name: 'everything', field: '*' }); + expect(issues.map((i) => [i.code, i.path])).toEqual([['invalid_format', ['field']]]); + }); +}); + +describe("every door that carries a cube or a dataset refuses '*' outside a count", () => { + const cubeWithWildcards = { + ...CUBE, + measures: { ...CUBE.measures, total: { label: 'Total', type: 'sum', sql: '*' } }, + dimensions: { ...CUBE.dimensions, everything: { label: 'Everything', type: 'string', sql: '*' } }, + }; + const datasetWithWildcard = { ...BASE, dimensions: [STAGE], measures: [COUNT, { name: 'revenue', aggregate: 'sum', field: '*' }] }; + + it('CubeSchema and the `analytics_cube` write door — one issue per member, at its `sql`', () => { + const door = getMetadataTypeSchema('analytics_cube'); + expect(door).toBe(CubeSchema); + for (const schema of [CubeSchema, door!]) { + expect(refusalOf(schema, cubeWithWildcards).map((i) => [i.code, i.path])).toEqual([ + ['custom', ['measures', 'total', 'sql']], + ['invalid_format', ['dimensions', 'everything', 'sql']], + ]); + } + }); + + it('`defineCube()` refuses it', () => { + expect(() => defineCube(cubeWithWildcards as never)).toThrow(/is the row wildcard `'\*'` under `type: 'sum'`/); + }); + + it('DatasetSchema and the `dataset` write door — at `measures.N.field`', () => { + const door = getMetadataTypeSchema('dataset'); + expect(door).toBe(DatasetSchema); + for (const schema of [DatasetSchema, door!]) { + expect(refusalOf(schema, datasetWithWildcard).map((i) => [i.code, i.path])).toEqual([ + ['custom', ['measures', 1, 'field']], + ]); + } + }); + + it('the authoring door, defineStack, refuses both with the STACK_SCHEMA_INVALID envelope', () => { + const stack = (extra: Record) => ({ + manifest: { id: 'com.example.row-wildcard', name: 'row_wildcard', version: '1.0.0', type: 'app' }, + ...extra, + }); + let thrown: unknown; + try { + defineStack(stack({ analyticsCubes: [cubeWithWildcards], datasets: [datasetWithWildcard] }) as never); + } catch (e) { + thrown = e; + } + const refusal = thrown as { code?: string; status?: number; issues?: Issue[] }; + expect(refusal?.code).toBe('STACK_SCHEMA_INVALID'); + expect(refusal?.status).toBe(422); + expect(refusal.issues?.map((i) => i.path)).toEqual(expect.arrayContaining([ + ['analyticsCubes', 0, 'measures', 'total', 'sql'], + ['analyticsCubes', 0, 'dimensions', 'everything', 'sql'], + ['datasets', 0, 'measures', 1, 'field'], + ])); + expect(refusal.issues).toHaveLength(3); + // CONTROL: the same stack with its members counting '*' is accepted by the same door. + expect(() => defineStack(stack({ analyticsCubes: [CUBE], datasets: [{ ...BASE, dimensions: [STAGE], measures: [COUNT] }] }) as never)).not.toThrow(); + }); +}); + +describe('the published JSON Schema: the dimension half is a pattern, the measure half a declared dropped refinement', () => { + const propertiesOf = (schema: z.ZodType) => + (z.toJSONSchema(schema, { io: 'input', unrepresentable: 'any' }) as { + properties: Record; + }).properties; + + it("a cube dimension's `sql` publishes the dataset dimension's own pattern, which refuses '*'", () => { + const cubeDimension = propertiesOf(DimensionSchema).sql!.pattern!; + expect(cubeDimension).toBe(propertiesOf(DatasetDimensionSchema).field!.pattern!); + expect(new RegExp(cubeDimension).test('*')).toBe(false); + }); + + it("the measure slots' published pattern still admits '*' — the cross-field half is a refinement, and both roots are ledgered", () => { + expect(new RegExp(propertiesOf(MetricSchema).sql!.pattern!).test('*')).toBe(true); + expect(new RegExp(propertiesOf(DatasetMeasureSchema).field!.pattern!).test('*')).toBe(true); + const ledger = JSON.parse(readFileSync(new URL('../../dropped-refinements.baseline.json', import.meta.url), 'utf8')) as { + entries: Record; + }; + expect(ledger.entries['data/Metric']?.sites).toContain(''); + expect(ledger.entries['ui/DatasetMeasure']?.sites).toContain(''); + }); +}); + +describe('ADR-0087 registration', () => { + it('carries the family D3 entry under step 18, with no D2 conversion and no retired-key row', () => { + const d3 = MIGRATIONS_BY_MAJOR[18]!.semantic.find((s) => s.id === D3_ID); + expect(d3, 'the family D3 entry').toBeDefined(); + expect(d3!.reason.length).toBeGreaterThan(0); + expect(d3!.acceptanceCriteria.length).toBeGreaterThan(0); + // No lossless rewrite exists: `count` changes the figure, and a column is the author's to name. + expect(d3!.conversionIds ?? []).toEqual([]); + // No key left any shape, so no `${defKey}:${name}` entry is owed. + expect(RETIRED_KEYS_BY_MAJOR[18]!.filter((k) => /(Metric|Dimension):sql$|DatasetMeasure:field$/.test(k))).toEqual([]); + }); +}); diff --git a/packages/spec/src/data/analytics.zod.ts b/packages/spec/src/data/analytics.zod.ts index 9663bfa5037..2f1336febb1 100644 --- a/packages/spec/src/data/analytics.zod.ts +++ b/packages/spec/src/data/analytics.zod.ts @@ -22,7 +22,12 @@ import { DateGranularity } from './query.zod'; import { lazySchema } from '../shared/lazy-schema'; import { strictObject } from '../shared/strict-object'; import { retiredKey } from '../shared/retired-key'; -import { ANALYTICS_COLUMN_REFERENCE } from './analytics-column-reference'; +import { + ANALYTICS_COLUMN_PATH, + ANALYTICS_COLUMN_REFERENCE, + rowWildcardOutsideCount, + rowWildcardOutsideCountRefusal, +} from './analytics-column-reference'; import { MetadataProtectionFields } from '../kernel/metadata-protection.zod'; export const AggregationMetricType = z.enum([ 'count', @@ -205,7 +210,13 @@ const CUBE_DIMENSION_NAME_REMOVED = cubeMemberNameRemoved('dimensions. strictObject( * The column the measure aggregates — a field of the cube's object, a * relationship path ending in one, or `'*'` for a count. A SQL expression * is refused at parse (#20943, ruling D; see `CUBE_MEMBER_SQL`): a derived - * value is declared on an ADR-0021 dataset instead. + * value is declared on an ADR-0021 dataset instead. `'*'` under any `type` + * but `count` is refused by the schema's refinement below (#21409). */ sql: z.string().regex(CUBE_MEMBER_SQL, { error: () => CUBE_METRIC_SQL_EXPRESSION_REFUSED }).describe( 'Column reference: a field of the cube\'s object ("amount"), a relationship path ending in one ' @@ -359,7 +373,22 @@ export const MetricSchema = lazySchema(() => strictObject( + 'Relayed verbatim as fields[].format on POST /analytics/query results, and on the measure by GET /analytics/meta.', ), }, -)); +).superRefine((metric, ctx) => { + // [#21409] `'*'` is the row wildcard a `count` aggregates (`COUNT(*)`), and + // only a `count` consumes it: under any other `type` it names no column, and + // the strategies emitted `SUM(*)` / `AVG(*)` / … verbatim, which the database + // refused. Cross-field (the `sql` and the `type` beside it), so a refinement + // — declared as a dropped-refinement site, since no JSON-Schema keyword + // carries it. The rule is the ONE predicate both measure schemas share + // (`./analytics-column-reference.ts`); the dataset measure calls the same one. + if (rowWildcardOutsideCount(metric.sql, metric.type)) { + ctx.addIssue({ + code: 'custom', + path: ['sql'], + message: rowWildcardOutsideCountRefusal('measures..sql', 'type', metric.type), + }); + } +})); /** * Dimension Schema @@ -394,11 +423,13 @@ export const DimensionSchema = lazySchema(() => strictObject( /** * The column the dimension groups by — a field of the cube's object, or a - * relationship path ending in one (`'*'` is admitted with the measure's - * accept set). A SQL expression is refused at parse (#20943, ruling D; see - * `CUBE_MEMBER_SQL`). + * relationship path ending in one. A SQL expression is refused at parse + * (#20943, ruling D; see `CUBE_MEMBER_SQL`), and so is the row wildcard + * `'*'` (#21409): no aggregate consumes it on a dimension, so the slot + * takes {@link ANALYTICS_COLUMN_PATH} — the measure's path without the + * wildcard arm, the pattern a dataset dimension's `field` already takes. */ - sql: z.string().regex(CUBE_MEMBER_SQL, { error: () => CUBE_DIMENSION_SQL_EXPRESSION_REFUSED }).describe( + sql: z.string().regex(ANALYTICS_COLUMN_PATH, { error: () => CUBE_DIMENSION_SQL_EXPRESSION_REFUSED }).describe( 'Column reference: a field of the cube\'s object ("status") or a relationship path ending in one ' + '("account.industry"). Never a SQL expression.', ), diff --git a/packages/spec/src/data/cube-member-sql-column-reference.test.ts b/packages/spec/src/data/cube-member-sql-column-reference.test.ts index b4df04f92a9..f32de331fe3 100644 --- a/packages/spec/src/data/cube-member-sql-column-reference.test.ts +++ b/packages/spec/src/data/cube-member-sql-column-reference.test.ts @@ -8,14 +8,16 @@ * What is pinned here, door by door: * 1. The accept set the ruling's execution parameters name — a bare * identifier, a dotted identifier path, and `'*'` — parses - * byte-identically to before, on a measure and a dimension alike. + * byte-identically to before, on a measure and a dimension alike, with + * one later narrowing (#21409): `'*'` only on a `count` measure. That + * half is pinned in `analytics-row-wildcard-count-only.test.ts`. * 2. The refusal: every other value — an expression, a quoted or * `$`-prefixed spelling, an empty string, a broken path — is refused at * `…sql` with the prescription, whose first sentence states the contract * and whose body names the ADR-0021 dataset form. * 3. The rule is a `pattern` in the published JSON Schema too, so a * document validated against `json-schema/**` is judged as the parse - * judges it. + * judges it — a measure's with the `'*'` arm, a dimension's without it. * 4. Every door that carries a cube refuses it: `CubeSchema`, the * `analytics_cube` write-door binding, `defineCube()` and `defineStack()` * (the last with its STACK_SCHEMA_INVALID / 422 envelope). @@ -116,8 +118,9 @@ describe('cube member sql — the accept set is unchanged for every column refer }); it('a dimension admits a bare column and a relationship path — parsed byte-identically', () => { - // `'*'` is in the accept set the execution parameters name for both members. - for (const sql of ['status', 'account.industry', 'account.owner.region', '_private', '*']) { + // `'*'` left a dimension's accept set with #21409: no aggregate consumes it + // there (refused in `analytics-row-wildcard-count-only.test.ts`). + for (const sql of ['status', 'account.industry', 'account.owner.region', '_private']) { const dim = { label: 'D', type: 'string', sql }; const r = DimensionSchema.safeParse(dim); expect(r.success, sql).toBe(true); @@ -162,7 +165,7 @@ describe('cube member sql — an expression is refused at parse, with the prescr }); describe('cube member sql — the rule reaches the published JSON Schema as a pattern', () => { - it('both members carry the same `pattern` on `sql`, and it judges values as the parse does', () => { + it('both members carry a `pattern` on `sql` — the dimension\'s is the measure\'s without the `*` arm — and each judges values as the parse does', () => { const metricSql = (z.toJSONSchema(MetricSchema, { io: 'input', unrepresentable: 'any' }) as { properties: Record; }).properties.sql!; @@ -170,10 +173,22 @@ describe('cube member sql — the rule reaches the published JSON Schema as a pa properties: Record; }).properties.sql!; expect(metricSql.pattern).toBeDefined(); - expect(dimensionSql.pattern).toBe(metricSql.pattern); - const pattern = new RegExp(metricSql.pattern!); - for (const sql of ['amount', 'account.amount', '*']) expect(pattern.test(sql), sql).toBe(true); - for (const sql of EXPRESSIONS) expect(pattern.test(sql), sql).toBe(false); + // [#21409] One column path; the dimension states the one restriction, the + // row-wildcard arm, and nothing else. + expect(metricSql.pattern!.startsWith('^(?:\\*|') && metricSql.pattern!.endsWith(')$')).toBe(true); + expect(dimensionSql.pattern).toBe(`^${metricSql.pattern!.slice('^(?:\\*|'.length, -')$'.length)}$`); + const metric = new RegExp(metricSql.pattern!); + const dimension = new RegExp(dimensionSql.pattern!); + for (const sql of ['amount', 'account.amount']) { + expect(metric.test(sql), sql).toBe(true); + expect(dimension.test(sql), sql).toBe(true); + } + expect(metric.test('*')).toBe(true); + expect(dimension.test('*')).toBe(false); + for (const sql of EXPRESSIONS) { + expect(metric.test(sql), sql).toBe(false); + expect(dimension.test(sql), sql).toBe(false); + } }); }); diff --git a/packages/spec/src/migrations/entries/semantic/18.analytics-row-wildcard-outside-count-refused.ts b/packages/spec/src/migrations/entries/semantic/18.analytics-row-wildcard-outside-count-refused.ts new file mode 100644 index 00000000000..e4d7c977bc3 --- /dev/null +++ b/packages/spec/src/migrations/entries/semantic/18.analytics-row-wildcard-outside-count-refused.ts @@ -0,0 +1,64 @@ +// Copyright (c) 2026 ObjectStack. Licensed under the Apache-2.0 license. + +import type { SemanticMigration } from '../../types.js'; + +// #21409 (ADR-0049 enforce-or-remove) — the row wildcard `'*'` is admitted only +// where a `count` consumes it: a cube measure under `type: 'count'` and a dataset +// measure under `aggregate: 'count'`. A cube dimension's `sql` takes the column +// path without the wildcard arm (the dataset dimension's pattern since +// `dataset-member-field-expression-refused`), and both measure slots ask one +// shared predicate. Semantic only, with no D2 conversion: such a member never +// produced an answer, and there is no lossless rewrite — `count` changes the +// figure the author asked for, and a column is the author's to name. +export const entry: SemanticMigration = { + id: 'analytics-row-wildcard-outside-count-refused', + // No backticks and no pipes in `surface` — build-upgrade-guide.ts renders it + // inside a code span AND a table cell. + surface: + 'analyticsCubes[].measures..sql, analyticsCubes[].dimensions..sql and ' + + 'datasets[].measures[].field (data.MetricSchema.sql / data.DimensionSchema.sql / ' + + 'ui.DatasetMeasureSchema.field) authored as the row wildcard * where no count consumes it — a cube ' + + 'measure whose type is anything but count, any cube dimension, and a dataset measure whose ' + + 'aggregate is anything but count or that declares none (a derived measure)', + replacement: + 'what the member meant. A row count: `type: \'count\'` on a cube measure or `aggregate: \'count\'` ' + + 'on a dataset measure, keeping `\'*\'` (a dataset count may also omit `field`). An aggregate of ' + + 'values: the column it aggregates — a field of the object (`amount`) or a relationship path ' + + 'ending in one (`account.amount`). A cube dimension: the column it groups by; to count rows, ' + + 'declare a `count` measure instead. A `derived` measure: delete the `field` key, which nothing ' + + 'read — a derived measure combines other measures by name', + reason: + '`\'*\'` is the row wildcard a `count` aggregates (`COUNT(*)`): it reads no field value, so no ' + + 'other aggregate has a column to read over it, and a dimension has no aggregate at all. The ' + + 'contract nevertheless admitted it in a cube member\'s `sql` on any measure and on a dimension, ' + + 'and in a dataset measure\'s `field` under any aggregate, and the analytics strategies passed it ' + + 'to the database as written. Measured at POST /api/v1/analytics/dataset/query over a real ' + + 'SQLite driver, on the native-SQL and the ObjectQL strategy alike: a dataset measure aggregating ' + + '`\'*\'` under `sum`, `avg`, `min`, `max` or `count_distinct` answered 500 DATABASE_ERROR — a ' + + 'server fault for an authoring mistake the contract had admitted. A dataset measure compiles to ' + + 'the cube measure it names verbatim, so the same reading covers an authored cube measure; a ' + + 'dimension over `\'*\'` (GROUP BY *) was measured the same way when the dataset dimension was ' + + 'narrowed. Such a member never produced an answer, so no working document changes meaning: the ' + + 'failure moves from the query to the authoring parse, which names the slot and the aggregate and ' + + 'prescribes a `count` or a column. There is no D2 conversion: rewriting to `count` would change ' + + 'the figure the author asked for, and only the author knows which column a sum over `\'*\'` was ' + + 'meant to read. A STORED document is not rewritten: a metadata read still serves it as stored, ' + + 'with the refusal on its read diagnostics, and a re-save through the metadata write door is ' + + 'refused at the slot. The dataset query door parses every dataset it is handed, inline or saved, ' + + 'so a stored dataset carrying such a measure is refused 400 VALIDATION_FAILED on EVERY query — ' + + 'including a query that selects only its other measures, which used to answer: it fails closed ' + + 'until the member is fixed. An authored cube reaches the analytics runtime through the stack ' + + 'definition, whose parse refuses it when the stack is built. In-repo census before the change: no ' + + 'example, platform object, doc, skill or fixture authored one, and neither did objectui at the ' + + 'pinned commit; deployed metadata was NOT measured. ADR-0021 / ADR-0049 / ADR-0087', + acceptanceCriteria: + 'Every analytics cube and dataset parses: `CubeSchema`, `DatasetSchema`, the analytics_cube and ' + + 'dataset write doors, defineCube and defineStack refuse `\'*\'` on a cube measure whose `type` is ' + + 'not `count` and on a dataset measure whose `aggregate` is not `count` (at its `sql` / `field`, ' + + 'code custom), and on a cube dimension (at its `sql`, code invalid_format), each naming the slot ' + + 'and prescribing a `count` or a column, so the sweep is mechanical — parse each document, and each ' + + 'refusal is one member to change. For each changed member, a query that selects it returns a ' + + 'figure instead of a 500, and a dashboard bound to a stored dataset that carried one answers ' + + 'again on every widget. A `count` over `\'*\'`, a dataset count with no `field`, and every member ' + + 'that names a column parse byte-identically to before and run on both strategies.', +}; diff --git a/packages/spec/src/migrations/registry.ts b/packages/spec/src/migrations/registry.ts index 87acefd17c3..29eb7a4b7b7 100644 --- a/packages/spec/src/migrations/registry.ts +++ b/packages/spec/src/migrations/registry.ts @@ -6964,6 +6964,66 @@ const step18: MigrationStep = { + 'query that carried a refused value was never returning one window, so re-check what the ' + 'widget was meant to show rather than trusting the old result set.', }, + // #21409 (ADR-0049 enforce-or-remove) — the row wildcard `'*'` is admitted only + // where a `count` consumes it: a cube measure under `type: 'count'` and a dataset + // measure under `aggregate: 'count'`. A cube dimension's `sql` takes the column + // path without the wildcard arm (the dataset dimension's pattern since + // `dataset-member-field-expression-refused`), and both measure slots ask one + // shared predicate. Semantic only, with no D2 conversion: such a member never + // produced an answer, and there is no lossless rewrite — `count` changes the + // figure the author asked for, and a column is the author's to name. + { + id: 'analytics-row-wildcard-outside-count-refused', + // No backticks and no pipes in `surface` — build-upgrade-guide.ts renders it + // inside a code span AND a table cell. + surface: + 'analyticsCubes[].measures..sql, analyticsCubes[].dimensions..sql and ' + + 'datasets[].measures[].field (data.MetricSchema.sql / data.DimensionSchema.sql / ' + + 'ui.DatasetMeasureSchema.field) authored as the row wildcard * where no count consumes it — a cube ' + + 'measure whose type is anything but count, any cube dimension, and a dataset measure whose ' + + 'aggregate is anything but count or that declares none (a derived measure)', + replacement: + 'what the member meant. A row count: `type: \'count\'` on a cube measure or `aggregate: \'count\'` ' + + 'on a dataset measure, keeping `\'*\'` (a dataset count may also omit `field`). An aggregate of ' + + 'values: the column it aggregates — a field of the object (`amount`) or a relationship path ' + + 'ending in one (`account.amount`). A cube dimension: the column it groups by; to count rows, ' + + 'declare a `count` measure instead. A `derived` measure: delete the `field` key, which nothing ' + + 'read — a derived measure combines other measures by name', + reason: + '`\'*\'` is the row wildcard a `count` aggregates (`COUNT(*)`): it reads no field value, so no ' + + 'other aggregate has a column to read over it, and a dimension has no aggregate at all. The ' + + 'contract nevertheless admitted it in a cube member\'s `sql` on any measure and on a dimension, ' + + 'and in a dataset measure\'s `field` under any aggregate, and the analytics strategies passed it ' + + 'to the database as written. Measured at POST /api/v1/analytics/dataset/query over a real ' + + 'SQLite driver, on the native-SQL and the ObjectQL strategy alike: a dataset measure aggregating ' + + '`\'*\'` under `sum`, `avg`, `min`, `max` or `count_distinct` answered 500 DATABASE_ERROR — a ' + + 'server fault for an authoring mistake the contract had admitted. A dataset measure compiles to ' + + 'the cube measure it names verbatim, so the same reading covers an authored cube measure; a ' + + 'dimension over `\'*\'` (GROUP BY *) was measured the same way when the dataset dimension was ' + + 'narrowed. Such a member never produced an answer, so no working document changes meaning: the ' + + 'failure moves from the query to the authoring parse, which names the slot and the aggregate and ' + + 'prescribes a `count` or a column. There is no D2 conversion: rewriting to `count` would change ' + + 'the figure the author asked for, and only the author knows which column a sum over `\'*\'` was ' + + 'meant to read. A STORED document is not rewritten: a metadata read still serves it as stored, ' + + 'with the refusal on its read diagnostics, and a re-save through the metadata write door is ' + + 'refused at the slot. The dataset query door parses every dataset it is handed, inline or saved, ' + + 'so a stored dataset carrying such a measure is refused 400 VALIDATION_FAILED on EVERY query — ' + + 'including a query that selects only its other measures, which used to answer: it fails closed ' + + 'until the member is fixed. An authored cube reaches the analytics runtime through the stack ' + + 'definition, whose parse refuses it when the stack is built. In-repo census before the change: no ' + + 'example, platform object, doc, skill or fixture authored one, and neither did objectui at the ' + + 'pinned commit; deployed metadata was NOT measured. ADR-0021 / ADR-0049 / ADR-0087', + acceptanceCriteria: + 'Every analytics cube and dataset parses: `CubeSchema`, `DatasetSchema`, the analytics_cube and ' + + 'dataset write doors, defineCube and defineStack refuse `\'*\'` on a cube measure whose `type` is ' + + 'not `count` and on a dataset measure whose `aggregate` is not `count` (at its `sql` / `field`, ' + + 'code custom), and on a cube dimension (at its `sql`, code invalid_format), each naming the slot ' + + 'and prescribing a `count` or a column, so the sweep is mechanical — parse each document, and each ' + + 'refusal is one member to change. For each changed member, a query that selects it returns a ' + + 'figure instead of a 500, and a dashboard bound to a stored dataset that carried one answers ' + + 'again on every widget. A `count` over `\'*\'`, a dataset count with no `field`, and every member ' + + 'that names a column parse byte-identically to before and run on both strategies.', + }, { id: 'analytics-time-dimension-date-range-vocabulary-closed', // No backticks in `surface` — build-upgrade-guide.ts renders it inside a diff --git a/packages/spec/src/ui/dataset.zod.ts b/packages/spec/src/ui/dataset.zod.ts index 525625a9315..59baf54f097 100644 --- a/packages/spec/src/ui/dataset.zod.ts +++ b/packages/spec/src/ui/dataset.zod.ts @@ -9,7 +9,12 @@ import { analyticsCarrierFilter } from './analytics-carrier-filter'; import { SnakeCaseIdentifierSchema } from '../shared/identifiers.zod'; import { I18nLabelSchema } from './i18n.zod'; import { AggregationFunction, DateGranularity } from '../data/query.zod'; -import { ANALYTICS_COLUMN_PATH, ANALYTICS_COLUMN_REFERENCE } from '../data/analytics-column-reference'; +import { + ANALYTICS_COLUMN_PATH, + ANALYTICS_COLUMN_REFERENCE, + rowWildcardOutsideCount, + rowWildcardOutsideCountRefusal, +} from '../data/analytics-column-reference'; /** * Analytics Dataset — the one semantic layer (ADR-0021). @@ -98,11 +103,12 @@ const DATASET_NO_SQL = * empty `field` (it skips one); a stored `count` measure with `field: ''` is * repaired on load by the D2 conversion `dataset-count-measure-empty-field-removed`. * - * A measure admits the row wildcard `'*'` (a count's `COUNT(*)`); a dimension - * does not — the restriction, and its measurement, are stated on - * {@link ANALYTICS_COLUMN_PATH}. An empty string is refused on both: on a - * measure the wildcard's spelling is `'*'` or no `field` at all, and on a - * dimension it names nothing to group by. + * A measure admits the row wildcard `'*'` (a count's `COUNT(*)`) under + * `aggregate: 'count'` only — the measure's refinement asks the one shared + * predicate (#21409); a dimension does not admit it at all. Both restrictions, + * and their measurements, are stated in `../data/analytics-column-reference.ts`. + * An empty string is refused on both: on a measure the wildcard's spelling is + * `'*'` or no `field` at all, and on a dimension it names nothing to group by. */ const DATASET_FIELD_EXPRESSION_REFUSED = 'A SQL expression there names no single field, so no platform check can judge which fields it reads, ' @@ -239,15 +245,16 @@ export const DatasetMeasureSchema = lazySchema(() => strictObject({ aggregate: AggregationFunction.optional().describe('Aggregation (sum/avg/count/...); omit when `derived` is set') .meta({ title: 'Aggregate' }), /** - * Base field, or `relationship[.relationship].field` path, or `'*'`. Optional - * for `count` (count(*)). A column reference only (#21220, see - * `DATASET_FIELD_EXPRESSION_REFUSED`): a SQL expression or an empty string is - * refused at parse. + * Base field, or `relationship[.relationship].field` path, or `'*'` for a + * `count`. Optional for `count` (count(*)). A column reference only (#21220, + * see `DATASET_FIELD_EXPRESSION_REFUSED`): a SQL expression or an empty string + * is refused at parse, and `'*'` under any other aggregate is refused by the + * schema's refinement below (#21409). */ field: z.string() .regex(ANALYTICS_COLUMN_REFERENCE, { error: () => DATASET_MEASURE_FIELD_NOT_COLUMN }) .optional() - .describe('Aggregated field: a base field, a relationship path, or "*"; optional for count(*). Never a SQL expression.') + .describe('Aggregated field: a base field, a relationship path, or "*" for a count; optional for count(*). Never a SQL expression.') .meta({ title: 'Field' }), /** * Measure-scoped filter (e.g. only won deals for "won_amount"). [#20080] A @@ -395,6 +402,21 @@ export const DatasetMeasureSchema = lazySchema(() => strictObject({ /** Names of other measures in this dataset (2+ for ratio/difference). */ of: z.array(SnakeCaseIdentifierSchema).min(1), }).optional().meta({ title: 'Derived From' }), +}).superRefine((measure, ctx) => { + // [#21409] `'*'` is the row wildcard a `count` aggregates (`COUNT(*)`), and + // only a `count` consumes it: under any other aggregate — or none, as on a + // `derived` measure — it names no column. Measured at + // `POST /analytics/dataset/query` before this rule: `{ aggregate: 'sum', + // field: '*' }` compiled to `SUM(*)` and answered 500 on both strategies. + // Cross-field, so a refinement (a declared dropped-refinement site). The rule + // is the ONE predicate the cube measure calls too (`MetricSchema`). + if (rowWildcardOutsideCount(measure.field, measure.aggregate)) { + ctx.addIssue({ + code: 'custom', + path: ['field'], + message: rowWildcardOutsideCountRefusal('measures[].field', 'aggregate', measure.aggregate), + }); + } })); /** From b74aada7e0a75cd18ae895060c5de468f1d2bc1b Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 2 Oct 2026 14:36:15 +0000 Subject: [PATCH 3/6] chore(spec): declare the two measure refinements' dropped-refinement sites and regenerate the dataset reference page Claude-Session: https://claude.ai/code/session_01UtnxvdiN376GF3sgXwAw4d Co-authored-by: Claude --- content/docs/references/ui/dataset.mdx | 4 ++-- .../spec/dropped-refinements.baseline.json | 24 +++++++++++++++++-- 2 files changed, 24 insertions(+), 4 deletions(-) diff --git a/content/docs/references/ui/dataset.mdx b/content/docs/references/ui/dataset.mdx index be509f53cea..3b379a2c613 100644 --- a/content/docs/references/ui/dataset.mdx +++ b/content/docs/references/ui/dataset.mdx @@ -83,7 +83,7 @@ const result = DatasetSchema.parse(data); | **name** | `string` | ✅ | Measure name — e.g. "revenue"; defined once | | **label** | `string \| Record` | optional | Display label — the default-language string, or an inline locale map (`{ en, "zh-CN" }`) resolved at render time | | **aggregate** | `Enum<'count' \| 'sum' \| 'avg' \| 'min' \| 'max' \| 'count_distinct'>` | optional | Aggregation (sum/avg/count/...); omit when `derived` is set | -| **field** | `string` | optional | Aggregated field: a base field, a relationship path, or "*"; optional for count(*). Never a SQL expression. | +| **field** | `string` | optional | Aggregated field: a base field, a relationship path, or "*" for a count; optional for count(*). Never a SQL expression. | | **filter** | `any` | optional | | | **format** | `string` | optional | Numeral pattern for a NUMERIC measure — grouping, decimals, percent; e.g. "0,0.00", "0.0%". An amount takes its symbol from `currency`, not from a "$" in the pattern. A DATE-valued measure never reads a date pattern: `"YYYY-MM-DD"` renders that arm's default face. A date or datetime value reads `format` as a display style — `short` or `relative`, honoured on both. | | **currency** | `string` | optional | Display currency code (ISO 4217) | @@ -124,7 +124,7 @@ const result = DatasetSchema.parse(data); | **name** | `string` | ✅ | Measure name — e.g. "revenue"; defined once | | **label** | `string \| Record` | optional | Display label — the default-language string, or an inline locale map (`{ en, "zh-CN" }`) resolved at render time | | **aggregate** | `Enum<'count' \| 'sum' \| 'avg' \| 'min' \| 'max' \| 'count_distinct'>` | optional | Aggregation (sum/avg/count/...); omit when `derived` is set | -| **field** | `string` | optional | Aggregated field: a base field, a relationship path, or "*"; optional for count(*). Never a SQL expression. | +| **field** | `string` | optional | Aggregated field: a base field, a relationship path, or "*" for a count; optional for count(*). Never a SQL expression. | | **filter** | `any` | optional | | | **format** | `string` | optional | Numeral pattern for a NUMERIC measure — grouping, decimals, percent; e.g. "0,0.00", "0.0%". An amount takes its symbol from `currency`, not from a "$" in the pattern. A DATE-valued measure never reads a date pattern: `"YYYY-MM-DD"` renders that arm's default face. A date or datetime value reads `format` as a display style — `short` or `relative`, honoured on both. | | **currency** | `string` | optional | Display currency code (ISO 4217) | diff --git a/packages/spec/dropped-refinements.baseline.json b/packages/spec/dropped-refinements.baseline.json index a2ab87151b8..172d1501d2f 100644 --- a/packages/spec/dropped-refinements.baseline.json +++ b/packages/spec/dropped-refinements.baseline.json @@ -2,8 +2,8 @@ "description": "Shrink-only ledger of every PUBLISHED JSON Schema that is STILL WIDER than the Zod type it was generated from, because a rule written as `.refine()` reaches the runtime and not the file (#18670). `z.toJSONSchema()` has no arm for a `custom` check: a plain record, the same record with a `.refine()`, and the same record with an ABORTING `.refine()` all project byte-identically (measured on zod 4.4.3, the version packages/spec resolves). So a document one of these files ACCEPTS can still be refused at parse time, and an author -- or an AI -- validating against packages/spec/json-schema/** finds out a release later. Each `sites` path is a position under that schema at which a refinement is dropped; the same paths are written onto the artifact itself as `x-dropped-refinements`. Item 2 closed the first patterns: a refinement DECLARED through the closed list in src/shared/refinement-projection.ts is emitted into the published file, reads `projected` rather than `dropped`, and its row LEAVES this ledger in the same PR -- which is why the ledger shrinks and never grows on a repair. Every refinement outside that closed list stays here, and adding an arm to the list is a public-contract decision, not a refactor. Hand-edited on purpose and with no `gen:` script: a generator would let a new gap be admitted by running a command instead of by a decision, which is the silence this ledger exists to end. Adding, removing or moving a site fails packages/spec/scripts/build-schemas.ts until the line moves with it, and the failure prints the corrected entry in full. ⛔ Do not delete or weaken a refinement to shorten this file -- the runtime rule is correct; it is the projection that is silent, and the remedy is to teach the closed list a NAMED pattern, never to drop the rule.", "measured": { "zod": "4.4.3", - "publishedSchemasWithDroppedRefinements": 215, - "droppedRefinementSites": 635, + "publishedSchemasWithDroppedRefinements": 217, + "droppedRefinementSites": 647, "refinementSitesThatDidProject": 369, "refinementSitesWithNoJsonFormToCompare": 0 }, @@ -64,6 +64,7 @@ "manifest.actions.element.in.ai.outputSchema", "manifest.actions.element.in.params.element.in", "manifest.agents.element.structuredOutput.schema", + "manifest.analyticsCubes.element.measures.valueType", "manifest.connectors.element.out", "manifest.dashboards.element.globalFilters.element", "manifest.dashboards.element.widgets.element", @@ -71,6 +72,7 @@ "manifest.datasets.element", "manifest.datasets.element.filter", "manifest.datasets.element.include.element", + "manifest.datasets.element.measures.element", "manifest.datasets.element.measures.element.filter", "manifest.datasources.element", "manifest.flows.element", @@ -169,6 +171,7 @@ "data.options[1].manifest.actions.element.in.ai.outputSchema", "data.options[1].manifest.actions.element.in.params.element.in", "data.options[1].manifest.agents.element.structuredOutput.schema", + "data.options[1].manifest.analyticsCubes.element.measures.valueType", "data.options[1].manifest.connectors.element.out", "data.options[1].manifest.dashboards.element.globalFilters.element", "data.options[1].manifest.dashboards.element.widgets.element", @@ -176,6 +179,7 @@ "data.options[1].manifest.datasets.element", "data.options[1].manifest.datasets.element.filter", "data.options[1].manifest.datasets.element.include.element", + "data.options[1].manifest.datasets.element.measures.element", "data.options[1].manifest.datasets.element.measures.element.filter", "data.options[1].manifest.datasources.element", "data.options[1].manifest.flows.element", @@ -259,6 +263,7 @@ "options[1].manifest.actions.element.in.ai.outputSchema", "options[1].manifest.actions.element.in.params.element.in", "options[1].manifest.agents.element.structuredOutput.schema", + "options[1].manifest.analyticsCubes.element.measures.valueType", "options[1].manifest.connectors.element.out", "options[1].manifest.dashboards.element.globalFilters.element", "options[1].manifest.dashboards.element.widgets.element", @@ -266,6 +271,7 @@ "options[1].manifest.datasets.element", "options[1].manifest.datasets.element.filter", "options[1].manifest.datasets.element.include.element", + "options[1].manifest.datasets.element.measures.element", "options[1].manifest.datasets.element.measures.element.filter", "options[1].manifest.datasources.element", "options[1].manifest.flows.element", @@ -309,6 +315,7 @@ "data.packages.element.options[1].manifest.actions.element.in.ai.outputSchema", "data.packages.element.options[1].manifest.actions.element.in.params.element.in", "data.packages.element.options[1].manifest.agents.element.structuredOutput.schema", + "data.packages.element.options[1].manifest.analyticsCubes.element.measures.valueType", "data.packages.element.options[1].manifest.connectors.element.out", "data.packages.element.options[1].manifest.dashboards.element.globalFilters.element", "data.packages.element.options[1].manifest.dashboards.element.widgets.element", @@ -316,6 +323,7 @@ "data.packages.element.options[1].manifest.datasets.element", "data.packages.element.options[1].manifest.datasets.element.filter", "data.packages.element.options[1].manifest.datasets.element.include.element", + "data.packages.element.options[1].manifest.datasets.element.measures.element", "data.packages.element.options[1].manifest.datasets.element.measures.element.filter", "data.packages.element.options[1].manifest.datasources.element", "data.packages.element.options[1].manifest.flows.element", @@ -551,6 +559,11 @@ "" ] }, + "data/Cube": { + "sites": [ + "measures.valueType" + ] + }, "data/DataEngineAggregateOptions": { "sites": [ "filter.options[1].lazy" @@ -726,6 +739,11 @@ "key" ] }, + "data/Metric": { + "sites": [ + "" + ] + }, "data/MongoConfig": { "sites": [ "", @@ -1211,11 +1229,13 @@ "filter", "filter.lazy", "include.element", + "measures.element", "measures.element.filter" ] }, "ui/DatasetMeasure": { "sites": [ + "", "filter", "filter.lazy" ] From 934b70a2db7ec1c1a76fc8cf589dced1ba659196 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 2 Oct 2026 15:01:37 +0000 Subject: [PATCH 4/6] chore(changeset): the analytics row wildcard is count-only (breaking minor, narrowing) Claude-Session: https://claude.ai/code/session_01UtnxvdiN376GF3sgXwAw4d Co-authored-by: Claude --- ...21409-analytics-row-wildcard-count-only.md | 115 ++++++++++++++++++ 1 file changed, 115 insertions(+) create mode 100644 .changeset/21409-analytics-row-wildcard-count-only.md diff --git a/.changeset/21409-analytics-row-wildcard-count-only.md b/.changeset/21409-analytics-row-wildcard-count-only.md new file mode 100644 index 00000000000..141512a1103 --- /dev/null +++ b/.changeset/21409-analytics-row-wildcard-count-only.md @@ -0,0 +1,115 @@ +--- +'@objectstack/spec': minor +--- + +feat(spec)!: the analytics row wildcard `'*'` is admitted only where a `count` consumes it — a cube or dataset measure over `'*'` under any other aggregate, and a cube dimension over `'*'`, are refused at parse (#21409) + +Clause-②: no (narrowing) + +**BREAKING** — shipped as `minor` under the launch-window convention +(`check-changeset-no-major` refuses `major` until GA; breaking-ness is carried by +this banner, the `(narrowing)` arm above and the ADR-0087 disposition below, +never by the level). + +`'*'` is the row wildcard: what a `count` aggregates (`COUNT(*)`), reading no +field value. It is now admitted in exactly one place, a measure that counts: + +- `MetricSchema.sql` — a cube measure's `sql` — admits `'*'` under + `type: 'count'` only; under any other `type` it is refused at `sql` + (code `custom`). +- `DatasetMeasureSchema.field` — an ADR-0021 dataset measure's `field` — admits + `'*'` under `aggregate: 'count'` only; under any other aggregate, or on a + measure with no aggregate (a `derived` one), it is refused at `field` + (code `custom`). A count may still omit `field`. +- `DimensionSchema.sql` — a cube dimension's `sql` — never admits `'*'` + (code `invalid_format`): it takes the column path without the wildcard arm, + the pattern a dataset dimension's `field` already takes. + +Each refusal names the slot and the aggregate the author wrote, and prescribes +the two ways out: a `count`, or a column. A column or a relationship path parses +byte-identically to before on every slot, and so does a `count` over `'*'`. + +Why: no aggregate but `count` has a column to read over `'*'`, and a dimension +has no aggregate at all, yet the contract admitted the wildcard on any measure +and on a cube dimension, and the analytics strategies sent it to the database as +written. Measured at `POST /api/v1/analytics/dataset/query` over a real SQLite +driver, on the native-SQL and the ObjectQL strategy alike: a dataset measure +aggregating `'*'` under `sum`, `avg`, `min`, `max` or `count_distinct` answered +`500 DATABASE_ERROR`. A dataset measure compiles to the cube measure it names +verbatim, so the same reading covers an authored cube measure. Such a member +never produced an answer, so no working document changes meaning: the failure +moves from the query to the authoring parse. The two measure slots ask ONE +shared predicate; the rule is cross-field (the slot and its aggregate), so it is +a refinement, which the published JSON Schema cannot carry — both sites are +declared in `dropped-refinements.baseline.json`. The dimension half is a +`pattern`, so `json-schema/**` states it. + +## FROM → TO + +``` +FROM defineDataset({ name: 'deal_metrics', label: 'Deal Metrics', object: 'deal', + dimensions: [{ name: 'stage', field: 'stage' }], + measures: [{ name: 'deals', aggregate: 'sum', field: '*' }] }) + -> parsed; every query selecting `deals` answered 500 DATABASE_ERROR +TO -> ZodError at measures.0.field (custom): + `measures[].field` is the row wildcard `'*'` under `aggregate: 'sum'`. … + + measures: [{ name: 'deals', aggregate: 'count' }] // a row count + measures: [{ name: 'deal_value', aggregate: 'sum', field: 'amount' }] // an aggregate of a column + +FROM defineCube({ name: 'deals', sql: 'deal', + measures: { total: { label: 'Total', type: 'sum', sql: '*' } }, + dimensions: { everything: { label: 'All', type: 'string', sql: '*' } } }) +TO -> refused at measures.total.sql (custom) and dimensions.everything.sql (invalid_format) + + measures: { total: { label: 'Total', type: 'sum', sql: 'amount' } }, + dimensions: { stage: { label: 'Stage', type: 'string', sql: 'stage' } } +``` + +**The one-line fix:** parse each cube and dataset; every refusal at `…sql` / +`…field` naming `'*'` is one member to change — declare a `count` to count rows, +or name the column the measure aggregates (a dimension names the column it +groups by). On a `derived` dataset measure, delete `field`: nothing read it. +There is no mechanical rewrite, so `os migrate meta` lists nothing for it. + +**What a stored document meets.** A metadata read still serves it as stored, +with the refusal on its read diagnostics (`_diagnostics`), and a re-save through +the metadata write door is refused at the slot. `POST +/api/v1/analytics/dataset/query` parses every dataset it is handed, inline or +saved, so a stored dataset carrying such a measure answers `400 +VALIDATION_FAILED` at `measures.N.field` on every query — including a query that +selects only its other measures, which used to answer — until the member is +fixed: it fails closed. An authored cube reaches the analytics runtime through +the stack definition, whose parse refuses it. + +## The kit + +- **Schema.** `data/analytics-column-reference.ts` (not published API) declares + the predicate `rowWildcardOutsideCount` and its refusal once; `MetricSchema` + and `DatasetMeasureSchema` call both from a refinement, and + `DimensionSchema.sql` takes `ANALYTICS_COLUMN_PATH`. No export, key or enum + member changes, so the api-surface, authorable-surface and JSON-schema + manifest ratchets are unchanged. +- **ADR-0087.** D3 entry `analytics-row-wildcard-outside-count-refused`. No D2 + conversion: rewriting to `count` would change the figure the author asked for, + and only the author can name the column. No `RETIRED_KEYS_BY_MAJOR` row. +- **Dropped refinements.** `data/Metric` and `ui/DatasetMeasure` gain their root + site, and every published schema embedding them gains the embedded site. +- **Liveness.** `analytics_cube` `measures.sql` / `dimensions.sql` and `dataset` + `measures.field` stay `live`, re-verified, their notes re-pointed here. +- **Docs.** The `ui/dataset` reference page is regenerated. +- **Runtime.** Unchanged. + +## Reach, measured + +- This repository: no example, platform object, doc, skill, script or test + fixture authors `'*'` outside a `count` at the three slots (`git grep` of every + `field` / `sql` value spelled `'*'`, 173 hits, each read in its enclosing + object: 154 under a `count`, the rest QueryAST aggregations, comments and + strategy-level literals). One spec pin admitted `'*'` on a cube dimension; it + now pins the refusal. +- objectui at the pinned `.objectui-sha`: zero `field` / `sql` values spelled + `'*'` (lit controls: 51 `aggregate: 'sum'`, 438 `field: 'amount'`). +- Out-of-repo authored metadata: NOT MEASURED. + + From b79d1be80d2a37297f27f7c84716ca4b7001fcfc Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 2 Oct 2026 16:14:21 +0000 Subject: [PATCH 5/6] chore(changeset): attribute the dataset FROM/TO refusal to the doors that parse defineDataset is an identity function and parses nothing; the dataset half of the FROM/TO block now names DatasetSchema.parse, defineStack (422) and the dataset query route (400), as measured. Claude-Session: https://claude.ai/code/session_01UtnxvdiN376GF3sgXwAw4d Co-authored-by: Claude --- .../21409-analytics-row-wildcard-count-only.md | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/.changeset/21409-analytics-row-wildcard-count-only.md b/.changeset/21409-analytics-row-wildcard-count-only.md index 141512a1103..23398a9f2ae 100644 --- a/.changeset/21409-analytics-row-wildcard-count-only.md +++ b/.changeset/21409-analytics-row-wildcard-count-only.md @@ -47,12 +47,16 @@ declared in `dropped-refinements.baseline.json`. The dimension half is a ## FROM → TO ``` -FROM defineDataset({ name: 'deal_metrics', label: 'Deal Metrics', object: 'deal', +FROM { name: 'deal_metrics', label: 'Deal Metrics', object: 'deal', dimensions: [{ name: 'stage', field: 'stage' }], - measures: [{ name: 'deals', aggregate: 'sum', field: '*' }] }) - -> parsed; every query selecting `deals` answered 500 DATABASE_ERROR -TO -> ZodError at measures.0.field (custom): + measures: [{ name: 'deals', aggregate: 'sum', field: '*' }] } + -> DatasetSchema.parse accepted it; a dataset query selecting `deals` + answered 500 DATABASE_ERROR +TO -> DatasetSchema.parse throws a ZodError at measures.0.field (custom): `measures[].field` is the row wildcard `'*'` under `aggregate: 'sum'`. … + defineStack({ datasets }) refuses it at datasets.N.measures.0.field (422 + STACK_SCHEMA_INVALID), and POST /api/v1/analytics/dataset/query answers + 400 VALIDATION_FAILED for an inline or a saved copy measures: [{ name: 'deals', aggregate: 'count' }] // a row count measures: [{ name: 'deal_value', aggregate: 'sum', field: 'amount' }] // an aggregate of a column From fce0c9a0426da813ce02cc38cfa814dcc776f768 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 2 Oct 2026 16:40:35 +0000 Subject: [PATCH 6/6] chore(changeset): os migrate meta lists the entry as a manual change; it rewrites nothing Claude-Session: https://claude.ai/code/session_01UtnxvdiN376GF3sgXwAw4d Co-authored-by: Claude --- .changeset/21409-analytics-row-wildcard-count-only.md | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/.changeset/21409-analytics-row-wildcard-count-only.md b/.changeset/21409-analytics-row-wildcard-count-only.md index 23398a9f2ae..6eb81a6057a 100644 --- a/.changeset/21409-analytics-row-wildcard-count-only.md +++ b/.changeset/21409-analytics-row-wildcard-count-only.md @@ -74,7 +74,9 @@ TO -> refused at measures.total.sql (custom) and dimensions.everything.sql (i `…field` naming `'*'` is one member to change — declare a `count` to count rows, or name the column the measure aggregates (a dimension names the column it groups by). On a `derived` dataset measure, delete `field`: nothing read it. -There is no mechanical rewrite, so `os migrate meta` lists nothing for it. +There is no mechanical rewrite: `os migrate meta` rewrites nothing for it, and +lists the entry `analytics-row-wildcard-outside-count-refused` as a manual +change that requires your judgment. **What a stored document meets.** A metadata read still serves it as stored, with the refusal on its read diagnostics (`_diagnostics`), and a re-save through