Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
45 changes: 36 additions & 9 deletions packages/qa/dogfood/test/expression-conformance.ledger.ts
Original file line number Diff line number Diff line change
Expand Up @@ -56,8 +56,35 @@ export type ExprMode = 'compile' | 'interpret';
*/
export type ExprDialect = 'cel' | 'cron' | 'template' | 'js' | 'settings-visibility';
export type ExprState = 'enforced' | 'experimental' | 'removed';
/** ADR-0058 D5 fail-policy tiers. */
export type FailPolicy = 'compile-error' | 'fail-closed' | 'fail-soft-log' | 'throw';
/**
* ADR-0058 D5 fail-policy tiers, plus the one state D5 has no tier for.
*
* The four D5 tiers each describe what an EVALUATOR does when the expression is
* bad. `unevaluated` is the fifth member because some slots have no evaluator at
* all: nothing reads the value, so the honest answer to "what happens when the
* expression is bad" is "nothing happens", and every one of the four would be a
* stronger claim than the row can support.
*
* That is not a hypothetical shortfall — it is what the rows carrying it were
* doing before this member existed. `compile-error` claimed the Zod parse was
* the refusal (true as far as it goes, and weaker than it reads: the parse
* judges the value's SHAPE and never its grammar, so it is a refusal every row
* in this ledger shares), and `fail-closed` claimed a RUNTIME refusal on a slot
* whose own `enforcement` cell said `(no runtime consumer yet)`. On a
* security-flavoured row that second one reads as a security guarantee, which
* is the borrowing with a consequence past legibility.
*
* ⛔ `unevaluated` is NOT a synonym for `experimental`, and NOT a tier for "not
* measured here". `cel-inline-grid-cell` is `experimental` and stays
* `fail-soft-log`: its enforcement cell names an evaluator it could not reach
* (the objectui renderer, outside this checkout) and reports a MEASURED write
* path on which nothing refuses. An absence this checkout cannot establish is
* not an absence this ledger may assert — that is the invented cell the ledger
* exists to prevent, in the other direction. The companion test pins the
* distinction so the new word cannot be borrowed the way the old ones were:
* see `NAMES_RUNTIME_EVALUATOR` / `DECLARES_NO_EVALUATOR` there.
*/
export type FailPolicy = 'compile-error' | 'fail-closed' | 'fail-soft-log' | 'throw' | 'unevaluated';

export interface ExprSurface extends ConformanceRow {
dialect: ExprDialect;
Expand Down Expand Up @@ -346,7 +373,7 @@ export const EXPRESSION_SURFACE: ExprSurface[] = [
// The key and its documented hand-off arrived with #14825.
id: 'cron-knowledge-refresh',
summary: 'knowledge-source periodic reindex cron (KnowledgeRefreshPolicy.cron) — surfaced, deliberately not scheduled',
dialect: 'cron', mode: 'interpret', state: 'experimental', failPolicy: 'compile-error',
dialect: 'cron', mode: 'interpret', state: 'experimental', failPolicy: 'unevaluated',
enforcement:
'PARSE ONLY — `CronExpressionInputSchema` refuses a blank/non-string, non-envelope value and normalizes to `{dialect:"cron",source}`; nothing evaluates the result. service-knowledge/knowledge-service.ts reads `refresh.onRecordChange` and NEVER `refresh.cron` (measured: the only `refresh` reads in that package are the two `onRecordChange` sites)',
covers: ['ai/knowledge-source.zod.ts:KnowledgeRefreshPolicySchema.cron'],
Expand All @@ -357,7 +384,7 @@ export const EXPRESSION_SURFACE: ExprSurface[] = [
// and #15028 (the envelope arm now pins the dialect — the note's last sentence).
id: 'cron-declared-unwired',
summary: 'cron slots on subsystems that were declared but never built — export schedules, flow schedule state, connector sync, cache warmup, DR backup/test',
dialect: 'cron', mode: 'interpret', state: 'experimental', failPolicy: 'compile-error',
dialect: 'cron', mode: 'interpret', state: 'experimental', failPolicy: 'unevaluated',
enforcement:
'PARSE ONLY — `CronExpressionInputSchema` refuses a blank/non-string, non-envelope value and normalizes to the envelope; NO EVALUATOR FOUND for any of these five keys. Reader hunt, per key, walking out from each declaration (2026-09-04, `61821e54cf5`): `api/export.zod.ts:cronExpression` — the whole `ExportJobApiContracts` family has zero consumers and rest-server serves no `/api/v1/data/export` route, so `POST /api/v1/data/export/schedules` is a declared contract nothing implements; `IExportService` has no provider binding, which its own source already records. `automation/execution.zod.ts:cronExpression` — `ScheduleStateSchema` has no consumer outside packages/spec; the schedule TRIGGER that does work reads a flow start node `config.schedule` through trigger-schedule/schedule-trigger.ts `normalizeSchedule`, a different shape this key never reaches. `integration/connector.zod.ts:schedule` — `syncConfig` has no reader outside packages/spec. `system/cache.zod.ts:schedule` (CacheWarmup) and `system/disaster-recovery.zod.ts:schedule` (BackupConfig + the DR `testing` block) — neither schema has any consumer outside packages/spec',
covers: [
Expand All @@ -367,15 +394,15 @@ export const EXPRESSION_SURFACE: ExprSurface[] = [
'system/cache.zod.ts:CacheWarmupSchema.schedule',
'system/disaster-recovery.zod.ts:BackupConfigSchema.schedule', 'system/disaster-recovery.zod.ts:DisasterRecoveryPlanSchema.schedule',
],
note: 'EXPERIMENTAL — five declared cron slots with no runtime evaluator (ADR-0049 enforce-or-remove candidates; each wants its own look, and the card that surfaced them says so rather than proposing a sweep). ⚠️ TWO of these surfaces are declared TWICE: `api/export.zod.ts` `cronExpression` on `ScheduledExportSchema` and on `ScheduleExportRequestSchema`, and `system/disaster-recovery.zod.ts` `schedule` on `BackupConfigSchema` and on `DisasterRecoveryPlanSchema` (the DR `testing` block). Both pairs are genuinely the same surface twice, so one row is honest here — and now that each declaring position carries its OWN key, that judgement is written out as two `covers` entries instead of being assumed by a collapse. ⚠️ The `failPolicy` on this row is `compile-error` because the PARSE is the only thing that ever refuses one of these values; it is not a claim that cron SYNTAX is checked. It is not: `@objectstack/formula` cronEngine validates 5/6-field patterns and `@` aliases, and has ZERO consumers outside packages/formula — nothing routes these slots through it. The parse now DOES pin these slots to the cron dialect (the sibling finding on the dialect union is closed): the envelope arm of `CronExpressionInputSchema` accepts a `cron` envelope only and its bare-string arm refuses a blank string, each with one issue at the slot naming the fix — and it still judges no cron syntax, by position: no grammar is restated in spec; `croner` judges the pattern where a schedule is wired (`cron-job-schedule`).',
note: 'EXPERIMENTAL — five declared cron slots with no runtime evaluator (ADR-0049 enforce-or-remove candidates; each wants its own look, and the card that surfaced them says so rather than proposing a sweep). ⚠️ TWO of these surfaces are declared TWICE: `api/export.zod.ts` `cronExpression` on `ScheduledExportSchema` and on `ScheduleExportRequestSchema`, and `system/disaster-recovery.zod.ts` `schedule` on `BackupConfigSchema` and on `DisasterRecoveryPlanSchema` (the DR `testing` block). Both pairs are genuinely the same surface twice, so one row is honest here — and now that each declaring position carries its OWN key, that judgement is written out as two `covers` entries instead of being assumed by a collapse. ⚠️ The `failPolicy` on this row is `unevaluated`. It read `compile-error` until the vocabulary gained a member for "nothing evaluates this slot", and that value was the closest available rather than a true one: the PARSE is the only thing that ever refuses one of these values, which is a property every row in this ledger shares and says nothing about this one. It was never a claim that cron SYNTAX is checked. It is not: `@objectstack/formula` cronEngine validates 5/6-field patterns and `@` aliases, and has ZERO consumers outside packages/formula — nothing routes these slots through it. The parse now DOES pin these slots to the cron dialect (the sibling finding on the dialect union is closed): the envelope arm of `CronExpressionInputSchema` accepts a `cron` envelope only and its bare-string arm refuses a blank string, each with one issue at the slot naming the fix — and it still judges no cron syntax, by position: no grammar is restated in spec; `croner` judges the pattern where a schedule is wired (`cron-job-schedule`).',
},

// ── TEMPLATE dialect (#15027) ─────────────────────────────────────────────
{
// The apparent owner was #14797 (closed completed), delivered by PR #14819.
id: 'template-prompt',
summary: 'AI prompt-template system/user prompts (PromptTemplate.system, .user) — `{{var}}` interpolation',
dialect: 'template', mode: 'interpret', state: 'experimental', failPolicy: 'compile-error',
dialect: 'template', mode: 'interpret', state: 'experimental', failPolicy: 'unevaluated',
enforcement:
'PARSE ONLY — `TemplateExpressionInputSchema` refuses a blank/non-string, non-envelope value and normalizes to `{dialect:"template",source}`; NO EVALUATOR FOUND. `PromptTemplateSchema` has no consumer outside packages/spec (measured 2026-09-04), so nothing interpolates the `{{var}}` holes and nothing checks that the declared `variables` match them',
covers: [
Expand All @@ -389,7 +416,7 @@ export const EXPRESSION_SURFACE: ExprSurface[] = [
{
id: 'template-title-format',
summary: 'object record-title template (Object.titleFormat, deprecated → nameField per ADR-0079)',
dialect: 'template', mode: 'interpret', state: 'experimental', failPolicy: 'compile-error',
dialect: 'template', mode: 'interpret', state: 'experimental', failPolicy: 'unevaluated',
enforcement:
'PARSE ONLY in this repo — `TemplateExpressionInputSchema` refuses a blank/non-string, non-envelope value. The KEY has a build-time reader that is NOT an evaluator: lint/validate-record-title.ts `validateRecordTitle` (wired into authoring-rules.ts, run by `os build` / `os lint` / the MCP authoring surface) reports every declaration as `title-format-retired`, an advisory WARNING steering the author to `nameField` — it reads that the key is present and never looks at the template text. The server-side title resolver deliberately does NOT read it: spec/src/data/display-name.ts `objectTitleCompleteness` / `resolveRecordDisplayName` resolve `nameField` then the `displayNameField` alias then a derivation, and ADR-0079 states the reason (render-only; the server can neither return nor query it)',
covers: ['data/object.zod.ts:ObjectSchemaBase.titleFormat'],
Expand All @@ -398,12 +425,12 @@ export const EXPRESSION_SURFACE: ExprSurface[] = [
{
id: 'cel-advanced-policy',
summary: 'advanced security / versioning policy conditions',
dialect: 'cel', mode: 'interpret', state: 'experimental', failPolicy: 'fail-closed',
dialect: 'cel', mode: 'interpret', state: 'experimental', failPolicy: 'unevaluated',
enforcement: '(no runtime consumer yet)',
covers: [
'kernel/plugin-security-advanced.zod.ts:PluginPermissionSchema.condition',
'kernel/plugin-versioning.zod.ts:MultiVersionSupportSchema.condition',
],
note: 'EXPERIMENTAL — declared policy conditions with no runtime evaluator yet (ADR-0056 D8 / ADR-0049 tracking).',
note: 'EXPERIMENTAL — declared policy conditions with no runtime evaluator yet (ADR-0056 D8 / ADR-0049 tracking). This row is why `unevaluated` was minted: it carried `fail-closed` — a RUNTIME refusal — while its own `enforcement` cell said `(no runtime consumer yet)`, so on a security-flavoured row the ledger read as a security guarantee over a slot nothing evaluates. `fail-closed` here was the one borrowing with a consequence past legibility, and spreading it to the other four unwired rows was refused for that reason.',
},
];
84 changes: 83 additions & 1 deletion packages/qa/dogfood/test/expression-conformance.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -24,7 +24,35 @@ const REPO_ROOT = join(HERE, '../../../..');
const SPEC_SRC = join(REPO_ROOT, 'packages/spec/src');

const MODES = new Set(['compile', 'interpret']);
const FAIL_POLICIES = new Set(['compile-error', 'fail-closed', 'fail-soft-log', 'throw']);
const FAIL_POLICIES = new Set(['compile-error', 'fail-closed', 'fail-soft-log', 'throw', 'unevaluated']);

/**
* Runtime evaluator/compiler SITES, as this ledger's own enforcement cells
* name them. Used only as the negative half of the `unevaluated` pin below: a
* row claiming nothing evaluates its slot must not, in the same breath, name
* the thing that does.
*
* The vocabulary is drawn from the enforcement cells already in the ledger
* rather than invented, and it is deliberately a DETECTOR, not an inventory —
* a row may name an evaluator this list has never heard of (the objectui
* renderers are named in prose, with no callable token to match), so a MISS
* proves nothing on its own. That is why the pin's other half is a positive
* requirement rather than this one alone.
*/
const NAMES_RUNTIME_EVALUATOR =
/celEngine|cronEngine|ExpressionEngine\.evaluate|compileCelToFilter|celToFilter|matchesFilterCondition|evaluateVisibility|evaluateValidationRules|evalFieldPredicate|evalRowPredicate|useRowPredicate|resolveCascadingOptions|toBoundaryJobSchedule|croner/;

/**
* The closed set of spellings that STATE the absence `unevaluated` claims.
*
* ⚠️ This does not make the claim true — no regex reads prose for honesty. What
* it does is refuse the shape the four older members were borrowed in: a row
* whose enforcement cell simply describes a site and leaves the reader to infer
* what happens to a bad expression. An `unevaluated` row has to say, in the
* cell itself, that this ledger looked and found no evaluator — which is the
* sentence a reviewer can check and a future author can be held to.
*/
const DECLARES_NO_EVALUATOR = /NO EVALUATOR FOUND|PARSE ONLY|no runtime consumer/;
// `settings-visibility` is not one of the spec's `ExpressionDialect` members on
// purpose (#7327): it is a closed non-CEL grammar with its own evaluator, and
// the ledger's job is to say what a surface IS, not what its schema used to
Expand Down Expand Up @@ -214,4 +242,58 @@ describe('ADR-0058 D7 — expression surface conformance ledger', () => {
+ 'Give the colliding declarations distinguishable keys, then classify each on its own row.');
expect(collisions, collisions.join('\n')).toEqual([]);
});

// The pin that makes `unevaluated` worth minting. The card this member comes
// from is about a vocabulary with no word for "nothing evaluates this slot",
// which forced five rows to borrow a member claiming something stronger —
// `compile-error` on four, and `fail-closed` on a security-flavoured row
// whose own enforcement cell read `(no runtime consumer yet)`. A new word
// that could be borrowed just as loosely would reproduce that defect one
// member wider, so the word arrives with the assertions below.
//
// "Non-empty runtime enforcement" cannot be checked as `enforcement !== ''`:
// `ExprSurface` makes the cell REQUIRED, so every row has a non-empty one,
// the five honest `unevaluated` rows included. The checkable question is what
// the cell SAYS — it must state the absence, and it must not name the runtime
// site whose existence the row is denying.
it('`unevaluated` states an absence, and cannot be borrowed the way the old members were', () => {
// Positive control for the detector itself. An emptied or mistyped
// NAMES_RUNTIME_EVALUATOR makes the negative assertion below vacuously
// green — the "reports green because it never looked" failure the pin above
// guards against with its own control. The COMPILE rows are exactly the
// rows another pin in this file already requires to name the canonical
// compiler, so they are rows this detector MUST fire on.
const compileRows = EXPRESSION_SURFACE.filter((x) => x.mode === 'compile');
expect(
compileRows.length,
'no COMPILE rows in the ledger — the runtime-evaluator detector has nothing to be controlled against, so the assertions below prove nothing',
).toBeGreaterThan(0);
for (const s of compileRows) {
expect(
NAMES_RUNTIME_EVALUATOR.test(s.enforcement),
`${s.id}: the runtime-evaluator detector does not fire on a row that is required to name the canonical compiler — the DETECTOR is broken, not the row`,
).toBe(true);
}

for (const s of EXPRESSION_SURFACE.filter((x) => x.failPolicy === 'unevaluated')) {
// `enforced` means the platform enforces the surface; `unevaluated` means
// nothing reads it. Restricting this pin to `experimental` rows would
// leave `state: 'enforced'` as the escape hatch, so the contradiction is
// refused directly instead.
expect(
s.state,
`${s.id}: state 'enforced' and failPolicy 'unevaluated' contradict each other — nothing evaluates the slot, so nothing enforces it`,
).not.toBe('enforced');

expect(
DECLARES_NO_EVALUATOR.test(s.enforcement),
`${s.id}: failPolicy 'unevaluated' but the enforcement cell never states the absence it claims. Say it in the cell — 'NO EVALUATOR FOUND', 'PARSE ONLY', or 'no runtime consumer' — so the claim is reviewable rather than inferred from silence`,
).toBe(true);

expect(
NAMES_RUNTIME_EVALUATOR.test(s.enforcement),
`${s.id}: failPolicy 'unevaluated' says nothing evaluates this slot, but the enforcement cell names a runtime evaluator/compiler site. One of the two is wrong: if something evaluates it, classify it under the ADR-0058 D5 tier that describes what happens to a bad expression there`,
).toBe(false);
}
});
});
Loading