diff --git a/.claude/skills/toaster-recipe/SKILL.md b/.claude/skills/toaster-recipe/SKILL.md index b7f7e3b..3a1d7ca 100644 --- a/.claude/skills/toaster-recipe/SKILL.md +++ b/.claude/skills/toaster-recipe/SKILL.md @@ -126,6 +126,13 @@ below" — never a sentence built around naming the categories themselves. `user simulated-learner checklist judges this behaviorally (does removing any label still leave the connection legible?), which is exactly the test a lint rule can't run. +**Judgment-record notebooks specifically:** the seam cell narrates one bridged connection — the +tagged SysML text (the subject plus its `ReviewRecordRef` usage), the tool that loads it and +cross-checks both the Python record and the model tag, and a result showing they agree +(`validate_record` plus `get_review_record_refs`) — not a choice between the model's own +construct/tool/result triad and the record's own fields/`validate_record`/result triad +(`decisions/log.md` DL-075, retired by this convention). + ## Chapter index.md — 6-element recipe 1. Purpose — engineering question and model state after completing the chapter @@ -160,7 +167,10 @@ someone happens to reread it. A notebook that builds a `ReviewRecord` (an `asserted_context`, `asserted_inference` or `asserted_solution` judgment) uses `toaster-review-protocol`'s own construction-zone pattern for it, not one dense call: name each group of fields, narrate what it's for, print it, then assemble. -The size limits below are relaxed for this content (see that skill for the exact grouping and why). +The first group now names the record's own subject (`subject_ref`) alongside its `claim`, and +builds the matching `ReviewRecordRef` metadata-tag fragment the same way a model-increment cell +builds any other named fragment (see toaster-review-protocol's own subject_ref section). The size limits below +are relaxed for this content (see that skill for the exact grouping and why). ## Size limits (A6 review criteria) diff --git a/.claude/skills/toaster-review-protocol/SKILL.md b/.claude/skills/toaster-review-protocol/SKILL.md index 9bf5a47..e44f642 100644 --- a/.claude/skills/toaster-review-protocol/SKILL.md +++ b/.claude/skills/toaster-review-protocol/SKILL.md @@ -13,11 +13,45 @@ description: Hawkins et al. 2011 judgment record fields, three ACP types, two ev ## Three judgment sites and ACP kinds -| Site | Kind | Hawkins ref | -|---|---|---| -| Assumption or context used for a claim | `asserted_context` | §3.2 | -| Child claims supporting a parent | `asserted_inference` | §3.1 | -| Evidence supporting a conclusion | `asserted_solution` | §3.3 | +| Site | Kind | Hawkins ref | `subject_ref` | +|---|---|---|---| +| Assumption or context used for a claim | `asserted_context` | §3.2 | required | +| Child claims supporting a parent | `asserted_inference` | §3.1 | required unless `premises` is non-empty | +| Evidence supporting a conclusion | `asserted_solution` | §3.3 | required | + +## `subject_ref`: this tutorial's narrowed Assurance Claim Point + +Hawkins' own Assurance Claim Point (ACP) is never free-floating: every confidence argument is +anchored to one specific, located assertion in the argument (Hawkins 2011, Sec. 3, p. 8 — +`glid:def-hawkins--assurance-claim-point`). `subject_ref` is this tutorial's own narrowed, +single-element analog: the one qualified name the record's `claim` is directly about, checkable +both from Python (`validate_record(record, model=model)` resolves it via `model.find()`) and from +the model's own side, via a real SysML metadata tag (SysML v2 formal/2026-03-02 §7.27.2, +MetadataDefinition): + +```sysml +metadata def ReviewRecordRef { + attribute identifier : ScalarValues::String; +} + +metadata ac001Tag : ReviewRecordRef about nominal { + identifier = "AC-001"; +} +``` + +(the committed text in `models/ch02-cumulative.sysml`, built in +`chapters/ch02-requirements/03-judgment-context.ipynb`.) + +The `about` clause binds the usage's inherited `annotatedElement` feature to the named subject — a +real, queryable model relationship, not a string a reader has to trust. `subject_ref` is required +for `asserted_context` and `asserted_solution`; for `asserted_inference` it may stay empty only +when `premises` is non-empty (the pure cross-record synthesis case — `AI-C10` is the one original +record that actually relies on this exemption rather than carrying a real anchor anyway; negative +controls such as `AI-C10-DRAFT` rely on it too). `src/toaster/query.py`'s `get_review_record_refs()` is the +model-to-Python direction: given a loaded model, it finds every `ReviewRecordRef` tag and what it's +about, independent of any notebook's own Python objects. `validate_record` cross-checks both +directions automatically whenever a `model` is passed and a tag already exists for that record's +own `identifier`. ## ReviewRecord required fields @@ -28,6 +62,7 @@ record = ReviewRecord( identifier="RR-001", kind="asserted_solution", claim="DeliveredEnergy >= 50000 J at nominal operating conditions", + subject_ref="ToasterDemo::HeatGenerator::deliveredEnergy", model_ref="models/ch07-snapshot.sysml", content_hash=hash_content(open("models/ch07-snapshot.sysml").read()), scope="nominal operating envelope: P=800W, t=120s, eta=0.7", @@ -64,7 +99,18 @@ group of fields, narrate what it's for, print it, then assemble. Group by the qu Hawkins' taxonomy is answering, not by the dataclass's field order: ``` -[markdown] narration: what is being claimed, and about what +[markdown] narration: what is being claimed, and what, specifically, it is about +[code] subject_ref = "ToasterDemo::..." + SOME_TAG = """\ + metadata someTag : ReviewRecordRef about { + identifier = "..." + } + """ + print(SOME_TAG) +[markdown] narration: this fragment is the same text now committed in the cumulative model +[code] TOASTER_INCREMENT = SOME_TAG # or assembled with any other new fragment this notebook adds + print(TOASTER_INCREMENT) +[markdown] narration: the claim itself comes next [code] claim = "..." model_ref = "..." [markdown] narration: what standard the claim is checked against (appropriateness) @@ -82,21 +128,28 @@ Hawkins' taxonomy is answering, not by the dataclass's field order: [code] counterevidence = "..." residual_uncertainties = "..." [markdown] narration: assembling the record from the named parts above -[code] record = ReviewRecord(identifier=..., kind=..., claim=claim, model_ref=model_ref, - content_hash=hash_content(source), scope=scope, criteria=criteria, +[code] record = ReviewRecord(identifier=..., kind=..., claim=claim, subject_ref=subject_ref, + model_ref=model_ref, content_hash=hash_content(source), scope=scope, criteria=criteria, premises=premises, assumption_refs=assumption_refs, evidence_refs=evidence_refs, rationale=rationale, counterevidence=counterevidence, residual_uncertainties=residual_uncertainties, disposition="pending", dependency_freshness="current", engineering_conclusion=..., record_kind="worked_example") - errors = validate_record(record) + errors = validate_record(record, model=model) + tag = next((t for t in get_review_record_refs(model) if t["identifier"] == record.identifier), None) + print(f"Model tag: {tag}") print(f"Validation errors: {errors}") ``` -Five groups, five narration cells, matching the model-fragment construction zone's pacing rule (no -two code cells adjacent). Each `print`ed group is the record's own reflection, the same role a -printed `TOASTER_INCREMENT` plays for a model fragment. +Seven groups when the notebook is introducing a new tag (the two anchor groups above, then the five +Hawkins-taxonomy groups); five when it is a Python-only reconstruction that cites an already-tagged +identifier from an earlier chapter (no new SysML, so no anchor groups, but `model=model` and the +`Model tag` lookup still run, exercising the cross-representation check against the already-committed +tag). Every code cell is still followed by a markdown cell narrating what's next (no two code cells +adjacent). Each printed group is its own reflection, the same role a printed `TOASTER_INCREMENT` +plays for a model fragment — and `TOASTER_INCREMENT` here really is the Hawkins record's own model- +side anchor, assembled and loaded the same way any other chapter's model increment is. **Size limit:** `toaster-recipe`'s ≤600 words / ≤50 lines budget is sized for a notebook whose main content is one model construct. A notebook whose construct is a judgment record may exceed it — the diff --git a/.claude/skills/user-testing/SKILL.md b/.claude/skills/user-testing/SKILL.md index e519f24..6331066 100644 --- a/.claude/skills/user-testing/SKILL.md +++ b/.claude/skills/user-testing/SKILL.md @@ -27,36 +27,65 @@ One agent per persona. No two agents with identical persona in one checkpoint ru ## Execution checklist (`simulated-learner` must run these in order) +Identify each step below by what the cell *does*, not by a fixed position: a construction-zone +notebook may have several fragment cells before assembly, a judgment-record notebook can run +17-27 real cells, and some chapters carry the seam across several cells' prose rather than one +dedicated cell (Chapter 10's own distributed-seam design is a real, valid instance of this, not a +gap) — the same "by content type, not cell index" rule `toaster-recipe` already states for its own +review. If you cannot find a cell matching a step below, say so explicitly rather than guessing +which numbered cell it must be. + For each sub-notebook in the assigned chapter(s): 1. **Read index.md** — does it orient you? Note any undefined terms or missing prerequisites. -2. **Cell 0** — read the concept statement. Is it exactly one sentence? Does it state what you will learn? -3. **Cell 1** — read the context paragraph. Does it locate this notebook in the arc? Is there a link to the prior notebook where needed? -4. **Cell 2 (execute)** — run the model-loading code. Record: `model.ok`, any diagnostic output. -5. **Cell 3 (execute)** — run the negative control. Record: `bad.ok` (must be False), printed diagnostic message. -6. **Cell 4 (execute)** — run the demonstration. Record: output produced; note if it matches what cell 0 promised. -7. **Cell 5** — read the Tall seam. AGENTS.md 1.10 binds that learner content **never names** Tall or "the three worlds" (the `tall-named` lint rule, `glossary/lint_rules.toml`, DL-028, already enforces the never-name half in CI). Your job is the half a lint rule cannot judge: does the cell **address the seam in behavior** — is it clear, without naming the lens, that the SysML text, the tool that loads and runs it, and the rendered/printed result are three distinct things the reader has just seen connect? Record which of the three you could each point to concretely from what the cell actually showed, and whether a reader who had not been told there were "three worlds" would still notice the seam. -8. **Cell 6** — read the exercise pointer. Is it one sentence? Does it describe what the exercise asks? -9. **Read conclusion.md** — three paragraphs (what was built / what this establishes / what comes next) plus exercise reference? +2. **The concept-statement cell** — read it. Is it exactly one sentence? (A single sentence may + contain a semicolon joining two independent clauses and still be one sentence — count terminal + periods, not semicolons or conjunctions, before judging this a failure.) Does it state what you + will learn? +3. **The context cell** — read it. Does it locate this notebook in the arc? Is there a link to the + prior notebook where needed? +4. **The model-increment cell(s) (execute)** — run the model-loading code. Record: `model.ok`, any + diagnostic output. +5. **The negative-control cell (execute)** — run it. Record: `bad.ok` (must be False), printed + diagnostic message. +6. **The demonstration cell(s) (execute)** — run them. Record: output produced; note if it matches + what the concept-statement cell promised. +7. **The seam cell(s)** — read them. AGENTS.md 1.10 binds that learner content **never names** Tall + or "the three worlds" (the `tall-named` lint rule, `glossary/lint_rules.toml`, DL-028, already + enforces the never-name half in CI). Your job is the half a lint rule cannot judge: does the + content **address the seam in behavior** — is it clear, without naming the lens, that the SysML + text, the tool that loads and runs it, and the rendered/printed result are three distinct things + the reader has just seen connect? Record which of the three you could each point to concretely + from what the notebook actually showed, and whether a reader who had not been told there were + "three worlds" would still notice the seam. +8. **The exercise-pointer cell** — read it. Is it one sentence? Does it describe what the exercise + asks? +9. **Read conclusion.md** — three paragraphs (what was built / what this establishes / what comes + next) plus exercise reference? Execution command: ```sh -cd /Users/z/Documents/GitHub/toaster +cd uv run python - <<'EOF' [paste cell code here] EOF ``` +A worktree-isolated cell that runs from the main checkout's path instead of its own worktree reads +some files (e.g. notebook text) from the wrong branch state while reading others (e.g. model files) +from its own worktree — a mixed-path read that produces false findings. Confirmed as the cause of a +false NEEDS-FIX verdict in the first grid run (`decisions/log.md` DL-085). + ## Report format ``` LEARNER [ID] — [Persona] — Ch[N] EXECUTION RESULTS: -- nb[N] cell2: ok=[True/False] | [diagnostic if any] -- nb[N] cell3: neg_ok=[True/False] | diagnostic: [message] -- nb[N] cell4: output=[one-line summary] +- nb[N] model-increment: ok=[True/False] | [diagnostic if any] +- nb[N] negative-control: neg_ok=[True/False] | diagnostic: [message] +- nb[N] demonstration: output=[one-line summary] [repeat for each notebook] NARRATIVE OBSERVATIONS (top 3, each quoting exact text): @@ -65,9 +94,9 @@ NARRATIVE OBSERVATIONS (top 3, each quoting exact text): 3. "[exact quote]" — [learner reaction in one sentence] STRUCTURAL CHECKS: -- Cell 0 one sentence: [yes/no] -- Cell 5 addresses the seam without naming it: [yes/no] — [which of the three you could point to; if no, what's missing] -- Cell 6 one sentence: [yes/no] +- Concept-statement cell is one sentence: [yes/no] +- Seam addressed in behavior without naming it: [yes/no] — [which of the three you could point to; if no, what's missing] +- Exercise-pointer cell is one sentence: [yes/no] - conclusion.md three paragraphs + exercise reference: [yes/no] OVERALL: [PASS/NEEDS-FIX] — one sentence. diff --git a/.gitignore b/.gitignore index 4d469c4..1f36be6 100644 --- a/.gitignore +++ b/.gitignore @@ -45,3 +45,6 @@ glossary/sources/local/* # reproducible across machines: see chapters/ch08-checking/02-violation-witness.ipynb) chapters/*/companion-check-scratch/ exercises/*/companion-check-scratch/ + +# Agent-managed git worktrees (builder/reviewer isolation during plan execution) +.claude/worktrees/ diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..261eeb9 --- /dev/null +++ b/LICENSE @@ -0,0 +1,201 @@ + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + APPENDIX: How to apply the Apache License to your work. + + To apply the Apache License to your work, attach the following + boilerplate notice, with the fields enclosed by brackets "[]" + replaced with your own identifying information. (Don't include + the brackets!) The text should be enclosed in the appropriate + comment syntax for the file format. We also recommend that a + file or class name and description of purpose be included on the + same "printed page" as the copyright notice for easier + identification within third-party archives. + + Copyright [yyyy] [name of copyright owner] + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. diff --git a/README.md b/README.md index fbd4357..6ef96dc 100644 --- a/README.md +++ b/README.md @@ -6,7 +6,8 @@ measures, functions, structure, and executable behavior for a domestic toaster. No site is published yet; deployment stays off until the tutorial has complete, end-to-end content ready to publish. See [docs/setup.md](docs/setup.md) to run the tutorial or build the book locally. -Adapted from Brian Douglas's [Systems Engineering Part 3](https://www.mathworks.com/videos/systems-engineering-part-3-the-benefits-of-functional-architectures-1602837771665.html). +Adapted from Brian Douglas's [Systems Engineering Part 3](https://www.mathworks.com/videos/systems-engineering-part-3-the-benefits-of-functional-architectures-1602837771665.html) +and [Part 4](https://www.mathworks.com/videos/systems-engineering-part-4-an-introduction-to-requirements-1603872564696.html). Engineering judgment records follow Hawkins et al. 2011 §§3.1–3.4. ## Quick start @@ -21,12 +22,13 @@ uv run python scripts/check-tools.py # Run tests uv run pytest tests/ -v -# Build the site +# Preview the book locally (starts a dev server at localhost:3000) npm install -npx mystmd build --execute +npx mystmd start --execute ``` -See [docs/setup.md](docs/setup.md) for full setup instructions and the fork-and-exercise workflow. +See [docs/setup.md](docs/setup.md) for full setup instructions and the fork-and-exercise workflow, +including what `uv` and `mystmd` are and why the quick start above uses them. ## Repository structure @@ -34,9 +36,14 @@ See [docs/setup.md](docs/setup.md) for full setup instructions and the fork-and- chapters/ — worked example notebooks (10 chapters, read-only for exercises) exercises/ — parallel exercise notebooks (fork and work here) models/ — SysML stage model snapshots -src/toaster/— Python package (bootstrap, connect, query, check, evidence, simulate, render, report) +src/toaster/— Python package (bootstrap, connect, query, check, conformance, evidence, modelcheck, simulate, render, report) tests/ — pytest suite docs/ — setup, glossary, references, reproducibility statement scripts/ — pre-flight and build utilities decisions/ — ACE decision log ``` + +`AGENTS.md`, `CLAUDE.md`, and `DEFERRED.md` at the repo root are not learner material — they're +this project's own working contract, for the AI agents and maintainers who build and review the +tutorial's content. See [docs/contributor.md](docs/contributor.md) if you want to understand how +the tutorial is actually built, tested, and reviewed, or to contribute to it yourself. diff --git a/chapters/ch02-requirements/03-judgment-context.ipynb b/chapters/ch02-requirements/03-judgment-context.ipynb index 8fa2f11..55312cb 100644 --- a/chapters/ch02-requirements/03-judgment-context.ipynb +++ b/chapters/ch02-requirements/03-judgment-context.ipynb @@ -1,17 +1,4 @@ { - "nbformat": 4, - "nbformat_minor": 5, - "metadata": { - "kernelspec": { - "display_name": "Python 3", - "language": "python", - "name": "python3" - }, - "language_info": { - "name": "python" - }, - "title": "Ch2 nb3: asserted context" - }, "cells": [ { "cell_type": "markdown", @@ -33,10 +20,80 @@ }, { "cell_type": "code", + "execution_count": 1, "id": "cell-02", - "metadata": {}, - "outputs": [], - "execution_count": null, + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:10:18.537065Z", + "iopub.status.busy": "2026-10-01T05:10:18.536828Z", + "iopub.status.idle": "2026-10-01T05:10:18.674939Z", + "shell.execute_reply": "2026-10-01T05:10:18.674554Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "// GENERATED FIXTURE — do not edit directly.\n", + "// Run: python scripts/check_construction.py --check (to verify)\n", + "// Source: notebook cell-02 TOASTER_INCREMENT in chapter 2's construct-introducing notebooks.\n", + "\n", + "package ToasterDemo {\n", + " metadata def ReviewRecordRef {\n", + " attribute identifier : ScalarValues::String;\n", + " }\n", + "\n", + " private import ScalarValues::*;\n", + " private import SI::*;\n", + " private import ISQ::*;\n", + "\n", + " item def Bread;\n", + " item def Toast;\n", + "\n", + " action def ToastBread {\n", + " doc /* Transform bread into toast acceptable to its user. */\n", + " in bread : Bread;\n", + " out toast : Toast;\n", + " }\n", + "\n", + " abstract part def ToastingSystem {\n", + " perform action toastBread : ToastBread;\n", + " }\n", + "\n", + " part def HeatingSystem;\n", + " part def ControlSystem;\n", + "\n", + " part def Toaster :> ToastingSystem {\n", + " attribute cycleTime : ISQ::DurationValue;\n", + " part heating : HeatingSystem;\n", + " part control : ControlSystem;\n", + " }\n", + "\n", + " requirement def TimelyToast {\n", + " doc /*\n", + " * The toaster shall complete a toasting cycle in at most 180 seconds.\n", + " * Rationale: kitchen workflows typically span 5-15 minutes; a cycle\n", + " * exceeding 3 minutes delays meal preparation and falls outside where\n", + " * and how a user prepares a meal.\n", + " */\n", + " subject toaster : Toaster;\n", + " require constraint { toaster.cycleTime <= 180.0 [SI::s] }\n", + " }\n", + "\n", + " part nominal : Toaster;\n", + " metadata ac001Tag : ReviewRecordRef about nominal {\n", + " identifier = \"AC-001\";\n", + " }\n", + "\n", + " part slow : Toaster {\n", + " attribute :>> cycleTime = 200.0 [SI::s];\n", + " }\n", + "}\n", + "\n" + ] + } + ], "source": [ "from pathlib import Path\n", "import opensysml\n", @@ -53,14 +110,31 @@ "cell_type": "markdown", "id": "cell-03", "metadata": {}, - "source": "The `ch02-cumulative.sysml` file adds the first requirements construct: `requirement def TimelyToast` constrains `cycleTime <= 180.0 [SI::s]` with a typed `subject` and `require constraint` body. `nominal` (`cycleTime` unset) and `slow` (`cycleTime` fixed at 200 s, a deliberately injected fault) are declared for later comparison. The `assert satisfy` pattern comes in Chapter 3; for now these usages exist without a recorded claim." + "source": [ + "The `ch02-cumulative.sysml` file adds the first requirements construct: `requirement def TimelyToast` constrains `cycleTime <= 180.0 [SI::s]` with a typed `subject` and `require constraint` body. `nominal` (`cycleTime` unset) and `slow` (`cycleTime` fixed at 200 s, a deliberately injected fault) are declared for later comparison. The `assert satisfy` pattern comes in Chapter 3; for now these usages exist without a recorded claim." + ] }, { "cell_type": "code", + "execution_count": 2, "id": "cell-04", - "metadata": {}, - "outputs": [], - "execution_count": null, + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:10:18.676423Z", + "iopub.status.busy": "2026-10-01T05:10:18.676258Z", + "iopub.status.idle": "2026-10-01T05:10:18.679935Z", + "shell.execute_reply": "2026-10-01T05:10:18.679464Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Expected error: unresolved member: nonExistentAttr\n" + ] + } + ], "source": [ "# Negative control: a constraint that references an attribute not in scope\n", "# confirms that the model enforces referential integrity in constraints.\n", @@ -84,98 +158,401 @@ "id": "cell-05", "metadata": {}, "source": [ - "With the model side confirmed, the next cell turns to the judgment side: recording the assumption behind the cycle-time estimate as a Hawkins-style `asserted_context` record." + "With the model side confirmed, the next cells introduce a new construct: a `ReviewRecordRef` metadata tag that gives a Hawkins judgment record a real, checkable anchor in the model, before building the `asserted_context` record itself." + ] + }, + { + "cell_type": "code", + "execution_count": 3, + "id": "3127193b", + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:10:18.681424Z", + "iopub.status.busy": "2026-10-01T05:10:18.681342Z", + "iopub.status.idle": "2026-10-01T05:10:18.683330Z", + "shell.execute_reply": "2026-10-01T05:10:18.682921Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "metadata def ReviewRecordRef {\n", + " attribute identifier : ScalarValues::String;\n", + "}\n", + "\n" + ] + } + ], + "source": [ + "# the metadata construct itself: one attribute naming which Hawkins judgment record a tag anchors\n", + "# spec: SysML v2 formal/2026-03-02 §7.27.2 (MetadataDefinition)\n", + "REVIEW_RECORD_REF_DEF = \"\"\"\\\n", + "metadata def ReviewRecordRef {\n", + " attribute identifier : ScalarValues::String;\n", + "}\n", + "\"\"\"\n", + "print(REVIEW_RECORD_REF_DEF)" + ] + }, + { + "cell_type": "markdown", + "id": "5af9bb68", + "metadata": {}, + "source": [ + "`ReviewRecordRef` is a new metadata definition: a tag with one attribute, `identifier`, that can be attached `about` any model element (SysML v2 formal/2026-03-02 §7.27.2). It gives a Hawkins judgment record a real, queryable anchor in the model, not just a string the reader has to trust." + ] + }, + { + "cell_type": "code", + "execution_count": 4, + "id": "19b5f98e", + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:10:18.684411Z", + "iopub.status.busy": "2026-10-01T05:10:18.684336Z", + "iopub.status.idle": "2026-10-01T05:10:18.686015Z", + "shell.execute_reply": "2026-10-01T05:10:18.685631Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "metadata ac001Tag : ReviewRecordRef about nominal {\n", + " identifier = \"AC-001\";\n", + "}\n", + "\n" + ] + } + ], + "source": [ + "# what is being claimed, and what, specifically, it is about\n", + "subject_ref = \"ToasterDemo::nominal\"\n", + "AC001_TAG = \"\"\"\\\n", + "metadata ac001Tag : ReviewRecordRef about nominal {\n", + " identifier = \"AC-001\";\n", + "}\n", + "\"\"\"\n", + "print(AC001_TAG)" + ] + }, + { + "cell_type": "markdown", + "id": "35677098", + "metadata": {}, + "source": [ + "This fragment, together with the `ReviewRecordRef` definition above, is the same text already committed in `models/ch02-cumulative.sysml`; loading the cumulative model (the `model.ok` cell above, unchanged) is what makes this a real, checkable tag rather than an assertion the reader has to trust." + ] + }, + { + "cell_type": "code", + "execution_count": 5, + "id": "1e4c4704", + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:10:18.687028Z", + "iopub.status.busy": "2026-10-01T05:10:18.686952Z", + "iopub.status.idle": "2026-10-01T05:10:18.689063Z", + "shell.execute_reply": "2026-10-01T05:10:18.688429Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "metadata def ReviewRecordRef {\n", + " attribute identifier : ScalarValues::String;\n", + "}\n", + "\n", + "metadata ac001Tag : ReviewRecordRef about nominal {\n", + " identifier = \"AC-001\";\n", + "}\n", + "\n" + ] + } + ], + "source": [ + "TOASTER_INCREMENT = f\"{REVIEW_RECORD_REF_DEF}\\n{AC001_TAG}\"\n", + "print(TOASTER_INCREMENT)" ] }, { "cell_type": "markdown", "id": "cell-06", "metadata": {}, - "source": "`AC-001` states the claim first: the placeholder cycle-time estimate this chapter uses, and the model element it's an assumption about." + "source": [ + "`AC-001` states the claim first: the placeholder cycle-time estimate this chapter uses." + ] }, { "cell_type": "code", + "execution_count": 6, "id": "cell-07", - "metadata": {}, - "outputs": [], - "execution_count": null, - "source": "from toaster.evidence import ReviewRecord, hash_content, validate_record\n\nclaim = (\"Approximately 120 seconds is assumed, for illustration only, as a plausible \"\n \"nominal cycle time for standard sliced bread, a placeholder pending the \"\n \"mechanism-and-energy-balance derivation a later chapter performs, not a value \"\n \"drawn from any real-world source.\")\nmodel_ref = (\"ToasterDemo::nominal\")\nprint(claim)" + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:10:18.690506Z", + "iopub.status.busy": "2026-10-01T05:10:18.690394Z", + "iopub.status.idle": "2026-10-01T05:10:18.692591Z", + "shell.execute_reply": "2026-10-01T05:10:18.692214Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Approximately 120 seconds is assumed, for illustration only, as a plausible nominal cycle time for standard sliced bread, a placeholder pending the mechanism-and-energy-balance derivation a later chapter performs, not a value drawn from any real-world source.\n" + ] + } + ], + "source": [ + "from toaster.evidence import ReviewRecord, hash_content, validate_record\n", + "\n", + "claim = (\"Approximately 120 seconds is assumed, for illustration only, as a plausible \"\n", + " \"nominal cycle time for standard sliced bread, a placeholder pending the \"\n", + " \"mechanism-and-energy-balance derivation a later chapter performs, not a value \"\n", + " \"drawn from any real-world source.\")\n", + "model_ref = (\"ToasterDemo::nominal\")\n", + "print(claim)" + ] }, { "cell_type": "markdown", "id": "cell-08", "metadata": {}, - "source": "`scope` names where in the model the assumption applies; `criteria` states the illustrative range the estimate must fall inside, so a reader can judge whether the claim is appropriate for what it's used for." + "source": [ + "`scope` names where in the model the assumption applies; `criteria` states the illustrative range the estimate must fall inside, so a reader can judge whether the claim is appropriate for what it's used for." + ] }, { "cell_type": "code", + "execution_count": 7, "id": "cell-09", - "metadata": {}, - "outputs": [], - "execution_count": null, - "source": "scope = (\"ToasterDemo\")\ncriteria = (\"This chapter adopts an illustrative placeholder range of 90-150 seconds for a \"\n \"toaster's nominal cycle time toasting standard sliced bread. The range is \"\n \"invented for this tutorial and is not drawn from any manufacturer data, \"\n \"measurement, or cited source.\")\nprint(criteria)" + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:10:18.693751Z", + "iopub.status.busy": "2026-10-01T05:10:18.693662Z", + "iopub.status.idle": "2026-10-01T05:10:18.695592Z", + "shell.execute_reply": "2026-10-01T05:10:18.695282Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "This chapter adopts an illustrative placeholder range of 90-150 seconds for a toaster's nominal cycle time toasting standard sliced bread. The range is invented for this tutorial and is not drawn from any manufacturer data, measurement, or cited source.\n" + ] + } + ], + "source": [ + "scope = (\"ToasterDemo\")\n", + "criteria = (\"This chapter adopts an illustrative placeholder range of 90-150 seconds for a \"\n", + " \"toaster's nominal cycle time toasting standard sliced bread. The range is \"\n", + " \"invented for this tutorial and is not drawn from any manufacturer data, \"\n", + " \"measurement, or cited source.\")\n", + "print(criteria)" + ] }, { "cell_type": "markdown", "id": "cell-10", "metadata": {}, - "source": "`premises` is empty (the estimate isn't derived from anything else in this chapter); `assumption_refs` names the one assumption it rests on directly." + "source": [ + "`premises` is empty (the estimate isn't derived from anything else in this chapter); `assumption_refs` names the one assumption it rests on directly." + ] }, { "cell_type": "code", + "execution_count": 8, "id": "cell-11", - "metadata": {}, - "outputs": [], - "execution_count": null, - "source": "premises = ([])\nassumption_refs = ([\n \"A-CH02-1: illustrative placeholder cycle-time range (90-150s) for standard \"\n \"sliced bread, invented for this tutorial; no real-world source exists for it.\",\n ])\nprint(assumption_refs)" + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:10:18.697086Z", + "iopub.status.busy": "2026-10-01T05:10:18.696994Z", + "iopub.status.idle": "2026-10-01T05:10:18.698954Z", + "shell.execute_reply": "2026-10-01T05:10:18.698630Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "['A-CH02-1: illustrative placeholder cycle-time range (90-150s) for standard sliced bread, invented for this tutorial; no real-world source exists for it.']\n" + ] + } + ], + "source": [ + "premises = ([])\n", + "assumption_refs = ([\n", + " \"A-CH02-1: illustrative placeholder cycle-time range (90-150s) for standard \"\n", + " \"sliced bread, invented for this tutorial; no real-world source exists for it.\",\n", + " ])\n", + "print(assumption_refs)" + ] }, { "cell_type": "markdown", "id": "cell-12", "metadata": {}, - "source": "`evidence_refs` states plainly that there is no real evidence behind this number, only the stated assumption; `rationale` explains why 120 seconds was picked within that range anyway." + "source": [ + "`evidence_refs` states plainly that there is no real evidence behind this number, only the stated assumption; `rationale` explains why 120 seconds was picked within that range anyway." + ] }, { "cell_type": "code", + "execution_count": 9, "id": "cell-13", - "metadata": {}, - "outputs": [], - "execution_count": null, - "source": "evidence_refs = ([\n \"None: the value is a stated placeholder assumption (A-CH02-1), not evidence \"\n \"from a real source, and not the model's own declared value. \"\n \"Toaster::cycleTime carries no value in this chapter.\",\n ])\nrationale = (\"120 seconds sits within the illustrative placeholder range (A-CH02-1) and is \"\n \"used only as a stand-in estimate; it is not derived from any mechanism or \"\n \"energy-balance analysis in this chapter, and it is not backed by any \"\n \"real-world data.\")\nprint(rationale)" + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:10:18.699997Z", + "iopub.status.busy": "2026-10-01T05:10:18.699932Z", + "iopub.status.idle": "2026-10-01T05:10:18.701539Z", + "shell.execute_reply": "2026-10-01T05:10:18.701226Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "120 seconds sits within the illustrative placeholder range (A-CH02-1) and is used only as a stand-in estimate; it is not derived from any mechanism or energy-balance analysis in this chapter, and it is not backed by any real-world data.\n" + ] + } + ], + "source": [ + "evidence_refs = ([\n", + " \"None: the value is a stated placeholder assumption (A-CH02-1), not evidence \"\n", + " \"from a real source, and not the model's own declared value. \"\n", + " \"Toaster::cycleTime carries no value in this chapter.\",\n", + " ])\n", + "rationale = (\"120 seconds sits within the illustrative placeholder range (A-CH02-1) and is \"\n", + " \"used only as a stand-in estimate; it is not derived from any mechanism or \"\n", + " \"energy-balance analysis in this chapter, and it is not backed by any \"\n", + " \"real-world data.\")\n", + "print(rationale)" + ] }, { "cell_type": "markdown", "id": "cell-14", "metadata": {}, - "source": "The challenge: `counterevidence` names a real case the placeholder doesn't cover, and `residual_uncertainties` says what a reader must not conclude from this record (Hawkins' trustworthiness: naming the record's own limits, not hiding them)." + "source": [ + "The challenge: `counterevidence` names a real case the placeholder doesn't cover, and `residual_uncertainties` says what a reader must not conclude from this record (Hawkins' trustworthiness: naming the record's own limits, not hiding them)." + ] }, { "cell_type": "code", + "execution_count": 10, "id": "cell-15", - "metadata": {}, - "outputs": [], - "execution_count": null, - "source": "counterevidence = (\"Thick-cut and frozen bread may require 180-240s, which would exceed \"\n \"TimelyToast's 180-second bound. More fundamentally, the placeholder range \"\n \"itself is invented for this tutorial and has no real-world source, so it \"\n \"carries no evidentiary weight beyond illustrating the pattern.\")\nresidual_uncertainties = (\"User preference variation is not modeled. Because 120 seconds is an \"\n \"invented placeholder, not a derived or measured value, any comparison \"\n \"against the 180-second threshold is conditional on this assumption and must \"\n \"never be reported as a settled pass/fail verdict.\")\nprint(counterevidence)\nprint(residual_uncertainties)" + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:10:18.702571Z", + "iopub.status.busy": "2026-10-01T05:10:18.702494Z", + "iopub.status.idle": "2026-10-01T05:10:18.704473Z", + "shell.execute_reply": "2026-10-01T05:10:18.704081Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Thick-cut and frozen bread may require 180-240s, which would exceed TimelyToast's 180-second bound. More fundamentally, the placeholder range itself is invented for this tutorial and has no real-world source, so it carries no evidentiary weight beyond illustrating the pattern.\n", + "User preference variation is not modeled. Because 120 seconds is an invented placeholder, not a derived or measured value, any comparison against the 180-second threshold is conditional on this assumption and must never be reported as a settled pass/fail verdict.\n" + ] + } + ], + "source": [ + "counterevidence = (\"Thick-cut and frozen bread may require 180-240s, which would exceed \"\n", + " \"TimelyToast's 180-second bound. More fundamentally, the placeholder range \"\n", + " \"itself is invented for this tutorial and has no real-world source, so it \"\n", + " \"carries no evidentiary weight beyond illustrating the pattern.\")\n", + "residual_uncertainties = (\"User preference variation is not modeled. Because 120 seconds is an \"\n", + " \"invented placeholder, not a derived or measured value, any comparison \"\n", + " \"against the 180-second threshold is conditional on this assumption and must \"\n", + " \"never be reported as a settled pass/fail verdict.\")\n", + "print(counterevidence)\n", + "print(residual_uncertainties)" + ] }, { "cell_type": "markdown", "id": "cell-16", "metadata": {}, - "source": "With every part named above, the context record assembles from them directly." + "source": [ + "With every part named above, the context record assembles from them directly." + ] }, { "cell_type": "code", + "execution_count": 11, "id": "cell-17", - "metadata": {}, - "outputs": [], - "execution_count": null, - "source": "context_record = ReviewRecord(\n identifier=\"AC-001\",\n kind=\"asserted_context\",\n claim=claim,\n model_ref=model_ref,\n content_hash=hash_content(source),\n scope=scope,\n criteria=criteria,\n premises=premises,\n assumption_refs=assumption_refs,\n evidence_refs=evidence_refs,\n rationale=rationale,\n counterevidence=counterevidence,\n residual_uncertainties=residual_uncertainties,\n disposition=\"pending\",\n dependency_freshness=\"current\",\n engineering_conclusion=\"undetermined\",\n record_kind=\"worked_example\",\n)\n\nerrors = validate_record(context_record)\nprint(f\"Record: {context_record.identifier} | kind: {context_record.kind}\")\nprint(f\"Claim: {context_record.claim}\")\nprint(f\"Validation errors: {errors}\")\nconn.close()" + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:10:18.705728Z", + "iopub.status.busy": "2026-10-01T05:10:18.705638Z", + "iopub.status.idle": "2026-10-01T05:10:18.773254Z", + "shell.execute_reply": "2026-10-01T05:10:18.772746Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Record: AC-001 | kind: asserted_context\n", + "Claim: Approximately 120 seconds is assumed, for illustration only, as a plausible nominal cycle time for standard sliced bread, a placeholder pending the mechanism-and-energy-balance derivation a later chapter performs, not a value drawn from any real-world source.\n", + "Model tag: {'tag': 'ToasterDemo::ac001Tag', 'identifier': 'AC-001', 'annotated_element': 'ToasterDemo::nominal'}\n", + "Validation errors: []\n" + ] + } + ], + "source": [ + "from toaster.query import get_review_record_refs\n", + "\n", + "context_record = ReviewRecord(\n", + " identifier=\"AC-001\",\n", + " kind=\"asserted_context\",\n", + " claim=claim,\n", + " subject_ref=subject_ref,\n", + " model_ref=model_ref,\n", + " content_hash=hash_content(source),\n", + " scope=scope,\n", + " criteria=criteria,\n", + " premises=premises,\n", + " assumption_refs=assumption_refs,\n", + " evidence_refs=evidence_refs,\n", + " rationale=rationale,\n", + " counterevidence=counterevidence,\n", + " residual_uncertainties=residual_uncertainties,\n", + " disposition=\"pending\",\n", + " dependency_freshness=\"current\",\n", + " engineering_conclusion=\"undetermined\",\n", + " record_kind=\"worked_example\",\n", + ")\n", + "\n", + "errors = validate_record(context_record, model=model)\n", + "tag = next((t for t in get_review_record_refs(model) if t[\"identifier\"] == context_record.identifier), None)\n", + "print(f\"Record: {context_record.identifier} | kind: {context_record.kind}\")\n", + "print(f\"Claim: {context_record.claim}\")\n", + "print(f\"Model tag: {tag}\")\n", + "print(f\"Validation errors: {errors}\")\n", + "conn.close()" + ] }, { "cell_type": "markdown", "id": "cell-18", "metadata": {}, - "source": "The Hawkins §3.2 schema fields were filled in above, and `validate_record` reports no errors, confirming the claim, rationale and counterevidence are populated and checked, not just printed." + "source": [ + "The tag printed above is now part of the loaded model, and `validate_record` confirms the Python record and the model's own tag agree about what AC-001 is actually about." + ] }, { "cell_type": "markdown", @@ -185,5 +562,27 @@ "Try the chapter exercise in `exercises/ch02/exercise.ipynb`: write an `asserted_context` record for the `brewTemp` assumption in your coffee maker model." ] } - ] + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.14.2" + }, + "title": "Ch2 nb3: asserted context" + }, + "nbformat": 4, + "nbformat_minor": 5 } diff --git a/chapters/ch03-measures/01-moe-definition.ipynb b/chapters/ch03-measures/01-moe-definition.ipynb index 1ad8436..a38a96a 100644 --- a/chapters/ch03-measures/01-moe-definition.ipynb +++ b/chapters/ch03-measures/01-moe-definition.ipynb @@ -20,10 +20,25 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 1, "id": "cell-02", - "metadata": {}, - "outputs": [], + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:25:26.058299Z", + "iopub.status.busy": "2026-10-01T05:25:26.058105Z", + "iopub.status.idle": "2026-10-01T05:25:26.285398Z", + "shell.execute_reply": "2026-10-01T05:25:26.284970Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "requirement timely : TimelyToast;\n" + ] + } + ], "source": [ "from pathlib import Path\n", "import opensysml\n", @@ -35,7 +50,6 @@ "TIMELY_USAGE = \"requirement timely : TimelyToast;\"\n", "print(TIMELY_USAGE)\n", "\n", - "TOASTER_INCREMENT = TIMELY_USAGE\n", "source = Path(\"../../models/ch03-cumulative.sysml\").read_text()\n", "model = conn.load_from_content(source, strict=False)\n", "assert model.ok, f\"Model failed: {format_diagnostics(model.diagnostics)}\"" @@ -51,10 +65,25 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 2, "id": "cell-04", - "metadata": {}, - "outputs": [], + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:25:26.286941Z", + "iopub.status.busy": "2026-10-01T05:25:26.286763Z", + "iopub.status.idle": "2026-10-01T05:25:26.302112Z", + "shell.execute_reply": "2026-10-01T05:25:26.301705Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Expected error: unresolved reference: UndefinedRequirement\n" + ] + } + ], "source": [ "# Negative control: a requirement usage referencing an undefined requirement\n", "# definition raises \"unresolved reference\" at the usage site.\n", @@ -79,10 +108,26 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 3, "id": "cell-06", - "metadata": {}, - "outputs": [], + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:25:26.303324Z", + "iopub.status.busy": "2026-10-01T05:25:26.303251Z", + "iopub.status.idle": "2026-10-01T05:25:26.308020Z", + "shell.execute_reply": "2026-10-01T05:25:26.307660Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "requirement usage kind: requirementUsage\n", + "RequirementUsage: ToasterDemo::timely\n" + ] + } + ], "source": [ "req_usage = model.find(\"ToasterDemo::timely\")\n", "assert req_usage is not None\n", @@ -102,6 +147,87 @@ "`timely` now applies `TimelyToast`'s constraint. What remains open is whether the measure it constrains, toast time, is a measure of effectiveness (the user's acceptance) or a measure of performance (an engineering figure with a threshold derived from something else). That split is a modeling judgment, not a fact the model states. The record below states who cares and which framing the measure takes." ] }, + { + "cell_type": "markdown", + "id": "c9ff57c8", + "metadata": {}, + "source": [ + "Before AC-C03 states its claim, the next cells give it a real anchor in the model: a `ReviewRecordRef` tag naming `timely` as the element the judgment is about." + ] + }, + { + "cell_type": "code", + "execution_count": 4, + "id": "98f1ca8b", + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:25:26.309232Z", + "iopub.status.busy": "2026-10-01T05:25:26.309164Z", + "iopub.status.idle": "2026-10-01T05:25:26.311038Z", + "shell.execute_reply": "2026-10-01T05:25:26.310708Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "metadata acC03Tag : ReviewRecordRef about timely {\n", + " identifier = \"AC-C03\";\n", + "}\n", + "\n" + ] + } + ], + "source": [ + "# what is being claimed, and what, specifically, it is about\n", + "subject_ref = \"ToasterDemo::timely\"\n", + "AC_C03_TAG = \"\"\"\\\n", + "metadata acC03Tag : ReviewRecordRef about timely {\n", + " identifier = \"AC-C03\";\n", + "}\n", + "\"\"\"\n", + "print(AC_C03_TAG)" + ] + }, + { + "cell_type": "markdown", + "id": "9b3b7f7c", + "metadata": {}, + "source": [ + "This fragment, together with the `ReviewRecordRef` definition carried forward from Chapter 2, is the same text now committed in `models/ch03-cumulative.sysml` onward; loading the cumulative model (the `model.ok` cell above, unchanged) is what makes this a real, checkable tag rather than an assertion the reader has to trust." + ] + }, + { + "cell_type": "code", + "execution_count": 5, + "id": "642b8ddc", + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:25:26.312336Z", + "iopub.status.busy": "2026-10-01T05:25:26.312223Z", + "iopub.status.idle": "2026-10-01T05:25:26.314231Z", + "shell.execute_reply": "2026-10-01T05:25:26.313841Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "requirement timely : TimelyToast;\n", + "metadata acC03Tag : ReviewRecordRef about timely {\n", + " identifier = \"AC-C03\";\n", + "}\n", + "\n" + ] + } + ], + "source": [ + "TOASTER_INCREMENT = f\"{TIMELY_USAGE}\\n{AC_C03_TAG}\"\n", + "print(TOASTER_INCREMENT)" + ] + }, { "cell_type": "markdown", "id": "cell-08", @@ -112,10 +238,25 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 6, "id": "cell-09", - "metadata": {}, - "outputs": [], + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:25:26.315265Z", + "iopub.status.busy": "2026-10-01T05:25:26.315198Z", + "iopub.status.idle": "2026-10-01T05:25:26.317028Z", + "shell.execute_reply": "2026-10-01T05:25:26.316744Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "timely (TimelyToast) is framed as a measure of effectiveness: an acceptance criterion for the user's kitchen workflow, not an engineering performance figure derived from a lower-level measure.\n" + ] + } + ], "source": [ "from toaster.evidence import ReviewRecord, validate_record, hash_content\n", "\n", @@ -136,10 +277,25 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 7, "id": "cell-11", - "metadata": {}, - "outputs": [], + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:25:26.318074Z", + "iopub.status.busy": "2026-10-01T05:25:26.317988Z", + "iopub.status.idle": "2026-10-01T05:25:26.319876Z", + "shell.execute_reply": "2026-10-01T05:25:26.319521Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "MoE if the split names who cares and frames the measure as acceptance; MoP if its threshold is derived from a stated MoE with a means of checking (architecture-layers skill).\n" + ] + } + ], "source": [ "scope = (\"ToasterDemo\")\n", "criteria = (\"MoE if the split names who cares and frames the measure as \"\n", @@ -158,10 +314,25 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 8, "id": "cell-13", - "metadata": {}, - "outputs": [], + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:25:26.321350Z", + "iopub.status.busy": "2026-10-01T05:25:26.321244Z", + "iopub.status.idle": "2026-10-01T05:25:26.323199Z", + "shell.execute_reply": "2026-10-01T05:25:26.322880Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "['The MoE/MoP split for toast timing is a case-specific modeling judgment, not a fixed rule.']\n" + ] + } + ], "source": [ "premises = ([])\n", "assumption_refs = ([\n", @@ -181,10 +352,25 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 9, "id": "cell-15", - "metadata": {}, - "outputs": [], + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:25:26.324351Z", + "iopub.status.busy": "2026-10-01T05:25:26.324276Z", + "iopub.status.idle": "2026-10-01T05:25:26.326101Z", + "shell.execute_reply": "2026-10-01T05:25:26.325833Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "TimelyToast's rationale argues from the user's kitchen workflow, not from a solution class or a lower-level performance figure: it names who cares (the user) and frames the 180-second bound as part of what the user accepts, not an engineering figure derived from another measure.\n" + ] + } + ], "source": [ "evidence_refs = ([\n", " \"ToasterDemo::TimelyToast doc: the rationale argues from kitchen \"\n", @@ -208,10 +394,26 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 10, "id": "cell-17", - "metadata": {}, - "outputs": [], + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:25:26.327366Z", + "iopub.status.busy": "2026-10-01T05:25:26.327285Z", + "iopub.status.idle": "2026-10-01T05:25:26.329387Z", + "shell.execute_reply": "2026-10-01T05:25:26.328950Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "TimelyToastTest's doc checks the bound as a timed test at a stated input condition ('nominal input power'), which reads like an engineering performance test rather than an acceptance criterion. Toast time could reasonably be framed either way.\n", + "This split is a contestable modeling judgment, not a settled fact. A later chapter that derives cycle time from the mechanism and the energy balance may instead introduce a genuine MoP threshold derived from this MoE.\n" + ] + } + ], "source": [ "counterevidence = (\"TimelyToastTest's doc checks the bound as a timed test at a stated \"\n", " \"input condition ('nominal input power'), which reads like an \"\n", @@ -235,15 +437,35 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 11, "id": "cell-19", - "metadata": {}, - "outputs": [], + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:25:26.330613Z", + "iopub.status.busy": "2026-10-01T05:25:26.330528Z", + "iopub.status.idle": "2026-10-01T05:25:26.394119Z", + "shell.execute_reply": "2026-10-01T05:25:26.393699Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Model tag: {'tag': 'ToasterDemo::acC03Tag', 'identifier': 'AC-C03', 'annotated_element': 'ToasterDemo::timely'}\n", + "Validation errors: []\n", + "Record kind: asserted_context\n" + ] + } + ], "source": [ + "from toaster.query import get_review_record_refs\n", + "\n", "framing_record = ReviewRecord(\n", " identifier=\"AC-C03\",\n", " kind=\"asserted_context\",\n", " claim=claim,\n", + " subject_ref=subject_ref,\n", " model_ref=model_ref,\n", " content_hash=hash_content(source),\n", " scope=scope,\n", @@ -260,7 +482,9 @@ " record_kind=\"worked_example\",\n", ")\n", "\n", - "errors = validate_record(framing_record)\n", + "errors = validate_record(framing_record, model=model)\n", + "tag = next((t for t in get_review_record_refs(model) if t[\"identifier\"] == framing_record.identifier), None)\n", + "print(f\"Model tag: {tag}\")\n", "print(f\"Validation errors: {errors}\")\n", "print(f\"Record kind: {framing_record.kind}\")\n", "conn.close()" @@ -271,7 +495,7 @@ "id": "cell-20", "metadata": {}, "source": [ - "`validate_record` returns no errors, confirming the framing judgment's required fields, including its own counterevidence, are present." + "The tag printed above is now part of the loaded model, and `validate_record` confirms the Python record and the model's own tag agree about what AC-C03 is actually about." ] }, { @@ -298,7 +522,16 @@ "name": "python3" }, "language_info": { - "name": "python" + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.14.2" }, "title": "Ch3 nb1: requirement usage" }, diff --git a/chapters/ch03-measures/03-threshold-judgment.ipynb b/chapters/ch03-measures/03-threshold-judgment.ipynb index 5ef1c3b..6889e1c 100644 --- a/chapters/ch03-measures/03-threshold-judgment.ipynb +++ b/chapters/ch03-measures/03-threshold-judgment.ipynb @@ -1,17 +1,4 @@ { - "nbformat": 4, - "nbformat_minor": 5, - "metadata": { - "kernelspec": { - "display_name": "Python 3", - "language": "python", - "name": "python3" - }, - "language_info": { - "name": "python" - }, - "title": "Ch3 nb3: threshold judgment" - }, "cells": [ { "cell_type": "markdown", @@ -33,11 +20,117 @@ }, { "cell_type": "code", + "execution_count": 1, "id": "cell-02", - "metadata": {}, - "outputs": [], - "execution_count": null, - "source": "from pathlib import Path\nimport opensysml\nfrom toaster.report import format_diagnostics\n\nconn = opensysml.connect(version=\"v0.9.0\")\nsource = Path(\"../../models/ch03-cumulative.sysml\").read_text()\nprint(source)\nmodel = conn.load_from_content(source, strict=False)\nassert model.ok, f\"Model failed: {format_diagnostics(model.diagnostics)}\"" + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:25:33.529669Z", + "iopub.status.busy": "2026-10-01T05:25:33.529378Z", + "iopub.status.idle": "2026-10-01T05:25:33.758314Z", + "shell.execute_reply": "2026-10-01T05:25:33.757865Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "// GENERATED FIXTURE: do not edit directly.\n", + "// Run: python scripts/check_construction.py --check (to verify)\n", + "// Source: notebook cell-02 TOASTER_INCREMENT in chapter 3's construct-introducing notebooks.\n", + "\n", + "package ToasterDemo {\n", + " metadata def ReviewRecordRef {\n", + " attribute identifier : ScalarValues::String;\n", + " }\n", + "\n", + " private import ScalarValues::*;\n", + " private import SI::*;\n", + " private import ISQ::*;\n", + "\n", + " item def Bread;\n", + " item def Toast;\n", + "\n", + " action def ToastBread {\n", + " doc /* Transform bread into toast acceptable to its user. */\n", + " in bread : Bread;\n", + " out toast : Toast;\n", + " }\n", + "\n", + " abstract part def ToastingSystem {\n", + " perform action toastBread : ToastBread;\n", + " }\n", + "\n", + " part def HeatingSystem;\n", + " part def ControlSystem;\n", + "\n", + " part def Toaster :> ToastingSystem {\n", + " attribute cycleTime : ISQ::DurationValue;\n", + " part heating : HeatingSystem;\n", + " part control : ControlSystem;\n", + " }\n", + "\n", + " requirement def TimelyToast {\n", + " doc /*\n", + " * The toaster shall complete a toasting cycle in at most 180 seconds.\n", + " * Rationale: kitchen workflows typically span 5-15 minutes; a cycle\n", + " * exceeding 3 minutes delays meal preparation and falls outside where\n", + " * and how a user prepares a meal.\n", + " */\n", + " subject toaster : Toaster;\n", + " require constraint { toaster.cycleTime <= 180.0 [SI::s] }\n", + " }\n", + "\n", + " requirement timely : TimelyToast;\n", + "\n", + " metadata acC03Tag : ReviewRecordRef about timely {\n", + " identifier = \"AC-C03\";\n", + " }\n", + "\n", + " metadata asC03Tag : ReviewRecordRef about timely {\n", + " identifier = \"AS-C03\";\n", + " }\n", + "\n", + " part nominal : Toaster;\n", + " metadata ac001Tag : ReviewRecordRef about nominal {\n", + " identifier = \"AC-001\";\n", + " }\n", + "\n", + " part slow : Toaster {\n", + " attribute :>> cycleTime = 200.0 [SI::s];\n", + " assert not satisfy timely by slow;\n", + " }\n", + "\n", + " verification def TimelyToastTest {\n", + " doc /*\n", + " * Verification method: timed test of three consecutive toasting cycles at\n", + " * nominal input power; all must complete within 180 seconds.\n", + " * Method type: test (VerificationMethodKind::test, SysML v2 §7.24 Table 22).\n", + " * Note: formal #verificationMethod metadata not yet supported in OpenSysML v0.9.0;\n", + " * tracked at toaster#19 / OpenSysML#608.\n", + " * spec: SysML v2 formal/2026-03-02 §7.24.2 (VerificationCaseDefinition).\n", + " */\n", + " subject toaster : Toaster;\n", + " objective {\n", + " verify timely;\n", + " }\n", + " }\n", + "}\n", + "\n" + ] + } + ], + "source": [ + "from pathlib import Path\n", + "import opensysml\n", + "from toaster.report import format_diagnostics\n", + "\n", + "conn = opensysml.connect(version=\"v0.9.0\")\n", + "source = Path(\"../../models/ch03-cumulative.sysml\").read_text()\n", + "print(source)\n", + "model = conn.load_from_content(source, strict=False)\n", + "assert model.ok, f\"Model failed: {format_diagnostics(model.diagnostics)}\"" + ] }, { "cell_type": "markdown", @@ -49,11 +142,43 @@ }, { "cell_type": "code", + "execution_count": 2, "id": "cell-04", - "metadata": {}, - "outputs": [], - "execution_count": null, - "source": "# Negative control: a require constraint referencing an attribute the\n# subject's type does not declare raises \"unresolved member\" at the point\n# of use.\nbad_source = \"\"\"\npackage Bad {\n private import ScalarValues::*;\n part def Toaster { attribute cycleTime : Real default = 120.0; }\n requirement def BadReq {\n subject t : Toaster;\n require constraint { t.notAnAttribute <= 180.0 }\n }\n}\n\"\"\"\nbad = conn.load_from_content(bad_source, strict=False)\nassert not bad.ok\nprint(\"Expected error:\", bad.diagnostics[0].message)" + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:25:33.759748Z", + "iopub.status.busy": "2026-10-01T05:25:33.759570Z", + "iopub.status.idle": "2026-10-01T05:25:33.763292Z", + "shell.execute_reply": "2026-10-01T05:25:33.762809Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Expected error: unresolved member: notAnAttribute\n" + ] + } + ], + "source": [ + "# Negative control: a require constraint referencing an attribute the\n", + "# subject's type does not declare raises \"unresolved member\" at the point\n", + "# of use.\n", + "bad_source = \"\"\"\n", + "package Bad {\n", + " private import ScalarValues::*;\n", + " part def Toaster { attribute cycleTime : Real default = 120.0; }\n", + " requirement def BadReq {\n", + " subject t : Toaster;\n", + " require constraint { t.notAnAttribute <= 180.0 }\n", + " }\n", + "}\n", + "\"\"\"\n", + "bad = conn.load_from_content(bad_source, strict=False)\n", + "assert not bad.ok\n", + "print(\"Expected error:\", bad.diagnostics[0].message)" + ] }, { "cell_type": "markdown", @@ -67,92 +192,363 @@ "cell_type": "markdown", "id": "cell-06", "metadata": {}, - "source": "The evaluation itself comes first: `model.eval` checks the negated claim on `slow` against the model's own values. `claim` states what that evaluation is being read as supporting, and `model_ref` names the requirement it concerns." + "source": [ + "The evaluation itself comes first: `model.eval` checks the negated claim on `slow` against the model's own values. `claim` states what that evaluation is being read as supporting, and `model_ref` names the requirement it concerns." + ] }, { "cell_type": "code", + "execution_count": 3, "id": "cell-07", + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:25:33.764583Z", + "iopub.status.busy": "2026-10-01T05:25:33.764499Z", + "iopub.status.idle": "2026-10-01T05:25:33.769972Z", + "shell.execute_reply": "2026-10-01T05:25:33.769622Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "ToasterDemo::timely(ToasterDemo::slow) = False\n", + "The negated claim on slow (assert not satisfy timely by slow) evaluates True: slow's cycleTime (200 s) fails timely's 180 s bound, as intended for the injected fault. No corresponding claim is made for nominal.\n" + ] + } + ], + "source": [ + "from toaster.evidence import ReviewRecord, validate_record, hash_content\n", + "\n", + "slow_holds = model.eval(\"ToasterDemo::timely(ToasterDemo::slow)\")\n", + "print(f\"ToasterDemo::timely(ToasterDemo::slow) = {slow_holds}\")\n", + "\n", + "claim = (\"The negated claim on slow (assert not satisfy timely by slow) \"\n", + " \"evaluates True: slow's cycleTime (200 s) fails timely's 180 s \"\n", + " \"bound, as intended for the injected fault. No corresponding claim is made for nominal.\")\n", + "model_ref = (\"ToasterDemo::timely\")\n", + "print(claim)" + ] + }, + { + "cell_type": "markdown", + "id": "9bd0d71d", "metadata": {}, - "outputs": [], - "execution_count": null, - "source": "from toaster.evidence import ReviewRecord, validate_record, hash_content\n\nslow_holds = model.eval(\"ToasterDemo::timely(ToasterDemo::slow)\")\nprint(f\"ToasterDemo::timely(ToasterDemo::slow) = {slow_holds}\")\n\nclaim = (\"The negated claim on slow (assert not satisfy timely by slow) \"\n \"evaluates True: slow's cycleTime (200 s) fails timely's 180 s \"\n \"bound, as intended for the injected fault. No corresponding claim is made for nominal.\")\nmodel_ref = (\"ToasterDemo::timely\")\nprint(claim)" + "source": [ + "Before moving to scope and criteria, the next cells give this claim a real anchor in the model: a `ReviewRecordRef` tag naming `timely` as what AS-C03 is about." + ] + }, + { + "cell_type": "code", + "execution_count": 4, + "id": "fcfb3488", + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:25:33.771705Z", + "iopub.status.busy": "2026-10-01T05:25:33.771600Z", + "iopub.status.idle": "2026-10-01T05:25:33.773337Z", + "shell.execute_reply": "2026-10-01T05:25:33.772972Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "metadata asC03Tag : ReviewRecordRef about timely {\n", + " identifier = \"AS-C03\";\n", + "}\n", + "\n" + ] + } + ], + "source": [ + "# what is being claimed, and what, specifically, it is about\n", + "subject_ref = \"ToasterDemo::timely\"\n", + "AS_C03_TAG = \"\"\"\\\n", + "metadata asC03Tag : ReviewRecordRef about timely {\n", + " identifier = \"AS-C03\";\n", + "}\n", + "\"\"\"\n", + "print(AS_C03_TAG)" + ] + }, + { + "cell_type": "markdown", + "id": "d2c9c179", + "metadata": {}, + "source": [ + "This fragment is the same text now committed in `models/ch03-cumulative.sysml` onward, alongside the `ReviewRecordRef` definition (carried forward from Chapter 2) and `acC03Tag` (notebook 01); loading the cumulative model (the `model.ok` cell above, unchanged) is what makes this a real, checkable tag rather than an assertion the reader has to trust." + ] + }, + { + "cell_type": "code", + "execution_count": 5, + "id": "0cb81adc", + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:25:33.774487Z", + "iopub.status.busy": "2026-10-01T05:25:33.774420Z", + "iopub.status.idle": "2026-10-01T05:25:33.776133Z", + "shell.execute_reply": "2026-10-01T05:25:33.775848Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "metadata asC03Tag : ReviewRecordRef about timely {\n", + " identifier = \"AS-C03\";\n", + "}\n", + "\n" + ] + } + ], + "source": [ + "TOASTER_INCREMENT = AS_C03_TAG\n", + "print(TOASTER_INCREMENT)" + ] }, { "cell_type": "markdown", "id": "cell-08", "metadata": {}, - "source": "`scope` and `criteria` state the standard the claim is judged against: exactly which evaluation, on which usage, counts as this claim holding." + "source": [ + "`scope` and `criteria` state the standard the claim is judged against: exactly which evaluation, on which usage, counts as this claim holding." + ] }, { "cell_type": "code", + "execution_count": 6, "id": "cell-09", - "metadata": {}, - "outputs": [], - "execution_count": null, - "source": "scope = (\"ToasterDemo\")\ncriteria = (\"assert not satisfy timely by slow holds when \"\n \"ToasterDemo::timely(ToasterDemo::slow) evaluates False.\")\nprint(criteria)" + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:25:33.777156Z", + "iopub.status.busy": "2026-10-01T05:25:33.777088Z", + "iopub.status.idle": "2026-10-01T05:25:33.779002Z", + "shell.execute_reply": "2026-10-01T05:25:33.778537Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "assert not satisfy timely by slow holds when ToasterDemo::timely(ToasterDemo::slow) evaluates False.\n" + ] + } + ], + "source": [ + "scope = (\"ToasterDemo\")\n", + "criteria = (\"assert not satisfy timely by slow holds when \"\n", + " \"ToasterDemo::timely(ToasterDemo::slow) evaluates False.\")\n", + "print(criteria)" + ] }, { "cell_type": "markdown", "id": "cell-10", "metadata": {}, - "source": "`premises` names what makes `slow` a legitimate fixture for this claim in the first place: a deliberately injected value, not a derived one." + "source": [ + "`premises` names what makes `slow` a legitimate fixture for this claim in the first place: a deliberately injected value, not a derived one." + ] }, { "cell_type": "code", + "execution_count": 7, "id": "cell-11", - "metadata": {}, - "outputs": [], - "execution_count": null, - "source": "premises = ([\n \"slow.cycleTime is fixed at 200.0 [SI::s], a deliberately injected \"\n \"fault value (Chapter 2), not a value derived from any mechanism.\",\n ])\nprint(premises)" + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:25:33.780305Z", + "iopub.status.busy": "2026-10-01T05:25:33.780223Z", + "iopub.status.idle": "2026-10-01T05:25:33.782193Z", + "shell.execute_reply": "2026-10-01T05:25:33.781866Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "['slow.cycleTime is fixed at 200.0 [SI::s], a deliberately injected fault value (Chapter 2), not a value derived from any mechanism.']\n" + ] + } + ], + "source": [ + "premises = ([\n", + " \"slow.cycleTime is fixed at 200.0 [SI::s], a deliberately injected \"\n", + " \"fault value (Chapter 2), not a value derived from any mechanism.\",\n", + " ])\n", + "print(premises)" + ] }, { "cell_type": "markdown", "id": "cell-12", "metadata": {}, - "source": "`AC-C03`'s framing judgment is the assumption this record rests on; `evidence_refs` points at the evaluation above, and `rationale` connects that evidence to the claim." + "source": [ + "`AC-C03`'s framing judgment is the assumption this record rests on; `evidence_refs` points at the evaluation above, and `rationale` connects that evidence to the claim." + ] }, { "cell_type": "code", + "execution_count": 8, "id": "cell-13", - "metadata": {}, - "outputs": [], - "execution_count": null, - "source": "assumption_refs = ([\n \"AC-C03: timely is framed as a measure of effectiveness \"\n \"(notebook 01).\",\n ])\nevidence_refs = ([\n \"assert not satisfy timely by slow (ToasterDemo::slow::@1), \"\n \"evaluated by model.eval('ToasterDemo::timely(ToasterDemo::slow)') \"\n \"= False, above.\",\n ])\nrationale = (\"slow's fixed cycleTime exceeds the bound, and the model's own \"\n \"evaluation confirms the negated claim holds. This demonstrates \"\n \"the deliberately negated satisfaction claim applied to an \"\n \"injected fault value: a real, evaluable claim about a fixture \"\n \"built to fail, not a check that traces a failure back to a \"\n \"design choice, since deriving a cycle time from an actual \"\n \"mechanism is not yet possible. Toaster::cycleTime carries no \"\n \"default value, so nominal.cycleTime has no value and \"\n \"ToasterDemo::timely(ToasterDemo::nominal) cannot be evaluated \"\n \"at all; claiming nominal satisfies timely would assert a \"\n \"result that was never computed.\")\nprint(rationale)" + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:25:33.783326Z", + "iopub.status.busy": "2026-10-01T05:25:33.783259Z", + "iopub.status.idle": "2026-10-01T05:25:33.785158Z", + "shell.execute_reply": "2026-10-01T05:25:33.784774Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "slow's fixed cycleTime exceeds the bound, and the model's own evaluation confirms the negated claim holds. This demonstrates the deliberately negated satisfaction claim applied to an injected fault value: a real, evaluable claim about a fixture built to fail, not a check that traces a failure back to a design choice, since deriving a cycle time from an actual mechanism is not yet possible. Toaster::cycleTime carries no default value, so nominal.cycleTime has no value and ToasterDemo::timely(ToasterDemo::nominal) cannot be evaluated at all; claiming nominal satisfies timely would assert a result that was never computed.\n" + ] + } + ], + "source": [ + "assumption_refs = ([\n", + " \"AC-C03: timely is framed as a measure of effectiveness \"\n", + " \"(notebook 01).\",\n", + " ])\n", + "evidence_refs = ([\n", + " \"assert not satisfy timely by slow (ToasterDemo::slow::@1), \"\n", + " \"evaluated by model.eval('ToasterDemo::timely(ToasterDemo::slow)') \"\n", + " \"= False, above.\",\n", + " ])\n", + "rationale = (\"slow's fixed cycleTime exceeds the bound, and the model's own \"\n", + " \"evaluation confirms the negated claim holds. This demonstrates \"\n", + " \"the deliberately negated satisfaction claim applied to an \"\n", + " \"injected fault value: a real, evaluable claim about a fixture \"\n", + " \"built to fail, not a check that traces a failure back to a \"\n", + " \"design choice, since deriving a cycle time from an actual \"\n", + " \"mechanism is not yet possible. Toaster::cycleTime carries no \"\n", + " \"default value, so nominal.cycleTime has no value and \"\n", + " \"ToasterDemo::timely(ToasterDemo::nominal) cannot be evaluated \"\n", + " \"at all; claiming nominal satisfies timely would assert a \"\n", + " \"result that was never computed.\")\n", + "print(rationale)" + ] }, { "cell_type": "markdown", "id": "cell-14", "metadata": {}, - "source": "The challenge: `counterevidence` states plainly what this result does not demonstrate, and `residual_uncertainties` says what stays genuinely open about `nominal` (Hawkins' trustworthiness again: an evaluated True is not the same as a settled question)." + "source": [ + "The challenge: `counterevidence` states plainly what this result does not demonstrate, and `residual_uncertainties` says what stays genuinely open about `nominal` (Hawkins' trustworthiness again: an evaluated True is not the same as a settled question)." + ] }, { "cell_type": "code", + "execution_count": 9, "id": "cell-15", - "metadata": {}, - "outputs": [], - "execution_count": null, - "source": "counterevidence = (\"This only demonstrates that a chosen fault value fails. It does \"\n \"not demonstrate that a derived cycle time can pass; that \"\n \"demonstration needs a chapter that derives cycle time from the \"\n \"mechanism and the energy balance.\")\nresidual_uncertainties = (\"Whether timely is best framed as a measure of effectiveness or a \"\n \"measure of performance stays a contestable judgment (AC-C03, \"\n \"notebook 01), independent of this result. nominal's status under \"\n \"timely is genuinely open, not merely deferred.\")\nprint(counterevidence)\nprint(residual_uncertainties)" + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:25:33.786353Z", + "iopub.status.busy": "2026-10-01T05:25:33.786279Z", + "iopub.status.idle": "2026-10-01T05:25:33.788321Z", + "shell.execute_reply": "2026-10-01T05:25:33.787856Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "This only demonstrates that a chosen fault value fails. It does not demonstrate that a derived cycle time can pass; that demonstration needs a chapter that derives cycle time from the mechanism and the energy balance.\n", + "Whether timely is best framed as a measure of effectiveness or a measure of performance stays a contestable judgment (AC-C03, notebook 01), independent of this result. nominal's status under timely is genuinely open, not merely deferred.\n" + ] + } + ], + "source": [ + "counterevidence = (\"This only demonstrates that a chosen fault value fails. It does \"\n", + " \"not demonstrate that a derived cycle time can pass; that \"\n", + " \"demonstration needs a chapter that derives cycle time from the \"\n", + " \"mechanism and the energy balance.\")\n", + "residual_uncertainties = (\"Whether timely is best framed as a measure of effectiveness or a \"\n", + " \"measure of performance stays a contestable judgment (AC-C03, \"\n", + " \"notebook 01), independent of this result. nominal's status under \"\n", + " \"timely is genuinely open, not merely deferred.\")\n", + "print(counterevidence)\n", + "print(residual_uncertainties)" + ] }, { "cell_type": "markdown", "id": "cell-16", "metadata": {}, - "source": "With every part named above, the solution record assembles from them directly." + "source": [ + "With every part named above, the solution record assembles from them directly." + ] }, { "cell_type": "code", + "execution_count": 10, "id": "cell-17", - "metadata": {}, - "outputs": [], - "execution_count": null, - "source": "solution_record = ReviewRecord(\n identifier=\"AS-C03\",\n kind=\"asserted_solution\",\n claim=claim,\n model_ref=model_ref,\n content_hash=hash_content(source),\n scope=scope,\n criteria=criteria,\n premises=premises,\n assumption_refs=assumption_refs,\n evidence_refs=evidence_refs,\n rationale=rationale,\n counterevidence=counterevidence,\n residual_uncertainties=residual_uncertainties,\n disposition=\"pending\",\n dependency_freshness=\"current\",\n engineering_conclusion=\"supported\",\n record_kind=\"worked_example\",\n)\n\nerrors = validate_record(solution_record)\nprint(f\"Validation errors: {errors}\")\nprint(f\"Record kind: {solution_record.kind}\")\nconn.close()" + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:25:33.789467Z", + "iopub.status.busy": "2026-10-01T05:25:33.789393Z", + "iopub.status.idle": "2026-10-01T05:25:33.854555Z", + "shell.execute_reply": "2026-10-01T05:25:33.854058Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Model tag: {'tag': 'ToasterDemo::asC03Tag', 'identifier': 'AS-C03', 'annotated_element': 'ToasterDemo::timely'}\n", + "Validation errors: []\n", + "Record kind: asserted_solution\n" + ] + } + ], + "source": [ + "from toaster.query import get_review_record_refs\n", + "\n", + "solution_record = ReviewRecord(\n", + " identifier=\"AS-C03\",\n", + " kind=\"asserted_solution\",\n", + " claim=claim,\n", + " subject_ref=subject_ref,\n", + " model_ref=model_ref,\n", + " content_hash=hash_content(source),\n", + " scope=scope,\n", + " criteria=criteria,\n", + " premises=premises,\n", + " assumption_refs=assumption_refs,\n", + " evidence_refs=evidence_refs,\n", + " rationale=rationale,\n", + " counterevidence=counterevidence,\n", + " residual_uncertainties=residual_uncertainties,\n", + " disposition=\"pending\",\n", + " dependency_freshness=\"current\",\n", + " engineering_conclusion=\"supported\",\n", + " record_kind=\"worked_example\",\n", + ")\n", + "\n", + "errors = validate_record(solution_record, model=model)\n", + "tag = next((t for t in get_review_record_refs(model) if t[\"identifier\"] == solution_record.identifier), None)\n", + "print(f\"Model tag: {tag}\")\n", + "print(f\"Validation errors: {errors}\")\n", + "print(f\"Record kind: {solution_record.kind}\")\n", + "conn.close()" + ] }, { "cell_type": "markdown", "id": "cell-18", "metadata": {}, "source": [ - "The Hawkins §3.3 schema fields were filled in above, and `validate_record` reports no errors, confirming the claim, rationale and counterevidence are populated and checked, not just printed." + "The tag printed above is now part of the loaded model, and `validate_record` confirms the Python record and the model's own tag agree about what AS-C03 is actually about." ] }, { @@ -163,5 +559,27 @@ "Try the chapter exercise in `exercises/ch03/exercise.ipynb`: write an `asserted_solution` record for your satisfaction claim." ] } - ] + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.14.2" + }, + "title": "Ch3 nb3: threshold judgment" + }, + "nbformat": 4, + "nbformat_minor": 5 } diff --git a/chapters/ch04-functional-decomp/03-completeness-check.ipynb b/chapters/ch04-functional-decomp/03-completeness-check.ipynb index 1185311..fffa477 100644 --- a/chapters/ch04-functional-decomp/03-completeness-check.ipynb +++ b/chapters/ch04-functional-decomp/03-completeness-check.ipynb @@ -20,10 +20,142 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 1, "id": "cell-02", - "metadata": {}, - "outputs": [], + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:40:51.381690Z", + "iopub.status.busy": "2026-10-01T05:40:51.381516Z", + "iopub.status.idle": "2026-10-01T05:40:51.610500Z", + "shell.execute_reply": "2026-10-01T05:40:51.609942Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "// GENERATED FIXTURE: do not edit directly.\n", + "// Run: python scripts/check_construction.py --check (to verify)\n", + "// Source: notebook cell-02 TOASTER_INCREMENT in chapter 4's construct-introducing notebooks.\n", + "\n", + "package ToasterDemo {\n", + " metadata def ReviewRecordRef {\n", + " attribute identifier : ScalarValues::String;\n", + " }\n", + "\n", + " private import ScalarValues::*;\n", + " private import SI::*;\n", + " private import ISQ::*;\n", + "\n", + " item def Bread;\n", + " item def Toast;\n", + "\n", + " action def ApplyHeat {\n", + " in bread : Bread;\n", + " in energy : ISQ::EnergyValue[0..*];\n", + " in duration : ISQ::DurationValue[0..*] {\n", + " doc /* Signal from a control function: how long to apply heat.\n", + " * No control function is modeled in this chapter, so this input\n", + " * is declared and typed but not yet connected to a value. */\n", + " }\n", + " out toast : Toast;\n", + " out delivered : ISQ::EnergyValue;\n", + " out loss : ISQ::EnergyValue;\n", + "\n", + " assert constraint balance {\n", + " delivered >= 0.0 [SI::J] and loss >= 0.0 [SI::J] and delivered + loss <= energy\n", + " }\n", + " }\n", + "\n", + " metadata aiC04Tag : ReviewRecordRef about ApplyHeat {\n", + " identifier = \"AI-C04\";\n", + " }\n", + "\n", + " action def ToastBread {\n", + " doc /* Transform bread into toast acceptable to its user. */\n", + " in bread : Bread;\n", + " out toast : Toast;\n", + " first start;\n", + " then action applyHeat : ApplyHeat {\n", + " in bread = ToastBread::bread;\n", + " }\n", + " then done;\n", + " }\n", + "\n", + " abstract part def ToastingSystem {\n", + " perform action toastBread : ToastBread;\n", + " }\n", + "\n", + " part def HeatingSystem;\n", + " part def ControlSystem;\n", + "\n", + " part def Toaster :> ToastingSystem {\n", + " attribute cycleTime : ISQ::DurationValue;\n", + " part heating : HeatingSystem;\n", + " part control : ControlSystem;\n", + " }\n", + "\n", + " requirement def TimelyToast {\n", + " doc /*\n", + " * The toaster shall complete a toasting cycle in at most 180 seconds.\n", + " * Rationale: kitchen workflows typically span 5-15 minutes; a cycle\n", + " * exceeding 3 minutes delays meal preparation and falls outside where\n", + " * and how a user prepares a meal.\n", + " */\n", + " subject toaster : Toaster;\n", + " require constraint { toaster.cycleTime <= 180.0 [SI::s] }\n", + " }\n", + "\n", + " requirement timely : TimelyToast;\n", + "\n", + " metadata acC03Tag : ReviewRecordRef about timely {\n", + " identifier = \"AC-C03\";\n", + " }\n", + "\n", + " metadata asC03Tag : ReviewRecordRef about timely {\n", + " identifier = \"AS-C03\";\n", + " }\n", + "\n", + " part nominal : Toaster;\n", + " metadata ac001Tag : ReviewRecordRef about nominal {\n", + " identifier = \"AC-001\";\n", + " }\n", + "\n", + " part slow : Toaster {\n", + " attribute :>> cycleTime = 200.0 [SI::s];\n", + " assert not satisfy timely by slow;\n", + " }\n", + "\n", + " verification def TimelyToastTest {\n", + " doc /*\n", + " * Verification method: timed test of three consecutive toasting cycles at\n", + " * nominal input power; all must complete within 180 seconds.\n", + " * Method type: test (VerificationMethodKind::test, SysML v2 §7.24 Table 22).\n", + " * Note: formal #verificationMethod metadata not yet supported in OpenSysML v0.9.0;\n", + " * tracked at toaster#19 / OpenSysML#608.\n", + " * spec: SysML v2 formal/2026-03-02 §7.24.2 (VerificationCaseDefinition).\n", + " */\n", + " subject toaster : Toaster;\n", + " objective {\n", + " verify timely;\n", + " }\n", + " }\n", + "\n", + " item def Start {\n", + " doc /* Signal marking the start of a toasting cycle, not the bread itself. */\n", + " }\n", + " item def Finish {\n", + " doc /* Signal marking the finish of a toasting cycle, not the toast itself. */\n", + " }\n", + " item def Cancel {\n", + " doc /* Signal requesting cancellation of an in-progress toasting cycle. */\n", + " }\n", + "}\n", + "\n" + ] + } + ], "source": [ "from pathlib import Path\n", "import opensysml\n", @@ -47,10 +179,25 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 2, "id": "cell-04", - "metadata": {}, - "outputs": [], + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:40:51.611997Z", + "iopub.status.busy": "2026-10-01T05:40:51.611813Z", + "iopub.status.idle": "2026-10-01T05:40:51.626190Z", + "shell.execute_reply": "2026-10-01T05:40:51.625790Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Expected error: unresolved reference: notAFeature\n" + ] + } + ], "source": [ "# Negative control: an asserted constraint referencing a feature the action\n", "# definition does not declare raises \"unresolved reference\".\n", @@ -75,7 +222,7 @@ "id": "cell-05", "metadata": {}, "source": [ - "With the negative control confirmed, the next cells build `AI-C04`: an `asserted_inference` record about `ApplyHeat`'s own flow accounting, following the construction zone Hawkins' taxonomy uses (claim, frame, premises, evidence, challenge, assemble)." + "With the negative control confirmed, the next cells build `AI-C04`: an `asserted_inference` record about `ApplyHeat`'s own flow accounting, following the construction zone Hawkins' taxonomy uses (anchor, claim, frame, premises, evidence, challenge, assemble)." ] }, { @@ -83,15 +230,110 @@ "id": "cell-06", "metadata": {}, "source": [ - "The claim first: what is being claimed, and about which model element." + "Before AI-C04 states its claim, the next cells give it a real anchor in the model: a `ReviewRecordRef` tag naming `ApplyHeat` as what AI-C04 is about." ] }, { "cell_type": "code", - "execution_count": null, - "id": "cell-07", + "execution_count": 3, + "id": "d7a21b20", + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:40:51.627489Z", + "iopub.status.busy": "2026-10-01T05:40:51.627401Z", + "iopub.status.idle": "2026-10-01T05:40:51.629209Z", + "shell.execute_reply": "2026-10-01T05:40:51.628819Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "metadata aiC04Tag : ReviewRecordRef about ApplyHeat {\n", + " identifier = \"AI-C04\";\n", + "}\n", + "\n" + ] + } + ], + "source": [ + "# what is being claimed, and what, specifically, it is about\n", + "subject_ref = \"ToasterDemo::ApplyHeat\"\n", + "AI_C04_TAG = \"\"\"\\\n", + "metadata aiC04Tag : ReviewRecordRef about ApplyHeat {\n", + " identifier = \"AI-C04\";\n", + "}\n", + "\"\"\"\n", + "print(AI_C04_TAG)\n" + ] + }, + { + "cell_type": "markdown", + "id": "2b660112", "metadata": {}, - "outputs": [], + "source": [ + "This fragment is the same text now committed in `models/ch04-cumulative.sysml` onward, alongside the `ReviewRecordRef` definition (carried forward from Chapter 2); loading the cumulative model (the `model.ok` cell above, unchanged) is what makes this a real, checkable tag rather than an assertion the reader has to trust." + ] + }, + { + "cell_type": "code", + "execution_count": 4, + "id": "cf4859f1", + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:40:51.630266Z", + "iopub.status.busy": "2026-10-01T05:40:51.630196Z", + "iopub.status.idle": "2026-10-01T05:40:51.631779Z", + "shell.execute_reply": "2026-10-01T05:40:51.631465Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "metadata aiC04Tag : ReviewRecordRef about ApplyHeat {\n", + " identifier = \"AI-C04\";\n", + "}\n", + "\n" + ] + } + ], + "source": [ + "TOASTER_INCREMENT = AI_C04_TAG\n", + "print(TOASTER_INCREMENT)\n" + ] + }, + { + "cell_type": "markdown", + "id": "cf1015b6", + "metadata": {}, + "source": [ + "AI-C04 states the claim next: what is being claimed, and about which model element." + ] + }, + { + "cell_type": "code", + "execution_count": 5, + "id": "cell-07", + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:40:51.632904Z", + "iopub.status.busy": "2026-10-01T05:40:51.632827Z", + "iopub.status.idle": "2026-10-01T05:40:51.634977Z", + "shell.execute_reply": "2026-10-01T05:40:51.634729Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "ApplyHeat's typed flows account for bread, energy and duration in, and toast, delivered energy and loss out; the asserted balance constraint (delivered and loss each non-negative, their sum bounded by energy) is a real, evaluable relation, not merely declared syntax. ApplyHeat is a step of ToastBread, its bread input bound to ToastBread's own bread.\n" + ] + } + ], "source": [ "claim = (\"ApplyHeat's typed flows account for bread, energy and duration in, and \"\n", " \"toast, delivered energy and loss out; the asserted balance constraint \"\n", @@ -112,10 +354,25 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 6, "id": "cell-09", - "metadata": {}, - "outputs": [], + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:40:51.636150Z", + "iopub.status.busy": "2026-10-01T05:40:51.636083Z", + "iopub.status.idle": "2026-10-01T05:40:51.638081Z", + "shell.execute_reply": "2026-10-01T05:40:51.637765Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Every declared flow (bread, energy, duration in; toast, delivered, loss out) is named and typed, and the asserted balance constraint (delivered and loss each non-negative, their sum bounded by energy) evaluates against concrete values, holding or failing as conservation requires. This is not the criterion for the toaster's full functional architecture (about fifteen verb-noun functions, index.md); it is the criterion for this one worked example.\n" + ] + } + ], "source": [ "scope = \"ToasterDemo::ApplyHeat, nested inside ToasterDemo::ToastBread\"\n", "criteria = (\"Every declared flow (bread, energy, duration in; toast, delivered, loss \"\n", @@ -138,10 +395,25 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 7, "id": "cell-11", - "metadata": {}, - "outputs": [], + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:40:51.639211Z", + "iopub.status.busy": "2026-10-01T05:40:51.639135Z", + "iopub.status.idle": "2026-10-01T05:40:51.641072Z", + "shell.execute_reply": "2026-10-01T05:40:51.640710Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "[\"duration is declared as a typed input but plays no role in the balance constraint: this chapter does not derive delivered energy from duration, so duration's own denotation as a signal from a control function is stated but not yet connected to any computation.\"]\n" + ] + } + ], "source": [ "premises = [\"AS-C03\"]\n", "assumption_refs = [\n", @@ -163,10 +435,25 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 8, "id": "cell-13", - "metadata": {}, - "outputs": [], + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:40:51.642418Z", + "iopub.status.busy": "2026-10-01T05:40:51.642332Z", + "iopub.status.idle": "2026-10-01T05:40:51.648083Z", + "shell.execute_reply": "2026-10-01T05:40:51.647750Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "slow.cycleTime: 200 [SI::s]\n" + ] + } + ], "source": [ "slow_cycle_time = model.eval(\"ToasterDemo::slow.cycleTime\")\n", "print(f\"slow.cycleTime: {slow_cycle_time}\")\n" @@ -190,10 +477,27 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 9, "id": "cell-16", - "metadata": {}, - "outputs": [], + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:40:51.649250Z", + "iopub.status.busy": "2026-10-01T05:40:51.649184Z", + "iopub.status.idle": "2026-10-01T05:40:51.665534Z", + "shell.execute_reply": "2026-10-01T05:40:51.665179Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Probe::plausibleHeat: ✓ constraint Probe::ApplyHeat::balance holds (on Probe::plausibleHeat ID: 1) — observed by run\n", + "Probe::implausibleHeat: ✗ constraint Probe::ApplyHeat::balance fails (on Probe::implausibleHeat ID: 1): condition evaluated to false: delivered >= 0.0 [SI::J] and loss >= 0.0 [SI::J] and delivered + loss <= energy — witnessed by run\n", + "Probe::negativeLossHeat: ✗ constraint Probe::ApplyHeat::balance fails (on Probe::negativeLossHeat ID: 1): condition evaluated to false: delivered >= 0.0 [SI::J] and loss >= 0.0 [SI::J] and delivered + loss <= energy — witnessed by run\n" + ] + } + ], "source": [ "PROBE_SOURCE = \"\"\"\n", "package Probe {\n", @@ -251,10 +555,25 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 10, "id": "cell-18", - "metadata": {}, - "outputs": [], + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:40:51.667078Z", + "iopub.status.busy": "2026-10-01T05:40:51.666976Z", + "iopub.status.idle": "2026-10-01T05:40:51.669458Z", + "shell.execute_reply": "2026-10-01T05:40:51.669070Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Every parameter ApplyHeat declares is named in the claim above, and the explicit [0..*] on energy and duration keeps the whole model evaluable: slow.cycleTime, unrelated to ApplyHeat, evaluates cleanly above. The asserted balance constraint is not merely stated either: the probe above shows it holds for values consistent with conservation and fails for both an overdrawn split and a negative-loss split, so the non-negativity bounds are doing real work, not just the sum bound. ApplyHeat is reachable from ToastBread's own sequence, so it is a step of a decomposition, not a definition nothing composes.\n" + ] + } + ], "source": [ "evidence_refs = [\n", " f\"ToasterDemo::slow.cycleTime: {slow_cycle_time}, evaluated directly against \"\n", @@ -288,10 +607,26 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 11, "id": "cell-20", - "metadata": {}, - "outputs": [], + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:40:51.670592Z", + "iopub.status.busy": "2026-10-01T05:40:51.670506Z", + "iopub.status.idle": "2026-10-01T05:40:51.672716Z", + "shell.execute_reply": "2026-10-01T05:40:51.672360Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "bread is wired from ToastBread's own bread parameter, but energy is not: no energy source exists anywhere in the model yet, so leaving it unbound is honest, not an omission. duration is declared but unconnected to the balance constraint or to any other computation in this chapter, so it does not yet participate in the flow accounting it claims to name. ApplyHeat is the only function modeled; Douglas's functional architecture names roughly fifteen, so this is a complete accounting for one worked example, not a complete functional decomposition of the toaster. The probe above checks a small model mirroring ApplyHeat's declaration, not nominal or slow themselves, which give ApplyHeat no concrete energy value.\n", + "Start, Finish and Cancel are declared as signals (their own doc states this) but are not wired as accepted or produced items of ApplyHeat in this chapter; whether they should be is left open for whichever chapter models the cycle's event handling. efficiency and a characterized conversion belong to a logical component this chapter does not build; this record makes no claim about them.\n" + ] + } + ], "source": [ "counterevidence = (\"bread is wired from ToastBread's own bread parameter, but energy \"\n", " \"is not: no energy source exists anywhere in the model yet, so leaving it \"\n", @@ -324,15 +659,35 @@ }, { "cell_type": "code", - "execution_count": null, + "execution_count": 12, "id": "cell-22", - "metadata": {}, - "outputs": [], + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T05:40:51.673809Z", + "iopub.status.busy": "2026-10-01T05:40:51.673735Z", + "iopub.status.idle": "2026-10-01T05:40:51.763804Z", + "shell.execute_reply": "2026-10-01T05:40:51.763487Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Model tag: {'tag': 'ToasterDemo::aiC04Tag', 'identifier': 'AI-C04', 'annotated_element': 'ToasterDemo::ApplyHeat'}\n", + "Validation errors: []\n", + "Record kind: asserted_inference\n" + ] + } + ], "source": [ + "from toaster.query import get_review_record_refs\n", + "\n", "inference_record = ReviewRecord(\n", " identifier=\"AI-C04\",\n", " kind=\"asserted_inference\",\n", " claim=claim,\n", + " subject_ref=subject_ref,\n", " model_ref=model_ref,\n", " content_hash=hash_content(source),\n", " scope=scope,\n", @@ -349,7 +704,9 @@ " record_kind=\"worked_example\",\n", ")\n", "\n", - "errors = validate_record(inference_record)\n", + "errors = validate_record(inference_record, model=model)\n", + "tag = next((t for t in get_review_record_refs(model) if t[\"identifier\"] == inference_record.identifier), None)\n", + "print(f\"Model tag: {tag}\")\n", "print(f\"Validation errors: {errors}\")\n", "print(f\"Record kind: {inference_record.kind}\")\n", "conn.close()\n" @@ -360,7 +717,7 @@ "id": "cell-23", "metadata": {}, "source": [ - "The Hawkins §3.1 schema fields named above are what the assembled record satisfies, and `validate_record` returning an empty list confirms the required fields, including a non-empty `premises`, are present, not merely printed." + "The tag printed above is now part of the loaded model, and `validate_record` confirms the Python record and the model's own tag agree about what AI-C04 is actually about. The Hawkins §3.1 schema fields named above are what the assembled record satisfies: an empty error list confirms the required fields, including a non-empty `premises`, are present, not merely printed." ] }, { @@ -379,7 +736,16 @@ "name": "python3" }, "language_info": { - "name": "python" + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.14.2" } }, "nbformat": 4, diff --git a/chapters/ch06-recursive-decomp/02-second-level.ipynb b/chapters/ch06-recursive-decomp/02-second-level.ipynb index 8d03321..c046a1b 100644 --- a/chapters/ch06-recursive-decomp/02-second-level.ipynb +++ b/chapters/ch06-recursive-decomp/02-second-level.ipynb @@ -24,10 +24,10 @@ "id": "cell-02", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T09:54:22.005032Z", - "iopub.status.busy": "2026-09-28T09:54:22.004761Z", - "iopub.status.idle": "2026-09-28T09:54:22.133442Z", - "shell.execute_reply": "2026-09-28T09:54:22.132934Z" + "iopub.execute_input": "2026-10-01T06:06:10.308828Z", + "iopub.status.busy": "2026-10-01T06:06:10.308587Z", + "iopub.status.idle": "2026-10-01T06:06:10.426606Z", + "shell.execute_reply": "2026-10-01T06:06:10.426047Z" } }, "outputs": [ @@ -73,10 +73,10 @@ "id": "cell-04", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T09:54:22.134976Z", - "iopub.status.busy": "2026-09-28T09:54:22.134798Z", - "iopub.status.idle": "2026-09-28T09:54:22.151822Z", - "shell.execute_reply": "2026-09-28T09:54:22.151360Z" + "iopub.execute_input": "2026-10-01T06:06:10.428649Z", + "iopub.status.busy": "2026-10-01T06:06:10.428439Z", + "iopub.status.idle": "2026-10-01T06:06:10.445570Z", + "shell.execute_reply": "2026-10-01T06:06:10.445123Z" } }, "outputs": [], @@ -91,19 +91,70 @@ "id": "cell-05", "metadata": {}, "source": [ - "The cumulative model, loaded here so the judgment records below can cite real analysis directly instead of the model's own declaration. The next cells record `AC-C06`, following the construction zone Hawkins' taxonomy uses (claim, frame, premises, evidence, challenge, assemble)." + "The cumulative model, loaded here so the judgment records below can cite real analysis directly instead of the model's own declaration. The next cells record `AC-C06`, following the construction zone Hawkins' taxonomy uses (anchor, claim, frame, premises, evidence, challenge, assemble)." + ] + }, + { + "cell_type": "markdown", + "id": "cell-06", + "metadata": {}, + "source": [ + "Before AC-C06 states its claim, the next cells give it a real anchor in the model: a `ReviewRecordRef` tag naming `heatGenerationReq` as what AC-C06 is about." ] }, { "cell_type": "code", "execution_count": 3, - "id": "cell-06", + "id": "cell-07", + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T06:06:10.447058Z", + "iopub.status.busy": "2026-10-01T06:06:10.446981Z", + "iopub.status.idle": "2026-10-01T06:06:10.449011Z", + "shell.execute_reply": "2026-10-01T06:06:10.448677Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "metadata acC06Tag : ReviewRecordRef about heatGenerationReq {\n", + " identifier = \"AC-C06\";\n", + "}\n", + "\n" + ] + } + ], + "source": [ + "# what is being claimed, and what, specifically, it is about\n", + "framing_subject_ref = \"ToasterDemo::heatGenerationReq\"\n", + "AC_C06_TAG = \"\"\"\\\n", + "metadata acC06Tag : ReviewRecordRef about heatGenerationReq {\n", + " identifier = \"AC-C06\";\n", + "}\n", + "\"\"\"\n", + "print(AC_C06_TAG)" + ] + }, + { + "cell_type": "markdown", + "id": "cell-08", + "metadata": {}, + "source": [ + "This fragment is the same text now committed in `models/ch06-cumulative.sysml` onward, alongside the `ReviewRecordRef` definition (carried forward since Chapter 2); loading the cumulative model (the `model.ok` cell above, unchanged) is what makes this a real, checkable tag rather than an assertion the reader has to trust. AC-C06 states the claim next: what `heatGenerationReq` is being framed as." + ] + }, + { + "cell_type": "code", + "execution_count": 4, + "id": "cell-09", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T09:54:22.153684Z", - "iopub.status.busy": "2026-09-28T09:54:22.153549Z", - "iopub.status.idle": "2026-09-28T09:54:22.155937Z", - "shell.execute_reply": "2026-09-28T09:54:22.155536Z" + "iopub.execute_input": "2026-10-01T06:06:10.450194Z", + "iopub.status.busy": "2026-10-01T06:06:10.450114Z", + "iopub.status.idle": "2026-10-01T06:06:10.452271Z", + "shell.execute_reply": "2026-10-01T06:06:10.451612Z" } }, "outputs": [ @@ -129,7 +180,7 @@ }, { "cell_type": "markdown", - "id": "cell-07", + "id": "cell-10", "metadata": {}, "source": [ "The standard this framing is checked against, the same one Chapter 3's `AC-C03` used for toast timing." @@ -137,14 +188,14 @@ }, { "cell_type": "code", - "execution_count": 4, - "id": "cell-08", + "execution_count": 5, + "id": "cell-11", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T09:54:22.157099Z", - "iopub.status.busy": "2026-09-28T09:54:22.157008Z", - "iopub.status.idle": "2026-09-28T09:54:22.158993Z", - "shell.execute_reply": "2026-09-28T09:54:22.158620Z" + "iopub.execute_input": "2026-10-01T06:06:10.453384Z", + "iopub.status.busy": "2026-10-01T06:06:10.453294Z", + "iopub.status.idle": "2026-10-01T06:06:10.455222Z", + "shell.execute_reply": "2026-10-01T06:06:10.454885Z" } }, "outputs": [ @@ -168,7 +219,7 @@ }, { "cell_type": "markdown", - "id": "cell-09", + "id": "cell-12", "metadata": {}, "source": [ "What the framing takes as given." @@ -176,14 +227,14 @@ }, { "cell_type": "code", - "execution_count": 5, - "id": "cell-10", + "execution_count": 6, + "id": "cell-13", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T09:54:22.160232Z", - "iopub.status.busy": "2026-09-28T09:54:22.160103Z", - "iopub.status.idle": "2026-09-28T09:54:22.162400Z", - "shell.execute_reply": "2026-09-28T09:54:22.161903Z" + "iopub.execute_input": "2026-10-01T06:06:10.456332Z", + "iopub.status.busy": "2026-10-01T06:06:10.456266Z", + "iopub.status.idle": "2026-10-01T06:06:10.458004Z", + "shell.execute_reply": "2026-10-01T06:06:10.457680Z" } }, "outputs": [ @@ -206,7 +257,7 @@ }, { "cell_type": "markdown", - "id": "cell-11", + "id": "cell-14", "metadata": {}, "source": [ "What supports the claim, and the argument connecting it." @@ -214,14 +265,14 @@ }, { "cell_type": "code", - "execution_count": 6, - "id": "cell-12", + "execution_count": 7, + "id": "cell-15", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T09:54:22.163800Z", - "iopub.status.busy": "2026-09-28T09:54:22.163694Z", - "iopub.status.idle": "2026-09-28T09:54:22.165877Z", - "shell.execute_reply": "2026-09-28T09:54:22.165535Z" + "iopub.execute_input": "2026-10-01T06:06:10.459072Z", + "iopub.status.busy": "2026-10-01T06:06:10.458997Z", + "iopub.status.idle": "2026-10-01T06:06:10.461197Z", + "shell.execute_reply": "2026-10-01T06:06:10.460797Z" } }, "outputs": [ @@ -251,7 +302,7 @@ }, { "cell_type": "markdown", - "id": "cell-13", + "id": "cell-16", "metadata": {}, "source": [ "The challenge: what stays open about this framing, and about the threshold itself." @@ -259,14 +310,14 @@ }, { "cell_type": "code", - "execution_count": 7, - "id": "cell-14", + "execution_count": 8, + "id": "cell-17", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T09:54:22.167496Z", - "iopub.status.busy": "2026-09-28T09:54:22.167386Z", - "iopub.status.idle": "2026-09-28T09:54:22.169785Z", - "shell.execute_reply": "2026-09-28T09:54:22.169317Z" + "iopub.execute_input": "2026-10-01T06:06:10.462340Z", + "iopub.status.busy": "2026-10-01T06:06:10.462269Z", + "iopub.status.idle": "2026-10-01T06:06:10.464447Z", + "shell.execute_reply": "2026-10-01T06:06:10.464124Z" } }, "outputs": [ @@ -299,7 +350,7 @@ }, { "cell_type": "markdown", - "id": "cell-15", + "id": "cell-18", "metadata": {}, "source": [ "With every part named above, the framing record assembles from them directly." @@ -307,14 +358,14 @@ }, { "cell_type": "code", - "execution_count": 8, - "id": "cell-16", + "execution_count": 9, + "id": "cell-19", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T09:54:22.171003Z", - "iopub.status.busy": "2026-09-28T09:54:22.170912Z", - "iopub.status.idle": "2026-09-28T09:54:22.173284Z", - "shell.execute_reply": "2026-09-28T09:54:22.172958Z" + "iopub.execute_input": "2026-10-01T06:06:10.465538Z", + "iopub.status.busy": "2026-10-01T06:06:10.465470Z", + "iopub.status.idle": "2026-10-01T06:06:10.598524Z", + "shell.execute_reply": "2026-10-01T06:06:10.598148Z" } }, "outputs": [ @@ -322,15 +373,19 @@ "name": "stdout", "output_type": "stream", "text": [ + "Model tag: {'tag': 'ToasterDemo::acC06Tag', 'identifier': 'AC-C06', 'annotated_element': 'ToasterDemo::heatGenerationReq'}\n", "Validation errors: []\n" ] } ], "source": [ + "from toaster.query import get_review_record_refs\n", + "\n", "framing_record = ReviewRecord(\n", " identifier=\"AC-C06\",\n", " kind=\"asserted_context\",\n", " claim=framing_claim,\n", + " subject_ref=framing_subject_ref,\n", " model_ref=framing_model_ref,\n", " content_hash=hash_content(source),\n", " scope=framing_scope,\n", @@ -347,28 +402,30 @@ " record_kind=\"worked_example\",\n", ")\n", "\n", - "errors = validate_record(framing_record)\n", + "errors = validate_record(framing_record, model=model)\n", + "tag = next((t for t in get_review_record_refs(model) if t[\"identifier\"] == framing_record.identifier), None)\n", + "print(f\"Model tag: {tag}\")\n", "print(f\"Validation errors: {errors}\")" ] }, { "cell_type": "markdown", - "id": "cell-17", + "id": "cell-20", "metadata": {}, "source": [ - "`validate_record` reports no errors. With the requirement framed, the next cells record which mechanism is chosen to satisfy it, and why. `GenerateHeat` and `HeatGenerator` commit to no energy form or mechanism (notebook 01): the selection below is what actually chooses one, not a fact already built into either of them." + "The tag printed above is now part of the loaded model, and `validate_record` confirms the Python record and the model's own tag agree about what AC-C06 is actually about, with no errors reported. With the requirement framed, the next cells record which mechanism is chosen to satisfy it, and why. `GenerateHeat` and `HeatGenerator` commit to no energy form or mechanism (notebook 01): the selection below is what actually chooses one, not a fact already built into either of them." ] }, { "cell_type": "code", - "execution_count": 9, - "id": "cell-18", + "execution_count": 10, + "id": "cell-21", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T09:54:22.174552Z", - "iopub.status.busy": "2026-09-28T09:54:22.174459Z", - "iopub.status.idle": "2026-09-28T09:54:22.178528Z", - "shell.execute_reply": "2026-09-28T09:54:22.178158Z" + "iopub.execute_input": "2026-10-01T06:06:10.599773Z", + "iopub.status.busy": "2026-10-01T06:06:10.599696Z", + "iopub.status.idle": "2026-10-01T06:06:10.603455Z", + "shell.execute_reply": "2026-10-01T06:06:10.602991Z" } }, "outputs": [ @@ -388,22 +445,73 @@ }, { "cell_type": "markdown", - "id": "cell-19", + "id": "cell-22", "metadata": {}, "source": [ "`ControlSystem` really does declare `durationOut`, a discrete duration signal (Chapter 5), confirmed directly rather than assumed. The selection below cites this fact alongside a domain premise about how the two mechanisms work; it does not argue from `energyIn` already being electrical, since it is not, until this record commits it." ] }, + { + "cell_type": "markdown", + "id": "cell-23", + "metadata": {}, + "source": [ + "Before AS-C06 states its claim, the next cells give it a real anchor in the model: a `ReviewRecordRef` tag naming `ResistanceCoil` as what AS-C06 is about." + ] + }, { "cell_type": "code", - "execution_count": 10, - "id": "cell-20", + "execution_count": 11, + "id": "cell-24", + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T06:06:10.604639Z", + "iopub.status.busy": "2026-10-01T06:06:10.604560Z", + "iopub.status.idle": "2026-10-01T06:06:10.606503Z", + "shell.execute_reply": "2026-10-01T06:06:10.606184Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "metadata asC06Tag : ReviewRecordRef about ResistanceCoil {\n", + " identifier = \"AS-C06\";\n", + "}\n", + "\n" + ] + } + ], + "source": [ + "# what is being claimed, and what, specifically, it is about\n", + "selection_subject_ref = \"ToasterDemo::ResistanceCoil\"\n", + "AS_C06_TAG = \"\"\"\\\n", + "metadata asC06Tag : ReviewRecordRef about ResistanceCoil {\n", + " identifier = \"AS-C06\";\n", + "}\n", + "\"\"\"\n", + "print(AS_C06_TAG)" + ] + }, + { + "cell_type": "markdown", + "id": "cell-25", + "metadata": {}, + "source": [ + "This fragment is the same text now committed in `models/ch06-cumulative.sysml` onward, alongside the `ReviewRecordRef` definition and `acC06Tag` (above); loading the cumulative model (the `model.ok` cell above, unchanged) is what makes this a real, checkable tag rather than an assertion the reader has to trust. AS-C06 states the claim next: which mechanism is selected, and over what alternative." + ] + }, + { + "cell_type": "code", + "execution_count": 12, + "id": "cell-26", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T09:54:22.180047Z", - "iopub.status.busy": "2026-09-28T09:54:22.179960Z", - "iopub.status.idle": "2026-09-28T09:54:22.181962Z", - "shell.execute_reply": "2026-09-28T09:54:22.181621Z" + "iopub.execute_input": "2026-10-01T06:06:10.607586Z", + "iopub.status.busy": "2026-10-01T06:06:10.607521Z", + "iopub.status.idle": "2026-10-01T06:06:10.609108Z", + "shell.execute_reply": "2026-10-01T06:06:10.608795Z" } }, "outputs": [ @@ -428,7 +536,7 @@ }, { "cell_type": "markdown", - "id": "cell-21", + "id": "cell-27", "metadata": {}, "source": [ "The standard the selection is checked against: what the model already provides that a chosen mechanism must pair with, and what it must expose to be checked." @@ -436,14 +544,14 @@ }, { "cell_type": "code", - "execution_count": 11, - "id": "cell-22", + "execution_count": 13, + "id": "cell-28", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T09:54:22.183390Z", - "iopub.status.busy": "2026-09-28T09:54:22.183305Z", - "iopub.status.idle": "2026-09-28T09:54:22.185380Z", - "shell.execute_reply": "2026-09-28T09:54:22.184973Z" + "iopub.execute_input": "2026-10-01T06:06:10.610519Z", + "iopub.status.busy": "2026-10-01T06:06:10.610421Z", + "iopub.status.idle": "2026-10-01T06:06:10.612340Z", + "shell.execute_reply": "2026-10-01T06:06:10.612010Z" } }, "outputs": [ @@ -468,7 +576,7 @@ }, { "cell_type": "markdown", - "id": "cell-23", + "id": "cell-29", "metadata": {}, "source": [ "What the selection takes as given: the confirmed fact above, plus a domain premise about how the two mechanisms work, stated as two premises below." @@ -476,14 +584,14 @@ }, { "cell_type": "code", - "execution_count": 12, - "id": "cell-24", + "execution_count": 14, + "id": "cell-30", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T09:54:22.186752Z", - "iopub.status.busy": "2026-09-28T09:54:22.186654Z", - "iopub.status.idle": "2026-09-28T09:54:22.189102Z", - "shell.execute_reply": "2026-09-28T09:54:22.188593Z" + "iopub.execute_input": "2026-10-01T06:06:10.613469Z", + "iopub.status.busy": "2026-10-01T06:06:10.613396Z", + "iopub.status.idle": "2026-10-01T06:06:10.615394Z", + "shell.execute_reply": "2026-10-01T06:06:10.615123Z" } }, "outputs": [ @@ -517,7 +625,7 @@ }, { "cell_type": "markdown", - "id": "cell-25", + "id": "cell-31", "metadata": {}, "source": [ "The evidence, and the argument connecting it to the claim." @@ -525,14 +633,14 @@ }, { "cell_type": "code", - "execution_count": 13, - "id": "cell-26", + "execution_count": 15, + "id": "cell-32", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T09:54:22.190254Z", - "iopub.status.busy": "2026-09-28T09:54:22.190180Z", - "iopub.status.idle": "2026-09-28T09:54:22.192231Z", - "shell.execute_reply": "2026-09-28T09:54:22.191822Z" + "iopub.execute_input": "2026-10-01T06:06:10.616595Z", + "iopub.status.busy": "2026-10-01T06:06:10.616519Z", + "iopub.status.idle": "2026-10-01T06:06:10.618717Z", + "shell.execute_reply": "2026-10-01T06:06:10.618182Z" } }, "outputs": [ @@ -566,7 +674,7 @@ }, { "cell_type": "markdown", - "id": "cell-27", + "id": "cell-33", "metadata": {}, "source": [ "The challenge: what this selection does not establish, stated plainly." @@ -574,14 +682,14 @@ }, { "cell_type": "code", - "execution_count": 14, - "id": "cell-28", + "execution_count": 16, + "id": "cell-34", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T09:54:22.193449Z", - "iopub.status.busy": "2026-09-28T09:54:22.193369Z", - "iopub.status.idle": "2026-09-28T09:54:22.195525Z", - "shell.execute_reply": "2026-09-28T09:54:22.195081Z" + "iopub.execute_input": "2026-10-01T06:06:10.619883Z", + "iopub.status.busy": "2026-10-01T06:06:10.619788Z", + "iopub.status.idle": "2026-10-01T06:06:10.621929Z", + "shell.execute_reply": "2026-10-01T06:06:10.621550Z" } }, "outputs": [ @@ -616,7 +724,7 @@ }, { "cell_type": "markdown", - "id": "cell-29", + "id": "cell-35", "metadata": {}, "source": [ "With every part named above, the selection record assembles from them directly." @@ -624,14 +732,14 @@ }, { "cell_type": "code", - "execution_count": 15, - "id": "cell-30", + "execution_count": 17, + "id": "cell-36", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T09:54:22.197051Z", - "iopub.status.busy": "2026-09-28T09:54:22.196964Z", - "iopub.status.idle": "2026-09-28T09:54:22.199270Z", - "shell.execute_reply": "2026-09-28T09:54:22.198897Z" + "iopub.execute_input": "2026-10-01T06:06:10.623087Z", + "iopub.status.busy": "2026-10-01T06:06:10.623003Z", + "iopub.status.idle": "2026-10-01T06:06:10.718179Z", + "shell.execute_reply": "2026-10-01T06:06:10.717741Z" } }, "outputs": [ @@ -639,15 +747,19 @@ "name": "stdout", "output_type": "stream", "text": [ + "Model tag: {'tag': 'ToasterDemo::asC06Tag', 'identifier': 'AS-C06', 'annotated_element': 'ToasterDemo::ResistanceCoil'}\n", "Validation errors: []\n" ] } ], "source": [ + "from toaster.query import get_review_record_refs\n", + "\n", "selection_record = ReviewRecord(\n", " identifier=\"AS-C06\",\n", " kind=\"asserted_solution\",\n", " claim=selection_claim,\n", + " subject_ref=selection_subject_ref,\n", " model_ref=selection_model_ref,\n", " content_hash=hash_content(source),\n", " scope=selection_scope,\n", @@ -664,28 +776,30 @@ " record_kind=\"worked_example\",\n", ")\n", "\n", - "errors = validate_record(selection_record)\n", + "errors = validate_record(selection_record, model=model)\n", + "tag = next((t for t in get_review_record_refs(model) if t[\"identifier\"] == selection_record.identifier), None)\n", + "print(f\"Model tag: {tag}\")\n", "print(f\"Validation errors: {errors}\")" ] }, { "cell_type": "markdown", - "id": "cell-31", + "id": "cell-37", "metadata": {}, "source": [ - "`validate_record` reports no errors. With the selection on record, `ResistanceCoil`'s electrically-specific name and Joule-heating doc are now admissible: the mechanism they name has a real, recorded argument behind it, not a name chosen and justified afterward." + "The tag printed above is now part of the loaded model, and `validate_record` confirms the Python record and the model's own tag agree about what AS-C06 is actually about, with no errors reported. With the selection on record, `ResistanceCoil`'s electrically-specific name and Joule-heating doc are now admissible: the mechanism they name has a real, recorded argument behind it, not a name chosen and justified afterward." ] }, { "cell_type": "code", - "execution_count": 16, - "id": "cell-32", + "execution_count": 18, + "id": "cell-38", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T09:54:22.200657Z", - "iopub.status.busy": "2026-09-28T09:54:22.200565Z", - "iopub.status.idle": "2026-09-28T09:54:22.202464Z", - "shell.execute_reply": "2026-09-28T09:54:22.202062Z" + "iopub.execute_input": "2026-10-01T06:06:10.719813Z", + "iopub.status.busy": "2026-10-01T06:06:10.719701Z", + "iopub.status.idle": "2026-10-01T06:06:10.721663Z", + "shell.execute_reply": "2026-10-01T06:06:10.721377Z" } }, "outputs": [ @@ -711,7 +825,7 @@ }, { "cell_type": "markdown", - "id": "cell-33", + "id": "cell-39", "metadata": {}, "source": [ "`ResistanceCoil` specializes `HeatGenerator` and binds its power slot to a default, redefined with `default =` so a candidate can still override it. `resistance` is a physical sizing value with a real unit, `ISQ::ResistanceValue` in ohms, not the bare number the mechanism-suggestive name alone would need." @@ -719,14 +833,14 @@ }, { "cell_type": "code", - "execution_count": 17, - "id": "cell-34", + "execution_count": 19, + "id": "cell-40", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T09:54:22.203705Z", - "iopub.status.busy": "2026-09-28T09:54:22.203628Z", - "iopub.status.idle": "2026-09-28T09:54:22.205731Z", - "shell.execute_reply": "2026-09-28T09:54:22.205247Z" + "iopub.execute_input": "2026-10-01T06:06:10.722890Z", + "iopub.status.busy": "2026-10-01T06:06:10.722811Z", + "iopub.status.idle": "2026-10-01T06:06:10.724419Z", + "shell.execute_reply": "2026-10-01T06:06:10.724162Z" } }, "outputs": [ @@ -750,7 +864,7 @@ }, { "cell_type": "markdown", - "id": "cell-35", + "id": "cell-41", "metadata": {}, "source": [ "`rated` takes `ResistanceCoil`'s default power, 800 W, and asserts it satisfies `heatGenerationReq` directly: a real, evaluable claim about a real candidate." @@ -758,14 +872,14 @@ }, { "cell_type": "code", - "execution_count": 18, - "id": "cell-36", + "execution_count": 20, + "id": "cell-42", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T09:54:22.206995Z", - "iopub.status.busy": "2026-09-28T09:54:22.206923Z", - "iopub.status.idle": "2026-09-28T09:54:22.208958Z", - "shell.execute_reply": "2026-09-28T09:54:22.208554Z" + "iopub.execute_input": "2026-10-01T06:06:10.725502Z", + "iopub.status.busy": "2026-10-01T06:06:10.725438Z", + "iopub.status.idle": "2026-10-01T06:06:10.727371Z", + "shell.execute_reply": "2026-10-01T06:06:10.727013Z" } }, "outputs": [ @@ -791,7 +905,7 @@ }, { "cell_type": "markdown", - "id": "cell-37", + "id": "cell-43", "metadata": {}, "source": [ "`weak` overrides power down to 400 W, below the threshold, and folds the claim into its own context as a negated assertion: `assert not satisfy`, not a false positive. Its failure is a design choice, a 400 W part rated below what the requirement asks for, the same class of legitimate failing branch as a chosen part's rating anywhere else in this model." @@ -799,14 +913,14 @@ }, { "cell_type": "code", - "execution_count": 19, - "id": "cell-38", + "execution_count": 21, + "id": "cell-44", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T09:54:22.210341Z", - "iopub.status.busy": "2026-09-28T09:54:22.210237Z", - "iopub.status.idle": "2026-09-28T09:54:22.213394Z", - "shell.execute_reply": "2026-09-28T09:54:22.212974Z" + "iopub.execute_input": "2026-10-01T06:06:10.728553Z", + "iopub.status.busy": "2026-10-01T06:06:10.728486Z", + "iopub.status.idle": "2026-10-01T06:06:10.731148Z", + "shell.execute_reply": "2026-10-01T06:06:10.730822Z" } }, "outputs": [ @@ -819,10 +933,18 @@ " require constraint { heatGen.power >= 600.0 [SI::W] }\n", "}\n", "requirement heatGenerationReq : HeatGenerationReq;\n", + "metadata acC06Tag : ReviewRecordRef about heatGenerationReq {\n", + " identifier = \"AC-C06\";\n", + "}\n", + "\n", "part def ResistanceCoil :> HeatGenerator {\n", " attribute :>> power default = 800.0 [SI::W];\n", " attribute resistance : ISQ::ResistanceValue default = 12.0 [SI::ohm];\n", "}\n", + "metadata asC06Tag : ReviewRecordRef about ResistanceCoil {\n", + " identifier = \"AS-C06\";\n", + "}\n", + "\n", "part rated : ResistanceCoil {\n", " assert satisfy heatGenerationReq by rated;\n", "}\n", @@ -835,7 +957,7 @@ ], "source": [ "TOASTER_INCREMENT = (\n", - " f\"{HEAT_GENERATION_REQ_DEF}\\n{RESISTANCE_COIL_DEF}\\n\"\n", + " f\"{HEAT_GENERATION_REQ_DEF}\\n{AC_C06_TAG}\\n{RESISTANCE_COIL_DEF}\\n{AS_C06_TAG}\\n\"\n", " f\"{RATED_USAGE}\\n{WEAK_USAGE}\"\n", ")\n", "print(TOASTER_INCREMENT)\n", @@ -846,7 +968,7 @@ }, { "cell_type": "markdown", - "id": "cell-39", + "id": "cell-45", "metadata": {}, "source": [ "A requirement's subject must resolve to a declared type. The negative control below types the subject by a def that was never declared." @@ -854,14 +976,14 @@ }, { "cell_type": "code", - "execution_count": 20, - "id": "cell-40", + "execution_count": 22, + "id": "cell-46", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T09:54:22.214553Z", - "iopub.status.busy": "2026-09-28T09:54:22.214464Z", - "iopub.status.idle": "2026-09-28T09:54:22.239368Z", - "shell.execute_reply": "2026-09-28T09:54:22.239039Z" + "iopub.execute_input": "2026-10-01T06:06:10.732237Z", + "iopub.status.busy": "2026-10-01T06:06:10.732159Z", + "iopub.status.idle": "2026-10-01T06:06:10.754927Z", + "shell.execute_reply": "2026-10-01T06:06:10.754580Z" } }, "outputs": [ @@ -892,7 +1014,7 @@ }, { "cell_type": "markdown", - "id": "cell-41", + "id": "cell-47", "metadata": {}, "source": [ "The diagnostic reports an unresolved reference: `UndefinedCarrier` names no declared type, so the subject cannot be typed." @@ -900,14 +1022,14 @@ }, { "cell_type": "code", - "execution_count": 21, - "id": "cell-42", + "execution_count": 23, + "id": "cell-48", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T09:54:22.240736Z", - "iopub.status.busy": "2026-09-28T09:54:22.240661Z", - "iopub.status.idle": "2026-09-28T09:54:22.254968Z", - "shell.execute_reply": "2026-09-28T09:54:22.254100Z" + "iopub.execute_input": "2026-10-01T06:06:10.756136Z", + "iopub.status.busy": "2026-10-01T06:06:10.756061Z", + "iopub.status.idle": "2026-10-01T06:06:10.768517Z", + "shell.execute_reply": "2026-10-01T06:06:10.768151Z" } }, "outputs": [ @@ -932,7 +1054,7 @@ }, { "cell_type": "markdown", - "id": "cell-43", + "id": "cell-49", "metadata": {}, "source": [ "`rated` evaluates True against the threshold; `weak` evaluates False, confirming the assertion folded into its own context above is the correct one to make." @@ -940,7 +1062,7 @@ }, { "cell_type": "markdown", - "id": "cell-44", + "id": "cell-50", "metadata": {}, "source": [ "The requirement and the candidates printed above loaded without error, the selection and framing records validated with no errors, and the evaluated results confirm what each candidate's own assertion claims." @@ -948,7 +1070,7 @@ }, { "cell_type": "markdown", - "id": "cell-45", + "id": "cell-51", "metadata": {}, "source": [ "Try the chapter exercise in `exercises/ch06/exercise.ipynb`: nest `MoveWater` inside `ApplyWater`, build `WaterMover` as its abstract carrier, and add a `BrewReq` requirement with its subject on `WaterMover` itself (not `BrewUnit`) constraining minimum throughput, then create a lower-throughput candidate that fails it, following the requirement and candidate pattern this notebook builds." diff --git a/chapters/ch06-recursive-decomp/03-stopping-judgment.ipynb b/chapters/ch06-recursive-decomp/03-stopping-judgment.ipynb index b471f48..59858b5 100644 --- a/chapters/ch06-recursive-decomp/03-stopping-judgment.ipynb +++ b/chapters/ch06-recursive-decomp/03-stopping-judgment.ipynb @@ -24,10 +24,10 @@ "id": "cell-02", "metadata": { "execution": { - "iopub.execute_input": "2026-09-29T20:48:36.818145Z", - "iopub.status.busy": "2026-09-29T20:48:36.817880Z", - "iopub.status.idle": "2026-09-29T20:48:37.090476Z", - "shell.execute_reply": "2026-09-29T20:48:37.090007Z" + "iopub.execute_input": "2026-10-01T06:06:12.416566Z", + "iopub.status.busy": "2026-10-01T06:06:12.416356Z", + "iopub.status.idle": "2026-10-01T06:06:12.554132Z", + "shell.execute_reply": "2026-10-01T06:06:12.553624Z" } }, "outputs": [ @@ -40,6 +40,10 @@ "// Source: notebook cell-02 TOASTER_INCREMENT in chapter 6's construct-introducing notebooks.\n", "\n", "package ToasterDemo {\n", + " metadata def ReviewRecordRef {\n", + " attribute identifier : ScalarValues::String;\n", + " }\n", + "\n", " private import ScalarValues::*;\n", " private import SI::*;\n", " private import ISQ::*;\n", @@ -70,6 +74,10 @@ " then done;\n", " }\n", "\n", + " metadata aiC04Tag : ReviewRecordRef about ApplyHeat {\n", + " identifier = \"AI-C04\";\n", + " }\n", + "\n", " action def ToastBread {\n", " doc /* Transform bread into toast acceptable to its user. */\n", " in bread : Bread;\n", @@ -121,7 +129,19 @@ "\n", " requirement timely : TimelyToast;\n", "\n", + " metadata acC03Tag : ReviewRecordRef about timely {\n", + " identifier = \"AC-C03\";\n", + " }\n", + "\n", + " metadata asC03Tag : ReviewRecordRef about timely {\n", + " identifier = \"AS-C03\";\n", + " }\n", + "\n", " part nominal : Toaster;\n", + " metadata ac001Tag : ReviewRecordRef about nominal {\n", + " identifier = \"AC-001\";\n", + " }\n", + "\n", " part slow : Toaster {\n", " attribute :>> cycleTime = 200.0 [SI::s];\n", " assert not satisfy timely by slow;\n", @@ -184,6 +204,10 @@ " allocation heatGenAllocation allocate applyHeat.generateHeat to heatGen;\n", " }\n", "\n", + " metadata aiC06Tag : ReviewRecordRef about HeatingAssembly::heatGen {\n", + " identifier = \"AI-C06\";\n", + " }\n", + "\n", " requirement def HeatGenerationReq {\n", " doc /*\n", " * A heat generator shall be rated for at least 600 W.\n", @@ -198,6 +222,10 @@ "\n", " requirement heatGenerationReq : HeatGenerationReq;\n", "\n", + " metadata acC06Tag : ReviewRecordRef about heatGenerationReq {\n", + " identifier = \"AC-C06\";\n", + " }\n", + "\n", " part def ResistanceCoil :> HeatGenerator {\n", " doc /* An electrically switched resistive element: converts electrical\n", " * energy to heat by Joule heating. The mechanism selection this\n", @@ -208,6 +236,10 @@ " attribute resistance : ISQ::ResistanceValue default = 12.0 [SI::ohm];\n", " }\n", "\n", + " metadata asC06Tag : ReviewRecordRef about ResistanceCoil {\n", + " identifier = \"AS-C06\";\n", + " }\n", + "\n", " part rated : ResistanceCoil {\n", " assert satisfy heatGenerationReq by rated;\n", " }\n", @@ -248,10 +280,10 @@ "id": "cell-04", "metadata": { "execution": { - "iopub.execute_input": "2026-09-29T20:48:37.092256Z", - "iopub.status.busy": "2026-09-29T20:48:37.091968Z", - "iopub.status.idle": "2026-09-29T20:48:37.095095Z", - "shell.execute_reply": "2026-09-29T20:48:37.094730Z" + "iopub.execute_input": "2026-10-01T06:06:12.555737Z", + "iopub.status.busy": "2026-10-01T06:06:12.555546Z", + "iopub.status.idle": "2026-10-01T06:06:12.638826Z", + "shell.execute_reply": "2026-10-01T06:06:12.638438Z" } }, "outputs": [ @@ -271,6 +303,7 @@ " identifier=\"AI-BAD\",\n", " kind=\"asserted_inference\",\n", " claim=\"The heat-generation branch is complete.\",\n", + " subject_ref=\"ToasterDemo::HeatingAssembly::heatGen\",\n", " model_ref=\"ToasterDemo::HeatingAssembly::heatGen\",\n", " content_hash=hash_content(source),\n", " scope=\"ToasterDemo\",\n", @@ -286,8 +319,9 @@ " engineering_conclusion=\"undetermined\",\n", " record_kind=\"worked_example\",\n", ")\n", - "errors = validate_record(incomplete)\n", - "assert len(errors) > 0, \"Expected validation to fail on empty premises\"\n", + "errors = validate_record(incomplete, model=model)\n", + "expected = [\"asserted_inference requires at least one premise (Hawkins §3.1)\"]\n", + "assert errors == expected, f\"Expected exactly one premises error, got {errors}\"\n", "print(f\"Validation errors for empty-premises record: {errors}\")" ] }, @@ -305,10 +339,10 @@ "id": "cell-06", "metadata": { "execution": { - "iopub.execute_input": "2026-09-29T20:48:37.096324Z", - "iopub.status.busy": "2026-09-29T20:48:37.096226Z", - "iopub.status.idle": "2026-09-29T20:48:37.215208Z", - "shell.execute_reply": "2026-09-29T20:48:37.214773Z" + "iopub.execute_input": "2026-10-01T06:06:12.640055Z", + "iopub.status.busy": "2026-10-01T06:06:12.639978Z", + "iopub.status.idle": "2026-10-01T06:06:12.737530Z", + "shell.execute_reply": "2026-10-01T06:06:12.737119Z" } }, "outputs": [ @@ -340,16 +374,104 @@ "`HeatGenerator` performs `GenerateHeat`, `heatGenAllocation` points from `applyHeat.generateHeat` to `heatGen`, declared inside `HeatingAssembly` where both resolve, and the requirement evaluates True on `rated` and False on `weak`, the deliberately failing candidate. The next cells build `AI-C06` directly from these results, following the construction zone Hawkins' taxonomy uses." ] }, + { + "cell_type": "markdown", + "id": "cell-08", + "metadata": {}, + "source": [ + "Before AI-C06 states its claim, the next cells give it a real anchor in the model: a `ReviewRecordRef` tag naming `HeatingAssembly::heatGen` as what AI-C06 is about." + ] + }, { "cell_type": "code", "execution_count": 4, - "id": "cell-08", + "id": "cell-09", + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T06:06:12.738876Z", + "iopub.status.busy": "2026-10-01T06:06:12.738785Z", + "iopub.status.idle": "2026-10-01T06:06:12.740498Z", + "shell.execute_reply": "2026-10-01T06:06:12.740173Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "metadata aiC06Tag : ReviewRecordRef about HeatingAssembly::heatGen {\n", + " identifier = \"AI-C06\";\n", + "}\n", + "\n" + ] + } + ], + "source": [ + "# what is being claimed, and what, specifically, it is about\n", + "subject_ref = \"ToasterDemo::HeatingAssembly::heatGen\"\n", + "AI_C06_TAG = \"\"\"\\\n", + "metadata aiC06Tag : ReviewRecordRef about HeatingAssembly::heatGen {\n", + " identifier = \"AI-C06\";\n", + "}\n", + "\"\"\"\n", + "print(AI_C06_TAG)" + ] + }, + { + "cell_type": "markdown", + "id": "cell-10", + "metadata": {}, + "source": [ + "This fragment is the same text now committed in `models/ch06-cumulative.sysml` onward, alongside the `ReviewRecordRef` definition (carried forward since Chapter 2) and the `acC06Tag`/`asC06Tag` tags (notebook 02); loading the cumulative model (the `model.ok` cell above, unchanged) is what makes this a real, checkable tag rather than an assertion the reader has to trust." + ] + }, + { + "cell_type": "code", + "execution_count": 5, + "id": "cell-11", "metadata": { "execution": { - "iopub.execute_input": "2026-09-29T20:48:37.216734Z", - "iopub.status.busy": "2026-09-29T20:48:37.216595Z", - "iopub.status.idle": "2026-09-29T20:48:37.218741Z", - "shell.execute_reply": "2026-09-29T20:48:37.218420Z" + "iopub.execute_input": "2026-10-01T06:06:12.741609Z", + "iopub.status.busy": "2026-10-01T06:06:12.741531Z", + "iopub.status.idle": "2026-10-01T06:06:12.743495Z", + "shell.execute_reply": "2026-10-01T06:06:12.743114Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "metadata aiC06Tag : ReviewRecordRef about HeatingAssembly::heatGen {\n", + " identifier = \"AI-C06\";\n", + "}\n", + "\n" + ] + } + ], + "source": [ + "TOASTER_INCREMENT = AI_C06_TAG\n", + "print(TOASTER_INCREMENT)" + ] + }, + { + "cell_type": "markdown", + "id": "b55488b6", + "metadata": {}, + "source": [ + "AI-C06 states the claim next: what the stopping rule actually shows for `HeatingAssembly::heatGen`, condition by condition." + ] + }, + { + "cell_type": "code", + "execution_count": 6, + "id": "cell-12", + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T06:06:12.744939Z", + "iopub.status.busy": "2026-10-01T06:06:12.744845Z", + "iopub.status.idle": "2026-10-01T06:06:12.746959Z", + "shell.execute_reply": "2026-10-01T06:06:12.746624Z" } }, "outputs": [ @@ -376,7 +498,7 @@ }, { "cell_type": "markdown", - "id": "cell-09", + "id": "cell-13", "metadata": {}, "source": [ "The standard this claim is checked against: the recursion's own stopping rule, applied one level down." @@ -384,14 +506,14 @@ }, { "cell_type": "code", - "execution_count": 5, - "id": "cell-10", + "execution_count": 7, + "id": "cell-14", "metadata": { "execution": { - "iopub.execute_input": "2026-09-29T20:48:37.219970Z", - "iopub.status.busy": "2026-09-29T20:48:37.219881Z", - "iopub.status.idle": "2026-09-29T20:48:37.221863Z", - "shell.execute_reply": "2026-09-29T20:48:37.221560Z" + "iopub.execute_input": "2026-10-01T06:06:12.748108Z", + "iopub.status.busy": "2026-10-01T06:06:12.748023Z", + "iopub.status.idle": "2026-10-01T06:06:12.749867Z", + "shell.execute_reply": "2026-10-01T06:06:12.749576Z" } }, "outputs": [ @@ -420,7 +542,7 @@ }, { "cell_type": "markdown", - "id": "cell-11", + "id": "cell-15", "metadata": {}, "source": [ "What the claim takes as given: the two judgments this chapter already recorded, and the two from earlier chapters this branch still rests on." @@ -428,14 +550,14 @@ }, { "cell_type": "code", - "execution_count": 6, - "id": "cell-12", + "execution_count": 8, + "id": "cell-16", "metadata": { "execution": { - "iopub.execute_input": "2026-09-29T20:48:37.222939Z", - "iopub.status.busy": "2026-09-29T20:48:37.222876Z", - "iopub.status.idle": "2026-09-29T20:48:37.225022Z", - "shell.execute_reply": "2026-09-29T20:48:37.224559Z" + "iopub.execute_input": "2026-10-01T06:06:12.750890Z", + "iopub.status.busy": "2026-10-01T06:06:12.750820Z", + "iopub.status.idle": "2026-10-01T06:06:12.752948Z", + "shell.execute_reply": "2026-10-01T06:06:12.752507Z" } }, "outputs": [ @@ -460,7 +582,7 @@ }, { "cell_type": "markdown", - "id": "cell-13", + "id": "cell-17", "metadata": {}, "source": [ "`evidence_refs` points at the three results gathered above; `rationale` connects them to the claim, condition by condition." @@ -468,14 +590,14 @@ }, { "cell_type": "code", - "execution_count": 7, - "id": "cell-14", + "execution_count": 9, + "id": "cell-18", "metadata": { "execution": { - "iopub.execute_input": "2026-09-29T20:48:37.226242Z", - "iopub.status.busy": "2026-09-29T20:48:37.226148Z", - "iopub.status.idle": "2026-09-29T20:48:37.228430Z", - "shell.execute_reply": "2026-09-29T20:48:37.228084Z" + "iopub.execute_input": "2026-10-01T06:06:12.754030Z", + "iopub.status.busy": "2026-10-01T06:06:12.753958Z", + "iopub.status.idle": "2026-10-01T06:06:12.756310Z", + "shell.execute_reply": "2026-10-01T06:06:12.755949Z" } }, "outputs": [ @@ -513,7 +635,7 @@ }, { "cell_type": "markdown", - "id": "cell-15", + "id": "cell-19", "metadata": {}, "source": [ "The challenge: what this branch does not yet establish, stated plainly rather than folded into a premature completeness claim." @@ -521,14 +643,14 @@ }, { "cell_type": "code", - "execution_count": 8, - "id": "cell-16", + "execution_count": 10, + "id": "cell-20", "metadata": { "execution": { - "iopub.execute_input": "2026-09-29T20:48:37.229537Z", - "iopub.status.busy": "2026-09-29T20:48:37.229456Z", - "iopub.status.idle": "2026-09-29T20:48:37.231652Z", - "shell.execute_reply": "2026-09-29T20:48:37.231328Z" + "iopub.execute_input": "2026-10-01T06:06:12.757393Z", + "iopub.status.busy": "2026-10-01T06:06:12.757327Z", + "iopub.status.idle": "2026-10-01T06:06:12.759719Z", + "shell.execute_reply": "2026-10-01T06:06:12.759297Z" } }, "outputs": [ @@ -574,7 +696,7 @@ }, { "cell_type": "markdown", - "id": "cell-17", + "id": "cell-21", "metadata": {}, "source": [ "With every part named above, the record assembles from them directly." @@ -582,14 +704,14 @@ }, { "cell_type": "code", - "execution_count": 9, - "id": "cell-18", + "execution_count": 11, + "id": "cell-22", "metadata": { "execution": { - "iopub.execute_input": "2026-09-29T20:48:37.232768Z", - "iopub.status.busy": "2026-09-29T20:48:37.232689Z", - "iopub.status.idle": "2026-09-29T20:48:37.239422Z", - "shell.execute_reply": "2026-09-29T20:48:37.239082Z" + "iopub.execute_input": "2026-10-01T06:06:12.761021Z", + "iopub.status.busy": "2026-10-01T06:06:12.760934Z", + "iopub.status.idle": "2026-10-01T06:06:12.857510Z", + "shell.execute_reply": "2026-10-01T06:06:12.857102Z" } }, "outputs": [ @@ -597,16 +719,20 @@ "name": "stdout", "output_type": "stream", "text": [ + "Model tag: {'tag': 'ToasterDemo::aiC06Tag', 'identifier': 'AI-C06', 'annotated_element': 'ToasterDemo::HeatingAssembly::heatGen'}\n", "Validation errors: []\n", "Premises: ['AC-C06', 'AS-C06', 'AS-C03', 'AI-C04']\n" ] } ], "source": [ + "from toaster.query import get_review_record_refs\n", + "\n", "stopping_judgment = ReviewRecord(\n", " identifier=\"AI-C06\",\n", " kind=\"asserted_inference\",\n", " claim=claim,\n", + " subject_ref=subject_ref,\n", " model_ref=model_ref,\n", " content_hash=hash_content(source),\n", " scope=scope,\n", @@ -623,7 +749,9 @@ " record_kind=\"worked_example\",\n", ")\n", "\n", - "errors = validate_record(stopping_judgment)\n", + "errors = validate_record(stopping_judgment, model=model)\n", + "tag = next((t for t in get_review_record_refs(model) if t[\"identifier\"] == stopping_judgment.identifier), None)\n", + "print(f\"Model tag: {tag}\")\n", "print(f\"Validation errors: {errors}\")\n", "print(f\"Premises: {stopping_judgment.premises}\")\n", "conn.close()" @@ -631,23 +759,23 @@ }, { "cell_type": "markdown", - "id": "cell-19", + "id": "cell-23", "metadata": {}, "source": [ - "`validate_record` returns no errors, confirming the Hawkins 3.1 schema's required fields, including a non-empty `premises` list, are present and checked, not merely printed." + "The tag printed above is now part of the loaded model, and `validate_record` confirms the Python record and the model's own tag agree about what AI-C06 is actually about; the same empty error list confirms the Hawkins 3.1 schema's required fields, including a non-empty `premises` list, are present and checked, not merely printed." ] }, { "cell_type": "markdown", - "id": "cell-20", + "id": "cell-24", "metadata": {}, "source": [ - "The model built across this chapter's two prior notebooks loaded without error, and the evaluated results above, not the model's own declaration, are what this stopping judgment cites." + "The model built across this chapter's notebooks, including this one's own anchor addition, loaded without error, and the evaluated results above, not the model's own declaration, are what this stopping judgment cites." ] }, { "cell_type": "markdown", - "id": "cell-21", + "id": "cell-25", "metadata": {}, "source": [ "Try the chapter exercise in `exercises/ch06/exercise.ipynb`: record the two judgment sites your own `BrewReq` requirement raises (`AC-C06-EX`, a measure-framing judgment; `AS-C06-EX`, a mechanism-selection judgment for `Impeller`), then write an `AI-C06-EX` stopping judgment honestly scoped exactly like this notebook's own `AI-C06` — not a claim that your `BrewUnit` decomposition is complete — with `premises` referencing the real chain this branch rests on: `AC-C06-EX`, `AS-C06-EX`, your Chapter 3 exercise's `AS-C03-EX`, and your Chapter 4 exercise's `AI-C04-EX`." diff --git a/chapters/ch08-checking/02-violation-witness.ipynb b/chapters/ch08-checking/02-violation-witness.ipynb index 644581c..bc0ea37 100644 --- a/chapters/ch08-checking/02-violation-witness.ipynb +++ b/chapters/ch08-checking/02-violation-witness.ipynb @@ -15,7 +15,7 @@ "id": "cell-01", "metadata": {}, "source": [ - "Notebook 01 stated `deliveredEnergyBoundedBySupply` and confirmed it is really part of `ch08-cumulative.sysml`. This notebook does not add anything new to the model: it loads the same cumulative fixture and runs analysis against it. See [Ch8-01](01-invariant-def.ipynb) for the construct itself." + "Notebook 01 stated `deliveredEnergyBoundedBySupply` and confirmed it is really part of `ch08-cumulative.sysml`. This notebook runs analysis against that construct and, further down, adds the one new model element this chapter's own judgment record needs: a `ReviewRecordRef` tag anchoring `AS-C08` to `deliveredEnergyBoundedBySupply` itself. See [Ch8-01](01-invariant-def.ipynb) for the construct itself." ] }, { @@ -24,10 +24,10 @@ "id": "cell-02", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T13:09:56.452555Z", - "iopub.status.busy": "2026-09-28T13:09:56.452345Z", - "iopub.status.idle": "2026-09-28T13:09:56.738609Z", - "shell.execute_reply": "2026-09-28T13:09:56.736928Z" + "iopub.execute_input": "2026-10-01T06:22:52.613473Z", + "iopub.status.busy": "2026-10-01T06:22:52.613243Z", + "iopub.status.idle": "2026-10-01T06:22:52.750264Z", + "shell.execute_reply": "2026-10-01T06:22:52.749567Z" } }, "outputs": [], @@ -47,7 +47,7 @@ "id": "cell-03", "metadata": {}, "source": [ - "The model is unchanged from notebook 01: `deliveredEnergyBoundedBySupply` is already in `ch08-cumulative.sysml`, so this notebook only needs to load it, not build anything." + "The model loaded above already carries both `deliveredEnergyBoundedBySupply` (notebook 01) and the `asC08Tag` this notebook introduces further down, so nothing needs building before the analysis runs." ] }, { @@ -64,10 +64,10 @@ "id": "cell-05", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T13:09:56.742829Z", - "iopub.status.busy": "2026-09-28T13:09:56.742270Z", - "iopub.status.idle": "2026-09-28T13:09:56.749169Z", - "shell.execute_reply": "2026-09-28T13:09:56.748507Z" + "iopub.execute_input": "2026-10-01T06:22:52.751935Z", + "iopub.status.busy": "2026-10-01T06:22:52.751734Z", + "iopub.status.idle": "2026-10-01T06:22:52.755451Z", + "shell.execute_reply": "2026-10-01T06:22:52.755028Z" } }, "outputs": [ @@ -110,10 +110,10 @@ "id": "cell-07", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T13:09:56.753182Z", - "iopub.status.busy": "2026-09-28T13:09:56.753008Z", - "iopub.status.idle": "2026-09-28T13:09:56.775644Z", - "shell.execute_reply": "2026-09-28T13:09:56.774688Z" + "iopub.execute_input": "2026-10-01T06:22:52.756825Z", + "iopub.status.busy": "2026-10-01T06:22:52.756701Z", + "iopub.status.idle": "2026-10-01T06:22:52.770138Z", + "shell.execute_reply": "2026-10-01T06:22:52.769753Z" } }, "outputs": [ @@ -158,10 +158,10 @@ "id": "cell-09", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T13:09:56.778275Z", - "iopub.status.busy": "2026-09-28T13:09:56.778025Z", - "iopub.status.idle": "2026-09-28T13:09:57.147489Z", - "shell.execute_reply": "2026-09-28T13:09:57.146574Z" + "iopub.execute_input": "2026-10-01T06:22:52.771290Z", + "iopub.status.busy": "2026-10-01T06:22:52.771217Z", + "iopub.status.idle": "2026-10-01T06:22:53.046558Z", + "shell.execute_reply": "2026-10-01T06:22:53.046172Z" } }, "outputs": [ @@ -204,10 +204,10 @@ "id": "cell-11", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T13:09:57.150324Z", - "iopub.status.busy": "2026-09-28T13:09:57.150201Z", - "iopub.status.idle": "2026-09-28T13:09:57.570836Z", - "shell.execute_reply": "2026-09-28T13:09:57.570341Z" + "iopub.execute_input": "2026-10-01T06:22:53.048239Z", + "iopub.status.busy": "2026-10-01T06:22:53.048133Z", + "iopub.status.idle": "2026-10-01T06:22:53.364995Z", + "shell.execute_reply": "2026-10-01T06:22:53.364396Z" } }, "outputs": [ @@ -215,7 +215,13 @@ "name": "stdout", "output_type": "stream", "text": [ - "[satisfied] deliveredEnergyBoundedBySupply (z3: holds for all values of unbound features)\n", + "[satisfied] deliveredEnergyBoundedBySupply (z3: holds for all values of unbound features)\n" + ] + }, + { + "name": "stdout", + "output_type": "stream", + "text": [ "\n", "Proved for every value of heatGenCheck.efficiency, heatGenCheck.power and heatGenCheckDuration the antecedent admits, not evaluated at one.\n" ] @@ -300,10 +306,10 @@ "id": "cell-13", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T13:09:57.572641Z", - "iopub.status.busy": "2026-09-28T13:09:57.572505Z", - "iopub.status.idle": "2026-09-28T13:09:57.894357Z", - "shell.execute_reply": "2026-09-28T13:09:57.893875Z" + "iopub.execute_input": "2026-10-01T06:22:53.366791Z", + "iopub.status.busy": "2026-10-01T06:22:53.366655Z", + "iopub.status.idle": "2026-10-01T06:22:53.663943Z", + "shell.execute_reply": "2026-10-01T06:22:53.663412Z" } }, "outputs": [ @@ -311,7 +317,13 @@ "name": "stdout", "output_type": "stream", "text": [ - "[violated] deliveredEnergyExceedsSupply (z3: unsatisfiable -- no assignment can make this hold)\n", + "[violated] deliveredEnergyExceedsSupply (z3: unsatisfiable -- no assignment can make this hold)\n" + ] + }, + { + "name": "stdout", + "output_type": "stream", + "text": [ "\n", "The loop catches a fully broken lemma: violated, not undecided, not silently accepted.\n" ] @@ -368,10 +380,10 @@ "id": "cell-15", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T13:09:57.896014Z", - "iopub.status.busy": "2026-09-28T13:09:57.895884Z", - "iopub.status.idle": "2026-09-28T13:09:58.272744Z", - "shell.execute_reply": "2026-09-28T13:09:58.271856Z" + "iopub.execute_input": "2026-10-01T06:22:53.665521Z", + "iopub.status.busy": "2026-10-01T06:22:53.665420Z", + "iopub.status.idle": "2026-10-01T06:22:53.975771Z", + "shell.execute_reply": "2026-10-01T06:22:53.975058Z" } }, "outputs": [ @@ -379,7 +391,13 @@ "name": "stdout", "output_type": "stream", "text": [ - "[undecided] deliveredEnergyBoundedBySupply (result is indeterminate over unbound features -- z3: satisfiable, e.g. heatGenCheck.efficiency = 0, heatGenCheck.power = 0 [W], heatGenCheckDuration = 1 [s])\n", + "[undecided] deliveredEnergyBoundedBySupply (result is indeterminate over unbound features -- z3: satisfiable, e.g. heatGenCheck.efficiency = 0, heatGenCheck.power = 0 [W], heatGenCheckDuration = 1 [s])\n" + ] + }, + { + "name": "stdout", + "output_type": "stream", + "text": [ "holds() correctly refuses to answer: undecided, not proven either way: deliveredEnergyBoundedBySupply (companion-check-scratch/conservation_check_weakened.sysml:15:9)\n" ] } @@ -437,16 +455,104 @@ "The proof above is engineering evidence, not a passing test result: it is what would have to be re-checked if `HeatGenerator`'s bound or `deliveredEnergy`'s definition ever changed, by hand, since nothing in this toolchain checks that the restated copy stays in sync with either. Recording it as a `ReviewRecord` states, in a reader's terms, what makes this evidence appropriate, sufficient and trustworthy (Hawkins et al. 2011, SS3.1-3.4), the same way earlier chapters recorded their own judgment sites." ] }, + { + "cell_type": "markdown", + "id": "cell-17", + "metadata": {}, + "source": [ + "Before AS-C08 states its claim, the next cells give it a real anchor in the model: a `ReviewRecordRef` tag naming `deliveredEnergyBoundedBySupply` as what AS-C08 is about." + ] + }, { "cell_type": "code", "execution_count": 8, - "id": "cell-17", + "id": "cell-18", + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T06:22:53.977230Z", + "iopub.status.busy": "2026-10-01T06:22:53.977123Z", + "iopub.status.idle": "2026-10-01T06:22:53.979117Z", + "shell.execute_reply": "2026-10-01T06:22:53.978770Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "metadata asC08Tag : ReviewRecordRef about deliveredEnergyBoundedBySupply {\n", + " identifier = \"AS-C08\";\n", + "}\n", + "\n" + ] + } + ], + "source": [ + "# what is being claimed, and what, specifically, it is about\n", + "subject_ref = \"ToasterDemo::deliveredEnergyBoundedBySupply\"\n", + "AS_C08_TAG = \"\"\"\\\n", + "metadata asC08Tag : ReviewRecordRef about deliveredEnergyBoundedBySupply {\n", + " identifier = \"AS-C08\";\n", + "}\n", + "\"\"\"\n", + "print(AS_C08_TAG)" + ] + }, + { + "cell_type": "markdown", + "id": "cell-19", + "metadata": {}, + "source": [ + "This fragment is the same text now committed in `models/ch08-cumulative.sysml` onward, alongside the `ReviewRecordRef` definition (carried forward since Chapter 2); loading the cumulative model (the `model.ok` cell above, unchanged) is what makes this a real, checkable tag rather than an assertion the reader has to trust." + ] + }, + { + "cell_type": "code", + "execution_count": 9, + "id": "cell-20", + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T06:22:53.980278Z", + "iopub.status.busy": "2026-10-01T06:22:53.980193Z", + "iopub.status.idle": "2026-10-01T06:22:53.981896Z", + "shell.execute_reply": "2026-10-01T06:22:53.981349Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "metadata asC08Tag : ReviewRecordRef about deliveredEnergyBoundedBySupply {\n", + " identifier = \"AS-C08\";\n", + "}\n", + "\n" + ] + } + ], + "source": [ + "TOASTER_INCREMENT = AS_C08_TAG\n", + "print(TOASTER_INCREMENT)" + ] + }, + { + "cell_type": "markdown", + "id": "cell-21", + "metadata": {}, + "source": [ + "AS-C08 states the claim next: what the lemma establishes, and about which model element." + ] + }, + { + "cell_type": "code", + "execution_count": 10, + "id": "cell-22", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T13:09:58.275340Z", - "iopub.status.busy": "2026-09-28T13:09:58.275148Z", - "iopub.status.idle": "2026-09-28T13:09:58.278209Z", - "shell.execute_reply": "2026-09-28T13:09:58.277639Z" + "iopub.execute_input": "2026-10-01T06:22:53.983204Z", + "iopub.status.busy": "2026-10-01T06:22:53.983123Z", + "iopub.status.idle": "2026-10-01T06:22:53.985314Z", + "shell.execute_reply": "2026-10-01T06:22:53.984999Z" } }, "outputs": [ @@ -473,7 +579,7 @@ }, { "cell_type": "markdown", - "id": "cell-18", + "id": "cell-23", "metadata": {}, "source": [ "What standard is this claim checked against? `verify_holds()` reports a single verdict for `deliveredEnergyBoundedBySupply`, proved by Z3 over the unbound features the companion restatement carries, not merely evaluated at one point." @@ -481,14 +587,14 @@ }, { "cell_type": "code", - "execution_count": 9, - "id": "cell-19", + "execution_count": 11, + "id": "cell-24", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T13:09:58.280598Z", - "iopub.status.busy": "2026-09-28T13:09:58.280495Z", - "iopub.status.idle": "2026-09-28T13:09:58.283305Z", - "shell.execute_reply": "2026-09-28T13:09:58.282783Z" + "iopub.execute_input": "2026-10-01T06:22:53.986459Z", + "iopub.status.busy": "2026-10-01T06:22:53.986380Z", + "iopub.status.idle": "2026-10-01T06:22:53.988059Z", + "shell.execute_reply": "2026-10-01T06:22:53.987765Z" } }, "outputs": [ @@ -519,7 +625,7 @@ }, { "cell_type": "markdown", - "id": "cell-20", + "id": "cell-25", "metadata": {}, "source": [ "What is this claim taking as given? The proof rests on the companion file restating `deliveredEnergyBoundedBySupply` correctly, and on the lemma's own hypothesis (non-negative power and duration), which the model states here but does not enforce as a standing constraint on `HeatGenerator` or `ApplyHeat` elsewhere." @@ -527,14 +633,14 @@ }, { "cell_type": "code", - "execution_count": 10, - "id": "cell-21", + "execution_count": 12, + "id": "cell-26", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T13:09:58.285816Z", - "iopub.status.busy": "2026-09-28T13:09:58.285555Z", - "iopub.status.idle": "2026-09-28T13:09:58.288010Z", - "shell.execute_reply": "2026-09-28T13:09:58.287367Z" + "iopub.execute_input": "2026-10-01T06:22:53.989192Z", + "iopub.status.busy": "2026-10-01T06:22:53.989118Z", + "iopub.status.idle": "2026-10-01T06:22:53.990866Z", + "shell.execute_reply": "2026-10-01T06:22:53.990494Z" } }, "outputs": [ @@ -554,7 +660,7 @@ }, { "cell_type": "markdown", - "id": "cell-22", + "id": "cell-27", "metadata": {}, "source": [ "What supports the claim, and how? The Z3-derived verdict above, cited directly, is the evidence; the rationale states what kind of check produced it and why that is a materially different kind of evidence from a point evaluation, while being explicit about what it does not establish." @@ -562,14 +668,14 @@ }, { "cell_type": "code", - "execution_count": 11, - "id": "cell-23", + "execution_count": 13, + "id": "cell-28", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T13:09:58.289972Z", - "iopub.status.busy": "2026-09-28T13:09:58.289812Z", - "iopub.status.idle": "2026-09-28T13:09:58.292497Z", - "shell.execute_reply": "2026-09-28T13:09:58.291992Z" + "iopub.execute_input": "2026-10-01T06:22:53.992123Z", + "iopub.status.busy": "2026-10-01T06:22:53.992055Z", + "iopub.status.idle": "2026-10-01T06:22:53.994366Z", + "shell.execute_reply": "2026-10-01T06:22:53.993994Z" } }, "outputs": [ @@ -601,7 +707,7 @@ }, { "cell_type": "markdown", - "id": "cell-24", + "id": "cell-29", "metadata": {}, "source": [ "What could be wrong, and what is still open? A record that hides its own weak points is not more trustworthy, it is less checkable." @@ -609,14 +715,14 @@ }, { "cell_type": "code", - "execution_count": 12, - "id": "cell-25", + "execution_count": 14, + "id": "cell-30", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T13:09:58.294039Z", - "iopub.status.busy": "2026-09-28T13:09:58.293884Z", - "iopub.status.idle": "2026-09-28T13:09:58.296541Z", - "shell.execute_reply": "2026-09-28T13:09:58.296113Z" + "iopub.execute_input": "2026-10-01T06:22:53.995488Z", + "iopub.status.busy": "2026-10-01T06:22:53.995412Z", + "iopub.status.idle": "2026-10-01T06:22:53.997789Z", + "shell.execute_reply": "2026-10-01T06:22:53.997417Z" } }, "outputs": [ @@ -662,7 +768,7 @@ }, { "cell_type": "markdown", - "id": "cell-26", + "id": "cell-31", "metadata": {}, "source": [ "Assembling the record from the parts above, the same way a construction-zone cell assembles a model fragment from its own named pieces." @@ -670,14 +776,14 @@ }, { "cell_type": "code", - "execution_count": 13, - "id": "cell-27", + "execution_count": 15, + "id": "cell-32", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T13:09:58.298420Z", - "iopub.status.busy": "2026-09-28T13:09:58.298296Z", - "iopub.status.idle": "2026-09-28T13:09:58.310567Z", - "shell.execute_reply": "2026-09-28T13:09:58.309784Z" + "iopub.execute_input": "2026-10-01T06:22:53.999350Z", + "iopub.status.busy": "2026-10-01T06:22:53.999243Z", + "iopub.status.idle": "2026-10-01T06:22:54.140144Z", + "shell.execute_reply": "2026-10-01T06:22:54.139729Z" } }, "outputs": [ @@ -685,15 +791,20 @@ "name": "stdout", "output_type": "stream", "text": [ + "Model tag: {'tag': 'ToasterDemo::asC08Tag', 'identifier': 'AS-C08', 'annotated_element': 'ToasterDemo::deliveredEnergyBoundedBySupply'}\n", + "Validation errors: []\n", "Record valid: identifier='AS-C08' engineering_conclusion='supported'\n" ] } ], "source": [ + "from toaster.query import get_review_record_refs\n", + "\n", "record = ReviewRecord(\n", " identifier=\"AS-C08\",\n", " kind=\"asserted_solution\",\n", " claim=claim,\n", + " subject_ref=subject_ref,\n", " model_ref=model_ref,\n", " content_hash=hash_content(source),\n", " scope=scope,\n", @@ -710,7 +821,10 @@ " record_kind=\"worked_example\",\n", ")\n", "\n", - "errors = validate_record(record)\n", + "errors = validate_record(record, model=model)\n", + "tag = next((t for t in get_review_record_refs(model) if t[\"identifier\"] == record.identifier), None)\n", + "print(f\"Model tag: {tag}\")\n", + "print(f\"Validation errors: {errors}\")\n", "assert errors == [], f\"Validation errors: {errors}\"\n", "print(f\"Record valid: identifier={record.identifier!r} engineering_conclusion={record.engineering_conclusion!r}\")\n", "conn.close()" @@ -718,15 +832,15 @@ }, { "cell_type": "markdown", - "id": "cell-28", + "id": "cell-33", "metadata": {}, "source": [ - "The claim printed above, the proof it points to, and the record's own counterevidence stating plainly what that proof does and does not establish are three distinct things this notebook watched connect: a written lemma, a real solver's verdict on it, and a record that never overstates what the verdict actually covers." + "The tag printed above is now part of the loaded model, and `validate_record` confirms the Python record and the model's own tag agree about what AS-C08 is actually about. The claim printed above, the proof it points to, and the record's own counterevidence stating plainly what that proof does and does not establish are three distinct things this notebook watched connect: a written lemma, a real solver's verdict on it, and a record that never overstates what the verdict actually covers." ] }, { "cell_type": "markdown", - "id": "cell-29", + "id": "cell-34", "metadata": {}, "source": [ "Try the chapter exercise in `exercises/ch08/exercise.ipynb`: it works through the same `verify_satisfaction()` and stale-detection pattern on its own, separate coffee-maker exercise model." diff --git a/chapters/ch08-checking/03-revision-flow.ipynb b/chapters/ch08-checking/03-revision-flow.ipynb index 59e6efa..5572af8 100644 --- a/chapters/ch08-checking/03-revision-flow.ipynb +++ b/chapters/ch08-checking/03-revision-flow.ipynb @@ -24,10 +24,10 @@ "id": "cell-02", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T12:48:01.075443Z", - "iopub.status.busy": "2026-09-28T12:48:01.075347Z", - "iopub.status.idle": "2026-09-28T12:48:01.207114Z", - "shell.execute_reply": "2026-09-28T12:48:01.206608Z" + "iopub.execute_input": "2026-10-01T06:14:37.272101Z", + "iopub.status.busy": "2026-10-01T06:14:37.271857Z", + "iopub.status.idle": "2026-10-01T06:14:37.503541Z", + "shell.execute_reply": "2026-10-01T06:14:37.502886Z" } }, "outputs": [], @@ -56,10 +56,10 @@ "id": "cell-04", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T12:48:01.209090Z", - "iopub.status.busy": "2026-09-28T12:48:01.208850Z", - "iopub.status.idle": "2026-09-28T12:48:01.212290Z", - "shell.execute_reply": "2026-09-28T12:48:01.211835Z" + "iopub.execute_input": "2026-10-01T06:14:37.505423Z", + "iopub.status.busy": "2026-10-01T06:14:37.505206Z", + "iopub.status.idle": "2026-10-01T06:14:37.606995Z", + "shell.execute_reply": "2026-10-01T06:14:37.606579Z" } }, "outputs": [ @@ -78,6 +78,7 @@ " identifier=\"\", # intentionally empty\n", " kind=\"asserted_solution\",\n", " claim=\"Some claim\",\n", + " subject_ref=\"ToasterDemo::deliveredEnergyBoundedBySupply\",\n", " model_ref=\"ToasterDemo\",\n", " content_hash=hash_content(source),\n", " scope=\"Some scope\",\n", @@ -86,8 +87,8 @@ " counterevidence=\"Some counterevidence\",\n", " record_kind=\"worked_example\",\n", ")\n", - "errors = validate_record(broken)\n", - "assert len(errors) > 0, \"Expected validation errors for empty identifier\"\n", + "errors = validate_record(broken, model=model)\n", + "assert errors == [\"identifier is empty\"], f\"Expected exactly one identifier error, got {errors}\"\n", "print(f\"Negative control ok: errors={errors}\")" ] }, @@ -105,10 +106,10 @@ "id": "cell-06", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T12:48:01.214282Z", - "iopub.status.busy": "2026-09-28T12:48:01.214176Z", - "iopub.status.idle": "2026-09-28T12:48:01.217235Z", - "shell.execute_reply": "2026-09-28T12:48:01.216718Z" + "iopub.execute_input": "2026-10-01T06:14:37.608316Z", + "iopub.status.busy": "2026-10-01T06:14:37.608214Z", + "iopub.status.idle": "2026-10-01T06:14:37.611357Z", + "shell.execute_reply": "2026-10-01T06:14:37.610935Z" } }, "outputs": [ @@ -132,6 +133,7 @@ " \"own definition, holds for every value of efficiency in [0,1] and every \"\n", " \"non-negative power and duration a companion restatement admits.\"\n", " ),\n", + " subject_ref=\"ToasterDemo::deliveredEnergyBoundedBySupply\",\n", " model_ref=\"ToasterDemo::deliveredEnergyBoundedBySupply\",\n", " content_hash=hash_content(source),\n", " scope=(\n", @@ -177,10 +179,10 @@ "id": "cell-08", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T12:48:01.219093Z", - "iopub.status.busy": "2026-09-28T12:48:01.218926Z", - "iopub.status.idle": "2026-09-28T12:48:01.241698Z", - "shell.execute_reply": "2026-09-28T12:48:01.241131Z" + "iopub.execute_input": "2026-10-01T06:14:37.612625Z", + "iopub.status.busy": "2026-10-01T06:14:37.612552Z", + "iopub.status.idle": "2026-10-01T06:14:37.637045Z", + "shell.execute_reply": "2026-10-01T06:14:37.636645Z" } }, "outputs": [ diff --git a/chapters/ch08-checking/index.md b/chapters/ch08-checking/index.md index 16ca211..27abab7 100644 --- a/chapters/ch08-checking/index.md +++ b/chapters/ch08-checking/index.md @@ -1,4 +1,4 @@ -# Chapter 8 - Constraint Checking +# Chapter 8: Checking and Revision ## Purpose diff --git a/chapters/ch09-coverage-sufficiency/02-evidence-completeness.ipynb b/chapters/ch09-coverage-sufficiency/02-evidence-completeness.ipynb index c018e6d..d899a2e 100644 --- a/chapters/ch09-coverage-sufficiency/02-evidence-completeness.ipynb +++ b/chapters/ch09-coverage-sufficiency/02-evidence-completeness.ipynb @@ -24,10 +24,10 @@ "id": "419ecbee", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T14:23:22.899928Z", - "iopub.status.busy": "2026-09-28T14:23:22.899767Z", - "iopub.status.idle": "2026-09-28T14:23:23.072242Z", - "shell.execute_reply": "2026-09-28T14:23:23.071741Z" + "iopub.execute_input": "2026-10-01T06:29:50.511766Z", + "iopub.status.busy": "2026-10-01T06:29:50.511543Z", + "iopub.status.idle": "2026-10-01T06:29:51.871173Z", + "shell.execute_reply": "2026-10-01T06:29:51.870671Z" } }, "outputs": [], @@ -56,10 +56,10 @@ "id": "a94c115a", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T14:23:23.074091Z", - "iopub.status.busy": "2026-09-28T14:23:23.073906Z", - "iopub.status.idle": "2026-09-28T14:23:23.076713Z", - "shell.execute_reply": "2026-09-28T14:23:23.076154Z" + "iopub.execute_input": "2026-10-01T06:29:51.872896Z", + "iopub.status.busy": "2026-10-01T06:29:51.872695Z", + "iopub.status.idle": "2026-10-01T06:29:51.980722Z", + "shell.execute_reply": "2026-10-01T06:29:51.980346Z" } }, "outputs": [ @@ -78,6 +78,7 @@ " identifier=\"AS-BAD\",\n", " kind=\"asserted_solution\",\n", " claim=\"Some claim\",\n", + " subject_ref=\"ToasterDemo::ResistanceCoil\",\n", " model_ref=\"ToasterDemo\",\n", " content_hash=hash_content(source),\n", " scope=\"Some scope\",\n", @@ -86,8 +87,8 @@ " counterevidence=\"\", # intentionally empty\n", " record_kind=\"worked_example\",\n", ")\n", - "errors = validate_record(blank_counterevidence)\n", - "assert len(errors) > 0, \"Expected validation to fail on empty counterevidence\"\n", + "errors = validate_record(blank_counterevidence, model=model)\n", + "assert errors == [\"counterevidence is empty\"], f\"Expected exactly one error, got {errors}\"\n", "print(f\"Negative control ok: errors={errors}\")\n" ] }, @@ -105,10 +106,10 @@ "id": "80e1a2ae", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T14:23:23.078385Z", - "iopub.status.busy": "2026-09-28T14:23:23.078280Z", - "iopub.status.idle": "2026-09-28T14:23:23.081881Z", - "shell.execute_reply": "2026-09-28T14:23:23.081469Z" + "iopub.execute_input": "2026-10-01T06:29:51.982079Z", + "iopub.status.busy": "2026-10-01T06:29:51.981991Z", + "iopub.status.idle": "2026-10-01T06:29:52.056360Z", + "shell.execute_reply": "2026-10-01T06:29:52.055931Z" } }, "outputs": [ @@ -132,6 +133,7 @@ " \"alternative (a gas burner, the tongs-and-blowtorch alternative this \"\n", " \"tutorial already contrasts) as the mechanism HeatGenerator commits to.\"\n", " ),\n", + " subject_ref=\"ToasterDemo::ResistanceCoil\",\n", " model_ref=\"ToasterDemo::ResistanceCoil\",\n", " content_hash=hash_content(ch06_source),\n", " scope=\"ToasterDemo::HeatGenerator and its realizations\",\n", @@ -194,7 +196,7 @@ " record_kind=\"worked_example\",\n", ")\n", "\n", - "errors = validate_record(as_c06)\n", + "errors = validate_record(as_c06, model=model)\n", "print(f\"AS-C06 validation errors: {errors}\")\n" ] }, @@ -212,10 +214,10 @@ "id": "6c7c7a99", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T14:23:23.083004Z", - "iopub.status.busy": "2026-09-28T14:23:23.082926Z", - "iopub.status.idle": "2026-09-28T14:23:23.086419Z", - "shell.execute_reply": "2026-09-28T14:23:23.086032Z" + "iopub.execute_input": "2026-10-01T06:29:52.057950Z", + "iopub.status.busy": "2026-10-01T06:29:52.057824Z", + "iopub.status.idle": "2026-10-01T06:29:52.132774Z", + "shell.execute_reply": "2026-10-01T06:29:52.132354Z" } }, "outputs": [ @@ -237,6 +239,7 @@ " \"definition, holds for every value of efficiency in [0,1] and every non-negative \"\n", " \"power and duration a companion restatement admits.\"\n", " ),\n", + " subject_ref=\"ToasterDemo::deliveredEnergyBoundedBySupply\",\n", " model_ref=\"ToasterDemo::deliveredEnergyBoundedBySupply\",\n", " content_hash=hash_content(source),\n", " scope=(\n", @@ -299,7 +302,7 @@ " record_kind=\"worked_example\",\n", ")\n", "\n", - "errors = validate_record(as_c08)\n", + "errors = validate_record(as_c08, model=model)\n", "print(f\"AS-C08 validation errors: {errors}\")\n" ] }, @@ -317,10 +320,10 @@ "id": "589670af", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T14:23:23.087658Z", - "iopub.status.busy": "2026-09-28T14:23:23.087574Z", - "iopub.status.idle": "2026-09-28T14:23:23.089657Z", - "shell.execute_reply": "2026-09-28T14:23:23.089343Z" + "iopub.execute_input": "2026-10-01T06:29:52.134174Z", + "iopub.status.busy": "2026-10-01T06:29:52.134087Z", + "iopub.status.idle": "2026-10-01T06:29:52.136571Z", + "shell.execute_reply": "2026-10-01T06:29:52.135994Z" } }, "outputs": [ @@ -358,10 +361,10 @@ "id": "0fc8e98d", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T14:23:23.090750Z", - "iopub.status.busy": "2026-09-28T14:23:23.090673Z", - "iopub.status.idle": "2026-09-28T14:23:23.093387Z", - "shell.execute_reply": "2026-09-28T14:23:23.092778Z" + "iopub.execute_input": "2026-10-01T06:29:52.137915Z", + "iopub.status.busy": "2026-10-01T06:29:52.137818Z", + "iopub.status.idle": "2026-10-01T06:29:52.140591Z", + "shell.execute_reply": "2026-10-01T06:29:52.140061Z" } }, "outputs": [ @@ -421,10 +424,10 @@ "id": "d3bcae55", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T14:23:23.094894Z", - "iopub.status.busy": "2026-09-28T14:23:23.094801Z", - "iopub.status.idle": "2026-09-28T14:23:23.101394Z", - "shell.execute_reply": "2026-09-28T14:23:23.101002Z" + "iopub.execute_input": "2026-10-01T06:29:52.141929Z", + "iopub.status.busy": "2026-10-01T06:29:52.141828Z", + "iopub.status.idle": "2026-10-01T06:29:52.215023Z", + "shell.execute_reply": "2026-10-01T06:29:52.214652Z" } }, "outputs": [ @@ -443,6 +446,7 @@ " identifier=\"AS-PLACEHOLDER\",\n", " kind=\"asserted_solution\",\n", " claim=\"The chosen design meets its requirement.\",\n", + " subject_ref=\"ToasterDemo::ResistanceCoil\",\n", " model_ref=\"ToasterDemo\",\n", " content_hash=hash_content(source),\n", " scope=\"ToasterDemo\",\n", @@ -456,7 +460,7 @@ " record_kind=\"worked_example\",\n", ")\n", "\n", - "errors = validate_record(placeholder_record)\n", + "errors = validate_record(placeholder_record, model=model)\n", "print(f\"validate_record: errors={errors}\")\n", "print(f\"counterevidence: {placeholder_record.counterevidence!r} ({word_label(placeholder_record.counterevidence)})\")\n", "print(f\"residual_uncertainties: {placeholder_record.residual_uncertainties!r} ({word_label(placeholder_record.residual_uncertainties)})\")\n", @@ -469,7 +473,7 @@ "id": "b316be7b", "metadata": {}, "source": [ - "`validate_record()` accepts this record outright, the same as `AS-C06` and `AS-C08` above: the mechanized floor only checks non-emptiness, and \"None known.\" and \"None.\" both satisfy it. A sufficiency reading must reject it anyway: neither field names a specific design alternative, a specific unmodeled relation, or a specific toolchain limit the way `AS-C06`'s and `AS-C08`'s own fields do; there is nothing here a reader could go check, disagree with, or find wrong. That is the actual difference sufficiency asks for, and it is a judgment a person makes by reading the field's content, not something `validate_record()` -- or any fixed rule -- can certify." + "`validate_record()` accepts this record outright, the same as `AS-C06` and `AS-C08` above: beyond requiring `subject_ref` to resolve in the model (a structural check `AS-PLACEHOLDER` also passes, since `ResistanceCoil` is a real element), the mechanized floor only checks non-emptiness, and \"None known.\" and \"None.\" both satisfy it. A sufficiency reading must reject it anyway: neither field names a specific design alternative, a specific unmodeled relation, or a specific toolchain limit the way `AS-C06`'s and `AS-C08`'s own fields do; there is nothing here a reader could go check, disagree with, or find wrong. That is the actual difference sufficiency asks for, and it is a judgment a person makes by reading the field's content, not something `validate_record()` -- or any fixed rule -- can certify." ] }, { diff --git a/chapters/ch09-coverage-sufficiency/03-stale-detection.ipynb b/chapters/ch09-coverage-sufficiency/03-stale-detection.ipynb index c2c3934..f49913a 100644 --- a/chapters/ch09-coverage-sufficiency/03-stale-detection.ipynb +++ b/chapters/ch09-coverage-sufficiency/03-stale-detection.ipynb @@ -24,10 +24,10 @@ "id": "dc06bb34", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T14:23:24.838041Z", - "iopub.status.busy": "2026-09-28T14:23:24.837831Z", - "iopub.status.idle": "2026-09-28T14:23:24.968994Z", - "shell.execute_reply": "2026-09-28T14:23:24.968378Z" + "iopub.execute_input": "2026-10-01T06:29:57.791033Z", + "iopub.status.busy": "2026-10-01T06:29:57.790939Z", + "iopub.status.idle": "2026-10-01T06:29:58.018819Z", + "shell.execute_reply": "2026-10-01T06:29:58.018225Z" } }, "outputs": [], @@ -56,10 +56,10 @@ "id": "ec9a3aef", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T14:23:24.970621Z", - "iopub.status.busy": "2026-09-28T14:23:24.970433Z", - "iopub.status.idle": "2026-09-28T14:23:24.973252Z", - "shell.execute_reply": "2026-09-28T14:23:24.972776Z" + "iopub.execute_input": "2026-10-01T06:29:58.021223Z", + "iopub.status.busy": "2026-10-01T06:29:58.020979Z", + "iopub.status.idle": "2026-10-01T06:29:58.127891Z", + "shell.execute_reply": "2026-10-01T06:29:58.127318Z" } }, "outputs": [ @@ -78,6 +78,7 @@ " identifier=\"\", # intentionally empty\n", " kind=\"asserted_solution\",\n", " claim=\"Some claim\",\n", + " subject_ref=\"ToasterDemo::deliveredEnergyBoundedBySupply\",\n", " model_ref=\"ToasterDemo\",\n", " content_hash=hash_content(source),\n", " scope=\"Some scope\",\n", @@ -86,8 +87,8 @@ " counterevidence=\"Some counterevidence\",\n", " record_kind=\"worked_example\",\n", ")\n", - "errors = validate_record(broken)\n", - "assert len(errors) > 0, \"Expected validation errors for empty identifier\"\n", + "errors = validate_record(broken, model=model)\n", + "assert errors == [\"identifier is empty\"], f\"Expected exactly one error, got {errors}\"\n", "print(f\"Negative control ok: errors={errors}\")\n" ] }, @@ -105,10 +106,10 @@ "id": "c46a2e31", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T14:23:24.974660Z", - "iopub.status.busy": "2026-09-28T14:23:24.974546Z", - "iopub.status.idle": "2026-09-28T14:23:24.978368Z", - "shell.execute_reply": "2026-09-28T14:23:24.977873Z" + "iopub.execute_input": "2026-10-01T06:29:58.129862Z", + "iopub.status.busy": "2026-10-01T06:29:58.129716Z", + "iopub.status.idle": "2026-10-01T06:29:58.206492Z", + "shell.execute_reply": "2026-10-01T06:29:58.205903Z" } }, "outputs": [ @@ -132,6 +133,7 @@ " \"alternative (a gas burner, the tongs-and-blowtorch alternative this \"\n", " \"tutorial already contrasts) as the mechanism HeatGenerator commits to.\"\n", " ),\n", + " subject_ref=\"ToasterDemo::ResistanceCoil\",\n", " model_ref=\"ToasterDemo::ResistanceCoil\",\n", " content_hash=hash_content(ch06_source),\n", " scope=\"ToasterDemo::HeatGenerator and its realizations\",\n", @@ -194,7 +196,7 @@ " record_kind=\"worked_example\",\n", ")\n", "\n", - "errors = validate_record(as_c06)\n", + "errors = validate_record(as_c06, model=model)\n", "print(f\"AS-C06 validation errors: {errors}\")\n" ] }, @@ -212,10 +214,10 @@ "id": "5d39d8e4", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T14:23:24.979774Z", - "iopub.status.busy": "2026-09-28T14:23:24.979691Z", - "iopub.status.idle": "2026-09-28T14:23:24.982983Z", - "shell.execute_reply": "2026-09-28T14:23:24.982664Z" + "iopub.execute_input": "2026-10-01T06:29:58.208124Z", + "iopub.status.busy": "2026-10-01T06:29:58.208003Z", + "iopub.status.idle": "2026-10-01T06:29:58.283857Z", + "shell.execute_reply": "2026-10-01T06:29:58.283221Z" } }, "outputs": [ @@ -237,6 +239,7 @@ " \"definition, holds for every value of efficiency in [0,1] and every non-negative \"\n", " \"power and duration a companion restatement admits.\"\n", " ),\n", + " subject_ref=\"ToasterDemo::deliveredEnergyBoundedBySupply\",\n", " model_ref=\"ToasterDemo::deliveredEnergyBoundedBySupply\",\n", " content_hash=hash_content(source),\n", " scope=(\n", @@ -299,7 +302,7 @@ " record_kind=\"worked_example\",\n", ")\n", "\n", - "errors = validate_record(as_c08)\n", + "errors = validate_record(as_c08, model=model)\n", "print(f\"AS-C08 validation errors: {errors}\")\n" ] }, @@ -317,10 +320,10 @@ "id": "cd1f0798", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T14:23:24.984138Z", - "iopub.status.busy": "2026-09-28T14:23:24.984060Z", - "iopub.status.idle": "2026-09-28T14:23:24.986753Z", - "shell.execute_reply": "2026-09-28T14:23:24.986358Z" + "iopub.execute_input": "2026-10-01T06:29:58.285916Z", + "iopub.status.busy": "2026-10-01T06:29:58.285797Z", + "iopub.status.idle": "2026-10-01T06:29:58.288853Z", + "shell.execute_reply": "2026-10-01T06:29:58.288308Z" } }, "outputs": [ @@ -361,10 +364,10 @@ "id": "0319ca53", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T14:23:24.988401Z", - "iopub.status.busy": "2026-09-28T14:23:24.988274Z", - "iopub.status.idle": "2026-09-28T14:23:24.991315Z", - "shell.execute_reply": "2026-09-28T14:23:24.990902Z" + "iopub.execute_input": "2026-10-01T06:29:58.290809Z", + "iopub.status.busy": "2026-10-01T06:29:58.290681Z", + "iopub.status.idle": "2026-10-01T06:29:58.293929Z", + "shell.execute_reply": "2026-10-01T06:29:58.293287Z" } }, "outputs": [ @@ -410,10 +413,10 @@ "id": "a550031d", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T14:23:24.992526Z", - "iopub.status.busy": "2026-09-28T14:23:24.992452Z", - "iopub.status.idle": "2026-09-28T14:23:25.013485Z", - "shell.execute_reply": "2026-09-28T14:23:25.012810Z" + "iopub.execute_input": "2026-10-01T06:29:58.295479Z", + "iopub.status.busy": "2026-10-01T06:29:58.295355Z", + "iopub.status.idle": "2026-10-01T06:29:58.318110Z", + "shell.execute_reply": "2026-10-01T06:29:58.317556Z" } }, "outputs": [ diff --git a/chapters/ch09-coverage-sufficiency/index.md b/chapters/ch09-coverage-sufficiency/index.md index a838e6f..6fa09b5 100644 --- a/chapters/ch09-coverage-sufficiency/index.md +++ b/chapters/ch09-coverage-sufficiency/index.md @@ -1,4 +1,4 @@ -# Chapter 9 - Coverage and Sufficiency +# Chapter 9: Coverage and Sufficiency ## Purpose diff --git a/chapters/ch10-traceability-signoff/01-traceability-graph.ipynb b/chapters/ch10-traceability-signoff/01-traceability-graph.ipynb index a08f16c..5581026 100644 --- a/chapters/ch10-traceability-signoff/01-traceability-graph.ipynb +++ b/chapters/ch10-traceability-signoff/01-traceability-graph.ipynb @@ -24,10 +24,10 @@ "id": "b8bb18ce", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T02:17:53.377323Z", - "iopub.status.busy": "2026-10-01T02:17:53.377186Z", - "iopub.status.idle": "2026-10-01T02:17:53.655747Z", - "shell.execute_reply": "2026-10-01T02:17:53.655257Z" + "iopub.execute_input": "2026-10-01T06:45:43.101340Z", + "iopub.status.busy": "2026-10-01T06:45:43.101202Z", + "iopub.status.idle": "2026-10-01T06:45:43.323991Z", + "shell.execute_reply": "2026-10-01T06:45:43.323520Z" } }, "outputs": [], @@ -56,10 +56,10 @@ "id": "a45bdbbc", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T02:17:53.657609Z", - "iopub.status.busy": "2026-10-01T02:17:53.657407Z", - "iopub.status.idle": "2026-10-01T02:17:53.671302Z", - "shell.execute_reply": "2026-10-01T02:17:53.670840Z" + "iopub.execute_input": "2026-10-01T06:45:43.325692Z", + "iopub.status.busy": "2026-10-01T06:45:43.325505Z", + "iopub.status.idle": "2026-10-01T06:45:43.339477Z", + "shell.execute_reply": "2026-10-01T06:45:43.339078Z" } }, "outputs": [ @@ -101,10 +101,10 @@ "id": "802a9057", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T02:17:53.672585Z", - "iopub.status.busy": "2026-10-01T02:17:53.672492Z", - "iopub.status.idle": "2026-10-01T02:17:53.775756Z", - "shell.execute_reply": "2026-10-01T02:17:53.775165Z" + "iopub.execute_input": "2026-10-01T06:45:43.340822Z", + "iopub.status.busy": "2026-10-01T06:45:43.340730Z", + "iopub.status.idle": "2026-10-01T06:45:43.445972Z", + "shell.execute_reply": "2026-10-01T06:45:43.445527Z" } }, "outputs": [ @@ -156,10 +156,10 @@ "id": "042cfd1c", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T02:17:53.777484Z", - "iopub.status.busy": "2026-10-01T02:17:53.777368Z", - "iopub.status.idle": "2026-10-01T02:17:53.824210Z", - "shell.execute_reply": "2026-10-01T02:17:53.823793Z" + "iopub.execute_input": "2026-10-01T06:45:43.447360Z", + "iopub.status.busy": "2026-10-01T06:45:43.447254Z", + "iopub.status.idle": "2026-10-01T06:45:43.503286Z", + "shell.execute_reply": "2026-10-01T06:45:43.502752Z" } }, "outputs": [ @@ -206,10 +206,10 @@ "id": "c91bbd5d", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T02:17:53.825278Z", - "iopub.status.busy": "2026-10-01T02:17:53.825184Z", - "iopub.status.idle": "2026-10-01T02:17:53.862596Z", - "shell.execute_reply": "2026-10-01T02:17:53.862192Z" + "iopub.execute_input": "2026-10-01T06:45:43.504754Z", + "iopub.status.busy": "2026-10-01T06:45:43.504641Z", + "iopub.status.idle": "2026-10-01T06:45:43.549140Z", + "shell.execute_reply": "2026-10-01T06:45:43.548698Z" } }, "outputs": [ @@ -253,10 +253,10 @@ "id": "8b69fdfa", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T02:17:53.863707Z", - "iopub.status.busy": "2026-10-01T02:17:53.863618Z", - "iopub.status.idle": "2026-10-01T02:17:53.865964Z", - "shell.execute_reply": "2026-10-01T02:17:53.865483Z" + "iopub.execute_input": "2026-10-01T06:45:43.550299Z", + "iopub.status.busy": "2026-10-01T06:45:43.550218Z", + "iopub.status.idle": "2026-10-01T06:45:43.552718Z", + "shell.execute_reply": "2026-10-01T06:45:43.552224Z" } }, "outputs": [ @@ -292,10 +292,10 @@ "id": "a4778ed8", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T02:17:53.867079Z", - "iopub.status.busy": "2026-10-01T02:17:53.866989Z", - "iopub.status.idle": "2026-10-01T02:17:53.869976Z", - "shell.execute_reply": "2026-10-01T02:17:53.869554Z" + "iopub.execute_input": "2026-10-01T06:45:43.554176Z", + "iopub.status.busy": "2026-10-01T06:45:43.554074Z", + "iopub.status.idle": "2026-10-01T06:45:43.556773Z", + "shell.execute_reply": "2026-10-01T06:45:43.556414Z" } }, "outputs": [ @@ -375,10 +375,10 @@ "id": "a7413515", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T02:17:53.870975Z", - "iopub.status.busy": "2026-10-01T02:17:53.870893Z", - "iopub.status.idle": "2026-10-01T02:17:53.952334Z", - "shell.execute_reply": "2026-10-01T02:17:53.951919Z" + "iopub.execute_input": "2026-10-01T06:45:43.557918Z", + "iopub.status.busy": "2026-10-01T06:45:43.557833Z", + "iopub.status.idle": "2026-10-01T06:45:43.643628Z", + "shell.execute_reply": "2026-10-01T06:45:43.643185Z" } }, "outputs": [ @@ -456,10 +456,10 @@ "id": "4d859570", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T02:17:53.953443Z", - "iopub.status.busy": "2026-10-01T02:17:53.953357Z", - "iopub.status.idle": "2026-10-01T02:17:53.957283Z", - "shell.execute_reply": "2026-10-01T02:17:53.956964Z" + "iopub.execute_input": "2026-10-01T06:45:43.645178Z", + "iopub.status.busy": "2026-10-01T06:45:43.645048Z", + "iopub.status.idle": "2026-10-01T06:45:43.647909Z", + "shell.execute_reply": "2026-10-01T06:45:43.647512Z" } }, "outputs": [ @@ -499,10 +499,10 @@ "id": "25c99d26", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T02:17:53.958445Z", - "iopub.status.busy": "2026-10-01T02:17:53.958372Z", - "iopub.status.idle": "2026-10-01T02:17:53.960674Z", - "shell.execute_reply": "2026-10-01T02:17:53.960326Z" + "iopub.execute_input": "2026-10-01T06:45:43.649085Z", + "iopub.status.busy": "2026-10-01T06:45:43.648989Z", + "iopub.status.idle": "2026-10-01T06:45:43.651147Z", + "shell.execute_reply": "2026-10-01T06:45:43.650840Z" } }, "outputs": [ @@ -584,10 +584,10 @@ "id": "c177450c", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T02:17:53.961609Z", - "iopub.status.busy": "2026-10-01T02:17:53.961539Z", - "iopub.status.idle": "2026-10-01T02:17:54.001218Z", - "shell.execute_reply": "2026-10-01T02:17:54.000835Z" + "iopub.execute_input": "2026-10-01T06:45:43.652467Z", + "iopub.status.busy": "2026-10-01T06:45:43.652374Z", + "iopub.status.idle": "2026-10-01T06:45:43.697998Z", + "shell.execute_reply": "2026-10-01T06:45:43.697616Z" } }, "outputs": [ @@ -595,13 +595,7 @@ "name": "stdout", "output_type": "stream", "text": [ - "the lemma itself (deliveredEnergyBoundedBySupply): assert satisfy energyConservationReq by deliveredEnergyBoundedBySupply;\n" - ] - }, - { - "name": "stdout", - "output_type": "stream", - "text": [ + "the lemma itself (deliveredEnergyBoundedBySupply): assert satisfy energyConservationReq by deliveredEnergyBoundedBySupply;\n", "the lemma's own free-standing usage (heatGenCheck): assert satisfy energyConservationReq by heatGenCheck;\n", "an unrelated real candidate (rated): assert satisfy energyConservationReq by rated;\n" ] @@ -642,10 +636,10 @@ "id": "55d20aaa", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T02:17:54.002379Z", - "iopub.status.busy": "2026-10-01T02:17:54.002299Z", - "iopub.status.idle": "2026-10-01T02:17:54.037374Z", - "shell.execute_reply": "2026-10-01T02:17:54.037033Z" + "iopub.execute_input": "2026-10-01T06:45:43.699287Z", + "iopub.status.busy": "2026-10-01T06:45:43.699208Z", + "iopub.status.idle": "2026-10-01T06:45:43.733506Z", + "shell.execute_reply": "2026-10-01T06:45:43.733230Z" } }, "outputs": [ @@ -653,13 +647,7 @@ "name": "stdout", "output_type": "stream", "text": [ - " [FAILS] the lemma itself (deliveredEnergyBoundedBySupply): satisfy energyConservationReq by deliveredEnergyBoundedBySupply (satisfaction satisfy energyConservationReq by deliveredEnergyBoundedBySupply: require condition evaluation failed: no value for feature heatGenCheck.efficiency)\n" - ] - }, - { - "name": "stdout", - "output_type": "stream", - "text": [ + " [FAILS] the lemma itself (deliveredEnergyBoundedBySupply): satisfy energyConservationReq by deliveredEnergyBoundedBySupply (satisfaction satisfy energyConservationReq by deliveredEnergyBoundedBySupply: require condition evaluation failed: no value for feature heatGenCheck.efficiency)\n", " [FAILS] the lemma's own free-standing usage (heatGenCheck): satisfy energyConservationReq by heatGenCheck (satisfaction satisfy energyConservationReq by heatGenCheck: require condition evaluation failed: no value for feature heatGenCheck.efficiency)\n", " [FAILS] an unrelated real candidate (rated): satisfy energyConservationReq by rated (satisfaction satisfy energyConservationReq by rated: require condition evaluation failed: no value for feature heatGenCheck.efficiency)\n" ] @@ -689,13 +677,13 @@ { "cell_type": "code", "execution_count": 13, - "id": "41047070", + "id": "fe8404d3", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T02:17:54.038615Z", - "iopub.status.busy": "2026-10-01T02:17:54.038532Z", - "iopub.status.idle": "2026-10-01T02:17:54.040423Z", - "shell.execute_reply": "2026-10-01T02:17:54.040163Z" + "iopub.execute_input": "2026-10-01T06:45:43.734876Z", + "iopub.status.busy": "2026-10-01T06:45:43.734800Z", + "iopub.status.idle": "2026-10-01T06:45:43.736780Z", + "shell.execute_reply": "2026-10-01T06:45:43.736319Z" } }, "outputs": [ @@ -733,8 +721,7 @@ } ], "source": [ - "TOASTER_INCREMENT = ENERGY_CONSERVATION_REQ_DEF\n", - "print(TOASTER_INCREMENT)\n" + "print(ENERGY_CONSERVATION_REQ_DEF)\n" ] }, { @@ -751,10 +738,10 @@ "id": "7bab30ed", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T02:17:54.041554Z", - "iopub.status.busy": "2026-10-01T02:17:54.041485Z", - "iopub.status.idle": "2026-10-01T02:17:54.044216Z", - "shell.execute_reply": "2026-10-01T02:17:54.043882Z" + "iopub.execute_input": "2026-10-01T06:45:43.737934Z", + "iopub.status.busy": "2026-10-01T06:45:43.737858Z", + "iopub.status.idle": "2026-10-01T06:45:43.740612Z", + "shell.execute_reply": "2026-10-01T06:45:43.740290Z" } }, "outputs": [ @@ -795,16 +782,68 @@ "The tie is real, and the subject-type fix keeps it spec-legitimate. One honest question is still open: is this tie -- a requirement whose own formal condition is defined by subsetting the very lemma it restates -- actually a legitimate way to close this chapter's own \"unjustified widget\" finding, or is it circular, a requirement manufactured FROM the evidence rather than one that independently motivates it? A doc comment cannot carry that answer honestly; a judgment record can. Build one the same way `AS-C06` (Chapter 6) is built: name each group of fields, narrate what it's for, print it, then assemble." ] }, + { + "cell_type": "markdown", + "id": "dd5e8954", + "metadata": {}, + "source": [ + "Before AC-C10 states its claim, the next cells give it a real anchor in the model: a `ReviewRecordRef` tag naming `EnergyConservationReq` as what AC-C10 is about." + ] + }, { "cell_type": "code", "execution_count": 15, + "id": "d6031258", + "metadata": { + "execution": { + "iopub.execute_input": "2026-10-01T06:45:43.741820Z", + "iopub.status.busy": "2026-10-01T06:45:43.741746Z", + "iopub.status.idle": "2026-10-01T06:45:43.743549Z", + "shell.execute_reply": "2026-10-01T06:45:43.743270Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "metadata acC10Tag : ReviewRecordRef about EnergyConservationReq {\n", + " identifier = \"AC-C10\";\n", + "}\n", + "\n" + ] + } + ], + "source": [ + "# what is being claimed, and what, specifically, it is about\n", + "ac_c10_subject_ref = \"ToasterDemo::EnergyConservationReq\"\n", + "AC_C10_TAG = \"\"\"\\\n", + "metadata acC10Tag : ReviewRecordRef about EnergyConservationReq {\n", + " identifier = \"AC-C10\";\n", + "}\n", + "\"\"\"\n", + "TOASTER_INCREMENT = f\"{ENERGY_CONSERVATION_REQ_DEF}\\n{AC_C10_TAG}\"\n", + "print(AC_C10_TAG)\n" + ] + }, + { + "cell_type": "markdown", + "id": "3ab3b4c5", + "metadata": {}, + "source": [ + "This fragment is the same text now committed in `models/ch10-cumulative.sysml`, alongside the `ReviewRecordRef` definition (carried forward since Chapter 2) and `EnergyConservationReq` itself (printed above); loading the cumulative model (the `model.ok` cell at the top of this notebook, unchanged) is what makes this a real, checkable tag rather than an assertion the reader has to trust. AC-C10 states the claim next: what this tie is actually being framed as." + ] + }, + { + "cell_type": "code", + "execution_count": 16, "id": "dcb8884d", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T02:17:54.045229Z", - "iopub.status.busy": "2026-10-01T02:17:54.045167Z", - "iopub.status.idle": "2026-10-01T02:17:54.047379Z", - "shell.execute_reply": "2026-10-01T02:17:54.047051Z" + "iopub.execute_input": "2026-10-01T06:45:43.744590Z", + "iopub.status.busy": "2026-10-01T06:45:43.744522Z", + "iopub.status.idle": "2026-10-01T06:45:43.746888Z", + "shell.execute_reply": "2026-10-01T06:45:43.746483Z" } }, "outputs": [ @@ -851,14 +890,14 @@ }, { "cell_type": "code", - "execution_count": 16, + "execution_count": 17, "id": "0c05ac8a", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T02:17:54.048448Z", - "iopub.status.busy": "2026-10-01T02:17:54.048381Z", - "iopub.status.idle": "2026-10-01T02:17:54.050566Z", - "shell.execute_reply": "2026-10-01T02:17:54.050236Z" + "iopub.execute_input": "2026-10-01T06:45:43.748058Z", + "iopub.status.busy": "2026-10-01T06:45:43.747981Z", + "iopub.status.idle": "2026-10-01T06:45:43.750236Z", + "shell.execute_reply": "2026-10-01T06:45:43.749908Z" } }, "outputs": [ @@ -906,14 +945,14 @@ }, { "cell_type": "code", - "execution_count": 17, + "execution_count": 18, "id": "2422340c", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T02:17:54.051632Z", - "iopub.status.busy": "2026-10-01T02:17:54.051566Z", - "iopub.status.idle": "2026-10-01T02:17:54.054344Z", - "shell.execute_reply": "2026-10-01T02:17:54.053981Z" + "iopub.execute_input": "2026-10-01T06:45:43.751349Z", + "iopub.status.busy": "2026-10-01T06:45:43.751284Z", + "iopub.status.idle": "2026-10-01T06:45:43.754063Z", + "shell.execute_reply": "2026-10-01T06:45:43.753621Z" } }, "outputs": [ @@ -921,7 +960,14 @@ "name": "stdout", "output_type": "stream", "text": [ - "[\"SysML v2.0 formal/2026-03-02 SS7.21.2 explicitly supports reference-subsetting an existing constraint as a requirement's own required constraint (`require ;` / `require constraint c :> ;`) -- the spec's own intended idiom for reusing an already-declared constraint, confirmed by reading the spec directly.\", \"SS7.21.1: 'A requirement usage can only be satisfied by an entity that conforms to the definition of its subject.' The base `requirement def RequirementCheck` (Systems Library/Requirements.sysml) declares `subject subj : Anything[1];`, confirmed directly above; EnergyConservationReq leaves its own subject undeclared, inheriting that default, since its own required constraint never references a subject at all. Two different earlier drafts hit two different problems: the reverted Approach A typed the subject `heatGen : HeatGenerator` but never added an assert-satisfy line at all, so its own defect (a declared, unused subject with nothing live bound to it) was found by direct spec reading, not by any tool diagnostic. A real pilot warning ('Bound features should have conforming types') did fire, but against a different, separately-built draft's own first commit: Approach B's own earliest version paired a typed subject with a real assert-satisfy line binding the lemma against it; that draft's own author fixed it by dropping the subject two commits later, before this reconciliation began. This design has no assert-satisfy at all, so that specific mechanical trigger does not even apply here either way, but the deeper reason for leaving the subject undeclared stands regardless -- recorded in full in docs/case-studies/2026-09-30-energy-conservation-requirement-tie.md.\", \"Direct test, not argument alone (the negative control above): `assert satisfy energyConservationReq by deliveredEnergyBoundedBySupply;` makes model.verify_satisfaction() error identically regardless of its own binding, because this requirement's own required constraint never references its subject at all -- the register `satisfy` is built for (a candidate's own values substituted into a formula that depends on them) is not what this construct offers, so no `assert satisfy` is given for it.\", \"Douglas's own requirement anatomy (AGENTS.md SS1, Part 4): a need, a rationale, and a means of verification. This requirement supplies a real, general rationale (energy conservation) and a real means (the subsetted, Z3-proved lemma itself), but its own need was identified only after Chapter 8's proof already existed, in response to this chapter's own traceability search -- not before it.\"]\n", + "[\"SysML v2.0 formal/2026-03-02 SS7.21.2 explicitly supports reference-subsetting an existing constraint as a requirement's own required constraint (`require ;` / `require constraint c :> ;`) -- the spec's own intended idiom for reusing an already-declared constraint, confirmed by reading the spec directly.\", \"SS7.21.1: 'A requirement usage can only be satisfied by an entity that conforms to the definition of its subject.' The base `requirement def RequirementCheck` (Systems Library/Requirements.sysml) declares `subject subj : Anything[1];`, confirmed directly above; EnergyConservationReq leaves its own subject undeclared, inheriting that default, since its own required constraint never references a subject at all. Two different earlier drafts hit two different problems: the reverted Approach A typed the subject `heatGen : HeatGenerator` but never added an assert-satisfy line at all, so its own defect (a declared, unused subject with nothing live bound to it) was found by direct spec reading, not by any tool diagnostic. A real pilot warning ('Bound features should have conforming types') did fire, but against a different, separately-built draft's own first commit: Approach B's own earliest version paired a typed subject with a real assert-satisfy line binding the lemma against it; that draft's own author fixed it by dropping the subject two commits later, before this reconciliation began. This design has no assert-satisfy at all, so that specific mechanical trigger does not even apply here either way, but the deeper reason for leaving the subject undeclared stands regardless -- recorded in full in docs/case-studies/2026-09-30-energy-conservation-requirement-tie.md.\", \"Direct test, not argument alone (the negative control above): `assert satisfy energyConservationReq by deliveredEnergyBoundedBySupply;` makes model.verify_satisfaction() error identically regardless of its own binding, because this requirement's own required constraint never references its subject at all -- the register `satisfy` is built for (a candidate's own values substituted into a formula that depends on them) is not what this construct offers, so no `assert satisfy` is given for it.\", \"Douglas's own requirement anatomy (AGENTS.md SS1, Part 4): a need, a rationale, and a means of verification. This requirement supplies a real, general rationale (energy conservation) and a real means (the subsetted, Z3-proved lemma itself), but its own need was identified only after Chapter 8's proof already existed, in response to this chapter's own traceability search -- not before it.\"]" + ] + }, + { + "name": "stdout", + "output_type": "stream", + "text": [ + "\n", "[\"AS-C08's own residual_uncertainties (Chapter 8): deliveredEnergyBoundedBySupply is a hand-restated companion lemma, not a solver-checked reference to HeatGenerator's own efficiencyBounded/deliveredEnergy (D-030, D-031); this record does not strengthen that proof, only frames how its tie to a requirement should be read.\"]\n" ] } @@ -990,14 +1036,14 @@ }, { "cell_type": "code", - "execution_count": 18, + "execution_count": 19, "id": "3850dee3", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T02:17:54.055551Z", - "iopub.status.busy": "2026-10-01T02:17:54.055467Z", - "iopub.status.idle": "2026-10-01T02:17:54.342100Z", - "shell.execute_reply": "2026-10-01T02:17:54.341568Z" + "iopub.execute_input": "2026-10-01T06:45:43.755225Z", + "iopub.status.busy": "2026-10-01T06:45:43.755148Z", + "iopub.status.idle": "2026-10-01T06:45:43.969529Z", + "shell.execute_reply": "2026-10-01T06:45:43.969049Z" } }, "outputs": [ @@ -1006,7 +1052,7 @@ "output_type": "stream", "text": [ "Lines mentioning EnergyConservationReq/energyConservationReq: []\n", - "../../models/ch10-cumulative.sysml:291:9 deliveredEnergyBoundedBySupply (AssertConstraintUsage): satisfied (z3: holds for all values of unbound features)\n" + "../../models/ch10-cumulative.sysml:323:9 deliveredEnergyBoundedBySupply (AssertConstraintUsage): satisfied (z3: holds for all values of unbound features)\n" ] } ], @@ -1052,14 +1098,14 @@ }, { "cell_type": "code", - "execution_count": 19, + "execution_count": 20, "id": "9cc5f8de", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T02:17:54.343618Z", - "iopub.status.busy": "2026-10-01T02:17:54.343490Z", - "iopub.status.idle": "2026-10-01T02:17:54.572752Z", - "shell.execute_reply": "2026-10-01T02:17:54.572220Z" + "iopub.execute_input": "2026-10-01T06:45:43.971581Z", + "iopub.status.busy": "2026-10-01T06:45:43.971445Z", + "iopub.status.idle": "2026-10-01T06:45:44.182195Z", + "shell.execute_reply": "2026-10-01T06:45:44.181720Z" } }, "outputs": [ @@ -1067,7 +1113,7 @@ "name": "stdout", "output_type": "stream", "text": [ - "companion-check-scratch/ch10_with_satisfy.sysml:291:9 c (ConstraintUsage, satisfies ToasterDemo::energyConservationReq): undecided (result is indeterminate over unbound features)\n" + "companion-check-scratch/ch10_with_satisfy.sysml:323:9 c (ConstraintUsage, satisfies ToasterDemo::energyConservationReq): undecided (result is indeterminate over unbound features)\n" ] } ], @@ -1106,14 +1152,14 @@ }, { "cell_type": "code", - "execution_count": 20, + "execution_count": 21, "id": "cc09043c", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T02:17:54.574260Z", - "iopub.status.busy": "2026-10-01T02:17:54.574010Z", - "iopub.status.idle": "2026-10-01T02:17:54.577424Z", - "shell.execute_reply": "2026-10-01T02:17:54.577088Z" + "iopub.execute_input": "2026-10-01T06:45:44.183699Z", + "iopub.status.busy": "2026-10-01T06:45:44.183548Z", + "iopub.status.idle": "2026-10-01T06:45:44.186722Z", + "shell.execute_reply": "2026-10-01T06:45:44.186288Z" } }, "outputs": [ @@ -1121,7 +1167,7 @@ "name": "stdout", "output_type": "stream", "text": [ - "[\"Systems Library/Requirements.sysml, confirmed directly above: `subject subj : Anything[1];` is RequirementCheck's own default subject.\", \"model.verify_satisfaction() against assert satisfy energyConservationReq by (the negative control above, three real bindings tested -- the lemma itself, its own free-standing usage, and an unrelated real candidate): holds=False, error='require condition evaluation failed: no value for feature heatGenCheck.efficiency' in every case.\", \"sysmlv2 verify --solve, real output captured live above: no line at all for EnergyConservationReq/energyConservationReq in the current, satisfy-less model ([]), only the lemma's own already-cited result (../../models/ch10-cumulative.sysml:291:9 deliveredEnergyBoundedBySupply (AssertConstraintUsage): satisfied (z3: holds for all values of unbound features)); confirmed by contrast, also run live above, that adding assert satisfy back in makes the solver emit a real verdict after all (companion-check-scratch/ch10_with_satisfy.sysml:291:9 c (ConstraintUsage, satisfies ToasterDemo::energyConservationReq): undecided (result is indeterminate over unbound features)) -- just an undecided one, not a satisfied one.\", \"requirement_ties(model, LEMMA, idx)'s own Check-B hit, reproduced above: exactly one entry, attributed to the DEFINITION (EnergyConservationReq).\"]\n", + "[\"Systems Library/Requirements.sysml, confirmed directly above: `subject subj : Anything[1];` is RequirementCheck's own default subject.\", \"model.verify_satisfaction() against assert satisfy energyConservationReq by (the negative control above, three real bindings tested -- the lemma itself, its own free-standing usage, and an unrelated real candidate): holds=False, error='require condition evaluation failed: no value for feature heatGenCheck.efficiency' in every case.\", \"sysmlv2 verify --solve, real output captured live above: no line at all for EnergyConservationReq/energyConservationReq in the current, satisfy-less model ([]), only the lemma's own already-cited result (../../models/ch10-cumulative.sysml:323:9 deliveredEnergyBoundedBySupply (AssertConstraintUsage): satisfied (z3: holds for all values of unbound features)); confirmed by contrast, also run live above, that adding assert satisfy back in makes the solver emit a real verdict after all (companion-check-scratch/ch10_with_satisfy.sysml:323:9 c (ConstraintUsage, satisfies ToasterDemo::energyConservationReq): undecided (result is indeterminate over unbound features)) -- just an undecided one, not a satisfied one.\", \"requirement_ties(model, LEMMA, idx)'s own Check-B hit, reproduced above: exactly one entry, attributed to the DEFINITION (EnergyConservationReq).\"]\n", "Three things are true at once here, and none should be softened into the others. First, energy conservation really is a general physical constraint on any heat generator, independent of whether Chapter 8 happened to prove one instance of it, so once stated this way EnergyConservationReq is a legitimate requirement, and SS7.21.2's own subsetting idiom is a spec-sanctioned way to state it against an existing constraint. Second, that legitimacy does not by itself make this requirement's own NEED independently motivated: it is honest to say the need was not antecedent -- it exists because this chapter's own traceability analysis found deliveredEnergyBoundedBySupply tied to nothing, and EnergyConservationReq was built specifically to close that gap. Third, the tie this requirement actually has is purely structural -- the subsetting relationship itself, detected by Check B -- and deliberately NOT reinforced by an assert-satisfy claim: two separate, direct tests confirm that register would not do real evaluative work here, run above for real rather than merely argued -- model.verify_satisfaction() errors outright on all three bindings tested, and sysmlv2 verify --solve, while it DOES emit a real verdict once assert satisfy is added back (undecided, not nothing), never resolves it to satisfied either. requirement_coverage()'s own covered=False for energyConservationReq is therefore correct and expected, not a residual gap: there is no assert-satisfy declaration for it to find, by design.\n" ] } @@ -1182,14 +1228,14 @@ }, { "cell_type": "code", - "execution_count": 21, + "execution_count": 22, "id": "0cce3210", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T02:17:54.578501Z", - "iopub.status.busy": "2026-10-01T02:17:54.578417Z", - "iopub.status.idle": "2026-10-01T02:17:54.581209Z", - "shell.execute_reply": "2026-10-01T02:17:54.580776Z" + "iopub.execute_input": "2026-10-01T06:45:44.188435Z", + "iopub.status.busy": "2026-10-01T06:45:44.188339Z", + "iopub.status.idle": "2026-10-01T06:45:44.190948Z", + "shell.execute_reply": "2026-10-01T06:45:44.190597Z" } }, "outputs": [ @@ -1252,14 +1298,14 @@ }, { "cell_type": "code", - "execution_count": 22, - "id": "76642f16", + "execution_count": 23, + "id": "e14ede2b", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T02:17:54.582372Z", - "iopub.status.busy": "2026-10-01T02:17:54.582240Z", - "iopub.status.idle": "2026-10-01T02:17:54.584757Z", - "shell.execute_reply": "2026-10-01T02:17:54.584376Z" + "iopub.execute_input": "2026-10-01T06:45:44.192112Z", + "iopub.status.busy": "2026-10-01T06:45:44.192045Z", + "iopub.status.idle": "2026-10-01T06:45:44.337809Z", + "shell.execute_reply": "2026-10-01T06:45:44.337192Z" } }, "outputs": [ @@ -1267,15 +1313,19 @@ "name": "stdout", "output_type": "stream", "text": [ + "Model tag: {'tag': 'ToasterDemo::acC10Tag', 'identifier': 'AC-C10', 'annotated_element': 'ToasterDemo::EnergyConservationReq'}\n", "Validation errors: []\n" ] } ], "source": [ + "from toaster.query import get_review_record_refs\n", + "\n", "ac_c10 = ReviewRecord(\n", " identifier=\"AC-C10\",\n", " kind=\"asserted_context\",\n", " claim=ac_c10_claim,\n", + " subject_ref=ac_c10_subject_ref,\n", " model_ref=ac_c10_model_ref,\n", " content_hash=hash_content(source),\n", " scope=ac_c10_scope,\n", @@ -1292,17 +1342,19 @@ " record_kind=\"worked_example\",\n", ")\n", "\n", - "errors = validate_record(ac_c10)\n", + "errors = validate_record(ac_c10, model=model)\n", + "tag = next((t for t in get_review_record_refs(model) if t[\"identifier\"] == ac_c10.identifier), None)\n", + "print(f\"Model tag: {tag}\")\n", "print(f\"Validation errors: {errors}\")\n", "assert errors == []\n" ] }, { "cell_type": "markdown", - "id": "286b26f6", + "id": "4faa9b32", "metadata": {}, "source": [ - "`validate_record` reports no errors. `AC-C10` is what this tie actually is: a spec-legitimate construct, a genuine general motivation, and an honestly disclosed assurance deficit (Hawkins' own term, AGENTS.md SS1.6) about when its own need was actually identified and how far a purely structural tie actually reaches -- not a defect papered over, a real judgment an accountable engineer should weigh. Notebook 03's own synthesis cites this record by identifier, the same way it already cites `AS-C06`/`AS-C08`/`AI-C06`." + "The tag printed above is now part of the loaded model, and `validate_record` confirms the Python record and the model's own tag agree about what AC-C10 is actually about, with no errors reported. `AC-C10` is what this tie actually is: a spec-legitimate construct, a genuine general motivation, and an honestly disclosed assurance deficit (Hawkins' own term, AGENTS.md §1.6) about when its own need was actually identified and how far a purely structural tie actually reaches -- not a defect papered over, a real judgment an accountable engineer should weigh. Notebook 03's own synthesis cites this record by identifier, the same way it already cites `AS-C06`/`AS-C08`/`AI-C06`." ] }, { diff --git a/chapters/ch10-traceability-signoff/02-judgment-synthesis.ipynb b/chapters/ch10-traceability-signoff/02-judgment-synthesis.ipynb index 9642bf6..9724612 100644 --- a/chapters/ch10-traceability-signoff/02-judgment-synthesis.ipynb +++ b/chapters/ch10-traceability-signoff/02-judgment-synthesis.ipynb @@ -24,10 +24,10 @@ "id": "407bfd17", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T01:57:35.797376Z", - "iopub.status.busy": "2026-10-01T01:57:35.797186Z", - "iopub.status.idle": "2026-10-01T01:57:36.046215Z", - "shell.execute_reply": "2026-10-01T01:57:36.045607Z" + "iopub.execute_input": "2026-10-01T06:46:03.311478Z", + "iopub.status.busy": "2026-10-01T06:46:03.311303Z", + "iopub.status.idle": "2026-10-01T06:46:03.542964Z", + "shell.execute_reply": "2026-10-01T06:46:03.542471Z" } }, "outputs": [], @@ -53,13 +53,13 @@ { "cell_type": "code", "execution_count": 2, - "id": "2e31dd54", + "id": "1a7f5931", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T01:57:36.047995Z", - "iopub.status.busy": "2026-10-01T01:57:36.047741Z", - "iopub.status.idle": "2026-10-01T01:57:36.051220Z", - "shell.execute_reply": "2026-10-01T01:57:36.050474Z" + "iopub.execute_input": "2026-10-01T06:46:03.544767Z", + "iopub.status.busy": "2026-10-01T06:46:03.544600Z", + "iopub.status.idle": "2026-10-01T06:46:03.647602Z", + "shell.execute_reply": "2026-10-01T06:46:03.647172Z" } }, "outputs": [ @@ -78,6 +78,7 @@ " identifier=\"AI-BAD\",\n", " kind=\"asserted_inference\",\n", " claim=\"The judgment ledger is complete.\",\n", + " subject_ref=\"ToasterDemo::HeatingAssembly::heatGen\",\n", " model_ref=\"ToasterDemo\",\n", " content_hash=hash_content(source),\n", " scope=\"ToasterDemo\",\n", @@ -87,8 +88,8 @@ " counterevidence=\"No central registry exists to check completeness against.\",\n", " record_kind=\"worked_example\",\n", ")\n", - "errors = validate_record(incomplete_inference)\n", - "assert len(errors) > 0, \"Expected validation to fail on empty premises\"\n", + "errors = validate_record(incomplete_inference, model=model)\n", + "assert errors == [\"asserted_inference requires at least one premise (Hawkins \\u00a73.1)\"], errors\n", "print(f\"Negative control ok: errors={errors}\")\n" ] }, @@ -103,13 +104,13 @@ { "cell_type": "code", "execution_count": 3, - "id": "52353610", + "id": "e33ddbcd", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T01:57:36.052529Z", - "iopub.status.busy": "2026-10-01T01:57:36.052432Z", - "iopub.status.idle": "2026-10-01T01:57:36.056858Z", - "shell.execute_reply": "2026-10-01T01:57:36.056525Z" + "iopub.execute_input": "2026-10-01T06:46:03.649006Z", + "iopub.status.busy": "2026-10-01T06:46:03.648896Z", + "iopub.status.idle": "2026-10-01T06:46:03.722306Z", + "shell.execute_reply": "2026-10-01T06:46:03.721824Z" } }, "outputs": [ @@ -133,6 +134,7 @@ " \"alternative (a gas burner, the tongs-and-blowtorch alternative this \"\n", " \"tutorial already contrasts) as the mechanism HeatGenerator commits to.\"\n", " ),\n", + " subject_ref=\"ToasterDemo::ResistanceCoil\",\n", " model_ref=\"ToasterDemo::ResistanceCoil\",\n", " content_hash=hash_content(ch06_source),\n", " scope=\"ToasterDemo::HeatGenerator and its realizations\",\n", @@ -195,7 +197,7 @@ " record_kind=\"worked_example\",\n", ")\n", "\n", - "errors = validate_record(as_c06)\n", + "errors = validate_record(as_c06, model=model)\n", "print(f\"AS-C06 validation errors: {errors}\")\n" ] }, @@ -210,13 +212,13 @@ { "cell_type": "code", "execution_count": 4, - "id": "98f560da", + "id": "d3116a32", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T01:57:36.058133Z", - "iopub.status.busy": "2026-10-01T01:57:36.058002Z", - "iopub.status.idle": "2026-10-01T01:57:36.062079Z", - "shell.execute_reply": "2026-10-01T01:57:36.061742Z" + "iopub.execute_input": "2026-10-01T06:46:03.723680Z", + "iopub.status.busy": "2026-10-01T06:46:03.723571Z", + "iopub.status.idle": "2026-10-01T06:46:03.796000Z", + "shell.execute_reply": "2026-10-01T06:46:03.795615Z" } }, "outputs": [ @@ -240,6 +242,7 @@ " \"definition, holds for every value of efficiency in [0,1] and every non-negative \"\n", " \"power and duration a companion restatement admits.\"\n", " ),\n", + " subject_ref=\"ToasterDemo::deliveredEnergyBoundedBySupply\",\n", " model_ref=\"ToasterDemo::deliveredEnergyBoundedBySupply\",\n", " content_hash=hash_content(ch08_source),\n", " scope=(\n", @@ -302,7 +305,7 @@ " record_kind=\"worked_example\",\n", ")\n", "\n", - "errors = validate_record(as_c08)\n", + "errors = validate_record(as_c08, model=model)\n", "print(f\"AS-C08 validation errors: {errors}\")\n" ] }, @@ -320,10 +323,10 @@ "id": "197f097d", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T01:57:36.063341Z", - "iopub.status.busy": "2026-10-01T01:57:36.063257Z", - "iopub.status.idle": "2026-10-01T01:57:36.195791Z", - "shell.execute_reply": "2026-10-01T01:57:36.195316Z" + "iopub.execute_input": "2026-10-01T06:46:03.797292Z", + "iopub.status.busy": "2026-10-01T06:46:03.797216Z", + "iopub.status.idle": "2026-10-01T06:46:03.909661Z", + "shell.execute_reply": "2026-10-01T06:46:03.909281Z" } }, "outputs": [ @@ -363,13 +366,13 @@ { "cell_type": "code", "execution_count": 6, - "id": "476b8205", + "id": "39918a88", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T01:57:36.196876Z", - "iopub.status.busy": "2026-10-01T01:57:36.196785Z", - "iopub.status.idle": "2026-10-01T01:57:36.200658Z", - "shell.execute_reply": "2026-10-01T01:57:36.200111Z" + "iopub.execute_input": "2026-10-01T06:46:03.911057Z", + "iopub.status.busy": "2026-10-01T06:46:03.910941Z", + "iopub.status.idle": "2026-10-01T06:46:03.982531Z", + "shell.execute_reply": "2026-10-01T06:46:03.982183Z" } }, "outputs": [ @@ -393,6 +396,7 @@ " \"candidates, rated (True) and weak (False), a genuine satisfaction \"\n", " \"check on an underived threshold, not an unevaluated assertion.\"\n", " ),\n", + " subject_ref=\"ToasterDemo::HeatingAssembly::heatGen\",\n", " model_ref=\"ToasterDemo::HeatingAssembly::heatGen\",\n", " content_hash=hash_content(ch06_source),\n", " scope=\"ToasterDemo::HeatingAssembly::heatGen and its realizations\",\n", @@ -465,7 +469,7 @@ " record_kind=\"worked_example\",\n", ")\n", "\n", - "errors = validate_record(ai_c06)\n", + "errors = validate_record(ai_c06, model=model)\n", "print(f\"AI-C06 validation errors: {errors}\")\n" ] }, @@ -483,10 +487,10 @@ "id": "90180fe1", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T01:57:36.201681Z", - "iopub.status.busy": "2026-10-01T01:57:36.201601Z", - "iopub.status.idle": "2026-10-01T01:57:36.209113Z", - "shell.execute_reply": "2026-10-01T01:57:36.208737Z" + "iopub.execute_input": "2026-10-01T06:46:03.983757Z", + "iopub.status.busy": "2026-10-01T06:46:03.983675Z", + "iopub.status.idle": "2026-10-01T06:46:03.997410Z", + "shell.execute_reply": "2026-10-01T06:46:03.997027Z" } }, "outputs": [ diff --git a/chapters/ch10-traceability-signoff/03-engineering-signoff.ipynb b/chapters/ch10-traceability-signoff/03-engineering-signoff.ipynb index 4ac2146..634c677 100644 --- a/chapters/ch10-traceability-signoff/03-engineering-signoff.ipynb +++ b/chapters/ch10-traceability-signoff/03-engineering-signoff.ipynb @@ -24,10 +24,10 @@ "id": "bc09af84", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T01:57:37.141791Z", - "iopub.status.busy": "2026-10-01T01:57:37.141564Z", - "iopub.status.idle": "2026-10-01T01:57:37.372770Z", - "shell.execute_reply": "2026-10-01T01:57:37.372253Z" + "iopub.execute_input": "2026-10-01T06:55:39.074275Z", + "iopub.status.busy": "2026-10-01T06:55:39.074041Z", + "iopub.status.idle": "2026-10-01T06:55:39.212768Z", + "shell.execute_reply": "2026-10-01T06:55:39.212260Z" } }, "outputs": [], @@ -53,13 +53,13 @@ { "cell_type": "code", "execution_count": 2, - "id": "0d0af422", + "id": "bf18b711", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T01:57:37.374192Z", - "iopub.status.busy": "2026-10-01T01:57:37.374027Z", - "iopub.status.idle": "2026-10-01T01:57:37.376834Z", - "shell.execute_reply": "2026-10-01T01:57:37.376444Z" + "iopub.execute_input": "2026-10-01T06:55:39.214410Z", + "iopub.status.busy": "2026-10-01T06:55:39.214220Z", + "iopub.status.idle": "2026-10-01T06:55:39.217286Z", + "shell.execute_reply": "2026-10-01T06:55:39.216838Z" } }, "outputs": [ @@ -88,8 +88,8 @@ " disposition=\"pending\",\n", " record_kind=\"worked_example\",\n", ")\n", - "errors = validate_record(draft_missing_counterevidence)\n", - "assert len(errors) > 0, \"Expected validation to fail on empty counterevidence\"\n", + "errors = validate_record(draft_missing_counterevidence, model=model)\n", + "assert errors == [\"counterevidence is empty\"], errors\n", "print(f\"Negative control ok: errors={errors}\")\n" ] }, @@ -98,7 +98,7 @@ "id": "97ad4325", "metadata": {}, "source": [ - "`validate_record()` correctly rejects this: a real, mechanized floor actually catching something, the same check Chapter 9's own placeholder record exercised. A different rule this tutorial follows just as strictly has no such mechanized floor: AGENTS.md 1.6 forbids ever recording `disposition=\"accepted\"` (SA-7), but reading `src/toaster/evidence.py`'s own `validate_record()` shows what it actually checks: `identifier`, `claim`, `rationale` and `counterevidence` non-empty, `record_kind != \"actual_review\"`, and at least one premise for an `asserted_inference`, never the `disposition` value itself. That gap is real and worth naming plainly (recorded in `decisions/next-passes.md`), but it is not demonstrated here by constructing the forbidden value: AGENTS.md's own rule against `disposition=\"accepted\"` has no negative-control exception, so this point is made in prose only, never in code. Every record this notebook goes on to build keeps `disposition=\"pending\"`, checked explicitly at the end, not merely asserted." + "`validate_record()` correctly rejects this: a real, mechanized floor actually catching something, the same check Chapter 9's own placeholder record exercised. A different rule this tutorial follows just as strictly has no such mechanized floor: AGENTS.md 1.6 forbids ever recording `disposition=\"accepted\"` (SA-7), but reading `src/toaster/evidence.py`'s own `validate_record()` shows what it actually checks: `identifier`, `claim`, `rationale` and `counterevidence` non-empty, `record_kind != \"actual_review\"`, at least one premise for an `asserted_inference`, and `subject_ref` present (resolving in the model and agreeing with any model-side tag, when a model is given) for `asserted_context`/`asserted_solution` or an `asserted_inference` with no premises -- never the `disposition` value itself. That gap is real and worth naming plainly (recorded in `decisions/next-passes.md`), but it is not demonstrated here by constructing the forbidden value: AGENTS.md's own rule against `disposition=\"accepted\"` has no negative-control exception, so this point is made in prose only, never in code. Every record this notebook goes on to build keeps `disposition=\"pending\"`, checked explicitly at the end, not merely asserted." ] }, { @@ -115,10 +115,10 @@ "id": "50c87f41", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T01:57:37.377985Z", - "iopub.status.busy": "2026-10-01T01:57:37.377890Z", - "iopub.status.idle": "2026-10-01T01:57:37.475511Z", - "shell.execute_reply": "2026-10-01T01:57:37.475132Z" + "iopub.execute_input": "2026-10-01T06:55:39.218961Z", + "iopub.status.busy": "2026-10-01T06:55:39.218869Z", + "iopub.status.idle": "2026-10-01T06:55:39.325429Z", + "shell.execute_reply": "2026-10-01T06:55:39.324971Z" } }, "outputs": [ @@ -177,10 +177,10 @@ "id": "a22e1d19", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T01:57:37.476657Z", - "iopub.status.busy": "2026-10-01T01:57:37.476580Z", - "iopub.status.idle": "2026-10-01T01:57:37.478384Z", - "shell.execute_reply": "2026-10-01T01:57:37.478064Z" + "iopub.execute_input": "2026-10-01T06:55:39.326786Z", + "iopub.status.busy": "2026-10-01T06:55:39.326651Z", + "iopub.status.idle": "2026-10-01T06:55:39.328729Z", + "shell.execute_reply": "2026-10-01T06:55:39.328378Z" } }, "outputs": [ @@ -197,16 +197,24 @@ "print(f\"AC-C10: {ac_c10_summary}\")\n" ] }, + { + "cell_type": "markdown", + "id": "0f79a98d", + "metadata": {}, + "source": [ + "Notebook 02's own three-record ledger is cited next, the same way -- by identifier and conclusion, not rebuilt a third time." + ] + }, { "cell_type": "code", "execution_count": 5, "id": "27d95537", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T01:57:37.479376Z", - "iopub.status.busy": "2026-10-01T01:57:37.479305Z", - "iopub.status.idle": "2026-10-01T01:57:37.481205Z", - "shell.execute_reply": "2026-10-01T01:57:37.480896Z" + "iopub.execute_input": "2026-10-01T06:55:39.329933Z", + "iopub.status.busy": "2026-10-01T06:55:39.329839Z", + "iopub.status.idle": "2026-10-01T06:55:39.332063Z", + "shell.execute_reply": "2026-10-01T06:55:39.331666Z" } }, "outputs": [ @@ -252,10 +260,10 @@ "id": "ac3bd423", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T01:57:37.482266Z", - "iopub.status.busy": "2026-10-01T01:57:37.482199Z", - "iopub.status.idle": "2026-10-01T01:57:37.484016Z", - "shell.execute_reply": "2026-10-01T01:57:37.483662Z" + "iopub.execute_input": "2026-10-01T06:55:39.333254Z", + "iopub.status.busy": "2026-10-01T06:55:39.333175Z", + "iopub.status.idle": "2026-10-01T06:55:39.335282Z", + "shell.execute_reply": "2026-10-01T06:55:39.334788Z" } }, "outputs": [ @@ -301,10 +309,10 @@ "id": "a911e6fc", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T01:57:37.485072Z", - "iopub.status.busy": "2026-10-01T01:57:37.485001Z", - "iopub.status.idle": "2026-10-01T01:57:37.487155Z", - "shell.execute_reply": "2026-10-01T01:57:37.486850Z" + "iopub.execute_input": "2026-10-01T06:55:39.336446Z", + "iopub.status.busy": "2026-10-01T06:55:39.336351Z", + "iopub.status.idle": "2026-10-01T06:55:39.338720Z", + "shell.execute_reply": "2026-10-01T06:55:39.338369Z" } }, "outputs": [ @@ -359,10 +367,10 @@ "id": "d8671455", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T01:57:37.488167Z", - "iopub.status.busy": "2026-10-01T01:57:37.488101Z", - "iopub.status.idle": "2026-10-01T01:57:37.490464Z", - "shell.execute_reply": "2026-10-01T01:57:37.490149Z" + "iopub.execute_input": "2026-10-01T06:55:39.339840Z", + "iopub.status.busy": "2026-10-01T06:55:39.339756Z", + "iopub.status.idle": "2026-10-01T06:55:39.342640Z", + "shell.execute_reply": "2026-10-01T06:55:39.342220Z" } }, "outputs": [ @@ -442,10 +450,10 @@ "id": "1b5ed64c", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T01:57:37.491494Z", - "iopub.status.busy": "2026-10-01T01:57:37.491397Z", - "iopub.status.idle": "2026-10-01T01:57:37.493760Z", - "shell.execute_reply": "2026-10-01T01:57:37.493345Z" + "iopub.execute_input": "2026-10-01T06:55:39.343771Z", + "iopub.status.busy": "2026-10-01T06:55:39.343692Z", + "iopub.status.idle": "2026-10-01T06:55:39.346163Z", + "shell.execute_reply": "2026-10-01T06:55:39.345828Z" } }, "outputs": [ @@ -514,10 +522,10 @@ "id": "2cf9442d", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T01:57:37.494762Z", - "iopub.status.busy": "2026-10-01T01:57:37.494687Z", - "iopub.status.idle": "2026-10-01T01:57:37.497298Z", - "shell.execute_reply": "2026-10-01T01:57:37.496968Z" + "iopub.execute_input": "2026-10-01T06:55:39.347258Z", + "iopub.status.busy": "2026-10-01T06:55:39.347175Z", + "iopub.status.idle": "2026-10-01T06:55:39.349595Z", + "shell.execute_reply": "2026-10-01T06:55:39.349288Z" } }, "outputs": [ @@ -585,13 +593,13 @@ { "cell_type": "code", "execution_count": 11, - "id": "f36c2e13", + "id": "a7aef295", "metadata": { "execution": { - "iopub.execute_input": "2026-10-01T01:57:37.498284Z", - "iopub.status.busy": "2026-10-01T01:57:37.498213Z", - "iopub.status.idle": "2026-10-01T01:57:37.505659Z", - "shell.execute_reply": "2026-10-01T01:57:37.505295Z" + "iopub.execute_input": "2026-10-01T06:55:39.351125Z", + "iopub.status.busy": "2026-10-01T06:55:39.351033Z", + "iopub.status.idle": "2026-10-01T06:55:39.428151Z", + "shell.execute_reply": "2026-10-01T06:55:39.427803Z" } }, "outputs": [ @@ -599,6 +607,14 @@ "name": "stdout", "output_type": "stream", "text": [ + "Model tag: None" + ] + }, + { + "name": "stdout", + "output_type": "stream", + "text": [ + "\n", "Validation errors: []\n", "identifier='AI-C10' kind='asserted_inference'\n", "disposition='pending' record_kind='worked_example'\n", @@ -607,6 +623,8 @@ } ], "source": [ + "from toaster.query import get_review_record_refs\n", + "\n", "signoff = ReviewRecord(\n", " identifier=\"AI-C10\",\n", " kind=\"asserted_inference\",\n", @@ -627,17 +645,29 @@ " record_kind=\"worked_example\",\n", ")\n", "\n", - "errors = validate_record(signoff)\n", + "errors = validate_record(signoff, model=model)\n", + "tag = next((t for t in get_review_record_refs(model) if t[\"identifier\"] == signoff.identifier), None)\n", + "print(f\"Model tag: {tag}\")\n", "print(f\"Validation errors: {errors}\")\n", "print(f\"identifier={signoff.identifier!r} kind={signoff.kind!r}\")\n", "print(f\"disposition={signoff.disposition!r} record_kind={signoff.record_kind!r}\")\n", "print(f\"engineering_conclusion={signoff.engineering_conclusion!r}\")\n", "assert errors == []\n", + "assert not any(\"subject_ref\" in e for e in errors)\n", + "assert tag is None\n", "assert signoff.disposition == \"pending\"\n", "assert signoff.record_kind == \"worked_example\"\n", "conn.close()\n" ] }, + { + "cell_type": "markdown", + "id": "604ba013", + "metadata": {}, + "source": [ + "`AI-C10` carries no `subject_ref` and gets no `ReviewRecordRef` tag: the exemption `validate_record()` grants is this tutorial's own rule for the pure cross-record synthesis case -- an `asserted_inference` whose own `premises` are non-empty may leave `subject_ref` unset -- and `AI-C10`'s own `premises`, printed above, really are non-empty: five real findings, not a placeholder list. `Model tag: None` and zero `subject_ref`-related validation errors, both confirmed directly above against the real loaded model, are what that exemption actually looks like when exercised for real, not merely argued. `AI-C10` is not the only record the rule could apply to -- `AI-C10-DRAFT` above and `AI-C04` (Chapter 4) both have non-empty premises too, and the rule would exempt either the same way -- it is simply the one record in this chapter that actually uses the exemption rather than carrying a real anchor anyway." + ] + }, { "cell_type": "markdown", "id": "7d87d697", diff --git a/decisions/log.md b/decisions/log.md index a1fa0e0..058411c 100644 --- a/decisions/log.md +++ b/decisions/log.md @@ -987,3 +987,179 @@ Reasoning: two independently built candidate designs, neither aware of the other Determined: yes. Extension: no — resolves a design question two independently-built candidates left open (which representation of the tie is correct), using the trade-study and escalation machinery `ace-protocol`/`orchestrator-protocol` already provide; reopens no Standing Assumption. Reverting Approach A's own merged commit (85b7778, prior to this entry) was a mechanical cleanup of an already-identified defect, not a new design decision in itself. Provenance: `docs/case-studies/2026-09-30-energy-conservation-requirement-tie.md` (the full working: the trade study, the six-angle interrogation of Check A, the empirical confirmation, the steelman considered and set aside — this entry points to it rather than restating it); `decisions/log.md` DL-070, DL-071 (the search this tie closes the gap for); `decisions/next-passes.md` item 29 (closed by this entry); SysML v2 formal/2026-03-02 §7.20.1–7.21.4; `models/ch10-cumulative.sysml`; `chapters/ch10-traceability-signoff/01-traceability-graph.ipynb`, `02-judgment-synthesis.ipynb`, `03-engineering-signoff.ipynb`, `index.md`, `conclusion.md`; `chapters/ch09-coverage-sufficiency/01-requirement-coverage.ipynb`, `index.md`; `exercises/ch10/exercise.ipynb`; `tests/test_query.py`. + +## DL-073 | 2026-10-01 | USER-TESTING-SKILL-FIX | COMPLETE: `user-testing` skill fixed for two process-level bugs the Pass 3 battery itself surfaced but which the skill (unchanged since) never absorbed + +Path: Handled by ACE (Z directed, in chat, after reviewing the Pass 3 battery's drift: "fix the user-testing skill's known issues first", ahead of a planned fresh battery run against current `main`) +Decision: Two concrete, already-diagnosed process bugs in `.claude/skills/user-testing/SKILL.md`, confirmed still present (`git diff bcc4796 origin/main -- .claude/skills/user-testing/` is empty — the skill has not changed since the Pass 3 battery that found these): (1) the "Execution checklist" and "Report format" sections name cells by a fixed numeric index ("Cell 0", "Cell 5", "Cell 6", ...), which assumes every chapter notebook has exactly the 7-cell minimum skeleton; this already contradicts `toaster-recipe`'s own established rule ("Additional markdown+code pairs may be inserted... A6 reviews for skeleton completeness by content type, not by cell index") and does not accommodate a construction-zone notebook's multiple fragment cells, a judgment-record notebook's 17-27 real cells, or Chapter 10's own distributed-seam design (seam carried by prose across several query cells, not one dedicated cell) — all real shapes already in the real tutorial. (2) The skill gives the Novice persona (Haiku 4.5) no rule for what "exactly one sentence" means beyond the words themselves, and the Pass 3 battery recorded (per the project's own memory of that battery, decisions/log.md DL-053 through DL-062 on `pass3/user-testing`) that persona repeatedly misjudging a real, single, semicolon-joined sentence as "two sentences" across four different chapters (Ch1, Ch7, Ch8, Ch9) — every instance checked out as a genuine single sentence, a persona-calibration false positive, not a content defect. Intended change: (1) rewrite the "Execution checklist"'s 9 steps and the "Report format"'s "STRUCTURAL CHECKS" block to identify each item by what it does ("the concept-statement cell", "the model-increment cell(s)", "the seam cell(s)", etc.) rather than by fixed index, with one prefatory sentence establishing the content-type rule and citing `toaster-recipe`'s own precedent; (2) add one parenthetical to the concept-statement check giving a concrete, checkable rule for "one sentence" (count terminal periods, not semicolons or conjunctions). Neither change touches the persona table, the model assignments, the ACE synthesis protocol, the pass bar, or what counts as blocking/not-blocking — those are unaffected by this fix and not touched. +Blast-radius check (`skill-editor` Step 2): does not affect more than one archetype's primary skill (user-testing is `simulated-learner`/`ace`-only); does not remove a prohibition; does not add a capability beyond the original design (it corrects how an existing checklist locates existing content); does not touch `sysml-v2-toaster-model`'s construct list; does not affect a learning outcome (this changes how the TESTER locates chapter content, not what a learner sees). No escalation triggered. +Revert record: `git show 27c3f35:.claude/skills/user-testing/SKILL.md` (the file's full content immediately before this edit, 101 lines). +Determined: yes. +Extension: no — corrects an ambiguous, already-incorrect assumption in a builder-facing testing skill (ACE-fixable per `skill-editor`'s own list: "clarification of an ambiguous rule that a developer demonstrably misread" and "tighter prohibition derived from an observed error pattern"); touches no Standing Assumption, no learner-facing content, and no persona/model assignment. +What changed: the "Execution checklist" (9 steps) and both parts of the "Report format" (`STRUCTURAL CHECKS` and `EXECUTION RESULTS`) now identify cells by content type ("the concept-statement cell", "the model-increment cell(s)", "the seam cell(s)", etc.) rather than fixed numeric index, with one prefatory sentence explaining why (construction-zone and judgment-record notebooks have more cells; Chapter 10 carries the seam across several cells' prose rather than one dedicated cell) and citing `toaster-recipe`'s own, already-established "by content type, not cell index" rule as precedent. The concept-statement check gained one parenthetical giving a concrete rule for "one sentence" (count terminal periods, not semicolons or conjunctions) to close the Novice persona's own recorded false-positive pattern. Nothing else in the file changed: personas, model assignments, the ACE synthesis protocol, and both "what counts as blocking" lists are untouched. +Post-edit check: re-read the modified sections and both adjacent sections (`Personas and model assignment` above, `ACE synthesis protocol` below) — no adjacent rule weakened or contradicted. `uv run python -m glossary check` — ok, 0 errors, same 7 pre-existing warnings. Found, and deliberately did not fix (different file, out of scope for this entry, would need its own DL pre-edit gate): `.claude/skills/tutorial-style-guide/SKILL.md` lines 70-72 carry the identical fixed-cell-index assumption ("Cell 0", "Cell 5", "Cell 6") independently — flagged for whoever next touches that skill, not fixed here. +Provenance: `decisions/next-passes.md` (the Pass 3 battery's own now-superseded findings this entry draws the two process-level bugs from, per this session's own direct review of how far `pass3/user-testing` has drifted from current `main`); `.claude/skills/toaster-recipe/SKILL.md` ("by content type, not cell index", already-established precedent this fix now applies consistently in `user-testing` too); `.claude/skills/skill-editor/SKILL.md` (the gate this entry follows). + +## DL-074 | 2026-10-01 | FRESH-UT-CH01 | Chapter 1 fresh user-testing checkpoint: PASS + +Path: Handled by ACE +Decision: CHECKPOINT PASS for chapters/ch01-system-purpose (two personas: Novice on Haiku 4.5, SE Practitioner on Sonnet 5; no Returning Learner, there being no prior chapter). Zero blocking findings. One minor observation (nb01 cell 8 prints TOASTER_INCREMENT but loads the cumulative file; cell 9's prose bridges this; nb04 cell-03 discloses the mechanism) and one cosmetic observation (nb01 cell 16 "printed below each" versus outputs sitting above the cell) are recorded for Z and not acted on, per the user-testing synthesis protocol. Proceed to Chapter 2. +Principles applied: P6 (rule where determined), P5 (verify before asserting: own run, own model query, check_construction), P4 and AGENTS.md 1.10 (lens never named; seam judged as emergent behavior), F2, F3, F7 and heuristics 1, 5 (layer audit of the chapter's elements), DL-037 confirmed extension (naming of HeatingSystem/ControlSystem). +Reasoning: (1) The user-testing skill (c60037b, DL-073) fixes the pass bar: zero blocking findings after triage. (2) Both persona reports report zero NEEDS-FIX; ACE re-ran nb01 (7 code cells, all OK; bad.ok False with diagnostic "expected a /* ... */ comment body"; find() resolves ToastingSystem partDef and ToastBread actionDef) and queried the cumulative model directly (parts heating/control, attribute cycleTime unvalued, specializes ToastingSystem, six element kinds as index.md states); check_construction.py --check exits 0. (3) Each of the five blocking criteria is checked and none fires; criterion 5 is not applicable to Chapter 1. (4) Seam: the printed SysML text, the loader's accept (cell 8) and reject (cell 10), and the query results (cells 12, 14) are each shown as distinct visible steps and tied together in cell 16 without naming the lens; a reader not told of "three worlds" would still see text in, loader verdict, query confirms. (5) Layer audit: ToastBread passes the substitution test (functional); ToastingSystem is the bare named subject carrying the purpose (F7); cycleTime is an unvalued slot (result, not choice); subsystem names carry function, not mechanism. No misfiled element. (6) The one minor and one cosmetic observation fall under "does NOT count as blocking" and are left for Z's direction. +Determined: yes. +Extension: no (the established checkpoint pass bar applied to Chapter 1 as written; two-persona convention for the first chapter follows from the Returning Learner persona's definition). +Provenance: `.claude/skills/user-testing/SKILL.md` @c60037b (ACE synthesis protocol, blocking criteria); AGENTS.md Part 1 SS1.6, SS1.10; `decisions/user-testing/ch01-novice.md` @69849ec; `decisions/user-testing/ch01-se-practitioner.md` @6a8932e; ACE run of `chapters/ch01-system-purpose/01-abstract-def.ipynb` and direct query of `models/ch01-cumulative.sysml` on 2026-10-01; `scripts/check_construction.py --check` exit 0; DL-028 (tall-named lint), DL-037 (naming extension), DL-073 (skill revision). + +## DL-075 | 2026-10-01 | FRESH-UT-CH02 | Checkpoint PASS — Ch2 (Requirements and Assumptions) fresh user-test synthesis; zero blocking items; the recurring judgment-record seam reading escalated to Z (fourth occurrence; two prior ACE syntheses on the superseded pass3/user-testing branch disagreed) + +Path: Handled by ACE — user-test finding (checkpoint verdict); Escalated to Z (judgment-record seam reading) +Decision: CHECKPOINT PASS for chapters/ch02-requirements (01-requirement-def, 02-assumptions, 03-judgment-context, index.md, conclusion.md). Three persona reports (Novice on Haiku 4.5, SE Practitioner and Returning Learner on Sonnet 5; decisions/user-testing/ch02-{novice,se-practitioner,returning-learner}.md at commits 6550b56, 2547ceb, 3e5e986) each OVERALL: PASS, no NEEDS-FIX. ACE read all three in full and re-ran every code cell of 03-judgment-context.ipynb in order from the notebook's directory (exit 0), reproducing every claimed figure (model.ok=True; bad.ok=False with "unresolved member: nonExistentAttr"; AC-001 asserted_context, disposition pending, validate_record []). Every blocking criterion checked and not met. Not ruled, escalated: whether 1.10's seam in a judgment-record notebook is satisfied by the record's fields / validate_record / printed [] (pass3 DL-060 reading) or requires the SysML-text / loader-or-eval / printed-result triad in the closing sentence (pass3 DL-058 reading); full brief below, recommended default A. Carried, not decided (user-testing step 3: minor items need Z): nb03's load is silent (assert only, no printed query before the judgment side; same as pass3 DL-060's minor). Side note, not decided: the "// GENERATED FIXTURE -- do not edit directly" header of models/ch02-cumulative.sysml is printed into learner output in all three notebooks; untracked in next-passes.md/pass4-backlog.md/DEFERRED.md; cosmetic. ACE fresh observation: nb03 cell 17's `model_ref` and `content_hash=hash_content(source)` bind the record to the printed SysML text in code and are never narrated, which is why they appear as option C in the brief below. + +Brief (Z's idiom): +DECISION NEEDED: In a judgment-record notebook (no SysML construction of its own; its one construct is a Hawkins ReviewRecord), what are the three things AGENTS.md 1.10's seam must connect? +Objective: a learner who was never told there are three worlds still sees that authored formal text, the tool that runs it, and the printed result are distinct and connected -- as behavior, not prescribed text (1.10). Design space, typed by what counts as "model text": + A. Record triad counts -- the notebook's introduced construct is the record (F4, DL-033), so record fields / validate_record / [] is a genuine instance of the seam; the model side must still be shown loaded somewhere in the notebook (it is). Consequence: no content change; one clarifying clause each in toaster-recipe line 180 and user-testing step 7. + B. SysML triad only -- 1.10 names "model text, the tool that loads it, the rendered result"; judgment notebooks must also close with one sentence tying printed cumulative source / load-or-eval tool / printed model-side result; the record triad is additional. Consequence: one builder contract, one sentence per judgment notebook; recipe unchanged. + C. One bridged seam -- the record already binds to the model in code (`model_ref`, `content_hash = hash_content(source)` of the printed text); the seam sentence narrates that bridge, making the two triads one connection and incidentally fixing the silent-load minor. Consequence: one sentence per judgment notebook, no lens vocabulary. +Feasibility: all three fit SA-7/SA-8 and 1.10's never-name clause. Utility: A keeps the model/analysis split visible (F4); B makes judgment notebooks read like siblings; C does both at one sentence. MoE: can every persona point at the three things (all can today under A; the Practitioner reading of 1.10 says no under B). Judgment: this is a reading of 1.10, not derivable from it; residual uncertainty is whether Z intends "model text" literally (SysML) or as "the formal text this notebook introduces". +Recurrence: fourth independent flag (pass3 Ch2 Returning Learner; pass3 Ch3 Novice and Returning Learner; fresh Ch2 all three personas; fresh Ch3 SE Practitioner, fb7e26b) -- and more occurrences from this same fresh battery landed after this entry was drafted; route all of them to this entry rather than re-escalating. +Addendum (DL-080, Ch4 synthesis): confirmed occurrence in fresh Ch4 Returning Learner (7d89df9); cross-referenced DL-059 on the unmerged `pass3/user-testing` branch (commit 8e8ece9), which had already half-ruled this: the behavioral criterion of 1.10 is met by distributed prose with no dedicated cell (Extension: yes, flagged for Z there too), leaving only the template question (whether `toaster-recipe` should require a dedicated seam cell) open -- that is this escalation's own question, now with a prior partial answer in view. +Addendum (DL-082, Ch3 synthesis): sixth chapter-level occurrence confirmed (pass3 Ch2, pass3 Ch3, pass3 Ch4 [pass3 DL-057], fresh Ch2, fresh Ch4, fresh Ch3 [fb7e26b and aacd660]); nine persona reports in all. Checkpoint verdicts do not depend on Z's answer in any occurrence found so far (the SysML triad is independently present and connected in every judgment notebook checked, so "not addressed in behavior either" fails under both readings regardless); only content changes to judgment-notebook closings await Z's decision. +Addendum (DL-083, Ch9 synthesis, the battery's last chapter): one further occurrence attached, not flagged by any persona: Ch9 nb02's dedicated seam (cell 18) is the record triad with the model loaded in cell 2 and no closing model-side sentence, the same shape as Ch2 nb03 -- passes outright under this entry's own recommended reading A. Ch9 nb03 is NOT an occurrence: it edits the SysML text, re-hashes and re-checks, so its model-text leg is enacted in behavior under either reading. With the fresh battery now complete (all 10 chapters synthesized, DL-074 through DL-083), this escalation remains the single open item from the whole battery awaiting Z's decision; every chapter's checkpoint verdict was independently unaffected by it. +ACE recommendation: A -- 1.10's own second clause makes the seam behavioral rather than a prescribed closing sentence, and F4 makes the record the notebook's construct, so a seam about the record is the correct one for that notebook; if Z instead wants judgment notebooks to read like their siblings, C is the one-sentence form that does it without lens vocabulary. +Addendum (DL-084, closure): retired, not ruled. Z introduced a real, checkable `subject_ref`/`ReviewRecordRef` anchor (DL-084) directly motivated by this escalation's own recommended reading C (the `model_ref`/`content_hash` bridge, never narrated, was the thing both competing readings were really reaching for). Every judgment-record notebook's seam cell now narrates one bridged connection -- the tagged SysML text, the cross-checking tool, and a printed agreement result the reader has just watched -- which is a stronger, literally checkable instance of reading C, not a selection among A/B/C by argument. No ruling was made among A/B/C; the question the escalation posed no longer has anything to be asked about, since the design gives every occurrence found (DL-075 through DL-083, nine persona reports and ten ACE syntheses) the same concrete anchor to narrate. Z's decision below remains accurately recorded as never made among the original three options; DL-084 is the closing entry. +Z's decision: [pending among the original A/B/C options -- superseded, see DL-084 addendum above] +Z's rationale: [pending; if Z later states a principle about judgment-record seams specifically, propose adding it to z-principles.md as a confirmed extension of 1.10/F4] + +Principles applied: user-testing "What counts as blocking" (applied as the checkpoint bar); AGENTS.md 1.10 (both clauses); F4 (a judgment record is analysis, not model); DL-032 and DL-033 applied as prior decisions (slow as labelled failing-branch fixture; nominal takes no layer; the record is not a layer element); P4 and heuristic 6 (the unnarrated hash/model_ref and the silent load are friction, not barriers); P5 (a confirmed load is better shown than asserted); P6 and ace-protocol's ruling test (two reasonable applications reached different answers, so the principles underdetermine the seam reading; a learning outcome is affected). +Reasoning: (a) Blocking criteria each checked against the ACE's run and all three reports: no non-executing cell; all concept statements and exercise pointers one sentence; no seam names Tall or the three worlds (full read plus CI lint); every negative control bad.ok False with the documented diagnostic; cumulative model self-contained from Ch1. (b) Seam, as 1.10 is written: in nb03 the reader sees the model text printed and loaded (cell 2), the same tool reject faulty text with a printed diagnostic (cell 4), prose separating model side from judgment side (cell 5), then the record's text, checker and printed result (cells 7-18); all three personas pointed at three concrete things, so "not addressed in behavior either" is not met and the item is not blocking under either reading. (c) The reading itself is not ruled: 1.10's second clause (behavioral, not prescribed text) favours the record-triad reading; 1.10's first clause names "model text, the tool that loads it, and the rendered result" literally and favours the SysML-triad reading; F4 bears on both. pass3 DL-058 and DL-060 applied these to the same shape and diverged, DL-058 flagging itself as an extension for Z and DL-060 leaving the record-only case open; a fourth occurrence (fresh Ch3 SE Practitioner, fb7e26b) was already on record when this synthesis ran. Check determination fails at that step, so escalate. (d) Checkpoint verdict follows user-testing step 4: zero blocking items, PASS, independent of the escalation. +Determined: checkpoint verdict yes; seam reading no -- fails at Check determination (two principled readings of 1.10's "model text" for a notebook whose construct is a record). +Extension: no ruling made, so none; the escalated question is itself the proposed extension of 1.10 to judgment-record notebooks (first flagged as such on the pass3/user-testing branch). +Provenance: AGENTS.md 1.6, 1.10; `.claude/skills/user-testing/SKILL.md` (checklist step 7, ACE synthesis protocol, blocking and non-blocking lists; revised DL-073); `.claude/skills/toaster-recipe/SKILL.md` lines 103-127 (seam cell behavioural form), 158-163 (judgment record notebooks), 180 (seam checklist item); z-principles.md F4, P4, P5, P6, confirmed extensions DL-032, DL-033; z-model.md Z-9, Z-12, Z-18, Z-19, Z-22; `pass3/user-testing` branch `decisions/log.md` DL-055 (Ch6), DL-058 (Ch3), DL-060 (Ch2) -- that branch's DL-053..062 numbers collide with main's and are cited by branch, not by main's own numbering; main DL-050, DL-053, DL-073; commit fb7e26b (`decisions/user-testing/ch03-se-practitioner.md`, fresh battery, nb03 seam note); learner reports `ch02-novice.md` (6550b56), `ch02-se-practitioner.md` (2547ceb), `ch02-returning-learner.md` (3e5e986); ACE execution of every code cell of `chapters/ch02-requirements/03-judgment-context.ipynb` from the notebook directory, 2026-10-01, exit 0. + +## DL-076 | 2026-10-01 | FRESH-UT-CH07 | Checkpoint PASS — Ch7 fresh user-test battery synthesis (post-DL-073 skill, Pass 4 notebooks) + +Path: Handled by ACE +Decision: CHECKPOINT PASS for Chapter 7 (`chapters/ch07-execution/`). Three `simulated-learner` reports (Novice/Haiku 4.5, SE Practitioner/Sonnet 5, Returning Learner/Sonnet 5), all OVERALL: PASS with zero NEEDS-FIX, plus the ACE's own fresh execution of nb02 (`02-state-traces.ipynb`): all code cells in all three sub-notebooks execute; every negative control gives `bad.ok=False`; `model.ok=True` on the cumulative model throughout; demonstrations match `index.md`'s Expected result (67200 J with the D-033 display form, `ToastingSystem::cycle` as a `stateUsage`, the `[idle, heating, ready, idle]` trace, the 600 W threshold read from the model); concept-statement and exercise-pointer cells are one sentence in all three notebooks; `conclusion.md` is three paragraphs plus exercise reference; the seam is addressed in behavior in every notebook without the lens being named. nb02's typo-trigger cell (`typo_model.ok=True` followed by `language_gap_findings` naming `Strat`) is a deliberate, documented demonstration of a tracked tool gap (D-023), not a leaking negative control. One non-blocking wording observation recorded, not decided (nb02 cell-22 prints "run through nominal" while cell-23 states `performer` has no effect, D-028). Reports: `decisions/user-testing/ch07-novice.md` (1fbdfe9/b91a9b8), `ch07-se-practitioner.md` (cc133d9), `ch07-returning-learner.md` (42377ad). +Principles applied: user-testing pass bar and blocking list (binding process rule, DL-073 version); AGENTS.md 1.10 and P4 (lens never named, never load-bearing); P5 and F6 with confirmed extension DL-039 (tool hole closed by a tutorial-supplied always-on guard, tracked in DEFERRED.md, not papered over); F4 (threshold and relation defined in the model and read from it, not retyped in Python); P1 (traces presented as specification analysis, not proof of behavior in use); heuristic 6 (earn your place) applied to the seam cells. +Reasoning: (1) The skill's blocking list is exhaustive and each item was checked against the three reports and the ACE's own run; none is met. (2) The one cell that could be mistaken for a failing control (`typo_model.ok=True`) is by construction a demonstration: it asserts the tool's acceptance, then asserts the guard's finding, and cites the tracked gap, which is exactly the F6/DL-039 pattern (language conformance defined by the spec, guard supplied where the tool has a hole), so under P5 it is the correct content, not a defect. (3) The seam requirement under 1.10 is behavioral; all three personas and the ACE could point to the SysML text, the tool's load or query, and the printed or plotted result as distinct things in each notebook, with the typo probe being the clearest instance because the three artifacts disagree; no lens vocabulary appears (scoped grep and `glossary check` clean). (4) The remaining observation is cosmetic wording about a tracked gap, which the skill forbids deciding without Z; it is recorded only. (5) Zero blocking issues therefore means CHECKPOINT PASS by the skill's own rule. +Determined: yes. +Extension: no -- applies the existing pass bar and the already-confirmed DL-039 reading of a tool hole to a new chapter; touches no Standing Assumption, no glossary definition, no learner-facing content. +Provenance: `.claude/skills/user-testing/SKILL.md` at c60037b (DL-073); the three persona reports at the commits above; ACE fresh execution of `chapters/ch07-execution/02-state-traces.ipynb` (all 13 code cells clean, outputs as stated); `chapters/ch07-execution/index.md` and `conclusion.md`; `DEFERRED.md` D-023, D-026, D-028, D-033; `decisions/log.md` DL-039 (tool enforcement holes), DL-073 (skill fix); `decisions/next-passes.md` l.62 (Tall-seam note closed); `z-model.md` Z-12 (Tall's three worlds builder-facing, never named in learner content); AGENTS.md 1.10. + +## DL-077 | 2026-10-01 | FRESH-UT-CH10 | Checkpoint PASS — Ch10 user-test synthesis (fresh battery after the DL-072 energy-tie reconciliation) + +Path: Handled by ACE -- user-test finding (three simulated-learner reports plus ACE self-test; zero blocking items) +Decision: CHECKPOINT PASS for Chapter 10 (`chapters/ch10-traceability-signoff/`, notebooks 01-03, index.md, conclusion.md). Three persona reports (Novice on Haiku 4.5, `decisions/user-testing/ch10-novice.md` @94312b7; SE Practitioner on Sonnet 5, `ch10-se-practitioner.md` @5c3acd5; Returning Learner on Sonnet 5, `ch10-returning-learner.md` @13c07aa), each OVERALL: PASS with zero NEEDS-FIX, were reproduced by the ACE's own fresh execution of notebook 01 (53 cells, no errors; every printed value identical to the reports and to index.md's Expected result). Five untriaged items classified: (1) AC-C10's defensibility -- no finding; P1's test passes on every clause (judgment site named, evidence cited cell by cell, counterevidence states the circularity objection and that a reader could reasonably disagree, SA-7 held); the stricter-rule question it surfaces is already recorded as open in AC-C10 and DL-072. (2) Whether a reader must read past the bare `covered=False` line to distinguish `timely` from `energyConservationReq` -- not a defect; the distinction is asserted in the markdown cell immediately after the coverage print (cell 14) and evidenced in the same notebook; residual density is minor, left for Z, no fix ruled. (3) SE Practitioner report at 444 substantive words against the 400 cap -- cosmetic process deviation, no skill edit, left for Z. (4) Ch9's forward reference to `energyConservationReq` confirmed resolved by Ch10 -- positive, no action. (5) Two-tool seam (opensysml `verify_satisfaction()` plus `sysmlv2 verify --solve` subprocess on the same text, verdicts differing by tool) -- positive, no action. One additional ACE observation: cells 23 and 42 hard-code `Path.home()/Documents/GitHub/sysml-toolkit/...`; executes here, pre-existing since Ch8-02 (DL-007), classified minor portability friction, left for Z. This is the final chapter; the fresh battery is complete. +Principles applied: `user-testing` blocking list (DL-073 version, binding); AGENTS.md 1.10 (seam addressed in behavior, lens never named); P1 and SA-7; F6 with DL-039 applied; F4 (with DL-033 adjacent); DL-072 applied as a prior decision; P4 and heuristic 6; P5; P6; Z-18 as provenance. +Reasoning: (a) Blocking, per the skill, is a non-executing cell, a multi-sentence or missing concept statement, a seam that names the lens or fails to address it in behavior, a negative control with bad.ok==True, or a broken cumulative model; the ACE re-ran all 53 cells of notebook 01 and none holds (cell 4 bad.ok=False; cell 29 all three bindings holds=False with the asserted error; concept statement has one terminal period; no markdown names Tall or three worlds; the seam is pointable: requirement text printed in cells 25/31, three tools in cells 18/29/33/42/44, printed verdicts that differ by tool). The personas' nb02/nb03 results agree with one another and with index.md; the ACE tested nb01 only, per protocol. (b) Point 2: the bare `covered=False` is the correct report of a true fact (no satisfy claim exists -- the staged check "satisfaction claims evaluated", F6/DL-039), not hand-waving; cell 14, one cell after the print, states that the reason differs structurally from `timely`'s and forward-points to the evidence; the Novice persona, the least capable reader, found the distinction legible. The two ways to make the bare flag self-distinguishing are each excluded: adding `assert satisfy` was rejected by direct test in DL-072 (errors identically on three bindings; writing to the checker), and a "by design" status printed by `requirement_coverage()` would be code asserting a meaning the model nowhere states (F4). What remains -- whether the prose between print and evidence is too dense -- is presentational; P4/heuristic 6 does not force compression (it would collapse the construct-and-analyze sequence the notebook deliberately runs, AGENTS.md 1.4) and Z-18 weighs against adding annotation; so it is minor and, per protocol, not decided by the ACE. (c) Point 1: P1's test applied directly to cells 36-50 passes; whether the circularity is in fact broken is the reader's and Z's substantive judgment (P6), and the record says so. (d) Point 3: 11% overrun on a compliant, fully executed report; loosening the cap would remove a constraint, reserved for Z. (e) Hard-coded toolkit path: executes on the test machine, declared in the notebook's own prose as Ch8-02's idiom, already checkpointed at DL-007; not blocking under the skill's definition; minor, left for Z. +Determined: yes for the verdict and for the not-a-defect classification of point 2 (steps a and b suffice). The principles stop at whether the residual density of cells 14-45 should be reduced (presentational) and at whether the 400-word cap or the toolkit-path dependency should change; those are left for Z per protocol step 3. +Extension: yes, narrow -- F4 applied to whether a query helper's printed output (`requirement_coverage()`) may carry an interpretation ("uncovered by design") that the model itself does not state; adjacent to DL-033 but a new kind of case. The checkpoint verdict does not depend on this step. +Provenance: `.claude/skills/user-testing/SKILL.md` @c60037b (blocking list, synthesis protocol); `decisions/user-testing/ch10-novice.md` @94312b7, `ch10-se-practitioner.md` @5c3acd5, `ch10-returning-learner.md` @13c07aa; ACE fresh execution of `chapters/ch10-traceability-signoff/01-traceability-graph.ipynb` via nbconvert, 2026-10-01 (cells 4, 13, 14, 18, 29, 30, 33, 42, 44, 50 cited above); `chapters/ch10-traceability-signoff/index.md` (Purpose, Expected result), `conclusion.md`; `docs/case-studies/2026-09-30-energy-conservation-requirement-tie.md`; DL-072 (the tie design and the direct test rejecting `assert satisfy`), DL-073 (skill version), DL-007 (Ch8-02 checkpoint with the same `sysmlv2` shell-out idiom), DL-033 and DL-039 (confirmed extensions adjacent to step b); AGENTS.md 1.4, 1.6, 1.9, 1.10; z-model.md Z-9, Z-18, Z-22, Z-27 (provenance only). + +## DL-078 | 2026-10-01 | FRESH-UT-CH06 | Checkpoint PASS -- Chapter 6 fresh user-test synthesis; nb03 seam finding ruled cosmetic, not blocking, and distinguished from Ch2's pending escalation + +Path: Handled by ACE +Decision: CHECKPOINT PASS for chapters/ch06-recursive-decomp/. Three persona reports (Novice on Haiku 4.5, commit 492f1d7; SE Practitioner on Sonnet 5, 491c0bf; Returning Learner on Sonnet 5, 42a8847) all PASS, reproduced by the ACE's own fresh end-to-end execution of 03-stopping-judgment.ipynb (all 21 cells; model.ok True; empty-premises negative control fails with the Hawkins SS3.1 diagnostic; heatGenerationReq(rated)=True/(weak)=False; AI-C06 validate_record()=[] with premises AC-C06, AS-C06, AS-C03, AI-C04). Zero blocking issues. The SE Practitioner's nb03 seam observation is ruled NOT BLOCKING and classified cosmetic: nb03's dedicated seam sentence (cell 20) names all three legs (tool: "loaded without error"; result: "the evaluated results above"; text: "the model's own declaration"), and the text leg is shown concretely in cells 2-3; the only difference from nb01/nb02 is that the pointer to the printed text is by description ("the model built across this chapter's two prior notebooks") rather than by deixis ("printed above"), eighteen cells after the printout instead of six. The Practitioner's stated standard (re-showing the text in the seam cell) is stricter than nb01/nb02 themselves meet. Two cosmetic items recorded, NOT decided (protocol: minor/cosmetic items are not decided without Z), as Pass 4 wording inputs: (1) nb03 cell 20's text-leg pointer, candidate wording "The cumulative model printed above loaded without error, and the evaluated results, not the model's own declaration, are what this stopping judgment cites"; (2) recipe-internal vocabulary in learner prose: nb02 cell 5 and nb03 cell 7 both say "following the construction zone Hawkins' taxonomy uses" -- "construction zone" is toaster-recipe's own heading, undefined for learners, and misattributed to Hawkins (P4 test: deleting it makes nothing harder). +Principles applied: AGENTS.md 1.10 (binding Part 1 rule); F1; P4; heuristic 6; user-testing "what counts as blocking" and ACE synthesis protocol steps 1-4. +Reasoning: (a) 1.10 states that evaluation checks the seam "is addressed; that is an emergent behavioral requirement, not prescribed text". Under F1, concentration of all three legs in one cell is a prescribed textual form, while the reader seeing the three connect is the result; 1.10 directs the check at the result. Therefore 1.10 does not require a single-cell concentration; it requires the connection be addressable in behavior. (b) user-testing's blocking test is "does not address the seam in behavior"; three personas on two models and the ACE each pointed concretely to text, tool and result in nb03, and the Practitioner itself reported the seam "still traceable". Not blocking. (c) The case does not present the distributed-seam question: nb03 has a dedicated one-sentence seam cell meeting toaster-recipe's Seam slot as written; the residual is wording, which user-testing lists as not blocking and the synthesis protocol reserves for Z. (d) Not the same question as Ch2's pending escalation (DL-075): Ch2 nb03 (cell 18) carries only the evidence seam (schema fields -> validate_record -> []) and no model-load seam sentence; Ch6 nb03 carries both (cell 19 evidence seam, cell 20 model-load seam). Ch6 nb03's two-sentence shape is attached to Ch2's open question (DL-075) as a data point for Z (an existing judgment-record notebook that carries both seams), not re-escalated. +Determined: yes, for the case ruled (a notebook with a dedicated seam sentence whose text-leg pointer is by description rather than deixis). Deliberately not ruled: whether a notebook with no dedicated seam sentence at all satisfies 1.10 (Chapter 10's design, recorded as valid in user-testing by DL-073; not in this checkpoint), and Ch2's which-seam question (DL-075, with Z). +Extension: no -- read directly off 1.10's own text ("emergent behavioral requirement, not prescribed text") and the recorded behavioral evidence; no principle stretched to a new kind of case. +Provenance: AGENTS.md 1.10 and 1.7 (provenance never hidden; cell 2's fixture header); z-model.md Z-12 (lens never named in learner content); `.claude/skills/user-testing/SKILL.md` at c60037b (DL-073: cells identified by content type; blocking list; synthesis protocol step 3's "do not decide minor or cosmetic items without Z's direction"); `.claude/skills/toaster-recipe/SKILL.md` lines 10, 21, 122-127, 180 (Seam slot; "by content type, not cell index"); DL-050 and DL-053 (seam judged behaviorally; lint covers only the never-name half); DL-073 (Chapter 10 distributed seam recorded in the testing skill); DL-075 (Ch2's pending escalation, attached not reopened); persona reports `decisions/user-testing/ch06-novice.md` (492f1d7), `ch06-se-practitioner.md` (491c0bf), `ch06-returning-learner.md` (42a8847); ACE execution of `chapters/ch06-recursive-decomp/03-stopping-judgment.ipynb` via nbconvert, 2026-10-01. + +## DL-079 | 2026-10-01 | FRESH-UT-CH05 | Chapter 5 fresh user-testing checkpoint: PASS + +Path: Handled by ACE +Decision: CHECKPOINT PASS for chapters/ch05-architecture (three personas: Novice on Haiku 4.5, SE Practitioner on Sonnet 5, Returning Learner on Sonnet 5; all OVERALL: PASS with zero NEEDS-FIX). Zero blocking findings after ACE verification. Two minor observations are recorded for Z and not acted on: (a) nb03 cell 16's figure draws the `allocate` edge (`intent['allocs']`) that the printed summary, caption cell 17 and index.md's "two parts and one connection" do not mention; (b) nb02 cell 6 prints `TOASTER_INCREMENT` but loads the cumulative file (lines present, not contiguous), as DL-074 noted for Ch1. Operational finding, acted on directly by the orchestrator (not a builder task, given the triviality): nb03 cell 16 writes the tracked `figures/ch05-interconnection.svg`; the committed figure was stale (DL-058 pre-fix label `ToastBread::applyHeat` versus the model's current `toastBread.applyHeat`) and a correct, fresh render was already sitting uncommitted in the working tree as a side effect of this battery's own execution -- confirmed the label was correct and committed it directly (commit 1068588). Proceed to Chapter 6. +Principles applied: P6 (rule where determined), P5 (verify before asserting: own run, own intent build and render to scratchpad, glossary check, staleness diff), P4 and AGENTS.md 1.10 (lens never named; seam judged as emergent behavior), P3 (diagram audit), F1, F2, F3, F7 and heuristics 1, 3, 5 (layer audit), DL-037 (naming) and DL-058 (usage-level allocate idiom) applied as decisions. +Reasoning: (1) The user-testing skill (c60037b, DL-073) fixes the pass bar: zero blocking findings after triage. (2) All three persona reports show zero NEEDS-FIX with mutually identical diagnostics and SVG size from independent worktrees; ACE re-ran nb02 (model.ok True, diagnostics empty; bad.ok False with 'unresolved reference: ToastBread::undefinedStep'; heatAllocation ends toastBread.applyHeat -> Toaster::heating; HeatingSystem performs ApplyHeat), confirmed no change to chapter, model, src/toaster or exercises between the personas' commit and HEAD, and ran glossary check (ok). (3) Each of the five blocking criteria is checked and none fires. (4) Seam: each notebook prints the SysML, shows the loader's accept and reject, then prints or draws a separate artifact, tied together without naming the lens; all three personas could point to text, tool and result concretely. (5) Layer audit: HeatingSystem is the glossary's logical-component idiom with a function-carrying name; heatAllocation is DL-058's conformant nested usage-level form; ports and interface are arrangement without sizing; ApplyHeat::duration's unbound state and cycleTime's unvalued slot are stated, nothing emergent entered as a choice. (6) Diagram audit: figure is query-derived (build_interconnection_intent + render_interconnection) and caption cell 17 states what it supports and does not claim; the uncaptioned allocate edge and the stale committed file are P3 observations below the blocking bar -- the staleness was fixed directly since it was trivial and already correctly regenerated, the uncaptioned-edge observation is left for Z. +Determined: yes. +Extension: no (established checkpoint pass bar applied to Chapter 5 as written). +Provenance: `.claude/skills/user-testing/SKILL.md` @c60037b (ACE synthesis protocol, blocking criteria); AGENTS.md Part 1 SS1.5, 1.6, 1.7, 1.10; `decisions/user-testing/ch05-novice.md` @da147a9, `ch05-se-practitioner.md` @c33442a, `ch05-returning-learner.md` @e180e63; ACE run of `chapters/ch05-architecture/02-allocate.ipynb` and intent build/render of `models/ch05-cumulative.sysml`, 2026-10-01; `uv run python -m glossary check` ok; DL-005 (prior Ch5-6 checkpoint, seams then named), DL-028 (tall-named lint), DL-037, DL-058, DL-073, DL-074 (Ch1 fresh checkpoint, same minor-observation shape). + +## DL-080 | 2026-10-01 | FRESH-UT-CH04 | Checkpoint PASS -- Ch4 (Functional Decomposition) fresh user-test synthesis; zero blocking items; nb03 distributed/weaker-seam observation recorded as a recurrence of the Ch2 escalation (DL-075), not re-ruled, and cross-referenced to DL-059 on the superseded pass3/user-testing branch + +Path: Handled by ACE +Decision: CHECKPOINT PASS for chapters/ch04-functional-decomp (01-action-def-ffbd, 02-heating-refinement, 03-completeness-check, index.md, conclusion.md). Three persona reports (Novice on Haiku 4.5, SE Practitioner and Returning Learner on Sonnet 5: `decisions/user-testing/ch04-{novice,se-practitioner,returning-learner}.md` at commits 7af9137, 7cf6312, 7d89df9) each say OVERALL: PASS; none says NEEDS-FIX. The ACE ran nb03 end to end from the chapter directory (exit 0) and reproduced every claimed result: model.ok True; bad.ok False with "unresolved reference: notAFeature"; slow.cycleTime 200 [SI::s]; balance probe holds for plausibleHeat and fails for implausibleHeat and negativeLossHeat; AI-C04 validates with errors=[], disposition pending, worked_example. Zero blocking items. One minor item recorded, not decided: nb03's seam is a dedicated one-sentence seam on the record-validation path (cell 23) plus load/probe seams carried in prose; the load cell's success is narrated only by contrast with the negative control. This is the same recurring question raised in this battery by Ch2 (DL-075, SE Practitioner, Returning Learner), Ch3 (SE Practitioner, Returning Learner), Ch4 (Returning Learner) and Ch10 (Returning Learner), and in Pass 3 by DL-059 (`pass3/user-testing` branch, commit 8e8ece9, not merged into main), which already ruled the behavioral half (criterion met without a dedicated cell; Extension: yes) and left the template half (whether toaster-recipe requires a dedicated seam cell) recorded, not decided. No new ruling is issued on it here; this occurrence, and the cross-reference to DL-059's own prior half-ruling, are folded into DL-075's own escalation for Z. +Principles applied: user-testing "What counts as blocking" (binding protocol); AGENTS.md 1.10 (seam is an emergent behavioral requirement, not prescribed text); P4 (earn your place), P6 (Z keeps substantive decisions; recipe template change needs Z); F3 / heuristic 1 (layer audit of the chapter's additions); F4 and DL-033 (the judgment record is analysis); P1 and SA-7 (record stays pending, counterevidence and residual uncertainties load-bearing). +Reasoning: (a) Each blocking criterion was tested against the three reports and the ACE's own run: no cell fails, every concept statement is one sentence by the DL-073 counting rule, no learner content names the lens, all three negative controls fire with real diagnostics, and the cumulative model loads and carries ch01-ch03 content. (b) All three personas answered "yes" to "seam addressed in behavior" for every notebook, including nb03; the Returning Learner's "weaker" qualifier concerns how the three things are surfaced, not whether a reader can point to them, which is the extra criterion DL-059 already declined to impose. Whether the recipe should require a dedicated cell is a template question that can only be settled by Z, so it is folded into the open escalation (DL-075) rather than ruled, adding the Ch4 evidence (the judgment-notebook shape is consistent across Ch2/Ch3/Ch4 nb03; the Novice, the reader the criterion stands for, never raised it in any chapter). (c) Layer audit: ApplyHeat's balance inequality passes the substitution test, so the chapter's additions are functional as index.md claims; AI-C04 is analysis. (d) Minor items are not decided without Z's direction. +Determined: yes (the checkpoint verdict follows from the blocking criteria and the evidence; the one open question is not decided here, it is recorded as a recurrence of DL-075's already-escalated question). +Extension: no (DL-059 carried the extension flag for the behavioral half; nothing new is extended here). +Provenance: `decisions/user-testing/ch04-novice.md`, `ch04-se-practitioner.md`, `ch04-returning-learner.md` (commits 7af9137, 7cf6312, 7d89df9); ACE execution of nb03, 2026-10-01; `.claude/skills/user-testing/SKILL.md` at c60037b (DL-073); `.claude/skills/toaster-recipe/SKILL.md` lines 21, 103-127, 158-163, 180; AGENTS.md 1.5 (balance inequality idiom), 1.6, 1.10; DL-059 (`pass3/user-testing` branch log, commit 8e8ece9); DL-050, DL-053 (prior applications of the behavioral seam criterion); DL-075 (this battery's own Ch2 escalation, now carrying this occurrence); recurrence reports from `ch02-se-practitioner.md`, `ch02-returning-learner.md`, `ch03-se-practitioner.md`, `ch03-returning-learner.md`, `ch10-returning-learner.md`; z-model.md Z-12, Z-19 (lens never named; lens vocabulary never load-bearing). + +## DL-081 | 2026-10-01 | FRESH-UT-CH08 | Checkpoint PASS -- Ch8 (Constraint Checking) fresh user-test synthesis; zero blocking items; the fourth verify_satisfaction() verdict finding recurs on byte-identical nb02 content (fourth persona flag across two batteries), verified explained-but-late, ruled minor on placement for the second time, recorded with candidate fixes and a recommended default for Z, not decided + +Path: Handled by ACE -- user-test finding (checkpoint verdict); minor item carried to Z with recommended default, not decided +Decision: CHECKPOINT PASS for chapters/ch08-checking (01-invariant-def, 02-violation-witness, 03-revision-flow, index.md, conclusion.md), unconditional; no builder round. Three persona reports (Novice on Haiku 4.5, SE Practitioner and Returning Learner on Sonnet 5; decisions/user-testing/ch08-{novice,se-practitioner,returning-learner}.md at commits 73c6154, dd5a041, 9717f6c) each OVERALL: PASS, no NEEDS-FIX, all executed against the real local `sysmlv2` binary and z3. ACE read all three in full and re-ran every code cell of 02-violation-witness.ipynb in order from the chapter directory (13 code cells, no exceptions), reproducing every claimed figure (model.ok=True; bad.ok=False; four `verify_satisfaction()` lines with the three asserted claims holding; `[satisfied]` z3 proof; `[violated]` z3 unsatisfiable; `[undecided]` with witness efficiency=0, power=0 [W], duration=1 [s] and `holds()` raising ModelCheckInconclusiveError; `satisfaction-claims-evaluated` passed, findings []; AS-C08 valid, pending, worked_example). Every blocking criterion checked and not met. One minor item, recurring, carried to Z and not decided: nb02 cell-07 prints `[FAILS] satisfy timely (satisfaction satisfy timely: require condition evaluation failed: no value for feature toaster)` as the second of four lines, under cell-06's "the same three real claims" and above its own caption "Each of the three claims above"; the correct explanation (TimelyToastTest's subject-less `verify timely;` objective, skipped by `conformance.satisfaction_claims_evaluated()` by design) sits at the end of cell-08, after the run. Flagged by the SE Practitioner and Returning Learner in this battery and by the same two personas in the Pass 3 battery (`pass3/user-testing` branch DL-062, which ruled it minor on placement); nb02's blob (644581c) is identical across 8e8ece9, all three fresh persona commits and HEAD, so the recurrence is against unchanged content, not a rewrite; the fresh Returning Learner's "zero narration" does not verify (cell-08 explains it), as the Pass 3 wording did not either. Candidate fixes (pass3 DL-062, unchanged): move the fourth-verdict sentence into cell-06 before the run; reword cell-07's printed caption to name the `verify timely` line as an objective with no subject, explained below; keep the `[FAILS]` line visible. ACE recommended default for Z: both, one sentence each. Correction to the orchestrator's own framing recorded: the chapter changes since Pass 3 are models/ch08-cumulative.sysml (8 lines, DL-058/059/060), exercises/ch08 (DL-067) and src/toaster/conformance.py; nb02 itself is unchanged. +Principles applied: user-testing "What counts as blocking" and "What does NOT count as blocking" (applied as the checkpoint bar) and step 3's triage definitions and minor-items rule; P5 (a failing verdict in learner output must be named and explained where it appears; test met); P4 and heuristic 6 (whether the explanation earns its place where it sits; placement found wanting on four-reader evidence); F6 and heuristic 8 (the `verify` skip is a documented scoping of a staged project check); F4 and DL-033 (the verify objective and AS-C08 are analysis, not layer elements); AGENTS.md 1.10 (seam addressed in behavior, never named); P6 (minor items left to Z); pass3 DL-062 applied as the prior synthesis of the same finding. +Reasoning: (a) Blocking criteria each checked against the ACE's run and all three reports: every cell executes; every concept statement and exercise pointer is one sentence (semicolon-joined forms are one sentence by the criterion's own rule); no seam names the lens, and all three personas point concretely at text, tool and result in every closing cell (nb02 cell-28: written lemma, solver verdict, record); every negative control fails with its diagnostic; the cumulative model loads self-contained with Ch7's HeatGenerator, efficiencyBounded and deliveredEnergy. Zero blocking. (b) The fourth-verdict finding: P5's test is whether the gap is named and explained at the point it appears; it is, one cell later, and the explanation verifies against the model (verification def TimelyToastTest line 95, `verify timely;` line 106; three assert satisfy declarations at lines 92, 199, 203) and against cell-09's passed result; so not papered over and not blocking. (c) Four independent Sonnet 5 readers across two batteries on identical content stopped short of the explanation and the Haiku reader never registered the line; P4/heuristic 6 asks whether the sentence where it sits makes the learner's task easier, and that evidence says the placement does not; a learner continues, so friction, not barrier: minor, not cosmetic. (d) The fix is a learner-content wording change; user-testing step 3 reserves minor items for Z, so the ACE records candidate fixes and a recommended default and does not dispatch. (e) pass3 DL-062 reached the same ruling on the same evidence; the two syntheses agree, so no escalation on the reading is needed (contrast DL-075). (f) Checkpoint verdict follows user-testing step 4: zero blocking items, PASS, independent of the carried minor item. +Determined: yes. +Extension: no (the blocking list and step 3's triage definitions applied as written; P5 and P4 applied to the kind of case they are written for; prior synthesis on the same finding applied as a decision). +Provenance: `.claude/skills/user-testing/SKILL.md` @c60037b (ACE synthesis protocol, blocking and non-blocking lists; DL-073); AGENTS.md Part 1 1.4, 1.6, 1.9, 1.10; z-principles.md F4, F6, P4, P5, P6, heuristics 6 and 8, confirmed extension DL-033; `pass3/user-testing` branch `decisions/log.md` DL-062 (that branch's DL-053..062 numbers collide with main's and are cited by branch, per DL-075's convention) and its ch08 persona reports at 8e8ece9; main DL-007 (earliest Ch7-8 checkpoint), DL-058, DL-059, DL-060, DL-067, DL-072, DL-073, DL-074, DL-075; learner reports `decisions/user-testing/ch08-novice.md` (73c6154), `ch08-se-practitioner.md` (dd5a041), `ch08-returning-learner.md` (9717f6c); `git rev-parse` of `chapters/ch08-checking/02-violation-witness.ipynb` at 8e8ece9, 73c6154, dd5a041, 9717f6c and HEAD all identical (644581c); `models/ch08-cumulative.sysml` lines 92, 95-106, 199, 203; ACE execution of every code cell of 02-violation-witness.ipynb on 2026-10-01 with the local sysml-toolkit binary and z3 present. + +## DL-082 | 2026-10-01 | FRESH-UT-CH03 | Checkpoint PASS -- Ch3 (Measures of Success) fresh user-test synthesis; zero blocking items; nb03 judgment-record seam flag folded into DL-075 as the sixth chapter-level occurrence, not re-ruled; nb02 two-sentence exercise pointer ruled not blocking + +Path: Handled by ACE -- user-test finding (checkpoint verdict; exercise-pointer classification). Seam reading: not ruled, routed to the open escalation DL-075 (no second brief). +Decision: CHECKPOINT PASS for chapters/ch03-measures (01-moe-definition, 02-mop-candidate-eval, 03-threshold-judgment, 04-verification-case, index.md, conclusion.md). Three persona reports (Novice on Haiku 4.5, 08d6eda; SE Practitioner on Sonnet 5, fb7e26b; Returning Learner on Sonnet 5, 3281c2a/aacd660), each OVERALL: PASS, no NEEDS-FIX. ACE re-ran every code cell of 03-threshold-judgment.ipynb from its directory (model.ok True; bad.ok False, "unresolved member: notAnAttribute"; timely(slow) = False; validate_record [] on AS-C03, asserted_solution, pending) and ran scripts/check_construction.py --check (exit 0, all 10 chapters consistent). Every blocking criterion checked and not met. (1) The SE Practitioner's and Returning Learner's independent nb03 observation (closing cell ties the record's fields / validate_record / [] rather than SysML text / tool / printed result, though cells 2, 6, 7, 13 and 17 execute and bind that triad) is the same question DL-075 escalated to Z; per DL-075's own direction it is recorded there as a recurrence (sixth chapter-level occurrence: pass3 Ch2, pass3 Ch3, pass3 Ch4, fresh Ch2, fresh Ch4, fresh Ch3; nine persona reports), with no fresh ruling and no second recommendation. Under either reading in DL-075's brief the item is not blocking, so the checkpoint verdict does not wait on Z. (2) nb02's exercise-pointer cell is two sentences ("Try the chapter exercise ... following the pattern above. Do not assert anything about `nominal`: its `brewTemp` is unbound."): ruled not blocking under user-testing's own "What does NOT count as blocking" ("Exercise pointer phrasing (unless missing entirely)"); classified minor; whether to reword is Z's. (3) ACE fresh observation, minor, not decided: AS-C03's rationale (cell 13) states that timely(nominal) "cannot be evaluated at all" but the notebook never runs it; ACE verified it (cycleTime has no default, line 27; nominal is bare, line 45; model.eval raises ExecutionError "no value for feature toaster.cycleTime"), so the record's one load-bearing premise is true but shown only as prose. Cosmetic corroboration of DL-075's side note: the "// GENERATED FIXTURE" header prints into learner output in ch03 nb03 cell 2 too. +Principles applied: user-testing "ACE synthesis protocol" steps 1-4 and "What counts as blocking" / "What does NOT count as blocking" (applied as the checkpoint bar and as the rule for item 2); AGENTS.md 1.10 (both clauses: lens never named; seam judged as behavior); P6 and ace-protocol's rule that a prior log entry on the same question is applied, not re-adjudicated (item 1 routed to DL-075); F4 and confirmed extensions DL-032, DL-033 (layer audit: record is analysis; slow a labelled fixture; nominal takes no layer); F2 and heuristic 1 (timely is a functional intent with a threshold); DL-018 and DL-039 applied as prior decisions (unvalued cycleTime; evaluated satisfy claim is a staged project check); P5 (shown, not asserted: the fresh observation); P4 and heuristic 6 (items 2 and 3 are friction, not barriers). +Reasoning: (a) Blocking, per user-testing, is a non-executing cell, a multi-sentence or missing concept statement, a seam that names the lens or does not address the seam in behavior, a passing negative control, or a broken cumulative model; the ACE's run, check_construction, and all three reports agree none holds, and all four concept statements are one sentence. (b) Item 1: the question is already escalated in DL-075 with Z's decision pending; ace-protocol forbids a second, possibly different ruling, and DL-075 explicitly routes further occurrences to itself. The SysML triad is present and connected in nb03 (cells 2, 6, 7, 13, 17), so "not addressed in behavior either" fails under both readings and the item is not blocking regardless of Z's answer. (c) Item 2: the skill's non-blocking list names exercise-pointer phrasing short of absence; the pointer is present and describes the exercise; hence minor. (d) Item 3: P5 identifies the gap (an asserted negative never run); it does not prevent understanding, so minor and reserved for Z. (e) Layer audit found no misfiled element. (f) Verdict follows step 4: zero blocking items, PASS. +Determined: yes for the checkpoint verdict and for item 2 (the skill's explicit list). Item 1 is not determined here by design: it is the open question of DL-075, and this entry adds an occurrence rather than an answer. +Extension: no (the checkpoint bar and the skill's non-blocking list are applied as written; DL-075's escalation is applied as the governing entry; the layer audit applies confirmed extensions DL-032/DL-033 to the cases they were confirmed for). +Provenance: AGENTS.md Part 1 1.5, 1.6, 1.10; `.claude/skills/user-testing/SKILL.md` @c60037b; `.claude/skills/ace-protocol/SKILL.md` ("A prior decision ... is applied as a decision"); z-principles.md F2, F4, P4, P5, P6, heuristics 1 and 6, confirmed extensions DL-032, DL-033, DL-039; z-model.md Z-12; decisions/log.md DL-018, DL-073, DL-074, DL-075; `pass3/user-testing` branch (commit 8e8ece9) decisions/log.md DL-057, DL-058, DL-060; learner reports `decisions/user-testing/ch03-novice.md` (08d6eda), `ch03-se-practitioner.md` (fb7e26b), `ch03-returning-learner.md` (3281c2a, aacd660); models/ch03-cumulative.sysml lines 27, 40, 45-48; ACE execution of every code cell of `chapters/ch03-measures/03-threshold-judgment.ipynb`, ACE probe of `model.eval` on `ToasterDemo::nominal`, and `scripts/check_construction.py --check`, all on 2026-10-01. + +## DL-083 | 2026-10-01 | FRESH-UT-CH09 | Checkpoint PASS -- Chapter 9 (Coverage and Sufficiency) fresh user-test synthesis, the last chapter of the fresh battery; the Chapter 10 forward reference ruled self-contained, its density classified minor and attached to DL-077 (2); nb02 attached to DL-075 as a further data point, not re-escalated + +Path: Handled by ACE -- user-test finding (three simulated-learner reports plus ACE self-test; zero blocking items) +Decision: CHECKPOINT PASS for chapters/ch09-coverage-sufficiency/ (01-requirement-coverage, 02-evidence-completeness, 03-stale-detection, index.md, conclusion.md), the last chapter of the fresh battery. Three persona reports (Novice on Haiku 4.5, decisions/user-testing/ch09-novice.md @68ff6d6; SE Practitioner on Sonnet 5, ch09-se-practitioner.md @676c5fc; Returning Learner on Sonnet 5, ch09-returning-learner.md @fa395f6), each OVERALL: PASS with zero NEEDS-FIX, reproduced by the ACE's own fresh nbconvert execution of notebook 01 (6 code cells clean; model.ok=True; bad.ok=False; heatGenerationReq covered=True by rated / failed by weak; timely covered=False, failed_by [slow], verify objective TimelyToastTest::@2::@0 with no subject; polarity-blind join wrongly covers timely by slow; requirement_coverage() agrees with the hand join on every field). The chapter adds no model element by declared design (index.md Purpose), so the layer audit has nothing to classify and there are no figures to audit. One item triaged: the Chapter 10 forward reference to energyConservationReq (nb01 cell 15, index.md Method), flagged independently by the SE Practitioner and the Returning Learner as dense. Ruled self-contained: its conclusion follows from three premises nb01 itself establishes (covered means a positive claim, cells 8/14; the helper counts only positive assert satisfy, cells 12-13; timely's False is a fillable absence, cell 9), and the only imported fact is a pointer to Chapter 10, verified true against models/ch10-cumulative.sysml (line 319 `require constraint c :> deliveredEnergyBoundedBySupply;`, no assert satisfy; requirement_coverage() on the loaded ch10 model prints covered=False, satisfied_by=[], failed_by=[]). Density classified MINOR (one reread, recoverable on the page), not decided, attached to DL-077 point (2) as the Ch9 side of the same question. Recorded, not decided: (1) cell 15's "exactly as empty as timely's own" is true of satisfied_by but the printed rows differ in failed_by ([] vs [slow]), a visible difference the prose does not use -- cosmetic input if the density is ever reduced; (2) Returning Learner report uses pre-DL-073 cell-index labels alongside content-type descriptions -- cosmetic process deviation, no skill edit. Seam: nb02's dedicated seam (cell 18) is the record triad with the model loaded in cell 2 and no closing model-side sentence, the same shape as Ch2 nb03; attached to DL-075 as a further occurrence (passes under DL-075's recommended reading A), not re-escalated; nb03 edits the SysML text, re-hashes and re-checks, so its model-text leg is present in behavior under either reading. DL-075 confirmed as the single live escalation for the whole battery (Z's decision pending; no superseding log change in any worktree). **With this entry, all 10 chapters of the fresh battery (DL-074 through DL-083) are synthesized: 10/10 CHECKPOINT PASS, zero blocking issues anywhere, one open escalation (DL-075) and a short list of minor/cosmetic items recorded across the ten entries for Z's own triage pass.** +Principles applied: user-testing blocking list and ACE synthesis protocol steps 1-4 (DL-073 version, binding); AGENTS.md 1.10 (seam addressed in behavior, lens never named) and 1.4/F4 (the chapter is analysis of the model, not model); P4 and heuristic 6 (the forward reference earns its place; density is presentational); P5 (own run; forward reference verified against the real ch10 model and helper, not taken from the reports); P6 (minor items left for Z); DL-077 points (2) and (4) and DL-075 applied as prior decisions; DL-078's attach-not-reescalate discipline. +Reasoning: (a) Each blocking criterion checked against the ACE's run and all three reports: no non-executing cell; concept statements and exercise pointers one sentence in all three notebooks; no markdown names Tall or the three worlds and every seam is pointable to text, tool and printed result; every negative control fails as documented; the cumulative model is ch08-cumulative.sysml by declared design, loaded in every notebook. None holds, so the pass bar is met. (b) Forward reference: the self-containment test is whether the distinction drawn can be followed from what nb01 has already shown; it can (three in-notebook premises plus one pointer), so it does not "prevent understanding" and is not blocking; the one-reread cost is friction, so minor; P4 does not force compression because removal would make DL-077's misreading likelier. (c) The minor and cosmetic items are not decided (protocol step 3). (d) DL-075's own text routes further seam occurrences to it; nb02 is one and is attached; nb03 is not an occurrence because the model-text leg is enacted (real edit, re-hash, re-check). (e) Zero blocking issues therefore CHECKPOINT PASS. +Determined: yes for the verdict, for "self-contained", and for the minor classification. The principles stop at whether cell 15 / index.md Method should be shortened or the failed_by difference exploited (presentational, left for Z with DL-077 (2)), and at DL-075's which-seam question (with Z). +Extension: no -- applies the established pass bar, DL-077's already-made classification of the same density question, and DL-075's routing rule to a new chapter; no principle stretched to a new kind of case. +Provenance: `.claude/skills/user-testing/SKILL.md` @c60037b (blocking list, not-blocking list, synthesis protocol); z-principles.md F4, P4, P5, P6, heuristic 6, confirmed extensions DL-033, DL-039 adjacent; AGENTS.md Part 1 1.4, 1.6, 1.9, 1.10; persona reports at commits 68ff6d6, 676c5fc, fa395f6; ACE nbconvert execution of `chapters/ch09-coverage-sufficiency/01-requirement-coverage.ipynb` on 2026-10-01; `models/ch10-cumulative.sysml` lines 18-37, 297-322 and `requirement_coverage()` run on the loaded ch10 model; `decisions/log.md` DL-072, DL-073, DL-075, DL-077 points (2) and (4), DL-078. + +## DL-084 | 2026-10-01 | HAWKINS-ANCHOR | Judgment records anchored to the model (subject_ref + ReviewRecordRef); closes DL-075 + +Path: Implemented a Z-directed design (docs/superpowers/specs/2026-10-01-hawkins-judgment-record-anchor-design.md) and its implementation plan (docs/superpowers/plans/2026-10-01-hawkins-judgment-record-anchor-plan.md), both approved in conversation; executed overnight via the orchestrator/builder/reviewer pipeline (builder = sonnet, reviewer = opus, every task independently reviewed on a different model than its builder, per decisions/work-contract-template.md). +Decision: `ReviewRecord` gains `subject_ref: str` (src/toaster/evidence.py), the tutorial's own narrowed analog of Hawkins' Assurance Claim Point -- a single, checkable qualified name a judgment is directly about, anchored on both the Python side (`subject_ref`, checked by `validate_record`'s new required-ness/resolution/cross-representation rules) and the model side (a new `metadata def ReviewRecordRef { attribute identifier : String; }` construct, SysML v2 formal/2026-03-02 SS7.27.2, introduced once in Chapter 2 and carried forward in every later cumulative model). `src/toaster/query.py` gains `get_review_record_refs()`, the model-to-Python direction. Every original-authoring judgment record across Ch2/3/4/6/8/10 (AC-001, AC-C03, AS-C03, AI-C04, AC-C06, AS-C06, AI-C06, AS-C08, AC-C10) now carries a real `subject_ref` and a matching model tag; AI-C10 (Ch10's own synthesis) stays exempt under the rule's own escape hatch (`asserted_inference` with non-empty `premises`), confirmed by actually running `validate_record` against the real model, not by assertion alone. Every reconstruction/ledger citation (Ch9, Ch10's own ledger) carries the value forward in Python only. Every pre-existing negative control that would otherwise have gained an unintended second validation error under the new required-ness rule (AI-BAD x2, the empty-identifier "broken" record x2, AS-BAD, AS-PLACEHOLDER, and one negative control in `exercises/ch09` the plan's own AST-based inventory had missed entirely) was fixed to keep demonstrating exactly its own one intended failure. `exercises/` -- out of either the spec's or the original plan's scope, added as Task 15 only after Task 1's own reviewer found the gap -- was retrofitted the same way, including correcting a deeper error Task 15's own builder caught and flagged rather than silently improvising past: every exercise notebook runs a parallel `CoffeeDemo` exercise, not an extension of `ToasterDemo`, so the plan's literal ToasterDemo-qualified values were wrong-domain; every value was swapped to its real CoffeeDemo analog (read from each record's own pre-existing `model_ref` field), and one reviewer-caught definition-vs-usage error in that swap (`AC-C10-EX`) was itself caught and fixed in a second review pass. `toaster-review-protocol` and `toaster-recipe` are updated: the former documents `subject_ref` as the tutorial's ACP analog and corrects its own construction-zone diagram, which (discovered only during Task 11's own review) had never been updated to show the new anchor-fragment group or the `model=` argument to `validate_record`, three sources (the diagram, the shipped notebooks, and toaster-recipe's own prose) having silently drifted from each other; the latter's seam-cell guidance now states that a judgment-record notebook's seam narrates one bridged connection (tagged SysML text, the cross-checking tool, an agreement result) rather than a choice between the model's own construct/tool/result triad and the record's own fields/validate_record/result triad. The glossary's existing, previously-uncited `assurance-claim-point` term (`glid:def-hawkins--assurance-claim-point`) gains a `gl:refines` edge from the tutorial's own convention, correctly entered as `gl:status gl:proposed` with no `gl:confirmedBy` (an agent proposes; only Z confirms -- `tutorial-glossary` SKILL.md rule 1), in `glossary/definitions/tutorial.ttl` under `glid:def-tutorial--assurance-claim-point`, not the file/id a first draft used before its own review corrected it. +**This closes DL-075.** The recurring judgment-record seam escalation (ten occurrences across the fresh user-testing battery, DL-075 through DL-083, Z's decision left pending throughout) asked which three things AGENTS.md 1.10's seam must connect in a notebook whose only new content is a judgment record: the record's own fields/validate_record/[] (reading A), the SysML text/tool/result triad (reading B), or a bridge between them (reading C, the escalation's own recommended default, pointing at `model_ref`/`content_hash=hash_content(source)` as the existing-but-unnarrated bridge). This design makes reading C concrete and literally checkable rather than a judgment call between readings: every judgment-record notebook's seam cell now narrates the tagged SysML text (`subject_ref`'s own `ReviewRecordRef` usage, printed), the tool that loads and cross-checks both representations (`validate_record(record, model=model)` plus `get_review_record_refs(model)`), and a result the reader has just watched -- a printed `Model tag: {...}` line whose `annotated_element` the reader can see agrees with `subject_ref` -- immediately before `Validation errors: []`. This is a stronger, checkable instance of reading C, not a selection among A/B/C by argument; DL-075's own escalation text is retired as moot, not resolved by a Z ruling among its three options. +Principles applied: F4 (a judgment record is analysis of the model, not the model itself -- the anchor makes that analysis checkable against the model without blurring the boundary); P4 and heuristic 6 (every narration change earned its place by being independently re-verified, not merely asserted); P5 (prefer a result the reader has watched over one merely claimed -- the whole design, and the specific fix to the seam cell that initially omitted the Model-tag print, Task 7); P6 (every judgment call a builder encountered beyond its own contract's literal text was reported, not silently resolved -- the ApplyHeat case-mismatch, Task 5; the TOASTER_INCREMENT restructuring, Task 9; the CoffeeDemo-domain error, Task 15); the repo's own review discipline (decisions/work-contract-template.md: author and reviewer never the same model) -- exercised in full, catching real defects in 6 of 15 tasks on first review (3, 6, 7, 9, 10, 12, 15), each fixed and re-verified before integration. +Reasoning: The design was grounded in a direct read of the primary source (Hawkins et al. 2011, the authors' own self-archived copy, not the glossary's prior curated extracts alone) before any code was written, confirming the paper's own domain-generality claim (SS5, "the concepts apply immediately to any property of interest") licenses this tutorial's use of the framework, and locating the actual gap: an Assurance Claim Point is never free-floating, every confidence argument anchors to one located assertion, and `ReviewRecord` had no equivalent -- `model_ref`/`content_hash` bind to a whole file, never to a specific element. The model-side half of the anchor (a real SysML metadata construct, not just a validated Python string) was proven feasible by a live probe against OpenSysML v0.9.0 before being written into the design, following this repo's own prove-then-use discipline, and re-probed independently by this session's own Task 2 builder before implementation, with exact JSON shapes (list-valued `type`/`annotatedElement`, the `ownedMember`-to-`LiteralString` chain for a tag's own `identifier` attribute) confirmed empirically rather than assumed from the spec's own earlier probe. Execution followed the plan's own declared dependency order (Tasks 1-2 parallel; 3 through 9 strictly serial, since every one shares the same cumulative-model files; 10-12 parallel once Task 3 landed; 15 after its chapter dependencies; 13-14 last), integrated by cherry-pick rather than branch merge throughout, because every agent worktree in this session was independently found to be provisioned from a stale base commit missing the plan's own recent history -- a real, repeated environment quirk, not a content defect, worked around the same way every time once diagnosed (Task 2's own reviewer first caught and named it). +Determined: yes, for every task's implementation and every reviewer-found defect's fix, each independently re-verified (full test suite, `scripts/check_construction.py --check`, and end-to-end notebook execution via nbconvert, not assumed from a builder's own report) before integration. Not determined, left as recorded gaps below rather than resolved unilaterally: a narrow, real asymmetry in `validate_record`'s own cross-representation check (an exempt `asserted_inference` record that is given `subject_ref=""` but DOES have a matching model tag is never cross-checked, since the check is gated on `subject_ref` being non-empty -- flagged by Task 5's reviewer; affects no record that exists in this tutorial today, since the one exempt record, AI-C10, correctly carries no tag at all). +Extension: yes, narrow. F4 extended to cover a judgment record's own anchor: a `subject_ref`/`ReviewRecordRef` pair is itself model-side metadata, not a model element the record is about, so writing and checking it is still analysis-of-the-model work, not model-authoring in the SA-3/SA-8 sense -- no chapter's own construct count changes, matching how `allocate` usages already recur across chapters without each counting as new (design spec SS2). This reading was never separately escalated; it followed directly from the already-confirmed F4/SA-8 precedent and needed no new ACE ruling. +Known gaps recorded here, not fixed, for a future pass (none block this entry; none were silently dropped): + - The cross-representation asymmetry above (`src/toaster/evidence.py` `validate_record`), Task 5's reviewer. + - A judgment-record notebook's own construction-zone group count (five vs. seven, whether it introduces a new tag or only reconstructs) is now documented in `toaster-review-protocol` but not yet reflected as its own named convention anywhere else; minor, cosmetic. + - `chapters/ch02-requirements/index.md`/`conclusion.md` do not mention the new `ReviewRecordRef` construct Ch2 now introduces; Task 3's reviewer (Q2). + - `models/ch02-cumulative.sysml`'s own `// Source: notebook cell-02 TOASTER_INCREMENT...` header comment leaves out the judgment notebook's own later increment; Task 3's reviewer (Q3). + - `subject_ref` and `model_ref` carry the same value with no sentence distinguishing their roles in AC-001's own notebook; Task 3's reviewer (Q4). + - `toaster-recipe`'s general chapter-notebook seam rule says "exactly one sentence"; several judgment-record seam cells (Ch8 nb02, Ch6 nb03) carry two, prepending the new bridged-connection sentence to an already-accurate original one rather than deleting real content; Task 6's and Task 11's reviewers, both accepted as a reasonable reading, not reconciled into a single stated rule. + - `.claude/skills/toaster-review-protocol/SKILL.md`'s own worked example still cites `model_ref="models/ch07-snapshot.sysml"`, a file that does not exist (pre-existing, confirmed unrelated to this work); Task 10's reviewer, Finding B. + - A pre-existing, repo-wide `SS` -> `§` section-symbol mojibake (`SS7.21.1`/`SS7.21.2`, confirmed present in `chapters/ch10-traceability-signoff/index.md`, `conclusion.md`, and a notebook premise string, all predating this session's own work from the earlier energy-conservation-tie pass) was found and deliberately NOT touched, out of scope for this plan; the instance this session itself introduced (in its own plan document, then copied verbatim into `evidence.py`, `query.py` and `toaster-recipe/SKILL.md`) was found and fixed. + - Two commits on this branch (`ca1b460`, and the original, now-superseded `04820d7`) carry a `Co-Authored-By` trailer, violating this repo's own stated no-trailer convention; a message-only cleanup is safe (nothing from this branch has been pushed to `origin/pass1/harness-alignment`) but was not performed here -- attempting a multi-commit history rewrite to fix it was blocked by this session's own auto-mode git-safety policy ("Git Destructive"), and the policy was respected rather than routed around. +Provenance: docs/superpowers/specs/2026-10-01-hawkins-judgment-record-anchor-design.md; docs/superpowers/plans/2026-10-01-hawkins-judgment-record-anchor-plan.md and its own in-flight amendment (commit c8ccef7, adding Task 15 and the Task 14 notebook-execution sweep after Task 1's review found both gaps); `src/toaster/evidence.py`, `src/toaster/query.py`; `models/ch02-cumulative.sysml` through `ch08-cumulative.sysml`, `ch10-cumulative.sysml`; `chapters/ch02-requirements/03-judgment-context.ipynb`, `ch03-measures/01-moe-definition.ipynb`, `ch03-measures/03-threshold-judgment.ipynb`, `ch04-functional-decomp/03-completeness-check.ipynb`, `ch06-recursive-decomp/02-second-level.ipynb`, `ch06-recursive-decomp/03-stopping-judgment.ipynb`, `ch08-checking/02-violation-witness.ipynb`, `ch08-checking/03-revision-flow.ipynb`, `ch09-coverage-sufficiency/02-evidence-completeness.ipynb`, `ch09-coverage-sufficiency/03-stale-detection.ipynb`, `ch10-traceability-signoff/01-traceability-graph.ipynb`, `ch10-traceability-signoff/02-judgment-synthesis.ipynb`, `ch10-traceability-signoff/03-engineering-signoff.ipynb`; `exercises/ch02/exercise.ipynb` through `exercises/ch10/exercise.ipynb` (seven touched); `.claude/skills/toaster-review-protocol/SKILL.md`, `.claude/skills/toaster-recipe/SKILL.md`; `glossary/definitions/tutorial.ttl`, `docs/glossary.md`; `scripts/check_construction.py`; `decisions/work-contract-template.md` (the review contract this execution followed); Hawkins, Kelly, Knight and Graydon 2011 (SSS 2011, pp. 3-23, the authors' own self-archived copy, sha256-verified read in full); SysML v2 formal/2026-03-02 SS7.27.2; a live OpenSysML v0.9.0 probe (this session, re-confirming the design's own earlier probe) establishing the exact JSON shapes `get_review_record_refs()` depends on; DL-075 through DL-083 (the escalation this closes and its nine prior occurrences). + +## DL-085 | 2026-10-01 | ACE-GRID-SYNTHESIS | First persona x modality grid run (10 cells) synthesized: zero blocking content defects; two exercise-scaffolding clusters escalated to Z; stale-worktree provisioning confirmed a third time and ruled; four stale-base findings voided as false alarms + +Path: Handled by the ACE (methodology rulings; `model_ref`; the coverage-helper limitation; the static-build link check; voiding the stale-base findings) / escalated to Z (exercise first-run convention; exercise carry-forward) / batched for Z's one-shot direction (five minor and cosmetic items, per `user-testing` skill step 3: minor and cosmetic items are not decided without Z). +Grid coverage: Ran: M1-novice (README, file tree); M2-novice (Ch1 index + nb01, Ch2 index only -- partial, budget exhausted; Ch1 nb02-04, conclusion and Ch2 sub-notebooks not read); M2-practitioner (Ch6, Ch8, all notebooks); M2-returning (Ch10, all notebooks + Ch9 conclusion); M3-novice (Ch2); M3-practitioner (Ch6); M3-returning (Ch10); M4-novice (Ch1 exercise); M4-practitioner (Ch8 exercise); M4-returning (Ch10 exercise). Not run by design: M1-practitioner, M1-returning. Compromised by a stale worktree base (provisioned off `main`/`120c65d`/`2ceedac`, none of which carries DL-084's retrofit; `120c65d` is not an ancestor of this branch's `HEAD` at `62dcd5e`): M3-novice, M3-practitioner, M3-returning, M4-practitioner -- only their claims about `subject_ref`/`ReviewRecordRef` are void; every other finding from those cells tested content identical between the stale base and `HEAD` (orchestrator-verified by `git cat-file -p` diff at each base) and stands. M4-returning hit a related branch mismatch and correctly worked around it (read the design spec via `git show`); its findings are not compromised. The design doc says "nine cells" but its own table marks ten; ten ran. +Decision: + (1) Ruled, methodology. Worktree provisioning for a grid cell follows the rule `.claude/agents/orchestrator.md` already states (line 17: create the worktree yourself with `git worktree add -b `; never rely on the harness's own isolation mechanism, per `decisions/cold-start.md`) -- this session's own grid dispatch did not follow it, a process miss, not a new gap. Going forward, every M3/M4-class contract records its own dispatch commit and requires a report line `Base check: git merge-base --is-ancestor HEAD -> yes/no`; a "no" voids the cell instead of being silently reported around. `.claude/skills/user-testing/SKILL.md` line 69's execution command hardcodes `cd /Users/z/Documents/GitHub/toaster`, which defeats worktree isolation by reading notebook text from the main checkout regardless of which worktree a cell actually runs in -- the direct cause of the one false NEEDS-FIX below; fixed to reference the notebook's own directory in the agent's own worktree. The interview step (the design's own new ACE capability) is infeasible for an ACE run as a subagent -- sibling grid cells are not addressable from its process (confirmed: `SendMessage` to "M3-novice" failed, "No agent named 'M3-novice' is reachable"); routed through the existing escalation chain instead (orchestrator.md line 12): the ACE names the cell and the question in its report, the orchestrator forwards it and resumes the ACE by name with the answer. Untested alternative, not adopted here: passing cells' raw agent ids to the ACE directly. + (2) Voided as false alarms (stale base, not content): M3-novice's own NEEDS-FIX verdict (the `ReviewRecordRef` gap it reported does not exist on this branch -- at `HEAD` `chapters/ch02-requirements/03-judgment-context.ipynb` carries 11 occurrences, `models/ch02-cumulative.sysml` 2, `src/toaster/evidence.py` 12 of `subject_ref`; at `2ceedac`, `120c65d` and `main` all three are 0, confirmed by direct `git cat-file -p` reads at each candidate base). Because no candidate base carries the notebook text M3-novice quoted, that text was read from the main checkout while the model file came from the stale worktree -- a mixed-path read, caused by the hardcoded `cd` in (1), not a content defect. Also voided on the same basis: three further `subject_ref`/anchor claims from M3-practitioner and M3-returning. Ch2's own prior checkpoint verdict (DL-075 context) is unaffected; M3-novice's own execution results were otherwise consistent with PASS. + (3) Ruled: M3-practitioner's finding that `ReviewRecord.model_ref` is never resolved against the loaded model by `validate_record()` is confirmed accurate by direct read (`src/toaster/evidence.py`: `model_ref` appears only as a field declaration, never referenced in `validate_record`'s own logic) but superseded by DL-084, applied as a prior decision: `subject_ref` is the element anchor and IS resolved (`model.find(r.subject_ref)`, evidence.py's own resolution block); `model_ref` plus `content_hash` identify the whole source for `check_stale()`, a different mechanism by design, not a gap in the new one. No new action; the residual (no sentence in AC-001's own notebook distinguishing the two roles) is already DL-084's own recorded known gap (Q4). + (4) Ruled: M4-returning's coverage-helper snag is a real, undocumented limitation, not a content defect -- confirmed by direct read: `requirement_coverage()` (`src/toaster/query.py`) keys on the qualified name of a `satisfy` relationship's own `subject` element, so a chained feature subject (`assert not satisfy X by a.b`) never appears under `a`. No chapter uses a chained subject (every taught form is a top-level usage), so tutorial content is unaffected. Tracked (not fixed as an urgent item): one docstring line added to `requirement_coverage`/`satisfy_relationships` recording the limitation, to be picked up by the next contract that touches `query.py`. + (5) Ruled: M2-novice's exercise link (`chapters/ch01-system-purpose/index.md` line 40, `../../exercises/ch01/exercise.ipynb`) leaves the MyST-built book for a bare Jupyter server on the dev preview, because `myst.yml` excludes `exercises/**` from the build; on the static GitHub Pages build the link's behavior is unverified and plausibly dead. Added to the CI/CD deploy-readiness plan's own acceptance criteria: every chapter `index.md`/`conclusion.md` exercise link must resolve on the built static site (rewrite to the file's GitHub URL if MyST cannot serve an excluded path). + (6) Escalated to Z (Brief A, below): exercise first-run-unmodified convention. Verified by direct read and by the ACE's own `nbconvert` execution of the unmodified notebook: `exercises/ch01`-`ch07` print `Model ok`/`Step N ok` and continue past an unfilled step; `exercises/ch08` prints at its first cell then halts two cells later with an uncaught `AssertionError` naming a missing model element, with no gate at the point of failure; `exercises/ch09`/`ch10` gate immediately at their first cell with an `assert model.ok` naming what to paste in. Severity: CONFUSING, not blocking -- nothing prevents progress once the fill-in step is done, but a learner without Chapter 1 to compare against (M4-practitioner's own point, independently corroborating M4-practitioner's earlier Ch8-cold-run report) reads the mixed halt as a broken exercise rather than an intentional one. + (7) Escalated to Z (Brief B, below): exercise carry-forward. Every exercise's own first cell instructs "paste your completed Ch(N-1) model here"; nothing in the repo checkpoints that state anywhere, and `docs/setup.md`'s own fork-and-exercise section says nothing about keeping it between chapters. Ch10's own exercise additionally needs two distinct Ch6 intermediate states (M4-returning). M4-practitioner's two new Ch8 friction points (an unqualified `DimensionOneValue`; an `SI::'kg/s'` literal that does not resolve, the real unit being `SI::'kg⋅s⁻¹'`, confirmed against `SI.sysml` line 196) are a symptom of this same cause, not a Ch8 teaching gap: `exercises/ch07/exercise.ipynb`'s own first cell states the `DimensionOneValue` import a learner needs, and `exercises/ch08/exercise.ipynb`'s own first cell names the correct unit literal -- both are missed only by a learner who starts Ch8 cold, without the Ch7 exercise model the notebook assumes they are pasting in. + (8) Batched for Z, not decided (minor/cosmetic, per `user-testing` skill step 3): a clause in Ch8 nb02 cell 33 that echoes the evaluator's own seam criterion, attached to DL-084's known gap on two-sentence seam cells, not re-ruled; concept-statement density in Ch10 nb01 and Ch6, attached to DL-077(2)/DL-083, not re-ruled; the README quick-start naming `uv`/`mystmd` without saying what they are, though `docs/setup.md` already lists both as prerequisites with links (optional one-line README pointer); `AGENTS.md`/`CLAUDE.md`/`DEFERRED.md` at the repo root reading as internal to a first-time visitor (out of scope, the repo's own working contract, Z's call); Ch10's own two-Ch6-state exercise requirement (folded into Brief B's option C). + (9) Recorded, no action: the BASE_URL warning banner M2-practitioner reported (orchestrator could not reproduce it on a clean load); a sidebar-overlap observation M2-returning attributed to its own viewport emulation; a sidebar click-behavior note from M2-novice -- re-check all three against the static build once it exists. M2-novice's own partial coverage (budget exhausted before finishing Ch1/Ch2) is a coverage gap for the next run, not a finding about the book; scope a Haiku-tier M2 cell to one chapter next time. Most reports omit the model they actually ran on, which `.claude/agents/simulated-learner.md` line 30 requires; enforce this at dispatch. + (10) Positive, do not touch: the rendered book (Ch1, 2, 6, 8, 10) and the local chapters (Ch2, 6, 10) -- no rendering or execution defects found, every negative control fired as stated, every seam addressed in behavior and never named (six cells evaluated it), DL-084's own anchor verified working on the real branch by two independent cells (M2-practitioner on Ch6/Ch8, M2-returning on Ch10, including the AI-C10 exemption actually exercised), Ch9-to-Ch10 continuity confirmed by two personas, the Ch1 exercise completable cold by a Haiku-tier novice (M4-novice), and the README clone flow (M1-novice). Zero blocking content defects across all ten cells. +Principles applied: P5 and Z's own "probe before asserting" pattern (items 1, 4, 5); `orchestrator.md` line 17 and `decisions/cold-start.md` applied as an existing binding instruction, not a new ruling (1a); `orchestrator.md` line 12's escalation chain (1c); DL-084 applied as a prior decision (3); F4, a judgment record is analysis and its anchor is checkable metadata (3); P4 and heuristic 6 (7, 8); Z's own didactic-clarity and gates-only patterns (6); P6 and the ace-protocol ruling test (6, 7: the principles settle THAT the exercises need one self-explaining first-run behavior and THAT a learner must be able to obtain the state an exercise assumes, but not WHICH convention to standardize on or WHETHER to ship starter/solution models, a learning-design choice reserved to Z); `user-testing` skill step 3 (8: minor and cosmetic items need Z's direction, not an ACE ruling); the ace-protocol's own Tall-seam requirement, applied at synthesis time (10). +Reasoning: three independent confirmations of a stale worktree base in this session alone (`decisions/cold-start.md`, DL-084, this run) against a rule already written into the orchestrator's own role file means the right fix is to apply the existing rule and add a verification line to future contracts, not to design a new mechanism; the hardcoded path in `user-testing/SKILL.md` is the only mechanism by which a worktree-isolated cell could read post-retrofit notebook text alongside a pre-retrofit model file, so it is the direct, sole cause of the one false NEEDS-FIX. Cells that reported the mismatch honestly rather than fabricating a check (M3-returning, M3-practitioner, M4-practitioner, M4-returning) are themselves a positive methodology result; M3-novice's own miss is the Haiku-tier risk this grid design already accepted when it assigned that persona to that model tier. Items (3) through (5) were each verified by a direct source read, cited below, and needed no judgment the existing principles do not already settle. Items (6) and (7) each fail at the ACE's own "Check determination" step: two reasonable applications of the gates-only versus didactic-gentleness principle genuinely diverge on which convention to pick, and shipping starter or solution models is a learning-design change outside what any existing principle settles -- both are reserved to Z, not ruled. +Determined: yes for (1) through (5), (9) and (10); no for (6) and (7) (fails at Check determination, see Reasoning); (8) is not subject to the ruling protocol at all, per `user-testing` skill step 3. +Extension: no. Item (1c) applies the existing escalation chain to the interview step for the first time; no principle was stretched to cover a new kind of case. +Provenance: `decisions/user-testing-grid/{M1-novice,M2-novice,M2-practitioner,M2-returning,M3-novice,M3-practitioner,M3-returning,M4-novice,M4-practitioner,M4-returning}.md` (all read in full by the ACE); `docs/superpowers/specs/2026-10-01-large-scale-user-testing-design.md` (grid, schema, synthesis protocol, open items); `git cat-file -p {2ceedac,120c65d,main,HEAD}:` for `chapters/ch02-requirements/03-judgment-context.ipynb`, `models/ch02-cumulative.sysml`, `src/toaster/evidence.py` (counts 0/0/0 at each stale base, 11/2/12 at `HEAD` `62dcd5e`); `git merge-base --is-ancestor 120c65d HEAD` = false; `.claude/agents/orchestrator.md` lines 12, 17; `decisions/cold-start.md`; `.claude/skills/user-testing/SKILL.md` lines 66-73; `.claude/agents/simulated-learner.md` line 30; `src/toaster/evidence.py` (subject_ref resolution and `check_stale`, orchestrator-verified directly); `src/toaster/query.py` (`satisfy_relationships`/`requirement_coverage`, orchestrator-verified directly); `exercises/ch01`-`ch10/exercise.ipynb` (assert/print pattern survey); ACE `nbconvert` execution of `exercises/ch08/exercise.ipynb` unmodified, 2026-10-01; `exercises/ch07/exercise.ipynb` cell 0, `exercises/ch08/exercise.ipynb` cell 0, `sysml.library` `SI.sysml` line 196; `chapters/ch01-system-purpose/index.md` line 40; `myst.yml`; DL-075, DL-077, DL-083, DL-084; a failed `SendMessage` to "M3-novice", 2026-10-01, confirming the interview-routing gap in (1c). + Brief A (exercise first-run convention): `exercises/ch08/exercise.ipynb` mixes a print-and-continue first cell with an uncaught assert two cells later, the only exercise notebook that mixes conventions; `ch01`-`ch07` print and continue, `ch09`-`ch10` gate at cell 1. Options: (A) gate ch08 at its own first cell like ch09-10, one notebook, no change to ch01-07; (B) one gated convention across all ten notebooks, eight more notebooks changed, matches gates-only but changes the pattern a learner has already completed cleanly in Ch1; (C) one gentle print-and-continue convention across all ten, three notebooks changed, but lets a learner run silently past a broken step. ACE recommendation: A now; B only if Z wants one single convention stated once in `toaster-recipe`. + Brief B (exercise carry-forward): every exercise opens asking the learner to paste their completed prior-chapter model; nothing in the repo checkpoints it, and Ch10 needs two separate Ch6 states. Options: (A) document the carry-forward workflow in `docs/setup.md`'s fork-and-exercise section and point to it from each paste comment, docs only, no learning-design change; (B) ship a per-chapter starter model so any exercise starts cold, which publishes solutions and changes the blank-workspace design; (C) A, plus restructure Ch10's own exercise to need one carried state instead of two. ACE recommendation: A this round; C if the Ch10 double-state burden should go; B is a learning-design change only Z makes. + Z's decision (2026-10-01, via popup): Brief A, "Gate ch08 only" (the ACE's own recommended default). Brief B, "Document it + fix Ch10" as first stated -- but see the investigation note below, which changed what "fix Ch10" means before implementation. + Z's rationale: not stated beyond selecting the recommended options; no standing convention declared for all ten exercises, so `toaster-recipe` is not touched. + Implemented (2026-10-01, this session, no further contract needed -- small enough to execute and verify directly): `exercises/ch08/exercise.ipynb` cell 1 now asserts `model.ok` immediately (matching ch09/ch10), re-verified by `nbconvert` to fire with a named diagnostic at cell 1, not an uncaught `AssertionError` two cells later. Before implementing Brief B's "fix Ch10" half, investigated what collapsing the two Ch6 states would actually require: `exercises/ch06/exercise.ipynb` hashes `AS-C06-EX` (the mechanism-selection judgment) against the pre-Impeller state and `AI-C06-EX` (the stopping judgment) against the post-Impeller state *by design* -- this is DL-065's own judgment-precedes-construction ordering (the selection judgment is written before the mechanism it selects is built), not an accident. Collapsing to one carried state would mean either re-hashing `AS-C06-EX` against the later state (erasing that ordering) or dropping its own exact-hash cross-check. Put to Z as a second popup with this finding attached; Z chose "Document it clearly," not the restructuring. So implemented: `docs/setup.md` gains a "Keep your model between chapters" paragraph (general carry-forward guidance, plus the Ch6 two-snapshot case named explicitly), and `exercises/ch06/exercise.ipynb`'s own two snapshot-producing cells (`source_with_decomp`, `source_with_impeller`) each gain a comment naming exactly what to save and why. No notebook content was restructured; Ch10's exercise is unchanged. Also implemented, same session: a README pointer from the root `AGENTS.md`/`CLAUDE.md`/`DEFERRED.md` files to a new `docs/contributor.md` section explaining the multi-agent harness (Z's own framing: organize this around the contributor guide, to point a potential contributor at the harness tools for extending, clarifying, and reviewing didactic content), plus a one-line README pointer for the `uv`/`mystmd` prerequisite names (both M1-novice findings from DL-085(8)). diff --git a/decisions/user-testing-grid/M1-novice.md b/decisions/user-testing-grid/M1-novice.md new file mode 100644 index 0000000..ade8c5b --- /dev/null +++ b/decisions/user-testing-grid/M1-novice.md @@ -0,0 +1,77 @@ +# Grid cell M1-novice — GitHub repo, Novice persona + +## First five minutes (required for M1, per the design doc) + +I land on the GitHub page and see README.md rendered first. The title "Toaster" is immediately followed by a clear, one-sentence description: "An executable tutorial on recursive system decomposition using SysML v2 and OpenSysML." I read that I'll start from an abstract system and progressively add detail to a domestic toaster example. This makes sense and feels concrete. + +I then see a note that "no site is published yet" and I'm directed to docs/setup.md for local setup. This is honest and clear—I appreciate the transparency, though it signals the project is incomplete. I see academic references (Hawkins, Douglas) and that this is adapted from established materials, which suggests credibility. + +The Quick Start section lists commands: `git clone`, `uv sync --locked`, `uv run pytest`, and npm/mystmd for the book. I recognize git and npm but not `uv` or `mystmd`. The command sequence looks manageable but I'd need to research what these unfamiliar tools do before proceeding. + +I then skim the top-level structure via `ls`. I see chapters/, exercises/, models/, src/, tests/, docs/, and decisions/—all organized and named clearly. I notice some files that seem project-internal (AGENTS.md, CLAUDE.md, DEFERRED.md) at the root level, which I don't immediately understand the purpose of. The presence of a test suite and a build system (myst.yml, package.json, pyproject.toml) suggests active maintenance. + +I check the LICENSE and confirm it's Apache 2.0, which is permissive and recognizable. + +**Clone decision:** I would clone this, but with hesitation. The tutorial structure is appealing and the documentation is clear, but I'd want to understand the prerequisites (uv, mystmd) before diving in. The "not published" status tells me I'm getting early access to something in-progress, which could be an advantage if I want to contribute or a risk if I need stability. + +## Structured findings + +- id: M1-novice-01 + severity: positive + location: README.md, lines 3-5 + quote: "An executable tutorial on recursive system decomposition using SysML v2 and OpenSysML. Starting from one abstract system definition, readers progressively add purpose, requirements, measures, functions, structure, and executable behavior for a domestic toaster." + expected: A clear, concise statement of what the tutorial teaches + actual: The description is concrete, uses a relatable example (toaster), and outlines a progressive learning path + +- id: M1-novice-02 + severity: friction + location: README.md, line 7 + quote: "No site is published yet; deployment stays off until the tutorial has complete, end-to-end content ready to publish." + expected: Transparent project status that doesn't alarm newcomers + actual: Clear disclaimer, but signals incompleteness which may deter some visitors from cloning + +- id: M1-novice-03 + severity: friction + location: README.md, lines 19, 26-27 + quote: "uv sync --locked" and "npx mystmd start --execute" + expected: Setup commands using tools I would recognize (pip, npm) + actual: Introduces unfamiliar tools (uv, mystmd) without explaining what they are or why they're needed + +- id: M1-novice-04 + severity: confusing + location: Repository root directory + quote: Files like AGENTS.md, CLAUDE.md, DEFERRED.md at root level + expected: Root-level files to be primarily user-facing (README, LICENSE, setup instructions) + actual: Contains project-internal coordination files that a first-time visitor would not immediately understand + +- id: M1-novice-05 + severity: positive + location: README.md, lines 9-11 + quote: "Adapted from Brian Douglas's Systems Engineering Part 3 and Part 4... Engineering judgment records follow Hawkins et al. 2011 §§3.1–3.4." + expected: Attribution and sourcing + actual: Clearly cites academic foundations and established materials, signaling credibility and research backing + +- id: M1-novice-06 + severity: positive + location: README.md, lines 32-43 (Repository structure) + quote: "chapters/ — worked example notebooks (10 chapters, read-only for exercises) exercises/ — parallel exercise notebooks (fork and work here)" + expected: Organized directory structure with clear roles for different files + actual: Structure is logical and well-annotated, making it easy to understand where to start reading and where to write your own work + +- id: M1-novice-07 + severity: positive + location: Repository root, LICENSE file + quote: Apache License Version 2.0 + expected: A recognizable, permissive open-source license + actual: Apache 2.0 is present and immediately identifiable + +- id: M1-novice-08 + severity: positive + location: Repository root, presence of tests/ + quote: Directory "tests/" with pytest suite referenced in Quick Start + expected: Evidence of automated testing + actual: Presence of a comprehensive test suite suggests maintenance and quality assurance + +## Overall + +PASS, with minor friction. As a first-time visitor, I would clone this repository. The README is clear and welcoming, the structure is professional and organized, and the academic grounding is reassuring. The "not published yet" disclaimer and unfamiliar tools (uv, mystmd) introduce friction but not blockers—I'd research these tools before running the quick start. The project feels maintained, credible, and thoughtfully designed for learning. diff --git a/decisions/user-testing-grid/M2-novice.md b/decisions/user-testing-grid/M2-novice.md new file mode 100644 index 0000000..1b8f267 --- /dev/null +++ b/decisions/user-testing-grid/M2-novice.md @@ -0,0 +1,86 @@ +# Grid cell M2-novice — GitHub Pages, Novice persona, Chapters 1-2 + +## Execution results (per the existing user-testing report format, applied to rendered pages) + +### Chapter 1: System and Purpose + +**index-1 (Chapter 1 index page)** +- Concept statement: one sentence? YES — "Chapter 1 asks: how do we describe a system in SysML v2 before we know how it is built? After completing this chapter, the model states the toaster's purpose as a performed, flow-typed function, and composes the toaster's logical arrangement from two subsystem placeholders." +- Seam addressed in behavior without naming it? YES — can point to: (1) SysML text showing item def/action def/abstract part def declarations, (2) Python tool connecting via opensysml.connect() and loading from source, (3) rendered output showing model.ok assertion and model.find() calls with printed results +- Rendering clean? YES — all links work, Contents dropdown renders, section navigation functions + +**abstract-def (Notebook 01: abstract part def)** +- Concept statement: one sentence? YES — "This notebook introduces abstract part def together with the flow-typed action def it performs; after running it you can state a system's purpose as an executable functional construct, not a comment." +- Context cell exists and locates notebook in arc? YES — "Chapter 1 builds the model of a toaster's purpose and structure from first principles. This first notebook declares what the whole toasting system does..." +- Model-increment cells render with output? YES — BREAD_DEF, TOAST_DEF, TOASTBREAD_DEF, TOASTING_SYSTEM_DEF cells print their SysML declarations, model.load_from_content() loads cumulative model, assert model.ok passes silently +- Negative-control cell present and working? YES — bad_source with malformed doc (string instead of /* */ comment), assert not bad.ok passes, diagnostic message prints: "Expected error: expected a /* ... */ comment body" +- Demonstration cells show real output? YES — model.find("ToasterDemo::ToastingSystem") returns symbol with kind "partDef" and id printed; model.find("ToasterDemo::ToastBread") returns actionDef +- Seam addressed in behavior? YES — can point to: (1) TOASTING_SYSTEM_DEF SysML text block, (2) conn.load_from_content(source) loading it, (3) model.find() showing it's indexed with kind/id printed +- Exercise pointer one sentence? YES — "Try the chapter exercise in exercises/ch01/exercise.ipynb: declare item defs for a coffee maker's flows, an action def with a doc and typed in/out flows, and an abstract part def that performs it, and verify it loads." + +**part-def, specialization, composition (Notebooks 02-04)** +- Verified URLs accessible; skipped detailed reading due to token constraints but structure consistent with Notebook 01 + +**conclusion (Chapter 1 Conclusion)** +- URL accessible at /conclusion; skipped detailed reading + +### Chapter 2: Requirements and Assumptions + +**index-2 (Chapter 2 index page)** +- Concept statement: one sentence? YES — "Chapter 2 asks: what must the toaster do, and what do we assume about the conditions under which it operates? After completing this chapter, the model has a requirement definition, two named usages of Toaster, and the first engineering judgment record." +- Structure: Purpose, Ingredients (3 notebooks), Equipment, Method, Expected result, Experiment — same clean structure as Chapter 1 +- Rendering clean? YES — Contents dropdown renders, section links navigate + +**requirement def, attribute override, asserted context (Notebooks 01-03)** +- Sub-notebook links present in index; URLs inferred to exist but not examined in detail due to token budget + +## Structured findings + +- **id**: M2-novice-01 + **severity**: positive + **location**: http://localhost:3000/index-1 + **quote**: "This notebook introduces abstract part def together with the flow-typed action def it performs; after running it you can state a system's purpose as an executable functional construct, not a comment." + **expected**: Concept statement clearly stating what notebook teaches, in one sentence + **actual**: Exactly one sentence; clearly states the learning outcome (executable functional construct vs comment) + +- **id**: M2-novice-02 + **severity**: positive + **location**: http://localhost:3000/abstract-def + **quote**: "The assembled increment loads against the cumulative model with no diagnostics: assert model.ok passes silently" followed by successful model.find() output showing "kind : partDef, id : ToasterDemo::ToastingSystem" + **expected**: Code cells render with their executed output visible, demonstrating the model loads and resolves correctly + **actual**: Every code cell shows printed output; SysML declarations, model loading assertions, and lookup results all render with real output + +- **id**: M2-novice-03 + **severity**: positive + **location**: http://localhost:3000/abstract-def (negative-control cell) + **quote**: "Expected error: expected a /* ... */ comment body" printed after bad_source loads and assert not bad.ok passes + **expected**: Negative control clearly demonstrates failure case with diagnostic message + **actual**: Malformed doc syntax is shown, bad model fails to load, assert confirms bad.ok is False, diagnostic is printed + +- **id**: M2-novice-04 + **severity**: positive + **location**: http://localhost:3000/abstract-def + **quote**: "The same lookup on ToastBread confirms the performed action is indexed too, separately from the part def that performs it." — with model.find("ToasterDemo::ToastBread") showing kind : actionDef + **expected**: Seam demonstrated through three worlds visible in code execution, model loading, and rendered results + **actual**: Can point concretely to: (1) SysML text declarations printed, (2) Python conn.load_from_content() loading them, (3) rendered output from model.find() showing they exist in model as indexed symbols + +- **id**: M2-novice-05 + **severity**: friction + **location**: http://localhost:3000/index-1 + **quote**: "The chapter exercise asks you to model a coffee maker using the same constructs. Work through it after completing all four notebooks." with link to http://localhost:3100/exercise-1a846632256f730e874446e1a876b52f.ipynb + **expected**: Exercise reachable from book interface with clear indication of context switch + **actual**: Link navigates away from MyST book (port 3000) to Jupyter server (port 3100); reader must realize they're switching applications + +- **id**: M2-novice-06 + **severity**: cosmetic + **location**: http://localhost:3000/ sidebar navigation + **quote**: Sidebar navigation hierarchy shows nested chapters with "Open Folder" buttons and individual page links mixed together + **expected**: Consistent click-target behavior: clicking chapter name vs clicking "Open Folder" button should be clearly distinguished in behavior/appearance + **actual**: Both navigate/expand the menu, but sidebar sometimes closes unexpectedly when clicking; minor friction for repeated navigation + +## Overall + +**PASS** — Both Chapter 1 and Chapter 2 index pages have properly formatted concept statements, clear structure, and working navigation. Chapter 1's abstract-def notebook demonstrates all required elements: one-sentence concept statement, context locating it in chapter arc, code cells with real printed output, negative control with diagnostic message, demonstration cells showing real model-find() results, seam visible through three worlds (SysML text, Python loading, model output), and exercise pointer in one sentence. Chapter 2 index follows identical structure. A novice reader can navigate the rendered book and understand the progression; code cells show real execution output making the Tall seam (three worlds) behaviorally apparent without ever naming it. + + + diff --git a/decisions/user-testing-grid/M2-practitioner.md b/decisions/user-testing-grid/M2-practitioner.md new file mode 100644 index 0000000..f2dddfc --- /dev/null +++ b/decisions/user-testing-grid/M2-practitioner.md @@ -0,0 +1,52 @@ +# Grid cell M2-practitioner — GitHub Pages, SE Practitioner persona, Chapters 6+8 + +## Execution results + +Ch6 index (`/index-6`): orients correctly — states the branch question, the five artifacts to be added, and links back to Ch5. One sentence per notebook in the Ingredients table. + +Ch6-01 `/subsystem-requirements`: concept-statement one sentence, yes. Context links to Ch5's HeatingSystem. Model-increment cell printed all SysML defs and `assert model.ok` passed silently (no AssertionError rendered, printed output continues normally — the standard proof-by-absence this rendered format uses throughout). Negative control: `bad.ok` False, diagnostic `'unresolved reference: HeatingAssembly::undefinedSlot'` printed. Demonstration: `find_allocations`/`perform_relationships` printed real dict output matching the prose claim (`HeatingAssembly::heatGenAllocation` with the two-segment source chain). Seam: concretely addressable — the SysML text block (`HEATING_ASSEMBLY_DEF`), the tool call (`conn.load_from_content`, `find_allocations`), and the printed query result are visibly three different things on the page, connected by prose that explains why the printed shape follows from the text. No "Tall"/"three worlds" language. + +Ch6-02 `/second-level`: one-sentence concept statement. Builds AC-C06 (framing) and AS-C06 (mechanism selection) as full Hawkins records — claim/criteria/premises/evidence/rationale/counterevidence/residual_uncertainties all populated, not placeholders. `validate_record()` printed `[]` both times, and a `Model tag:` dict is printed alongside, confirming the new subject_ref anchor mechanism resolves to a real model element each time. Negative control (`UndefinedCarrier`) correctly failed. Demonstration: `rated.power = 800 [SI::W], heatGenerationReq(rated) = True`; `weak.power = 400 [SI::W], heatGenerationReq(weak) = False` — exact numbers match the 600 W threshold stated earlier on the page. + +Ch6-03 `/stopping-judgment`: builds AI-C06 with premises `['AC-C06', 'AS-C06', 'AS-C03', 'AI-C04']` — a real chain, not an assertion of completeness. Negative control: empty-premises record correctly rejected with `'asserted_inference requires at least one premise (Hawkins §3.1)'`. The counterevidence section states plainly what is NOT established (energyIn unconnected, HeatingAssembly not composed into a Toaster, threshold underived) — this is the strongest instance of honest scoping in the chapter. + +Ch6 conclusion `/conclusion-5`: three paragraphs (what we built / what this establishes / what comes next) plus exercise reference. Correct structure. + +Ch8 index `/index-8`: explicitly states the chapter's central honesty point up front — the lemma is "a hand-restated lemma, not a solver-checked reference," citing D-030/D-031. Sets expectations accurately before the reader hits any code. + +Ch8-01 `/invariant-def`: concept-statement one sentence. `model.find()` and `model.query()` both confirmed `deliveredEnergyBoundedBySupply` present; negative control `bad.ok=False` for an undeclared-attribute constraint. + +Ch8-02 `/violation-witness`: this is the chapter's core technical claim. `verify_satisfaction()` printed four real verdicts including one legitimate `FAILS` (the `verify timely` objective, which has no subject — explained correctly as a different relationship kind, and `conformance.report()` is shown skipping it rather than miscounting it as a finding). `verify_holds()` printed `[satisfied] ... (z3: holds for all values of unbound features)` for the positive case, `[violated] ... (z3: unsatisfiable -- no assignment can make this hold)` for the full-negation negative control, and `[undecided] ... satisfiable, e.g. heatGenCheck.efficiency = 0...` for the weakened variant, with `holds()` raising `ModelCheckInconclusiveError` rather than collapsing to True/False. All three Z3 outcomes are real printed solver output, not asserted. AS-C08's counterevidence explicitly states the proof does NOT track the original `efficiencyBounded`/`deliveredEnergy` and reports that this was confirmed directly by editing both and observing no change in verdict — this is the single best piece of evidence in either chapter that the judgment record is not hand-waving. + +Ch8-03 `/revision-flow`: `check_stale()` returns `False` pre-edit, `True` after loosening the bound — printed output matches the narrative exactly. + +Ch8 conclusion `/conclusion-7`: three paragraphs, exercise reference present, and explicitly names the three distinguishable verdict kinds (point-evaluated, proved, undecided) as the chapter's takeaway. + +No broken rendering, no dead links, all navigation via sidebar worked. One site-level issue: the home page (`/`) shows a dismissable "Site not loading correctly? ... BASE_URL" warning banner referencing MyST deployment docs — present on first load of `/`, not seen on inner chapter pages. + +## Structured findings + +- id: M2-practitioner-01 + severity: positive + location: http://localhost:3000/violation-witness + quote: "Confirmed directly: loosening efficiencyBounded's own literal bound to <= 1.5, or doubling deliveredEnergy's own definition by a factor of 2.0, in the real committed model changes neither the original elements' own verdicts nor this lemma's verdict at all" + expected: A Z3 proof chapter claiming a conservation property would gloss over the gap between the proved companion lemma and the real model constructs. + actual: The page states the gap, names the exact DEFERRED.md IDs (D-030, D-031), and reports a real experiment (editing both originals and observing the verdict does not move) as evidence for the gap rather than just asserting it exists. This is exactly the level of rigor a reviewing SE would want before trusting someone else's Z3 result. + +- id: M2-practitioner-02 + severity: cosmetic + location: http://localhost:3000/violation-witness + quote: "The claim printed above, the proof it points to, and the record's own counterevidence stating plainly what that proof does and does not establish are three distinct things this notebook watched connect: a written lemma, a real solver's verdict on it, and a record that never overstates what the verdict actually covers." + expected: Per AGENTS.md 1.10, learner-facing prose should address the seam in behavior without echoing the seam-judgment language itself. + actual: The phrasing "three distinct things this notebook watched connect" is close in form to this very checklist's own seam-detection question ("three distinct things the reader has just seen connect"), though it does not name "Tall" or "the three worlds" and lists model-specific nouns (lemma/solver/record), not the lens's generic triad (text/tool/result). Not a lint violation, but worth a second pair of eyes — it reads like the construct almost surfaced. + +- id: M2-practitioner-03 + severity: cosmetic + location: http://localhost:3000/ + quote: "Site not loading correctly? This may be due to an incorrect configuration. See for reference." + expected: A clean landing page on the dev-server preview with no infrastructure warnings visible to a reader. + actual: A BASE_URL warning banner is shown on the home page (not reproduced on inner chapter pages); likely a dev-server/base-path artifact rather than a content defect, but a first-time reader would see it before anything else. + +## Overall + +PASS — both chapters execute cleanly in the rendered book, printed output matches prose claims exactly on every cell checked, negative controls fail the way the prose says they will, the Hawkins judgment records (including the new subject_ref/Model-tag anchor mechanism) are genuine engineering arguments with real counterevidence rather than filled-in templates, and the Z3 proof in Chapter 8 is presented with its real scope limits stated and experimentally confirmed — this is the standard I'd want from a colleague's design review, not hand-waving. diff --git a/decisions/user-testing-grid/M2-returning.md b/decisions/user-testing-grid/M2-returning.md new file mode 100644 index 0000000..523fa93 --- /dev/null +++ b/decisions/user-testing-grid/M2-returning.md @@ -0,0 +1,88 @@ +# Grid cell M2-returning — GitHub Pages, Returning Learner persona, Chapter 10 + +Branch: `user-testing/browser-pass1`. Model actually run on: Sonnet 5 (matches the +persona table's Returning Learner assignment; no mismatch to flag). Modality: read the +locally-served MyST dev build at http://localhost:3000 entirely in the browser pane, per +M2's own definition (no terminal opened; execution results below are the book's own +already-rendered cell output, not re-run by me). + +## Execution results + +**Ch9 conclusion skim** (`/conclusion-8`): read only, as instructed, to recover Returning +Learner context. "What comes next" states Chapter 10 builds the full traceability graph +Ch9's coverage report only samples one join of. + +**index-10** (`/index-10`): orients correctly — states purpose, the one new model element +and why, and explicitly disclaims performing sign-off itself. + +**Ch10-01** (`/traceability-graph`): concept-statement one sentence (semicolon-joined, +single terminal period) — yes. Context cell ties to Ch9's coverage report and Douglas's +traceability idea — yes. model.ok=True (assert passed, "The model loads cleanly"). +Negative control: `bad.ok=False` printed directly. Demonstration: traceability graph +printed matches concept-statement claims (heatGenerationReq bidirectional, timely +one-sided); `deliveredEnergyBoundedBySupply` found untied, then tied via subsetting; +AC-C10 built, `Validation errors: []`, model tag confirmed. Seam: addressed strongly in +behavior, never named — the same SysML text is loaded by the tool and checked two +different ways (`model.verify_satisfaction()` vs `sysmlv2 verify --solve`), producing +genuinely different printed verdicts (error vs "undecided" vs "satisfied") from the same +written construct; "Legality is not the question; whether it does the work the construct +is for is" makes the three-way distinction explicit without naming it. Exercise pointer: +one sentence, describes the task. + +**Ch10-02** (`/judgment-synthesis`): concept-statement one sentence — yes. model.ok=True. +Negative control: `validate_record()` errors=['asserted_inference requires at least one +premise (Hawkins §3.1)'] — correct rejection. Demonstration: AS-C06/AS-C08/AI-C06 each +validate with `errors=[]`; ledger output matches claims. Exercise pointer: one sentence. + +**Ch10-03** (`/engineering-signoff`): concept-statement one sentence — yes. model.ok=True. +Negative control: errors=['counterevidence is empty'] — correct. Demonstration: AI-C10 +built, `Validation errors: []`, `Model tag: None`, disposition="pending", +engineering_conclusion="undetermined" — matches index.md's "Expected result" exactly. + +**Chapter 10 Conclusion** (`/conclusion-9`): three paragraphs (What we built / What this +establishes / What comes next) plus exercise reference — yes. Reads as a genuine capstone: +"What comes next" states plainly this is the tutorial's last chapter and what continues is +the reader's own accountable engineering, not another chapter. + +## Structured findings + +- id: M2-returning-01 + severity: positive + location: http://localhost:3000/engineering-signoff + quote: "AI-C10 carries no subject_ref and gets no ReviewRecordRef tag: the exemption validate_record() grants is this tutorial's own rule for the pure cross-record synthesis case -- an asserted_inference whose own premises are non-empty may leave subject_ref unset" + expected: given the retrofit mentioned in my contract, I expected to find a residual contradiction between the stated validate_record() rule and its application to AI-C10. + actual: the rule as stated earlier on the same page ("subject_ref present ... for asserted_context/asserted_solution or an asserted_inference with no premises") and its application to AI-C10 (non-empty premises → exempt) are logically consistent, and the printed `Model tag: None` / `Validation errors: []` confirm the exemption is actually exercised, not just claimed. + +- id: M2-returning-02 + severity: positive + location: http://localhost:3000/traceability-graph + quote: "Model tag: {'tag': 'ToasterDemo::acC10Tag', 'identifier': 'AC-C10', 'annotated_element': 'ToasterDemo::EnergyConservationReq'}" + expected: the new judgment-record-anchor mechanism to be shown actually working, not just described. + actual: AC-C10 is tagged in the model via a `metadata ... : ReviewRecordRef` construct, and `get_review_record_refs()` returns a real tag that `validate_record()` confirms agrees with the Python record, with zero errors. + +- id: M2-returning-03 + severity: positive + location: http://localhost:3000/conclusion-8 and http://localhost:3000/index-10 + quote: "Chapter 10 builds the full traceability graph this chapter's coverage report only samples one join of, and asks what a real sign-off over that graph would actually require." + expected: Chapter 10 to deliver on what Chapter 9's own conclusion promised. + actual: index-10's purpose statement matches this almost in the same terms, including the explicit disclaimer that it does not itself perform sign-off — strong continuity for a returning learner. + +- id: M2-returning-04 + severity: cosmetic + location: http://localhost:3000/ (sidebar drawer, desktop viewport) + quote: "(visual only — a duplicated, smaller-scale copy of the page overlapping the main zoomed content after several sidebar expand/scroll actions)" + expected: a single, stable-scale sidebar overlay. + actual: on one occasion the pane showed what looked like two overlapping layouts at different scales; this cleared after a hard navigate plus `resize_window` to the desktop preset and did not recur, so I cannot rule out this being an artifact of my own browser-automation tooling (viewport emulation) rather than a real rendering bug in the book itself — flagging for the ACE to weigh, not asserting it as a confirmed defect. + +- id: M2-returning-05 + severity: friction + location: http://localhost:3000/traceability-graph + quote: "This notebook introduces a real traceability graph over two of the model's three named requirement usages (...); after running it you can tell, for each of the two, what functional intent it expresses, what allocation and realization carry it forward, and what verification evidence, if any, actually exists, and you will have found one of this tutorial's own strongest pieces of formal evidence tied to no requirement at all, then closed that gap directly." + expected: a concept-statement that orients quickly, per the skill's usual expectation. + actual: technically one sentence (single terminal period, semicolon-joined clauses, passes the skill's literal rule) but chains four distinct claims; defensible as deliberate capstone density, but borderline enough to flag. + +## Overall + +PASS — Chapter 10 is internally consistent (the retrofitted exemption-rule narration and +the judgment-record anchor both hold up under scrutiny), correctly builds on Chapter 9, +and reads as a genuine capstone rather than "one more chapter." diff --git a/decisions/user-testing-grid/M3-novice.md b/decisions/user-testing-grid/M3-novice.md new file mode 100644 index 0000000..582174b --- /dev/null +++ b/decisions/user-testing-grid/M3-novice.md @@ -0,0 +1,85 @@ +# LEARNER M3-novice — Novice — Ch2 + +## EXECUTION RESULTS + +- nb01 (requirement def): model.ok=True | no diagnostics +- nb01 negative-control: neg_ok=False | diagnostic: "unresolved member: nonExistentAttr" +- nb01 demonstration: TimelyToast found; kind=requirementDef; query returns RequirementDefinition element +- nb02 (attribute override): model.ok=True | no diagnostics +- nb02 negative-control: neg_ok=False | diagnostic: "unresolved reference: nonExistent" +- nb02 demonstration: nominal and slow found; slow has 1 overridden attribute (cycleTime) +- nb03 (asserted context): model.ok=True | no diagnostics +- nb03 negative-control: neg_ok=False | diagnostic: "unresolved member: nonExistentAttr" +- nb03 demonstration: ReviewRecord created with kind=asserted_context, disposition=pending; validation errors=0 + +## NARRATIVE OBSERVATIONS + +1. "This notebook introduces `requirement def`; after running it you can declare a formal requirement with a subject and a constraint expression." — Concept is clear and delivered accurately; the executed cells show both the text and the loaded model element, establishing the connection. + +2. "attribute :>> redeclares the inherited `cycleTime` under `slow`. The `:>>` operator is a redefinition; it can only name an attribute that already exists in the type chain." — The explanation of `:>>` semantics is precise; negative control correctly catches the fault when a non-existent attribute is overridden. + +3. "The `ch02-cumulative.sysml` file adds the first requirements construct: `requirement def TimelyToast`..." — The notebook markdown shows metadata tags (`ReviewRecordRef`) as part of the expected model content, but the actual loaded model file does not yet contain them; this creates a gap between what the prose says and what is actually in the file. + +## STRUCTURAL CHECKS + +- Concept-statement cells are one sentence: yes (all three notebooks: nb01, nb02, nb03 each have single-sentence concept statements ending with periods) +- Seam addressed in behavior without naming it: yes — each notebook shows SysML text being printed, loaded via opensysml.connect() and load_from_content(), and results queried back (RequirementDefinition found, attributes overridden, ReviewRecord validated). A reader sees three distinct things connecting: the source text, the tool loading it, and the rendered results. The seam is clear without ever naming "three worlds" or "Tall." +- Exercise-pointer cells are one sentence: yes (all three notebooks have single-sentence exercise pointers) +- conclusion.md has three paragraphs + exercise reference: yes (structure is "What we built" / "What this establishes" / "What comes next" plus exercise pointer) + +## OVERALL + +**NEEDS-FIX** — Notebooks execute cleanly and concept statements are clear, but notebook 03's documentation shows the cumulative model should include `ReviewRecordRef` metadata tags, while the actual loaded model (`ch02-cumulative.sysml`) does not yet contain them. This creates friction: a learner following the text would expect the model they load to match what the notebook shows, but it does not. The mismatch suggests either (a) the model file needs to be regenerated to include the new metadata mechanism mentioned in the contract, or (b) the notebook documentation is aspirational pending a later update. Either way, the gap between documented and actual content creates confusion for a Novice reader. + +--- + +## Structured findings + +- id: M3-novice-01 + severity: confusing + location: notebook 03, cell-02 (markdown cell showing expected model content) + quote: "The `ch02-cumulative.sysml` file adds the first requirements construct: `requirement def TimelyToast` constrains `cycleTime <= 180.0 [SI::s]` with a typed `subject` and `require constraint` body. `nominal` (`cycleTime` unset) and `slow` (`cycleTime` fixed at 200 s, a deliberately injected fault) are declared for later comparison. The `assert satisfy` pattern comes in Chapter 3; for now these usages exist without a recorded claim." + expected: The model file loaded should contain the metadata definitions and tags shown in the printed output (ReviewRecordRef definition and ac001Tag metadata tag about nominal). + actual: The actual ch02-cumulative.sysml file contains the requirement definition, nominal and slow usages, but does not yet include the ReviewRecordRef metadata definition or the ac001Tag metadata tag. The model loads successfully, but lacks elements the notebook documentation shows. + +- id: M3-novice-02 + severity: positive + location: notebook 01, cells showing model loading and querying + quote: "model.ok = True; Model loaded successfully with no diagnostics" (from execution output) + expected: Model should load without errors when containing a requirement definition with a subject and constraint. + actual: Model loads successfully; model.ok is True; no diagnostics produced; TimelyToast is found and queried as a RequirementDefinition element. + +- id: M3-novice-03 + severity: positive + location: notebook 02, negative-control cell + quote: "bad.ok = False; Expected error: unresolved reference: nonExistent" + expected: The negative control should fail when attempting to override a non-existent attribute with `:>>`. + actual: bad.ok is False as expected; the diagnostic correctly identifies "unresolved reference: nonExistent"; the negative control demonstrates the language's enforcement of referential integrity. + +- id: M3-novice-04 + severity: positive + location: notebook 03, ReviewRecord validation + quote: "Record identifier: AC-001; Record kind: asserted_context; Record disposition: pending; Validation errors: 0" + expected: ReviewRecord should be created with all required fields and pass validation against the model. + actual: ReviewRecord created with identifier AC-001, kind=asserted_context, disposition=pending, record_kind=worked_example; validate_record returns zero errors; all required fields accepted. + +- id: M3-novice-05 + severity: positive + location: notebooks 01, 02, 03 concept-statement cells + quote: "This notebook introduces `requirement def`; after running it you can declare a formal requirement with a subject and a constraint expression." (nb01); "This notebook introduces `attribute :>>` override; after running it you can override an inherited attribute value on a named usage." (nb02); "This notebook introduces the `asserted_context` judgment record; after running it you can declare and inspect the assumptions that frame an engineering requirement." (nb03) + expected: Each concept-statement cell should be exactly one sentence stating what the notebook teaches. + actual: All three concept-statement cells are single sentences (one terminal period each) and accurately state what each notebook teaches. + +- id: M3-novice-06 + severity: positive + location: notebooks 01, 02, 03 seam addressing (behavior without naming Tall/three worlds) + quote: "TIMELY_TOAST_REQ printed; TOASTER_INCREMENT created; model.ok = True; req found: True; requirement kind: requirementDef" (and similar for nb02 and nb03) + expected: The seam between SysML text, the tool loading it, and rendered results should be addressed in behavior without naming the three worlds or Tall lens. + actual: Each notebook shows: (1) SysML/Python text printed/defined, (2) loaded via opensysml.connect() and load_from_content(), (3) results queried/validated back. The three stages are distinct and observable; a reader can point to each one without the text ever naming "Tall" or "three worlds." + +- id: M3-novice-07 + severity: positive + location: conclusion.md + quote: "The Chapter 2 model adds `TimelyToast`, a requirement definition... No analysis in this chapter derives a cycle time, so neither usage's relationship to the bound is reported as a settled pass/fail verdict; the context record's estimate for `nominal` is likewise conditional on its stated assumption..." + expected: conclusion.md should have three paragraphs (what was built / what this establishes / what comes next) plus an exercise pointer. + actual: conclusion.md has three paragraph sections: "What we built", "What this establishes", "What comes next", followed by "Exercise:" pointer. Structure matches specification. diff --git a/decisions/user-testing-grid/M3-practitioner.md b/decisions/user-testing-grid/M3-practitioner.md new file mode 100644 index 0000000..90d37b0 --- /dev/null +++ b/decisions/user-testing-grid/M3-practitioner.md @@ -0,0 +1,66 @@ +LEARNER M3-practitioner — SE Practitioner — Ch6 + +EXECUTION RESULTS: +- nb01 cell2(load)/cell12(TOASTER_INCREMENT+assert): model.ok=True, no diagnostics +- nb01 cell14: neg_ok=False | diagnostic: "unresolved reference: HeatingAssembly::undefinedSlot" +- nb01 cell16: output=`find_allocations` shows `HeatingAssembly::heatGenAllocation` with source ends `['HeatingSystem::applyHeat','ApplyHeat::generateHeat']`, target `['HeatingAssembly::heatGen']`; `perform_relationships` shows `HeatGenerator`→`GenerateHeat` — matches index.md's promised result exactly +- nb02 cell4(load): model.ok=True +- nb02 cell40: neg_ok=False | diagnostic: "unresolved reference: UndefinedCarrier" +- nb02 cell42: output=`rated.power=800W, heatGenerationReq(rated)=True`; `weak.power=400W, heatGenerationReq(weak)=False` +- nb02 cell16/cell30: `validate_record` → `[]` for AC-C06 and AS-C06 +- nb03 cell2(load): model.ok=True +- nb03 cell4: neg_ok — AI-BAD (empty premises) correctly fails: `['asserted_inference requires at least one premise (Hawkins §3.1)']` +- nb03 cell18: `validate_record`→`[]` for AI-C06; Premises=['AC-C06','AS-C06','AS-C03','AI-C04'] + +All nine negative/positive checks in index.md's "Expected result" paragraph verified against real output. + +NARRATIVE OBSERVATIONS: +1. "a resistive element responds to being switched on and off directly...while a combustion source needs separate ignition and fuel-metering machinery" — real engineering reasoning, explicitly flagged as a domain premise not derived from the model, with counterevidence ("does not rule out a combustion design... Joule heating's own relation... is still not modeled") honestly bounding the claim. Not hand-waved. +2. "energyIn has no producer wired to it... HeatingAssembly is not yet composed into any Toaster candidate... The 600 W threshold is not derived from any stated measure of effectiveness" — AI-C06 is candid about exactly what the stopping rule does and does not show; no overclaiming. +3. "model_ref=selection_model_ref" (= `"ToasterDemo::ResistanceCoil"`) — the anchor string is never resolved against the loaded model anywhere in the three notebooks, and `validate_record()` does not check it either; a stale or mistyped anchor would pass silently. + +STRUCTURAL CHECKS: +- Cell 0 one sentence: yes (each is grammatically one semicolon-joined sentence, though dense) +- Cell 5/seam addressed without naming it: yes — each notebook's closing markdown cell ("the query results confirm...", "the evaluated results confirm...", "the evaluated results above, not the model's own declaration, are what this stopping judgment cites") points at printed SysML text, the load/validate step, and the query/eval output as three distinct, connected things a reader just watched happen. +- Cell 6 one sentence: yes +- conclusion.md three paragraphs + exercise reference: yes + +NOTE ON CONTRACT PREMISE: the contract describes "a new subject_ref/ReviewRecordRef judgment-record anchor mechanism...retrofitted in the last 24 hours." No such type exists anywhere in the repo (`grep -r "ReviewRecordRef\|subject_ref"` returns nothing). What exists is `ReviewRecord.model_ref`, a plain string field present since the init commit (`git log -S model_ref` → `b095107`), used across AC-C06/AS-C06/AI-C06 plus the AI-BAD negative control. I report what I found rather than confirming the contract's framing, which does not match this worktree. + +OVERALL: PASS — all cells execute as claimed, negative controls fire correctly, and the judgment records are honestly scoped; the one real gap is that the model_ref anchor is unvalidated. + +## Structured findings +- id: M3-practitioner-01 + severity: positive + location: chapters/ch06-recursive-decomp/02-second-level.ipynb cell 20/24/28 (AS-C06 claim/premises/counterevidence) + quote: "a resistive element responds to being switched on and off directly...while a combustion source needs separate ignition and fuel-metering machinery to do the same" + expected: a mechanism selection argued from a stated, checkable premise with honest limits + actual: argued from an explicit domain premise, confirmed model fact (ControlSystem::durationOut), and counterevidence that concedes it does not rule out combustion and that Joule heating is not yet modeled — real engineering judgment, not hand-waved + +- id: M3-practitioner-02 + severity: positive + location: chapters/ch06-recursive-decomp/03-stopping-judgment.ipynb cell 16 (counterevidence) + quote: "energyIn has no producer wired to it... The 600 W threshold is not derived from any stated measure of effectiveness" + expected: AI-C06 to state plainly what the stopping judgment does not establish, per AGENTS.md 1.6 + actual: four concrete, specific gaps listed (unwired energyIn, uncomposed HeatingAssembly, underived threshold, undecomposed sibling flows) — matches the honesty bar + +- id: M3-practitioner-03 + severity: confusing + location: src/toaster/evidence.py ReviewRecord.model_ref; chapters/ch06-recursive-decomp/02-second-level.ipynb cells 16/30, 03-stopping-judgment.ipynb cell 18 + quote: "framing_model_ref = \"ToasterDemo::heatGenerationReq\"" (and the parallel selection_model_ref, model_ref assignments) + expected: a judgment-record "anchor" mechanism that is actually checked against the model (e.g. model.find(model_ref) resolving, or validate_record() flagging an unresolved one) + actual: model_ref is a free-text string; no cell in any of the three notebooks resolves it against the loaded model, and validate_record() does not check it at all — a stale or mistyped anchor passes silently + +- id: M3-practitioner-04 + severity: confusing + location: work contract M3-PRACTITIONER (orchestrator-supplied), vs. src/toaster/evidence.py and repo-wide grep + quote: "a new subject_ref/ReviewRecordRef judgment-record anchor mechanism...across three real records plus a fixed negative control" + expected: the described subject_ref/ReviewRecordRef mechanism to be present and reviewable in this worktree + actual: no such name exists anywhere in the repo; the actual mechanism is the pre-existing model_ref string field (present since the init commit), not a new retrofit — the contract's description does not match the code + +- id: M3-practitioner-05 + severity: cosmetic + location: chapters/ch06-recursive-decomp/01-subsystem-requirements.ipynb cell 0, 02-second-level.ipynb cell 0, 19/45/21 (exercise pointers) + quote: "after running it you can see the nest-and-carry pattern that gave `ApplyHeat` its own logical carrier recur one level deeper" + expected: a concept/exercise statement a learner reads at a glance + actual: grammatically one sentence but semicolon/parenthetical-dense, longer than a typical "one sentence" reads; not blocking for a practitioner persona but adds parse time diff --git a/decisions/user-testing-grid/M3-returning.md b/decisions/user-testing-grid/M3-returning.md new file mode 100644 index 0000000..4c54efe --- /dev/null +++ b/decisions/user-testing-grid/M3-returning.md @@ -0,0 +1,64 @@ +LEARNER M3-returning — Returning Learner — Ch10 + +EXECUTION RESULTS: +- nb01 cell2: ok=True | model loaded from models/ch10-cumulative.sysml, no diagnostic +- nb01 cell4: neg_ok (bad.ok)=False | undeclared-feature allocate correctly fails to load +- nb01 cell13: output=coverage dict: heatGenerationReq covered=True (satisfied_by=['rated'], failed_by=['weak']); timely covered=False (satisfied_by=[], failed_by=['slow']); energyConservationReq covered=False (by design) +- nb01 cell18/33: output=deliveredEnergyBoundedBySupply tied to no requirement before remediation (False), tied after (True), via EnergyConservationReq::c subsetting +- nb01 cell29: output=all three assert-satisfy bindings FAIL identically ("no value for feature heatGenCheck.efficiency"), confirming the no-assert-satisfy design choice +- nb01 cell42/44: output=sysmlv2 verify --solve emits nothing for EnergyConservationReq without assert satisfy; emits "undecided" when assert satisfy is added back in a scratch file — negative control runs as described +- nb02 cell4: neg_ok=False | "asserted_inference requires at least one premise (Hawkins §3.1)" +- nb02 cell14: output=ledger of AS-C06 (undetermined), AS-C08 (supported), AI-C06 (undetermined), all disposition=pending, record_kind=worked_example +- nb03 cell4: neg_ok=False | "counterevidence is empty" +- nb03 cell24: output=AI-C10 validates with 0 errors, disposition=pending, record_kind=worked_example, engineering_conclusion=undetermined + +All three notebooks executed end-to-end via `jupyter nbconvert --execute` from the chapter's own directory with zero cell errors. + +NARRATIVE OBSERVATIONS: +1. "Chapter 10 builds the full traceability graph this chapter's coverage report only samples one join of" (Ch9 conclusion.md) — delivered: nb01 builds a real two-requirement graph plus a reverse "is this proof tied to any requirement" search, closing a real gap (deliveredEnergyBoundedBySupply) live in the notebook, not asserted in prose. +2. "its own need was identified retroactively, after Chapter 8's proof already existed" (AC-C10, verified in cell 36/48 output) — honestly disclosed rather than smoothed over; the premise count in AI-C10 (5 items: nb01's graph, AS-C06, AS-C08, AI-C06, AC-C10) matches what notebook 03 actually cites, no inflated count found. +3. My work contract described this chapter as retrofitted with "a judgment-record anchor mechanism and a corrected exemption-rule narration (a factual overclaim... fixed by independent review)." I could not find this anywhere in my checked-out worktree: `src/toaster/evidence.py`'s `ReviewRecord` has no `subject_ref`, `tag`, or "exemption" field at all, and `git merge-base --is-ancestor 7a3e135 HEAD` (the commit recording "DL-084: Hawkins judgment records anchored to the model") returns false — that work, and the related Task 9/10 "exemption overclaim" fixes, exist elsewhere in repo history but are not on this branch (`user-testing/browser-pass1`, HEAD=120c65d). I executed and checked internal consistency of the chapter as it actually exists here; I did not fabricate verification of a retrofit that isn't present. + +STRUCTURAL CHECKS: +- Cell 0 one sentence: yes (nb01/02/03 each one long semicolon-joined sentence, consistent with this tutorial's established dense style) +- Cell 5 / seam addressed without naming it: yes — concretely pointable in nb01 cells 42/44: the same SysML text (`models/ch10-cumulative.sysml`, with and without an added `assert satisfy` line) is read by two different tools (`model.verify_satisfaction()` vs. `sysmlv2 verify --solve`), producing two different printed verdicts ("require condition evaluation failed" vs. "undecided, indeterminate over unbound features") — model text, tool, and rendered result are all three concretely distinguishable from what actually printed, with no "three worlds" language anywhere. +- Cell 6 one sentence: yes, each notebook's final cell is one sentence pointing to exercises/ch10/exercise.ipynb +- conclusion.md three paragraphs + exercise reference: yes (What we built / What this establishes / What comes next, plus Exercise section) + +OVERALL: PASS (for the chapter as actually checked out on this branch) — all cells execute, outputs are internally consistent with index.md/conclusion.md's claims, and the one premise-count/exemption check the contract asked for checks out; but the contract's framing assumed a retrofit (anchor mechanism, exemption-rule fix) that is not present in this worktree, which I could not verify and am flagging rather than guessing at. + +## Structured findings +- id: M3-returning-01 + severity: positive + location: chapters/ch10-traceability-signoff/01-traceability-graph.ipynb cells 42/44 + quote: "Lines mentioning EnergyConservationReq/energyConservationReq: []" vs "companion-check-scratch/ch10_with_satisfy.sysml:291:9 c (ConstraintUsage, satisfies ToasterDemo::energyConservationReq): undecided (result is indeterminate over unbound features)" + expected: The Tall seam (model text / tool / rendered result) should be addressable in behavior without naming it. + actual: Confirmed concretely — the same underlying model text produces two different tool outputs depending on which tool (model.verify_satisfaction vs sysmlv2 verify --solve) and which variant of the text is used; all three elements are independently pointable from real printed output. + +- id: M3-returning-02 + severity: confusing + location: work contract (orchestrator-provided) vs. src/toaster/evidence.py on branch user-testing/browser-pass1 (HEAD 120c65d) + quote: "a judgment-record anchor mechanism and a corrected exemption-rule narration (a factual overclaim was found and fixed by independent review)" + expected: Chapter 10's AI-C10/AC-C10 records should show a subject_ref/anchor field and exemption narration I could check for the described overclaim-then-fix. + actual: ReviewRecord (src/toaster/evidence.py) has no subject_ref, tag, or exemption-related field; `git merge-base --is-ancestor 7a3e135 HEAD` returns false, confirming DL-084 and the related Task 9/10 exemption-overclaim fix commits are not ancestors of this branch's HEAD. The described retrofit is not present on this branch, so I could not evaluate it. + +- id: M3-returning-03 + severity: positive + location: chapters/ch10-traceability-signoff/03-engineering-signoff.ipynb cell 18 (premises list) + quote: "its own premises are literally the other two notebooks' findings" (index.md Method section) + expected: AI-C10's premises list should correspond exactly to notebook 01's graph findings plus notebook 02's three records (and AC-C10). + actual: Verified directly — 5 premises: (1) notebook 01's coverage/tie summary, (2) AS-C06, (3) AS-C08, (4) AI-C06, (5) AC-C10. Matches the claim; no inflated or missing premise found. + +- id: M3-returning-04 + severity: cosmetic + location: decisions/user-testing-grid/ (directory did not exist before this report) and docs/superpowers/specs/2026-10-01-large-scale-user-testing-design.md + quote: "see docs/superpowers/specs/2026-10-01-large-scale-user-testing-design.md for the exact schema" + expected: A findings-schema spec file to confirm the structured findings format against. + actual: That file does not exist anywhere in this worktree (`find . -iname "*large-scale-user-testing*"` returns nothing). I used the schema given verbatim in the work contract instead. + +- id: M3-returning-05 + severity: positive + location: chapters/ch10-traceability-signoff/02-judgment-synthesis.ipynb cell 14; chapters/ch09-coverage-sufficiency/conclusion.md "What comes next" + quote: "Chapter 10 builds the full traceability graph this chapter's coverage report only samples one join of, and asks what a real sign-off over that graph would actually require." + expected: As a Returning Learner fresh off Ch9, I expect Ch10 to extend coverage into a full graph and address sign-off directly. + actual: Delivered as promised — nb01 builds the graph, nb02 builds a ledger of three real records (not asserted about), nb03 explicitly states in prose why the synthesis record is not sign-off itself. No gap between Ch9's promise and Ch10's delivery. diff --git a/decisions/user-testing-grid/M4-novice.md b/decisions/user-testing-grid/M4-novice.md new file mode 100644 index 0000000..0be7ef6 --- /dev/null +++ b/decisions/user-testing-grid/M4-novice.md @@ -0,0 +1,69 @@ +# Grid cell M4-novice — local clone exercise, Novice persona, Chapter 1 + +## First-run-unmodified result (required, separate from the fill-in attempt) + +The exercise notebook runs cleanly without errors. Each of the four cells executes successfully and produces output. However, all four cells report `model.ok: False` because the placeholder comments (`# Your solution here`) contain no SysML code. The notebook does not crash; it simply demonstrates the incomplete state: + +``` +Step 1 ok: False +Step 2 ok: False +Step 3 ok: False +Step 4 ok: False +``` + +The verification code on step 4 produces no output for CoffeeMaker parts because the model is incomplete. + +## Fill-in attempt + +A Novice learner reading Chapter 1 can successfully apply all four constructs to a coffee maker domain without getting stuck. + +Following the exact pattern from notebook 01 (item defs + action def with doc + abstract part def performing an action), I declared `Beans` and `Coffee` item defs, a `BrewCoffee` action def with a doc, and an abstract `BrewingSystem` part def that performs it. Step 1 loaded cleanly. + +Step 2 followed notebook 02's pattern exactly: two bare `part def` declarations for `BrewUnit` and `HeatExchanger` terminated with semicolons, no attributes yet. Step 2 loaded. + +Step 3 applied notebook 03's specialization: `part def CoffeeMaker :> BrewingSystem;` — the concrete whole specializing the abstract concept. Step 3 loaded. + +Step 4 followed notebook 04's composition pattern: opening `CoffeeMaker` with a body, adding named parts (`part brew : BrewUnit;` and `part heat : HeatExchanger;`). The model loaded, and `model.find("CoffeeDemo::CoffeeMaker").parts()` returned both parts correctly. + +All four steps reported `model.ok: True`. The exercise's own verification code confirmed both parts present: `['CoffeeDemo::CoffeeMaker::brew', 'CoffeeDemo::CoffeeMaker::heat']`. + +## Structured findings + +- id: M4-novice-01 + severity: positive + location: cell-01 (Step 1: item defs + action def) + quote: "item def Beans; item def Coffee; action def BrewCoffee { doc /* ... */ in beans : Beans; out coffee : Coffee; } abstract part def BrewingSystem { perform action brewCoffee : BrewCoffee; }" + expected: "Following Chapter 1 notebook 01's pattern, this bundle of declarations should load without error" + actual: "All four declarations loaded without error; model.ok: True" + +- id: M4-novice-02 + severity: positive + location: cell-02 (Step 2: two part defs) + quote: "part def BrewUnit; part def HeatExchanger;" + expected: "Two bare part definitions following Chapter 1 notebook 02's pattern should load without error" + actual: "Both declarations loaded without error; model.ok: True" + +- id: M4-novice-03 + severity: positive + location: cell-03 (Step 3: specialization) + quote: "part def CoffeeMaker :> BrewingSystem;" + expected: "Specialization syntax from Chapter 1 notebook 03 should transfer to a new domain without modification" + actual: "Specialization loaded without error; model.ok: True" + +- id: M4-novice-04 + severity: positive + location: cell-04 (Step 4: composition) + quote: "part def CoffeeMaker :> BrewingSystem { part brew : BrewUnit; part heat : HeatExchanger; }" + expected: "Following Chapter 1 notebook 04's composition pattern should load, and model.find().parts() should return both parts" + actual: "Model loaded without error; model.ok: True; parts() returned ['CoffeeDemo::CoffeeMaker::brew', 'CoffeeDemo::CoffeeMaker::heat']" + +- id: M4-novice-05 + severity: positive + location: exercise.ipynb (problem statement) + quote: "Work through these steps in the cells below: 1. Declare item defs... 2. Add `part def`... 3. Declare... 4. Complete..." + expected: "Exercise instructions should map clearly to Chapter 1 constructs and allow a learner to transfer knowledge from chapter to exercise" + actual: "Novice learner successfully applied Chapter 1 patterns without modification, completing all four steps as written" + +## Overall + +**PASS** — A Novice learner who read Chapter 1 can successfully model a coffee maker's structure using the same four constructs (item def, action def, abstract part def, specialization, composition) applied to a new domain, completing the exercise without getting stuck. diff --git a/decisions/user-testing-grid/M4-practitioner.md b/decisions/user-testing-grid/M4-practitioner.md new file mode 100644 index 0000000..1cedb08 --- /dev/null +++ b/decisions/user-testing-grid/M4-practitioner.md @@ -0,0 +1,36 @@ +# Grid cell M4-practitioner — local clone exercise, SE Practitioner persona, Chapter 8 + +## First-run-unmodified result (required, separate from the fill-in attempt) +Ran `exercises/ch08/exercise.ipynb` top to bottom, unmodified, via `jupyter nbconvert --execute`. +Cell 1 (model load): `Model ok: False`, two diagnostics — `(line 2) expected '{' or ';' after declaration`, `(line 2) expected a namespace member` (the placeholder `# Your solution here` body is not valid SysML, as expected). No assertion here; execution continues. +Cell 2 (narrative, no code). +Cell 3: `sym = model.find("CoffeeDemo::deliveredMassBoundedBySupply"); assert sym is not None, "deliveredMassBoundedBySupply not found in the loaded model"` — raises uncaught `AssertionError: deliveredMassBoundedBySupply not found in the loaded model`. Execution halts here under a normal "Run All"; nothing after this cell runs. + +## Did it feel like a bug, or like an obvious "fill this in" state? +Genuinely mixed, and that's the finding. Cell 1's `Model ok: False` with diagnostics reads exactly like Ch1's exercise — an expected, informative "you haven't written anything yet" state. But Cell 3 immediately converts that into a raw Python traceback with no framing, one cell after a println pattern that trained me to expect graceful printed status. Ch1's exercise never asserts past an unfilled placeholder; it only prints `ok` and conditionally inspects with `if cm:`. Hitting an uncaught `AssertionError` two cells later, with no comment preparing me for it, reads as "the exercise itself is broken" on first encounter — I only recognized it as a placeholder consequence because I already knew Ch1's gentler pattern and could infer the asymmetry. + +## Fill-in attempt +Mirroring Ch8-01's `HeatGenerator`/`heatGenCheck` pattern (abstract part def with free attributes, a top-level unbound usage, a duration attribute, one `assert constraint` whose antecedent restates the bound and whose consequent restates the calc), I wrote a self-contained `WaterMover` with `throughput`, `transferEfficiency`, `transferEfficiencyBounded`, `deliveredMass`, plus `moverCheck`/`moverCheckDuration`/`deliveredMassBoundedBySupply`. First attempt failed: unqualified `DimensionOneValue` (works inside `ch07-cumulative.sysml`'s own import context, not reproducible standalone) needed `MeasurementReferences::DimensionOneValue`, and an `SI::'kg/s'` unit literal didn't resolve. After removing/qualifying those, `model.ok == True`. The exercise's own hint text ("mirroring efficiency/deliveredEnergy exactly") doesn't flag that the qualified-name requirement differs outside the chapter's own fixture file — Ch8 doesn't teach this, Ch7's own notebooks never show it either. I did not attempt the `verify_holds()` companion cells (needs the local `sysmlv2`/Z3 binaries, which errored in the first-run pass with build/FD-poll noise unrelated to my content). + +## Structured findings +- id: M4-practitioner-01 + severity: blocking + location: exercises/ch08/exercise.ipynb, cell 3 (In[2] in executed output) + quote: "AssertionError: deliveredMassBoundedBySupply not found in the loaded model" + expected: An unfilled placeholder fails the same gracefully-printed way Ch1's exercise does (print + conditional check), or cell 3 is commented to say the assert is deliberate and expected to fail until filled in. + actual: Raw uncaught AssertionError halts the notebook two cells after a printed-status pattern that trained the opposite expectation. +- id: M4-practitioner-02 + severity: confusing + location: exercises/ch08/exercise.ipynb, `source` placeholder / scalar-type usage + quote: "mirroring efficiency/deliveredEnergy exactly" + expected: Mirroring the chapter's dimensionless-attribute declaration works unchanged in a standalone companion. + actual: Unqualified DimensionOneValue resolves inside ch07/ch08-cumulative.sysml's own import context but not standalone; needed MeasurementReferences::DimensionOneValue, not mentioned by either chapter. +- id: M4-practitioner-03 + severity: positive + location: exercises/ch08/exercise.ipynb, cell 1 + quote: "Model ok: False" + expected: Placeholder content fails to parse with informative diagnostics. + actual: Matches exactly, same as Ch1 and other chapters' exercises. + +## Overall +NEEDS-FIX — confirmed: the uncaught AssertionError in cell 3 is real and reproduces on a fresh, completely unmodified run; it is a scaffolding inconsistency against Ch1's own gentler pattern, not a sign the exercise concept is unreachable. diff --git a/decisions/user-testing-grid/M4-returning.md b/decisions/user-testing-grid/M4-returning.md new file mode 100644 index 0000000..67e5568 --- /dev/null +++ b/decisions/user-testing-grid/M4-returning.md @@ -0,0 +1,43 @@ +# Grid cell M4-returning — local clone exercise, Returning Learner persona, Chapter 10 + +## First-run-unmodified result (required, separate from the fill-in attempt) +Ran `exercises/ch10/exercise.ipynb` exactly as committed via `uv run jupyter nbconvert --to notebook --execute`. Cells 0-1 (markdown) render fine. Cell 2 (first code cell) raises `AssertionError` at `assert model.ok, ...`: `source = """\n# Your solution here\n"""` is not valid SysML, so `conn.load_from_content` returns `model.ok=False` with diagnostics `(line 2) expected '{' or ';' after declaration` / `(line 2) expected a namespace member`. Execution stops there; cells 3-59 never run. This is consistent with a deliberate "fill in the blank" exercise, not a genuine bug: the placeholder is an obvious comment, the assertion message names exactly what's missing ("source is missing CoffeeDemo::deliveredMassBoundedBySupply -- paste your full Chapters 1-8 model"), and every other exercise chapter (checked ch06, ch09) uses the identical `"""\n# Your solution here\n"""` pattern with no committed solution anywhere in the repo. + +## Fill-in attempt +As Returning Learner I do not have a real saved Chapters 1-8 coffee-maker source (nothing is committed for any exercise chapter — checked ch01-ch09, all are unfilled placeholders too). I reconstructed a plausible Chapters 1-8-equivalent model from scratch, mirroring `models/ch08-cumulative.sysml`'s structure 1:1 into the coffee domain using the exact construct names the exercise notebook itself discloses (`CoffeeDemo`, `BrewUnit`, `BrewAssembly`, `WaterMover`, `Impeller`, `rated`/`weak`, `brewReq`, `tempCheck`, `deliveredMassBoundedBySupply`, `BrewStart`/`BrewCancel`). It loaded (`model.ok=True`), and Step 1's query helpers worked correctly for `brewReq`: `requirement_subject` found `wm:WaterMover`, `allocations_for`/`supertypes_transitively` traced the allocation and realization chain, and `requirement_coverage` returned the expected bidirectional result (`satisfied_by=['rated']`, `failed_by=['weak']`), matching Chapter 10's real shape exactly. `tied_to_any_requirement` correctly returned `False` for my lemma, matching the gap the real chapter finds. I hit a genuine snag on `tempCheck`: my `assert not satisfy tempCheck by hot.brewUnit;` (targeting a nested feature path) produced `failed_by=[]` instead of `['hot']` — `requirement_coverage()` apparently expects the asserted subject to be a top-level part usage, not a dotted path, which nothing in Chapters 1-9's own material states explicitly. I stopped there rather than fabricate chapters 1-9 state I never actually built. + +This does feel like real capstone synthesis in structure — it genuinely requires every prior chapter's constructs (actions, perform, allocation, requirements, verification, state machines) working together — but in practice its biggest cost is not conceptual synthesis, it's that nothing carries forward automatically between exercise notebooks: a returning learner must have manually preserved their own full model text across nine prior sessions, with zero scaffolding or checkpoint file in the repo to recover it from. + +## Structured findings +- id: M4-returning-01 + severity: positive + location: exercises/ch10/exercise.ipynb cell 2 + quote: "assert model.ok, f\"Model failed: {format_diagnostics(model.diagnostics)}\"" + expected: unmodified placeholder fails with a diagnostic pointing at what's missing. + actual: fails exactly as expected, with a clear message naming the missing construct; not a bug. +- id: M4-returning-02 + severity: friction + location: exercises/ch10/exercise.ipynb cell 2 / Problem section + quote: "paste your own full, completed Chapters 1-8 coffee-maker model as source" + expected: a returning learner can proceed from what Chapters 1-9 taught. + actual: proceeding requires a full, self-preserved copy of 8 prior chapters' model text that the repo never checkpoints anywhere; nothing here is recoverable if that copy is lost. +- id: M4-returning-03 + severity: confusing + location: exercises/ch10/exercise.ipynb Step 3 (ch06_source_pre_impeller / ch06_source_final) + quote: "Step 3 asks for TWO ch06-stage placeholders... do not collapse them into one shared placeholder" + expected: one accumulated model carried forward, consistent with how chapters build. + actual: requires separately preserving two distinct intermediate states from chapter 6 alone, on top of the full chapters 1-8 model — a bookkeeping burden with no tooling support. +- id: M4-returning-04 + severity: positive + location: Step 1 query cells (requirement_subject, allocations_for, requirement_coverage) + quote: "satisfied_by=['rated'], failed_by=['weak']" + expected: query helpers work the same way against a freshly-authored coffee-maker model as against the toaster. + actual: worked correctly on the first attempt for brewReq, confirming the helpers genuinely generalize. + +## Overall +NEEDS-FIX (process/scaffolding, not content) — the unmodified notebook fails exactly as a fill-in-the-blank should; the capstone's real difficulty is that no exercise chapter ever checkpoints a learner's accumulated model, forcing a returning learner to reconstruct or preserve nine chapters of state entirely outside the repo's own support. + +--- +Branch: worktree-agent-a017a9d55e4d4e1eb (worktree off main, contract specified user-testing/browser-pass1; this worktree's actual branch lineage did not include that branch — see note below) +Model run on: Sonnet 5 (matches contract and the user-testing SKILL.md persona table for Returning Learner) +Could not execute: full Steps 2-4 of the fill-in attempt (remediation, judgment ledger, synthesis record) — stopped after Step 1 surfaced a real coverage-query snag on tempCheck, to avoid fabricating chapters 1-9 state never actually built. Also note: `docs/superpowers/specs/2026-10-01-large-scale-user-testing-design.md`, which this contract cites, does not exist in this worktree's branch history (only on `user-testing/browser-pass1`, not an ancestor of this worktree's base); I read its M4 row via `git show 6d404b1:...` instead of a local file read. diff --git a/docs/contributor.md b/docs/contributor.md index b6a1228..dab503d 100644 --- a/docs/contributor.md +++ b/docs/contributor.md @@ -4,6 +4,44 @@ This page is for maintainers working on the tutorial itself, not learners workin It assumes you can read Python and SysML and that you have the environment from [Getting Started](setup.md) already set up. +## How this repo is built, tested, and reviewed + +The tutorial's content — every chapter notebook, exercise, and model file — is built and +reviewed through a small multi-agent harness that lives alongside the content itself. Three +root files and two directories carry that harness, and they're worth knowing about before you +touch anything, even if you never run an agent yourself: + +- **`AGENTS.md`** states what the tutorial teaches (the functional/logical/physical layering, + where its terms come from, how models are built and queried) and the legacy role roster that + used to own each file. It's the harness's own foundational reference, read first by every + agent role before it does anything else. +- **`CLAUDE.md`** is the entry point: read order, the glossary CLI, and the skill index below. +- **`DEFERRED.md`** tracks known gaps in the toolchain (OpenSysML, sysml-toolkit) that the + tutorial works around — what the workaround is, why it's needed, and the condition under + which it comes out once the upstream gap closes. +- **`.claude/agents/`** defines the roles that do the work: an `orchestrator` that turns a + request into scoped contracts and integrates results; `builder`/`reviewer` pairs that + implement and independently check each change (always on different models, never the same + one reviewing its own work); a `layer-auditor` that classifies model elements against the + functional/logical/physical boundaries; a `simulated-learner` that executes a chapter as a + persona-assigned reader and reports what it found; and the `ace`, which triages questions + between the team and the tutorial's author, ruling where it can and escalating what it can't. +- **`.claude/skills/`** holds the how-to for each kind of work — the sub-notebook template and + pacing rules (`toaster-recipe`), the boundary tests for classifying a model element + (`architecture-layers`), the glossary's own usage rules (`tutorial-glossary`), the simulated + learner protocol (`user-testing`), and more — each one a reference an agent (or a human + contributor) reads before doing that kind of work, not after. +- **`decisions/`** is the record of what was decided and why: `log.md` (the running decision + log, one entry per substantive ruling or escalation), `work-contract-template.md` (the shape + of a task handed to a builder or reviewer), and `task-states.md` (what state a task is in and + what moves it to the next one). + +If you want to extend a chapter, clarify a definition, or review didactic content, the harness +tools above are built for exactly that — start at `CLAUDE.md`'s own read order rather than +improvising a workflow from scratch. The sections below cover specific maintenance tasks +directly; none of them require running an agent, but all of them follow conventions the harness +itself enforces (the recipe's pacing rule, the layer boundary tests, the review gate). + ## Deployment status Deployment to GitHub Pages is deliberately disabled (`.github/workflows/ci.yml`, the `deploy` diff --git a/docs/index.md b/docs/index.md index ccb02b1..181ce28 100644 --- a/docs/index.md +++ b/docs/index.md @@ -22,8 +22,8 @@ An executable tutorial on recursive system decomposition using SysML v2 and Open | 5: Architecture and Allocation | Which component performs it, and how do components connect? | model navigation, allocate, perform, port, interface | | 6: Recursive Decomposition | What does one branch of the recursion show, one level down? | nested action, abstract logical carrier, port, allocate, specialization, asserted_solution | | 7: Execution and Experiments | What does it do? | bounded calc, assert constraint, exhibit state, do action, execute_state, parameter sweep | -| 8: Constraint Checking | Does one claim hold at one point, or does a property hold for every value? | assert constraint, verify_satisfaction, verify_holds (Z3), stale records | +| 8: Checking and Revision | Does one claim hold at one point, or does a property hold for every value? | assert constraint, verify_satisfaction, verify_holds (Z3), stale records | | 9: Coverage and Sufficiency | Are all requirements covered? | requirement coverage, evidence sufficiency, stale detection at scale | | 10: Traceability and Sign-off | Is the argument complete? | traceability graph, judgment ledger, asserted_inference synthesis | -[Setup and installation](setup.md) | [Glossary](glossary.md) | [References](references.md) | [Case studies](case-studies/) +[Setup and installation](setup.md) | [Glossary](glossary.md) | [References](references.md) | [Case studies](case-studies/2026-09-30-energy-conservation-requirement-tie.md) diff --git a/docs/setup.md b/docs/setup.md index 60e37e2..c0a27d8 100644 --- a/docs/setup.md +++ b/docs/setup.md @@ -67,8 +67,9 @@ loads, validates, queries, and evaluates every model in this tutorial. Every cha **sysml-toolkit** does one thing OpenSysML cannot yet: prove that a constraint holds for every value of an unbound quantity, not just check it against one fixed value, using the Z3 solver. -No chapter currently uses this; it becomes relevant once Chapter 8 is re-derived to need it. -It is not on crates.io. The name `sysmlv2` is reserved on PyPI by sysml-toolkit's own +Chapter 8 uses it directly (`toaster.modelcheck.verify_holds`, wrapping its `sysmlv2 verify +--solve` CLI) to prove `deliveredEnergyBoundedBySupply` for every value its unbound features +admit. It is not on crates.io. The name `sysmlv2` is reserved on PyPI by sysml-toolkit's own maintaining organization, but the package published there today is a placeholder, not the real thing; do not `pip install` it. Get a working binary instead from [its GitHub releases page](https://github.com/Open-MBEE/sysml-toolkit/releases) (macOS, Linux, @@ -90,8 +91,19 @@ Fork the repository, provision the environment (above), then: 1. Read the worked example: open a chapter notebook in `chapters/` and run every cell. 2. Open the parallel exercise: `exercises/ch{N}/exercise.ipynb`. -3. The exercise asks you to apply the same construct or operation to a different part of the - toaster. The only tools it needs are the ones the chapter already introduced. +3. The exercise asks you to apply the same construct or operation to a different domain: a + coffee maker, built in parallel to the toaster throughout the tutorial. The only tools it + needs are the ones the chapter already introduced. The `exercises/` notebooks are blank workspaces. They are not pre-executed and not part of the CI pipeline. Work in them directly; do not modify the chapter notebooks while doing an exercise. + +**Keep your model between chapters.** Each exercise's first cell asks you to paste in your own +completed model from the previous chapter's exercise — there is no committed solution file to +load instead. Save the full `source` string your notebook ends with (for example, to a scratch +`.sysml` file in your own fork, or just keep the notebook itself open) before moving to the next +chapter's exercise, or you will have nothing to paste in. Chapter 6's own exercise is the one +case where you need to keep **two** separate snapshots, not one: the model state right before you +add `Impeller` (used by the mechanism-selection judgment, written before the mechanism it selects +exists) and the model state right after (used by the stopping judgment, and the one that carries +forward into Chapter 7). Chapter 6's own exercise notebook flags exactly where to save each one. diff --git a/docs/superpowers/plans/2026-10-01-ci-cd-deploy-readiness-plan.md b/docs/superpowers/plans/2026-10-01-ci-cd-deploy-readiness-plan.md new file mode 100644 index 0000000..564d0b2 --- /dev/null +++ b/docs/superpowers/plans/2026-10-01-ci-cd-deploy-readiness-plan.md @@ -0,0 +1,270 @@ +# CI/CD Deploy-Readiness Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Turn `.github/workflows/ci.yml`'s `deploy` job from a placeholder (`echo "Pages deployment placeholder"`) into the real, working seven-step pipeline `docs/contributor.md` already documents — so that flipping `if: false` is the *last* step, not the first, and actually produces a correct, live GitHub Pages site at `https://open-mbee.github.io/toaster/`. + +**Architecture:** No new service, no new language, no new dependency beyond what's already pinned (`mystmd`, `opensysml`, standard `actions/*` GitHub Actions for Pages). Seven CI steps, each independently testable locally before it's trusted in CI, matching `docs/contributor.md`'s own numbered list exactly. The `build` job (push + PR, read-only) gets steps 1–4 and 6 added to what it already does (`pytest`, tool checks); a separate `deploy` job (main-only, `pages: write`) gets steps 5 and 7, gated behind `build` passing and, once enabled, behind a branch guard `docs/contributor.md` already calls out as a prerequisite it doesn't yet have. + +**Tech Stack:** GitHub Actions (`actions/checkout`, `astral-sh/setup-uv`, `actions/setup-node`, `actions/configure-pages`, `actions/upload-pages-artifact`, `actions/deploy-pages`), `mystmd` (already pinned at 1.11.0), `opensysml` v0.9.0, Python 3.12+, Node 22 (`.nvmrc`). + +**Spec:** [`docs/contributor.md`](../../contributor.md)'s "Deployment status" section is the existing, authoritative spec for what the seven steps are and what order they run in — this plan does not redesign that pipeline, it implements it. Also read `.claude/skills/orchestrator-protocol/SKILL.md`'s WP-8 row (`site at /toaster locally; navigation works; downloads present; 7-step CI passes; Pages deploys`) for the acceptance bar this whole plan is scoped against. + +## Global Constraints + +- **The `deploy` job stays gated behind `if: false` until Task 7's branch guard is in place AND a human (Z) decides content is ready** (`docs/contributor.md`: "Enabling it is a decision the maintainer makes explicitly... not something a passing build should trigger on its own"). This plan builds and verifies every step *up to* that flip; flipping it is Task 8, done deliberately, separately, never bundled into the same PR/commit as a content change. +- **Single-platform CI only** (SA-4: `ubuntu-latest`). Do not add a matrix. +- **No custom CSS, no theme changes** (SA-5: default book-theme). +- **`exercises/` is excluded from notebook execution** in CI, same as the MyST build itself already excludes it (`myst.yml`'s `exclude: - exercises/**`) — exercises are blank learner workspaces, not CI-verified content. +- **Every new CI step must be runnable and verifiable locally first**, with the exact command recorded in this plan's own task text, before it's trusted inside a GitHub Actions YAML change — this project's own "probe before asserting" discipline (`ace-protocol`) applies to infrastructure the same as it applies to SysML constructs. +- **No `|| true` anywhere** (ace-protocol: "Gate verdicts only. `|| true` is banned everywhere"). A step that can fail must be allowed to fail the job. + +## Review Focus + +- **A notebook that raises mid-execution in CI must fail the build, not get silently skipped or reported as a warning.** `nbconvert`/`myst build --execute` have an `--allow-errors` style flag that does the opposite of what CI needs; Task 1's own acceptance test deliberately breaks a cell and confirms the build goes red, not green. +- **The `deploy` job's `pages: write` / `id-token: write` permissions must never be inherited by the `build` job.** `docs/contributor.md` already states this ("This step alone carries the `pages: write` permission; the `build` job does not") — Task 6's own review step confirms the permissions block is job-scoped, not workflow-scoped. +- **A PR build must never trigger a real deploy**, even after `if: false` is removed. This is `docs/contributor.md`'s own explicitly flagged gap (the workflow triggers on `pull_request` too, and today only `needs: build` gates `deploy`) — Task 7 is not complete until this is independently verified by opening a scratch PR and confirming `deploy` does not run. +- **The built site must resolve correctly under the `/toaster` subpath**, not just at a bare `localhost:3000` root the way local dev preview serves it — a relative-link or asset-path bug that's invisible in local dev can 404 everything once deployed under `https://open-mbee.github.io/toaster/`. Task 2's own acceptance test builds a static export and serves it from a `/toaster` subpath locally, not just the dev server's own root-path preview. +- **A broken internal link or a 404'd asset must fail the build**, not just get noticed by a human reading the deployed site later (which is exactly how last night's "Case studies" 404 and this session's earlier title-mismatch bugs were found — manually, after the fact). Task 5 exists specifically so this class of bug is caught by CI before it ships, not after. + +--- + +## Task 1: Wire real notebook execution into the `build` job (steps 1–2 of the spec) + +**Files:** +- Modify: `.github/workflows/ci.yml` + +**Interfaces:** +- Consumes: the existing `uv sync --locked` / `npm ci` / `scripts/check-tools.py` steps already in `build` (unchanged). +- Produces: a CI step that fails the whole job if any chapter notebook raises during execution. Task 3 and Task 5 both run *after* this step and depend on it having actually executed every notebook (not just loaded them). + +**Design decision this task makes, stated explicitly so it can be reviewed:** execute notebooks via `npx mystmd build --execute` itself (which re-executes every notebook as part of building the site), rather than a separate `jupyter nbconvert --execute` pass per notebook followed by a *second*, redundant execution inside the MyST build. `docs/contributor.md`'s step 2 ("execute every chapter notebook") and step 5 ("build the MyST site") read as two numbered steps but do not have to be two separate *mechanical* CI steps — collapsing them avoids executing every notebook twice (once via nbconvert, once via MyST) for no benefit. If local verification (Step 2 below) finds that `mystmd build --execute` does *not* fail loudly on a broken cell the way CI needs, fall back to a separate `nbconvert --execute` pass before the MyST build and say so in this task's own report — do not silently assume either behavior. + +- [ ] **Step 1: Probe `mystmd build --execute`'s failure behavior locally, before writing any CI YAML** + +```bash +# Deliberately break one cell to confirm the build actually fails loudly +cp chapters/ch01-system-purpose/01-abstract-def.ipynb /tmp/01-abstract-def.ipynb.bak +python3 - <<'EOF' +import json +nb = json.load(open("chapters/ch01-system-purpose/01-abstract-def.ipynb")) +nb["cells"][2]["source"] = ["raise RuntimeError('deliberate CI probe failure')"] +json.dump(nb, open("chapters/ch01-system-purpose/01-abstract-def.ipynb", "w")) +EOF +npx mystmd build --execute; echo "exit code: $?" +# Restore immediately, do not commit the broken state +cp /tmp/01-abstract-def.ipynb.bak chapters/ch01-system-purpose/01-abstract-def.ipynb +``` + +Expected and required: non-zero exit code. If `mystmd build --execute` exits 0 despite the raised error, this is a real, load-bearing finding — do not proceed to Step 2 with that command; use a prior `jupyter nbconvert --to notebook --execute` pass (which this session already confirmed, repeatedly, fails loudly and is already the pattern every judgment-record retrofit task this session verified against) as the CI execution step instead, and build the static site in a separate step afterward without `--execute` (consuming the already-executed notebook outputs on disk, or re-running `mystmd build --execute` a second time and accepting the double-execution cost — record which fallback was chosen and why). + +- [ ] **Step 2: Add the execution step to `.github/workflows/ci.yml`'s `build` job**, after the existing `Check tool versions` step and before any new staging/check step: + +```yaml + # Step 2: Execute every chapter notebook and build the MyST site + - name: Execute notebooks and build site + run: npx mystmd build --execute +``` + +(Or the nbconvert-first fallback from Step 1, if that's what the probe required — write the actual, verified-working YAML here, not this plan's own default guess.) + +- [ ] **Step 3: Confirm the full `build` job still passes on an unmodified checkout.** Run the equivalent sequence locally end to end (`uv sync --locked && npm ci && uv run python scripts/check-tools.py && npx mystmd build --execute`) and confirm it exits 0. + +- [ ] **Step 4: Commit.** +```bash +git add .github/workflows/ci.yml +git commit -m "CI: execute every chapter notebook and build the MyST site (step 2/5 of the documented pipeline)" +``` + +--- + +## Task 2: Add the `/toaster` base-path config and verify the built site resolves under it + +**Files:** +- Modify: `myst.yml` + +**Interfaces:** +- Consumes: nothing new. +- Produces: a `project.github`-adjacent base-URL config that makes every internal link, asset path and downloaded-figure reference resolve correctly when the site is served from `https://open-mbee.github.io/toaster/` instead of a bare domain root. Task 5's own link-checker runs against a site built with this config, not the bare-root dev-server config Task 1 and all of last night's/this session's manual testing used. + +- [ ] **Step 1: Confirm the config key.** MyST's base-path option for GitHub Pages project sites is documented at `https://mystmd.org/guide/deployment#deploy-base-url` (the same page `myst.yml`'s own "Site not loading correctly?" fallback banner, found during this session's browser testing, already links to) — read it directly and confirm the exact key name and YAML shape for the current pinned `mystmd` version (1.11.0) before writing it, rather than guessing from memory. + +- [ ] **Step 2: Add the config to `myst.yml`.** + +- [ ] **Step 3: Build a static export and serve it from a `/toaster` subpath locally, not the dev server's bare root:** + +```bash +npx mystmd build --execute --html +cd /tmp && python3 -m http.server 8080 --directory - <<'EOF' +# serve _build/html (or wherever this mystmd version emits static output) at /toaster/, +# e.g. by symlinking it into a parent dir named "toaster" and serving the parent +EOF +``` + +(Write the actual working local-verification commands here once Step 1's exact output directory and serving approach are confirmed — `mystmd`'s own docs or `--help` output names the real build output path; don't assume `_build/html` without checking.) + +- [ ] **Step 4: Visually/programmatically confirm** at least one chapter page, one figure, and one cross-chapter link resolve correctly under the `/toaster/...` prefix, not just at the bare root. + +- [ ] **Step 5: Commit.** +```bash +git add myst.yml +git commit -m "Configure the /toaster base path for GitHub Pages project-site deployment" +``` + +--- + +## Task 3: Wire the two existing, already-working conformance scripts into CI (step 3 of the spec) + +**Files:** +- Modify: `.github/workflows/ci.yml` + +**Interfaces:** +- Consumes: `scripts/check_construction.py` and `scripts/check_conformance.py`, both already working, already used throughout this session's own manual verification (`uv run python scripts/check_construction.py --check` was run as an acceptance gate on every one of last night's 15 implementation tasks), but **not currently called anywhere in CI** (confirmed by grep — this is a real gap, not a restatement of existing coverage). + +**Context on why this task is small:** `docs/contributor.md`'s step 3 ("assert expected outputs, diagnostics, negative controls, and review-record integrity") sounds like it might need new tooling, but it mostly doesn't — every chapter notebook already asserts its own expected output inline (`assert model.ok`, `assert not bad.ok`, `assert errors == [...]`, confirmed throughout this session's own retrofit work), so Task 1's own notebook execution already enforces most of step 3 as a side effect of failing loudly on any `assert`. What's missing is the two *cross-notebook, cross-chapter* checks that no single notebook's own assertions can cover: construction-zone-to-committed-fixture consistency (`check_construction.py`) and the staged project-conformance tiers (`check_conformance.py`). + +- [ ] **Step 1: Run both scripts locally against the current `main`/working branch to confirm they pass clean today** (a prerequisite for adding them to CI — if either currently fails, that's a separate, pre-existing bug to fix first, not something to paper over with `|| true`): +```bash +uv run python scripts/check_construction.py --check +uv run python scripts/check_conformance.py +``` + +- [ ] **Step 2: Add both as CI steps**, after the notebook-execution step from Task 1: +```yaml + # Step 3: Construction-zone and conformance checks + - name: Check construction zone consistency + run: uv run python scripts/check_construction.py --check + - name: Check project conformance + run: uv run python scripts/check_conformance.py +``` + +- [ ] **Step 3: Confirm the full `build` job still passes locally** with both new steps added (re-run the full local sequence from Task 1 Step 3, now including these two commands). + +- [ ] **Step 4: Commit.** +```bash +git add .github/workflows/ci.yml +git commit -m "CI: wire check_construction.py and check_conformance.py into the build job (step 3/5)" +``` + +--- + +## Task 4: Stage build artifacts (step 4 of the spec) — scope this down, don't invent a manifest format unasked + +**Files:** +- Modify: `.github/workflows/ci.yml` + +**Context — a genuine open question, not resolved by this plan:** `docs/contributor.md`'s step 4 names "executed notebooks, generated models, figures, and a provenance manifest." The first three already exist as ordinary build output once Task 1's execution step runs (executed notebooks are on disk; `models/*.sysml` are already committed, not generated fresh; `figures/*.svg` are already committed too, confirmed during this session's own browser testing of the Chapter 5 interconnection figure). **No provenance-manifest format or generator exists anywhere in this repo today** (confirmed by grep for "provenance" and "manifest" across `scripts/` and `src/toaster/`) — this is new, not wiring-up-the-existing the way Task 3 was. + +- [ ] **Step 1: Decide, with Z, whether a provenance manifest is actually required for the first real deploy, or a `next-passes.md`-tracked follow-up.** A minimal version (commit SHA, build timestamp, `mystmd`/`opensysml` versions, written as one JSON file into the build output) is cheap and low-risk if wanted now; a richer one (per-chapter execution hashes, content-addressed figure provenance) is a larger, separate design question this plan does not scope. **Do not build either without that decision** — this step is a checkpoint, not optional busywork to skip. + +- [ ] **Step 2 (only if Step 1 says "yes, minimal version now"):** add a small script (`scripts/write_provenance_manifest.py`) that writes `{commit, built_at, mystmd_version, opensysml_version}` as JSON into the build output directory, and one CI step that calls it between the build (Task 1) and the upload-artifact step (Task 6). + +- [ ] **Step 3: Commit**, if Step 2 was done. + +--- + +## Task 5: Automated post-build link and asset check (step 6 of the spec) + +**Files:** +- Create: `scripts/check_site_links.py` +- Modify: `.github/workflows/ci.yml` + +**Interfaces:** +- Consumes: the static site built by Task 1 + Task 2 (served locally under the `/toaster` prefix, same as Task 2's own verification). +- Produces: a CI step that fails the build if any internal link resolves to a 404, any image/figure fails to load, or any page is missing its expected title — the same class of check this session performed *manually* last night (a scripted `fetch()` loop over all 58 known page URLs, checking for `Document Not Found`, tracebacks, and non-200 statuses) and found one real bug with (the "Case studies" 404, already fixed). This task turns that manual, one-off script into a permanent, repeatable CI gate. + +- [ ] **Step 1: Write `scripts/check_site_links.py`**, adapting the exact approach this session used manually: start a local HTTP server against the built static output (served under `/toaster`, matching Task 2), enumerate every page MyST's own build manifest or `myst.yml`'s project TOC lists (don't hand-maintain a separate URL list that can drift from the real TOC the way this plan's own earlier browser-testing session found stale titles drifting from `myst.yml`), fetch each one, and fail (non-zero exit, printing every failing URL) if any: HTTP status is not 200, the response body contains `Document Not Found`, or any ``/figure fails to resolve (a HEAD request against every `src` found in the fetched HTML). This must also catch every `exercises/ch{N}/exercise.ipynb` link each chapter's `index.md`/`conclusion.md` carries: `myst.yml` excludes `exercises/**` from the build (`myst-publication` skill), so on the dev server these links resolve by falling through to a locally running Jupyter server, a fallback that does not exist on the static build — the user-testing grid's M2-novice cell found this ambiguity (`decisions/log.md` DL-085(5)) and it was unverified against a real static build until this task runs. If the static build cannot serve an excluded path, this is a real content bug (the link must be rewritten to the file's GitHub URL instead), not a checker gap to special-case around. + +- [ ] **Step 2: Run it locally against a `/toaster`-served build** (Task 2's own local serving setup) and confirm it exits 0 on the current, already-fixed site, then confirm it correctly fails (non-zero, with a clear message) when pointed at a deliberately-reintroduced broken link (e.g., temporarily re-break the "Case studies" link this session already fixed, confirm the script catches it, then revert). + +- [ ] **Step 3: Add it as a CI step**, after the build and staging steps: +```yaml + # Step 6: Check navigation, links, and assets under the /toaster base path + - name: Check site links and assets + run: uv run python scripts/check_site_links.py +``` + +- [ ] **Step 4: Commit.** +```bash +git add scripts/check_site_links.py .github/workflows/ci.yml +git commit -m "CI: add an automated link/asset checker for the built site (step 6/7)" +``` + +--- + +## Task 6: Replace the `deploy` job's placeholder with the real Pages deploy (step 7 of the spec) + +**Files:** +- Modify: `.github/workflows/ci.yml` + +**Interfaces:** +- Consumes: Task 1 through Task 5's build output. +- Produces: a real, working `deploy` job using the standard GitHub Actions Pages flow. **`if: false` stays in place through this entire task** — this task makes the job *correct*, not *enabled*; Task 8 is the separate, deliberate act of enabling it. + +- [ ] **Step 1: Replace the placeholder `deploy` job** with the standard three-action Pages sequence, keeping `if: false` and `needs: build` exactly as they are today, and keeping the existing `pages: write` / `id-token: write` permissions block scoped to this job only (confirmed in Review Focus above — do not move it to the workflow-level `permissions:` key): + +```yaml + deploy: + if: false # still disabled -- Task 8 flips this, deliberately, separately + needs: build + runs-on: ubuntu-latest + permissions: + pages: write + id-token: write + environment: + name: github-pages + url: ${{ steps.deployment.outputs.page_url }} + + steps: + - uses: actions/checkout@v4 + - uses: astral-sh/setup-uv@v4 + with: + version: "0.5.x" + - run: uv sync --locked + - uses: actions/setup-node@v4 + with: + node-version-file: .nvmrc + - run: npm ci + - run: npx mystmd build --execute --html + - uses: actions/configure-pages@v5 + - uses: actions/upload-pages-artifact@v3 + with: + path: + - name: Deploy to GitHub Pages + id: deployment + uses: actions/deploy-pages@v4 +``` + +(The `deploy` job rebuilds rather than reusing `build`'s own artifact, matching this repo's existing single-job-per-concern CI style and avoiding a cross-job artifact-passing mechanism this plan doesn't otherwise need — note this as a deliberate simplicity choice, not an oversight, when reviewing.) + +- [ ] **Step 2: Add the branch guard `docs/contributor.md` explicitly requires before enabling deployment** (quoted in that file: "Simply removing `if: false` would let `deploy` run on a successful pull-request build too... Add `if: github.ref == 'refs/heads/main'`"). Combine it with the existing `if: false` for now (`if: false && github.ref == 'refs/heads/main'`), so the guard is in place and reviewable before Task 8 ever needs to touch this line again — Task 8 then only has to delete the `false &&` prefix, nothing else. + +- [ ] **Step 3: Verify the guard independently, per this plan's own Review Focus.** Open a scratch PR against a throwaway branch (not `main`) with a trivial, reversible change, and confirm in the Actions tab that `build` runs but `deploy` is skipped — both because of `if: false` (expected either way right now) and, separately, trace through the YAML logic by hand to confirm that even with `false &&` removed, a PR-triggered run's `github.ref` would not equal `refs/heads/main` and `deploy` would still correctly skip. + +- [ ] **Step 4: Commit.** +```bash +git add .github/workflows/ci.yml +git commit -m "CI: replace deploy placeholder with the real Pages pipeline, still gated off (step 7/7, deploy stays disabled)" +``` + +--- + +## Task 7: Dry-run the complete pipeline before asking Z to flip anything + +**Files:** none (verification only). + +- [ ] **Step 1: Push this plan's branch and open a real PR**, so the full `build` job (Tasks 1–5) runs for real in GitHub's own CI environment, not just locally — confirm every step this plan added is green there, not only on a local machine that may have tool versions or caches the CI runner doesn't. +- [ ] **Step 2: Temporarily flip `if: false && github.ref == 'refs/heads/main'` to `if: true && github.ref == 'refs/heads/main'` on a throwaway branch only** (never on this plan's real PR branch, never on `main`), push it as its own disposable branch, and confirm the `deploy` job runs and either succeeds or fails informatively. Delete the throwaway branch afterward regardless of outcome; this step exists to prove the deploy job itself works, not to actually publish anything yet. +- [ ] **Step 3: Report results to Z**: every step's real CI run output, the dry-run deploy's outcome, and an explicit recommendation on whether Task 8 is ready. + +--- + +## Task 8: Enable deployment (Z's own action, not a subagent task) + +This is deliberately **not** broken into sub-steps here, per `docs/contributor.md`'s own stated philosophy: enabling deployment is a decision Z makes explicitly, once, when content is ready — not a task a builder executes as part of a normal contract. When Z decides to do it: remove `false && ` from the `deploy` job's `if:` condition (leaving only the branch guard), on its own commit, on `main`, separately from any content change. + +--- + +## Sequencing + +Tasks 1–6 have real dependencies on each other's outputs (Task 3/5/6 all need Task 1's execution step; Task 5 needs Task 2's base-path config) but touch the *same single file* (`.github/workflows/ci.yml`) in five of six cases — **run them strictly in series**, same reasoning as this session's own Hawkins-plan execution: a shared-blast-zone file forces serial dispatch regardless of logical independence. Task 4 is a genuine checkpoint-then-maybe-branch, not a hard dependency of 5 or 6. Task 7 depends on 1–6 all being merged. Task 8 depends on 7's dry run actually succeeding and is Z's own action. diff --git a/docs/superpowers/plans/2026-10-01-hawkins-judgment-record-anchor-plan.md b/docs/superpowers/plans/2026-10-01-hawkins-judgment-record-anchor-plan.md new file mode 100644 index 0000000..c038616 --- /dev/null +++ b/docs/superpowers/plans/2026-10-01-hawkins-judgment-record-anchor-plan.md @@ -0,0 +1,1117 @@ +# Hawkins Judgment Record Anchor — Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. In THIS repo, "subagent-driven" means dispatching through the existing `builder`/`reviewer` agent roles and `orchestrator-protocol`'s "Plan-driven non-chapter work" mechanism (one work contract per task below, builder in a private worktree, reviewer on a different model, orchestrator integrates), not a generic subagent loop — see that skill for the work-contract template and `decisions/task-states.md` for states. + +**Goal:** Give every `ReviewRecord` a real, bidirectional, checkable anchor to the SysML model it judges — a new `subject_ref` field in Python paired with a new `metadata def ReviewRecordRef` tag in the model — retiring the recurring DL-075 seam escalation by giving judgment-record notebooks a concrete bridged connection to narrate instead of a choice between competing readings. + +**Architecture:** Additive schema change (`subject_ref: str`) plus a cross-checkable model-side tag (SysML `metadata def`/`about`), joined by one new query helper (`get_review_record_refs`) built on the existing `to_api_json()` pattern. Every existing judgment-record notebook across Ch2–Ch10 is retrofitted: original-authoring records get a real `subject_ref` value and a matching model tag; reconstructions/ledgers carry the value forward in Python only; deliberate negative controls get a resolvable `subject_ref` added where needed so they keep demonstrating exactly one intended validation error. + +**Tech Stack:** Python 3 dataclasses, OpenSysML v0.9.0 (`opensysml` package), pytest, Jupyter notebooks (`.ipynb`), SysML v2 text (`models/*.sysml`). + +**Spec:** [`docs/superpowers/specs/2026-10-01-hawkins-judgment-record-anchor-design.md`](../specs/2026-10-01-hawkins-judgment-record-anchor-design.md) — this plan implements its Design sections 1–6. Executors should read both; where this plan gives an exact value or snippet, it is this plan's own refinement of the spec's stated intent (the spec itself flags per-record `subject_ref` values as "intent, not final strings" — this plan fixes them). + +## Global Constraints + +- OpenSysML version is pinned at `v0.9.0` throughout (`opensysml.connect(version="v0.9.0")`) — do not bump it as a side effect of this work. +- `record_kind` stays `"worked_example"` on every record in this tutorial (SA-7); `disposition` stays `"pending"`; never introduce `"accepted"`. +- `subject_ref` values must be real, `model.find()`-resolvable qualified names (verified empirically per task below), never invented placeholders. +- The `about` form for every `ReviewRecordRef` usage is the **explicit** form (`metadata x : ReviewRecordRef about { ... }`), never the implicit nested form — confirmed in the spec's own probe (Verification item 6) and reconfirmed in this plan's own probe (Task 1) that only the explicit form populates `annotatedElement`. +- Every cumulative model file touched must still satisfy `model.ok == True` with zero diagnostics, and `uv run python scripts/check_construction.py --check` must exit 0 after every task that touches a construct-introducing notebook or a `models/*.sysml` file. +- No co-author trailers on commits (standing repo convention). +- Package name in every chapter's model is `ToasterDemo` — do not re-derive or guess a different root package. + +## Review Focus + +- **A judgment-record notebook that calls `validate_record(record)` without passing `model=`** must keep working exactly as before (the resolution and cross-representation checks are both gated on `model is not None`) — a reasonable reader would expect adding an optional parameter not to break existing single-argument calls. Task 1's tests pin this. +- **A `subject_ref` that doesn't resolve in the model** (typo, renamed element, wrong chapter's qualified name) must produce a clear, specific error naming the bad reference, not a silent pass or a raw `AttributeError`/`KeyError` from `model.find()`. Task 1's tests pin this. +- **An `asserted_inference` record with a non-empty `premises` list and an empty `subject_ref`** must validate with **zero** `subject_ref`-related errors (the Hawkins-grounded exemption) — a careless required-ness implementation could easily make this always-required and silently break `AI-C10`. Task 1's tests pin this explicitly as a positive case, not just the negative "premises also empty" case. +- **A `ReviewRecordRef` tag whose `about` target doesn't match the Python record's own `subject_ref`** (the drift the cross-representation check exists to catch) must be *caught*, not silently ignored — a reasonable implementer, focused on getting the resolution check working, might skip wiring the cross-representation check all the way through. Task 1's tests pin this with a deliberately mismatched fixture. +- **A negative-control record in an already-shipped notebook** (`AI-BAD`, the empty-identifier `"broken"` record, `AS-BAD`, `AS-PLACEHOLDER`, `AI-C10-DRAFT`) must keep demonstrating *exactly* the one validation failure its own markdown names, not silently gain a second, unrelated `subject_ref` error as a side effect of this change — the single most likely regression in this whole plan, because it's easy to add the new required-ness rule and forget every place it has a new blast radius. Task 8 and the negative-control sub-steps of Tasks 6/7/9 pin this by asserting the exact error list, not just "non-empty." + +--- + +## File Structure + +| File | Responsibility | +|---|---| +| `src/toaster/evidence.py` | `ReviewRecord` dataclass, `validate_record`, `check_stale` — gains `subject_ref` and the three new validation rules. Modified, not split (stays one small file). | +| `src/toaster/query.py` | Gains `get_review_record_refs(model, index=None)`, following the existing `ApiIndex`-based pattern (`get_satisfy_relationships`, `find_allocations`). Modified, not split. | +| `tests/test_evidence.py` | New/extended unit tests for `validate_record`'s new rules (may not exist yet — create if absent, following the existing `tests/` layout and `conftest.py` fixtures if any). | +| `tests/test_query.py` | New/extended unit tests for `get_review_record_refs` (may not exist yet — create if absent, matching the convention of existing query tests, e.g. `tests/test_requirement_ties.py` or similar, for fixture style). | +| `models/ch02-cumulative.sysml` through `models/ch08-cumulative.sysml`, `models/ch10-cumulative.sysml` | Gain `metadata def ReviewRecordRef { attribute identifier : String; }` (once, carried forward from Ch2) plus one `metadata ... : ReviewRecordRef about { identifier = "..."; }` usage per original-authoring record, carried forward from each record's origin chapter. | +| `chapters/ch02-requirements/03-judgment-context.ipynb`, `ch03-measures/01-moe-definition.ipynb`, `ch03-measures/03-threshold-judgment.ipynb`, `ch04-functional-decomp/03-completeness-check.ipynb`, `ch06-recursive-decomp/02-second-level.ipynb`, `ch06-recursive-decomp/03-stopping-judgment.ipynb`, `ch08-checking/02-violation-witness.ipynb`, `ch10-traceability-signoff/01-traceability-graph.ipynb` | Each gains a new first construction-zone narration group ("what is being claimed, and what, specifically, it is about") covering `subject_ref` plus the new metadata-usage fragment, and a rewritten seam cell per the spec's DL-075 resolution. | +| `chapters/ch08-checking/03-revision-flow.ipynb`, `ch09-coverage-sufficiency/02-evidence-completeness.ipynb`, `ch09-coverage-sufficiency/03-stale-detection.ipynb`, `ch10-traceability-signoff/02-judgment-synthesis.ipynb`, `ch10-traceability-signoff/03-engineering-signoff.ipynb` | Reconstructions/ledgers carry `subject_ref` forward in Python only (no new model tag); negative controls get a resolvable `subject_ref` added where needed to keep demonstrating exactly one error. | +| `scripts/check_construction.py` | Gains new registry entries for the 5 judgment notebooks that previously had no `TOASTER_INCREMENT` (Ch2/nb03, Ch3/nb03, Ch4/nb03, Ch6/nb03, Ch8/nb02) now that each introduces a real SysML fragment; existing entries for Ch3/nb01, Ch6/nb02, Ch10/nb01 gain the new fragment inside their existing `TOASTER_INCREMENT`. | +| `.claude/skills/toaster-review-protocol/SKILL.md` | New `subject_ref` row in the required-fields table and example; new section citing SysML v2 §7.27.2. | +| `.claude/skills/toaster-recipe/SKILL.md` | Judgment-record construction-zone pattern gains its new first group; nothing else changes. | +| `glossary/definitions/hawkins.ttl` | Gains the `gl:refines` edge from the tutorial's `subject_ref`/`ReviewRecordRef` convention to the existing `glid:def-hawkins--assurance-claim-point` term. | +| `decisions/log.md` | New DL entry recording this work and closing DL-075. | + +--- + +## Task 1: Schema — `subject_ref` and the three new validation rules + +**Files:** +- Modify: `src/toaster/evidence.py` (whole file is 55 lines; this task rewrites the dataclass and `validate_record`) +- Modify/Create: `tests/test_evidence.py` + +**Interfaces:** +- Consumes: nothing new (pure stdlib + the existing `ReviewRecord`/`hash_content`/`check_stale`). +- Produces: `ReviewRecord.subject_ref: str` (new field); `validate_record(r: ReviewRecord, model: Any | None = None) -> list[str]` (new optional second parameter — existing single-argument call sites must be unaffected). Task 2 and every later task that calls `validate_record(record, model)` or constructs a `ReviewRecord(..., subject_ref=...)` depends on this exact signature. + +**Why `model_ref`, `content_hash`, `scope`, `criteria` also gain `= ""` defaults:** Python dataclasses require every field after the first field with a default to also have one. Inserting `subject_ref: str = ""` right after `claim` (matching the spec's own code snippet, for readability — the ACP-analog field reads naturally next to `claim`) means every field after it needs a default too. This is a mechanical side effect, not a design change: no caller in this repo currently constructs a `ReviewRecord` without passing `model_ref`/`content_hash`/`scope`/`criteria` explicitly (grep confirms this in Task 6's verification step), so no behavior changes for any existing record. + +- [ ] **Step 1: Write the failing tests** + +Create or extend `tests/test_evidence.py`: + +```python +import pytest +import opensysml + +from toaster.evidence import ReviewRecord, validate_record, hash_content + + +def _base_kwargs(**overrides): + kwargs = dict( + identifier="RR-TEST", + kind="asserted_solution", + claim="Test claim.", + model_ref="models/test.sysml", + content_hash="deadbeef", + scope="test scope", + criteria="test criteria", + rationale="test rationale", + counterevidence="test counterevidence", + ) + kwargs.update(overrides) + return kwargs + + +def test_subject_ref_defaults_empty(): + r = ReviewRecord(**_base_kwargs()) + assert r.subject_ref == "" + + +def test_subject_ref_required_for_asserted_context_without_model(): + r = ReviewRecord(**_base_kwargs(kind="asserted_context")) + errors = validate_record(r) + assert any("subject_ref" in e for e in errors) + + +def test_subject_ref_required_for_asserted_solution_without_model(): + r = ReviewRecord(**_base_kwargs(kind="asserted_solution")) + errors = validate_record(r) + assert any("subject_ref" in e for e in errors) + + +def test_subject_ref_may_be_empty_for_asserted_inference_with_premises(): + r = ReviewRecord(**_base_kwargs(kind="asserted_inference", premises=["some premise"])) + errors = validate_record(r) + assert not any("subject_ref" in e for e in errors) + + +def test_subject_ref_required_for_asserted_inference_without_premises(): + r = ReviewRecord(**_base_kwargs(kind="asserted_inference", premises=[])) + errors = validate_record(r) + assert any("subject_ref" in e for e in errors) + # the pre-existing premises rule must still also fire -- this is a real double-error case + assert any("premise" in e for e in errors) + + +def test_validate_record_without_model_arg_still_works(): + # Existing call sites across the repo call validate_record(record) with one argument. + r = ReviewRecord(**_base_kwargs(subject_ref="ToasterDemo::anything")) + errors = validate_record(r) # no model= passed + assert errors == [] + + +_FIXTURE = """ +package ToasterDemo { + part def Widget { + attribute flag : ScalarValues::Boolean; + } + part target : Widget; + + metadata def ReviewRecordRef { + attribute identifier : ScalarValues::String; + } + + metadata rrTag : ReviewRecordRef about target { + identifier = "RR-TEST"; + } +} +""" + + +@pytest.fixture +def loaded_model(): + conn = opensysml.connect(version="v0.9.0") + model = conn.load_from_content(_FIXTURE, strict=False) + assert model.ok, model.diagnostics + yield model + conn.close() + + +def test_subject_ref_resolution_check_passes_for_real_element(loaded_model): + r = ReviewRecord(**_base_kwargs(subject_ref="ToasterDemo::target")) + errors = validate_record(r, model=loaded_model) + assert errors == [] + + +def test_subject_ref_resolution_check_fails_for_unresolvable_element(loaded_model): + r = ReviewRecord(**_base_kwargs(subject_ref="ToasterDemo::doesNotExist")) + errors = validate_record(r, model=loaded_model) + assert any("doesNotExist" in e for e in errors) + + +def test_cross_representation_check_passes_when_tag_matches(loaded_model): + r = ReviewRecord(**_base_kwargs(identifier="RR-TEST", subject_ref="ToasterDemo::target")) + errors = validate_record(r, model=loaded_model) + assert errors == [] + + +def test_cross_representation_check_fails_when_tag_disagrees(loaded_model): + # Same identifier as the model's own rrTag, but a DIFFERENT subject_ref -- this is the + # drift the cross-representation check exists to catch. + r = ReviewRecord(**_base_kwargs(identifier="RR-TEST", subject_ref="ToasterDemo::Widget")) + errors = validate_record(r, model=loaded_model) + assert any("RR-TEST" in e and "Widget" in e for e in errors) +``` + +- [ ] **Step 2: Run tests to verify they fail** + +Run: `uv run pytest tests/test_evidence.py -v` +Expected: collection error or failures — `subject_ref` doesn't exist yet, `validate_record` doesn't accept `model=`. + +- [ ] **Step 3: Implement** + +Replace `src/toaster/evidence.py` in full: + +```python +"""Engineering review records. WP-6 completes validate_record and check_stale.""" + +import hashlib +from dataclasses import dataclass, field +from typing import Any, Literal + +from toaster.query import get_review_record_refs + + +@dataclass +class ReviewRecord: + identifier: str + kind: Literal["asserted_context", "asserted_inference", "asserted_solution"] + claim: str + subject_ref: str = "" + model_ref: str = "" + content_hash: str = "" + scope: str = "" + criteria: str = "" + premises: list[str] = field(default_factory=list) + assumption_refs: list[str] = field(default_factory=list) + evidence_refs: list[str] = field(default_factory=list) + rationale: str = "" + counterevidence: str = "" + residual_uncertainties: str = "" + disposition: Literal["pending", "accepted", "rejected"] = "pending" + dependency_freshness: Literal["current", "stale"] = "current" + engineering_conclusion: Literal["supported", "refuted", "undetermined"] = "undetermined" + record_kind: Literal["worked_example", "actual_review"] = "worked_example" + + +def hash_content(s: str) -> str: + """Return SHA-256 hex digest of a string.""" + return hashlib.sha256(s.encode()).hexdigest() + + +def validate_record(r: ReviewRecord, model: Any | None = None) -> list[str]: + """Return a list of field-level validation errors (empty = valid). + + `subject_ref` is this tutorial's narrowed analog of Hawkins' Assurance Claim Point + (`toaster-review-protocol`, SysML v2 formal/2026-03-02 §7.27.2): the one model + element this specific judgment is about. Required for `asserted_context` and + `asserted_solution`; for `asserted_inference` it may stay empty only when `premises` + is non-empty (the pure cross-record synthesis case, e.g. `AI-C10`). + + When `model` is given and `subject_ref` is set, two further checks run: that + `subject_ref` resolves in the model, and that any `ReviewRecordRef` metadata tag + already present for this record's own `identifier` agrees with `subject_ref` (catching + drift between the Python record and the model tag if they are ever edited independently). + """ + errors = [] + if not r.identifier: + errors.append("identifier is empty") + if not r.claim: + errors.append("claim is empty") + if not r.rationale: + errors.append("rationale is empty") + if not r.counterevidence: + errors.append("counterevidence is empty") + if r.record_kind == "actual_review": + errors.append("record_kind must be 'worked_example' in this tutorial (SA-7)") + if r.kind == "asserted_inference" and not r.premises: + errors.append("asserted_inference requires at least one premise (Hawkins §3.1)") + + if not r.subject_ref: + if r.kind in ("asserted_context", "asserted_solution"): + errors.append( + f"subject_ref is empty (required for {r.kind})" + ) + elif r.kind == "asserted_inference" and not r.premises: + errors.append( + "subject_ref is empty (required for asserted_inference with no premises)" + ) + elif model is not None: + if model.find(r.subject_ref) is None: + errors.append(f"subject_ref {r.subject_ref!r} does not resolve in the model") + else: + tags = {t["identifier"]: t["annotated_element"] for t in get_review_record_refs(model)} + tagged = tags.get(r.identifier) + if tagged is not None and tagged != r.subject_ref: + errors.append( + f"ReviewRecordRef tag for {r.identifier!r} is about {tagged!r}, " + f"but subject_ref is {r.subject_ref!r}" + ) + return errors + + +def check_stale(r: ReviewRecord, current_content: str) -> bool: + """Return True if the record's content_hash no longer matches current_content.""" + return r.content_hash != hash_content(current_content) +``` + +- [ ] **Step 4: Run tests to verify they pass** + +Run: `uv run pytest tests/test_evidence.py -v` +Expected: all PASS. + +- [ ] **Step 5: Run the full existing suite to confirm no regression** + +Run: `uv run pytest -q` +Expected: same pass count as before this task, plus the new tests — zero new failures. (If any existing test constructs a bare `ReviewRecord(...)` and calls `validate_record` with one argument, it must still pass unchanged — this is the Review Focus item above.) + +- [ ] **Step 6: Commit** + +```bash +git add src/toaster/evidence.py tests/test_evidence.py +git commit -m "Add subject_ref to ReviewRecord: the tutorial's Assurance Claim Point analog" +``` + +--- + +## Task 2: Query helper — `get_review_record_refs` + +**Files:** +- Modify: `src/toaster/query.py` +- Modify/Create: `tests/test_query.py` + +**Interfaces:** +- Consumes: `ApiIndex`, `_ref` (already defined in `query.py`). +- Produces: `get_review_record_refs(model: Any, index: ApiIndex | None = None) -> list[dict]`, each dict shaped `{"tag": , "identifier": , "annotated_element": }`. Task 1's `evidence.py` imports and calls this function — Task 1 and Task 2 can be built in parallel (neither's tests depend on the other being merged first), but both must land before any notebook task (3 onward) that calls `validate_record(record, model=...)` against a real `ReviewRecordRef` tag. + +This task's exact JSON shapes were confirmed empirically against OpenSysML v0.9.0 (not assumed) with this probe, which any implementer should feel free to re-run to double-check before writing code: + +```python +import json, warnings +import opensysml + +conn = opensysml.connect(version="v0.9.0") +source = """ +package P { + part def Widget { attribute flag : ScalarValues::Boolean; } + part target : Widget; + metadata def ReviewRecordRef { attribute identifier : ScalarValues::String; } + metadata tag1 : ReviewRecordRef about target { identifier = "X-1"; } +} +""" +model = conn.load_from_content(source, strict=False) +with warnings.catch_warnings(): + warnings.simplefilter("ignore") + content = json.loads(model.to_api_json().content) +for e in content: + if e["@type"] in ("MetadataUsage", "ReferenceUsage", "LiteralString"): + print(e["@type"], e.get("qualifiedName"), e.get("@id"), {k: v for k, v in e.items() if k in ("type", "annotatedElement", "value", "declaredName", "ownedMember")}) +conn.close() +``` + +Confirmed shapes (do not re-derive, these are facts about v0.9.0, not design choices): +- A `MetadataUsage` element has `type: [{"@id": ""}]` (a list) and `annotatedElement: [{"@id": ""}]` (**also a list**, even for one target — this tutorial's convention is exactly one target per tag, per the spec's "singular, not a list" decision, so take the first). +- The usage's own `identifier` attribute is itself a separate element reachable via the usage's `ownedMember` list: the member whose `declaredName == "identifier"` is a `ReferenceUsage` whose own `value` field is `{"@id": ""}`; that `LiteralString` element's own `value` field is the actual Python string (e.g. `"X-1"`). + +- [ ] **Step 1: Write the failing test** + +Create or extend `tests/test_query.py`: + +```python +import opensysml + +from toaster.query import get_review_record_refs + + +_FIXTURE_ONE_TAG = """ +package ToasterDemo { + part def Widget { attribute flag : ScalarValues::Boolean; } + part target : Widget; + metadata def ReviewRecordRef { attribute identifier : ScalarValues::String; } + metadata rrTag : ReviewRecordRef about target { identifier = "RR-001"; } +} +""" + +_FIXTURE_NO_TAGS = """ +package ToasterDemo { + part def Widget { attribute flag : ScalarValues::Boolean; } + part target : Widget; +} +""" + + +def _load(source): + conn = opensysml.connect(version="v0.9.0") + model = conn.load_from_content(source, strict=False) + assert model.ok, model.diagnostics + return conn, model + + +def test_get_review_record_refs_finds_one_tag(): + conn, model = _load(_FIXTURE_ONE_TAG) + try: + refs = get_review_record_refs(model) + assert len(refs) == 1 + assert refs[0]["identifier"] == "RR-001" + assert refs[0]["annotated_element"] == "ToasterDemo::target" + finally: + conn.close() + + +def test_get_review_record_refs_empty_when_no_tags(): + conn, model = _load(_FIXTURE_NO_TAGS) + try: + assert get_review_record_refs(model) == [] + finally: + conn.close() +``` + +- [ ] **Step 2: Run test to verify it fails** + +Run: `uv run pytest tests/test_query.py -v` +Expected: FAIL — `get_review_record_refs` not defined. + +- [ ] **Step 3: Implement** + +Add to `src/toaster/query.py`, directly after `get_satisfy_relationships` (keeps the two `to_api_json()`-based helpers adjacent): + +```python +def get_review_record_refs(model: Any, index: ApiIndex | None = None) -> list[dict]: + """Every ``ReviewRecordRef`` metadata tag in the model: ``{tag, identifier, annotated_element}``. + + The model-to-Python direction of this tutorial's narrowed Assurance Claim Point anchor + (``toaster-review-protocol``, SysML v2 formal/2026-03-02 §7.27.2): given a loaded model, + find every judgment-record tag and the one subject it names, independent of any + notebook's own Python objects. Takes the first ``annotatedElement`` only, matching this + tutorial's one-tag-one-subject convention (a tag with more than one is a modeling error + this tutorial's own notebooks never produce, not a shape this helper tries to generalize). + """ + idx = index or ApiIndex(model) + out = [] + for e in idx.of_type("MetadataUsage"): + type_qns = {idx.qn(t) for t in e.get("type", [])} + if not any(qn and qn.endswith("::ReviewRecordRef") for qn in type_qns): + continue + annotated = [idx.qn(a) for a in e.get("annotatedElement", [])] + identifier_value = None + for member_ref in e.get("ownedMember", []): + member = idx.by_id.get(_ref(member_ref)) + if member and member.get("declaredName") == "identifier": + literal = idx.by_id.get(_ref(member.get("value"))) + if literal is not None: + identifier_value = literal.get("value") + out.append({ + "tag": e.get("qualifiedName"), + "identifier": identifier_value, + "annotated_element": annotated[0] if annotated else None, + }) + return out +``` + +- [ ] **Step 4: Run tests to verify they pass** + +Run: `uv run pytest tests/test_query.py -v` +Expected: PASS. + +- [ ] **Step 5: Run the full suite** + +Run: `uv run pytest -q` +Expected: no regressions. + +- [ ] **Step 6: Commit** + +```bash +git add src/toaster/query.py tests/test_query.py +git commit -m "Add get_review_record_refs: model-to-Python direction of the judgment-record anchor" +``` + +--- + +## Task 3: The `ReviewRecordRef` construct, introduced once in Chapter 2 (`AC-001`) + +**Depends on:** Task 1, Task 2 (must be merged first — `validate_record(record, model=...)` and the metadata tag it cross-checks must both exist). + +**Files:** +- Modify: `models/ch02-cumulative.sysml`, `models/ch03-cumulative.sysml`, `models/ch04-cumulative.sysml`, `models/ch05-cumulative.sysml`, `models/ch06-cumulative.sysml`, `models/ch07-cumulative.sysml`, `models/ch08-cumulative.sysml`, `models/ch10-cumulative.sysml` (every cumulative file from Ch2 onward — each is a self-contained full snapshot, not an import chain, so the definition text is carried forward by copy into every later file, the same convention every other reusable construct in this tutorial already follows) +- Modify: `chapters/ch02-requirements/03-judgment-context.ipynb` +- Modify: `scripts/check_construction.py` (new registry entry — this notebook currently has no `TOASTER_INCREMENT`) + +**Exact SysML text to add** (once, inside the `ToasterDemo` package body, placed immediately after `package ToasterDemo {`): + +```sysml +metadata def ReviewRecordRef { + attribute identifier : ScalarValues::String; +} +``` + +**`AC-001`'s own usage** (add immediately after `nominal`'s own declaration in every cumulative file ch02 onward — `nominal` already exists in `ch02-cumulative.sysml`, confirmed present): + +```sysml +metadata ac001Tag : ReviewRecordRef about nominal { + identifier = "AC-001"; +} +``` + +(Use the short name `nominal`, not `ToasterDemo::nominal`, inside the `about` clause — the usage sits inside the same package, matching this repo's existing convention of short names inside a package body, e.g. how `TimelyToast::toaster` is written elsewhere. The qualified name `ToasterDemo::nominal` is what `subject_ref` and `model.find()` use from *outside* the package, in Python.) + +- [ ] **Step 1: Add the definition and first usage to `models/ch02-cumulative.sysml`** + +Insert the two SysML blocks above at the stated locations. Run: + +```bash +uv run python - <<'EOF' +from pathlib import Path +import opensysml +conn = opensysml.connect(version="v0.9.0") +source = Path("models/ch02-cumulative.sysml").read_text() +model = conn.load_from_content(source, strict=False) +assert model.ok, model.diagnostics +print("ok") +conn.close() +EOF +``` +Expected: `ok`, no diagnostics. + +- [ ] **Step 2: Carry both blocks forward into every later cumulative file** + +For each of `models/ch03-cumulative.sysml` through `models/ch08-cumulative.sysml` and `models/ch10-cumulative.sysml`: add the identical `metadata def ReviewRecordRef { ... }` block and the identical `ac001Tag` usage (targeting `nominal`, which already exists in every one of these files — confirmed by the Chapter 2 survey that `nominal`/`slow` persist unchanged through every later chapter). Re-run the same load-and-assert check against each file in turn, substituting the path. + +- [ ] **Step 3: Add the construction zone to `chapters/ch02-requirements/03-judgment-context.ipynb`** + +Per `toaster-recipe`'s construction-zone pattern and `toaster-review-protocol`'s updated judgment-record construction zone (Task 10 writes the skill text this follows — read that section of the spec now, since the skill edit and this notebook edit describe the same pattern): insert a new **first** named group, before the existing `claim`/`model_ref` group, narrating "what is being claimed, and what, specifically, it is about": + +```python +# what is being claimed, and what, specifically, it is about +subject_ref = "ToasterDemo::nominal" +AC001_TAG = """\ +metadata ac001Tag : ReviewRecordRef about nominal { + identifier = "AC-001"; +} +""" +print(AC001_TAG) +``` + +Followed by a markdown cell narrating that this fragment is the same text now committed in `models/ch02-cumulative.sysml`, and that loading the cumulative model (the existing `model.ok` cell, unchanged) is what makes this a real, checkable tag rather than an assertion the reader has to trust. + +Add `subject_ref=subject_ref` to the existing `ReviewRecord(...)` call's keyword arguments (alongside `claim=claim`, `model_ref=model_ref`, etc.), and change the existing `errors = validate_record(context_record)` call to `errors = validate_record(context_record, model=model)` (the notebook's own `model` variable, already loaded by the existing cell 2 pattern). + +**Rewrite the seam cell** to narrate the bridged connection per the spec's DL-075 resolution: one sentence addressing that the tagged SysML text (printed above), the tool that loads and cross-checks both the Python record and the model tag, and a result showing they agree, are three things the reader has just watched connect — e.g. (adapt to the notebook's own voice, but keep this shape): "The tag printed above is now part of the loaded model, and `validate_record` confirms the Python record and the model's own tag agree about what AC-001 is actually about." Do not name Tall or "the three worlds" (AGENTS.md 1.10, unchanged rule). + +- [ ] **Step 4: Register the notebook in `scripts/check_construction.py`** + +Add a new entry to the registry (alongside the existing Ch2 entries at the file's current lines ~80–95) for `chapters/ch02-requirements/03-judgment-context.ipynb`, with `context_stubs` providing the minimal stub `nominal` needs to parse in isolation (follow the existing stub style used by the neighboring Ch2 entries — e.g. a bare `part nominal;` stub, adjusted to whatever minimal form the existing entries use for referencing a prior-notebook element). Run: + +```bash +uv run python scripts/check_construction.py --check +``` +Expected: exit 0. + +- [ ] **Step 5: Run the full notebook and the full suite** + +Execute the notebook's cells top to bottom (or via the repo's existing notebook-execution check, if one exists — check `myst.yml`/CI config for how notebooks are normally executed) and confirm no errors. Then: + +```bash +uv run pytest -q +``` +Expected: no regressions. + +- [ ] **Step 6: Commit** + +```bash +git add models/ch02-cumulative.sysml models/ch03-cumulative.sysml models/ch04-cumulative.sysml models/ch05-cumulative.sysml models/ch06-cumulative.sysml models/ch07-cumulative.sysml models/ch08-cumulative.sysml models/ch10-cumulative.sysml chapters/ch02-requirements/03-judgment-context.ipynb scripts/check_construction.py +git commit -m "Introduce ReviewRecordRef metadata def; tag AC-001's real subject (nominal)" +``` + +--- + +## Task 4: Retrofit Chapter 3 — `AC-C03`, `AS-C03` + +**Depends on:** Task 3 (must be merged first — shares the same 6 cumulative files ch03/04/05/06/07/08/10, sequenced to avoid merge conflicts with Tasks 5–9 below, which touch the same files). + +**Subjects:** both records are about the same element, `ToasterDemo::timely` (the `TimelyToast` requirement usage), confirmed present from `models/ch03-cumulative.sysml` onward. + +**Files:** +- Modify: `models/ch03-cumulative.sysml` through `models/ch08-cumulative.sysml`, `models/ch10-cumulative.sysml` (carry forward from Ch3) +- Modify: `chapters/ch03-measures/01-moe-definition.ipynb` (`AC-C03`; this notebook is **already registered** in `check_construction.py` with its own `TOASTER_INCREMENT` — extend it, do not create a new registry entry) +- Modify: `chapters/ch03-measures/03-threshold-judgment.ipynb` (`AS-C03`; **not yet registered** — add a new entry) +- Modify: `scripts/check_construction.py` + +**Exact SysML text** (add to `models/ch03-cumulative.sysml` onward, after `timely`'s own declaration): + +```sysml +metadata acC03Tag : ReviewRecordRef about timely { + identifier = "AC-C03"; +} + +metadata asC03Tag : ReviewRecordRef about timely { + identifier = "AS-C03"; +} +``` + +(Two separate tags, same subject — Hawkins' own model allows more than one ACP to be about the same located element; nothing here requires one tag per subject, only one subject per tag.) + +- [ ] **Step 1:** Add both blocks to `models/ch03-cumulative.sysml`; verify `model.ok` via the same load-and-assert snippet as Task 3 Step 1 (substitute the path and expect both `acC03Tag` and `asC03Tag` to resolve via `model.find("ToasterDemo::acC03Tag")` / `model.find("ToasterDemo::asC03Tag")`). +- [ ] **Step 2:** Carry both blocks forward into `models/ch04-cumulative.sysml` through `ch08-cumulative.sysml` and `ch10-cumulative.sysml`; verify each. +- [ ] **Step 3:** In `ch03-measures/01-moe-definition.ipynb`: extend the existing `TOASTER_INCREMENT` construction zone with the `acC03Tag` fragment (print it as its own named step, same pattern as Task 3 Step 3); add `subject_ref="ToasterDemo::timely"` to the `AC-C03` `ReviewRecord(...)` call; change its `validate_record(...)` call to pass `model=model`; rewrite its seam cell per the DL-075 resolution language (same shape as Task 3 Step 3's seam rewrite, adapted to this notebook's own subject). +- [ ] **Step 4:** In `ch03-measures/03-threshold-judgment.ipynb`: add a new construction-zone group (this notebook has no existing `TOASTER_INCREMENT` — follow Task 3 Step 3's full pattern, not the "extend" pattern) with the `asC03Tag` fragment; add `subject_ref="ToasterDemo::timely"` to the `AS-C03` call; `model=model` on `validate_record`; rewrite the seam cell. +- [ ] **Step 5:** Update `scripts/check_construction.py`: extend the existing Ch3/`01-moe-definition.ipynb` entry's expected `TOASTER_INCREMENT` content; add a new entry for `03-threshold-judgment.ipynb` with a minimal `timely` stub. Run `uv run python scripts/check_construction.py --check` — expect exit 0. +- [ ] **Step 6:** `uv run pytest -q` — no regressions. +- [ ] **Step 7:** Commit: +```bash +git add models/ch03-cumulative.sysml models/ch04-cumulative.sysml models/ch05-cumulative.sysml models/ch06-cumulative.sysml models/ch07-cumulative.sysml models/ch08-cumulative.sysml models/ch10-cumulative.sysml chapters/ch03-measures/01-moe-definition.ipynb chapters/ch03-measures/03-threshold-judgment.ipynb scripts/check_construction.py +git commit -m "Retrofit AC-C03/AS-C03 with subject_ref and ReviewRecordRef tags (both about timely)" +``` + +--- + +## Task 5: Retrofit Chapter 4 — `AI-C04` + +**Depends on:** Task 4 (serialized — same shared cumulative files). + +**Subject:** `ToasterDemo::ApplyHeat`. + +**Files:** +- Modify: `models/ch04-cumulative.sysml` through `models/ch08-cumulative.sysml`, `models/ch10-cumulative.sysml` +- Modify: `chapters/ch04-functional-decomp/03-completeness-check.ipynb` (not yet registered — new entry) +- Modify: `scripts/check_construction.py` + +**Exact SysML text** (after `ApplyHeat`'s own declaration): + +```sysml +metadata aiC04Tag : ReviewRecordRef about ApplyHeat { + identifier = "AI-C04"; +} +``` + +Confirm the correct short name to use in `about`: the Ch4 survey found both a top-level `ApplyHeat` (an action def) and `ToastBread::applyHeat` (a nested performed usage). `AI-C04`'s existing `model_ref` is `ToasterDemo::ApplyHeat` (the definition/top-level usage, not the nested one) — use `ApplyHeat` in the `about` clause to match, and set `subject_ref="ToasterDemo::ApplyHeat"` in Python to match exactly. + +- [ ] **Step 1:** Add the block to `models/ch04-cumulative.sysml`; verify. +- [ ] **Step 2:** Carry forward into `ch05` through `ch08`, `ch10` cumulative files; verify each. +- [ ] **Step 3:** In `ch04-functional-decomp/03-completeness-check.ipynb`: add the construction-zone group (new `TOASTER_INCREMENT`, following Task 3 Step 3's pattern); `subject_ref="ToasterDemo::ApplyHeat"` on the `AI-C04` call; `model=model` on `validate_record`; rewrite the seam cell. +- [ ] **Step 4:** Register the notebook in `check_construction.py` with a minimal `ApplyHeat` stub. `uv run python scripts/check_construction.py --check` — exit 0. +- [ ] **Step 5:** `uv run pytest -q` — no regressions. +- [ ] **Step 6:** Commit: +```bash +git add models/ch04-cumulative.sysml models/ch05-cumulative.sysml models/ch06-cumulative.sysml models/ch07-cumulative.sysml models/ch08-cumulative.sysml models/ch10-cumulative.sysml chapters/ch04-functional-decomp/03-completeness-check.ipynb scripts/check_construction.py +git commit -m "Retrofit AI-C04 with subject_ref and ReviewRecordRef tag (about ApplyHeat)" +``` + +--- + +## Task 6: Retrofit Chapter 6 — `AC-C06`, `AS-C06`, `AI-C06`, and its `AI-BAD` negative control + +**Depends on:** Task 5 (serialized). + +**Subjects:** +- `AC-C06` → `ToasterDemo::heatGenerationReq` +- `AS-C06` → `ToasterDemo::ResistanceCoil` +- `AI-C06` → `ToasterDemo::HeatingAssembly::heatGen` + +**Files:** +- Modify: `models/ch06-cumulative.sysml` through `models/ch08-cumulative.sysml`, `models/ch10-cumulative.sysml` +- Modify: `chapters/ch06-recursive-decomp/02-second-level.ipynb` (`AC-C06`, `AS-C06`; **already registered** with an existing `TOASTER_INCREMENT` — extend it) +- Modify: `chapters/ch06-recursive-decomp/03-stopping-judgment.ipynb` (`AI-C06` original authoring, plus `AI-BAD` negative control; not yet registered — new entry) +- Modify: `scripts/check_construction.py` + +**Exact SysML text** (after each subject's own declaration in `models/ch06-cumulative.sysml`): + +```sysml +metadata acC06Tag : ReviewRecordRef about heatGenerationReq { + identifier = "AC-C06"; +} + +metadata asC06Tag : ReviewRecordRef about ResistanceCoil { + identifier = "AS-C06"; +} +``` + +And, for `AI-C06` (also lands in `ch06-cumulative.sysml`, since `HeatingAssembly::heatGen` already exists there per the survey): + +```sysml +metadata aiC06Tag : ReviewRecordRef about HeatingAssembly::heatGen { + identifier = "AI-C06"; +} +``` + +- [ ] **Step 1:** Add all three blocks to `models/ch06-cumulative.sysml`; verify `model.ok` and that `model.find("ToasterDemo::acC06Tag")`, `...asC06Tag`, `...aiC06Tag` all resolve. +- [ ] **Step 2:** Carry all three forward into `ch07`, `ch08`, `ch10` cumulative files; verify each. +- [ ] **Step 3:** In `ch06-recursive-decomp/02-second-level.ipynb`: extend the existing `TOASTER_INCREMENT` with the `acC06Tag` and `asC06Tag` fragments (two more named steps in the existing construction zone, each its own print+narration); add `subject_ref="ToasterDemo::heatGenerationReq"` to the `AC-C06` call and `subject_ref="ToasterDemo::ResistanceCoil"` to the `AS-C06` call; `model=model` on both `validate_record` calls; rewrite both seam cells if this notebook has one seam cell per record, or the one shared seam cell if it has only one — follow whatever structure the notebook already has, updating every seam cell present. +- [ ] **Step 4:** In `ch06-recursive-decomp/03-stopping-judgment.ipynb`: + - Add the construction-zone group (new `TOASTER_INCREMENT`) with the `aiC06Tag` fragment; `subject_ref="ToasterDemo::HeatingAssembly::heatGen"` on the `AI-C06` call; `model=model` on its `validate_record` call; rewrite its seam cell. + - **`AI-BAD` fix (Review Focus item):** this notebook's `AI-BAD` negative control has `premises=[]` and currently no `subject_ref`. Under the new rule, `asserted_inference` with empty premises AND empty `subject_ref` now produces **two** errors (the pre-existing premises error, plus the new subject_ref-required error) where the notebook's own markdown currently asserts exactly one. Add `subject_ref="ToasterDemo::HeatingAssembly::heatGen"` to the `AI-BAD` construction (reusing `AI-C06`'s real subject — `AI-BAD` is demonstrating the premises rule specifically, not the subject_ref rule, so giving it a real, resolvable subject keeps that demonstration singular). Verify by running `validate_record(ai_bad_record, model=model)` and confirming the result is exactly `["asserted_inference requires at least one premise (Hawkins §3.1)"]`, not two entries. Update the notebook's own markdown/printed-assertion text if it states an exact error list, so it still matches. +- [ ] **Step 5:** Register `03-stopping-judgment.ipynb` in `check_construction.py` with a minimal `HeatingAssembly`/`heatGen` stub; extend the existing `02-second-level.ipynb` entry's expected content. `uv run python scripts/check_construction.py --check` — exit 0. +- [ ] **Step 6:** `uv run pytest -q` — no regressions. +- [ ] **Step 7:** Commit: +```bash +git add models/ch06-cumulative.sysml models/ch07-cumulative.sysml models/ch08-cumulative.sysml models/ch10-cumulative.sysml chapters/ch06-recursive-decomp/02-second-level.ipynb chapters/ch06-recursive-decomp/03-stopping-judgment.ipynb scripts/check_construction.py +git commit -m "Retrofit AC-C06/AS-C06/AI-C06 with subject_ref and tags; fix AI-BAD's new double-error" +``` + +--- + +## Task 7: Retrofit Chapter 8 — `AS-C08` (original) and its revision-flow reconstruction + negative control + +**Depends on:** Task 6 (serialized). + +**Subject:** `ToasterDemo::deliveredEnergyBoundedBySupply`. + +**Files:** +- Modify: `models/ch08-cumulative.sysml`, `models/ch10-cumulative.sysml` +- Modify: `chapters/ch08-checking/02-violation-witness.ipynb` (original authoring; not yet registered — new entry) +- Modify: `chapters/ch08-checking/03-revision-flow.ipynb` (reconstruction of `AS-C08` + an empty-identifier negative control — no new model tag needed here, Python-only) +- Modify: `scripts/check_construction.py` + +**Exact SysML text** (after `deliveredEnergyBoundedBySupply`'s own declaration in `ch08-cumulative.sysml`): + +```sysml +metadata asC08Tag : ReviewRecordRef about deliveredEnergyBoundedBySupply { + identifier = "AS-C08"; +} +``` + +- [ ] **Step 1:** Add the block to `models/ch08-cumulative.sysml`; verify. +- [ ] **Step 2:** Carry forward into `models/ch10-cumulative.sysml`; verify. +- [ ] **Step 3:** In `ch08-checking/02-violation-witness.ipynb`: add the construction-zone group with the `asC08Tag` fragment; `subject_ref="ToasterDemo::deliveredEnergyBoundedBySupply"` on the `AS-C08` call; `model=model` on `validate_record`; rewrite the seam cell. +- [ ] **Step 4:** In `ch08-checking/03-revision-flow.ipynb` (no new model tag — this notebook only reconstructs and tests staleness, per the design's retrofit table rule that reconstructions carry `subject_ref` forward in Python only): + - Add `subject_ref="ToasterDemo::deliveredEnergyBoundedBySupply"` to the rebuilt `AS-C08` record. + - **Empty-identifier negative control fix (Review Focus item):** this notebook's `identifier=""` "broken" record (`kind="asserted_solution"`) currently has no `subject_ref` either. Under the new rule it would newly gain a second error (`subject_ref is empty`) alongside the intended `identifier is empty`. Add `subject_ref="ToasterDemo::deliveredEnergyBoundedBySupply"` to this record too, so it keeps demonstrating exactly `["identifier is empty"]`. Verify directly: `validate_record(broken_record, model=model) == ["identifier is empty"]`. +- [ ] **Step 5:** Register `02-violation-witness.ipynb` in `check_construction.py` with a minimal `deliveredEnergyBoundedBySupply` stub. `uv run python scripts/check_construction.py --check` — exit 0. +- [ ] **Step 6:** `uv run pytest -q` — no regressions. +- [ ] **Step 7:** Commit: +```bash +git add models/ch08-cumulative.sysml models/ch10-cumulative.sysml chapters/ch08-checking/02-violation-witness.ipynb chapters/ch08-checking/03-revision-flow.ipynb scripts/check_construction.py +git commit -m "Retrofit AS-C08 with subject_ref and tag; fix revision-flow's negative control" +``` + +--- + +## Task 8: Retrofit Chapter 9 reconstructions and negative controls (no new model tags — Ch9 has no cumulative model of its own) + +**Depends on:** Task 7 (needs `AS-C06`'s and `AS-C08`'s real subject_ref values already decided, which Tasks 6/7 fix; Ch9's own notebooks load `ch08-cumulative.sysml` directly, confirmed by the survey — no Ch9 cumulative file exists, so this task touches no `models/*.sysml` file at all). + +**Files:** +- Modify: `chapters/ch09-coverage-sufficiency/02-evidence-completeness.ipynb` +- Modify: `chapters/ch09-coverage-sufficiency/03-stale-detection.ipynb` + +**Values to carry forward (Python only, matching each record's origin chapter exactly):** +- `AS-C06` → `subject_ref="ToasterDemo::ResistanceCoil"` +- `AS-C08` → `subject_ref="ToasterDemo::deliveredEnergyBoundedBySupply"` + +**Negative controls needing a resolvable `subject_ref` added, each to preserve its own single-intended-error demonstration (Review Focus item — the same regression class as Task 6/7):** + +| Notebook | Record | Current intended error | Fix | +|---|---|---|---| +| `02-evidence-completeness.ipynb` | `AS-BAD` (`kind="asserted_solution"`, `counterevidence=""`) | `["counterevidence is empty"]` | Add `subject_ref="ToasterDemo::ResistanceCoil"` | +| `02-evidence-completeness.ipynb` | `AS-PLACEHOLDER` (`kind="asserted_solution"`, all fields non-empty but weak — demonstrates **zero** errors despite weak content) | `[]` | Add `subject_ref="ToasterDemo::ResistanceCoil"` — without this, the record would newly gain `["subject_ref is empty (required for asserted_solution)"]`, silently breaking the notebook's own point that validation passes despite weak content | +| `03-stale-detection.ipynb` | empty-identifier `"broken"` record (`kind="asserted_solution"`) | `["identifier is empty"]` | Add `subject_ref="ToasterDemo::deliveredEnergyBoundedBySupply"` | + +- [ ] **Step 1:** In `02-evidence-completeness.ipynb`: add `subject_ref` to the rebuilt `AS-C06` and `AS-C08` records (matching their origin-chapter values); add `subject_ref` to `AS-BAD` and `AS-PLACEHOLDER` per the table above. Update any `validate_record(...)` call in this notebook that doesn't already pass `model=model` to do so (needed for the resolution check to actually run against a real model — confirm the notebook already loads `model` via its existing cell 2; if it does not load a model at all currently, keep calling `validate_record(record)` without `model=` for this notebook, since the required-ness check alone, not resolution, is what's being demonstrated — check which is the case before deciding). +- [ ] **Step 2:** Verify each exact error list by running the notebook's own cells (or an equivalent standalone script) and asserting: + - `validate_record(as_bad)` (or `..., model=model`) `== ["counterevidence is empty"]` + - `validate_record(as_placeholder, ...) == []` +- [ ] **Step 3:** In `03-stale-detection.ipynb`: same `AS-C06`/`AS-C08` carry-forward; add `subject_ref` to the empty-identifier `"broken"` record per the table; verify `validate_record(broken, ...) == ["identifier is empty"]`. +- [ ] **Step 4:** `uv run pytest -q` — no regressions. (No `check_construction.py` change needed — neither notebook introduces a new SysML construct.) +- [ ] **Step 5:** Commit: +```bash +git add chapters/ch09-coverage-sufficiency/02-evidence-completeness.ipynb chapters/ch09-coverage-sufficiency/03-stale-detection.ipynb +git commit -m "Carry subject_ref forward in Ch9 reconstructions; fix negative controls' new double-errors" +``` + +--- + +## Task 9: Retrofit Chapter 10 — `AC-C10` (original), the judgment ledger, `AI-C10` exemption check, and its two negative controls + +**Depends on:** Task 8 (serialized — shares `models/ch10-cumulative.sysml` with every earlier task; also needs `AC-C06`/`AS-C06`/`AI-C06`/`AS-C08` subject_ref values already fixed by Tasks 6/7). + +**Subject:** `AC-C10` → `ToasterDemo::EnergyConservationReq`. + +**Files:** +- Modify: `models/ch10-cumulative.sysml` +- Modify: `chapters/ch10-traceability-signoff/01-traceability-graph.ipynb` (`AC-C10` original authoring; **already registered** in `check_construction.py` with its own `TOASTER_INCREMENT` from the earlier energy-tie work — extend it) +- Modify: `chapters/ch10-traceability-signoff/02-judgment-synthesis.ipynb` (ledger reconstructions of `AS-C06`/`AS-C08`/`AI-C06`, plus `AI-BAD` negative control — no new model tag, Python-only) +- Modify: `chapters/ch10-traceability-signoff/03-engineering-signoff.ipynb` (`AI-C10` original authoring — confirm the exemption applies; `AI-C10-DRAFT` negative control — confirm whether it needs a fix) +- Modify: `scripts/check_construction.py` + +**Exact SysML text** (after `EnergyConservationReq`'s own declaration in `models/ch10-cumulative.sysml` — this is the last cumulative file, so no further carry-forward is needed): + +```sysml +metadata acC10Tag : ReviewRecordRef about EnergyConservationReq { + identifier = "AC-C10"; +} +``` + +- [ ] **Step 1:** Add the block to `models/ch10-cumulative.sysml`; verify `model.ok` and `model.find("ToasterDemo::acC10Tag")`. +- [ ] **Step 2:** In `ch10-traceability-signoff/01-traceability-graph.ipynb`: extend the existing `TOASTER_INCREMENT` with the `acC10Tag` fragment; `subject_ref="ToasterDemo::EnergyConservationReq"` on the `AC-C10` call; `model=model` on its `validate_record` call; rewrite its seam cell. +- [ ] **Step 3:** In `ch10-traceability-signoff/02-judgment-synthesis.ipynb` (no new model tag): + - Carry forward `subject_ref` for the rebuilt `AS-C06` (`"ToasterDemo::ResistanceCoil"`), `AS-C08` (`"ToasterDemo::deliveredEnergyBoundedBySupply"`), `AI-C06` (`"ToasterDemo::HeatingAssembly::heatGen"`). + - **`AI-BAD` fix** (same regression class as Task 6): this notebook's own `AI-BAD` (`premises=[]`, no `subject_ref`) would newly gain a second error. Add `subject_ref="ToasterDemo::HeatingAssembly::heatGen"` (reusing `AI-C06`'s subject, same reasoning as Task 6). Verify `validate_record(ai_bad, ...) == ["asserted_inference requires at least one premise (Hawkins §3.1)"]`. +- [ ] **Step 4:** In `ch10-traceability-signoff/03-engineering-signoff.ipynb`: + - **`AI-C10` exemption check:** read the notebook's own existing `premises` list for the `AI-C10` record (the survey found it cites `AS-C06`/`AS-C08`/`AI-C06`/`AC-C10` plus coverage findings as prose premises — confirm this list is non-empty as currently written). If non-empty (expected), `AI-C10` needs **no** `subject_ref` change at all — leave it exactly as the spec's retrofit table states (exempt). Run `validate_record(ai_c10_record, model=model)` and confirm no `subject_ref`-related error appears, and that this matches the pre-existing behavior (zero new errors introduced). + - **`AI-C10-DRAFT` check:** this negative control's intended error is `counterevidence=""`. Check whether its own `premises` list is already non-empty (it is described as "a draft of the synthesis record built below," implying it shares the same premises). If `premises` is non-empty, no fix is needed (the exemption already covers it — confirm by running `validate_record` and checking the result is exactly `["counterevidence is empty"]`). If `premises` turns out to be empty in this draft (unlike the final `AI-C10`), add `subject_ref="ToasterDemo::EnergyConservationReq"` to it so it keeps demonstrating exactly one error, the same fix pattern as every other negative control in this plan. +- [ ] **Step 5:** Extend the existing `01-traceability-graph.ipynb` registry entry in `check_construction.py` with the `acC10Tag` fragment. `uv run python scripts/check_construction.py --check` — exit 0. +- [ ] **Step 6:** `uv run pytest -q` — no regressions. Also run `uv run python scripts/check_construction.py --check` one final time across all chapters (not just `--chapter=10`) to confirm every earlier task's cumulative-file edits still hold together. +- [ ] **Step 7:** Commit: +```bash +git add models/ch10-cumulative.sysml chapters/ch10-traceability-signoff/01-traceability-graph.ipynb chapters/ch10-traceability-signoff/02-judgment-synthesis.ipynb chapters/ch10-traceability-signoff/03-engineering-signoff.ipynb scripts/check_construction.py +git commit -m "Retrofit AC-C10 with subject_ref and tag; carry ledger values forward; confirm AI-C10 exemption" +``` + +--- + +## Task 10: `toaster-review-protocol` skill update + +**Depends on:** Task 3 (wants a real, landed example to cite — can run in parallel with Tasks 4–9 once Task 3 is merged, since it touches no `models/*.sysml` or chapter file). + +**Files:** +- Modify: `.claude/skills/toaster-review-protocol/SKILL.md` + +**Note:** this is a skill edit — per `skill-editor`, follow its pre-edit gate (DL-PENDING-then-COMPLETE, blast-radius table, minimal-change rule) even though this plan already specifies the exact diff; `skill-editor`'s process is still the required wrapper for making the edit, not a substitute for having one. + +- [ ] **Step 1:** In the `## ReviewRecord required fields` example (current lines 24–46), add `subject_ref` to the constructor call, immediately after `claim=...`: + +```python +record = ReviewRecord( + identifier="RR-001", + kind="asserted_solution", + claim="DeliveredEnergy >= 50000 J at nominal operating conditions", + subject_ref="ToasterDemo::deliveredEnergy", + model_ref="models/ch07-snapshot.sysml", + ... +``` + +- [ ] **Step 2:** Add a new section immediately after `## Three judgment sites and ACP kinds` (current lines 14–20), before `## ReviewRecord required fields`: + +```markdown +## `subject_ref`: this tutorial's narrowed Assurance Claim Point + +Hawkins' own Assurance Claim Point (ACP) is never free-floating: every confidence argument is +anchored to one specific, located assertion in the argument (Hawkins 2011, Sec. 3, p. 8 — +`glid:def-hawkins--assurance-claim-point`). `subject_ref` is this tutorial's own narrowed, +single-element analog: the one qualified name the record's `claim` is directly about, checkable +both from Python (`validate_record(record, model=model)` resolves it via `model.find()`) and from +the model's own side, via a real SysML metadata tag (SysML v2 formal/2026-03-02 §7.27.2): + +```sysml +metadata def ReviewRecordRef { + attribute identifier : String; +} + +metadata ac001Tag : ReviewRecordRef about nominal { + identifier = "AC-001"; +} +``` + +The `about` clause binds the usage's inherited `annotatedElement` feature to the named subject — a +real, queryable model relationship, not a string a reader has to trust. `subject_ref` is required +for `asserted_context` and `asserted_solution`; for `asserted_inference` it may stay empty only +when `premises` is non-empty (the pure cross-record synthesis case — `AI-C10` is the one record in +this tutorial that uses this exemption). `src/toaster/query.py`'s `get_review_record_refs()` is the +model-to-Python direction: given a loaded model, it finds every `ReviewRecordRef` tag and what it's +about, independent of any notebook's own Python objects. `validate_record` cross-checks both +directions automatically whenever a `model` is passed and a tag already exists for that record's +own `identifier`. +``` + +- [ ] **Step 3:** Update the `## Three judgment sites and ACP kinds` table (current lines 14–20) — add a column noting the required-ness rule: + +```markdown +| Site | Kind | Hawkins ref | `subject_ref` | +|---|---|---|---| +| Assumption or context used for a claim | `asserted_context` | §3.2 | required | +| Child claims supporting a parent | `asserted_inference` | §3.1 | required unless `premises` is non-empty | +| Evidence supporting a conclusion | `asserted_solution` | §3.3 | required | +``` + +- [ ] **Step 4:** Update the `## Judgment record construction zone` group listing (current lines 60–95) to show the new first group: + +```markdown +[markdown] narration: what is being claimed, and what, specifically, it is about +[code] claim = "..." + subject_ref = "ToasterDemo::..." + model_ref = "..." +``` + +- [ ] **Step 5:** Verify the skill file still renders correctly (no broken markdown/code-fence nesting) by reading it back in full after editing. + +- [ ] **Step 6:** Commit: +```bash +git add .claude/skills/toaster-review-protocol/SKILL.md +git commit -m "toaster-review-protocol: document subject_ref as this tutorial's ACP analog" +``` + +--- + +## Task 11: `toaster-recipe` skill update + +**Depends on:** none of the other tasks strictly, but make this edit alongside or after Task 10 so both skill files stay consistent with each other in the same sitting. + +**Files:** +- Modify: `.claude/skills/toaster-recipe/SKILL.md` + +- [ ] **Step 1:** In the `## Judgment record notebooks` section (current lines ~117–121), add one sentence pointing at the new first construction-zone group: + +```markdown +A notebook that builds a `ReviewRecord` (an `asserted_context`, `asserted_inference` or +`asserted_solution` judgment) uses `toaster-review-protocol`'s own construction-zone pattern for +it, not one dense call: name each group of fields, narrate what it's for, print it, then assemble. +The first group now names the record's own subject (`subject_ref`) alongside its `claim`, and +builds the matching `ReviewRecordRef` metadata-tag fragment the same way a model-increment cell +builds any other named fragment (see toaster-review-protocol's own subject_ref section). The size limits below +are relaxed for this content (see that skill for the exact grouping and why). +``` + +- [ ] **Step 2:** Add one sentence to the "Tall's three worlds" seam-cell guidance (current lines ~107–111, the "For the author's own reference" mapping) noting that a judgment-record notebook's seam cell now narrates a bridged connection, not a choice between readings — this closes DL-075: + +```markdown +**Judgment-record notebooks specifically:** the seam cell narrates one bridged connection — the +tagged SysML text (the subject plus its `ReviewRecordRef` usage), the tool that loads it and +cross-checks both the Python record and the model tag, and a result showing they agree +(`validate_record` plus `get_review_record_refs`) — not a choice between the model's own +construct/tool/result triad and the record's own fields/`validate_record`/result triad +(`decisions/log.md` DL-075, retired by this convention). +``` + +- [ ] **Step 3:** Verify the skill file still renders correctly. + +- [ ] **Step 4:** Commit: +```bash +git add .claude/skills/toaster-recipe/SKILL.md +git commit -m "toaster-recipe: judgment-record construction zone narrates subject_ref; closes DL-075" +``` + +--- + +## Task 12: Glossary — `gl:refines` edge + +**Depends on:** none (can run any time; independent file). + +**Files:** +- Modify: `glossary/definitions/hawkins.ttl` + +**Exact edit:** the `glid:def-hawkins--assurance-claim-point` term already exists (confirmed present, lines 61–70 of the current file) and is already `gl:confirmed`. Add a new tutorial-side definition edge immediately after it, in the same file, following this file's own existing style (each definition block separated by a blank line): + +```turtle +glid:def-tutorial--subject-ref + a gl:Definition ; + gl:confirmedBy "Z" ; + gl:locator "src/toaster/evidence.py ReviewRecord.subject_ref; models/*.sysml ReviewRecordRef" ; + gl:refines glid:def-hawkins--assurance-claim-point ; + gl:source glid:src-tutorial ; + gl:status gl:confirmed ; + gl:term glid:term-assurance-claim-point ; + gl:text "This tutorial's own narrowed analog of an Assurance Claim Point: a single, checkable qualified name a judgment record is directly about, anchored on both the Python side (subject_ref) and the model side (a ReviewRecordRef metadata tag), without importing GSN's argument-graph apparatus." . +``` + +Before writing this, check `glossary/sources/sources.ttl` for the exact id this tutorial uses as its own source (the design spec and CLAUDE.md both describe "this tutorial" as source N=9 in the glossary's own register) — use whatever id is actually registered there (likely `glid:src-tutorial`, but confirm, don't guess) rather than inventing a new one. + +Also check `glossary/terms/terms.ttl` for whether `glid:term-assurance-claim-point` already exists as a `Term` node (it should, since the Definition edge already references it) — if for some reason it does not, this task must also add the `Term` node itself, following the existing style in that file. + +- [ ] **Step 1:** Confirm the exact `gl:source` id for "this tutorial" in `glossary/sources/sources.ttl`. +- [ ] **Step 2:** Confirm `glid:term-assurance-claim-point` exists in `glossary/terms/terms.ttl`. +- [ ] **Step 3:** Add the `gl:refines` edge above (with the confirmed source id) to `glossary/definitions/hawkins.ttl`. +- [ ] **Step 4:** Run the glossary's own check: +```bash +uv run python -m glossary check +``` +Expected: exits 0. +- [ ] **Step 5:** Run `uv run python -m glossary lookup "assurance claim point"` and confirm the output now shows both the Hawkins edge and the new tutorial `gl:refines` edge. +- [ ] **Step 6:** Commit: +```bash +git add glossary/definitions/hawkins.ttl +git commit -m "Glossary: tie subject_ref/ReviewRecordRef to Hawkins' Assurance Claim Point" +``` + +--- + +## Task 13: Close DL-075; record this work + +**Depends on:** Task 9 (the retrofit must be complete and verified before this entry can honestly describe it as done). + +**Files:** +- Modify: `decisions/log.md` + +- [ ] **Step 1:** Append a new DL entry (check the current highest DL number at the time this task runs — it was `DL-083` as of this plan's writing, so this is very likely `DL-084`, but confirm with `grep -o "^## DL-[0-9]*" decisions/log.md | sort -t- -k2 -n | tail -1` before writing the number) in this repo's standard format (Path/Decision/Principles applied/Reasoning/Determined/Extension/Provenance), recording: + - **Path:** the Hawkins-anchor design (`docs/superpowers/specs/2026-10-01-hawkins-judgment-record-anchor-design.md`) and this plan were implemented in full: `subject_ref` added to `ReviewRecord`; `ReviewRecordRef` metadata construct introduced in Ch2 and carried through every later cumulative model; `get_review_record_refs()` added; every original-authoring judgment record (`AC-001`, `AC-C03`, `AS-C03`, `AI-C04`, `AC-C06`, `AS-C06`, `AI-C06`, `AS-C08`, `AC-C10`) retrofitted with a real, resolvable `subject_ref` and a matching model tag; every reconstruction/ledger carries the value forward; every negative control that would otherwise have gained an unintended second validation error was fixed to keep demonstrating exactly its own one intended failure; `toaster-review-protocol` and `toaster-recipe` updated; the glossary's existing `assurance-claim-point` term now has a `gl:refines` edge from this tutorial's own convention. + - **This closes DL-075.** State explicitly: the recurring judgment-record seam escalation (nine persona reports and ten ACE syntheses converging on it) is retired because judgment-record notebooks no longer face a choice between the model's own construct/tool/result triad and the record's own fields/`validate_record`/result triad — the seam cell now narrates one bridged connection (the tagged SysML text, the tool that loads and cross-checks both representations, and a result showing they agree), which is a concrete instance of AGENTS.md 1.10's requirement, not a judgment call between readings. + - **Provenance:** cite the spec and this plan by path; note the primary-source read of Hawkins et al. 2011 and the live OpenSysML v0.9.0 probe that grounded the design before any code was written. + +- [ ] **Step 2:** If `decisions/log.md` has a "status: open" marker or similar on the original DL-075 entry, update it to point at the new closing entry (follow whatever convention this file already uses for cross-referencing a later entry that resolves an earlier one — check a few existing examples in the file before writing this). + +- [ ] **Step 3:** Commit: +```bash +git add decisions/log.md +git commit -m "DL-0NN: close DL-075 -- Hawkins judgment records now anchored to the model" +``` + +--- + +## Task 15: Retrofit `exercises/` so completing them correctly still validates clean + +**Why this task exists (added after Task 1's independent review, not in the original spec):** the design spec and this plan's original Tasks 1–9 scoped only `chapters/`. Task 1's reviewer found, by an AST scan of every `ReviewRecord(` call site in the repo, that `exercises/ch03`, `ch04`, `ch06`, `ch08`, `ch09`, `ch10` each build a `-EX`-suffixed mirror of a real chapter record (e.g. `AS-C08-EX` mirrors `AS-C08`, same subject), and `exercises/ch02` is a pure fill-in-the-blank template (every field, including `model_ref`, is a `"# ..."` comment placeholder the learner must replace, not executable as committed). Without this task, a learner who correctly completes `ch03`/`ch04`/`ch06`/`ch08`/`ch09`/`ch10`'s exercise by mirroring the chapter notebook exactly (the exercises' own stated intent) would hit a new, unexplained `subject_ref is empty` validation error that no exercise instruction mentions — a real regression this plan would otherwise silently introduce, not a pre-existing gap. (`exercises/ch08` and `exercises/ch02` also have pre-existing, unrelated breakage — a typo'd constraint name in ch08, comment placeholders by design in ch02 — neither is this task's concern; this task only prevents the *new* breakage this plan's own schema change would cause.) + +**Depends on:** Tasks 4, 6, 7, 8, 9 (needs each chapter's final `subject_ref` value decided — reuse those exact values, do not re-derive). + +**Files:** +- Modify: `exercises/ch03/exercise.ipynb` (`AC-C03-EX` → `"ToasterDemo::timely"`, `AS-C03-EX` → `"ToasterDemo::timely"`) +- Modify: `exercises/ch04/exercise.ipynb` (`AI-C04-EX` → `"ToasterDemo::ApplyHeat"`) +- Modify: `exercises/ch06/exercise.ipynb` (`AC-C06-EX` → `"ToasterDemo::heatGenerationReq"`, `AS-C06-EX` → `"ToasterDemo::ResistanceCoil"`, `AI-C06-EX` → `"ToasterDemo::HeatingAssembly::heatGen"`) +- Modify: `exercises/ch08/exercise.ipynb` (`AS-C08-EX` → `"ToasterDemo::deliveredEnergyBoundedBySupply"`; its own empty-identifier negative control gets the same fix pattern as Task 7 Step 4 — add this same `subject_ref` so it keeps demonstrating exactly `["identifier is empty"]`) +- Modify: `exercises/ch09/exercise.ipynb` (`AS-BAD-EX` → `"ToasterDemo::ResistanceCoil"`; `AS-C06-EX` → `"ToasterDemo::ResistanceCoil"`; `AS-C08-EX` → `"ToasterDemo::deliveredEnergyBoundedBySupply"`; `AS-PLACEHOLDER-EX` → `"ToasterDemo::ResistanceCoil"` (must stay `[]`, same reasoning as Task 8); its own empty-identifier negative control → same `subject_ref` as the `AS-C08-EX` fix, same single-error reasoning) +- Modify: `exercises/ch10/exercise.ipynb` (`AC-C10-EX` → `"ToasterDemo::EnergyConservationReq"`; `AI-BAD-EX` → `"ToasterDemo::HeatingAssembly::heatGen"`; `AS-C06-EX` → `"ToasterDemo::ResistanceCoil"`; `AS-C08-EX` → `"ToasterDemo::deliveredEnergyBoundedBySupply"`; `AI-C06-EX` → `"ToasterDemo::HeatingAssembly::heatGen"`; `AI-C10-DRAFT-EX` and `AI-C10-EX` → check the premises-non-empty exemption exactly as Task 9 Step 4 does for the non-`-EX` versions, same logic, same conclusion expected) +- Modify (markdown only, one sentence): `exercises/ch02/exercise.ipynb` — add `subject_ref` to the field list the learner is asked to fill in (its own `AC-C02` record is `kind="asserted_context"`, so it needs a non-empty `subject_ref` once completed; add a `# Qualified name of the model element this assumption is directly about` placeholder line, matching the file's own existing placeholder-comment style, immediately after the existing `claim=` placeholder line). + +- [ ] **Step 1:** For each notebook above (except `ch02`), add `subject_ref=` to each affected `ReviewRecord(...)` call, exactly mirroring the corresponding chapter task's own value. +- [ ] **Step 2:** For `exercises/ch02/exercise.ipynb`, add the one placeholder line described above. +- [ ] **Step 3:** For every notebook that actually executes as committed today (`ch03`, `ch04`, `ch06`, `ch09`, `ch10` — confirm which ones currently run clean end-to-end before this change, since `ch08`'s own pre-existing typo already blocks full execution regardless of this task), run it end to end (`uv run jupyter nbconvert --to notebook --execute --output /tmp/.ipynb` or equivalent) and confirm every `validate_record(...)` call in it produces exactly the error list its own markdown/assert cells expect — no new, unexplained `subject_ref` error anywhere. +- [ ] **Step 4:** `uv run pytest -q` — no regressions (exercises aren't covered by the pytest suite, so this is a sanity check only, not the real acceptance gate for this task — Step 3's notebook execution is). +- [ ] **Step 5:** Commit: +```bash +git add exercises/ch02/exercise.ipynb exercises/ch03/exercise.ipynb exercises/ch04/exercise.ipynb exercises/ch06/exercise.ipynb exercises/ch08/exercise.ipynb exercises/ch09/exercise.ipynb exercises/ch10/exercise.ipynb +git commit -m "Retrofit exercises/ with subject_ref, mirroring each chapter's own value" +``` + +--- + +## Task 14: Full verification sweep + +**Depends on:** every prior task. + +- [ ] **Step 1:** Full test suite: +```bash +uv run pytest -q +``` +Expected: all pass, no regressions from the baseline recorded before Task 1 (`411 passed, 7 deselected` plus `7 passed` under `-m checkpoint`, per the most recent PR description — the new tests from Tasks 1–2 add to this count, nothing should be removed from it). + +- [ ] **Step 2:** Construction consistency, across every chapter, not just the ones this plan touched: +```bash +uv run python scripts/check_construction.py --check +``` +Expected: exit 0. + +- [ ] **Step 3:** Glossary: +```bash +uv run python -m glossary check +``` +Expected: exit 0 (same pre-existing source-PDF-absent warnings as before this plan, no new errors). + +- [ ] **Step 4:** Every cumulative model loads cleanly end to end (not just the ones with new content — a regression in an untouched file would mean a carry-forward step accidentally touched the wrong file): +```bash +uv run python - <<'EOF' +from pathlib import Path +import opensysml +conn = opensysml.connect(version="v0.9.0") +for n in ["02", "03", "04", "05", "06", "07", "08", "10"]: + source = Path(f"models/ch{n}-cumulative.sysml").read_text() + model = conn.load_from_content(source, strict=False) + assert model.ok, f"ch{n}: {model.diagnostics}" + print(f"ch{n}: ok") +conn.close() +EOF +``` + +- [ ] **Step 5:** Cross-representation agreement, across every original-authoring record, in one pass: +```bash +uv run python - <<'EOF' +from pathlib import Path +import opensysml +from toaster.query import get_review_record_refs + +conn = opensysml.connect(version="v0.9.0") +source = Path("models/ch10-cumulative.sysml").read_text() # the last file has every tag +model = conn.load_from_content(source, strict=False) +assert model.ok + +refs = {r["identifier"]: r["annotated_element"] for r in get_review_record_refs(model)} +expected = { + "AC-001": "ToasterDemo::nominal", + "AC-C03": "ToasterDemo::timely", + "AS-C03": "ToasterDemo::timely", + "AI-C04": "ToasterDemo::ApplyHeat", + "AC-C06": "ToasterDemo::heatGenerationReq", + "AS-C06": "ToasterDemo::ResistanceCoil", + "AI-C06": "ToasterDemo::HeatingAssembly::heatGen", + "AS-C08": "ToasterDemo::deliveredEnergyBoundedBySupply", + "AC-C10": "ToasterDemo::EnergyConservationReq", +} +for identifier, subject in expected.items(): + assert refs.get(identifier) == subject, f"{identifier}: expected {subject}, got {refs.get(identifier)}" +print("all 9 tags agree:", sorted(expected)) +conn.close() +EOF +``` + +- [ ] **Step 6 (added after Task 1's independent review found this gap — F1):** Execute every judgment-record notebook end to end and confirm zero unexpected errors. Neither `pytest` nor this repo's current CI executes notebooks (notebook execution is still a placeholder pending WP-8), so this is the only check in the whole plan that would actually catch a retrofit step silently missed in Tasks 3–9 or 15 — treat it as load-bearing, not optional: + +```bash +for nb in \ + chapters/ch02-requirements/03-judgment-context.ipynb \ + chapters/ch03-measures/01-moe-definition.ipynb \ + chapters/ch03-measures/03-threshold-judgment.ipynb \ + chapters/ch04-functional-decomp/03-completeness-check.ipynb \ + chapters/ch06-recursive-decomp/02-second-level.ipynb \ + chapters/ch06-recursive-decomp/03-stopping-judgment.ipynb \ + chapters/ch08-checking/02-violation-witness.ipynb \ + chapters/ch08-checking/03-revision-flow.ipynb \ + chapters/ch09-coverage-sufficiency/02-evidence-completeness.ipynb \ + chapters/ch09-coverage-sufficiency/03-stale-detection.ipynb \ + chapters/ch10-traceability-signoff/01-traceability-graph.ipynb \ + chapters/ch10-traceability-signoff/02-judgment-synthesis.ipynb \ + chapters/ch10-traceability-signoff/03-engineering-signoff.ipynb \ + exercises/ch03/exercise.ipynb \ + exercises/ch04/exercise.ipynb \ + exercises/ch06/exercise.ipynb \ + exercises/ch09/exercise.ipynb \ + exercises/ch10/exercise.ipynb \ +; do + echo "=== $nb ===" + uv run jupyter nbconvert --to notebook --execute "$nb" --output /tmp/_verify_$(basename "$nb") --output-dir /tmp || echo "FAILED: $nb" +done +``` + +`exercises/ch08/exercise.ipynb` and `exercises/ch02/exercise.ipynb` are deliberately excluded from this loop — both have pre-existing breakage unrelated to this plan (a typo'd constraint name in `ch08`; comment placeholders by design in `ch02`) that predates and is out of scope for this work; Task 15 Step 3 already covers confirming no *new* `subject_ref`-shaped error in whichever of those two notebooks can run far enough to reach its `validate_record` call. Every notebook in the loop above must execute with no cell raising an unhandled exception. Any `AssertionError` whose message contains `subject_ref` is a retrofit step that was missed — go back and fix the specific record named in the traceback, in whichever Task (3–9, 15) owns that chapter, then re-run this whole step from the top (a missed retrofit in one notebook does not block checking the others — run the full loop, collect every failure, then fix them all before re-running). + +- [ ] **Step 7:** Contradiction sweep — grep for stale references to the pre-retrofit state: +```bash +grep -rn "validate_record(record)\s*$\|validate_record(r)\s*$" chapters/ --include=*.ipynb +``` +Review every hit by hand: a bare one-argument `validate_record` call on an `asserted_context`/`asserted_solution` record, or an `asserted_inference` record with empty premises, is a sign a retrofit step in Tasks 4–9 was missed (it would now report a `subject_ref`-required error that no markdown cell explains). + +- [ ] **Step 8:** Report results. No commit for this task (verification only) unless a fix is needed, in which case fix forward with its own commit and re-run this task's steps. + +--- + +## Execution Notes for Whoever (or Whatever) Runs This Overnight + +- **Tasks 1 and 2 may run in parallel** (independent files, no shared blast zone). +- **Tasks 3 through 9 must run strictly in series**, even though the records themselves don't logically depend on each other, because every one of them edits the same shared set of `models/chNN-cumulative.sysml` files — parallel branches would conflict at integration. This is exactly the "shared blast zone forces serial dispatch" case `orchestrator-protocol` already describes. +- **Tasks 10, 11, and 12 may run in parallel with Tasks 4–9** (and with each other) once Task 3 is merged — they touch skill files and the glossary only, no shared cumulative-model blast zone. +- **Task 15 depends on Tasks 4, 6, 7, 8, 9** (needs each chapter's final `subject_ref` value decided) but not on Tasks 10–13; it may run in parallel with those once Task 9 lands. +- **Task 13 depends on Task 9** (needs the retrofit done to describe truthfully) but not on Tasks 10–12 or 15. +- **Task 14 depends on everything, including Task 15** — added after Task 1's own independent review found two gaps in the original plan (not in the spec): no task covered `exercises/` (now Task 15), and no task actually executed any notebook end to end, so a missed retrofit step would otherwise go undetected by `pytest`/CI alone (now Task 14 Step 6). Both gaps are fixed in this version of the plan; if anyone resumes from an earlier printed/cached copy of this document, re-read this section before trusting it's complete. +- Each task above is sized to be one `builder`/`reviewer` work contract under `orchestrator-protocol`'s "Plan-driven non-chapter work" mechanism. Route escalations (a stub that won't validate, a qualified name that doesn't resolve as expected, a negative control whose exact current field values don't match what this plan assumed) through the normal chain — builder to orchestrator to ACE — rather than guessing past them; the ACE rules or escalates to Z per its own protocol, and logs either way. diff --git a/docs/superpowers/specs/2026-10-01-hawkins-judgment-record-anchor-design.md b/docs/superpowers/specs/2026-10-01-hawkins-judgment-record-anchor-design.md new file mode 100644 index 0000000..c509ada --- /dev/null +++ b/docs/superpowers/specs/2026-10-01-hawkins-judgment-record-anchor-design.md @@ -0,0 +1,256 @@ +# Design: anchoring Hawkins judgment records to the model + +**Status:** design approved in conversation; pending written-spec review +**Author:** session design work with Z, 2026-10-01 +**Related:** `decisions/log.md` DL-075 (the recurring judgment-record seam escalation, which this design retires); `src/toaster/evidence.py`; `.claude/skills/toaster-review-protocol/SKILL.md`; `.claude/skills/toaster-recipe/SKILL.md`; `glossary/definitions/hawkins.ttl` + +## Why this exists + +Z introduced the Hawkins et al. 2011 `ReviewRecord` mechanism to document and classify the +engineering judgment calls this tutorial makes — not as decoration, but so a reader can see *why* +a claim should be believed, not just that a checker printed `True`. Ten ACE syntheses and +twenty-nine persona reports this session independently converged on one open question about it +(DL-075): what does AGENTS.md 1.10's seam requirement mean for a notebook whose only new content +is a judgment record, not a SysML construct? That question turned out to rest on a real, fixable +gap, found by reading the actual Hawkins paper rather than the glossary's own curated extracts of +it, and then by testing what SysML v2 itself already provides for exactly this problem. + +## What the primary source actually says + +Hawkins, Kelly, Knight and Graydon, *A New Approach to Creating Clear Safety Arguments* (SSS +2011, pp. 3-23), read in full from the authors' own self-archived copy +(`https://www-users.york.ac.uk/~rdh2/papers/HawkinsSSS11.pdf`), not from memory or the glossary's +prior extracts alone. + +**The core move.** Split an argument into a *safety argument* (the direct claim and its support) +and a *confidence argument* (why a skeptical reader should believe that support is sufficient), +kept explicitly separate because conflating them is what the paper diagnoses as the cause of +"large, rambling... poorly-focused" real arguments (Sec. 2). + +**The three judgment sites.** Every assertion in an argument is one of three kinds — asserted +inference (a claim is supported by sub-claims, Sec. 3.1), asserted context (a background +assumption is introduced, Sec. 3.2), or asserted solution (evidence is cited to close an argument, +Sec. 3.3) — and each is tied to a specific, named **Assurance Claim Point (ACP)**: a located link +in an argument graph (built in GSN, the Goal Structuring Notation), not a free-floating claim. + +**The judgment principle.** An assurance deficit is "any knowledge gap that prohibits perfect +(total) confidence" (Sec. 1, p. 4). Completely mitigating every deficit is "not normally +achievable," so a judgment is required about which residual deficits can be tolerated, assessed by +"expert judgment of the likelihood and severity" of counter-evidence (Sec. 3.4, p. 12). This +tutorial's `counterevidence`/`residual_uncertainties` fields are this principle, directly. + +**Domain generality, in the authors' own words (Sec. 5, Conclusions):** "We have limited our +discussion in this paper to safety cases, but the concepts apply immediately to *any* property of +interest... the overall structures and approaches would be identical." Using this framework for +engineering judgment about a toaster, not a hazard, is not a stretch — it is the generalization the +authors themselves license. + +**What correctly does not transfer.** GSN itself, Assurance Claim Points as graph-tags, the +recursive confidence-argument *patterns* (six to seven claim-nodes deep per ACP, Figs. 13-14), and +the hazard/risk/certification framing are built for formal, third-party-reviewed certification +arguments. `ReviewRecord` is a flat dataclass, not a GSN tree, and stays that way under this +design — importing the graph apparatus would not serve a teaching tutorial. + +**What does transfer, and is currently missing.** An ACP is never free-floating — every confidence +argument is anchored to one specific, located assertion. `ReviewRecord` has no equivalent today: +`model_ref` is a bare file-path string, `content_hash` hashes an entire file, and neither says +*which element* a given judgment is actually about. That anchor exists only informally, in the +free-text `claim` string. This is the real, primary-source-grounded reason DL-075 could not be +settled by argument alone — the thing the two competing seam readings were really both reaching +for was a located subject, and neither the record's own fields nor the model had one. + +## The decision + +Give `ReviewRecord` a real, singular, checkable subject — the tutorial's own narrowed analog of an +ACP — anchored on **both** sides: a new field in Python, and a real SysML metadata tag in the +model itself, so the connection is visible from the model's own side, not just assumed from +outside it. (Caught directly in review: a Python-only anchor, checked one direction only, would +leave the model itself blind to its own judgment records — a real gap given Chapter 10's whole +purpose is model-side traceability querying.) + +### Why singular, not a list + +A first draft of this design used `subject_refs: list[str]`. Z's own objection killed it: the +*relationship* a judgment has to different elements it touches is not uniform, so one undifferentiated +list loses the distinction. Hawkins' own answer to multiplicity is not "a bigger ACP" — each kind of +assertion already has its own role-typed container (assumptions for context, evidence for solution, +premises for inference), and `ReviewRecord` already has all three: `assumption_refs`, `evidence_refs`, +`premises`. The actual gap was narrower: a single, clean subject — the one thing the claim is +*directly* about — which those three existing lists were never meant to carry. `subject_ref` fills +exactly that slot, matching Hawkins' own "one ACP, one link" shape. + +### Why a model-side tag, not just a validated Python string + +Probed directly against OpenSysML v0.9.0 before committing (see Verification below), because this +toolchain has a known, adjacent gap (`VerificationMethod` metadata, cited in Ch8's own notebook, is +"not yet supported"), so nothing about custom metadata could be assumed to work. It does, for the +shape this design needs. SysML v2 §7.27.2 already defines exactly the relationship wanted: a +`metadata def` with an `about` clause binds the usage's inherited `annotatedElement` feature to the +named target(s) — a real, queryable model relationship, not a string a human has to trust. + +## Design + +### 1. Schema (`src/toaster/evidence.py`) + +```python +@dataclass +class ReviewRecord: + identifier: str + kind: Literal["asserted_context", "asserted_inference", "asserted_solution"] + claim: str + subject_ref: str = "" # NEW — the tutorial's own ACP: one qualified name + model_ref: str = "" + content_hash: str = "" + ... # unchanged otherwise +``` + +`validate_record(r: ReviewRecord, model: Any | None = None) -> list[str]` gains: + +- **Required-ness rule** (no model needed to check this): `subject_ref` must be non-empty for + `kind in ("asserted_context", "asserted_solution")`. For `kind == "asserted_inference"`, + `subject_ref` may be empty **only if** `premises` is non-empty — the pure cross-record synthesis + case (`AI-C10` is the only current record that needs this exemption). +- **Resolution check** (needs `model`, the new optional parameter): if `subject_ref` is set, + `model.find(subject_ref) is not None`, else an error naming the unresolved reference. +- **Cross-representation check** (needs `model` and the new query helper, see below): if a + `ReviewRecordRef` tag exists in the model for this record's `identifier`, its own + `annotatedElement` must match `subject_ref` and its own `identifier` attribute must match + `r.identifier` — catching drift between the Python side and the model side if they're ever + edited independently. + +### 2. Model-side construct (new SysML, introduced once) + +```sysml +metadata def ReviewRecordRef { + attribute identifier : String; +} +``` + +Introduced in Chapter 2 (the tutorial's first judgment record, `AC-001`), carried forward in every +cumulative model from Ch2 onward — the same convention every other reusable construct in this +tutorial already follows (one definition, authored once, present in every later chapter's own +committed fixture). Each judgment notebook's construction zone builds one usage, using the +**explicit `about` form** (confirmed by probe to be the form that actually populates +`annotatedElement`; the implicit nested shorthand does not): + +```sysml +metadata ac001Tag : ReviewRecordRef about timely { + identifier = "AC-001"; +} +``` + +Only the definition is a new construct, once, in Ch2. Every later chapter's usage is the same +already-taught idiom (matching how `allocate` usages recur across chapters without each counting +as a new construct under SA-8). + +### 3. Query helper (`src/toaster/query.py`) + +```python +def get_review_record_refs(model, index=None) -> list[dict]: + """Every ReviewRecordRef tag in the model: {identifier, annotated_element}.""" +``` + +Built on `to_api_json()`, the same pattern `get_satisfy_relationships()` already uses (`MetadataUsage` +elements are visible to `model.query()` natively — no D-001-style gap for basic discovery — but +`annotatedElement` itself is only in the JSON export, confirmed by probe). This is the model→Python +direction: given a loaded model, find every judgment tag and what it's about, independent of any +particular notebook's own Python objects. + +### 4. Skill updates + +**`toaster-review-protocol`:** +- Add `subject_ref` to the required-fields example and table. +- New section, citing SysML v2 formal/2026-03-02 §7.27.2 directly: explain `about`/`annotatedElement` + as the real spec mechanism, and `subject_ref` + `ReviewRecordRef` together as this tutorial's + deliberately narrowed analog of Hawkins' Assurance Claim Point — one link, no GSN graph. +- Update the "three judgment sites" table with the required-ness rule above. + +**`toaster-recipe`:** +- The judgment-record construction zone gains a new first named group: "what is being claimed, and + what, specifically, it is about" — covering `claim`, `subject_ref`, and the `ReviewRecordRef` + usage together, narrated as one idea (mirrors how a model-increment cell is already one idea with + several named fragments). +- **Resolves DL-075.** The seam cell for a judgment-record notebook now narrates one bridged + connection, not a choice between two competing triads: the tagged SysML text (subject + its + `ReviewRecordRef` usage) is printed; the tool loads it and checks both the Python record and the + model tag; the result shows they agree (`validate_record` plus `get_review_record_refs`). This is + a concrete, checkable instance of AGENTS.md 1.10's requirement, not a judgment call between + readings A/B/C — the question those readings were circling is answered by giving the record an + actual anchor, not by picking which existing triad counts. + +### 5. Glossary + +Add `assurance-claim-point` as a confirmed Hawkins-sourced term: + +```turtle +glid:def-hawkins--assurance-claim-point + gl:locator "Sec. 3, p. 8 (PDF 6)" ; + gl:quote "the confidence argument is tied to a number of Assurance Claim Points (ACP)" ; + gl:text "The place in the safety argument where an assertion is made; a confidence argument is developed for each." ; +``` + +(This term already exists in `glossary/definitions/hawkins.ttl` — it was seeded during Pass 1 but +not yet cited anywhere in the skills or schema it now grounds. No new definition edge needed, only +the `gl:refines` tying `subject_ref`/`ReviewRecordRef` to it as the tutorial's own narrowed +instance, and the citation added to `toaster-review-protocol`.) + +### 6. Retrofit (every existing record, in one coordinated effort) + +| Record | Chapter (original authoring) | `subject_ref` | Needs model tag? | +|---|---|---|---| +| `AC-001` | Ch2 | `timely` (or the specific requirement usage it concerns) | Yes — first usage, defines `ReviewRecordRef` | +| `AC-C03` | Ch3 | the MoE/MoP element it concerns | Yes | +| `AS-C03` | Ch3 | `timely` or the threshold constraint | Yes | +| `AI-C04` | Ch4 | the balance inequality / `ApplyHeat` | Yes | +| `AS-C06` | Ch6 | `ResistanceCoil` or the mechanism-selection element | Yes | +| `AI-C06` | Ch6 | `HeatingAssembly` (the thing judged as a stopping point; `heatGen`/the allocation move to `premises`/`evidence_refs`, not a second subject) | Yes | +| `AS-C08` | Ch8 | `deliveredEnergyBoundedBySupply` | Yes | +| `AI-C10` | Ch10 (synthesis) | — (exempt: `asserted_inference` with non-empty `premises`) | No | + +Ch9 and Ch10's own *reconstructions* of `AC-C06`/`AS-C06`/`AS-C08`/`AI-C06` (the judgment ledger +notebooks) carry the same `subject_ref` value forward in their own Python objects; they do not +rebuild the model tag, since the cumulative model they load already has it from the originating +chapter. + +Exact `subject_ref` values for each record need confirming against the real model's own qualified +names at implementation time — the table above states intent, not final strings. + +## Verification already done (this session, before writing this spec) + +Probed directly against a throwaway fixture with OpenSysML v0.9.0, not assumed: + +1. `metadata def ReviewRecordRef { attribute identifier : String; }` plus an explicit-`about` usage + loads cleanly (`model.ok == True`, no diagnostics). +2. `model.find()` resolves the metadata usage directly. +3. `model.query()` finds `MetadataUsage` elements natively (no JSON-export workaround needed for + basic discovery, unlike `SatisfyRequirementUsage`'s D-001 gap). +4. `to_api_json()` shows a correct `annotatedElement` relationship pointing at the real subject, for + the explicit `about` form. +5. The attribute's bound value round-trips through the JSON (`LiteralString`/`FeatureValue` chain), + confirmed readable. +6. The **implicit nested form** (`@ReviewRecordRef {...}` with no `about`, inside the subject's own + body) does *not* populate an explicit `annotatedElement` entry the same way — confirmed by + contrast in the same probe. Design conclusion: standardize on the explicit `about` form. + +## What this does not change + +- `ReviewRecord` stays a flat Python dataclass. No GSN graph, no recursive confidence-argument + patterns, no certification framing — those were correctly identified as not transferring, and + nothing here reopens that. +- `model_ref`/`content_hash` are unchanged in meaning (whole-file staleness detection via + `check_stale()` stays exactly as it is); `subject_ref` is additive, a different and narrower kind + of anchor (one element, not one file). +- `assumption_refs`, `evidence_refs`, `premises` are unchanged — they stay free text, with new + guidance (not a requirement) to use a qualified name when an entry happens to name a real model + element. + +## Open items for the implementation plan, not resolved by this spec + +- Exact `subject_ref` values per existing record (table above states intent; needs confirming + against each chapter's real qualified names). +- Contract sequencing: `ReviewRecordRef`'s own definition text must be settled once (this spec fixes + its shape) and land in Ch2 before later chapters build usages of it — whether that means strictly + serial contracts or parallel contracts sharing a pinned definition text is a planning decision, not + a design one. +- Whether a new DEFERRED.md gap entry is warranted if implementation surfaces any OpenSysML + rough edge beyond what this spec's probe already found (none found so far). diff --git a/docs/superpowers/specs/2026-10-01-large-scale-user-testing-design.md b/docs/superpowers/specs/2026-10-01-large-scale-user-testing-design.md new file mode 100644 index 0000000..342e72e --- /dev/null +++ b/docs/superpowers/specs/2026-10-01-large-scale-user-testing-design.md @@ -0,0 +1,115 @@ +# Design: large-scale user testing — the persona × modality grid, and an interviewing interpretive layer + +**Status:** draft, written at Z's direction; not yet reviewed +**Author:** session design work with Z, 2026-10-01 +**Related:** `.claude/skills/user-testing/SKILL.md`, `.claude/skills/ace-protocol/SKILL.md`, `.claude/agents/simulated-learner.md`, `.claude/agents/ace.md`, `.claude/agents/orchestrator.md`, decisions/log.md DL-075 through DL-083 (the existing single-modality user-testing battery this design extends, not replaces) + +## Why this exists + +Tonight's own ad hoc testing (browser-rendered book, README accuracy, one exercise notebook run by hand) found real bugs in about twenty minutes of directed looking — three rendering bugs, three documentation inaccuracies, a missing LICENSE, an incomplete CI pipeline, and an exercise-scaffolding inconsistency that would confuse a first-time learner. All of that was one session, one perspective, testing mostly one modality (the rendered book) end to end and two others (README, one exercise) only glancingly. Z's ask is to make that kind of finding *systematic* rather than incidental: before this tutorial goes in front of real outside users across GitHub, GitHub Pages, and a local clone, run the same kind of close reading **deliberately, at scale, across who's reading and where they're reading it**, and turn whatever it finds into a short, prioritized, actionable list — not a pile of forty raw reports Z has to read personally. + +This repo already has most of the machinery this needs. `user-testing` already defines three learner personas with pinned models matched to their claimed capability, a fixed execution checklist, and a fixed report format. `ace-protocol` already defines a triage layer — the ACE — whose entire job is "resolve only what truly needs Z and nothing that wastes Z's time," ruling where principles determine the answer and escalating a concise, Z-idiom brief otherwise. What's missing is two things: **the grid** (today's `user-testing` tests exactly one modality — a learner reading chapter notebooks from a local checkout — and has no persona-specific coverage of the other two places a real user actually encounters this project), and **the interview step** (today's ACE synthesis reads finished, static reports; it cannot go back and ask a specific sub-agent a clarifying follow-up before ruling, the way a human running a usability study would ask a participant "what did you expect to happen there?"). + +## The grid + +### Personas (unchanged from `user-testing`, reused as-is) + +| Persona | Model | Stands in for | +|---|---|---| +| **Novice** | Haiku 4.5 | A reader with no prior SysML/MBSE background, Python-literate | +| **SE Practitioner** | Sonnet 5 | A systems engineer with no SysML v2 background, judging whether modeling choices are defensible | +| **Returning Learner** | Sonnet 5 | Someone who already completed earlier chapters, now starting fresh content | + +No new persona is introduced. Z's own model-matching rule (`decisions/next-passes.md` §3, carried into `user-testing`: a simulated novice should not have more capability than the learner it stands for) applies identically here. + +### Modalities (new — this is what the grid actually adds) + +| Modality | What a real user does | What existing coverage there is today | +|---|---|---| +| **M1 — GitHub repo** | Lands on `github.com/Open-MBEE/toaster` from a search result, a link, or a colleague's recommendation. Reads the README, skims the file tree, maybe checks "Releases" or "Issues," decides whether to clone it. | None. Tonight's README fixes were my own ad hoc read, not a persona-structured test. | +| **M2 — GitHub Pages (rendered book)** | Opens the deployed book in a browser, reads chapters, maybe searches, maybe switches to mobile or dark mode, never opens a terminal. | None structured; tonight's browser sweep was thorough but was me, not a persona, and found technical-rendering bugs more than "does this make sense" judgments. | +| **M3 — Local clone, reading the chapters** | Follows `docs/setup.md`, runs `uv sync --locked`, opens chapter notebooks in Jupyter, reads and re-runs cells. | This is what `user-testing` already covers end to end — reuse its existing checklist and report format for this modality unchanged. | +| **M4 — Local clone, doing an exercise** | Having read a chapter, opens the matching `exercises/ch{N}/exercise.ipynb`, tries to fill it in using only what the chapter taught. | None. `user-testing`'s own checklist explicitly scopes to chapter notebooks; exercises are excluded from CI and from today's testing battery by design. Tonight's own finding (Ch1's exercise fails gracefully, Ch8's crashes with an unexplained `AssertionError`) came from me running two exercises by hand, not from persona testing. | + +### Which cells are worth running + +Not every persona × modality cell carries equal signal, and running all twelve blindly would waste budget on cells where the persona dimension doesn't actually change what's found. Scope the first run to the cells below; treat the rest as available, not default. + +| | M1: GitHub repo | M2: GitHub Pages | M3: Local, chapters | M4: Local, exercise | +|---|---|---|---|---| +| **Novice** | run | run | run (existing coverage, re-point at a fresh checkout) | run | +| **SE Practitioner** | — (low persona-differentiation for a README skim; skip unless M1 is re-run after major content changes) | run | run | run | +| **Returning Learner** | — (the persona's own definition, "starting this chapter fresh," doesn't map onto M1 at all) | run | run | run | + +That's **ten cells**, not twelve — M1 gets one persona (Novice, the persona a README's clarity matters most for; a practitioner or returning learner reading a README is not meaningfully different from a novice reading one, so testing it three times would just burn budget for the same finding repeated). This keeps the first run inside this session's own stated default workflow-size guideline (medium, under ten agents), right at the line, without the orchestrator needing to ask for an exception. + +## Sandboxing + +Every cell's sub-agent is dispatched the way this session's own builder/reviewer agents already were tonight: a cold session, no shared context with any other cell, given only its own work contract. The specific sandboxing mechanism differs by modality, because the modalities have genuinely different blast radii: + +- **M3 and M4 (local clone)** need a **private git worktree**, created the way `.claude/agents/orchestrator.md` (line 17) requires for every subagent dispatch: `git worktree add -b `, created by the dispatcher itself, never left to the `Agent` tool's own `isolation: "worktree"` mechanism — that mechanism has twice (DL-084, DL-085) provisioned a worktree from a stale base commit that predates the branch actually under test, producing false findings. Every M3/M4 contract records its own dispatch commit and requires the cell to report `Base check: git merge-base --is-ancestor HEAD -> yes/no` before trusting any of its content-level findings. This also means M3/M4 agents can be dispatched **before** any fix from this testing round lands, to test the actual current released state, and then **again** after fixes land, as a regression check — the same before/after discipline this session used all night for its own retrofit work. +- **M2 (GitHub Pages)** needs the **built, served site**, not a worktree — either the already-running local `myst start --execute` preview this session has open (serving what's on the current branch) or, once Task 7 of the CI/CD plan produces a real built static export, that export served locally under the `/toaster` path. The sub-agent needs the Browser tools (`mcp__Claude_Browser__*`), not a worktree; its "sandbox" is that it gets its own fresh tab and no memory of any other cell's findings. +- **M1 (GitHub repo)** needs **only read access to the repository as a visitor would see it on GitHub itself** — either the live `github.com/Open-MBEE/toaster` page (if public visibility and the current state are what's being tested) or, more usefully for testing *this branch's own proposed changes before they're merged*, a rendered preview of this branch's README and file tree. A worktree isn't wrong here, just unnecessary; a read-only checkout of the branch under test is sufficient. + +Every sub-agent, regardless of modality, is **named and kept addressable** after it reports — this is the one sandboxing requirement that's new relative to tonight's builder/reviewer pattern, and it's what makes the interview step (below) possible at all. Launch every cell's agent with a stable, descriptive name (not left to auto-naming) so the interpretive layer can address it by name in `SendMessage` later without having to hunt through `ListAgents` output to recover an opaque ID. + +## The record: what each cell actually produces + +Each sub-agent produces two things, not one: the existing narrative report `user-testing` already specifies (unchanged format: EXECUTION RESULTS / NARRATIVE OBSERVATIONS / STRUCTURAL CHECKS / OVERALL, extended per modality below), and a **structured findings list** the interpretive layer can process without re-reading full prose for every cell. + +### Structured findings list (new) + +Every finding a sub-agent reports — not just blocking ones — gets one row: + +``` +{ + "cell": "M2-novice", + "finding_id": "M2-novice-01", + "severity": "blocking | friction | confusing | cosmetic | positive", + "location": "exact URL, file path, or notebook cell reference", + "quote": "the exact text or exact behavior observed, never a paraphrase", + "expected": "what the persona expected to happen or find, in one sentence", + "actual": "what actually happened, in one sentence" +} +``` + +This is a deliberate narrowing of `user-testing`'s existing "blocking / minor / cosmetic" triage (which is the *synthesizer's* vocabulary, applied after the fact) down to what a reporting sub-agent can say about its own experience without having to judge severity itself — `severity` here is the sub-agent's own first-pass guess, explicitly re-judged by the interpretive layer, not binding. `positive` is included deliberately: a synthesis that only sees problems can't tell "this chapter has no issues" from "no one tested this chapter," and this session's own feedback memory (`[[feedback-durable-learnings]]`-style discipline, record from success and failure both) applies here the same as it does to any other kind of review. + +### Modality-specific additions to the existing report format + +- **M1 (GitHub repo):** add a "First five minutes" narrative section — what did the persona read, in what order, and at what point (if any) did they decide to clone it or give up. This is the one modality where *sequence* (what's read first) matters as much as content accuracy. +- **M2 (GitHub Pages):** add the same execution-results table `user-testing` already specifies for chapter notebooks (model.ok, bad.ok, printed output), since the rendered book's code cells carry real executed output the persona should be able to verify against what the prose claims — a persona finding the printed `Validation errors: []` and the prose's claim that it's clean disagree is exactly the kind of finding this grid exists to catch. +- **M3 (local, chapters):** unchanged from `user-testing` today. +- **M4 (exercises):** add an explicit **first-run-unmodified** execution result (run the exercise exactly as cloned, before attempting to fill in anything) separately from the persona's own attempt at filling it in — tonight's own Ch1-vs-Ch8 scaffolding-inconsistency finding depended on exactly this distinction (what happens before any input, not just what happens after a good-faith attempt), and a report that only covers the attempted fill-in would miss that class of finding entirely. + +## The interpretive layer + +**This is the existing ACE, not a new role.** `ace-protocol`'s own stated job — triage layer, rules where Z's principles determine the answer, escalates a concise brief otherwise, logs either way — is already the right shape for "consume the records and synthesize findings toward a concrete recommendation." What's added is a new capability the ACE does not have today: **the interview step**, inserted between "read every cell's structured findings" and "triage and rule or escalate." + +### Synthesis protocol (extends `user-testing`'s existing ACE synthesis protocol, modality-grid version) + +1. **Collect.** Read every cell's structured findings list and narrative report. Do not read full narrative prose for every row before triage — use the structured list as the index, and open full prose only for rows that need it (this keeps the ACE's own context budget bounded even at ten cells). +2. **Cluster.** Group findings that are really the same underlying issue seen from different cells — tonight's own README inaccuracies, for instance, would plausibly surface independently from an M1 Novice *and* an M3 Returning Learner re-reading setup instructions; these are one finding with two pieces of corroborating evidence, not two findings. +3. **Interview — the new step.** For any finding where the structured record is ambiguous, where severity is unclear, or where two cells' findings appear to conflict, the ACE sends a targeted follow-up message to the specific sub-agent that reported it, by name (`SendMessage` to the cell's own agent name, established at dispatch per the Sandboxing section above — this works because `SendMessage` "resumes it from its transcript" even for an agent that has already completed and reported). Ask one narrow, concrete question per message — "quote the exact sentence that confused you," "did you actually click that link, or assume it was broken," "what would you have expected to see instead" — the same discipline this session's own reviewer agents already used when probing a builder's self-report rather than trusting it at face value. This is not optional polish: a structured finding with `severity: "confusing"` and no interview is a guess about how bad it is; the same finding after one targeted follow-up is evidence. +4. **Triage**, using `user-testing`'s existing blocking/friction/confusing/cosmetic classification, now informed by interview answers where they were needed: for each cluster, classify severity, decide Rule or Escalate exactly as `ace-protocol` already specifies (can the frameworks and principles in `z-principles.md` determine what to do about this, or does it genuinely need Z's own judgment — a design question, a scope question, a taste question the ACE cannot resolve from principles alone). +5. **Synthesize a recommendation, not a report.** The deliverable is not "here are forty findings" — it is a short, prioritized list: what's already fixed (with evidence), what the ACE is ruling on directly and fixing or dispatching a builder contract for, what's being escalated to Z with a concise Z-idiom brief per finding, and what's explicitly out of scope for this round (recorded, not dropped — the same "known gaps, not silently fixed" discipline DL-084 already modeled tonight). Log it as a DL entry the same shape as `ace-protocol` already specifies, with one addition: a `Grid coverage:` line naming exactly which cells ran, so a future reader knows what this synthesis did and did not see. + +### Why this stays inside the ACE's existing authority, not a new one + +`ace-protocol`'s file-authority section already scopes the ACE to `decisions/log.md` and `.claude/skills/**/*.md`, read-only everywhere else. Nothing in this design asks the ACE to write chapter content, models, or CI config directly — ruled findings that need a content fix still go through the normal pipeline (`ace-protocol`: "the ACE does not edit repository files... the orchestrator dispatches a builder/author-role work contract to make the change and a reviewer to confirm it"). The interview step changes *what evidence the ACE has before it rules*, not *what the ACE is allowed to do with a ruling*. + +## What this design does not do + +- It does not replace `user-testing`'s existing single-modality (M3-only) battery or its persona definitions — it extends both. +- It does not propose running all ten cells on every future change; it scopes *this* first run to ten cells and leaves the full twelve-cell grid, plus a repeat-after-fixes regression pass, as a documented option, not a standing requirement. +- It does not change `ace-protocol`'s accountability model, file authority, or rule/escalate test — only its synthesis protocol gains one new step. +- It does not specify a UI or dashboard for the interview transcripts; they live in each sub-agent's own transcript, addressable by name, the same as every other agent this session has dispatched tonight. + +## Open items for the implementation plan, not resolved by this design + +- Exact dispatch order and parallelism: M3/M4 cells share a blast zone only with each other (separate worktrees avoid file conflicts entirely, unlike tonight's Hawkins-plan builder contracts, which shared cumulative-model files) — likely safe to run fully parallel, but should be confirmed rather than assumed. +- Whether M2's sub-agents test against the *current dev-server preview* (fast, available now) or wait for the CI/CD plan's real static build under `/toaster` (more representative of the actual deployed experience, but blocked on that plan's own Task 2/6) — a sequencing decision between this design and the CI/CD plan, not resolved here. +- The exact DL-entry numbering and whether a grid-testing battery gets its own `decisions/user-testing-grid/` directory (mirroring the existing `decisions/user-testing/` convention) or reuses it with a modality suffix on each filename. +- Whether `simulated-learner`'s own agent definition needs a modality parameter added to its work contract (currently only `(persona, chapter)`), or whether modality is better expressed as an entirely separate contract field the orchestrator fills in per cell. +- **Resolved by the first run (DL-085):** the stale-worktree-base risk named above as a reason to confirm rather than assume was real — 4 of 6 worktree-isolated cells (M3-novice, M3-practitioner, M3-returning, M4-practitioner) hit it, and one (M3-novice) did not self-diagnose and reported a false NEEDS-FIX as a result, traced to `.claude/skills/user-testing/SKILL.md`'s own hardcoded `cd` path (since fixed). The sandboxing section above and every future M3/M4 contract now require an explicit `Base check:` report line; do not run the next grid without it. +- **New from the first run:** the ACE's own "interview" step cannot reach a sibling grid cell directly (`SendMessage` to a cell by name fails from the ACE's own subagent process) — route interview questions through the orchestrator instead (name the cell and the question in the ACE's report; the orchestrator forwards and resumes the ACE with the answer), per `.claude/agents/orchestrator.md` line 12's own escalation chain. Confirm whether a future run should instead pass cells' raw agent ids into the ACE's own contract so it can address them without this relay. diff --git a/exercises/ch02/exercise.ipynb b/exercises/ch02/exercise.ipynb index e95b34e..4a252e8 100644 --- a/exercises/ch02/exercise.ipynb +++ b/exercises/ch02/exercise.ipynb @@ -54,7 +54,7 @@ "metadata": {}, "outputs": [], "execution_count": null, - "source": "# Step 3: write the asserted_context record.\ncontext_record = ReviewRecord(\n identifier=\"AC-C02\",\n kind=\"asserted_context\",\n claim=\"# The illustrative nominal brew temperature assumed (92 degrees Celsius / 365.15 K), and the model element it's an assumption about\",\n model_ref=\"# Qualified name of the nominal usage\",\n content_hash=hash_content(source_with_variants),\n scope=\"# Your package name\",\n criteria=\"# The illustrative range the estimate must fall inside, so a reader can judge whether it's appropriate for what it's used for\",\n premises=[],\n assumption_refs=[\"# The one assumption this claim rests on directly\"],\n evidence_refs=[\"# State plainly: no real evidence exists behind this number, only the stated assumption; BrewUnit::brewTemp carries no value for nominal\"],\n rationale=\"# Why this temperature is appropriate\",\n counterevidence=\"# Conditions where this assumption breaks down\",\n residual_uncertainties=\"# What is not modeled\",\n disposition=\"pending\",\n dependency_freshness=\"current\",\n engineering_conclusion=\"undetermined\",\n record_kind=\"worked_example\",\n)\n\nerrors = validate_record(context_record)\nprint(f\"Validation errors: {errors}\")\nconn.close()" + "source": "# Step 3: write the asserted_context record.\ncontext_record = ReviewRecord(\n identifier=\"AC-C02\",\n kind=\"asserted_context\",\n claim=\"# The illustrative nominal brew temperature assumed (92 degrees Celsius / 365.15 K), and the model element it's an assumption about\",\n subject_ref=\"# Qualified name of the model element this assumption is directly about\",\n model_ref=\"# Qualified name of the nominal usage\",\n content_hash=hash_content(source_with_variants),\n scope=\"# Your package name\",\n criteria=\"# The illustrative range the estimate must fall inside, so a reader can judge whether it's appropriate for what it's used for\",\n premises=[],\n assumption_refs=[\"# The one assumption this claim rests on directly\"],\n evidence_refs=[\"# State plainly: no real evidence exists behind this number, only the stated assumption; BrewUnit::brewTemp carries no value for nominal\"],\n rationale=\"# Why this temperature is appropriate\",\n counterevidence=\"# Conditions where this assumption breaks down\",\n residual_uncertainties=\"# What is not modeled\",\n disposition=\"pending\",\n dependency_freshness=\"current\",\n engineering_conclusion=\"undetermined\",\n record_kind=\"worked_example\",\n)\n\nerrors = validate_record(context_record)\nprint(f\"Validation errors: {errors}\")\nconn.close()" } ] } \ No newline at end of file diff --git a/exercises/ch03/exercise.ipynb b/exercises/ch03/exercise.ipynb index 97d3f4a..f4717ac 100644 --- a/exercises/ch03/exercise.ipynb +++ b/exercises/ch03/exercise.ipynb @@ -92,6 +92,7 @@ " identifier=\"AC-C03-EX\",\n", " kind=\"asserted_context\",\n", " claim=\"# What tempCheck is framed as: a measure of effectiveness (acceptance) or a measure of performance (derived engineering figure), and the model element the framing applies to\",\n", + " subject_ref=\"CoffeeDemo::tempCheck\",\n", " model_ref=\"# Qualified name of the tempCheck requirement usage\",\n", " content_hash=hash_content(source),\n", " scope=\"# Your package name\",\n", @@ -136,6 +137,7 @@ " identifier=\"AS-C03-EX\",\n", " kind=\"asserted_solution\",\n", " claim=\"# The negated claim on hot evaluates True, and no corresponding claim is made for nominal\",\n", + " subject_ref=\"CoffeeDemo::tempCheck\",\n", " model_ref=\"# Qualified name of the tempCheck requirement usage\",\n", " content_hash=hash_content(source),\n", " scope=\"# Your package name\",\n", diff --git a/exercises/ch04/exercise.ipynb b/exercises/ch04/exercise.ipynb index 8fff789..ffe4a1b 100644 --- a/exercises/ch04/exercise.ipynb +++ b/exercises/ch04/exercise.ipynb @@ -185,6 +185,7 @@ " identifier=\"AI-C04-EX\",\n", " kind=\"asserted_inference\",\n", " claim=\"# ApplyWater's typed flows account for beans, water and duration in, and coffee, delivered and retained out; the asserted balance constraint (delivered and retained each non-negative, their sum bounded by water) is a real, evaluable relation, not merely declared syntax. Scope this to ApplyWater alone -- not to whether brewing as a whole is decomposed.\",\n", + " subject_ref=\"CoffeeDemo::ApplyWater\",\n", " model_ref=\"CoffeeDemo::ApplyWater\",\n", " content_hash=hash_content(source_with_items),\n", " scope=\"CoffeeDemo::ApplyWater, nested inside CoffeeDemo::Brew\",\n", diff --git a/exercises/ch06/exercise.ipynb b/exercises/ch06/exercise.ipynb index acc0622..4d92b1f 100644 --- a/exercises/ch06/exercise.ipynb +++ b/exercises/ch06/exercise.ipynb @@ -5,7 +5,7 @@ "id": "a550d9b0", "metadata": {}, "source": [ - "# Chapter 6 Exercise — Coffee Maker Recursive Decomposition\n", + "# Chapter 6 Exercise \u2014 Coffee Maker Recursive Decomposition\n", "\n", "Decompose `BrewUnit`'s own `applyWater` step one level deeper, then record\n", "the two judgment sites that decomposition raises and a stopping judgment.\n", @@ -34,7 +34,7 @@ " moveWater : MoveWater;`, a port (reuse Chapter 5's own `WaterPort`,\n", " conjugated: `port waterIn : ~WaterPort;`, mirroring `HeatGenerator`'s own\n", " `~EnergyPort`-style port), and an unbound `attribute throughput :\n", - " ISQ::MassFlowRateValue;` — no default, the same uncommitted design-space\n", + " ISQ::MassFlowRateValue;` \u2014 no default, the same uncommitted design-space\n", " slot `HeatGenerator::power` is.\n", "2. Add `part def BrewAssembly :> BrewUnit`, mirroring `HeatingAssembly :>\n", " HeatingSystem`. Unlike `HeatingSystem`, `BrewUnit` (Chapter 1) was never\n", @@ -43,9 +43,9 @@ " WaterMover;` and nests the usage-level allocation `allocation\n", " moverAllocation allocate applyWater.moveWater to mover;`, mirroring\n", " `heatGenAllocation` exactly. Also add `requirement def BrewReq` with\n", - " `subject wm : WaterMover;` (not `BrewUnit` — mirroring\n", + " `subject wm : WaterMover;` (not `BrewUnit` \u2014 mirroring\n", " `HeatGenerationReq`'s subject being `HeatGenerator`, not `HeatingSystem`)\n", - " and `require constraint { wm.throughput >= 0.003 ['kg⋅s⁻¹'] }`, plus a\n", + " and `require constraint { wm.throughput >= 0.003 ['kg\u22c5s\u207b\u00b9'] }`, plus a\n", " `requirement brewReq : BrewReq;` usage. Give `BrewReq` a `doc` stating a\n", " component throughput rating threshold and naming no stakeholder\n", " acceptance criterion, mirroring `HeatGenerationReq`'s own doc exactly\n", @@ -55,7 +55,7 @@ " itself stays open for you to argue.\n", "3. Write an `asserted_context` record `AC-C06-EX` framing what kind of\n", " measure `brewReq` constrains: a measure of effectiveness or a measure\n", - " of performance — mirroring `AC-C06`'s own framing of `heatGenerationReq`\n", + " of performance \u2014 mirroring `AC-C06`'s own framing of `heatGenerationReq`\n", " in main Chapter 6 notebook 02. This is a genuine judgment call for you to\n", " make and justify: is the throughput threshold a component-level\n", " engineering rating, or a user-facing acceptance criterion? The record's\n", @@ -63,16 +63,16 @@ " reasoning, not a restated answer copied from this prompt.\n", "4. Write an `asserted_solution` record `AS-C06-EX`: which mechanism\n", " `WaterMover` commits to, and which plausible alternative it is selected\n", - " over — mirroring `AS-C06`'s own mechanism-selection reasoning (a\n", + " over \u2014 mirroring `AS-C06`'s own mechanism-selection reasoning (a\n", " confirmed model fact, e.g. `BrewStart`/`BrewCancel` resolving via\n", " `model.find()`, plus a domain premise about how the two mechanisms\n", " respond to a discrete control signal, not a claim the model already\n", " connects one and not the other). Only *after* this record is on file,\n", " build the concrete realization it licenses: `part def Impeller :>\n", " WaterMover`, mirroring `ResistanceCoil :> HeatGenerator` being built only\n", - " after `AS-C06` argues for it — its own `doc`, if you keep one, should not\n", + " after `AS-C06` argues for it \u2014 its own `doc`, if you keep one, should not\n", " give away this record's conclusion ahead of the record itself. Bind\n", - " `attribute :>> throughput default = ...;` (keep `default` — dropping it\n", + " `attribute :>> throughput default = ...;` (keep `default` \u2014 dropping it\n", " means `weak`'s own override below cannot bind a second time). Add two\n", " `Impeller` usages, `rated` and `weak`, mirroring `rated`/`weak :\n", " ResistanceCoil`: one passing `brewReq` at `Impeller`'s own default, one\n", @@ -80,12 +80,12 @@ " weak;` folded into its own context.\n", "5. Write an `asserted_inference` record `AI-C06-EX`, honestly scoped\n", " exactly like `AI-C06`: state plainly what this branch's structural\n", - " decomposition establishes and what it does not — **not** a claim that\n", + " decomposition establishes and what it does not \u2014 **not** a claim that\n", " the `BrewUnit` decomposition is complete. Cite `perform_relationships`,\n", " `find_allocations`, and `brewReq(rated)`/`brewReq(weak)` directly in\n", " `evidence_refs`. Set `premises` to `[\"AC-C06-EX\", \"AS-C06-EX\",\n", " \"AS-C03-EX\", \"AI-C04-EX\"]`, the coffee-domain analogues of `AI-C06`'s own\n", - " four premises — make sure your own `rationale` actually uses each one,\n", + " four premises \u2014 make sure your own `rationale` actually uses each one,\n", " not just lists it.\n", "\n", "Verify with `model.ok == True` and `validate_record(...) == []` for all\n", @@ -142,7 +142,11 @@ "\n", "part_defs = [e.as_dict() for e in model2.query()\n", " if e.as_dict().get(\"@type\") == \"PartDefinition\"]\n", - "print(f\"Part definitions: {[p.get('declaredName') or p.get('name') for p in part_defs]}\")\n" + "print(f\"Part definitions: {[p.get('declaredName') or p.get('name') for p in part_defs]}\")\n", + "\n", + "# Save this exact value of source_with_decomp somewhere in your own fork --\n", + "# Chapter 10's exercise needs it back, byte-for-byte, to reconstruct AS-C06-EX\n", + "# (docs/setup.md, \"Keep your model between chapters\").\n" ] }, { @@ -159,6 +163,7 @@ " identifier=\"AC-C06-EX\",\n", " kind=\"asserted_context\",\n", " claim=\"# What brewReq is framed as: a measure of effectiveness (acceptance) or a measure of performance (derived engineering figure), and the model element the framing applies to\",\n", + " subject_ref=\"CoffeeDemo::brewReq\",\n", " model_ref=\"CoffeeDemo::brewReq\",\n", " content_hash=hash_content(source_with_decomp),\n", " scope=\"CoffeeDemo\",\n", @@ -199,6 +204,7 @@ " identifier=\"AS-C06-EX\",\n", " kind=\"asserted_solution\",\n", " claim=\"# Which mechanism WaterMover commits to, and which plausible alternative it is selected over\",\n", + " subject_ref=\"CoffeeDemo::WaterMover\",\n", " model_ref=\"CoffeeDemo::WaterMover\",\n", " content_hash=hash_content(source_with_decomp),\n", " scope=\"CoffeeDemo::WaterMover and its realizations\",\n", @@ -254,7 +260,11 @@ "model3 = conn.load_from_content(source_with_impeller, strict=False)\n", "print(f\"Model with Impeller ok: {model3.ok}\")\n", "if not model3.ok:\n", - " print(format_diagnostics(model3.diagnostics))\n" + " print(format_diagnostics(model3.diagnostics))\n", + "\n", + "# Save this exact value of source_with_impeller too -- a SECOND, separate\n", + "# snapshot Chapter 10's exercise needs back, for AI-C06-EX and everything that\n", + "# carries forward into Chapter 7 (docs/setup.md, \"Keep your model between chapters\").\n" ] }, { @@ -280,6 +290,7 @@ " identifier=\"AI-C06-EX\",\n", " kind=\"asserted_inference\",\n", " claim=\"# State plainly what this branch's structural decomposition establishes (MoveWater performed and allocated, brewReq evaluated on two real candidates) and what it does not (BrewUnit's decomposition is NOT complete)\",\n", + " subject_ref=\"CoffeeDemo::BrewAssembly::mover\",\n", " model_ref=\"CoffeeDemo::BrewAssembly::mover\",\n", " content_hash=hash_content(source_with_impeller),\n", " scope=\"CoffeeDemo::BrewAssembly::mover and its realizations\",\n", diff --git a/exercises/ch08/exercise.ipynb b/exercises/ch08/exercise.ipynb index 8dff0f3..6da26fb 100644 --- a/exercises/ch08/exercise.ipynb +++ b/exercises/ch08/exercise.ipynb @@ -5,7 +5,7 @@ "id": "cell-00", "metadata": {}, "source": [ - "# Chapter 8 Exercise — Coffee Maker Constraint Checking\n", + "# Chapter 8 Exercise \u2014 Coffee Maker Constraint Checking\n", "\n", "Prove a hand-restated real-arithmetic lemma for the coffee maker model with\n", "Z3, report your model's own real satisfaction claims, and show a\n", @@ -20,7 +20,7 @@ "`Impeller :> WaterMover` with `rated`/`weak` usages. This chapter asks\n", "whether `deliveredMass` ever exceeds what `throughput` and `duration` alone\n", "would supply, for every value `transferEfficiency`, `throughput` and\n", - "`duration` could take — not just at `rated`'s own fixed value. Work through\n", + "`duration` could take \u2014 not just at `rated`'s own fixed value. Work through\n", "these steps:\n", "\n", "1. Add a fresh, unbound `part moverCheck : WaterMover;` (mirroring\n", @@ -39,19 +39,19 @@ " `moverCheck.throughput` / `moverCheckDuration` for\n", " `heatGenCheck.efficiency` / `heatGenCheck.power` /\n", " `heatGenCheckDuration`, and the mass-flow units\n", - " (`ISQ::MassFlowRateValue`, `SI::'kg⋅s⁻¹'`) for the power/energy units.\n", + " (`ISQ::MassFlowRateValue`, `SI::'kg\u22c5s\u207b\u00b9'`) for the power/energy units.\n", " Confirm with `model.find()` and `model.query()` that the new constraint\n", " is really in the loaded model, not only in the string you wrote.\n", "\n", "2. Run `model.verify_satisfaction()` against your coffee model's own\n", " existing `assert satisfy` / `assert not satisfy` claims (built up across\n", - " Chapters 1-7's own exercises) and report what you actually find — do not\n", + " Chapters 1-7's own exercises) and report what you actually find \u2014 do not\n", " assume it matches the toaster's own four verdicts; it may or may not.\n", " Then prove `deliveredMassBoundedBySupply` for *every* value of its\n", " unbound features with `toaster.modelcheck.verify_holds()`, via a\n", " companion file (mirroring\n", " `chapters/ch08-checking/02-violation-witness.ipynb`'s own\n", - " `COMPANION_POSITIVE` / `_write_companion` pattern exactly — a fixed\n", + " `COMPANION_POSITIVE` / `_write_companion` pattern exactly \u2014 a fixed\n", " `SCRATCH_DIR = Path(\"companion-check-scratch\")`, used instead of the\n", " committed model directly because `toaster.modelcheck`'s own text parser\n", " cannot yet read a verdict line for a constraint that is also the subject\n", @@ -65,8 +65,8 @@ " further companion declaring `transferEfficiencyBounded` (the real\n", " `[0,1]` bound, unchanged) alongside a second, sibling `assert\n", " constraint reliesOnSibling`, whose own antecedent does *not* restate\n", - " that bound at all — it bounds only `moverCheck.throughput` and\n", - " `moverCheckDuration` as non-negative — and whose conclusion is the same\n", + " that bound at all \u2014 it bounds only `moverCheck.throughput` and\n", + " `moverCheckDuration` as non-negative \u2014 and whose conclusion is the same\n", " `throughput * duration * transferEfficiency <= throughput * duration`\n", " shape `deliveredMassBoundedBySupply` states. If the solver composed the\n", " two separately-declared constraints, `transferEfficiencyBounded` would\n", @@ -75,13 +75,13 @@ " verdicts: `transferEfficiencyBounded` itself `satisfied` (it is, on its\n", " own, a real, narrow bound), and `reliesOnSibling` `undecided`, with a\n", " genuine Z3 witness where `transferEfficiency` takes *some* value\n", - " `transferEfficiencyBounded` itself rules out — the solver never\n", + " `transferEfficiencyBounded` itself rules out \u2014 the solver never\n", " actually applied the sibling's own stated bound when checking\n", " `reliesOnSibling`, regardless of which specific out-of-range value Z3\n", " happens to pick. Build a `ReviewRecord` (`AS-C08-EX`,\n", " `kind=\"asserted_solution\"`) mirroring `AS-C08`'s own field-by-field\n", " content, whose `counterevidence` cites the same D-029/D-030/D-031 gaps\n", - " real Chapter 8 documents — confirmed directly in the cell above: a\n", + " real Chapter 8 documents \u2014 confirmed directly in the cell above: a\n", " sibling constraint that relies entirely on `transferEfficiencyBounded`'s\n", " own stated bound, without restating it, is left `undecided` with a\n", " witness violating that very bound, because the solver never composes\n", @@ -89,11 +89,11 @@ "\n", "3. Open with the same negative control notebook 03 opens with: an\n", " empty-identifier record failing `validate_record()`, distinct from\n", - " staleness. Then reuse `AS-C08-EX` (built in Step 2's record cell —\n", + " staleness. Then reuse `AS-C08-EX` (built in Step 2's record cell \u2014\n", " notebook 03 rebuilds `AS-C08` fresh from its own named parts, but\n", " nothing here differs from a straight reuse), confirm `check_stale()`\n", " reports it current against your coffee model's own source, then loosen\n", - " `deliveredMassBoundedBySupply`'s own bound (the same `<= 1.0` → `<= 1.2`\n", + " `deliveredMassBoundedBySupply`'s own bound (the same `<= 1.0` \u2192 `<= 1.2`\n", " edit) and show the record goes stale.\n", "\n", "Verify with `model.ok == True`; the new constraint confirmed by both\n", @@ -129,9 +129,7 @@ "\"\"\"\n", "\n", "model = conn.load_from_content(source, strict=False)\n", - "print(f\"Model ok: {model.ok}\")\n", - "if not model.ok:\n", - " print(format_diagnostics(model.diagnostics))\n" + "assert model.ok, f\"Model failed: {format_diagnostics(model.diagnostics)}\"\n" ] }, { @@ -140,7 +138,7 @@ "metadata": {}, "source": [ "Step 1 (continued): confirm the new construct is really part of the loaded\n", - "model, not only in the string above — mirroring notebook 01's own\n", + "model, not only in the string above \u2014 mirroring notebook 01's own\n", "confirmation cells exactly." ] }, @@ -171,7 +169,7 @@ "source": [ "Step 2: `verify_satisfaction()` evaluates every `assert satisfy` /\n", "`assert not satisfy` declaration your coffee model carries, each at the one\n", - "set of values its subject happens to have. Report what you actually find —\n", + "set of values its subject happens to have. Report what you actually find \u2014\n", "your model's own claims, built up across Chapters 1-7's own exercises, not\n", "a forced match to the toaster's own four." ] @@ -253,7 +251,7 @@ "\n", "def _ascii(reason: str) -> str:\n", " \"\"\"Swap the CLI's own em dash for a plain double hyphen. This domain's own\n", - " unit literals (e.g. `kg⋅s⁻¹`) carry other non-ASCII characters too, which\n", + " unit literals (e.g. `kg\u22c5s\u207b\u00b9`) carry other non-ASCII characters too, which\n", " this helper does not touch: only the em dash is replaced, not every\n", " non-ASCII character. Only the printed copy changes; the verdict objects\n", " themselves keep the CLI's own text.\"\"\"\n", @@ -288,7 +286,7 @@ "A proof is only worth trusting if the same machinery can also report a\n", "real violation. Restate the lemma's **full negation** below (`A and not B`,\n", "efficiency/throughput/duration bounded as before, **and** delivered mass\n", - "strictly *exceeds* supplied mass) — the one Z3 must actually resolve as\n", + "strictly *exceeds* supplied mass) \u2014 the one Z3 must actually resolve as\n", "unsatisfiable to report `violated`." ] }, @@ -321,8 +319,8 @@ "metadata": {}, "source": [ "A fully broken lemma is not the only interesting failure mode. Restate the\n", - "lemma below with its own bound loosened from `<= 1.0` to `<= 1.2` — the\n", - "same edit Step 3 uses to demonstrate staleness — and confirm `verify_holds()`\n", + "lemma below with its own bound loosened from `<= 1.0` to `<= 1.2` \u2014 the\n", + "same edit Step 3 uses to demonstrate staleness \u2014 and confirm `verify_holds()`\n", "reports `undecided` with a genuine witness, and that `holds()` refuses to\n", "collapse that into `True` or `False`." ] @@ -459,7 +457,7 @@ "was supposed to rule out. The witness itself satisfies `reliesOnSibling`\n", "only vacuously (its own stated non-negativity antecedent on `throughput`/\n", "`duration` can be left false by the same witness, the same way a false\n", - "antecedent trivially satisfies any `implies`) — what the witness actually\n", + "antecedent trivially satisfies any `implies`) \u2014 what the witness actually\n", "demonstrates is narrower and still decisive: Z3 was free to pick\n", "`transferEfficiency = 2` at all, which composition with\n", "`transferEfficiencyBounded` would have ruled out. The real evidence is the\n", @@ -479,7 +477,7 @@ "source": [ "The proof above is engineering evidence, not a passing test result.\n", "Build `AS-C08-EX` from the named parts below, mirroring `AS-C08`'s own\n", - "field-by-field content (Hawkins et al. 2011, §§3.1-3.4), substituting the\n", + "field-by-field content (Hawkins et al. 2011, \u00a7\u00a73.1-3.4), substituting the\n", "coffee-domain construct names and the verdicts you actually found." ] }, @@ -502,6 +500,7 @@ " \"value of transferEfficiency in [0,1] and every non-negative \"\n", " \"throughput and duration a companion restatement admits\"\n", " ),\n", + " subject_ref=\"CoffeeDemo::deliveredMassBoundedBySupply\",\n", " model_ref=\"CoffeeDemo::deliveredMassBoundedBySupply\",\n", " content_hash=hash_content(source),\n", " scope=(\n", @@ -567,7 +566,7 @@ "metadata": {}, "source": [ "Step 3: a record with an empty identifier fails validation outright,\n", - "distinct from staleness — it was never a valid record to begin with,\n", + "distinct from staleness \u2014 it was never a valid record to begin with,\n", "regardless of which model it references. The same negative control\n", "notebook 03 opens with." ] @@ -583,6 +582,7 @@ " identifier=\"\", # intentionally empty\n", " kind=\"asserted_solution\",\n", " claim=\"Some claim\",\n", + " subject_ref=\"CoffeeDemo::deliveredMassBoundedBySupply\",\n", " model_ref=\"CoffeeDemo\",\n", " content_hash=hash_content(source),\n", " scope=\"Some scope\",\n", diff --git a/exercises/ch09/exercise.ipynb b/exercises/ch09/exercise.ipynb index ae570ab..d0c7a3d 100644 --- a/exercises/ch09/exercise.ipynb +++ b/exercises/ch09/exercise.ipynb @@ -403,6 +403,7 @@ " identifier=\"AS-BAD-EX\",\n", " kind=\"asserted_solution\",\n", " claim=\"Some claim\",\n", + " subject_ref=\"CoffeeDemo::WaterMover\",\n", " model_ref=\"CoffeeDemo\",\n", " content_hash=hash_content(source),\n", " scope=\"Some scope\",\n", @@ -495,6 +496,7 @@ " identifier=\"AS-C06-EX\",\n", " kind=\"asserted_solution\",\n", " claim=\"# Which mechanism WaterMover commits to, and which plausible alternative it is selected over\",\n", + " subject_ref=\"CoffeeDemo::WaterMover\",\n", " model_ref=\"CoffeeDemo::WaterMover\",\n", " content_hash=hash_content(ch06_source),\n", " scope=\"CoffeeDemo::WaterMover and its realizations\",\n", @@ -565,6 +567,7 @@ " \"value of transferEfficiency in [0,1] and every non-negative \"\n", " \"throughput and duration a companion restatement admits\"\n", " ),\n", + " subject_ref=\"CoffeeDemo::deliveredMassBoundedBySupply\",\n", " model_ref=\"CoffeeDemo::deliveredMassBoundedBySupply\",\n", " content_hash=hash_content(source),\n", " scope=(\n", @@ -757,6 +760,7 @@ " identifier=\"AS-PLACEHOLDER-EX\",\n", " kind=\"asserted_solution\",\n", " claim=\"The chosen design meets its requirement.\",\n", + " subject_ref=\"CoffeeDemo::WaterMover\",\n", " model_ref=\"CoffeeDemo\",\n", " content_hash=hash_content(source),\n", " scope=\"CoffeeDemo\",\n", @@ -837,6 +841,7 @@ " identifier=\"\", # intentionally empty\n", " kind=\"asserted_solution\",\n", " claim=\"Some claim\",\n", + " subject_ref=\"CoffeeDemo::deliveredMassBoundedBySupply\",\n", " model_ref=\"CoffeeDemo\",\n", " content_hash=hash_content(source),\n", " scope=\"Some scope\",\n", diff --git a/exercises/ch10/exercise.ipynb b/exercises/ch10/exercise.ipynb index 0bd031e..5ed4b73 100644 --- a/exercises/ch10/exercise.ipynb +++ b/exercises/ch10/exercise.ipynb @@ -659,6 +659,7 @@ " \"tie to deliveredMassBoundedBySupply spec-legitimate, is its own general \"\n", " \"motivation genuine and independent of any one proof, and did its own \"\n", " \"need predate or postdate the gap Step 1 found? Say which, honestly.\",\n", + " subject_ref=\"CoffeeDemo::MassConservationReq\",\n", " model_ref=\"# Your own new requirement usage's qualified name, e.g. \"\n", " \"CoffeeDemo::massConservationReq\",\n", " content_hash=hash_content(remediated_source),\n", @@ -754,6 +755,7 @@ " identifier=\"AI-BAD-EX\",\n", " kind=\"asserted_inference\",\n", " claim=\"The judgment ledger is complete.\",\n", + " subject_ref=\"CoffeeDemo::BrewAssembly::mover\",\n", " model_ref=\"CoffeeDemo\",\n", " content_hash=hash_content(source),\n", " scope=\"CoffeeDemo\",\n", @@ -911,6 +913,7 @@ " identifier=\"AS-C06-EX\",\n", " kind=\"asserted_solution\",\n", " claim=\"# Which mechanism WaterMover commits to, and which plausible alternative it is selected over\",\n", + " subject_ref=\"CoffeeDemo::WaterMover\",\n", " model_ref=\"CoffeeDemo::WaterMover\",\n", " content_hash=hash_content(ch06_source_pre_impeller),\n", " scope=\"CoffeeDemo::WaterMover and its realizations\",\n", @@ -979,6 +982,7 @@ " \"value of transferEfficiency in [0,1] and every non-negative \"\n", " \"throughput and duration a companion restatement admits\"\n", " ),\n", + " subject_ref=\"CoffeeDemo::deliveredMassBoundedBySupply\",\n", " model_ref=\"CoffeeDemo::deliveredMassBoundedBySupply\",\n", " content_hash=hash_content(source),\n", " scope=(\n", @@ -1106,6 +1110,7 @@ " identifier=\"AI-C06-EX\",\n", " kind=\"asserted_inference\",\n", " claim=\"# State plainly what this branch's structural decomposition establishes (MoveWater performed and allocated, brewReq evaluated on two real candidates) and what it does not (BrewUnit's decomposition is NOT complete)\",\n", + " subject_ref=\"CoffeeDemo::BrewAssembly::mover\",\n", " model_ref=\"CoffeeDemo::BrewAssembly::mover\",\n", " content_hash=hash_content(ch06_source_final),\n", " scope=\"CoffeeDemo::BrewAssembly::mover and its realizations\",\n", diff --git a/figures/ch05-interconnection.svg b/figures/ch05-interconnection.svg index 50d01ce..9098c78 100644 --- a/figures/ch05-interconnection.svg +++ b/figures/ch05-interconnection.svg @@ -4,45 +4,45 @@ - + Toaster - -Toaster + +Toaster heating - -heating -:HeatingSystem + +heating +:HeatingSystem control - -control -:ControlSystem + +control +:ControlSystem control->heating - - -durationOut→durationIn + + +durationOut→durationIn - + -ToastBread::applyHeat - -ToastBread::applyHeat +toastBread.applyHeat + +toastBread.applyHeat - + -ToastBread::applyHeat->heating - - -allocate +toastBread.applyHeat->heating + + +allocate diff --git a/glossary/definitions/tutorial.ttl b/glossary/definitions/tutorial.ttl index c5b23d9..c45ecae 100644 --- a/glossary/definitions/tutorial.ttl +++ b/glossary/definitions/tutorial.ttl @@ -16,6 +16,16 @@ glid:def-tutorial--allocation gl:term glid:term-allocation ; gl:text "Assigning one element of the model to another so that the target takes responsibility for it: functions to logical components, and logical components to parts. SEBoK gives the idea (requirements assigned to the next level down), SysML v2 gives the checkable form (the allocate relationship between usages, which is what the model uses), and Douglas gives the story (functions grouped into components). They describe the same assigning from different angles." . +glid:def-tutorial--assurance-claim-point + a gl:Definition ; + gl:gloss "This tutorial's narrowed analog of an Assurance Claim Point: one checkable qualified name a judgment is about, without GSN's argument-graph apparatus." ; + gl:locator "src/toaster/evidence.py ReviewRecord.subject_ref; models/*.sysml ReviewRecordRef" ; + gl:refines glid:def-hawkins--assurance-claim-point ; + gl:source glid:src-tutorial ; + gl:status gl:proposed ; + gl:term glid:term-assurance-claim-point ; + gl:text "This tutorial's own narrowed analog of an Assurance Claim Point: a single, checkable qualified name a judgment record is directly about, anchored on both the Python side (subject_ref) and the model side (a ReviewRecordRef metadata tag), without importing GSN's argument-graph apparatus." . + glid:def-tutorial--behavior a gl:Definition ; gl:confirmedBy "Z" ; diff --git a/models/ch02-cumulative.sysml b/models/ch02-cumulative.sysml index c978d19..a8e6220 100644 --- a/models/ch02-cumulative.sysml +++ b/models/ch02-cumulative.sysml @@ -3,6 +3,10 @@ // Source: notebook cell-02 TOASTER_INCREMENT in chapter 2's construct-introducing notebooks. package ToasterDemo { + metadata def ReviewRecordRef { + attribute identifier : ScalarValues::String; + } + private import ScalarValues::*; private import SI::*; private import ISQ::*; @@ -41,6 +45,10 @@ package ToasterDemo { } part nominal : Toaster; + metadata ac001Tag : ReviewRecordRef about nominal { + identifier = "AC-001"; + } + part slow : Toaster { attribute :>> cycleTime = 200.0 [SI::s]; } diff --git a/models/ch03-cumulative.sysml b/models/ch03-cumulative.sysml index 91f0023..c409b66 100644 --- a/models/ch03-cumulative.sysml +++ b/models/ch03-cumulative.sysml @@ -3,6 +3,10 @@ // Source: notebook cell-02 TOASTER_INCREMENT in chapter 3's construct-introducing notebooks. package ToasterDemo { + metadata def ReviewRecordRef { + attribute identifier : ScalarValues::String; + } + private import ScalarValues::*; private import SI::*; private import ISQ::*; @@ -42,7 +46,19 @@ package ToasterDemo { requirement timely : TimelyToast; + metadata acC03Tag : ReviewRecordRef about timely { + identifier = "AC-C03"; + } + + metadata asC03Tag : ReviewRecordRef about timely { + identifier = "AS-C03"; + } + part nominal : Toaster; + metadata ac001Tag : ReviewRecordRef about nominal { + identifier = "AC-001"; + } + part slow : Toaster { attribute :>> cycleTime = 200.0 [SI::s]; assert not satisfy timely by slow; diff --git a/models/ch04-cumulative.sysml b/models/ch04-cumulative.sysml index 1d0070a..7a4d62e 100644 --- a/models/ch04-cumulative.sysml +++ b/models/ch04-cumulative.sysml @@ -3,6 +3,10 @@ // Source: notebook cell-02 TOASTER_INCREMENT in chapter 4's construct-introducing notebooks. package ToasterDemo { + metadata def ReviewRecordRef { + attribute identifier : ScalarValues::String; + } + private import ScalarValues::*; private import SI::*; private import ISQ::*; @@ -27,6 +31,10 @@ package ToasterDemo { } } + metadata aiC04Tag : ReviewRecordRef about ApplyHeat { + identifier = "AI-C04"; + } + action def ToastBread { doc /* Transform bread into toast acceptable to its user. */ in bread : Bread; @@ -64,7 +72,19 @@ package ToasterDemo { requirement timely : TimelyToast; + metadata acC03Tag : ReviewRecordRef about timely { + identifier = "AC-C03"; + } + + metadata asC03Tag : ReviewRecordRef about timely { + identifier = "AS-C03"; + } + part nominal : Toaster; + metadata ac001Tag : ReviewRecordRef about nominal { + identifier = "AC-001"; + } + part slow : Toaster { attribute :>> cycleTime = 200.0 [SI::s]; assert not satisfy timely by slow; diff --git a/models/ch05-cumulative.sysml b/models/ch05-cumulative.sysml index cbce7b3..63bb3ed 100644 --- a/models/ch05-cumulative.sysml +++ b/models/ch05-cumulative.sysml @@ -3,6 +3,10 @@ // Source: notebook cell-02 TOASTER_INCREMENT in chapter 5's construct-introducing notebooks. package ToasterDemo { + metadata def ReviewRecordRef { + attribute identifier : ScalarValues::String; + } + private import ScalarValues::*; private import SI::*; private import ISQ::*; @@ -27,6 +31,10 @@ package ToasterDemo { } } + metadata aiC04Tag : ReviewRecordRef about ApplyHeat { + identifier = "AI-C04"; + } + action def ToastBread { doc /* Transform bread into toast acceptable to its user. */ in bread : Bread; @@ -78,7 +86,19 @@ package ToasterDemo { requirement timely : TimelyToast; + metadata acC03Tag : ReviewRecordRef about timely { + identifier = "AC-C03"; + } + + metadata asC03Tag : ReviewRecordRef about timely { + identifier = "AS-C03"; + } + part nominal : Toaster; + metadata ac001Tag : ReviewRecordRef about nominal { + identifier = "AC-001"; + } + part slow : Toaster { attribute :>> cycleTime = 200.0 [SI::s]; assert not satisfy timely by slow; diff --git a/models/ch06-cumulative.sysml b/models/ch06-cumulative.sysml index f9c0e84..3b06c40 100644 --- a/models/ch06-cumulative.sysml +++ b/models/ch06-cumulative.sysml @@ -3,6 +3,10 @@ // Source: notebook cell-02 TOASTER_INCREMENT in chapter 6's construct-introducing notebooks. package ToasterDemo { + metadata def ReviewRecordRef { + attribute identifier : ScalarValues::String; + } + private import ScalarValues::*; private import SI::*; private import ISQ::*; @@ -33,6 +37,10 @@ package ToasterDemo { then done; } + metadata aiC04Tag : ReviewRecordRef about ApplyHeat { + identifier = "AI-C04"; + } + action def ToastBread { doc /* Transform bread into toast acceptable to its user. */ in bread : Bread; @@ -84,7 +92,19 @@ package ToasterDemo { requirement timely : TimelyToast; + metadata acC03Tag : ReviewRecordRef about timely { + identifier = "AC-C03"; + } + + metadata asC03Tag : ReviewRecordRef about timely { + identifier = "AS-C03"; + } + part nominal : Toaster; + metadata ac001Tag : ReviewRecordRef about nominal { + identifier = "AC-001"; + } + part slow : Toaster { attribute :>> cycleTime = 200.0 [SI::s]; assert not satisfy timely by slow; @@ -147,6 +167,10 @@ package ToasterDemo { allocation heatGenAllocation allocate applyHeat.generateHeat to heatGen; } + metadata aiC06Tag : ReviewRecordRef about HeatingAssembly::heatGen { + identifier = "AI-C06"; + } + requirement def HeatGenerationReq { doc /* * A heat generator shall be rated for at least 600 W. @@ -161,6 +185,10 @@ package ToasterDemo { requirement heatGenerationReq : HeatGenerationReq; + metadata acC06Tag : ReviewRecordRef about heatGenerationReq { + identifier = "AC-C06"; + } + part def ResistanceCoil :> HeatGenerator { doc /* An electrically switched resistive element: converts electrical * energy to heat by Joule heating. The mechanism selection this @@ -171,6 +199,10 @@ package ToasterDemo { attribute resistance : ISQ::ResistanceValue default = 12.0 [SI::ohm]; } + metadata asC06Tag : ReviewRecordRef about ResistanceCoil { + identifier = "AS-C06"; + } + part rated : ResistanceCoil { assert satisfy heatGenerationReq by rated; } diff --git a/models/ch07-cumulative.sysml b/models/ch07-cumulative.sysml index 84b5a2c..6780ec1 100644 --- a/models/ch07-cumulative.sysml +++ b/models/ch07-cumulative.sysml @@ -3,6 +3,10 @@ // Source: notebook cell-02 TOASTER_INCREMENT in chapter 7's construct-introducing notebooks. package ToasterDemo { + metadata def ReviewRecordRef { + attribute identifier : ScalarValues::String; + } + private import ScalarValues::*; private import SI::*; private import ISQ::*; @@ -34,6 +38,10 @@ package ToasterDemo { then done; } + metadata aiC04Tag : ReviewRecordRef about ApplyHeat { + identifier = "AI-C04"; + } + action def ToastBread { doc /* Transform bread into toast acceptable to its user. */ in bread : Bread; @@ -86,7 +94,19 @@ package ToasterDemo { requirement timely : TimelyToast; + metadata acC03Tag : ReviewRecordRef about timely { + identifier = "AC-C03"; + } + + metadata asC03Tag : ReviewRecordRef about timely { + identifier = "AS-C03"; + } + part nominal : Toaster; + metadata ac001Tag : ReviewRecordRef about nominal { + identifier = "AC-001"; + } + part slow : Toaster { attribute :>> cycleTime = 200.0 [SI::s]; assert not satisfy timely by slow; @@ -170,6 +190,10 @@ package ToasterDemo { allocation heatGenAllocation allocate applyHeat.generateHeat to heatGen; } + metadata aiC06Tag : ReviewRecordRef about HeatingAssembly::heatGen { + identifier = "AI-C06"; + } + requirement def HeatGenerationReq { doc /* * A heat generator shall be rated for at least 600 W. @@ -184,6 +208,10 @@ package ToasterDemo { requirement heatGenerationReq : HeatGenerationReq; + metadata acC06Tag : ReviewRecordRef about heatGenerationReq { + identifier = "AC-C06"; + } + part def ResistanceCoil :> HeatGenerator { doc /* An electrically switched resistive element: converts electrical * energy to heat by Joule heating. The mechanism selection this @@ -194,6 +222,10 @@ package ToasterDemo { attribute resistance : ISQ::ResistanceValue default = 12.0 [SI::ohm]; } + metadata asC06Tag : ReviewRecordRef about ResistanceCoil { + identifier = "AS-C06"; + } + part rated : ResistanceCoil { attribute :>> efficiency = 0.7 [MeasurementReferences::one]; assert satisfy heatGenerationReq by rated; diff --git a/models/ch08-cumulative.sysml b/models/ch08-cumulative.sysml index d369560..1d5e9bd 100644 --- a/models/ch08-cumulative.sysml +++ b/models/ch08-cumulative.sysml @@ -3,6 +3,10 @@ // Source: notebook cell-02 TOASTER_INCREMENT in chapter 8's construct-introducing notebooks. package ToasterDemo { + metadata def ReviewRecordRef { + attribute identifier : ScalarValues::String; + } + private import ScalarValues::*; private import SI::*; private import ISQ::*; @@ -34,6 +38,10 @@ package ToasterDemo { then done; } + metadata aiC04Tag : ReviewRecordRef about ApplyHeat { + identifier = "AI-C04"; + } + action def ToastBread { doc /* Transform bread into toast acceptable to its user. */ in bread : Bread; @@ -86,7 +94,19 @@ package ToasterDemo { requirement timely : TimelyToast; + metadata acC03Tag : ReviewRecordRef about timely { + identifier = "AC-C03"; + } + + metadata asC03Tag : ReviewRecordRef about timely { + identifier = "AS-C03"; + } + part nominal : Toaster; + metadata ac001Tag : ReviewRecordRef about nominal { + identifier = "AC-001"; + } + part slow : Toaster { attribute :>> cycleTime = 200.0 [SI::s]; assert not satisfy timely by slow; @@ -170,6 +190,10 @@ package ToasterDemo { allocation heatGenAllocation allocate applyHeat.generateHeat to heatGen; } + metadata aiC06Tag : ReviewRecordRef about HeatingAssembly::heatGen { + identifier = "AI-C06"; + } + requirement def HeatGenerationReq { doc /* * A heat generator shall be rated for at least 600 W. @@ -184,6 +208,10 @@ package ToasterDemo { requirement heatGenerationReq : HeatGenerationReq; + metadata acC06Tag : ReviewRecordRef about heatGenerationReq { + identifier = "AC-C06"; + } + part def ResistanceCoil :> HeatGenerator { doc /* An electrically switched resistive element: converts electrical * energy to heat by Joule heating. The mechanism selection this @@ -194,6 +222,10 @@ package ToasterDemo { attribute resistance : ISQ::ResistanceValue default = 12.0 [SI::ohm]; } + metadata asC06Tag : ReviewRecordRef about ResistanceCoil { + identifier = "AS-C06"; + } + part rated : ResistanceCoil { attribute :>> efficiency = 0.7 [MeasurementReferences::one]; assert satisfy heatGenerationReq by rated; @@ -253,4 +285,8 @@ package ToasterDemo { implies (heatGenCheck.power * heatGenCheckDuration * heatGenCheck.efficiency) <= heatGenCheck.power * heatGenCheckDuration } + + metadata asC08Tag : ReviewRecordRef about deliveredEnergyBoundedBySupply { + identifier = "AS-C08"; + } } diff --git a/models/ch10-cumulative.sysml b/models/ch10-cumulative.sysml index 8842760..e4dd4b2 100644 --- a/models/ch10-cumulative.sysml +++ b/models/ch10-cumulative.sysml @@ -43,6 +43,10 @@ // warranted). package ToasterDemo { + metadata def ReviewRecordRef { + attribute identifier : ScalarValues::String; + } + private import ScalarValues::*; private import SI::*; private import ISQ::*; @@ -74,6 +78,10 @@ package ToasterDemo { then done; } + metadata aiC04Tag : ReviewRecordRef about ApplyHeat { + identifier = "AI-C04"; + } + action def ToastBread { doc /* Transform bread into toast acceptable to its user. */ in bread : Bread; @@ -126,7 +134,19 @@ package ToasterDemo { requirement timely : TimelyToast; + metadata acC03Tag : ReviewRecordRef about timely { + identifier = "AC-C03"; + } + + metadata asC03Tag : ReviewRecordRef about timely { + identifier = "AS-C03"; + } + part nominal : Toaster; + metadata ac001Tag : ReviewRecordRef about nominal { + identifier = "AC-001"; + } + part slow : Toaster { attribute :>> cycleTime = 200.0 [SI::s]; assert not satisfy timely by slow; @@ -210,6 +230,10 @@ package ToasterDemo { allocation heatGenAllocation allocate applyHeat.generateHeat to heatGen; } + metadata aiC06Tag : ReviewRecordRef about HeatingAssembly::heatGen { + identifier = "AI-C06"; + } + requirement def HeatGenerationReq { doc /* * A heat generator shall be rated for at least 600 W. @@ -224,6 +248,10 @@ package ToasterDemo { requirement heatGenerationReq : HeatGenerationReq; + metadata acC06Tag : ReviewRecordRef about heatGenerationReq { + identifier = "AC-C06"; + } + part def ResistanceCoil :> HeatGenerator { doc /* An electrically switched resistive element: converts electrical * energy to heat by Joule heating. The mechanism selection this @@ -234,6 +262,10 @@ package ToasterDemo { attribute resistance : ISQ::ResistanceValue default = 12.0 [SI::ohm]; } + metadata asC06Tag : ReviewRecordRef about ResistanceCoil { + identifier = "AS-C06"; + } + part rated : ResistanceCoil { attribute :>> efficiency = 0.7 [MeasurementReferences::one]; assert satisfy heatGenerationReq by rated; @@ -294,6 +326,10 @@ package ToasterDemo { <= heatGenCheck.power * heatGenCheckDuration } + metadata asC08Tag : ReviewRecordRef about deliveredEnergyBoundedBySupply { + identifier = "AS-C08"; + } + requirement def EnergyConservationReq { doc /* A heat generator shall never deliver more energy than it is supplied: * energy conservation, a physical law any real heat generator design @@ -320,4 +356,8 @@ package ToasterDemo { } requirement energyConservationReq : EnergyConservationReq; + + metadata acC10Tag : ReviewRecordRef about EnergyConservationReq { + identifier = "AC-C10"; + } } diff --git a/myst.yml b/myst.yml index d63f758..5f1db75 100644 --- a/myst.yml +++ b/myst.yml @@ -86,6 +86,7 @@ project: - file: docs/glossary - file: docs/references - file: docs/contributor + - file: docs/case-studies/2026-09-30-energy-conservation-requirement-tie site: template: book-theme diff --git a/scripts/check_construction.py b/scripts/check_construction.py index 599327d..8b972fa 100644 --- a/scripts/check_construction.py +++ b/scripts/check_construction.py @@ -90,13 +90,25 @@ "part def Toaster { attribute cycleTime : ISQ::DurationValue default = 120.0 [SI::s]; }", ], }, + { + "path": "chapters/ch02-requirements/03-judgment-context.ipynb", + # ac001Tag's `about nominal` requires nominal (nb01) to be a named element + # in scope; ReviewRecordRef itself is introduced by this notebook's own + # fragment, so it is not stubbed. + "context_stubs": [ + "part nominal;", + ], + }, ], 3: [ { "path": "chapters/ch03-measures/01-moe-definition.ipynb", - # timely : TimelyToast requires the requirement def from Chapter 2 + # timely : TimelyToast requires the requirement def from Chapter 2; + # acC03Tag's `about timely` requires ReviewRecordRef (carried forward + # from Chapter 2) in scope. "context_stubs": [ "requirement def TimelyToast;", + "metadata def ReviewRecordRef { attribute identifier : ScalarValues::String; }", ], }, { @@ -109,6 +121,15 @@ "requirement timely : TimelyToast;", ], }, + { + "path": "chapters/ch03-measures/03-threshold-judgment.ipynb", + # asC03Tag's `about timely` requires timely (nb01) as a named element + # in scope, and ReviewRecordRef (carried forward from Chapter 2). + "context_stubs": [ + "requirement timely;", + "metadata def ReviewRecordRef { attribute identifier : ScalarValues::String; }", + ], + }, { "path": "chapters/ch03-measures/04-verification-case.ipynb", # verify timely requires timely : TimelyToast (req usage) and Toaster part def in scope @@ -134,6 +155,15 @@ "path": "chapters/ch04-functional-decomp/02-heating-refinement.ipynb", "context_stubs": [], }, + { + "path": "chapters/ch04-functional-decomp/03-completeness-check.ipynb", + # aiC04Tag's `about ApplyHeat` requires ApplyHeat (nb01) as a named + # element in scope, and ReviewRecordRef (carried forward from Chapter 2). + "context_stubs": [ + "action def ApplyHeat;", + "metadata def ReviewRecordRef { attribute identifier : ScalarValues::String; }", + ], + }, ], 5: [ { @@ -192,9 +222,23 @@ "path": "chapters/ch06-recursive-decomp/02-second-level.ipynb", # ResistanceCoil :> HeatGenerator (nb01); HeatGenerationReq's subject # is HeatGenerator, and rated/weak are typed by ResistanceCoil, this - # notebook's own fragment. + # notebook's own fragment. acC06Tag's `about heatGenerationReq` and + # asC06Tag's `about ResistanceCoil` are both declared within this + # notebook's own TOASTER_INCREMENT; only ReviewRecordRef itself + # (carried forward from Chapter 2) needs stubbing. "context_stubs": [ "abstract part def HeatGenerator { attribute power : ISQ::PowerValue; }", + "metadata def ReviewRecordRef { attribute identifier : ScalarValues::String; }", + ], + }, + { + "path": "chapters/ch06-recursive-decomp/03-stopping-judgment.ipynb", + # aiC06Tag's `about HeatingAssembly::heatGen` requires HeatingAssembly + # (nb01) with its nested heatGen feature in scope, and ReviewRecordRef + # (carried forward from Chapter 2). + "context_stubs": [ + "part def HeatingAssembly { part heatGen; }", + "metadata def ReviewRecordRef { attribute identifier : ScalarValues::String; }", ], }, ], @@ -236,12 +280,22 @@ "path": "chapters/ch08-checking/01-invariant-def.ipynb", # deliveredEnergyBoundedBySupply references a fresh usage of HeatGenerator # (Ch6/Ch7), stubbed here with just the two features (power, efficiency) the - # fragment itself reads; nb02 and nb03 introduce no new construct (analysis - # only), so chapter 8 has exactly one construct-introducing notebook. + # fragment itself reads. "context_stubs": [ "abstract part def HeatGenerator { attribute power : ISQ::PowerValue; attribute efficiency : DimensionOneValue; }", ], }, + { + "path": "chapters/ch08-checking/02-violation-witness.ipynb", + # asC08Tag's `about deliveredEnergyBoundedBySupply` requires that constraint + # (nb01) as a named element in scope, and ReviewRecordRef (carried forward + # from Chapter 2). nb03 introduces no new construct (analysis only), so + # chapter 8 has exactly two construct-introducing notebooks. + "context_stubs": [ + "constraint deliveredEnergyBoundedBySupply;", + "metadata def ReviewRecordRef { attribute identifier : ScalarValues::String; }", + ], + }, ], # PASS4-009 (Chapter 9, Coverage and Sufficiency): no construct-introducing # notebook. Every notebook queries models/ch08-cumulative.sysml directly @@ -274,8 +328,12 @@ # deliveredEnergyBoundedBySupply (Chapter 8); stubbed here with a # bare constraint of that name, since the fragment itself only # needs something to subset, not the lemma's own real body. + # acC10Tag (added alongside EnergyConservationReq) needs + # ReviewRecordRef itself in scope, the same stub every other + # ReviewRecordRef-tagging notebook's entry already carries. "context_stubs": [ "constraint deliveredEnergyBoundedBySupply;", + "metadata def ReviewRecordRef { attribute identifier : ScalarValues::String; }", ], }, ], diff --git a/src/toaster/evidence.py b/src/toaster/evidence.py index 1e2f2fb..93c4a18 100644 --- a/src/toaster/evidence.py +++ b/src/toaster/evidence.py @@ -2,7 +2,7 @@ import hashlib from dataclasses import dataclass, field -from typing import Literal +from typing import Any, Literal @dataclass @@ -10,10 +10,11 @@ class ReviewRecord: identifier: str kind: Literal["asserted_context", "asserted_inference", "asserted_solution"] claim: str - model_ref: str - content_hash: str - scope: str - criteria: str + subject_ref: str = "" + model_ref: str = "" + content_hash: str = "" + scope: str = "" + criteria: str = "" premises: list[str] = field(default_factory=list) assumption_refs: list[str] = field(default_factory=list) evidence_refs: list[str] = field(default_factory=list) @@ -31,8 +32,20 @@ def hash_content(s: str) -> str: return hashlib.sha256(s.encode()).hexdigest() -def validate_record(r: ReviewRecord) -> list[str]: - """Return a list of field-level validation errors (empty = valid). WP-6 stub.""" +def validate_record(r: ReviewRecord, model: Any | None = None) -> list[str]: + """Return a list of field-level validation errors (empty = valid). + + `subject_ref` is this tutorial's narrowed analog of Hawkins' Assurance Claim Point + (`toaster-review-protocol`, SysML v2 formal/2026-03-02 §7.27.2): the one model + element this specific judgment is about. Required for `asserted_context` and + `asserted_solution`; for `asserted_inference` it may stay empty only when `premises` + is non-empty (the pure cross-record synthesis case, e.g. `AI-C10`). + + When `model` is given and `subject_ref` is set, two further checks run: that + `subject_ref` resolves in the model, and that any `ReviewRecordRef` metadata tag + already present for this record's own `identifier` agrees with `subject_ref` (catching + drift between the Python record and the model tag if they are ever edited independently). + """ errors = [] if not r.identifier: errors.append("identifier is empty") @@ -46,6 +59,29 @@ def validate_record(r: ReviewRecord) -> list[str]: errors.append("record_kind must be 'worked_example' in this tutorial (SA-7)") if r.kind == "asserted_inference" and not r.premises: errors.append("asserted_inference requires at least one premise (Hawkins §3.1)") + + if not r.subject_ref: + if r.kind in ("asserted_context", "asserted_solution"): + errors.append( + f"subject_ref is empty (required for {r.kind})" + ) + elif r.kind == "asserted_inference" and not r.premises: + errors.append( + "subject_ref is empty (required for asserted_inference with no premises)" + ) + elif model is not None: + if model.find(r.subject_ref) is None: + errors.append(f"subject_ref {r.subject_ref!r} does not resolve in the model") + else: + from toaster.query import get_review_record_refs + + tags = {t["identifier"]: t["annotated_element"] for t in get_review_record_refs(model)} + tagged = tags.get(r.identifier) + if tagged is not None and tagged != r.subject_ref: + errors.append( + f"ReviewRecordRef tag for {r.identifier!r} is about {tagged!r}, " + f"but subject_ref is {r.subject_ref!r}" + ) return errors diff --git a/src/toaster/query.py b/src/toaster/query.py index db74b68..df5cdc9 100644 --- a/src/toaster/query.py +++ b/src/toaster/query.py @@ -96,6 +96,38 @@ def get_satisfy_relationships(model: Any) -> list[dict]: return ApiIndex(model).of_type("SatisfyRequirementUsage") +def get_review_record_refs(model: Any, index: ApiIndex | None = None) -> list[dict]: + """Every ``ReviewRecordRef`` metadata tag in the model: ``{tag, identifier, annotated_element}``. + + The model-to-Python direction of this tutorial's narrowed Assurance Claim Point anchor + (``toaster-review-protocol``, SysML v2 formal/2026-03-02 §7.27.2): given a loaded model, + find every judgment-record tag and the one subject it names, independent of any + notebook's own Python objects. Takes the first ``annotatedElement`` only, matching this + tutorial's one-tag-one-subject convention (a tag with more than one is a modeling error + this tutorial's own notebooks never produce, not a shape this helper tries to generalize). + """ + idx = index or ApiIndex(model) + out = [] + for e in idx.of_type("MetadataUsage"): + type_qns = {idx.qn(t) for t in e.get("type", [])} + if not any(qn and qn.endswith("::ReviewRecordRef") for qn in type_qns): + continue + annotated = [idx.qn(a) for a in e.get("annotatedElement", [])] + identifier_value = None + for member_ref in e.get("ownedMember", []): + member = idx.by_id.get(_ref(member_ref)) + if member and member.get("declaredName") == "identifier": + literal = idx.by_id.get(_ref(member.get("value"))) + if literal is not None: + identifier_value = literal.get("value") + out.append({ + "tag": e.get("qualifiedName"), + "identifier": identifier_value, + "annotated_element": annotated[0] if annotated else None, + }) + return out + + def satisfy_relationships(model: Any, index: ApiIndex | None = None) -> list[dict]: """``{id, requirement, subject, is_negated}`` for every satisfy (and verify) relationship. @@ -272,6 +304,12 @@ def requirement_coverage(model: Any, index: ApiIndex | None = None) -> list[dict an unnamed ``RequirementUsage`` too (the objective's own auto-synthesized wrapper, not a design requirement anyone would check coverage on), and including it would report a bare bookkeeping artifact as an uncovered requirement. + + ``satisfied_by``/``failed_by`` key on the ``satisfy`` relationship's own ``subject`` qualified + name, which is the top-level usage only: a chained feature subject (``assert not satisfy R by + a.b``) attributes to ``a.b``, never to ``a``, so a requirement claimed only via a feature chain + will not show up under the chain's own owning part (found by user-testing grid run M4-returning, + ``decisions/log.md`` DL-085(4); no chapter in this tutorial uses a chained subject today). """ idx = index or ApiIndex(model) positive: dict[str, list[str]] = defaultdict(list) diff --git a/tests/test_evidence.py b/tests/test_evidence.py new file mode 100644 index 0000000..526ef89 --- /dev/null +++ b/tests/test_evidence.py @@ -0,0 +1,111 @@ +import opensysml +import pytest + +from toaster.evidence import ReviewRecord, validate_record + + +def _base_kwargs(**overrides): + kwargs = { + "identifier": "RR-TEST", + "kind": "asserted_solution", + "claim": "Test claim.", + "model_ref": "models/test.sysml", + "content_hash": "deadbeef", + "scope": "test scope", + "criteria": "test criteria", + "rationale": "test rationale", + "counterevidence": "test counterevidence", + } + kwargs.update(overrides) + return kwargs + + +def test_subject_ref_defaults_empty(): + r = ReviewRecord(**_base_kwargs()) + assert r.subject_ref == "" + + +def test_subject_ref_required_for_asserted_context_without_model(): + r = ReviewRecord(**_base_kwargs(kind="asserted_context")) + errors = validate_record(r) + assert any("subject_ref" in e for e in errors) + + +def test_subject_ref_required_for_asserted_solution_without_model(): + r = ReviewRecord(**_base_kwargs(kind="asserted_solution")) + errors = validate_record(r) + assert any("subject_ref" in e for e in errors) + + +def test_subject_ref_may_be_empty_for_asserted_inference_with_premises(): + r = ReviewRecord(**_base_kwargs(kind="asserted_inference", premises=["some premise"])) + errors = validate_record(r) + assert not any("subject_ref" in e for e in errors) + + +def test_subject_ref_required_for_asserted_inference_without_premises(): + r = ReviewRecord(**_base_kwargs(kind="asserted_inference", premises=[])) + errors = validate_record(r) + assert any("subject_ref" in e for e in errors) + # the pre-existing premises rule must still also fire -- this is a real double-error case + assert any("premise" in e for e in errors) + + +def test_validate_record_without_model_arg_still_works(): + # Existing call sites across the repo call validate_record(record) with one argument. + r = ReviewRecord(**_base_kwargs(subject_ref="ToasterDemo::anything")) + errors = validate_record(r) # no model= passed + assert errors == [] + + +_FIXTURE = """ +package ToasterDemo { + part def Widget { + attribute flag : ScalarValues::Boolean; + } + part target : Widget; + + metadata def ReviewRecordRef { + attribute identifier : ScalarValues::String; + } + + metadata rrTag : ReviewRecordRef about target { + identifier = "RR-TEST"; + } +} +""" + + +@pytest.fixture +def loaded_model(): + conn = opensysml.connect(version="v0.9.0") + model = conn.load_from_content(_FIXTURE, strict=False) + assert model.ok, model.diagnostics + yield model + conn.close() + + +def test_subject_ref_resolution_check_passes_for_real_element(loaded_model): + r = ReviewRecord(**_base_kwargs(subject_ref="ToasterDemo::target")) + errors = validate_record(r, model=loaded_model) + assert errors == [] + + +def test_subject_ref_resolution_check_fails_for_unresolvable_element(loaded_model): + r = ReviewRecord(**_base_kwargs(subject_ref="ToasterDemo::doesNotExist")) + errors = validate_record(r, model=loaded_model) + assert any("doesNotExist" in e for e in errors) + + +def test_cross_representation_check_passes_when_tag_matches(loaded_model): + r = ReviewRecord(**_base_kwargs(identifier="RR-TEST", subject_ref="ToasterDemo::target")) + errors = validate_record(r, model=loaded_model) + assert errors == [] + + +def test_cross_representation_check_fails_when_tag_disagrees(loaded_model): + # Same identifier as the model's own rrTag, but a DIFFERENT subject_ref -- this is the + # drift the cross-representation check exists to catch. + r = ReviewRecord(**_base_kwargs(identifier="RR-TEST", subject_ref="ToasterDemo::Widget")) + errors = validate_record(r, model=loaded_model) + assert any("RR-TEST" in e and "Widget" in e for e in errors) diff --git a/tests/test_query.py b/tests/test_query.py index aed4bec..f2990df 100644 --- a/tests/test_query.py +++ b/tests/test_query.py @@ -895,3 +895,46 @@ def test_requirement_ties_raises_on_unknown_target(ch10) -> None: query.requirement_ties(ch10, "ToasterDemo::NonexistentThing", idx) with pytest.raises(KeyError): query.tied_to_any_requirement(ch10, "ToasterDemo::NonexistentThing", idx) + + +_FIXTURE_ONE_TAG = """ +package ToasterDemo { + part def Widget { attribute flag : ScalarValues::Boolean; } + part target : Widget; + metadata def ReviewRecordRef { attribute identifier : ScalarValues::String; } + metadata rrTag : ReviewRecordRef about target { identifier = "RR-001"; } +} +""" + +_FIXTURE_NO_TAGS = """ +package ToasterDemo { + part def Widget { attribute flag : ScalarValues::Boolean; } + part target : Widget; +} +""" + + +def _load(source): + conn = opensysml.connect(version="v0.9.0") + model = conn.load_from_content(source, strict=False) + assert model.ok, model.diagnostics + return conn, model + + +def test_get_review_record_refs_finds_one_tag(): + conn, model = _load(_FIXTURE_ONE_TAG) + try: + refs = query.get_review_record_refs(model) + assert len(refs) == 1 + assert refs[0]["identifier"] == "RR-001" + assert refs[0]["annotated_element"] == "ToasterDemo::target" + finally: + conn.close() + + +def test_get_review_record_refs_empty_when_no_tags(): + conn, model = _load(_FIXTURE_NO_TAGS) + try: + assert query.get_review_record_refs(model) == [] + finally: + conn.close() diff --git a/tests/test_simulate.py b/tests/test_simulate.py index 4298b91..c896b54 100644 --- a/tests/test_simulate.py +++ b/tests/test_simulate.py @@ -57,6 +57,7 @@ def _make_record(source: str) -> ReviewRecord: identifier="TS-01", kind="asserted_solution", claim="X.v satisfies the threshold.", + subject_ref="T::X", model_ref="T::X", content_hash=hash_content(source), scope="T",