diff --git a/.github/APPROVED_CONTRIBUTORS b/.github/APPROVED_CONTRIBUTORS
new file mode 100644
index 00000000..37dc4fee
--- /dev/null
+++ b/.github/APPROVED_CONTRIBUTORS
@@ -0,0 +1,385 @@
+# GitHub handles approved to bypass contribution auto-close
+# Format:
+# capability:
+# issue future issues stay open
+# pr future issues and PRs stay open
+
+herrnel pr
+julien-c pr
+barapa pr
+alasano pr
+aadishv pr
+airtonix pr
+aliou pr
+aos pr
+austinm911 pr
+banteg pr
+ben-vargas pr
+butelo pr
+can1357 pr
+CarlosGtrz pr
+cau1k pr
+cmf pr
+crcatala pr
+Cursivez pr
+cv pr
+dannote pr
+default-anton pr
+dnouri pr
+DronNick pr
+enisdenjo pr
+ferologics pr
+fightbulc pr
+ghoulr pr
+gnattu pr
+HACKE-RC pr
+hewliyang pr
+hjanuschka pr
+iamd3vil pr
+jblwilliams pr
+joshp123 pr
+jsinge97 pr
+justram pr
+kaofelix pr
+kiliman pr
+kim0 pr
+lockmeister pr
+LukeFost pr
+lukele pr
+m-box-mr pr
+marckrenn pr
+markusylisiurunen pr
+mcinteerj pr
+melihmucuk pr
+mitsuhiko pr
+mrexodia pr
+nathyong pr
+nickseelert pr
+nicobailon pr
+ninlds pr
+ogulcancelik pr
+patrick-kidger pr
+paulbettner pr
+Perlence pr
+pjtf93 pr
+prateekmedia pr
+prathamdby pr
+ribelo pr
+richardgill pr
+robinwander pr
+ronyrus pr
+roshanasingh4 pr
+scutifer pr
+skuridin pr
+steipete pr
+svkozak pr
+tallshort pr
+theBucky pr
+thomasmhr pr
+tiagoefreitas pr
+timolins pr
+tmustier pr
+tudoroancea pr
+unexge pr
+vaayne pr
+VaclavSynacek pr
+vsabavat pr
+w-winter pr
+Whamp pr
+WismutHansen pr
+XesGaDeus pr
+yevhen pr
+badlogictest pr
+terrorobe pr
+zedrdave pr
+mrud pr
+toorusr pr
+andresaraujo pr
+lightningRalf pr
+williballenthin pr
+masonc15 pr
+4h9fbZ pr
+haoqixu pr
+Graffioh pr
+charles-cooper pr
+emanuelst pr
+juanibiapina pr
+liby pr
+pasky pr
+odysseus0 pr
+giuseppeg pr
+michaelpersonal pr
+academo pr
+PriNova pr
+semtexzv pr
+jasonish pr
+markusn pr
+SamFold pr
+Soleone pr
+virtuald pr
+NateSmyth pr
+7Sageer pr
+MatthieuBizien pr
+sumeet pr
+marchellodev pr
+vedang pr
+lucemia pr
+mcollina pr
+lajarre pr
+smithbm2316 pr
+drewburr pr
+gordonhwc pr
+deybhayden pr
+tintinweb pr
+asoules pr
+zhahaoyu pr
+in0vik pr
+jtac pr
+yzhg1983 pr
+smcllns pr
+dmmulroy pr
+zmberber pr
+andresvi94 pr
+sudosubin pr
+Mic92 pr
+pmateusz pr
+wirjo pr
+jay-aye-see-kay pr
+lucasmeijer pr
+Evizero pr
+
+ofa1 pr
+
+crisog issue
+
+mpazik pr
+
+vekexasia pr
+
+Michaelliv pr
+
+cmraible pr
+
+dljsjr pr
+
+drio pr
+
+jlaneve pr
+
+tantara pr
+
+Nutlope pr
+
+xl0 pr
+
+mdsjip pr
+
+Exrun94 pr
+
+marcbloech pr
+
+pidalf pr
+
+injaneity pr
+
+thirtythreeforty pr
+
+justinpbarnett pr
+
+cristinaponcela pr
+
+LooSik pr
+
+mchenco pr
+
+Phoen1xCode pr
+
+louis030195 pr
+
+technocidal pr
+
+pandada8 pr
+
+npupko issue
+
+chrisvariety pr
+
+maximilianzuern pr
+
+brianmichel pr
+
+abhinavmathur-atlan pr
+
+mattiacerutti pr
+
+josephyoung pr
+
+mbazso pr
+
+AJM10565 pr
+
+DanielThomas pr
+
+MichaelYochpaz pr
+
+stephanmck pr
+
+rolfvreijdenberger pr
+
+psoukie pr
+
+vastxie pr
+
+ItsumoSeito pr
+
+davidlifschitz pr
+
+vdxz pr
+
+dangooddd pr
+
+Mearman pr
+
+dodiego pr
+
+any-victor pr
+
+geraschenko pr
+
+skhoroshavin pr
+
+cyzlmh pr
+
+xz-dev pr
+
+rajp152k pr
+
+affanali2k3 pr
+
+ArcadiaLin pr
+
+anilgulecha pr
+
+DeviosLang pr
+
+HarrodRen pr
+
+aaronkyriesenbach pr
+
+farid-fari pr
+
+petrroll pr
+
+vibeinging pr
+
+DivineDominion pr
+
+ananthakumaran pr
+
+andrebreijao pr
+
+anh-chu pr
+
+rsaryev pr
+
+QuintinShaw pr
+
+R-Taneja pr
+
+zaycruz pr
+
+mteam88 pr
+
+christianbasch pr
+
+cpacker pr
+
+rgarcia pr
+
+renaudhartert-db pr
+
+HyeokjaeLee pr
+
+arajkumar pr
+
+brianstanley pr
+
+scruffymongrel pr
+
+SI-RUI-ZHANG pr
+
+sunnyyoung pr
+
+muyiyr pr
+
+hi-neason pr
+
+acmerfight pr
+
+tizmagik pr
+
+jingtao-wisdomgraph pr
+
+XRX193 pr
+
+autopeasant pr
+
+Snail-Turbo pr
+
+futile pr
+
+arasovic pr
+
+xXJSONDeruloXx pr
+
+Marvae pr
+
+skkdevcraft pr
+
+PierrunoYT pr
+
+zhichli pr
+
+wesleyzhangwq pr
+
+dgokeeffe pr
+
+vipentti pr
+
+midastruth pr
+
+Maximo-Guk pr
+
+bigoldcat123 pr
+
+johnatbasicas pr
+
+powerfooI pr
+
+yearth pr
+
+pablasso pr
+
+bilby91 pr
+
+giannisCKS pr
+
+Panoplos pr
+
+haoyongchun1125-maker pr
+
+gwokhou pr
+
+gaoyk19 pr
+
+cad0p pr
+
+Jaaneek pr
+
+CaiJichang212 pr
+
+Mallikarjun-0 pr
+
+wutongyuonce pr
+
+Terminator666666 pr
diff --git a/.github/ISSUE_TEMPLATE/bug.yml b/.github/ISSUE_TEMPLATE/bug.yml
index 763179da..0fd4964e 100644
--- a/.github/ISSUE_TEMPLATE/bug.yml
+++ b/.github/ISSUE_TEMPLATE/bug.yml
@@ -5,11 +5,13 @@ body:
- type: markdown
attributes:
value: |
- **Before you start:** Read [CONTRIBUTING.md](https://github.com/stepfun-ai/Step-Code/blob/main/CONTRIBUTING.md).
+ **Before you start:** Read [CONTRIBUTING.md](https://github.com/stepfun-ai/step-harness/blob/main/CONTRIBUTING.md).
+
+ New issues from new contributors are auto-closed by default. Maintainers review auto-closed issues daily. Issues that do not meet the quality bar in [CONTRIBUTING.md](https://github.com/stepfun-ai/step-harness/blob/main/CONTRIBUTING.md) will not be reopened or receive a reply.
Keep this short. If it doesn't fit on one screen, it's too long. Write in your own voice.
- **Important:** before reporting an issue in core, please validate first with `step -ne` that this is not caused by an extension you loaded.
+ **Important:** before reporting an issue in core, please validate first with `pi -ne` that this is not caused by an extension you loaded.
- type: textarea
id: description
@@ -38,6 +40,6 @@ body:
id: version
attributes:
label: Version
- description: e.g. 0.84.4
+ description: e.g. 0.49.0
validations:
required: false
diff --git a/.github/ISSUE_TEMPLATE/config.yml b/.github/ISSUE_TEMPLATE/config.yml
new file mode 100644
index 00000000..66b6798c
--- /dev/null
+++ b/.github/ISSUE_TEMPLATE/config.yml
@@ -0,0 +1,5 @@
+blank_issues_enabled: false
+contact_links:
+ - name: Questions
+ url: https://discord.com/invite/3cU7Bz4UPx
+ about: Ask questions on Discord instead of opening an issue
diff --git a/.github/ISSUE_TEMPLATE/contribution.yml b/.github/ISSUE_TEMPLATE/contribution.yml
index fcc3535b..94503b32 100644
--- a/.github/ISSUE_TEMPLATE/contribution.yml
+++ b/.github/ISSUE_TEMPLATE/contribution.yml
@@ -1,11 +1,13 @@
name: Contribution Proposal
-description: Propose a change or feature
+description: Propose a change or feature (required for new contributors before submitting a PR)
labels: []
body:
- type: markdown
attributes:
value: |
- **Before you start:** Read [CONTRIBUTING.md](https://github.com/stepfun-ai/Step-Code/blob/main/CONTRIBUTING.md).
+ **Before you start:** Read [CONTRIBUTING.md](https://github.com/stepfun-ai/step-harness/blob/main/CONTRIBUTING.md).
+
+ New issues from new contributors are auto-closed by default. Maintainers review auto-closed issues daily. Issues that do not meet the quality bar in [CONTRIBUTING.md](https://github.com/stepfun-ai/step-harness/blob/main/CONTRIBUTING.md) will not be reopened or receive a reply.
Keep this short. If it doesn't fit on one screen, it's too long. Write in your own voice.
diff --git a/.github/ISSUE_TEMPLATE/package-report.yml b/.github/ISSUE_TEMPLATE/package-report.yml
new file mode 100644
index 00000000..bf3d4c09
--- /dev/null
+++ b/.github/ISSUE_TEMPLATE/package-report.yml
@@ -0,0 +1,49 @@
+name: Package Report
+description: Report a problematic Step package
+labels: ["package-report"]
+body:
+ - type: markdown
+ attributes:
+ value: |
+ Use this form to report a Step package. For Step core bugs, use the bug report template instead.
+
+ New issues from new contributors are auto-closed by default. Maintainers review auto-closed issues daily. Issues that do not meet the quality bar in [CONTRIBUTING.md](https://github.com/stepfun-ai/step-harness/blob/main/CONTRIBUTING.md) will not be reopened or receive a reply.
+
+ Keep this short. If it doesn't fit on one screen, it's too long. Write in your own voice.
+
+ - type: input
+ id: package-name
+ attributes:
+ label: Package name
+ description: The npm package name.
+ placeholder: "@scope/package"
+ validations:
+ required: true
+
+ - type: input
+ id: package-version
+ attributes:
+ label: Version
+ description: The reported package version.
+ placeholder: "0.1.0"
+ validations:
+ required: false
+
+ - type: dropdown
+ id: report-type
+ attributes:
+ label: What are you reporting?
+ options:
+ - Malicious or unsafe behavior
+ - Impersonation
+ - Trademark / TOS Violations
+ validations:
+ required: true
+
+ - type: textarea
+ id: details
+ attributes:
+ label: Details
+ description: Describe the concern and include links, logs, or screenshots if helpful.
+ validations:
+ required: true
diff --git a/.github/workflows/approve-contributor.yml b/.github/workflows/approve-contributor.yml
new file mode 100644
index 00000000..debe87a7
--- /dev/null
+++ b/.github/workflows/approve-contributor.yml
@@ -0,0 +1,223 @@
+name: Approve Contributor
+
+on:
+ issue_comment:
+ types: [created]
+
+jobs:
+ approve:
+ runs-on: ubuntu-latest
+ permissions:
+ contents: write
+ issues: write
+ pull-requests: write
+ steps:
+ - name: Checkout
+ uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
+ with:
+ ref: ${{ github.event.repository.default_branch }}
+
+ - name: Update contributor approval
+ id: update
+ uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
+ with:
+ script: |
+ const fs = require('fs');
+
+ const APPROVED_FILE = '.github/APPROVED_CONTRIBUTORS';
+ const VALID_CAPABILITIES = new Set(['issue', 'pr']);
+ const issueAuthor = context.payload.issue.user.login;
+ const commenter = context.payload.comment.user.login;
+ const commentBody = (context.payload.comment.body || '').trim();
+
+ const approvalAtStartPattern = /^[\s.]*(?:@[A-Za-z0-9](?:[A-Za-z0-9-]{0,37}[A-Za-z0-9])?(?:\s*,\s*|[.:]\s*|\s+))*(lgtmi|lgtm)(?=$|[\s]|[^\p{L}\p{N}_\s])/iu;
+ const approvalAtEndPattern = /(?:^|[\s.])(lgtmi|lgtm)\s*(?:[^\p{L}\p{N}_\s])?\s*$/iu;
+ const approvalMatch = commentBody.match(approvalAtStartPattern) ?? commentBody.match(approvalAtEndPattern);
+
+ if (!approvalMatch) {
+ console.log('Comment does not start or end with lgtm or lgtmi');
+ core.setOutput('status', 'skipped');
+ return;
+ }
+
+ const targetCapability = approvalMatch[1].toLowerCase() === 'lgtmi' ? 'issue' : 'pr';
+
+ try {
+ const { data: permissionLevel } = await github.rest.repos.getCollaboratorPermissionLevel({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ username: commenter,
+ });
+
+ if (!['admin', 'maintain', 'write'].includes(permissionLevel.permission)) {
+ console.log(`${commenter} does not have write access`);
+ core.setOutput('status', 'skipped');
+ return;
+ }
+ } catch {
+ console.log(`${commenter} does not have collaborator access`);
+ core.setOutput('status', 'skipped');
+ return;
+ }
+
+ function parseMentionedUsers(body) {
+ const users = [];
+ const seenUsers = new Set();
+ const mentionPattern = /(^|[^A-Za-z0-9_])@([A-Za-z0-9](?:[A-Za-z0-9-]{0,37}[A-Za-z0-9])?)(?![A-Za-z0-9-]|\/)/g;
+
+ for (const match of body.matchAll(mentionPattern)) {
+ const username = match[2];
+ const normalizedUser = username.toLowerCase();
+ if (seenUsers.has(normalizedUser)) {
+ continue;
+ }
+ seenUsers.add(normalizedUser);
+ users.push(username);
+ }
+
+ return users;
+ }
+
+ function parseApprovedUsers(content) {
+ const lines = content.split('\n');
+ const entries = [];
+ const users = new Map();
+
+ for (const line of lines) {
+ const trimmed = line.trim();
+ if (!trimmed || trimmed.startsWith('#')) {
+ entries.push({ type: 'other', line });
+ continue;
+ }
+
+ const parts = trimmed.split(/\s+/);
+ if (parts.length !== 2) {
+ console.log(`Skipping malformed line: ${line}`);
+ entries.push({ type: 'other', line });
+ continue;
+ }
+
+ const [username, capability] = parts;
+ const normalizedCapability = capability.toLowerCase();
+ if (!VALID_CAPABILITIES.has(normalizedCapability)) {
+ console.log(`Skipping line with invalid capability: ${line}`);
+ entries.push({ type: 'other', line });
+ continue;
+ }
+
+ const normalizedUser = username.toLowerCase();
+ const entry = { type: 'user', username, normalizedUser, capability: normalizedCapability };
+ entries.push(entry);
+ users.set(normalizedUser, entry);
+ }
+
+ return { entries, users };
+ }
+
+ function stringifyApprovedUsers(entries) {
+ const normalizedEntries = [...entries];
+
+ while (normalizedEntries.length > 0) {
+ const lastEntry = normalizedEntries[normalizedEntries.length - 1];
+ if (lastEntry.type !== 'other' || lastEntry.line.trim() !== '') {
+ break;
+ }
+ normalizedEntries.pop();
+ }
+
+ return `${normalizedEntries
+ .map((entry) => (entry.type === 'user' ? `${entry.username} ${entry.capability}` : entry.line))
+ .join('\n')}\n`;
+ }
+
+ const content = fs.readFileSync(APPROVED_FILE, 'utf8');
+ const { entries, users } = parseApprovedUsers(content);
+ const mentionedUsers = parseMentionedUsers(commentBody);
+ const approvalTargets = mentionedUsers.length > 0 ? mentionedUsers : [issueAuthor];
+ const changedTargets = [];
+ const alreadyTargets = [];
+
+ for (const username of approvalTargets) {
+ const normalizedUser = username.toLowerCase();
+ const existingEntry = users.get(normalizedUser);
+ const existingCapability = existingEntry?.capability ?? null;
+
+ if (existingCapability === 'pr' || existingCapability === targetCapability) {
+ alreadyTargets.push(existingEntry?.username ?? username);
+ console.log(`${username} is already approved for ${existingCapability}`);
+ continue;
+ }
+
+ if (existingEntry) {
+ existingEntry.capability = targetCapability;
+ changedTargets.push(existingEntry.username);
+ } else {
+ const entry = { type: 'user', username, normalizedUser, capability: targetCapability };
+ entries.push(entry);
+ users.set(normalizedUser, entry);
+ changedTargets.push(username);
+ }
+
+ console.log(`Set ${username} capability to ${targetCapability}`);
+ }
+
+ core.setOutput('capability', targetCapability);
+ core.setOutput('changed_targets', JSON.stringify(changedTargets));
+ core.setOutput('already_targets', JSON.stringify(alreadyTargets));
+
+ if (changedTargets.length === 0) {
+ core.setOutput('status', 'already');
+ return;
+ }
+
+ fs.writeFileSync(APPROVED_FILE, stringifyApprovedUsers(entries));
+ core.setOutput('status', 'changed');
+
+ - name: Commit and push
+ if: steps.update.outputs.status == 'changed'
+ run: |
+ git config user.name "github-actions[bot]"
+ git config user.email "github-actions[bot]@users.noreply.github.com"
+ git add .github/APPROVED_CONTRIBUTORS
+ git diff --staged --quiet || git commit -m "chore: approve contributors from issue #${{ github.event.issue.number }}"
+ git push
+
+ - name: Comment on issue
+ if: steps.update.outputs.status == 'changed' || steps.update.outputs.status == 'already'
+ uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
+ env:
+ CAPABILITY: ${{ steps.update.outputs.capability }}
+ CHANGED_TARGETS: ${{ steps.update.outputs.changed_targets }}
+ ALREADY_TARGETS: ${{ steps.update.outputs.already_targets }}
+ with:
+ script: |
+ const capability = process.env.CAPABILITY;
+ const changedTargets = JSON.parse(process.env.CHANGED_TARGETS || '[]');
+ const alreadyTargets = JSON.parse(process.env.ALREADY_TARGETS || '[]');
+ const defaultBranch = context.payload.repository.default_branch;
+ const formatTargets = (targets) => targets.map((target) => `@${target}`).join(', ');
+ const bodyLines = [];
+
+ if (changedTargets.length > 0) {
+ if (capability === 'issue') {
+ bodyLines.push(`${formatTargets(changedTargets)} approved for issues. Future issues will not be auto-closed. PRs still require \`lgtm\` at the start of a maintainer reply (optionally after one or more \`@username\` mentions) or at the end.`);
+ } else {
+ bodyLines.push(`${formatTargets(changedTargets)} approved for issues and PRs. Future issues and PRs will not be auto-closed.`);
+ }
+ }
+
+ if (alreadyTargets.length > 0) {
+ const verb = alreadyTargets.length === 1 ? 'is' : 'are';
+ bodyLines.push(`${formatTargets(alreadyTargets)} ${verb} already approved.`);
+ }
+
+ bodyLines.push('', `See [CONTRIBUTING.md](https://github.com/${context.repo.owner}/${context.repo.repo}/blob/${defaultBranch}/CONTRIBUTING.md).`);
+ const body = bodyLines.join('\n');
+
+ await github.rest.issues.createComment({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ issue_number: context.issue.number,
+ body,
+ });
+
diff --git a/.github/workflows/issue-gate.yml b/.github/workflows/issue-gate.yml
new file mode 100644
index 00000000..03372aca
--- /dev/null
+++ b/.github/workflows/issue-gate.yml
@@ -0,0 +1,129 @@
+name: Issue Gate
+
+on:
+ issues:
+ types: [opened]
+
+jobs:
+ check-contributor:
+ runs-on: ubuntu-latest
+ permissions:
+ contents: read
+ issues: write
+ steps:
+ - name: Check issue author
+ uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
+ with:
+ script: |
+ const APPROVED_FILE = '.github/APPROVED_CONTRIBUTORS';
+ const VALID_CAPABILITIES = new Set(['issue', 'pr']);
+ const TRUSTED_BOT_AUTHORS = new Set(['dependabot[bot]', 'sentry[bot]', 'claude[bot]']);
+ const issueAuthor = context.payload.issue.user.login;
+ const defaultBranch = context.payload.repository.default_branch;
+ const isBotAuthor = issueAuthor.endsWith('[bot]');
+
+ if (TRUSTED_BOT_AUTHORS.has(issueAuthor)) {
+ console.log(`Skipping trusted bot: ${issueAuthor}`);
+ return;
+ }
+
+ async function getPermission(username) {
+ try {
+ const { data: permissionLevel } = await github.rest.repos.getCollaboratorPermissionLevel({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ username,
+ });
+ return permissionLevel.permission;
+ } catch {
+ return null;
+ }
+ }
+
+ async function getTextFile(path) {
+ const { data: fileContent } = await github.rest.repos.getContent({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ path,
+ ref: defaultBranch,
+ });
+
+ if (!('content' in fileContent) || typeof fileContent.content !== 'string') {
+ throw new Error(`Expected file content for ${path}`);
+ }
+
+ return Buffer.from(fileContent.content, 'base64').toString('utf8');
+ }
+
+ function parseApprovedUsers(content) {
+ const users = new Map();
+
+ for (const rawLine of content.split('\n')) {
+ const line = rawLine.trim();
+ if (!line || line.startsWith('#')) continue;
+
+ const parts = line.split(/\s+/);
+ if (parts.length !== 2) {
+ console.log(`Skipping malformed line: ${rawLine}`);
+ continue;
+ }
+
+ const [username, capability] = parts;
+ const normalizedCapability = capability.toLowerCase();
+ if (!VALID_CAPABILITIES.has(normalizedCapability)) {
+ console.log(`Skipping line with invalid capability: ${rawLine}`);
+ continue;
+ }
+
+ users.set(username.toLowerCase(), normalizedCapability);
+ }
+
+ return users;
+ }
+
+ const permission = await getPermission(issueAuthor);
+ if (!isBotAuthor && ['admin', 'maintain', 'write'].includes(permission)) {
+ console.log(`${issueAuthor} is a collaborator with ${permission} access`);
+ return;
+ }
+
+ const approvedContent = await getTextFile(APPROVED_FILE);
+ const approvedUsers = parseApprovedUsers(approvedContent);
+ const capability = approvedUsers.get(issueAuthor.toLowerCase());
+
+ if (!isBotAuthor && (capability === 'issue' || capability === 'pr')) {
+ console.log(`${issueAuthor} is approved for ${capability}`);
+ return;
+ }
+
+ const message = [
+ 'This issue was auto-closed. All issues from new contributors are auto-closed by default.',
+ '',
+ `Maintainers review auto-closed issues daily and reopen worthwhile ones. Issues that do not meet the quality bar in [CONTRIBUTING.md](https://github.com/${context.repo.owner}/${context.repo.repo}/blob/${defaultBranch}/CONTRIBUTING.md) will not be reopened or receive a reply.`,
+ '',
+ 'If a maintainer replies `lgtmi` on one of your issues, your future issues will stay open. If a maintainer replies `lgtm`, your future issues and PRs will stay open. The command must be at the start of the reply (optionally after one or more `@username` mentions) or at the end.',
+ '',
+ `See [CONTRIBUTING.md](https://github.com/${context.repo.owner}/${context.repo.repo}/blob/${defaultBranch}/CONTRIBUTING.md).`,
+ ].join('\n');
+
+ await github.rest.issues.createComment({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ issue_number: context.issue.number,
+ body: message,
+ });
+
+ await github.rest.issues.addLabels({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ issue_number: context.issue.number,
+ labels: ['untriaged'],
+ });
+
+ await github.rest.issues.update({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ issue_number: context.issue.number,
+ state: 'closed',
+ state_reason: 'not_planned',
+ });
diff --git a/.github/workflows/issue-triage-labels.yml b/.github/workflows/issue-triage-labels.yml
new file mode 100644
index 00000000..1a44253f
--- /dev/null
+++ b/.github/workflows/issue-triage-labels.yml
@@ -0,0 +1,142 @@
+name: Issue Triage Labels
+
+on:
+ issues:
+ types: [reopened, labeled]
+
+jobs:
+ update-labels:
+ runs-on: ubuntu-latest
+ permissions:
+ issues: write
+ steps:
+ - name: Update triage labels
+ uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
+ with:
+ script: |
+ const UNTRIAGED_LABEL = 'untriaged';
+ const NO_ACTION_LABEL = 'no-action';
+ const LAST_READ_LABEL = 'last-read';
+ const TO_DISCUSS_LABEL = 'to-discuss';
+ const INPROGRESS_LABEL = 'inprogress';
+
+ function issueHasLabel(issue, labelName) {
+ return (issue.labels ?? []).some((label) => label.name === labelName);
+ }
+
+ async function removeLabelIfPresent(issueNumber, issue, labelName) {
+ if (!issueHasLabel(issue, labelName)) {
+ console.log(`Issue #${issueNumber} does not have ${labelName}`);
+ return;
+ }
+
+ try {
+ await github.rest.issues.removeLabel({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ issue_number: issueNumber,
+ name: labelName,
+ });
+ console.log(`Removed ${labelName} from #${issueNumber}`);
+ } catch (error) {
+ if (error.status === 404) {
+ console.log(`Label ${labelName} was already absent from #${issueNumber}`);
+ return;
+ }
+ throw error;
+ }
+ }
+
+ if (context.payload.action === 'reopened') {
+ await removeLabelIfPresent(context.issue.number, context.payload.issue, UNTRIAGED_LABEL);
+ await removeLabelIfPresent(context.issue.number, context.payload.issue, NO_ACTION_LABEL);
+ return;
+ }
+
+ if (context.payload.action === 'labeled' && context.payload.label?.name === NO_ACTION_LABEL) {
+ await removeLabelIfPresent(context.issue.number, context.payload.issue, UNTRIAGED_LABEL);
+ return;
+ }
+
+ if (context.payload.action !== 'labeled' || context.payload.label?.name !== LAST_READ_LABEL) {
+ console.log('Not a last-read label event');
+ return;
+ }
+
+ const currentIssueNumber = context.issue.number;
+ const lastReadIssues = await github.paginate(github.rest.issues.listForRepo, {
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ state: 'all',
+ labels: LAST_READ_LABEL,
+ per_page: 100,
+ });
+
+ const previousIssueNumbers = lastReadIssues
+ .filter((issue) => !issue.pull_request)
+ .map((issue) => issue.number)
+ .filter((issueNumber) => issueNumber !== currentIssueNumber);
+
+ if (previousIssueNumbers.length === 0) {
+ console.log('No previous last-read issue found');
+ return;
+ }
+
+ const previousIssueNumber = Math.max(...previousIssueNumbers);
+ if (currentIssueNumber <= previousIssueNumber) {
+ console.log(
+ `Last-read was added to old issue #${currentIssueNumber}; latest last-read is #${previousIssueNumber}`,
+ );
+ return;
+ }
+
+ const untriagedIssues = await github.paginate(github.rest.issues.listForRepo, {
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ state: 'all',
+ labels: UNTRIAGED_LABEL,
+ per_page: 100,
+ });
+
+ const issuesToMark = untriagedIssues
+ .filter((issue) => !issue.pull_request)
+ .filter((issue) => issue.number >= previousIssueNumber && issue.number <= currentIssueNumber)
+ .sort((a, b) => a.number - b.number);
+
+ if (issuesToMark.length === 0) {
+ console.log(`No untriaged issues found from #${previousIssueNumber} to #${currentIssueNumber}`);
+ return;
+ }
+
+ for (const issue of issuesToMark) {
+ if (issueHasLabel(issue, TO_DISCUSS_LABEL)) {
+ console.log(`Skipped ${NO_ACTION_LABEL} for #${issue.number} because it has ${TO_DISCUSS_LABEL}`);
+ } else {
+ await github.rest.issues.addLabels({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ issue_number: issue.number,
+ labels: [NO_ACTION_LABEL],
+ });
+ console.log(`Added ${NO_ACTION_LABEL} to #${issue.number}`);
+ }
+
+ await github.rest.issues.update({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ issue_number: issue.number,
+ state: 'closed',
+ state_reason: 'not_planned',
+ });
+ console.log(`Closed #${issue.number} as not planned`);
+
+ await removeLabelIfPresent(issue.number, issue, INPROGRESS_LABEL);
+
+ await github.rest.issues.removeLabel({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ issue_number: issue.number,
+ name: UNTRIAGED_LABEL,
+ });
+ console.log(`Removed ${UNTRIAGED_LABEL} from #${issue.number}`);
+ }
diff --git a/.github/workflows/pr-gate.yml b/.github/workflows/pr-gate.yml
new file mode 100644
index 00000000..bc6f7eef
--- /dev/null
+++ b/.github/workflows/pr-gate.yml
@@ -0,0 +1,128 @@
+name: PR Gate
+
+on:
+ pull_request_target:
+ types: [opened]
+
+jobs:
+ check-contributor:
+ runs-on: ubuntu-latest
+ permissions:
+ contents: read
+ issues: write
+ pull-requests: write
+ steps:
+ - name: Check if contributor is approved
+ uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
+ with:
+ script: |
+ const APPROVED_FILE = '.github/APPROVED_CONTRIBUTORS';
+ const VALID_CAPABILITIES = new Set(['issue', 'pr']);
+ const TRUSTED_BOT_AUTHORS = new Set(['dependabot[bot]', 'sentry[bot]', 'claude[bot]']);
+ const prAuthor = context.payload.pull_request.user.login;
+ const defaultBranch = context.payload.repository.default_branch;
+ const isBotAuthor = prAuthor.endsWith('[bot]');
+
+ if (TRUSTED_BOT_AUTHORS.has(prAuthor)) {
+ console.log(`Skipping trusted bot: ${prAuthor}`);
+ return;
+ }
+
+ async function getPermission(username) {
+ try {
+ const { data: permissionLevel } = await github.rest.repos.getCollaboratorPermissionLevel({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ username,
+ });
+ return permissionLevel.permission;
+ } catch {
+ return null;
+ }
+ }
+
+ async function getTextFile(path) {
+ const { data: fileContent } = await github.rest.repos.getContent({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ path,
+ ref: defaultBranch,
+ });
+
+ if (!('content' in fileContent) || typeof fileContent.content !== 'string') {
+ throw new Error(`Expected file content for ${path}`);
+ }
+
+ return Buffer.from(fileContent.content, 'base64').toString('utf8');
+ }
+
+ function parseApprovedUsers(content) {
+ const users = new Map();
+
+ for (const rawLine of content.split('\n')) {
+ const line = rawLine.trim();
+ if (!line || line.startsWith('#')) continue;
+
+ const parts = line.split(/\s+/);
+ if (parts.length !== 2) {
+ console.log(`Skipping malformed line: ${rawLine}`);
+ continue;
+ }
+
+ const [username, capability] = parts;
+ const normalizedCapability = capability.toLowerCase();
+ if (!VALID_CAPABILITIES.has(normalizedCapability)) {
+ console.log(`Skipping line with invalid capability: ${rawLine}`);
+ continue;
+ }
+
+ users.set(username.toLowerCase(), normalizedCapability);
+ }
+
+ return users;
+ }
+
+ async function closePullRequest(message) {
+ await github.rest.issues.createComment({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ issue_number: context.payload.pull_request.number,
+ body: message,
+ });
+
+ await github.rest.pulls.update({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ pull_number: context.payload.pull_request.number,
+ state: 'closed',
+ });
+ }
+
+ const permission = await getPermission(prAuthor);
+ if (!isBotAuthor && ['admin', 'maintain', 'write'].includes(permission)) {
+ console.log(`${prAuthor} is a collaborator with ${permission} access`);
+ return;
+ }
+
+ const approvedContent = await getTextFile(APPROVED_FILE);
+ const approvedUsers = parseApprovedUsers(approvedContent);
+ const capability = approvedUsers.get(prAuthor.toLowerCase());
+
+ if (!isBotAuthor && capability === 'pr') {
+ console.log(`${prAuthor} is approved for PRs`);
+ return;
+ }
+
+ console.log(`${prAuthor} is not approved, closing PR`);
+
+ const message = [
+ 'This PR was auto-closed. Only contributors approved with `lgtm` can open PRs. Open an issue first and ask a maintainer for approval.',
+ '',
+ `Maintainers review auto-closed issues daily. Issues that do not meet the quality bar in [CONTRIBUTING.md](https://github.com/${context.repo.owner}/${context.repo.repo}/blob/${defaultBranch}/CONTRIBUTING.md) will not be reopened or receive a reply.`,
+ '',
+ 'If a maintainer replies `lgtmi`, your future issues will stay open. If a maintainer replies `lgtm`, your future issues and PRs will stay open. The command must be at the start of the reply (optionally after one or more `@username` mentions) or at the end.',
+ '',
+ `See [CONTRIBUTING.md](https://github.com/${context.repo.owner}/${context.repo.repo}/blob/${defaultBranch}/CONTRIBUTING.md).`,
+ ].join('\n');
+
+ await closePullRequest(message);
diff --git a/.github/workflows/remove-inprogress-on-close.yml b/.github/workflows/remove-inprogress-on-close.yml
new file mode 100644
index 00000000..4904067e
--- /dev/null
+++ b/.github/workflows/remove-inprogress-on-close.yml
@@ -0,0 +1,31 @@
+name: Remove In Progress Label On Close
+
+on:
+ issues:
+ types: [closed]
+
+jobs:
+ remove-label:
+ runs-on: ubuntu-latest
+ permissions:
+ issues: write
+ steps:
+ - name: Remove inprogress label
+ uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
+ with:
+ script: |
+ const labelName = 'inprogress';
+ const labels = context.payload.issue.labels ?? [];
+ const hasLabel = labels.some((label) => label.name === labelName);
+
+ if (!hasLabel) {
+ console.log(`Issue does not have ${labelName} label`);
+ return;
+ }
+
+ await github.rest.issues.removeLabel({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ issue_number: context.issue.number,
+ name: labelName,
+ });
diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md
index 8939aa7d..a03daec7 100644
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -18,10 +18,21 @@ Using AI to write code is fine. Submitting AI-generated slop without understandi
If you use an agent, run it from the repository root directory so it picks up `AGENTS.md` automatically. Your agent must follow the rules and guidelines in that file.
-## Issue Review
+## Contribution Gate
+
+All issues and PRs from new contributors are auto-closed by default.
Issues submitted Friday through Sunday are not guaranteed to be reviewed. If something is urgent, ask on Discord: https://discord.com/invite/3cU7Bz4UPx
+Maintainers review auto-closed issues daily and reopen worthwhile ones. Issues that do not meet the quality bar below will not be reopened or receive a reply.
+
+Approval happens through maintainer replies on issues:
+
+- `lgtmi`: your future issues will not be auto-closed
+- `lgtm`: your future issues and PRs will not be auto-closed
+
+The command must be at the start of the reply (optionally after one or more `@username` mentions) or at the end. `lgtmi` does not grant rights to submit PRs. Only `lgtm` grants rights to submit PRs.
+
## Quality Bar For Issues
If you open an issue, you must use one of the two GitHub issue templates.
@@ -34,6 +45,8 @@ If you open an issue, keep it short, concrete, and worth reading.
- Explain why it matters.
- If you want to implement the change yourself, say so.
+If the issue is real and written well, a maintainer may reopen it or reply with `lgtmi` or `lgtm` in the command position described above.
+
## Blocking
If you ignore this document twice, or if you spam the tracker with agent-generated issues, your GitHub account will be permanently blocked.
@@ -42,6 +55,8 @@ If you send a large volume of issues through automation, your GitHub account wil
## Before Submitting a PR
+Do not open a PR unless you have already been approved by a maintainer using `lgtm` in the command position described above.
+
Before submitting a PR:
```bash
@@ -59,6 +74,10 @@ Ask on [Discord](https://discord.com/invite/nKXTsAcmbT).
## FAQ
+### Why are new issues and PRs auto-closed?
+
+StepCode receives more issues than the maintainers can responsibly review in real time. Many reports do not meet the quality bar in this guide or do not follow CONTRIBUTING.md. Some are slung at the repository mindlessly via an agent instead of being reviewed and shaped by the person submitting them. Auto-closing creates a buffer so maintainers can review the tracker on their own schedule and reopen the issues that meet the quality bar.
+
### Why are weekend issues lower priority?
We triage the tracker during working hours. That means more issues can accumulate over the weekend. Anything submitted Friday through Sunday may be missed or given lower priority in the Monday review queue. If a problem is urgent, ask on Discord and include the short version, a repro, and the relevant logs.
diff --git a/README.md b/README.md
index 6935a3e6..5709c0f5 100644
--- a/README.md
+++ b/README.md
@@ -12,14 +12,16 @@
Step Code runs in your terminal and handles the full task loop—reading code, making changes, and running tests\. It works with the Step provider and discovers the available Step models after sign\-in\. MCP servers, Agent Skills, plugins, and multi\-agent orchestration work out of the box; you can delegate long\-running tasks to `/goal` and let Step Code drive them forward autonomously\.
-Step Code comes with built-in [StepPage](https://platform.stepfun.ai/docs/en/step-code/reference/steppage) publishing. Once your local page is ready, you can publish it as an accessible static website with a single command—bringing development, debugging, and delivery all within the same terminal\.
+Step Code comes with built-in StepPage publishing. Once your local page is ready, you can publish it as an accessible static website with a single command—bringing development, debugging, and delivery all within the same terminal\.
-
+
## Why Step Code
- **Token efficiency** — tuned alongside Step models to consume fewer tokens for the same task; Long\-horizon tasks are split across parallel subagents, each with its own isolated context, so redundant content never enters the main conversation\.
+- **Jev skill routing** — optional [Jev suggestions](packages/coding-agent/docs/jev-skill-routing.md) help the coding model choose among installed skills before a task starts; manual skill commands and the complete catalog remain available.
+
- **Static site publishing** — the built\-in steppage plugin publishes a local directory to a shareable static URL in one command, with version management and rollback\.
- **Long\-running task delegation** — `/goal` hands an objective to Step Code, which works toward it autonomously, and `/cron` runs on a schedule; the status line shows a live timer for the active task\.
diff --git a/README.zh-CN.md b/README.zh-CN.md
index 5f404ca5..546eca60 100644
--- a/README.zh-CN.md
+++ b/README.zh-CN.md
@@ -12,14 +12,16 @@
Step Code 运行于终端,单轮任务即可完成代码阅读、修改与测试验证的完整闭环。它与 Step(StepFun)provider 深度协同,登录后从 Step 服务发现可用模型。MCP 工具、Agent Skills、插件与多代理编排开箱即用;长任务可通过 /goal 托管,由 Step Code 持续自主推进。
-Step Code 内置 [StepPage](https://platform.stepfun.com/docs/zh/step-code/reference/steppage) 发布能力:本地页面构建完成后,只需一条命令即可发布为可访问的静态网站,让开发、调试与交付在同一个终端中完成。
+Step Code 内置 StepPage 发布能力:本地页面构建完成后,只需一条命令即可发布为可访问的静态网站,让开发、调试与交付在同一个终端中完成。
-
+
## Why Step Code
- **token 效率**——与阶跃模型深度协同,同等任务消耗更少 token;长任务拆分为子代理并行执行,子代理持有独立上下文,冗余内容不进入主对话。
+- **Jev skill 路由**——可选的 [Jev 建议](packages/coding-agent/docs/jev-skill-routing.md)在任务开始前辅助挑选已安装的 skill,帮助减少选错或漏选;保留手动调用和完整目录,默认关闭。
+
- **静态网站发布**——内置 steppage 插件,一条指令将本地目录发布为可分享的静态网址,并支持版本管理与回滚。
- **长任务托管**——`/goal` 将目标交由 Step Code 持续自主推进,`/cron` 按计划定时执行;状态行实时显示活跃任务计时。
diff --git a/apps/cli/src/bootstrap/extensions.ts b/apps/cli/src/bootstrap/extensions.ts
index 92c33f41..7fd342bf 100644
--- a/apps/cli/src/bootstrap/extensions.ts
+++ b/apps/cli/src/bootstrap/extensions.ts
@@ -12,6 +12,7 @@ import {
createStepCronExtension,
createStepExtensionInline,
createStepGoalExtension,
+ createStepJevSkillRouterExtensionInline,
type InlineExtension,
type StepExtensionOptions,
type StepTelemetryReporter,
@@ -48,6 +49,7 @@ export function createStepExtensionFactories(deps: StepExtensionFactoryDeps): In
traceHeaderPolicy: deps.traceHeaderPolicy,
}),
createStepCapabilitiesExtensionInline({ telemetry: deps.telemetry }),
+ createStepJevSkillRouterExtensionInline(),
createStepCronExtension({ telemetry: deps.telemetry }),
createStepGoalExtension({ telemetry: deps.telemetry }),
...(deps.stepCodeProviderExtension ? [deps.stepCodeProviderExtension] : []),
diff --git a/apps/cli/test/jev-skill-router-bootstrap.test.ts b/apps/cli/test/jev-skill-router-bootstrap.test.ts
new file mode 100644
index 00000000..1421934b
--- /dev/null
+++ b/apps/cli/test/jev-skill-router-bootstrap.test.ts
@@ -0,0 +1,20 @@
+import type { ExtensionAPI } from "@step-harness/coding-agent";
+import { expect, it, vi } from "vitest";
+import { createStepExtensionFactories } from "../src/bootstrap/extensions.ts";
+
+it("mounts Jev skill routing in the Step CLI without enabling outbound calls by default", async () => {
+ vi.stubEnv("STEP_JEV_SKILL_ROUTING", "");
+ const extensions = createStepExtensionFactories({
+ telemetry: { track: vi.fn() },
+ traceHeaderPolicy: { allowedBaseUrls: [], highSensitivityFields: [] },
+ stepSettings: undefined,
+ feedbackIdentity: undefined,
+ permission: undefined,
+ stepCodeProviderExtension: undefined,
+ });
+ const router = extensions.find((extension) => extension.name === "Jev skill routing");
+ if (!router || typeof router === "function") throw new Error("Jev inline extension is missing");
+ const on = vi.fn();
+ await router.factory({ on } as unknown as ExtensionAPI);
+ expect(on).not.toHaveBeenCalled();
+});
diff --git a/packages/coding-agent/README.md b/packages/coding-agent/README.md
index 6e5ae582..ee100816 100644
--- a/packages/coding-agent/README.md
+++ b/packages/coding-agent/README.md
@@ -3,6 +3,8 @@
+> New issues and PRs from new contributors are auto-closed by default. Maintainers review auto-closed issues daily. See [CONTRIBUTING.md](../../CONTRIBUTING.md).
+
---
Pi is a minimal terminal coding harness. Adapt pi to your workflows, not the other way around, without having to fork and modify pi internals. Extend it with TypeScript [Extensions](#extensions), [Skills](#skills), [Prompt Templates](#prompt-templates), and [Themes](#themes). Put your extensions, skills, prompt templates, and themes in [Pi Packages](#pi-packages) and share them with others via npm or git.
diff --git a/packages/coding-agent/docs/docs.json b/packages/coding-agent/docs/docs.json
index bbc9e74b..5bbaf1a7 100644
--- a/packages/coding-agent/docs/docs.json
+++ b/packages/coding-agent/docs/docs.json
@@ -56,6 +56,10 @@
"title": "Skills",
"path": "skills.md"
},
+ {
+ "title": "Jev Skill Routing",
+ "path": "jev-skill-routing.md"
+ },
{
"title": "Prompt Templates",
"path": "prompt-templates.md"
diff --git a/packages/coding-agent/docs/environment-variables.md b/packages/coding-agent/docs/environment-variables.md
index e73d5366..d10731ef 100644
--- a/packages/coding-agent/docs/environment-variables.md
+++ b/packages/coding-agent/docs/environment-variables.md
@@ -8,6 +8,9 @@ Step reads the variables below. Configuration shared across launches belongs in
| `STEP_CODING_AGENT_DIR` | Override the agent directory; default is `~/.stepcode/agent` |
| `STEP_CODING_AGENT_SESSION_DIR` | Override session storage; `--session-dir` takes precedence |
| `STEP_API_KEY` | StepFun API credential |
+| `STEP_JEV_SKILL_ROUTING` | Set to `1` to enable optional [Jev skill routing](jev-skill-routing.md); disabled by default |
+| `TYPESAFE_API_KEY` | TypeSafe credential for Jev routing; requires `STEP_JEV_SKILL_ROUTING=1` |
+| `JEV_API_KEY` | Fallback Jev credential when `TYPESAFE_API_KEY` is empty |
| `STEP_BASE_URL` | Override the StepFun API endpoint |
| `STEP_PROVIDER`, `STEP_MODEL` | Default provider and model selection |
| `VISUAL`, `EDITOR` | External editor fallback when `externalEditor` is unset |
diff --git a/packages/coding-agent/docs/extensions.md b/packages/coding-agent/docs/extensions.md
index 54b68cdf..f429c360 100644
--- a/packages/coding-agent/docs/extensions.md
+++ b/packages/coding-agent/docs/extensions.md
@@ -1018,14 +1018,14 @@ Access to models, providers, and resolved authentication. `ctx.modelRegistry.get
### ctx.signal
-The current agent abort signal, or `undefined` when no agent turn is active.
+The current operation's abort signal, including prompt preparation, or `undefined` when idle.
Use this for abort-aware nested work started by extension handlers, for example:
- `fetch(..., { signal: ctx.signal })`
- model calls that accept `signal`
- file or process helpers that accept `AbortSignal`
-`ctx.signal` is typically defined during active turn events such as `tool_call`, `tool_result`, `message_update`, and `turn_end`.
+`ctx.signal` is defined during `before_agent_start` and active turn events such as `tool_call`, `tool_result`, `message_update`, and `turn_end`. Pass it to asynchronous preparation work so cancellation stops that work before the coding model starts. A cancelled `before_agent_start` skips remaining handlers and does not start the agent loop.
It is usually `undefined` in idle or non-turn contexts such as session events, extension commands, and shortcuts fired while step is idle.
```typescript
@@ -1043,7 +1043,7 @@ pi.on("tool_result", async (event, ctx) => {
### ctx.isIdle() / ctx.abort() / ctx.hasPendingMessages()
-Control flow helpers. `ctx.isIdle()` is false while Step is processing an agent run, automatic retry, auto-compaction retry, or queued continuation.
+Control flow helpers. `ctx.isIdle()` is false while Step is processing `before_agent_start` hooks, an agent run, automatic retry, auto-compaction retry, or queued continuation.
### ctx.shutdown()
diff --git a/packages/coding-agent/docs/jev-skill-routing.md b/packages/coding-agent/docs/jev-skill-routing.md
new file mode 100644
index 00000000..2b10f60f
--- /dev/null
+++ b/packages/coding-agent/docs/jev-skill-routing.md
@@ -0,0 +1,75 @@
+# Jev skill routing
+
+Step Code can use [TypeSafe Jev](https://typesafe.ai/) to suggest an installed skill before the coding model starts a request. This is optional and disabled by default. It is intended to reduce missed or unnecessary skill loads when several skills are available.
+
+## Enable
+
+Install at least two [skills](skills.md), obtain a TypeSafe API key, and start Step Code with:
+
+```bash
+export TYPESAFE_API_KEY=""
+export STEP_JEV_SKILL_ROUTING=1
+step
+```
+
+For a single prompt:
+
+```bash
+STEP_JEV_SKILL_ROUTING=1 TYPESAFE_API_KEY="" \
+ step -p "Use a browser to check the checkout page"
+```
+
+Use `pnpm step` in place of `step` when running from source. `JEV_API_KEY` is accepted as a fallback when `TYPESAFE_API_KEY` is empty. A key alone does not enable routing. Unset `STEP_JEV_SKILL_ROUTING` or set it to `0` before starting Step Code to disable it.
+
+Jev uses a separate TypeSafe credential and billing account. Step sign-in and the coding model selected in `/model` continue to work as usual. This extension does not increase a Step Plan allowance.
+
+## What happens
+
+1. Before a new text request starts, Jev ranks the names and descriptions of the eligible skills and checks whether any skill is needed.
+2. If there is a match, Step reads a short excerpt from each of up to three shortlisted skills. A second Jev call checks their suitability.
+3. A sufficiently confident result adds a small suggestion to the current system prompt. The coding model decides whether to read and follow the suggested skill.
+
+The full skill catalog stays available, including skills Jev did not select. Explicit skill requests and project instructions take precedence. The suggestion grants no tool permissions and does not execute a skill. It is reset before the next request and is not recomputed on every tool call within a request.
+
+Both API calls and excerpt reads share a 1,500 ms deadline and honor task cancellation. Cancelling or closing the session while routing is pending stops further Jev calls and prevents the coding model from starting. There are no retries. A timeout, API error, malformed answer, uncertain result, or no match leaves the original prompt in place so the task can continue.
+
+Routing is skipped for:
+
+- `/skill:name` commands, expanded skill commands, and `$name` mentions of an installed skill.
+- Requests with images, empty requests, or requests over 12,000 characters.
+- Sessions without a `read` or `read_file` tool.
+- Catalogs with fewer than two or more than 254 eligible skills, or duplicate names.
+
+Skills with `disable-model-invocation: true` are excluded from automatic routing; explicit `/skill:name` commands still work. Existing discovery and project-trust rules determine which skills are available. `/reload` refreshes the catalog used for the next request.
+
+## Data sent to TypeSafe
+
+Enabling this feature sends data to `https://api.typesafe.ai/v1/systemone`, using the `jev-latest` model:
+
+| Request | Data |
+| --- | --- |
+| Ranking | Current user request, eligible skill names and descriptions |
+| Verification | Current user request, shortlisted names and descriptions, and up to 700 characters from each of at most three skill bodies |
+
+The excerpt reader skips YAML frontmatter and reads at most the first 16 KiB of a skill file. A frontmatter block that extends beyond that prefix yields an empty excerpt. The extension does not separately attach local file paths, the system prompt, project context files, conversation history, tool results, images, or Step credentials. Text already present in the user request or skill metadata/body is sent as described above, so use this feature only for content you can share with TypeSafe.
+
+The endpoint is fixed, and HTTP redirects are rejected. The extension does not persist routing requests or responses.
+
+## Efficiency and measurement
+
+The potential benefit is better skill selection: avoiding an irrelevant full skill read or finding a useful skill earlier. Keeping the complete catalog preserves its existing system-prompt prefix, but does not remove its tokens. The changing recommendation can affect caching after that prefix; it does not guarantee a cache hit for the rest of the conversation. Jev also adds API calls, billable input, and latency.
+
+TypeSafe's [skill-suggestion cookbook](https://docs.typesafe.ai/cookbooks/skill_suggestion.md) reports an evaluation using 182 Hermes skills and 488 constructed requests. On 315 requests with a matching skill, the rate of a wrong or missing first skill load fell from 16.8% to 7.3%; on 173 requests without a match, unnecessary loads fell from 9.8% to 4.0%. Those results used Jev 1.12 and Claude Haiku 4.5. They are evidence from that experiment, not measured Step Code results. This integration uses its own conservative confidence gates and the `jev-latest` alias.
+
+As of September 22, 2026, TypeSafe lists Jev 1.13.0 input at $0.042 per million tokens and output as free. Consult the current [models and pricing](https://docs.typesafe.ai/models.md) before enabling it.
+
+To measure the effect on your workflow, compare the same tasks, skill catalog, coding model, and initial repository state with routing disabled and enabled. Include both matching and no-match tasks, and repeat runs to account for model variation. TypeSafe currently reports its best accuracy in English, so evaluate Chinese and other languages separately when they are part of your workflow. Record:
+
+- Task completion quality and whether the first skill read was appropriate.
+- Unnecessary skill reads and coding-model input, output, and cache usage.
+- Jev charges and total cost per completed task.
+- Time to the first coding-model response and total task latency, including slow or unavailable Jev calls.
+
+Step's `/session` statistics cover the coding session; this extension's TypeSafe usage is not added to those totals. Include TypeSafe account usage separately when comparing costs. The automated tests exercise the real request format with a fake transport and a scripted coding model; they do not measure live Jev accuracy, production latency, token savings, or user growth.
+
+See the [TypeSafe API reference](https://docs.typesafe.ai/api.md) for the typed Choice and Noul response formats.
diff --git a/packages/coding-agent/docs/skills.md b/packages/coding-agent/docs/skills.md
index bed45c72..2ad54d34 100644
--- a/packages/coding-agent/docs/skills.md
+++ b/packages/coding-agent/docs/skills.md
@@ -71,6 +71,8 @@ For project-level Claude Code skills, add to `.stepcode/settings.json`:
This is progressive disclosure: only descriptions are always in context, full instructions load on-demand.
+For catalogs with several skills, optional [Jev skill routing](jev-skill-routing.md) can suggest a relevant skill before the coding model starts. It is disabled by default, keeps the full catalog available, and respects explicit skill commands.
+
## Skill Commands
Skills register as `/skill:name` commands:
diff --git a/packages/coding-agent/src/core/agent-session-runtime.ts b/packages/coding-agent/src/core/agent-session-runtime.ts
index 4da727a2..cba913ad 100644
--- a/packages/coding-agent/src/core/agent-session-runtime.ts
+++ b/packages/coding-agent/src/core/agent-session-runtime.ts
@@ -531,6 +531,8 @@ export class AgentSessionRuntime implements AgentSessionRuntimeHost {
const session = this._session;
try {
if (!this.currentSessionDisposed) {
+ // Settle prompt hooks and model work before invalidating their context.
+ await session.abort();
await emitSessionShutdownEvent(session.extensionRunner, {
type: "session_shutdown",
reason: "quit",
diff --git a/packages/coding-agent/src/core/agent-session.ts b/packages/coding-agent/src/core/agent-session.ts
index 945f3200..b053dc6f 100644
--- a/packages/coding-agent/src/core/agent-session.ts
+++ b/packages/coding-agent/src/core/agent-session.ts
@@ -952,6 +952,7 @@ export class AgentSession {
*/
dispose(): void {
try {
+ this._agentRunAbortController?.abort();
this.abortRetry();
this.abortCompaction();
this.abortBranchSummary();
@@ -988,12 +989,12 @@ export class AgentSession {
return this.agent.state.thinkingLevel;
}
- /** Whether the session is currently processing an agent run or post-run continuation. */
+ /** Whether the session is processing prompt hooks, an agent run, or post-run continuation. */
get isStreaming(): boolean {
return this._isAgentRunActive;
}
- /** Whether the session has no active agent run, retry, auto-compaction, or queued continuation. */
+ /** Whether the session has no active prompt hooks, agent run, retry, or queued continuation. */
get isIdle(): boolean {
return !this._isAgentRunActive;
}
@@ -1203,11 +1204,16 @@ export class AgentSession {
// Prompting
// =========================================================================
- private async _runAgentPrompt(messages: AgentMessage | AgentMessage[]): Promise {
+ private async _runAgentPrompt(
+ messages: AgentMessage | AgentMessage[],
+ beforeStart?: (signal: AbortSignal) => Promise,
+ ): Promise {
const runAbortController = new AbortController();
this._agentRunAbortController = runAbortController;
this._isAgentRunActive = true;
try {
+ if (beforeStart) await beforeStart(runAbortController.signal);
+ if (runAbortController.signal.aborted) return;
await this.agent.prompt(messages);
while (
!runAbortController.signal.aborted &&
@@ -1276,6 +1282,7 @@ export class AgentSession {
const expandPromptTemplates = options?.expandPromptTemplates ?? true;
const preflightResult = options?.preflightResult;
let messages: AgentMessage[] | undefined;
+ let beforeStart: ((signal: AbortSignal) => Promise) | undefined;
try {
// Handle extension commands first (execute immediately, even during streaming)
@@ -1389,36 +1396,49 @@ export class AgentSession {
}
this._pendingNextTurnMessages = [];
- // Emit before_agent_start extension event
- const result = await this._extensionRunner.emitBeforeAgentStart(
- expandedText,
- currentImages,
- this._baseSystemPrompt,
- this._baseSystemPromptOptions,
- );
- // Add all custom messages from extensions
- if (result?.messages) {
- for (const msg of result.messages) {
- messages.push({
- role: "custom",
- customType: msg.customType,
- // Untyped extensions can pass null/missing content; normalize at ingestion.
- content: msg.content ?? [],
- display: msg.display,
- details: msg.details,
- timestamp: Date.now(),
- });
+ const promptMessages = messages;
+ beforeStart = async (signal) => {
+ try {
+ // Emit before_agent_start extension event
+ const result = await this._extensionRunner.emitBeforeAgentStart(
+ expandedText,
+ currentImages,
+ this._baseSystemPrompt,
+ this._baseSystemPromptOptions,
+ );
+ if (signal.aborted) {
+ preflightResult?.(false);
+ return;
+ }
+ // Add all custom messages from extensions
+ if (result?.messages) {
+ for (const msg of result.messages) {
+ promptMessages.push({
+ role: "custom",
+ customType: msg.customType,
+ // Untyped extensions can pass null/missing content; normalize at ingestion.
+ content: msg.content ?? [],
+ display: msg.display,
+ details: msg.details,
+ timestamp: Date.now(),
+ });
+ }
+ }
+ // Apply extension-modified system prompt, or reset to base
+ if (result?.systemPrompt !== undefined) {
+ this._systemPromptOverride = result.systemPrompt;
+ this.agent.state.systemPrompt = result.systemPrompt;
+ } else {
+ // Ensure we're using the base prompt (in case previous turn had modifications)
+ this._systemPromptOverride = undefined;
+ this.agent.state.systemPrompt = this._baseSystemPrompt;
+ }
+ preflightResult?.(true);
+ } catch (error) {
+ preflightResult?.(false);
+ throw error;
}
- }
- // Apply extension-modified system prompt, or reset to base
- if (result?.systemPrompt !== undefined) {
- this._systemPromptOverride = result.systemPrompt;
- this.agent.state.systemPrompt = result.systemPrompt;
- } else {
- // Ensure we're using the base prompt (in case previous turn had modifications)
- this._systemPromptOverride = undefined;
- this.agent.state.systemPrompt = this._baseSystemPrompt;
- }
+ };
} catch (error) {
preflightResult?.(false);
throw error;
@@ -1428,8 +1448,7 @@ export class AgentSession {
return;
}
- preflightResult?.(true);
- await this._runAgentPrompt(messages);
+ await this._runAgentPrompt(messages, beforeStart);
}
/**
@@ -2753,7 +2772,7 @@ export class AgentSession {
getScopedModels: () => this._scopedModels,
isIdle: () => this.isIdle,
isProjectTrusted: () => this.settingsManager.isProjectTrusted(),
- getSignal: () => this.agent.signal,
+ getSignal: () => this.agent.signal ?? this._agentRunAbortController?.signal,
abort: () => {
this._abortCurrentRun();
// Hosts may restore queued input after the session has been cancelled.
diff --git a/packages/coding-agent/src/core/extensions/runner.ts b/packages/coding-agent/src/core/extensions/runner.ts
index b91f96e0..828333f0 100644
--- a/packages/coding-agent/src/core/extensions/runner.ts
+++ b/packages/coding-agent/src/core/extensions/runner.ts
@@ -1155,6 +1155,8 @@ export class ExtensionRunner {
this.assertActive();
return currentSystemPrompt;
};
+ // Keep this operation's signal usable after disposal invalidates its context.
+ const signal = ctx.signal;
const messages: NonNullable[] = [];
let systemPromptModified = false;
@@ -1163,6 +1165,7 @@ export class ExtensionRunner {
if (!handlers || handlers.length === 0) continue;
for (const handler of handlers) {
+ if (signal?.aborted) return undefined;
try {
const event: BeforeAgentStartEvent = {
type: "before_agent_start",
@@ -1172,6 +1175,7 @@ export class ExtensionRunner {
systemPromptOptions,
};
const handlerResult = await handler(event, ctx);
+ if (signal?.aborted) return undefined;
if (handlerResult) {
const result = handlerResult as BeforeAgentStartEventResult;
@@ -1184,6 +1188,7 @@ export class ExtensionRunner {
}
}
} catch (err) {
+ if (signal?.aborted) return undefined;
const message = err instanceof Error ? err.message : String(err);
const stack = err instanceof Error ? err.stack : undefined;
this.emitError({
diff --git a/packages/coding-agent/src/core/extensions/types.ts b/packages/coding-agent/src/core/extensions/types.ts
index ea9c6946..0d01b5fa 100644
--- a/packages/coding-agent/src/core/extensions/types.ts
+++ b/packages/coding-agent/src/core/extensions/types.ts
@@ -374,7 +374,7 @@ export interface ExtensionContext {
isIdle(): boolean;
/** Whether project-local trust is active for this context. */
isProjectTrusted(): boolean;
- /** The current abort signal, or undefined when the agent is not streaming. */
+ /** The current operation's abort signal, including before_agent_start; undefined when idle. */
signal: AbortSignal | undefined;
/** Abort the current agent operation */
abort(): void;
diff --git a/packages/coding-agent/src/features/jev-skill-router/router.ts b/packages/coding-agent/src/features/jev-skill-router/router.ts
new file mode 100644
index 00000000..e84d3195
--- /dev/null
+++ b/packages/coding-agent/src/features/jev-skill-router/router.ts
@@ -0,0 +1,251 @@
+export const JEV_MAX_SKILLS = 254;
+export const JEV_MAX_REQUEST_CHARS = 12000;
+export const JEV_EXCERPT_CHARS = 700;
+export const JEV_TIMEOUT_MS = 1500;
+
+export interface JevSkill {
+ name: string;
+ description: string;
+}
+
+export interface JevRoutingOptions {
+ apiKey: string;
+ fetch?: typeof globalThis.fetch;
+ timeoutMs?: number;
+ signal?: AbortSignal;
+ loadExcerpt: (skill: JevSkill, signal: AbortSignal) => Promise;
+}
+
+export interface JevRoutingResult {
+ skillName?: string;
+ reason: "selected" | "no-match" | "low-confidence" | "unavailable" | "unsupported-input";
+}
+
+const ENDPOINT = "https://api.typesafe.ai/v1/systemone";
+const MIN_CONFIDENCE = 0.6;
+const MIN_FIT = 0.5;
+const SHORTLIST_SIZE = 3;
+// Allow rounding in an otherwise complete probability distribution.
+const PROBABILITY_SUM_TOLERANCE = 0.01;
+const NONE = "No documented skill would be useful for this specific task.";
+const RANK_INSTRUCTIONS =
+ "Which documented skill would be most useful for the user's specific request? Choose none if no skill would help.";
+const VERIFY_INSTRUCTIONS =
+ "Which skill best fits the user's specific request, considering its description and excerpt? Choose none if none fits.";
+const NEEDS_INSTRUCTIONS = "Would any of these documented skills be useful for the user's specific request?\n";
+const FIT_INSTRUCTIONS = "Does this skill, as documented below, fit the user's specific request?\n";
+
+type Questions = Record<
+ string,
+ { type: "choice"; instructions: string; criteria: Record } | { type: "noul"; instructions: string }
+>;
+
+interface ChoiceAnswer {
+ type: "choice";
+ choice: string;
+ probabilities: Record;
+ confidence: number;
+}
+
+function isRecord(value: unknown): value is Record {
+ return typeof value === "object" && value !== null && !Array.isArray(value);
+}
+
+function isProbability(value: unknown): value is number {
+ return typeof value === "number" && Number.isFinite(value) && value >= 0 && value <= 1;
+}
+
+function readChoice(value: unknown, allowed: readonly string[]): ChoiceAnswer {
+ if (
+ !isRecord(value) ||
+ value.type !== "choice" ||
+ typeof value.choice !== "string" ||
+ !allowed.includes(value.choice) ||
+ !isProbability(value.confidence) ||
+ !isRecord(value.probabilities) ||
+ Object.keys(value.probabilities).length !== allowed.length
+ ) {
+ throw new Error("Invalid Jev choice");
+ }
+ const probabilities: Record = {};
+ let total = 0;
+ for (const id of allowed) {
+ const probability = value.probabilities[id];
+ if (!Object.hasOwn(value.probabilities, id) || !isProbability(probability)) {
+ throw new Error("Invalid Jev probability");
+ }
+ probabilities[id] = probability;
+ total += probability;
+ }
+ if (Math.abs(total - 1) > PROBABILITY_SUM_TOLERANCE) throw new Error("Invalid Jev distribution");
+ return { type: "choice", choice: value.choice, probabilities, confidence: value.confidence };
+}
+
+function readNoul(value: unknown): number {
+ if (!isRecord(value) || value.type !== "noul" || !isProbability(value.noul)) {
+ throw new Error("Invalid Jev noul");
+ }
+ return value.noul;
+}
+
+export async function suggestJevSkill(
+ request: string,
+ skills: readonly JevSkill[],
+ options: JevRoutingOptions,
+): Promise {
+ if (
+ typeof request !== "string" ||
+ !request.trim() ||
+ request.length > JEV_MAX_REQUEST_CHARS ||
+ !Array.isArray(skills) ||
+ skills.length < 2 ||
+ skills.length > JEV_MAX_SKILLS ||
+ !isRecord(options) ||
+ typeof options.apiKey !== "string" ||
+ !options.apiKey.trim() ||
+ typeof options.loadExcerpt !== "function" ||
+ (options.fetch !== undefined && typeof options.fetch !== "function")
+ ) {
+ return { reason: "unsupported-input" };
+ }
+ const names = new Set();
+ for (const skill of skills) {
+ if (
+ !isRecord(skill) ||
+ typeof skill.name !== "string" ||
+ !skill.name.trim() ||
+ typeof skill.description !== "string" ||
+ names.has(skill.name)
+ ) {
+ return { reason: "unsupported-input" };
+ }
+ names.add(skill.name);
+ }
+ const timeoutMs = options.timeoutMs === undefined ? JEV_TIMEOUT_MS : options.timeoutMs;
+ if (!Number.isFinite(timeoutMs) || timeoutMs <= 0 || timeoutMs > 2_147_483_647) {
+ return { reason: "unsupported-input" };
+ }
+
+ let callerSignal: AbortSignal | undefined;
+ try {
+ callerSignal = options.signal;
+ if (
+ callerSignal !== undefined &&
+ (!(callerSignal instanceof AbortSignal) ||
+ typeof callerSignal.aborted !== "boolean" ||
+ typeof callerSignal.addEventListener !== "function" ||
+ typeof callerSignal.removeEventListener !== "function")
+ ) {
+ return { reason: "unsupported-input" };
+ }
+ if (callerSignal?.aborted) return { reason: "unavailable" };
+ // A forged prototype can pass instanceof without being a usable native signal.
+ if (callerSignal) AbortSignal.prototype.throwIfAborted.call(callerSignal);
+ } catch {
+ return { reason: "unsupported-input" };
+ }
+
+ const controller = new AbortController();
+ const fetch = options.fetch ?? globalThis.fetch;
+ const deadline = performance.now() + timeoutMs;
+ let timer: ReturnType | undefined;
+ let cancel!: () => void;
+ // Both interruption sources win even when injected readers/transports ignore abort.
+ const interrupted = new Promise((resolve) => {
+ cancel = () => {
+ controller.abort();
+ resolve({ reason: "unavailable" });
+ };
+ });
+
+ function checkDeadline(): void {
+ if (controller.signal.aborted || performance.now() >= deadline) throw new Error("Jev routing timed out");
+ }
+
+ async function post(questions: Questions): Promise> {
+ checkDeadline();
+ const body = JSON.stringify({ model: "jev-latest", state: { request }, questions });
+ checkDeadline();
+ const response = await fetch(ENDPOINT, {
+ method: "POST",
+ headers: { Authorization: `Bearer ${options.apiKey}`, "Content-Type": "application/json" },
+ redirect: "error",
+ signal: controller.signal,
+ body,
+ });
+ checkDeadline();
+ if (!response.ok || response.redirected) throw new Error("Jev request unavailable");
+ const payload: unknown = await response.json();
+ checkDeadline();
+ if (!isRecord(payload) || !isRecord(payload.answers)) throw new Error("Invalid Jev answers");
+ return payload.answers;
+ }
+
+ async function route(): Promise {
+ const candidates = skills.map((skill, index) => ({ id: `skill_${index}`, skill, index }));
+ const criteria: Record = Object.fromEntries(
+ candidates.map(({ id, skill }) => [id, JSON.stringify({ name: skill.name, description: skill.description })]),
+ );
+ const first = await post({
+ which: { type: "choice", instructions: RANK_INSTRUCTIONS, criteria: { ...criteria, none: NONE } },
+ // Each question is independent, so the Noul must carry its own documentation.
+ needs_skill: { type: "noul", instructions: NEEDS_INSTRUCTIONS + Object.values(criteria).join("\n") },
+ });
+ const ranking = readChoice(first.which, [...candidates.map(({ id }) => id), "none"]);
+ const needsSkill = readNoul(first.needs_skill);
+ if (ranking.choice === "none" || needsSkill < MIN_FIT) return { reason: "no-match" };
+
+ const shortlist = candidates
+ .sort((a, b) => ranking.probabilities[b.id] - ranking.probabilities[a.id] || a.index - b.index)
+ .slice(0, SHORTLIST_SIZE);
+ const excerpts = await Promise.all(
+ shortlist.map(async ({ id, skill }) => {
+ checkDeadline();
+ const excerpt = await options.loadExcerpt(skill, controller.signal);
+ checkDeadline();
+ if (typeof excerpt !== "string") throw new Error("Invalid skill excerpt");
+ return [
+ id,
+ JSON.stringify({
+ name: skill.name,
+ description: skill.description,
+ excerpt: excerpt.slice(0, JEV_EXCERPT_CHARS),
+ }),
+ ] as const;
+ }),
+ );
+ const questions: Questions = {
+ which: {
+ type: "choice",
+ instructions: VERIFY_INSTRUCTIONS,
+ criteria: { ...Object.fromEntries(excerpts), none: NONE },
+ },
+ };
+ for (const [id, description] of excerpts) {
+ questions[`fits_${id}`] = { type: "noul", instructions: FIT_INSTRUCTIONS + description };
+ }
+ const second = await post(questions);
+ const selection = readChoice(second.which, [...shortlist.map(({ id }) => id), "none"]);
+ const fits = new Map(shortlist.map(({ id }) => [id, readNoul(second[`fits_${id}`])]));
+ if (selection.choice === "none") return { reason: "no-match" };
+ if (selection.confidence < MIN_CONFIDENCE || fits.get(selection.choice)! < MIN_FIT) {
+ return { reason: "low-confidence" };
+ }
+ return { reason: "selected", skillName: shortlist.find(({ id }) => id === selection.choice)!.skill.name };
+ }
+
+ try {
+ timer = setTimeout(cancel, timeoutMs);
+ callerSignal?.addEventListener("abort", cancel, { once: true });
+ if (callerSignal?.aborted) cancel();
+ const result = await Promise.race([route(), interrupted]);
+ checkDeadline();
+ return result;
+ } catch {
+ controller.abort();
+ return { reason: "unavailable" };
+ } finally {
+ clearTimeout(timer);
+ callerSignal?.removeEventListener("abort", cancel);
+ }
+}
diff --git a/packages/coding-agent/src/features/step-jev-skill-router.ts b/packages/coding-agent/src/features/step-jev-skill-router.ts
new file mode 100644
index 00000000..2f75e309
--- /dev/null
+++ b/packages/coding-agent/src/features/step-jev-skill-router.ts
@@ -0,0 +1,108 @@
+import { open } from "node:fs/promises";
+import type { ExtensionAPI, ExtensionFactory, InlineExtension } from "../core/extensions/types.ts";
+import type { Skill } from "../core/skills.ts";
+import { stripFrontmatter } from "../utils/frontmatter.ts";
+import {
+ JEV_EXCERPT_CHARS,
+ JEV_MAX_REQUEST_CHARS,
+ JEV_MAX_SKILLS,
+ suggestJevSkill,
+} from "./jev-skill-router/router.ts";
+
+export interface StepJevSkillRouterOptions {
+ env?: NodeJS.ProcessEnv;
+ fetch?: typeof globalThis.fetch;
+}
+
+// Read just the beginning, even when a skill bundles a very large reference.
+const MAX_SKILL_PREFIX_BYTES = 16_384;
+
+async function readSkillExcerpt(skill: Skill, signal: AbortSignal): Promise {
+ signal.throwIfAborted();
+ const file = await open(skill.filePath, "r");
+ try {
+ signal.throwIfAborted();
+ const buffer = Buffer.alloc(MAX_SKILL_PREFIX_BYTES);
+ const { bytesRead } = await file.read(buffer, 0, buffer.length, 0);
+ signal.throwIfAborted();
+ const prefix = buffer
+ .subarray(0, bytesRead)
+ .toString("utf8")
+ .replace(/^\uFEFF/, "")
+ .replace(/\r\n?/g, "\n");
+ // A frontmatter block larger than the read limit has no body to disclose.
+ if (prefix.startsWith("---") && !prefix.includes("\n---", 3)) return "";
+ return stripFrontmatter(prefix).slice(0, JEV_EXCERPT_CHARS);
+ } finally {
+ await file.close();
+ }
+}
+
+function escapeXml(value: string): string {
+ return value
+ .replace(/&/g, "&")
+ .replace(//g, ">")
+ .replace(/"/g, """)
+ .replace(/'/g, "'");
+}
+
+/** Optional skill suggestions, using Pi's existing per-prompt extension hook. */
+export function createStepJevSkillRouterExtension(options: StepJevSkillRouterOptions = {}): ExtensionFactory {
+ return (pi: ExtensionAPI): void => {
+ const env = options.env ?? process.env;
+ if (env.STEP_JEV_SKILL_ROUTING !== "1") return;
+ const apiKey = env.TYPESAFE_API_KEY?.trim() || env.JEV_API_KEY?.trim();
+ if (!apiKey) return;
+
+ pi.on("before_agent_start", async (event, ctx) => {
+ const prompt = event.prompt.trim();
+ if (!prompt || prompt.length > JEV_MAX_REQUEST_CHARS || event.images?.length) return;
+ // Expanded manual commands contain full skill bodies; never classify them.
+ if (prompt.startsWith("/skill:") || prompt.startsWith(" name === "read" || name === "read_file")) return;
+ if (loadedSkills.some((skill) => prompt.includes(`$${skill.name}`))) return;
+
+ const skills = loadedSkills.filter((skill) => !skill.disableModelInvocation);
+ if (skills.length < 2 || skills.length > JEV_MAX_SKILLS) return;
+ const byName = new Map(skills.map((skill) => [skill.name, skill]));
+ if (byName.size !== skills.length) return;
+
+ const result = await suggestJevSkill(
+ event.prompt,
+ skills.map(({ name, description }) => ({ name, description })),
+ {
+ apiKey,
+ fetch: options.fetch,
+ signal: ctx.signal,
+ loadExcerpt: async (candidate, signal) => {
+ const skill = byName.get(candidate.name);
+ if (!skill) throw new Error("Skill is not eligible for automatic routing");
+ return readSkillExcerpt(skill, signal);
+ },
+ },
+ );
+ if (result.reason !== "selected" || !result.skillName || !byName.has(result.skillName)) return;
+
+ // Preserve the catalog and the preceding cache prefix. The runtime resets
+ // this override before the next prompt, including after a failed route.
+ return {
+ systemPrompt:
+ `${event.systemPrompt}\n\n\n` +
+ `Jev suggests considering ${escapeXml(result.skillName)} for this request. ` +
+ "Read its skill file if it fits the user's task. " +
+ "Explicit skill requests and project instructions take precedence; other skills remain available.\n" +
+ "",
+ };
+ });
+ };
+}
+
+export function createStepJevSkillRouterExtensionInline(options: StepJevSkillRouterOptions = {}): InlineExtension {
+ return {
+ name: "Jev skill routing",
+ factory: createStepJevSkillRouterExtension(options),
+ hidden: true,
+ };
+}
diff --git a/packages/coding-agent/src/index.ts b/packages/coding-agent/src/index.ts
index 5b9a78f4..2451a926 100644
--- a/packages/coding-agent/src/index.ts
+++ b/packages/coding-agent/src/index.ts
@@ -463,6 +463,11 @@ export {
type StepCronRuntimeOptions,
stepCronExtensionInline,
} from "./features/step-cron.ts";
+export {
+ createStepJevSkillRouterExtension,
+ createStepJevSkillRouterExtensionInline,
+ type StepJevSkillRouterOptions,
+} from "./features/step-jev-skill-router.ts";
export { createStepProviderConfig, STEP_PROVIDER_ID } from "./features/step-provider/index.ts";
export {
CreateGoalParams,
diff --git a/packages/coding-agent/test/jev-skill-router.test.ts b/packages/coding-agent/test/jev-skill-router.test.ts
new file mode 100644
index 00000000..dea99c6b
--- /dev/null
+++ b/packages/coding-agent/test/jev-skill-router.test.ts
@@ -0,0 +1,991 @@
+import { getEventListeners } from "node:events";
+import { afterEach, describe, expect, test, vi } from "vitest";
+import {
+ JEV_EXCERPT_CHARS,
+ JEV_MAX_REQUEST_CHARS,
+ JEV_MAX_SKILLS,
+ JEV_TIMEOUT_MS,
+ type JevRoutingOptions,
+ type JevSkill,
+ suggestJevSkill,
+} from "../src/features/jev-skill-router/router.ts";
+
+const REQUEST = "Investigate why the TypeScript build fails.";
+const API_KEY = "test-key-never-in-the-body";
+const SKILLS: readonly JevSkill[] = Object.freeze([
+ Object.freeze({ name: "debug", description: "Investigate failures and verify their causes." }),
+ Object.freeze({ name: "review", description: "Review a proposed code change." }),
+]);
+
+function choice(probabilities: Record, selected = "skill_0", confidence = 0.9) {
+ return { type: "choice", choice: selected, probabilities, confidence };
+}
+
+function noul(value: number) {
+ return { type: "noul", noul: value };
+}
+
+function firstAnswers(selected = "skill_0", needsSkill = 0.9) {
+ return {
+ which: choice({ skill_0: 0.7, skill_1: 0.2, none: 0.1 }, selected),
+ needs_skill: noul(needsSkill),
+ };
+}
+
+function secondAnswers(selected = "skill_0", confidence = 0.9, chosenFit = 0.9) {
+ return {
+ which: choice({ skill_0: 0.7, skill_1: 0.2, none: 0.1 }, selected, confidence),
+ fits_skill_0: noul(chosenFit),
+ fits_skill_1: noul(1),
+ };
+}
+
+function response(answers: Record) {
+ return Response.json({ answers, model: "jev-latest", usage: { input_tokens: 80, output_tokens: 12 } });
+}
+
+function setup(first = firstAnswers(), second = secondAnswers()) {
+ const fetch = vi
+ .fn()
+ .mockResolvedValueOnce(response(first))
+ .mockResolvedValueOnce(response(second));
+ const loadExcerpt = vi
+ .fn()
+ .mockImplementation(async (skill) => `Guide for ${skill.name}`);
+ return { fetch, loadExcerpt, apiKey: API_KEY };
+}
+
+function requestBody(fetch: ReturnType["fetch"], call: number) {
+ return JSON.parse(fetch.mock.calls[call][1]!.body as string);
+}
+
+afterEach(() => {
+ vi.useRealTimers();
+ vi.restoreAllMocks();
+ vi.unstubAllGlobals();
+});
+
+describe("Jev skill recommendations", () => {
+ test("exports the agreed routing bounds", () => {
+ expect([JEV_MAX_SKILLS, JEV_MAX_REQUEST_CHARS, JEV_EXCERPT_CHARS, JEV_TIMEOUT_MS]).toEqual([
+ 254, 12000, 700, 1500,
+ ]);
+ });
+
+ test("selects a skill using two batched requests with self-contained questions and no host fields", async () => {
+ const skills = [
+ {
+ ...SKILLS[0],
+ description: `Full description: ${"D".repeat(1600)}`,
+ path: "/private/debug/SKILL.md",
+ secret: "host-secret",
+ },
+ { ...SKILLS[1], cwd: "/private/worktree", hidden: false },
+ ];
+ const options = setup();
+
+ await expect(suggestJevSkill(REQUEST, skills, options)).resolves.toEqual({
+ reason: "selected",
+ skillName: "debug",
+ });
+ expect(options.fetch).toHaveBeenCalledTimes(2);
+ const [first, second] = [requestBody(options.fetch, 0), requestBody(options.fetch, 1)];
+ const signal = options.fetch.mock.calls[0][1]!.signal;
+ expect(signal).toBeInstanceOf(AbortSignal);
+ for (const [url, init] of options.fetch.mock.calls) {
+ expect(url).toBe("https://api.typesafe.ai/v1/systemone");
+ expect(init).toMatchObject({
+ method: "POST",
+ redirect: "error",
+ headers: { Authorization: `Bearer ${API_KEY}`, "Content-Type": "application/json" },
+ signal,
+ });
+ expect(init!.signal).toBe(signal);
+ for (const excluded of [API_KEY, "host-secret", "/private/", '"hidden"', '"cwd"', '"path"']) {
+ expect(init!.body).not.toContain(excluded);
+ }
+ }
+ for (const body of [first, second]) {
+ expect(Object.keys(body).sort()).toEqual(["model", "questions", "state"]);
+ expect(body.model).toBe("jev-latest");
+ expect(body.state).toEqual({ request: REQUEST });
+ expect(body.questions.which).toEqual({
+ type: "choice",
+ instructions: expect.any(String),
+ criteria: expect.objectContaining({ none: expect.any(String) }),
+ });
+ expect(Object.keys(body.questions.which.criteria)).toEqual(["skill_0", "skill_1", "none"]);
+ }
+ expect(Object.keys(first.questions)).toEqual(["which", "needs_skill"]);
+ expect(first.questions.needs_skill).toEqual({ type: "noul", instructions: expect.any(String) });
+ expect(Object.keys(second.questions)).toEqual(["which", "fits_skill_0", "fits_skill_1"]);
+ for (const [index, skill] of skills.entries()) {
+ const id = `skill_${index}`;
+ expect(JSON.parse(first.questions.which.criteria[id])).toEqual({
+ name: skill.name,
+ description: skill.description,
+ });
+ expect(first.questions.needs_skill.instructions).toContain(skill.name);
+ expect(first.questions.needs_skill.instructions).toContain(skill.description);
+ expect(JSON.parse(second.questions.which.criteria[id])).toEqual({
+ name: skill.name,
+ description: skill.description,
+ excerpt: `Guide for ${skill.name}`,
+ });
+ expect(second.questions[`fits_${id}`]).toEqual({ type: "noul", instructions: expect.any(String) });
+ expect(second.questions[`fits_${id}`].instructions).toContain(skill.description);
+ expect(second.questions[`fits_${id}`].instructions).toContain(`Guide for ${skill.name}`);
+ expect(options.loadExcerpt).toHaveBeenNthCalledWith(index + 1, skill, signal);
+ }
+ expect(options.loadExcerpt).toHaveBeenCalledTimes(2);
+ });
+
+ test.each([
+ ["none", 0.99],
+ ["skill_0", 0.499],
+ ])("stops after the first pass for choice %s and needs_skill %s", async (selected, needsSkill) => {
+ const options = setup(firstAnswers(selected, needsSkill));
+ await expect(suggestJevSkill(REQUEST, SKILLS, options)).resolves.toEqual({ reason: "no-match" });
+ expect(options.fetch).toHaveBeenCalledTimes(1);
+ expect(options.loadExcerpt).not.toHaveBeenCalled();
+ });
+
+ test("returns no-match for the second pass's explicit none even with high fits", async () => {
+ const options = setup(firstAnswers(), secondAnswers("none"));
+ await expect(suggestJevSkill(REQUEST, SKILLS, options)).resolves.toEqual({ reason: "no-match" });
+ expect(options.fetch).toHaveBeenCalledTimes(2);
+ });
+
+ test.each([
+ [0.6, 0.5, "selected"],
+ [0.599, 1, "low-confidence"],
+ [1, 0.499, "low-confidence"],
+ [0, 0, "low-confidence"],
+ ])("gates the chosen candidate on confidence %s and its own fit %s", async (confidence, fit, reason) => {
+ const first = firstAnswers("skill_0", 0.5);
+ first.which.confidence = 0;
+ const options = setup(first, secondAnswers("skill_0", confidence, fit));
+ await expect(suggestJevSkill(REQUEST, SKILLS, options)).resolves.toEqual(
+ reason === "selected" ? { reason, skillName: "debug" } : { reason },
+ );
+ expect(options.fetch).toHaveBeenCalledTimes(2);
+ });
+
+ test("loads only the three highest probabilities with catalog-order ties and 700-character excerpts", async () => {
+ const skills = Object.freeze(
+ Array.from({ length: 5 }, (_, index) =>
+ Object.freeze({
+ name: ["none", "skill_0", "constructor", "__proto__", "fifth"][index],
+ description: `Documentation ${index}`,
+ }),
+ ),
+ );
+ const first = {
+ which: choice(
+ { skill_0: 0.02, skill_1: 0.15, skill_2: 0.15, skill_3: 0.25, skill_4: 0.03, none: 0.4 },
+ "skill_0",
+ 0.01,
+ ),
+ needs_skill: noul(0.9),
+ };
+ const second = {
+ which: choice({ skill_3: 0.6, skill_1: 0.2, skill_2: 0.1, none: 0.1 }, "skill_3"),
+ fits_skill_3: noul(0.8),
+ fits_skill_1: noul(0.9),
+ fits_skill_2: noul(0.9),
+ };
+ const options = setup();
+ options.fetch.mockReset().mockResolvedValueOnce(response(first)).mockResolvedValueOnce(response(second));
+ options.loadExcerpt.mockResolvedValue(`${"E".repeat(700)}EXCERPT_TAIL_MUST_NOT_LEAK`);
+
+ await expect(suggestJevSkill(REQUEST, skills, options)).resolves.toEqual({
+ reason: "selected",
+ skillName: "__proto__",
+ });
+ expect(options.fetch).toHaveBeenCalledTimes(2);
+ expect(options.loadExcerpt.mock.calls.map(([skill]) => skill.name)).toEqual([
+ "__proto__",
+ "skill_0",
+ "constructor",
+ ]);
+ const body = requestBody(options.fetch, 1);
+ expect(Object.keys(body.questions)).toEqual(["which", "fits_skill_3", "fits_skill_1", "fits_skill_2"]);
+ expect(Object.keys(body.questions.which.criteria)).toEqual(["skill_3", "skill_1", "skill_2", "none"]);
+ for (const id of ["skill_3", "skill_1", "skill_2"]) {
+ expect(JSON.parse(body.questions.which.criteria[id]).excerpt).toBe("E".repeat(700));
+ }
+ expect(JSON.stringify(body)).not.toContain("EXCERPT_TAIL_MUST_NOT_LEAK");
+ expect(skills.map((skill) => skill.name)).toEqual(["none", "skill_0", "constructor", "__proto__", "fifth"]);
+ });
+
+ test("uses global fetch when no transport is injected", async () => {
+ const options = setup();
+ vi.stubGlobal("fetch", options.fetch);
+ await expect(
+ suggestJevSkill(REQUEST, SKILLS, { apiKey: API_KEY, loadExcerpt: options.loadExcerpt }),
+ ).resolves.toEqual({
+ reason: "selected",
+ skillName: "debug",
+ });
+ expect(options.fetch).toHaveBeenCalledTimes(2);
+ });
+});
+
+describe("input bounds", () => {
+ test.each(["", " \n\t", "x".repeat(12001), null, 42])(
+ "rejects unsupported request %# before doing any work",
+ async (request) => {
+ const options = setup();
+ await expect(suggestJevSkill(request as string, SKILLS, options)).resolves.toEqual({
+ reason: "unsupported-input",
+ });
+ expect(options.fetch).not.toHaveBeenCalled();
+ expect(options.loadExcerpt).not.toHaveBeenCalled();
+ },
+ );
+
+ test.each([0, 1, 255])("rejects a catalog of %s skills without contacting Jev", async (count) => {
+ const options = setup();
+ const skills = Array.from({ length: count }, (_, index) => ({
+ name: `name-${index}`,
+ description: "Description",
+ }));
+ await expect(suggestJevSkill(REQUEST, skills, options)).resolves.toEqual({ reason: "unsupported-input" });
+ expect(options.fetch).not.toHaveBeenCalled();
+ expect(options.loadExcerpt).not.toHaveBeenCalled();
+ });
+
+ test.each([
+ null,
+ {},
+ [SKILLS[0], SKILLS[0]],
+ [SKILLS[0], { name: "", description: "Empty name" }],
+ [SKILLS[0], { name: " \t", description: "Blank name" }],
+ [SKILLS[0], { name: 42, description: "Non-string name" }],
+ [SKILLS[0], { name: "missing-description" }],
+ [SKILLS[0], { name: "invalid-description", description: 12 }],
+ [SKILLS[0], null],
+ new Array(2),
+ ])("rejects invalid or duplicate skill metadata %# without throwing", async (skills) => {
+ const options = setup();
+ await expect(suggestJevSkill(REQUEST, skills as readonly JevSkill[], options)).resolves.toEqual({
+ reason: "unsupported-input",
+ });
+ expect(options.fetch).not.toHaveBeenCalled();
+ expect(options.loadExcerpt).not.toHaveBeenCalled();
+ });
+
+ test.each(["", " \n\t", undefined, null, 42])("rejects invalid API key %# without HTTP", async (apiKey) => {
+ const options = setup();
+ await expect(suggestJevSkill(REQUEST, SKILLS, { ...options, apiKey: apiKey as string })).resolves.toEqual({
+ reason: "unsupported-input",
+ });
+ expect(options.fetch).not.toHaveBeenCalled();
+ expect(options.loadExcerpt).not.toHaveBeenCalled();
+ });
+
+ test.each([0, -1, Number.NaN, Number.POSITIVE_INFINITY, 2 ** 31, "100", null])(
+ "rejects invalid timeout %# without HTTP",
+ async (timeoutMs) => {
+ const options = setup();
+ await expect(
+ suggestJevSkill(REQUEST, SKILLS, { ...options, timeoutMs: timeoutMs as number }),
+ ).resolves.toEqual({ reason: "unsupported-input" });
+ expect(options.fetch).not.toHaveBeenCalled();
+ expect(options.loadExcerpt).not.toHaveBeenCalled();
+ },
+ );
+
+ test.each([null, undefined, {}, { apiKey: API_KEY, loadExcerpt: "not a function" }])(
+ "rejects invalid options %# without throwing",
+ async (invalid) => {
+ const options = setup();
+ vi.stubGlobal("fetch", options.fetch);
+ await expect(suggestJevSkill(REQUEST, SKILLS, invalid as unknown as JevRoutingOptions)).resolves.toEqual({
+ reason: "unsupported-input",
+ });
+ expect(options.fetch).not.toHaveBeenCalled();
+ },
+ );
+
+ test("accepts exactly 254 skills and a 12000-character request without truncating either", async () => {
+ const skills = Array.from({ length: 254 }, (_, index) => ({
+ name: `name-${index}`,
+ description: `Full description ${index}`,
+ }));
+ const request = ` ${"x".repeat(11998)} `;
+ const probabilities = Object.fromEntries(skills.map((_, index) => [`skill_${index}`, 0]));
+ const options = setup();
+ options.fetch.mockReset().mockResolvedValueOnce(
+ response({
+ which: choice({ ...probabilities, none: 1 }, "none"),
+ needs_skill: noul(0),
+ }),
+ );
+ await expect(suggestJevSkill(request, skills, options)).resolves.toEqual({ reason: "no-match" });
+ const body = requestBody(options.fetch, 0);
+ expect(body.state).toEqual({ request });
+ expect(Object.keys(body.questions.which.criteria)).toHaveLength(255);
+ expect(JSON.parse(body.questions.which.criteria.skill_253)).toEqual(skills[253]);
+ expect(options.fetch).toHaveBeenCalledTimes(1);
+ expect(options.loadExcerpt).not.toHaveBeenCalled();
+ });
+});
+
+describe("response validation", () => {
+ const malformedChoices: [string, unknown][] = [
+ ["missing choice answer", undefined],
+ ["null answer", null],
+ ["array answer", []],
+ ["wrong answer type", { ...firstAnswers().which, type: "score" }],
+ ["unknown chosen ID", { ...firstAnswers().which, choice: "skill_999" }],
+ ["skill name used as ID", { ...firstAnswers().which, choice: "debug" }],
+ ["non-string chosen ID", { ...firstAnswers().which, choice: 0 }],
+ ["prototype ID", { ...firstAnswers().which, choice: "constructor" }],
+ ["missing probabilities", { ...firstAnswers().which, probabilities: undefined }],
+ ["array probabilities", { ...firstAnswers().which, probabilities: [0.7, 0.2, 0.1] }],
+ ["missing none probability", { ...firstAnswers().which, probabilities: { skill_0: 0.8, skill_1: 0.2 } }],
+ ["missing candidate probability", { ...firstAnswers().which, probabilities: { skill_0: 0.9, none: 0.1 } }],
+ [
+ "extra probability",
+ { ...firstAnswers().which, probabilities: { skill_0: 0.7, skill_1: 0.2, none: 0.1, extra: 0 } },
+ ],
+ [
+ "unknown probability replacing candidate",
+ { ...firstAnswers().which, probabilities: { skill_0: 0.7, other: 0.2, none: 0.1 } },
+ ],
+ ["negative probability", { ...firstAnswers().which, probabilities: { skill_0: 0.9, skill_1: 0.2, none: -0.1 } }],
+ ["probability over one", { ...firstAnswers().which, probabilities: { skill_0: 1.1, skill_1: 0, none: 0 } }],
+ ["string probability", { ...firstAnswers().which, probabilities: { skill_0: "0.7", skill_1: 0.2, none: 0.1 } }],
+ ["null probability", { ...firstAnswers().which, probabilities: { skill_0: null, skill_1: 0.9, none: 0.1 } }],
+ [
+ "unnormalized distribution",
+ { ...firstAnswers().which, probabilities: { skill_0: 0.1, skill_1: 0.1, none: 0.1 } },
+ ],
+ ["missing confidence", { ...firstAnswers().which, confidence: undefined }],
+ ["negative confidence", { ...firstAnswers().which, confidence: -0.01 }],
+ ["confidence over one", { ...firstAnswers().which, confidence: 1.01 }],
+ ["string confidence", { ...firstAnswers().which, confidence: "0.9" }],
+ ["null confidence", { ...firstAnswers().which, confidence: null }],
+ ];
+
+ test.each(malformedChoices)("rejects first-pass %s before reading excerpts", async (_label, which) => {
+ const options = setup();
+ options.fetch.mockReset().mockResolvedValueOnce(response({ ...firstAnswers(), which }));
+ await expect(suggestJevSkill(REQUEST, SKILLS, options)).resolves.toEqual({ reason: "unavailable" });
+ expect(options.fetch).toHaveBeenCalledTimes(1);
+ expect(options.loadExcerpt).not.toHaveBeenCalled();
+ });
+
+ test.each(malformedChoices)("rejects second-pass %s", async (_label, which) => {
+ const options = setup();
+ options.fetch
+ .mockReset()
+ .mockResolvedValueOnce(response(firstAnswers()))
+ .mockResolvedValueOnce(response({ ...secondAnswers(), which }));
+ await expect(suggestJevSkill(REQUEST, SKILLS, options)).resolves.toEqual({ reason: "unavailable" });
+ expect(options.fetch).toHaveBeenCalledTimes(2);
+ });
+
+ const malformedNouls: [string, unknown][] = [
+ ["missing", undefined],
+ ["null", null],
+ ["array", []],
+ ["wrong type", { type: "score", noul: 0.9 }],
+ ["missing noul", { type: "noul", confidence: 0.9 }],
+ ["string noul", { type: "noul", noul: "0.9" }],
+ ["negative noul", noul(-0.1)],
+ ["noul over one", noul(1.1)],
+ ];
+
+ test.each(malformedNouls)("rejects %s needs_skill even when the choice is none", async (_label, needs_skill) => {
+ const options = setup();
+ options.fetch.mockReset().mockResolvedValueOnce(response({ ...firstAnswers("none"), needs_skill }));
+ await expect(suggestJevSkill(REQUEST, SKILLS, options)).resolves.toEqual({ reason: "unavailable" });
+ expect(options.fetch).toHaveBeenCalledTimes(1);
+ expect(options.loadExcerpt).not.toHaveBeenCalled();
+ });
+
+ test.each(malformedNouls)("rejects %s fit for an unchosen candidate", async (_label, fits_skill_1) => {
+ const options = setup();
+ options.fetch
+ .mockReset()
+ .mockResolvedValueOnce(response(firstAnswers()))
+ .mockResolvedValueOnce(response({ ...secondAnswers(), fits_skill_1 }));
+ await expect(suggestJevSkill(REQUEST, SKILLS, options)).resolves.toEqual({ reason: "unavailable" });
+ expect(options.fetch).toHaveBeenCalledTimes(2);
+ });
+
+ test.each([null, [], {}, { answers: null }, { answers: [] }, { answers: "wrong" }])(
+ "rejects malformed response envelope %#",
+ async (payload) => {
+ const options = setup();
+ options.fetch.mockReset().mockResolvedValueOnce(Response.json(payload));
+ await expect(suggestJevSkill(REQUEST, SKILLS, options)).resolves.toEqual({ reason: "unavailable" });
+ expect(options.fetch).toHaveBeenCalledTimes(1);
+ expect(options.loadExcerpt).not.toHaveBeenCalled();
+ },
+ );
+
+ test.each([Number.NaN, Number.POSITIVE_INFINITY, Number.NEGATIVE_INFINITY])(
+ "rejects non-finite numerical answers %# from injected JSON readers",
+ async (value) => {
+ for (const field of ["probability", "confidence", "noul"]) {
+ const answers = firstAnswers();
+ if (field === "probability") answers.which.probabilities.skill_0 = value;
+ if (field === "confidence") answers.which.confidence = value;
+ if (field === "noul") answers.needs_skill.noul = value;
+ const reply = response(firstAnswers());
+ vi.spyOn(reply, "json").mockResolvedValue({ answers });
+ const options = setup();
+ options.fetch.mockReset().mockResolvedValueOnce(reply);
+ await expect(suggestJevSkill(REQUEST, SKILLS, options)).resolves.toEqual({ reason: "unavailable" });
+ expect(options.fetch).toHaveBeenCalledTimes(1);
+ expect(options.loadExcerpt).not.toHaveBeenCalled();
+ }
+ },
+ );
+
+ test("allows small probability rounding error", async () => {
+ const first = firstAnswers();
+ first.which.probabilities = { skill_0: 0.7, skill_1: 0.2, none: 0.099 };
+ const second = secondAnswers();
+ second.which.probabilities = { skill_0: 0.7, skill_1: 0.2, none: 0.101 };
+ await expect(suggestJevSkill(REQUEST, SKILLS, setup(first, second))).resolves.toEqual({
+ reason: "selected",
+ skillName: "debug",
+ });
+ });
+
+ test("rejects a valid catalog skill that was not shortlisted", async () => {
+ const skills = [...SKILLS, { name: "third", description: "Third" }, { name: "fourth", description: "Fourth" }];
+ const options = setup();
+ options.fetch
+ .mockReset()
+ .mockResolvedValueOnce(
+ response({
+ which: choice({ skill_0: 0.5, skill_1: 0.25, skill_2: 0.15, skill_3: 0.05, none: 0.05 }),
+ needs_skill: noul(1),
+ }),
+ )
+ .mockResolvedValueOnce(
+ response({
+ which: choice({ skill_0: 0.5, skill_1: 0.3, skill_2: 0.1, none: 0.1 }, "skill_3"),
+ fits_skill_0: noul(1),
+ fits_skill_1: noul(1),
+ fits_skill_2: noul(1),
+ fits_skill_3: noul(1),
+ }),
+ );
+ await expect(suggestJevSkill(REQUEST, skills, options)).resolves.toEqual({ reason: "unavailable" });
+ expect(options.fetch).toHaveBeenCalledTimes(2);
+ });
+});
+
+describe("fail-open errors", () => {
+ test.each([301, 302, 307, 308, 401, 429, 500, 503, 529])(
+ "stops without retrying HTTP %s in either pass",
+ async (status) => {
+ for (const pass of [1, 2]) {
+ const options = setup();
+ options.fetch.mockReset();
+ if (pass === 2) options.fetch.mockResolvedValueOnce(response(firstAnswers()));
+ const failed = new Response(API_KEY, {
+ status,
+ headers: { Location: "https://other.example", "Retry-After": "0" },
+ });
+ const read = vi.spyOn(failed, "json");
+ options.fetch.mockResolvedValueOnce(failed);
+ await expect(suggestJevSkill(REQUEST, SKILLS, options)).resolves.toEqual({ reason: "unavailable" });
+ expect(options.fetch).toHaveBeenCalledTimes(pass);
+ expect(options.loadExcerpt).toHaveBeenCalledTimes(pass === 1 ? 0 : 2);
+ expect(read).not.toHaveBeenCalled();
+ }
+ },
+ );
+
+ test("rejects a redirected response from an injected transport", async () => {
+ const reply = response(firstAnswers());
+ Object.defineProperty(reply, "redirected", { value: true });
+ const options = setup();
+ options.fetch.mockReset().mockResolvedValueOnce(reply);
+ await expect(suggestJevSkill(REQUEST, SKILLS, options)).resolves.toEqual({ reason: "unavailable" });
+ expect(options.fetch).toHaveBeenCalledTimes(1);
+ expect(options.loadExcerpt).not.toHaveBeenCalled();
+ });
+
+ test.each([1, 2])("handles network rejection in pass %s without retrying or logging secrets", async (pass) => {
+ const logs = ["error", "warn", "log", "info", "debug"].map((method) =>
+ vi.spyOn(console, method as "error").mockImplementation(() => {}),
+ );
+ const options = setup();
+ options.fetch.mockReset();
+ if (pass === 2) options.fetch.mockResolvedValueOnce(response(firstAnswers()));
+ options.fetch.mockRejectedValueOnce(new Error(`Transport failed with ${API_KEY}`));
+ await expect(suggestJevSkill(REQUEST, SKILLS, options)).resolves.toEqual({ reason: "unavailable" });
+ expect(options.fetch).toHaveBeenCalledTimes(pass);
+ for (const log of logs) expect(log).not.toHaveBeenCalled();
+ });
+
+ test.each([1, 2])("handles invalid JSON in pass %s", async (pass) => {
+ const options = setup();
+ options.fetch.mockReset();
+ if (pass === 2) options.fetch.mockResolvedValueOnce(response(firstAnswers()));
+ options.fetch.mockResolvedValueOnce(new Response("{invalid json", { status: 200 }));
+ await expect(suggestJevSkill(REQUEST, SKILLS, options)).resolves.toEqual({ reason: "unavailable" });
+ expect(options.fetch).toHaveBeenCalledTimes(pass);
+ });
+
+ test.each(["sync", "async"])("handles a %s excerpt failure without the second HTTP request", async (kind) => {
+ const options = setup();
+ if (kind === "sync")
+ options.loadExcerpt.mockImplementation(() => {
+ throw new Error("read failed");
+ });
+ else options.loadExcerpt.mockRejectedValue(new Error("read failed"));
+ await expect(suggestJevSkill(REQUEST, SKILLS, options)).resolves.toEqual({ reason: "unavailable" });
+ expect(options.fetch).toHaveBeenCalledTimes(1);
+ });
+
+ test("handles a non-string excerpt as unavailable", async () => {
+ const options = setup();
+ options.loadExcerpt.mockResolvedValue(null as unknown as string);
+ await expect(suggestJevSkill(REQUEST, SKILLS, options)).resolves.toEqual({ reason: "unavailable" });
+ expect(options.fetch).toHaveBeenCalledTimes(1);
+ });
+});
+
+function deferred() {
+ let resolve!: (value: T) => void;
+ let reject!: (reason: unknown) => void;
+ const promise = new Promise((resolvePromise, rejectPromise) => {
+ resolve = resolvePromise;
+ reject = rejectPromise;
+ });
+ return { promise, resolve, reject };
+}
+
+describe("one shared deadline", () => {
+ test("returns at the default deadline when the first transport ignores abort and ignores its late response", async () => {
+ vi.useFakeTimers();
+ const first = deferred();
+ const options = setup();
+ options.fetch.mockReset().mockReturnValueOnce(first.promise);
+ const settled = vi.fn();
+ const result = suggestJevSkill(REQUEST, SKILLS, options);
+ void result.then(settled);
+ const signal = options.fetch.mock.calls[0][1]!.signal!;
+
+ await vi.advanceTimersByTimeAsync(1499);
+ expect(settled).not.toHaveBeenCalled();
+ expect(signal.aborted).toBe(false);
+ await vi.advanceTimersByTimeAsync(1);
+ expect(settled).toHaveBeenCalledExactlyOnceWith({ reason: "unavailable" });
+ expect(signal.aborted).toBe(true);
+ expect(vi.getTimerCount()).toBe(0);
+
+ const late = response(firstAnswers());
+ const parse = vi.spyOn(late, "json");
+ first.resolve(late);
+ await vi.advanceTimersByTimeAsync(0);
+ expect(parse).not.toHaveBeenCalled();
+ expect(options.loadExcerpt).not.toHaveBeenCalled();
+ expect(options.fetch).toHaveBeenCalledTimes(1);
+ });
+
+ test("includes the first request's time in the excerpt budget and never sends a late second request", async () => {
+ vi.useFakeTimers();
+ const first = deferred();
+ const excerpt = deferred();
+ const options = setup();
+ options.fetch.mockReset().mockReturnValueOnce(first.promise);
+ options.loadExcerpt.mockReturnValue(excerpt.promise);
+ const settled = vi.fn();
+ void suggestJevSkill(REQUEST, SKILLS, { ...options, timeoutMs: 1000 }).then(settled);
+ const signal = options.fetch.mock.calls[0][1]!.signal!;
+
+ await vi.advanceTimersByTimeAsync(600);
+ first.resolve(response(firstAnswers()));
+ await vi.advanceTimersByTimeAsync(0);
+ expect(options.loadExcerpt).toHaveBeenCalledTimes(2);
+ for (const [, excerptSignal] of options.loadExcerpt.mock.calls) expect(excerptSignal).toBe(signal);
+ await vi.advanceTimersByTimeAsync(399);
+ expect(settled).not.toHaveBeenCalled();
+ await vi.advanceTimersByTimeAsync(1);
+ expect(settled).toHaveBeenCalledExactlyOnceWith({ reason: "unavailable" });
+ expect(signal.aborted).toBe(true);
+ expect(vi.getTimerCount()).toBe(0);
+
+ excerpt.resolve("Eventually read, despite ignoring abort.");
+ await vi.advanceTimersByTimeAsync(0);
+ expect(options.fetch).toHaveBeenCalledTimes(1);
+ expect(settled).toHaveBeenCalledTimes(1);
+ });
+
+ test("includes both earlier phases in the second request's budget and uses the identical signal", async () => {
+ vi.useFakeTimers();
+ const first = deferred();
+ const excerpt = deferred();
+ const second = deferred();
+ const options = setup();
+ options.fetch.mockReset().mockReturnValueOnce(first.promise).mockReturnValueOnce(second.promise);
+ options.loadExcerpt.mockReturnValue(excerpt.promise);
+ const settled = vi.fn();
+ void suggestJevSkill(REQUEST, SKILLS, options).then(settled);
+ const signal = options.fetch.mock.calls[0][1]!.signal!;
+
+ await vi.advanceTimersByTimeAsync(500);
+ first.resolve(response(firstAnswers()));
+ await vi.advanceTimersByTimeAsync(500);
+ excerpt.resolve("Documented instructions.");
+ await vi.advanceTimersByTimeAsync(0);
+ expect(options.fetch).toHaveBeenCalledTimes(2);
+ expect(options.fetch.mock.calls[1][1]!.signal).toBe(signal);
+ for (const [, excerptSignal] of options.loadExcerpt.mock.calls) expect(excerptSignal).toBe(signal);
+ await vi.advanceTimersByTimeAsync(499);
+ expect(settled).not.toHaveBeenCalled();
+ await vi.advanceTimersByTimeAsync(1);
+ expect(settled).toHaveBeenCalledExactlyOnceWith({ reason: "unavailable" });
+ expect(signal.aborted).toBe(true);
+ expect(vi.getTimerCount()).toBe(0);
+
+ const late = response(secondAnswers());
+ const parse = vi.spyOn(late, "json");
+ second.resolve(late);
+ await vi.advanceTimersByTimeAsync(0);
+ expect(parse).not.toHaveBeenCalled();
+ expect(options.fetch).toHaveBeenCalledTimes(2);
+ expect(settled).toHaveBeenCalledTimes(1);
+ });
+
+ test.each([1, 2])("bounds JSON body parsing that ignores abort in pass %s", async (pass) => {
+ vi.useFakeTimers();
+ const body = deferred();
+ const reply = response(pass === 1 ? firstAnswers() : secondAnswers());
+ vi.spyOn(reply, "json").mockReturnValue(body.promise);
+ const options = setup();
+ options.fetch.mockReset();
+ if (pass === 2) options.fetch.mockResolvedValueOnce(response(firstAnswers()));
+ options.fetch.mockResolvedValueOnce(reply);
+ const settled = vi.fn();
+ void suggestJevSkill(REQUEST, SKILLS, { ...options, timeoutMs: 100 }).then(settled);
+
+ await vi.advanceTimersByTimeAsync(100);
+ expect(settled).toHaveBeenCalledExactlyOnceWith({ reason: "unavailable" });
+ expect(options.fetch.mock.calls[0][1]!.signal!.aborted).toBe(true);
+ expect(vi.getTimerCount()).toBe(0);
+ body.resolve({ answers: pass === 1 ? firstAnswers() : secondAnswers() });
+ await vi.advanceTimersByTimeAsync(0);
+ expect(options.fetch).toHaveBeenCalledTimes(pass);
+ if (pass === 1) expect(options.loadExcerpt).not.toHaveBeenCalled();
+ });
+
+ test("does not send the second request if elapsed time exceeds the budget before a timer callback runs", async () => {
+ vi.useFakeTimers();
+ let elapsed = 0;
+ vi.spyOn(performance, "now").mockImplementation(() => elapsed);
+ const options = setup();
+ options.loadExcerpt.mockImplementation(async () => {
+ elapsed = 1500;
+ return "A read that consumed the remaining wall-clock budget.";
+ });
+ await expect(suggestJevSkill(REQUEST, SKILLS, options)).resolves.toEqual({ reason: "unavailable" });
+ expect(options.fetch).toHaveBeenCalledTimes(1);
+ expect(options.fetch.mock.calls[0][1]!.signal!.aborted).toBe(true);
+ expect(vi.getTimerCount()).toBe(0);
+ });
+
+ test("does not accept an answer parsed after the absolute deadline even before timers run", async () => {
+ vi.useFakeTimers();
+ let elapsed = 0;
+ vi.spyOn(performance, "now").mockImplementation(() => elapsed);
+ const late = response(secondAnswers());
+ vi.spyOn(late, "json").mockImplementation(async () => {
+ elapsed = 1500;
+ return { answers: secondAnswers() };
+ });
+ const options = setup();
+ options.fetch.mockReset().mockResolvedValueOnce(response(firstAnswers())).mockResolvedValueOnce(late);
+ await expect(suggestJevSkill(REQUEST, SKILLS, options)).resolves.toEqual({ reason: "unavailable" });
+ expect(options.fetch).toHaveBeenCalledTimes(2);
+ expect(vi.getTimerCount()).toBe(0);
+ });
+
+ test.each(["selected", "no-match", "low-confidence", "unavailable", "unsupported-input"])(
+ "clears its timer when returning %s early",
+ async (reason) => {
+ vi.useFakeTimers();
+ const options = setup(
+ firstAnswers(reason === "no-match" ? "none" : "skill_0"),
+ secondAnswers("skill_0", reason === "low-confidence" ? 0.5 : 0.9),
+ );
+ if (reason === "unavailable")
+ options.fetch.mockReset().mockResolvedValueOnce(new Response(null, { status: 401 }));
+ const result = await suggestJevSkill(reason === "unsupported-input" ? "" : REQUEST, SKILLS, options);
+ expect(result.reason).toBe(reason);
+ expect(vi.getTimerCount()).toBe(0);
+ const calls = options.fetch.mock.calls.length;
+ await vi.advanceTimersByTimeAsync(3000);
+ expect(options.fetch).toHaveBeenCalledTimes(calls);
+ if (["selected", "no-match", "low-confidence"].includes(reason)) {
+ expect(options.fetch.mock.calls[0][1]!.signal!.aborted).toBe(false);
+ }
+ },
+ );
+
+ test.each(["resolve", "reject"])(
+ "aborts other reads after an excerpt fails and consumes their late %s",
+ async (outcome) => {
+ vi.useFakeTimers();
+ const other = deferred();
+ const options = setup();
+ options.loadExcerpt.mockRejectedValueOnce(new Error("First read failed")).mockReturnValueOnce(other.promise);
+ await expect(suggestJevSkill(REQUEST, SKILLS, options)).resolves.toEqual({ reason: "unavailable" });
+ expect(options.loadExcerpt).toHaveBeenCalledTimes(2);
+ expect(options.loadExcerpt.mock.calls[1][1].aborted).toBe(true);
+ expect(vi.getTimerCount()).toBe(0);
+ if (outcome === "resolve") other.resolve("Too late");
+ else other.reject(new Error("Late read failure"));
+ await vi.advanceTimersByTimeAsync(0);
+ expect(options.fetch).toHaveBeenCalledTimes(1);
+ },
+ );
+});
+
+describe("caller cancellation", () => {
+ test("returns unavailable for an initially aborted signal without HTTP, reads, listeners, or timers", async () => {
+ vi.useFakeTimers();
+ const caller = new AbortController();
+ caller.abort(new Error("Already cancelled"));
+ const options = setup();
+ await expect(suggestJevSkill(REQUEST, SKILLS, { ...options, signal: caller.signal })).resolves.toEqual({
+ reason: "unavailable",
+ });
+ expect(options.fetch).not.toHaveBeenCalled();
+ expect(options.loadExcerpt).not.toHaveBeenCalled();
+ expect(getEventListeners(caller.signal, "abort")).toEqual([]);
+ expect(vi.getTimerCount()).toBe(0);
+ });
+
+ test.each([
+ null,
+ false,
+ 42,
+ "cancel",
+ {},
+ new AbortController(),
+ { aborted: true },
+ { aborted: false, addEventListener() {}, removeEventListener() {} },
+ Object.create(AbortSignal.prototype),
+ Object.create(AbortSignal.prototype, { aborted: { value: false } }),
+ ])("rejects invalid signal %# without throwing or starting work", async (signal) => {
+ vi.useFakeTimers();
+ const options = setup();
+ await expect(suggestJevSkill(REQUEST, SKILLS, { ...options, signal: signal as AbortSignal })).resolves.toEqual({
+ reason: "unsupported-input",
+ });
+ expect(options.fetch).not.toHaveBeenCalled();
+ expect(options.loadExcerpt).not.toHaveBeenCalled();
+ expect(vi.getTimerCount()).toBe(0);
+ });
+
+ test.each([1, 2])(
+ "immediately cancels HTTP pass %s that ignores abort and discards its late response",
+ async (pass) => {
+ vi.useFakeTimers();
+ const caller = new AbortController();
+ const transport = deferred();
+ const options = setup();
+ options.fetch.mockReset();
+ if (pass === 2) options.fetch.mockResolvedValueOnce(response(firstAnswers()));
+ options.fetch.mockReturnValueOnce(transport.promise);
+ const settled = vi.fn();
+ void suggestJevSkill(REQUEST, SKILLS, { ...options, signal: caller.signal }).then(settled);
+ await vi.advanceTimersByTimeAsync(0);
+ expect(options.fetch).toHaveBeenCalledTimes(pass);
+ const requestSignal = options.fetch.mock.calls[0][1]!.signal!;
+ for (const [, init] of options.fetch.mock.calls) expect(init!.signal).toBe(requestSignal);
+ for (const [, signal] of options.loadExcerpt.mock.calls) expect(signal).toBe(requestSignal);
+
+ caller.abort(new Error("Session cancelled"));
+ expect(requestSignal.aborted).toBe(true);
+ await vi.advanceTimersByTimeAsync(0);
+ expect(settled).toHaveBeenCalledExactlyOnceWith({ reason: "unavailable" });
+ expect(getEventListeners(caller.signal, "abort")).toEqual([]);
+ expect(vi.getTimerCount()).toBe(0);
+
+ const late = response(pass === 1 ? firstAnswers() : secondAnswers());
+ const parse = vi.spyOn(late, "json");
+ transport.resolve(late);
+ await vi.advanceTimersByTimeAsync(0);
+ expect(parse).not.toHaveBeenCalled();
+ expect(options.fetch).toHaveBeenCalledTimes(pass);
+ expect(options.loadExcerpt).toHaveBeenCalledTimes(pass === 1 ? 0 : 2);
+ expect(settled).toHaveBeenCalledTimes(1);
+ },
+ );
+
+ test.each([1, 2])("immediately cancels JSON parsing in pass %s even when it ignores abort", async (pass) => {
+ vi.useFakeTimers();
+ const caller = new AbortController();
+ const body = deferred();
+ const reply = response(pass === 1 ? firstAnswers() : secondAnswers());
+ const parse = vi.spyOn(reply, "json").mockReturnValue(body.promise);
+ const options = setup();
+ options.fetch.mockReset();
+ if (pass === 2) options.fetch.mockResolvedValueOnce(response(firstAnswers()));
+ options.fetch.mockResolvedValueOnce(reply);
+ const settled = vi.fn();
+ void suggestJevSkill(REQUEST, SKILLS, { ...options, signal: caller.signal }).then(settled);
+ await vi.advanceTimersByTimeAsync(0);
+ expect(parse).toHaveBeenCalledTimes(1);
+
+ caller.abort();
+ expect(options.fetch.mock.calls[0][1]!.signal!.aborted).toBe(true);
+ await vi.advanceTimersByTimeAsync(0);
+ expect(settled).toHaveBeenCalledExactlyOnceWith({ reason: "unavailable" });
+ expect(getEventListeners(caller.signal, "abort")).toEqual([]);
+ expect(vi.getTimerCount()).toBe(0);
+ body.resolve({ answers: pass === 1 ? firstAnswers() : secondAnswers() });
+ await vi.advanceTimersByTimeAsync(0);
+ expect(options.fetch).toHaveBeenCalledTimes(pass);
+ expect(options.loadExcerpt).toHaveBeenCalledTimes(pass === 1 ? 0 : 2);
+ expect(settled).toHaveBeenCalledTimes(1);
+ });
+
+ test.each(["resolve", "reject"])(
+ "cancels pending excerpts and consumes their late %s without a second request",
+ async (outcome) => {
+ vi.useFakeTimers();
+ const caller = new AbortController();
+ const excerpt = deferred();
+ const options = setup();
+ options.loadExcerpt.mockReturnValue(excerpt.promise);
+ const settled = vi.fn();
+ void suggestJevSkill(REQUEST, SKILLS, { ...options, signal: caller.signal }).then(settled);
+ await vi.advanceTimersByTimeAsync(0);
+ expect(options.loadExcerpt).toHaveBeenCalledTimes(2);
+ const requestSignal = options.fetch.mock.calls[0][1]!.signal!;
+ for (const [, signal] of options.loadExcerpt.mock.calls) expect(signal).toBe(requestSignal);
+
+ caller.abort();
+ expect(requestSignal.aborted).toBe(true);
+ await vi.advanceTimersByTimeAsync(0);
+ expect(settled).toHaveBeenCalledExactlyOnceWith({ reason: "unavailable" });
+ expect(getEventListeners(caller.signal, "abort")).toEqual([]);
+ expect(vi.getTimerCount()).toBe(0);
+ if (outcome === "resolve") excerpt.resolve("Late documentation");
+ else excerpt.reject(new Error("Late read failure"));
+ await vi.advanceTimersByTimeAsync(0);
+ expect(options.fetch).toHaveBeenCalledTimes(1);
+ expect(settled).toHaveBeenCalledTimes(1);
+ },
+ );
+
+ test.each(["first-fetch", "excerpt", "second-body"])(
+ "prevents selection when %s cancels synchronously while completing",
+ async (phase) => {
+ vi.useFakeTimers();
+ const caller = new AbortController();
+ const options = setup();
+ if (phase === "first-fetch") {
+ options.fetch.mockReset().mockImplementationOnce(async () => {
+ caller.abort();
+ return response(firstAnswers());
+ });
+ } else if (phase === "excerpt") {
+ options.loadExcerpt.mockImplementation(async () => {
+ caller.abort();
+ return "Completed documentation";
+ });
+ } else {
+ const reply = response(secondAnswers());
+ vi.spyOn(reply, "json").mockImplementation(async () => {
+ caller.abort();
+ return { answers: secondAnswers() };
+ });
+ options.fetch.mockReset().mockResolvedValueOnce(response(firstAnswers())).mockResolvedValueOnce(reply);
+ }
+ await expect(suggestJevSkill(REQUEST, SKILLS, { ...options, signal: caller.signal })).resolves.toEqual({
+ reason: "unavailable",
+ });
+ expect(options.fetch).toHaveBeenCalledTimes(phase === "second-body" ? 2 : 1);
+ expect(options.loadExcerpt).toHaveBeenCalledTimes(phase === "first-fetch" ? 0 : phase === "excerpt" ? 1 : 2);
+ expect(options.fetch.mock.calls[0][1]!.signal!.aborted).toBe(true);
+ expect(getEventListeners(caller.signal, "abort")).toEqual([]);
+ expect(vi.getTimerCount()).toBe(0);
+ },
+ );
+
+ test.each(["selected", "first-none", "second-none", "low-confidence", "unavailable", "unsupported-input"])(
+ "removes only its own external listener and timer after %s",
+ async (outcome) => {
+ vi.useFakeTimers();
+ const caller = new AbortController();
+ const otherListener = vi.fn();
+ caller.signal.addEventListener("abort", otherListener);
+ const options = setup(
+ firstAnswers(outcome === "first-none" ? "none" : "skill_0"),
+ secondAnswers(outcome === "second-none" ? "none" : "skill_0", outcome === "low-confidence" ? 0.5 : 0.9),
+ );
+ if (outcome === "unavailable")
+ options.fetch.mockReset().mockResolvedValueOnce(new Response(null, { status: 401 }));
+ const result = await suggestJevSkill(outcome === "unsupported-input" ? "" : REQUEST, SKILLS, {
+ ...options,
+ signal: caller.signal,
+ });
+ expect(result.reason).toBe(outcome.endsWith("-none") ? "no-match" : outcome);
+ expect(getEventListeners(caller.signal, "abort")).toEqual([otherListener]);
+ expect(vi.getTimerCount()).toBe(0);
+
+ const requestSignal = options.fetch.mock.calls[0]?.[1]?.signal;
+ const wasAborted = requestSignal?.aborted;
+ const calls = options.fetch.mock.calls.length;
+ caller.abort();
+ await vi.advanceTimersByTimeAsync(3000);
+ expect(otherListener).toHaveBeenCalledTimes(1);
+ expect(requestSignal?.aborted).toBe(wasAborted);
+ expect(options.fetch).toHaveBeenCalledTimes(calls);
+ },
+ );
+
+ test("preserves the shared 1500ms deadline with an active caller signal and detaches on timeout", async () => {
+ vi.useFakeTimers();
+ const caller = new AbortController();
+ const first = deferred();
+ const options = setup();
+ options.fetch.mockReset().mockReturnValueOnce(first.promise);
+ const settled = vi.fn();
+ void suggestJevSkill(REQUEST, SKILLS, { ...options, signal: caller.signal }).then(settled);
+ expect(getEventListeners(caller.signal, "abort")).toHaveLength(1);
+ await vi.advanceTimersByTimeAsync(1499);
+ expect(settled).not.toHaveBeenCalled();
+ await vi.advanceTimersByTimeAsync(1);
+ expect(settled).toHaveBeenCalledExactlyOnceWith({ reason: "unavailable" });
+ expect(options.fetch.mock.calls[0][1]!.signal!.aborted).toBe(true);
+ expect(caller.signal.aborted).toBe(false);
+ expect(getEventListeners(caller.signal, "abort")).toEqual([]);
+ expect(vi.getTimerCount()).toBe(0);
+ first.reject(new Error("Late transport failure"));
+ await vi.advanceTimersByTimeAsync(0);
+ expect(options.fetch).toHaveBeenCalledTimes(1);
+ });
+
+ test("fails open and removes a listener even if caller listener registration throws after attaching it", async () => {
+ vi.useFakeTimers();
+ const caller = new AbortController();
+ const addListener = caller.signal.addEventListener.bind(caller.signal);
+ vi.spyOn(caller.signal, "addEventListener").mockImplementation((...args) => {
+ addListener(...args);
+ throw new Error("Invalid signal listener registration");
+ });
+ const options = setup();
+ await expect(suggestJevSkill(REQUEST, SKILLS, { ...options, signal: caller.signal })).resolves.toEqual({
+ reason: "unavailable",
+ });
+ expect(options.fetch).not.toHaveBeenCalled();
+ expect(options.loadExcerpt).not.toHaveBeenCalled();
+ expect(getEventListeners(caller.signal, "abort")).toEqual([]);
+ expect(vi.getTimerCount()).toBe(0);
+ });
+});
diff --git a/packages/coding-agent/test/step-jev-skill-router.test.ts b/packages/coding-agent/test/step-jev-skill-router.test.ts
new file mode 100644
index 00000000..1f232a40
--- /dev/null
+++ b/packages/coding-agent/test/step-jev-skill-router.test.ts
@@ -0,0 +1,210 @@
+import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import { join } from "node:path";
+import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
+import type {
+ BeforeAgentStartEvent,
+ BeforeAgentStartEventResult,
+ ExtensionAPI,
+ ExtensionContext,
+} from "../src/core/extensions/types.ts";
+import { loadSkillsFromDir, type Skill } from "../src/core/skills.ts";
+import { suggestJevSkill } from "../src/features/jev-skill-router/router.ts";
+import { createStepJevSkillRouterExtension } from "../src/features/step-jev-skill-router.ts";
+
+vi.mock("../src/features/jev-skill-router/router.ts", () => ({
+ suggestJevSkill: vi.fn(),
+ JEV_EXCERPT_CHARS: 700,
+ JEV_MAX_SKILLS: 254,
+ JEV_MAX_REQUEST_CHARS: 12_000,
+}));
+
+type Handler = (event: BeforeAgentStartEvent) => Promise;
+const enabledEnv = { STEP_JEV_SKILL_ROUTING: "1", TYPESAFE_API_KEY: "test-typesafe-key" };
+
+describe("Step Jev skill routing extension", () => {
+ let root: string;
+ let skills: Skill[];
+ let event: BeforeAgentStartEvent;
+
+ beforeEach(() => {
+ vi.clearAllMocks();
+ vi.mocked(suggestJevSkill).mockResolvedValue({ reason: "selected", skillName: "browser" });
+ root = mkdtempSync(join(tmpdir(), "step-jev-skills-"));
+ for (const name of ["browser", "documents", "manual"]) {
+ mkdirSync(join(root, name));
+ writeFileSync(
+ join(root, name, "SKILL.md"),
+ `---\nname: ${name}\ndescription: Instructions for ${name}\ndisable-model-invocation: ${name === "manual"}\n---\n${name} body ${"x".repeat(1000)}`,
+ );
+ }
+ skills = loadSkillsFromDir({ dir: root, source: "test" }).skills;
+ event = {
+ type: "before_agent_start",
+ prompt: "Check the checkout page in a browser",
+ systemPrompt: "Original system prompt and complete skill catalog",
+ systemPromptOptions: {
+ cwd: root,
+ skills,
+ selectedTools: ["read_file", "run_command"],
+ contextFiles: [{ path: "/private/AGENTS.md", content: "private project instructions" }],
+ },
+ };
+ });
+
+ afterEach(() => rmSync(root, { recursive: true, force: true }));
+
+ function register(env: NodeJS.ProcessEnv = enabledEnv): Handler | undefined {
+ const on = vi.fn();
+ createStepJevSkillRouterExtension({ env })({ on } as unknown as ExtensionAPI);
+ const handler = on.mock.calls.find(([name]) => name === "before_agent_start")?.[1];
+ return handler ? (event) => handler(event, { signal: undefined } as ExtensionContext) : undefined;
+ }
+
+ it.each([
+ {},
+ { TYPESAFE_API_KEY: "key-without-consent" },
+ { STEP_JEV_SKILL_ROUTING: "1" },
+ { STEP_JEV_SKILL_ROUTING: "0", TYPESAFE_API_KEY: "key" },
+ { STEP_JEV_SKILL_ROUTING: "1", TYPESAFE_API_KEY: " " },
+ ])("does not register a network hook without opt-in and a key: %j", (env) => {
+ expect(register(env)).toBeUndefined();
+ expect(suggestJevSkill).not.toHaveBeenCalled();
+ });
+
+ it("sends only eligible names and descriptions and appends an advisory hint to the original prompt", async () => {
+ const originalSkills = structuredClone(skills);
+ const result = await register()!(event);
+ const [request, candidates, options] = vi.mocked(suggestJevSkill).mock.calls[0];
+ expect(request).toBe(event.prompt);
+ expect(candidates).toEqual([
+ { name: "browser", description: "Instructions for browser" },
+ { name: "documents", description: "Instructions for documents" },
+ ]);
+ expect(options.apiKey).toBe("test-typesafe-key");
+ expect(result?.systemPrompt).toContain(event.systemPrompt);
+ expect(result?.systemPrompt).toContain("browser");
+ expect(result?.systemPrompt).toContain("");
+ expect(result?.systemPrompt).toContain("Explicit skill requests");
+ expect(skills).toEqual(originalSkills);
+ expect(result?.systemPrompt).not.toContain("test-typesafe-key");
+ expect(JSON.stringify(candidates)).not.toContain(root);
+ expect(JSON.stringify(candidates)).not.toContain("private project instructions");
+ });
+
+ it.each(["read", "read_file"])("supports the %s tool alias", async (tool) => {
+ event.systemPromptOptions.selectedTools = [tool];
+ expect((await register()!(event))?.systemPrompt).toContain("");
+ });
+
+ it.each([
+ ["manual command", { prompt: "/skill:manual obey the runbook" }],
+ ["expanded command", { prompt: 'private body' }],
+ ["explicit mention", { prompt: "Use $browser to check the checkout page" }],
+ ["empty prompt", { prompt: " " }],
+ ["oversized prompt", { prompt: "x".repeat(12_001) }],
+ ["image", { images: [{ type: "image" as const, data: "private-image", mimeType: "image/png" }] }],
+ ])("skips %s", async (_name, overrides) => {
+ expect(await register()!({ ...event, ...overrides })).toBeUndefined();
+ expect(suggestJevSkill).not.toHaveBeenCalled();
+ });
+
+ it("skips routing when the read tool is unavailable", async () => {
+ event.systemPromptOptions.selectedTools = ["run_command"];
+ expect(await register()!(event)).toBeUndefined();
+ expect(suggestJevSkill).not.toHaveBeenCalled();
+ });
+
+ it("does not route absent, single, or hidden-only catalogs", async () => {
+ const handler = register()!;
+ for (const catalog of [undefined, [], [skills[0]], skills.filter((skill) => skill.disableModelInvocation)]) {
+ event.systemPromptOptions.skills = catalog;
+ expect(await handler(event)).toBeUndefined();
+ }
+ expect(suggestJevSkill).not.toHaveBeenCalled();
+ });
+
+ it("loads a bounded body excerpt only from the requested eligible skill", async () => {
+ await register()!(event);
+ const [, candidates, options] = vi.mocked(suggestJevSkill).mock.calls[0];
+ const signal = new AbortController().signal;
+ const excerpt = await options.loadExcerpt(candidates[0], signal);
+ expect(excerpt).toBe(`browser body ${"x".repeat(1000)}`.slice(0, 700));
+ await expect(options.loadExcerpt({ name: "manual", description: "hidden" }, signal)).rejects.toThrow();
+ });
+
+ it("uses the documented fallback key without forwarding the Step key", async () => {
+ await register({ STEP_JEV_SKILL_ROUTING: "1", JEV_API_KEY: "jev-key", STEP_API_KEY: "step-secret" })!(event);
+ expect(vi.mocked(suggestJevSkill).mock.calls[0][2].apiKey).toBe("jev-key");
+ expect(JSON.stringify(vi.mocked(suggestJevSkill).mock.calls)).not.toContain("step-secret");
+ });
+
+ it.each(["no-match", "low-confidence", "unavailable", "unsupported-input"] as const)(
+ "leaves the prompt unchanged after %s",
+ async (reason) => {
+ vi.mocked(suggestJevSkill).mockResolvedValue({ reason });
+ expect(await register()!(event)).toBeUndefined();
+ },
+ );
+
+ it("never inserts an unrecognized or hidden skill returned by the router", async () => {
+ const handler = register()!;
+ for (const skillName of ["unknown", "manual"]) {
+ vi.mocked(suggestJevSkill).mockResolvedValue({ reason: "selected", skillName });
+ expect(await handler(event)).toBeUndefined();
+ }
+ });
+
+ it("takes the current event catalog after a resource reload", async () => {
+ const handler = register()!;
+ await handler(event);
+ event.systemPromptOptions.skills = [
+ { ...skills[0], name: "new-browser" },
+ { ...skills[1], name: "new-documents" },
+ ];
+ vi.mocked(suggestJevSkill).mockResolvedValue({ reason: "selected", skillName: "new-browser" });
+ const result = await handler(event);
+ expect(vi.mocked(suggestJevSkill).mock.calls[1][1].map((skill) => skill.name)).toEqual([
+ "new-browser",
+ "new-documents",
+ ]);
+ expect(result?.systemPrompt).toContain("new-browser");
+ expect(result?.systemPrompt?.split("")).toHaveLength(2);
+ });
+
+ it("skips duplicate or oversized eligible catalogs before routing", async () => {
+ const handler = register()!;
+ for (const catalog of [
+ [skills[0], skills[0]],
+ Array.from({ length: 255 }, (_, index) => ({ ...skills[0], name: `skill-${index}` })),
+ ]) {
+ event.systemPromptOptions.skills = catalog;
+ expect(await handler(event)).toBeUndefined();
+ }
+ expect(suggestJevSkill).not.toHaveBeenCalled();
+ });
+
+ it("does not disclose incomplete frontmatter from a large skill file", async () => {
+ writeFileSync(skills[0].filePath, `---\nprivate_metadata: ${"x".repeat(20_000)}\n---\nPublic body`);
+ await register()!(event);
+ const [, candidates, options] = vi.mocked(suggestJevSkill).mock.calls[0];
+ expect(await options.loadExcerpt(candidates[0], new AbortController().signal)).toBe("");
+ });
+
+ it("refuses an excerpt read after the deadline has expired", async () => {
+ await register()!(event);
+ const [, candidates, options] = vi.mocked(suggestJevSkill).mock.calls[0];
+ const controller = new AbortController();
+ controller.abort();
+ await expect(options.loadExcerpt(candidates[0], controller.signal)).rejects.toThrow();
+ });
+
+ it("escapes leniently loaded skill names in the advisory XML", async () => {
+ const name = 'browser<&"';
+ event.systemPromptOptions.skills = [{ ...skills[0], name }, skills[1]];
+ vi.mocked(suggestJevSkill).mockResolvedValue({ reason: "selected", skillName: name });
+ const result = await register()!(event);
+ expect(result?.systemPrompt).toContain("browser<&"");
+ expect(result?.systemPrompt).not.toContain(name);
+ });
+});
diff --git a/packages/coding-agent/test/suite/jev-skill-routing-cancellation.test.ts b/packages/coding-agent/test/suite/jev-skill-routing-cancellation.test.ts
new file mode 100644
index 00000000..68a2b7a0
--- /dev/null
+++ b/packages/coding-agent/test/suite/jev-skill-routing-cancellation.test.ts
@@ -0,0 +1,425 @@
+import { fileURLToPath } from "node:url";
+import { fauxAssistantMessage } from "@step-harness/providers";
+import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
+import { AgentSessionRuntime } from "../../src/core/agent-session-runtime.ts";
+import type { ExtensionContext, ExtensionFactory } from "../../src/core/extensions/types.ts";
+import { loadSkillsFromDir } from "../../src/core/skills.ts";
+import { type JevRoutingResult, type JevSkill, suggestJevSkill } from "../../src/features/jev-skill-router/router.ts";
+import { createStepJevSkillRouterExtension } from "../../src/features/step-jev-skill-router.ts";
+import { createTestExtensionsResult, createTestResourceLoader } from "../utilities.ts";
+import { createHarness, getUserTexts, type Harness } from "./harness.ts";
+
+const FIRST_PROMPT = "Investigate the build failure";
+const NEXT_PROMPT = "What is two plus two?";
+const skills = ["valid-skill", "multiline-description"].flatMap(
+ (name) =>
+ loadSkillsFromDir({
+ dir: fileURLToPath(new URL(`../fixtures/skills/${name}/`, import.meta.url)),
+ source: "test",
+ }).skills,
+);
+
+function deferred() {
+ let resolve!: (value: T) => void;
+ const promise = new Promise((complete) => {
+ resolve = complete;
+ });
+ return { promise, resolve };
+}
+
+// Observe rejection immediately: a cancelled prompt may reject before abort() settles.
+function observe(promise: Promise) {
+ let settled = false;
+ const result = promise.then(
+ (value) => {
+ settled = true;
+ return { status: "fulfilled" as const, value };
+ },
+ (error: unknown) => {
+ settled = true;
+ return { status: "rejected" as const, error };
+ },
+ );
+ return {
+ result,
+ get settled() {
+ return settled;
+ },
+ };
+}
+
+interface JevRequest {
+ state: { request: string };
+ questions: Record }>;
+}
+
+function reply(request: JevRequest, selected = "skill_0"): Response {
+ const answers = Object.fromEntries(
+ Object.entries(request.questions).map(([id, question]) => [
+ id,
+ question.type === "choice"
+ ? {
+ type: "choice",
+ choice: selected,
+ confidence: 0.99,
+ probabilities: Object.fromEntries(
+ Object.keys(question.criteria ?? {}).map((option) => [option, option === selected ? 1 : 0]),
+ ),
+ }
+ : { type: "noul", noul: selected === "none" ? 0 : 0.99 },
+ ]),
+ );
+ return Response.json({ model: "jev-test", answers, usage: { input_tokens: 0, output_tokens: 0 } });
+}
+
+function blockedTransport() {
+ const started = deferred();
+ const firstReply = deferred();
+ const requests: JevRequest[] = [];
+ const signals: Array = [];
+ const fetch = vi.fn(async (_url, init) => {
+ const request = JSON.parse(String(init?.body)) as JevRequest;
+ requests.push(request);
+ signals.push(init?.signal);
+ if (requests.length === 1) {
+ started.resolve();
+ // Deliberately ignore abort to exercise session and engine cancellation races.
+ return firstReply.promise;
+ }
+ return reply(request, request.state.request === FIRST_PROMPT ? "skill_0" : "none");
+ });
+ return {
+ fetch,
+ requests,
+ signals,
+ started: started.promise,
+ release: (selected = "skill_0") => firstReply.resolve(reply(requests[0], selected)),
+ };
+}
+
+describe("Jev routing cancellation in an AgentSession", () => {
+ const harnesses: Harness[] = [];
+
+ beforeEach(() => {
+ vi.useFakeTimers({ toFake: ["setTimeout", "clearTimeout", "performance"] });
+ });
+
+ afterEach(() => {
+ while (harnesses.length) harnesses.pop()!.cleanup();
+ vi.useRealTimers();
+ });
+
+ async function session(factories: ExtensionFactory[]): Promise {
+ const extensionsResult = await createTestExtensionsResult(factories);
+ const harness = await createHarness({
+ resourceLoader: {
+ ...createTestResourceLoader({ extensionsResult }),
+ getSkills: () => ({ skills, diagnostics: [] }),
+ },
+ });
+ harnesses.push(harness);
+ return harness;
+ }
+
+ async function routedSession(
+ transport: ReturnType,
+ additionalFactories: ExtensionFactory[] = [],
+ ) {
+ const contextSignals: Array = [];
+ const harness = await session([
+ (pi) => {
+ pi.on("before_agent_start", (_event, ctx) => {
+ contextSignals.push(ctx.signal);
+ });
+ },
+ createStepJevSkillRouterExtension({
+ env: { STEP_JEV_SKILL_ROUTING: "1", TYPESAFE_API_KEY: "test-typesafe-key" },
+ fetch: transport.fetch,
+ }),
+ ...additionalFactories,
+ ]);
+ return { harness, contextSignals };
+ }
+
+ it("aborts the real hook during its first HTTP request, ignores the late reply, and permits a fresh prompt", async () => {
+ const transport = blockedTransport();
+ const { harness, contextSignals } = await routedSession(transport);
+ harness.setResponses([fauxAssistantMessage("The cancelled task must not run")]);
+ const pending = observe(harness.session.prompt(FIRST_PROMPT));
+ try {
+ await transport.started;
+ expect.soft(harness.session.isStreaming).toBe(true);
+ expect.soft(harness.session.isIdle).toBe(false);
+ expect.soft(contextSignals[0]).toBeInstanceOf(AbortSignal);
+ expect.soft(contextSignals[0]?.aborted).toBe(false);
+ expect.soft(transport.signals[0]?.aborted).toBe(false);
+
+ const abort = observe(harness.session.abort());
+ await vi.advanceTimersByTimeAsync(0);
+ // No deadline has advanced and the transport is still blocked.
+ expect.soft(abort.settled).toBe(true);
+ expect.soft(pending.settled).toBe(true);
+ expect.soft(contextSignals[0]?.aborted).toBe(true);
+ expect.soft(transport.signals[0]?.aborted).toBe(true);
+ expect.soft(harness.session.isIdle).toBe(true);
+
+ transport.release();
+ await pending.result;
+ await abort.result;
+ await vi.advanceTimersByTimeAsync(0);
+ expect.soft(transport.fetch).toHaveBeenCalledTimes(1);
+ expect.soft(harness.faux.state.callCount).toBe(0);
+ expect.soft(harness.eventsOfType("tool_execution_start")).toHaveLength(0);
+ expect.soft(harness.session.systemPrompt).not.toContain("");
+
+ harness.setResponses([fauxAssistantMessage("Four")]);
+ await harness.session.prompt(NEXT_PROMPT);
+ expect.soft(transport.fetch).toHaveBeenCalledTimes(2);
+ expect.soft(harness.faux.state.callCount).toBe(1);
+ expect.soft(contextSignals[1]).toBeInstanceOf(AbortSignal);
+ expect.soft(contextSignals[1]).not.toBe(contextSignals[0]);
+ expect.soft(contextSignals[1]?.aborted).toBe(false);
+ expect.soft(getUserTexts(harness)).toEqual([NEXT_PROMPT]);
+ } finally {
+ transport.release();
+ await pending.result;
+ }
+ });
+
+ it("direct session.dispose cancels blocked Jev HTTP and settles the prompt without stale context errors", async () => {
+ const transport = blockedTransport();
+ const { harness, contextSignals } = await routedSession(transport);
+ harness.setResponses([fauxAssistantMessage("The disposed task must not run")]);
+ const pending = observe(harness.session.prompt(FIRST_PROMPT));
+ try {
+ await transport.started;
+ harness.session.dispose();
+ // Synchronous disposal must cancel the operation before invalidating its context.
+ expect.soft(contextSignals[0]?.aborted).toBe(true);
+ expect.soft(transport.signals[0]?.aborted).toBe(true);
+ await vi.advanceTimersByTimeAsync(0);
+ expect.soft(pending.settled).toBe(true);
+ expect.soft(harness.session.isIdle).toBe(true);
+
+ transport.release();
+ const outcome = await pending.result;
+ await vi.advanceTimersByTimeAsync(0);
+ expect.soft(outcome).toEqual({ status: "fulfilled", value: undefined });
+ expect.soft(transport.fetch).toHaveBeenCalledTimes(1);
+ expect.soft(harness.faux.state.callCount).toBe(0);
+ expect.soft(harness.session.messages).toEqual([]);
+ } finally {
+ transport.release();
+ await pending.result;
+ }
+ });
+
+ it("awaited runtime.dispose settles blocked Jev routing before shutdown and context invalidation", async () => {
+ const transport = blockedTransport();
+ const phases: string[] = [];
+ const { harness, contextSignals } = await routedSession(transport, [
+ (pi) => {
+ pi.on("agent_settled", () => {
+ phases.push("settled");
+ });
+ pi.on("session_shutdown", (_event, ctx) => {
+ phases.push("shutdown");
+ expect.soft(ctx.isIdle()).toBe(true);
+ expect.soft(ctx.signal).toBeUndefined();
+ expect.soft(contextSignals[0]?.aborted).toBe(true);
+ expect.soft(transport.signals[0]?.aborted).toBe(true);
+ });
+ },
+ ]);
+ const runtime = new AgentSessionRuntime(
+ harness.session,
+ {
+ cwd: harness.tempDir,
+ agentDir: harness.tempDir,
+ modelRuntime: harness.session.modelRuntime,
+ settingsManager: harness.settingsManager,
+ resourceLoader: harness.session.resourceLoader,
+ diagnostics: [],
+ },
+ async () => {
+ throw new Error("Disposal must not create a replacement runtime");
+ },
+ );
+ runtime.setBeforeSessionInvalidate(() => {
+ phases.push("invalidate");
+ expect.soft(harness.session.isIdle).toBe(true);
+ });
+ harness.setResponses([fauxAssistantMessage("The disposed task must not run")]);
+ const pending = observe(harness.session.prompt(FIRST_PROMPT));
+ try {
+ await transport.started;
+ const disposal = observe(runtime.dispose());
+ // Let disposal start; keep both the transport gate and the 1500ms deadline untouched.
+ await vi.advanceTimersByTimeAsync(0);
+ expect.soft(contextSignals[0]?.aborted).toBe(true);
+ expect.soft(transport.signals[0]?.aborted).toBe(true);
+ expect.soft(pending.settled).toBe(true);
+ expect.soft(disposal.settled).toBe(true);
+ expect.soft(phases).toEqual(["settled", "shutdown", "invalidate"]);
+
+ transport.release();
+ const [promptOutcome, disposalOutcome] = await Promise.all([pending.result, disposal.result]);
+ await vi.advanceTimersByTimeAsync(0);
+ expect.soft(promptOutcome).toEqual({ status: "fulfilled", value: undefined });
+ expect.soft(disposalOutcome).toEqual({ status: "fulfilled", value: undefined });
+ expect.soft(transport.fetch).toHaveBeenCalledTimes(1);
+ expect.soft(harness.faux.state.callCount).toBe(0);
+ expect.soft(harness.session.messages).toEqual([]);
+ } finally {
+ transport.release();
+ await pending.result;
+ await runtime.dispose();
+ }
+ });
+
+ it("cancels a real engine excerpt wait through before_agent_start's existing ctx.signal", async () => {
+ const excerptStarted = deferred();
+ const excerpt = deferred();
+ const contextSignals: Array = [];
+ let routingResult: JevRoutingResult | undefined;
+ const fetch = vi.fn(async (_url, init) => {
+ const request = JSON.parse(String(init?.body)) as JevRequest;
+ return reply(request, request.state.request === FIRST_PROMPT ? "skill_0" : "none");
+ });
+ const loadExcerpt = vi.fn(async (_skill: JevSkill, _signal: AbortSignal) => {
+ excerptStarted.resolve();
+ return excerpt.promise;
+ });
+ const harness = await session([
+ (pi) => {
+ pi.on("before_agent_start", async (event, ctx) => {
+ contextSignals.push(ctx.signal);
+ const options = { apiKey: "test-typesafe-key", fetch, loadExcerpt, signal: ctx.signal };
+ routingResult = await suggestJevSkill(event.prompt, skills, options);
+ });
+ },
+ ]);
+ harness.setResponses([fauxAssistantMessage("The cancelled task must not run")]);
+ const pending = observe(harness.session.prompt(FIRST_PROMPT));
+ try {
+ await excerptStarted.promise;
+ expect.soft(harness.session.isStreaming).toBe(true);
+ expect.soft(harness.session.isIdle).toBe(false);
+ expect.soft(contextSignals[0]).toBeInstanceOf(AbortSignal);
+ const abort = observe(harness.session.abort());
+ await vi.advanceTimersByTimeAsync(0);
+ expect.soft(abort.settled).toBe(true);
+ expect.soft(pending.settled).toBe(true);
+ expect.soft(contextSignals[0]?.aborted).toBe(true);
+ for (const [, signal] of loadExcerpt.mock.calls) expect.soft(signal.aborted).toBe(true);
+ expect.soft(routingResult).toEqual({ reason: "unavailable" });
+
+ excerpt.resolve("Late excerpt must not trigger verification");
+ await pending.result;
+ await abort.result;
+ await vi.advanceTimersByTimeAsync(0);
+ expect.soft(fetch).toHaveBeenCalledTimes(1);
+ expect.soft(harness.faux.state.callCount).toBe(0);
+ expect.soft(harness.eventsOfType("tool_execution_start")).toHaveLength(0);
+
+ harness.setResponses([fauxAssistantMessage("Four")]);
+ await harness.session.prompt(NEXT_PROMPT);
+ expect.soft(fetch).toHaveBeenCalledTimes(2);
+ expect.soft(harness.faux.state.callCount).toBe(1);
+ expect.soft(getUserTexts(harness)).toEqual([NEXT_PROMPT]);
+ } finally {
+ excerpt.resolve("Cleanup");
+ await pending.result;
+ }
+ });
+
+ it("honors ctx.abort() inside a hook and discards its result and all remaining handlers", async () => {
+ let hookContext: ExtensionContext | undefined;
+ let hookSignal: AbortSignal | undefined;
+ const laterInSameExtension = vi.fn();
+ const laterExtension = vi.fn();
+ const harness = await session([
+ (pi) => {
+ pi.on("before_agent_start", (_event, ctx) => {
+ hookContext = ctx;
+ hookSignal = ctx.signal;
+ expect.soft(ctx.isIdle()).toBe(false);
+ expect.soft(hookSignal).toBeInstanceOf(AbortSignal);
+ expect.soft(hookSignal?.aborted).toBe(false);
+ ctx.abort();
+ expect.soft(ctx.signal).toBe(hookSignal);
+ return {
+ systemPrompt: "Discard this cancelled system prompt",
+ message: {
+ customType: "cancelled-hook-result",
+ content: "Discard this cancelled message",
+ display: false,
+ },
+ };
+ });
+ pi.on("before_agent_start", laterInSameExtension);
+ },
+ (pi) => {
+ pi.on("before_agent_start", laterExtension);
+ },
+ ]);
+ const originalSystemPrompt = harness.session.systemPrompt;
+ harness.setResponses([fauxAssistantMessage("The cancelled task must not run")]);
+ await harness.session.prompt(FIRST_PROMPT);
+
+ expect(hookSignal?.aborted).toBe(true);
+ expect(laterInSameExtension).not.toHaveBeenCalled();
+ expect(laterExtension).not.toHaveBeenCalled();
+ expect(harness.faux.state.callCount).toBe(0);
+ expect(harness.session.messages).toEqual([]);
+ expect(harness.session.systemPrompt).toBe(originalSystemPrompt);
+ expect(harness.session.isStreaming).toBe(false);
+ expect(harness.session.isIdle).toBe(true);
+ // The existing context getters follow the run back to idle.
+ expect(hookContext?.isIdle()).toBe(true);
+ expect(hookContext?.signal).toBeUndefined();
+ });
+
+ it.each([undefined, "followUp"] as const)(
+ "applies concurrent prompt policy %s while the routing hook is pending",
+ async (streamingBehavior) => {
+ const transport = blockedTransport();
+ const { harness, contextSignals } = await routedSession(transport);
+ harness.setResponses([fauxAssistantMessage("First answer"), fauxAssistantMessage("Follow-up answer")]);
+ const pending = observe(harness.session.prompt(FIRST_PROMPT));
+ try {
+ await transport.started;
+ const concurrent = observe(harness.session.prompt(NEXT_PROMPT, { streamingBehavior }));
+ await vi.advanceTimersByTimeAsync(0);
+ expect.soft(concurrent.settled).toBe(true);
+ expect.soft(transport.fetch).toHaveBeenCalledTimes(1);
+ expect.soft(contextSignals).toHaveLength(1);
+ expect.soft(harness.faux.state.callCount).toBe(0);
+ if (streamingBehavior === "followUp") {
+ expect.soft(harness.session.getFollowUpMessages()).toEqual([NEXT_PROMPT]);
+ }
+
+ transport.release("none");
+ const outcome = await concurrent.result;
+ await pending.result;
+ if (streamingBehavior === undefined) {
+ expect.soft(outcome).toMatchObject({
+ status: "rejected",
+ error: expect.objectContaining({ message: expect.stringContaining("already processing") }),
+ });
+ expect.soft(getUserTexts(harness)).toEqual([FIRST_PROMPT]);
+ expect.soft(harness.faux.state.callCount).toBe(1);
+ } else {
+ expect.soft(outcome.status).toBe("fulfilled");
+ expect.soft(getUserTexts(harness)).toEqual([FIRST_PROMPT, NEXT_PROMPT]);
+ expect.soft(harness.faux.state.callCount).toBe(2);
+ expect.soft(harness.session.getFollowUpMessages()).toEqual([]);
+ }
+ expect.soft(harness.session.isIdle).toBe(true);
+ } finally {
+ transport.release("none");
+ await pending.result;
+ }
+ },
+ );
+});
diff --git a/packages/coding-agent/test/suite/step-jev-skill-routing.test.ts b/packages/coding-agent/test/suite/step-jev-skill-routing.test.ts
new file mode 100644
index 00000000..78be9381
--- /dev/null
+++ b/packages/coding-agent/test/suite/step-jev-skill-routing.test.ts
@@ -0,0 +1,169 @@
+import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import { join } from "node:path";
+import { fauxAssistantMessage, fauxToolCall } from "@step-harness/providers";
+import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
+import { loadSkillsFromDir, type Skill } from "../../src/core/skills.ts";
+import { createStepJevSkillRouterExtension } from "../../src/features/step-jev-skill-router.ts";
+import { createTestExtensionsResult, createTestResourceLoader } from "../utilities.ts";
+import { createHarness, getUserTexts, type Harness } from "./harness.ts";
+
+interface JevRequest {
+ state: { request: string };
+ questions: Record }>;
+}
+
+describe("Jev routing in an AgentSession", () => {
+ let root: string;
+ let skills: Skill[];
+ const harnesses: Harness[] = [];
+
+ beforeEach(() => {
+ root = mkdtempSync(join(tmpdir(), "step-jev-session-"));
+ for (const name of ["browser", "documents", "manual"]) {
+ mkdirSync(join(root, name));
+ writeFileSync(
+ join(root, name, "SKILL.md"),
+ `---\nname: ${name}\ndescription: Use ${name} instructions\ndisable-model-invocation: ${name === "manual"}\n---\n${name} instructions ${"x".repeat(1000)}\nOUTSIDE_EXCERPT`,
+ );
+ }
+ skills = loadSkillsFromDir({ dir: root, source: "test" }).skills;
+ });
+
+ afterEach(() => {
+ while (harnesses.length) harnesses.pop()!.cleanup();
+ rmSync(root, { recursive: true, force: true });
+ });
+
+ function transport() {
+ const requests: JevRequest[] = [];
+ let noMatch = false;
+ const fetch = vi.fn(async (url, init) => {
+ expect(url).toBe("https://api.typesafe.ai/v1/systemone");
+ expect(new Headers(init?.headers).get("Authorization")).toBe("Bearer typesafe-test-key");
+ const request = JSON.parse(String(init?.body)) as JevRequest;
+ requests.push(request);
+ const answers: Record = {};
+ for (const [id, question] of Object.entries(request.questions)) {
+ if (question.type === "choice") {
+ const options = Object.keys(question.criteria ?? {});
+ const choice = noMatch ? "none" : "skill_0";
+ expect(options).toContain(choice);
+ answers[id] = {
+ type: "choice",
+ choice,
+ confidence: 0.99,
+ probabilities: Object.fromEntries(options.map((option) => [option, option === choice ? 1 : 0])),
+ };
+ } else {
+ answers[id] = { type: "noul", noul: noMatch ? 0 : 0.99 };
+ }
+ }
+ return Response.json({ model: "jev-test", answers, usage: { input_tokens: 100, output_tokens: 20 } });
+ });
+ return {
+ fetch,
+ requests,
+ setNoMatch: () => {
+ noMatch = true;
+ },
+ };
+ }
+
+ async function session(fetch: typeof globalThis.fetch): Promise {
+ const extensionsResult = await createTestExtensionsResult(
+ [
+ createStepJevSkillRouterExtension({
+ env: { STEP_JEV_SKILL_ROUTING: "1", TYPESAFE_API_KEY: "typesafe-test-key" },
+ fetch,
+ }),
+ ],
+ root,
+ );
+ const harness = await createHarness({
+ resourceLoader: {
+ ...createTestResourceLoader({ extensionsResult }),
+ getSkills: () => ({ skills, diagnostics: [] }),
+ getAgentsFiles: () => ({ agentsFiles: [{ path: "/private/AGENTS.md", content: "PRIVATE_PROJECT_GUIDE" }] }),
+ },
+ });
+ harnesses.push(harness);
+ return harness;
+ }
+
+ it("routes once before the tool loop and clears the suggestion on the next no-match prompt", async () => {
+ const jev = transport();
+ const harness = await session(jev.fetch);
+ const prompts: string[] = [];
+ harness.setResponses([
+ (context) => {
+ prompts.push(context.systemPrompt ?? "");
+ return fauxAssistantMessage(fauxToolCall("read", { path: skills[0].filePath }), { stopReason: "toolUse" });
+ },
+ (context) => {
+ prompts.push(context.systemPrompt ?? "");
+ return fauxAssistantMessage("Checked the browser instructions");
+ },
+ ]);
+
+ await harness.session.prompt("Check the checkout page");
+
+ expect(jev.requests).toHaveLength(2);
+ expect(harness.faux.state.callCount).toBe(2);
+ expect(harness.eventsOfType("tool_execution_start")).toHaveLength(1);
+ expect(prompts[0]).toBe(prompts[1]);
+ expect(prompts[0]).toContain("");
+ expect(prompts[0]).toContain("browser");
+ expect(prompts[0]).toContain("documents");
+ expect(prompts[0]).not.toContain("manual");
+ const transmitted = JSON.stringify(jev.requests);
+ expect(transmitted).not.toContain(root);
+ expect(transmitted).not.toContain("PRIVATE_PROJECT_GUIDE");
+ expect(transmitted).not.toContain("OUTSIDE_EXCERPT");
+ expect(transmitted).not.toContain("manual");
+ expect(transmitted).not.toContain("faux-key");
+
+ jev.setNoMatch();
+ harness.setResponses([
+ (context) => {
+ prompts.push(context.systemPrompt ?? "");
+ return fauxAssistantMessage("The answer is four");
+ },
+ ]);
+ await harness.session.prompt("What is two plus two?");
+ expect(jev.requests).toHaveLength(3);
+ expect(prompts[2]).not.toContain("");
+ expect(prompts[0].startsWith(prompts[2])).toBe(true);
+ });
+
+ it("allows an explicit hidden skill without sending its expanded body to Jev", async () => {
+ const jev = transport();
+ const harness = await session(jev.fetch);
+ harness.setResponses([fauxAssistantMessage("Following the manual skill")]);
+ await harness.session.prompt("/skill:manual follow these instructions");
+
+ expect(jev.fetch).not.toHaveBeenCalled();
+ expect(getUserTexts(harness)[0]).toContain(' {
+ const fetch = vi.fn(async () => new Response("rate limited", { status: 429 }));
+ const harness = await session(fetch);
+ let prompt = "";
+ harness.setResponses([
+ (context) => {
+ prompt = context.systemPrompt ?? "";
+ return fauxAssistantMessage("The task can continue");
+ },
+ ]);
+ await harness.session.prompt("Check the checkout page");
+
+ expect(fetch).toHaveBeenCalledTimes(1);
+ expect(harness.faux.state.callCount).toBe(1);
+ expect(prompt).not.toContain("");
+ expect(prompt).toContain("browser");
+ expect(prompt).toContain("documents");
+ });
+});