diff --git a/.claude-plugin/plugin.json b/.claude-plugin/plugin.json index 33807fe..a986862 100644 --- a/.claude-plugin/plugin.json +++ b/.claude-plugin/plugin.json @@ -2,7 +2,7 @@ "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "skillhook", "displayName": "skillhook", - "version": "0.3.0", + "version": "0.5.0", "description": "Turn this machine into a permanent, secure webhook endpoint that runs Agent Skills with Claude Code or Codex. Two skills teach the agent to install and expose skillhook and to write good webhook skills; the bundled MCP server manages skills, secrets, jobs and exposure.", "author": { "name": "Meter", diff --git a/.codex-plugin/plugin.json b/.codex-plugin/plugin.json index c195774..61c6f33 100644 --- a/.codex-plugin/plugin.json +++ b/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "skillhook", - "version": "0.3.0", + "version": "0.5.0", "description": "Turn this machine into a permanent, secure webhook endpoint that runs Agent Skills with Claude Code or Codex. Two skills teach the agent to install and expose skillhook and to write good webhook skills; the bundled MCP server manages skills, secrets, jobs and exposure.", "author": { "name": "Meter", diff --git a/.cursor-plugin/plugin.json b/.cursor-plugin/plugin.json index 96656e4..6f7f0ec 100644 --- a/.cursor-plugin/plugin.json +++ b/.cursor-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "skillhook", - "version": "0.3.0", + "version": "0.5.0", "description": "Turn this machine into a permanent, secure webhook endpoint that runs Agent Skills with Claude Code or Codex. Two skills teach the agent to install and expose skillhook and to write good webhook skills; the bundled MCP server manages skills, secrets, jobs and exposure.", "author": { "name": "Meter", diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index ca82195..14f383e 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -62,6 +62,15 @@ jobs: node dist/cli.js link . --dir "$RUNNER_TEMP/skillhook" --json node dist/cli.js projects --dir "$RUNNER_TEMP/skillhook" --json node dist/cli.js run pull-after-merge --dir "$RUNNER_TEMP/skillhook" --payload '{}' --dry-run --json > /dev/null + node dist/cli.js run --file examples/skills/hello/SKILL.md --dir "$RUNNER_TEMP/skillhook" --payload '{"name":"ci"}' --dry-run --json > /dev/null + node dist/cli.js deliveries list --dir "$RUNNER_TEMP/skillhook" --json > /dev/null + node dist/cli.js jobs list --waiting --dir "$RUNNER_TEMP/skillhook" --json > /dev/null + node dist/cli.js stats --since 7d --dir "$RUNNER_TEMP/skillhook" --json > /dev/null + node dist/cli.js config path --dir "$RUNNER_TEMP/skillhook" --json > /dev/null + node dist/cli.js cloud status --dir "$RUNNER_TEMP/skillhook" --json > /dev/null + node dist/cli.js runners --local --dir "$RUNNER_TEMP/skillhook" --json > /dev/null || true # no claude/codex on the runner + SKILLHOOK_NO_UPDATE_CHECK=1 node dist/cli.js health --quick --local --dir "$RUNNER_TEMP/skillhook" --json > "$RUNNER_TEMP/health.json" || true + node -e 'const r = JSON.parse(require("node:fs").readFileSync(process.argv[1], "utf8")); if (!Array.isArray(r.checks) || !r.groups) process.exit(1);' "$RUNNER_TEMP/health.json" - name: Audit production dependencies run: npm audit --omit=dev @@ -95,7 +104,7 @@ jobs: const fs = require("node:fs"); const [info] = JSON.parse(fs.readFileSync(process.argv[2], "utf8")); const files = new Set(info.files.map((f) => f.path)); - const required = ["package.json", "README.md", "CHANGELOG.md", "LICENSE", "dist/cli.js", "dist/index.js", "dist/index.d.ts", "dist/update.js", "schema/skillhook.schema.json", "schema/skillhook.yaml.schema.json", "examples/skills/hello/SKILL.md", "examples/skills/sentry-triage/SKILL.md"]; + const required = ["package.json", "README.md", "CHANGELOG.md", "LICENSE", "dist/cli.js", "dist/index.js", "dist/index.d.ts", "dist/update.js", "dist/cloud/protocol.js", "dist/cloud/protocol.d.ts", "schema/skillhook.schema.json", "schema/skillhook.yaml.schema.json", "examples/skills/hello/SKILL.md", "examples/skills/sentry-triage/SKILL.md"]; const missing = required.filter((f) => !files.has(f)); const unwanted = [...files].filter((f) => /^(src|test|scripts|docs|skills|\.github)\//.test(f) || /\.test\.|\.env|\.tgz$|\.map$/.test(f)); if (missing.length || unwanted.length) { @@ -123,3 +132,7 @@ jobs: skillhook doctor --dir "$home" --json > "$RUNNER_TEMP/doctor.json" || true node -e 'const r = JSON.parse(require("node:fs").readFileSync(process.argv[1], "utf8")); if (!Array.isArray(r.checks)) process.exit(1);' "$RUNNER_TEMP/doctor.json" skillhook update --dir "$home" --json || true + skillhook stats --dir "$home" --json > /dev/null + skillhook deliveries list --dir "$home" --json > /dev/null + skillhook run --file examples/skills/hello/SKILL.md --dir "$home" --payload '{"name":"ci"}' --dry-run --json > /dev/null + SKILLHOOK_NO_UPDATE_CHECK=1 skillhook health --quick --local --dir "$home" --json > /dev/null || true diff --git a/AGENTS.md b/AGENTS.md index 2c1c12d..d4fdc5b 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -27,10 +27,16 @@ is `skillhook`. User docs: `README.md`, `docs/`, `llms.txt`. | `src/prompt.ts` | Placeholders, event block, unattended-run guardrails. | | `src/schedule.ts`, `src/scheduler.ts` | Cron parsing and next/previous occurrence in an IANA zone (pure, no deps); the scheduler that fires `schedule:` hooks from `serve` (wall-clock tick, `catch_up` / `overlap`, exactly-once slots via the delivery index, state in `jobs/.schedules.json`). `src/commands/schedules.ts` is the CLI. | | `src/jobs.ts`, `src/queue.ts`, `src/run.ts` | Job directories on disk, the concurrency queue, invocation preparation. | -| `src/runners/` | `claude.ts`, `codex.ts`, `shell.ts`: build argv, parse output; `env.ts` is the env allow-list. | +| `src/events.ts` | The in-process event bus (`Events`, `EventMap`): the queue publishes `job.*`, the scheduler `schedule.*`, the registry `skill.changed`, `serve` `server.*`; `GET /events` and `GET /jobs//events` stream it (SSE, `openEventStream` in `src/server.ts`). The cloud link will subscribe to the same bus. | +| `src/progress.ts`, `src/answer.ts`, `src/mcp-job.ts`, `src/commands/job.ts` | The job API for the running agent and the human loop. `progress.ts` is the file model in the job directory (`progress.jsonl`, `progress.json`, `question.json`, `answer.json`) that the queue watches; `mcp-job.ts` serves it as the per-run MCP server (`skillhook mcp --job`, injected by the runners) and `commands/job.ts` as `skillhook job progress\|ask\|outcome\|note\|context`; `answer.ts` (leaf, like `manual.ts`) delivers a person's answer live or as a `trigger: resume` job that reopens the session. | +| `src/runners/` | `claude.ts`, `codex.ts`, `shell.ts`: build argv, parse output; `env.ts` is the env allow-list (`baseRunEnv` is also what probes run with); `failure.ts` classifies a failed run (`failure.kind`, from the CLIs' captured lines) and holds the `fallback` / `retry` schemas. | +| `src/cloud/` | The Skillhook Cloud side of this machine (`control.ts`: the commands that act on it, with the config keys the cloud may never change; `seal.ts`: X25519 + AES-GCM sealing for secrets). `protocol.ts` is the wire protocol as pure zod (no `node:` imports; exported as `@meterapp/skillhook/protocol`, the cloud repo imports it), with the vocabulary repeated as literals and a drift test; `config.ts` holds the URL rules, the kill switch and `commandAllowed`; `link.ts` is the sync loop `serve` runs (idle until `cloud.enabled`), `outbox.ts` the spool and ledgers in `jobs/.cloud/`, `redact.ts` what is removed before anything leaves, `commands.ts` the command dispatcher, `ingress.ts` hosted deliveries replayed to the local server, `pair.ts` / `src/commands/cloud.ts` pairing. Tests talk to `src/test-support/fake-cloud.ts`, never to a real cloud. | +| `src/stats.ts` | Pure aggregation over job records and delivery records (`computeStats`) and `collectStats` over the store and the log: `GET /stats`, `skillhook stats`, MCP `get_stats`. New numbers go here with a unit test on synthetic records. | +| `src/readiness.ts` | Is a runner installed and logged in (`checkReadiness`, `ReadinessCache`): the queue's pre-flight before every job, `GET /runners`, `skillhook runners`, `runners.changed`. A not-ready runner fails the job fast or hands it to a `fallback:` runner; a failed run may be retried or handed over only before the agent produced anything. | | `src/ops.ts` | Shared operations (create skill, run locally, sign+send, resolve URLs). CLI and MCP both call this; do not duplicate logic in either. | | `src/mcp.ts` | MCP server (`@modelcontextprotocol/server` v2, stdio). Tools wrap `ops.ts`. | -| `src/tailscale.ts`, `src/service.ts`, `src/doctor.ts` | Funnel/Serve, launchd/systemd, diagnostics. | +| `src/tailscale.ts`, `src/service.ts` | Funnel/Serve, launchd/systemd. | +| `src/health.ts`, `src/doctor.ts`, `src/tools.ts` | The grouped health report (`runHealth`, `HealthCache` behind `GET /health/checks`, `health.changed`); `doctor.ts` is its quick flavour printed flat; `tools.ts` probes `claude` / `codex` (version, login, `mcp list`, `plugin list`, `codex doctor`) with the job environment (`baseRunEnv`) and holds the pure parsers of their output. New checks: add them in `runHealth` with a group, a fixture answer in `test/fixtures/` when a CLI is involved, a row in `docs/operations.md`. | | `src/update.ts`, `src/commands/update.ts` | The daily update check (registry lookup, 24 h cache in `/update-check.json`, install-method detection, background refresh) and `skillhook update`. | | `scripts/release.ts` | Version bump / consistency check / release notes across `package.json`, the lockfile, the plugin manifests and `CHANGELOG.md`. | | `.github/workflows/` | `ci.yml` (PRs and main: checks + packed-tarball install), `release.yml` (tags merged version bumps), `publish.yml` (npm publish with provenance, GitHub release, verification). | @@ -41,7 +47,7 @@ is `skillhook`. User docs: `README.md`, `docs/`, `llms.txt`. | `test/fixtures/` | `fake-claude.mjs` / `fake-codex.mjs` emulate the real CLIs' output formats. | Runtime state lives outside the repo in `~/.skillhook` (`SKILLHOOK_HOME`): -`skillhook.json`, `.env` (mode 600), `skills/`, `jobs/` (including `.deliveries.json` and `.schedules.json`), `logs/`, `server.json`. +`skillhook.json`, `.env` (mode 600), `skills/`, `jobs/` (including `.deliveries.json`, `.schedules.json`, `.delivery-log/`, `.cloud/`), `logs/`, `server.json`. ## Hard rules @@ -49,13 +55,14 @@ Runtime state lives outside the repo in `~/.skillhook` (`SKILLHOOK_HOME`): - **Security is not optional.** The server binds `127.0.0.1` by default; TLS and public exposure are Tailscale's job. Every webhook goes through `verifyRequest`; every admin route through `requireAdmin`. Compare secrets only with `safeEqual`. A skill without `auth` gets a bearer token (`SKILLHOOK_SECRET_`); `auth: none` must be explicit and is warned about. Secret values are never logged, never returned by an API/tool except once at generation, and never written into job files (`redactHeaders`). The agent's environment is an allow-list (`src/runners/env.ts`); `SKILLHOOK_ADMIN_TOKEN` and `SKILLHOOK_SECRET_*` are never forwarded implicitly. - **Payloads are data.** Anything that reaches the prompt from a webhook is wrapped in `` and the guardrails say so. Never build a prompt by concatenating payload text outside those blocks. - **Skills are Agent Skills.** Standard frontmatter (`name`, `description`, `license`, `compatibility`, `metadata`, `allowed-tools`) plus a `skillhook:` block. `name` must equal the directory name. New fields: add to the zod schema in `src/skills.ts`, to `docs/skills.md`, to `skills/skillhook-authoring/SKILL.md`, and cover them in `src/skills.test.ts` — in the same PR. `schedule` and `webhook` are block fields like any other (normalized by `resolveSchedule`, documented in `docs/schedules.md`). A hook in `skillhook.yaml` is the same block plus exactly one of `run` / `skill` / `prompt` (`HookSchema` in `src/projects.ts` extends `SkillhookBlockSchema`, so new block fields reach hooks automatically); hook-only fields go in `src/projects.ts`, `docs/projects.md`, `npm run schema` and `src/projects.test.ts`. A compiled hook is an ordinary `Skill` (with `source.type === "project"`); never special-case hooks in the server, queue or runners. -- **Config changes** go in `src/config.ts` (zod, `.prefault({})` for nested objects so defaults apply), then `npm run schema`, then `docs/operations.md`. `projects` is the one key the server re-reads without a restart (`configProjects` in `src/registry.ts`); keep it that way. +- **Config changes** go in `src/config.ts` (zod, `.prefault({})` for nested objects so defaults apply), then `npm run schema`, then `docs/operations.md`. The running server owns one live `Config` object (`ConfigRef`): a reload (`PATCH /config`, `POST /config/reload`, `skillhook config set`, a file edit noticed within 5 s) patches that object in place, so read config values at use time, never copy them at construction (the rate limiter takes a getter; the logger has `setLevel`, the job store `configure`). Only `host` and `port` need a restart (`RESTART_CONFIG_KEYS`); a new key is hot unless it is added there, and `config.changed` says what a reload did. - **Runners never shell-interpolate.** Argv arrays only; the prompt travels on stdin; parse the CLI's structured output (`stream-json`, JSONL). When Claude Code or Codex change flags, update the runner, `test/fixtures/`, `docs/runners.md` and the version note in `README.md` together. -- **Jobs are directories.** `job.json` is the record; artifacts sit next to it; nothing outside `~/.skillhook/jobs` is written by the server. Statuses: `queued running succeeded failed timed_out cancelled interrupted`. +- **Jobs are directories.** `job.json` is the record; artifacts sit next to it; nothing outside `~/.skillhook/jobs` is written by the server, with the cloud link's control commands as the documented exceptions (`skill.put` writes `skills//SKILL.md`, `secret.generate` / `secret.set` and a token rotation write `.env`, `config.patch` writes `skillhook.json`); its own state is `jobs/.cloud/`, skills it removes go to `jobs/.removed-skills/`. Statuses: `queued running succeeded failed timed_out cancelled interrupted`. The running agent talks to skillhook only through files in its job directory (`src/progress.ts`): no token, no HTTP, so the shell runner and a restart are covered; the queue turns them into events and record fields. +- **State changes are events.** Whatever the server learns (a job changing state, a schedule firing or skipping, a skill file appearing or changing) is emitted on `Events` (`src/events.ts`) at the place it happens, after the record on disk is updated, with the full record in the payload. Consumers (the SSE routes, later the cloud link) subscribe; they never poll job files. A new kind of state change gets a new `EventMap` entry, an emit, a row in `docs/api.md` and a test. Listener errors are logged, never thrown into the publisher. - **Every CLI command supports `--json`** and returns non-zero on failure. Register new commands in `COMMANDS` and `HELP` in `src/commands/main.ts`, then in the README table. - **Third-party facts** (Granola, Sentry, GitHub, Tailscale) are stated in `docs/` and the examples with the exact header names; change them only with a source. - **Tests are hermetic**: `tempHome()` from `src/test-support/helpers.ts`, fake runners, ephemeral ports. Never touch `~/.skillhook`, the real `claude`/`codex`, `launchctl` or `tailscale` from a test. Never reach the real npm registry either: point `SKILLHOOK_NPM_REGISTRY` at a local `node:http` server or set `SKILLHOOK_NO_UPDATE_CHECK=1`. -- **The CLI phones home exactly once a day, and only for the update check** (`src/update.ts`: the registry's `latest` dist-tag, cached 24 h, never on `--json`, in CI, or when `SKILLHOOK_NO_UPDATE_CHECK` / `update_check: false` say so). Do not add other outbound requests the user did not ask for, and never auto-install anything. +- **Outbound requests are opt-in and enumerated.** By default the CLI phones home once a day, and only for the update check (`src/update.ts`: the registry's `latest` dist-tag, cached 24 h, never on `--json`, in CI, or when `SKILLHOOK_NO_UPDATE_CHECK` / `update_check: false` say so). The one other outbound connection is the Skillhook Cloud link (`src/cloud/link.ts`), and only after `skillhook cloud connect` wrote `cloud.enabled` and `SKILLHOOK_CLOUD_TOKEN`: it talks to `cloud.url` over HTTPS only, sends only what `docs/cloud.md` lists (redacted, scrubbed of every `.env` value), obeys `cloud.mode` and the local allow/deny lists (which the cloud cannot change), and stops on `cloud.enabled: false`, `SKILLHOOK_NO_CLOUD=1` or `cloud disconnect`. Never enable it by default, from `init` or from a job; do not add other outbound requests the user did not ask for, and never auto-install anything. ## Checks diff --git a/CHANGELOG.md b/CHANGELOG.md index c69868f..f09d5f3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,150 @@ All notable changes to skillhook, newest first. The format follows [Keep a Chang ## Unreleased +## 0.5.0 (2026-09-28) + +- The cloud link checks the runners as soon as it connects, so the dashboard shows whether `claude` and + `codex` are installed and signed in before the first job runs (previously only after one). +- MCP tools `cloud_status` and `cloud_disconnect`. There is deliberately no `cloud_connect`: pairing hands + the machine to an account, so the person runs `skillhook cloud connect --code …` themselves. The setup + skill gains a Skillhook Cloud step that says so. +- The Skillhook Cloud link, opt-in. `skillhook cloud connect --code XXXX-XXXX [--control]` pairs the + machine (the token goes to `.env` as `SKILLHOOK_CLOUD_TOKEN`, `cloud.*` to skillhook.json; observe + mode unless `--control`), `cloud status` and `cloud disconnect` (which revokes the token) complete + it. The running server then keeps one outbound HTTPS connection to `cloud.url`: it uploads + deliveries, jobs, progress, schedule, config and health changes (headers redacted, command lines + dropped, every string scrubbed of every `.env` value; bodies only with `cloud.upload_payloads` and + at most 256 KiB), snapshots and periodic health reports; runs read commands (health, jobs, + deliveries, stats, logs, skills) and, in `cloud.mode: control` or when allow-listed, the ones that + act on the machine (run, test, replay and cancel jobs, answer a job waiting for a person, patch the + configuration except the bind address, runner commands and the link itself, fire a schedule, + install an update, restart a service-run server once the cloud has the answer, write or remove + skills, generate a secret returned only sealed to the requester's key, and, allow-listed only, set a + secret sealed to the machine's own key); streams a watched job's output (`job.output`) and uploads + large artifacts in chunks; and replays webhooks that arrived at the machine's hosted + URLs to the local server, where the signature is checked with the local secret (`via: "ingress"` + and `ingress_id` on the delivery record). Events wait in `jobs/.cloud/` while the cloud is + unreachable. `GET /health` (admin) reports the link as `cloud`, and doctor/health gain a + `cloud link` check. Kill switches: `cloud.enabled: false`, `SKILLHOOK_NO_CLOUD=1`, + `cloud disconnect`. The settings and the wire protocol, exported as `@meterapp/skillhook/protocol` + for the cloud to validate against, are documented in [docs/cloud.md](docs/cloud.md) and + [docs/cloud-protocol.md](docs/cloud-protocol.md). `SKILLHOOK_CLOUD_*` variables never reach a run's + environment, even when a skill lists them. + +## 0.4.0 (2026-09-28) + +- An event bus inside `skillhook serve` (`src/events.ts`): the queue publishes `job.queued`, + `job.started`, `job.updated`, `job.cancelled` and `job.finished`, the scheduler + `schedule.registered`, `schedule.fired` and `schedule.skipped`, the registry `skill.changed` (a + `SKILL.md` or `skillhook.yaml` that appeared, changed or disappeared, noticed on the next lookup or + listing) and the server `server.started` / `server.stopping`. Every event carries a `seq`, a + timestamp and the full record. +- Two streaming admin routes (server-sent events): `GET /events` (the whole bus, `?types=` to filter) + and `GET /jobs//events` (one job: `status` snapshots, `stdout`/`stderr` as they are written, + `end`). `GET /jobs//artifacts/` returns one artifact file as-is (`?tail=`). + `skillhook jobs logs -f` follows a running job through the server when one is running. +- Many senders waiting with `?wait=` on the same server no longer trigger Node's + `MaxListenersExceededWarning`. +- A delivery log. Every request to `/hooks/` is now recorded in `jobs/.delivery-log/` with its + outcome (`accepted`, `duplicate`, `in_flight`, `skipped`, `rejected`, `challenge`, `error`), the HTTP + status and error code the sender got, the reason (the failing `when` condition, the auth error), the + redacted headers, the client IP and the job it created or was folded into. Refused deliveries keep + their body (`deliveries.store_bodies`, `deliveries.body_max_bytes`, 64 KiB) so what arrived can be + inspected and, later, replayed; the newest `deliveries.max` (2000) records are kept. New: + `skillhook deliveries list|show`, `GET /deliveries` and `GET /deliveries/?include=body`, the MCP + tools `list_deliveries` and `get_delivery`, `recent_deliveries` in `skillhook_status`, `deliveries` + in `GET /health` (admin) and the `delivery.received` event. +- `GET /jobs`, `skillhook jobs list` and the MCP tool `list_jobs` page with `after` (`next_after` in + the response) and filter by `trigger` and `since`; the route caps `limit` at 500 and answers + `400 bad_request` for an unknown `status` or `trigger`. A malformed skill name in a hook URL is + `404 unknown_skill` instead of `500`. +- Task outcomes. Every finished job now carries `outcome` (`completed`, `partial`, `needs_human`, + `nothing_to_do`, `failed`, `unknown`) next to `status`: the agent reports it by writing + `response.json` (`{outcome, summary, links, data}`) in the job directory (`SKILLHOOK_RESPONSE_PATH`, + `{{response_path}}`; the guardrails say so), and the report is kept as `job.response`. A new + `response:` field in the `skillhook:` block chooses how firmly it is asked for: `mode: file` asks for + the file, `mode: structured` makes the runner answer with JSON (`claude -p --json-schema`, + `codex exec --output-schema /response.schema.json`; the answer is stored as `response.json` + too), optionally against your own `schema`. A shell command that exits 0 is `completed`; a run that + reports nothing is `unknown`; every non-succeeded status is `failed`. Surfaces: the `?wait=` + response (`outcome`, `response`), `GET /jobs?outcome=`, `?include=response`, the `response` + artifact, `skillhook jobs list --outcome` (new column) and `jobs show --response`, the MCP + `list_jobs` filter and `skillhook_status`. +- Replay. `skillhook deliveries replay ` (`POST /deliveries//replay`, MCP `replay_delivery`) + runs a recorded delivery again through the skill as it is now, and `skillhook jobs replay ` + (`POST /jobs//replay`, MCP `replay_job`) does the same for any earlier job: a new job with + `trigger: replay`, `source.method: REPLAY` and `replay_of: {delivery, job}`, the original payload, + redacted headers (plus `x-skillhook-replay-of`), query string and sender IP. The signature is not + checked again (a delivery that was rejected needs `--force` / `force`), `when` filters apply unless + `--skip-filters`, nothing is de-duplicated, and `runner`/`model`/`effort` can be overridden. Through + the running server when there is one, in the CLI process otherwise. The guardrails tell the agent it + is replaying. `src/manual.ts` (manual runs) and `src/replay.ts` (the planner) are new leaf modules, + re-exported from `src/ops.ts`. +- A job API for the running agent, and a human in the loop. Every Claude and Codex run now gets a + per-run MCP server (`skillhook mcp --job`, injected with `claude --mcp-config` / + `codex -c mcp_servers.skillhook_job.*`, nothing to configure) with `job_progress`, `job_ask_human`, + `job_set_outcome`, `job_note` and `job_context`; the same is available as + `skillhook job progress|ask|outcome|note|context` (`$SKILLHOOK_BIN`) for shell skills and agents + that prefer a CLI. The guardrails explain both. Everything is files in the job directory + (`progress.jsonl`, `progress.json`, `question.json`, `answer.json`), which the queue watches: they + become the `progress`, `question` and `answer` fields of the job, the events `job.progress`, + `job.waiting_human` and `job.answered`, `GET /jobs//progress` and the timeline in + `skillhook jobs show`. `job_ask_human` waits for a person (`human_wait_seconds`, default 300; the + job's timeout clock is paused meanwhile). A person answers with `skillhook jobs answer "…"`, + `POST /jobs//answer` or the MCP tool `answer_job`: live when the agent is still waiting, + otherwise as a new job with `trigger: resume` that continues the session + (`claude -p --resume `, `codex exec resume `) with the answer in a + `` block; the two jobs are linked by `resume_of` / `resolved_by`, and a run without a + session runs the skill afresh (`runner_reason`). `skillhook jobs list --waiting`, + `GET /jobs?waiting=1` and `list_jobs {waiting: true}` show what waits for a person (an open + question, or outcome `needs_human`); a run that ends with its question unanswered counts as + `needs_human`. New block fields `agent_api` (`mcp` | `cli` | `none`) and `human_wait_seconds`; new + job variables `SKILLHOOK_BIN`, `SKILLHOOK_HOME`, `SKILLHOOK_HUMAN_WAIT_SECONDS`. +- Live configuration and remote control. The running server now holds one live `skillhook.json`: + `skillhook config set` / `unset` tell it to re-read the file (`skillhook config reload`, + `POST /config/reload`, `PATCH /config {set, unset}`, MCP `update_config`; a hand edit is noticed + within five seconds), every key but `host` and `port` applies at once, and those two are reported + as `pending_restart` (`GET /config`, MCP `get_config`). An invalid change is refused and nothing is + written. `POST /control/restart` (MCP `restart_server`) stops a service-run server gracefully and + lets launchd / systemd start it again; `GET /service`, `GET /logs` and `POST /update` (MCP + `check_update`) expose the service status, its log and the update check to the admin API. New event + `config.changed`. Internally `ConfigRef` patches the live config in place, the logger and the job + store take new settings, the rate limiter reads its limit at use, and the queue can `drain`. +- Stats. `skillhook stats [--since 24h|7d|ISO] [--until ISO] [--skill S]`, `GET /stats` and the MCP tool + `get_stats` sum up the job directories and the delivery log: jobs by status, outcome, trigger, runner + and failure kind, success and completion rates, duration and queue-wait percentiles, cost and + tokens (Claude and Codex usage added up), deliveries by outcome and HTTP status, and the same per + skill. +- Runner readiness, failure kinds and fallback. Before a job spawns, skillhook checks that its runner is + installed and logged in (or has an API key), with the job environment, cached for + `health.readiness_cache_seconds` (60): `skillhook runners`, `GET /runners`, MCP `get_runners`, event + `runners.changed`. A runner that is not ready fails the job at once (`failure.kind: auth`, no + process started) unless the skill's new `fallback: { runners: [codex] }` (or `defaults.fallback` in + `skillhook.json`) names a ready runner, which then takes over (`runner_requested`, `runner_reason`). + Every `failed` or `timed_out` job now carries `failure: {kind, code, retryable, message}` (`auth`, + `usage_limit`, `rate_limit`, `budget`, `max_turns`, `not_found`, `timeout`, `crash`, `unknown`), + classified from what the CLI printed; `jobs list --failure`, `GET /jobs?failure=` and `list_jobs` + filter by it. `fallback.on` may add `auth`, `usage_limit`, `rate_limit`, `crash`, and the new + `retry: { attempts, on, backoff_seconds }` repeats a run on the same runner; both act only on a run + that failed before the agent produced anything, and record the earlier runs in `attempts`. +- Deep health. `skillhook health` (`GET /health/checks`, MCP `get_health`) is the doctor plus what the + agents actually depend on, grouped (`system`, `skillhook`, `runners`, `tools`, `skills`, `exposure`): + `claude` / `codex` versions and logins, one check per MCP server Claude Code and Codex know + (connected, needs authentication, failed to connect, with the CLI's reason), Claude's MCP config + diagnostics and installed plugins, `codex doctor`, free disk space, and per skill the last run and + any `env:` name that is not set. The probes run with the same environment as a job, so + `CLAUDE_CONFIG_DIR`, `CODEX_HOME` or an API key in `.env` apply to the diagnosis. The server keeps + one report per flavour for `health.cache_seconds` (60; `health.probe_timeout_seconds`, 20, bounds + `claude mcp list`), answers `GET /doctor` and `GET /health/checks` from it (`?refresh=1`, + `?deep=0`, `?network=1`) and publishes `health.changed` when a check changes status. `doctor` + gained `disk` and shows the CLI versions; every check now carries `group` and `data`. +- Ad-hoc runs. `skillhook run --file SKILL.md` (or `--stdin`), `POST /skills/test` and the MCP tool + `test_skill` run a SKILL.md that is not installed: the document is validated, kept at + `jobs//skill//SKILL.md` and run from there, as a job with `trigger: test`, `adhoc: true`, + `skill_file` and `source.method: TEST`; nothing is added to `/skills`. `--dry-run` works with + `--file` too. Every job now records `skill_file` (the SKILL.md or skillhook.yaml it ran from), and + `skillhook run --cwd` applies to real runs, not only to `--dry-run`. + ## 0.3.0 (2026-09-23) - Scheduled hooks. A `schedule:` key on any skill (`skillhook:` block) or hook (`skillhook.yaml`) runs it diff --git a/README.md b/README.md index e0ff068..94cb1ba 100644 --- a/README.md +++ b/README.md @@ -32,15 +32,17 @@ Give everything that has a trigger a webhook. Anything that can call a URL can s in the skill's cwd, prompt = SKILL.md body + payload, unattended-run guardrails │ ▼ - ~/.skillhook/jobs// job.json · payload.json · event.json · prompt.md · stdout.log · result.md + ~/.skillhook/jobs// job.json · payload.json · event.json · prompt.md · stdout.log · result.md · response.json ``` - A skill is a directory `~/.skillhook/skills//SKILL.md`: standard Agent Skills frontmatter plus a `skillhook:` block that sets the runner, model, authentication, filters and working directory. Edits apply to the next delivery without a restart. - A repository can carry its own hooks in a version-controlled `skillhook.yaml` (webhook name → a shell command, a `SKILL.md` in the repository, or inline instructions); `skillhook link ` serves them. See [Version-controlled hooks](#version-controlled-hooks-in-a-repository). - Any skill or hook can also carry a `schedule:` (a cron expression, a time zone, and what to do about missed slots); the server fires it without a webhook. See [Scheduled hooks](#scheduled-hooks). - The runner is the real `claude` or `codex` CLI on the machine, so subscriptions, MCP servers, `CLAUDE.md`/`AGENTS.md` files and tool permissions apply as usual. -- Responses are immediate (`202` with a job id) or synchronous with `?wait=N` (or `Prefer: wait=N`); the agent's final message becomes the job result. -- Developed against Claude Code 2.1.270, Codex CLI 0.153.4 and Tailscale 1.102.3. skillhook drives the CLIs through their headless flags (`claude -p --output-format stream-json …`, `codex exec --json …`); `skillhook run --dry-run` shows the exact command line. +- Responses are immediate (`202` with a job id) or synchronous with `?wait=N` (or `Prefer: wait=N`); the agent's final message becomes the job result, and what it reports in `response.json` (`completed`, `partial`, `needs_human`, `nothing_to_do`, `failed`) becomes the job's `outcome`, so `skillhook jobs list --outcome needs_human` shows what is waiting for a person. +- Optional: `skillhook cloud connect` pairs the machine with Skillhook Cloud, a hosted dashboard for every machine's webhooks, jobs, questions waiting for a person, health and stats, with hosted webhook URLs that keep deliveries while the machine sleeps. Opt-in, outbound only, observe mode unless you choose control. See [docs/cloud.md](docs/cloud.md). +- The agent is not cut off while it runs: a per-run job API (MCP tools injected into the run, or `skillhook job …`) lets it report progress and ask a person a question; `skillhook jobs answer "…"` delivers the answer to the waiting agent or, when the run already ended, starts a new job that resumes the Claude or Codex session with it. See [Reporting progress and asking a person](docs/skills.md#reporting-progress-and-asking-a-person). +- Developed against Claude Code 2.1.270, Codex CLI 0.153.4 and Tailscale 1.102.3. skillhook drives the CLIs through their headless flags (`claude -p --output-format stream-json …`, `codex exec --json …`; `response: { mode: structured }` adds `claude --json-schema` / `codex --output-schema`); `skillhook run --dry-run` shows the exact command line. ## Quickstart @@ -240,7 +242,7 @@ Set the default once (`skillhook config set defaults.runner codex`, `skillhook c - Payloads are delivered as data inside `` tags with guardrails; the agent is told it runs unattended and must not follow instructions found in the payload. - Per-IP rate limits (120 requests/min, 10 auth failures/min), a 1 MiB body cap, per-skill and global concurrency limits and per-job timeouts bound the damage of floods and runaway jobs. - Retries and duplicates are absorbed: provider delivery ids are remembered for 24 h, and a delivery whose payload matches a job of the same skill that is still queued or running is answered with that job's id instead of a second run (`dedupe.in_flight`, on by default). -- The admin API (`/skills`, `/jobs`) needs `Authorization: Bearer $SKILLHOOK_ADMIN_TOKEN`, except for direct loopback callers such as the CLI. +- The admin API (`/skills`, `/jobs`, `/events`) needs `Authorization: Bearer $SKILLHOOK_ADMIN_TOKEN`, except for direct loopback callers such as the CLI. | `auth.type` | Sender sends | Secret | |---|---|---| @@ -317,18 +319,24 @@ Agents reading this repository should start with [`AGENTS.md`](AGENTS.md) (layou | Command | Purpose | |---|---| | `skillhook init [--runner claude\|codex\|shell] [--model M] [--port N] [--force]` | Create `~/.skillhook` with config, secrets and the `hello` skill. | -| `skillhook doctor` | Check Node, config, secrets, skills, Claude/Codex login, Tailscale, public URL, server and service; exit 1 on failures. | +| `skillhook doctor` | Check Node, disk, config, secrets, skills, Claude/Codex login, Tailscale, public URL, server and service; exit 1 on failures. | +| `skillhook runners [--refresh] [--local]` | Is each runner installed and logged in (or given an API key): what every job checks before it starts. | +| `skillhook health [--quick] [--refresh] [--no-network] [--local]` | The doctor plus every MCP server Claude Code and Codex know, plugins, `codex doctor` and each skill's last run, grouped; via the running server's cached report when there is one. | | `skillhook serve [--port N] [--host H] [--pretty] [--log-level L]` | Run the webhook server in the foreground. | | `skillhook service install\|uninstall\|status\|restart\|logs [--lines N] [-f]` | Run the server at login (launchd on macOS, systemd `--user` on Linux). | | `skillhook expose tailscale [--serve] [--port N]` · `expose status` · `expose off` · `expose cloudflare\|ngrok` | Get a permanent HTTPS URL via Tailscale Funnel or Serve; print recipes for other tunnels. | | `skillhook url [skill] [--public\|--local]` | Print webhook URLs. | | `skillhook skills list\|show \|new \|validate [name]\|examples\|add [--as NAME]\|path ` | Manage `SKILL.md` files (`new` takes `--description`, `--runner`, `--model`, `--effort`, `--auth`, `--secret-env`, `--cwd`, `--timeout`, `--env`, `--no-secret`, `--force`). | | `skillhook secret set [--value V\|--stdin]` · `secret generate [--force] [--bytes N]` · `secret list` · `secret unset ` | Manage `.env` (values are shown once at generation, never afterwards). | -| `skillhook run [--payload JSON\|@file\|-] [--header "N: v"] [--runner R] [--model M] [--effort E] [--cwd DIR] [--dry-run]` | Run a skill locally, no HTTP, no authentication. | +| `skillhook run [--payload JSON\|@file\|-] [--header "N: v"] [--runner R] [--model M] [--effort E] [--cwd DIR] [--wait S] [--dry-run]` · `run --file SKILL.md \| --stdin [same options]` | Run a skill locally, no HTTP, no authentication; `--file`/`--stdin` run a SKILL.md that is not installed (kept with the job). | | `skillhook send [--payload …] [--wait N] [--url BASE\|--public\|--local] [--header "N: v"]` | POST a correctly signed test webhook to the running server or the public URL. | -| `skillhook jobs list [--skill S] [--status ST] [--limit N]` · `jobs show [--result] [--prompt] [--stdout] [--stderr]` · `jobs logs [-f] [--stderr]` · `jobs cancel ` · `jobs resume [--exec]` · `jobs path ` · `jobs prune [--keep N]` | Inspect and manage jobs. | -| `skillhook mcp [--print-config]` | MCP server over stdio; `--print-config` prints client configuration. | -| `skillhook config show\|get \|set \|unset \|path` | Read and edit `skillhook.json`. | +| `skillhook jobs list [--skill S] [--status ST] [--outcome O] [--failure K] [--trigger T] [--waiting] [--since ISO] [--after ID] [--limit N]` · `jobs show [--result] [--response] [--prompt] [--stdout] [--stderr]` · `jobs logs [-f] [--stderr]` · `jobs answer "" [--option X] [--by NAME] [--no-resume] [--wait S]` · `jobs cancel ` · `jobs replay [--skip-filters] [--wait S]` · `jobs resume [--exec]` · `jobs path ` · `jobs prune [--keep N]` | Inspect and manage jobs (`--waiting`: what is waiting for a person; `answer`: reply to a waiting job, live or by resuming its session; `replay`: the same request again as a new job). | +| `skillhook job progress "" [--state working\|blocked] [--percent N] [--step S]` · `job ask "" [--option A]... [--context T] [--wait S]` · `job outcome [--summary S] [--link URL]... [--data JSON]` · `job note ""` · `job context` | The job API for the agent inside a run (`$SKILLHOOK_BIN job …`; also the `job_*` MCP tools of `skillhook mcp --job`): report progress, ask a person and wait for the answer, report the outcome. | +| `skillhook stats [--since 24h\|7d\|ISO] [--until ISO] [--skill S]` | Jobs by status, outcome, runner and failure kind; durations, cost, tokens; deliveries by outcome; per skill. | +| `skillhook deliveries list [--skill S] [--outcome O] [--since ISO] [--after ID] [--limit N]` · `deliveries show [--body]` · `deliveries replay [--force] [--skip-filters] [--wait S]` | Every webhook the server received, whatever became of it: accepted, duplicate, in flight, skipped by a filter, rejected (with the status and reason), Slack challenge; replay one through the skill as it is now. | +| `skillhook mcp [--print-config]` · `mcp --job` | MCP server over stdio; `--print-config` prints client configuration; `--job` serves one run's job API (the runners start it). | +| `skillhook cloud connect --code XXXX-XXXX [--control] [--url U]` · `cloud status` · `cloud disconnect [--keep-token]` | Pair this machine with Skillhook Cloud (opt-in, outbound only; observe mode unless `--control`): webhooks, jobs, health and stats of every machine in one place, hosted webhook URLs. See [docs/cloud.md](docs/cloud.md). | +| `skillhook config show\|get \|set \|unset \|reload\|path` | Read and edit `skillhook.json`; `set`/`unset` tell the running server, which applies every key but `host` and `port` live. | | `skillhook link [dir] [--no-secret]` / `skillhook unlink ` | Serve the hooks a repository declares in its `skillhook.yaml` (default `.`); stop serving them. | | `skillhook projects [list]` / `skillhook projects init [dir] [--force]` | List linked repositories and their hooks; write a starter `skillhook.yaml` and link it. | | `skillhook schedules [list]` · `schedules next [--count N]` · `schedules run [--wait S]` | Every skill or hook with a `schedule:`, its next and last runs; preview occurrences; fire one now. | diff --git a/docs/api.md b/docs/api.md index fbd2974..e4ee887 100644 --- a/docs/api.md +++ b/docs/api.md @@ -8,6 +8,7 @@ Conventions: - Errors are `{"ok": false, "error": "", "message": ""}`; codes are listed at the end. - Timeouts: keep-alive 65 s, headers 70 s, whole request `max(300 s, max_wait_seconds + 30 s)`. - Every request counts against the per-IP limit `rate_limit.requests_per_minute` (120); beyond it the answer is `429 rate_limited`. +- `GET /events` and `GET /jobs//events` answer `text/event-stream` and stay open; every other route is one JSON (or text) response. Related: [security.md](security.md) (authentication), [skills.md](skills.md) (filters, dedupe, placeholders), [operations.md](operations.md) (job files). @@ -17,13 +18,33 @@ Related: [security.md](security.md) (authentication), [skills.md](skills.md) (fi |---|---|---|---| | `GET` | `/` | none | Banner: `skillhook ` plus a hint. | | `GET` | `/health` | none; admin for details | Liveness. Public callers get `{ok, version}`; admin callers also get `uptime_seconds`, `queue` and `schedules`. | +| `GET` | `/health/checks` | admin | The grouped health report (`skillhook health`), cached; `?deep=0`, `?network=1`, `?refresh=1`. | +| `GET` | `/doctor` | admin | The quick report (`skillhook doctor`), cached; `?network=0`, `?refresh=1`. | +| `GET` | `/runners` | admin | Is each runner installed and logged in (what a job checks before it starts); `?refresh=1`. | +| `GET` | `/stats` | admin | Numbers over jobs and deliveries: by status, outcome, runner, failure kind; durations, cost, tokens; per skill (`?since=24h`, `?until=`, `?skill=`). | +| `GET`, `PATCH` | `/config` | admin | The live `skillhook.json` (defaults applied) with what applies live and what needs a restart; change keys in one validated write. | +| `POST` | `/config/reload` | admin | Re-read `skillhook.json` (after editing it by hand). | +| `POST` | `/control/restart` | admin | Stop, let running jobs finish, exit; launchd / systemd starts the server again. `409` when not run as the service. | +| `GET` | `/service` | admin | The launchd / systemd service status. | +| `GET` | `/logs` | admin | The last lines of the service log (`?lines=200`). | +| `POST` | `/update` | admin | Ask npm for a newer skillhook; `{"install": true}` installs it (the server does not restart itself). | | `GET`, `HEAD` | `/hooks/` | none | `200` text when the skill exists, is enabled and has a webhook, `404` otherwise (`schedule_only` for a `webhook: false` skill). Lets providers "test" the URL. | | `POST`, `PUT` | `/hooks/` | the skill's `auth` | Deliver a webhook. `404 schedule_only` for a skill with `webhook: false`. | | `GET` | `/skills` | admin | Every skill with its effective settings. | | `POST` | `/skills//run` | admin | Run a skill with an arbitrary payload, bypassing webhook auth. | -| `GET` | `/jobs` | admin | Recent jobs. | +| `POST` | `/skills/test` | admin | Run a SKILL.md that is not installed (the document travels in the body). | +| `GET` | `/jobs` | admin | Recent jobs (`?waiting=1`: only those waiting for a person). | | `GET` | `/jobs/` | admin | One job, optionally with artifacts. | | `POST` | `/jobs//cancel` | admin | Cancel a queued or running job. | +| `GET` | `/jobs//progress` | admin | What the agent reported: current state, pending question, answer, timeline. | +| `POST` | `/jobs//answer` | admin | A person's answer: delivered live to a waiting job, or a new job continues the session. | +| `GET` | `/jobs//artifacts/` | admin | One artifact file as it is on disk (`?tail=` for its end). | +| `GET` | `/jobs//events` | admin | Server-sent events for one job: `status` snapshots, `stdout`/`stderr` as they are written, `end`. | +| `GET` | `/events` | admin | Server-sent events for the whole server: `delivery.received`, `job.*`, `schedule.*`, `skill.changed`, `server.*` (`?types=` to filter). | +| `GET` | `/deliveries` | admin | Every webhook received, newest first, whatever became of it. | +| `GET` | `/deliveries/` | admin | One delivery, optionally with its body. | +| `POST` | `/deliveries//replay` | admin | Run a recorded delivery again, as a new job. | +| `POST` | `/jobs//replay` | admin | Run the request an earlier job received again, as a new job. | Anything else is `404 not_found`; another method on `/hooks/` is `405 method_not_allowed`. @@ -42,6 +63,8 @@ Processing order: 9. In-flight check (`dedupe.in_flight`, default `jobs.dedupe_in_flight` = `true`): a payload and query string identical to a job of this skill that is still queued or running -> `200` with `duplicate: true`, `in_flight: true` and that job's `job_id`; with `?wait=` the response waits for that job instead. 10. The job is written to disk and queued; the response is sent. +Whatever the step it stopped at, every request to `/hooks/` is recorded in the delivery log with its outcome, the status it was answered with and the reason ([`GET /deliveries`](#get-deliveries)); a refused delivery keeps its body so it can be inspected and replayed. + ### Responses Asynchronous (default): `202 Accepted` @@ -56,7 +79,7 @@ Asynchronous (default): `202 Accepted` } ``` -Synchronous: add `?wait=` or send a `Prefer: wait=` header (clamped to `max_wait_seconds`, default 120). When the job finishes in time the answer is `200`; `ok` reflects the job outcome: +Synchronous: add `?wait=` or send a `Prefer: wait=` header (clamped to `max_wait_seconds`, default 120). When the job finishes in time the answer is `200`; `ok` reflects the job status, and the body also carries `outcome` and `response` (whether the task was done and what the agent reported, `null` when it reported nothing; see [skills.md](skills.md#reporting-the-outcome)): ```json { @@ -153,7 +176,7 @@ Admin routes accept `Authorization: Bearer `. Without a t ## `GET /health` -Public: `{"ok": true, "version": "0.1.0"}`. Admin or direct local: adds `"uptime_seconds"`, `"queue": {"running": 0, "queued": 0, "running_ids": []}` and `"schedules"`, one entry per skill or hook with a `schedule:`: +Public: `{"ok": true, "version": "0.1.0"}`. Admin or direct local: adds `"cloud"` (the Skillhook Cloud link: `{state, reason?, mode, enabled, url, machine_id, last_sync_at, last_error, connected_since, syncs, outbox_depth, dropped_total, ingress_urls, …}`, or `null` for a server without one), `"uptime_seconds"`, `"queue": {"running": 0, "queued": 0, "running_ids": []}`, `"deliveries": {"total": 412, "last_received_at": "2026-09-28T10:00:02.000Z"}` (the delivery log) and `"schedules"`, one entry per skill or hook with a `schedule:`: ```json { "skill": "weekly-review", "cron": "0 16 * * 5", "timezone": "America/New_York", "catch_up": "latest", "overlap": "skip", "enabled": true, "webhook": false, "next_due": "2026-09-25T20:00:00.000Z", "last_slot": "2026-09-18T20:00:00.000Z", "last_fired_at": "2026-09-18T20:00:09.120Z", "last_job": "20260918T200009Z-k3x9q2", "last_status": "succeeded", "skipped": 0 } @@ -161,6 +184,18 @@ Public: `{"ok": true, "version": "0.1.0"}`. Admin or direct local: adds `"uptime The CLI and MCP server use this route to detect a running server, and `skillhook schedules list` prefers its live `schedules` over the state file. +## `GET /health/checks` + +The report of [`skillhook health`](operations.md#health): `{checks, ok, summary, groups, generated_at, duration_ms, deep, network, cached, public_url?, server?}`. Each check is `{name, status: ok|warn|fail|skip, detail, hint?, group: system|skillhook|runners|tools|skills|exposure, data?}`. The server keeps one report per flavour for `health.cache_seconds` (60) and answers from it (`cached: true`); `?refresh=1` probes again, `?deep=0` leaves out the slow probes (MCP servers, plugins, `codex doctor`, last runs), and `?network=1` also asks the npm registry for a newer version and probes the public URL (off by default: the server makes no outbound request unless asked). The `server` check describes this very process (uptime, queue). Concurrent callers share one probe run. + +## `GET /doctor` + +The quick flavour, as `skillhook doctor` prints it: `GET /health/checks?deep=0` with `network` on by default (`?network=0` to turn it off). + +## `GET /runners` + +`{runners: [{runner, found, path?, version?, authenticated, method?, detail, hint?, ready, checked_at}], default_runner}` for `claude`, `codex` and `shell`: whether each is installed and logged in or given an API key, as the queue checks before every job ([runners.md](runners.md#readiness)). Answers are cached for `health.readiness_cache_seconds`; `?refresh=1` probes again. + ## `GET /skills` ```json @@ -236,20 +271,40 @@ curl -sS -X POST http://127.0.0.1:8787/skills/hello/run \ -d '{"payload":{"name":"Dee"},"wait":60,"model":"sonnet"}' ``` +## `POST /skills/test` + +Runs a SKILL.md that is not installed: the document is validated like any skill file, written to `jobs//skill//SKILL.md` (the server writes nothing outside the jobs directory) and run from there, with `trigger: "test"`, `adhoc: true`, `skill_file` pointing at that copy and `source.method: "TEST"`. Nothing is added to `/skills`, and the job's default working directory is the copy's own directory unless the document or `cwd` says otherwise. + +| Field | Type | Meaning | +|---|---|---| +| `skill_md` | string | The whole SKILL.md text, frontmatter included. Its `name` must be a valid skill name; the frontmatter and `skillhook:` block are validated as usual. | +| `payload`, `headers`, `runner`, `model`, `effort`, `wait` | | As in `POST /skills//run`. | +| `cwd` | string | Working directory for the run. | + +Responses are the webhook shapes plus `adhoc: true`. An invalid document is `400 invalid_skill_document` with the validation message; a missing `skill_md` is `400 bad_request`. This is what `skillhook run --file` / `--stdin` and the MCP `test_skill` tool use when a server is running. + +```bash +curl -sS -X POST -H "Authorization: Bearer $SKILLHOOK_ADMIN_TOKEN" -H "Content-Type: application/json" \ + -d "$(jq -n --rawfile md draft/SKILL.md '{skill_md: $md, payload: {name: "Dee"}, wait: 120}')" http://127.0.0.1:8787/skills/test +``` + ## `GET /jobs` -Query: `skill=`, `status=`, `limit=` (default 50). Newest first. +Query: `skill=`, `status=`, `outcome=` (derived for jobs recorded before outcomes existed; queued and running jobs never match), `trigger=`, `failure=` (jobs that failed that way, see [runners.md](runners.md#failure-kinds)), `waiting=1` (only jobs waiting for a person: an unanswered question, or a finished job with outcome `needs_human` that nobody answered or resumed yet), `since=` (created at or after; whole seconds), `after=` (only older jobs: the `next_after` of the previous page), `limit=` (default 50, at most 500). Newest first. An unknown `status`, `outcome`, `trigger` or `since` value is `400 bad_request`. ```json { "jobs": [ { "id": "20260916T025443Z-z1y3m4", "skill": "hello", "status": "succeeded", "…": "…" } ], - "queue": { "running": 0, "queued": 0, "running_ids": [] } + "queue": { "running": 0, "queued": 0, "running_ids": [] }, + "next_after": "20260916T025443Z-z1y3m4" } ``` +`next_after` is the last id of a full page (pass it as `after` for the next one) and `null` when the page was not full. + ## `GET /jobs/` -`?include=result,stdout,stderr,prompt,payload,event` adds an `artifacts` object with file contents (each capped to its last 512 KiB and prefixed with `… [N bytes omitted]` when truncated; a missing file is `null`). +`?include=result,stdout,stderr,prompt,payload,event,response` adds an `artifacts` object with file contents (each capped to its last 512 KiB and prefixed with `… [N bytes omitted]` when truncated; a file the job did not write is left out). ```bash curl -sS "http://127.0.0.1:8787/jobs/20260916T025442Z-r1wn6g?include=result,prompt" @@ -271,6 +326,177 @@ Ids that do not exist (or do not look like `YYYYMMDDTHHMMSSZ-xxxxxx`) are `404 u `200 {"ok": true, "job_id": "…", "status": "…"}` when the job was queued (it becomes `cancelled` at once) or running (SIGTERM now, SIGKILL after 10 s, then `cancelled`). `409 {"ok": false, "job_id": "…", "status": "succeeded"}` when it had already finished. +## `GET /jobs//progress` + +What the running (or finished) agent reported through the job API ([skills.md](skills.md#reporting-progress-and-asking-a-person)): `{job_id, status, outcome, waiting, progress?, question?, answer?, timeline}`. `progress` is the current state (`{state: working|blocked|waiting_human|done, message, percent?, step?, updated_at}`), `question` the pending or last question (`{id, text, options?, context?, asked_at, wait_until?, answered_at?}`), `answer` the person's answer (`{question_id?, text, option?, by?, at}`) and `timeline` the entries of `progress.jsonl`, oldest first (`?limit=` keeps the last N, default 200). All of it is also on the job record. + +## `POST /jobs//answer` + +A person answers a job. Body: + +| Field | Type | Meaning | +|---|---|---| +| `answer` | string | The answer (required). | +| `option` | string | One of the question's options, when it had any. | +| `by` | string | Who answered, for the record and the agent. | +| `resume` | `auto` \| `never` | For a job that already ended: `auto` (default) starts a new job that continues the session, `never` only records the answer. | +| `wait` | number | Seconds to wait for the resume job (also `?wait=`); clamped to `max_wait_seconds`. | + +Response: `{ok, job_id, delivered, answer, resume_job_id, resume_job?, job}` with `delivered` one of `live` (the job is running and waiting; the agent's `ask` call returns the answer), `resumed` (`resume_job` is the new job with `trigger: "resume"` and `resume_of`; the original gets `resolved_by`) or `recorded`. `409 not_waiting` when the job is not waiting for a person (nothing asked, already answered, already resumed, still queued); `404 unknown_job`; `400 bad_request` without `answer` or with another `resume` value. + +```bash +curl -sS -X POST -H "Authorization: Bearer $SKILLHOOK_ADMIN_TOKEN" -H "content-type: application/json" \ + -d '{"answer":"Go with the smaller change","option":"A","by":"ada","wait":120}' \ + http://127.0.0.1:8787/jobs/20260916T025442Z-r1wn6g/answer +``` + +## `GET /jobs//artifacts/` + +`` is one of `stdout`, `stderr`, `prompt`, `result`, `payload`, `event`, `response`. The body is the file as written, with no JSON envelope: `application/json` for `event` and for a `payload` that was parsed as JSON, `text/plain` otherwise. `x-artifact-bytes` carries the file's full size. `?tail=` returns only the last `` bytes and adds `x-artifact-truncated: true`. A name outside the list, or a file the job has not written yet, is `404 unknown_artifact`. + +```bash +curl -sS -H "Authorization: Bearer $SKILLHOOK_ADMIN_TOKEN" "http://127.0.0.1:8787/jobs/20260916T025442Z-r1wn6g/artifacts/result" +``` + +## `GET /jobs//events` + +A `text/event-stream` that follows one job. Messages, in order: + +- `event: status`, `data:` the job record: once at connect, then after each change (`running`, `pid`/`session_id` captured, cancel requested); +- `event: stdout` / `event: stderr`, `data:` a JSON string with the new bytes, sent as the files grow (`?streams=stdout,stderr`; default `stdout`; a file already larger than 512 KiB starts at its tail); +- `event: end`, `data:` the final record, after which the server closes the stream. + +A job that has already finished gets `status`, the whole output and `end` at once. A comment line (`: ping`) every 15 s keeps proxies from closing an idle stream. `skillhook jobs logs -f` uses this route when a server is running, and reads the file otherwise. + +## `GET /events` + +A `text/event-stream` of the server's event bus. Each message carries `id` (the event's `seq`, increasing by one per event in this server process), `event` (the type) and `data` (the whole event, `{"seq", "type", "at", "data"}`). `?types=job.finished,schedule.fired` limits it to those types; an unknown type is `400 bad_request`. Events that happened before the connection, or while it was down, are not replayed: a consumer that reconnects should reconcile through `/jobs` and `/health`. + +| Type | `data` | +|---|---| +| `server.started`, `server.stopping` | `{state}` (the `server.json` record) and `{reason, running}` | +| `job.queued`, `job.started`, `job.finished` | `{job}` | +| `job.updated` | `{job, fields}`: `pid`, `session_id`, `resume_command` captured while running; `runner`, `runner_requested`, `runner_reason` when a fallback runner takes over; `attempts` when a run is repeated | +| `job.cancelled` | `{job, state}` with `state` `queued` or `running`; `job.finished` follows | +| `job.progress` | `{job, entry}`: the agent reported progress, a note or its outcome (`entry` is the `progress.jsonl` line) | +| `job.waiting_human` | `{job, question}`: the agent asked a person and waits | +| `job.answered` | `{job, answer, delivered, resume_job_id?}` with `delivered` `live`, `resumed` or `recorded` | +| `schedule.registered` | `{skill, cron, timezone, next_due}` | +| `schedule.fired` | `{skill, slot, job, caught_up}` | +| `schedule.skipped` | `{skill, slot, reason}`: `in_flight`, `caught_up`, `too_old` or `duplicate` | +| `skill.changed` | `{name, action, source}` with `action` `added`, `changed` or `removed`, noticed when a lookup or listing reads the changed file | +| `health.changed` | `{report, changed}`: a fresh health report whose checks differ from the previous one of the same flavour (`changed` lists `{name, from, to}`; the first report of a flavour has `from: null`) | +| `runners.changed` | `{runner, readiness, previous?}`: a runner became usable or stopped being so (installed, logged in), as the readiness check sees it | +| `config.changed` | `{changed, applied, restart_required, pending_restart, config}`: `skillhook.json` was re-read; `applied` took effect now, `restart_required` at the next start | + +```bash +curl -sN -H "Authorization: Bearer $SKILLHOOK_ADMIN_TOKEN" "http://127.0.0.1:8787/events?types=job.finished,schedule.fired" +``` + +## `GET /deliveries` + +The delivery log: one record per request to `/hooks/`, newest first, whatever became of it. Query: `skill=`, `outcome=`, `since=`, `after=` (the `next_after` of the previous page), `limit=` (default 50, at most 500). + +```json +{ + "deliveries": [ + { "id": "20260928T100002Z-q7m2ka", "skill": "gh", "received_at": "2026-09-28T10:00:02.418Z", "outcome": "rejected", "http_status": 401, "code": "invalid_signature", "reason": "signature mismatch", "ip": "140.82.115.6", "method": "POST", "path": "/hooks/gh", "query": {}, "headers": { "content-type": "application/json", "x-github-event": "pull_request", "x-github-delivery": "b3e4…" }, "user_agent": "GitHub-Hookshot/abc", "content_type": "application/json", "bytes": 9412, "body_stored": true, "duration_ms": 2 }, + { "id": "20260928T095910Z-x1p0ll", "skill": "hello", "received_at": "2026-09-28T09:59:10.101Z", "outcome": "accepted", "http_status": 202, "delivery_id": null, "job_id": "20260928T095910Z-k3x9q2", "ip": "127.0.0.1", "method": "POST", "path": "/hooks/hello", "query": {}, "headers": { "content-type": "application/json", "user-agent": "skillhook-send" }, "user_agent": "skillhook-send", "content_type": "application/json", "bytes": 15, "body_kind": "json", "body_stored": false, "duration_ms": 4 } + ], + "next_after": null +} +``` + +The log lives in `jobs/.delivery-log/` and keeps the newest `deliveries.max` (2000) records. It is the answer to "why did that webhook not run": a `rejected` record carries the error code and message the sender got, a `skipped` one the `when` condition that did not match, a `duplicate` or `in_flight` one the job it was folded into. Rate-limited requests to a hook are recorded too (`429 rate_limited`). CLI: `skillhook deliveries list|show`; MCP: `list_deliveries`, `get_delivery`. + +## `GET /deliveries/` + +`{"delivery": {…}}`; `?include=body` adds `"body": {"encoding": "utf8" | "base64", "text": "…", "truncated": false, "source": "log" | "job"}`: the body the log kept for a refused delivery (`skipped`, `rejected`, `error`; at most `deliveries.body_max_bytes`, 64 KiB, and only while `deliveries.store_bodies` is on), or the payload of the job an accepted delivery created; `null` when neither exists. An unknown id is `404 unknown_delivery`. + +## `POST /deliveries//replay` + +Runs a recorded delivery again through the skill as it is now: the original payload, headers (redacted, plus `x-skillhook-replay-of: `), query string and sender IP, as a new job with `trigger: "replay"`, `source.method: "REPLAY"` and `replay_of: {"delivery": "", "job": ""}` (the job is present when the delivery had been accepted; its `event.json` and `body.bin` are then what is replayed). The signature is not checked again, `when` filters apply unless skipped, and nothing is de-duplicated: a replay never counts as a duplicate and is never folded into a job still in flight, and it carries no `delivery_id` itself. + +Body: a JSON object, all fields optional. + +| Field | Type | Meaning | +|---|---|---| +| `force` | boolean | Replay a delivery that was `rejected` or `error`, whose body was therefore never verified. Without it the answer is `409 replay_needs_force`. | +| `skip_filters` | boolean | Run even when the skill's `when` conditions do not match; otherwise a non-match answers `200 {"ok": true, "skipped": true, "reason", "replay_of"}`. | +| `runner`, `model`, `effort` | string | Overrides for this run, as in `POST /skills//run`. | +| `wait` | number | Seconds to wait for the result (also `?wait=`); clamped to `max_wait_seconds`. | + +Responses are the webhook shapes (`202` queued, `200` finished when waiting) plus `replay_of`. Errors: `404 unknown_delivery`, `404 unknown_skill` (the skill is gone or disabled), `409 replay_needs_force`, `409 no_body` (the body was not kept: `deliveries.store_bodies` was off, it was cut at `deliveries.body_max_bytes`, or the record was compacted away), `400 bad_request` (not a JSON object, or an unknown `runner`). + +```bash +curl -sS -X POST -H "Authorization: Bearer $SKILLHOOK_ADMIN_TOKEN" -H "Content-Type: application/json" \ + -d '{"skip_filters": true, "wait": 60}' "http://127.0.0.1:8787/deliveries/20260928T100002Z-q7m2ka/replay" +``` + +## `POST /jobs//replay` + +The same for an earlier job, whatever its trigger: its `event.json` (payload, redacted headers, query) is run again as a new job with `trigger: "replay"` and `replay_of: {"job": ""}`. The body takes `skip_filters`, `runner`, `model`, `effort` and `wait` as above (`force` is not needed: a job's request was accepted). `404 unknown_job` / `404 unknown_skill`. + +## `GET /config`, `PATCH /config`, `POST /config/reload` + +`GET` answers `{config, file, exists, hot_keys, restart_keys, pending_restart}`: the live configuration with defaults applied, which top-level keys the running server applies on reload (`hot_keys`, everything but `host` and `port`) and which wait for a restart (`restart_keys`), and the restart-only keys whose file value differs from what this process started with (`pending_restart`). + +`PATCH` takes `{"set": {"": value, …}, "unset": ["", …]}` (at least one of them), writes the file once after validating the whole result, re-reads it into the live config and answers `{ok, applied, restart_required, restart_required_keys, pending_restart, config, …}`. `applied` lists the top-level keys that changed and took effect now; `restart_required_keys` those written for the next start. Errors: `400 config_invalid` (the result would not validate: nothing is written), `400 config_key_not_allowed` (`$schema`, `__proto__` and friends), `400 bad_request` (shape). + +```bash +curl -sS -X PATCH -H "Authorization: Bearer $SKILLHOOK_ADMIN_TOKEN" -H "content-type: application/json" \ + -d '{"set":{"concurrency":4,"defaults.model":"sonnet"},"unset":["defaults.effort"]}' http://127.0.0.1:8787/config +``` + +`POST /config/reload` re-reads a file edited by hand (the server also notices a changed file within five seconds on its own) and answers the same shape; `400 config_invalid` when the file does not parse, in which case the running settings are kept. Every reload that changed something is a `config.changed` event. + +## `POST /control/restart` + +Body `{"force"?: boolean, "wait_seconds"?: number}` (default 30, at most 600). The server answers `202 {ok, restarting: true, force, wait_seconds, running}`, stops accepting requests, lets running jobs finish for up to `wait_seconds` (terminates them at once with `force`, or when the wait runs out), removes `server.json` and exits 0; launchd (`KeepAlive`) or systemd (`Restart=always`) starts it again, and queued jobs are re-queued at start. `409 not_a_service` when this server is not the one the service supervises (a `skillhook serve` in a terminal), since nothing would start it again. + +## `GET /service`, `GET /logs` + +`/service`: `{service: {platform, installed, running, pid, file, logFile, detail?}, this_pid, supervised}` (`supervised` is what `/control/restart` checks). `/logs?lines=200` (at most 2000): `{file, exists, lines}` from `/logs/service.log`. + +## `POST /update` + +Body `{"install"?: boolean}`. Answers what `skillhook update` prints: `{ok, current, latest, available, checked_at, registry, cached, install: {method, command}, release_notes, installed, service_restarted, service_note, error?}`. With `install: true` and a newer version, the package manager that installed skillhook runs the upgrade; the server itself keeps running the old code until `POST /control/restart` (it never restarts itself from this route), and `service_note` says so. `ok: false` with `error` when the registry did not answer, when skillhook runs from a source checkout, or when the install failed. + +## `GET /stats` + +Query: `since=<24h|7d|2w|ISO-8601>` (default: everything on disk, newest 5000 jobs and deliveries), `until=`, `skill=`. Response: + +```json +{ + "window": { "since": "2026-09-27T12:00:00.000Z", "until": null, "skill": null }, + "jobs": { "total": 42, "finished": 40, "queued": 1, "running": 1, "by_status": { "succeeded": 37, "failed": 2, "timed_out": 1, "…": 0 }, "by_outcome": { "completed": 30, "partial": 2, "needs_human": 3, "nothing_to_do": 2, "failed": 3, "unknown": 0 }, "by_trigger": { "webhook": 38, "…": 0 }, "by_runner": { "claude": 40, "codex": 2, "shell": 0 }, "by_failure_kind": { "rate_limit": 1, "auth": 1, "…": 0 }, "success_rate": 0.925, "completion_rate": 0.865, "duration_ms": { "count": 40, "p50": 42000, "p95": 190000, "avg": 61000, "max": 300000 }, "queue_wait_ms": { "count": 41, "p50": 200, "p95": 4000, "avg": 700, "max": 9000 }, "cost_usd": 1.2345, "tokens": { "input": 120000, "output": 30000, "cached_input": 80000 }, "waiting_for_human": 2 }, + "deliveries": { "total": 60, "by_outcome": { "accepted": 42, "duplicate": 3, "in_flight": 0, "skipped": 5, "rejected": 10, "challenge": 0, "error": 0 }, "by_http_status": { "202": 42, "200": 8, "401": 9, "413": 1 }, "accepted_rate": 0.7, "last_received_at": "2026-09-28T11:58:00.000Z" }, + "skills": { "hello": { "jobs": 30, "by_status": { "…": 0 }, "by_outcome": { "…": 0 }, "success_rate": 0.97, "cost_usd": 0.9, "tokens": { "…": 0 }, "duration_ms": { "…": 0 }, "deliveries": 41, "last_job": { "id": "20260928T115800Z-k3x9q2", "status": "succeeded", "outcome": "completed", "created_at": "2026-09-28T11:58:00.000Z" } } }, + "generated_at": "2026-09-28T12:00:00.000Z" +} +``` + +`success_rate` is succeeded over finished jobs; `completion_rate` is completed plus nothing_to_do over finished jobs with a reported outcome; percentiles are nearest-rank over finished jobs (`queue_wait_ms`: creation to start). Tokens add Claude's `input_tokens` / `output_tokens` / `cache_read_input_tokens` and Codex's `input_tokens` / `output_tokens` / `cached_input_tokens`. `400 bad_request` for a `since` or `until` that does not parse. + +## Delivery record + +| Field | Type | Notes | +|---|---|---| +| `id` | string | Same format as job ids; the delivery log's own id, distinct from the provider's `delivery_id`. | +| `skill` | string | The name in the URL, as requested, also when no such skill exists. | +| `received_at` | ISO-8601 | | +| `outcome` | string | `accepted` (a job was created), `duplicate` (delivery id seen before), `in_flight` (folded into a queued or running job), `skipped` (a `when` filter), `rejected` (any error answer: 401, 403, 404, 413, 429, 500, 503), `challenge` (Slack URL verification), `error` (unexpected server error). | +| `http_status` | number | What the sender was answered at decision time; a `?wait=` request may have ended as `200` with the result instead of `202`. | +| `code`, `reason` | string, optional | The error code and message of a rejected delivery; `duplicate`, `in_flight`, `skipped` (with the condition as `reason`), `challenge` or `internal_error` otherwise. Absent when accepted. | +| `delivery_id` | string, optional | Provider delivery id or `dedupe` value, when one was found. | +| `job_id` | string, optional | The job created, or the one the delivery was folded into. | +| `ip`, `method`, `path`, `query` | | The request (`token` and `wait` removed from `query`). | +| `headers` | object | Redacted like `event.json` (no authorization, signature, token or cookie headers); values over 512 characters are shortened. | +| `via`, `ingress_id` | string, optional | `via: "ingress"` and the cloud's id for a webhook that arrived at a hosted URL and was handed over by the cloud link ([cloud.md](cloud.md#hosted-urls)). | +| `user_agent`, `content_type`, `bytes`, `body_kind` | | The body as received (`body_kind` is only known once the body was parsed). | +| `body_stored`, `body_truncated` | boolean | Whether the log kept the body, and whether it was cut at `deliveries.body_max_bytes`. | +| `duration_ms` | number | From arrival to the decision (a `?wait=` is not counted). | + ## Job record | Field | Type | Notes | @@ -278,7 +504,7 @@ Ids that do not exist (or do not look like `YYYYMMDDTHHMMSSZ-xxxxxx`) are `404 u | `id` | string | `YYYYMMDDTHHMMSSZ-<6 chars>`, UTC, sortable; also the directory name under `jobs/`. | | `skill` | string | | | `status` | string | `queued`, `running`, `succeeded`, `failed`, `timed_out`, `cancelled`, `interrupted`. | -| `trigger` | string | `webhook`, `api`, `cli`, `mcp`, `schedule` (fired by a `schedule:`). | +| `trigger` | string | `webhook`, `api`, `cli`, `mcp`, `schedule` (fired by a `schedule:`), `replay` (an operator replayed a delivery or job), `test` (a SKILL.md supplied with the request), `resume` (a person answered an earlier job; this run continues it). | | `runner` | string | `claude`, `codex`, `shell`. | | `model`, `effort` | string, optional | Resolved values when set. | | `created_at`, `started_at`, `finished_at` | ISO-8601 | | @@ -290,9 +516,23 @@ Ids that do not exist (or do not look like `YYYYMMDDTHHMMSSZ-xxxxxx`) are `404 u | `cost_usd`, `usage`, `num_turns` | optional | As reported by the runner (Claude reports all three, Codex `usage` only). | | `result` | string, optional | Final agent message, truncated to 20 000 characters here; complete in `result.md`. | | `error` | string, optional | Failure reason. | +| `outcome` | string, optional | Whether the task was done, set when the job ends: `completed`, `partial`, `needs_human`, `nothing_to_do`, `failed` (also every status other than `succeeded`) or `unknown` (the agent reported nothing). See [skills.md](skills.md#reporting-the-outcome). | +| `response` | object, optional | What the agent reported: `{"outcome", "summary", "links"?, "data"?}` (`data` is capped at 64 KiB here; complete in `response.json`). | +| `replay_of` | object, optional | For `trigger: replay`: `{"delivery"?: "", "job"?: ""}`. | +| `progress` | object, optional | What the agent last reported: `{"state": "working"\|"blocked"\|"waiting_human"\|"done", "message", "percent"?, "step"?, "updated_at"}`. | +| `question` | object, optional | The question the agent asked a person: `{"id", "text", "options"?, "context"?, "asked_at", "wait_until"?, "answered_at"?}`; pending until `answered_at` is set. | +| `answer` | object, optional | The person's answer: `{"question_id"?, "text", "option"?, "by"?, "at"}`. | +| `resume_of`, `resume` | optional | For `trigger: resume`: the job whose answer this run carries, and `{"session_id", "runner"}` when that job's session is continued (absent when the skill had to run afresh; `runner_reason` then says why). | +| `resolved_by` | string, optional | The resume job an answer to this job started. | +| `runner_reason` | string, optional | Why the run differs from what was asked: a resume without a session, or a fallback runner (`fallback: claude not logged in`, `fallback: claude failed (rate_limit)`). | +| `runner_requested` | string, optional | The runner the skill asked for, when `runner` is a fallback that took over. | +| `failure` | object, optional | For `failed` and `timed_out` jobs: `{"kind", "code"?, "retryable", "message"?}` with `kind` one of `auth`, `usage_limit`, `rate_limit`, `budget`, `max_turns`, `not_found`, `timeout`, `crash`, `unknown` ([runners.md](runners.md#failure-kinds)). | +| `attempts` | array, optional | Earlier runs of this job that a `retry:` or `fallback:` repeated: `[{runner, started_at, finished_at, status, error?, failure?}]`; the record itself is the last attempt. | +| `adhoc` | `true`, optional | The SKILL.md came with the request (`POST /skills/test`, `skillhook run --file`) and lives in `jobs//skill//`. | +| `skill_file` | string, optional | The `SKILL.md` (or `skillhook.yaml`) the job ran from. | | `delivery_id` | string, optional | Provider delivery id when known; `schedule:` for scheduled runs. | | `fingerprint` | string, optional | SHA-256 of the payload and query string of a webhook delivery; what the in-flight duplicate check compares. | -| `source` | object | `ip`, `method` (`POST`, `PUT`, `LOCAL` for CLI/MCP runs, `SCHEDULE` for scheduled runs), `path`, `content_type`, `user_agent`. | +| `source` | object | `ip`, `method` (`POST`, `PUT`, `LOCAL` for CLI/MCP runs, `SCHEDULE` for scheduled runs, `REPLAY` for replays, whose `ip` is the original sender's, `TEST` for ad-hoc runs, `RESUME` for resumed runs, `CLOUD` for runs started from Skillhook Cloud), `path`, `content_type`, `user_agent`. | `job.json` on disk also contains `command` (the exact argv); API responses omit it. @@ -302,14 +542,19 @@ Ids that do not exist (or do not look like `YYYYMMDDTHHMMSSZ-xxxxxx`) are `404 u |---|---|---| | 200 | — | Result available, duplicate, skipped, Slack challenge, admin reads, successful cancel. | | 202 | — | Job queued (or still running after `wait`). | -| 400 | `bad_request` | `/skills//run` body is not a JSON object. | +| 400 | `bad_request` | `/skills//run` body is not a JSON object; unknown `?types=` (`/events`), `?streams=` (`/jobs//events`), `?status=`/`?trigger=` (`/jobs`), `?outcome=` (`/deliveries`) or malformed `?since=` value; `/skills/test` without `skill_md`; `/jobs//answer` without `answer` or with a `resume` other than `auto`/`never`; an unknown `?failure=` kind (`/jobs`). | +| 400 | `invalid_skill_document` | `/skills/test`: the SKILL.md does not validate (the message says why). | +| 400 | `config_invalid`, `config_key_not_allowed` | `PATCH /config` would not validate (nothing written) or names `$schema` / a prototype key; `POST /config/reload` found an unparsable file. | | 401 | `missing_token`, `invalid_token`, `missing_credentials`, `invalid_credentials`, `missing_signature`, `invalid_signature`, `missing_timestamp`, `invalid_timestamp`, `stale_timestamp` | Webhook authentication failed. | | 401 | `unauthorized` | Admin route without a valid token. | | 403 | `ip_not_allowed` | Client IP not in the skill's `allow_ips`. | -| 404 | `unknown_skill`, `unknown_job`, `not_found` | | +| 404 | `unknown_skill`, `unknown_job`, `unknown_artifact`, `unknown_delivery`, `not_found` | | | 404 | `schedule_only` | The skill has `webhook: false`; it runs only on its `schedule:`. | | 405 | `method_not_allowed` | | | 409 | — (`ok: false`) | Cancel on a finished job. | +| 409 | `replay_needs_force`, `no_body` | Replaying a rejected delivery without `force`; a delivery whose body was not kept. | +| 409 | `not_waiting`, `unknown_skill` | Answering a job that is not waiting for a person; the skill of the job to resume no longer exists. | +| 409 | `not_a_service` | `POST /control/restart` on a server that launchd / systemd would not start again. | | 413 | `payload_too_large` | Body over `max_body_bytes`. | | 429 | `rate_limited`, `too_many_failures` | Per-IP limits. | | 500 | `invalid_skill`, `internal_error` | `SKILL.md` failed to parse; unexpected error (see the server log). | diff --git a/docs/cloud-protocol.md b/docs/cloud-protocol.md new file mode 100644 index 0000000..70c5f84 --- /dev/null +++ b/docs/cloud-protocol.md @@ -0,0 +1,68 @@ +# Skillhook Cloud protocol + +The messages between a machine and Skillhook Cloud, as zod schemas in `src/cloud/protocol.ts`, exported as `@meterapp/skillhook/protocol` (no Node built-ins, so the cloud can import it in any runtime). `PROTOCOL_VERSION` is 1; the cloud answers `426 upgrade_required` with `min_protocol_version` to a machine that is too old, and keeps accepting older versions within its supported range. + +## Transport + +Outbound HTTPS from the machine only: + +| Request | Purpose | +|---|---| +| `POST /api/agent/pair` | `PairRequest` (a pairing code from the dashboard, or a token) → `PairResponse` (`machine_id`, `machine_token` shown once, `mode`, `dashboard_url`). | +| `POST /api/agent/sync` | `SyncRequest` → `SyncResponse` (or `SyncError`). The machine's heartbeat, event upload, command channel and hosted-ingress channel, all in one; `wait: true` lets the cloud hold the request up to `LIMITS.long_poll_seconds` (25) when it has nothing to say. `Authorization: Bearer `, `x-skillhook-protocol: 1`. | +| `PUT /api/agent/artifacts//` | Chunked upload of a job artifact for `job.artifact`: `application/octet-stream` bodies of `LIMITS.artifact_chunk_bytes` (1 MiB), in order, each with `Content-Range: bytes -/` and `x-skillhook-sha256` (hex SHA-256 of the whole, already scrubbed, file); at most `LIMITS.max_artifact_bytes` (32 MiB). | +| `POST /api/agent/disconnect` | Revoke the token (`skillhook cloud disconnect`). | + +## `SyncRequest` + +| Field | Content | +|---|---| +| `protocol_version` | `1` | +| `sent_at` | The machine clock (the cloud derives skew). | +| `wait` | Nothing is pending; the cloud may hold the request. | +| `machine` | `{id, hostname, os, arch, skillhook_version, node_version, started_at, public_url?}` | +| `status` | `{queue: {running, queued}, running_jobs, link: {state, reason?, mode, outbox_depth, dropped_total, watched_jobs}}` | +| `snapshot?` | On connect and every `cloud.snapshot_interval_seconds`: skills, skill errors, projects, schedules, effective config, health and readiness summaries, stats. | +| `events` | Up to 200 `EventEnvelope`s: `{id: ":", seq, ts, machine_id, type, data}`; `seq` increases by one per durable event, `null` for transient `job.output`. Types: the server's own `delivery.received`, `job.*`, `schedule.*`, `skill.changed`, `config.changed`, `health.changed`, `runners.changed`, plus `link.started`, `link.stopped`, `health.report`, `job.output`. | +| `command_results` | Up to 50 `{command_id, ok, result?, error?: {code, message, hint?}, sensitive?, sealed?, started_at, finished_at, duration_ms}`; re-sent until acknowledged. | +| `ingress_acks` | `{id, outcome, http_status, job_id?, code?, reason?}` for hosted-ingress deliveries processed since the last sync. | +| `ack.commands_received` | Command ids received (the cloud stops re-sending them). | + +## `SyncResponse` + +| Field | Content | +|---|---| +| `ack.events_through` | Every event with `seq` ≤ this is durable on the cloud; the machine drops it from its outbox. `ack.command_results` lists result ids stored. | +| `commands` | Up to 50 `{id, type, args?, issued_at, expires_at?, timeout_ms?, requested_by?}`. The machine validates `args` against `COMMAND_ARGS[type]`, checks the policy (`commandAllowed`), runs them one at a time, and answers with a `CommandResult` in a later sync. | +| `ingress` | Up to 20 hosted-ingress deliveries `{id, skill, received_at, method, path, query, headers (raw), body_base64 (≤ 1 MiB), content_type, source_ip}`, fed through the ordinary webhook pipeline (signature verified with the local secret, dedupe, `when` filters, queue) and acknowledged in the next sync. | +| `next_poll_ms` | When to sync again if nothing is pending. | +| `hints?` | Only ever reduce or inform: `snapshot_interval_s`, `health_interval_s`, `upload_payloads: false`, `upload_artifacts: false`, `max_event_bytes`, `max_batch_events`, `mode: interactive \| idle`, `ingress_urls` (per skill). | +| `rotate?` | `{token, old_valid_until}`: a new machine token to store; the old one keeps working until then. | +| `notice?` | A line for the server log. | + +Errors are `SyncError` `{ok: false, error, message?, retry_after_ms?, min_protocol_version?}` with HTTP status: `401 invalid_token` and `403 machine_disabled` stop the link until the config or the token changes; `413 payload_too_large` halves the batch; `426 upgrade_required` retries in ten minutes; `429 rate_limited` honours `retry_after_ms`; `5xx` and network errors back off exponentially (1 s to 60 s with full jitter). + +## Transient output + +`job.watch` makes the machine send `job.output` events with `seq: null` and an id of the form `:out:`: `{job_id, stream, offset, chunk, eof?, status?, expired?}`. They ride along with the next sync, are never spooled to disk and are not re-sent if that request fails. + +## Command results worth knowing + +- `skill.run`, `skill.test`, `delivery.replay`, `job.replay`, `schedule.run`: `{accepted: true, job_id, …}` (a replay whose filters do not match: `{accepted: false, skipped: true, reason}`); the job itself is followed through its events. +- `job.answer`: `{job_id, delivered: live|resumed|recorded, answer, resume_job_id}`. +- `config.patch`: `{applied, restart_required_keys, pending_restart}`. +- `secret.generate`: `{secret_env, existed, generated}` with the value in the result's `sealed` field only, `sensitive: true`. +- `job.artifact`: `{job_id, name, bytes, text, truncated}` inline, or `{job_id, name, uploaded: true, bytes, sha256, chunks}`. +- `service.restart`: `{restarting: true, when, wait_seconds, running}`; the restart begins once a sync response acknowledges this result (or 15 seconds later). + +## Ordering and idempotency + +Events carry a per-machine `seq`; the cloud de-duplicates on `(machine_id, seq)`, the machine on command ids and ingress ids, and both sides re-send until acknowledged, so a lost response is never lost work. + +## Command classes and policy + +`COMMAND_CLASS` says what each command type needs: `read` (both modes), `control` (`cloud.mode: control` or an entry in `cloud.allow_commands`) or `allow_list` (`secret.set`: only with an explicit entry). `cloud.deny_commands` wins over everything; patterns are exact types, `prefix.*` or `*`. `commandAllowed(type, policy)` in `src/cloud/config.ts` is the single implementation. + +## Sealed values + +A `secret.generate` result (and a `secret.set` argument) is sealed to a recipient's X25519 public key: an ephemeral X25519 key pair, HKDF-SHA256 over the shared secret, AES-256-GCM; `{recipient_key, ephemeral_public_key, nonce, ciphertext}` as base64url. The cloud stores a sealed result for at most two minutes and only the recipient can open it. diff --git a/docs/cloud.md b/docs/cloud.md new file mode 100644 index 0000000..ef8665b --- /dev/null +++ b/docs/cloud.md @@ -0,0 +1,106 @@ +# Skillhook Cloud + +Skillhook Cloud is the hosted control plane for machines running skillhook: every webhook and job of every machine in one place, health of the CLIs and their MCP servers, replay, stats, a playground for skills, remote configuration from a browser or from an MCP client, alerts, and hosted webhook URLs that keep deliveries while a machine is asleep. It is a separate service (`MeterApp/skillhook-cloud`); this document is about the machine side. + +**Status.** The link is in this version: `skillhook cloud connect` pairs a machine, the running server keeps one outbound connection to the cloud, uploads what happens, runs the commands below as far as `cloud.mode` and the allow/deny lists permit, and delivers webhooks that arrived at the machine's hosted URLs. + +## Principles + +- **Opt-in, outbound only.** A machine talks to the cloud only after `skillhook cloud connect` pairs it (a code from the dashboard) and only by opening HTTPS requests to `cloud.url` (plain `http` only to a loopback address or with `SKILLHOOK_CLOUD_ALLOW_INSECURE=1`); the cloud never connects to the machine and never holds the admin token. It works behind NAT without Tailscale. +- **Observe by default.** A freshly paired machine is in `mode: observe`: the cloud can read, not act. `--control` at pairing (what the dashboard's pairing page prints) or `cloud.mode: control` later lets it run skills, answer jobs, change the configuration and restart the server. `cloud.allow_commands` / `cloud.deny_commands` refine either mode per command type; the cloud cannot raise a machine's exposure, only the machine's own config can. +- **Payloads are data, secrets stay home.** Headers are redacted on the machine before anything is uploaded; every uploaded string is scrubbed against every value in `.env`; webhook bodies travel only when both `cloud.upload_payloads` and the organisation's policy allow, and never beyond 256 KiB. `SKILLHOOK_CLOUD_*` variables never reach a run, even when a skill lists them in `env:`. A secret the cloud asks skillhook to generate is sealed to the requester's key; the cloud never stores it in the clear. +- **Kill switches.** `cloud.enabled: false`, `SKILLHOOK_NO_CLOUD=1` in the server's environment, or `skillhook cloud disconnect` stop all traffic; the link never starts from `init`, from a job, or on its own. + +## Settings + +| Key | Default | Meaning | +|---|---|---| +| `cloud.enabled` | `false` | Whether the running server keeps a link open. Written by `skillhook cloud connect` / `disconnect`; the server follows it within seconds, without a restart. | +| `cloud.url` | `https://cloud.skillhook.dev` (placeholder) | The service. `SKILLHOOK_CLOUD_URL` overrides it; plain `http` is accepted only for loopback addresses or with `SKILLHOOK_CLOUD_ALLOW_INSECURE=1`. | +| `cloud.machine_id` | unset | Assigned at pairing. | +| `cloud.mode` | `observe` | `observe` or `control`. | +| `cloud.allow_commands`, `cloud.deny_commands` | `[]` | Command types (`skill.run`, patterns like `job.*`, `*`) allowed regardless of mode, or refused regardless of anything. `secret.set` is never allowed without an explicit allow entry. | +| `cloud.upload_payloads` | `true` | Upload webhook payloads with deliveries (redacted headers; bodies at most 256 KiB). | +| `cloud.upload_artifacts` | `true` | Let the cloud fetch job artifacts and live output. | +| `cloud.ingress` | `true` | Accept hosted-ingress deliveries (webhooks the cloud received for this machine). | +| `cloud.snapshot_interval_seconds` | `60` | How often the full snapshot (skills, schedules, config, health summary) is sent. | +| `cloud.health_interval_seconds` | `600` | How often a deep health report is sent. | +| `cloud.outbox_max_events` | `5000` | Events kept on disk while the cloud is unreachable. | + +The machine token lives in `.env` as `SKILLHOOK_CLOUD_TOKEN` (an optional X25519 private key as `SKILLHOOK_CLOUD_PRIVATE_KEY`); both are written once by `skillhook cloud connect` and never printed again. + +## Connecting + +On the dashboard's pairing page choose *Control* or *Observe* and copy the command it prints: + +```bash +skillhook cloud connect --code ABCD-EFGH --control +``` + +`connect` sends the code with a description of the machine (hostname, OS, architecture, skillhook and Node versions, public URL), receives a machine id and a machine token, stores the token in `.env` as `SKILLHOOK_CLOUD_TOKEN` (mode 600, never printed), writes `cloud.url`, `cloud.machine_id`, `cloud.mode` and finally `cloud.enabled: true` to `skillhook.json`, and tells a running server to re-read its configuration; the link is up within seconds. Without `--control` the machine is paired in `observe` mode. `--url` (or `SKILLHOOK_CLOUD_URL`) points at another deployment; `--token` pairs with a machine token instead of a code; `--force` pairs a machine that is already connected again. + +```bash +skillhook cloud status # enabled, URL, machine id, mode, token present, and the running server's link state +``` + +```bash +skillhook cloud disconnect # cloud.enabled: false, token removed from .env and revoked, local spool deleted +``` + +`disconnect --keep-token` leaves the token in `.env`. `skillhook doctor` and `skillhook health` report a `cloud link` check: skipped when not connected, failing when `cloud.enabled` has no token, an `http` URL, or a revoked token or disabled machine, warning when no server runs, the link is degraded or events were dropped. + +## What the link does + +The running server opens HTTPS requests to `cloud.url` (`POST /api/agent/sync`); the cloud may hold a request up to 25 seconds when it has nothing to say, which makes the link both the heartbeat and the command channel ([cloud-protocol.md](cloud-protocol.md)). Each request carries: + +- **Events**: every delivery (the delivery record with redacted headers, and the body when `cloud.upload_payloads` allows it and it is at most 256 KiB), every job change (the job record without its command line; the result up to 8 KiB), progress lines (at most one per job every five seconds, questions and answers always), schedule and skill changes, configuration changes, health changes and runner readiness (checked once when the link connects, then whenever a job checks it), plus `link.started` / `link.stopped`. +- **A snapshot** on connect and every `cloud.snapshot_interval_seconds`: skill summaries, projects, schedules, the effective configuration, the last health and readiness answers and a day of stats. +- **A deep health report** every `cloud.health_interval_seconds`. +- **Command results** and **hosted-ingress acknowledgements** (below). + +Every string is scrubbed of every value in `.env` before it leaves, `authorization`, cookie, signature and token headers never leave, and command lines, environments and `.env` itself never do. Events wait in `jobs/.cloud/outbox.jsonl` while the cloud is unreachable (at most `cloud.outbox_max_events`, oldest dropped first and counted); a server restart loses nothing that was spooled. On errors the link backs off exponentially up to a minute, is reported `degraded` after three failures, stops on a revoked token (`401`) or a disabled machine (`403`) until the configuration or the token changes, halves its batches on `413`, honours `429`'s retry delay and waits ten minutes on `426` (update skillhook). + +## Hosted URLs + +A skill can have a hosted webhook URL on the cloud (the dashboard creates it) in addition to, or instead of, its Tailscale URL. The cloud accepts the request, keeps it sealed until this machine collects it, and hands it over in a sync response; the link replays it to the local server as the original request (method, headers, body, query string without `wait`, the sender's address as `X-Forwarded-For`), so the signature is verified here with the local secret and deduplication, `when` filters and queueing apply exactly as for a direct webhook. The delivery record says `via: "ingress"` with the cloud's `ingress_id`, and the outcome goes back to the cloud with the next sync. A delivery the cloud sends twice is acknowledged again from `jobs/.cloud/ingress.json` and never run twice. `cloud.ingress: false` declines them (`503 ingress_disabled`). `?wait=` does not apply to hosted deliveries: the cloud has already answered the sender. + +## What never leaves the machine + +`.env` and every value in it, the admin token, command lines and run environments, `authorization` / cookie / signature / token headers, job artifacts unless `cloud.upload_artifacts` allows them and a command asks, webhook bodies unless `cloud.upload_payloads` allows them, and anything a command policy refuses. + +## Commands the cloud may send + +Read commands (both modes): `ping`, `health.get`, `snapshot.get`, `runners.get`, `skills.list`, `skill.get`, `delivery.list`, `delivery.get`, `job.list`, `job.get`, `job.artifact`, `job.watch`, `job.unwatch`, `job.progress.get`, `stats.get`, `config.get`, `secret.list` (names only), `service.status`, `logs.tail`, `schedules.list`, `update.check`, `expose.status`. + +Control commands (`mode: control` or an allow entry): `skill.put`, `skill.delete`, `skill.run`, `skill.test`, `delivery.replay`, `job.cancel`, `job.replay`, `job.answer`, `config.patch`, `secret.generate`, `service.restart`, `schedule.run`, `update.install`. + +Results are scrubbed like events. Each command runs once: its id is remembered in `jobs/.cloud/commands.json`, and a command the cloud sends again is answered from the kept result. + +## Control mode + +**Control mode gives the cloud, and everyone with access to this machine on the dashboard, the power to run code on the machine as the user who runs skillhook:** `skill.test` runs any SKILL.md, including `runner: shell` commands and agents with `bypassPermissions`, and `skill.put` installs one. Turn it on only for machines you would give those people a shell on. `cloud.deny_commands` narrows it (for example `["skill.put", "skill.test", "update.install"]` keeps running and answering installed skills while refusing new code), and `cloud.allow_commands` lets an `observe` machine accept a few chosen ones (`["job.answer"]` to answer the agents' questions from a phone and nothing else). + +What each control command does, and the rules it adds on top of the policy: + +| Command | Effect | +|---|---| +| `skill.run` | Runs an installed skill with the given payload as a new job (`trigger: api`, `source.method: CLOUD`, header `x-skillhook-cloud-user` with the requester's name). Answers `{accepted, job_id}` at once; the job's progress arrives as events. | +| `skill.test` | Runs a SKILL.md that is not installed, like `skillhook run --file` (`trigger: test`). | +| `delivery.replay`, `job.replay` | Replays a recorded delivery or an earlier job, like `skillhook deliveries replay` / `jobs replay` (`force` for a rejected delivery, `skip_filters`). | +| `job.cancel` | Cancels a queued or running job. | +| `job.answer` | A person's answer to a job waiting for one: delivered live, or a new job resumes the agent's session ([skills.md](skills.md#reporting-progress-and-asking-a-person)); `by` defaults to the requester's name. | +| `config.patch` | Changes `skillhook.json` like `PATCH /config`, except for `host`, `port`, `trust_proxy`, `runners`, `env_passthrough`, `projects` and `cloud`, which the cloud may never change. | +| `service.restart` | Restarts a server run by launchd / systemd once the cloud has the answer (`when: idle` lets running jobs finish, up to `wait_seconds`; `now` does not wait). | +| `schedule.run` | Fires a scheduled skill now. | +| `update.install` | Installs a newer skillhook with the package manager that installed it; the server keeps running the old version until `service.restart`. | +| `secret.generate` | Generates a skill's secret (or any `ENV_NAME`) and returns it only sealed to the requester's key (`recipient_key`, required); the value never travels or rests in the clear. `SKILLHOOK_CLOUD_*` names are refused. | +| `skill.put` | Writes `skills//SKILL.md` after validating it. Never for a name that comes from a linked repository; `auth: none` needs `allow_unauthenticated`. No secret is created: `secret.generate` does that, sealed. | +| `skill.delete` | Removes a skill of `skills/` by moving its directory to `jobs/.removed-skills/-