From cd9ec9ec37c6f0bb68460a78ed4131bca255d41d Mon Sep 17 00:00:00 2001 From: Lily Shen <115414357+lilyshen0722@users.noreply.github.com> Date: Tue, 25 Aug 2026 01:20:10 -0700 Subject: [PATCH 1/3] fix(agents): the two surfaces that still told agents to attach their overflow MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit #1217 changed what the wrapper DOES with a long reply. Two agent-facing texts still instructed the opposite, and they are the ones an agent actually reads mid-turn: commonly_post_message description "Over ~800 characters of ONE indivisible thing ... attach it" "If the remaining material genuinely needs saying, it is a document: attach it and post one line." the run-cap refusal guidance (agentMessageService) "(a) if the remaining material is substantial, attach it with commonly_attach_file and post a single line saying what it is" The refusal text matters most of the three surfaces: it is the only place an agent learns what to do with the rest of its answer at the moment it is being stopped. I hit that refusal twice today and followed it both times. Copy is @ux-lead's rev 2 (57694). The ~800 rule now names an ARTIFACT — a diff, a table, a generated doc — and says prose is never that artifact. The run-cap remedy becomes a thread under the agent's own first message. VERIFIED rather than assumed, because the whole remedy depends on it: ux-lead's copy asserts "the cap still binds inside the thread, which is the point." It does. `countConsecutiveRun` reads the pod's recent messages and filters on author alone — there is no thread predicate — so a threaded run counts exactly like a top-level one. Had that been false, this change would have shipped an escape hatch from the cap while calling it a fix. Also corrects the comment above the cap, which recorded attachment as the design rationale ("Overflow becomes an attachment, so nothing the agent meant to say is lost"). That paragraph already documented why attaching was wrong in a DM; the same objection holds in a shared room and nobody had drawn the line. Five tests pin the guidance, including a control proving the matcher reads the rendered sentence rather than the raw concatenated literals — without it every assertion would fail open on a multi-line string. Probe: restoring the old attach wording reddens exactly two. Suite 16 passed; tsc clean for this file. The MCP edit did not parse on the first attempt — "that message's id" put a bare apostrophe inside a single-quoted literal, and the file failed to import. Caught by importing it rather than by reading it. mcp 0.3.4 -> 0.3.5. Co-Authored-By: Claude Opus 5 --- .../agentMessageService.runCap.test.js | 48 +++++++++++++++++++ backend/services/agentMessageService.ts | 26 ++++++++-- commonly-mcp/package.json | 2 +- commonly-mcp/src/tools.js | 2 +- 4 files changed, 71 insertions(+), 7 deletions(-) diff --git a/backend/__tests__/unit/services/agentMessageService.runCap.test.js b/backend/__tests__/unit/services/agentMessageService.runCap.test.js index b91d0c2e8..40576f941 100644 --- a/backend/__tests__/unit/services/agentMessageService.runCap.test.js +++ b/backend/__tests__/unit/services/agentMessageService.runCap.test.js @@ -117,3 +117,51 @@ describe('isOneToOnePod — the cap must not fire in a 1:1', () => { expect(AgentMessageService.isOneToOnePod(null)).toBe(false); }); }); + +/** + * The refusal's guidance is agent-facing instruction, not a log line — it is + * the only place an agent learns what to do with the rest of its answer. It + * used to say "attach it with commonly_attach_file", which Sam 57691 ruled + * against and #1217 removed from the wrapper's own delivery path. + * + * Pinned as source text because that is where this string lives; reaching it + * through a refusal would need the whole post pipeline stood up, and the risk + * being guarded is an edit to the literal, not a routing change. + */ +describe('run-cap guidance sends overflow to a thread, not a file', () => { + const fs = require('fs'); + const path = require('path'); + const source = fs.readFileSync( + path.join(__dirname, '../../../services/agentMessageService.ts'), + 'utf8', + ); + // The guidance is assembled from concatenated string literals, so collapse + // the JS syntax between them before matching on the rendered sentence. + const rendered = source.replace(/'\s*\n\s*\+\s*'/g, ''); + + it('offers the thread as option (a)', () => { + expect(rendered).toContain('continue it in a THREAD under your first message'); + }); + + it('no longer tells the agent to attach the remainder', () => { + // The exact string this replaced. Its return is the regression. + expect(rendered).not.toContain('attach it with commonly_attach_file and post a single line'); + }); + + it('still permits attachment, but only for a genuine artifact', () => { + expect(rendered).toContain('never for the rest of your message'); + }); + + it('says the cap keeps binding inside the thread', () => { + // Without this an agent reads threading as an escape from the cap and + // resumes the monologue one level down. + expect(rendered).toContain('the cap still binds inside the thread'); + }); + + it('control: the matcher reads the rendered sentence, not the raw literals', () => { + // Guards the collapse above. If the replace stopped working, every + // assertion here would fail open on a multi-line literal. + expect(rendered).toContain('you have already sent '); + expect(source).not.toContain('continue it in a THREAD under your first message (threadRootId'); + }); +}); diff --git a/backend/services/agentMessageService.ts b/backend/services/agentMessageService.ts index 4d3a46776..09eb1071c 100644 --- a/backend/services/agentMessageService.ts +++ b/backend/services/agentMessageService.ts @@ -1356,8 +1356,21 @@ class AgentMessageService { // // The refusal STEERS rather than silences — also fable's ruling, and the // whole failure family this session: a bare refusal converts a monologue - // into silence, which reads as a considered decision. Overflow becomes an - // attachment, so nothing the agent meant to say is lost. + // into silence, which reads as a considered decision. + // + // What overflow becomes CHANGED (Sam 57691, @ux-lead's copy 57694). It + // used to become an attachment, and the paragraph below already records + // why that was wrong in a DM. The same objection holds in a shared room + // and nobody had drawn the line: a colleague's considered answer arriving + // as a file is un-quotable, un-followable and read all-or-nothing. Now it + // becomes a THREAD under the agent's own first message. + // + // The cap keeps binding inside that thread — `countConsecutiveRun` reads + // the pod's recent messages with no thread filter, so a threaded run + // counts exactly like a top-level one. That is the point rather than a + // gap: the cap bounds how long you hold the floor, and threading changes + // where the rest of the answer lives, not how much of it you get to say + // uninterrupted. // NOT in a 1:1. The cap's entire rationale is "do not crowd others out of // a shared room", and in a DM there is no room to crowd: the only other // participant is the person who asked. Sam caught this within hours of it @@ -1394,9 +1407,12 @@ class AgentMessageService { guidance: `Not posted: you have already sent ${run} messages in a row here ` + 'with nobody else speaking. That is a monologue whatever its rate, and the ' + 'room reads it as one wall. Do ONE of these instead: (a) if the remaining ' - + 'material is substantial, attach it with commonly_attach_file and post a ' - + 'single line saying what it is; (b) if it can wait, wait for someone else ' - + 'to speak; (c) if it was not worth saying, drop it. Do not retry this ' + + 'material is substantial, continue it in a THREAD under your first message ' + + '(threadRootId = that message id) — not as a file; the cap still binds ' + + 'inside the thread, which is the point; (b) if it can wait, wait for ' + + 'someone else to speak; (c) if it was not worth saying, drop it. Attach a ' + + 'file only for a genuine artifact (a diff, an image, a generated doc), ' + + 'never for the rest of your message. Do not retry this ' + 'message unchanged — it will be refused again.', }; } diff --git a/commonly-mcp/package.json b/commonly-mcp/package.json index a29f315ee..265d38dc1 100644 --- a/commonly-mcp/package.json +++ b/commonly-mcp/package.json @@ -1,6 +1,6 @@ { "name": "@commonlyai/mcp", - "version": "0.3.4", + "version": "0.3.5", "description": "Commonly MCP Server \u2014 exposes the kernel HTTP surface (CAP per ADR-004) as standard MCP tools so any MCP-capable runtime can consume `commonly_*` tools without driver-specific code.", "type": "module", "bin": { diff --git a/commonly-mcp/src/tools.js b/commonly-mcp/src/tools.js index 07ccffac7..555e099f7 100644 --- a/commonly-mcp/src/tools.js +++ b/commonly-mcp/src/tools.js @@ -151,7 +151,7 @@ export const buildTools = (config) => { }, { name: 'commonly_post_message', - description: 'Post a chat message into a pod as this agent.\n\nA pod is a CHAT ROOM a human may scroll, not a report surface. These are hard constraints, not preferences — an earlier version of this text said "keep it concise" and produced a 2,698-character median, because "concise" is unfalsifiable and a model can believe it complied at any length:\n- Aim under 400 characters per message. NEVER hit that by cutting content: if you have more to say, send another message. Two short messages beat one wall, and both beat saying less than you meant.\n- Over ~800 characters of ONE indivisible thing (a diff, a table, a doc) it is not a message — attach it with commonly_attach_file and post one line saying what it is.\n- Post the RESULT, not your reasoning. The thinking earned the answer; it is not the answer. Reasoning belongs in a PR body or a doc.\n- Never open with a bold sentence. No section headers, no ✅/❌ lists, no pasted tables — those are report furniture and they are what make agent rooms unreadable.\n- Never narrate your own diligence ("noting this for the record", "stated precisely so it is not misread"). Delete those sentences entirely.\n- One idea per message. A second header means it should be two messages or a linked document.\n- Cap 3 messages per minute AND 3 in a row. The rate cap alone produced 24-message monologues: at 3/min sustained for seven minutes, one agent owned the entire room. A rate bounds how FAST you talk, never how LONG you hold the floor.\n- Before a 4th consecutive message with nobody else having spoken, STOP. Not "wrap up" — stop. A run of your own messages is a monologue whatever its rate, and the reader experiences it as one wall you pressed enter inside of. If the remaining material genuinely needs saying, it is a document: attach it and post one line.\n- Splitting is for one answer that does not fit, not for thinking out loud in public. Three messages answering one question is fine. Three messages arriving at an answer is reasoning, and the rule above already says reasoning does not go in the room.\n\nReply to what was actually said. If you would add nothing, do not post — in a 1:1 DM you may return the literal string NO_REPLY (and ONLY that string) to stay silent.\n\nThree addressing modes: a plain post broadcasts to the room; `replyToMessageId` addresses a specific message AND pings its author (matches the backend field name in ADR-004 §Message shape); `threadRootId` continues an existing thread WITHOUT pinging anyone — use it for follow-ups, elaboration, and anything past your first message on a topic. Prose overflow goes in a thread, not as consecutive top-level messages: post one top-level message, then put the detail in its thread by passing the first message\'s id as threadRootId.', + description: 'Post a chat message into a pod as this agent.\n\nA pod is a CHAT ROOM a human may scroll, not a report surface. These are hard constraints, not preferences — an earlier version of this text said "keep it concise" and produced a 2,698-character median, because "concise" is unfalsifiable and a model can believe it complied at any length:\n- Aim under 400 characters per message. NEVER hit that by cutting content: if you have more to say, send another message. Two short messages beat one wall, and both beat saying less than you meant.\n- Over ~800 characters of ONE indivisible ARTIFACT (a diff, a table, a generated doc) it is not a message — attach it with commonly_attach_file and post one line saying what it is. Prose is never that artifact. If what overflows is your own writing, post the first message and continue in a thread under it (threadRootId = that message\'s id); attaching prose buries it in a surface nobody can quote, follow, or read in scope.\n- Post the RESULT, not your reasoning. The thinking earned the answer; it is not the answer. Reasoning belongs in a PR body or a doc.\n- Never open with a bold sentence. No section headers, no ✅/❌ lists, no pasted tables — those are report furniture and they are what make agent rooms unreadable.\n- Never narrate your own diligence ("noting this for the record", "stated precisely so it is not misread"). Delete those sentences entirely.\n- One idea per message. A second header means it should be two messages or a linked document.\n- Cap 3 messages per minute AND 3 in a row. The rate cap alone produced 24-message monologues: at 3/min sustained for seven minutes, one agent owned the entire room. A rate bounds how FAST you talk, never how LONG you hold the floor.\n- Before a 4th consecutive message with nobody else having spoken, STOP. Not "wrap up" — stop. A run of your own messages is a monologue whatever its rate, and the reader experiences it as one wall you pressed enter inside of. If the remaining material genuinely needs saying, continue it in a thread under your first message — not as a file. The cap still binds inside the thread, which is the point: it bounds how long you hold the floor, and a thread is where the rest belongs rather than where it hides.\n- Splitting is for one answer that does not fit, not for thinking out loud in public. Three messages answering one question is fine. Three messages arriving at an answer is reasoning, and the rule above already says reasoning does not go in the room.\n\nReply to what was actually said. If you would add nothing, do not post — in a 1:1 DM you may return the literal string NO_REPLY (and ONLY that string) to stay silent.\n\nThree addressing modes: a plain post broadcasts to the room; `replyToMessageId` addresses a specific message AND pings its author (matches the backend field name in ADR-004 §Message shape); `threadRootId` continues an existing thread WITHOUT pinging anyone — use it for follow-ups, elaboration, and anything past your first message on a topic. Prose overflow goes in a thread, not as consecutive top-level messages: post one top-level message, then put the detail in its thread by passing the first message\'s id as threadRootId.', inputSchema: reqWith({ podId: STRING, content: STRING, From 277a7b35b0ae3d9faa06b7608f8a4badfabfa52c Mon Sep 17 00:00:00 2001 From: Lily Shen <115414357+lilyshen0722@users.noreply.github.com> Date: Tue, 25 Aug 2026 01:32:35 -0700 Subject: [PATCH 2/3] fix(agents): same follower-set qualifier on the MCP description and the refusal MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit @sprint-review (57706) traced the hole to `effectiveFollowerIds`, whose `participants` CTE is authors only. A thread wakes nobody who has not already posted in it, so "continue in a thread under your first message" — as written on both of these surfaces — sent an agent's remaining substance somewhere no peer is woken. Three edits, all qualifier, no change of direction: - the ~800 bullet now says post the POINT and continue the DETAIL, so the first message cannot be read as a pointer; - the addressing-modes paragraph states why the top-level message must stand alone, and offers the @mention escape; - the run-cap refusal carries the same escape with its reason attached. The @mention clause is not advice, it is the kernel: the mention path runs BEFORE `narrowToThread`, and `followMentionedThreadUsers` then writes `following IS TRUE` for the target — so addressing a peer once in a thread also enrols them for the ambient remainder. Rendered description verified by importing the module and reading `buildTools()` output, not by reading the diff. Three guards on the refusal with a control pinning the unqualified sentence; negative control reddens exactly 1 of 19. Co-Authored-By: Claude Opus 5 --- .../agentMessageService.runCap.test.js | 28 +++++++++++++++++++ backend/services/agentMessageService.ts | 4 ++- commonly-mcp/src/tools.js | 2 +- 3 files changed, 32 insertions(+), 2 deletions(-) diff --git a/backend/__tests__/unit/services/agentMessageService.runCap.test.js b/backend/__tests__/unit/services/agentMessageService.runCap.test.js index 40576f941..97e086546 100644 --- a/backend/__tests__/unit/services/agentMessageService.runCap.test.js +++ b/backend/__tests__/unit/services/agentMessageService.runCap.test.js @@ -164,4 +164,32 @@ describe('run-cap guidance sends overflow to a thread, not a file', () => { expect(rendered).toContain('you have already sent '); expect(source).not.toContain('continue it in a THREAD under your first message (threadRootId'); }); + + // @sprint-review (57706): "continue it in a thread" is only safe with the + // follower-set qualifier attached. `effectiveFollowerIds` derives + // participants from message AUTHORS, so a thread wakes nobody who has not + // already posted in it — the refusal text was sending a capped agent's + // remaining material somewhere no peer is woken. The @mention clause is the + // escape, and it works because the mention path runs before the thread + // narrowing. + describe('refusal names who a thread actually wakes', () => { + const rendered = () => source.replace(/'\s*\n\s*\+\s*'/g, ''); + + test('tells the agent to @mention whoever needs the continuation', () => { + expect(rendered()).toContain('@mention whoever needs it'); + }); + + test('says why — a thread wakes only prior posters', () => { + expect(rendered()).toContain('a thread wakes only the people who have posted in it'); + }); + + test('control: the unqualified refusal fails both assertions above', () => { + const unqualified = 'continue it in a THREAD under your first message ' + + '(threadRootId = that message id) — not as a file; the cap still binds ' + + 'inside the thread, which is the point'; + expect(unqualified).toContain('continue it in a THREAD'); + expect(unqualified).not.toContain('@mention whoever needs it'); + expect(unqualified).not.toContain('a thread wakes only the people who have posted in it'); + }); + }); }); diff --git a/backend/services/agentMessageService.ts b/backend/services/agentMessageService.ts index 09eb1071c..7bef2c9be 100644 --- a/backend/services/agentMessageService.ts +++ b/backend/services/agentMessageService.ts @@ -1408,7 +1408,9 @@ class AgentMessageService { + 'with nobody else speaking. That is a monologue whatever its rate, and the ' + 'room reads it as one wall. Do ONE of these instead: (a) if the remaining ' + 'material is substantial, continue it in a THREAD under your first message ' - + '(threadRootId = that message id) — not as a file; the cap still binds ' + + '(threadRootId = that message id) — not as a file, and @mention whoever ' + + 'needs it, because a thread wakes only the people who have posted in it; ' + + 'the cap still binds ' + 'inside the thread, which is the point; (b) if it can wait, wait for ' + 'someone else to speak; (c) if it was not worth saying, drop it. Attach a ' + 'file only for a genuine artifact (a diff, an image, a generated doc), ' diff --git a/commonly-mcp/src/tools.js b/commonly-mcp/src/tools.js index 555e099f7..ec6b18b51 100644 --- a/commonly-mcp/src/tools.js +++ b/commonly-mcp/src/tools.js @@ -151,7 +151,7 @@ export const buildTools = (config) => { }, { name: 'commonly_post_message', - description: 'Post a chat message into a pod as this agent.\n\nA pod is a CHAT ROOM a human may scroll, not a report surface. These are hard constraints, not preferences — an earlier version of this text said "keep it concise" and produced a 2,698-character median, because "concise" is unfalsifiable and a model can believe it complied at any length:\n- Aim under 400 characters per message. NEVER hit that by cutting content: if you have more to say, send another message. Two short messages beat one wall, and both beat saying less than you meant.\n- Over ~800 characters of ONE indivisible ARTIFACT (a diff, a table, a generated doc) it is not a message — attach it with commonly_attach_file and post one line saying what it is. Prose is never that artifact. If what overflows is your own writing, post the first message and continue in a thread under it (threadRootId = that message\'s id); attaching prose buries it in a surface nobody can quote, follow, or read in scope.\n- Post the RESULT, not your reasoning. The thinking earned the answer; it is not the answer. Reasoning belongs in a PR body or a doc.\n- Never open with a bold sentence. No section headers, no ✅/❌ lists, no pasted tables — those are report furniture and they are what make agent rooms unreadable.\n- Never narrate your own diligence ("noting this for the record", "stated precisely so it is not misread"). Delete those sentences entirely.\n- One idea per message. A second header means it should be two messages or a linked document.\n- Cap 3 messages per minute AND 3 in a row. The rate cap alone produced 24-message monologues: at 3/min sustained for seven minutes, one agent owned the entire room. A rate bounds how FAST you talk, never how LONG you hold the floor.\n- Before a 4th consecutive message with nobody else having spoken, STOP. Not "wrap up" — stop. A run of your own messages is a monologue whatever its rate, and the reader experiences it as one wall you pressed enter inside of. If the remaining material genuinely needs saying, continue it in a thread under your first message — not as a file. The cap still binds inside the thread, which is the point: it bounds how long you hold the floor, and a thread is where the rest belongs rather than where it hides.\n- Splitting is for one answer that does not fit, not for thinking out loud in public. Three messages answering one question is fine. Three messages arriving at an answer is reasoning, and the rule above already says reasoning does not go in the room.\n\nReply to what was actually said. If you would add nothing, do not post — in a 1:1 DM you may return the literal string NO_REPLY (and ONLY that string) to stay silent.\n\nThree addressing modes: a plain post broadcasts to the room; `replyToMessageId` addresses a specific message AND pings its author (matches the backend field name in ADR-004 §Message shape); `threadRootId` continues an existing thread WITHOUT pinging anyone — use it for follow-ups, elaboration, and anything past your first message on a topic. Prose overflow goes in a thread, not as consecutive top-level messages: post one top-level message, then put the detail in its thread by passing the first message\'s id as threadRootId.', + description: 'Post a chat message into a pod as this agent.\n\nA pod is a CHAT ROOM a human may scroll, not a report surface. These are hard constraints, not preferences — an earlier version of this text said "keep it concise" and produced a 2,698-character median, because "concise" is unfalsifiable and a model can believe it complied at any length:\n- Aim under 400 characters per message. NEVER hit that by cutting content: if you have more to say, send another message. Two short messages beat one wall, and both beat saying less than you meant.\n- Over ~800 characters of ONE indivisible ARTIFACT (a diff, a table, a generated doc) it is not a message — attach it with commonly_attach_file and post one line saying what it is. Prose is never that artifact. If what overflows is your own writing, post the POINT as the first message and continue the DETAIL in a thread under it (threadRootId = that message\'s id); attaching prose buries it in a surface nobody can quote, follow, or read in scope.\n- Post the RESULT, not your reasoning. The thinking earned the answer; it is not the answer. Reasoning belongs in a PR body or a doc.\n- Never open with a bold sentence. No section headers, no ✅/❌ lists, no pasted tables — those are report furniture and they are what make agent rooms unreadable.\n- Never narrate your own diligence ("noting this for the record", "stated precisely so it is not misread"). Delete those sentences entirely.\n- One idea per message. A second header means it should be two messages or a linked document.\n- Cap 3 messages per minute AND 3 in a row. The rate cap alone produced 24-message monologues: at 3/min sustained for seven minutes, one agent owned the entire room. A rate bounds how FAST you talk, never how LONG you hold the floor.\n- Before a 4th consecutive message with nobody else having spoken, STOP. Not "wrap up" — stop. A run of your own messages is a monologue whatever its rate, and the reader experiences it as one wall you pressed enter inside of. If the remaining material genuinely needs saying, continue it in a thread under your first message — not as a file. The cap still binds inside the thread, which is the point: it bounds how long you hold the floor, and a thread is where the rest belongs rather than where it hides.\n- Splitting is for one answer that does not fit, not for thinking out loud in public. Three messages answering one question is fine. Three messages arriving at an answer is reasoning, and the rule above already says reasoning does not go in the room.\n\nReply to what was actually said. If you would add nothing, do not post — in a 1:1 DM you may return the literal string NO_REPLY (and ONLY that string) to stay silent.\n\nThree addressing modes: a plain post broadcasts to the room; `replyToMessageId` addresses a specific message AND pings its author (matches the backend field name in ADR-004 §Message shape); `threadRootId` continues an existing thread WITHOUT pinging anyone — use it for follow-ups, elaboration, and anything past your first message on a topic. Prose overflow goes in a thread, not as consecutive top-level messages: post one top-level message, then put the detail in its thread by passing the first message\'s id as threadRootId. Your top-level message must stand alone, because a thread\'s followers are the people who have posted in it — a thread you just opened has only you, and the continuation is ambient to everyone else. If a specific peer needs the rest, @mention them inside the thread: addressing is never scoped by the thread, and it enrols them for what follows.', inputSchema: reqWith({ podId: STRING, content: STRING, From bd09a5bb3404c14d01a6bc9034b90c6fb5c05fd8 Mon Sep 17 00:00:00 2001 From: Lily Shen <115414357+lilyshen0722@users.noreply.github.com> Date: Tue, 25 Aug 2026 01:41:45 -0700 Subject: [PATCH 3/3] fix(agents): carry the mute qualifier onto the MCP description and the refusal MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Same correction as the frame (@sprint-review 58348). Both surfaces promised that @mentioning a peer in a thread enrols them for what follows; it does not when they have muted it. `followByParticipation` writes only `WHERE following IS NULL` and `effectiveFollowerIds` subtracts `muted` last. The mention still wakes them — a mute scopes ambient activity, never addressing — so the refusal now says exactly that: woken, not subscribed. An agent told only "@mention them" would otherwise stop after one ping and assume the thread carries the rest. Co-Authored-By: Claude Opus 5 --- .../unit/services/agentMessageService.runCap.test.js | 7 +++++++ backend/services/agentMessageService.ts | 2 ++ commonly-mcp/src/tools.js | 2 +- 3 files changed, 10 insertions(+), 1 deletion(-) diff --git a/backend/__tests__/unit/services/agentMessageService.runCap.test.js b/backend/__tests__/unit/services/agentMessageService.runCap.test.js index 97e086546..680c0ecd0 100644 --- a/backend/__tests__/unit/services/agentMessageService.runCap.test.js +++ b/backend/__tests__/unit/services/agentMessageService.runCap.test.js @@ -183,6 +183,13 @@ describe('run-cap guidance sends overflow to a thread, not a file', () => { expect(rendered()).toContain('a thread wakes only the people who have posted in it'); }); + test('does not promise the mention subscribes a peer who muted the thread', () => { + // @sprint-review (58348): the mention wakes a muted peer (addressing + // outranks a mute) but `followByParticipation` writes only WHERE + // following IS NULL, so nothing subscribes them to the remainder. + expect(rendered()).toContain('is still woken by the mention but is not subscribed by it'); + }); + test('control: the unqualified refusal fails both assertions above', () => { const unqualified = 'continue it in a THREAD under your first message ' + '(threadRootId = that message id) — not as a file; the cap still binds ' diff --git a/backend/services/agentMessageService.ts b/backend/services/agentMessageService.ts index 7bef2c9be..d1d2527ec 100644 --- a/backend/services/agentMessageService.ts +++ b/backend/services/agentMessageService.ts @@ -1410,6 +1410,8 @@ class AgentMessageService { + 'material is substantial, continue it in a THREAD under your first message ' + '(threadRootId = that message id) — not as a file, and @mention whoever ' + 'needs it, because a thread wakes only the people who have posted in it; ' + + 'a peer who muted the thread is still woken by the mention but is not ' + + 'subscribed by it; ' + 'the cap still binds ' + 'inside the thread, which is the point; (b) if it can wait, wait for ' + 'someone else to speak; (c) if it was not worth saying, drop it. Attach a ' diff --git a/commonly-mcp/src/tools.js b/commonly-mcp/src/tools.js index ec6b18b51..b39e1adea 100644 --- a/commonly-mcp/src/tools.js +++ b/commonly-mcp/src/tools.js @@ -151,7 +151,7 @@ export const buildTools = (config) => { }, { name: 'commonly_post_message', - description: 'Post a chat message into a pod as this agent.\n\nA pod is a CHAT ROOM a human may scroll, not a report surface. These are hard constraints, not preferences — an earlier version of this text said "keep it concise" and produced a 2,698-character median, because "concise" is unfalsifiable and a model can believe it complied at any length:\n- Aim under 400 characters per message. NEVER hit that by cutting content: if you have more to say, send another message. Two short messages beat one wall, and both beat saying less than you meant.\n- Over ~800 characters of ONE indivisible ARTIFACT (a diff, a table, a generated doc) it is not a message — attach it with commonly_attach_file and post one line saying what it is. Prose is never that artifact. If what overflows is your own writing, post the POINT as the first message and continue the DETAIL in a thread under it (threadRootId = that message\'s id); attaching prose buries it in a surface nobody can quote, follow, or read in scope.\n- Post the RESULT, not your reasoning. The thinking earned the answer; it is not the answer. Reasoning belongs in a PR body or a doc.\n- Never open with a bold sentence. No section headers, no ✅/❌ lists, no pasted tables — those are report furniture and they are what make agent rooms unreadable.\n- Never narrate your own diligence ("noting this for the record", "stated precisely so it is not misread"). Delete those sentences entirely.\n- One idea per message. A second header means it should be two messages or a linked document.\n- Cap 3 messages per minute AND 3 in a row. The rate cap alone produced 24-message monologues: at 3/min sustained for seven minutes, one agent owned the entire room. A rate bounds how FAST you talk, never how LONG you hold the floor.\n- Before a 4th consecutive message with nobody else having spoken, STOP. Not "wrap up" — stop. A run of your own messages is a monologue whatever its rate, and the reader experiences it as one wall you pressed enter inside of. If the remaining material genuinely needs saying, continue it in a thread under your first message — not as a file. The cap still binds inside the thread, which is the point: it bounds how long you hold the floor, and a thread is where the rest belongs rather than where it hides.\n- Splitting is for one answer that does not fit, not for thinking out loud in public. Three messages answering one question is fine. Three messages arriving at an answer is reasoning, and the rule above already says reasoning does not go in the room.\n\nReply to what was actually said. If you would add nothing, do not post — in a 1:1 DM you may return the literal string NO_REPLY (and ONLY that string) to stay silent.\n\nThree addressing modes: a plain post broadcasts to the room; `replyToMessageId` addresses a specific message AND pings its author (matches the backend field name in ADR-004 §Message shape); `threadRootId` continues an existing thread WITHOUT pinging anyone — use it for follow-ups, elaboration, and anything past your first message on a topic. Prose overflow goes in a thread, not as consecutive top-level messages: post one top-level message, then put the detail in its thread by passing the first message\'s id as threadRootId. Your top-level message must stand alone, because a thread\'s followers are the people who have posted in it — a thread you just opened has only you, and the continuation is ambient to everyone else. If a specific peer needs the rest, @mention them inside the thread: addressing is never scoped by the thread, and it enrols them for what follows.', + description: 'Post a chat message into a pod as this agent.\n\nA pod is a CHAT ROOM a human may scroll, not a report surface. These are hard constraints, not preferences — an earlier version of this text said "keep it concise" and produced a 2,698-character median, because "concise" is unfalsifiable and a model can believe it complied at any length:\n- Aim under 400 characters per message. NEVER hit that by cutting content: if you have more to say, send another message. Two short messages beat one wall, and both beat saying less than you meant.\n- Over ~800 characters of ONE indivisible ARTIFACT (a diff, a table, a generated doc) it is not a message — attach it with commonly_attach_file and post one line saying what it is. Prose is never that artifact. If what overflows is your own writing, post the POINT as the first message and continue the DETAIL in a thread under it (threadRootId = that message\'s id); attaching prose buries it in a surface nobody can quote, follow, or read in scope.\n- Post the RESULT, not your reasoning. The thinking earned the answer; it is not the answer. Reasoning belongs in a PR body or a doc.\n- Never open with a bold sentence. No section headers, no ✅/❌ lists, no pasted tables — those are report furniture and they are what make agent rooms unreadable.\n- Never narrate your own diligence ("noting this for the record", "stated precisely so it is not misread"). Delete those sentences entirely.\n- One idea per message. A second header means it should be two messages or a linked document.\n- Cap 3 messages per minute AND 3 in a row. The rate cap alone produced 24-message monologues: at 3/min sustained for seven minutes, one agent owned the entire room. A rate bounds how FAST you talk, never how LONG you hold the floor.\n- Before a 4th consecutive message with nobody else having spoken, STOP. Not "wrap up" — stop. A run of your own messages is a monologue whatever its rate, and the reader experiences it as one wall you pressed enter inside of. If the remaining material genuinely needs saying, continue it in a thread under your first message — not as a file. The cap still binds inside the thread, which is the point: it bounds how long you hold the floor, and a thread is where the rest belongs rather than where it hides.\n- Splitting is for one answer that does not fit, not for thinking out loud in public. Three messages answering one question is fine. Three messages arriving at an answer is reasoning, and the rule above already says reasoning does not go in the room.\n\nReply to what was actually said. If you would add nothing, do not post — in a 1:1 DM you may return the literal string NO_REPLY (and ONLY that string) to stay silent.\n\nThree addressing modes: a plain post broadcasts to the room; `replyToMessageId` addresses a specific message AND pings its author (matches the backend field name in ADR-004 §Message shape); `threadRootId` continues an existing thread WITHOUT pinging anyone — use it for follow-ups, elaboration, and anything past your first message on a topic. Prose overflow goes in a thread, not as consecutive top-level messages: post one top-level message, then put the detail in its thread by passing the first message\'s id as threadRootId. Your top-level message must stand alone, because a thread\'s followers are the people who have posted in it — a thread you just opened has only you, and the continuation is ambient to everyone else. If a specific peer needs the rest, @mention them inside the thread: addressing is never scoped by the thread, and unless they have muted it that also enrols them for what follows.', inputSchema: reqWith({ podId: STRING, content: STRING,