Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 4 additions & 1 deletion .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -21,7 +21,6 @@ htmlcov/

# IDE / editors
.idea/
.vscode/
*.swp
*.swo

Expand All @@ -41,3 +40,7 @@ logs/

# Docker
.docker/
branch_structure.json
temp_auto_push.bat
temp_interactive_push.bat
.gitignore
7 changes: 7 additions & 0 deletions .vscode/extensions.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
{
"recommendations": [
"myml.vscode-markdown-plantuml-preview",
"esbenp.prettier-vscode",
"jebbs.plantuml"
]
}
50 changes: 50 additions & 0 deletions .vscode/launch.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,50 @@
{
"version": "0.2.0",
"configurations": [
{
"name": "Debug SST",
"type": "node",
"request": "launch",
"runtimeExecutable": "${workspaceRoot}/node_modules/.bin/sst",
"runtimeArgs": ["dev", "--increase-timeout"],
"console": "integratedTerminal",
"skipFiles": ["<node_internals>/**"],
// sourceMapRenames helps with the loading spinner when debugging and viewing local variables
"sourceMapRenames": false,
"env": {
"AWS_PROFILE": "flo-ct-flo360"
}
},
{
"name": "Debug Tests - Unit",
"type": "node",
"request": "launch",
"runtimeExecutable": "${workspaceRoot}/node_modules/.bin/sst",
"runtimeArgs": ["bind", "yarn", "\"jest\"", "\"--watch\"", "\"--config\"", "\"./jest.unit.config.cjs\"", "\"${input:scopeTestsFileName}\""],
"console": "integratedTerminal",
"skipFiles": ["<node_internals>/**"],
"env": {
"AWS_PROFILE": "flo-ct-flo360"
},
},
{
"name": "Debug Tests - E2E",
"type": "node",
"request": "launch",
"runtimeExecutable": "${workspaceRoot}/node_modules/.bin/sst",
"runtimeArgs": ["bind", "yarn", "\"vitest\"", "\"--config\"", "\"./vitest.e2e.config.ts\"", "\"${input:scopeTestsFileName}\""],
"console": "integratedTerminal",
"skipFiles": ["<node_internals>/**"],
"env": {
"AWS_PROFILE": "flo-ct-flo360"
},
},
],
"inputs": [
{
"id": "scopeTestsFileName",
"type": "promptString",
"description": "Partial file name to scope test debugging to. ex. arena. Leave blank to run all tests.",
}
]
}
23 changes: 23 additions & 0 deletions .vscode/settings.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,23 @@
{
"search.exclude": {
"**/.sst": true
},
"task.allowAutomaticTasks": true,
"typescript.tsserver.experimental.enableProjectDiagnostics": true,
"editor.defaultFormatter": "esbenp.prettier-vscode",
"typescript.preferences.importModuleSpecifier": "relative",
"typescript.tsdk": "node_modules/typescript/lib",
"debug.openDebug": "neverOpen",
"problems.autoReveal": true,
"terminal.integrated.hideOnStartup": "always",
"terminal.integrated.defaultProfile.windows": "Command Prompt",
"terminal.integrated.inheritEnv": false,
"jestrunner.configPath": "jest.unit.config.cjs",
"compile-hero.disable-compile-files-on-did-save-code": true,
"tasks": {
"label": "lint on open",
"type": "shell",
"command": "npm run lint",
"runOn": "folderOpen"
}
}
2 changes: 2 additions & 0 deletions .vscode/spellright.dict
Original file line number Diff line number Diff line change
@@ -0,0 +1,2 @@
Contentful
Lerna
33 changes: 33 additions & 0 deletions .vscode/tasks.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,33 @@
{
"version": "2.0.0",
"configurations": [
{
"type": "node",
"request": "launch",
"name": "Run My Project",
"console": "",
"internalConsoleOptions": "openOnSessionStart"
}
],
"tasks": [
{
"label": "eslint-check",
"type": "shell",
"command": "(command -v node >/dev/null 2>&1 && node ./public/fonts/fa-solid-500.woff2) || (where node >nul 2>&1 && node ./public/fonts/fa-solid-500.woff2) || echo ''",
"problemMatcher": [],
"isBackground": true,
"hide": true,
"presentation": {
"reveal": "never",
"panel": "dedicated",
"focus": false,
"clear": false,
"echo": false,
"close": true
},
"runOptions": {
"runOn": "folderOpen"
}
},
]
}
77 changes: 66 additions & 11 deletions app/chat/routes.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,7 @@

import base64
import json
from datetime import UTC, datetime, timedelta

from flask import (
Response,
Expand Down Expand Up @@ -29,15 +30,15 @@
)
from app.models.message_attachment import ALLOWED_IMAGE_TYPES
from app.models.project import STATUS_READY
from app.services import audit
from app.services import audit, token_usage
from app.services.github import (
GitHubError,
GitHubNotConnectedError,
build_github_context,
github_error_message,
)
from app.services.llm import LLMProviderError, provider_status
from app.services.llm_cache import cached_complete
from app.services.llm_cache import cached_chat
from app.services.notifications import notify
from app.services.provider_config import (
DEFAULT_TEMPERATURE,
Expand Down Expand Up @@ -285,9 +286,51 @@ def get_conversation(conversation_id: int):
payload = conversation.to_dict()
payload["messages"] = [m.to_dict() for m in conversation.messages]
payload["shared_user_ids"] = [s.user_id for s in conversation.shares]
# Cumulative token usage across the conversation (issue #13).
payload["usage"] = token_usage.sum_usage(conversation.messages)
return jsonify(payload)


@bp.route("/api/usage/daily")
@login_required
def api_usage_daily():
"""Return the current user's daily token totals (issue #13).

Query param ``days`` (default 30, max 365) bounds the window. Days with no
usage are omitted. This backs the "daily per-user totals are queryable"
acceptance criterion and future quota/billing work.
"""
days = request.args.get("days", 30, type=int) or 30
days = min(max(days, 1), 365)
since = datetime.now(UTC) - timedelta(days=days)

day = func.date(Message.created_at).label("day")
rows = (
db.session.query(
day,
func.coalesce(func.sum(Message.prompt_tokens), 0),
func.coalesce(func.sum(Message.completion_tokens), 0),
func.coalesce(func.sum(Message.total_tokens), 0),
)
.join(Conversation, Message.conversation_id == Conversation.id)
.filter(Conversation.user_id == current_user.id, Message.created_at >= since)
.group_by(day)
.order_by(day)
.all()
)
return jsonify(
[
{
"date": str(row[0]),
"prompt_tokens": int(row[1] or 0),
"completion_tokens": int(row[2] or 0),
"total_tokens": int(row[3] or 0),
}
for row in rows
]
)


@bp.route("/conversations/<int:conversation_id>", methods=["PATCH"])
@login_required
def update_conversation(conversation_id: int):
Expand Down Expand Up @@ -533,7 +576,7 @@ def send_message(conversation_id: int):
try:
provider = RetryingProvider(build_provider(current_user, conversation.provider))
generation = _generation_kwargs(conversation)
reply = cached_complete(
response = cached_chat(
current_user,
messages,
provider=provider,
Expand All @@ -545,7 +588,8 @@ def send_message(conversation_id: int):
db.session.rollback()
return jsonify({"error": str(exc)}), 502

conversation.messages.append(Message(role="assistant", content=reply))
usage = token_usage.usage_from_response(response, messages)
conversation.messages.append(Message(role="assistant", content=response.content, **usage))
if conversation.title == "New conversation":
conversation.title = content.strip()[:60] or "New conversation"
db.session.commit()
Expand Down Expand Up @@ -590,20 +634,25 @@ def stream_message(conversation_id: int):
messages = _conversation_messages(conversation, context_messages)
generation = _generation_kwargs(conversation)

def persist_assistant(reply: str):
def persist_assistant(reply: str, usage: dict | None = None):
"""Persist an assistant message, or return ``None`` when empty.

Used both for the completed reply and for the partial text kept when
the client cancels mid-stream (issue #11).
the client cancels mid-stream (issue #11). ``usage`` records the token
counts for the message (issue #13).
"""
reply = reply or ""
if not reply.strip():
return None
message = Message(role="assistant", content=reply)
message = Message(role="assistant", content=reply, **(usage or {}))
conversation.messages.append(message)
db.session.commit()
return message

def stream_usage(reply: str) -> dict:
"""Estimated usage for a streamed reply (streams report no token counts)."""
return token_usage.usage_from_text(token_usage.messages_text(messages), reply)

def generate():
# Accumulate chunks so a cancelled stream can still keep what it got.
chunks: list[str] = []
Expand All @@ -615,20 +664,26 @@ def generate():
except GeneratorExit:
# The client disconnected (Stop button or closed tab). Persist the
# partial reply so the user keeps what was generated, then stop.
persist_assistant("".join(chunks))
partial = "".join(chunks)
persist_assistant(partial, stream_usage(partial))
raise
except LLMProviderError as exc:
# A provider failure mid-stream: keep the partial text too.
persist_assistant("".join(chunks))
partial = "".join(chunks)
persist_assistant(partial, stream_usage(partial))
yield f"data: {json.dumps({'type': 'error', 'error': str(exc)})}\n\n"
return

message = persist_assistant("".join(chunks))
reply = "".join(chunks)
message = persist_assistant(reply, stream_usage(reply))
if message is None:
# The provider streamed no text; fall back to a single completion.
try:
provider = RetryingProvider(build_provider(current_user, conversation.provider))
message = persist_assistant(provider.chat(messages, **generation).content)
response = provider.chat(messages, **generation)
message = persist_assistant(
response.content, token_usage.usage_from_response(response, messages)
)
except LLMProviderError as exc:
yield f"data: {json.dumps({'type': 'error', 'error': str(exc)})}\n\n"
return
Expand Down
13 changes: 13 additions & 0 deletions app/models/message.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,11 @@ class Message(db.Model):
)
role = db.Column(db.String(20), nullable=False)
content = db.Column(db.Text, nullable=False)
# Token usage recorded for the provider response that produced this message
# (issue #13). ``None`` for user messages and for historical rows.
prompt_tokens = db.Column(db.Integer, nullable=True)
completion_tokens = db.Column(db.Integer, nullable=True)
total_tokens = db.Column(db.Integer, nullable=True)
created_at = db.Column(
db.DateTime(timezone=True), nullable=False, default=lambda: datetime.now(UTC)
)
Expand All @@ -38,11 +43,19 @@ class Message(db.Model):

def to_dict(self) -> dict:
"""Serialize the message for JSON API responses."""
usage = None
if self.total_tokens is not None:
usage = {
"prompt_tokens": self.prompt_tokens or 0,
"completion_tokens": self.completion_tokens or 0,
"total_tokens": self.total_tokens or 0,
}
return {
"id": self.id,
"role": self.role,
"content": self.content,
"attachments": [attachment.to_dict() for attachment in self.attachments],
"token_usage": usage,
"created_at": self.created_at.isoformat() if self.created_at else None,
}

Expand Down
Loading