Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
15 changes: 14 additions & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -148,7 +148,9 @@ Remove-Item -Recurse -Force (Join-Path $env:APPDATA "bt") -ErrorAction SilentlyC
| `bt projects` | Manage projects (list, create, view, delete) |
| `bt datasets` | Manage remote datasets (list, create, update, view, delete) |
| `bt prompts` | Manage prompts (list, view, versions, assign, delete) |
| `bt scorers` | Manage scorers (list, create, view, invoke, delete) |
| `bt functions` | Manage functions (list, view, invoke, update, push, pull, delete) |
| `bt tools` | Manage tools (list, view, invoke, update, delete) |
| `bt scorers` | Manage scorers (list, create, view, invoke, update, delete) |
| `bt environments` | Manage deployment environments (list, view, create, update, delete) |
| `bt sync` | Synchronize project logs between Braintrust and local NDJSON files |
| `bt update` | Update bt in-place |
Expand All @@ -174,6 +176,17 @@ Use `--if-exists error|ignore|replace` to control slug conflicts. Text and struc

Before writing a scorer, `bt` sends the complete candidate definition to Braintrust for validation. The backend applies the same model-parameter and replacement checks as the write and returns structured issues with normalization suggestions when available.

Update only the fields you specify, or use `--patch` for fields without dedicated flags:

```bash
bt scorers update helpfulness --messages @messages.json
bt scorers update helpfulness --new-slug answer-helpfulness
bt functions update my-function --name "Updated function" --description "Updated"
bt tools update my-tool --prompt @prompt.txt --new-slug lookup-order
```

The API replaces `prompt_data` rather than merging it, so `bt` reads the current definition and sends a materialized replacement with your changes. A concurrent edit can therefore be overwritten.

For TypeScript and Python code scorers, use the Braintrust SDK and `bt functions push`.

## `bt eval`
Expand Down
14 changes: 14 additions & 0 deletions src/functions/api.rs
Original file line number Diff line number Diff line change
Expand Up @@ -261,6 +261,20 @@ pub async fn delete_function(client: &ApiClient, function_id: &str) -> Result<()
client.delete(&path).await
}

/// Partially update a function (scorer/tool/prompt/...) by id.
///
/// Top-level fields are patched, but object-valued fields such as `prompt_data`
/// are replaced wholesale. Callers updating `prompt_data` must materialize the
/// complete value before sending the request.
pub async fn patch_function(
client: &ApiClient,
function_id: &str,
body: &serde_json::Value,
) -> Result<serde_json::Value> {
let path = format!("/v1/function/{}", encode(function_id));
client.patch(&path, body).await
}

pub async fn list_functions_page(
client: &ApiClient,
query: &FunctionListQuery,
Expand Down
109 changes: 23 additions & 86 deletions src/functions/create.rs
Original file line number Diff line number Diff line change
@@ -1,20 +1,17 @@
use anyhow::{bail, Context, Result};
use clap::{builder::BoolishValueParser, ArgGroup, Args};
use dialoguer::Input;
use serde_json::{json, Map, Value};
use serde_json::{json, Value};

use crate::{
error::user_error,
ui::{is_interactive, print_command_status, with_spinner, CommandStatus},
utils::{merge_json_objects, read_text_source, read_yaml_object_source},
};

use super::{
api,
prompt_config::{
parse_choice_scores_source, parse_classifications_source, validate_unit_interval,
PromptConfigArgs,
},
prompt_config::PromptConfigArgs,
scorer_config::{build_scorer_config, ScorerConfig},
IfExistsMode, ResolvedContext,
};

Expand Down Expand Up @@ -275,101 +272,41 @@ fn build_scorer_definition(
name: &str,
slug: &str,
) -> Result<Value> {
let prompt = resolve_prompt_block(args)?;
let (function_type, parser) = resolve_output_parser(args)?;

let mut prompt_data = json!({
"prompt": prompt,
"parser": parser,
})
.as_object()
.expect("prompt data is an object")
.clone();
let prompt_config = args
.prompt_config
.build_prompt_data_patch(Some(&args.model))?;
merge_json_objects(&mut prompt_data, &prompt_config);
let config = build_scorer_config(
&ScorerConfig {
messages: Some(&args.messages),
model: Some(&args.model),
prompt_config: &args.prompt_config,
choice_scores: args.choice_scores.as_deref(),
classifications: args.classifications.as_deref(),
use_cot: Some(args.use_cot),
allow_no_match: args.classifications.as_ref().map(|_| args.allow_no_match),
pass_threshold: args.pass_threshold,
metadata: args.metadata.as_deref(),
metadata_label: "scorer metadata",
},
true,
)?;

let mut definition = json!({
"project_id": project_id,
"name": name,
"slug": slug,
"function_data": {
"type": "prompt",
},
"prompt_data": prompt_data,
"function_data": { "type": "prompt" },
"if_exists": args.if_exists.as_str(),
"function_type": function_type,
});
definition
.as_object_mut()
.expect("scorer definition is an object")
.extend(config);

if let Some(description) = args.description.as_deref() {
definition["description"] = Value::String(description.to_string());
}

let metadata = resolve_metadata(args)?;
if !metadata.is_empty() {
definition["metadata"] = Value::Object(metadata);
}

Ok(definition)
}

fn resolve_output_parser(args: &CreateArgs) -> Result<(&'static str, Value)> {
match (
args.choice_scores.as_deref(),
args.classifications.as_deref(),
) {
(Some(source), None) => Ok((
"scorer",
json!({
"type": "llm_classifier",
"use_cot": args.use_cot,
"choice_scores": parse_choice_scores_source(source)?,
}),
)),
(None, Some(source)) => Ok((
"classifier",
json!({
"type": "llm_classifier",
"use_cot": args.use_cot,
"choice": parse_classifications_source(source)?,
"allow_no_match": args.allow_no_match,
}),
)),
(Some(_), Some(_)) => bail!(
"use either --choice-scores for score output or --classifications for classification output, not both"
),
(None, None) => bail!(
"output choices required. Pass --choice-scores <SOURCE> or --classifications <SOURCE>"
),
}
}

fn resolve_metadata(args: &CreateArgs) -> Result<Map<String, Value>> {
let mut metadata = match args.metadata.as_deref() {
Some(source) => read_yaml_object_source(source, "scorer metadata")?,
None => Map::new(),
};
if let Some(pass_threshold) = args.pass_threshold {
validate_unit_interval(pass_threshold, "--pass-threshold")?;
metadata.insert("__pass_threshold".to_string(), json!(pass_threshold));
}
Ok(metadata)
}

fn resolve_prompt_block(args: &CreateArgs) -> Result<Value> {
let raw = read_text_source(&args.messages, "messages")?;
parse_messages(&raw)
}

fn parse_messages(raw: &str) -> Result<Value> {
let messages: Value = serde_json::from_str(raw).context("invalid JSON in scorer messages")?;
match messages {
Value::Array(_) => Ok(json!({ "type": "chat", "messages": messages })),
_ => bail!("scorer messages must be a JSON array"),
}
}

#[cfg(test)]
mod tests {
use clap::Parser;
Expand Down
107 changes: 103 additions & 4 deletions src/functions/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -18,9 +18,12 @@ mod delete;
mod invoke;
mod list;
pub(crate) mod prompt_config;
pub(crate) mod prompt_patch;
mod pull;
mod push;
pub(crate) mod report;
mod scorer_config;
mod update;
mod view;

use api::Function;
Expand Down Expand Up @@ -114,7 +117,6 @@ pub enum PushLanguage {
fn build_web_path(function: &Function) -> String {
let id = &function.id;
match function.function_type.as_deref() {
Some("tool") => format!("tools?pr={}", urlencoding::encode(id)),
Some("scorer") => format!("scorers/{}", urlencoding::encode(id)),
Some("classifier") if function.prompt_data.is_some() => {
format!("scorers/{}", urlencoding::encode(id))
Expand All @@ -128,7 +130,8 @@ fn build_web_path(function: &Function) -> String {
)
}
Some("parameters") => format!("parameters/{}", urlencoding::encode(id)),
_ => format!("functions/{}", urlencoding::encode(id)),
Some("llm") => format!("prompts/{}", urlencoding::encode(id)),
_ => format!("tools?pr={}", urlencoding::encode(id)),
}
}

Expand Down Expand Up @@ -173,8 +176,7 @@ Examples:
bt tools view my-tool
bt tools view fn_123
bt tools view --id fn_123
bt scorers list
bt scorers delete my-scorer
bt tools update my-tool --name \"Lookup order\" --new-slug lookup-order
")]
pub struct FunctionArgs {
#[command(subcommand)]
Expand All @@ -191,6 +193,22 @@ pub(crate) enum FunctionCommands {
Delete(DeleteArgs),
/// Invoke by slug
Invoke(invoke::InvokeArgs),
/// Update tool fields or prompt content
Update(Box<update::ToolUpdateArgs>),
}

#[derive(Debug, Clone, Subcommand)]
pub(crate) enum ScorerFunctionCommands {
/// List all in the current project
List,
/// View details
View(ViewArgs),
/// Delete by slug
Delete(DeleteArgs),
/// Invoke by slug
Invoke(invoke::InvokeArgs),
/// Update scorer configuration
Update(Box<update::ScorerUpdateArgs>),
}

#[derive(Debug, Clone, Args)]
Expand Down Expand Up @@ -223,6 +241,8 @@ enum FunctionsCommands {
Delete(FunctionsDeleteArgs),
/// Invoke a function
Invoke(FunctionsInvokeArgs),
/// Update common function fields or apply an arbitrary patch
Update(Box<FunctionsUpdateArgs>),
/// Push local function definitions
Push(PushArgs),
/// Pull remote function definitions
Expand Down Expand Up @@ -272,6 +292,15 @@ struct FunctionsInvokeArgs {
function_type: Option<FunctionTypeFilter>,
}

#[derive(Debug, Clone, Args)]
struct FunctionsUpdateArgs {
#[command(flatten)]
inner: update::GenericUpdateArgs,
/// Filter by function type (for interactive selection)
#[arg(long = "type", short = 't', value_enum)]
function_type: Option<FunctionTypeFilter>,
}

#[derive(Debug, Clone, Args)]
pub(crate) struct PushArgs {
/// File or directory path(s) to scan for function definitions.
Expand Down Expand Up @@ -656,6 +685,7 @@ pub(crate) async fn run_typed_command(
None | Some(FunctionCommands::List) => list::run(&ctx, base.json, ft).await,
Some(FunctionCommands::Delete(d)) => delete::run(&ctx, d.slug(), d.force, ft).await,
Some(FunctionCommands::Invoke(i)) => invoke::run(&ctx, &i, base.json, ft).await,
Some(FunctionCommands::Update(u)) => update::run_tool(&ctx, &u, base.json).await,
Some(FunctionCommands::View(_)) => {
unreachable!("handled before context resolution")
}
Expand All @@ -670,6 +700,41 @@ pub(crate) async fn run_scorer_create(base: BaseArgs, args: create::CreateArgs)
create::run(&ctx, &args, json_output).await
}

pub(crate) async fn run_scorer_command(
base: BaseArgs,
command: Option<ScorerFunctionCommands>,
) -> Result<()> {
let ft = Some(FunctionTypeFilter::Scorer);
match command {
Some(ScorerFunctionCommands::View(v)) => match v.selector()? {
ViewSelector::Id(id) => {
let auth_ctx = resolve_auth_context(&base).await?;
view::run_by_id(&auth_ctx, id, v.options(&base), ft).await
}
ViewSelector::Slug(slug) => {
let ctx = resolve_context(&base).await?;
view::run(&ctx, slug, v.options(&base), ft).await
}
},
command => {
let ctx = resolve_context(&base).await?;
match command {
None | Some(ScorerFunctionCommands::List) => list::run(&ctx, base.json, ft).await,
Some(ScorerFunctionCommands::Delete(d)) => {
delete::run(&ctx, d.slug(), d.force, ft).await
}
Some(ScorerFunctionCommands::Invoke(i)) => {
invoke::run(&ctx, &i, base.json, ft).await
}
Some(ScorerFunctionCommands::Update(u)) => {
update::run_scorer(&ctx, &u, base.json).await
}
Some(ScorerFunctionCommands::View(_)) => unreachable!("handled above"),
}
}
}
}

pub async fn run(base: BaseArgs, args: FunctionsArgs) -> Result<()> {
let function_type = args.function_type;
match args.command {
Expand Down Expand Up @@ -701,6 +766,15 @@ pub async fn run(base: BaseArgs, args: FunctionsArgs) -> Result<()> {
Some(FunctionsCommands::Invoke(i)) => {
invoke::run(&ctx, &i.inner, base.json, i.function_type.or(function_type)).await
}
Some(FunctionsCommands::Update(u)) => {
update::run_generic(
&ctx,
&u.inner,
base.json,
u.function_type.or(function_type),
)
.await
}
Some(FunctionsCommands::Push(_))
| Some(FunctionsCommands::Pull(_))
| Some(FunctionsCommands::View(_)) => {
Expand Down Expand Up @@ -1218,6 +1292,31 @@ mod tests {
)
}

#[test]
fn typed_function_update_command_is_not_read_only() {
let _guard = test_lock();
let parsed = FunctionArgsHarness::try_parse_from(["bt-tools", "update", "my-tool", "-y"])
.expect("parse");
assert!(!function_command_is_read_only(parsed.args.command.as_ref()));
}

#[test]
fn functions_update_command_is_not_read_only() {
let _guard = test_lock();
let parsed = FunctionsArgsHarness::try_parse_from([
"bt-functions",
"update",
"my-fn",
"--description",
"x",
"-y",
])
.expect("parse");
assert!(!functions_command_is_read_only(
parsed.args.command.as_ref()
));
}

#[test]
fn typed_function_commands_map_to_expected_auth_mode() {
let _guard = test_lock();
Expand Down
Loading
Loading