From f11743afdaab542914ee36d6eda4ea79848c7363 Mon Sep 17 00:00:00 2001 From: Elijah Kurien <80718858+elijah0528@users.noreply.github.com> Date: Wed, 30 Sep 2026 22:36:09 +0000 Subject: [PATCH 1/2] feat: add `project_group_name` to create projects inside a project group Python port of braintrustdata/braintrust-sdk-javascript#2541. `init_logger`, `init`, `init_dataset`, and `Eval`/`EvalAsync` now accept `project_group_name`. When the named project does not exist yet, it is created inside that project group, which lets callers who only hold project-creation permission on a group (rather than on the whole organization) register projects. The option is only honored when the project is looked up by name -- it is ignored when `project_id` is supplied -- and it is omitted from the wire payload entirely when unset, so existing callers see no change in behavior or requests. Notes on the port: - The JS SDK registers projects through `api/project/register`; the Python SDK was migrated to the generated `POST /v1/project` resource in #705. Both are backed by the same `createProjectSchema`, which the API half of this change extends, so the new field is threaded into that body via a shared `_register_project` helper. - `init`/`init_dataset` need no extra round trip here. Unlike the JS versions, they already resolve the project via project registration and then register the experiment/dataset by `project_id`. - The JS change also keys its project-metadata LRU cache on the group name. Python's `_compute_logger_metadata` has no such cache, so there is nothing to key. - New parameters are appended to each signature rather than inserted next to `project_id`, so existing positional callers are unaffected. `test_project_group.py` drives the real entrypoints against a local HTTP server (`api/_test_server.scripted_server`) and asserts on the request bodies the SDK actually sends, including that the field is absent when unset. Co-Authored-By: Claude Opus 5 --- py/src/braintrust/cli/eval.py | 1 + py/src/braintrust/framework.py | 16 ++ py/src/braintrust/logger.py | 42 +++++- py/src/braintrust/test_helpers.py | 2 +- py/src/braintrust/test_project_group.py | 189 ++++++++++++++++++++++++ 5 files changed, 241 insertions(+), 9 deletions(-) create mode 100644 py/src/braintrust/test_project_group.py diff --git a/py/src/braintrust/cli/eval.py b/py/src/braintrust/cli/eval.py index f0e5dc890..1fe950d18 100644 --- a/py/src/braintrust/cli/eval.py +++ b/py/src/braintrust/cli/eval.py @@ -143,6 +143,7 @@ async def run_evaluator_task(evaluator, position, opts: EvaluatorOpts): experiment = init_experiment( project_name=evaluator.project_name, project_id=evaluator.project_id, + project_group_name=evaluator.project_group_name, experiment_name=evaluator.experiment_name, description=evaluator.description, metadata=evaluator.metadata, diff --git a/py/src/braintrust/framework.py b/py/src/braintrust/framework.py index b352252e9..3a9cb365d 100644 --- a/py/src/braintrust/framework.py +++ b/py/src/braintrust/framework.py @@ -462,6 +462,13 @@ class Evaluator(Generic[Input, Output, Expected]): parameter_values: dict[str, Any] | None = None + project_group_name: str | None = None + """ + If specified, creates the project inside the project group with this name when the project does + not already exist. Requires permission to create projects in that group. Ignored if `project_id` + is specified. + """ + @dataclasses.dataclass class EvalResultWithSummary(SerializableDataClass, Generic[Input, Output, Expected]): @@ -704,6 +711,7 @@ def _EvalCommon( parent: str | None = None, state: BraintrustState | None = None, enable_cache: bool = True, + project_group_name: str | None = None, ) -> Callable[[], Coroutine[Any, Any, EvalResultWithSummary[Input, Output, Expected]]]: """ This helper is needed because in case of `_lazy_load`, we need to update @@ -743,6 +751,7 @@ def _EvalCommon( description=description, summarize_scores=summarize_scores, parameters=parameters, + project_group_name=project_group_name, ) if _lazy_load: @@ -781,6 +790,7 @@ async def make_empty_summary(): experiment = init_experiment( project_name=evaluator.project_name if evaluator.project_id is None else None, project_id=evaluator.project_id, + project_group_name=evaluator.project_group_name, experiment_name=evaluator.experiment_name, description=evaluator.description, metadata=evaluator.metadata, @@ -845,6 +855,7 @@ async def EvalAsync( parent: str | None = None, state: BraintrustState | None = None, enable_cache: bool = True, + project_group_name: str | None = None, ) -> EvalResultWithSummary[Input, Output, Expected]: """ A function you can use to define an evaluator. This is a convenience wrapper around the `Evaluator` class. @@ -885,6 +896,7 @@ async def EvalAsync( :param timeout: (Optional) The duration, in seconds, after which to time out the evaluation. Defaults to None, in which case there is no timeout. :param project_id: (Optional) If specified, uses the given project ID instead of the evaluator's name to identify the project. + :param project_group_name: (Optional) Creates the project inside the project group with this name, if the project does not already exist. Requires permission to create projects in that group. Ignored if `project_id` is specified. :param base_experiment_name: An optional experiment name to use as a base. If specified, the new experiment will be summarized and compared to this experiment. :param base_experiment_id: An optional experiment id to use as a base. If specified, the new experiment will be @@ -937,6 +949,7 @@ async def EvalAsync( parent=parent, state=state, enable_cache=enable_cache, + project_group_name=project_group_name, ) return await f() @@ -974,6 +987,7 @@ def Eval( parent: str | None = None, state: BraintrustState | None = None, enable_cache: bool = True, + project_group_name: str | None = None, ) -> EvalResultWithSummary[Input, Output, Expected]: """ A function you can use to define an evaluator. This is a convenience wrapper around the `Evaluator` class. @@ -1014,6 +1028,7 @@ def Eval( :param timeout: (Optional) The duration, in seconds, after which to time out the evaluation. Defaults to None, in which case there is no timeout. :param project_id: (Optional) If specified, uses the given project ID instead of the evaluator's name to identify the project. + :param project_group_name: (Optional) Creates the project inside the project group with this name, if the project does not already exist. Requires permission to create projects in that group. Ignored if `project_id` is specified. :param base_experiment_name: An optional experiment name to use as a base. If specified, the new experiment will be summarized and compared to this experiment. :param base_experiment_id: An optional experiment id to use as a base. If specified, the new experiment will be @@ -1068,6 +1083,7 @@ def Eval( parent=parent, state=state, enable_cache=enable_cache, + project_group_name=project_group_name, ) # https://stackoverflow.com/questions/55409641/asyncio-run-cannot-be-called-from-a-running-event-loop-when-using-jupyter-no try: diff --git a/py/src/braintrust/logger.py b/py/src/braintrust/logger.py index 2e7a7adf4..8ad612834 100644 --- a/py/src/braintrust/logger.py +++ b/py/src/braintrust/logger.py @@ -1545,6 +1545,23 @@ class OrgProjectMetadata: project: ObjectMetadata +def _register_project( + state: "BraintrustState", + name: str | None, + project_group_name: str | None = None, +) -> Mapping[str, Any]: + """Register (get-or-create) a project by name, optionally inside a project group. + + `project_group_name` is omitted from the request body entirely when unset, so callers that do + not use project groups send exactly the same payload as before. + """ + + body: dict[str, Any] = {"name": name, "org_name": state.org_name} + if project_group_name is not None: + body["project_group_name"] = project_group_name + return state.api_client().projects.post_project(body=cast(Any, body)) + + # Pyright produces an error for overlapping overloads # (reportOverlappingOverload) because of the default argument to `open`. It # thinks a call like `init()` with no arguments could match both overloads. @@ -1573,6 +1590,7 @@ def init( base_experiment_id: str | None = ..., repo_info: RepoInfo | None = ..., state: BraintrustState | None = ..., + project_group_name: str | None = ..., ) -> "Experiment": ... @@ -1598,6 +1616,7 @@ def init( base_experiment_id: str | None = ..., repo_info: RepoInfo | None = ..., state: BraintrustState | None = ..., + project_group_name: str | None = ..., ) -> "ReadonlyExperiment": ... @@ -1622,6 +1641,7 @@ def init( base_experiment_id: str | None = None, repo_info: RepoInfo | None = None, state: BraintrustState | None = None, + project_group_name: str | None = None, ) -> "Experiment | ReadonlyExperiment": """ Log in, and then initialize a new experiment in a specified project. If the project does not exist, it will be created. @@ -1646,6 +1666,7 @@ def init( :param set_current: If true (the default), set the global current-experiment to the newly-created one. :param open: If the experiment already exists, open it in read-only mode. Throws an error if the experiment does not already exist. :param project_id: The id of the project to create the experiment in. This takes precedence over `project` if specified. + :param project_group_name: (Optional) Create the project inside the project group with this name, if the project does not already exist. Requires permission to create projects in that group. Ignored if `project_id` is specified. :param base_experiment_id: An optional experiment id to use as a base. If specified, the new experiment will be summarized and compared to this. This takes precedence over `base_experiment` if specified. :param repo_info: (Optional) Explicitly specify the git metadata for this experiment. This takes precedence over `git_metadata_settings` if specified. :param state: (Optional) A BraintrustState object to use. If not specified, will use the global state. This is for advanced use only. @@ -1696,9 +1717,7 @@ def compute_metadata(): def compute_metadata(): state.login(org_name=org_name, api_key=api_key, app_url=app_url) if project_id is None: - project_info = state.api_client().projects.post_project( - body={"name": project or GLOBAL_PROJECT, "org_name": state.org_name} - ) + project_info = _register_project(state, project or GLOBAL_PROJECT, project_group_name=project_group_name) else: project_info = state.api_client().projects.get_project_id(project_id) @@ -1812,6 +1831,7 @@ def init_dataset( state: BraintrustState | None = None, environment: str | None = None, dataset_id: str | None = None, + project_group_name: str | None = None, ) -> "Dataset": """ Create or load a dataset. When creating a dataset, its project will be created if it does not exist. @@ -1827,6 +1847,7 @@ def init_dataset( key is specified, will prompt the user to login. :param org_name: (Optional) The name of a specific organization to connect to. This is useful if you belong to multiple. :param project_id: The id of the project to create the dataset in. This takes precedence over `project` if specified. + :param project_group_name: (Optional) Create the project inside the project group with this name, if the project does not already exist. Requires permission to create projects in that group. Ignored if `project_id` is specified. :param metadata: (Optional) a dictionary, or an object that serializes to a dictionary (such as a Pydantic model), with additional data about the dataset. The values in `metadata` can be any JSON-serializable type, but its keys must be strings. :param use_output: (Deprecated) If True, records will be fetched from this dataset in the legacy format, with the "expected" field renamed to "output". This option will be removed in a future version of Braintrust. @@ -1854,7 +1875,7 @@ def compute_metadata(): if project_id is not None: resp_project = api_client.projects.get_project_id(project_id) else: - resp_project = api_client.projects.post_project(body={"name": project, "org_name": state.org_name}) + resp_project = _register_project(state, project, project_group_name=project_group_name) body = _populate_args( {"project_id": resp_project["id"], "name": name or "logs"}, description=description, @@ -1879,15 +1900,14 @@ def compute_metadata(): def _compute_logger_metadata( project_name: str | None = None, project_id: str | None = None, + project_group_name: str | None = None, state: BraintrustState | None = None, ): state = state or _state state.login() org_id = state.org_id if project_id is None: - response = state.api_client().projects.post_project( - body={"name": project_name or GLOBAL_PROJECT, "org_name": state.org_name} - ) + response = _register_project(state, project_name or GLOBAL_PROJECT, project_group_name=project_group_name) return OrgProjectMetadata( org_id=org_id, project=ObjectMetadata(id=response["id"], name=response["name"], full_info=dict(response)), @@ -1915,6 +1935,7 @@ def init_logger( set_current: bool = True, state: BraintrustState | None = None, environment: SpanOriginEnvironment | None = None, + project_group_name: str | None = None, ) -> "Logger": """ Create a new logger in a specified project. If the project does not exist, it will be created. @@ -1928,12 +1949,17 @@ def init_logger( :param org_name: (Optional) The name of a specific organization to connect to. This is useful if you belong to multiple. :param force_login: Login again, even if you have already logged in (by default, the logger will not login if you are already logged in) :param set_current: If true (the default), set the global current-experiment to the newly-created one. + :param project_group_name: (Optional) Create the project inside the project group with this name, if the project does not already exist. Requires permission to create projects in that group. Ignored if `project_id` is specified. :returns: The newly created Logger. """ state = state or _state state.span_origin_environment = detect_environment(environment) - compute_metadata_args = dict(project_name=project, project_id=project_id) + compute_metadata_args: dict[str, Any] = dict(project_name=project, project_id=project_id) + if project_group_name is not None: + # Only present when set, so exported span components (and therefore anything that resolves + # them with an older SDK) are byte-for-byte unchanged for callers not using project groups. + compute_metadata_args["project_group_name"] = project_group_name link_args = { "app_url": app_url, diff --git a/py/src/braintrust/test_helpers.py b/py/src/braintrust/test_helpers.py index 585454bc3..5d1ba3adb 100644 --- a/py/src/braintrust/test_helpers.py +++ b/py/src/braintrust/test_helpers.py @@ -104,7 +104,7 @@ def init_test_logger(project_name: str): l._lazy_metadata = lazy_metadata # Skip actual login by setting fake metadata directly # Replace the global _compute_logger_metadata function with a resolved LazyValue - def fake_compute_logger_metadata(project_name=None, project_id=None, state=None): + def fake_compute_logger_metadata(project_name=None, project_id=None, project_group_name=None, state=None): if project_id: project_metadata = ObjectMetadata(id=project_id, name=project_name, full_info=dict()) else: diff --git a/py/src/braintrust/test_project_group.py b/py/src/braintrust/test_project_group.py new file mode 100644 index 000000000..168fec33c --- /dev/null +++ b/py/src/braintrust/test_project_group.py @@ -0,0 +1,189 @@ +"""Tests for the `project_group_name` option on the entrypoints that can create a project. + +These run against a real local HTTP server (`scripted_server`) rather than a mocked client, so the +assertions are made on the bytes the SDK actually puts on the wire. +""" + +import json +from contextlib import contextmanager +from unittest.mock import patch + +import braintrust +import pytest +from braintrust import logger +from braintrust.api._test_server import scripted_server +from braintrust.logger import BraintrustState, span_components_to_object_id +from braintrust.span_identifier_v4 import SpanComponentsV4 + + +PROJECT_ID = "00000000-0000-0000-0000-000000000001" + +# `test_helpers.init_test_logger` replaces this module-level function with a fake and never puts it +# back, so capture the real one at import time (before any test body has run) and restore it for +# each test here. These tests are about the requests the real implementation makes. +_REAL_COMPUTE_LOGGER_METADATA = logger._compute_logger_metadata + + +@pytest.fixture(autouse=True) +def use_real_compute_logger_metadata(): + with patch.object(logger, "_compute_logger_metadata", _REAL_COMPUTE_LOGGER_METADATA): + yield + + +@contextmanager +def braintrust_server(): + """Serve the handful of endpoints these entrypoints touch, and yield a logged-in state.""" + + base_url = [] + + def handle(command, path, body, headers): + route = path.split("?")[0] + if route == "/api/apikey/login": + response = { + "org_info": [{"id": "org-id", "name": "org-name", "api_url": base_url[0], "proxy_url": None}] + } + elif route == "/v1/project": + response = {"id": PROJECT_ID, "name": "project"} + elif route == f"/v1/project/{PROJECT_ID}": + response = {"id": PROJECT_ID, "name": "project"} + elif route == "/v1/experiment": + response = {"id": "experiment-id", "name": "experiment", "project_id": PROJECT_ID} + elif route == "/v1/dataset": + response = {"id": "dataset-id", "name": "dataset", "project_id": PROJECT_ID} + else: + response = {} + return (200, {"Content-Type": "application/json"}, json.dumps(response).encode()) + + with scripted_server(handle) as (url, handler): + base_url.append(url) + state = BraintrustState() + state.login(api_key="test-key", app_url=url) + try: + yield state, handler + finally: + state.flush() + + +def bodies_for(handler, path): + return [json.loads(body) for method, request_path, body, _ in handler.requests if request_path.split("?")[0] == path] + + +def test_init_logger_forwards_project_group_name(): + with braintrust_server() as (state, handler): + log = braintrust.init_logger( + project="project", project_group_name="my-group", set_current=False, state=state + ) + assert log.id == PROJECT_ID + + assert bodies_for(handler, "/v1/project") == [ + {"name": "project", "org_name": "org-name", "project_group_name": "my-group"} + ] + + +def test_init_logger_omits_project_group_name_when_unspecified(): + with braintrust_server() as (state, handler): + braintrust.init_logger(project="project", set_current=False, state=state).id + + assert bodies_for(handler, "/v1/project") == [{"name": "project", "org_name": "org-name"}] + + +def test_init_creates_the_project_in_the_group_then_registers_by_id(): + with braintrust_server() as (state, handler): + experiment = braintrust.init( + project="project", + experiment="experiment", + project_group_name="my-group", + set_current=False, + state=state, + ) + assert experiment.id == "experiment-id" + + assert bodies_for(handler, "/v1/project") == [ + {"name": "project", "org_name": "org-name", "project_group_name": "my-group"} + ] + (experiment_body,) = bodies_for(handler, "/v1/experiment") + assert experiment_body["project_id"] == PROJECT_ID + assert "project_name" not in experiment_body + + +def test_init_omits_project_group_name_when_unspecified(): + with braintrust_server() as (state, handler): + braintrust.init(project="project", experiment="experiment", set_current=False, state=state).id + + assert bodies_for(handler, "/v1/project") == [{"name": "project", "org_name": "org-name"}] + + +def test_init_dataset_creates_the_project_in_the_group_then_registers_by_id(): + with braintrust_server() as (state, handler): + dataset = braintrust.init_dataset( + project="project", name="dataset", project_group_name="my-group", state=state + ) + assert dataset.id == "dataset-id" + + assert bodies_for(handler, "/v1/project") == [ + {"name": "project", "org_name": "org-name", "project_group_name": "my-group"} + ] + (dataset_body,) = bodies_for(handler, "/v1/dataset") + assert dataset_body["project_id"] == PROJECT_ID + assert "project_name" not in dataset_body + + +def test_project_group_name_is_ignored_when_project_id_is_specified(): + with braintrust_server() as (state, handler): + braintrust.init_dataset( + project_id=PROJECT_ID, name="dataset", project_group_name="my-group", state=state + ).id + braintrust.init_logger( + project_id=PROJECT_ID, project_group_name="my-group", set_current=False, state=state + ).id + + assert bodies_for(handler, "/v1/project") == [] + assert bodies_for(handler, "/v1/dataset")[0]["project_id"] == PROJECT_ID + + +def test_eval_creates_the_project_in_the_group_before_the_experiment(): + with braintrust_server() as (state, handler): + result = braintrust.Eval( + "project", + data=[{"input": 1, "expected": 2}], + task=lambda input: input * 2, + scores=[], + project_group_name="my-group", + state=state, + ) + assert result.summary.experiment_id == "experiment-id" + + assert bodies_for(handler, "/v1/project") == [ + {"name": "project", "org_name": "org-name", "project_group_name": "my-group"} + ] + assert bodies_for(handler, "/v1/experiment")[0]["project_id"] == PROJECT_ID + + +def test_exported_span_components_carry_the_project_group_for_lazy_resolution(): + with braintrust_server() as (state, handler): + log = braintrust.init_logger( + project="project", project_group_name="my-group", set_current=False, state=state + ) + + # The logger has not resolved its id yet, so `export()` defers project resolution (and + # creation) to whoever consumes the exported components. + components = SpanComponentsV4.from_str(log.export()) + assert components.compute_object_metadata_args == { + "project_name": "project", + "project_id": None, + "project_group_name": "my-group", + } + + with patch.object(logger, "_state", state): + assert span_components_to_object_id(components) == PROJECT_ID + + assert bodies_for(handler, "/v1/project") == [ + {"name": "project", "org_name": "org-name", "project_group_name": "my-group"} + ] + + +def test_exported_span_components_omit_the_project_group_when_unspecified(): + with braintrust_server() as (state, _handler): + log = braintrust.init_logger(project="project", set_current=False, state=state) + components = SpanComponentsV4.from_str(log.export()) + assert components.compute_object_metadata_args == {"project_name": "project", "project_id": None} From ea4727238056e12518fad497afb5f4812d030277 Mon Sep 17 00:00:00 2001 From: lforst <8118419+lforst@users.noreply.github.com> Date: Thu, 1 Oct 2026 16:03:52 +0000 Subject: [PATCH 2/2] Update PR #835 --- py/src/braintrust/cli/eval.py | 1 - py/src/braintrust/framework.py | 16 -- py/src/braintrust/framework2.py | 23 ++- py/src/braintrust/logger.py | 42 +----- py/src/braintrust/test_framework2.py | 32 ++++ py/src/braintrust/test_helpers.py | 2 +- py/src/braintrust/test_project_group.py | 189 ------------------------ 7 files changed, 57 insertions(+), 248 deletions(-) delete mode 100644 py/src/braintrust/test_project_group.py diff --git a/py/src/braintrust/cli/eval.py b/py/src/braintrust/cli/eval.py index 1fe950d18..f0e5dc890 100644 --- a/py/src/braintrust/cli/eval.py +++ b/py/src/braintrust/cli/eval.py @@ -143,7 +143,6 @@ async def run_evaluator_task(evaluator, position, opts: EvaluatorOpts): experiment = init_experiment( project_name=evaluator.project_name, project_id=evaluator.project_id, - project_group_name=evaluator.project_group_name, experiment_name=evaluator.experiment_name, description=evaluator.description, metadata=evaluator.metadata, diff --git a/py/src/braintrust/framework.py b/py/src/braintrust/framework.py index 3a9cb365d..b352252e9 100644 --- a/py/src/braintrust/framework.py +++ b/py/src/braintrust/framework.py @@ -462,13 +462,6 @@ class Evaluator(Generic[Input, Output, Expected]): parameter_values: dict[str, Any] | None = None - project_group_name: str | None = None - """ - If specified, creates the project inside the project group with this name when the project does - not already exist. Requires permission to create projects in that group. Ignored if `project_id` - is specified. - """ - @dataclasses.dataclass class EvalResultWithSummary(SerializableDataClass, Generic[Input, Output, Expected]): @@ -711,7 +704,6 @@ def _EvalCommon( parent: str | None = None, state: BraintrustState | None = None, enable_cache: bool = True, - project_group_name: str | None = None, ) -> Callable[[], Coroutine[Any, Any, EvalResultWithSummary[Input, Output, Expected]]]: """ This helper is needed because in case of `_lazy_load`, we need to update @@ -751,7 +743,6 @@ def _EvalCommon( description=description, summarize_scores=summarize_scores, parameters=parameters, - project_group_name=project_group_name, ) if _lazy_load: @@ -790,7 +781,6 @@ async def make_empty_summary(): experiment = init_experiment( project_name=evaluator.project_name if evaluator.project_id is None else None, project_id=evaluator.project_id, - project_group_name=evaluator.project_group_name, experiment_name=evaluator.experiment_name, description=evaluator.description, metadata=evaluator.metadata, @@ -855,7 +845,6 @@ async def EvalAsync( parent: str | None = None, state: BraintrustState | None = None, enable_cache: bool = True, - project_group_name: str | None = None, ) -> EvalResultWithSummary[Input, Output, Expected]: """ A function you can use to define an evaluator. This is a convenience wrapper around the `Evaluator` class. @@ -896,7 +885,6 @@ async def EvalAsync( :param timeout: (Optional) The duration, in seconds, after which to time out the evaluation. Defaults to None, in which case there is no timeout. :param project_id: (Optional) If specified, uses the given project ID instead of the evaluator's name to identify the project. - :param project_group_name: (Optional) Creates the project inside the project group with this name, if the project does not already exist. Requires permission to create projects in that group. Ignored if `project_id` is specified. :param base_experiment_name: An optional experiment name to use as a base. If specified, the new experiment will be summarized and compared to this experiment. :param base_experiment_id: An optional experiment id to use as a base. If specified, the new experiment will be @@ -949,7 +937,6 @@ async def EvalAsync( parent=parent, state=state, enable_cache=enable_cache, - project_group_name=project_group_name, ) return await f() @@ -987,7 +974,6 @@ def Eval( parent: str | None = None, state: BraintrustState | None = None, enable_cache: bool = True, - project_group_name: str | None = None, ) -> EvalResultWithSummary[Input, Output, Expected]: """ A function you can use to define an evaluator. This is a convenience wrapper around the `Evaluator` class. @@ -1028,7 +1014,6 @@ def Eval( :param timeout: (Optional) The duration, in seconds, after which to time out the evaluation. Defaults to None, in which case there is no timeout. :param project_id: (Optional) If specified, uses the given project ID instead of the evaluator's name to identify the project. - :param project_group_name: (Optional) Creates the project inside the project group with this name, if the project does not already exist. Requires permission to create projects in that group. Ignored if `project_id` is specified. :param base_experiment_name: An optional experiment name to use as a base. If specified, the new experiment will be summarized and compared to this experiment. :param base_experiment_id: An optional experiment id to use as a base. If specified, the new experiment will be @@ -1083,7 +1068,6 @@ def Eval( parent=parent, state=state, enable_cache=enable_cache, - project_group_name=project_group_name, ) # https://stackoverflow.com/questions/55409641/asyncio-run-cannot-be-called-from-a-running-event-loop-when-using-jupyter-no try: diff --git a/py/src/braintrust/framework2.py b/py/src/braintrust/framework2.py index 4bda4dbb5..51fb23630 100644 --- a/py/src/braintrust/framework2.py +++ b/py/src/braintrust/framework2.py @@ -1,7 +1,7 @@ import dataclasses import json from collections.abc import Callable, Mapping, Sequence -from typing import Any, overload +from typing import Any, cast, overload import slugify from braintrust.logger import _internal_get_global_state, api_conn, login @@ -26,16 +26,19 @@ def __init__(self): self._cache: dict[Project, str] = {} self._name_cache: dict[str, str] = {} - def get_by_name(self, project_name: str) -> str: + def get_by_name(self, project_name: str, project_group_name: str | None = None) -> str: if project_name not in self._name_cache: state = _internal_get_global_state() - project = state.api_client().projects.post_project(body={"name": project_name, "org_name": state.org_name}) + body: dict[str, Any] = {"name": project_name, "org_name": state.org_name} + if project_group_name is not None: + body["project_group_name"] = project_group_name + project = state.api_client().projects.post_project(body=cast(Any, body)) self._name_cache[project_name] = project["id"] return self._name_cache[project_name] def get(self, project: "Project") -> str: if project not in self._cache: - self._cache[project] = self.get_by_name(project.name) + self._cache[project] = self.get_by_name(project.name, project.project_group_name) return self._cache[project] @@ -607,8 +610,9 @@ def create( class Project: """A handle to a Braintrust project.""" - def __init__(self, name: str): + def __init__(self, name: str, project_group_name: str | None = None): self.name = name + self.project_group_name = project_group_name self.tools = ToolBuilder(self) self.prompts = PromptBuilder(self) self.parameters = ParametersBuilder(self) @@ -659,8 +663,13 @@ def publish(self): class ProjectBuilder: """Creates handles to Braintrust projects.""" - def create(self, name: str) -> Project: - return Project(name) + def create(self, name: str, project_group_name: str | None = None) -> Project: + """Create a handle to a Braintrust project. + + :param name: The name of the project. + :param project_group_name: (Optional) If specified, creates the project inside the project group with this name when the project does not already exist. Requires permission to create projects in that group. + """ + return Project(name, project_group_name=project_group_name) projects = ProjectBuilder() diff --git a/py/src/braintrust/logger.py b/py/src/braintrust/logger.py index 8ad612834..2e7a7adf4 100644 --- a/py/src/braintrust/logger.py +++ b/py/src/braintrust/logger.py @@ -1545,23 +1545,6 @@ class OrgProjectMetadata: project: ObjectMetadata -def _register_project( - state: "BraintrustState", - name: str | None, - project_group_name: str | None = None, -) -> Mapping[str, Any]: - """Register (get-or-create) a project by name, optionally inside a project group. - - `project_group_name` is omitted from the request body entirely when unset, so callers that do - not use project groups send exactly the same payload as before. - """ - - body: dict[str, Any] = {"name": name, "org_name": state.org_name} - if project_group_name is not None: - body["project_group_name"] = project_group_name - return state.api_client().projects.post_project(body=cast(Any, body)) - - # Pyright produces an error for overlapping overloads # (reportOverlappingOverload) because of the default argument to `open`. It # thinks a call like `init()` with no arguments could match both overloads. @@ -1590,7 +1573,6 @@ def init( base_experiment_id: str | None = ..., repo_info: RepoInfo | None = ..., state: BraintrustState | None = ..., - project_group_name: str | None = ..., ) -> "Experiment": ... @@ -1616,7 +1598,6 @@ def init( base_experiment_id: str | None = ..., repo_info: RepoInfo | None = ..., state: BraintrustState | None = ..., - project_group_name: str | None = ..., ) -> "ReadonlyExperiment": ... @@ -1641,7 +1622,6 @@ def init( base_experiment_id: str | None = None, repo_info: RepoInfo | None = None, state: BraintrustState | None = None, - project_group_name: str | None = None, ) -> "Experiment | ReadonlyExperiment": """ Log in, and then initialize a new experiment in a specified project. If the project does not exist, it will be created. @@ -1666,7 +1646,6 @@ def init( :param set_current: If true (the default), set the global current-experiment to the newly-created one. :param open: If the experiment already exists, open it in read-only mode. Throws an error if the experiment does not already exist. :param project_id: The id of the project to create the experiment in. This takes precedence over `project` if specified. - :param project_group_name: (Optional) Create the project inside the project group with this name, if the project does not already exist. Requires permission to create projects in that group. Ignored if `project_id` is specified. :param base_experiment_id: An optional experiment id to use as a base. If specified, the new experiment will be summarized and compared to this. This takes precedence over `base_experiment` if specified. :param repo_info: (Optional) Explicitly specify the git metadata for this experiment. This takes precedence over `git_metadata_settings` if specified. :param state: (Optional) A BraintrustState object to use. If not specified, will use the global state. This is for advanced use only. @@ -1717,7 +1696,9 @@ def compute_metadata(): def compute_metadata(): state.login(org_name=org_name, api_key=api_key, app_url=app_url) if project_id is None: - project_info = _register_project(state, project or GLOBAL_PROJECT, project_group_name=project_group_name) + project_info = state.api_client().projects.post_project( + body={"name": project or GLOBAL_PROJECT, "org_name": state.org_name} + ) else: project_info = state.api_client().projects.get_project_id(project_id) @@ -1831,7 +1812,6 @@ def init_dataset( state: BraintrustState | None = None, environment: str | None = None, dataset_id: str | None = None, - project_group_name: str | None = None, ) -> "Dataset": """ Create or load a dataset. When creating a dataset, its project will be created if it does not exist. @@ -1847,7 +1827,6 @@ def init_dataset( key is specified, will prompt the user to login. :param org_name: (Optional) The name of a specific organization to connect to. This is useful if you belong to multiple. :param project_id: The id of the project to create the dataset in. This takes precedence over `project` if specified. - :param project_group_name: (Optional) Create the project inside the project group with this name, if the project does not already exist. Requires permission to create projects in that group. Ignored if `project_id` is specified. :param metadata: (Optional) a dictionary, or an object that serializes to a dictionary (such as a Pydantic model), with additional data about the dataset. The values in `metadata` can be any JSON-serializable type, but its keys must be strings. :param use_output: (Deprecated) If True, records will be fetched from this dataset in the legacy format, with the "expected" field renamed to "output". This option will be removed in a future version of Braintrust. @@ -1875,7 +1854,7 @@ def compute_metadata(): if project_id is not None: resp_project = api_client.projects.get_project_id(project_id) else: - resp_project = _register_project(state, project, project_group_name=project_group_name) + resp_project = api_client.projects.post_project(body={"name": project, "org_name": state.org_name}) body = _populate_args( {"project_id": resp_project["id"], "name": name or "logs"}, description=description, @@ -1900,14 +1879,15 @@ def compute_metadata(): def _compute_logger_metadata( project_name: str | None = None, project_id: str | None = None, - project_group_name: str | None = None, state: BraintrustState | None = None, ): state = state or _state state.login() org_id = state.org_id if project_id is None: - response = _register_project(state, project_name or GLOBAL_PROJECT, project_group_name=project_group_name) + response = state.api_client().projects.post_project( + body={"name": project_name or GLOBAL_PROJECT, "org_name": state.org_name} + ) return OrgProjectMetadata( org_id=org_id, project=ObjectMetadata(id=response["id"], name=response["name"], full_info=dict(response)), @@ -1935,7 +1915,6 @@ def init_logger( set_current: bool = True, state: BraintrustState | None = None, environment: SpanOriginEnvironment | None = None, - project_group_name: str | None = None, ) -> "Logger": """ Create a new logger in a specified project. If the project does not exist, it will be created. @@ -1949,17 +1928,12 @@ def init_logger( :param org_name: (Optional) The name of a specific organization to connect to. This is useful if you belong to multiple. :param force_login: Login again, even if you have already logged in (by default, the logger will not login if you are already logged in) :param set_current: If true (the default), set the global current-experiment to the newly-created one. - :param project_group_name: (Optional) Create the project inside the project group with this name, if the project does not already exist. Requires permission to create projects in that group. Ignored if `project_id` is specified. :returns: The newly created Logger. """ state = state or _state state.span_origin_environment = detect_environment(environment) - compute_metadata_args: dict[str, Any] = dict(project_name=project, project_id=project_id) - if project_group_name is not None: - # Only present when set, so exported span components (and therefore anything that resolves - # them with an older SDK) are byte-for-byte unchanged for callers not using project groups. - compute_metadata_args["project_group_name"] = project_group_name + compute_metadata_args = dict(project_name=project, project_id=project_id) link_args = { "app_url": app_url, diff --git a/py/src/braintrust/test_framework2.py b/py/src/braintrust/test_framework2.py index c49b6d796..aa7079da9 100644 --- a/py/src/braintrust/test_framework2.py +++ b/py/src/braintrust/test_framework2.py @@ -30,6 +30,38 @@ def test_project_id_cache_uses_generated_project_registration(): mock_state.app_conn.assert_not_called() +def test_project_id_cache_creates_the_project_in_its_project_group(): + mock_state = MagicMock() + mock_state.org_name = "test-org" + mock_state.api_client.return_value.projects.post_project.return_value = { + "id": "generated-project-id", + "name": "test-project", + } + project = projects.create("test-project", project_group_name="my-group") + with patch("braintrust.logger._state", mock_state): + project_id = ProjectIdCache().get(project) + + assert project_id == "generated-project-id" + mock_state.api_client.return_value.projects.post_project.assert_called_once_with( + body={"name": "test-project", "org_name": "test-org", "project_group_name": "my-group"} + ) + + +def test_project_id_cache_omits_project_group_name_when_unspecified(): + mock_state = MagicMock() + mock_state.org_name = "test-org" + mock_state.api_client.return_value.projects.post_project.return_value = { + "id": "generated-project-id", + "name": "test-project", + } + with patch("braintrust.logger._state", mock_state): + ProjectIdCache().get(projects.create("test-project")) + + mock_state.api_client.return_value.projects.post_project.assert_called_once_with( + body={"name": "test-project", "org_name": "test-org"} + ) + + class TestCodeFunctionMetadata: """Tests for CodeFunction metadata support.""" diff --git a/py/src/braintrust/test_helpers.py b/py/src/braintrust/test_helpers.py index 5d1ba3adb..585454bc3 100644 --- a/py/src/braintrust/test_helpers.py +++ b/py/src/braintrust/test_helpers.py @@ -104,7 +104,7 @@ def init_test_logger(project_name: str): l._lazy_metadata = lazy_metadata # Skip actual login by setting fake metadata directly # Replace the global _compute_logger_metadata function with a resolved LazyValue - def fake_compute_logger_metadata(project_name=None, project_id=None, project_group_name=None, state=None): + def fake_compute_logger_metadata(project_name=None, project_id=None, state=None): if project_id: project_metadata = ObjectMetadata(id=project_id, name=project_name, full_info=dict()) else: diff --git a/py/src/braintrust/test_project_group.py b/py/src/braintrust/test_project_group.py deleted file mode 100644 index 168fec33c..000000000 --- a/py/src/braintrust/test_project_group.py +++ /dev/null @@ -1,189 +0,0 @@ -"""Tests for the `project_group_name` option on the entrypoints that can create a project. - -These run against a real local HTTP server (`scripted_server`) rather than a mocked client, so the -assertions are made on the bytes the SDK actually puts on the wire. -""" - -import json -from contextlib import contextmanager -from unittest.mock import patch - -import braintrust -import pytest -from braintrust import logger -from braintrust.api._test_server import scripted_server -from braintrust.logger import BraintrustState, span_components_to_object_id -from braintrust.span_identifier_v4 import SpanComponentsV4 - - -PROJECT_ID = "00000000-0000-0000-0000-000000000001" - -# `test_helpers.init_test_logger` replaces this module-level function with a fake and never puts it -# back, so capture the real one at import time (before any test body has run) and restore it for -# each test here. These tests are about the requests the real implementation makes. -_REAL_COMPUTE_LOGGER_METADATA = logger._compute_logger_metadata - - -@pytest.fixture(autouse=True) -def use_real_compute_logger_metadata(): - with patch.object(logger, "_compute_logger_metadata", _REAL_COMPUTE_LOGGER_METADATA): - yield - - -@contextmanager -def braintrust_server(): - """Serve the handful of endpoints these entrypoints touch, and yield a logged-in state.""" - - base_url = [] - - def handle(command, path, body, headers): - route = path.split("?")[0] - if route == "/api/apikey/login": - response = { - "org_info": [{"id": "org-id", "name": "org-name", "api_url": base_url[0], "proxy_url": None}] - } - elif route == "/v1/project": - response = {"id": PROJECT_ID, "name": "project"} - elif route == f"/v1/project/{PROJECT_ID}": - response = {"id": PROJECT_ID, "name": "project"} - elif route == "/v1/experiment": - response = {"id": "experiment-id", "name": "experiment", "project_id": PROJECT_ID} - elif route == "/v1/dataset": - response = {"id": "dataset-id", "name": "dataset", "project_id": PROJECT_ID} - else: - response = {} - return (200, {"Content-Type": "application/json"}, json.dumps(response).encode()) - - with scripted_server(handle) as (url, handler): - base_url.append(url) - state = BraintrustState() - state.login(api_key="test-key", app_url=url) - try: - yield state, handler - finally: - state.flush() - - -def bodies_for(handler, path): - return [json.loads(body) for method, request_path, body, _ in handler.requests if request_path.split("?")[0] == path] - - -def test_init_logger_forwards_project_group_name(): - with braintrust_server() as (state, handler): - log = braintrust.init_logger( - project="project", project_group_name="my-group", set_current=False, state=state - ) - assert log.id == PROJECT_ID - - assert bodies_for(handler, "/v1/project") == [ - {"name": "project", "org_name": "org-name", "project_group_name": "my-group"} - ] - - -def test_init_logger_omits_project_group_name_when_unspecified(): - with braintrust_server() as (state, handler): - braintrust.init_logger(project="project", set_current=False, state=state).id - - assert bodies_for(handler, "/v1/project") == [{"name": "project", "org_name": "org-name"}] - - -def test_init_creates_the_project_in_the_group_then_registers_by_id(): - with braintrust_server() as (state, handler): - experiment = braintrust.init( - project="project", - experiment="experiment", - project_group_name="my-group", - set_current=False, - state=state, - ) - assert experiment.id == "experiment-id" - - assert bodies_for(handler, "/v1/project") == [ - {"name": "project", "org_name": "org-name", "project_group_name": "my-group"} - ] - (experiment_body,) = bodies_for(handler, "/v1/experiment") - assert experiment_body["project_id"] == PROJECT_ID - assert "project_name" not in experiment_body - - -def test_init_omits_project_group_name_when_unspecified(): - with braintrust_server() as (state, handler): - braintrust.init(project="project", experiment="experiment", set_current=False, state=state).id - - assert bodies_for(handler, "/v1/project") == [{"name": "project", "org_name": "org-name"}] - - -def test_init_dataset_creates_the_project_in_the_group_then_registers_by_id(): - with braintrust_server() as (state, handler): - dataset = braintrust.init_dataset( - project="project", name="dataset", project_group_name="my-group", state=state - ) - assert dataset.id == "dataset-id" - - assert bodies_for(handler, "/v1/project") == [ - {"name": "project", "org_name": "org-name", "project_group_name": "my-group"} - ] - (dataset_body,) = bodies_for(handler, "/v1/dataset") - assert dataset_body["project_id"] == PROJECT_ID - assert "project_name" not in dataset_body - - -def test_project_group_name_is_ignored_when_project_id_is_specified(): - with braintrust_server() as (state, handler): - braintrust.init_dataset( - project_id=PROJECT_ID, name="dataset", project_group_name="my-group", state=state - ).id - braintrust.init_logger( - project_id=PROJECT_ID, project_group_name="my-group", set_current=False, state=state - ).id - - assert bodies_for(handler, "/v1/project") == [] - assert bodies_for(handler, "/v1/dataset")[0]["project_id"] == PROJECT_ID - - -def test_eval_creates_the_project_in_the_group_before_the_experiment(): - with braintrust_server() as (state, handler): - result = braintrust.Eval( - "project", - data=[{"input": 1, "expected": 2}], - task=lambda input: input * 2, - scores=[], - project_group_name="my-group", - state=state, - ) - assert result.summary.experiment_id == "experiment-id" - - assert bodies_for(handler, "/v1/project") == [ - {"name": "project", "org_name": "org-name", "project_group_name": "my-group"} - ] - assert bodies_for(handler, "/v1/experiment")[0]["project_id"] == PROJECT_ID - - -def test_exported_span_components_carry_the_project_group_for_lazy_resolution(): - with braintrust_server() as (state, handler): - log = braintrust.init_logger( - project="project", project_group_name="my-group", set_current=False, state=state - ) - - # The logger has not resolved its id yet, so `export()` defers project resolution (and - # creation) to whoever consumes the exported components. - components = SpanComponentsV4.from_str(log.export()) - assert components.compute_object_metadata_args == { - "project_name": "project", - "project_id": None, - "project_group_name": "my-group", - } - - with patch.object(logger, "_state", state): - assert span_components_to_object_id(components) == PROJECT_ID - - assert bodies_for(handler, "/v1/project") == [ - {"name": "project", "org_name": "org-name", "project_group_name": "my-group"} - ] - - -def test_exported_span_components_omit_the_project_group_when_unspecified(): - with braintrust_server() as (state, _handler): - log = braintrust.init_logger(project="project", set_current=False, state=state) - components = SpanComponentsV4.from_str(log.export()) - assert components.compute_object_metadata_args == {"project_name": "project", "project_id": None}