diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index e9a3d95..40e198a 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -4,6 +4,21 @@ on: push: tags: - 'rel-*' + # Manual dispatch is a rehearsal by default: it assembles the tarball end to end, including + # downloading the three sibling JARs, but stops before publishing. Publishing is gated on + # dry_run, and the tag push path is unaffected. Without this the workflow -- and in + # particular the release-notes wiring below -- could only be exercised by cutting a real + # release. Follows the pattern in dp-desktop-app. + workflow_dispatch: + inputs: + version: + description: "Version to build, without the rel- prefix (e.g. 1.16.0). Must exist as rel- in dp-grpc, dp-service, and dp-desktop-app." + required: true + dry_run: + description: "Build only; do not publish a GitHub Release" + required: false + type: boolean + default: true env: DP_SERVICE_REPO: osprey-dcs/dp-service @@ -13,6 +28,12 @@ env: jobs: build-release: runs-on: ubuntu-latest + + env: + # A dry run assembles and verifies the tarball but never publishes a release. A rel-* + # tag push always publishes; a manual dispatch defaults to a dry run and must opt in. + DRY_RUN: ${{ github.event_name == 'workflow_dispatch' && inputs.dry_run }} + steps: # Checkout data-platform repo @@ -23,10 +44,54 @@ jobs: # convention in CLAUDE.md. Dependabot (.github/dependabot.yml) keeps the pins current. uses: actions/checkout@a37ce9120846195fa4ece8f58b268e6043cb2f26 # v3.7.0 - # Extract version from tag + # On a tag push the version comes from the tag. On a manual dispatch GITHUB_REF_NAME is + # the branch, not a tag, so the version has to be supplied as an input -- it selects + # which rel- of the sibling repos to download JARs from. - name: Set VERSION env + shell: bash + run: | + set -euo pipefail + if [[ "${GITHUB_EVENT_NAME}" == "workflow_dispatch" ]]; then + VERSION="${{ inputs.version }}" + VERSION="${VERSION#rel-}" + else + TAG="${GITHUB_REF_NAME}" + if [[ "$TAG" != rel-* ]]; then + echo "::error::Tag does not start with 'rel-': $TAG" + exit 1 + fi + VERSION="${TAG#rel-}" + fi + echo "VERSION=${VERSION}" >> $GITHUB_ENV + echo "Building version ${VERSION}" + + # Checked up front rather than left to action-gh-release, which fails the job on a + # missing body_path only after the sibling-repo JARs have been downloaded and the + # tarball uploaded. Every release from 1.16.0 on ships notes; see + # doc/developer/release.md. + # + # The path is derived from VERSION rather than GITHUB_REF_NAME because on a manual + # dispatch GITHUB_REF_NAME is the branch, so it would look for doc/release-notes/main.md. + # On a tag push the two are identical. + - name: Verify release notes exist + shell: bash run: | - echo "VERSION=${GITHUB_REF_NAME#rel-}" >> $GITHUB_ENV + set -euo pipefail + NOTES="doc/release-notes/rel-${VERSION}.md" + if [ ! -f "$NOTES" ]; then + # A dry run rehearses the build before the notes are written, so it warns rather + # than failing; the publishing path always fails. DRY_RUN is a job-level env and + # so always exported -- as the empty string on a tag push, which set -u would + # otherwise trip on. + if [[ "${DRY_RUN:-}" == "true" ]]; then + echo "::warning::Release notes not found at $NOTES -- a real run would fail here." + exit 0 + fi + echo "::error::Release notes not found at $NOTES." + echo "::error::Add the notes for ${VERSION} and re-run, or retag once they are on the tagged commit." + exit 1 + fi + echo "NOTES_PATH=$NOTES" >> $GITHUB_ENV # Create directory structure - name: Create installer directories @@ -122,12 +187,17 @@ jobs: sha256sum data-platform-${{ env.VERSION }}.tar.gz \ > data-platform-${{ env.VERSION }}.tar.gz.sha256 + # Everything above this point -- downloading the sibling JARs, assembling the tarball, + # checksumming it, and verifying the notes -- runs in a dry run too, so a rehearsal + # exercises the whole job except the one step that writes. - name: Publish GitHub Release + if: env.DRY_RUN != 'true' uses: softprops/action-gh-release@3bb12739c298aeb8a4eeaf626c5b8d85266b0e65 # v2.6.2 with: files: | data-platform-${{ env.VERSION }}.tar.gz data-platform-${{ env.VERSION }}.tar.gz.sha256 + body_path: ${{ env.NOTES_PATH }} overwrite_files: true env: GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} diff --git a/README.md b/README.md index 4c62644..68d4345 100644 --- a/README.md +++ b/README.md @@ -9,7 +9,7 @@ This document includes the following details: - [Requirements and objectives](#requirements-and-objectives) - [MLDP Project elements](#data-platform-project-elements) - [Status and milestones](#status-and-milestones) -- [Todo and road map](#todo-and-road-map) +- [Todo and road map](#mldp-todo-and-road-map) - [Additional documentation](#additional-documentation) - [Installation and getting started](#installation-and-getting-started) @@ -233,11 +233,11 @@ Performance benchmark applications were developed and utilized to evaluate candi ### Data Platform v1.0 (November 2023) -Version 1.0 of the Data Platform includes an initial Java implementation of the Ingestion Service providing a gRPC API and using MongoDB for storing time-series data. The initial ingestion service implementation focuses only on scalar data and with timestamps specified using the "sampling clock" mechanism with start time and sample period. It is accompanied by a performance benchmark application that is used at each stage of development to measure ingestion performance relative to the project goal. The initial implementation exceeds our goal by a comfortable margin, but this will continue to be a focus as the project evolves. [section "Data Platform API"](#data-platform-api) provides more information about the ingestion API. +Version 1.0 of the Data Platform includes an initial Java implementation of the Ingestion Service providing a gRPC API and using MongoDB for storing time-series data. The initial ingestion service implementation focuses only on scalar data and with timestamps specified using the "sampling clock" mechanism with start time and sample period. It is accompanied by a performance benchmark application that is used at each stage of development to measure ingestion performance relative to the project goal. The initial implementation exceeds our goal by a comfortable margin, but this will continue to be a focus as the project evolves. [section "gRPC API"](#grpc-api) provides more information about the ingestion API. ### v1.1 (January 2024) -Version 1.1 includes a Java implementation of the Query Service gRPC API, using the MongoDB database managed by the ingestion service to fulfill client query requests. A variety of API RPC methods for querying time-series data are provided to support the development of clients with varying performance requirements, ranging from streaming methods that return bucketed result data down to simple single response methods that return tabular data. See [section "Data Platform API"](#data-platform-api) for a detailed description of the query API. +Version 1.1 includes a Java implementation of the Query Service gRPC API, using the MongoDB database managed by the ingestion service to fulfill client query requests. A variety of API RPC methods for querying time-series data are provided to support the development of clients with varying performance requirements, ranging from streaming methods that return bucketed result data down to simple single response methods that return tabular data. See [section "gRPC API"](#grpc-api) for a detailed description of the query API. ### v1.2 (February 2024) @@ -245,7 +245,7 @@ Version 1.2 saw changes to the "proto" files defining the gRPC API for the Data ### v1.3 (April 2024) -Version 1.3 provides an initial implementation of the annotation service for adding annotations to archived data and performing queries against those annotations. The primary focus for the initial annotation service implementation was on the data model for associating annotations with data in the archive. The only type of annotation currently supported is a simple user comment, but we will be adding many other types of annotations using the same underlying data model. See [section "Data Platform API"](#data-platform-api) for more details about the annotation data model. +Version 1.3 provides an initial implementation of the annotation service for adding annotations to archived data and performing queries against those annotations. The primary focus for the initial annotation service implementation was on the data model for associating annotations with data in the archive. The only type of annotation currently supported is a simple user comment, but we will be adding many other types of annotations using the same underlying data model. See [section "gRPC API"](#grpc-api) for more details about the annotation data model. ### v1.4 (July 2024) @@ -291,23 +291,24 @@ The main focus of v1.13 is a new set of column-oriented data structures in the p Version 1.14 includes three main new features. First, support for column-level metadata in the Ingestion Service: an optional `ColumnMetadata` field (carrying provenance, tags, and key/value attributes) has been added to all 16 column types in the gRPC API. When present, the metadata is persisted in MongoDB alongside the column data and is restored on query, so the retrieved column equals the original ingested column. A validation layer enforces limits on field lengths and collection sizes. Second, a new PV Metadata API is added to the Annotation Service for creating, querying, retrieving, and deleting `PvMetadata` records that describe the properties of archived PVs. As part of this work, the Query Service methods `queryPvMetadata()` and `queryProviderMetadata()` are renamed to `queryPvStats()` and `queryProviderStats()` to better reflect that they return archive ingestion statistics rather than user-defined metadata. Third, a new Machine Configuration API is added to the Annotation Service for managing named machine configuration records and time-bounded configuration activations. The API supports full CRUD operations for both `Configuration` and `ConfigurationActivation` records, enforces non-overlapping activation intervals per configuration and category, and provides a `getActiveConfigurations()` method for retrieving all configurations active at a given point in time. -### v1.15 (Auguest 2026) +### v1.15 (August 2026) The two primary features included in version 1.15 are 1) a new "v2 query API", and 2) the initial release of the Python client API library. The v2 query API provides an interface that leverages archive metadata for filtering queries by time, PV metadata, temporal machine configuration, and sample status, a major update to the initial query API that supported only a list of PV names and a time range as search criteria. The initial implementation of the Python client library includes low-level wrappers for calling the MLDP PV metadata, machine configuration, and v2 query APIs, and provides a foundation for building higher-level features and conveniences to support data science applications. The v1.15 release also includes a number of performance improvements and bug fixes. +### v1.16 (September 2026) + +Version 1.16 centers on two new API capabilities that span every repository in the ecosystem. First, a new Sample Status API assigns status codes to individual PV samples at specific timestamps, keyed by (PV name, timestamp, domain, layer), supporting automated data cleaning, quality assessment, and MLOps workflows; query methods can filter samples by status, and the API replaces the `DataValue.ValueStatus` field removed in this release, which was never queryable. Second, the DataSet and Annotation APIs — the oldest generation of the Annotation Service — are modernized to the CRUD conventions established by the PV metadata, machine configuration, and sample status APIs, gaining single-record get and delete methods, paging, audit fields, typed calculations columns, and column-level provenance. The release also includes substantial query performance work driven by SLAC deployment reports: the hours-long startup bucket scan is removed, query index bounds are now maintained per PV rather than sized to the longest bucket in the archive, and every bucket query is pinned to the compound index and bounded on both sides. A new metrics framework exports request rates, latency histograms, and per-stage query breakdowns from every service over a Prometheus endpoint, with a slow query log for diagnosing individual queries. The desktop application gains a deployment mode for running against remote services rather than only in-process demonstration services, along with new views for authoring and exploring curated metadata. This release is delivered through a new schema migration mechanism that migrates the database at first startup — see the [release notes](doc/release-notes/rel-1.16.0.md) for the upgrade procedure. + --- ## MLDP TODO and Road Map -#### v1.16 Development Plans - * Sample Status API for marking the disposition of individual sample values for automated data cleaning tools (e.g, "suspect value"). - * v2 modernization of the original Annotation API to follow conventions established for the new PV metadata and machine configuration APIs - * Python client library - * sample status API interface - * v2 annotation API interface +#### v1.17 Development Plans + * Python client library * ingestion API interface * bucket-oriented query interface - * improved metrics capture and observability + * tags / attribute usage API: new API and annotation service handling + * data subscription enhancements to support multiple ingestion servers #### FY27 Development Priorities * ingestion API and database schema improvements for handling timestamps with jitter (using shared timestamps between data buckets) @@ -315,7 +316,6 @@ The two primary features included in version 1.15 are 1) a new "v2 query API", a * age-based archival of data to external storage while maintaining Mongo indexes * enhance query API to support query by PV data value * Python client library enhancements for building ML applications - * Modify data subscription mechanism to support multiple ingestion service instances (e.g., REDIS registry). * support for large atomic data values that exceed the 16MB Mongo object limit * tools for ingesting data from EPICS environment @@ -362,3 +362,19 @@ current, since pinning otherwise trades supply-chain risk for silent staleness. If you are adding or editing a workflow, the full convention — how to resolve a SHA correctly, how to verify one, and why pinning is deliberately kept separate from upgrading — is in [CLAUDE.md](CLAUDE.md#github-actions-pin-every-uses-to-a-commit-sha). + +## release notes + +Per-release notes live under [`doc/release-notes/`](doc/release-notes/), one document per +release, covering what changed across the whole ecosystem since the previous release and what +upgrading requires. These are the master notes: the releases in the `dp-grpc`, `dp-service`, +`dp-desktop-app`, and `dp-python-lib` repos point back here. + +| Release | Notes | +|---|---| +| 1.16.0 | [rel-1.16.0](doc/release-notes/rel-1.16.0.md) | + +Releases before 1.16.0 were documented on the +[GitHub release](https://github.com/osprey-dcs/data-platform/releases) itself, and summarized in +[status and milestones](#status-and-milestones) above. The `rel-*` tags remain the authority on +what any past release contained. diff --git a/doc/developer/release.md b/doc/developer/release.md index 90b57e0..f8d0958 100644 --- a/doc/developer/release.md +++ b/doc/developer/release.md @@ -2,55 +2,78 @@ This document contains notes to help streamline the process of creating a Data Platform release. +Previously, we used scripts to add tags to reach repo and build the release artifacts. -## complete development work +Now, this is largely accomplished by GHA workflows that are triggered when new release tags are added. So in a nutshell, assuming the development work and documentation is complete for each repo and merged to the main branch using pull requests, take the following general steps in each repo: -Make sure pom.xml for dp-grpc, dp-service reflect current version number! -### data-platform -- add/modify scripts as needed +## write the release notes first +Starting with 1.16.0, every release ships notes, and they must be **merged to `main` before the +`rel-*` tag is pushed**. The tag is what the release workflow builds from, so notes that land +after the tag are not on the tagged commit and the workflow will not see them. -## update documentation -- dp-grpc - - update README.md API docs - - in-line comments in proto files -- dp-service - - update README.md - - update developer-notes.md and UML diagrams for any important new features / frameworks - - add comments to application.yml for any new config resources - - update java command line docs for running new applications etc - - update running.md with details about running applications and configuration -- data-platform - - create release notes - - update doc/install/quick-start.md, installation.md - - update README.md - - update README.md with new scripts etc - - update release process (this doc) +The master notes for the whole ecosystem live in this repo at +`doc/release-notes/rel-.md` — for example `doc/release-notes/rel-1.16.0.md`. The +`release.yml` workflow reads that file and publishes it as the body of the GitHub release, via +`body_path`. A `Verify release notes exist` step runs immediately after the version is extracted +from the tag and fails the job with a clear error if the file is missing, so a forgotten document +fails within seconds of the tag push rather than after the sibling-repo JARs have been downloaded +and the tarball published. -Make sure this is all done before adding tags etc because it's a pain removing tags, deleting releases, adding tags, creating releases etc. +Each of `dp-grpc`, `dp-service`, `dp-desktop-app`, and `dp-python-lib` follows the same pattern +with its own per-repo notes, and their releases point back to the master notes here. +Organize the notes by issue ticket rather than by PR, since a ticket often spans several PRs. A +breaking release leads with an "Upgrading from " checklist that calls out silent +behavior changes separately from compile errors. Add each new document to the table in the +[release notes](../../README.md#release-notes) section of `README.md`. -## merge changes from fork's development branch to upstream's main branch +If the workflow does fail on a missing document, add the notes, then move the tag onto the new +commit and force-push it — the release job re-runs from the retagged commit: -### create pull requests to merge dev branch in fork to upstream - -If working in a fork, create a github pull request to merge the development branch in the fork to the development branch in the upstream repo. - -### create pull requests to merge dev branch in upstream to main branch in upstream - -Create a pull request to merge the dev branch on the upstream to the main branch on the upstream. - - -## create git tags for release in upstream/main - -Need to do this in the upstream for all repos: dp-grpc, dp-service, dp-desktop-app, data-platform. Creating a "rel-" tag in each repo will automatically run the release workflow to create a release with the same name as the tag and publish artifacts appropriate for the repo. The tag should be created manually to avoid a race condition between the repo workflows as they publish their releases: ``` -git tag rel-1.13.0 -git push origin rel-1.13.0 +git tag -f rel-1.16.0 +git push -f origin rel-1.16.0 ``` +## release process steps + +* update the release notes as described above, in e.g., doc/release-notes/rel-1.16.0, making sure to cover all the PRs / issues / features since the previous release + * the release.yml workflow in each repo assumes this file exists, and uses it for the body of the published release + * release notes must be merged to main before the release workflow runs +* make sure the README and other repo documents cover all the key issues / features since the previous release +* create the release tag and push, e.g., + * git tag rel-1.16.0 && git push origin rel-1.16.0 +* check the actions for the repo to make sure all workflows triggered by the new tag complete successfully +* check the published release, check the links within the release notes resolve + +### tag order matters: data-platform goes last + +Create the tags manually, one repo at a time, rather than scripting them together. Two ordering +constraints: + +* **`data-platform` must be tagged last, and only after the `dp-grpc`, `dp-service`, and + `dp-desktop-app` releases are actually published.** Its release job downloads + `dp-grpc-.jar`, `dp-service-.jar`, and `dp-desktop-app-.jar` from + those repos' releases at the matching `rel-` tag. Tagging it while a sibling's CI is still + running fails the download step. +* Tagging by hand also avoids a race between the sibling workflows as they publish their own + releases. + +The master release notes in this repo link to the child repos at `blob/rel-/...`, so +those links resolve only once each child tag exists — another reason this repo goes last. + +The artifacts created for the published release vary by repo: + +* dp-grpc + * CI creates regular and sha256 jar and tarball files +* dp-service + * CI creates regular and sha256 jar files +* dp-desktop-app + * CI creates regular and sha256 jar files +* dp-python-lib + * CI uses sigstore so creates wheel and tarball unsigned and signed with sigstore +* data-platform + * creates tarball with sha256 -## edit github release docs for each repo - -Need to do this in all repos: dp-grpc, dp-service, dp-desktop-app, data-platform. The master release notes should be added to the data-platform repo release. Minimally, edit the release notes for each of the other repos to point to the data-platform release notes. The release for each repo should already contain the artifacts appropriate for that repo (as generated by the release workflow). diff --git a/doc/release-notes/rel-1.16.0.md b/doc/release-notes/rel-1.16.0.md new file mode 100644 index 0000000..139da9c --- /dev/null +++ b/doc/release-notes/rel-1.16.0.md @@ -0,0 +1,445 @@ +# Data Platform 1.16.0 Release Notes + +Changes since rel-1.15.0. These are the master notes for the release: the MLDP ships as five +repositories tagged in lockstep, and this document describes what changed across all of them, +organized by feature. Each section links to the repository notes that carry the detail. + +| Repository | What it ships | 1.16.0 notes | +|---|---|---| +| [dp-grpc](https://github.com/osprey-dcs/dp-grpc) | gRPC API definition and generated Java stubs | [notes](https://github.com/osprey-dcs/dp-grpc/blob/rel-1.16.0/doc/release-notes/rel-1.16.0.md) | +| [dp-service](https://github.com/osprey-dcs/dp-service) | Java service implementations | [notes](https://github.com/osprey-dcs/dp-service/blob/rel-1.16.0/doc/release-notes/rel-1.16.0.md) | +| [dp-python-lib](https://github.com/osprey-dcs/dp-python-lib) | Python client library | [notes](https://github.com/osprey-dcs/dp-python-lib/blob/rel-1.16.0/doc/release-notes/rel-1.16.0.md) | +| [dp-desktop-app](https://github.com/osprey-dcs/dp-desktop-app) | JavaFX desktop GUI | [notes](https://github.com/osprey-dcs/dp-desktop-app/blob/rel-1.16.0/doc/release-notes/rel-1.16.0.md) | +| [data-platform](https://github.com/osprey-dcs/data-platform) | Installation tarball, deployment tooling, documentation | this document | + +**1.16.0 is a breaking release, and it is the first one that migrates your database.** The first +1.16.0 service to start against an existing archive changes stored data, with no downgrade path. +Read [Upgrading from 1.15.0](#upgrading-from-1150) before installing anything. + +## Contents + +- [Upgrading from 1.15.0](#upgrading-from-1150) +- [Sample Status API](#sample-status-api) +- [DataSet and Annotation API modernization](#dataset-and-annotation-api-modernization) +- [Query performance](#query-performance) +- [Service metrics and observability](#service-metrics-and-observability) +- [Schema migration mechanism](#schema-migration-mechanism) +- [Desktop application: deployment mode and metadata authoring](#desktop-application-deployment-mode-and-metadata-authoring) +- [Python client library](#python-client-library) +- [Correctness fixes worth knowing](#correctness-fixes-worth-knowing) +- [Release process and supply chain](#release-process-and-supply-chain) + +## Upgrading from 1.15.0 + +This upgrade is not a drop-in binary replacement. The database is migrated in place, each service +binds a second port, two query changes alter results without raising an error, and Python clients +that pin `grpcio` need a new floor. + +**Before you start:** + +1. **Free ports 9464–9467 on every service host**, or set the metrics-port variables. Each + service now binds a Prometheus endpoint and **refuses to start if it cannot**. On Kubernetes a + collision is `CrashLoopBackOff`, not a pod running without metrics. The bind interface is + `DP_TELEMETRY_PROMETHEUS_HOST` (default `0.0.0.0`; set `127.0.0.1` to expose metrics only to a + local scraper). If you cannot free a port and cannot move one, `DP_TELEMETRY_ENABLED=false` + binds no port at all. +2. **Take a restorable backup.** There are no downgrade migrations, and a 1.15.0 binary against a + migrated database *misreads* it rather than refusing: it sees every annotation's comment as + empty (v1 renamed the field) and repeats its hours-long startup bucket scan (v5 dropped the + marker that skipped it). Restoring the backup is the only way back. +3. **Stop every service** — all of them, on every host. A 1.15.0 *ingestion* process writing after + migration v5 has seeded `pvStats` produces buckets that queries can silently miss; a 1.15.0 + *query* or *annotation* process keeps serving wrong answers against the migrated schema. The + migration claim coordinates the processes that are *starting*; it does nothing about one already + running. **If a full stop is impossible**, upgrade ingestion first and let it migrate, but stop + or upgrade query and annotation *before* it does — leaving them up is a live wrong answer, not + just a risk window. + +**The upgrade itself:** + +4. **Start one service and let it migrate.** Five migrations run in the first upgraded process, + two of them full scans of the `buckets` collection — minutes to a couple of hours on archives in + the tens of millions of buckets. **Budget the window from a read-only measurement of your own + archive rather than from that range**; the [SLAC runbook](https://github.com/osprey-dcs/dp-service/blob/rel-1.16.0/doc/runbooks/upgrade-1.16-slac.md) + has the query. Other services started meanwhile wait five minutes on the claim and then exit; + restart them once the migration finishes. +5. **Verify** the schema marker (`version: 5`), the `pvStats` count, and the index set before + starting the rest. + +**Then, in client code:** + +6. **Rebuild against the 1.16.0 stubs and fix the compile errors.** `SaveDataSetRequest` is flat, + `Annotation` moved to the top level, `Annotation.comment` is now `description`, + `CalculationsDataFrame` carries its frame under a `frame` submessage of type + `common.DataFrame`, and `DataValue.ValueStatus` is gone. In Python these surface at runtime + rather than at build time — grep for `valueStatus`. +7. **Python only: raise your `grpcio` floor if you pin it.** The regenerated stubs raise a + `RuntimeError` **at import** on grpcio older than 1.84.0 — not at call time, and not as a + warning. A fresh install resolves to the newest release and never sees this, which is why it + tends to surface first in a pinned environment. `protobuf` moves to `>=7.35.1`. +8. **Audit every query that passes more than one criterion.** Criteria now combine with **AND**; + values within one criterion with **OR**. Two tag criteria used to match *either* tag and now + match *both*. **This is silent** — no error, a different result set. +9. **Add paging loops wherever a result was assumed complete.** An unset `limit` now means a + server default page size, not an unbounded result. `queryPvMetadata` in particular was + previously unbounded. +10. **Check anywhere an empty criteria list was relied on to fail.** It now matches all records + and returns the first page of the collection. **`ConfigurationSelector` is the exception and + goes the other way**: an empty criteria list there is *rejected*, not treated as match-all. To + query the full `TimeRange` unconditionally, omit the selector entirely, and guard any code that + builds one conditionally so that dropping the last criterion drops the whole selector. + +Runbooks for the migration, including a rehearsal procedure against a restored copy and the SLAC +sequence with measured numbers: +[`schema-migration.md`](https://github.com/osprey-dcs/dp-service/blob/rel-1.16.0/doc/runbooks/schema-migration.md), +[`schema-migration-rehearsal.md`](https://github.com/osprey-dcs/dp-service/blob/rel-1.16.0/doc/runbooks/schema-migration-rehearsal.md), +[`upgrade-1.16-slac.md`](https://github.com/osprey-dcs/dp-service/blob/rel-1.16.0/doc/runbooks/upgrade-1.16-slac.md). + +## Sample Status API + +*data-platform [#47](https://github.com/osprey-dcs/data-platform/issues/47) · dp-grpc #121 · +dp-service #238 · dp-python-lib #8 · dp-desktop-app #37, #38 — all four repos* + +The headline feature of 1.16.0. The Sample Status API assigns status codes to **individual PV +samples at specific timestamps**, supporting automated data cleaning, quality assessment, and MLOps +workflows — an ML model labeling samples as anomalous, a rule engine flagging out-of-range values, +an operator marking a handful of suspect points. + +A status is keyed by **(pvName, timestamp, domain, layer)**. A *domain* names the contract that +gives the int32 status codes their meaning (`data_quality`, `ml_anomaly`); the MLDP neither +validates nor interprets them, following the `EnumColumn` precedent. A *layer* names the producing +stream (`ml_model_v1`, `rule_engine`, `operator_override`), so several independent interpretations +of the same samples coexist within one domain. + +Two properties shape everything built on it. **Sparse labeling is first-class**: a save supplies +only the timestamps being labeled, and the absence of a status means "no assertion" — there is no +implicit default. **Matching is by exact timestamp at nanosecond precision**, with no tolerance +window, so producers must label using timestamps taken from query results or exact `SamplingClock` +arithmetic; a recomputed or rounded timestamp silently matches nothing. + +Queries can filter on status through `QuerySpec.sampleStatusSelector`, in one of two modes: +`MODE_INCLUDE_MATCHING` returns only samples carrying a matching status ("return only the +anomalies"), `MODE_EXCLUDE_MATCHING` drops them ("drop the bad data"). Because unlabeled samples +have no assertion attached, **the two modes are not complements** over a sparsely labeled archive. +The selector is supported on the sample-oriented query methods only; a bucket query rejects it, +since a storage bucket is returned whole and cannot represent per-sample filtering. + +This release also **removes `DataValue.ValueStatus`** and its `StatusCode` / `Severity` enums +(dp-grpc #143), for which the Sample Status API is the designated replacement. No server path ever +read the removed field, so there is no stored behavior to preserve and no migration is required; +archived values that carried it still parse, and field 15 is reserved permanently so they are never +misread. Acquisition-time alarm information is now captured in a status domain, where — unlike the +removed field — it is queryable and correctable after ingestion. + +Two domain-registry methods (`saveSampleStatusDomain`, `querySampleStatusDomains`) are defined as +deferred stubs and return "not yet implemented"; the registry record shape is deferred to a later +release. + +Details: [dp-grpc](https://github.com/osprey-dcs/dp-grpc/blob/rel-1.16.0/doc/release-notes/rel-1.16.0.md#sample-status-api-dp-grpc-issue-121) (model, messages, selector semantics) · +[dp-service](https://github.com/osprey-dcs/dp-service/blob/rel-1.16.0/doc/release-notes/rel-1.16.0.md#sample-status-api-issues-dp-grpc-121-dp-service-238) (storage, page sizes, limits) · +[dp-python-lib](https://github.com/osprey-dcs/dp-python-lib/blob/rel-1.16.0/doc/release-notes/rel-1.16.0.md#sample-status-api-issue-8) (client API) · +[dp-desktop-app](https://github.com/osprey-dcs/dp-desktop-app/blob/rel-1.16.0/doc/release-notes/rel-1.16.0.md#sample-status-generation-37-38) (generation and the new Explore view). +Worked examples: [sample status cookbook](https://github.com/osprey-dcs/dp-grpc/blob/rel-1.16.0/doc/cookbook/sample-status.md). + +## DataSet and Annotation API modernization + +*data-platform [#83](https://github.com/osprey-dcs/data-platform/issues/83) · dp-grpc #132 · +dp-service #248 · dp-python-lib #6 · dp-desktop-app #42 — all four repos* + +The DataSet and Annotation APIs are the oldest generation of `DpAnnotationService`, predating the +conventions established by the PV metadata, machine configuration, and sample status APIs. 1.16.0 +brings them into line. **Method names are unchanged, but message shapes and query semantics +changed incompatibly** — this is the bulk of the compile errors in the upgrade checklist above. + +**Message shapes.** `SaveDataSetRequest` is flat rather than embedding a `DataSet`, which is the +house rule for `Save*Request` messages: the domain message carries server-set audit fields that +must not be accepted as input. `Annotation` is a top-level message rather than nested inside a +query response, and its `comment` field is renamed `description`, matching every other entity. +Both entities gain server-set `createdTime` / `updatedTime` and a last-writer `modifiedBy`, and +DataSets gain tags and attributes. + +**Query results carry references, not content.** `queryAnnotations` now returns `dataSetIds` and +`calculationsId` rather than embedding the DataSets and Calculations themselves; fetch content by +id with `queryDataSets` or `getCalculations`. This removes an N+1 fan-out on the server, which +previously issued one dataset lookup per dataset id per annotation, serially and unbatched, then +embedded every frame, column, and value into each returned annotation. + +**Criteria combine with AND.** All query criteria now AND together, values within a single +criterion OR, matching the convention used everywhere else in the API. The previous scheme ORed +some criteria together with per-method bucket assignments. **This is the silent one** — two tag +criteria that used to match "either tag" now match "both tags", with no error raised. + +**Paging.** `queryDataSets` and `queryAnnotations` are now paged, and across every paged +`DpAnnotationService` query an unset `limit` means a server-configured default page size rather +than an unbounded result. Result ordering is now part of the API contract, and a malformed page +token is rejected rather than silently restarting at page one. + +**New methods.** `getDataSet` / `getAnnotation` / `getCalculations` for single-record lookup, and +`deleteDataSet` / `deleteAnnotation`. Deleting a DataSet is **rejected** while any Annotation +references it; deleting an Annotation is not blocked by incoming references, which are soft +associations and may dangle. `patchDataSet` / `patchAnnotation` are reserved placeholders that +return "not yet implemented". + +**Typed calculations and column provenance.** `CalculationsDataFrame` is now `name` plus a +`frame` submessage of type `common.DataFrame`, replacing the previous `DataTimestamps` + +`repeated DataColumn` pair, so calculation output gets the full set of typed scalar, array, image, +struct, and serialized column types plus per-column metadata. Frame names must be distinct within +a `Calculations` object, since they are an addressing key. Alongside it, `ColumnProvenance` gains a +structured `derivedFrom` list naming the columns a derived column was computed from — either an +archived PV or a Calculations column, with an optional source interval, which matters for +aggregations whose input window is not implied by the output's own timestamps. Links are stored as +supplied and never validated. + +**Export** gains inline `dataBlocks` as an ad-hoc source alongside `dataSetId` and +`calculationsSpec`, at least one of which is now required. Note the documented restriction: the +tabular formats (CSV, XLSX) can represent scalar columns only, so data containing array, image, or +struct columns exports to HDF5 only. + +Details: [dp-grpc](https://github.com/osprey-dcs/dp-grpc/blob/rel-1.16.0/doc/release-notes/rel-1.16.0.md#modernized-dataset-and-annotation-apis-dp-grpc-issue-132) · +[dp-service](https://github.com/osprey-dcs/dp-service/blob/rel-1.16.0/doc/release-notes/rel-1.16.0.md#modernized-datasets-and-annotations-apis-issues-dp-grpc-132-dp-service-248) · +[dp-python-lib](https://github.com/osprey-dcs/dp-python-lib/blob/rel-1.16.0/doc/release-notes/rel-1.16.0.md#datasets-annotations-and-export-issue-6) · +[dp-desktop-app](https://github.com/osprey-dcs/dp-desktop-app/blob/rel-1.16.0/doc/release-notes/rel-1.16.0.md#annotation-api-modernization-42). +Worked examples: [datasets and annotations cookbook](https://github.com/osprey-dcs/dp-grpc/blob/rel-1.16.0/doc/cookbook/datasets-and-annotations.md). + +## Query performance + +*dp-service [#232](https://github.com/osprey-dcs/dp-service/issues/232), +[#271](https://github.com/osprey-dcs/dp-service/issues/271), +[#274](https://github.com/osprey-dcs/dp-service/issues/274), +[#275](https://github.com/osprey-dcs/dp-service/issues/275) — primarily dp-service* + +A sustained effort against bucket query cost, prompted by query-performance reports from the SLAC +deployment. Four changes, each independently significant on a large archive. + +**The startup full-collection scan is gone (#232).** Services no longer verify the whole `buckets` +collection against the configured span limit before binding their gRPC port — a scan that took +hours on archives in the tens of millions of buckets, during which the port was unbound and +Kubernetes liveness probes failed. **Startup cost is now independent of archive size.** + +**Query lower bounds are per PV rather than archive-wide (#232).** Every bucket time-range query +carries a lower bound so the index scan has a floor. That bound previously had to cover the +longest bucket *anywhere in the archive*, so a handful of over-long buckets on a few PVs imposed +that same lookback on every query for every other PV. It is now derived from a per-PV statistic +maintained at ingestion. On an archive where the configured limit had been raised to accommodate +outliers, this is the difference between a lookback measured in weeks and one measured in seconds +for the well-behaved majority. A consequence worth noting: `Buckets.maxBucketSpanSeconds` is now +**ingestion-only** — raising it no longer widens any query's scan, and a deployment that raised it +to accommodate outliers can lower it back after upgrading. + +**Every bucket query is pinned to the compound index and bounded on both sides (#271).** All +bucket retrieval now hints the shipped compound index. Previously the planner chose among every +index on the collection, and a long-lived archive still carries `pvName`-led indexes retired in +beta-1.6.0 and 1.15.0 — startup never drops an index — alongside anything added by hand, each of +which is a planner candidate. On +recent-window queries the planner was measured choosing a `lastTime`-led index whose plan needs a +blocking in-memory sort. Separately, the index scan now carries an upper bound as well as a lower +one; before, the scan ran to the end of each PV's history and discarded everything past the window +by filter, which on a historical query against a still-active PV is most of that PV's archive. +**Behavior change:** if the compound index is missing, bucket queries now fail with an error naming +the hint rather than silently degrading to a collection scan. Operators are encouraged to drop the +leftover `pvName`-led indexes, which no longer affect plan choice but still cost a write per +ingested bucket and their share of disk; the +[SLAC runbook](https://github.com/osprey-dcs/dp-service/blob/rel-1.16.0/doc/runbooks/upgrade-1.16-slac.md) +lists them by name. Any operational procedure that pre-seeded the `bucketSpanVerification` marker +to skip the startup scan (the "option 0" runbook on dp-service #257) is obsolete and should be +retired. + +**Bucket retrieval is partitioned by span class (#274).** The per-PV bound was initially applied +as one maximum over all PVs in a request, so one long-span PV made every other PV's retrieval fetch +that span's worth of history. Requests are now partitioned into power-of-two span classes, each +bounded by its own maximum. + +Five new query benchmark clients cover `queryTable`, `querySamples`, `querySamplesStream`, +`queryBuckets`, and `queryBucketsStream` (#275), with a loader that takes options for history depth +and long-span PVs. + +Details: [dp-service](https://github.com/osprey-dcs/dp-service/blob/rel-1.16.0/doc/release-notes/rel-1.16.0.md#per-pv-bucket-span-bound-the-startup-span-scan-is-removed-issue-232). + +## Service metrics and observability + +*dp-service [#212](https://github.com/osprey-dcs/dp-service/issues/212) — dp-service* + +Every service now collects and exports metrics: request rates, error rates, latency histograms, a +per-stage breakdown of query handling, MongoDB command durations, handler queue and worker +saturation, gRPC call metrics, and JVM runtime metrics. The operator reference, including the +PromQL for diagnosing a slow query, is +[`doc/metrics.md`](https://github.com/osprey-dcs/dp-service/blob/rel-1.16.0/doc/metrics.md). + +**Deployment change: each service binds a second port and fails to start if it cannot.** Metrics +are on by default, on a Prometheus scrape endpoint per service — ingestion 9464, query 9465, +annotation 9466, ingestion stream 9467. The fail-closed choice is deliberate: the alternative is a +service an operator believes is instrumented and is not. Each port is configurable, the bind +interface is settable (use `127.0.0.1` on a shared host), and the whole feature switches off with +`DP_TELEMETRY_ENABLED=false`. To push to an OpenTelemetry collector instead of being scraped, the +standard OTel environment variables apply with no rebuild. + +**There is no authentication on the scrape endpoint.** It exposes no data values and no PV names, +but it does reveal request rates, latencies, and collection names. + +**New: slow query log.** A query whose total handling time reaches a configurable threshold +(default 1000 ms, so this is on by default) writes one WARN line to a dedicated `dp.slowquery` +logger carrying the per-stage breakdown and the shape of the request. It answers "why was *this* +query slow" without a trace backend. + +**Ingestion latency is now measurable.** `dp.ingest.duration` measures from request arrival +through the end of persistence — the number the gRPC call duration cannot show, since ingestion +acknowledges as soon as a request is validated and enqueued. An operator watching call duration +alone would see a healthy few milliseconds while the queue behind it fell arbitrarily far behind. + +**Two cautions for anyone writing alerts**, both verified against a real MongoDB: the +`db.client.operation.duration` `error.type` label covers only commands the server *refused*, so +duplicate keys and validation failures never appear there; and a total database outage makes the +metric **go silent** rather than raising an error rate, so alert on absence of data. + +Details: [dp-service](https://github.com/osprey-dcs/dp-service/blob/rel-1.16.0/doc/release-notes/rel-1.16.0.md#service-metrics-issue-212). + +## Schema migration mechanism + +*dp-service [#254](https://github.com/osprey-dcs/dp-service/issues/254) — dp-service* + +1.16.0 is the first release delivered through a schema migration mechanism, and the mechanism +itself is new infrastructure worth understanding before the upgrade. + +Every service records the database's schema version and applies pending migrations during startup, +**before its port binds**. A database whose version the binary cannot establish — newer than the +build, a migration that failed partway, a claim held by a process that did not finish — **stops the +service** rather than being served from. The choice is deliberate: every migration in this release +exists because the unmigrated shape reads as a *wrong answer* rather than an error — a null +description, an unmatchable tag, an invisible bucket — and a mechanism that logged and continued +would compound one silent failure with another. + +Concurrent startup is the normal case: one process wins an atomic claim and migrates, the others +wait and then proceed. A database with no marker is classified by content — empty means a fresh +install stamped at the current version, while any document in any managed collection means a legacy +database migrated from version 0. That classification is why you should **restore backups before +the first start**, never underneath a marker. + +Migrations can be disabled with `DP_MONGO_RUN_SCHEMA_MIGRATIONS_ON_STARTUP=false`, which skips +*applying* them but **not the version check** — a mismatched database still refuses to start. + +Five migrations ship in 1.16.0: three on the annotations collection (the `comment` → `description` +rename and its text index, tag normalization, and id canonicalization) and two on buckets, both +full scans, seeding the per-PV statistics the query work above depends on. + +Details: [dp-service](https://github.com/osprey-dcs/dp-service/blob/rel-1.16.0/doc/release-notes/rel-1.16.0.md#schema-migration-mechanism-issue-254). + +## Desktop application: deployment mode and metadata authoring + +*data-platform [#88](https://github.com/osprey-dcs/data-platform/issues/88) · dp-desktop-app #4, +#17, #18, #27, #36, #39 — dp-desktop-app* + +**The desktop app is no longer demonstration-only.** Previous releases ran the MLDP services +in-process alongside the GUI, against a local demo database. It now also runs against +**already-running remote services**, selected with `--mode=deployment` plus four configurable +connect strings. Deployment mode never constructs a MongoDB client at all, and the status bar, +window title, and startup log all name the query target, so "which archive am I pointed at" is +answerable from the UI. An absent or unrecognized mode resolves to demo — deliberately, so a +misspelling cannot become a connection attempt against production. + +Deployment mode writes no PV time-series data and authors no curated metadata in this release: +ingestion, metadata authoring, `Explore → Data Events`, and demo-data deletion are disabled. It is **not** a read-only +mode — dataset save, annotation save, and export remain available, since those are the analysis +workflow. + +**Behavior change: the demo database is no longer dropped at launch.** Every previous release +wiped it at startup, so a demo session always began empty. This is the first release where a demo +session starts with the previous session's data still present; `Tools → Delete Demo Data` clears it +on request. + +**The data query migrated to Query API V2**, bringing several visible changes: missing values +render blank rather than `N/A` (V2 distinguishes "no sample here" from a decode failure), the +1-minute query chopping is gone, results are capped at 50,000 rows and say so, and a running query +can be stopped. PVs can be selected three ways — name list, name pattern, or metadata criteria — +and two optional filters compose by intersection: machine configuration activations restrict the +time axis, then sample status drops individual samples. + +**Three new Explore views** cover PV metadata, machine configurations and their activations, and +sample statuses. The old `Explore → PVs` is renamed **PV Statistics** to distinguish curated +metadata records from statistics derived by aggregation over ingested buckets — a PV can appear in +one and not the other. **Two new authoring views**, `Metadata → PV` and +`Metadata → Machine Configuration`, create and update curated metadata records; both saves are +full-replace upserts, so an existing record is confirmed before being overwritten. + +The ingestion views' request-level "Request Details" panel is replaced by a **Column Metadata** +panel, attaching tags, attributes, and provenance to every column rather than to the request, which +is where the archive actually stores and queries them. + +This release also adds the repo's **first automated test coverage and CI build** — unit tests, +view-load smoke tests enumerated from the classpath, and live integration tests that skip rather +than fail when MongoDB is unreachable. + +Details: [dp-desktop-app](https://github.com/osprey-dcs/dp-desktop-app/blob/rel-1.16.0/doc/release-notes/rel-1.16.0.md). + +## Python client library + +*dp-python-lib #6, #8, #14, #40, #41 — dp-python-lib* + +**The largest release the library has had.** It adds three new API areas — sample status, +datasets/annotations/export, and the `DataFrame` builders and conversions both depend on — and +brings `AnnotationClient` to **full coverage of every implemented `DpAnnotationService` feature +area**: PV metadata, machine configuration, sample status, datasets, annotations, and export. + +It is a breaking release only narrowly: every hand-written client signature is unchanged or +widened, and both breaking changes are inherited from the protocol — the removal of +`DataValue.valueStatus`, and the AND/OR criteria change underneath an unchanged client API. Python +has no compile step, so the first surfaces as an `AttributeError` at runtime; grep for it. + +Two client-side conveniences worth calling out. `get_datasets(ids)` is the batch fetch that avoids +the annotation-listing N+1 created by references-not-content, chunking its id list so a full page of +annotation ids does not become an oversized request. And the `iter_*` methods follow +`next_page_token` for you, which is the migration path for any code that treated an unpaged +`query_*` result as complete. + +Details: [dp-python-lib](https://github.com/osprey-dcs/dp-python-lib/blob/rel-1.16.0/doc/release-notes/rel-1.16.0.md). + +## Correctness fixes worth knowing + +Several fixes in this release closed defects that produced **wrong answers rather than errors**. +They are collected here because each one means results from 1.15.0 and earlier may have been +incomplete in ways nothing reported. + +- **Unary `querySamples` silently omitted PVs on large pages** (dp-service #274). A page was + assembled by draining the bucket cursor until the message budget tripped; since the cursor is + ordered by PV, a budget that tripped partway through the first PV emitted every later PV as + all-unset values with no error, and the page token resumed at the same position — those PVs were + never returned. With default settings this affected any request whose first PV in name order had + more than roughly 455,000 samples in the window: about 7.5 minutes at 1 kHz, 12 hours at 10 Hz, + or 5 days at 1 Hz. + Pages are now retrieved in time slices, each covering every selected PV. +- **Annotation edits silently destroyed stored Calculations** (dp-desktop-app #42). Loading an + annotation from a query result populated the editor with no calculations, and saving any + unrelated edit then destroyed the stored object — no error, no warning, nothing in the UI. The + Annotation Builder's tags and attributes had the same shape of bug from the opposite direction, + and are fixed in the same release. +- **`querySamplesStream` is now bounded in memory**, emitting as slices are retrieved rather than + assembling the whole window first, and the streaming methods now apply outbound flow control so a + slow client no longer causes the server to buffer an entire result. +- **Blank criterion values no longer reach the server** (dp-service #243). A blank `prefix` or + `contains` value was a silent match-all. +- **Business-rule failures are now rejections rather than errors** (dp-service #235), so a caller + can distinguish a client mistake from a service failure without matching on the message text. + +## Release process and supply chain + +*data-platform [#90](https://github.com/osprey-dcs/data-platform/issues/90)* + +**Every GitHub Actions reference in every repository in the organization is now pinned to a full +commit SHA** with a trailing version comment, rather than to a floating tag. A tag is mutable: +whoever controls an action's repository can repoint it at different code, and every workflow picks +that up on its next run with no diff, no review, and no notification. Release jobs run with +`contents: write` and publish the artifacts users download, so this is where it matters most. Each +repository also carries a Dependabot configuration for the `github-actions` ecosystem, since +pinning otherwise trades supply-chain risk for silent staleness. The convention is documented in +[CLAUDE.md](https://github.com/osprey-dcs/data-platform/blob/rel-1.16.0/CLAUDE.md#github-actions-pin-every-uses-to-a-commit-sha). + +Several release workflows gained a **dry-run rehearsal path** (`workflow_dispatch` with a `dry_run` +input defaulting to true), so a workflow whose write side previously could only be exercised by +cutting a real release can now be rehearsed. + +**Release notes are now version-controlled**, one document per release under `doc/release-notes/` +in each repository, published as the GitHub release body. Release jobs verify the notes exist +before building rather than failing at the publish step. This document is the master set; the +child repositories' notes carry the per-repository detail. + +Plan documents for substantial tickets are likewise now version-controlled under `plan/tickets/`, +so a design record gets PR review and a stable cross-repo URL.