From eb7b1ac8b23152c349c552686bb508a51076a982 Mon Sep 17 00:00:00 2001 From: Andrey Fedorov Date: Tue, 29 Sep 2026 15:42:44 -0400 Subject: [PATCH] feat: release history via get_release_changes (#40) Add GET /v3/releases/changes and the MCP tool get_release_changes, which diff IDC release N against N-1 (series added/revised/removed, new/updated/ removed collections, analysis results) from index + prior_versions_index. Improve discoverability of version history: get_idc_version description, a whats_new MCP prompt, an INSTRUCTIONS line and idc://guide section, descriptions for prior_versions_index, and notable_columns in list_tables. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 20 +++ docs/user-guide.md | 40 +++++- src/idc_api/core/context.py | 2 + src/idc_api/core/models.py | 62 ++++++++++ src/idc_api/core/schema.py | 50 +++++++- src/idc_api/core/services/__init__.py | 2 + src/idc_api/core/services/query.py | 2 + src/idc_api/core/services/releases.py | 167 ++++++++++++++++++++++++++ src/idc_api/mcp/server.py | 58 ++++++++- src/idc_api/rest/app.py | 19 +++ tests/test_releases.py | 115 ++++++++++++++++++ 11 files changed, 531 insertions(+), 6 deletions(-) create mode 100644 src/idc_api/core/services/releases.py create mode 100644 tests/test_releases.py diff --git a/CHANGELOG.md b/CHANGELOG.md index cff056d..d1da239 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -13,6 +13,26 @@ Refactors, CI, and formatting land in the git history, not here. ## [Unreleased] +### Added + +- **Release history: `GET /v3/releases/changes?version=N` and the MCP tool + `get_release_changes(version)`** — what changed in an IDC release versus the previous one: + series added / revised / removed (with size in TB and patients affected), collections that are + new / updated / removed, and analysis results that gained series. Any past release can be + described, computed from the served release's `index` + `prior_versions_index`; `version` + defaults to the served release. +- MCP prompt **`whats_new`** (optional `version`) — a ready-made "summarize this release" request. +- `GET /v3/tables` / `list_tables`: each table now carries **`notable_columns`**, a few columns + worth knowing before writing SQL (e.g. `series_init_idc_version` on `index`). + +### Changed + +- `prior_versions_index` is now documented: a table description and descriptions for + `min_idc_version`, `max_idc_version`, `crdc_series_uuid`, and `series_size_MB` (upstream ships + them empty). +- The `get_idc_version` tool description no longer implies the server can only speak to one + release; it points at `get_release_changes` and the version-history columns. + ## [3.0.0b3] — 2026-08-10 Beta iteration: one shape for every cohort filter, and no request that silently answers with the diff --git a/docs/user-guide.md b/docs/user-guide.md index d7e7637..93895f5 100644 --- a/docs/user-guide.md +++ b/docs/user-guide.md @@ -54,7 +54,7 @@ understanding, because picking the right one makes everything else easy: | Surface | Answers | REST | MCP tools | |---|---|---|---| -| **Discovery** | "What exists? What can I filter on?" | `GET /v3/version`, `/v3/stats`, `/v3/collections`, `/v3/collections/{id}`, `/v3/analysis_results`, `/v3/attributes`, `/v3/attributes/{attr}/values` | `get_idc_version`, `get_stats`, `list_collections`, `get_collection`, `list_analysis_results`, `list_attributes`, `get_attribute_values` | +| **Discovery** | "What exists? What can I filter on? What changed in a release?" | `GET /v3/version`, `/v3/releases/changes`, `/v3/stats`, `/v3/collections`, `/v3/collections/{id}`, `/v3/analysis_results`, `/v3/attributes`, `/v3/attributes/{attr}/values` | `get_idc_version`, `get_release_changes`, `get_stats`, `list_collections`, `get_collection`, `list_analysis_results`, `list_attributes`, `get_attribute_values` | | **Cohort** | "How big is *my* selection, and what's in it?" | `POST /v3/cohort/counts`, `POST /v3/cohort/manifest` | `build_cohort` | | **Retrieval** | "Give me the download links" | `POST /v3/cohort/manifest.txt` | `get_cohort_urls` | | **SQL** | "Run my custom query" + schema | `GET /v3/tables`, `/v3/tables/{table}`, `POST /v3/sql` | `list_tables`, `get_table_schema`, `run_sql` | @@ -144,7 +144,7 @@ per-collection, keyed by `collection_id`). | `index` | one row per **series** — the main table | | `collections_index` | one row per collection (curated metadata) | | `analysis_results_index` | one row per analysis result | -| `version_metadata_index` / `prior_versions_index` | IDC release versions / removed series | +| `version_metadata_index` / `prior_versions_index` | IDC release dates / superseded or removed series versions (`min_idc_version`–`max_idc_version`) — see [Release history](#release-history) | | `seg_index`, `ann_index`, `ann_group_index`, `rtstruct_index` | segmentations / annotations / RT structures: **what was segmented** (`SegmentedPropertyType_CodeMeanings` — `BodyPartExamined` reflects the source acquisition, not this) and the **reference** to the image series they derive from (`segmented_SeriesInstanceUID` / `referenced_SeriesInstanceUID`) | | `ct_index`, `mr_index`, `pt_index` | per-modality acquisition parameters (slice thickness, kVp, TE/TR, injected dose…) | | `sm_index`, `sm_instance_index` | slide-microscopy (pathology) series / instance metadata | @@ -215,6 +215,34 @@ WHERE i.collection_id = 'nlst' AND i.Modality = 'CT' > `IDC_API_INCLUDE_INDICES=all` includes it; the clinical tools return a clear "not included" > error otherwise). +### Release history + +The server serves **one** IDC data release (see `GET /v3/version` / `get_idc_version`), but that +release carries the history of every earlier one, so "what's new in v24" or "what changed since +v20" are answerable without external release notes: + +- On `index`, `series_init_idc_version` is the release a series first appeared in, and + `series_revised_idc_version` the release its current content dates from. +- `prior_versions_index` holds every **superseded or removed version** of a series — one row per + old `crdc_series_uuid`, valid from `min_idc_version` through `max_idc_version`. If its + `SeriesInstanceUID` is still in `index` the series was revised; otherwise it was removed. +- `version_metadata_index` dates each release. + +`GET /v3/releases/changes?version=N` / `get_release_changes(version=N)` does the diff of release +N against N-1 for you: series **added / revised / removed** (with TB and patients affected), +collections that are **new / updated / removed**, and the analysis results that gained series. +Omit `version` for the served release. For per-series detail, query the columns above with SQL — +e.g. the series of a collection that changed since v20: + +```sql +SELECT SeriesInstanceUID, series_init_idc_version, series_revised_idc_version +FROM index +WHERE collection_id = 'nlst' AND series_revised_idc_version > 20 +``` + +The diff tells you *what* changed, not *why*; for the narrative, see the +[IDC release notes](https://learn.canceridc.dev/data/data-release-notes). + --- ## 2. Using the REST API @@ -230,13 +258,14 @@ uv run idc-api # http://127.0.0.1:8000 — Swagger UI at /v3/docs | Method & path | Purpose | |---|---| | `GET /v3/version` | IDC data release served (e.g. `v24`) + pinned index version, **and** this server's own software version (`api_version`, plus `build` if the deploy stamped one) | +| `GET /v3/releases/changes?version=` | What changed in an IDC release vs. the previous one: series added / revised / removed, new / updated / removed collections, analysis results (default: the served release) | | `GET /v3/stats` | Headline totals (collections, patients, studies, series, size_TB) | | `GET /v3/collections` | List collections (datasets) | | `GET /v3/collections/{id}` | Collection detail: counts, modalities, license breakdown | | `GET /v3/analysis_results` | Derived datasets (segmentations/annotations) | | `GET /v3/attributes` | Filterable attributes (name, type, term/range, categorical) | | `GET /v3/attributes/{attr}/values?limit=` | Distinct values + counts for an attribute, plus a `note` caveat when one applies (e.g. `BodyPartExamined` ≠ segmented anatomy) | -| `GET /v3/tables` | Tables available to SQL | +| `GET /v3/tables` | Tables available to SQL, each with a few `notable_columns` | | `GET /v3/tables/{table}` | Column schema for a table | | `GET /v3/clinical/tables?collection_id=` | Per-collection clinical tables (optionally one collection) | | `GET /v3/clinical/tables/{table}` | Clinical table columns + human-readable labels | @@ -285,6 +314,8 @@ curl -s 'localhost:8000/v3/attributes/Modality/values?limit=10' ```bash curl -s localhost:8000/v3/version # data release + this server's build +curl -s localhost:8000/v3/releases/changes # what's new in the served release +curl -s 'localhost:8000/v3/releases/changes?version=23' # …or in any earlier one curl -s localhost:8000/v3/stats # headline totals curl -s localhost:8000/v3/collections # list datasets curl -s localhost:8000/v3/collections/nlst # one collection's detail @@ -368,6 +399,7 @@ uv run idc-mcp --http --host 0.0.0.0 --port 8080 # hosted/shared - **Discovery:** `get_idc_version`, `get_stats`, `list_collections`, `get_collection`, `list_analysis_results`, `list_attributes`, `get_attribute_values` +- **Release history:** `get_release_changes` - **Schema (for SQL):** `list_tables`, `get_table_schema` - **Clinical data:** `list_clinical_tables`, `get_clinical_table_schema`, `get_clinical_table` - **Cohort / query:** `build_cohort`, `run_sql` @@ -375,6 +407,8 @@ uv run idc-mcp --http --host 0.0.0.0 --port 8080 # hosted/shared `get_citations`, `get_licenses` - **Resources:** `idc://guide` (data model + recommended workflow), `idc://tables`, `idc://schema/{table}` +- **Prompts:** `whats_new` (optional `version`) — a ready-made "summarize what's new in this + release" request for clients that show MCP prompts Tool descriptions are prescriptive about *when* to call each one, and the server ships an `idc://guide` resource with the same conceptual model as this document — so a capable agent can diff --git a/src/idc_api/core/context.py b/src/idc_api/core/context.py index ee8fa54..13f4838 100644 --- a/src/idc_api/core/context.py +++ b/src/idc_api/core/context.py @@ -13,6 +13,7 @@ LicenseService, ManifestService, QueryService, + ReleaseService, ViewerService, ) @@ -22,6 +23,7 @@ def __init__(self, settings: Settings | None = None): self.settings = settings or get_settings() self.backend = DuckDBBackend(self.settings) self.discovery = DiscoveryService(self.backend) + self.releases = ReleaseService(self.backend) self.cohort = CohortService(self.backend, self.settings) self.manifest = ManifestService(self.backend, self.settings) self.query = QueryService(self.backend, self.settings) diff --git a/src/idc_api/core/models.py b/src/idc_api/core/models.py index c7afba8..baadc14 100644 --- a/src/idc_api/core/models.py +++ b/src/idc_api/core/models.py @@ -36,6 +36,63 @@ class Stats(BaseModel): size_TB: float +class CollectionChange(BaseModel): + collection_id: str + status: str = Field( + ..., + description="'new' (no series in the previous release), 'removed' (no series in this " + "release), or 'updated' (present in both, with series added/revised/removed).", + ) + series_added: int + series_revised: int + series_removed: int + size_TB_added: float + size_TB_revised: float + size_TB_removed: float + patients_affected: int = Field( + ..., description="Distinct patients with at least one added, revised, or removed series." + ) + + +class AnalysisResultChange(BaseModel): + analysis_result_id: str + is_new: bool = Field(..., description="True if this analysis result first appeared here.") + series_added: int + + +class ReleaseChanges(BaseModel): + """What changed in one IDC release relative to the one before it.""" + + idc_version: str = Field(..., description="The release described, e.g. 'v24'.") + release_date: str | None = None + previous_version: str | None = Field(None, description="The release compared against.") + previous_release_date: str | None = None + current_version: str = Field(..., description="The release this server serves.") + series_added: int + series_revised: int + series_removed: int + size_TB_added: float + size_TB_revised: float + size_TB_removed: float + patients_affected: int + new_collections: list[str] = Field( + default_factory=list, description="Collections that first appeared in this release." + ) + removed_collections: list[str] = Field( + default_factory=list, description="Collections with no series left in this release." + ) + collections: list[CollectionChange] = Field( + default_factory=list, description="Per-collection change, largest additions first." + ) + analysis_results: list[AnalysisResultChange] = Field( + default_factory=list, + description="Analysis results that gained series in this release. Counted from series " + "still present in the current release, so for an older release it omits series that were " + "later removed.", + ) + note: str = "" + + class CollectionSummary(BaseModel): collection_id: str collection_name: str | None = None @@ -117,6 +174,11 @@ class TableInfo(BaseModel): name: str description: str = "" column_count: int + notable_columns: list[str] = Field( + default_factory=list, + description="A few columns worth knowing about before writing SQL (not the full list — " + "get_table_schema has every column with its description).", + ) class TableList(BaseModel): diff --git a/src/idc_api/core/schema.py b/src/idc_api/core/schema.py index 83d42aa..1b29769 100644 --- a/src/idc_api/core/schema.py +++ b/src/idc_api/core/schema.py @@ -151,6 +151,50 @@ def _column_type(c: dict) -> str: "(join to index on dicom_patient_id = index.PatientID). Use this table's column_label and " "value mappings to interpret their often-cryptic coded columns." ), + # Upstream ships this table with no description at all. + "prior_versions_index": ( + "Series versions IDC served in an earlier release but no longer serves: one row per " + "superseded or removed version of a series (identified by crdc_series_uuid), valid from " + "min_idc_version through max_idc_version. A SeriesInstanceUID that is also in `index` was " + "revised (its current version starts at index.series_revised_idc_version); one that is " + "not was removed. Together with `index` this reconstructs the content of any past release " + "— get_release_changes does that diff for you." + ), +} + +# Fill-ins for upstream column descriptions that are empty. Applied only where upstream has no +# text, so an upstream description takes over automatically once it lands. +COLUMN_DESCRIPTION_FILLINS: dict[str, dict[str, str]] = { + "prior_versions_index": { + "crdc_series_uuid": "Identifier of this specific version of the series (changes on " + "every revision); never equal to a crdc_series_uuid in `index`.", + "min_idc_version": "First IDC release (integer) that served this version of the series.", + "max_idc_version": "Last IDC release (integer) that served this version of the series; " + "the next release revised or removed it.", + "series_size_MB": "Size of this version of the series, in MB.", + }, +} + +# A handful of columns per table surfaced by list_tables, so a caller learns they exist without +# having to call get_table_schema first (a caller that guesses the obvious columns right never +# does). Not a full listing; names absent from a table's schema are dropped. +NOTABLE_COLUMNS: dict[str, list[str]] = { + "index": [ + "collection_id", + "analysis_result_id", + "PatientID", + "SeriesInstanceUID", + "Modality", + "BodyPartExamined", + "SeriesDescription", + "license_short_name", + "series_size_MB", + "series_init_idc_version", + "series_revised_idc_version", + ], + "version_metadata_index": ["idc_version", "version_timestamp"], + "prior_versions_index": ["SeriesInstanceUID", "min_idc_version", "max_idc_version"], + "seg_index": ["segmented_SeriesInstanceUID", "SegmentedPropertyType_CodeMeanings"], } @@ -158,14 +202,16 @@ def _column_type(c: dict) -> str: def table_schema(table: str) -> dict: """Return ``{name, description, columns:[{name,type,description}]}`` for a table, sourced from the idc-index schema JSON shipped in INDEX_METADATA (table descriptions may be - repointed inward via ``TABLE_DESCRIPTION_OVERRIDES``).""" + repointed inward via ``TABLE_DESCRIPTION_OVERRIDES``; empty column descriptions are filled + from ``COLUMN_DESCRIPTION_FILLINS``).""" meta = idc_index_data.INDEX_METADATA[metadata_key(table)] schema = meta.get("schema", {}) or {} + fillins = COLUMN_DESCRIPTION_FILLINS.get(table, {}) columns = [ { "name": c["name"], "type": _column_type(c), - "description": c.get("description", "") or "", + "description": c.get("description", "") or fillins.get(c["name"], ""), } for c in schema.get("columns", []) ] diff --git a/src/idc_api/core/services/__init__.py b/src/idc_api/core/services/__init__.py index 69643e1..1785462 100644 --- a/src/idc_api/core/services/__init__.py +++ b/src/idc_api/core/services/__init__.py @@ -11,6 +11,7 @@ from .licenses import LicenseService from .manifest import ManifestService from .query import QueryService +from .releases import ReleaseService from .viewer import ViewerService __all__ = [ @@ -21,5 +22,6 @@ "LicenseService", "ManifestService", "QueryService", + "ReleaseService", "ViewerService", ] diff --git a/src/idc_api/core/services/query.py b/src/idc_api/core/services/query.py index 084293d..1288508 100644 --- a/src/idc_api/core/services/query.py +++ b/src/idc_api/core/services/query.py @@ -17,11 +17,13 @@ def list_tables(self) -> TableList: tables = [] for name in self.backend.list_tables(): sch = schema.table_schema(name) + names = {c["name"] for c in sch["columns"]} tables.append( TableInfo( name=name, description=sch["description"], column_count=len(sch["columns"]), + notable_columns=[c for c in schema.NOTABLE_COLUMNS.get(name, []) if c in names], ) ) return TableList(tables=tables) diff --git a/src/idc_api/core/services/releases.py b/src/idc_api/core/services/releases.py new file mode 100644 index 0000000..e20c714 --- /dev/null +++ b/src/idc_api/core/services/releases.py @@ -0,0 +1,167 @@ +"""Release history: what changed in a given IDC data release. + +Computed entirely from the current release's bundled tables. Every *series version* IDC has +ever served is either the current row in ``index`` (valid from ``series_revised_idc_version`` +to now) or a row in ``prior_versions_index`` (a superseded or removed version, valid from +``min_idc_version`` to ``max_idc_version``). A series is in release N if one of its versions +spans N, so comparing release N with N-1 gives added / revised / removed series exactly. +""" + +from __future__ import annotations + +from ..backend.base import QueryBackend +from ..errors import InvalidQueryError +from ..models import AnalysisResultChange, CollectionChange, ReleaseChanges + +_MB_PER_TB = 1_000_000 + +# Params: current, N, N, N-1, N-1, N. +_DIFF_SQL = """ +WITH v AS ( + SELECT SeriesInstanceUID, collection_id, PatientID, series_size_MB, + series_revised_idc_version AS lo, ? AS hi + FROM index + UNION ALL + SELECT SeriesInstanceUID, collection_id, PatientID, series_size_MB, + min_idc_version AS lo, max_idc_version AS hi + FROM prior_versions_index +), +cur AS (SELECT * FROM v WHERE lo <= ? AND hi >= ?), +prev AS (SELECT * FROM v WHERE lo <= ? AND hi >= ?), +s AS ( + SELECT COALESCE(c.collection_id, p.collection_id) AS collection_id, + COALESCE(c.PatientID, p.PatientID) AS PatientID, + c.SeriesInstanceUID IS NOT NULL AS in_cur, + p.SeriesInstanceUID IS NOT NULL AS in_prev, + CASE WHEN p.SeriesInstanceUID IS NULL THEN 'added' + WHEN c.SeriesInstanceUID IS NULL THEN 'removed' + WHEN c.lo = ? THEN 'revised' END AS change, + c.series_size_MB AS cur_mb, + p.series_size_MB AS prev_mb + FROM cur c FULL OUTER JOIN prev p ON c.SeriesInstanceUID = p.SeriesInstanceUID +) +SELECT GROUPING(collection_id) AS is_total, + collection_id, + count(*) FILTER (WHERE in_prev) AS prev_series, + count(*) FILTER (WHERE in_cur) AS cur_series, + count(*) FILTER (WHERE change = 'added') AS series_added, + count(*) FILTER (WHERE change = 'revised') AS series_revised, + count(*) FILTER (WHERE change = 'removed') AS series_removed, + COALESCE(sum(cur_mb) FILTER (WHERE change = 'added'), 0) AS added_mb, + COALESCE(sum(cur_mb) FILTER (WHERE change = 'revised'), 0) AS revised_mb, + COALESCE(sum(prev_mb) FILTER (WHERE change = 'removed'), 0) AS removed_mb, + count(DISTINCT PatientID) FILTER (WHERE change IS NOT NULL) AS patients_affected +FROM s +GROUP BY GROUPING SETS ((collection_id), ()) +HAVING count(*) FILTER (WHERE change IS NOT NULL) > 0 OR GROUPING(collection_id) = 1 +""" + +_NOTE = ( + "Computed from the current release: series in `index` plus superseded/removed series " + "versions in `prior_versions_index`. 'revised' means the series is in both releases but its " + "content changed (new crdc_series_uuid). For per-series detail, query those tables with " + "run_sql (series_init_idc_version / series_revised_idc_version on `index`; min_idc_version / " + "max_idc_version on `prior_versions_index`)." +) + + +def _tb(mb: float) -> float: + return round(mb / _MB_PER_TB, 3) + + +def _parse_version(version: int | str) -> int: + s = str(version).strip().lower().removeprefix("v") + if not s.isdigit(): + raise InvalidQueryError(f"Invalid IDC version: {version!r}. Use an integer such as 24.") + return int(s) + + +class ReleaseService: + def __init__(self, backend: QueryBackend): + self.backend = backend + + def _release_dates(self) -> dict[int, str | None]: + rows = self.backend.query( + "SELECT idc_version, version_timestamp FROM version_metadata_index" + ).rows + return { + int(r["idc_version"]): (str(r["version_timestamp"]) if r["version_timestamp"] else None) + for r in rows + } + + def release_changes(self, version: int | str | None = None) -> ReleaseChanges: + dates = self._release_dates() + current = max(dates) + n = current if version is None else _parse_version(version) + if n not in dates: + raise InvalidQueryError( + f"Unknown IDC version: {version!r}. Releases available: v{min(dates)}–v{current}." + ) + + rows = self.backend.query(_DIFF_SQL, [current, n, n, n - 1, n - 1, n]).rows + total = next(r for r in rows if r["is_total"]) + per_coll = [r for r in rows if not r["is_total"]] + + def status(r: dict) -> str: + if r["prev_series"] == 0: + return "new" + if r["cur_series"] == 0: + return "removed" + return "updated" + + collections = sorted( + ( + CollectionChange( + collection_id=r["collection_id"], + status=status(r), + series_added=r["series_added"], + series_revised=r["series_revised"], + series_removed=r["series_removed"], + size_TB_added=_tb(r["added_mb"]), + size_TB_revised=_tb(r["revised_mb"]), + size_TB_removed=_tb(r["removed_mb"]), + patients_affected=r["patients_affected"], + ) + for r in per_coll + ), + key=lambda c: (-c.size_TB_added, -c.series_added, c.collection_id), + ) + + analysis = self.backend.query( + "SELECT analysis_result_id, " + "count(*) FILTER (WHERE series_init_idc_version = ?) AS series_added, " + "min(series_init_idc_version) = ? AS is_new " + "FROM index WHERE analysis_result_id IS NOT NULL " + "GROUP BY 1 HAVING series_added > 0 ORDER BY series_added DESC, 1", + [n, n], + ).rows + + prev = n - 1 if (n - 1) in dates else None + return ReleaseChanges( + idc_version=f"v{n}", + release_date=dates[n], + previous_version=f"v{prev}" if prev is not None else None, + previous_release_date=dates.get(prev) if prev is not None else None, + current_version=f"v{current}", + series_added=total["series_added"], + series_revised=total["series_revised"], + series_removed=total["series_removed"], + size_TB_added=_tb(total["added_mb"]), + size_TB_revised=_tb(total["revised_mb"]), + size_TB_removed=_tb(total["removed_mb"]), + patients_affected=total["patients_affected"], + new_collections=sorted(c.collection_id for c in collections if c.status == "new"), + removed_collections=sorted( + c.collection_id for c in collections if c.status == "removed" + ), + collections=collections, + analysis_results=[ + AnalysisResultChange( + analysis_result_id=r["analysis_result_id"], + is_new=bool(r["is_new"]), + series_added=r["series_added"], + ) + for r in analysis + ], + note=_NOTE, + ) diff --git a/src/idc_api/mcp/server.py b/src/idc_api/mcp/server.py index a5db696..42377a0 100644 --- a/src/idc_api/mcp/server.py +++ b/src/idc_api/mcp/server.py @@ -64,6 +64,8 @@ buckets, no server involved. 4. Trust `filters_applied`, not your intent: surface any `warnings` a result carries, and treat an empty `filters_applied` as the whole archive rather than a cohort. +5. For "what's new / what changed in release N", call get_release_changes — release history is + queryable even though one release is served. Cite with get_citations (per-dataset citations plus the IDC paper to acknowledge IDC itself); respect get_licenses (CC-BY vs CC-BY-NC). See `idc://guide` for the data model, the full tool list, and join examples.""" @@ -242,7 +244,10 @@ def get_idc_version() -> dict: """Return the IDC data release served (e.g. 'v24') and pinned idc-index version, plus this server's own software version (`api_version`, and `build` if the deploy stamped one). Call this to confirm which IDC version your answers are based on — and which build of the server - produced them.""" + produced them. The server serves one release but is NOT limited to it: every series records + the release it first appeared in and was last revised in, and superseded/removed series are + kept in `prior_versions_index`, so past releases and what changed between them are + queryable — use get_release_changes for "what's new in vN".""" return ctx.discovery.version().model_dump(mode="json") @@ -254,6 +259,21 @@ def get_stats() -> dict: return ctx.discovery.stats().model_dump(mode="json") +@mcp.tool() +@guard +def get_release_changes(version: int | None = None) -> dict: + """What's new in an IDC data release — what changed since the previous version. Returns the + series added, revised, and removed (with size in TB and patients affected), the collections + that were newly added, updated, or removed, and the analysis results that gained series. + `version` is the release number (e.g. 24 for v24); omit it for the release this server + serves. Call this whenever the user asks what is new, added, changed, updated, or removed in + IDC, in a given release, or since a version — do not conclude that history is unavailable + because the server serves a single release. For per-series detail beyond this summary, use + run_sql on `index` (series_init_idc_version / series_revised_idc_version) and + `prior_versions_index` (min_idc_version / max_idc_version).""" + return ctx.releases.release_changes(version).model_dump(mode="json") + + @mcp.tool() @guard def list_collections() -> list[dict]: @@ -486,6 +506,29 @@ def get_licenses(terms: dict | None = None, ranges: dict | None = None) -> dict: return ctx.licenses.get_licenses(f).model_dump(mode="json") +# --- prompts ------------------------------------------------------------------------------ + + +@mcp.prompt() +def whats_new(version: str = "") -> str: + """Summarize what is new in an IDC data release (the latest one by default).""" + n = version.strip().lower().removeprefix("v") + target, call = ( + (f"IDC release v{n}", f"get_release_changes(version={n})") + if n + else ( + "the latest IDC release", + "get_release_changes()", + ) + ) + return ( + f"Summarize what changed in {target}. Call {call}, then report: the release date; the " + "headline totals (series added / revised / removed, TB added, patients affected); the new " + "collections, each with a one-line description from get_collection; notable updates or " + "removals in existing collections; and any new analysis results." + ) + + # --- resources ---------------------------------------------------------------------------- _GUIDE = """\ @@ -502,6 +545,8 @@ def get_licenses(terms: dict | None = None, ranges: dict | None = None) -> dict: - *Discovery* (`get_stats`, `list_collections`, `get_collection`, `list_analysis_results`, `list_attributes`, `get_attribute_values`) — what exists, and the *vocabulary* (attribute names + valid values) you filter on. +- *Release history* (`get_idc_version`, `get_release_changes`) — which release is served, and + what each release added, revised, or removed (see *Release history* below). - *Cohort* (`build_cohort`) — turn a chosen combination of that vocabulary into distinct counts + a sample of series + a download payload. - *Retrieval* (`get_cohort_urls`) — the download half: public URLs for direct S3/GCS transfer. @@ -583,6 +628,17 @@ def get_licenses(terms: dict | None = None, ranges: dict | None = None) -> dict: `index JOIN clinical.nlst_canc ON index.PatientID = clinical.nlst_canc.dicom_patient_id` filtered on the relevant staging column. Use `get_clinical_table` to read a whole small table. These tables are present only when `clinical_index` is included in the build. + +**Release history.** The server serves one IDC release, but that release carries the history of +all earlier ones. On `index`, `series_init_idc_version` is the release a series first appeared in +and `series_revised_idc_version` the release its current content dates from. +`prior_versions_index` holds every superseded or removed *version* of a series (one row per old +`crdc_series_uuid`, valid from `min_idc_version` through `max_idc_version`); a SeriesInstanceUID +also in `index` was revised, one that isn't was removed. `version_metadata_index` dates each +release. `get_release_changes(version=N)` diffs release N against N-1 from these tables — series +added / revised / removed, new and removed collections, per-collection size — so start there for +"what's new in vN"; drop to `run_sql` on those columns for per-series detail (e.g. "which of my +cohort's series changed since v20": `series_revised_idc_version > 20`). """ diff --git a/src/idc_api/rest/app.py b/src/idc_api/rest/app.py index 5a41e4c..af6405f 100644 --- a/src/idc_api/rest/app.py +++ b/src/idc_api/rest/app.py @@ -32,6 +32,7 @@ CollectionSummary, LicensesResult, ManifestResponse, + ReleaseChanges, SqlResult, Stats, TableList, @@ -425,6 +426,24 @@ def stats(): patients, studies, series, and instances, plus the total size in TB.""" return C().discovery.stats() + @app.get( + f"{API_PREFIX}/releases/changes", + response_model=ReleaseChanges, + tags=["discovery"], + summary="What changed in a release", + ) + def release_changes( + version: int | None = Query( + None, ge=1, description="IDC release number, e.g. 24; omit for the served release." + ), + ): + """What's new in an IDC data release relative to the previous one: series added, + revised, and removed (with size in TB and patients affected), collections that were + newly added, updated, or removed, and analysis results that gained series. Computed from + the served release's own tables (`index` plus `prior_versions_index`), so any past + release can be described, not just the latest.""" + return C().releases.release_changes(version) + @app.get( f"{API_PREFIX}/collections", response_model=list[CollectionSummary], diff --git a/tests/test_releases.py b/tests/test_releases.py new file mode 100644 index 0000000..7f887b7 --- /dev/null +++ b/tests/test_releases.py @@ -0,0 +1,115 @@ +"""Release history: get_release_changes diffs release N against N-1, and the version-history +tables/columns are discoverable from the tool surface (issue #40).""" + +from __future__ import annotations + +import pytest + +from idc_api.core import schema +from idc_api.core.errors import InvalidQueryError +from idc_api.mcp.server import mcp + + +def _current(ctx) -> int: + return int( + ctx.backend.query("SELECT max(idc_version) v FROM version_metadata_index").rows[0]["v"] + ) + + +def test_latest_release_is_default_and_self_consistent(ctx): + cur = _current(ctx) + rc = ctx.releases.release_changes() + assert rc.idc_version == rc.current_version == f"v{cur}" + assert rc.previous_version == f"v{cur - 1}" + assert rc.release_date and rc.previous_release_date + # Totals are the sum of the per-collection rows. + assert rc.series_added == sum(c.series_added for c in rc.collections) + assert rc.series_revised == sum(c.series_revised for c in rc.collections) + assert rc.series_removed == sum(c.series_removed for c in rc.collections) + assert rc.series_added > 0 + # In the served release, a series whose current content dates from this release was either + # added (absent before) or revised (present before) — nothing else. + n = ctx.backend.query( + "SELECT count(*) n FROM index WHERE series_revised_idc_version = ?", [cur] + ).rows[0]["n"] + assert rc.series_added + rc.series_revised == n + revised = ctx.backend.query( + "SELECT count(*) n FROM index WHERE series_init_idc_version < ? " + "AND series_revised_idc_version = ?", + [cur, cur], + ).rows[0]["n"] + assert rc.series_revised == revised + by_id = {c.collection_id: c for c in rc.collections} + for cid in rc.new_collections: + c = by_id[cid] + assert c.status == "new" and c.series_added > 0 and c.series_removed == 0 + + +def test_first_release_has_no_predecessor(ctx): + rc = ctx.releases.release_changes(1) + assert rc.idc_version == "v1" and rc.previous_version is None + assert rc.series_revised == rc.series_removed == 0 + assert rc.series_added > 0 + assert all(c.status == "new" for c in rc.collections) + + +def test_version_accepts_v_prefix(ctx): + cur = _current(ctx) + assert ctx.releases.release_changes(f"v{cur}") == ctx.releases.release_changes(cur) + + +@pytest.mark.parametrize("bad", [0, 9999, "latest", "v"]) +def test_unknown_version_is_a_clean_error(ctx, bad): + with pytest.raises(InvalidQueryError): + ctx.releases.release_changes(bad) + + +async def test_release_changes_parity(ctx, client, parse_mcp): + v = _current(ctx) - 1 + core = ctx.releases.release_changes(v).model_dump(mode="json") + rest = client.get("/v3/releases/changes", params={"version": v}).json() + mcp_out = parse_mcp(await mcp.call_tool("get_release_changes", {"version": v})) + assert core == rest == mcp_out + + +def test_rest_unknown_version_is_400(client): + r = client.get("/v3/releases/changes", params={"version": 9999}) + assert r.status_code == 400 + + +# --- discoverability ------------------------------------------------------------------------ + + +async def test_version_tool_points_at_release_history(): + tools = {t.name: t for t in await mcp.list_tools()} + assert "get_release_changes" in tools + assert "get_release_changes" in tools["get_idc_version"].description + desc = " ".join(tools["get_release_changes"].description.split()).lower() + for word in ("new", "changed", "added", "removed", "since"): + assert word in desc + + +async def test_whats_new_prompt(): + prompts = {p.name for p in await mcp.list_prompts()} + assert "whats_new" in prompts + msg = await mcp.get_prompt("whats_new", {"version": "v24"}) + assert "get_release_changes(version=24)" in msg.messages[0].content.text + + +def test_prior_versions_index_is_documented(): + sch = schema.table_schema("prior_versions_index") + assert sch["description"] + cols = {c["name"]: c["description"] for c in sch["columns"]} + assert cols["min_idc_version"] and cols["max_idc_version"] + + +def test_notable_columns_exist_in_schema(): + # list_tables drops unknown names silently, so catch curation drift here. + for table, cols in schema.NOTABLE_COLUMNS.items(): + names = {c["name"] for c in schema.table_schema(table)["columns"]} + assert set(cols) <= names, (table, set(cols) - names) + + +def test_list_tables_surfaces_version_columns(ctx): + tables = {t.name: t for t in ctx.query.list_tables().tables} + assert "series_init_idc_version" in tables["index"].notable_columns