diff --git a/ensembl-track-api.openapi.yaml b/ensembl-track-api.openapi.yaml index 3226cef..35981ce 100644 --- a/ensembl-track-api.openapi.yaml +++ b/ensembl-track-api.openapi.yaml @@ -26,6 +26,64 @@ servers: - url: http://www.ensembl.org/api/tracks description: Production server paths: + /transcriptomic/{genome_id}/configuration: + get: + summary: Returns prepared transcriptomic selections and static filter counts. + description: > + Resolved through the same dataset/release selection as track_categories, or + pinned with dataset_id. Counts are occurrences across + handover records, grouped case-insensitively, and do not change with + selection. Selections contain display and filtering metadata. Each product includes + a track_id referencing the registered track. File locations are returned + by /track/{track_id}, not by this configuration endpoint. Rendering settings may not yet be configured. + parameters: + - name: genome_id + in: path + required: true + schema: + type: string + format: uuid + - name: dataset_id + in: query + description: Pin a registered dataset for this genome; mutually exclusive with release. + schema: + type: string + format: uuid + - name: release + in: query + description: Select the catalogue as of this release; defaults to latest when neither parameter is given. + schema: + type: string + format: date + responses: + '400': + description: Invalid dataset UUID or conflicting query parameters. + '200': + description: Prepared coverage catalogue. + content: + application/json: + schema: + type: object + properties: + track_count: + type: integer + filters: + type: object + additionalProperties: + type: array + items: + type: object + properties: + value: + type: string + count: + type: integer + selections: + type: array + items: + $ref: '#/components/schemas/TranscriptomicSelection' + '404': + description: No coverage configuration for the requested genome. /track_categories/{genome_id}: get: summary: Returns all track categories (and tracks) for a given genome at a specific release. @@ -33,6 +91,8 @@ paths: Returns tracks filtered by release date and browser type. Automatically deduplicates tracks with overlapping specifications, keeping only the most recent version of each track type. + Configured transcriptomic categories use the same release selection and + return an empty track_list. parameters: - name: genome_id in: path @@ -264,6 +324,55 @@ components: - Variation - Regulation + TranscriptomicSelection: + type: object + additionalProperties: false + required: + - selection_id + - parent_sample_id + - display_label + - selection_level + - data_type + - metadata + - track_products + properties: + selection_id: + type: string + parent_sample_id: + type: string + display_label: + type: string + selection_level: + type: string + enum: [run] + data_type: + type: string + enum: [rnaseq] + metadata: + type: object + additionalProperties: + type: string + track_products: + type: array + minItems: 1 + maxItems: 1 + items: + $ref: '#/components/schemas/TranscriptomicTrackProduct' + + TranscriptomicTrackProduct: + type: object + additionalProperties: false + required: + - track_type + - track_id + properties: + track_type: + type: string + enum: [rnaseq_coverage] + track_id: + type: string + format: uuid + TrackList: type: object properties: diff --git a/tracks/management/commands/seed_transcriptomic_configuration.py b/tracks/management/commands/seed_transcriptomic_configuration.py new file mode 100644 index 0000000..25ecedd --- /dev/null +++ b/tracks/management/commands/seed_transcriptomic_configuration.py @@ -0,0 +1,178 @@ +"""Register the coverage pilot and its prepared discovery configuration.""" + +import json +import re +import uuid +from datetime import date +from pathlib import Path + +from django.core.management.base import BaseCommand, CommandError +from django.db import transaction + +from tracks.models import ( + Category, + DatasetRelease, + Specifications, + Track, + TranscriptomicConfiguration, +) +from tracks.transcriptomic import prepare_configuration + + +class Command(BaseCommand): + help = "Seed run-level coverage tracks and static counts from the Genebuild JSON." + + def add_arguments(self, parser): + parser.add_argument("--genome-id", required=True, type=uuid.UUID) + parser.add_argument("--records", required=True, type=Path) + parser.add_argument( + "--release", required=True, help="Release label, YYYY-MM-DD" + ) + parser.add_argument( + "--dataset-id", + type=uuid.UUID, + help="Existing dataset for an identical retry; omit to create a new version.", + ) + + def handle(self, *args, **options): + try: + records = json.loads(options["records"].read_text(encoding="utf-8-sig")) + prepared = prepare_configuration(records) + except (OSError, ValueError) as error: + raise CommandError(str(error)) from error + + # Validate options + genome_id = options["genome_id"] + + release = options["release"] + if not re.fullmatch(r"[0-9]{4}-[0-9]{2}(?:-[0-9]{2})?", release): + raise CommandError("--release must use YYYY-MM or YYYY-MM-DD.") + + try: + # Add a day only for calendar validation; preserve the release label. + date.fromisoformat(release if len(release) == 10 else f"{release}-01") + except ValueError as error: + raise CommandError( + "--release must contain a valid year, month and optional day." + ) from error + + dataset_id = options.get("dataset_id") or uuid.uuid4() + for record in records: + supplied_uuid = record["target_genome"].get("genome_uuid") + if supplied_uuid and supplied_uuid != str(genome_id): + raise CommandError( + "The supplied genome UUID conflicts with the handover." + ) + + # Seed the configuration and tracks + with transaction.atomic(): + category, _ = Category.objects.get_or_create( + track_category_id="transcriptomic", + defaults={"label": "Transcriptomic data", "type": "Genomic"}, + ) + + spec, _ = Specifications.objects.get_or_create( + name="rnaseq-coverage-genomebrowser", + defaults={ + "label": "RNA-seq coverage", + "category": category, + "browser": "GenomeBrowser", + "type": "regular", + "discovery_mode": "configured", + "files": ["rnaseq_coverage"], + "trigger": [], + "settings": {}, + "on_by_default": False, + "description": "Run-level RNA-seq coverage supplied by Genebuild.", + }, + ) + if ( + spec.category_id != category.pk + or spec.browser != "GenomeBrowser" + or spec.discovery_mode != "configured" + or spec.files != ["rnaseq_coverage"] + ): + raise CommandError( + "Existing coverage specification conflicts with this importer." + ) + + config, _ = TranscriptomicConfiguration.objects.get_or_create( + genome_id=genome_id, + dataset_id=dataset_id, + defaults={"specification": spec}, + ) + if config.specification_id != spec.pk: + raise CommandError( + "This genome already has a different configured specification." + ) + + # The existing selector has no tie-breaker for competing versions in one release. + competing = DatasetRelease.objects.filter( + genome_id=genome_id, + release_label=release, + dataset_id__in=Track.objects.filter( + genome_id=genome_id, + specifications=spec, + ).values("dataset_id"), + ).exclude(dataset_id=dataset_id) + if competing.exists(): + raise CommandError( + "A coverage dataset already exists for this release; retry with its --dataset-id." + ) + + for record in prepared["selections"]: + record["track_products"][0]["track_id"] = str( + uuid.uuid5( + config.dataset_id, record["selection_id"] + ":rnaseq_coverage" + ) + ) + + if config.configuration and config.configuration != prepared: + raise CommandError( + "Dataset content differs. Omit --dataset-id to create a new version at a new release." + ) + + created_count = 0 + for record in records: + product = record["track_products"][0] + + # Retained catalogue UUID gives stable track IDs across repeated seeds. + track_id = uuid.uuid5( + config.dataset_id, record["selection_id"] + ":rnaseq_coverage" + ) + + existing_track = Track.objects.filter(track_id=track_id).first() + if existing_track is not None and existing_track.datafiles != { + "rnaseq_coverage": product["track_file"] + }: + raise CommandError( + f"Track {track_id} already exists with a different file path. Omit --dataset-id to create a new version at a new release." + ) + + track, created = Track.objects.update_or_create( + track_id=track_id, + defaults={ + "genome_id": genome_id, + "dataset_id": config.dataset_id, + "datafiles": {"rnaseq_coverage": product["track_file"]}, + }, + ) + track.specifications.add(spec) + created_count += int(created) + + config.configuration = prepared + config.track_count = prepared["track_count"] + config.save(update_fields=["configuration", "track_count"]) + DatasetRelease.objects.get_or_create( + genome_id=genome_id, + dataset_id=config.dataset_id, + release_label=release, + ) + + self.stdout.write( + self.style.SUCCESS( + f"Seeded {config.track_count} tracks for genome {genome_id} " + f"({created_count} created); dataset {config.dataset_id}; release {release}. " + "File paths are preserved; trigger/settings remain unchanged." + ) + ) diff --git a/tracks/migrations/0004_transcriptomicconfiguration_and_more.py b/tracks/migrations/0004_transcriptomicconfiguration_and_more.py new file mode 100644 index 0000000..1d0138e --- /dev/null +++ b/tracks/migrations/0004_transcriptomicconfiguration_and_more.py @@ -0,0 +1,113 @@ +# Generated by Django 5.2.17 on 2026-09-29 09:26 + +import uuid +from typing import ClassVar + +import django.db.models.deletion +from django.db import migrations, models + +import tracks.fields + + +class Migration(migrations.Migration): + + dependencies: ClassVar = [ + ("tracks", "0003_query_indexes"), + ] + + operations: ClassVar = [ + migrations.CreateModel( + name="TranscriptomicConfiguration", + fields=[ + ( + "id", + models.AutoField( + auto_created=True, + primary_key=True, + serialize=False, + verbose_name="ID", + ), + ), + ("genome_id", tracks.fields.HyphenatedUUIDField()), + ( + "dataset_id", + tracks.fields.HyphenatedUUIDField( + default=uuid.uuid4, editable=False + ), + ), + ("track_count", models.PositiveIntegerField(default=0)), + ("configuration", models.JSONField(default=dict)), + ], + ), + migrations.RenameIndex( + model_name="datasetrelease", + new_name="tracks_data_genome__34d615_idx", + old_name="tracks_release_genome_label_idx", + ), + migrations.RenameIndex( + model_name="track", + new_name="tracks_trac_dataset_95bf62_idx", + old_name="tracks_track_dataset_idx", + ), + migrations.RenameIndex( + model_name="track", + new_name="tracks_trac_genome__f2675d_idx", + old_name="tracks_track_genome_dataset_idx", + ), + migrations.AddField( + model_name="specifications", + name="discovery_mode", + field=models.CharField( + choices=[("inline", "Inline"), ("configured", "Configured")], + default="inline", + max_length=20, + ), + ), + migrations.AlterField( + model_name="specifications", + name="browser", + field=models.CharField( + choices=[ + ("GenomeBrowser", "GenomeBrowser"), + ("StructuralVariant", "StructuralVariant"), + ], + max_length=20, + ), + ), + migrations.AlterField( + model_name="specifications", + name="strand", + field=models.CharField( + blank=True, + choices=[("forward", "forward"), ("reverse", "reverse")], + max_length=20, + null=True, + ), + ), + migrations.AlterField( + model_name="specifications", + name="type", + field=models.CharField( + choices=[ + ("gene", "gene"), + ("variant", "variant"), + ("regular", "regular"), + ], + max_length=8, + ), + ), + migrations.AddField( + model_name="transcriptomicconfiguration", + name="specification", + field=models.ForeignKey( + on_delete=django.db.models.deletion.PROTECT, to="tracks.specifications" + ), + ), + migrations.AddConstraint( + model_name="transcriptomicconfiguration", + constraint=models.UniqueConstraint( + fields=("genome_id", "dataset_id"), + name="unique_transcriptomic_genome_dataset", + ), + ), + ] diff --git a/tracks/models.py b/tracks/models.py index 5a282fe..7a94765 100644 --- a/tracks/models.py +++ b/tracks/models.py @@ -54,6 +54,10 @@ class BrowserType(models.TextChoices): GENOME_BROWSER = "GenomeBrowser", "GenomeBrowser" STRUCTURAL_VARIANT = "StructuralVariant", "StructuralVariant" + class DiscoveryMode(models.TextChoices): + INLINE = "inline", "Inline" + CONFIGURED = "configured", "Configured" + name = models.CharField(max_length=50, unique=True) label = models.CharField(max_length=50) @@ -89,6 +93,12 @@ class BrowserType(models.TextChoices): max_length=20, ) + discovery_mode = models.CharField( + max_length=20, + choices=DiscoveryMode.choices, + default=DiscoveryMode.INLINE, + ) + class Track(models.Model): specifications: models.ManyToManyField = models.ManyToManyField( @@ -144,3 +154,19 @@ class Meta: indexes: ClassVar = [ models.Index(fields=["genome_id", "-release_label"]), ] + + +class TranscriptomicConfiguration(models.Model): + genome_id = HyphenatedUUIDField() + dataset_id = HyphenatedUUIDField(default=uuid.uuid4, editable=False) + specification = models.ForeignKey(Specifications, on_delete=models.PROTECT) + track_count = models.PositiveIntegerField(default=0) + configuration = models.JSONField(default=dict) + + class Meta: + constraints: ClassVar = [ + models.UniqueConstraint( + fields=["genome_id", "dataset_id"], + name="unique_transcriptomic_genome_dataset", + ), + ] diff --git a/tracks/transcriptomic.py b/tracks/transcriptomic.py new file mode 100644 index 0000000..8b08f7d --- /dev/null +++ b/tracks/transcriptomic.py @@ -0,0 +1,112 @@ +"""Preparation of the current run-level RNA-seq coverage handover.""" + +import re +from collections import Counter, defaultdict + + +def prepare_configuration(records): + """Validate the coverage pilot and count records, preserving source metadata. + + This deliberately accepts the current string-valued run-level handover. + New selection levels and metadata shapes require an explicit extension. + """ + if not isinstance(records, list) or not records: + raise ValueError("Expected a non-empty list of selection records.") + + identifiers = set() + assemblies = set() + counts = defaultdict(Counter) + + for index, record in enumerate(records): + if not isinstance(record, dict): + raise TypeError(f"Record {index} must be an object.") + for key in ("selection_id", "parent_sample_id", "display_label"): + if not isinstance(record.get(key), str) or not record[key]: + raise ValueError(f"Record {index}: missing or invalid {key}.") + + selection_id = record["selection_id"] + if selection_id in identifiers: + raise ValueError(f"Duplicate selection_id: {selection_id}") + + identifiers.add(selection_id) + if ( + record.get("schema_version") != "transcriptomic-selection-0.1" + or record.get("selection_level") != "run" + or record.get("data_type") != "rnaseq" + ): + raise ValueError( + f"{selection_id}: expected a current run-level RNA-seq record." + ) + + genome = record.get("target_genome") + if not isinstance(genome, dict) or not all( + isinstance(genome.get(key), str) and genome[key] + for key in ("species", "assembly") + ): + raise ValueError(f"{selection_id}: missing target species/assembly.") + + assemblies.add(genome["assembly"]) + provenance = record.get("provenance", {}) + if ( + not isinstance(provenance, dict) + or provenance.get("biosample_accession") != record["parent_sample_id"] + or provenance.get("run_accessions") != [selection_id] + ): + raise ValueError(f"{selection_id}: inconsistent sample/run provenance.") + + products = record.get("track_products") + if not isinstance(products, list) or len(products) != 1: + raise ValueError(f"{selection_id}: expected one coverage product.") + product = products[0] + if not isinstance(product, dict) or ( + product.get("track_type"), + product.get("format"), + product.get("status"), + ) != ("rnaseq_coverage", "BigWig", "available"): + raise ValueError(f"{selection_id}: expected an available coverage BigWig.") + if not isinstance(product.get("track_file"), str) or not product["track_file"]: + raise ValueError(f"{selection_id}: missing track_file.") + if not isinstance(product.get("md5"), str) or not re.fullmatch( + r"[0-9a-f]{32}", product["md5"] + ): + raise ValueError(f"{selection_id}: missing or invalid md5.") + + metadata = record.get("metadata") + if not isinstance(metadata, dict) or any( + not isinstance(v, str) for v in metadata.values() + ): + raise ValueError( + f"{selection_id}: this importer requires string metadata values." + ) + + for key, value in metadata.items(): + counts[key][value.casefold()] += 1 + + if len(assemblies) != 1: + raise ValueError("One genome configuration must contain exactly one assembly.") + + return { + "track_count": len(records), + "filters": { + key: [ + {"value": value, "count": count} + for value, count in sorted(values.items()) + ] + for key, values in sorted(counts.items()) + }, + "selections": [ + { + "selection_id": record["selection_id"], + "parent_sample_id": record["parent_sample_id"], + "display_label": record["display_label"], + "selection_level": record["selection_level"], + "data_type": record["data_type"], + "metadata": dict(record["metadata"]), + "track_products": [ + {"track_type": product["track_type"]} + for product in record["track_products"] + ], + } + for record in records + ], + } diff --git a/tracks/urls.py b/tracks/urls.py index dfea7fe..6b5fa61 100644 --- a/tracks/urls.py +++ b/tracks/urls.py @@ -34,4 +34,9 @@ # New ingress endpoints path("tracks/create", views.CreateTrack.as_view(), name="create_track"), path("tracks/link_type", views.LinkTypeToTrack.as_view(), name="link_type"), + path( + "transcriptomic//configuration", + views.TranscriptomicConfigurationView.as_view(), + name="transcriptomic_configuration", + ), ] diff --git a/tracks/views.py b/tracks/views.py index 9835e78..282a5ab 100644 --- a/tracks/views.py +++ b/tracks/views.py @@ -24,7 +24,12 @@ from rest_framework.response import Response from rest_framework.views import APIView -from tracks.models import DatasetRelease, Specifications, Track +from tracks.models import ( + DatasetRelease, + Specifications, + Track, + TranscriptomicConfiguration, +) from tracks.serializers import ( CategorySerializer, CreateTrackSerializer, @@ -239,6 +244,39 @@ def combine_track_and_specification( return data +def get_transcriptomic_category(genome_id: str, browser: str, dataset_ids): + if browser != "GenomeBrowser": + return None + + configuration = ( + TranscriptomicConfiguration.objects.filter( + genome_id=genome_id, + dataset_id__in=dataset_ids, + track_count__gt=0, + specification__browser=browser, + specification__discovery_mode="configured", + ) + .select_related("specification__category") + .defer("configuration") + .first() + ) + + if ( + configuration is None + or not Track.objects.filter( + genome_id=genome_id, + dataset_id=configuration.dataset_id, + specifications=configuration.specification, + ).exists() + ): + return None + + return { + **CategorySerializer(configuration.specification.category).data, + "track_list": [], + } + + # ── Views ───────────────────────────────────────────────────────────────────── @@ -260,19 +298,20 @@ class GenomeTrackList(APIView): def get(self, request, genome_id): browser = request.query_params.get("browser", "GenomeBrowser") release_param = request.query_params.get("release") + # Validate browser if browser not in ["GenomeBrowser", "StructuralVariant"]: return Response( {"error": "browser must be 'GenomeBrowser' or 'StructuralVariant'"}, status=status.HTTP_400_BAD_REQUEST, ) + try: # Step 1: Determine target release target_release = get_target_release(genome_id, release_param) # Step 2: Get all datasets up to target release datasets = get_datasets_up_to_release(genome_id, target_release) - if not datasets: return Response( {"error": "No datasets found for this genome and release."}, @@ -289,14 +328,14 @@ def get(self, request, genome_id): # Step 5: Select latest dataset from each bin selected_dataset_ids = select_latest_dataset_from_bins(bins) - # Step 6: Get all tracks from selected datasets + # Step 6: Get all inline tracks from selected datasets tracks = Track.objects.filter( genome_id=genome_id, dataset_id__in=selected_dataset_ids ).prefetch_related( Prefetch( "specifications", queryset=Specifications.objects.filter( - browser=browser + browser=browser, discovery_mode="inline" ).select_related("category"), to_attr="browser_specifications", ), @@ -310,7 +349,7 @@ def get(self, request, genome_id): status=status.HTTP_404_NOT_FOUND, ) - # Step 7: For each track, get specification and group by category + # Step 7: For each inline track, get specification and group by category categories = {} for track in tracks: @@ -335,17 +374,36 @@ def get(self, request, genome_id): track_data = combine_track_and_specification(track, spec) categories[category_id]["track_list"].append(track_data) + # Sort track_list by display_order within each category + for cat_data in categories.values(): + cat_data["track_list"].sort(key=lambda x: x["display_order"]) + + # Step 8: Get transcriptomic category + configured_category = get_transcriptomic_category( + genome_id, browser, selected_dataset_ids + ) + if configured_category is not None: + existing = next( + ( + category + for category in categories.values() + if category["track_category_id"] + == configured_category["track_category_id"] + ), + None, + ) + if existing is None: + categories[configured_category["track_category_id"]] = ( + configured_category + ) + if not categories: return Response( {"error": "No tracks found for this genome."}, status=status.HTTP_404_NOT_FOUND, ) - # Sort track_list by display_order within each category - for cat_data in categories.values(): - cat_data["track_list"].sort(key=lambda x: x["display_order"]) - - # Step 8: Return + # Step 9: Return return Response( {"track_categories": list(categories.values())}, status=status.HTTP_200_OK, @@ -512,3 +570,77 @@ def post(self, request): {"error": "Validation failed", "details": serializer.errors}, status=status.HTTP_400_BAD_REQUEST, ) + + +class TranscriptomicConfigurationView(APIView): + """ + Return the prepared catalogue for a pinned dataset or resolved release. + """ + + http_method_names: list[str] = ["get"] # noqa: RUF012 + + @redis_cache( + "transcriptomic_configuration", + params=(("dataset_id", None), ("release", None)), + ) + def get(self, request, genome_id): + dataset_param = request.query_params.get("dataset_id") + release_param = request.query_params.get("release") + if dataset_param is not None and release_param is not None: + return Response( + {"error": "Use dataset_id or release, not both."}, status=400 + ) + if dataset_param is not None: + try: + dataset_id = UUID(dataset_param) + except ValueError: + return Response({"error": "dataset_id must be a UUID."}, status=400) + # A pinned link keeps resolving its historical configuration after a new release. + dataset_ids = list( + DatasetRelease.objects.filter( + genome_id=genome_id, + dataset_id=dataset_id, + ).values_list("dataset_id", flat=True) + ) + else: + try: + target_release = get_target_release(genome_id, release_param) + except ValueError: + return Response( + {"error": "No releases found for this genome."}, + status=status.HTTP_404_NOT_FOUND, + ) + + datasets = get_datasets_up_to_release(genome_id, target_release) + if not datasets: + return Response( + {"error": "No datasets found for this genome and release."}, + status=status.HTTP_404_NOT_FOUND, + ) + + dataset_specs = get_specifications_for_datasets( + [dataset["dataset_id"] for dataset in datasets], + "GenomeBrowser", + ) + bins = bin_datasets_by_overlapping_specs(datasets, dataset_specs) + dataset_ids = select_latest_dataset_from_bins(bins) + + configuration = ( + TranscriptomicConfiguration.objects.filter( + genome_id=genome_id, + dataset_id__in=dataset_ids, + track_count__gt=0, + specification__browser="GenomeBrowser", + specification__discovery_mode="configured", + ) + .values_list("configuration", flat=True) + .first() + ) + if configuration is None: + return Response( + { + "error": "No transcriptomic configuration for this genome and dataset/release." + }, + status=status.HTTP_404_NOT_FOUND, + ) + return Response(configuration)