Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
109 changes: 109 additions & 0 deletions ensembl-track-api.openapi.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -26,13 +26,73 @@ servers:
- url: http://www.ensembl.org/api/tracks
description: Production server
paths:
/transcriptomic/{genome_id}/configuration:
get:
summary: Returns prepared transcriptomic selections and static filter counts.
description: >
Resolved through the same dataset/release selection as track_categories, or
pinned with dataset_id. Counts are occurrences across
handover records, grouped case-insensitively, and do not change with
selection. Selections contain display and filtering metadata. Each product includes
a track_id referencing the registered track. File locations are returned
by /track/{track_id}, not by this configuration endpoint. Rendering settings may not yet be configured.
parameters:
- name: genome_id
in: path
required: true
schema:
type: string
format: uuid
- name: dataset_id
in: query
description: Pin a registered dataset for this genome; mutually exclusive with release.
schema:
type: string
format: uuid
- name: release
in: query
description: Select the catalogue as of this release; defaults to latest when neither parameter is given.
schema:
type: string
format: date
responses:
'400':
description: Invalid dataset UUID or conflicting query parameters.
'200':
description: Prepared coverage catalogue.
content:
application/json:
schema:
type: object
properties:
track_count:
type: integer
filters:
type: object
additionalProperties:
type: array
items:
type: object
properties:
value:
type: string
count:
type: integer
selections:
type: array
items:
$ref: '#/components/schemas/TranscriptomicSelection'
'404':
description: No coverage configuration for the requested genome.
/track_categories/{genome_id}:
get:
summary: Returns all track categories (and tracks) for a given genome at a specific release.
description: >
Returns tracks filtered by release date and browser type.
Automatically deduplicates tracks with overlapping specifications,
keeping only the most recent version of each track type.
Configured transcriptomic categories use the same release selection and
return an empty track_list.
parameters:
- name: genome_id
in: path
Expand Down Expand Up @@ -264,6 +324,55 @@ components:
- Variation
- Regulation

TranscriptomicSelection:
type: object
additionalProperties: false
required:
- selection_id
- parent_sample_id
- display_label
- selection_level
- data_type
- metadata
- track_products
properties:
selection_id:
type: string
parent_sample_id:
type: string
display_label:
type: string
selection_level:
type: string
enum: [run]
data_type:
type: string
enum: [rnaseq]
metadata:
type: object
additionalProperties:
type: string
track_products:
type: array
minItems: 1
maxItems: 1
items:
$ref: '#/components/schemas/TranscriptomicTrackProduct'

TranscriptomicTrackProduct:
type: object
additionalProperties: false
required:
- track_type
- track_id
properties:
track_type:
type: string
enum: [rnaseq_coverage]
track_id:
type: string
format: uuid

TrackList:
type: object
properties:
Expand Down
178 changes: 178 additions & 0 deletions tracks/management/commands/seed_transcriptomic_configuration.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,178 @@
"""Register the coverage pilot and its prepared discovery configuration."""

import json
import re
import uuid
from datetime import date
from pathlib import Path

from django.core.management.base import BaseCommand, CommandError
from django.db import transaction

from tracks.models import (
Category,
DatasetRelease,
Specifications,
Track,
TranscriptomicConfiguration,
)
from tracks.transcriptomic import prepare_configuration


class Command(BaseCommand):
help = "Seed run-level coverage tracks and static counts from the Genebuild JSON."

def add_arguments(self, parser):
parser.add_argument("--genome-id", required=True, type=uuid.UUID)
parser.add_argument("--records", required=True, type=Path)
parser.add_argument(
"--release", required=True, help="Release label, YYYY-MM-DD"
)
parser.add_argument(
"--dataset-id",
type=uuid.UUID,
help="Existing dataset for an identical retry; omit to create a new version.",
)

def handle(self, *args, **options):
try:
records = json.loads(options["records"].read_text(encoding="utf-8-sig"))
prepared = prepare_configuration(records)
except (OSError, ValueError) as error:
raise CommandError(str(error)) from error

# Validate options
genome_id = options["genome_id"]

release = options["release"]
if not re.fullmatch(r"[0-9]{4}-[0-9]{2}(?:-[0-9]{2})?", release):
raise CommandError("--release must use YYYY-MM or YYYY-MM-DD.")

try:
# Add a day only for calendar validation; preserve the release label.
date.fromisoformat(release if len(release) == 10 else f"{release}-01")
except ValueError as error:
raise CommandError(
"--release must contain a valid year, month and optional day."
) from error

dataset_id = options.get("dataset_id") or uuid.uuid4()
for record in records:
supplied_uuid = record["target_genome"].get("genome_uuid")
if supplied_uuid and supplied_uuid != str(genome_id):
raise CommandError(
"The supplied genome UUID conflicts with the handover."
)

# Seed the configuration and tracks
with transaction.atomic():
category, _ = Category.objects.get_or_create(
track_category_id="transcriptomic",
defaults={"label": "Transcriptomic data", "type": "Genomic"},
)

spec, _ = Specifications.objects.get_or_create(
name="rnaseq-coverage-genomebrowser",
defaults={
"label": "RNA-seq coverage",
"category": category,
"browser": "GenomeBrowser",
"type": "regular",
"discovery_mode": "configured",
"files": ["rnaseq_coverage"],
"trigger": [],
"settings": {},
"on_by_default": False,
"description": "Run-level RNA-seq coverage supplied by Genebuild.",
},
)
if (
spec.category_id != category.pk
or spec.browser != "GenomeBrowser"
or spec.discovery_mode != "configured"
or spec.files != ["rnaseq_coverage"]
):
raise CommandError(
"Existing coverage specification conflicts with this importer."
)

config, _ = TranscriptomicConfiguration.objects.get_or_create(
genome_id=genome_id,
dataset_id=dataset_id,
defaults={"specification": spec},
)
if config.specification_id != spec.pk:
raise CommandError(
"This genome already has a different configured specification."
)

# The existing selector has no tie-breaker for competing versions in one release.
competing = DatasetRelease.objects.filter(
genome_id=genome_id,
release_label=release,
dataset_id__in=Track.objects.filter(
genome_id=genome_id,
specifications=spec,
).values("dataset_id"),
).exclude(dataset_id=dataset_id)
if competing.exists():
raise CommandError(
"A coverage dataset already exists for this release; retry with its --dataset-id."
)

for record in prepared["selections"]:
record["track_products"][0]["track_id"] = str(
uuid.uuid5(
config.dataset_id, record["selection_id"] + ":rnaseq_coverage"
)
)

if config.configuration and config.configuration != prepared:
raise CommandError(
"Dataset content differs. Omit --dataset-id to create a new version at a new release."
)

created_count = 0
for record in records:
product = record["track_products"][0]

# Retained catalogue UUID gives stable track IDs across repeated seeds.
track_id = uuid.uuid5(
config.dataset_id, record["selection_id"] + ":rnaseq_coverage"
)

existing_track = Track.objects.filter(track_id=track_id).first()
if existing_track is not None and existing_track.datafiles != {
"rnaseq_coverage": product["track_file"]
}:
raise CommandError(
f"Track {track_id} already exists with a different file path. Omit --dataset-id to create a new version at a new release."
)

track, created = Track.objects.update_or_create(
track_id=track_id,
defaults={
"genome_id": genome_id,
"dataset_id": config.dataset_id,
"datafiles": {"rnaseq_coverage": product["track_file"]},
},
)
track.specifications.add(spec)
created_count += int(created)

config.configuration = prepared
config.track_count = prepared["track_count"]
config.save(update_fields=["configuration", "track_count"])
DatasetRelease.objects.get_or_create(
genome_id=genome_id,
dataset_id=config.dataset_id,
release_label=release,
)

self.stdout.write(
self.style.SUCCESS(
f"Seeded {config.track_count} tracks for genome {genome_id} "
f"({created_count} created); dataset {config.dataset_id}; release {release}. "
"File paths are preserved; trigger/settings remain unchanged."
)
)
Loading
Loading