Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
99 changes: 99 additions & 0 deletions bbconf/alembic/versions/c7f3a9e1b204_added_analysis_files_table.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,99 @@
"""Added analysis_files table

Revision ID: c7f3a9e1b204
Revises: 845d978eac7d
Create Date: 2026-08-17 21:00:00.000000

"""

from typing import Sequence, Union

import sqlalchemy as sa
from alembic import op
from sqlalchemy.dialects import postgresql

# revision identifiers, used by Alembic.
revision: str = "c7f3a9e1b204"
down_revision: Union[str, None] = "845d978eac7d"
branch_labels: Union[str, Sequence[str], None] = None
depends_on: Union[str, Sequence[str], None] = None


def upgrade() -> None:
"""Upgrade schema."""
op.create_table(
"analysis_files",
sa.Column("id", sa.Integer(), autoincrement=True, nullable=False),
sa.Column(
"name",
sa.String(),
nullable=False,
comment="Logical name/key, e.g. openSignalMatrix",
),
sa.Column(
"file_path",
sa.String(),
nullable=False,
comment="S3 object key, relative to the bucket root",
),
sa.Column(
"file_type",
sa.String(),
nullable=True,
comment="Category, e.g. openSignalMatrix | reference | model",
),
sa.Column(
"genome",
sa.String(),
nullable=True,
comment="Genome/assembly, e.g. hg38 (optional)",
),
sa.Column("description", sa.String(), nullable=True),
sa.Column(
"tags",
postgresql.ARRAY(sa.String()),
nullable=True,
comment="Free-form tags",
),
sa.Column(
"file_size",
sa.Integer(),
nullable=True,
comment="Size of the file in bytes",
),
sa.Column("checksum", sa.String(), nullable=True, comment="SHA256 of the file"),
sa.Column(
"creation_date",
sa.TIMESTAMP(timezone=True),
nullable=False,
comment="Upload date",
),
sa.PrimaryKeyConstraint("id"),
)
op.create_index(
op.f("ix_analysis_files_id"), "analysis_files", ["id"], unique=False
)
op.create_index(
op.f("ix_analysis_files_name"), "analysis_files", ["name"], unique=False
)
op.create_index(
op.f("ix_analysis_files_file_type"),
"analysis_files",
["file_type"],
unique=False,
)
op.create_index(
op.f("ix_analysis_files_genome"),
"analysis_files",
["genome"],
unique=False,
)


def downgrade() -> None:
"""Downgrade schema."""
op.drop_index(op.f("ix_analysis_files_genome"), table_name="analysis_files")
op.drop_index(op.f("ix_analysis_files_file_type"), table_name="analysis_files")
op.drop_index(op.f("ix_analysis_files_name"), table_name="analysis_files")
op.drop_index(op.f("ix_analysis_files_id"), table_name="analysis_files")
op.drop_table("analysis_files")
6 changes: 6 additions & 0 deletions bbconf/bbagent.py
Original file line number Diff line number Diff line change
Expand Up @@ -35,6 +35,7 @@
UsageModel,
UsageStats,
)
from bbconf.modules.analysis_files import BedAgentAnalysisFile
from bbconf.modules.bedfiles import BedAgentBedFile
from bbconf.modules.bedsets import BedAgentBedSet
from bbconf.modules.objects import BBObjects
Expand Down Expand Up @@ -66,6 +67,7 @@ def __init__(
self._bedset = BedAgentBedSet(self.config)
self._objects = BBObjects(self.config)
self._snapshot = BedAgentSnapshot(self.config)
self._analysis_files = BedAgentAnalysisFile(self.config)

# get_stats() runs three uncached COUNT queries on the multi-hundred-
# thousand-row bed table and is called on hot paths (the stats endpoint
Expand All @@ -91,6 +93,10 @@ def objects(self) -> BBObjects:
def snapshot(self) -> BedAgentSnapshot:
return self._snapshot

@property
def analysis_files(self) -> BedAgentAnalysisFile:
return self._analysis_files

def __repr__(self) -> str:
repr = f"BedBaseAgent(config={self.config})"
repr += f"\n{self.bed}"
Expand Down
42 changes: 42 additions & 0 deletions bbconf/db_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -718,6 +718,48 @@ class BedSnapshot(Base):
)


class AnalysisFile(Base):
"""
Registry of standalone analysis files (openSignalMatrix, models, other
analysis inputs) stored in S3. Not tied to any bed file or bedset.

Append-only: one row per uploaded file, so name-based lookups resolve the
newest matching row (same model as ``bed_snapshots``). This is a new table,
so ``Base.metadata.create_all()`` creates it on the next connection.
"""

__tablename__ = "analysis_files"

id: Mapped[int] = mapped_column(primary_key=True, index=True, autoincrement=True)
name: Mapped[str] = mapped_column(
nullable=False, index=True, comment="Logical name/key, e.g. openSignalMatrix"
)
file_path: Mapped[str] = mapped_column(
nullable=False, comment="S3 object key, relative to the bucket root"
)
file_type: Mapped[Optional[str]] = mapped_column(
nullable=True,
index=True,
comment="Category, e.g. openSignalMatrix | reference | model",
)
genome: Mapped[Optional[str]] = mapped_column(
nullable=True, index=True, comment="Genome/assembly, e.g. hg38 (optional)"
)
description: Mapped[Optional[str]] = mapped_column(nullable=True)
tags: Mapped[Optional[list]] = mapped_column(
ARRAY(String), nullable=True, comment="Free-form tags"
)
file_size: Mapped[Optional[int]] = mapped_column(
nullable=True, comment="Size of the file in bytes"
)
checksum: Mapped[Optional[str]] = mapped_column(
nullable=True, comment="SHA256 of the file"
)
creation_date: Mapped[datetime.datetime] = mapped_column(
default=deliver_update_date, comment="Upload date"
)


class BaseEngine:
"""
A class with base methods, that are used in several classes.
Expand Down
7 changes: 7 additions & 0 deletions bbconf/exceptions.py
Original file line number Diff line number Diff line change
Expand Up @@ -77,6 +77,13 @@ class SnapshotNotFoundError(BedBaseConfError):
pass


class AnalysisFileNotFoundError(BedBaseConfError):
"""
Error type for missing analysis file"""

pass


class UniverseNotFoundError(BedBaseConfError):
"""
Error type for missing universe"""
Expand Down
33 changes: 33 additions & 0 deletions bbconf/models/base_models.py
Original file line number Diff line number Diff line change
Expand Up @@ -131,3 +131,36 @@ class BedSnapshotResult(BaseModel):
class BedSnapshotListResult(BaseModel):
count: int
results: list[BedSnapshotResult]


class AnalysisFileArtifact(BaseModel):
"""A standalone analysis file to publish (upload to S3 + record in the database)."""

path: str # local file path to upload
name: str
file_type: str | None = None
genome: str | None = None
description: str | None = None
tags: list[str] | None = None
file_size: int | None = None
checksum: str | None = None


class AnalysisFileResult(BaseModel):
"""One registered standalone analysis file."""

id: int | None = None
name: str
file_path: str
file_type: str | None = None
genome: str | None = None
description: str | None = None
tags: list[str] | None = None
file_size: int | None = None
checksum: str | None = None
creation_date: datetime.datetime


class AnalysisFileListResult(BaseModel):
count: int
results: list[AnalysisFileResult]
Loading