Skip to content

Commit bbd0c2e

Browse files
committed
Fix cacao gene ID validation bypass and simplify eFP query service
query_efp_database_dynamic() validated cacao gene IDs by species alone, but cacao is split across three patterns (ccn/sca/tc) with no single "efp_cacao" entry in the registry, so validation silently passed any string through. Switch to the per-database lookup, which resolves the correct pattern. Also removes an unused EFPDataService class wrapper (nothing called it, only the plain module-level functions), collapses a multi-engine candidate loop down to a single lookup now that sqlite-mirror fallback is gone, and collapses eFP-utils' 8 near-identical per-species gene validation blocks into one. Trims docstrings on newly-introduced code to one line each.
1 parent d865d41 commit bbd0c2e

9 files changed

Lines changed: 227 additions & 501 deletions

File tree

‎api/models/efp_dynamic.py‎

Lines changed: 3 additions & 44 deletions
Original file line numberDiff line numberDiff line change
@@ -1,10 +1,4 @@
1-
"""
2-
Dynamic SQLAlchemy model generation for all eFP databases.
3-
4-
At import time, one ORM model class is generated per database entry in
5-
SIMPLE_EFP_DATABASE_SCHEMAS and stored in SIMPLE_EFP_SAMPLE_MODELS.
6-
This replaces ~1,984 lines of hand-written boilerplate with a single registry.
7-
"""
1+
"""Generates one SQLAlchemy model class per database in SIMPLE_EFP_DATABASE_SCHEMAS at import time, instead of hand-writing each one."""
82

93
from __future__ import annotations
104

@@ -18,23 +12,7 @@
1812

1913

2014
def _to_sqla_type(column_spec):
21-
"""
22-
Map a column specification dictionary to a SQLAlchemy column type.
23-
24-
Converts the simple type descriptors used in schema definitions to the
25-
appropriate SQLAlchemy type objects for ORM model generation.
26-
27-
:param column_spec: Column specification with 'type', 'length', and 'unsigned' keys
28-
:type column_spec: Dict[str, Any]
29-
:return: SQLAlchemy column type (String, Integer, Float, or Text)
30-
:rtype: sqlalchemy.types.TypeEngine
31-
:raises ValueError: If column type is not one of: string, integer, float, text
32-
33-
Example::
34-
35-
col_spec = {"type": "string", "length": 24}
36-
sqla_type = _to_sqla_type(col_spec) # Returns String(24)
37-
"""
15+
"""Map a schema column spec ('type', 'length', 'unsigned') to a SQLAlchemy column type."""
3816
col_type = column_spec.get("type")
3917
if col_type == "string":
4018
return String(column_spec["length"])
@@ -50,26 +28,7 @@ def _to_sqla_type(column_spec):
5028

5129

5230
def _generate_model(bind_key: str, spec) -> db.Model:
53-
"""
54-
Build a concrete SQLAlchemy model class for the given schema specification.
55-
56-
Dynamically creates an ORM model with the specified table name, bind key,
57-
and columns based on the schema definition. The generated model class can
58-
be used like any Flask-SQLAlchemy model.
59-
60-
:param bind_key: Database bind key (e.g., 'cannabis', 'embryo')
61-
:type bind_key: str
62-
:param spec: Database schema specification from SIMPLE_EFP_DATABASE_SCHEMAS
63-
:type spec: Dict[str, Any]
64-
:return: Dynamically generated SQLAlchemy model class
65-
:rtype: db.Model
66-
67-
Example::
68-
69-
schema = SIMPLE_EFP_DATABASE_SCHEMAS['cannabis']
70-
CannabisModel = _generate_model('cannabis', schema)
71-
# Returns class: CannabisSampleData(db.Model)
72-
"""
31+
"""Build a concrete SQLAlchemy model class for the given database's schema spec."""
7332
attrs = {"__bind_key__": bind_key, "__tablename__": spec["table_name"]}
7433

7534
for column in spec["columns"]:

‎api/models/efp_schemas.py‎

Lines changed: 2 additions & 17 deletions
Original file line numberDiff line numberDiff line change
@@ -1,13 +1,4 @@
1-
"""
2-
Schema definitions for all eFP databases that expose a sample_data table.
3-
4-
Every database shares the same three-column structure:
5-
data_probeset_id (VARCHAR 255), data_signal (FLOAT), data_bot_id (VARCHAR 255).
6-
7-
Species and probeset/gene-model classification are sourced from combined_master.json.
8-
To add a new database, add its name to _DATABASE_NAMES (and an entry to combined_master.json
9-
if one doesn't already exist) — no other changes needed.
10-
"""
1+
"""Schema definitions for eFP databases sharing the same sample_data (data_probeset_id, data_signal, data_bot_id) table -- to add a database, just add its name to _DATABASE_NAMES."""
112

123
from __future__ import annotations
134

@@ -33,13 +24,7 @@
3324

3425

3526
def _schema(species: str, charset: str = "latin1") -> DatabaseSpec:
36-
"""Build a schema entry for one eFP database.
37-
38-
:param species: Species name stored in metadata (e.g., 'arabidopsis').
39-
:param charset: MySQL character set — 'latin1' for most, 'utf8mb4' for non-Latin labels.
40-
:returns: Full database schema dict ready for model generation.
41-
:rtype: DatabaseSpec
42-
"""
27+
"""Build a schema entry for one eFP database, defaulting to latin1 (utf8mb4 for non-Latin labels)."""
4328
return {
4429
**_SCHEMA_TEMPLATE,
4530
"charset": charset,

‎api/resources/gene_density.py‎

Lines changed: 3 additions & 19 deletions
Original file line numberDiff line numberDiff line change
@@ -1,17 +1,4 @@
1-
"""
2-
Gene density endpoint for the BAR API.
3-
4-
Returns per-bin gene density across all Arabidopsis thaliana chromosomes for a
5-
given bin size (in base pairs), as used by Eplant's ChromosomeView to colour
6-
chromosomes by gene density.
7-
8-
Reads: eplant2.tair10_gff3
9-
Writes: JSON — list of chromosomes with density arrays
10-
11-
Usage::
12-
13-
GET /gene_density?species=Arabidopsis_thaliana&bin_size=143061.51645207437
14-
"""
1+
"""Per-bin gene density across Arabidopsis chromosomes, used by Eplant's ChromosomeView to colour chromosomes by gene density."""
152

163
from flask import request
174
from flask_restx import Namespace, Resource
@@ -67,9 +54,7 @@ def get(self):
6754
start_bin_expr = func.floor(TAIR10GFF3.Start / bin_size)
6855
end_bin_expr = func.floor(TAIR10GFF3.End / bin_size)
6956

70-
# Aggregated query for single-bin genes (~98%+ of all genes at typical zoom levels).
71-
# FLOOR(start/binSize) == FLOOR(end/binSize) means the gene fits within one bin,
72-
# so GROUP BY is safe and avoids fetching one row per gene.
57+
# genes that fit in one bin (~98%+) can be counted with GROUP BY instead of fetched row by row
7358
single_bin_rows = db.session.execute(
7459
db.select(chr_expr, start_bin_expr, func.count())
7560
.where(
@@ -85,8 +70,7 @@ def get(self):
8570
if 0 <= idx < len(bins[chr_char]):
8671
bins[chr_char][idx] += cnt
8772

88-
# Individual rows for genes that span multiple bins (rare — typically <2% of genes).
89-
# Each such gene is counted once in every bin it spans, matching the original behaviour.
73+
# genes spanning multiple bins (rare) are counted once in every bin they touch
9074
multi_bin_rows = db.session.execute(
9175
db.select(chr_expr, start_bin_expr, end_bin_expr)
9276
.where(

‎api/resources/microarray_gene_expression.py‎

Lines changed: 3 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -67,11 +67,7 @@ def get(self, species="", gene_id=""):
6767
class GetDatabases(Resource):
6868
@microarray_gene_expression.param("species", _in="path", default="arabidopsis")
6969
def get(self, species=""):
70-
"""This endpoint returns the databases and views available for a given
71-
species, and how many of each -- sourced live from combined_master.json
72-
(data/efp_info/combined_master.json) instead of a hardcoded mapping, so
73-
it can't drift out of sync with the actual database/frontend catalog.
74-
"""
70+
"""Returns the databases and views available for a given species, sourced live from combined_master.json."""
7571
species = str(escape(species)).lower()
7672

7773
master = load_combined_master()
@@ -84,9 +80,7 @@ def get(self, species=""):
8480
if db_info["species"] == species
8581
}
8682

87-
# A database can be exposed under different view display names by
88-
# different frontend instances (efp vs eplant); collapse to a single
89-
# view_name -> database_name mapping like the endpoint has always returned.
83+
# collapse each database's views (efp/eplant may name them differently) into one view_name -> database_name mapping
9084
views = {}
9185
for db_name, db_info in databases.items():
9286
for used_by in db_info["used_by"]:
@@ -103,8 +97,6 @@ def get(self, species=""):
10397

10498
@microarray_gene_expression.route("/<string:species>/<string:view>/samples")
10599
class GetSamples1(Resource):
106-
"""This endpoint returns control and sample group mappings for a given species and view (or all views)"""
107-
108100
@microarray_gene_expression.param("species", _in="path", default="arabidopsis")
109101
@microarray_gene_expression.param("view", _in="path", default="Abiotic_Stress")
110102
def get(self, species="", view=""):
@@ -122,8 +114,7 @@ def get(self, species="", view=""):
122114
if db_info["species"] == species
123115
}
124116

125-
# Collapse every database's views to a single view_name -> {database,
126-
# platform, groups} mapping, same view-name normalization as /databases.
117+
# same view_name -> {database, platform, groups} collapsing as /databases
127118
all_views = {}
128119
for db_name, db_info in species_databases.items():
129120
for view_info in db_info["views"].values():

0 commit comments

Comments
 (0)