From e39749547fe43984ae78fcc92ed1d43ffd720104 Mon Sep 17 00:00:00 2001 From: Vin Date: Sun, 20 Sep 2026 19:51:54 -0400 Subject: [PATCH 01/13] Add arabidopsis_NIE_pseudobulk CI fixture from PR #327 Cherry-picks only the arabidopsis_NIE_pseudobulk dump. Its CREATE TABLE already matches the prod schema exactly, including data_signal_std. The other six pseudobulk dumps and all seven UMAP dumps from that branch are out of scope here and stay uncommitted. Contributions: Steven Qiao (took some of his code logic and I implemented it post Reena's major changes in PR #328) --- .../arabidopsis_NIE_pseudobulk_dump.sql | 63 +++++++++++++++++++ 1 file changed, 63 insertions(+) create mode 100644 config/databases/arabidopsis_NIE_pseudobulk_dump.sql diff --git a/config/databases/arabidopsis_NIE_pseudobulk_dump.sql b/config/databases/arabidopsis_NIE_pseudobulk_dump.sql new file mode 100644 index 0000000..6ad7a21 --- /dev/null +++ b/config/databases/arabidopsis_NIE_pseudobulk_dump.sql @@ -0,0 +1,63 @@ +-- MySQL dump 10.13 Distrib 9.4.0, for Linux (x86_64) +-- +-- Host: localhost Database: arabidopsis_NIE_pseudobulk +-- ------------------------------------------------------ +-- Server version 9.4.0 + +/*!40101 SET @OLD_CHARACTER_SET_CLIENT=@@CHARACTER_SET_CLIENT */; +/*!40101 SET @OLD_CHARACTER_SET_RESULTS=@@CHARACTER_SET_RESULTS */; +/*!40101 SET @OLD_COLLATION_CONNECTION=@@COLLATION_CONNECTION */; +/*!50503 SET NAMES utf8mb4 */; +/*!40103 SET @OLD_TIME_ZONE=@@TIME_ZONE */; +/*!40103 SET TIME_ZONE='+00:00' */; +/*!40014 SET @OLD_UNIQUE_CHECKS=@@UNIQUE_CHECKS, UNIQUE_CHECKS=0 */; +/*!40014 SET @OLD_FOREIGN_KEY_CHECKS=@@FOREIGN_KEY_CHECKS, FOREIGN_KEY_CHECKS=0 */; +/*!40101 SET @OLD_SQL_MODE=@@SQL_MODE, SQL_MODE='NO_AUTO_VALUE_ON_ZERO' */; +/*!40111 SET @OLD_SQL_NOTES=@@SQL_NOTES, SQL_NOTES=0 */; + +-- +-- Current Database: `arabidopsis_NIE_pseudobulk` +-- + +CREATE DATABASE /*!32312 IF NOT EXISTS*/ `arabidopsis_NIE_pseudobulk` /*!40100 DEFAULT CHARACTER SET latin1 */ /*!80016 DEFAULT ENCRYPTION='N' */; + +USE `arabidopsis_NIE_pseudobulk`; + +-- +-- Table structure for table `sample_data` +-- + +DROP TABLE IF EXISTS `sample_data`; +/*!40101 SET @saved_cs_client = @@character_set_client */; +/*!50503 SET character_set_client = utf8mb4 */; +CREATE TABLE `sample_data` ( + `data_probeset_id` varchar(16) NOT NULL, + `data_signal` float DEFAULT '0', + `data_signal_std` float DEFAULT '0', + `data_bot_id` varchar(64) NOT NULL, + KEY `data_probeset_id` (`data_probeset_id`,`data_bot_id`,`data_signal`) +) ENGINE=InnoDB DEFAULT CHARSET=latin1; +/*!40101 SET character_set_client = @saved_cs_client */; + +-- +-- Dumping data for table `sample_data` +-- + +LOCK TABLES `sample_data` WRITE; +/*!40000 ALTER TABLE `sample_data` DISABLE KEYS */; +INSERT INTO `sample_data` VALUES ('AT1G01010',0.0346533,0.224356,'D0_Mesophyll'),('AT1G01010',0.0251518,0.185286,'D0_Sieve element_responsive'),('AT1G01010',0.0297745,0.203837,'D0_Guard'),('AT1G01010',0.0550332,0.260401,'D0_Defense state'),('AT1G01010',0.0424581,0.239094,'D0_Epidermal'),('AT1G01010',0.0163655,0.146177,'D0_Phloem Parenchyma'),('AT1G01010',0.0389615,0.234467,'D0_Metabolic stress state'),('AT1G01010',0.0292404,0.198967,'D0_Phloem companion'),('AT1G01010',0.0510094,0.226118,'D0_Trichome'),('AT1G01010',0.027103,0.181704,'D0_Dividing'),('AT1G01010',0.0692417,0.276578,'D0_Stress responsive'),('AT1G01010',0.0110158,0.119154,'D0_Sugar metabolic state'),('AT1G01010',0.0565404,0.272046,'D0_Immune active'),('AT1G01010',0.0513775,0.273271,'D0_Hydathode'),('AT1G01010',0.045102,0.247266,'D0_Vascular'),('AT1G01010',0.0256331,0.169289,'D0_Myrosin'),('AT1G01010',0.0274436,0.177235,'W0_Vascular'),('AT1G01010',0.024174,0.172252,'W0_Mesophyll'),('AT1G01010',0.0213485,0.15354,'W0_Phloem Parenchyma'),('AT1G01010',0.0286998,0.178188,'W0_Dividing'),('AT1G01010',0.0278313,0.182473,'W0_Epidermal'),('AT1G01010',0.0563447,0.251333,'W0_Immune active'),('AT1G01010',0.0343054,0.201204,'W0_Sieve element_responsive'),('AT1G01010',0.0574457,0.253535,'W0_Defense state'),('AT1G01010',0.0136855,0.12117,'W0_Guard'),('AT1G01010',0.0272221,0.169833,'W0_Phloem companion'),('AT1G01010',0.0715313,0.271057,'W0_Stress responsive'),('AT1G01010',0.00820466,0.0673068,'W0_Myrosin'),('AT1G01010',0.0393423,0.199799,'W0_Sugar metabolic state'),('AT1G01010',0.0339956,0.217678,'W0_Metabolic stress state'),('AT1G01010',0.0214199,0.133693,'W0_Trichome'),('AT1G01010',0.0571251,0.236552,'W0_Hydathode'),('AT1G01010',0.0422542,0.250415,'W15_Dividing'),('AT1G01010',0.0446,0.254963,'W15_Mesophyll'),('AT1G01010',0.037824,0.236808,'W15_Phloem Parenchyma'),('AT1G01010',0.0420744,0.251218,'W15_Immune active'),('AT1G01010',0.0387755,0.241218,'W15_Epidermal'),('AT1G01010',0.104241,0.43291,'W15_Hydathode'),('AT1G01010',0.104467,0.324257,'W15_Stress responsive'),('AT1G01010',0.03282,0.228891,'W15_Guard'),('AT1G01010',0.0783275,0.325689,'W15_Defense state'),('AT1G01010',0.0205945,0.15169,'W15_Myrosin'),('AT1G01010',0.0290853,0.203305,'W15_Phloem companion'),('AT1G01010',0.0409241,0.246531,'W15_Vascular'),('AT1G01010',0,0,'W15_Trichome'),('AT1G01010',0.0304135,0.206024,'W15_Sieve element_responsive'),('AT1G01010',0,0,'W15_Sugar metabolic state'),('AT1G01010',0.0573618,0.262866,'W15_Metabolic stress state'),('AT1G01010',0.0187317,0.156752,'R15_Sieve element_responsive'),('AT1G01010',0.0287312,0.191696,'R15_Immune active'),('AT1G01010',0.0329913,0.213788,'R15_Mesophyll'),('AT1G01010',0.0130775,0.14382,'R15_Guard'),('AT1G01010',0.0547408,0.257222,'R15_Defense state'),('AT1G01010',0.0541722,0.243631,'R15_Stress responsive'),('AT1G01010',0.02531,0.180358,'R15_Phloem Parenchyma'),('AT1G01010',0.0400045,0.239473,'R15_Epidermal'),('AT1G01010',0.0300884,0.196556,'R15_Vascular'),('AT1G01010',0.0282214,0.199286,'R15_Dividing'),('AT1G01010',0.034919,0.214859,'R15_Phloem companion'),('AT1G01010',0.018792,0.157063,'R15_Metabolic stress state'),('AT1G01010',0.0528136,0.261553,'R15_Trichome'),('AT1G01010',0,0,'R15_Myrosin'),('AT1G01010',0.0564275,0.281679,'R15_Hydathode'),('AT1G01010',0.0462055,0.222434,'R15_Sugar metabolic state'); +INSERT INTO `sample_data` VALUES ('AT1G01010',0.0242853,0.114475,'W0_Phloem average'); +INSERT INTO `sample_data` VALUES ('AT1G01010',0.0228029,0.123446,'D0_Phloem average'); +INSERT INTO `sample_data` VALUES ('AT1G01010',0.0301145,0.140262,'R15_Phloem average'); +INSERT INTO `sample_data` VALUES ('AT1G01010',0.0334546,0.156053,'W15_Phloem average'); +/*!40000 ALTER TABLE `sample_data` ENABLE KEYS */; +UNLOCK TABLES; +/*!40103 SET TIME_ZONE=@OLD_TIME_ZONE */; +/*!40101 SET SQL_MODE=@OLD_SQL_MODE */; +/*!40014 SET FOREIGN_KEY_CHECKS=@OLD_FOREIGN_KEY_CHECKS */; +/*!40014 SET UNIQUE_CHECKS=@OLD_UNIQUE_CHECKS */; +/*!40101 SET CHARACTER_SET_CLIENT=@OLD_CHARACTER_SET_CLIENT */; +/*!40101 SET CHARACTER_SET_RESULTS=@OLD_CHARACTER_SET_RESULTS */; +/*!40101 SET COLLATION_CONNECTION=@OLD_COLLATION_CONNECTION */; +/*!40111 SET SQL_NOTES=@OLD_SQL_NOTES */; +-- Dump completed on 2026-07-17 01:32:31 From 05d5ffa81a0dbe84777038f0e9f7497c362df119 Mon Sep 17 00:00:00 2001 From: Vin Date: Sun, 20 Sep 2026 20:16:10 -0400 Subject: [PATCH 02/13] Catalog arabidopsis_NIE_pseudobulk under a new pseudobulk_std variant Adds a pseudobulk_std schema variant declaring sample_data's fourth column, data_signal_std, under a new 'value_std' role, plus one database entry using it. used_by/views are empty deliberately: GetDatabases and GetSamples1 in microarray_gene_expression.py filter the catalog on species alone, so any used_by entry here would inject a new key into those two live endpoints. Empty also matches the file's own invariant that used_by and views never disagree on emptiness across all 193 existing entries. source is 'super_viewer', a new value. This database is deployed but serves neither eFP nor ePlant, so it breaks the existing used_by-nonempty <-> active <-> source-names-a-frontend correlation whichever way it is written. A novel value fails loudly if code ever switches on the field; filing it as legacy_not_in_dropdown would be silently wrong and would group a live database with the known-broken hnahal/rohan/rpatel entries. --- data/efp_info/combined_master.json | 46 ++++++++++++++++++++++++++++++ 1 file changed, 46 insertions(+) diff --git a/data/efp_info/combined_master.json b/data/efp_info/combined_master.json index 642ed29..e8b1530 100644 --- a/data/efp_info/combined_master.json +++ b/data/efp_info/combined_master.json @@ -706,6 +706,31 @@ } } } + }, + "pseudobulk_std": { + "assigned_databases": 1, + "tables": { + "sample_data": { + "columns": { + "data_bot_id": { + "type": "varchar", + "role": "sample" + }, + "data_probeset_id": { + "type": "varchar", + "role": "gene_id" + }, + "data_signal": { + "type": "float", + "role": "value" + }, + "data_signal_std": { + "type": "float", + "role": "value_std" + } + } + } + } } }, "gene_id_patterns": { @@ -1393,6 +1418,27 @@ "schema_source": "prod_information_schema", "status": "active" }, + "arabidopsis_NIE_pseudobulk": { + "species": "arabidopsis", + "source": "super_viewer", + "platform": "rna_seq", + "schema_variant": "pseudobulk_std", + "schema_verified": true, + "value_semantics": { + "quantity": null, + "unit": null, + "scale": null + }, + "identifier_type": "gene_model", + "gene_id_pattern": "arabidopsis", + "used_by": [], + "views": {}, + "tables_present": [ + "sample_data" + ], + "schema_source": "prod_information_schema", + "status": "active" + }, "arabidopsis_ecotypes": { "species": "arabidopsis", "source": "efp", From 59d84b083b94adc7e373d605fd78d1f22b469677 Mon Sep 17 00:00:00 2001 From: Vin Date: Sun, 20 Sep 2026 20:16:10 -0400 Subject: [PATCH 03/13] Make _sample_data_model variant-aware for optional sample_data columns The generator hardcoded the same three columns for every catalogued database. It now also maps columns whose schema_variant declares a role listed in _OPTIONAL_COLUMNS_BY_ROLE, currently just 'value_std'. Driven by role rather than by 'every column the variant declares', so this is a genuine no-op. Verified by running the generator over all 194 databases: import api: OK -- generator ran for 194 databases models mapping data_signal_std: ['arabidopsis_NIE_pseudobulk'] models unchanged: 193 NIE cols: [data_probeset_id, data_bot_id, data_signal, data_signal_std] klepikova cols: [data_probeset_id, data_bot_id, data_signal] The three databases naming a variant absent from schema_variants fall back to {} and generate the three required columns exactly as before: rohan variant='legacy_microarray_projinfo' present=False -> 3 cols rpatel variant='legacy_microarray_projinfo' present=False -> 3 cols hnahal variant='legacy_..._minus_data_call_data_p_val' present=False -> 3 cols data_signal_std maps as FLOAT, nullable=True, primary_key=False. --- api/models/efp_dynamic.py | 53 ++++++++++++++++++++++++++++----------- 1 file changed, 38 insertions(+), 15 deletions(-) diff --git a/api/models/efp_dynamic.py b/api/models/efp_dynamic.py index 8d3ff38..1f20314 100644 --- a/api/models/efp_dynamic.py +++ b/api/models/efp_dynamic.py @@ -4,20 +4,43 @@ from api import db from api.utils.bar_utils import load_combined_master +# Optional sample_data columns, keyed by the schema_variants role that declares them. +# A database only gets one of these if its assigned variant declares a column with that +# role, so every variant predating this mapping generates exactly the columns it did +# before. Each value is a factory: mapped_column objects cannot be shared between models. +_OPTIONAL_COLUMNS_BY_ROLE = { + "value_std": lambda: db.mapped_column(db.Float, nullable=True), +} -def _sample_data_model(database): + +def _optional_columns(variant): + """Extra mapped columns this variant's sample_data declares, keyed by column name.""" + columns = variant.get("tables", {}).get("sample_data", {}).get("columns", {}) + return { + column: _OPTIONAL_COLUMNS_BY_ROLE[spec["role"]]() + for column, spec in columns.items() + if spec.get("role") in _OPTIONAL_COLUMNS_BY_ROLE + } + + +def _sample_data_model(database, variant): class_name = "".join(part.capitalize() for part in database.split("_")) + "SampleData" - return type( - class_name, - (db.Model,), - { - "__bind_key__": database, - "__tablename__": "sample_data", - "data_probeset_id": db.mapped_column(db.String(255), primary_key=True), - "data_bot_id": db.mapped_column(db.String(255), primary_key=True), - "data_signal": db.mapped_column(db.Float, primary_key=True), - }, - ) - - -SAMPLE_DATA_MODELS = {database: _sample_data_model(database) for database in load_combined_master()["databases"]} + attributes = { + "__bind_key__": database, + "__tablename__": "sample_data", + "data_probeset_id": db.mapped_column(db.String(255), primary_key=True), + "data_bot_id": db.mapped_column(db.String(255), primary_key=True), + "data_signal": db.mapped_column(db.Float, primary_key=True), + } + attributes.update(_optional_columns(variant)) + return type(class_name, (db.Model,), attributes) + + +_MASTER = load_combined_master() + +# A few catalogued databases name a schema_variant that is not in schema_variants; they +# fall back to {} and so generate the three required columns only, exactly as before. +SAMPLE_DATA_MODELS = { + database: _sample_data_model(database, _MASTER["schema_variants"].get(info["schema_variant"], {})) + for database, info in _MASTER["databases"].items() +} From 9bc969a5874559c4f2d8ff2ad627e9372645ee27 Mon Sep 17 00:00:00 2001 From: Vin Date: Sun, 20 Sep 2026 20:16:10 -0400 Subject: [PATCH 04/13] Expose data_signal_std as value_std for databases that declare it Selects the column and adds a third row key only when the model for that database actually maps data_signal_std, so every other database keeps the exact two-key row shape. --- api/resources/gene_expression.py | 16 ++++++++++++++-- 1 file changed, 14 insertions(+), 2 deletions(-) diff --git a/api/resources/gene_expression.py b/api/resources/gene_expression.py index 7e68749..ea642a1 100644 --- a/api/resources/gene_expression.py +++ b/api/resources/gene_expression.py @@ -54,19 +54,31 @@ def get(self, database, gene_id): query_id = rows[0][0] + # only databases whose schema_variant declares a standard deviation column have + # data_signal_std mapped; every other database keeps the two-key row shape + has_signal_std = hasattr(model, "data_signal_std") + columns = [model.data_bot_id, model.data_signal] + if has_signal_std: + columns.append(model.data_signal_std) + rows = db.session.execute( - db.select(model.data_bot_id, model.data_signal).where(func.upper(model.data_probeset_id) == query_id.upper()) + db.select(*columns).where(func.upper(model.data_probeset_id) == query_id.upper()) ).all() if len(rows) == 0: return BARUtils.error_exit("There are no data found for the given gene"), 400 + if has_signal_std: + data = [{"name": name, "value": str(value), "value_std": str(std)} for name, value, std in rows] + else: + data = [{"name": name, "value": str(value)} for name, value in rows] + res = { "gene_id": gene_id, "probset_id": query_id, "database": database, "record_count": len(rows), - "data": [{"name": name, "value": str(value)} for name, value in rows], + "data": data, } return BARUtils.success_exit(res) From ba9124cdf581b4e651c81e12c21d97af07d1683c Mon Sep 17 00:00:00 2001 From: Vin Date: Sun, 20 Sep 2026 20:16:39 -0400 Subject: [PATCH 05/13] Wire arabidopsis_NIE_pseudobulk into CI's MySQL bind and seed list Without the bind, CI cannot connect to the database no matter what the catalog says. The seed line uses the committed filename, _dump.sql. The seven pseudobulk lines on PR #327 that reference .sql rather than _dump.sql are not on dev, so there is no such bug here to fix. --- config/BAR_API.cfg | 1 + config/init.sh | 1 + 2 files changed, 2 insertions(+) diff --git a/config/BAR_API.cfg b/config/BAR_API.cfg index 950d4d4..908adcd 100755 --- a/config/BAR_API.cfg +++ b/config/BAR_API.cfg @@ -12,6 +12,7 @@ SQLALCHEMY_TRACK_MODIFICATIONS = False SQLALCHEMY_BINDS = { 'annotations_lookup': 'mysql://root:root@localhost/annotations_lookup', 'arabidopsis_ecotypes': 'mysql://root:root@localhost/arabidopsis_ecotypes', + 'arabidopsis_NIE_pseudobulk': 'mysql://root:root@localhost/arabidopsis_NIE_pseudobulk', 'arachis': 'mysql://root:root@localhost/arachis', 'cannabis': 'mysql://root:root@localhost/cannabis', 'canola_nssnp' : 'mysql://root:root@localhost/canola_nssnp', diff --git a/config/init.sh b/config/init.sh index e85d22c..ef2502a 100755 --- a/config/init.sh +++ b/config/init.sh @@ -11,6 +11,7 @@ echo "Welcome to the BAR API. Running init!" mysql -u $DB_USER -p$DB_PASS < ./config/databases/annotations_lookup.sql mysql -u $DB_USER -p$DB_PASS < ./config/databases/arabidopsis_ecotypes.sql +mysql -u $DB_USER -p$DB_PASS < ./config/databases/arabidopsis_NIE_pseudobulk_dump.sql mysql -u $DB_USER -p$DB_PASS < ./config/databases/arachis.sql mysql -u $DB_USER -p$DB_PASS < ./config/databases/cannabis.sql mysql -u $DB_USER -p$DB_PASS < ./config/databases/canola_nssnp.sql From 3661cfe5b1497f36157b6608d7d9283f26fd6c2b Mon Sep 17 00:00:00 2001 From: Vin Date: Sun, 20 Sep 2026 20:16:39 -0400 Subject: [PATCH 06/13] Test value_std exposure and the unchanged shape for other databases test_only_the_pseudobulk_model_maps_data_signal_std asserts the no-op over every generated model without needing a database. test_other_databases_keep_the_two_key_row_shape pins klepikova's exact top-level keys and per-row keys, so a regression that leaks value_std into another database's response fails rather than passing unnoticed. --- tests/resources/test_gene_expression.py | 40 +++++++++++++++++++++++++ 1 file changed, 40 insertions(+) diff --git a/tests/resources/test_gene_expression.py b/tests/resources/test_gene_expression.py index ebeed59..226cdac 100644 --- a/tests/resources/test_gene_expression.py +++ b/tests/resources/test_gene_expression.py @@ -6,6 +6,7 @@ from api import app, db from api.models.annotations_lookup import AtAgiLookup +from api.models.efp_dynamic import SAMPLE_DATA_MODELS class TestGeneExpression(TestCase): @@ -52,6 +53,45 @@ def test_probeset_database_accepts_probeset_directly(self): self.assertEqual(response.json["data"]["probset_id"], "261585_AT") +class TestSignalStdColumn(TestCase): + """data_signal_std is exposed only for databases whose schema_variant declares it.""" + + def setUp(self): + self.client = app.test_client() + + def test_only_the_pseudobulk_model_maps_data_signal_std(self): + """The variant-aware model generator has to stay a no-op for every other database.""" + with_std = {name for name, model in SAMPLE_DATA_MODELS.items() if hasattr(model, "data_signal_std")} + self.assertEqual(with_std, {"arabidopsis_NIE_pseudobulk"}) + + def test_pseudobulk_rows_carry_value_std(self): + response = self.client.get("/gene_expression/expression/arabidopsis_NIE_pseudobulk/AT1G01010") + self.assertEqual(response.status_code, 200) + rows = response.json["data"]["data"] + self.assertTrue(rows) + for row in rows: + self.assertEqual(set(row), {"name", "value", "value_std"}) + + def test_pseudobulk_value_std_is_the_stored_column(self): + """value_std must be data_signal_std, not a second copy of data_signal.""" + response = self.client.get("/gene_expression/expression/arabidopsis_NIE_pseudobulk/AT1G01010") + self.assertEqual(response.status_code, 200) + rows = {row["name"]: row for row in response.json["data"]["data"]} + self.assertIn("D0_Mesophyll", rows) + self.assertAlmostEqual(float(rows["D0_Mesophyll"]["value"]), 0.0346533, places=5) + self.assertAlmostEqual(float(rows["D0_Mesophyll"]["value_std"]), 0.224356, places=5) + + def test_other_databases_keep_the_two_key_row_shape(self): + """Guards the additive-only promise: no new key may appear for the other databases.""" + response = self.client.get("/gene_expression/expression/klepikova/AT1G01010") + self.assertEqual(response.status_code, 200) + payload = response.json["data"] + self.assertEqual(set(payload), {"gene_id", "probset_id", "database", "record_count", "data"}) + self.assertTrue(payload["data"]) + for row in payload["data"]: + self.assertEqual(set(row), {"name", "value"}) + + class TestAgiToProbesetConversion(TestCase): """Covers the AGI -> probeset lookup path for probeset-keyed databases.""" From 7c0587073cb5fcd31bfb133dd81f7a8c897a743c Mon Sep 17 00:00:00 2001 From: Vin Date: Sun, 20 Sep 2026 23:46:26 -0400 Subject: [PATCH 07/13] Add a Mean_CTRL 'fixture' row and pin its inline handling The committed dump previously predates Mean_CTRL, so AT1G01010 had 68 of its 69 rows. The added row's values are taken from an example row from prod: AT1G01010 | 0.036322 | 0.224118 | Mean_CTRL Endpoint now returns record_count 69 with Mean_CTRL as row 69 of the flat list. test_mean_ctrl_is_returned_inline_as_an_ordinary_row guards the design decision the endpoint rests on: Mean_CTRL is the eFP view XML's denominator but is stored like any other data_bot_id, so the endpoint returns it inline rather than partitioning it into a separate response key. PR #327's _split_rows() did the latter; that shape is deliberately not ported, and this test makes it a regression guard rather than a comment. --- .../arabidopsis_NIE_pseudobulk_dump.sql | 1 + tests/resources/test_gene_expression.py | 17 +++++++++++++++++ 2 files changed, 18 insertions(+) diff --git a/config/databases/arabidopsis_NIE_pseudobulk_dump.sql b/config/databases/arabidopsis_NIE_pseudobulk_dump.sql index 6ad7a21..3af0882 100644 --- a/config/databases/arabidopsis_NIE_pseudobulk_dump.sql +++ b/config/databases/arabidopsis_NIE_pseudobulk_dump.sql @@ -50,6 +50,7 @@ INSERT INTO `sample_data` VALUES ('AT1G01010',0.0242853,0.114475,'W0_Phloem aver INSERT INTO `sample_data` VALUES ('AT1G01010',0.0228029,0.123446,'D0_Phloem average'); INSERT INTO `sample_data` VALUES ('AT1G01010',0.0301145,0.140262,'R15_Phloem average'); INSERT INTO `sample_data` VALUES ('AT1G01010',0.0334546,0.156053,'W15_Phloem average'); +INSERT INTO `sample_data` VALUES ('AT1G01010',0.036322,0.224118,'Mean_CTRL'); /*!40000 ALTER TABLE `sample_data` ENABLE KEYS */; UNLOCK TABLES; /*!40103 SET TIME_ZONE=@OLD_TIME_ZONE */; diff --git a/tests/resources/test_gene_expression.py b/tests/resources/test_gene_expression.py index 226cdac..0b0e164 100644 --- a/tests/resources/test_gene_expression.py +++ b/tests/resources/test_gene_expression.py @@ -81,6 +81,23 @@ def test_pseudobulk_value_std_is_the_stored_column(self): self.assertAlmostEqual(float(rows["D0_Mesophyll"]["value"]), 0.0346533, places=5) self.assertAlmostEqual(float(rows["D0_Mesophyll"]["value_std"]), 0.224356, places=5) + def test_mean_ctrl_is_returned_inline_as_an_ordinary_row(self): + """Mean_CTRL is the eFP view XML's denominator, but it is stored like any + other data_bot_id. Rendering that distinction belongs to the view layer, so the + endpoint must not partition it out into a separate response key.""" + response = self.client.get("/gene_expression/expression/arabidopsis_NIE_pseudobulk/AT1G01010") + self.assertEqual(response.status_code, 200) + payload = response.json["data"] + self.assertEqual(set(payload), {"gene_id", "probset_id", "database", "record_count", "data"}) + + names = [row["name"] for row in payload["data"]] + self.assertIn("Mean_CTRL", names) + self.assertEqual(names.count("Mean_CTRL"), 1) + self.assertEqual(payload["record_count"], len(payload["data"])) + + mean_row = next(row for row in payload["data"] if row["name"] == "Mean_CTRL") + self.assertEqual(set(mean_row), {"name", "value", "value_std"}) + def test_other_databases_keep_the_two_key_row_shape(self): """Guards the additive-only promise: no new key may appear for the other databases.""" response = self.client.get("/gene_expression/expression/klepikova/AT1G01010") From c6e4f75c4e79be9ff9c37b0b5467da08e92e021e Mon Sep 17 00:00:00 2001 From: Vin Date: Mon, 21 Sep 2026 02:07:24 -0400 Subject: [PATCH 08/13] Add arabidopsis_NIE_umap CI fixture from PR #327 Copies only the arabidopsis_NIE_umap dump. Its umap_coords and umap_expression schema already matches prod exactly. The other six UMAP dumps from that branch are out of scope and stay uncommitted. Contributions: Steven Qiao (took some of his code logic and I implemented it post Reena's major changes in PR #328) --- config/databases/arabidopsis_NIE_umap.sql | 83 +++++++++++++++++++++++ 1 file changed, 83 insertions(+) create mode 100644 config/databases/arabidopsis_NIE_umap.sql diff --git a/config/databases/arabidopsis_NIE_umap.sql b/config/databases/arabidopsis_NIE_umap.sql new file mode 100644 index 0000000..ed5b9d6 --- /dev/null +++ b/config/databases/arabidopsis_NIE_umap.sql @@ -0,0 +1,83 @@ +-- MySQL dump 10.13 Distrib 9.4.0, for Linux (x86_64) +-- +-- Host: localhost Database: arabidopsis_NIE_umap +-- ------------------------------------------------------ +-- Server version 9.4.0 + +/*!40101 SET @OLD_CHARACTER_SET_CLIENT=@@CHARACTER_SET_CLIENT */; +/*!40101 SET @OLD_CHARACTER_SET_RESULTS=@@CHARACTER_SET_RESULTS */; +/*!40101 SET @OLD_COLLATION_CONNECTION=@@COLLATION_CONNECTION */; +/*!50503 SET NAMES utf8mb4 */; +/*!40103 SET @OLD_TIME_ZONE=@@TIME_ZONE */; +/*!40103 SET TIME_ZONE='+00:00' */; +/*!40014 SET @OLD_UNIQUE_CHECKS=@@UNIQUE_CHECKS, UNIQUE_CHECKS=0 */; +/*!40014 SET @OLD_FOREIGN_KEY_CHECKS=@@FOREIGN_KEY_CHECKS, FOREIGN_KEY_CHECKS=0 */; +/*!40101 SET @OLD_SQL_MODE=@@SQL_MODE, SQL_MODE='NO_AUTO_VALUE_ON_ZERO' */; +/*!40111 SET @OLD_SQL_NOTES=@@SQL_NOTES, SQL_NOTES=0 */; + +-- +-- Current Database: `arabidopsis_NIE_umap` +-- + +CREATE DATABASE /*!32312 IF NOT EXISTS*/ `arabidopsis_NIE_umap` /*!40100 DEFAULT CHARACTER SET latin1 */ /*!80016 DEFAULT ENCRYPTION='N' */; + +USE `arabidopsis_NIE_umap`; + +-- +-- Table structure for table `umap_coords` +-- + +DROP TABLE IF EXISTS `umap_coords`; +/*!40101 SET @saved_cs_client = @@character_set_client */; +/*!50503 SET character_set_client = utf8mb4 */; +CREATE TABLE `umap_coords` ( + `cell_id` INT NOT NULL, + `umap_1` FLOAT NOT NULL, + `umap_2` FLOAT NOT NULL, + `cell_type` VARCHAR(128) NOT NULL, + PRIMARY KEY (`cell_id`) +) ENGINE=InnoDB DEFAULT CHARSET=latin1; +/*!40101 SET character_set_client = @saved_cs_client */; + +-- +-- Dumping data for table `umap_coords` +-- + +LOCK TABLES `umap_coords` WRITE; +/*!40000 ALTER TABLE `umap_coords` DISABLE KEYS */; +INSERT INTO `umap_coords` VALUES (0,-3.696337,-0.631605,'Vascular'); +/*!40000 ALTER TABLE `umap_coords` ENABLE KEYS */; +UNLOCK TABLES; + +-- +-- Table structure for table `umap_expression` +-- + +DROP TABLE IF EXISTS `umap_expression`; +/*!40101 SET @saved_cs_client = @@character_set_client */; +/*!50503 SET character_set_client = utf8mb4 */; +CREATE TABLE `umap_expression` ( + `gene_id` VARCHAR(32) NOT NULL, + `expression` JSON NOT NULL, + PRIMARY KEY (`gene_id`) +) ENGINE=InnoDB DEFAULT CHARSET=latin1; +/*!40101 SET character_set_client = @saved_cs_client */; + +-- +-- Dumping data for table `umap_expression` +-- + +LOCK TABLES `umap_expression` WRITE; +/*!40000 ALTER TABLE `umap_expression` DISABLE KEYS */; +INSERT INTO `umap_expression` VALUES ('AT3G55980','{}'); +/*!40000 ALTER TABLE `umap_expression` ENABLE KEYS */; +UNLOCK TABLES; +/*!40103 SET TIME_ZONE=@OLD_TIME_ZONE */; +/*!40101 SET SQL_MODE=@OLD_SQL_MODE */; +/*!40014 SET FOREIGN_KEY_CHECKS=@OLD_FOREIGN_KEY_CHECKS */; +/*!40014 SET UNIQUE_CHECKS=@OLD_UNIQUE_CHECKS */; +/*!40101 SET CHARACTER_SET_CLIENT=@OLD_CHARACTER_SET_CLIENT */; +/*!40101 SET CHARACTER_SET_RESULTS=@OLD_CHARACTER_SET_RESULTS */; +/*!40101 SET COLLATION_CONNECTION=@OLD_COLLATION_CONNECTION */; +/*!40111 SET SQL_NOTES=@OLD_SQL_NOTES */; +-- Dump completed on 2026-07-17 01:12:53 From 0d84ae23b9ebe80665dda113a4f3e36ff52323f5 Mon Sep 17 00:00:00 2001 From: Vin Date: Mon, 21 Sep 2026 02:17:39 -0400 Subject: [PATCH 09/13] Replace arabidopsis_NIE_umap stub rows with prod-sourced fixture rows Four umap_coords cells (43-46) and two genes' expression restricted to those cells, with values taken from prod. I.e. take Steven's older dump and fill it with prod values. AT1G01010 has no key 45 because prod stores no zeros: that is the zero-expression case. AT3G18780 (ACT2) has all four. The two genes differ exactly at cell 45, so a query ignoring its WHERE clause, or a server filling in zeros, fails the tests. --- config/databases/arabidopsis_NIE_umap.sql | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/config/databases/arabidopsis_NIE_umap.sql b/config/databases/arabidopsis_NIE_umap.sql index ed5b9d6..59f495e 100644 --- a/config/databases/arabidopsis_NIE_umap.sql +++ b/config/databases/arabidopsis_NIE_umap.sql @@ -45,7 +45,7 @@ CREATE TABLE `umap_coords` ( LOCK TABLES `umap_coords` WRITE; /*!40000 ALTER TABLE `umap_coords` DISABLE KEYS */; -INSERT INTO `umap_coords` VALUES (0,-3.696337,-0.631605,'Vascular'); +INSERT INTO `umap_coords` VALUES (43,-6.89231,7.56863,'Metabolic stress state'),(44,2.87577,-4.82349,'Dividing'),(45,-0.262505,-8.55344,'Guard'),(46,-4.5876,6.05261,'Defense state'); /*!40000 ALTER TABLE `umap_coords` ENABLE KEYS */; UNLOCK TABLES; @@ -69,7 +69,7 @@ CREATE TABLE `umap_expression` ( LOCK TABLES `umap_expression` WRITE; /*!40000 ALTER TABLE `umap_expression` DISABLE KEYS */; -INSERT INTO `umap_expression` VALUES ('AT3G55980','{}'); +INSERT INTO `umap_expression` VALUES ('AT1G01010','{\"43\": 1.152292, \"44\": 1.546603, \"46\": 1.392931}'),('AT3G18780','{\"43\": 1.673516, \"44\": 3.142986, \"45\": 1.490115, \"46\": 1.953491}'); /*!40000 ALTER TABLE `umap_expression` ENABLE KEYS */; UNLOCK TABLES; /*!40103 SET TIME_ZONE=@OLD_TIME_ZONE */; From 32374730c99a704b9ec77abcc0216dd2bed88129 Mon Sep 17 00:00:00 2001 From: Vin Date: Mon, 21 Sep 2026 02:27:38 -0400 Subject: [PATCH 10/13] Add static models for arabidopsis_NIE_umap UmapCoords and UmapExpression match prod's umap_coords and umap_expression exactly. expression maps as db.JSON so it comes back as a dict, following FigureModels.data in gaia.py. --- api/models/arabidopsis_NIE_umap.py | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) create mode 100644 api/models/arabidopsis_NIE_umap.py diff --git a/api/models/arabidopsis_NIE_umap.py b/api/models/arabidopsis_NIE_umap.py new file mode 100644 index 0000000..d73edfa --- /dev/null +++ b/api/models/arabidopsis_NIE_umap.py @@ -0,0 +1,19 @@ +from api import db + + +class UmapCoords(db.Model): + __bind_key__ = "arabidopsis_NIE_umap" + __tablename__ = "umap_coords" + + cell_id: db.Mapped[int] = db.mapped_column(db.Integer, nullable=False, primary_key=True) + umap_1: db.Mapped[float] = db.mapped_column(db.Float, nullable=False) + umap_2: db.Mapped[float] = db.mapped_column(db.Float, nullable=False) + cell_type: db.Mapped[str] = db.mapped_column(db.String(128), nullable=False) + + +class UmapExpression(db.Model): + __bind_key__ = "arabidopsis_NIE_umap" + __tablename__ = "umap_expression" + + gene_id: db.Mapped[str] = db.mapped_column(db.String(32), nullable=False, primary_key=True) + expression: db.Mapped[dict] = db.mapped_column(db.JSON, nullable=False) From 61e2b0efa9d3d76cebe21433a908fa00ed790170 Mon Sep 17 00:00:00 2001 From: Vin Date: Mon, 21 Sep 2026 02:58:32 -0400 Subject: [PATCH 11/13] Add UMAP coordinates and sparse expression endpoints GET /umap_gene_expression/ returns every cell's coordinates keyed by cell_id. GET /umap_gene_expression// returns the gene's stored sparse expression object as-is, also keyed by cell_id, so the client joins the two directly and a cell with no expression key is a zero. Nothing is densified and no zeros are filled in. UMAPUtils.get_tables is the registry: one if/elif chain that sets the tables and the species literal per database, as the original RNASeqUtils did. The database name is compared exactly, never lowercased, since arabidopsis_NIE_umap is mixed-case. Gene IDs validate against gene_id_patterns[species], the same pattern arabidopsis_NIE_pseudobulk uses. --- api/__init__.py | 2 + api/resources/umap_gene_expression.py | 78 +++++++++++++++++++++++++++ 2 files changed, 80 insertions(+) create mode 100644 api/resources/umap_gene_expression.py diff --git a/api/__init__.py b/api/__init__.py index 1f12433..cb05b36 100644 --- a/api/__init__.py +++ b/api/__init__.py @@ -81,6 +81,7 @@ def create_app(): from api.resources.gaia import gaia from api.resources.rnaseq_gene_expression import rnaseq_gene_expression from api.resources.microarray_gene_expression import microarray_gene_expression + from api.resources.umap_gene_expression import umap_gene_expression from api.resources.proxy import bar_proxy from api.resources.thalemine import thalemine from api.resources.snps import snps @@ -98,6 +99,7 @@ def create_app(): bar_api.add_namespace(gaia) bar_api.add_namespace(rnaseq_gene_expression) bar_api.add_namespace(microarray_gene_expression) + bar_api.add_namespace(umap_gene_expression) bar_api.add_namespace(bar_proxy) bar_api.add_namespace(thalemine) bar_api.add_namespace(snps) diff --git a/api/resources/umap_gene_expression.py b/api/resources/umap_gene_expression.py new file mode 100644 index 0000000..683d136 --- /dev/null +++ b/api/resources/umap_gene_expression.py @@ -0,0 +1,78 @@ +from flask_restx import Namespace, Resource +from markupsafe import escape +from api import db +from api.models.arabidopsis_NIE_umap import UmapCoords as ArabidopsisNIEUmapCoords +from api.models.arabidopsis_NIE_umap import UmapExpression as ArabidopsisNIEUmapExpression +from api.utils.bar_utils import BARUtils, load_combined_master + +umap_gene_expression = Namespace( + "UMAP Gene Expression", + description="UMAP coordinates and single cell gene expression data from the BAR Databases", + path="/umap_gene_expression", +) + + +class UMAPUtils: + @staticmethod + def get_tables(database): + """This function sets the tables and species for a UMAP database + :param database: name of BAR database + :return: dict with the coordinates table, expression table and species + """ + # Set database + if database == "arabidopsis_NIE_umap": + coords_table = ArabidopsisNIEUmapCoords + expression_table = ArabidopsisNIEUmapExpression + species = "arabidopsis" + + else: + return {"success": False, "error": "Invalid database", "error_code": 400} + + return {"success": True, "coords_table": coords_table, "expression_table": expression_table, "species": species} + + +@umap_gene_expression.route("/") +class GetUMAPCoordinates(Resource): + @umap_gene_expression.param("database", _in="path", default="arabidopsis_NIE_umap") + def get(self, database=""): + """This end point returns the UMAP coordinates of every cell""" + database = escape(database) + + tables = UMAPUtils.get_tables(database) + if not tables["success"]: + return BARUtils.error_exit(tables["error"]), tables["error_code"] + + table = tables["coords_table"] + rows = db.session.execute(db.select(table.cell_id, table.umap_1, table.umap_2, table.cell_type)).all() + + data = {} + for row in rows: + data[row[0]] = {"umap_1": row[1], "umap_2": row[2], "cell_type": row[3]} + + return BARUtils.success_exit(data) + + +@umap_gene_expression.route("//") +class GetUMAPGeneExpression(Resource): + @umap_gene_expression.param("database", _in="path", default="arabidopsis_NIE_umap") + @umap_gene_expression.param("gene_id", _in="path", default="At1g01010") + def get(self, database="", gene_id=""): + """This end point returns the sparse UMAP gene expression data, keyed by cell id""" + database = escape(database) + gene_id = escape(gene_id) + + tables = UMAPUtils.get_tables(database) + if not tables["success"]: + return BARUtils.error_exit(tables["error"]), tables["error_code"] + + pattern = load_combined_master()["gene_id_patterns"][tables["species"]] + if not BARUtils.is_valid_gene_id(pattern, gene_id): + return BARUtils.error_exit("Invalid gene id"), 400 + + table = tables["expression_table"] + rows = db.session.execute(db.select(table.expression).where(table.gene_id == gene_id)).scalars().all() + + if len(rows) == 0: + return BARUtils.error_exit("There are no data found for the given gene"), 400 + + return BARUtils.success_exit(rows[0]) From 42060f0d9f190523d1713b4073ea10cb8f3afd74 Mon Sep 17 00:00:00 2001 From: Vin Date: Mon, 21 Sep 2026 02:58:44 -0400 Subject: [PATCH 12/13] Wire arabidopsis_NIE_umap into CI's MySQL bind and seed list --- config/BAR_API.cfg | 1 + config/init.sh | 1 + 2 files changed, 2 insertions(+) diff --git a/config/BAR_API.cfg b/config/BAR_API.cfg index 908adcd..debbc6a 100755 --- a/config/BAR_API.cfg +++ b/config/BAR_API.cfg @@ -13,6 +13,7 @@ SQLALCHEMY_BINDS = { 'annotations_lookup': 'mysql://root:root@localhost/annotations_lookup', 'arabidopsis_ecotypes': 'mysql://root:root@localhost/arabidopsis_ecotypes', 'arabidopsis_NIE_pseudobulk': 'mysql://root:root@localhost/arabidopsis_NIE_pseudobulk', + 'arabidopsis_NIE_umap': 'mysql://root:root@localhost/arabidopsis_NIE_umap', 'arachis': 'mysql://root:root@localhost/arachis', 'cannabis': 'mysql://root:root@localhost/cannabis', 'canola_nssnp' : 'mysql://root:root@localhost/canola_nssnp', diff --git a/config/init.sh b/config/init.sh index ef2502a..1e92e5c 100755 --- a/config/init.sh +++ b/config/init.sh @@ -12,6 +12,7 @@ echo "Welcome to the BAR API. Running init!" mysql -u $DB_USER -p$DB_PASS < ./config/databases/annotations_lookup.sql mysql -u $DB_USER -p$DB_PASS < ./config/databases/arabidopsis_ecotypes.sql mysql -u $DB_USER -p$DB_PASS < ./config/databases/arabidopsis_NIE_pseudobulk_dump.sql +mysql -u $DB_USER -p$DB_PASS < ./config/databases/arabidopsis_NIE_umap.sql mysql -u $DB_USER -p$DB_PASS < ./config/databases/arachis.sql mysql -u $DB_USER -p$DB_PASS < ./config/databases/cannabis.sql mysql -u $DB_USER -p$DB_PASS < ./config/databases/canola_nssnp.sql From 12d4c56bfc0fefd2221cb95d7b01365798fba5d3 Mon Sep 17 00:00:00 2001 From: Vin Date: Mon, 21 Sep 2026 03:57:18 -0400 Subject: [PATCH 13/13] Test the UMAP coordinates and sparse expression endpoints Whole-response equality against literal dicts, as in test_rnaseq_gene_expression.py. AT1G01010's expected dict has no key 45, so anything filling in zeros fails; it is requested as At1g01010 to exercise the case-insensitive collation the plain-equality query relies on. --- tests/resources/test_umap_gene_expression.py | 64 ++++++++++++++++++++ 1 file changed, 64 insertions(+) create mode 100644 tests/resources/test_umap_gene_expression.py diff --git a/tests/resources/test_umap_gene_expression.py b/tests/resources/test_umap_gene_expression.py new file mode 100644 index 0000000..96f2023 --- /dev/null +++ b/tests/resources/test_umap_gene_expression.py @@ -0,0 +1,64 @@ +from api import app +from unittest import TestCase + + +class TestIntegrations(TestCase): + def setUp(self): + self.app_client = app.test_client() + + def test_get_arabidopsis_nie_umap_coordinates(self): + """This tests the UMAP coordinates returned for the Arabidopsis NIE UMAP database + :return: + """ + # Valid data + response = self.app_client.get("/umap_gene_expression/arabidopsis_NIE_umap") + expected = { + "wasSuccessful": True, + "data": { + "43": {"umap_1": -6.89231, "umap_2": 7.56863, "cell_type": "Metabolic stress state"}, + "44": {"umap_1": 2.87577, "umap_2": -4.82349, "cell_type": "Dividing"}, + "45": {"umap_1": -0.262505, "umap_2": -8.55344, "cell_type": "Guard"}, + "46": {"umap_1": -4.5876, "umap_2": 6.05261, "cell_type": "Defense state"}, + }, + } + self.assertEqual(response.json, expected) + + # Invalid database + response = self.app_client.get("/umap_gene_expression/arabidopsis_NIE_um;ap") + expected = {"wasSuccessful": False, "error": "Invalid database"} + self.assertEqual(response.json, expected) + + def test_get_arabidopsis_nie_umap_gene(self): + """This tests the sparse UMAP expression data returned for a gene. Cells with no key are zero. + :return: + """ + # Valid data, with no expression in cell 45 + response = self.app_client.get("/umap_gene_expression/arabidopsis_NIE_umap/At1g01010") + expected = {"wasSuccessful": True, "data": {"43": 1.152292, "44": 1.546603, "46": 1.392931}} + self.assertEqual(response.json, expected) + + # Valid data, with expression in every cell + response = self.app_client.get("/umap_gene_expression/arabidopsis_NIE_umap/AT3G18780") + expected = { + "wasSuccessful": True, + "data": {"43": 1.673516, "44": 3.142986, "45": 1.490115, "46": 1.953491}, + } + self.assertEqual(response.json, expected) + + # Invalid gene + response = self.app_client.get("/umap_gene_expression/arabidopsis_NIE_umap/At1g0101x") + expected = {"wasSuccessful": False, "error": "Invalid gene id"} + self.assertEqual(response.json, expected) + + # Invalid database + response = self.app_client.get("/umap_gene_expression/arabidopsis_NIE_um;ap/At1g01010") + expected = {"wasSuccessful": False, "error": "Invalid database"} + self.assertEqual(response.json, expected) + + # No data for a valid gene + response = self.app_client.get("/umap_gene_expression/arabidopsis_NIE_umap/At1g01011") + expected = { + "wasSuccessful": False, + "error": "There are no data found for the given gene", + } + self.assertEqual(response.json, expected)